diff --git a/README.md b/README.md index b642eba..bf698f9 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ LuaPyre is a clean-slate Lua runtime written in Python. It targets **Lua 5.5.1** semantics, a sandbox-first embedding model, and optional gradual type annotations that feed runtime optimization without creating a second language/runtime. -**Python 3.13+** · **current pre-alpha: 0.35.0a1** +**Python 3.13+** · **current pre-alpha: 0.37.0a1** LuaPyre implements Lua 5.5.1 language semantics for its supported sandboxed embedding profile. The runtime is built around a register VM and explicit Lua frames, with a guarded tiered JIT that specializes proven hot paths and deoptimizes back to the same interpreter. @@ -185,10 +185,18 @@ and hook callbacks do not recursively invoke themselves. The [0.35 performance record](docs/speed-0.35.md) contains paired measurements, code-shape checks, rejected experiments, and correctness gates. +0.37 indexes table iteration/deletion and broadens scalar Python entry to +strict floats and two integers. See the [0.37 performance record](docs/speed-0.37.md) +for paired Python 3.13/3.14 results. + The [0.36 performance roadmap](docs/performance-roadmap-0.36.md) adds fresh Python-headroom measurements, 11 focused speed probes, and the next ordered work on scalar entry, cross-block facts, nested regions, tables, and recursion. -This planning pass leaves the runtime at 0.35.0a1. +That planning pass left the runtime at 0.35.0a1; the implemented 0.37 tranche +below advances the package version. + +The [0.37 roadmap](docs/performance-roadmap-0.37.md) retains the deferred call, +region, table-proof, and recursion work without representing it as completed. Pinned tests and workloads from real packages provide an additional compatibility gate. Penlight's portable upstream suite is 23/23 green, diff --git a/benchmarks/README.md b/benchmarks/README.md index c5dc3d1..516b062 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -109,6 +109,9 @@ For performance comparisons, compare runs on the same runner class and Python ve The results, interpretation, and next implementation priorities are in [`performance-roadmap-0.35.md`](../docs/performance-roadmap-0.35.md), with the earlier comparison retained in the 0.34 roadmap. +- `native_headroom.py` compares that same corpus three ways: LuaPyre, native + Lua 5.5 through `lupa.lua55`, and the reduced-contract Python lower bounds. + It runs isolated processes and rotates implementation order. - `inspect_codegen.py` captures final generated AST/source, selects hot generated functions using checked profiled executions, and records generic and warmed adaptive opcode counts plus pooled frame/register-slot counts. diff --git a/benchmarks/inspect_codegen.py b/benchmarks/inspect_codegen.py index 2c6d8cd..69e847e 100644 --- a/benchmarks/inspect_codegen.py +++ b/benchmarks/inspect_codegen.py @@ -110,7 +110,9 @@ def main(): parser.add_argument("--warmups", type=int, default=7) parser.add_argument("--executions", type=int, default=3) parser.add_argument("--top", type=int, default=3) - parser.add_argument("--suite", choices=("headroom", "036"), default="headroom") + parser.add_argument( + "--suite", choices=("headroom", "036", "037"), default="headroom" + ) parser.add_argument("--case", action="append") parser.add_argument("--include-source", action="store_true") args = parser.parse_args() @@ -123,6 +125,14 @@ def main(): cases = CASES + def prepare_case(name): + runtime, run, _reference = prepare_probe(name) + return runtime, run, lambda value: validate(CASES[name], value) + elif args.suite == "037": + from speed_037_ab import CASES, prepare as prepare_probe, validate + + cases = CASES + def prepare_case(name): runtime, run, _reference = prepare_probe(name) return runtime, run, lambda value: validate(CASES[name], value) diff --git a/benchmarks/native_headroom.py b/benchmarks/native_headroom.py new file mode 100644 index 0000000..eb51619 --- /dev/null +++ b/benchmarks/native_headroom.py @@ -0,0 +1,227 @@ +"""Compare LuaPyre with native Lua 5.5 and reduced-contract Python. + +The native column runs the same untyped Lua source through Lupa's pinned +``lupa.lua55`` backend. The Python functions preserve each algorithm but omit +Lua runtime semantics, so they are engineering lower bounds rather than +equivalent implementations or mathematical limits. +""" +from __future__ import annotations + +import argparse +from dataclasses import asdict +import gc +import json +import os +from pathlib import Path +import platform +import statistics +import subprocess +import sys +import time + +from lupa.lua55 import LuaRuntime as NativeLuaRuntime + +from python_headroom import REFERENCES, prepare as prepare_luapyre +from vm_programs import WORKLOADS + + +ORDERS = ( + ("luapyre", "native_lua55", "python_lower_bound"), + ("native_lua55", "python_lower_bound", "luapyre"), + ("python_lower_bound", "luapyre", "native_lua55"), +) + + +def prepare_native(name): + runtime = NativeLuaRuntime(unpack_returned_tuples=True) + version = runtime.eval("_VERSION") + if version != "Lua 5.5": + raise RuntimeError(f"expected Lua 5.5, got {version!r}") + if name == "python_calls_1000": + function = runtime.execute("return function(x) return x + 1 end") + + def run(): + for value in range(1000): + assert function(value) == value + 1 + return 1000 + + validate = lambda result: result == 1000 + else: + workload = WORKLOADS[name] + function = runtime.execute( + "return function()\n" + workload.source_for("lua") + "\nend" + ) + run = function + validate = workload.validate + return runtime, run, validate + + +def measure(run, validate, warmups, repeats): + for _ in range(warmups): + assert validate(run()) + samples = [] + was_enabled = gc.isenabled() + gc.disable() + try: + for _ in range(repeats): + start = time.perf_counter_ns() + result = run() + samples.append((time.perf_counter_ns() - start) / 1_000_000) + assert validate(result), result + finally: + if was_enabled: + gc.enable() + return {"median_ms": statistics.median(samples), "samples_ms": samples} + + +def run_worker(names, order, warmups, repeats): + affinity = None + if hasattr(os, "sched_getaffinity"): + affinity = min(os.sched_getaffinity(0)) + os.sched_setaffinity(0, {affinity}) + results = {} + for name in names: + luapyre_runtime, luapyre_run, validate = prepare_luapyre(name) + native_runtime, native_run, native_validate = prepare_native(name) + runners = { + "luapyre": luapyre_run, + "native_lua55": native_run, + "python_lower_bound": REFERENCES[name], + } + row = { + label: measure(runners[label], validate, warmups, repeats) + for label in order + } + # Retain both runtimes until all samples and the warm-counter probe end. + assert native_runtime.eval("_VERSION") == "Lua 5.5" + assert native_validate(native_run()) + before = asdict(luapyre_runtime.jit_stats) + assert validate(luapyre_run()) + after = asdict(luapyre_runtime.jit_stats) + row["luapyre_warm_counters"] = { + key: after[key] - value + for key, value in before.items() + if type(value) is int and after[key] != value + } + results[name] = row + return { + "python": platform.python_version(), + "platform": platform.platform(), + "affinity_cpu": affinity, + "hash_seed": os.environ.get("PYTHONHASHSEED"), + "order": list(order), + "workloads": results, + } + + +def aggregate(revision, runs, warmups, repeats): + results = {} + for name in runs[0]["workloads"]: + row = {} + for label in ORDERS[0]: + medians = [run["workloads"][name][label]["median_ms"] for run in runs] + row[label] = { + "median_ms": statistics.median(medians), + "process_medians_ms": medians, + } + row["luapyre_over_native"] = ( + row["luapyre"]["median_ms"] / row["native_lua55"]["median_ms"] + ) + row["luapyre_over_python_lower_bound"] = ( + row["luapyre"]["median_ms"] / row["python_lower_bound"]["median_ms"] + ) + row["native_over_python_lower_bound"] = ( + row["native_lua55"]["median_ms"] + / row["python_lower_bound"]["median_ms"] + ) + results[name] = row + return { + "schema_version": 1, + "revision": revision, + "python": runs[0]["python"], + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": runs[0]["platform"], + "warmups": warmups, + "repeats": repeats, + "processes": len(runs), + "method": ( + f"{len(runs)} isolated processes; rotated implementation order; fixed " + "CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; " + "Lua GC active; every result checked; median of process medians" + ), + "interpretation": ( + "Native Lua uses the same untyped Lua algorithm. Python lower bounds " + "use the same algorithm but omit Lua runtime semantics and are not " + "attainable-speed guarantees or mathematical lower bounds." + ), + "runs": runs, + "workloads": results, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--revision", required=True) + parser.add_argument("--json", type=Path) + parser.add_argument("--warmups", type=int, default=7) + parser.add_argument("--repeats", type=int, default=31) + parser.add_argument("--processes", type=int, default=3) + parser.add_argument("--case", action="append", choices=REFERENCES) + parser.add_argument("--worker", action="store_true", help=argparse.SUPPRESS) + parser.add_argument( + "--order", choices=range(len(ORDERS)), type=int, help=argparse.SUPPRESS + ) + args = parser.parse_args() + if args.warmups < 1 or args.repeats < 1 or args.processes < 1: + parser.error("warmups, repeats, and processes must be positive") + names = args.case or list(REFERENCES) + if args.worker: + if args.order is None: + parser.error("workers require --order") + json.dump( + run_worker(names, ORDERS[args.order], args.warmups, args.repeats), + sys.stdout, + ) + return + if args.json is None: + parser.error("parent runs require --json") + + runs = [] + for process_index in range(args.processes): + command = [ + sys.executable, + str(Path(__file__).resolve()), + "--revision", + args.revision, + "--worker", + "--order", + str(process_index % len(ORDERS)), + "--warmups", + str(args.warmups), + "--repeats", + str(args.repeats), + ] + for name in args.case or (): + command.extend(("--case", name)) + environment = os.environ.copy() + environment["PYTHONHASHSEED"] = "0" + completed = subprocess.run( + command, check=True, capture_output=True, text=True, env=environment + ) + runs.append(json.loads(completed.stdout)) + print( + f"{runs[-1]['python']} process {process_index + 1}/{args.processes}: " + f"{'/'.join(runs[-1]['order'])}", + flush=True, + ) + report = aggregate(args.revision, runs, args.warmups, args.repeats) + args.json.write_text(json.dumps(report, indent=2) + "\n") + for name, row in report["workloads"].items(): + print( + f"{name}: {row['luapyre_over_native']:.2f}x native Lua 5.5; " + f"{row['luapyre_over_python_lower_bound']:.2f}x Python lower bound" + ) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/results/codegen_037_313.json b/benchmarks/results/codegen_037_313.json new file mode 100644 index 0000000..6c52ded --- /dev/null +++ b/benchmarks/results/codegen_037_313.json @@ -0,0 +1,175 @@ +{ + "schema_version": 1, + "revision": "0.37.0a1", + "suite": "037", + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "profiled_executions": 3, + "interpretation": "Diagnostic static code counts and pool slot counts; no elapsed-time or transitive-memory claims", + "workloads": [ + { + "name": "python_leaf_local", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 126, + "locals": 12, + "names": [ + "proto", + "constants" + ], + "generic_opcodes": { + "BINARY_OP": 2, + "BINARY_SUBSCR": 2, + "LOAD_ATTR": 2, + "LOAD_CONST": 8, + "LOAD_FAST": 8, + "LOAD_FAST_LOAD_FAST": 2, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 15 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_INT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_SUBSCR_LIST_INT": 2, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST": 8, + "LOAD_FAST": 8, + "LOAD_FAST_LOAD_FAST": 2, + "STORE_FAST": 15 + }, + "ast_stores": { + "_r0": 1, + "_r1": 2, + "_r2": 2, + "_r3": 2, + "_r4": 2, + "_r5": 2, + "_r6": 2, + "_v": 1, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + }, + { + "name": "python_leaf_float", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 106, + "locals": 8, + "names": [ + "proto", + "constants", + "float" + ], + "generic_opcodes": { + "BINARY_OP": 1, + "BINARY_SUBSCR": 1, + "CALL": 1, + "LOAD_ATTR": 2, + "LOAD_CONST": 4, + "LOAD_FAST": 5, + "LOAD_FAST_LOAD_FAST": 1, + "LOAD_GLOBAL": 1, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 8 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_FLOAT": 1, + "BINARY_SUBSCR_LIST_INT": 1, + "INSTRUMENTED_CALL": 1, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST": 4, + "LOAD_FAST": 5, + "LOAD_FAST_LOAD_FAST": 1, + "LOAD_GLOBAL_BUILTIN": 1, + "STORE_FAST": 8 + }, + "ast_stores": { + "_r0": 1, + "_r1": 2, + "_r2": 2, + "_r3": 2, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + }, + { + "name": "python_leaf_binary", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 76, + "locals": 9, + "names": [ + "proto", + "constants" + ], + "generic_opcodes": { + "BINARY_OP": 1, + "LOAD_ATTR": 2, + "LOAD_CONST": 2, + "LOAD_FAST": 4, + "LOAD_FAST_LOAD_FAST": 1, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 6, + "STORE_FAST_LOAD_FAST": 1 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_INT": 1, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST": 2, + "LOAD_FAST": 4, + "LOAD_FAST_LOAD_FAST": 1, + "STORE_FAST": 6, + "STORE_FAST_LOAD_FAST": 1 + }, + "ast_stores": { + "_r0": 1, + "_r1": 1, + "_r2": 2, + "_r3": 2, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + } + ] +} diff --git a/benchmarks/results/codegen_037_314.json b/benchmarks/results/codegen_037_314.json new file mode 100644 index 0000000..2e0a0e4 --- /dev/null +++ b/benchmarks/results/codegen_037_314.json @@ -0,0 +1,183 @@ +{ + "schema_version": 1, + "revision": "0.37.0a1", + "suite": "037", + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "profiled_executions": 3, + "interpretation": "Diagnostic static code counts and pool slot counts; no elapsed-time or transitive-memory claims", + "workloads": [ + { + "name": "python_leaf_local", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 158, + "locals": 12, + "names": [ + "proto", + "constants" + ], + "generic_opcodes": { + "BINARY_OP": 4, + "LOAD_ATTR": 2, + "LOAD_CONST": 6, + "LOAD_FAST": 4, + "LOAD_FAST_BORROW": 4, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 2, + "LOAD_SMALL_INT": 2, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 15 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_INT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_OP_SUBSCR_LIST_INT": 2, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST_IMMORTAL": 6, + "LOAD_FAST": 4, + "LOAD_FAST_BORROW": 4, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 2, + "LOAD_SMALL_INT": 2, + "STORE_FAST": 15 + }, + "ast_stores": { + "_r0": 1, + "_r1": 2, + "_r2": 2, + "_r3": 2, + "_r4": 2, + "_r5": 2, + "_r6": 2, + "_v": 1, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + }, + { + "name": "python_leaf_float", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 122, + "locals": 8, + "names": [ + "proto", + "constants", + "float" + ], + "generic_opcodes": { + "BINARY_OP": 2, + "CALL": 1, + "LOAD_ATTR": 2, + "LOAD_CONST": 3, + "LOAD_FAST": 2, + "LOAD_FAST_BORROW": 3, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 1, + "LOAD_GLOBAL": 1, + "LOAD_SMALL_INT": 1, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 8 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_FLOAT": 1, + "BINARY_OP_SUBSCR_LIST_INT": 1, + "INSTRUMENTED_CALL": 1, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST_IMMORTAL": 3, + "LOAD_FAST": 2, + "LOAD_FAST_BORROW": 3, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 1, + "LOAD_GLOBAL_BUILTIN": 1, + "LOAD_SMALL_INT": 1, + "STORE_FAST": 8 + }, + "ast_stores": { + "_r0": 1, + "_r1": 2, + "_r2": 2, + "_r3": 2, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + }, + { + "name": "python_leaf_binary", + "hot_generated_functions": [ + { + "file": "", + "function": "_jit_leaf_scalar", + "profiled_calls": 3000, + "profiled_executions": 3, + "bytecode_bytes": 84, + "locals": 9, + "names": [ + "proto", + "constants" + ], + "generic_opcodes": { + "BINARY_OP": 1, + "LOAD_ATTR": 2, + "LOAD_CONST": 2, + "LOAD_FAST": 2, + "LOAD_FAST_BORROW": 2, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 1, + "RESUME": 1, + "RETURN_VALUE": 1, + "STORE_FAST": 6, + "STORE_FAST_LOAD_FAST": 1 + }, + "adaptive_opcodes": { + "BINARY_OP_ADD_INT": 1, + "INSTRUMENTED_RESUME": 1, + "INSTRUMENTED_RETURN_VALUE": 1, + "LOAD_ATTR_SLOT": 2, + "LOAD_CONST_IMMORTAL": 2, + "LOAD_FAST": 2, + "LOAD_FAST_BORROW": 2, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 1, + "STORE_FAST": 6, + "STORE_FAST_LOAD_FAST": 1 + }, + "ast_stores": { + "_r0": 1, + "_r1": 1, + "_r2": 2, + "_r3": 2, + "consts": 1 + } + } + ], + "cached_function_entries": 0, + "cached_call_entries": 0, + "retained_pooled_frames": 0, + "retained_register_slots": 0, + "retention_note": "Slot counts only; not transitive heap bytes or a leak test" + } + ] +} diff --git a/benchmarks/results/native_headroom_037_313.json b/benchmarks/results/native_headroom_037_313.json new file mode 100644 index 0000000..6c3c198 --- /dev/null +++ b/benchmarks/results/native_headroom_037_313.json @@ -0,0 +1,3397 @@ +{ + "schema_version": 1, + "revision": "3312f76", + "python": "3.13.15", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "Three isolated processes; Latin-square implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "typed_arith": { + "luapyre": { + "median_ms": 1.01464, + "samples_ms": [ + 0.966348, + 1.009652, + 0.966318, + 0.950615, + 1.175255, + 1.071913, + 0.961802, + 1.098803, + 0.941952, + 2.228741, + 1.05638, + 1.958725, + 1.174464, + 1.033067, + 1.002662, + 1.200752, + 1.01464, + 0.979908, + 0.998665, + 0.965627, + 1.079484, + 0.987499, + 1.033847, + 0.964686, + 0.985286, + 1.190517, + 0.94736, + 1.184288, + 1.24059, + 1.058945, + 0.958467 + ] + }, + "native_lua55": { + "median_ms": 0.127818, + "samples_ms": [ + 0.123201, + 0.127117, + 0.127006, + 0.162518, + 0.108449, + 0.144923, + 0.155839, + 0.12256, + 0.131222, + 0.113957, + 0.123742, + 0.127838, + 0.131924, + 0.127818, + 0.133195, + 0.137452, + 0.132014, + 0.123291, + 0.186253, + 0.115389, + 0.131152, + 0.141397, + 0.131282, + 0.107397, + 0.113577, + 0.12287, + 0.122861, + 0.126976, + 0.126957, + 0.227744, + 0.133336 + ] + }, + "python_lower_bound": { + "median_ms": 0.731033, + "samples_ms": [ + 0.731764, + 0.706657, + 0.72903, + 0.719887, + 0.8928, + 0.887943, + 0.705085, + 0.718114, + 0.722471, + 0.735579, + 0.778212, + 0.741779, + 0.726175, + 0.715831, + 0.71525, + 0.7909, + 0.777671, + 0.7361, + 0.706667, + 0.714058, + 0.739776, + 0.713928, + 0.725906, + 0.780685, + 0.736551, + 0.736331, + 0.731033, + 0.782158, + 0.728319, + 0.703963, + 0.746195 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "luapyre": { + "median_ms": 2.215192, + "samples_ms": [ + 2.26231, + 2.572495, + 2.285304, + 2.239867, + 2.210584, + 2.217965, + 2.215192, + 2.147122, + 2.139821, + 2.566047, + 2.48699, + 2.218476, + 2.139019, + 2.127722, + 2.13937, + 2.204546, + 2.402627, + 2.434383, + 2.213549, + 2.140201, + 2.160882, + 2.324802, + 2.249211, + 2.204426, + 2.120863, + 2.537073, + 2.15957, + 2.188222, + 2.239527, + 2.129085, + 2.389277 + ] + }, + "native_lua55": { + "median_ms": 0.262975, + "samples_ms": [ + 0.478553, + 0.262075, + 0.31983, + 0.263316, + 0.262815, + 0.262756, + 0.277337, + 0.263097, + 0.313961, + 0.262975, + 0.291157, + 0.261574, + 0.261414, + 0.27225, + 0.261333, + 0.260983, + 0.275915, + 0.276245, + 0.262495, + 0.262295, + 0.262525, + 0.277658, + 0.261955, + 0.261824, + 0.274352, + 0.262275, + 0.262205, + 0.261955, + 0.478442, + 0.263997, + 0.280481 + ] + }, + "python_lower_bound": { + "median_ms": 1.5534, + "samples_ms": [ + 1.481854, + 1.619988, + 1.483447, + 1.736208, + 1.60134, + 1.586208, + 1.568562, + 1.492831, + 1.694577, + 1.489556, + 1.56715, + 1.503947, + 1.476417, + 1.674337, + 1.538938, + 1.507082, + 1.763518, + 1.479571, + 1.509515, + 1.465651, + 1.68327, + 1.5534, + 1.497649, + 1.482156, + 1.552669, + 1.47149, + 2.573988, + 1.88699, + 1.94179, + 2.967075, + 1.577565 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "luapyre": { + "median_ms": 31.395689, + "samples_ms": [ + 31.06958, + 32.060565, + 31.99617, + 32.333114, + 33.140079, + 31.679335, + 31.888842, + 30.972067, + 30.958737, + 31.970902, + 31.210997, + 31.638575, + 32.79349, + 31.638145, + 31.011004, + 31.791139, + 32.257443, + 30.940912, + 32.082467, + 30.99492, + 30.677244, + 31.071423, + 30.978926, + 30.75585, + 31.067437, + 31.188164, + 30.809689, + 32.302058, + 31.395689, + 33.146258, + 30.815077 + ] + }, + "native_lua55": { + "median_ms": 0.326259, + "samples_ms": [ + 0.315433, + 0.325788, + 0.31348, + 0.339558, + 0.326259, + 0.34746, + 0.351626, + 0.389211, + 0.318598, + 0.316234, + 0.548655, + 0.31978, + 0.342763, + 0.318277, + 0.332879, + 0.347601, + 0.315844, + 0.370194, + 0.342383, + 0.316604, + 0.316074, + 0.330275, + 0.315402, + 0.32043, + 0.331827, + 0.315183, + 0.330555, + 0.330846, + 0.316725, + 0.315013, + 0.326439 + ] + }, + "python_lower_bound": { + "median_ms": 0.761137, + "samples_ms": [ + 0.758873, + 0.950875, + 0.772574, + 0.863297, + 0.771112, + 0.761137, + 0.834505, + 0.768207, + 0.747216, + 0.758193, + 0.756169, + 1.052825, + 0.756811, + 0.744152, + 0.761487, + 0.753516, + 0.855466, + 0.761969, + 0.785943, + 0.763381, + 0.756511, + 0.772734, + 0.755329, + 0.76291, + 0.75645, + 0.755569, + 0.766716, + 0.759886, + 0.757402, + 0.759284, + 0.747036 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "luapyre": { + "median_ms": 103.168043, + "samples_ms": [ + 101.076408, + 104.49197, + 103.164141, + 102.480557, + 103.431544, + 105.024832, + 106.30553, + 104.863311, + 106.372865, + 103.141535, + 105.389473, + 101.913748, + 101.811213, + 102.516687, + 110.916061, + 102.185061, + 104.673312, + 103.521011, + 102.967339, + 103.367065, + 102.662101, + 101.288897, + 103.789466, + 102.757611, + 102.011347, + 102.174405, + 101.988272, + 103.168043, + 106.339501, + 103.816553, + 103.974886 + ] + }, + "native_lua55": { + "median_ms": 1.763648, + "samples_ms": [ + 1.868573, + 1.705743, + 1.700406, + 1.729348, + 1.732353, + 1.905507, + 1.781866, + 1.764169, + 1.733114, + 1.754705, + 1.729338, + 1.741256, + 1.830917, + 1.731572, + 1.789757, + 2.034386, + 1.833521, + 1.757118, + 1.753854, + 1.739262, + 1.842184, + 1.763648, + 1.73803, + 1.926217, + 1.767704, + 1.814262, + 1.842193, + 1.755836, + 1.870535, + 1.875563, + 1.731942 + ] + }, + "python_lower_bound": { + "median_ms": 2.155493, + "samples_ms": [ + 2.092361, + 2.111829, + 2.323349, + 2.388896, + 2.239367, + 2.181641, + 2.258665, + 2.168112, + 2.3439, + 2.30349, + 2.201321, + 2.114784, + 2.172147, + 2.216393, + 2.114243, + 2.172638, + 2.087163, + 2.101545, + 2.165338, + 2.087704, + 2.088465, + 2.099381, + 2.136926, + 2.118389, + 2.161602, + 2.150175, + 2.434773, + 2.14623, + 2.155493, + 2.084919, + 2.15301 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "luapyre": { + "median_ms": 23.015227, + "samples_ms": [ + 25.4975, + 23.671491, + 22.913298, + 22.368909, + 22.693886, + 22.749969, + 22.906098, + 22.800803, + 22.985595, + 22.578326, + 23.260968, + 23.099842, + 23.31696, + 23.740593, + 26.427515, + 23.983369, + 22.980717, + 23.388536, + 23.015227, + 23.285835, + 23.076518, + 22.639166, + 22.823977, + 25.297807, + 23.609109, + 23.571905, + 22.907129, + 22.506621, + 23.210184, + 22.499822, + 22.575732 + ] + }, + "native_lua55": { + "median_ms": 0.192603, + "samples_ms": [ + 0.192603, + 0.202907, + 0.19065, + 0.192423, + 0.189678, + 0.191431, + 0.189729, + 0.266191, + 0.194024, + 0.193414, + 0.21122, + 0.214655, + 0.191591, + 0.192031, + 0.1904, + 0.191541, + 0.220133, + 0.289155, + 0.199492, + 0.195006, + 0.191411, + 0.258159, + 0.19093, + 0.195226, + 0.19122, + 0.191571, + 0.212602, + 0.194235, + 0.191761, + 0.191481, + 0.224339 + ] + }, + "python_lower_bound": { + "median_ms": 0.586331, + "samples_ms": [ + 0.579871, + 0.597878, + 0.578109, + 0.598999, + 0.600502, + 0.578579, + 0.616044, + 0.574303, + 0.597306, + 0.598078, + 0.575775, + 0.59237, + 0.593101, + 0.595024, + 0.596426, + 0.57189, + 0.59262, + 0.571949, + 0.58638, + 0.597667, + 0.574704, + 0.584718, + 0.573442, + 0.584448, + 0.577498, + 0.583586, + 0.604677, + 0.571909, + 0.586791, + 0.577728, + 0.586331 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "luapyre": { + "median_ms": 29.627071, + "samples_ms": [ + 30.344734, + 30.669552, + 30.187564, + 56.04309, + 29.239563, + 29.627071, + 34.901572, + 29.570268, + 29.31984, + 29.583187, + 29.914463, + 29.499264, + 29.932469, + 30.737862, + 29.837169, + 29.568345, + 32.252003, + 29.723704, + 29.850349, + 29.048923, + 29.196079, + 29.003926, + 29.204952, + 29.140086, + 29.287794, + 29.245792, + 29.70763, + 29.474748, + 29.422571, + 31.065482, + 29.722231 + ] + }, + "native_lua55": { + "median_ms": 0.586611, + "samples_ms": [ + 0.574203, + 0.590877, + 0.574873, + 0.602274, + 0.574483, + 0.5858, + 0.58599, + 0.576807, + 0.589235, + 0.587352, + 0.586831, + 0.58592, + 0.573993, + 0.586651, + 0.573872, + 0.589435, + 0.589014, + 0.587462, + 0.587082, + 0.573622, + 0.586611, + 0.574403, + 0.58581, + 0.602054, + 0.574153, + 0.586832, + 0.574133, + 0.588403, + 0.573722, + 0.587362, + 0.601082 + ] + }, + "python_lower_bound": { + "median_ms": 3.370947, + "samples_ms": [ + 3.64505, + 3.402403, + 3.535168, + 3.423715, + 3.371037, + 3.357177, + 3.357798, + 3.346552, + 3.352821, + 3.369855, + 3.358579, + 3.355335, + 3.430545, + 3.380621, + 3.582718, + 3.44098, + 3.368994, + 3.352981, + 3.36619, + 3.371097, + 3.353842, + 3.71371, + 3.490864, + 3.441491, + 3.390607, + 3.369726, + 3.369716, + 3.371298, + 3.369906, + 3.370947, + 3.349165 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "luapyre": { + "median_ms": 0.53158, + "samples_ms": [ + 0.527995, + 0.53184, + 0.545851, + 0.730513, + 0.566511, + 0.527885, + 0.535526, + 0.615573, + 0.578158, + 0.726787, + 0.528585, + 0.528646, + 0.518151, + 0.53112, + 0.520845, + 0.529137, + 0.547734, + 0.514285, + 0.53158, + 0.515347, + 0.529007, + 0.518581, + 0.55803, + 0.540423, + 0.539102, + 0.513093, + 0.523788, + 0.577597, + 0.574353, + 0.537438, + 0.513734 + ] + }, + "native_lua55": { + "median_ms": 1.171349, + "samples_ms": [ + 1.155576, + 1.148396, + 1.143418, + 1.173472, + 1.168555, + 1.168745, + 1.194613, + 1.237927, + 1.163418, + 1.162136, + 1.180793, + 1.171489, + 1.164489, + 1.176737, + 1.162646, + 1.155236, + 1.176647, + 1.173402, + 1.163157, + 1.252428, + 1.196306, + 1.171349, + 1.391943, + 1.250966, + 1.168145, + 1.166051, + 1.181074, + 1.171509, + 1.158711, + 1.170447, + 1.183697 + ] + }, + "python_lower_bound": { + "median_ms": 0.47691, + "samples_ms": [ + 0.467486, + 0.472203, + 0.484882, + 0.471913, + 0.484391, + 0.470811, + 0.481227, + 0.485422, + 0.480545, + 0.466294, + 0.482257, + 0.465924, + 0.486564, + 0.469218, + 0.481066, + 0.469299, + 0.494536, + 0.465834, + 0.477912, + 0.465884, + 0.484151, + 0.470791, + 0.464822, + 0.481507, + 0.481717, + 0.484601, + 0.464762, + 0.47691, + 0.471893, + 0.479574, + 0.472453 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 28.78119, + "samples_ms": [ + 29.610097, + 28.305542, + 28.78119, + 28.683416, + 28.544371, + 29.259612, + 29.320311, + 28.124706, + 28.758877, + 29.062283, + 28.279694, + 31.571475, + 29.448859, + 28.160649, + 30.195035, + 29.282896, + 28.693772, + 28.402974, + 28.592423, + 29.106397, + 28.703116, + 28.169381, + 28.198004, + 28.859305, + 28.224413, + 28.448592, + 29.725886, + 28.929317, + 30.331856, + 28.904521, + 28.781791 + ] + }, + "native_lua55": { + "median_ms": 0.842537, + "samples_ms": [ + 0.889145, + 0.866842, + 0.874013, + 0.869026, + 0.887823, + 0.859932, + 0.883767, + 0.873401, + 0.86508, + 0.928423, + 0.877658, + 0.870488, + 0.874744, + 1.11103, + 0.83096, + 0.857017, + 0.822908, + 0.818251, + 0.826613, + 0.8179, + 0.817921, + 0.797441, + 0.820264, + 0.832352, + 0.81114, + 0.817059, + 0.803049, + 0.810249, + 0.832532, + 0.842537, + 0.826663 + ] + }, + "python_lower_bound": { + "median_ms": 2.060364, + "samples_ms": [ + 1.973315, + 2.059793, + 3.020132, + 2.631161, + 2.792438, + 2.044681, + 2.039413, + 2.321307, + 2.319234, + 2.113953, + 1.997281, + 2.075827, + 2.008608, + 2.300867, + 1.979735, + 2.078831, + 1.947788, + 2.239778, + 2.045782, + 2.277392, + 2.033354, + 2.060364, + 2.280768, + 2.09172, + 2.053624, + 2.063899, + 1.96227, + 2.062937, + 1.978123, + 2.042799, + 1.96246 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.68139, + "samples_ms": [ + 0.696102, + 0.70135, + 0.671616, + 0.661281, + 0.672317, + 0.679687, + 0.659428, + 0.685987, + 0.669212, + 0.657284, + 0.68139, + 0.672227, + 0.666699, + 0.687169, + 0.857478, + 0.666378, + 0.672376, + 0.739085, + 0.916825, + 0.664535, + 0.706387, + 0.676362, + 0.746175, + 0.763891, + 0.719556, + 0.716271, + 0.66081, + 0.700878, + 0.709341, + 0.72937, + 0.664946 + ] + }, + "native_lua55": { + "median_ms": 0.199083, + "samples_ms": [ + 0.198752, + 0.197791, + 0.205692, + 0.402781, + 0.196027, + 0.250467, + 0.214525, + 0.244419, + 0.196228, + 0.198942, + 0.196638, + 0.200354, + 0.237759, + 0.19798, + 0.199042, + 0.199083, + 0.219742, + 0.205121, + 0.19746, + 0.199794, + 0.198902, + 0.232772, + 0.201356, + 0.19734, + 0.197099, + 0.212231, + 0.234594, + 0.196779, + 0.196619, + 0.199172, + 0.19765 + ] + }, + "python_lower_bound": { + "median_ms": 0.050153, + "samples_ms": [ + 0.050514, + 0.050153, + 0.050293, + 0.050093, + 0.050403, + 0.050153, + 0.050203, + 0.050223, + 0.050144, + 0.050194, + 0.050084, + 0.050184, + 0.050033, + 0.072316, + 0.058906, + 0.049834, + 0.049963, + 0.049864, + 0.049993, + 0.050414, + 0.050224, + 0.050144, + 0.049923, + 0.050143, + 0.050054, + 0.049953, + 0.049753, + 0.050264, + 0.049703, + 0.050344, + 0.050234 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "typed_arith": { + "native_lua55": { + "median_ms": 0.126186, + "samples_ms": [ + 0.126796, + 0.124533, + 0.120787, + 0.113006, + 0.133296, + 0.126365, + 0.126186, + 0.130201, + 0.12258, + 0.112966, + 0.126075, + 0.126065, + 0.14272, + 0.107568, + 0.126186, + 0.109471, + 0.126215, + 0.126465, + 0.174786, + 0.155788, + 0.130321, + 0.107307, + 0.112415, + 0.126115, + 0.126085, + 0.130181, + 0.122039, + 0.140056, + 0.126615, + 0.107257, + 0.130182 + ] + }, + "python_lower_bound": { + "median_ms": 0.720988, + "samples_ms": [ + 0.716822, + 0.721229, + 0.724002, + 0.716842, + 0.757162, + 0.719005, + 0.766174, + 0.746496, + 0.724964, + 0.704393, + 0.727798, + 0.719956, + 0.719726, + 0.712726, + 0.778612, + 0.71536, + 0.705335, + 0.718755, + 0.734778, + 0.776159, + 0.706367, + 0.725736, + 0.718615, + 0.720497, + 0.7361, + 0.720988, + 0.728589, + 0.712936, + 0.718965, + 0.731975, + 0.721348 + ] + }, + "luapyre": { + "median_ms": 0.95367, + "samples_ms": [ + 0.950124, + 1.003562, + 0.946129, + 0.956614, + 0.945678, + 0.966819, + 0.948432, + 0.949944, + 0.941702, + 0.953941, + 0.95367, + 0.939509, + 0.936865, + 0.941592, + 0.95348, + 0.938858, + 0.95352, + 0.959839, + 0.96111, + 0.945227, + 1.193762, + 1.02181, + 1.239228, + 0.98105, + 0.960048, + 0.955853, + 1.029491, + 0.950675, + 0.939999, + 0.957505, + 0.958567 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "native_lua55": { + "median_ms": 0.262045, + "samples_ms": [ + 0.262195, + 0.261774, + 0.261744, + 0.282104, + 0.263136, + 0.263597, + 0.34041, + 0.262535, + 0.261914, + 0.262045, + 0.276616, + 0.263406, + 0.263176, + 0.263517, + 0.278058, + 0.260843, + 0.260883, + 0.261144, + 0.276856, + 0.261223, + 0.260913, + 0.292509, + 0.261193, + 0.260772, + 0.261514, + 0.273161, + 0.261024, + 0.261163, + 0.261074, + 0.275284, + 0.260973 + ] + }, + "python_lower_bound": { + "median_ms": 1.454455, + "samples_ms": [ + 1.448636, + 1.429087, + 1.466502, + 1.438381, + 1.624224, + 1.443548, + 1.515834, + 1.460814, + 1.451691, + 1.428296, + 1.44413, + 1.452301, + 1.440644, + 1.547691, + 1.463869, + 1.52639, + 1.622681, + 1.453814, + 1.552808, + 1.448015, + 4.252882, + 1.827072, + 1.446413, + 1.552258, + 1.469036, + 1.445652, + 1.490869, + 1.454455, + 1.428296, + 1.536635, + 1.436037 + ] + }, + "luapyre": { + "median_ms": 2.205587, + "samples_ms": [ + 2.301437, + 2.16647, + 2.198898, + 2.158318, + 2.375797, + 2.137036, + 2.219577, + 2.157877, + 2.202162, + 2.314717, + 2.205587, + 2.154792, + 2.137367, + 2.254018, + 2.128184, + 2.233357, + 2.460371, + 2.285865, + 2.192118, + 2.170495, + 2.300536, + 2.262721, + 2.225126, + 2.408384, + 2.159049, + 2.130016, + 2.175182, + 2.136596, + 2.954035, + 2.927536, + 2.224084 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "native_lua55": { + "median_ms": 0.317146, + "samples_ms": [ + 0.314842, + 0.417473, + 0.341081, + 0.315724, + 0.315663, + 0.331647, + 0.318057, + 0.314662, + 0.326228, + 0.315443, + 0.317146, + 0.328302, + 0.314312, + 0.400508, + 0.34049, + 0.322334, + 0.31367, + 0.330585, + 0.31366, + 0.314552, + 0.325678, + 0.316434, + 0.3138, + 0.328422, + 0.315213, + 0.414598, + 0.339097, + 0.315903, + 0.316324, + 0.330254, + 0.314441 + ] + }, + "python_lower_bound": { + "median_ms": 0.795076, + "samples_ms": [ + 0.791332, + 0.798281, + 0.789439, + 0.969392, + 0.847003, + 0.979858, + 0.804951, + 0.779834, + 0.791381, + 0.855305, + 0.819803, + 0.79041, + 0.775107, + 0.789248, + 0.789639, + 0.802267, + 0.795527, + 0.778092, + 0.789348, + 0.796799, + 0.927581, + 0.795728, + 0.779724, + 0.804651, + 0.794296, + 0.871339, + 0.795076, + 0.775819, + 0.788557, + 0.791631, + 0.807415 + ] + }, + "luapyre": { + "median_ms": 30.326368, + "samples_ms": [ + 29.665218, + 32.952182, + 29.822398, + 33.415341, + 30.496387, + 30.201815, + 29.910507, + 30.415158, + 30.32116, + 30.476769, + 30.41632, + 30.115759, + 29.655472, + 29.588384, + 30.439303, + 29.573903, + 31.074846, + 30.533191, + 29.699537, + 29.753006, + 34.015852, + 30.326368, + 30.343724, + 31.38287, + 30.508517, + 30.442941, + 29.802501, + 30.569697, + 29.856871, + 29.892252, + 29.94525 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "native_lua55": { + "median_ms": 1.747605, + "samples_ms": [ + 1.690441, + 1.728957, + 1.697431, + 1.733414, + 1.730129, + 1.739914, + 1.725222, + 1.719573, + 1.758711, + 2.066032, + 1.768896, + 1.734616, + 1.759853, + 1.753494, + 1.752692, + 1.722037, + 1.746544, + 1.758831, + 1.715067, + 1.756658, + 1.754685, + 2.028237, + 1.733725, + 1.840922, + 1.725382, + 1.747605, + 1.770259, + 1.841413, + 1.913328, + 1.735306, + 2.079773 + ] + }, + "python_lower_bound": { + "median_ms": 2.049568, + "samples_ms": [ + 2.029649, + 2.061596, + 2.032003, + 2.048817, + 2.031752, + 2.048407, + 2.036169, + 2.051751, + 2.087864, + 2.047535, + 2.11903, + 2.056398, + 2.049568, + 2.047836, + 2.032042, + 2.056808, + 2.036699, + 2.060294, + 2.035888, + 2.048967, + 2.036218, + 2.293687, + 2.131769, + 2.035378, + 2.052663, + 2.036099, + 2.132891, + 3.427981, + 2.051992, + 3.216221, + 2.076437 + ] + }, + "luapyre": { + "median_ms": 101.757409, + "samples_ms": [ + 105.312897, + 102.99132, + 104.677154, + 100.058425, + 99.637808, + 100.782908, + 105.063801, + 105.025455, + 102.123987, + 101.198789, + 104.218371, + 101.873961, + 104.868595, + 100.085004, + 100.438873, + 101.093734, + 104.573232, + 100.931046, + 101.249513, + 99.636255, + 103.207627, + 105.130349, + 103.649997, + 98.979572, + 100.826884, + 99.343796, + 104.944336, + 102.476324, + 100.186293, + 101.013838, + 101.757409 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "native_lua55": { + "median_ms": 0.19782, + "samples_ms": [ + 0.216828, + 0.195718, + 0.201476, + 0.200545, + 0.19726, + 0.21137, + 0.19782, + 0.198952, + 0.195667, + 0.196238, + 0.2184, + 0.19807, + 0.195126, + 0.194966, + 0.212442, + 0.212532, + 0.19784, + 0.202417, + 0.19781, + 0.19804, + 0.219712, + 0.19739, + 0.195667, + 0.197189, + 0.194435, + 0.198341, + 0.194545, + 0.196338, + 0.193274, + 0.193074, + 0.206653 + ] + }, + "python_lower_bound": { + "median_ms": 0.612139, + "samples_ms": [ + 0.588784, + 0.589726, + 0.592639, + 0.644486, + 0.611898, + 0.591529, + 0.578279, + 0.612139, + 0.576957, + 0.833843, + 0.60658, + 0.577358, + 0.676613, + 0.629244, + 0.672527, + 0.617576, + 0.57896, + 0.646329, + 0.614773, + 0.648942, + 0.605208, + 0.596996, + 0.761467, + 0.615874, + 0.592079, + 0.625408, + 0.91999, + 0.795287, + 0.580542, + 0.665497, + 0.612058 + ] + }, + "luapyre": { + "median_ms": 24.661486, + "samples_ms": [ + 24.689426, + 23.968558, + 23.963431, + 24.034716, + 24.815892, + 24.197285, + 26.390964, + 25.131626, + 24.892995, + 24.805276, + 24.537644, + 24.678621, + 25.282628, + 24.661486, + 25.081372, + 24.573356, + 24.537383, + 24.13995, + 24.54186, + 24.048946, + 24.796884, + 24.501701, + 24.408103, + 23.945715, + 25.347703, + 24.257883, + 24.74602, + 27.989179, + 25.500567, + 24.606174, + 25.064457 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "native_lua55": { + "median_ms": 0.606891, + "samples_ms": [ + 0.576446, + 0.589645, + 0.57259, + 0.592549, + 0.574723, + 0.628983, + 0.613461, + 0.575625, + 0.584938, + 0.574163, + 0.588234, + 0.76969, + 0.675131, + 0.630415, + 0.609905, + 0.595795, + 0.606891, + 0.671576, + 0.66085, + 0.654271, + 0.611228, + 0.597076, + 0.607371, + 0.595404, + 0.608734, + 0.620981, + 0.592309, + 0.608423, + 0.593962, + 0.610135, + 0.605829 + ] + }, + "python_lower_bound": { + "median_ms": 3.306803, + "samples_ms": [ + 3.579203, + 3.387943, + 3.302587, + 3.321985, + 3.299533, + 3.306093, + 3.301696, + 3.280024, + 3.316497, + 3.306803, + 3.353702, + 3.286804, + 3.438547, + 3.320083, + 3.280805, + 3.310168, + 3.288507, + 3.27737, + 3.305481, + 3.524904, + 3.401762, + 3.315245, + 3.296007, + 3.285422, + 3.342936, + 3.272984, + 3.307545, + 3.271662, + 3.314555, + 3.269959, + 3.756263 + ] + }, + "luapyre": { + "median_ms": 29.970898, + "samples_ms": [ + 31.502766, + 30.791052, + 30.50379, + 58.487902, + 29.554987, + 29.478365, + 30.099447, + 30.066849, + 29.361433, + 29.750785, + 29.87035, + 29.636588, + 31.167675, + 29.75373, + 29.516611, + 32.919386, + 30.262646, + 30.160066, + 31.871368, + 30.489549, + 29.970898, + 29.613544, + 29.258623, + 29.694743, + 29.174089, + 29.485125, + 30.51722, + 29.411127, + 29.736774, + 33.623669, + 31.223226 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "native_lua55": { + "median_ms": 1.186922, + "samples_ms": [ + 1.205038, + 1.147575, + 1.202245, + 1.186922, + 1.155917, + 1.379835, + 1.262483, + 1.167594, + 1.241372, + 1.297895, + 1.165571, + 1.156627, + 1.285146, + 1.17872, + 1.166542, + 1.156397, + 1.428947, + 1.245869, + 1.325625, + 1.243526, + 1.245348, + 1.172891, + 1.270274, + 1.164439, + 1.164439, + 1.291125, + 1.189626, + 1.157339, + 1.152461, + 1.168355, + 1.16502 + ] + }, + "python_lower_bound": { + "median_ms": 0.380628, + "samples_ms": [ + 0.37513, + 0.395691, + 0.380628, + 0.376582, + 0.391115, + 0.375752, + 0.389682, + 0.37505, + 0.596595, + 0.386567, + 0.379937, + 0.38863, + 0.373828, + 0.385435, + 0.378085, + 0.375982, + 0.464171, + 0.382411, + 0.406817, + 0.377735, + 0.379156, + 0.433346, + 0.379647, + 0.378385, + 0.41566, + 0.379387, + 0.389772, + 0.376112, + 0.389992, + 0.387499, + 0.376473 + ] + }, + "luapyre": { + "median_ms": 0.548996, + "samples_ms": [ + 0.545711, + 0.785182, + 0.547233, + 0.555105, + 0.555756, + 0.556978, + 0.643815, + 0.542106, + 0.548996, + 0.691244, + 0.554904, + 0.568474, + 0.537378, + 0.546241, + 0.526923, + 0.546362, + 0.530729, + 0.543437, + 0.554063, + 0.549667, + 0.784481, + 0.545971, + 0.566441, + 0.669913, + 0.566902, + 0.571589, + 0.534605, + 0.548555, + 0.532812, + 0.539512, + 0.521386 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "native_lua55": { + "median_ms": 0.832051, + "samples_ms": [ + 0.856327, + 0.828025, + 0.838671, + 0.832051, + 0.843748, + 0.848535, + 0.818661, + 0.83817, + 0.952008, + 0.850067, + 0.855616, + 0.827214, + 0.836298, + 0.816558, + 1.203326, + 0.800875, + 0.824701, + 0.809899, + 1.030793, + 0.848335, + 0.815717, + 0.804661, + 1.009232, + 0.866391, + 0.80364, + 0.795267, + 0.795267, + 0.793134, + 0.840634, + 0.815577, + 0.792192 + ] + }, + "python_lower_bound": { + "median_ms": 1.962039, + "samples_ms": [ + 1.957373, + 1.945595, + 1.96263, + 1.940548, + 1.961168, + 1.947989, + 2.135634, + 2.000807, + 2.011973, + 1.96251, + 1.964894, + 1.949461, + 1.966075, + 1.938906, + 2.073393, + 1.938014, + 2.026224, + 1.955339, + 2.059673, + 2.002769, + 1.965815, + 2.017651, + 1.962831, + 1.942451, + 1.962039, + 1.942321, + 1.959325, + 1.937793, + 1.936603, + 1.938225, + 1.964233 + ] + }, + "luapyre": { + "median_ms": 28.674444, + "samples_ms": [ + 30.764481, + 29.157452, + 28.674444, + 28.788231, + 28.528148, + 29.680971, + 28.530262, + 28.286234, + 28.125267, + 28.573425, + 28.457635, + 28.436294, + 30.446895, + 29.00583, + 28.596119, + 31.607308, + 29.147017, + 28.850142, + 28.667253, + 29.371827, + 28.78137, + 28.78121, + 28.067292, + 28.149653, + 28.623118, + 28.203663, + 28.885694, + 28.682946, + 28.278963, + 28.40577, + 28.888627 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "native_lua55": { + "median_ms": 0.205231, + "samples_ms": [ + 0.206594, + 0.199202, + 0.206443, + 0.200344, + 0.221595, + 0.212352, + 0.205202, + 0.206503, + 0.228616, + 0.222506, + 0.206954, + 0.203769, + 0.208095, + 0.208136, + 0.219643, + 0.204711, + 0.202307, + 0.204911, + 0.21823, + 0.2043, + 0.205161, + 0.204871, + 0.20432, + 0.218631, + 0.204951, + 0.20454, + 0.205061, + 0.225501, + 0.220644, + 0.205231, + 0.205051 + ] + }, + "python_lower_bound": { + "median_ms": 0.052327, + "samples_ms": [ + 0.050574, + 0.068381, + 0.052327, + 0.053849, + 0.051455, + 0.053788, + 0.050804, + 0.053729, + 0.051356, + 0.05461, + 0.051496, + 0.053969, + 0.051225, + 0.053929, + 0.050965, + 0.118194, + 0.050504, + 0.053508, + 0.050664, + 0.056643, + 0.050354, + 0.053238, + 0.050585, + 0.053408, + 0.050294, + 0.053688, + 0.050484, + 0.053208, + 0.050284, + 0.053559, + 0.050324 + ] + }, + "luapyre": { + "median_ms": 0.66077, + "samples_ms": [ + 0.667259, + 0.975191, + 0.741178, + 0.644366, + 0.65426, + 0.662192, + 0.646699, + 0.733947, + 0.68854, + 0.648091, + 0.661521, + 0.657946, + 0.649683, + 0.72279, + 0.686788, + 0.643224, + 0.657896, + 0.656073, + 0.643064, + 0.712025, + 0.690873, + 0.642783, + 0.663914, + 0.665627, + 0.653419, + 0.713267, + 0.697895, + 0.659338, + 0.650685, + 0.66077, + 0.644186 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "typed_arith": { + "python_lower_bound": { + "median_ms": 0.726406, + "samples_ms": [ + 0.790941, + 0.707689, + 0.724824, + 0.840073, + 0.754037, + 0.712085, + 0.721559, + 0.723371, + 0.726406, + 0.897237, + 0.704865, + 0.723872, + 0.723031, + 0.705846, + 0.818001, + 0.754738, + 0.726957, + 0.710302, + 0.724273, + 0.973189, + 0.756811, + 0.751653, + 0.707679, + 0.726676, + 0.802728, + 0.819222, + 0.70847, + 0.722691, + 0.719626, + 0.729881, + 0.731664 + ] + }, + "luapyre": { + "median_ms": 0.993698, + "samples_ms": [ + 0.993698, + 0.98794, + 0.984504, + 0.999927, + 0.984945, + 0.987089, + 0.990584, + 1.086244, + 1.006356, + 0.988551, + 0.999948, + 1.059044, + 1.008771, + 0.985126, + 1.066215, + 1.019316, + 0.999367, + 0.984264, + 1.107165, + 0.98779, + 0.987099, + 0.991204, + 1.007959, + 0.98794, + 1.0018, + 1.00863, + 0.998415, + 0.986118, + 0.969132, + 0.986247, + 1.002822 + ] + }, + "native_lua55": { + "median_ms": 0.126105, + "samples_ms": [ + 0.139975, + 0.124172, + 0.111053, + 0.126105, + 0.109751, + 0.126265, + 0.126146, + 0.126435, + 0.112825, + 0.137953, + 0.112536, + 0.126125, + 0.126716, + 0.107168, + 0.112325, + 0.126105, + 0.112816, + 0.133837, + 0.126346, + 0.112866, + 0.126095, + 0.130301, + 0.112625, + 0.148949, + 0.126356, + 0.129731, + 0.126286, + 0.106667, + 0.112565, + 0.120637, + 0.122069 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "python_lower_bound": { + "median_ms": 1.459672, + "samples_ms": [ + 1.465861, + 1.456287, + 1.426965, + 1.450218, + 1.447374, + 1.466272, + 1.467074, + 1.43765, + 1.48546, + 1.454114, + 1.449397, + 1.677272, + 1.68988, + 1.602161, + 1.448516, + 1.453233, + 1.562613, + 1.459672, + 1.436528, + 1.460333, + 1.456868, + 1.461074, + 1.459922, + 1.457128, + 1.474825, + 1.497037, + 1.471049, + 1.452632, + 1.455246, + 1.440995, + 1.469556 + ] + }, + "luapyre": { + "median_ms": 2.215801, + "samples_ms": [ + 2.180581, + 2.204275, + 2.194561, + 2.186439, + 2.182934, + 2.216293, + 2.184396, + 2.202753, + 2.435224, + 2.481321, + 2.243903, + 2.226778, + 2.43228, + 2.273196, + 2.235491, + 2.228471, + 2.206338, + 2.183514, + 2.1876, + 2.211796, + 2.1873, + 2.22111, + 2.178347, + 2.400573, + 2.317652, + 2.185707, + 2.220639, + 2.22098, + 2.215801, + 2.225536, + 2.183405 + ] + }, + "native_lua55": { + "median_ms": 0.262846, + "samples_ms": [ + 0.30647, + 0.277347, + 0.284457, + 0.260983, + 0.260803, + 0.260752, + 0.27269, + 0.261003, + 0.260702, + 0.260532, + 0.277948, + 0.26568, + 0.26575, + 0.280081, + 0.262846, + 0.262705, + 0.276205, + 0.276776, + 0.261954, + 0.262395, + 0.261714, + 0.276737, + 0.265971, + 0.263447, + 0.274934, + 0.262836, + 0.262756, + 0.262686, + 0.274683, + 0.262526, + 0.262455 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "python_lower_bound": { + "median_ms": 0.791271, + "samples_ms": [ + 0.768458, + 0.870828, + 0.88636, + 0.811501, + 0.791662, + 0.799063, + 0.815727, + 0.791271, + 0.784561, + 0.772854, + 0.802398, + 0.80444, + 0.792513, + 0.794826, + 0.761077, + 0.784983, + 0.8035, + 0.779734, + 0.778552, + 0.760756, + 0.77699, + 0.800315, + 1.059035, + 0.77013, + 0.83162, + 0.774016, + 0.772193, + 0.768949, + 0.833864, + 0.755238, + 0.766535 + ] + }, + "luapyre": { + "median_ms": 29.979179, + "samples_ms": [ + 30.06874, + 30.043993, + 29.75522, + 30.011627, + 29.8169, + 30.444151, + 30.235935, + 29.547955, + 29.534315, + 30.214283, + 29.75539, + 32.522241, + 30.183278, + 30.219762, + 29.630257, + 30.694918, + 29.729211, + 29.698076, + 29.703093, + 29.979179, + 29.68058, + 29.55778, + 30.084964, + 29.655703, + 29.599261, + 30.686737, + 29.832453, + 29.976715, + 32.920795, + 30.485811, + 30.095038 + ] + }, + "native_lua55": { + "median_ms": 0.316094, + "samples_ms": [ + 0.328913, + 0.316094, + 0.315043, + 0.328883, + 0.315193, + 0.315964, + 0.377985, + 0.314451, + 0.316605, + 0.328372, + 0.313831, + 0.323885, + 0.334541, + 0.314752, + 0.313961, + 0.343374, + 0.31373, + 0.315292, + 0.343504, + 0.316265, + 0.315132, + 0.315392, + 0.328932, + 0.318307, + 0.314813, + 0.330235, + 0.31356, + 0.31324, + 0.327941, + 0.314661, + 0.329744 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "python_lower_bound": { + "median_ms": 2.106292, + "samples_ms": [ + 2.288809, + 2.102505, + 2.151568, + 2.097238, + 2.076267, + 2.257984, + 2.107874, + 2.076007, + 2.089527, + 2.167531, + 2.246907, + 2.138068, + 2.100863, + 2.086242, + 2.092842, + 2.151658, + 2.10503, + 2.072361, + 2.095084, + 2.167801, + 2.13236, + 2.166589, + 2.070969, + 2.103878, + 2.11952, + 2.179839, + 2.106292, + 2.118429, + 2.10522, + 2.108515, + 2.095466 + ] + }, + "luapyre": { + "median_ms": 102.173583, + "samples_ms": [ + 101.056624, + 100.110005, + 99.241279, + 99.26839, + 98.940048, + 108.989246, + 100.254757, + 100.059019, + 100.841268, + 104.656316, + 103.527811, + 100.394963, + 103.105321, + 102.173583, + 103.429956, + 102.212791, + 102.349862, + 100.622817, + 102.627449, + 105.997805, + 103.20035, + 101.244449, + 104.435553, + 102.296483, + 103.857915, + 100.993181, + 100.045049, + 99.789284, + 101.137312, + 107.614408, + 104.582648 + ] + }, + "native_lua55": { + "median_ms": 1.750539, + "samples_ms": [ + 1.728857, + 1.830767, + 1.757249, + 1.727336, + 1.736769, + 1.71699, + 1.731862, + 1.761265, + 1.734566, + 1.721736, + 1.727345, + 1.774874, + 1.710731, + 1.733664, + 1.964393, + 1.865678, + 1.795425, + 1.744379, + 1.748066, + 1.738351, + 1.764089, + 1.980616, + 1.842244, + 1.808174, + 1.763058, + 1.767574, + 1.750539, + 1.741796, + 1.756688, + 1.772341, + 1.740464 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "python_lower_bound": { + "median_ms": 0.582765, + "samples_ms": [ + 0.570046, + 0.582815, + 0.765133, + 0.580102, + 0.644887, + 0.873692, + 0.572079, + 0.587713, + 0.569286, + 0.57921, + 0.579501, + 0.566462, + 0.766755, + 0.607692, + 0.568925, + 0.588875, + 0.57195, + 0.641651, + 0.649103, + 0.569595, + 0.638527, + 0.569426, + 0.590026, + 0.594833, + 0.57242, + 0.60655, + 0.569175, + 0.580371, + 0.567072, + 0.588594, + 0.582765 + ] + }, + "luapyre": { + "median_ms": 22.902893, + "samples_ms": [ + 23.136365, + 23.458308, + 22.884696, + 22.985054, + 22.756538, + 22.558247, + 23.17387, + 22.729768, + 23.273407, + 23.15343, + 22.902893, + 22.844227, + 22.473853, + 22.899017, + 24.303768, + 23.24812, + 23.198727, + 25.066208, + 23.082106, + 23.669699, + 25.050525, + 22.920048, + 22.88698, + 22.769327, + 22.752723, + 22.73051, + 22.71665, + 22.354878, + 22.896604, + 22.93484, + 22.859349 + ] + }, + "native_lua55": { + "median_ms": 0.196959, + "samples_ms": [ + 0.194856, + 0.196959, + 0.194215, + 0.23844, + 0.21777, + 0.197179, + 0.195146, + 0.2381, + 0.210099, + 0.195808, + 0.193334, + 0.238651, + 0.194115, + 0.208726, + 0.194836, + 0.239792, + 0.193944, + 0.253552, + 0.235906, + 0.241565, + 0.194636, + 0.196128, + 0.242857, + 0.243858, + 0.196829, + 0.19743, + 0.194296, + 0.34061, + 0.194485, + 0.196679, + 0.194025 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "python_lower_bound": { + "median_ms": 3.331429, + "samples_ms": [ + 3.399028, + 3.317419, + 3.592252, + 3.419709, + 3.353522, + 3.331429, + 3.317188, + 3.321836, + 3.315616, + 3.33163, + 3.313884, + 3.325731, + 3.322747, + 3.30478, + 3.320003, + 3.603299, + 3.499536, + 3.342696, + 3.452798, + 3.376756, + 3.31116, + 3.302638, + 3.319151, + 3.343386, + 3.309818, + 3.425977, + 3.295196, + 3.334494, + 3.345821, + 3.354503, + 3.29797 + ] + }, + "luapyre": { + "median_ms": 30.426767, + "samples_ms": [ + 30.706586, + 31.00933, + 30.684464, + 56.844476, + 30.474144, + 30.462768, + 29.725326, + 31.917242, + 29.958855, + 30.426767, + 30.709602, + 29.891712, + 30.611027, + 29.952541, + 30.618709, + 29.852374, + 30.539583, + 30.967691, + 30.194566, + 29.869218, + 30.389662, + 29.680752, + 30.56471, + 29.702704, + 30.861054, + 30.279111, + 30.409772, + 30.114799, + 29.7136, + 29.811914, + 33.366202 + ] + }, + "native_lua55": { + "median_ms": 0.584948, + "samples_ms": [ + 0.602435, + 0.621913, + 0.686768, + 0.583888, + 0.604538, + 0.582945, + 0.596225, + 0.596445, + 0.62673, + 0.612759, + 0.561975, + 0.574863, + 0.558149, + 0.573141, + 0.575655, + 0.609816, + 0.601934, + 0.556717, + 0.572721, + 0.566942, + 0.614813, + 0.580982, + 0.621943, + 0.584948, + 0.555796, + 0.634962, + 0.579021, + 0.560673, + 0.569105, + 0.627391, + 0.597658 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "python_lower_bound": { + "median_ms": 0.377584, + "samples_ms": [ + 0.377083, + 0.389922, + 0.377584, + 0.375291, + 0.385886, + 0.376503, + 0.385566, + 0.372967, + 0.390203, + 0.389652, + 0.376352, + 0.374219, + 0.389662, + 0.375752, + 0.393117, + 0.377214, + 0.378335, + 0.387889, + 0.376733, + 0.401119, + 0.372016, + 0.371706, + 0.389232, + 0.373248, + 0.375942, + 0.387509, + 0.374821, + 0.387018, + 0.373498, + 0.387119, + 0.3881 + ] + }, + "luapyre": { + "median_ms": 0.568755, + "samples_ms": [ + 0.563417, + 0.571589, + 0.724634, + 0.588294, + 0.540023, + 0.570237, + 0.567022, + 0.556226, + 0.548265, + 0.70853, + 0.556878, + 0.568755, + 0.530228, + 0.581664, + 0.559772, + 0.595033, + 0.531481, + 0.566612, + 0.541465, + 0.577788, + 0.76953, + 0.643364, + 0.575956, + 0.578879, + 0.533463, + 0.635473, + 0.555605, + 0.576386, + 0.528706, + 0.57182, + 0.548926 + ] + }, + "native_lua55": { + "median_ms": 1.159382, + "samples_ms": [ + 1.154485, + 1.171259, + 1.162236, + 1.487503, + 1.262323, + 1.186362, + 1.174214, + 1.192401, + 1.15802, + 1.152732, + 1.151861, + 1.161054, + 1.155967, + 1.170508, + 1.169968, + 1.152422, + 1.159382, + 1.150499, + 1.174704, + 1.165701, + 1.147925, + 1.162366, + 1.150539, + 1.143499, + 1.142397, + 1.157099, + 1.163869, + 1.143459, + 1.163979, + 1.145762, + 1.15146 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "python_lower_bound": { + "median_ms": 2.004413, + "samples_ms": [ + 2.037781, + 2.011863, + 2.046504, + 2.024432, + 2.000457, + 2.027766, + 2.005114, + 2.034887, + 2.011714, + 2.123938, + 2.011974, + 2.261821, + 2.004413, + 2.19336, + 1.987257, + 2.06383, + 1.977603, + 2.006265, + 1.95483, + 2.076929, + 1.981779, + 2.000227, + 1.979926, + 1.969732, + 1.969742, + 1.987938, + 1.979005, + 1.987438, + 1.951926, + 1.993556, + 1.967999 + ] + }, + "luapyre": { + "median_ms": 28.717348, + "samples_ms": [ + 29.504083, + 28.867839, + 28.601608, + 28.407143, + 28.782704, + 28.764477, + 28.717348, + 28.339092, + 35.286491, + 30.480356, + 28.941917, + 28.50038, + 28.259446, + 28.779078, + 29.470934, + 28.891774, + 28.789814, + 28.834049, + 28.605003, + 28.346834, + 28.631182, + 28.24899, + 29.212996, + 28.20808, + 28.352712, + 29.050727, + 28.435653, + 28.371189, + 28.269791, + 28.900236, + 28.172718 + ] + }, + "native_lua55": { + "median_ms": 0.799073, + "samples_ms": [ + 0.820704, + 0.788137, + 0.802078, + 0.811521, + 0.800285, + 0.791973, + 0.780276, + 0.80365, + 0.811321, + 0.802137, + 0.795447, + 0.780145, + 0.796719, + 0.821005, + 0.796489, + 0.802608, + 0.782549, + 0.799433, + 0.81052, + 0.800265, + 0.800895, + 0.778703, + 0.79683, + 0.812943, + 0.79036, + 0.799073, + 0.782028, + 0.791692, + 0.826593, + 0.794516, + 0.796089 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "python_lower_bound": { + "median_ms": 0.050794, + "samples_ms": [ + 0.051526, + 0.051015, + 0.071715, + 0.051005, + 0.050825, + 0.050424, + 0.050644, + 0.050374, + 0.050564, + 0.050605, + 0.050414, + 0.050794, + 0.050484, + 0.050574, + 0.051226, + 0.050744, + 0.050705, + 0.073107, + 0.051185, + 0.051045, + 0.057063, + 0.152774, + 0.051415, + 0.051115, + 0.050995, + 0.050675, + 0.050594, + 0.050525, + 0.050534, + 0.050765, + 0.050844 + ] + }, + "luapyre": { + "median_ms": 0.675482, + "samples_ms": [ + 0.676964, + 0.675392, + 0.66079, + 0.677906, + 0.677735, + 0.680449, + 0.677465, + 0.705105, + 0.667921, + 0.697234, + 0.67411, + 0.677665, + 0.671105, + 0.678646, + 0.663464, + 0.671095, + 0.678766, + 0.696342, + 0.663734, + 0.676753, + 0.672718, + 0.659789, + 0.674841, + 0.691606, + 0.664966, + 0.685396, + 0.673489, + 0.665046, + 0.675482, + 0.693028, + 0.664806 + ] + }, + "native_lua55": { + "median_ms": 0.199583, + "samples_ms": [ + 0.200514, + 0.201836, + 0.202186, + 0.213814, + 0.199303, + 0.200284, + 0.200734, + 0.214996, + 0.2113, + 0.19804, + 0.1979, + 0.199543, + 0.198241, + 0.209698, + 0.196459, + 0.19783, + 0.201086, + 0.200354, + 0.428399, + 0.19747, + 0.19743, + 0.199583, + 0.221986, + 0.196659, + 0.196759, + 0.195797, + 0.267373, + 0.226172, + 0.198031, + 0.197741, + 0.196749 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + } + ], + "workloads": { + "typed_arith": { + "luapyre": { + "median_ms": 0.993698, + "process_medians_ms": [ + 1.01464, + 0.95367, + 0.993698 + ] + }, + "native_lua55": { + "median_ms": 0.126186, + "process_medians_ms": [ + 0.127818, + 0.126186, + 0.126105 + ] + }, + "python_lower_bound": { + "median_ms": 0.726406, + "process_medians_ms": [ + 0.731033, + 0.720988, + 0.726406 + ] + }, + "luapyre_over_native": 7.874867259442411, + "luapyre_over_python_lower_bound": 1.3679650223153443, + "native_over_python_lower_bound": 0.17371277219626488 + }, + "typed_branch": { + "luapyre": { + "median_ms": 2.215192, + "process_medians_ms": [ + 2.215192, + 2.205587, + 2.215801 + ] + }, + "native_lua55": { + "median_ms": 0.262846, + "process_medians_ms": [ + 0.262975, + 0.262045, + 0.262846 + ] + }, + "python_lower_bound": { + "median_ms": 1.459672, + "process_medians_ms": [ + 1.5534, + 1.454455, + 1.459672 + ] + }, + "luapyre_over_native": 8.427718131529488, + "luapyre_over_python_lower_bound": 1.5175957338360946, + "native_over_python_lower_bound": 0.18007196137214387 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 30.326368, + "process_medians_ms": [ + 31.395689, + 30.326368, + 29.979179 + ] + }, + "native_lua55": { + "median_ms": 0.317146, + "process_medians_ms": [ + 0.326259, + 0.317146, + 0.316094 + ] + }, + "python_lower_bound": { + "median_ms": 0.791271, + "process_medians_ms": [ + 0.761137, + 0.795076, + 0.791271 + ] + }, + "luapyre_over_native": 95.62273527019101, + "luapyre_over_python_lower_bound": 38.32614616231354, + "native_over_python_lower_bound": 0.40080579220014384 + }, + "binary_trees": { + "luapyre": { + "median_ms": 102.173583, + "process_medians_ms": [ + 103.168043, + 101.757409, + 102.173583 + ] + }, + "native_lua55": { + "median_ms": 1.750539, + "process_medians_ms": [ + 1.763648, + 1.747605, + 1.750539 + ] + }, + "python_lower_bound": { + "median_ms": 2.106292, + "process_medians_ms": [ + 2.155493, + 2.049568, + 2.106292 + ] + }, + "luapyre_over_native": 58.36692755774078, + "luapyre_over_python_lower_bound": 48.508745700975936, + "native_over_python_lower_bound": 0.8310998664952439 + }, + "sieve": { + "luapyre": { + "median_ms": 23.015227, + "process_medians_ms": [ + 23.015227, + 24.661486, + 22.902893 + ] + }, + "native_lua55": { + "median_ms": 0.196959, + "process_medians_ms": [ + 0.192603, + 0.19782, + 0.196959 + ] + }, + "python_lower_bound": { + "median_ms": 0.586331, + "process_medians_ms": [ + 0.586331, + 0.612139, + 0.582765 + ] + }, + "luapyre_over_native": 116.85288308734305, + "luapyre_over_python_lower_bound": 39.25295950580815, + "native_over_python_lower_bound": 0.33591776658576805 + }, + "table_mix": { + "luapyre": { + "median_ms": 29.970898, + "process_medians_ms": [ + 29.627071, + 29.970898, + 30.426767 + ] + }, + "native_lua55": { + "median_ms": 0.586611, + "process_medians_ms": [ + 0.586611, + 0.606891, + 0.584948 + ] + }, + "python_lower_bound": { + "median_ms": 3.331429, + "process_medians_ms": [ + 3.370947, + 3.306803, + 3.331429 + ] + }, + "luapyre_over_native": 51.09160585123702, + "luapyre_over_python_lower_bound": 8.9964090484894, + "native_over_python_lower_bound": 0.17608389673020197 + }, + "string_build": { + "luapyre": { + "median_ms": 0.548996, + "process_medians_ms": [ + 0.53158, + 0.548996, + 0.568755 + ] + }, + "native_lua55": { + "median_ms": 1.171349, + "process_medians_ms": [ + 1.171349, + 1.186922, + 1.159382 + ] + }, + "python_lower_bound": { + "median_ms": 0.380628, + "process_medians_ms": [ + 0.47691, + 0.380628, + 0.377584 + ] + }, + "luapyre_over_native": 0.46868695837022106, + "luapyre_over_python_lower_bound": 1.4423426547705371, + "native_over_python_lower_bound": 3.0774115409270992 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 28.717348, + "process_medians_ms": [ + 28.78119, + 28.674444, + 28.717348 + ] + }, + "native_lua55": { + "median_ms": 0.832051, + "process_medians_ms": [ + 0.842537, + 0.832051, + 0.799073 + ] + }, + "python_lower_bound": { + "median_ms": 2.004413, + "process_medians_ms": [ + 2.060364, + 1.962039, + 2.004413 + ] + }, + "luapyre_over_native": 34.51392763183988, + "luapyre_over_python_lower_bound": 14.327061339155154, + "native_over_python_lower_bound": 0.41510956075419586 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.675482, + "process_medians_ms": [ + 0.68139, + 0.66077, + 0.675482 + ] + }, + "native_lua55": { + "median_ms": 0.199583, + "process_medians_ms": [ + 0.199083, + 0.205231, + 0.199583 + ] + }, + "python_lower_bound": { + "median_ms": 0.050794, + "process_medians_ms": [ + 0.050153, + 0.052327, + 0.050794 + ] + }, + "luapyre_over_native": 3.3844666128878713, + "luapyre_over_python_lower_bound": 13.29846044808442, + "native_over_python_lower_bound": 3.929263298814821 + } + } +} diff --git a/benchmarks/results/native_headroom_037_314.json b/benchmarks/results/native_headroom_037_314.json new file mode 100644 index 0000000..d3a9085 --- /dev/null +++ b/benchmarks/results/native_headroom_037_314.json @@ -0,0 +1,3397 @@ +{ + "schema_version": 1, + "revision": "3312f76", + "python": "3.14.7", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "Three isolated processes; Latin-square implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "typed_arith": { + "luapyre": { + "median_ms": 0.72231, + "samples_ms": [ + 0.717354, + 0.702641, + 0.719786, + 0.77636, + 0.727057, + 0.706157, + 0.720668, + 0.726907, + 0.743391, + 0.731684, + 0.719987, + 0.72231, + 0.707889, + 0.73574, + 0.778723, + 0.723632, + 0.706087, + 0.721079, + 0.729772, + 0.737773, + 0.734758, + 0.739295, + 0.721369, + 0.705435, + 0.718705, + 0.768308, + 0.717163, + 0.702612, + 0.712616, + 0.779053, + 0.773816 + ] + }, + "native_lua55": { + "median_ms": 0.126085, + "samples_ms": [ + 0.136851, + 0.126316, + 0.122099, + 0.126075, + 0.126075, + 0.122059, + 0.126085, + 0.107147, + 0.134237, + 0.126286, + 0.126165, + 0.126146, + 0.126075, + 0.126105, + 0.122811, + 0.122169, + 0.126095, + 0.121248, + 0.126135, + 0.112985, + 0.130131, + 0.126085, + 0.147577, + 0.124002, + 0.155298, + 0.122259, + 0.126165, + 0.112415, + 0.128739, + 0.107147, + 0.112946 + ] + }, + "python_lower_bound": { + "median_ms": 0.483029, + "samples_ms": [ + 0.480636, + 0.492222, + 0.479844, + 0.492794, + 0.479705, + 0.491141, + 0.479133, + 0.541175, + 0.478833, + 0.496659, + 0.479864, + 0.492003, + 0.478643, + 0.493565, + 0.479424, + 0.53759, + 0.479313, + 0.511721, + 0.477821, + 0.492803, + 0.482959, + 0.494296, + 0.483029, + 0.543468, + 0.478002, + 0.672798, + 0.479163, + 0.495056, + 0.480496, + 0.497701, + 0.482569 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "luapyre": { + "median_ms": 1.856225, + "samples_ms": [ + 1.85946, + 1.831509, + 1.844057, + 2.085011, + 1.952777, + 1.894581, + 1.843867, + 1.841493, + 1.857937, + 2.010612, + 1.85951, + 1.838509, + 1.853461, + 1.92136, + 1.915752, + 1.840141, + 1.896814, + 1.875503, + 1.851879, + 1.894531, + 1.864106, + 1.845429, + 1.815775, + 1.856225, + 1.830517, + 1.835745, + 1.861693, + 1.842875, + 1.815015, + 1.827523, + 1.917114 + ] + }, + "native_lua55": { + "median_ms": 0.337035, + "samples_ms": [ + 0.335924, + 0.336615, + 0.350424, + 0.334581, + 0.334451, + 0.346058, + 0.336794, + 0.334892, + 0.34747, + 0.334762, + 0.348341, + 0.349302, + 0.336514, + 0.3337, + 0.346068, + 0.337345, + 0.337035, + 0.34744, + 0.334331, + 0.334161, + 0.34713, + 0.335313, + 0.356043, + 0.349854, + 0.335363, + 0.338878, + 0.353128, + 0.334792, + 0.432976, + 0.338027, + 0.334311 + ] + }, + "python_lower_bound": { + "median_ms": 1.142818, + "samples_ms": [ + 1.141787, + 1.119033, + 1.122668, + 1.30747, + 1.153894, + 1.202225, + 1.124791, + 1.134165, + 1.127666, + 1.391713, + 1.152051, + 1.17144, + 1.140915, + 1.121386, + 1.142818, + 1.138191, + 1.122819, + 1.156147, + 1.173713, + 1.143508, + 1.121577, + 1.149017, + 1.12418, + 1.144299, + 1.400516, + 1.142498, + 1.196107, + 1.124471, + 1.134726, + 1.149518, + 1.189577 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "luapyre": { + "median_ms": 24.045179, + "samples_ms": [ + 23.751147, + 23.808681, + 23.804695, + 24.143173, + 24.171624, + 23.840939, + 23.842622, + 24.086941, + 24.130925, + 24.225454, + 23.616249, + 24.042084, + 24.201488, + 23.615067, + 23.653864, + 24.045179, + 24.142011, + 23.673113, + 24.646441, + 24.120169, + 26.045195, + 24.010799, + 24.132257, + 24.14127, + 24.21663, + 24.845764, + 23.886526, + 23.862901, + 24.070406, + 23.926665, + 23.782293 + ] + }, + "native_lua55": { + "median_ms": 0.324176, + "samples_ms": [ + 0.323845, + 0.335343, + 0.322243, + 0.321822, + 0.335393, + 0.324176, + 0.357945, + 0.334952, + 0.322563, + 0.339538, + 0.337216, + 0.322323, + 0.323465, + 0.337966, + 0.322853, + 0.337375, + 0.335162, + 0.322533, + 0.323616, + 0.336404, + 0.322714, + 0.325918, + 0.351727, + 0.322784, + 0.32689, + 0.335843, + 0.322774, + 0.322404, + 0.336935, + 0.321963, + 0.320361 + ] + }, + "python_lower_bound": { + "median_ms": 0.619399, + "samples_ms": [ + 0.640571, + 0.604337, + 0.620271, + 0.610576, + 0.619399, + 0.615824, + 0.619319, + 0.62018, + 0.618979, + 0.607842, + 0.616705, + 0.610085, + 0.622644, + 0.633941, + 0.60622, + 0.617927, + 0.80415, + 0.606971, + 0.621102, + 0.634261, + 0.608453, + 0.671566, + 0.605579, + 0.630035, + 0.885319, + 0.682762, + 0.605388, + 0.676603, + 0.711735, + 0.607852, + 0.619739 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "luapyre": { + "median_ms": 85.286983, + "samples_ms": [ + 85.660662, + 87.776227, + 84.350699, + 84.997278, + 84.412969, + 84.479578, + 84.969407, + 84.844513, + 86.34062, + 86.249716, + 84.646062, + 86.832281, + 85.599111, + 85.44129, + 85.800087, + 84.573736, + 86.496659, + 84.158337, + 84.661385, + 84.693883, + 87.21945, + 85.286983, + 84.110246, + 87.027288, + 87.354058, + 85.795559, + 85.401822, + 84.128112, + 85.22361, + 85.466666, + 84.389235 + ] + }, + "native_lua55": { + "median_ms": 1.758001, + "samples_ms": [ + 1.696651, + 1.729108, + 1.746835, + 1.726445, + 1.745863, + 1.777629, + 1.785862, + 1.739614, + 2.08456, + 1.834643, + 1.743489, + 1.790649, + 1.751801, + 1.795155, + 1.73703, + 1.763308, + 1.745212, + 1.758001, + 1.735919, + 1.75737, + 1.778681, + 1.88703, + 2.035518, + 1.930664, + 1.796127, + 1.753924, + 1.797629, + 1.825009, + 1.776167, + 1.746564, + 1.752623 + ] + }, + "python_lower_bound": { + "median_ms": 2.014708, + "samples_ms": [ + 2.015529, + 1.992895, + 2.003942, + 2.065653, + 2.07781, + 2.054196, + 2.148854, + 1.998704, + 2.133572, + 2.092922, + 2.003401, + 1.985574, + 2.018793, + 2.004403, + 2.011403, + 1.99575, + 2.0162, + 2.007727, + 2.012244, + 1.99569, + 2.014217, + 2.020686, + 2.328018, + 2.024432, + 2.014708, + 1.997703, + 2.00984, + 1.992535, + 2.143758, + 2.059113, + 2.071311 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "luapyre": { + "median_ms": 19.843341, + "samples_ms": [ + 20.317256, + 20.416342, + 19.825654, + 20.174727, + 19.854487, + 19.995083, + 20.196139, + 19.689755, + 19.620122, + 19.917059, + 20.005669, + 20.012709, + 19.972931, + 19.513836, + 19.763463, + 19.99946, + 20.030585, + 19.77534, + 19.477023, + 19.704837, + 19.675584, + 19.814027, + 19.415252, + 19.523622, + 19.735953, + 19.843341, + 19.960617, + 19.560823, + 19.552851, + 20.130882, + 19.861665 + ] + }, + "native_lua55": { + "median_ms": 0.193936, + "samples_ms": [ + 0.193895, + 0.211561, + 0.193094, + 0.194346, + 0.194115, + 0.2178, + 0.209979, + 0.19707, + 0.194896, + 0.193936, + 0.200275, + 0.219723, + 0.195037, + 0.195157, + 0.192323, + 0.193414, + 0.20444, + 0.191261, + 0.192172, + 0.193324, + 0.192063, + 0.207165, + 0.191983, + 0.191622, + 0.191442, + 0.208707, + 0.204851, + 0.192853, + 0.193034, + 0.193084, + 0.191812 + ] + }, + "python_lower_bound": { + "median_ms": 0.508668, + "samples_ms": [ + 0.514446, + 0.499815, + 0.510281, + 0.500897, + 0.51047, + 0.500336, + 0.578431, + 0.502579, + 0.512534, + 0.498934, + 0.510941, + 0.500406, + 0.518563, + 0.49676, + 0.717175, + 0.506405, + 0.529569, + 0.588586, + 0.498634, + 0.50978, + 0.497041, + 0.533835, + 0.498153, + 0.521467, + 0.49685, + 0.511242, + 0.498834, + 0.508668, + 0.499825, + 0.529529, + 0.500606 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "luapyre": { + "median_ms": 21.879974, + "samples_ms": [ + 22.555197, + 22.686451, + 23.015515, + 48.00393, + 23.990929, + 21.723253, + 21.757895, + 21.609326, + 21.595086, + 21.615594, + 21.616977, + 21.634693, + 21.719518, + 21.640481, + 21.887776, + 21.687331, + 21.786496, + 27.95496, + 28.048858, + 21.828328, + 22.227745, + 22.111214, + 22.801349, + 21.802569, + 22.489601, + 22.280954, + 21.85023, + 21.879974, + 21.664828, + 22.359159, + 22.769822 + ] + }, + "native_lua55": { + "median_ms": 0.589988, + "samples_ms": [ + 0.589988, + 0.59207, + 0.587484, + 0.573363, + 0.58481, + 0.574135, + 0.58447, + 0.600433, + 0.571551, + 0.587304, + 0.761991, + 0.602015, + 0.585602, + 0.615656, + 0.732337, + 0.608405, + 0.573284, + 0.585821, + 0.574756, + 0.588576, + 0.649536, + 0.574925, + 0.601795, + 0.624088, + 0.772266, + 0.60489, + 0.599962, + 0.640843, + 0.604319, + 0.573464, + 0.586583 + ] + }, + "python_lower_bound": { + "median_ms": 2.529781, + "samples_ms": [ + 2.493968, + 2.490714, + 2.498275, + 2.730216, + 2.49472, + 2.485816, + 2.507508, + 2.50158, + 2.470745, + 2.503412, + 2.550562, + 2.558413, + 2.554788, + 2.759259, + 2.806247, + 2.765988, + 2.600425, + 2.513297, + 2.503122, + 2.529781, + 2.506137, + 2.533106, + 2.480348, + 2.949679, + 2.520798, + 2.908919, + 2.860257, + 2.775893, + 2.638321, + 2.496432, + 2.770285 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "luapyre": { + "median_ms": 0.524421, + "samples_ms": [ + 0.538752, + 0.510251, + 0.514527, + 0.531512, + 0.514337, + 0.50278, + 0.515569, + 0.497582, + 0.506956, + 0.700861, + 0.535057, + 0.558231, + 0.502719, + 0.515939, + 0.49574, + 0.506575, + 0.494828, + 0.510131, + 1.527607, + 0.565382, + 0.779096, + 0.522268, + 0.699889, + 0.586743, + 0.538813, + 0.685368, + 0.524421, + 2.548799, + 0.612591, + 0.54399, + 0.511482 + ] + }, + "native_lua55": { + "median_ms": 1.316476, + "samples_ms": [ + 1.27796, + 1.183531, + 1.189159, + 1.245892, + 1.338289, + 1.538483, + 1.514558, + 1.433649, + 1.18917, + 1.42798, + 1.580745, + 1.185044, + 1.203631, + 1.384507, + 1.190772, + 1.204932, + 1.235678, + 1.190842, + 1.316476, + 1.183581, + 1.375083, + 1.28488, + 1.180165, + 1.170913, + 1.465596, + 1.33219, + 1.632121, + 1.535819, + 1.4204, + 1.432377, + 1.351498 + ] + }, + "python_lower_bound": { + "median_ms": 0.431464, + "samples_ms": [ + 0.386117, + 0.449301, + 0.365097, + 0.449662, + 0.366939, + 0.416442, + 0.393919, + 0.457974, + 0.467978, + 0.367651, + 0.494247, + 0.413989, + 0.364466, + 0.453538, + 0.367711, + 1.038588, + 0.519594, + 0.369733, + 0.48289, + 0.417675, + 0.363815, + 0.451193, + 0.368632, + 0.440157, + 0.431464, + 0.493666, + 0.681452, + 0.444424, + 0.3671, + 0.38068, + 1.601355 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 24.02563, + "samples_ms": [ + 24.751087, + 23.535349, + 23.464856, + 23.711387, + 23.735614, + 25.32427, + 23.800628, + 23.456573, + 24.797376, + 24.063616, + 23.967004, + 23.958301, + 24.02563, + 25.414612, + 28.695379, + 23.873296, + 23.721122, + 23.863602, + 24.321316, + 23.702434, + 23.633344, + 25.728285, + 24.086429, + 24.66494, + 23.463684, + 24.52218, + 23.307053, + 25.26957, + 24.227468, + 24.635176, + 24.594807 + ] + }, + "native_lua55": { + "median_ms": 0.846205, + "samples_ms": [ + 0.849519, + 0.858402, + 1.100088, + 0.846205, + 0.831113, + 0.995645, + 0.840797, + 0.833806, + 0.825354, + 0.905242, + 0.816341, + 0.88394, + 0.845444, + 0.831082, + 0.845494, + 0.831143, + 0.810903, + 0.828889, + 0.822149, + 0.996616, + 0.941155, + 0.893294, + 0.85632, + 0.95201, + 0.993922, + 0.844562, + 0.864241, + 0.827357, + 1.342985, + 0.785055, + 0.886454 + ] + }, + "python_lower_bound": { + "median_ms": 2.431947, + "samples_ms": [ + 2.413831, + 2.350828, + 2.810214, + 2.340172, + 2.34543, + 2.554598, + 2.805557, + 2.613474, + 2.362655, + 2.402274, + 2.403716, + 2.448622, + 2.785768, + 2.438066, + 2.500919, + 2.486057, + 2.477014, + 2.431947, + 2.380632, + 2.35907, + 2.520277, + 2.5425, + 2.396976, + 2.352881, + 2.333883, + 2.332541, + 2.333542, + 2.496723, + 2.349486, + 2.593655, + 2.614966 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.600834, + "samples_ms": [ + 0.599773, + 0.808509, + 0.730815, + 0.588456, + 0.600514, + 0.731235, + 0.585891, + 0.666991, + 0.664487, + 0.610458, + 0.600534, + 0.589306, + 0.600022, + 0.767439, + 0.632139, + 0.608004, + 0.585141, + 0.600834, + 0.602316, + 0.587544, + 0.615225, + 0.587895, + 0.603277, + 0.660722, + 0.586583, + 0.613723, + 0.616366, + 0.583218, + 0.594835, + 0.583919, + 0.59769 + ] + }, + "native_lua55": { + "median_ms": 0.19675, + "samples_ms": [ + 0.191092, + 0.230019, + 0.194076, + 0.192544, + 0.215998, + 0.194056, + 0.194276, + 0.195127, + 0.193926, + 0.207926, + 0.196459, + 0.196329, + 0.197781, + 0.225792, + 0.282255, + 0.193795, + 0.19717, + 0.213454, + 0.221696, + 0.19669, + 0.19761, + 0.193905, + 0.192553, + 0.208617, + 0.196829, + 0.19675, + 0.197611, + 0.196779, + 0.463713, + 0.195858, + 0.195357 + ] + }, + "python_lower_bound": { + "median_ms": 0.04067, + "samples_ms": [ + 0.04067, + 0.040569, + 0.040509, + 0.040469, + 0.04055, + 0.093618, + 0.041261, + 0.04044, + 0.04072, + 0.04072, + 0.040439, + 0.0405, + 0.057034, + 0.040901, + 0.04065, + 0.04062, + 0.065867, + 0.04083, + 0.0404, + 0.0407, + 0.04064, + 0.04063, + 0.04057, + 0.04062, + 0.04061, + 0.04074, + 0.04101, + 0.0408, + 0.04069, + 0.04076, + 0.04102 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "typed_arith": { + "native_lua55": { + "median_ms": 0.126115, + "samples_ms": [ + 0.12251, + 0.127538, + 0.112135, + 0.112606, + 0.112616, + 0.152274, + 0.184622, + 0.107528, + 0.107227, + 0.126085, + 0.111363, + 0.12216, + 0.126115, + 0.108139, + 0.153966, + 0.126516, + 0.329836, + 0.127207, + 0.121118, + 0.130151, + 0.154407, + 0.113097, + 0.331548, + 0.12181, + 0.126396, + 0.126096, + 0.15678, + 0.112165, + 0.176359, + 0.126446, + 0.126115 + ] + }, + "python_lower_bound": { + "median_ms": 0.504312, + "samples_ms": [ + 0.477562, + 0.511983, + 0.478944, + 0.48973, + 0.479916, + 0.551911, + 0.482199, + 0.509439, + 0.480967, + 0.508297, + 0.481338, + 1.335104, + 0.514266, + 0.478644, + 0.492254, + 0.555528, + 0.775671, + 1.117313, + 0.512204, + 0.561086, + 0.500836, + 0.535428, + 0.514477, + 0.482219, + 0.557109, + 0.478294, + 0.504312, + 0.479315, + 0.493135, + 0.569147, + 0.478414 + ] + }, + "luapyre": { + "median_ms": 0.74815, + "samples_ms": [ + 1.102521, + 0.899032, + 0.709814, + 0.882398, + 0.759357, + 0.715923, + 0.702243, + 1.21008, + 0.778946, + 0.798955, + 0.7956, + 0.735101, + 0.703364, + 0.758636, + 0.748611, + 0.716534, + 0.705027, + 0.716264, + 0.72029, + 0.793718, + 0.706519, + 0.715673, + 0.743654, + 0.721442, + 0.74815, + 0.755071, + 0.716474, + 0.949366, + 0.757324, + 0.780018, + 0.747269 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "native_lua55": { + "median_ms": 0.301464, + "samples_ms": [ + 0.292971, + 0.305549, + 0.292911, + 0.372268, + 0.320923, + 0.292821, + 0.483482, + 0.318087, + 0.293281, + 0.292791, + 0.305159, + 0.292621, + 0.345448, + 0.314382, + 0.29266, + 0.29245, + 0.309976, + 0.30562, + 0.294003, + 0.293842, + 0.355924, + 0.292831, + 0.301464, + 0.304928, + 0.300993, + 0.295234, + 0.304669, + 0.370255, + 0.293202, + 0.343395, + 0.288694 + ] + }, + "python_lower_bound": { + "median_ms": 1.143441, + "samples_ms": [ + 1.122381, + 1.118695, + 1.143441, + 1.177022, + 1.138084, + 1.276888, + 1.120999, + 1.115661, + 1.186535, + 1.14252, + 1.375433, + 1.144743, + 1.13553, + 1.401531, + 1.140919, + 1.272732, + 1.137313, + 1.321674, + 1.179745, + 1.154428, + 1.201697, + 1.184132, + 1.145415, + 1.124444, + 1.280784, + 1.150583, + 1.133998, + 1.12835, + 1.11474, + 1.133076, + 1.12207 + ] + }, + "luapyre": { + "median_ms": 1.945001, + "samples_ms": [ + 1.853607, + 2.051477, + 1.919133, + 2.439178, + 2.18888, + 2.113809, + 2.275437, + 2.071587, + 2.078016, + 2.363106, + 1.868429, + 1.949717, + 2.059399, + 1.866025, + 1.831635, + 1.830773, + 1.833947, + 2.759199, + 1.9241, + 1.926353, + 2.113158, + 1.928908, + 1.846877, + 1.962407, + 1.920125, + 1.846466, + 1.830333, + 1.938201, + 2.00577, + 1.945001, + 2.256129 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "native_lua55": { + "median_ms": 0.318198, + "samples_ms": [ + 0.314533, + 0.327471, + 0.315624, + 0.313462, + 0.314192, + 0.376764, + 0.330947, + 0.33295, + 0.318198, + 0.314443, + 0.330536, + 0.332028, + 0.314062, + 0.314372, + 0.324647, + 0.32661, + 0.316696, + 0.32626, + 0.313491, + 0.316465, + 0.324868, + 0.315754, + 0.328994, + 0.325278, + 0.315614, + 0.316605, + 0.333932, + 0.314162, + 0.314162, + 0.328032, + 0.501137 + ] + }, + "python_lower_bound": { + "median_ms": 0.619632, + "samples_ms": [ + 0.710735, + 0.650016, + 0.617829, + 0.608104, + 0.617378, + 0.608645, + 0.668693, + 0.647272, + 0.607794, + 0.616487, + 0.604349, + 0.619582, + 0.617138, + 0.619632, + 0.665489, + 0.66642, + 0.60463, + 0.620333, + 0.671948, + 0.604749, + 0.646241, + 0.607163, + 0.617378, + 0.620603, + 0.60495, + 0.784924, + 0.769782, + 0.610108, + 0.646741, + 0.640903, + 0.620313 + ] + }, + "luapyre": { + "median_ms": 24.569299, + "samples_ms": [ + 24.88208, + 27.444098, + 35.69089, + 31.931869, + 25.019051, + 24.514179, + 24.470034, + 24.357858, + 24.649908, + 25.318298, + 24.950148, + 24.769353, + 24.270631, + 24.336607, + 24.251443, + 26.792682, + 27.093674, + 24.661204, + 24.725048, + 24.569299, + 24.063317, + 23.985282, + 25.417364, + 24.518375, + 24.357829, + 24.3788, + 26.029307, + 23.906076, + 24.439398, + 24.084598, + 23.728736 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "native_lua55": { + "median_ms": 1.761134, + "samples_ms": [ + 1.799391, + 1.730039, + 1.726063, + 1.732903, + 1.696089, + 2.251385, + 1.796036, + 1.842164, + 1.892407, + 1.834622, + 1.73703, + 1.754695, + 1.73009, + 1.728177, + 1.773363, + 1.731031, + 1.772892, + 1.798159, + 1.778721, + 1.735967, + 1.7447, + 1.77145, + 1.747675, + 1.861252, + 1.734496, + 1.761134, + 1.751551, + 1.768566, + 1.73122, + 1.808154, + 1.781975 + ] + }, + "python_lower_bound": { + "median_ms": 1.952826, + "samples_ms": [ + 1.967097, + 1.930573, + 1.954238, + 1.915741, + 1.951234, + 1.936813, + 1.946426, + 1.947177, + 1.951434, + 1.939176, + 1.9553, + 1.940889, + 1.95532, + 1.952646, + 1.965125, + 2.116767, + 2.0302, + 1.940959, + 1.949091, + 1.963642, + 2.161342, + 2.040134, + 1.952826, + 2.130887, + 2.001638, + 1.937464, + 1.953738, + 1.945986, + 1.944564, + 1.96233, + 1.984713 + ] + }, + "luapyre": { + "median_ms": 86.546099, + "samples_ms": [ + 92.322407, + 88.220948, + 90.61364, + 93.039731, + 93.830931, + 93.868977, + 93.039129, + 90.700703, + 86.157029, + 86.664083, + 88.106059, + 87.815889, + 85.626345, + 85.307021, + 85.233724, + 85.681541, + 85.602314, + 85.388441, + 84.942837, + 85.04756, + 85.066609, + 86.812, + 86.454084, + 85.48963, + 107.394557, + 116.411292, + 86.546099, + 89.148729, + 85.237659, + 84.716284, + 84.687122 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "native_lua55": { + "median_ms": 0.193995, + "samples_ms": [ + 0.201495, + 0.200033, + 0.208756, + 0.197029, + 0.194996, + 0.193755, + 0.193424, + 0.207515, + 0.193034, + 0.193655, + 0.193935, + 0.194385, + 0.206714, + 0.19694, + 0.194385, + 0.193954, + 0.217529, + 0.209137, + 0.193544, + 0.193995, + 0.192142, + 0.193304, + 0.206863, + 0.194025, + 0.192483, + 0.193234, + 0.193063, + 0.206422, + 0.193204, + 0.193854, + 0.193083 + ] + }, + "python_lower_bound": { + "median_ms": 0.512172, + "samples_ms": [ + 0.495157, + 0.514775, + 0.522557, + 0.5179, + 0.512172, + 0.497249, + 0.506804, + 0.49746, + 0.506904, + 0.495307, + 0.528546, + 0.499904, + 0.513613, + 0.502658, + 0.520234, + 0.512813, + 0.518641, + 0.498321, + 0.528526, + 0.495297, + 0.609565, + 0.499133, + 0.523017, + 0.496629, + 0.509678, + 0.49748, + 0.526062, + 0.558169, + 0.589165, + 0.498472, + 0.527664 + ] + }, + "luapyre": { + "median_ms": 19.68132, + "samples_ms": [ + 19.196795, + 19.069086, + 19.747232, + 19.741584, + 19.986723, + 20.03397, + 19.403983, + 20.406587, + 19.549788, + 19.530449, + 19.487726, + 19.430252, + 19.68132, + 19.889918, + 20.557988, + 19.215085, + 19.845822, + 19.401589, + 21.520252, + 20.424443, + 19.375691, + 19.881034, + 19.988552, + 19.627362, + 19.10929, + 19.662313, + 19.360519, + 19.385126, + 20.112824, + 20.479653, + 19.772414 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "native_lua55": { + "median_ms": 0.57151, + "samples_ms": [ + 0.584489, + 0.589215, + 0.584669, + 0.585671, + 0.574694, + 0.586973, + 0.57151, + 0.664356, + 0.798463, + 0.547014, + 0.572311, + 0.542677, + 0.67423, + 0.629054, + 0.609345, + 0.570618, + 0.542126, + 0.562567, + 0.554765, + 0.541125, + 0.560504, + 0.543819, + 0.566072, + 0.543078, + 0.5516, + 0.540744, + 0.554104, + 0.555866, + 0.901804, + 0.581995, + 0.763381 + ] + }, + "python_lower_bound": { + "median_ms": 2.4987, + "samples_ms": [ + 2.489357, + 2.510488, + 2.47782, + 2.503808, + 2.579409, + 2.506482, + 2.493743, + 2.497108, + 2.485701, + 2.509426, + 2.476588, + 2.554322, + 2.514374, + 2.479121, + 2.512471, + 2.480214, + 2.489798, + 2.489617, + 2.480383, + 2.514073, + 2.515525, + 2.475537, + 2.483278, + 2.481896, + 2.4987, + 2.611777, + 2.495586, + 2.57359, + 2.574762, + 2.499121, + 2.562595 + ] + }, + "luapyre": { + "median_ms": 22.342781, + "samples_ms": [ + 22.525639, + 22.591415, + 22.222784, + 45.001183, + 22.148375, + 22.2407, + 22.594089, + 21.908061, + 21.701959, + 21.610334, + 21.834253, + 22.831097, + 22.754706, + 22.675919, + 22.075357, + 21.58045, + 22.348348, + 22.342781, + 24.484696, + 22.122657, + 22.521753, + 22.660487, + 22.385063, + 22.12528, + 22.246478, + 21.773033, + 22.980376, + 22.841884, + 21.561443, + 21.752914, + 22.934228 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "native_lua55": { + "median_ms": 1.166594, + "samples_ms": [ + 1.167315, + 1.163079, + 1.162778, + 1.175527, + 1.159784, + 1.1772, + 1.179132, + 1.168837, + 1.175146, + 1.177159, + 1.160605, + 1.172863, + 1.158392, + 1.180914, + 1.156679, + 1.158112, + 1.172593, + 1.178902, + 1.159863, + 1.166594, + 1.176078, + 1.15732, + 1.153424, + 1.191049, + 1.158292, + 1.155718, + 1.15084, + 1.169508, + 1.154576, + 1.168806, + 1.166484 + ] + }, + "python_lower_bound": { + "median_ms": 0.363284, + "samples_ms": [ + 0.373488, + 0.363284, + 0.389572, + 0.358246, + 0.357655, + 0.375642, + 0.359698, + 0.361111, + 0.372127, + 0.360149, + 0.363293, + 0.383373, + 0.366699, + 0.393578, + 0.361421, + 0.363184, + 0.378386, + 0.363764, + 0.359078, + 0.369893, + 0.359358, + 0.371626, + 0.358607, + 0.357695, + 0.384925, + 0.358616, + 0.358456, + 0.371605, + 0.357465, + 0.358337, + 0.373579 + ] + }, + "luapyre": { + "median_ms": 0.522989, + "samples_ms": [ + 0.556327, + 0.531541, + 0.51779, + 0.600402, + 0.521085, + 0.540954, + 0.518071, + 0.526804, + 0.503509, + 0.682853, + 0.552091, + 0.513484, + 0.524791, + 0.509769, + 0.524871, + 0.515567, + 0.518952, + 0.511781, + 0.539262, + 0.508747, + 0.522989, + 0.507366, + 0.519603, + 0.785484, + 0.529588, + 0.553003, + 0.512052, + 0.531752, + 0.514997, + 0.526173, + 0.507906 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "native_lua55": { + "median_ms": 0.829529, + "samples_ms": [ + 0.843539, + 0.817561, + 0.866803, + 0.833485, + 0.938488, + 0.834175, + 0.833815, + 0.82342, + 0.940141, + 0.831682, + 0.814066, + 0.854625, + 0.840484, + 0.814757, + 0.903056, + 0.825703, + 0.842208, + 0.829529, + 0.808798, + 0.901794, + 0.816799, + 0.803951, + 0.799875, + 0.812573, + 0.822228, + 0.81658, + 0.811091, + 0.792914, + 0.913162, + 0.816869, + 0.864701 + ] + }, + "python_lower_bound": { + "median_ms": 2.407256, + "samples_ms": [ + 2.410541, + 2.397061, + 2.618517, + 2.431051, + 2.381748, + 2.388518, + 2.564146, + 2.513803, + 2.410961, + 2.374147, + 2.526431, + 2.601672, + 2.360197, + 2.587821, + 2.423029, + 2.439443, + 2.618817, + 2.440335, + 2.618477, + 2.398934, + 2.350412, + 2.402289, + 2.361479, + 2.407256, + 2.367357, + 2.381358, + 2.381529, + 2.765021, + 2.397051, + 2.389861, + 2.391122 + ] + }, + "luapyre": { + "median_ms": 23.59514, + "samples_ms": [ + 24.33718, + 23.542983, + 24.147321, + 24.740733, + 23.070409, + 24.222471, + 24.025642, + 24.460061, + 23.592496, + 23.743929, + 24.213268, + 23.556683, + 23.497926, + 23.551585, + 24.246587, + 24.129164, + 23.69106, + 23.426181, + 24.088724, + 24.332162, + 23.499009, + 23.142555, + 23.59514, + 23.587698, + 23.394115, + 23.352373, + 23.749437, + 23.359273, + 23.143156, + 24.113742, + 23.482284 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "native_lua55": { + "median_ms": 0.196929, + "samples_ms": [ + 0.223348, + 0.195347, + 0.195266, + 0.195347, + 0.212031, + 0.206844, + 0.195777, + 0.195567, + 0.195637, + 0.196449, + 0.209878, + 0.195938, + 0.20428, + 0.244029, + 0.196929, + 0.217029, + 0.193103, + 0.193534, + 0.195347, + 0.194776, + 0.208276, + 0.195387, + 0.194916, + 0.196338, + 0.212592, + 0.212462, + 0.197951, + 0.196979, + 0.197761, + 0.19773, + 0.212291 + ] + }, + "python_lower_bound": { + "median_ms": 0.0406, + "samples_ms": [ + 0.04074, + 0.04038, + 0.04069, + 0.0407, + 0.04049, + 0.04065, + 0.040659, + 0.040709, + 0.040269, + 0.040509, + 0.04047, + 0.0404, + 0.051666, + 0.04069, + 0.04062, + 0.04071, + 0.04053, + 0.04072, + 0.0405, + 0.04052, + 0.040379, + 0.0406, + 0.04065, + 0.040449, + 0.040479, + 0.04051, + 0.04066, + 0.04075, + 0.04072, + 0.0406, + 0.040509 + ] + }, + "luapyre": { + "median_ms": 0.597848, + "samples_ms": [ + 0.59173, + 0.603025, + 0.585059, + 0.594794, + 0.609345, + 0.588705, + 0.602826, + 0.586641, + 0.59915, + 0.601193, + 0.584659, + 0.608544, + 0.585891, + 0.59238, + 0.598489, + 0.689262, + 0.609335, + 0.614142, + 0.587733, + 0.593793, + 0.583818, + 0.598769, + 0.597848, + 0.592701, + 0.610297, + 0.587633, + 0.601694, + 0.600132, + 0.580082, + 0.598269, + 0.588885 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "typed_arith": { + "python_lower_bound": { + "median_ms": 0.489238, + "samples_ms": [ + 0.480125, + 0.492042, + 0.479845, + 0.489238, + 0.479094, + 0.490281, + 0.493846, + 0.490711, + 0.478693, + 0.491732, + 0.478423, + 0.492984, + 0.478774, + 0.49, + 0.479575, + 0.515868, + 0.48304, + 0.492573, + 0.480025, + 0.493855, + 0.481097, + 0.492383, + 0.481147, + 0.50407, + 0.481257, + 0.493445, + 0.478222, + 0.490981, + 0.478163, + 0.580773, + 0.479224 + ] + }, + "luapyre": { + "median_ms": 0.719877, + "samples_ms": [ + 0.71465, + 0.733948, + 0.703083, + 0.719868, + 0.726998, + 0.709101, + 0.724123, + 0.738795, + 0.717143, + 0.70157, + 0.717074, + 0.717594, + 0.715972, + 0.719877, + 0.82395, + 0.743602, + 0.702321, + 0.715071, + 0.805964, + 0.783681, + 0.708732, + 0.731875, + 0.719296, + 0.74935, + 0.707739, + 0.717604, + 0.875907, + 0.924488, + 0.744284, + 0.754969, + 0.805864 + ] + }, + "native_lua55": { + "median_ms": 0.124082, + "samples_ms": [ + 0.12214, + 0.126236, + 0.121358, + 0.121859, + 0.126306, + 0.121088, + 0.112946, + 0.121137, + 0.130131, + 0.138694, + 0.121409, + 0.126406, + 0.107408, + 0.126235, + 0.122029, + 0.107137, + 0.126075, + 0.107177, + 0.126066, + 0.133676, + 0.126155, + 0.112976, + 0.113056, + 0.126195, + 0.126075, + 0.126085, + 0.112365, + 0.112716, + 0.127868, + 0.126296, + 0.124082 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "typed_branch": { + "python_lower_bound": { + "median_ms": 1.132423, + "samples_ms": [ + 1.120806, + 1.142228, + 1.120807, + 1.117842, + 1.119895, + 1.138643, + 1.263266, + 1.176989, + 1.148076, + 1.115469, + 1.118704, + 1.119304, + 1.211419, + 1.136479, + 1.119694, + 1.258599, + 1.132593, + 1.427507, + 1.141507, + 1.193904, + 1.11642, + 1.11603, + 1.138963, + 1.121958, + 1.126014, + 1.13671, + 1.119013, + 1.132423, + 1.118934, + 1.133144, + 1.118383 + ] + }, + "luapyre": { + "median_ms": 1.846312, + "samples_ms": [ + 1.840964, + 1.849927, + 1.836578, + 1.834585, + 1.833724, + 1.846582, + 1.827634, + 1.847494, + 1.908754, + 2.096389, + 1.905319, + 1.854985, + 1.833573, + 1.834956, + 1.854044, + 2.30773, + 2.050682, + 1.846312, + 1.852301, + 1.829438, + 1.843778, + 1.83131, + 1.827785, + 2.031894, + 1.872491, + 1.832372, + 1.843138, + 1.83113, + 1.853132, + 1.900752, + 1.83858 + ] + }, + "native_lua55": { + "median_ms": 0.293301, + "samples_ms": [ + 0.2926, + 0.292009, + 0.291979, + 0.305399, + 0.29321, + 0.314211, + 0.305989, + 0.2925, + 0.29259, + 0.304197, + 0.293091, + 0.29293, + 0.295424, + 0.304497, + 0.292249, + 0.347841, + 0.324828, + 0.293241, + 0.308283, + 0.305398, + 0.29301, + 0.29267, + 0.304417, + 0.29292, + 0.293982, + 0.293301, + 0.305148, + 0.293572, + 0.29302, + 0.304617, + 0.29321 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 29999 + } + }, + "fib_recursive": { + "python_lower_bound": { + "median_ms": 0.622544, + "samples_ms": [ + 0.620982, + 0.624167, + 0.610066, + 0.622865, + 0.645529, + 0.614223, + 0.621944, + 0.635043, + 0.625619, + 0.621503, + 0.624107, + 0.620031, + 0.620221, + 0.608714, + 0.750813, + 0.63327, + 0.616846, + 0.638357, + 0.608684, + 0.621523, + 0.68111, + 0.610116, + 0.850759, + 0.692818, + 0.617076, + 0.635353, + 0.623346, + 0.618849, + 0.622544, + 0.651677, + 0.610557 + ] + }, + "luapyre": { + "median_ms": 24.270311, + "samples_ms": [ + 24.225636, + 24.758719, + 24.50075, + 24.162834, + 24.405, + 24.756746, + 23.947487, + 24.09916, + 24.521471, + 24.270311, + 24.320315, + 24.028546, + 24.597462, + 24.209873, + 24.123286, + 24.37825, + 25.06597, + 24.525567, + 23.889272, + 23.967597, + 24.139619, + 23.993074, + 24.627186, + 24.091138, + 24.435835, + 24.103376, + 24.520709, + 25.110626, + 23.9081, + 24.090928, + 24.75218 + ] + }, + "native_lua55": { + "median_ms": 0.326109, + "samples_ms": [ + 0.321372, + 0.321332, + 0.334712, + 0.324587, + 0.325988, + 0.323665, + 0.372867, + 0.322163, + 0.351877, + 0.326109, + 0.323835, + 0.336224, + 0.322654, + 0.322935, + 0.336734, + 0.336975, + 0.324327, + 0.338397, + 0.327431, + 0.325078, + 0.351186, + 0.322394, + 0.3265, + 0.322534, + 0.336294, + 0.326299, + 0.324576, + 0.344746, + 0.500896, + 0.35352, + 0.326069 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + } + }, + "binary_trees": { + "python_lower_bound": { + "median_ms": 1.981461, + "samples_ms": [ + 2.205189, + 2.038485, + 2.00187, + 1.96726, + 1.981931, + 1.977394, + 1.972968, + 2.153283, + 1.981461, + 2.036622, + 1.986358, + 1.963184, + 1.978717, + 2.195234, + 2.003062, + 2.040166, + 2.229485, + 2.121857, + 1.968922, + 1.972147, + 1.976233, + 1.979408, + 1.979878, + 2.104722, + 1.976072, + 1.966809, + 2.065805, + 1.974611, + 2.041148, + 1.973449, + 1.978065 + ] + }, + "luapyre": { + "median_ms": 86.255617, + "samples_ms": [ + 86.011476, + 85.523948, + 86.241854, + 88.035949, + 85.53185, + 87.658034, + 85.410202, + 87.405995, + 85.900466, + 86.317868, + 85.599895, + 85.451707, + 85.426961, + 85.593044, + 86.636957, + 88.96947, + 84.945494, + 85.553237, + 86.287073, + 86.725226, + 86.89797, + 85.879145, + 86.053641, + 86.41464, + 86.141399, + 91.822428, + 93.215613, + 88.877806, + 86.255617, + 90.49507, + 88.629451 + ] + }, + "native_lua55": { + "median_ms": 1.762737, + "samples_ms": [ + 1.715578, + 1.736168, + 1.712925, + 1.723329, + 1.692254, + 1.720896, + 1.736309, + 1.702398, + 1.773092, + 1.726775, + 1.749658, + 1.815064, + 1.724221, + 1.762737, + 1.734396, + 1.770018, + 1.730511, + 1.820772, + 1.7303, + 1.820542, + 3.947624, + 2.501422, + 1.878918, + 1.84593, + 1.776708, + 1.793523, + 1.770318, + 1.766933, + 1.832219, + 1.819812, + 1.744882 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "sieve": { + "python_lower_bound": { + "median_ms": 0.508927, + "samples_ms": [ + 0.512823, + 0.495277, + 0.553172, + 0.500514, + 0.515176, + 0.498541, + 0.514445, + 0.505122, + 0.513654, + 0.496248, + 0.535757, + 0.495437, + 0.509258, + 0.501446, + 0.526182, + 0.499333, + 0.5111, + 0.494746, + 0.536257, + 0.499282, + 0.510259, + 0.495388, + 0.507905, + 0.495147, + 0.51062, + 0.508927, + 0.512833, + 0.496389, + 0.509138, + 0.495427, + 0.509287 + ] + }, + "luapyre": { + "median_ms": 19.481853, + "samples_ms": [ + 19.904033, + 19.673454, + 19.639935, + 19.481853, + 20.022307, + 19.461183, + 19.505698, + 19.803406, + 20.931762, + 19.771589, + 19.063029, + 19.280818, + 22.497531, + 19.327747, + 21.40051, + 19.589642, + 19.15299, + 19.13932, + 19.054716, + 19.783938, + 19.235932, + 19.591103, + 19.222784, + 19.250944, + 19.136366, + 20.164656, + 19.335388, + 19.502554, + 19.053284, + 19.394285, + 19.303191 + ] + }, + "native_lua55": { + "median_ms": 0.191431, + "samples_ms": [ + 0.192052, + 0.192743, + 0.191431, + 0.192213, + 0.20433, + 0.190399, + 0.190269, + 0.19066, + 0.189939, + 0.207564, + 0.19103, + 0.192753, + 0.191131, + 0.206694, + 0.205702, + 0.19106, + 0.189878, + 0.193464, + 0.189859, + 0.203769, + 0.191701, + 0.191111, + 0.191201, + 0.19106, + 0.190239, + 0.215707, + 0.191271, + 0.191922, + 0.191551, + 0.190059, + 0.214615 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 4137 + } + }, + "table_mix": { + "python_lower_bound": { + "median_ms": 2.507822, + "samples_ms": [ + 2.490537, + 2.645474, + 2.507822, + 2.580889, + 2.501853, + 2.518939, + 2.492811, + 2.529554, + 2.597483, + 2.551727, + 2.528732, + 2.494603, + 2.516324, + 2.510466, + 2.499159, + 2.50668, + 2.509545, + 2.493431, + 2.51338, + 2.504728, + 2.556593, + 2.495564, + 2.500641, + 2.498559, + 2.508213, + 2.493631, + 2.527781, + 2.491718, + 2.518147, + 2.501303, + 2.505749 + ] + }, + "luapyre": { + "median_ms": 22.065917, + "samples_ms": [ + 22.638538, + 22.689302, + 22.616165, + 46.530096, + 22.076472, + 21.938099, + 22.065917, + 21.639501, + 21.919152, + 21.988934, + 21.914915, + 21.801899, + 22.376434, + 22.564639, + 23.098563, + 23.38957, + 22.35402, + 22.219343, + 21.898952, + 22.073809, + 22.043734, + 22.172674, + 21.725478, + 21.915846, + 22.051646, + 22.028102, + 22.049023, + 21.751465, + 21.797364, + 22.385937, + 22.158844 + ] + }, + "native_lua55": { + "median_ms": 0.58587, + "samples_ms": [ + 0.577528, + 0.591578, + 0.573773, + 0.5992, + 0.654361, + 0.575776, + 0.58593, + 0.574643, + 0.58587, + 0.573142, + 0.593261, + 0.637556, + 0.572862, + 0.589876, + 0.572491, + 0.586281, + 0.587422, + 0.575495, + 0.641121, + 0.575165, + 0.583227, + 0.58554, + 0.571119, + 0.589035, + 0.604307, + 0.600231, + 0.58616, + 0.57217, + 0.584057, + 0.571419, + 0.586792 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 3, + "loop_iterations": 24665 + } + }, + "string_build": { + "python_lower_bound": { + "median_ms": 0.371836, + "samples_ms": [ + 0.396392, + 0.363784, + 0.362081, + 0.400578, + 0.371836, + 0.499233, + 0.372737, + 0.366809, + 0.382662, + 0.368441, + 0.386197, + 0.377063, + 0.362702, + 0.376172, + 0.363033, + 0.419125, + 0.38814, + 0.363053, + 0.361431, + 0.376653, + 0.360469, + 0.391074, + 0.36076, + 0.360029, + 0.37418, + 0.362422, + 0.360168, + 0.372376, + 0.364225, + 0.36107, + 0.374299 + ] + }, + "luapyre": { + "median_ms": 0.518331, + "samples_ms": [ + 0.53085, + 0.549637, + 0.513264, + 0.530319, + 0.503759, + 0.528366, + 0.514966, + 0.522787, + 0.512152, + 0.70836, + 0.51749, + 0.525252, + 0.518401, + 0.533293, + 0.528165, + 0.509157, + 0.536768, + 0.511571, + 0.522247, + 0.510199, + 0.520674, + 0.509248, + 0.517921, + 0.511211, + 0.548906, + 0.51102, + 0.520785, + 0.503599, + 0.514946, + 0.511151, + 0.518331 + ] + }, + "native_lua55": { + "median_ms": 1.119944, + "samples_ms": [ + 1.158771, + 1.288902, + 1.154555, + 1.143158, + 1.202145, + 1.128588, + 1.112954, + 1.130129, + 1.112223, + 1.117161, + 1.129478, + 1.107506, + 1.107446, + 1.104662, + 1.120555, + 1.101047, + 1.11639, + 1.141156, + 1.129778, + 1.112554, + 1.114567, + 1.125663, + 1.109138, + 1.111892, + 1.386305, + 1.119944, + 1.187073, + 1.113555, + 1.132623, + 1.105733, + 1.108888 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_iterations": 2874 + } + }, + "spectral_norm": { + "python_lower_bound": { + "median_ms": 2.363981, + "samples_ms": [ + 2.363981, + 2.348468, + 2.368879, + 2.340476, + 2.69001, + 2.355588, + 2.358512, + 2.366996, + 2.367185, + 2.359835, + 2.354497, + 2.339094, + 2.365553, + 2.343471, + 2.35682, + 2.38334, + 2.344802, + 2.400043, + 2.548792, + 2.454103, + 2.374657, + 2.357762, + 2.37713, + 2.357902, + 2.363039, + 2.374787, + 2.333175, + 2.657472, + 2.403819, + 2.437028, + 2.351603 + ] + }, + "luapyre": { + "median_ms": 23.591748, + "samples_ms": [ + 24.372914, + 23.644235, + 23.54624, + 23.706016, + 24.381336, + 24.836415, + 23.874663, + 23.222425, + 23.526842, + 23.497869, + 23.21163, + 23.591748, + 23.252539, + 23.221804, + 23.910285, + 23.509547, + 23.254091, + 23.508796, + 23.701829, + 23.442548, + 23.870547, + 23.332877, + 23.751122, + 23.632787, + 23.254853, + 24.034748, + 23.452934, + 23.151822, + 23.650884, + 23.652197, + 24.058021 + ] + }, + "native_lua55": { + "median_ms": 0.836328, + "samples_ms": [ + 0.844409, + 0.833414, + 0.821245, + 0.832182, + 0.852562, + 0.839092, + 0.834536, + 0.838702, + 0.829307, + 0.836328, + 0.834516, + 0.833474, + 0.834195, + 0.850989, + 0.814516, + 0.832642, + 0.838641, + 0.897217, + 0.862867, + 0.834115, + 0.840464, + 0.813504, + 0.837159, + 0.851851, + 0.934262, + 0.908014, + 0.84434, + 0.85135, + 0.822698, + 0.834725, + 0.835998 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_iterations": 58, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + }, + "python_calls_1000": { + "python_lower_bound": { + "median_ms": 0.04089, + "samples_ms": [ + 0.040779, + 0.0409, + 0.04094, + 0.04086, + 0.041101, + 0.04075, + 0.04088, + 0.04088, + 0.061561, + 0.041301, + 0.04096, + 0.040901, + 0.040659, + 0.040759, + 0.04097, + 0.04067, + 0.04081, + 0.041411, + 0.04081, + 0.041051, + 0.04079, + 0.0409, + 0.041081, + 0.04084, + 0.04089, + 0.04064, + 0.04061, + 0.060319, + 0.04089, + 0.04081, + 0.04092 + ] + }, + "luapyre": { + "median_ms": 0.595184, + "samples_ms": [ + 0.597978, + 0.592079, + 0.732275, + 0.66082, + 0.582546, + 0.656354, + 0.582565, + 0.597107, + 0.615584, + 0.57927, + 0.611217, + 0.580983, + 0.597117, + 0.59947, + 0.580762, + 0.61332, + 0.581674, + 0.595364, + 0.59942, + 0.583036, + 0.595164, + 0.593571, + 0.591208, + 0.580672, + 0.598088, + 0.593191, + 0.586711, + 0.595184, + 0.595784, + 0.600241, + 0.594252 + ] + }, + "native_lua55": { + "median_ms": 0.199242, + "samples_ms": [ + 0.196869, + 0.216358, + 0.19725, + 0.199023, + 0.199222, + 0.219442, + 0.210399, + 0.197781, + 0.199363, + 0.19763, + 0.199332, + 0.211991, + 0.199403, + 0.19812, + 0.198421, + 0.199333, + 0.212182, + 0.196418, + 0.19709, + 0.198621, + 0.199773, + 0.21129, + 0.199373, + 0.199242, + 0.199453, + 0.22463, + 0.21066, + 0.195938, + 0.196298, + 0.199222, + 0.198281 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + } + } + } + } + ], + "workloads": { + "typed_arith": { + "luapyre": { + "median_ms": 0.72231, + "process_medians_ms": [ + 0.72231, + 0.74815, + 0.719877 + ] + }, + "native_lua55": { + "median_ms": 0.126085, + "process_medians_ms": [ + 0.126085, + 0.126115, + 0.124082 + ] + }, + "python_lower_bound": { + "median_ms": 0.489238, + "process_medians_ms": [ + 0.483029, + 0.504312, + 0.489238 + ] + }, + "luapyre_over_native": 5.728754411706388, + "luapyre_over_python_lower_bound": 1.476397990344168, + "native_over_python_lower_bound": 0.25771710292332156 + }, + "typed_branch": { + "luapyre": { + "median_ms": 1.856225, + "process_medians_ms": [ + 1.856225, + 1.945001, + 1.846312 + ] + }, + "native_lua55": { + "median_ms": 0.301464, + "process_medians_ms": [ + 0.337035, + 0.301464, + 0.293301 + ] + }, + "python_lower_bound": { + "median_ms": 1.142818, + "process_medians_ms": [ + 1.142818, + 1.143441, + 1.132423 + ] + }, + "luapyre_over_native": 6.157368707374678, + "luapyre_over_python_lower_bound": 1.6242525056483186, + "native_over_python_lower_bound": 0.26379003480869223 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 24.270311, + "process_medians_ms": [ + 24.045179, + 24.569299, + 24.270311 + ] + }, + "native_lua55": { + "median_ms": 0.324176, + "process_medians_ms": [ + 0.324176, + 0.318198, + 0.326109 + ] + }, + "python_lower_bound": { + "median_ms": 0.619632, + "process_medians_ms": [ + 0.619399, + 0.619632, + 0.622544 + ] + }, + "luapyre_over_native": 74.86769841073983, + "luapyre_over_python_lower_bound": 39.168911547499164, + "native_over_python_lower_bound": 0.5231750458336561 + }, + "binary_trees": { + "luapyre": { + "median_ms": 86.255617, + "process_medians_ms": [ + 85.286983, + 86.546099, + 86.255617 + ] + }, + "native_lua55": { + "median_ms": 1.761134, + "process_medians_ms": [ + 1.758001, + 1.761134, + 1.762737 + ] + }, + "python_lower_bound": { + "median_ms": 1.981461, + "process_medians_ms": [ + 2.014708, + 1.952826, + 1.981461 + ] + }, + "luapyre_over_native": 48.97731632005288, + "luapyre_over_python_lower_bound": 43.531322090114315, + "native_over_python_lower_bound": 0.8888057852261538 + }, + "sieve": { + "luapyre": { + "median_ms": 19.68132, + "process_medians_ms": [ + 19.843341, + 19.68132, + 19.481853 + ] + }, + "native_lua55": { + "median_ms": 0.193936, + "process_medians_ms": [ + 0.193936, + 0.193995, + 0.191431 + ] + }, + "python_lower_bound": { + "median_ms": 0.508927, + "process_medians_ms": [ + 0.508668, + 0.512172, + 0.508927 + ] + }, + "luapyre_over_native": 101.48358221268872, + "luapyre_over_python_lower_bound": 38.67218677727847, + "native_over_python_lower_bound": 0.3810684047024426 + }, + "table_mix": { + "luapyre": { + "median_ms": 22.065917, + "process_medians_ms": [ + 21.879974, + 22.342781, + 22.065917 + ] + }, + "native_lua55": { + "median_ms": 0.58587, + "process_medians_ms": [ + 0.589988, + 0.57151, + 0.58587 + ] + }, + "python_lower_bound": { + "median_ms": 2.507822, + "process_medians_ms": [ + 2.529781, + 2.4987, + 2.507822 + ] + }, + "luapyre_over_native": 37.66350384897673, + "luapyre_over_python_lower_bound": 8.79883699879816, + "native_over_python_lower_bound": 0.23361705894596985 + }, + "string_build": { + "luapyre": { + "median_ms": 0.522989, + "process_medians_ms": [ + 0.524421, + 0.522989, + 0.518331 + ] + }, + "native_lua55": { + "median_ms": 1.166594, + "process_medians_ms": [ + 1.316476, + 1.166594, + 1.119944 + ] + }, + "python_lower_bound": { + "median_ms": 0.371836, + "process_medians_ms": [ + 0.431464, + 0.363284, + 0.371836 + ] + }, + "luapyre_over_native": 0.4483042086621396, + "luapyre_over_python_lower_bound": 1.4065044804698847, + "native_over_python_lower_bound": 3.13738852612442 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 23.59514, + "process_medians_ms": [ + 24.02563, + 23.59514, + 23.591748 + ] + }, + "native_lua55": { + "median_ms": 0.836328, + "process_medians_ms": [ + 0.846205, + 0.829529, + 0.836328 + ] + }, + "python_lower_bound": { + "median_ms": 2.407256, + "process_medians_ms": [ + 2.431947, + 2.407256, + 2.363981 + ] + }, + "luapyre_over_native": 28.212782544647556, + "luapyre_over_python_lower_bound": 9.801674603781235, + "native_over_python_lower_bound": 0.3474196346379446 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.597848, + "process_medians_ms": [ + 0.600834, + 0.597848, + 0.595184 + ] + }, + "native_lua55": { + "median_ms": 0.196929, + "process_medians_ms": [ + 0.19675, + 0.196929, + 0.199242 + ] + }, + "python_lower_bound": { + "median_ms": 0.04067, + "process_medians_ms": [ + 0.04067, + 0.0406, + 0.04089 + ] + }, + "luapyre_over_native": 3.03585556215692, + "luapyre_over_python_lower_bound": 14.69997541185149, + "native_over_python_lower_bound": 4.842119498401771 + } + } +} diff --git a/benchmarks/results/speed_037.json b/benchmarks/results/speed_037.json new file mode 100644 index 0000000..8a15bf2 --- /dev/null +++ b/benchmarks/results/speed_037.json @@ -0,0 +1,31 @@ +{ + "schema_version": 1, + "baseline_revision": "20fb9756cb65b55f29607aacc2ae5aa7ca80a2f4", + "candidate": "codex/037-performance-roadmap worktree", + "method": "Three isolated processes per Python; Lua/Python, Python/Lua, Lua/Python; seven warmups and 31 checked samples; fixed affinity and PYTHONHASHSEED=0; median of process medians", + "interpretation": "Reduced-contract same-algorithm Python references are comparison points, not promised speedups. Improvement is calculated from LuaPyre elapsed times.", + "summary": [ + {"python": "3.13.15", "case": "next_dense", "baseline_ms": 2092.125983, "candidate_ms": 63.960816, "python_ms": 0.141869, "improvement_percent": 96.942784, "candidate_ratio": 450.844201}, + {"python": "3.13.15", "case": "delete_hash", "baseline_ms": 93.027238, "candidate_ms": 8.619113, "python_ms": 0.489208, "improvement_percent": 90.734850, "candidate_ratio": 17.618504}, + {"python": "3.13.15", "case": "python_leaf_float", "baseline_ms": 1.662360, "candidate_ms": 1.110228, "python_ms": 0.084053, "improvement_percent": 33.213744, "candidate_ratio": 13.208666}, + {"python": "3.13.15", "case": "python_leaf_binary", "baseline_ms": 2.161388, "candidate_ms": 1.204636, "python_ms": 0.092356, "improvement_percent": 44.265629, "candidate_ratio": 13.043397}, + {"python": "3.14.7", "case": "next_dense", "baseline_ms": 1992.278984, "candidate_ms": 59.118938, "python_ms": 0.111022, "improvement_percent": 97.032597, "candidate_ratio": 532.497505}, + {"python": "3.14.7", "case": "delete_hash", "baseline_ms": 94.613735, "candidate_ms": 7.699087, "python_ms": 0.539831, "improvement_percent": 91.862612, "candidate_ratio": 14.262032}, + {"python": "3.14.7", "case": "python_leaf_float", "baseline_ms": 1.551137, "candidate_ms": 1.072873, "python_ms": 0.078995, "improvement_percent": 30.833124, "candidate_ratio": 13.581530}, + {"python": "3.14.7", "case": "python_leaf_binary", "baseline_ms": 2.063980, "candidate_ms": 1.194673, "python_ms": 0.080578, "improvement_percent": 42.117995, "candidate_ratio": 14.826293} + ], + "process_medians": { + "3.13.15": { + "next_dense": {"baseline_ms": [2092.125983, 2050.654498, 2100.914270], "candidate_ms": [63.960816, 64.259415, 60.986797]}, + "delete_hash": {"baseline_ms": [95.222899, 93.027238, 91.676879], "candidate_ms": [8.727673, 8.619113, 8.610342]}, + "python_leaf_float": {"baseline_ms": [1.640024, 1.728015, 1.662360], "candidate_ms": [1.112183, 1.080666, 1.110228]}, + "python_leaf_binary": {"baseline_ms": [2.161388, 2.162262, 2.130418], "candidate_ms": [1.226150, 1.191519, 1.204636]} + }, + "3.14.7": { + "next_dense": {"baseline_ms": [1974.565288, 2046.980876, 1992.278984], "candidate_ms": [61.082194, 59.118938, 58.734674]}, + "delete_hash": {"baseline_ms": [94.613735, 96.932285, 93.360822], "candidate_ms": [7.840991, 7.699087, 7.615855]}, + "python_leaf_float": {"baseline_ms": [1.566387, 1.551137, 1.548914], "candidate_ms": [1.073477, 1.067295, 1.072873]}, + "python_leaf_binary": {"baseline_ms": [2.083025, 2.032323, 2.063980], "candidate_ms": [1.194673, 1.181823, 1.206769]} + } + } +} diff --git a/benchmarks/speed_037_ab.py b/benchmarks/speed_037_ab.py new file mode 100644 index 0000000..3b592a4 --- /dev/null +++ b/benchmarks/speed_037_ab.py @@ -0,0 +1,235 @@ +"""Measure the 0.37 table and scalar-boundary candidates. + +The 0.36 cases remain available unchanged. New cases add table traversal and +deletion scaling; every timed result is checked against a same-algorithm +Python reference. Run this file against both baseline and candidate sources. +""" +from __future__ import annotations + +import argparse +from dataclasses import asdict +import gc +import json +import os +from pathlib import Path +import platform +import statistics +import time + +from luapyre import LuaRuntime +from speed_036_ab import CASES as CASES_036, Case, FUEL, call_batch, validate + + +_DENSE = tuple(range(1, 2049)) +_HASH = {f"key-{index}": index for index in range(1, 2049)} + + +def next_dense(): + total = 0 + for _ in range(4): + for value in _DENSE: + total += value + return total + + +def next_hash(): + total = 0 + for _ in range(4): + for value in _HASH.values(): + total += value + return total + + +def delete_dense(): + values = {index: index for index in range(1, 2049)} + total = 0 + for index in range(1, 2049): + total += values[index] + del values[index] + return total + + +def delete_hash(): + values = {f"key-{index}": index for index in range(1, 2049)} + total = 0 + for index in range(1, 2049): + key = f"key-{index}" + total += values[key] + del values[key] + return total + + +NEW_CASES = { + "next_dense": Case(""" +local values: table = {} +for i = 1, 2048 do values[i] = i end +return function(): integer + local total: integer = 0 + for round = 1, 4 do + for _, value in pairs(values) do total = total + value end + end + return total +end +""", next_dense, 8392704, "Repeated full traversal of an unchanged dense table", ((),)), + "next_hash": Case(""" +local values: table = {} +for i = 1, 2048 do values["key-" .. i] = i end +return function(): integer + local total: integer = 0 + for round = 1, 4 do + for _, value in pairs(values) do total = total + value end + end + return total +end +""", next_hash, 8392704, "Repeated full traversal of an unchanged hash table", ((),)), + "delete_dense": Case(""" +return function(): integer + local values: table = {} + for i = 1, 2048 do values[i] = i end + local total: integer = 0 + for i = 1, 2048 do total = total + values[i]; values[i] = nil end + return total +end +""", delete_dense, 2098176, "Forward deletion sweep over a dense table", ((),)), + "delete_hash": Case(""" +return function(): integer + local values: table = {} + for i = 1, 2048 do values["key-" .. i] = i end + local total: integer = 0 + for i = 1, 2048 do + local key = "key-" .. i + total = total + values[key] + values[key] = nil + end + return total +end +""", delete_hash, 2098176, "Forward deletion sweep over a hash table", ((),)), +} + +CASES = {**CASES_036, **NEW_CASES} +UNTYPED_CASES = frozenset(("next_dense", "next_hash")) + + +def prepare(name, *, jit=True, threshold=32): + case = CASES[name] + runtime = LuaRuntime(jit=jit, jit_threshold=threshold, fuel=FUEL) + source = case.source if name in UNTYPED_CASES else "-- luapyre: typed\n" + case.source + if case.arguments is None: + proto = runtime.compile(source) + + def run(): + return runtime.vm.run(proto, fuel=FUEL) + reference = case.reference + else: + function = runtime.execute_python(source) + expected = tuple(case.reference(*args) for args in case.arguments) + run = call_batch(function, case.arguments, expected) + reference = call_batch(case.reference, case.arguments, expected) + return runtime, run, reference + + +def counter_delta(before, runtime): + return { + key: value - before[key] + for key, value in asdict(runtime.jit_stats).items() + if type(value) is int and value != before[key] + } + + +def timed_samples(run, case, count, *, disable_gc): + was_enabled = gc.isenabled() + if disable_gc: + gc.disable() + samples = [] + try: + for _ in range(count): + started = time.perf_counter_ns() + result = run() + samples.append((time.perf_counter_ns() - started) / 1_000_000) + assert validate(case, result), (result, case.expected) + finally: + if was_enabled: + gc.enable() + else: + gc.disable() + return {"median_ms": statistics.median(samples), "samples_ms": samples} + + +def measure(name, warmups, repeats, *, python_first=False): + case = CASES[name] + started = time.perf_counter_ns() + runtime, run, reference = prepare(name) + setup_ms = (time.perf_counter_ns() - started) / 1_000_000 + before = asdict(runtime.jit_stats) + cold = timed_samples(run, case, 1, disable_gc=False) + cold_counters = counter_delta(before, runtime) + before = asdict(runtime.jit_stats) + warm = timed_samples(run, case, warmups, disable_gc=False) + warm_counters = counter_delta(before, runtime) + timed_samples(reference, case, warmups, disable_gc=False) + variants = [("luapyre", run), ("python_reference", reference)] + if python_first: + variants.reverse() + before = asdict(runtime.jit_stats) + steady = { + label: timed_samples(fn, case, repeats, disable_gc=True) + for label, fn in variants + } + steady_counters = counter_delta(before, runtime) + return { + "purpose": case.purpose, + "setup_ms": setup_ms, + "cold": cold, + "cold_counters": cold_counters, + "warmup": warm, + "warmup_counters": warm_counters, + "steady": steady, + "steady_counters": steady_counters, + "compilation_during_steady": { + key: value for key, value in steady_counters.items() if "compile" in key + }, + "ratio": steady["luapyre"]["median_ms"] / steady["python_reference"]["median_ms"], + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--revision", required=True) + parser.add_argument("--json", type=Path, required=True) + parser.add_argument("--warmups", type=int, default=7) + parser.add_argument("--repeats", type=int, default=31) + parser.add_argument("--python-first", action="store_true") + parser.add_argument("--case", action="append", choices=CASES) + args = parser.parse_args() + if args.warmups < 1 or args.repeats < 1: + parser.error("warmups and repeats must be positive") + affinity = None + if hasattr(os, "sched_getaffinity"): + affinity = min(os.sched_getaffinity(0)) + os.sched_setaffinity(0, {affinity}) + results = {} + for name in args.case or CASES: + results[name] = measure( + name, args.warmups, args.repeats, python_first=args.python_first + ) + print(f"{name}: {results[name]['ratio']:.2f}x Python reference", flush=True) + args.json.write_text(json.dumps({ + "schema_version": 1, + "revision": args.revision, + "python": platform.python_version(), + "platform": platform.platform(), + "affinity_cpu": affinity, + "hash_seed": os.environ.get("PYTHONHASHSEED"), + "jit_threshold": 32, + "fuel": FUEL, + "warmups": args.warmups, + "repeats": args.repeats, + "python_first": args.python_first, + "method": "Setup, first execution, warmup, and steady phases separated; GC disabled only during steady timing; every result checked", + "interpretation": "Reduced-contract Python references, not promised speedups; boundary cases use the same checked driver on both sides.", + "workloads": results, + }, indent=2) + "\n") + + +if __name__ == "__main__": + main() diff --git a/docs/performance-roadmap-0.36.md b/docs/performance-roadmap-0.36.md index 9e20bc2..d73fc8a 100644 --- a/docs/performance-roadmap-0.36.md +++ b/docs/performance-roadmap-0.36.md @@ -1,5 +1,9 @@ # Performance roadmap: 0.36 +The next [0.37 roadmap](performance-roadmap-0.37.md) distinguishes the +unfinished work below from new iteration/deletion and compiled-call +experiments. It does not mark this roadmap's runtime work as completed. + ## Starting point and scope The accumulated 0.33–0.35 performance work is merged into `main` by diff --git a/docs/performance-roadmap-0.37.md b/docs/performance-roadmap-0.37.md new file mode 100644 index 0000000..916742e --- /dev/null +++ b/docs/performance-roadmap-0.37.md @@ -0,0 +1,279 @@ +# Performance roadmap: 0.37 + +## Goal and actual starting point + +Reduce the remaining cost of table operations, compiled calls, and control +flow while keeping LuaPyre entirely in Python. Continue generating lean +Python source/AST and inspect warmed code with `dis(..., adaptive=True)`. +Direct CPython bytecode generation is outside this plan. + +The original planning baseline is `main` at +`20fb9756cb65b55f29607aacc2ae5aa7ca80a2f4`, tree +`f57d313df710d58258981202d934b66e99c6abd7`, after +[PR #45](https://github.com/rbroderi/LuaPyre/pull/45). The runtime is still +**0.35.0a1**: 0.36 added measurements, probes, and a roadmap, not a runtime +tranche. The implemented subset and measurements are now recorded in +[`speed-0.37.md`](speed-0.37.md); unfinished items below remain future work. + +The [0.36 roadmap](performance-roadmap-0.36.md) remains the detailed source +for unfinished prerequisites. This plan can start from current main. If a +0.36 runtime tranche lands first, remeasure it and remove only prerequisites +that its code and tests actually satisfy. Do not count the same improvement +in both releases. + +## Where the gap remains + +Recorded medians from the 0.36 planning pass, running the 0.35 runtime: + +| Workload | LuaPyre / Python, 3.13 | LuaPyre / Python, 3.14 | Primary investigation | +| --- | ---: | ---: | --- | +| Binary trees | 47.34× | 45.13× | Calls, record writes, allocation | +| Sieve | 40.42× | 43.58× | Nested control flow and table operations | +| Recursive Fibonacci | 41.43× | 36.83× | Compiled entries and frame materialization | +| Table mix | 8.99× | 9.76× | Repeated representation checks and writes | +| Spectral norm | 14.06× | 11.14× | Tier differences, helper calls, numeric proofs | +| Python two-integer leaf | 27.91× | 29.74× | Entry validation, argument/result containers | + +Sources: [original workload samples](../benchmarks/results/python_headroom_035.json), +[focused probe samples](../benchmarks/results/speed_036_baseline.json), +[profiles](../benchmarks/results/profile_035.json), and +[generated-code audits](../benchmarks/results/codegen_036_baseline.json). +The boundary row uses the newer harness's identical per-call checking driver; +its ratio is not directly comparable to the older Python-call benchmark. + +Arithmetic, branches, and string building are already approximately 1.3–1.7× +their direct Python references. Prioritize the larger gaps. These references +use the same application algorithms but omit parts of Lua's contract, including +metamethods, fuel, hooks, GC accounting, and conversion. Their timings are +comparison points, not a promise that every workload can reach 1×. + +## Ordered work + +Each row is a separately measured change. Finish only the prerequisite needed +for that change; avoid making every optimization depend on a complete new IR. + +| Step | Deliverable | Relationship to 0.36 | Dependency | +| --- | --- | --- | --- | +| 0 | Baseline, size sweeps, and cost attribution | Reuses its 20 workloads and phase-aware harness | None | +| 1 | Efficient table traversal and deletion bookkeeping | New source-backed candidate | Step 0 | +| 2 | Facts across call continuations and longer structured regions | Explicit carryover | Step 0 | +| 3 | Scalar compiled-call interfaces and broader Python entry | New internal interface; boundary cleanup is carryover | Required call facts from step 2 | +| 4 | Table access proofs across a region | Carryover, using the shared facts | Step 2 | +| 5 | Lazy scalar recursive activations | Carryover, separately gated prototype | Proven call/recovery contract from step 3 | + +### 0. Establish an attributable baseline + +Keep all nine original controls and eleven 0.36 probes. Record the actual +runtime revision/tree, harness revision, Python build, input size, selected +tier, and counter changes. A revision label does not select a source checkout. +Separate setup, first execution, warmup, and steady state; add isolated JIT +compile timing before evaluating larger generated functions. + +For table and call probes, add diagnostic Python variants using `LuaTable` +operations or an explicit argument/result wrapper, alongside the direct +Python algorithm. These comparisons help locate representation and wrapper +costs, but do not reproduce the entire Lua contract. Do not subtract their +timings as if all costs were independent or advertise them as native Python. + +Measure allocations and retained memory separately from elapsed time. Count +entry-wrapper calls, argument/result containers, frame acquisitions, spills, +state transitions, generic table operations, and entry-list construction. +Use `dis(adaptive=True)` to check specialization after warmup; fewer opcodes +alone do not establish a speedup. + +### 1. Remove repeated whole-table work from iteration and deletion + +Current source exposes a candidate absent from the 0.36 probe set: +[`next_fn`](../src/luapyre/stdlib.py) builds `list(table.items())` on **every** +call and scans it for the previous key. Traversing an unchanged table of +`n` entries therefore does quadratic work. Both `rawset` and +`rawset_prehashed` in [`table.py`](../src/luapyre/table.py) also materialize the +entries when deleting a present key to record its successor. Repeated deletion +can have the same scaling problem. This is a source-derived diagnosis; +there is no measured 0.37 iteration speedup yet. + +First benchmark dense, hash, and mixed tables across sizes. Prototype a +table-owned order/successor index that can be reused across `next` calls and +maintained on deletion. A cached list with a linear search is insufficient. +Target linear total work for an unchanged full traversal and a deletion sweep, +including index construction. Start with unchanged traversal; accept mutation +support as a second change only if its maintenance cost is justified. + +Preserve independent/interleaved traversals, invalid-key errors, deletion of +the current key, chains of deleted successors, nil versus false, numeric key +normalization, and `pairs` metamethod behavior. Do not introduce a single +mutable cursor shared by all callers. Value overwrite must return the current +value; an index must not freeze a snapshot of values. + +Audit every mutation route, including generated writes, raw writes, array/hash +migration, weak-table clearing, and supported Python access. The existing +`version` increments on every write; it cannot by itself distinguish an +unchanged layout from a changed value. Fall back for unproved mutation paths. +Test metadata lifetime, object keys, weak tables and finalizers: the collector +currently traces deleted-successor references. An index must neither retain +dead objects indefinitely nor bypass required roots/accounting. + +Acceptance evidence: scaling curves and allocation counts, traversal/deletion +speed, plus unchanged-table read/write and construction controls showing the +cost of maintaining any new metadata. Do not change the user's algorithm or +public iteration API to obtain the gain. + +### 2. Carry facts and structured execution across calls and nested loops + +[`function_jit.py`](../src/luapyre/function_jit.py) still starts each block +with an empty `known_constants`. Implement conservative fixed-point facts +using the existing CFG/value IR: intersect at joins, widen at loops, and +invalidate heap/upvalue facts on effects. Preserve an uncaptured scalar or +constant key across a call only when its definition remains valid. + +Land constant record-key propagation first, with a branch using different +keys as a negative test. Extend argument/return and numeric range facts only +with explicit proofs. Then use these facts in shared structured lowering for +nested reducible loops and call-free continuations, across both the loop and +whole-function tiers. Target `record_continuations`, `sparse_nested`, sieve, +and the `numeric_helper`/`numeric_inline` pair. + +The helper/inline difference combines tier selection and type/range contracts; +hold both constant in diagnostic probes. Do not disable whole-function +promotion globally to favor one benchmark. Preserve exact fuel, PCs, error +locations, pending results, and live registers at every suspension or guard +miss. Later-iteration exits are required tests. + +### 3. Remove containers and redundant work at compiled call boundaries + +The whole-function CALL emitter currently packs arguments into a tuple, invokes +an entry wrapper, unpacks status/results, and extracts a scalar from the result +sequence. Real-frame entries also acquire/recycle a frame. Caching a target +does not remove these per-call costs. + +Add an explicitly admitted fixed-arity scalar interface for compiled-to-compiled +calls: begin with one/two scalar arguments and one scalar result. Retain real +frames and the existing suspension path in the first experiment, so argument +and result packaging can be measured independently of frame elimination. +Then evaluate fixed two-result calls separately. Keep dynamic arity, open +results, varargs, unknown targets, and observable captures on existing paths. + +Do not reinterpret `DirectCallSite` as permission to erase a frame: its current +contract proves identity only. Extend admission/effect/recovery facts +explicitly in [`call_ir.py`](../src/luapyre/call_ir.py) and +[`escape_analysis.py`](../src/luapyre/escape_analysis.py). Small pure callees +can instead be inlined when a bounded cost model proves that profitable; +benchmark calls too large for current inlining to expose the new interface. + +In a separate change, finish 0.36's direct-leaf definite-assignment cleanup +and strict float/two-integer Python adapters. Reuse `python_leaf_local`, +`python_leaf_float`, and `python_leaf_binary`. Fuse an adapter and body only +if profiles after cleanup show enough removable work to pass the speed gate. + +Every path retains live JIT/hook/re-entry checks, fuel and stack limits, +argument/result conversion, target mutation handling, safepoints, and roots. +Do not cache mutable runtime settings as immutable contracts. Distinguish +nil, false, empty returns, multiple returns, suspension, and DEOPT. Check +invalid arguments and int64 boundaries as well as successful calls. + +### 4. Prove table representation once per admitted region + +Start with 0.36's read-only dense region, then existing-slot primitive writes +through aliases. Retain bounds/representation facts across the region only +while effects prove them stable. Aliasing may invalidate a cached value even +when storage and bounds remain valid. Repeated per-access shape guards failed +the earlier speed gate; merely moving that approach into a new helper is not +a new justification. + +Use step 2's key facts for record writes across recursive call continuations +before attempting a different record storage layout. Preserve write versions, +GC barriers/accounting, nil deletion, metatable fallback, and array/hash +migration. Growth, holes, weak modes, collectable writes and callbacks need +proved handling or the existing path. Coordinate any structural token with +step 1 instead of adding incompatible invalidation schemes. + +Targets: `dense_read`, `dense_alias_write`, `record_continuations`, table mix, +binary trees and sieve. Attribute wins to checks, hashing, bookkeeping, or +dispatch using counters and code audits. A primitive-array microbenchmark +alone does not justify a global table representation change. + +### 5. Prototype lazy scalar recursion only after recovery is proven + +The recorded Fibonacci profile still includes roughly 21,888 materialized +entry-wrapper calls per execution. Existing escape analysis admits typed +lexical leaves, not general recursive frame elimination. + +Begin with fixed-arity scalar self-recursion whose only capture is a guarded +recursive target. Keep activation state in Python locals and materialize exact +logical Lua frames on suspension/error when required. Count logical frames +even while physical Lua frames are absent. Guard against reaching CPython's +own recursion limit before the Lua limit, with a proven transition to the +existing execution path before Python stack exhaustion. + +Compare linear and balanced recursion at multiple depths; exhaust small fuel +budgets and test lowered frame limits, target mutation, exceptions, and +resumption through every ancestor. A returned value alone is insufficient +evidence. Effectful binary-tree construction, mutual recursion, escaping +captures, and yields are later admissions, not part of the first prototype. +Reject the prototype if recovery or retained-memory costs erase the win. + +## Planned probes and correctness coverage + +The reusable [`speed_037_ab.py`](../benchmarks/speed_037_ab.py) harness retains +the 0.36 probes and adds table traversal/deletion cases. The code audit now +accepts `inspect_codegen.py --suite 037`, and +[`test_performance_037.py`](../tests/test_performance_037.py) supplies +deterministic semantics. The remaining rows describe coverage required when +their deferred optimizations are attempted. + +| Probe family | Inputs and comparison | Required failure/mutation cases | +| --- | --- | --- | +| `next` / `pairs` scaling | Dense, hash, mixed; 32, 256, 2,048 entries; checked count and checksum | Interleaved traversal, invalid key, `__pairs`, value overwrite | +| Deletion scaling | Forward/reverse/alternating deletion; same sizes; raw and prehashed paths | Delete current/next, delete/reinsert, nil versus false, weak keys | +| Compiled scalar calls | Unary, binary, two-result callees over the inline budget; 100/1,000/10,000 calls | Target change, short fuel, wrong type, exception after earlier effects | +| Continuation/region facts | Same numeric contracts with helper versus inline body; varying trip counts | Conflicting join facts, nested early exit, later-iteration guard miss | +| Table proof lifetime | Existing 0.36 dense/record probes plus repeated value overwrite | Alias deletion, growth, metatable change, callback, GC collection | +| Recursive recovery | Existing linear/balanced probes plus varying depth | Every small fuel budget, frame limit, target mutation, ancestor resume | +| Python boundary | Existing local/float/binary probes; same checking driver on both sides | Live config changes, bool/int distinction, conversion, re-entry | + +Use the same input and algorithm in A/B and Python-reference runs. Keep +scaling separate from fixed-work throughput. Do not compare a fully checked +Lua API call with an unchecked Python call in the new boundary probes. + +## Acceptance and release gates + +- Establish baseline and candidate in three independent paired processes per + CPython version (3.13 and 3.14), in A/B, B/A, A/B order, with seven warmups, + 31 checked samples, fixed CPU affinity and `PYTHONHASHSEED=0`. Run elapsed-time + measurements sequentially; profile and audit separately. Require zero + steady-state compilation or report and resolve incomplete warmup. +- Accept an optimization only with at least 5% elapsed-time reduction on its + named target on both versions. Investigate process spread and reject + repeatable control regressions above 5%. For a broad call/frame or table + rewrite, require a representative workload win as well as its isolated probe. +- Report both speedup and the remaining Python ratio. When baseline exceeds + the same-run Python reference, also report the fraction of excess time + removed: `(baseline - candidate) / (baseline - Python)`. Use paired current + measurements, not a historical denominator. Never equate that fraction with + the proportion of all Lua semantics that can be eliminated. +- Report preparation/compile cost, cold execution, generated source/bytecode + size, allocations, retained memory, cache limits and break-even invocation + count. A larger specialization must justify its startup and memory costs. +- Differentially test interpreter, cold JIT and warmed JIT, then every fuel + budget through completion for small admitted shapes. Check errors, side + effects, hooks, GC roots/finalizers and recovery, not just final results. +- Before a runtime release, run the full Python suite on both versions (576 + tests at this planning baseline, plus new tests), all 24 required official + Lua 5.5.1 probes, Penlight 23/23, luatest 5/5, LuaCov scanner 24/24 and + Are We Fast Yet 11/11. Preserve established exclusions. + +Ship accepted changes independently and record rejected/deferred experiments +in `docs/speed-0.37.md` when implementation occurs. Bump the runtime version +only with that tranche. This plan does not claim that every candidate must +land, or that 1× direct Python is a reachable universal floor. + +## Keep out of the critical path + +Do not revisit the sub-5% already-owned GC-root shortcut without new profiles. +Defer custom record layouts and temporary-table scalar replacement until the +safer table/call work is measured; allocation and identity can affect GC, +weak tables and finalizers. Apply direct division only after finite/nonzero +and conversion/range proofs justify it, preserving NaN, infinities, signed +zero and error order. No memoization, changed benchmark algorithms, removed +fuel/GC/hook semantics, batch-call substitutions, native extensions, or direct +bytecode generation are performance shortcuts in this roadmap. diff --git a/docs/speed-0.37.md b/docs/speed-0.37.md new file mode 100644 index 0000000..0a092b5 --- /dev/null +++ b/docs/speed-0.37.md @@ -0,0 +1,137 @@ +# LuaPyre 0.37 performance tranche + +0.37 accepts two Python-only optimization groups from the +[0.37 roadmap](performance-roadmap-0.37.md): indexed table iteration/deletion +and broader allocation-free scalar entry from Python. It continues to emit +Python source/AST and relies on CPython's adaptive specialization; it does not +generate bytecode directly. + +## Accepted changes + +1. **Index table iteration order once.** `next` no longer rebuilds and scans a + complete entry list for every returned pair. Each table lazily caches its + current iteration keys and token positions. Independent callers pass their + own keys, values remain live, and any uncoordinated generated write changes + the table version and forces a safe rebuild. +2. **Keep deletion sweeps linear.** Raw and pre-hashed deletion use the same + index to record the deleted key's successor. The index remains valid while + removed keys are skipped, preserving LuaPyre's `next(table, deleted_key)` + behavior without rebuilding the table per deletion. Reinsertions and layout + changes invalidate it. GC pacing charges index allocation. +3. **Broaden scalar Python entry.** Certified one-float and two-integer, + one-result leaves now receive positional Python scalars and return a scalar. + They avoid argument indexing and result tuples while retaining live JIT, + hook, re-entry, stack, fuel, conversion, overflow, safepoint and type checks. +4. **Remove unreachable direct-leaf cell work.** Direct leaf ASTs delete LOCAL + cell-materialization branches when no cell dictionary exists. Generated + pre-hashed deletion routes through the semantic helper so iteration + continuation metadata is not bypassed. + +## Measurements + +The baseline is merged `main` at +`20fb9756cb65b55f29607aacc2ae5aa7ca80a2f4`, which contains the 0.35 runtime +and the 0.36/0.37 planning work. Each row is the median of three process +medians on an isolated CPU, using A/B, B/A, A/B ordering, seven warmups and 31 +checked samples. Python GC is disabled only during steady timing; Lua GC stays +active. Full process medians are in +[`speed_037.json`](../benchmarks/results/speed_037.json). + +| Target | Python | Baseline | 0.37 | Improvement | 0.37 / Python | +| --- | --- | ---: | ---: | ---: | ---: | +| Dense `next` traversal, 2,048 entries ×4 | 3.13.15 | 2,092.126 ms | 63.961 ms | **96.9% faster** | 450.84× | +| Dense `next` traversal, 2,048 entries ×4 | 3.14.7 | 1,992.279 ms | 59.119 ms | **97.0% faster** | 532.50× | +| Hash deletion sweep, 2,048 entries | 3.13.15 | 93.027 ms | 8.619 ms | **90.7% faster** | 17.62× | +| Hash deletion sweep, 2,048 entries | 3.14.7 | 94.614 ms | 7.699 ms | **91.9% faster** | 14.26× | +| 1,000 strict-float Python calls | 3.13.15 | 1.662 ms | 1.110 ms | **33.2% faster** | 13.21× | +| 1,000 strict-float Python calls | 3.14.7 | 1.551 ms | 1.073 ms | **30.8% faster** | 13.58× | +| 1,000 two-integer Python calls | 3.13.15 | 2.161 ms | 1.205 ms | **44.3% faster** | 13.04× | +| 1,000 two-integer Python calls | 3.14.7 | 2.064 ms | 1.195 ms | **42.1% faster** | 14.83× | + +The very large remaining traversal ratio is useful evidence: after eliminating +quadratic list construction/search, generic-for VM and host-function calls +dominate this reduced-contract comparison. It is not evidence that table +iteration still has quadratic scaling. Direct Python omits Lua fuel, iterator +protocol, multiple values, callability checks, and table semantics. + +## Native Lua 5.5 and Python lower bounds + +The reusable [`native_headroom.py`](../benchmarks/native_headroom.py) harness +also compares the nine established algorithms with native Lua 5.5 and direct +Python. Native execution uses the untyped version of the same Lua source +through Lupa 2.8's pinned `lupa.lua55` runtime. The Python functions preserve +the algorithm but omit Lua fuel, values, tables, metatables, stack/debug state, +GC, dynamic guards, and other runtime contracts; they are engineering lower +bounds, not equivalent implementations or promised attainable speeds. + +Each result is the median of three process medians on one isolated CPU. The +three implementations occupy each position once in a Latin-square order, with +seven warmups and 31 checked samples. Python GC is disabled during timing while +both Lua collectors remain active. Each sample crosses Python once to invoke +the workload; the explicit 1,000-call case crosses the boundary 1,000 times. + +| Workload | Python | LuaPyre | Native Lua 5.5 | Python lower bound | vs native | vs Python | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Typed arithmetic | 3.13.15 | 0.994 ms | 0.126 ms | 0.726 ms | 7.87× | 1.37× | +| Typed arithmetic | 3.14.7 | 0.722 ms | 0.126 ms | 0.489 ms | 5.73× | 1.48× | +| Typed branch | 3.13.15 | 2.215 ms | 0.263 ms | 1.460 ms | 8.43× | 1.52× | +| Typed branch | 3.14.7 | 1.856 ms | 0.301 ms | 1.143 ms | 6.16× | 1.62× | +| Recursive Fibonacci | 3.13.15 | 30.326 ms | 0.317 ms | 0.791 ms | 95.62× | 38.33× | +| Recursive Fibonacci | 3.14.7 | 24.270 ms | 0.324 ms | 0.620 ms | 74.87× | 39.17× | +| Binary trees | 3.13.15 | 102.174 ms | 1.751 ms | 2.106 ms | 58.37× | 48.51× | +| Binary trees | 3.14.7 | 86.256 ms | 1.761 ms | 1.981 ms | 48.98× | 43.53× | +| Sieve | 3.13.15 | 23.015 ms | 0.197 ms | 0.586 ms | 116.85× | 39.25× | +| Sieve | 3.14.7 | 19.681 ms | 0.194 ms | 0.509 ms | 101.48× | 38.67× | +| Table mix | 3.13.15 | 29.971 ms | 0.587 ms | 3.331 ms | 51.09× | 9.00× | +| Table mix | 3.14.7 | 22.066 ms | 0.586 ms | 2.508 ms | 37.66× | 8.80× | +| String build | 3.13.15 | 0.549 ms | 1.171 ms | 0.381 ms | **0.47×** | 1.44× | +| String build | 3.14.7 | 0.523 ms | 1.167 ms | 0.372 ms | **0.45×** | 1.41× | +| Spectral norm | 3.13.15 | 28.717 ms | 0.832 ms | 2.004 ms | 34.51× | 14.33× | +| Spectral norm | 3.14.7 | 23.595 ms | 0.836 ms | 2.407 ms | 28.21× | 9.80× | +| 1,000 Python calls | 3.13.15 | 0.675 ms | 0.200 ms | 0.051 ms | 3.38× | 13.30× | +| 1,000 Python calls | 3.14.7 | 0.598 ms | 0.197 ms | 0.041 ms | 3.04× | 14.70× | + +The results split the remaining work cleanly. Straight typed loops are already +within 1.37–1.62× of reduced-contract Python, so their 5.73–8.43× native-Lua +gap is primarily the ceiling of executing the loop as Python. Recursive frame +creation and Lua table semantics dominate Fibonacci, binary trees, sieve, and +spectral norm. The Python-to-Lua boundary is about 3× native Lua but already +far below the cost of 1,000 ordinary Python calls through the full LuaPyre +contract. String construction is the exception: LuaPyre is about 2.1–2.2× +faster than native Lua for this repeated-concatenation shape. + +Raw samples and per-process medians are retained in +[`native_headroom_037_313.json`](../benchmarks/results/native_headroom_037_313.json) +and +[`native_headroom_037_314.json`](../benchmarks/results/native_headroom_037_314.json). + +## Rejected and deferred work + +A conservative fixed-point constant analysis across whole-function blocks was +implemented as an experiment, corrected to invalidate loop-written registers, +and removed. `record_continuations` changed by about +1% on 3.13 and -7% on +3.14 in diagnostic runs, failing the cross-version 5% gate. + +Longer nested regions, general table-representation proofs, scalar internal +compiled-call interfaces, and lazy recursive activations remain deferred. They +need narrower recovery/effect contracts and independent measurements; 0.37 +does not label them complete. Direct bytecode generation remains out of scope. + +## Validation + +`benchmarks/speed_037_ab.py` retains all 0.36 cases and adds dense/hash +traversal and deletion probes. `inspect_codegen.py --suite 037` audits the new +scalar runners with adaptive disassembly; the final 3.13 and 3.14 audits are +[`codegen_037_313.json`](../benchmarks/results/codegen_037_313.json) and +[`codegen_037_314.json`](../benchmarks/results/codegen_037_314.json). +Deterministic tests cover cold and +warmed results, independent traversal, live value replacement, numeric key +normalization, invalid keys, deleted-successor chains, reinsertion, pre-hashed +deletion, scalar code shape, argument validation, JIT toggles, and the +successor-preserving generated deletion route. + +- All 592 Python tests pass on CPython 3.13.15 and 3.14.7. +- All 24 required unchanged official Lua 5.5.1 probes pass on both versions. +- Penlight passes 23/23, luatest 5/5, LuaCov scanner specs 24/24, and Are We + Fast Yet 11/11 on both versions. Established native-module and unsafe-I/O + exclusions remain unchanged. diff --git a/pyproject.toml b/pyproject.toml index fb6b59a..6324f90 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "luapyre" -version = "0.35.0a1" +version = "0.37.0a1" description = "A high-performance, sandboxed Lua 5.5 runtime for Python with optional gradual typing" readme = "README.md" requires-python = ">=3.13" diff --git a/src/luapyre/__init__.py b/src/luapyre/__init__.py index d6bc585..56d024b 100644 --- a/src/luapyre/__init__.py +++ b/src/luapyre/__init__.py @@ -27,4 +27,4 @@ "LuaTraceFrame", ] -__version__ = "0.35.0a1" +__version__ = "0.37.0a1" diff --git a/src/luapyre/dense_jit.py b/src/luapyre/dense_jit.py index d5dc20a..bbfdab3 100644 --- a/src/luapyre/dense_jit.py +++ b/src/luapyre/dense_jit.py @@ -10,6 +10,7 @@ generated_namespace, optimize_generated_ast, promote_constant_registers, + remove_empty_cell_branches, unwrap_scalar_return, ) from .range_analysis import analyze_integer_ranges @@ -184,10 +185,9 @@ def _compile_leaf(self, proto: Proto): exec(compile(tree, "", "exec"), namespace) direct_lines = ["def _jit_leaf_direct(vm, closure, args):"] direct_lines.append(" consts = closure.proto.constants") - if any(item.ins.op is Op.LOCAL for item in sequence): - direct_lines.append(" cells = {}") direct_lines.extend(lines[4:]) direct_tree = optimize_generated_ast(ast.parse("\n".join(direct_lines))) + direct_tree = remove_empty_cell_branches(direct_tree) direct_initial = { index: ast.parse( f"args[{index}] if {index} < len(args) else None", mode="eval" @@ -204,19 +204,23 @@ def _compile_leaf(self, proto: Proto): namespace, ) scalar_runner = None - if ( - proto.param_count == 1 - and proto.param_types - and proto.param_types[0].name == "integer" - and return_ins.ins.b == 1 - ): - scalar_lines = ["def _jit_leaf_scalar(vm, closure, value):"] + scalar_types = tuple(item.name for item in proto.param_types) + scalar_parameters = ( + scalar_types in (("integer",), ("float",), ("integer", "integer")) + ) + if scalar_parameters and return_ins.ins.b == 1: + names = ("value",) if proto.param_count == 1 else ("left", "right") + scalar_lines = [ + f"def _jit_leaf_scalar(vm, closure, {', '.join(names)}):" + ] scalar_lines.append(" consts = closure.proto.constants") - if any(item.ins.op is Op.LOCAL for item in sequence): - scalar_lines.append(" cells = {}") scalar_lines.extend(lines[4:]) scalar_tree = optimize_generated_ast(ast.parse("\n".join(scalar_lines))) - scalar_initial = {0: ast.Name(id="value", ctx=ast.Load())} + scalar_tree = remove_empty_cell_branches(scalar_tree) + scalar_initial = { + index: ast.Name(id=name, ctx=ast.Load()) + for index, name in enumerate(names) + } if proto.env_reg >= 0: scalar_initial[proto.env_reg] = ast.parse( "closure.env", mode="eval" diff --git a/src/luapyre/jit_codegen.py b/src/luapyre/jit_codegen.py index 0b55d13..029bea1 100644 --- a/src/luapyre/jit_codegen.py +++ b/src/luapyre/jit_codegen.py @@ -614,6 +614,30 @@ def unwrap_scalar_return(tree: ast.AST) -> ast.AST: return tree +class _RemoveEmptyCellBranches(ast.NodeTransformer): + def visit_If(self, node: ast.If): + node = self.generic_visit(node) + test = node.test + if ( + isinstance(test, ast.Compare) + and len(test.ops) == 1 + and isinstance(test.ops[0], ast.In) + and len(test.comparators) == 1 + and isinstance(test.comparators[0], ast.Name) + and test.comparators[0].id == "cells" + and not node.orelse + ): + return None + return node + + +def remove_empty_cell_branches(tree: ast.AST) -> ast.AST: + """Delete LOCAL cell materialization when a direct leaf has no cells.""" + tree = _RemoveEmptyCellBranches().visit(tree) + ast.fix_missing_locations(tree) + return tree + + def optimize_generated_ast(tree: ast.AST) -> ast.AST: tree = _InlineI64Assignments().visit(tree) tree = optimize_semantic_helpers(tree) diff --git a/src/luapyre/runtime.py b/src/luapyre/runtime.py index aa4d0f9..0105769 100644 --- a/src/luapyre/runtime.py +++ b/src/luapyre/runtime.py @@ -29,7 +29,7 @@ from .jit import DEOPT from .typed_parser import TypedParser from .typesys import ANY -from .values import MultiValue, i64, truthy +from .values import MultiValue, i64, static_value_type, truthy from .vm import HostFunction @@ -564,7 +564,7 @@ def _call_bound( return self._from_lua(result, return_type) def _make_bound_adapter(self, function, leaf_cache): - """Create the common fixed one-integer Python entry once per binding.""" + """Create a fixed scalar Python entry once per typed leaf binding.""" if ( not isinstance(function, Closure) or not hasattr(self.vm, "call_compiled_leaf") @@ -574,17 +574,119 @@ def _make_bound_adapter(self, function, leaf_cache): if ( not proto.jit_fully_typed or proto.is_vararg - or proto.param_count != 1 - or proto.param_types[0].name != "integer" + or tuple(item.name for item in proto.param_types) + not in (("integer",), ("float",), ("integer", "integer")) ): return None vm = self.vm minimum = -(1 << 63) maximum = (1 << 63) - 1 + parameter_types = tuple(item.name for item in proto.param_types) + + if parameter_types == ("integer",): + def call_integer(args, return_type, fuel): + if len(args) != 1 or type(args[0]) is not int: + return self._call_bound( + function, + args, + return_type=return_type, + fuel=fuel, + leaf_cache=leaf_cache, + ) + value = args[0] + if not minimum <= value <= maximum: + raise OverflowError("LuaInt result is outside signed 64-bit range") + if ( + not vm.jit.enabled + or vm._active_frames is not None + or vm.hooks_active() + or vm.max_frames < 2 + ): + return self._call_bound( + function, + args, + return_type=return_type, + fuel=fuel, + leaf_cache=leaf_cache, + ) + cached = leaf_cache[0] + if cached is not None and cached[0] is proto: + compiled = cached[1] + else: + compiled = vm.jit.get_leaf(proto) + leaf_cache[0] = (proto, compiled) + budget = vm.default_fuel if fuel is None else fuel + if compiled is None or budget < compiled.instruction_cost + 4: + return self._call_bound( + function, + args, + return_type=return_type, + fuel=fuel, + leaf_cache=leaf_cache, + ) + scalar_runner = compiled.scalar_runner + if scalar_runner is not None: + result = scalar_runner(vm, function, value) + if result is DEOPT: + return self._call_bound( + function, + args, + return_type=return_type, + fuel=fuel, + leaf_cache=leaf_cache, + ) + vm.jit.stats.leaf_executions += 1 + vm.jit.stats.leaf_frame_elisions += 1 + vm.jit.stats.python_direct_entries += 1 + vm.gc.safepoint() + if return_type is None and ( + result is None or type(result) in (bool, int, float) + ): + return result + if return_type is int and type(result) is int: + return result + return self._from_lua(result, return_type) + values = compiled.direct_runner(vm, function, args) + if values is DEOPT: + return self._call_bound( + function, + args, + return_type=return_type, + fuel=fuel, + leaf_cache=leaf_cache, + ) + vm.jit.stats.leaf_executions += 1 + vm.jit.stats.leaf_frame_elisions += 1 + vm.jit.stats.python_direct_entries += 1 + vm.gc.safepoint(values) + result = ( + None if not values + else values[0] if len(values) == 1 + else values + ) + if return_type is None and ( + result is None or type(result) in (bool, int, float) + ): + return result + if return_type is int and type(result) is int: + return result + return self._from_lua(result, return_type) + + return call_integer def call(args, return_type, fuel): - if len(args) != 1 or type(args[0]) is not int: + valid = len(args) == len(parameter_types) + if valid: + for index, (value, expected) in enumerate(zip(args, parameter_types)): + if (expected == "integer" and type(value) is not int) or ( + expected == "float" and type(value) is not float + ): + raise LuaRuntimeError( + f"argument {index + 1}: expected {expected}, " + f"got {static_value_type(value).name}" + ) + if not valid: return self._call_bound( function, args, @@ -592,9 +694,9 @@ def call(args, return_type, fuel): fuel=fuel, leaf_cache=leaf_cache, ) - value = args[0] - if not minimum <= value <= maximum: - raise OverflowError("LuaInt result is outside signed 64-bit range") + for value, expected in zip(args, parameter_types): + if expected == "integer" and not minimum <= value <= maximum: + raise OverflowError("LuaInt result is outside signed 64-bit range") if ( not vm.jit.enabled or vm._active_frames is not None @@ -625,7 +727,7 @@ def call(args, return_type, fuel): ) scalar_runner = compiled.scalar_runner if scalar_runner is not None: - result = scalar_runner(vm, function, value) + result = scalar_runner(vm, function, *args) if result is DEOPT: return self._call_bound( function, diff --git a/src/luapyre/stdlib.py b/src/luapyre/stdlib.py index 19e67a2..4bcce4f 100644 --- a/src/luapyre/stdlib.py +++ b/src/luapyre/stdlib.py @@ -139,16 +139,8 @@ def setmetatable(table, mt=_MISSING, *_ignored): def next_fn(table, key=None, *_ignored): if not isinstance(table, LuaTable): raise LuaRuntimeError("bad argument #1 to 'next' (table expected)") - items = list(table.items()) - if key is None: - return MultiValue(items[0]) if items else None - for index, (current, value) in enumerate(items): - if lua_equal(current, key): - return MultiValue(items[index + 1]) if index + 1 < len(items) else None - known_deleted, successor = table.successor_after_deleted(key) - if known_deleted: - return None if successor is None else MultiValue((successor, table.rawget(successor))) - raise LuaRuntimeError("invalid key to 'next'") + item = table.next_item(key) + return None if item is None else MultiValue(item) next_host = put("next", next_fn) diff --git a/src/luapyre/table.py b/src/luapyre/table.py index bfd8816..a04f105 100644 --- a/src/luapyre/table.py +++ b/src/luapyre/table.py @@ -8,6 +8,7 @@ _NUM = object() _STR = object() _OBJ = object() +_ITERATION_STATE = object() def _hash_key(key): @@ -46,6 +47,54 @@ def __init__(self): self._deleted_successors: dict[object, object | None] = {} self._reserved_bytes = 0 + def _ensure_iteration_index(self) -> None: + state = self._deleted_successors.get(_ITERATION_STATE) + if state is not None and state[0] == self.version: + return + keys = tuple(key for key, _value in self.items()) + positions = { + _hash_key(key): index for index, key in enumerate(keys) + } + self._deleted_successors[_ITERATION_STATE] = (self.version, keys, positions) + collector = self._gc_owner + if collector is not None and keys: + collector.account_bytes(40 * len(keys)) + + def _next_indexed_key(self, position: int): + keys = self._deleted_successors[_ITERATION_STATE][1] + while position < len(keys): + key = keys[position] + if self.rawhas(key): + return key + position += 1 + return None + + def next_item(self, key=None): + """Return the next raw entry without rebuilding the key order per call.""" + self._ensure_iteration_index() + if key is None: + successor = self._next_indexed_key(0) + else: + token = _hash_key(key) + position = self._deleted_successors[_ITERATION_STATE][2].get(token) + if position is None: + known_deleted, successor = self.successor_after_deleted(key) + if not known_deleted: + raise LuaRuntimeError("invalid key to 'next'") + elif self.rawhas(key) or token in self._deleted_successors: + successor = self._next_indexed_key(position + 1) + else: + raise LuaRuntimeError("invalid key to 'next'") + if successor is None: + return None + return successor, self.rawget(successor) + + def _remember_deleted_successor(self, token) -> None: + self._ensure_iteration_index() + position = self._deleted_successors[_ITERATION_STATE][2].get(token) + if position is not None: + self._deleted_successors[token] = self._next_indexed_key(position + 1) + def rawget(self, key): if type(key) is int and key >= 1: idx = key - 1 @@ -85,16 +134,21 @@ def rawset(self, key, value): h = _hash_key(key) if h is None: raise LuaRuntimeError("table index is nil") - if value is None and self.rawhas(key): - entries = list(self.items()) - for index, (current, _item) in enumerate(entries): - if _hash_key(current) == h: - successor = entries[index + 1][0] if index + 1 < len(entries) else None - self._deleted_successors[h] = successor - break - elif value is not None: + present = False + if value is None: + present = self.rawhas(key) + if present: + self._remember_deleted_successor(h) + else: self._deleted_successors.pop(h, None) self.version += 1 + if value is None and present: + # Deletion leaves the indexed order usable; missing keys are + # skipped and remain valid inputs to next(). + state = self._deleted_successors[_ITERATION_STATE] + self._deleted_successors[_ITERATION_STATE] = ( + self.version, state[1], state[2] + ) collector = self._gc_owner if collector is not None: collector.table_write_barrier(self, key, value) @@ -126,20 +180,19 @@ def rawset(self, key, value): def rawset_prehashed(self, key, token, value): """Set a compiler-validated non-array constant key.""" - if value is None and token in self.hash: - entries = list(self.items()) - for index, (current, _item) in enumerate(entries): - if _hash_key(current) == token: - successor = ( - entries[index + 1][0] - if index + 1 < len(entries) - else None - ) - self._deleted_successors[token] = successor - break - elif value is not None: + present = False + if value is None: + present = token in self.hash + if present: + self._remember_deleted_successor(token) + else: self._deleted_successors.pop(token, None) self.version += 1 + if value is None and present: + state = self._deleted_successors[_ITERATION_STATE] + self._deleted_successors[_ITERATION_STATE] = ( + self.version, state[1], state[2] + ) collector = self._gc_owner if collector is not None: collector.table_write_barrier(self, key, value) diff --git a/src/luapyre/typed_ir_function_jit.py b/src/luapyre/typed_ir_function_jit.py index 798d4bd..bcf2a00 100644 --- a/src/luapyre/typed_ir_function_jit.py +++ b/src/luapyre/typed_ir_function_jit.py @@ -409,10 +409,10 @@ def emit_forloop( if array_index is None: lines.extend( [ - f"{indent}{table_tmp}.version += 1", f"{indent}if {c} is None:", - f"{indent} {table_tmp}.hash.pop({token}, None)", + f"{indent} {table_tmp}.rawset_prehashed({key_expr}, {token}, None)", f"{indent}else:", + f"{indent} {table_tmp}.version += 1", f"{indent} {table_tmp}.hash[{token}] = ({key_expr}, {c})", ] ) diff --git a/src/luapyre/typed_ir_jit.py b/src/luapyre/typed_ir_jit.py index 02aeba5..61a7c42 100644 --- a/src/luapyre/typed_ir_jit.py +++ b/src/luapyre/typed_ir_jit.py @@ -247,10 +247,10 @@ def _emit_typed_ir_instruction( # compiled. Version changes still invalidate every read cache. out.extend( [ - f"{indent}{table_tmp}.version += 1", f"{indent}if {value} is None:", - f"{indent} {table_tmp}.hash.pop({token}, None)", + f"{indent} {table_tmp}.rawset_prehashed({key_expr}, {token}, None)", f"{indent}else:", + f"{indent} {table_tmp}.version += 1", f"{indent} {table_tmp}.hash[{token}] = ({key_expr}, {value})", ] ) diff --git a/tests/test_performance_037.py b/tests/test_performance_037.py new file mode 100644 index 0000000..f020e54 --- /dev/null +++ b/tests/test_performance_037.py @@ -0,0 +1,146 @@ +"""Correctness coverage for the accepted 0.37 performance paths.""" +from __future__ import annotations + +import importlib.util +import dis +from pathlib import Path +import sys + +import pytest + +from luapyre import LuaRuntime, LuaRuntimeError +from luapyre.table import LuaTable, _hash_key + + +_PATH = Path(__file__).resolve().parents[1] / "benchmarks" / "speed_037_ab.py" +sys.path.insert(0, str(_PATH.parent)) +_SPEC = importlib.util.spec_from_file_location("luapyre_speed_037", _PATH) +assert _SPEC is not None and _SPEC.loader is not None +probes = importlib.util.module_from_spec(_SPEC) +sys.modules[_SPEC.name] = probes +_SPEC.loader.exec_module(probes) + + +@pytest.mark.parametrize("name", probes.NEW_CASES) +def test_new_probe_reference_interpreter_and_warmed_jit(name): + case = probes.CASES[name] + _, interpreted, reference = probes.prepare(name, jit=False) + expected = interpreted() + assert probes.validate(case, expected) + assert probes.validate(case, reference()) + _, warmed, _ = probes.prepare(name, threshold=1) + for _ in range(4): + assert warmed() == expected + + +def test_next_index_supports_independent_traversals_and_live_values(): + table = LuaTable.from_sequence((10, 20, 30)) + first = table.next_item() + assert first == (1, 10) + assert table.next_item() == first + table.rawset(2, 99) + assert table.next_item(1) == (2, 99) + assert table.next_item(2) == (3, 30) + + +def test_next_index_preserves_deleted_key_chains_and_reinsertion(): + table = LuaTable() + for key in (b"a", b"b", b"c", b"d"): + table.rawset(key, key) + table.rawset(b"b", None) + table.rawset(b"c", None) + assert table.next_item(b"b") == (b"d", b"d") + assert table.next_item(b"c") == (b"d", b"d") + table.rawset(b"b", 42) + assert dict(table.items())[b"b"] == 42 + assert table.next_item(b"b") is None + + +def test_next_index_rejects_unknown_and_normalizes_numeric_keys(): + table = LuaTable.from_sequence((10, 20)) + assert table.next_item(1.0) == (2, 20) + with pytest.raises(LuaRuntimeError, match="invalid key"): + table.next_item(b"unknown") + + +def test_prehashed_deletion_uses_indexed_successor(): + table = LuaTable() + for key in (b"a", b"b", b"c"): + table.rawset_prehashed(key, _hash_key(key), key) + table.rawset_prehashed(b"a", _hash_key(b"a"), None) + assert table.successor_after_deleted(b"a") == (True, b"b") + + +@pytest.mark.parametrize("name,args", [ + ("python_leaf_local", (3,)), + ("python_leaf_float", (3.0,)), + ("python_leaf_binary", (3, 4)), +]) +def test_scalar_boundary_shapes_use_direct_entry(name, args): + case = probes.CASES[name] + runtime = LuaRuntime(jit_threshold=1) + function = runtime.execute_python("-- luapyre: typed\n" + case.source) + before = runtime.jit_stats.python_direct_entries + for _ in range(4): + assert function(*args) == case.reference(*args) + assert runtime.jit_stats.python_direct_entries > before + + +def test_float_and_binary_scalar_entries_keep_validation_and_live_jit_toggle(): + float_fn = LuaRuntime(jit_threshold=1).execute_python("""-- luapyre: typed +return function(value: float): float return value + 0.5 end +""") + assert float_fn(2.0) == 2.5 + with pytest.raises(LuaRuntimeError, match="expected float"): + float_fn(2) + + runtime = LuaRuntime(jit_threshold=1) + binary = runtime.execute_python("""-- luapyre: typed +return function(left: integer, right: integer): integer return left + right end +""") + assert binary(2, 3) == 5 + runtime.vm.jit.enabled = False + assert binary(4, 5) == 9 + runtime.vm.jit.enabled = True + with pytest.raises(LuaRuntimeError, match="expected integer"): + binary(2, False) + + +@pytest.mark.parametrize("name", [ + "python_leaf_local", "python_leaf_float", "python_leaf_binary", +]) +def test_scalar_entries_have_no_argument_or_result_containers(name): + case = probes.CASES[name] + runtime = LuaRuntime(jit_threshold=1) + function = runtime.execute_python("-- luapyre: typed\n" + case.source) + compiled = runtime.vm.jit.get_leaf(function.raw.proto) + assert compiled is not None and compiled.scalar_runner is not None + opnames = { + instruction.opname + for instruction in dis.get_instructions(compiled.scalar_runner) + } + assert not {"BUILD_LIST", "BUILD_MAP", "BUILD_TUPLE"} & opnames + assert "cells" not in compiled.scalar_runner.__code__.co_varnames + + +def test_warmed_prehashed_deletion_uses_successor_preserving_helper(monkeypatch): + original = LuaTable.rawset_prehashed + deletions = 0 + + def observed(table, key, token, value): + nonlocal deletions + if value is None: + deletions += 1 + return original(table, key, token, value) + + monkeypatch.setattr(LuaTable, "rawset_prehashed", observed) + runtime = LuaRuntime(jit_threshold=1) + remove = runtime.execute_python("""-- luapyre: typed +return function(t: table): table + t.a = nil + return t +end +""") + for _ in range(40): + assert remove({"a": 1, "b": 2}) == {"b": 2} + assert deletions > 0