From 254ea5aa0fca3cec18249d829468bdef74b4971d Mon Sep 17 00:00:00 2001 From: rbroderi Date: Mon, 14 Sep 2026 21:32:40 -0400 Subject: [PATCH] Implement 0.39 and corrected 0.40 performance tranches --- README.md | 25 +- benchmarks/README.md | 15 +- benchmarks/check_speed_generality.py | 71 + benchmarks/native_headroom.py | 34 + benchmarks/python_headroom.py | 195 +- benchmarks/results/headroom_040_controls.json | 4698 +++++++++++++++++ .../results/language_benchmarks_040_313.json | 2293 ++++++++ .../results/language_benchmarks_040_314.json | 2293 ++++++++ .../results/native_headroom_040_313.json | 803 +++ .../results/native_headroom_040_314.json | 803 +++ benchmarks/results/profile_040_plan.json | 1502 ++++++ benchmarks/results/speed_039.json | 27 + .../results/speed_040_string_control.json | 16 + benchmarks/speed_037_ab.py | 1 + benchmarks/speed_039_ab.py | 55 + benchmarks/speed_040_ab.py | 304 ++ benchmarks/vm_programs.py | 269 +- docs/language-benchmarks.md | 83 + docs/performance-generality-policy.md | 33 + docs/performance-roadmap-0.40.md | 36 + docs/speed-0.39.md | 71 + docs/speed-0.40.md | 39 + pyproject.toml | 2 +- src/luapyre/__init__.py | 2 +- src/luapyre/ast_jit.py | 201 +- src/luapyre/function_jit.py | 514 +- src/luapyre/gc.py | 28 +- src/luapyre/jit.py | 7 + src/luapyre/jitvm.py | 39 + src/luapyre/structured_jit.py | 2 + src/luapyre/table.py | 43 +- src/luapyre/typed_ir_jit.py | 27 + tests/test_language_benchmarks.py | 68 + tests/test_performance_039.py | 199 + tests/test_performance_040.py | 235 + 35 files changed, 14951 insertions(+), 82 deletions(-) create mode 100644 benchmarks/check_speed_generality.py create mode 100644 benchmarks/results/headroom_040_controls.json create mode 100644 benchmarks/results/language_benchmarks_040_313.json create mode 100644 benchmarks/results/language_benchmarks_040_314.json create mode 100644 benchmarks/results/native_headroom_040_313.json create mode 100644 benchmarks/results/native_headroom_040_314.json create mode 100644 benchmarks/results/profile_040_plan.json create mode 100644 benchmarks/results/speed_039.json create mode 100644 benchmarks/results/speed_040_string_control.json create mode 100644 benchmarks/speed_039_ab.py create mode 100644 benchmarks/speed_040_ab.py create mode 100644 docs/language-benchmarks.md create mode 100644 docs/performance-generality-policy.md create mode 100644 docs/performance-roadmap-0.40.md create mode 100644 docs/speed-0.39.md create mode 100644 docs/speed-0.40.md create mode 100644 tests/test_language_benchmarks.py create mode 100644 tests/test_performance_039.py create mode 100644 tests/test_performance_040.py diff --git a/README.md b/README.md index ef64d00..69f4f52 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ LuaPyre is a clean-slate Lua runtime written in Python. It targets **Lua 5.5.1** semantics, a sandbox-first embedding model, and optional gradual type annotations that feed runtime optimization without creating a second language/runtime. -**Python 3.13+** · **current pre-alpha: 0.38.0a1** +**Python 3.13+** · **current pre-alpha: 0.40.0a1** LuaPyre implements Lua 5.5.1 language semantics for its supported sandboxed embedding profile. The runtime is built around a register VM and explicit Lua frames, with a guarded tiered JIT that specializes proven hot paths and deoptimizes back to the same interpreter. @@ -194,6 +194,11 @@ proved dense primitive-write regions, and cheaper fresh record construction. See the [0.38 performance record](docs/speed-0.38.md) and [implementation roadmap](docs/performance-roadmap-0.38.md). +0.39 enters cached numeric loops immediately after `FORPREP` and specializes +proved hash-only integer-to-boolean regions without tagged keys or primitive GC +barriers. The typed Sieve workload is 64–67% faster than 0.38. See the +[0.39 performance record](docs/speed-0.39.md). + The [0.36 performance roadmap](docs/performance-roadmap-0.36.md) adds fresh Python-headroom measurements, 11 focused speed probes, and the next ordered work on scalar entry, cross-block facts, nested regions, tables, and recursion. @@ -363,11 +368,19 @@ It compares: - PUC Lua 5.5 through `lupa.lua55` - LuaJIT through `lupa.luajit21` (falling back to `lupa.luajit20` when appropriate) -The corpus contains `micro`, `typed`, and `algorithm` groups. Algorithm workloads include recursive Fibonacci, Sieve, binary trees, table mixing, string construction, and spectral norm. Where LuaPyre uses typed annotations, the native engines receive an equivalent standard-Lua spelling and all backends must produce the same result. +The corpus contains `micro`, `typed`, `algorithm`, and `language` groups. The +language group adds bounded n-body, Mandelbrot, spectral-norm, fannkuch-redux, +binary-trees, FASTA, k-nucleotide, and reverse-complement workloads. Output-heavy +cases use deterministic in-memory checksums. Where LuaPyre uses typed +annotations, the native engines receive an equivalent standard-Lua spelling +and all backends must produce the same result. The exact parameters, +adaptations, and pidigits portability decision are documented in +[`docs/language-benchmarks.md`](docs/language-benchmarks.md). ```bash python benchmarks/compare_runtimes.py --group typed --require-all python benchmarks/compare_runtimes.py --group algorithm --require-all +python benchmarks/compare_runtimes.py --group language --require-all ``` Use `--json PATH` for a versioned machine-readable report. The permanent **Four-way runtime benchmark** Actions workflow records text and JSON artifacts. See [`benchmarks/README.md`](benchmarks/README.md) for methodology and comparison guidance. @@ -385,6 +398,14 @@ LuaPyre-native `string.dump` output, cross-runtime PUC emission, and additional ## Performance roadmap +The corrected [0.40 performance roadmap](docs/performance-roadmap-0.40.md) +removes three whole-algorithm matrix/tree compilers and retains only reusable +typed-IR, scalar-entry, record, GC/accounting, and sparse-set improvements. +Published benchmark reports now include admission telemetry and classify paths +seen in fewer than three workloads as experimental. See the +[0.40 corrective record](docs/speed-0.40.md) and permanent +[performance generality policy](docs/performance-generality-policy.md). + 0.17 establishes the intended optimizer architecture: **a small statically typed IR compiler targeting optimized Python AST today, with the IR remaining reusable by future backends.** 1. exact table-dispatched interpreter as Tier 0 and universal deoptimization target diff --git a/benchmarks/README.md b/benchmarks/README.md index 516b062..a043414 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -25,6 +25,7 @@ python benchmarks/compare_runtimes.py --warmups 5 --repeats 15 --require-all --j python benchmarks/compare_runtimes.py --group micro --require-all python benchmarks/compare_runtimes.py --group typed --require-all python benchmarks/compare_runtimes.py --group algorithm --require-all +python benchmarks/compare_runtimes.py --group language --require-all ``` Groups may be supplied more than once. With no `--group`, the complete corpus runs. @@ -33,7 +34,8 @@ Groups may be supplied more than once. With no `--group`, the complete corpus ru - **micro** keeps the small VM/JIT kernels: arithmetic, tables, calls, branches, and coroutines. These are useful for locating dispatch and specialization overhead. - **typed** contains direct typed-vs-dynamic optimization probes. LuaPyre receives a source-equivalent file headed by `-- luapyre: typed`; Lua 5.5 and LuaJIT receive ordinary Lua because LuaPyre's annotations are intentionally a source extension. -- **algorithm** contains larger end-to-end programs: recursive Fibonacci, Sieve of Eratosthenes, binary trees, a table-mixing kernel, string construction, and spectral norm. Their LuaPyre variants use the fully typed source contract wherever useful, while the reference engines execute equivalent standard Lua. +- **algorithm** contains the project-specific larger programs: recursive Fibonacci, Sieve of Eratosthenes, a table-mixing kernel, and string construction. Their LuaPyre variants use the fully typed source contract wherever useful, while the reference engines execute equivalent standard Lua. +- **language** contains bounded, deterministic ports of n-body, Mandelbrot, spectral-norm, fannkuch-redux, binary-trees, FASTA, k-nucleotide, and reverse-complement. Output-heavy cases use in-memory checksums so every backend runs under the same sandbox. See [`docs/language-benchmarks.md`](../docs/language-benchmarks.md) for parameters, adaptations, and the pidigits disposition. A workload can therefore carry two spellings of the same algorithm: standard Lua for the reference engines and a fully typed LuaPyre spelling for the optimization target. Both must produce the same expected result. Floating-point workloads may declare a tight absolute tolerance; integer/string workloads remain exact. @@ -101,7 +103,7 @@ For performance comparisons, compare runs on the same runner class and Python ve timings, raw samples, and per-phase tier counters. The committed baseline uses three processes on each Python version; it does not claim a 0.36 speedup. See [`performance-roadmap-0.36.md`](../docs/performance-roadmap-0.36.md). -- `python_headroom.py` compares nine algorithms against reduced-contract direct +- `python_headroom.py` compares the headroom corpus against reduced-contract direct Python implementations. Use `PYTHONPATH=src PYTHONHASHSEED=0`, `--revision`, `--json`, and optionally `--python-first` or repeated `--case` selections. Every result is checked. Ratios include Lua semantic and representation @@ -111,7 +113,14 @@ For performance comparisons, compare runs on the same runner class and Python ve earlier comparison retained in the 0.34 roadmap. - `native_headroom.py` compares that same corpus three ways: LuaPyre, native Lua 5.5 through `lupa.lua55`, and the reduced-contract Python lower bounds. - It runs isolated processes and rotates implementation order. + It runs isolated processes and rotates implementation order. Both headroom + tools report per-workload fast-path admission counts; the native aggregate + labels paths seen in fewer than three distinct workloads as experimental. +- `speed_040_ab.py` is the corrected 0.40 generality corpus. It contains three + structurally different scalar DAGs, nested reductions, and sparse-set uses. + Run the unchanged file on baseline and candidate revisions, then use + `check_speed_generality.py`; a group passes only when all three cases admit + the intended path and improve over baseline. - `inspect_codegen.py` captures final generated AST/source, selects hot generated functions using checked profiled executions, and records generic and warmed adaptive opcode counts plus pooled frame/register-slot counts. diff --git a/benchmarks/check_speed_generality.py b/benchmarks/check_speed_generality.py new file mode 100644 index 0000000..faa5b73 --- /dev/null +++ b/benchmarks/check_speed_generality.py @@ -0,0 +1,71 @@ +"""Enforce the three-program performance/admission gate on two A/B reports.""" +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +from speed_040_ab import GENERALITY_GROUPS + + +def evaluate( + baseline: dict, candidate: dict, *, minimum_improvement: float = 0.02 +) -> dict: + groups = {} + for path, names in GENERALITY_GROUPS.items(): + cases = {} + for name in names: + old = baseline["workloads"][name]["steady"]["luapyre"]["median_ms"] + new = candidate["workloads"][name]["steady"]["luapyre"]["median_ms"] + admissions = candidate["workloads"][name].get( + "fast_path_admissions", {} + ).get(path, 0) + cases[name] = { + "baseline_ms": old, + "candidate_ms": new, + "improvement_fraction": old / new - 1.0, + "admissions": admissions, + "passes": admissions > 0 and old / new - 1.0 >= minimum_improvement, + } + passed = sum(case["passes"] for case in cases.values()) + groups[path] = { + "cases": cases, + "passing_cases": passed, + "status": "accepted" if passed >= 3 else "experimental", + } + return { + "schema_version": 1, + "minimum_improvement_fraction": minimum_improvement, + "groups": groups, + "accepted": all(group["status"] == "accepted" for group in groups.values()), + } + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("baseline", type=Path) + parser.add_argument("candidate", type=Path) + parser.add_argument("--json", type=Path) + parser.add_argument( + "--minimum-improvement", + type=float, + default=0.02, + help="minimum improvement required in each case (default: 0.02)", + ) + args = parser.parse_args() + if args.minimum_improvement <= 0: + parser.error("--minimum-improvement must be positive") + report = evaluate( + json.loads(args.baseline.read_text()), + json.loads(args.candidate.read_text()), + minimum_improvement=args.minimum_improvement, + ) + rendered = json.dumps(report, indent=2) + "\n" + if args.json is not None: + args.json.write_text(rendered) + print(rendered, end="") + return 0 if report["accepted"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/benchmarks/native_headroom.py b/benchmarks/native_headroom.py index eb51619..e217be1 100644 --- a/benchmarks/native_headroom.py +++ b/benchmarks/native_headroom.py @@ -103,6 +103,9 @@ def run_worker(names, order, warmups, repeats): for key, value in before.items() if type(value) is int and after[key] != value } + row["fast_path_admissions"] = dict( + luapyre_runtime.jit_stats.fast_path_admissions + ) results[name] = row return { "python": platform.python_version(), @@ -134,7 +137,37 @@ def aggregate(revision, runs, warmups, repeats): row["native_lua55"]["median_ms"] / row["python_lower_bound"]["median_ms"] ) + row["fast_path_admissions"] = { + path: max( + run["workloads"][name].get("fast_path_admissions", {}).get(path, 0) + for run in runs + ) + for path in sorted({ + path + for run in runs + for path in run["workloads"][name].get("fast_path_admissions", {}) + }) + } results[name] = row + coverage = {} + paths = sorted({ + path + for run in runs + for workload in run["workloads"].values() + for path in workload.get("fast_path_admissions", {}) + }) + for path in paths: + workloads = sorted({ + name + for run in runs + for name, row in run["workloads"].items() + if row.get("fast_path_admissions", {}).get(path, 0) > 0 + }) + coverage[path] = { + "workloads": workloads, + "workload_count": len(workloads), + "status": "established" if len(workloads) >= 3 else "experimental", + } return { "schema_version": 1, "revision": revision, @@ -155,6 +188,7 @@ def aggregate(revision, runs, warmups, repeats): "attainable-speed guarantees or mathematical lower bounds." ), "runs": runs, + "fast_path_coverage": coverage, "workloads": results, } diff --git a/benchmarks/python_headroom.py b/benchmarks/python_headroom.py index daf8ada..4bda4ad 100644 --- a/benchmarks/python_headroom.py +++ b/benchmarks/python_headroom.py @@ -145,6 +145,194 @@ def multiply_at_av(x, n): return math.sqrt(vbv / vv) +def n_body(): + pi = 3.141592653589793 + solar_mass = 4.0 * pi * pi + days_per_year = 365.24 + bodies = [ + [0.0, 0.0, 0.0, 0.0, 0.0, 0.0, solar_mass], + [4.841431442464721, -1.1603200440274284, -0.10362204447112311, + 0.001660076642744037 * days_per_year, + 0.007699011184197404 * days_per_year, + -0.0000690460016972063 * days_per_year, + 0.0009547919384243266 * solar_mass], + [8.34336671824458, 4.124798564124305, -0.4035234171143214, + -0.002767425107268624 * days_per_year, + 0.004998528012349172 * days_per_year, + 0.00002304172975737639 * days_per_year, + 0.0002858859806661308 * solar_mass], + [12.894369562139131, -15.111151401698631, -0.22330757889265573, + 0.002964601375647616 * days_per_year, + 0.0023784717395948095 * days_per_year, + -0.000029658956854023756 * days_per_year, + 0.00004366244043351563 * solar_mass], + [15.379697114850917, -25.919314609987964, 0.17925877295037118, + 0.0026806777249038932 * days_per_year, + 0.001628241700382423 * days_per_year, + -0.00009515922545197159 * days_per_year, + 0.000051513890204661145 * solar_mass], + ] + px = py = pz = 0.0 + for body in bodies: + px += body[3] * body[6] + py += body[4] * body[6] + pz += body[5] * body[6] + bodies[0][3] = -px / solar_mass + bodies[0][4] = -py / solar_mass + bodies[0][5] = -pz / solar_mass + for _ in range(1000): + for i in range(len(bodies) - 1): + bi = bodies[i] + for j in range(i + 1, len(bodies)): + bj = bodies[j] + dx, dy, dz = bi[0] - bj[0], bi[1] - bj[1], bi[2] - bj[2] + distance2 = dx * dx + dy * dy + dz * dz + magnitude = 0.01 / (distance2 * math.sqrt(distance2)) + bi[3] -= dx * bj[6] * magnitude + bi[4] -= dy * bj[6] * magnitude + bi[5] -= dz * bj[6] * magnitude + bj[3] += dx * bi[6] * magnitude + bj[4] += dy * bi[6] * magnitude + bj[5] += dz * bi[6] * magnitude + for body in bodies: + body[0] += 0.01 * body[3] + body[1] += 0.01 * body[4] + body[2] += 0.01 * body[5] + energy = 0.0 + for i, bi in enumerate(bodies): + energy += 0.5 * bi[6] * (bi[3] * bi[3] + bi[4] * bi[4] + bi[5] * bi[5]) + for bj in bodies[i + 1:]: + dx, dy, dz = bi[0] - bj[0], bi[1] - bj[1], bi[2] - bj[2] + energy -= bi[6] * bj[6] / math.sqrt(dx * dx + dy * dy + dz * dz) + return energy + + +def mandelbrot(): + size = 40 + checksum = 0 + for y in range(size): + ci = 2.0 * y / size - 1.0 + byte = bits = 0 + for x in range(size): + cr = 2.0 * x / size - 1.5 + zr = zi = tr = ti = 0.0 + iterations = 0 + while iterations < 50 and tr + ti <= 4.0: + zi = 2.0 * zr * zi + ci + zr = tr - ti + cr + tr, ti = zr * zr, zi * zi + iterations += 1 + byte = byte * 2 + int(tr + ti <= 4.0) + bits += 1 + if bits == 8: + checksum = (checksum * 131 + byte) % 2147483647 + byte = bits = 0 + return checksum + + +def fannkuch_redux(): + n = 7 + perm1 = list(range(n)) + count = [0] * (n + 1) + max_flips, checksum, sign, r = 0, 0, 1, n + while True: + while r != 1: + count[r - 1] = r + r -= 1 + perm = perm1.copy() + flips = 0 + k = perm[0] + while k != 0: + perm[:k + 1] = reversed(perm[:k + 1]) + flips += 1 + k = perm[0] + checksum += sign * flips + max_flips = max(max_flips, flips) + while True: + if r == n: + return checksum * 100 + max_flips + first = perm1[0] + perm1[:r] = perm1[1:r + 1] + perm1[r] = first + count[r] -= 1 + if count[r] > 0: + sign = -sign + break + r += 1 + + +def fasta(): + alu = b"GGCCGGGCGCGGTGGCTCACGCCTGTAATCCCAGCACTTTGGGAGGCCGAGGCGGGCGGATCACCTGAGGTCAGGAGTTCGAGACCAGCCTGGCCAACATGGTGAAACCCCGTCTCTACTAAAAATACAAAAATTAGCCGGGCGTGGTGGCGCGCGCCTGTAATCCCAGCTACTCGGGAGGCTGAGGCAGGAGAATCGCTTGAACCCGGGAGGCGGAGGTTGCAGTGAGCCGAGATCGCGCCACTGCACTCCAGCCTGGGCGACAGAGCGAGACTCCGTCTCAAAAA" + iub = [(b"a", 97, .27), (b"c", 99, .12), (b"g", 103, .12), + (b"t", 116, .27), (b"B", 66, .02), (b"D", 68, .02), + (b"H", 72, .02), (b"K", 75, .02), (b"M", 77, .02), + (b"N", 78, .02), (b"R", 82, .02), (b"S", 83, .02), + (b"V", 86, .02), (b"W", 87, .02), (b"Y", 89, .02)] + homo = [(b"a", 97, .3029549426680), (b"c", 99, .1979883004921), + (b"g", 103, .1975473066391), (b"t", 116, .3015094502008)] + + def cumulative(dist): + total = 0.0 + out = [] + for char, code, probability in dist: + total += probability + out.append((char, code, total)) + return out + + iub, homo = cumulative(iub), cumulative(homo) + seed = 42 + + def pick(dist): + nonlocal seed + seed = (seed * 3877 + 29573) % 139968 + value = seed / 139968 + for item in dist: + if value < item[2]: + return item + return dist[-1] + + chunks = [] + checksum = 0 + for i in range(1000): + char = alu[i % len(alu):i % len(alu) + 1] + chunks.append(char) + checksum += char[0] + for dist, count_ in ((iub, 1500), (homo, 2500)): + for _ in range(count_): + char, code, _ = pick(dist) + chunks.append(char) + checksum += code + output = b"".join(chunks) + return len(output) * 1000000 + checksum + + +def k_nucleotide(): + motif = b"GGTATTTTAATTTATAGTGGTAAGATATTAAGATAATATTTGGTGGTAGTTTTAATGTGTAA" + sequence = motif * 20 + checksum = 0 + for k in (1, 2, 3, 4, 6, 12, 18): + counts = {} + for i in range(len(sequence) - k + 1): + key = sequence[i:i + k] + counts[key] = counts.get(key, 0) + 1 + checksum = (checksum * 131 + counts.get(sequence[:k], 0)) % 2147483647 + checksum = (checksum * 131 + counts.get(sequence[6:6 + k], 0)) % 2147483647 + return checksum + + +def reverse_complement(): + sequence = b"ACGTUMRWSYKVHDBN" * 250 + complement = dict(zip(b"ACGTUMRWSYKVHDBN", b"TGCAAKYWSRMBDHVN")) + output = bytearray() + for i in range(len(sequence) - 1, -1, -1): + output.append(complement[sequence[i]]) + reversed_ = bytes(output) + checksum = 0 + for i, byte in enumerate(reversed_, 1): + checksum = (checksum + i * byte) % 2147483647 + return checksum + + def python_calls_1000(): def bump(value): return value + 1 @@ -157,7 +345,9 @@ def bump(value): REFERENCES = { fn.__name__: fn for fn in ( typed_arith, typed_branch, fib_recursive, binary_trees, sieve, - table_mix, string_build, spectral_norm, python_calls_1000, + table_mix, string_build, spectral_norm, n_body, mandelbrot, + fannkuch_redux, fasta, k_nucleotide, reverse_complement, + python_calls_1000, ) } @@ -237,6 +427,9 @@ def main(): key: value - before[key] for key, value in after.items() if type(value) is int and value != before[key] } + row["fast_path_admissions"] = dict( + runtime.jit_stats.fast_path_admissions + ) row["ratio"] = row["luapyre"]["median_ms"] / row["python_reference"]["median_ms"] results[name] = row print(f"{platform.python_version()} {name}: {row['ratio']:.2f}x Python reference", flush=True) diff --git a/benchmarks/results/headroom_040_controls.json b/benchmarks/results/headroom_040_controls.json new file mode 100644 index 0000000..1cc9255 --- /dev/null +++ b/benchmarks/results/headroom_040_controls.json @@ -0,0 +1,4698 @@ +{ + "interpretation": "Reduced-contract same-algorithm Python references are engineering lower bounds, not guaranteed attainable speeds.", + "method": "Three isolated pinned processes per CPython version; alternating implementation order; 7 warmups and 31 checked samples; Python GC disabled only during steady timing", + "revision": "0.40.0a1", + "schema_version": 1, + "versions": [ + { + "affinity_cpu": 0, + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.13.15", + "repeats": 31, + "runs": [ + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.13.15", + "python_first": false, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 22.504055, + "samples_ms": [ + 23.407628, + 23.324828, + 23.296064, + 22.851534, + 22.313036, + 22.05714, + 22.539357, + 22.141184, + 22.045503, + 21.946319, + 23.07411, + 22.039675, + 22.251235, + 22.038994, + 22.594127, + 22.229493, + 22.346905, + 21.79671, + 23.881163, + 25.410811, + 22.696206, + 22.504055, + 22.408105, + 22.170076, + 26.981311, + 22.293988, + 22.456396, + 22.590001, + 22.76654, + 25.80565, + 23.65469 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 2.082579, + "samples_ms": [ + 2.140685, + 2.181014, + 2.112293, + 2.212209, + 2.232279, + 2.082579, + 2.121046, + 2.097692, + 2.187794, + 2.082038, + 2.122559, + 2.091533, + 2.12366, + 2.075159, + 2.138372, + 2.108507, + 2.060627, + 2.071894, + 2.326927, + 2.086345, + 2.078604, + 2.072355, + 2.078704, + 2.059426, + 2.080626, + 2.073307, + 2.080307, + 2.054769, + 2.080958, + 2.058013, + 2.072926 + ] + }, + "ratio": 10.805858985421443 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 23.55869, + "samples_ms": [ + 23.328014, + 23.526315, + 24.151841, + 23.614815, + 23.914323, + 23.275497, + 23.168571, + 23.194268, + 23.413239, + 23.272443, + 23.977425, + 23.528879, + 23.774158, + 23.29748, + 23.451546, + 25.015506, + 24.966905, + 23.567315, + 23.455051, + 23.584721, + 24.199553, + 23.219783, + 23.55869, + 23.954209, + 23.529987, + 23.459654, + 24.037211, + 23.904496, + 26.335796, + 23.988298, + 23.433967 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.774133, + "samples_ms": [ + 0.848483, + 0.808163, + 0.774133, + 0.757359, + 0.780083, + 0.840731, + 0.876894, + 0.80556, + 0.77166, + 0.761315, + 0.822504, + 0.874161, + 0.797589, + 0.77168, + 0.761856, + 0.772882, + 0.783537, + 0.779271, + 0.75784, + 0.770228, + 0.769427, + 0.787664, + 0.776037, + 0.760384, + 0.772271, + 0.770789, + 0.793102, + 0.758471, + 0.773642, + 0.773232, + 0.775846 + ] + }, + "ratio": 30.43235464706969 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.672054, + "samples_ms": [ + 0.672154, + 0.670272, + 0.642822, + 0.657813, + 0.658294, + 0.698803, + 0.677351, + 0.670542, + 0.863775, + 0.658454, + 0.654078, + 0.787192, + 0.745192, + 0.672054, + 0.660728, + 0.644234, + 0.657903, + 0.726294, + 0.704141, + 0.660898, + 0.654158, + 0.641149, + 0.715177, + 0.937453, + 0.686045, + 0.638205, + 0.654478, + 0.718442, + 0.681678, + 0.731712, + 0.691072 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.052848, + "samples_ms": [ + 0.052567, + 0.052777, + 0.052386, + 0.052577, + 0.066638, + 0.054871, + 0.052537, + 0.05421, + 0.052487, + 0.05447, + 0.052687, + 0.05452, + 0.051966, + 0.054389, + 0.052397, + 0.053799, + 0.052517, + 0.05473, + 0.052848, + 0.05424, + 0.052567, + 0.054269, + 0.052487, + 0.06818, + 0.052838, + 0.05475, + 0.053178, + 0.05433, + 0.052797, + 0.053959, + 0.052627 + ] + }, + "ratio": 12.716734786557677 + }, + "sieve": { + "luapyre": { + "median_ms": 8.503432, + "samples_ms": [ + 8.729825, + 9.632516, + 8.759658, + 8.776993, + 8.415584, + 9.394056, + 8.448853, + 8.451036, + 8.492937, + 8.381735, + 8.499166, + 8.60459, + 8.620194, + 8.503432, + 8.397467, + 8.661494, + 8.427881, + 8.416195, + 8.41312, + 8.441401, + 8.598261, + 8.622727, + 8.420181, + 8.729024, + 8.677147, + 8.422814, + 8.659361, + 8.400322, + 8.524523, + 8.932381, + 8.450375 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.5871, + "samples_ms": [ + 0.588081, + 0.570205, + 0.586969, + 0.572839, + 0.587401, + 0.605888, + 0.575653, + 0.588432, + 0.572879, + 0.584796, + 0.578377, + 0.650313, + 0.624905, + 0.572979, + 0.58705, + 0.571757, + 0.590145, + 0.58721, + 0.57405, + 0.614119, + 0.574982, + 0.590154, + 0.590585, + 0.572498, + 0.5871, + 0.591636, + 0.588251, + 0.573249, + 0.592988, + 0.589002, + 0.572388 + ] + }, + "ratio": 14.483788111054336 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.733213, + "samples_ms": [ + 3.547515, + 2.742837, + 2.715637, + 2.726553, + 2.730669, + 2.864384, + 2.820691, + 2.827421, + 2.732431, + 2.733213, + 2.671993, + 2.768995, + 2.972293, + 3.057207, + 2.988236, + 2.819739, + 2.700054, + 2.652224, + 2.716438, + 2.783025, + 2.662939, + 2.69776, + 3.094121, + 2.833129, + 2.693794, + 3.251321, + 2.844556, + 2.701006, + 2.691471, + 2.703929, + 2.668958 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.043041, + "samples_ms": [ + 2.009453, + 2.022332, + 2.036432, + 2.035721, + 2.035461, + 2.038826, + 2.005757, + 2.029111, + 2.124581, + 2.010515, + 2.017905, + 2.094247, + 2.047909, + 4.970579, + 2.747103, + 2.304885, + 2.153804, + 2.086446, + 2.079055, + 2.024274, + 2.038024, + 2.065875, + 2.025035, + 2.043041, + 2.116089, + 2.010995, + 2.049682, + 2.066476, + 2.332335, + 2.061839, + 1.996885 + ] + }, + "ratio": 1.3378160301237225 + }, + "string_build": { + "luapyre": { + "median_ms": 0.581091, + "samples_ms": [ + 0.554402, + 0.599548, + 0.550996, + 0.726424, + 0.591265, + 0.55313, + 0.56706, + 0.656962, + 0.557436, + 0.562603, + 1.415853, + 0.657212, + 1.393241, + 0.610384, + 0.562333, + 0.568332, + 0.597324, + 0.597325, + 0.583034, + 0.54654, + 0.557336, + 0.542645, + 0.560151, + 0.591466, + 0.599387, + 0.581091, + 0.544367, + 0.557676, + 0.540611, + 0.650432, + 0.638735 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.398093, + "samples_ms": [ + 0.388099, + 0.542594, + 0.387587, + 0.405704, + 0.396881, + 0.384883, + 0.397422, + 0.381528, + 0.380797, + 0.394087, + 0.638645, + 0.519661, + 0.447335, + 0.42989, + 0.385435, + 0.401388, + 0.384333, + 0.432693, + 0.408568, + 0.427276, + 0.412043, + 0.385565, + 0.398093, + 0.39617, + 0.380798, + 0.444071, + 0.388769, + 0.384783, + 0.418994, + 0.436479, + 0.419645 + ] + }, + "ratio": 1.459686555654081 + }, + "table_mix": { + "luapyre": { + "median_ms": 31.279617, + "samples_ms": [ + 33.237633, + 33.120302, + 32.023225, + 58.855158, + 31.636078, + 31.9037, + 32.431132, + 32.574532, + 31.217696, + 30.374501, + 30.580172, + 31.129616, + 30.497771, + 31.012084, + 31.279617, + 33.268058, + 32.462839, + 31.733911, + 30.961861, + 30.762369, + 31.269681, + 31.130588, + 30.368082, + 30.481528, + 31.342078, + 30.343325, + 32.154297, + 31.940143, + 31.062208, + 30.973808, + 31.358331 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 3.436252, + "samples_ms": [ + 3.517101, + 3.472434, + 3.357648, + 3.389163, + 3.363606, + 4.496664, + 3.372889, + 3.708271, + 3.436252, + 3.386939, + 3.425536, + 3.367992, + 3.826914, + 3.629406, + 3.594463, + 3.475229, + 3.568887, + 3.35914, + 3.64676, + 3.444784, + 3.329596, + 3.666009, + 3.506114, + 3.430454, + 3.342224, + 3.588675, + 3.335574, + 3.332721, + 3.361042, + 3.603577, + 3.422412 + ] + }, + "ratio": 9.102829769178744 + }, + "typed_arith": { + "luapyre": { + "median_ms": 1.013655, + "samples_ms": [ + 1.052472, + 1.002438, + 1.007136, + 1.071089, + 1.019634, + 1.002339, + 1.015137, + 1.014416, + 1.027436, + 1.011252, + 1.000636, + 1.007936, + 1.013655, + 1.005113, + 1.000215, + 1.000536, + 1.05803, + 1.007466, + 0.999074, + 0.999614, + 1.027805, + 1.024701, + 0.997512, + 1.007867, + 1.020796, + 1.387953, + 1.321736, + 1.033755, + 1.012163, + 1.087162, + 1.014476 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.734055, + "samples_ms": [ + 0.779992, + 0.761095, + 0.852409, + 0.755537, + 0.777339, + 0.754766, + 0.751961, + 0.747525, + 0.766733, + 0.742477, + 0.734055, + 0.704352, + 0.74417, + 0.775085, + 0.72999, + 0.700076, + 0.717141, + 0.719724, + 0.792381, + 0.719794, + 0.731201, + 0.719013, + 0.708798, + 0.715338, + 0.777318, + 0.731772, + 0.702659, + 0.720155, + 0.72373, + 0.808945, + 0.718593 + ] + }, + "ratio": 1.3808978891227497 + }, + "typed_branch": { + "luapyre": { + "median_ms": 2.196998, + "samples_ms": [ + 2.169568, + 2.313479, + 2.196919, + 2.173915, + 2.249165, + 2.140285, + 2.195186, + 2.178952, + 2.248564, + 2.166363, + 2.196998, + 2.208856, + 2.168657, + 2.247252, + 2.21894, + 2.198942, + 2.178611, + 2.201935, + 2.164721, + 2.201715, + 2.141778, + 2.17749, + 2.219321, + 2.190949, + 2.401397, + 2.211359, + 2.262514, + 2.380888, + 2.275253, + 2.182968, + 2.169869 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.489011, + "samples_ms": [ + 1.448703, + 1.520968, + 1.530983, + 1.492897, + 1.450104, + 1.470995, + 1.498615, + 1.445688, + 1.472787, + 1.459898, + 1.499847, + 1.496032, + 1.472688, + 1.469774, + 1.503202, + 1.472457, + 1.459988, + 1.461491, + 1.456994, + 1.45387, + 1.489011, + 1.456714, + 1.494079, + 1.446148, + 2.787453, + 1.599123, + 2.565489, + 1.85686, + 1.581587, + 1.764425, + 1.72672 + ] + }, + "ratio": 1.4754746606976037 + } + } + }, + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.13.15", + "python_first": true, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 22.890352, + "samples_ms": [ + 23.8783, + 22.468224, + 22.002801, + 22.887238, + 23.562267, + 22.216374, + 47.614322, + 26.616081, + 22.518438, + 23.10048, + 22.351272, + 23.16147, + 22.954085, + 22.749316, + 23.083886, + 22.613547, + 22.991009, + 22.890352, + 22.617973, + 22.460873, + 23.336436, + 22.374536, + 24.378252, + 26.472481, + 23.470963, + 23.142742, + 22.754442, + 22.106184, + 23.129753, + 22.878705, + 22.270254 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 2.168307, + "samples_ms": [ + 2.059588, + 2.089582, + 2.165573, + 3.010031, + 2.111945, + 2.420547, + 2.242026, + 2.108019, + 2.09525, + 2.141938, + 2.433747, + 2.539571, + 2.257888, + 2.850587, + 2.098976, + 2.100468, + 2.118935, + 2.091104, + 2.168307, + 2.047741, + 2.541164, + 2.213444, + 2.21056, + 2.177731, + 2.231941, + 2.30006, + 2.133426, + 2.236217, + 2.021051, + 2.18401, + 2.065557 + ] + }, + "ratio": 10.556785547434012 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 23.751985, + "samples_ms": [ + 24.204368, + 24.435466, + 23.937407, + 23.700399, + 23.707499, + 24.021931, + 23.185124, + 23.97402, + 23.656625, + 23.86453, + 23.555087, + 23.372479, + 23.881325, + 24.111712, + 23.544481, + 23.297028, + 23.855758, + 23.613962, + 24.490347, + 23.496601, + 23.686589, + 23.231412, + 24.665424, + 23.730263, + 28.981908, + 24.652785, + 24.006908, + 23.395763, + 23.077747, + 24.51958, + 23.751985 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.770668, + "samples_ms": [ + 0.760133, + 0.804768, + 0.770879, + 0.786051, + 0.770668, + 0.760073, + 0.768775, + 0.773823, + 0.801203, + 0.767965, + 1.61181, + 0.773413, + 0.853429, + 0.742718, + 0.758661, + 0.768696, + 0.755016, + 0.797027, + 0.772992, + 0.755276, + 0.756057, + 0.753464, + 0.742708, + 0.823826, + 0.755727, + 0.753523, + 0.742116, + 0.804788, + 0.995838, + 1.650376, + 0.816606 + ] + }, + "ratio": 30.819996418691318 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.661439, + "samples_ms": [ + 0.868902, + 0.905796, + 0.735647, + 0.690822, + 0.704983, + 0.663952, + 0.65537, + 1.001627, + 0.643893, + 0.68965, + 0.654328, + 0.643172, + 0.656692, + 0.678083, + 0.879939, + 0.647658, + 0.665534, + 0.661439, + 0.64139, + 0.756127, + 0.684813, + 0.644093, + 0.656992, + 0.652856, + 0.657553, + 0.714917, + 0.671593, + 0.64211, + 0.658885, + 0.657492, + 0.660978 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.051155, + "samples_ms": [ + 0.051316, + 0.051154, + 0.050634, + 0.050885, + 0.050634, + 0.051084, + 0.053077, + 0.050944, + 0.050534, + 0.051025, + 0.050544, + 0.051566, + 0.050634, + 0.051135, + 0.081129, + 0.051565, + 0.051005, + 0.051405, + 0.051125, + 0.051185, + 0.051024, + 0.051345, + 0.051155, + 0.051215, + 0.051165, + 0.051306, + 0.051315, + 0.051435, + 0.051165, + 0.051385, + 0.050935 + ] + }, + "ratio": 12.930094809891507 + }, + "sieve": { + "luapyre": { + "median_ms": 8.582626, + "samples_ms": [ + 8.442471, + 8.384957, + 8.335795, + 8.341713, + 8.610388, + 9.371322, + 8.365128, + 8.383886, + 8.611449, + 8.372218, + 8.355394, + 8.75578, + 8.655023, + 8.460889, + 8.408371, + 8.526254, + 8.660721, + 8.603968, + 8.351939, + 8.582626, + 8.676304, + 8.427189, + 8.52327, + 9.290094, + 9.28112, + 8.936986, + 8.787146, + 9.481133, + 9.61525, + 8.532513, + 8.63229 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.58039, + "samples_ms": [ + 0.563105, + 0.595132, + 0.567111, + 0.575273, + 0.567852, + 0.581402, + 0.58684, + 0.571818, + 0.597265, + 0.564066, + 0.579749, + 0.578478, + 0.581833, + 0.581292, + 0.563606, + 0.60017, + 0.567472, + 0.58039, + 0.576054, + 0.572529, + 0.580611, + 0.580481, + 0.583535, + 0.570826, + 0.596664, + 0.584196, + 0.568503, + 0.579369, + 0.58692, + 0.58699, + 0.583165 + ] + }, + "ratio": 14.787687589379555 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.722357, + "samples_ms": [ + 3.562948, + 2.716529, + 2.667027, + 2.93608, + 3.021865, + 2.665253, + 2.670121, + 2.657753, + 3.817681, + 3.342866, + 2.713033, + 2.788704, + 2.907638, + 2.783076, + 2.721716, + 2.722357, + 2.836234, + 2.788363, + 2.712443, + 2.655299, + 2.655649, + 2.677991, + 2.791828, + 2.659725, + 2.672084, + 2.78562, + 2.849584, + 2.741405, + 2.783717, + 2.675208, + 2.661958 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.008091, + "samples_ms": [ + 2.012857, + 2.166803, + 2.008091, + 2.068469, + 1.996834, + 2.045887, + 1.992168, + 2.04192, + 1.992588, + 2.099264, + 1.987932, + 2.045565, + 2.02148, + 1.976595, + 1.966039, + 1.980862, + 1.988612, + 1.994121, + 1.991386, + 1.998387, + 1.971377, + 2.119934, + 2.031896, + 2.020009, + 2.006749, + 2.069721, + 2.166943, + 2.043623, + 2.067638, + 1.964707, + 1.9937 + ] + }, + "ratio": 1.3556940397621424 + }, + "string_build": { + "luapyre": { + "median_ms": 0.557327, + "samples_ms": [ + 0.529125, + 0.619808, + 0.664132, + 0.574031, + 0.557327, + 0.585057, + 0.616693, + 0.536526, + 0.658685, + 0.563595, + 0.529806, + 0.542575, + 0.536065, + 0.569034, + 0.531248, + 0.626027, + 0.540161, + 0.562584, + 0.555704, + 0.530537, + 0.557557, + 0.532951, + 0.633488, + 0.559879, + 0.56656, + 0.52553, + 0.541002, + 0.555934, + 0.529816, + 0.589723, + 0.53308 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.388739, + "samples_ms": [ + 0.534893, + 0.38237, + 0.410532, + 0.379366, + 0.39586, + 0.380508, + 0.379055, + 0.39579, + 0.432663, + 0.429358, + 0.378885, + 0.380347, + 0.393677, + 0.377754, + 0.376802, + 0.390772, + 0.376622, + 0.453995, + 0.382871, + 0.396461, + 0.412595, + 0.567831, + 0.418713, + 0.383612, + 0.393487, + 0.377143, + 0.377193, + 0.388739, + 0.381779, + 0.41635, + 0.376472 + ] + }, + "ratio": 1.4336791523361434 + }, + "table_mix": { + "luapyre": { + "median_ms": 29.603409, + "samples_ms": [ + 30.011988, + 29.603409, + 30.178241, + 57.654324, + 29.000326, + 29.346363, + 31.524434, + 30.260132, + 30.225551, + 31.452007, + 29.357159, + 29.11281, + 29.531824, + 28.993586, + 29.199798, + 30.997602, + 29.876659, + 29.706801, + 30.857267, + 29.710927, + 29.122245, + 29.024241, + 29.336238, + 29.293905, + 29.088765, + 30.063393, + 29.554828, + 29.306914, + 30.057905, + 30.076092, + 29.589114 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 3.35085, + "samples_ms": [ + 3.358501, + 3.348246, + 3.329109, + 3.351321, + 3.338342, + 3.426822, + 3.41891, + 3.320004, + 3.320445, + 3.346384, + 3.437397, + 3.94431, + 3.515892, + 3.402616, + 3.362677, + 3.319595, + 3.383619, + 3.451428, + 3.431849, + 3.445519, + 3.696767, + 3.325343, + 3.35085, + 3.333846, + 3.332123, + 3.345012, + 3.322819, + 3.363048, + 3.317021, + 3.339133, + 3.327997 + ] + }, + "ratio": 8.834596893325575 + }, + "typed_arith": { + "luapyre": { + "median_ms": 0.976509, + "samples_ms": [ + 0.979915, + 1.141431, + 0.974296, + 0.94879, + 0.983831, + 1.071589, + 0.968438, + 0.957532, + 0.966195, + 1.016258, + 0.963341, + 0.94865, + 0.964833, + 1.005843, + 0.976509, + 0.948929, + 0.967076, + 0.991492, + 0.973496, + 0.953376, + 0.971833, + 1.03142, + 1.165247, + 1.028477, + 1.093732, + 1.112539, + 1.05849, + 0.9696, + 0.964642, + 1.01079, + 1.042827 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.764139, + "samples_ms": [ + 0.850676, + 0.734536, + 0.731121, + 0.739683, + 0.745942, + 0.720505, + 0.771369, + 0.747224, + 0.801975, + 0.738461, + 0.755065, + 0.751901, + 0.748656, + 0.773713, + 0.746693, + 0.745432, + 0.769847, + 0.766252, + 1.582177, + 0.992163, + 0.809235, + 1.457054, + 0.761986, + 0.733214, + 0.766372, + 0.775676, + 0.856314, + 0.764139, + 0.748136, + 1.059282, + 0.860661 + ] + }, + "ratio": 1.2779206400929672 + }, + "typed_branch": { + "luapyre": { + "median_ms": 2.20596, + "samples_ms": [ + 2.193662, + 2.128647, + 2.224978, + 2.149308, + 2.31445, + 2.305126, + 2.178901, + 2.281481, + 2.275452, + 2.220832, + 2.350172, + 2.187874, + 2.20557, + 2.146814, + 2.131922, + 2.165941, + 2.136629, + 2.127135, + 2.277035, + 2.382389, + 2.49912, + 2.23993, + 2.162117, + 2.310073, + 2.21913, + 2.20596, + 2.243185, + 2.195445, + 2.129369, + 2.130931, + 2.317334 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.446629, + "samples_ms": [ + 1.46119, + 1.450113, + 1.711376, + 1.437434, + 1.52941, + 1.431667, + 1.457675, + 1.424025, + 1.42746, + 1.457174, + 1.44069, + 1.446629, + 1.544982, + 1.443204, + 1.432978, + 1.455972, + 1.436464, + 1.456183, + 1.425488, + 1.427971, + 1.463263, + 1.429944, + 1.445868, + 1.456833, + 1.451696, + 1.44069, + 1.456383, + 1.425498, + 1.526795, + 1.429604, + 1.478525 + ] + }, + "ratio": 1.5248968463925445 + } + } + }, + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.13.15", + "python_first": false, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 22.910943, + "samples_ms": [ + 22.910943, + 22.720364, + 22.215113, + 23.577379, + 23.627763, + 26.078212, + 25.807525, + 23.467738, + 22.668107, + 23.299382, + 22.203026, + 50.986961, + 24.664111, + 22.788414, + 23.054503, + 22.402909, + 23.227026, + 26.637239, + 23.140709, + 23.976584, + 22.565778, + 22.857815, + 22.949881, + 22.62926, + 22.672664, + 22.822324, + 22.296333, + 23.242999, + 22.640878, + 22.117712, + 22.774223 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 2.148527, + "samples_ms": [ + 2.686815, + 4.51353, + 2.250806, + 2.085354, + 2.194664, + 2.307419, + 2.163087, + 2.233201, + 2.216366, + 2.106905, + 2.364964, + 2.176247, + 2.081869, + 2.118062, + 2.071734, + 2.162237, + 2.101548, + 2.112564, + 2.077583, + 2.222455, + 2.124972, + 2.256926, + 2.076571, + 2.112283, + 2.122939, + 2.148527, + 2.249685, + 2.077162, + 2.120465, + 2.317514, + 2.121677 + ] + }, + "ratio": 10.663558335548029 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 23.862116, + "samples_ms": [ + 23.758004, + 24.040598, + 24.009372, + 24.855711, + 25.465584, + 23.816098, + 25.118516, + 23.788168, + 23.539123, + 23.390856, + 23.862116, + 23.307223, + 24.244125, + 23.883547, + 24.558666, + 24.029922, + 25.447017, + 24.658922, + 24.923941, + 23.540906, + 23.614013, + 23.73536, + 23.456112, + 23.681982, + 23.88534, + 23.675733, + 24.894798, + 23.717575, + 23.792244, + 23.513685, + 26.495803 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.77156, + "samples_ms": [ + 0.750489, + 0.779872, + 0.912105, + 0.848283, + 0.879598, + 0.867831, + 0.764119, + 0.760174, + 0.761865, + 0.810607, + 0.891907, + 0.762026, + 0.759863, + 0.77156, + 0.851316, + 0.874631, + 0.762727, + 0.932886, + 1.09265, + 0.774685, + 0.875402, + 0.787994, + 0.763749, + 0.743148, + 0.759012, + 0.757679, + 0.774534, + 0.743529, + 0.761465, + 0.756558, + 0.76517 + ] + }, + "ratio": 30.927103530509616 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.669351, + "samples_ms": [ + 0.711342, + 0.679896, + 0.680337, + 0.790328, + 0.653438, + 0.723029, + 0.665605, + 0.653117, + 0.685073, + 0.664324, + 0.648811, + 0.663492, + 0.666006, + 0.759263, + 0.685545, + 0.662611, + 0.688559, + 0.679296, + 0.669351, + 0.737651, + 0.678594, + 0.684002, + 0.667007, + 0.649341, + 0.664243, + 0.678954, + 0.648109, + 0.665755, + 0.648931, + 0.668039, + 1.000165 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.050304, + "samples_ms": [ + 0.050474, + 0.050514, + 0.050194, + 0.050063, + 0.050064, + 0.050193, + 0.049974, + 0.050183, + 0.049983, + 0.050214, + 0.050104, + 0.050193, + 0.050194, + 0.049963, + 0.050083, + 0.050164, + 0.050304, + 0.050103, + 0.064043, + 0.053138, + 0.050955, + 0.050775, + 0.050534, + 0.050444, + 0.051055, + 0.050725, + 0.051035, + 0.050564, + 0.050484, + 0.050764, + 0.050874 + ] + }, + "ratio": 13.306118797709924 + }, + "sieve": { + "luapyre": { + "median_ms": 8.73985, + "samples_ms": [ + 8.566496, + 8.73985, + 8.716024, + 8.609468, + 8.509322, + 17.586425, + 8.498165, + 9.059878, + 8.619984, + 8.747801, + 8.580997, + 8.786358, + 8.640805, + 9.583064, + 8.780249, + 8.910409, + 8.824343, + 8.916669, + 9.164091, + 8.655165, + 8.715224, + 8.438809, + 8.561128, + 8.595739, + 8.670818, + 9.643152, + 9.009023, + 8.905052, + 8.546637, + 9.396151, + 8.853606 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.585828, + "samples_ms": [ + 0.58708, + 0.758772, + 0.574421, + 0.586859, + 0.657913, + 0.57365, + 0.583946, + 0.585828, + 0.590665, + 0.584076, + 0.567, + 0.589994, + 0.567772, + 0.58725, + 0.591285, + 0.588993, + 0.591917, + 0.569074, + 0.583475, + 0.572329, + 0.58023, + 0.618316, + 0.575683, + 0.604295, + 0.572578, + 0.586218, + 0.58003, + 0.592788, + 0.604986, + 0.574922, + 0.583194 + ] + }, + "ratio": 14.918798691766185 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.840803, + "samples_ms": [ + 3.85626, + 3.139851, + 2.728828, + 2.854182, + 3.05063, + 2.806131, + 2.901231, + 2.873961, + 2.822815, + 2.696672, + 2.715989, + 2.793042, + 3.089877, + 2.959125, + 2.810998, + 2.694258, + 2.731472, + 2.711593, + 2.840803, + 2.656422, + 2.696401, + 2.877546, + 2.688249, + 4.610615, + 5.583721, + 2.775226, + 2.862133, + 3.067274, + 2.957773, + 2.858939, + 2.781745 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.049563, + "samples_ms": [ + 2.123431, + 2.107438, + 2.024265, + 2.072497, + 1.977437, + 2.056033, + 1.983947, + 2.233092, + 2.167756, + 2.21181, + 2.159734, + 2.105585, + 2.029524, + 1.97239, + 2.38178, + 2.056383, + 2.096672, + 1.976656, + 1.970287, + 1.960773, + 1.988934, + 2.03387, + 1.98578, + 2.220003, + 2.049563, + 1.979971, + 1.993812, + 2.125444, + 2.117001, + 2.01307, + 2.000902 + ] + }, + "ratio": 1.386053026913542 + }, + "string_build": { + "luapyre": { + "median_ms": 0.557998, + "samples_ms": [ + 0.55974, + 0.538479, + 0.622462, + 0.557637, + 0.567682, + 0.557998, + 0.542455, + 0.556715, + 0.536206, + 0.620288, + 0.550667, + 0.556104, + 0.540292, + 0.550647, + 0.555775, + 0.537618, + 0.618246, + 0.551498, + 0.852169, + 0.594861, + 0.550156, + 0.632617, + 0.593199, + 0.578738, + 0.566981, + 0.540482, + 0.55981, + 0.547192, + 0.641099, + 0.576665, + 0.584666 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.396772, + "samples_ms": [ + 0.383091, + 0.415118, + 0.383372, + 0.383071, + 0.50494, + 0.442518, + 0.412985, + 0.388309, + 0.399385, + 0.384083, + 0.40268, + 0.411583, + 0.38907, + 0.400577, + 0.385555, + 0.385174, + 0.396772, + 0.386497, + 0.404372, + 0.385695, + 0.473394, + 0.41608, + 0.394037, + 0.396891, + 0.382721, + 0.383071, + 0.406976, + 0.384033, + 0.400888, + 0.382481, + 0.405784 + ] + }, + "ratio": 1.406344197675239 + }, + "table_mix": { + "luapyre": { + "median_ms": 30.416635, + "samples_ms": [ + 30.544482, + 30.716314, + 31.083662, + 59.131548, + 29.663712, + 29.227013, + 32.520526, + 31.659295, + 30.416635, + 30.879022, + 33.026437, + 31.811038, + 29.659316, + 29.184811, + 29.695729, + 29.265009, + 29.896153, + 31.149147, + 30.132469, + 30.479067, + 28.95843, + 29.373325, + 30.42718, + 29.235023, + 31.422078, + 31.10244, + 29.700956, + 29.283705, + 30.02462, + 30.518805, + 29.049581 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 3.356178, + "samples_ms": [ + 3.354365, + 3.321006, + 3.626453, + 5.068937, + 3.558824, + 3.356178, + 3.437626, + 3.670238, + 3.354074, + 3.556, + 3.358911, + 3.290862, + 3.509212, + 3.480479, + 3.322087, + 3.593775, + 3.404078, + 3.30293, + 3.281477, + 3.636178, + 3.334776, + 3.33052, + 3.305513, + 3.282009, + 3.561638, + 3.409655, + 3.32311, + 3.309458, + 3.329228, + 3.313475, + 3.434101 + ] + }, + "ratio": 9.062878965299218 + }, + "typed_arith": { + "luapyre": { + "median_ms": 0.943201, + "samples_ms": [ + 0.994978, + 0.952575, + 0.936862, + 0.927358, + 1.008838, + 0.933106, + 0.959115, + 0.927258, + 0.978262, + 0.932435, + 0.945775, + 0.927559, + 0.983691, + 0.932145, + 0.947698, + 0.926247, + 0.974107, + 0.93585, + 1.050458, + 0.936161, + 0.990831, + 0.924794, + 0.944113, + 0.926046, + 0.928199, + 0.97652, + 0.930162, + 0.943201, + 0.952094, + 0.988127, + 0.929041 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.709579, + "samples_ms": [ + 0.708928, + 0.929321, + 0.874761, + 0.871637, + 0.881871, + 0.854822, + 0.917073, + 0.711742, + 0.886939, + 0.705083, + 0.713936, + 0.765211, + 0.716239, + 0.70354, + 0.705673, + 0.703841, + 0.798069, + 0.689911, + 0.712704, + 0.735637, + 0.687877, + 0.70383, + 0.706845, + 0.705643, + 0.691313, + 0.721567, + 0.709579, + 0.688418, + 0.700166, + 0.704572, + 0.707766 + ] + }, + "ratio": 1.3292402960065053 + }, + "typed_branch": { + "luapyre": { + "median_ms": 2.155247, + "samples_ms": [ + 2.17818, + 2.106225, + 2.126825, + 2.121267, + 2.201544, + 2.099555, + 2.365784, + 2.155247, + 2.222295, + 2.161346, + 2.124561, + 2.120455, + 2.381828, + 2.261812, + 2.331285, + 2.230566, + 2.154936, + 2.114147, + 2.155797, + 2.138422, + 2.125552, + 2.122718, + 2.109319, + 2.261813, + 2.245388, + 2.241403, + 2.1577, + 2.167815, + 2.148477, + 2.130961, + 2.121477 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.450154, + "samples_ms": [ + 1.442012, + 1.468461, + 1.461932, + 1.47471, + 1.432377, + 1.531453, + 1.436404, + 1.446198, + 1.453549, + 1.433549, + 1.469112, + 1.441831, + 1.576408, + 1.541838, + 1.472637, + 1.574486, + 1.573615, + 1.446198, + 1.46793, + 1.432759, + 1.444946, + 1.450154, + 1.446248, + 1.427291, + 1.489602, + 1.432117, + 1.458806, + 1.448011, + 1.44775, + 1.459908, + 1.435041 + ] + }, + "ratio": 1.486219394629812 + } + } + } + ], + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre_median_ms": 22.890352, + "luapyre_process_medians_ms": [ + 22.504055, + 22.890352, + 22.910943 + ], + "python_median_ms": 2.148527, + "python_process_medians_ms": [ + 2.082579, + 2.148527, + 2.168307 + ], + "ratio": 10.653974560245228 + }, + "fib_recursive": { + "luapyre_median_ms": 23.751985, + "luapyre_process_medians_ms": [ + 23.55869, + 23.751985, + 23.862116 + ], + "python_median_ms": 0.77156, + "python_process_medians_ms": [ + 0.770668, + 0.77156, + 0.774133 + ], + "ratio": 30.78436544144331 + }, + "python_calls_1000": { + "luapyre_median_ms": 0.669351, + "luapyre_process_medians_ms": [ + 0.661439, + 0.669351, + 0.672054 + ], + "python_median_ms": 0.051155, + "python_process_medians_ms": [ + 0.050304, + 0.051155, + 0.052848 + ], + "ratio": 13.084761997849673 + }, + "sieve": { + "luapyre_median_ms": 8.582626, + "luapyre_process_medians_ms": [ + 8.503432, + 8.582626, + 8.73985 + ], + "python_median_ms": 0.585828, + "python_process_medians_ms": [ + 0.58039, + 0.585828, + 0.5871 + ], + "ratio": 14.65041957707723 + }, + "spectral_norm": { + "luapyre_median_ms": 2.733213, + "luapyre_process_medians_ms": [ + 2.722357, + 2.733213, + 2.840803 + ], + "python_median_ms": 2.043041, + "python_process_medians_ms": [ + 2.008091, + 2.043041, + 2.049563 + ], + "ratio": 1.3378160301237225 + }, + "string_build": { + "luapyre_median_ms": 0.557998, + "luapyre_process_medians_ms": [ + 0.557327, + 0.557998, + 0.581091 + ], + "python_median_ms": 0.396772, + "python_process_medians_ms": [ + 0.388739, + 0.396772, + 0.398093 + ], + "ratio": 1.406344197675239 + }, + "table_mix": { + "luapyre_median_ms": 30.416635, + "luapyre_process_medians_ms": [ + 29.603409, + 30.416635, + 31.279617 + ], + "python_median_ms": 3.356178, + "python_process_medians_ms": [ + 3.35085, + 3.356178, + 3.436252 + ], + "ratio": 9.062878965299218 + }, + "typed_arith": { + "luapyre_median_ms": 0.976509, + "luapyre_process_medians_ms": [ + 0.943201, + 0.976509, + 1.013655 + ], + "python_median_ms": 0.734055, + "python_process_medians_ms": [ + 0.709579, + 0.734055, + 0.764139 + ], + "ratio": 1.3302940515356478 + }, + "typed_branch": { + "luapyre_median_ms": 2.196998, + "luapyre_process_medians_ms": [ + 2.155247, + 2.196998, + 2.20596 + ], + "python_median_ms": 1.450154, + "python_process_medians_ms": [ + 1.446629, + 1.450154, + 1.489011 + ], + "ratio": 1.5150101299586112 + } + } + }, + { + "affinity_cpu": 0, + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.14.7", + "repeats": 31, + "runs": [ + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.14.7", + "python_first": false, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 21.352334, + "samples_ms": [ + 21.132863, + 20.637567, + 21.154534, + 22.256669, + 21.352334, + 21.994515, + 23.092213, + 21.551906, + 22.205034, + 21.275672, + 21.078423, + 23.481755, + 22.317278, + 21.69077, + 21.457458, + 20.515608, + 20.789379, + 21.719192, + 20.827295, + 20.937376, + 21.637933, + 21.16559, + 21.111662, + 20.541857, + 21.024904, + 21.051814, + 22.085268, + 22.391106, + 21.188845, + 21.800601, + 22.508158 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 1.955856, + "samples_ms": [ + 1.941815, + 2.071315, + 1.962435, + 1.994272, + 2.292559, + 1.969825, + 1.959581, + 1.920965, + 1.944089, + 1.940754, + 1.970306, + 1.955856, + 1.964468, + 1.926753, + 1.948094, + 1.964589, + 2.004667, + 2.213173, + 2.147407, + 1.906082, + 1.947243, + 1.944289, + 1.93224, + 1.943357, + 1.920935, + 1.960652, + 1.939842, + 1.968053, + 1.934775, + 1.963888, + 1.940513 + ] + }, + "ratio": 10.917129890953117 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 18.999467, + "samples_ms": [ + 21.326005, + 19.265147, + 18.999467, + 20.006742, + 19.437429, + 19.087666, + 19.803906, + 18.952097, + 19.071973, + 18.609335, + 19.587358, + 18.68794, + 18.547205, + 18.548816, + 18.85132, + 18.761438, + 18.995221, + 18.934381, + 18.816809, + 19.805798, + 19.32916, + 19.250034, + 18.856267, + 20.325059, + 19.825697, + 19.111201, + 18.859222, + 18.901523, + 18.683665, + 19.255562, + 18.803299 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.642531, + "samples_ms": [ + 0.636353, + 0.646176, + 0.642792, + 0.647379, + 0.643363, + 0.642531, + 0.63457, + 0.6413, + 0.643183, + 0.628651, + 0.658215, + 0.63522, + 0.644835, + 0.642582, + 0.631145, + 0.645416, + 0.65586, + 0.634079, + 0.662651, + 0.6414, + 0.630775, + 0.651785, + 0.800543, + 0.624685, + 0.636132, + 0.639897, + 0.623814, + 0.692104, + 0.666256, + 0.621691, + 0.632367 + ] + }, + "ratio": 29.569728153194166 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.598396, + "samples_ms": [ + 0.603043, + 0.593379, + 0.635331, + 0.588432, + 0.653828, + 0.714297, + 0.583895, + 0.612878, + 0.586088, + 0.754755, + 0.605036, + 0.637353, + 0.612356, + 0.680577, + 0.637454, + 0.605827, + 0.601902, + 0.581973, + 0.597766, + 0.61426, + 0.596794, + 0.597756, + 0.581903, + 0.596243, + 0.586929, + 0.598396, + 0.616092, + 0.583094, + 0.596073, + 0.58727, + 0.596864 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.04069, + "samples_ms": [ + 0.04085, + 0.04061, + 0.04089, + 0.04057, + 0.040589, + 0.040479, + 0.040509, + 0.056032, + 0.04081, + 0.04066, + 0.04069, + 0.040629, + 0.04056, + 0.040639, + 0.040619, + 0.04053, + 0.0409, + 0.04069, + 0.04082, + 0.043093, + 0.04061, + 0.040699, + 0.040689, + 0.04049, + 0.040549, + 0.040739, + 0.05426, + 0.04081, + 0.04086, + 0.04076, + 0.0409 + ] + }, + "ratio": 14.706217743917426 + }, + "sieve": { + "luapyre": { + "median_ms": 6.633669, + "samples_ms": [ + 6.682351, + 6.869706, + 6.677233, + 6.630695, + 6.549386, + 6.56595, + 6.848003, + 6.55919, + 6.556226, + 6.535405, + 7.10511, + 6.886851, + 6.643233, + 6.560091, + 6.993176, + 6.676212, + 6.633669, + 6.691414, + 6.591638, + 6.62003, + 6.576175, + 6.786994, + 6.556736, + 6.544258, + 6.66119, + 6.567502, + 6.738753, + 6.56564, + 6.542586, + 7.573897, + 6.79097 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.510999, + "samples_ms": [ + 0.509176, + 0.499882, + 0.510999, + 0.497269, + 0.510118, + 0.494775, + 0.565789, + 0.498461, + 0.519812, + 0.495857, + 0.517108, + 0.499011, + 0.511379, + 0.539861, + 0.538319, + 0.501014, + 0.524859, + 0.496438, + 0.511419, + 0.497599, + 0.513343, + 0.575644, + 0.500874, + 0.515645, + 0.496878, + 0.512641, + 0.497049, + 0.506913, + 0.604245, + 0.567221, + 0.558298 + ] + }, + "ratio": 12.981765130655834 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.92258, + "samples_ms": [ + 4.134526, + 2.936981, + 2.958132, + 2.945845, + 2.968287, + 2.914158, + 2.875582, + 2.946765, + 2.925164, + 2.933676, + 3.105949, + 2.949199, + 2.92258, + 2.90196, + 2.878887, + 2.864886, + 2.887459, + 2.828853, + 2.804777, + 2.778369, + 2.875832, + 3.123184, + 2.924593, + 2.845387, + 2.803206, + 2.814753, + 2.781945, + 3.020544, + 3.071528, + 2.904163, + 3.035506 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.343372, + "samples_ms": [ + 2.463928, + 2.425282, + 2.44412, + 2.309352, + 2.334108, + 2.312237, + 2.343372, + 2.316182, + 2.467894, + 2.550806, + 2.49267, + 2.353477, + 2.342611, + 2.329031, + 2.32841, + 2.624905, + 2.405222, + 2.326939, + 2.330523, + 2.618305, + 2.542354, + 2.361599, + 2.337233, + 2.318105, + 2.326848, + 2.326217, + 2.327328, + 2.354309, + 2.516626, + 2.387918, + 2.320368 + ] + }, + "ratio": 1.2471686100200907 + }, + "string_build": { + "luapyre": { + "median_ms": 0.537468, + "samples_ms": [ + 0.51949, + 0.585427, + 0.548133, + 0.537468, + 0.683852, + 0.561352, + 1.334815, + 0.617274, + 0.525349, + 0.549905, + 0.513662, + 0.780413, + 1.007535, + 0.598276, + 0.521444, + 0.554833, + 0.504028, + 0.510738, + 0.493473, + 0.521935, + 0.623443, + 0.544878, + 0.530687, + 0.516877, + 0.574732, + 0.53889, + 0.49851, + 0.519681, + 0.50501, + 0.522496, + 0.533461 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.370292, + "samples_ms": [ + 0.357944, + 0.376612, + 0.36154, + 0.391113, + 0.364654, + 0.361359, + 0.370292, + 0.358696, + 0.358736, + 0.429209, + 0.428508, + 0.393096, + 0.36205, + 0.383642, + 0.372356, + 0.357163, + 0.371223, + 0.357934, + 0.487284, + 0.395709, + 0.363933, + 0.377613, + 0.365124, + 0.36196, + 0.391413, + 0.359977, + 0.41001, + 0.391934, + 0.361209, + 0.370833, + 0.360488 + ] + }, + "ratio": 1.4514707312067232 + }, + "table_mix": { + "luapyre": { + "median_ms": 22.023836, + "samples_ms": [ + 22.581395, + 22.649715, + 22.340061, + 46.634285, + 21.80141, + 21.888949, + 22.330064, + 21.478357, + 22.380108, + 21.773419, + 21.720692, + 22.099537, + 24.374699, + 22.150532, + 21.68519, + 23.508741, + 22.543438, + 22.356724, + 21.797415, + 21.804986, + 22.042333, + 21.645151, + 21.626886, + 21.877872, + 21.962446, + 22.023836, + 21.682126, + 22.382391, + 21.451818, + 22.430893, + 21.832516 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 2.567891, + "samples_ms": [ + 2.499461, + 2.69045, + 2.584315, + 2.571707, + 2.511258, + 2.494714, + 2.850685, + 2.624233, + 2.50591, + 2.545378, + 2.516796, + 2.577565, + 2.49893, + 2.529535, + 2.496066, + 2.513822, + 2.664202, + 2.622711, + 2.577085, + 2.699684, + 2.950642, + 2.694957, + 2.552829, + 2.693004, + 2.621309, + 2.553911, + 2.56086, + 2.54001, + 2.626497, + 2.567891, + 2.538728 + ] + }, + "ratio": 8.576624163564574 + }, + "typed_arith": { + "luapyre": { + "median_ms": 0.731561, + "samples_ms": [ + 0.855113, + 0.775987, + 0.719153, + 0.701758, + 0.71683, + 0.716169, + 0.731983, + 0.70356, + 0.718483, + 0.717802, + 0.920158, + 0.74412, + 1.862979, + 0.865147, + 0.753924, + 0.740064, + 0.702499, + 0.867001, + 0.756689, + 0.753615, + 0.740966, + 0.718253, + 0.715438, + 0.731561, + 0.700516, + 0.735708, + 0.738212, + 0.719163, + 0.717621, + 0.725072, + 0.715418 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.511109, + "samples_ms": [ + 0.502727, + 0.76409, + 0.664784, + 0.524338, + 0.532711, + 0.584997, + 0.540361, + 0.526912, + 0.479633, + 0.524248, + 0.480274, + 0.502206, + 0.478931, + 0.749368, + 0.47828, + 0.522516, + 0.511109, + 0.608521, + 0.507574, + 0.490178, + 0.569855, + 0.613259, + 0.513422, + 0.506973, + 0.492572, + 0.489997, + 0.49182, + 0.488045, + 0.54614, + 0.489227, + 0.491981 + ] + }, + "ratio": 1.4313209119776797 + }, + "typed_branch": { + "luapyre": { + "median_ms": 1.896749, + "samples_ms": [ + 1.844673, + 1.913153, + 2.043554, + 1.861397, + 1.843461, + 1.857632, + 1.828008, + 1.915306, + 1.937939, + 2.024196, + 1.925361, + 1.834138, + 1.853065, + 1.835719, + 1.990907, + 1.887065, + 1.908997, + 1.882347, + 1.923778, + 1.837603, + 2.04766, + 2.817417, + 1.896749, + 1.935396, + 1.866465, + 1.92459, + 2.094088, + 1.8907, + 1.84257, + 1.920414, + 1.882518 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.14004, + "samples_ms": [ + 1.121173, + 1.12632, + 1.175432, + 1.131868, + 1.122374, + 1.148002, + 1.14004, + 1.175172, + 1.11972, + 1.222962, + 1.176914, + 1.1536, + 1.136365, + 1.120742, + 1.120112, + 1.120642, + 1.148102, + 1.130636, + 1.294556, + 1.142764, + 1.181801, + 1.179028, + 1.191596, + 1.121904, + 1.13346, + 1.122305, + 1.193419, + 1.155083, + 1.176564, + 1.136575, + 1.120642 + ] + }, + "ratio": 1.6637565348584262 + } + } + }, + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.14.7", + "python_first": true, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 20.590567, + "samples_ms": [ + 20.483, + 20.498102, + 20.392837, + 20.590567, + 20.834634, + 20.390524, + 24.350264, + 21.061016, + 20.992095, + 20.857107, + 20.556657, + 21.877723, + 21.087355, + 21.114094, + 20.657124, + 20.8384, + 20.478743, + 20.315845, + 20.479544, + 20.386748, + 20.584197, + 21.177888, + 21.15223, + 21.044151, + 20.379147, + 20.286782, + 20.475067, + 20.657265, + 20.315585, + 20.410443, + 21.667925 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 1.957617, + "samples_ms": [ + 2.018396, + 1.98002, + 1.951378, + 1.960682, + 1.957617, + 1.965889, + 1.936586, + 1.956706, + 1.929085, + 2.108558, + 2.004726, + 1.991787, + 1.93889, + 1.959289, + 2.033969, + 1.951398, + 1.93186, + 1.951668, + 1.957387, + 1.95261, + 1.934483, + 2.107697, + 2.030544, + 2.156799, + 1.998897, + 1.953832, + 1.924809, + 2.133084, + 2.012718, + 1.952139, + 1.937607 + ] + }, + "ratio": 10.518179500893178 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 19.054858, + "samples_ms": [ + 21.837093, + 19.09013, + 19.339054, + 19.564444, + 19.497746, + 18.896877, + 19.246058, + 19.054858, + 18.559442, + 19.036271, + 18.799885, + 19.120564, + 19.196475, + 18.971195, + 19.093794, + 18.515157, + 18.64008, + 19.055499, + 18.834205, + 19.851144, + 22.03979, + 18.808497, + 18.836027, + 19.196905, + 19.343731, + 18.7904, + 18.88572, + 18.980529, + 18.795208, + 19.028519, + 19.103059 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.635831, + "samples_ms": [ + 0.642271, + 0.663482, + 0.654839, + 0.649932, + 0.622151, + 0.634699, + 0.631856, + 0.62102, + 0.634159, + 0.646457, + 0.62164, + 0.648139, + 0.620709, + 0.650242, + 0.633157, + 0.705724, + 0.648059, + 0.700857, + 0.622702, + 0.640157, + 0.643492, + 0.636352, + 0.63471, + 0.634939, + 0.6213, + 0.635831, + 0.633287, + 0.623463, + 0.645385, + 0.620159, + 0.656982 + ] + }, + "ratio": 29.968431863183767 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.610845, + "samples_ms": [ + 0.594662, + 0.610845, + 0.626398, + 0.595863, + 0.774304, + 0.612407, + 0.593769, + 0.610494, + 0.66857, + 0.632207, + 0.610083, + 0.59408, + 0.609343, + 0.658325, + 0.601721, + 0.623494, + 0.629082, + 0.62143, + 0.608942, + 0.595652, + 0.613027, + 0.625146, + 0.598086, + 0.649691, + 0.597125, + 0.623944, + 0.613689, + 0.593399, + 0.622532, + 0.597445, + 0.610044 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.04045, + "samples_ms": [ + 0.04081, + 0.040409, + 0.040599, + 0.04069, + 0.040449, + 0.040259, + 0.040559, + 0.040539, + 0.040349, + 0.040359, + 0.040259, + 0.040419, + 0.04045, + 0.040359, + 0.04068, + 0.040439, + 0.053389, + 0.040459, + 0.04054, + 0.040489, + 0.0405, + 0.04101, + 0.040429, + 0.040599, + 0.040419, + 0.040519, + 0.04066, + 0.040429, + 0.04043, + 0.040399, + 0.040429 + ] + }, + "ratio": 15.10123609394314 + }, + "sieve": { + "luapyre": { + "median_ms": 6.592786, + "samples_ms": [ + 6.555832, + 6.715065, + 6.578365, + 6.501372, + 6.846668, + 6.600006, + 6.530975, + 6.517366, + 7.041413, + 6.928968, + 6.754743, + 6.625254, + 6.639615, + 6.809493, + 6.718199, + 6.571826, + 6.52057, + 6.539598, + 6.614147, + 6.58166, + 6.545046, + 6.592786, + 6.551705, + 6.518457, + 6.515452, + 6.63714, + 6.703548, + 6.54138, + 6.528472, + 6.817014, + 6.856993 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.506532, + "samples_ms": [ + 0.517037, + 0.498049, + 0.514563, + 0.500553, + 0.510698, + 0.495286, + 0.516316, + 0.511429, + 0.511239, + 0.494954, + 0.508094, + 0.493283, + 0.506532, + 0.495926, + 0.506682, + 0.511119, + 0.601531, + 0.524518, + 0.49876, + 0.509466, + 0.493813, + 0.507232, + 0.493012, + 0.531469, + 0.498149, + 0.508094, + 0.493843, + 0.50534, + 0.495556, + 0.506281, + 0.493222 + ] + }, + "ratio": 13.015537024314359 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.808774, + "samples_ms": [ + 3.751405, + 3.018541, + 2.888782, + 2.835604, + 2.807282, + 2.874711, + 2.796717, + 2.863735, + 2.780322, + 2.928509, + 2.836054, + 2.79915, + 2.787853, + 2.771029, + 2.813701, + 2.793361, + 2.772621, + 2.780453, + 2.808774, + 2.889643, + 2.876593, + 2.90133, + 2.793141, + 2.784919, + 2.766863, + 2.863695, + 2.816516, + 2.803076, + 2.770048, + 2.961177, + 2.795594 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.367538, + "samples_ms": [ + 2.326107, + 2.340658, + 2.346246, + 2.326138, + 2.354449, + 2.344244, + 2.34155, + 2.62873, + 3.241257, + 2.576694, + 2.448316, + 2.367538, + 2.353697, + 2.414437, + 2.339877, + 2.354459, + 2.480694, + 2.773112, + 3.206707, + 2.361359, + 2.349421, + 2.350233, + 2.346647, + 2.374228, + 3.045962, + 2.45786, + 2.378334, + 2.388629, + 2.367248, + 2.604865, + 2.386365 + ] + }, + "ratio": 1.1863691311396058 + }, + "string_build": { + "luapyre": { + "median_ms": 0.516006, + "samples_ms": [ + 0.574562, + 0.627479, + 0.710711, + 0.6484, + 0.555764, + 0.504108, + 0.516006, + 0.496518, + 0.50486, + 0.491189, + 0.504819, + 0.830777, + 0.536175, + 0.759382, + 0.632706, + 0.591567, + 0.555995, + 0.512251, + 0.543096, + 0.514434, + 0.526301, + 0.508976, + 0.51944, + 0.501595, + 0.505691, + 0.505931, + 0.528013, + 0.511909, + 0.511379, + 0.495166, + 0.503147 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.365165, + "samples_ms": [ + 0.377634, + 0.360929, + 0.356332, + 0.370173, + 0.362311, + 0.392876, + 0.365576, + 0.360278, + 0.45047, + 0.365335, + 0.359727, + 0.369301, + 0.357013, + 0.3551, + 0.373948, + 0.363402, + 0.391143, + 0.363603, + 0.359317, + 0.368219, + 0.361399, + 0.361809, + 0.376932, + 0.491479, + 0.475637, + 0.365165, + 0.411974, + 0.36134, + 0.356302, + 0.372556, + 0.355722 + ] + }, + "ratio": 1.4130762805854886 + }, + "table_mix": { + "luapyre": { + "median_ms": 22.343714, + "samples_ms": [ + 22.743761, + 22.952166, + 22.508907, + 48.75028, + 22.147857, + 23.068486, + 23.705579, + 22.343714, + 22.64792, + 24.779702, + 22.098986, + 23.073766, + 22.683606, + 24.062546, + 22.057268, + 21.907839, + 21.985754, + 22.09982, + 22.207017, + 22.113981, + 22.332361, + 22.201119, + 22.274537, + 26.71446, + 22.303849, + 21.927768, + 22.330749, + 22.879662, + 22.167509, + 22.712198, + 24.1959 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 2.507523, + "samples_ms": [ + 2.559038, + 2.675068, + 2.611756, + 2.511077, + 2.491439, + 2.507523, + 2.49174, + 2.801863, + 2.541752, + 2.539119, + 2.495715, + 2.494323, + 2.506221, + 2.468856, + 2.50625, + 2.56061, + 2.503267, + 2.549314, + 2.517538, + 2.492931, + 2.652785, + 2.515645, + 2.580049, + 3.827105, + 2.921959, + 2.480202, + 2.507012, + 2.501374, + 2.486352, + 2.504227, + 2.491099 + ] + }, + "ratio": 8.910671607000214 + }, + "typed_arith": { + "luapyre": { + "median_ms": 0.731111, + "samples_ms": [ + 0.719994, + 0.73041, + 0.722608, + 0.722148, + 0.704823, + 0.717871, + 0.719965, + 0.734857, + 0.701017, + 0.718573, + 0.718062, + 0.838999, + 0.815725, + 0.738772, + 0.719624, + 0.721908, + 0.704202, + 1.019473, + 1.014827, + 1.04462, + 1.858292, + 0.765531, + 0.758811, + 0.981417, + 0.731111, + 0.724691, + 0.81199, + 0.734917, + 1.080954, + 0.8328, + 0.729849 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.49155, + "samples_ms": [ + 0.49155, + 0.49145, + 0.519752, + 0.51951, + 0.479402, + 0.491931, + 0.478551, + 0.513713, + 0.47849, + 0.49163, + 0.4781, + 0.491059, + 0.479402, + 0.479122, + 0.492331, + 0.493202, + 0.493633, + 0.478591, + 0.490759, + 0.478571, + 0.493122, + 0.479933, + 0.494093, + 0.529215, + 0.528104, + 0.478731, + 0.492532, + 0.482237, + 0.494915, + 0.478712, + 0.496006 + ] + }, + "ratio": 1.48735835622012 + }, + "typed_branch": { + "luapyre": { + "median_ms": 1.892722, + "samples_ms": [ + 1.845904, + 1.889517, + 1.832694, + 1.900624, + 1.895356, + 1.981903, + 1.844211, + 1.940502, + 1.844282, + 1.919742, + 1.858552, + 2.131082, + 1.885902, + 1.912811, + 1.862047, + 1.916026, + 1.849498, + 1.892722, + 1.834667, + 1.836269, + 1.922135, + 1.830982, + 1.915816, + 1.832804, + 1.893573, + 1.943356, + 1.900003, + 1.902537, + 1.904159, + 1.834056, + 1.892322 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.155743, + "samples_ms": [ + 1.24985, + 1.205886, + 1.191335, + 1.161882, + 1.120701, + 1.30442, + 1.243471, + 1.135473, + 1.248659, + 1.153319, + 1.197463, + 1.179087, + 1.155743, + 1.141512, + 1.121773, + 1.18115, + 1.223202, + 1.129655, + 1.38602, + 1.14029, + 1.155212, + 1.255599, + 1.140501, + 1.14031, + 1.124277, + 1.129164, + 1.141652, + 1.165066, + 1.145978, + 1.128313, + 1.194269 + ] + }, + "ratio": 1.6376668515405242 + } + } + }, + { + "affinity_cpu": 0, + "hash_seed": "0", + "interpretation": "Reduced-contract same-algorithm references; omit Lua runtime semantics; ratios are not guaranteed available speedups", + "method": "Separate unprofiled samples; Python GC disabled during timing; every result checked", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "python": "3.14.7", + "python_first": false, + "repeats": 31, + "revision": "0.40.0a1-final", + "schema_version": 1, + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 21.113606, + "samples_ms": [ + 21.919005, + 20.476953, + 20.255899, + 20.237022, + 21.322132, + 21.130921, + 20.721942, + 20.84954, + 20.908806, + 21.214734, + 20.375555, + 20.275368, + 23.223407, + 21.314631, + 20.416815, + 21.021601, + 27.833088, + 21.374558, + 21.618295, + 20.518414, + 20.52268, + 20.253376, + 21.258729, + 21.257967, + 20.842448, + 48.876301, + 23.340177, + 21.04791, + 21.113606, + 22.472216, + 21.763388 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + }, + "python_reference": { + "median_ms": 1.994541, + "samples_ms": [ + 1.983665, + 2.062311, + 1.972489, + 1.958469, + 1.992499, + 1.957837, + 2.235234, + 2.05442, + 1.994541, + 1.959961, + 2.257456, + 2.844156, + 2.057934, + 1.970126, + 2.136489, + 2.018567, + 2.031966, + 1.918701, + 2.016704, + 1.92527, + 2.023414, + 1.938469, + 2.041651, + 1.929075, + 2.019138, + 2.211519, + 1.966741, + 1.994482, + 1.968744, + 1.928565, + 2.005788 + ] + }, + "ratio": 10.585696659030825 + }, + "fib_recursive": { + "luapyre": { + "median_ms": 19.119856, + "samples_ms": [ + 18.787058, + 18.615226, + 18.933693, + 19.247262, + 19.123781, + 18.527318, + 21.090292, + 18.802211, + 18.724516, + 19.851217, + 18.503423, + 18.773708, + 18.431457, + 19.119856, + 18.939882, + 19.339037, + 18.551844, + 18.519226, + 18.84339, + 19.504219, + 19.239891, + 18.404438, + 20.077468, + 19.232771, + 19.134868, + 19.288322, + 18.728422, + 19.966876, + 19.153314, + 19.198281, + 19.278878 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 3 + }, + "python_reference": { + "median_ms": 0.63497, + "samples_ms": [ + 0.626487, + 0.635772, + 0.63497, + 0.625777, + 0.650463, + 0.62125, + 0.637073, + 0.632236, + 0.619297, + 0.637584, + 0.645426, + 0.624565, + 0.633257, + 0.638395, + 0.632155, + 0.640798, + 0.637003, + 0.636272, + 0.636813, + 0.625496, + 0.640789, + 0.637825, + 0.623844, + 0.649512, + 0.636322, + 0.621601, + 0.628711, + 0.627098, + 0.633477, + 0.654799, + 0.627018 + ] + }, + "ratio": 30.111432036159187 + }, + "python_calls_1000": { + "luapyre": { + "median_ms": 0.601671, + "samples_ms": [ + 0.690091, + 0.597485, + 0.629873, + 0.601671, + 0.585497, + 0.606619, + 0.590115, + 0.612637, + 0.628982, + 0.587301, + 0.606398, + 0.58719, + 0.60716, + 0.612678, + 0.591536, + 0.623703, + 0.591346, + 0.610434, + 0.606979, + 0.585548, + 0.60096, + 0.60042, + 0.601832, + 0.602392, + 0.58707, + 0.600339, + 0.592287, + 0.600901, + 0.618106, + 0.588833, + 0.603945 + ] + }, + "luapyre_warm_counters": { + "leaf_executions": 1000, + "leaf_frame_elisions": 1000, + "python_direct_entries": 1000 + }, + "python_reference": { + "median_ms": 0.04061, + "samples_ms": [ + 0.040709, + 0.04098, + 0.040439, + 0.04056, + 0.040419, + 0.0406, + 0.040399, + 0.04068, + 0.04066, + 0.040619, + 0.052977, + 0.04077, + 0.040559, + 0.04063, + 0.04062, + 0.040469, + 0.04045, + 0.040409, + 0.04072, + 0.040439, + 0.04074, + 0.04071, + 0.040599, + 0.040429, + 0.040549, + 0.040519, + 0.04077, + 0.04061, + 0.040639, + 0.040789, + 0.040509 + ] + }, + "ratio": 14.815833538537305 + }, + "sieve": { + "luapyre": { + "median_ms": 6.84059, + "samples_ms": [ + 6.679493, + 6.767593, + 6.624583, + 6.794392, + 6.677822, + 6.758419, + 9.067051, + 7.035875, + 7.613302, + 8.626425, + 6.95677, + 7.185465, + 7.259774, + 7.128952, + 6.985922, + 6.84059, + 6.808783, + 6.737169, + 7.188559, + 6.726943, + 6.997389, + 6.582492, + 6.872436, + 6.637572, + 6.921128, + 7.102022, + 6.591906, + 6.571536, + 6.64324, + 6.686684, + 7.071709 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 5327 + }, + "python_reference": { + "median_ms": 0.508715, + "samples_ms": [ + 0.829935, + 0.64207, + 0.587461, + 0.503107, + 0.514533, + 0.501665, + 0.513412, + 0.494644, + 0.513302, + 0.531429, + 0.538238, + 0.499982, + 0.515295, + 0.497809, + 0.509826, + 0.624094, + 0.500413, + 0.564187, + 0.499511, + 0.509096, + 0.496056, + 0.507092, + 0.494734, + 0.508945, + 0.493743, + 0.565498, + 0.495055, + 0.508715, + 0.494864, + 0.505941, + 0.493593 + ] + }, + "ratio": 13.446802237008933 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.936982, + "samples_ms": [ + 3.728341, + 2.935029, + 2.884785, + 2.937122, + 2.848603, + 2.865698, + 2.843335, + 2.921569, + 2.940227, + 2.904925, + 2.944543, + 3.266164, + 3.04543, + 2.936982, + 3.219166, + 3.335556, + 2.907959, + 3.39334, + 2.991382, + 2.931774, + 2.955569, + 3.029638, + 2.865677, + 2.980496, + 2.980065, + 2.878095, + 2.996479, + 2.888311, + 2.868542, + 2.856544, + 2.845778 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 31, + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30, + "table_ic_hits": 2 + }, + "python_reference": { + "median_ms": 2.431602, + "samples_ms": [ + 2.44435, + 2.653246, + 2.477419, + 2.456598, + 2.422949, + 2.460394, + 2.407916, + 2.711401, + 2.420565, + 2.639095, + 2.438401, + 2.379956, + 2.412964, + 2.645485, + 2.485821, + 2.392715, + 2.416279, + 2.523427, + 2.529265, + 2.377813, + 2.394807, + 2.405023, + 2.431602, + 2.652205, + 2.427957, + 2.429599, + 2.398073, + 2.391563, + 2.501023, + 2.38936, + 2.468225 + ] + }, + "ratio": 1.207838289325309 + }, + "string_build": { + "luapyre": { + "median_ms": 0.509536, + "samples_ms": [ + 0.508956, + 0.527332, + 0.511399, + 0.517468, + 0.51889, + 0.52572, + 0.524158, + 0.505861, + 0.519101, + 0.494524, + 0.507994, + 0.72386, + 0.568653, + 0.509536, + 0.558699, + 0.564767, + 0.536826, + 0.522455, + 0.506431, + 0.578287, + 0.501364, + 0.507703, + 0.503357, + 0.507394, + 0.487905, + 0.497489, + 0.4844, + 0.514724, + 0.485792, + 0.495406, + 0.507283 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 2875 + }, + "python_reference": { + "median_ms": 0.363092, + "samples_ms": [ + 0.363092, + 0.359757, + 0.370513, + 0.578658, + 0.392295, + 0.360308, + 0.371103, + 0.357224, + 0.355411, + 0.394197, + 0.35465, + 0.35484, + 0.369721, + 0.40981, + 0.362541, + 0.386316, + 0.355901, + 0.369732, + 0.362241, + 0.383592, + 0.373206, + 0.355471, + 0.35482, + 0.370473, + 0.359697, + 0.370222, + 0.356722, + 0.355812, + 0.374518, + 0.355501, + 0.369672 + ] + }, + "ratio": 1.4033247771914554 + }, + "table_mix": { + "luapyre": { + "median_ms": 22.397316, + "samples_ms": [ + 22.397316, + 23.819059, + 22.934503, + 46.418732, + 23.316782, + 22.115273, + 21.837176, + 21.964833, + 21.761135, + 22.940302, + 21.980936, + 22.61156, + 22.842198, + 24.847866, + 23.352926, + 22.002548, + 23.711741, + 22.118358, + 22.395754, + 21.697761, + 21.952185, + 21.633377, + 22.378859, + 22.561197, + 21.645885, + 21.863945, + 22.749883, + 24.611789, + 22.839784, + 22.4477, + 21.572628 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 2, + "loop_executions": 2, + "loop_iterations": 22890 + }, + "python_reference": { + "median_ms": 2.520292, + "samples_ms": [ + 2.56673, + 2.524508, + 2.516826, + 2.669681, + 2.601891, + 2.528784, + 2.497999, + 2.509185, + 2.514323, + 2.503517, + 2.623322, + 2.504218, + 2.516706, + 2.507202, + 2.507884, + 2.521863, + 2.530757, + 2.493402, + 2.517297, + 2.498019, + 2.547031, + 2.515204, + 2.578787, + 2.517037, + 2.520292, + 2.585898, + 2.622632, + 2.61514, + 2.675009, + 2.56699, + 2.511849 + ] + }, + "ratio": 8.886794069893488 + }, + "typed_arith": { + "luapyre": { + "median_ms": 0.776518, + "samples_ms": [ + 0.866509, + 0.80586, + 0.926848, + 1.050909, + 0.776518, + 0.740715, + 0.823417, + 0.761446, + 0.845569, + 0.77137, + 0.73754, + 0.849754, + 0.774584, + 0.784408, + 0.738381, + 0.78512, + 0.836826, + 0.757289, + 0.781474, + 0.736418, + 0.748857, + 0.7511, + 0.751941, + 0.751902, + 0.750049, + 0.750359, + 0.819841, + 0.787874, + 0.77149, + 0.776868, + 0.832379 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 0.49857, + "samples_ms": [ + 0.757399, + 0.515275, + 1.323278, + 0.58033, + 0.490419, + 0.543756, + 0.489156, + 0.503387, + 0.489607, + 0.49886, + 0.489597, + 0.51242, + 0.58049, + 0.564137, + 0.489056, + 0.500202, + 0.489387, + 0.49839, + 0.490598, + 0.513482, + 0.490639, + 0.501295, + 0.49132, + 0.49871, + 0.489477, + 0.49857, + 0.49178, + 0.51195, + 0.489327, + 0.496197, + 0.489066 + ] + }, + "ratio": 1.5574904226086608 + }, + "typed_branch": { + "luapyre": { + "median_ms": 1.888075, + "samples_ms": [ + 1.842849, + 2.899937, + 2.001071, + 1.915556, + 1.853965, + 1.91824, + 1.912471, + 2.067377, + 1.888075, + 1.935405, + 1.89796, + 2.07579, + 1.901926, + 2.028171, + 1.852023, + 2.213802, + 1.865342, + 1.915386, + 1.853034, + 1.884731, + 1.929866, + 1.832845, + 1.837672, + 1.857331, + 1.833105, + 1.842389, + 1.813556, + 1.853855, + 1.890228, + 1.848908, + 1.828759 + ] + }, + "luapyre_warm_counters": { + "loop_entry_executions": 1, + "loop_executions": 1, + "loop_iterations": 30000 + }, + "python_reference": { + "median_ms": 1.135624, + "samples_ms": [ + 1.121883, + 1.137276, + 1.125629, + 1.124216, + 1.14002, + 1.134131, + 1.164797, + 1.245254, + 1.139058, + 1.122574, + 1.136495, + 1.149183, + 1.124177, + 1.125388, + 1.123726, + 1.131747, + 1.117216, + 1.121263, + 1.132439, + 1.135624, + 1.290961, + 1.12005, + 1.13355, + 1.194199, + 1.193068, + 1.200739, + 1.13989, + 1.120972, + 1.150475, + 1.342196, + 1.387071 + ] + }, + "ratio": 1.6625881453720597 + } + } + } + ], + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre_median_ms": 21.113606, + "luapyre_process_medians_ms": [ + 20.590567, + 21.113606, + 21.352334 + ], + "python_median_ms": 1.957617, + "python_process_medians_ms": [ + 1.955856, + 1.957617, + 1.994541 + ], + "ratio": 10.785360977147215 + }, + "fib_recursive": { + "luapyre_median_ms": 19.054858, + "luapyre_process_medians_ms": [ + 18.999467, + 19.054858, + 19.119856 + ], + "python_median_ms": 0.635831, + "python_process_medians_ms": [ + 0.63497, + 0.635831, + 0.642531 + ], + "ratio": 29.968431863183767 + }, + "python_calls_1000": { + "luapyre_median_ms": 0.601671, + "luapyre_process_medians_ms": [ + 0.598396, + 0.601671, + 0.610845 + ], + "python_median_ms": 0.04061, + "python_process_medians_ms": [ + 0.04045, + 0.04061, + 0.04069 + ], + "ratio": 14.815833538537305 + }, + "sieve": { + "luapyre_median_ms": 6.633669, + "luapyre_process_medians_ms": [ + 6.592786, + 6.633669, + 6.84059 + ], + "python_median_ms": 0.508715, + "python_process_medians_ms": [ + 0.506532, + 0.508715, + 0.510999 + ], + "ratio": 13.040049929724894 + }, + "spectral_norm": { + "luapyre_median_ms": 2.92258, + "luapyre_process_medians_ms": [ + 2.808774, + 2.92258, + 2.936982 + ], + "python_median_ms": 2.367538, + "python_process_medians_ms": [ + 2.343372, + 2.367538, + 2.431602 + ], + "ratio": 1.234438475749914 + }, + "string_build": { + "luapyre_median_ms": 0.516006, + "luapyre_process_medians_ms": [ + 0.509536, + 0.516006, + 0.537468 + ], + "python_median_ms": 0.365165, + "python_process_medians_ms": [ + 0.363092, + 0.365165, + 0.370292 + ], + "ratio": 1.4130762805854886 + }, + "table_mix": { + "luapyre_median_ms": 22.343714, + "luapyre_process_medians_ms": [ + 22.023836, + 22.343714, + 22.397316 + ], + "python_median_ms": 2.520292, + "python_process_medians_ms": [ + 2.507523, + 2.520292, + 2.567891 + ], + "ratio": 8.865525899379913 + }, + "typed_arith": { + "luapyre_median_ms": 0.731561, + "luapyre_process_medians_ms": [ + 0.731111, + 0.731561, + 0.776518 + ], + "python_median_ms": 0.49857, + "python_process_medians_ms": [ + 0.49155, + 0.49857, + 0.511109 + ], + "ratio": 1.467318530998656 + }, + "typed_branch": { + "luapyre_median_ms": 1.892722, + "luapyre_process_medians_ms": [ + 1.888075, + 1.892722, + 1.896749 + ], + "python_median_ms": 1.14004, + "python_process_medians_ms": [ + 1.135624, + 1.14004, + 1.155743 + ], + "ratio": 1.660224202659556 + } + } + } + ] +} diff --git a/benchmarks/results/language_benchmarks_040_313.json b/benchmarks/results/language_benchmarks_040_313.json new file mode 100644 index 0000000..4a44e8b --- /dev/null +++ b/benchmarks/results/language_benchmarks_040_313.json @@ -0,0 +1,2293 @@ +{ + "schema_version": 1, + "revision": "0.40-language", + "python": "3.13.15", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "n_body": { + "luapyre": { + "median_ms": 752.16908, + "samples_ms": [ + 745.487306, + 733.650185, + 744.993061, + 752.16908, + 738.68521, + 768.470179, + 735.939857, + 735.839285, + 719.732666, + 766.462105, + 763.825264, + 744.413823, + 757.119042, + 753.279234, + 763.992628, + 743.173076, + 747.996462, + 740.430078, + 755.634966, + 790.084028, + 846.05305, + 853.505807, + 764.300994, + 785.771747, + 736.179551, + 742.974967, + 724.585783, + 726.859378, + 752.433568, + 769.923218, + 779.569653 + ] + }, + "native_lua55": { + "median_ms": 2.091349, + "samples_ms": [ + 1.895152, + 1.872079, + 1.995949, + 1.865058, + 1.879008, + 1.936903, + 2.15412, + 1.934028, + 2.243662, + 2.234118, + 2.169634, + 2.101694, + 2.19496, + 2.060784, + 2.074494, + 2.092952, + 2.144356, + 2.091349, + 2.053464, + 2.055967, + 2.065581, + 2.116726, + 2.082757, + 2.070829, + 2.118128, + 2.080974, + 2.118409, + 2.135534, + 2.270781, + 2.097358, + 2.102876 + ] + }, + "python_lower_bound": { + "median_ms": 4.847831, + "samples_ms": [ + 4.583114, + 4.836765, + 5.33985, + 4.941337, + 4.485991, + 4.933015, + 4.566939, + 4.631344, + 4.448306, + 4.377182, + 4.847831, + 4.550115, + 4.581811, + 5.111697, + 4.852117, + 4.503988, + 4.521964, + 4.515124, + 4.500483, + 4.55325, + 4.326979, + 5.937194, + 6.988212, + 7.044874, + 7.015381, + 7.039616, + 6.984706, + 7.05625, + 7.015711, + 7.004215, + 6.950366 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "luapyre": { + "median_ms": 459.138868, + "samples_ms": [ + 539.471889, + 443.173703, + 427.726441, + 427.409851, + 437.450329, + 427.523146, + 426.32943, + 439.648039, + 500.303716, + 474.146758, + 452.718799, + 438.068433, + 464.992681, + 431.480655, + 469.226286, + 447.50456, + 543.855418, + 441.268461, + 442.273934, + 469.41906, + 543.203245, + 536.437191, + 459.138868, + 507.918422, + 462.408674, + 448.173327, + 546.720962, + 465.88074, + 434.175245, + 467.612259, + 461.224038 + ] + }, + "native_lua55": { + "median_ms": 1.325827, + "samples_ms": [ + 1.219562, + 1.194215, + 1.353828, + 1.256937, + 1.494504, + 1.201696, + 1.348851, + 1.309103, + 1.360709, + 1.236758, + 2.603334, + 1.207104, + 1.325827, + 1.202718, + 2.067101, + 1.625566, + 1.357294, + 1.180826, + 1.741275, + 1.25895, + 1.56722, + 1.452543, + 1.200825, + 1.563705, + 1.19701, + 1.499091, + 1.311526, + 1.44428, + 1.255675, + 1.277908, + 1.401738 + ] + }, + "python_lower_bound": { + "median_ms": 5.442693, + "samples_ms": [ + 5.514769, + 5.491676, + 5.48766, + 5.449905, + 5.46732, + 5.573615, + 5.445759, + 5.476634, + 5.457675, + 5.418187, + 5.489272, + 5.442693, + 5.466168, + 5.461492, + 5.45398, + 5.430796, + 5.492307, + 5.427933, + 5.485156, + 5.442003, + 4.218335, + 4.105189, + 4.035768, + 4.027976, + 4.049168, + 4.047835, + 4.344619, + 4.110077, + 4.038191, + 4.027395, + 4.041006 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "luapyre": { + "median_ms": 689.524841, + "samples_ms": [ + 739.480332, + 689.524841, + 712.729421, + 734.944155, + 695.429719, + 669.67856, + 681.646374, + 698.739993, + 673.849369, + 679.019297, + 709.663369, + 748.585558, + 724.088129, + 721.926082, + 685.47068, + 759.526363, + 723.567521, + 811.144429, + 708.478327, + 670.610488, + 675.263466, + 679.937145, + 719.332018, + 687.049151, + 667.188974, + 674.400286, + 682.014636, + 672.429855, + 742.930814, + 689.299103, + 684.795942 + ] + }, + "native_lua55": { + "median_ms": 3.187255, + "samples_ms": [ + 3.187255, + 3.244469, + 3.299319, + 3.680597, + 3.362061, + 3.208836, + 3.135239, + 3.138324, + 3.120908, + 3.414658, + 3.625417, + 3.461897, + 3.336053, + 3.280121, + 3.174016, + 3.174857, + 3.188697, + 3.18399, + 3.16337, + 3.247563, + 3.109891, + 3.128609, + 3.114459, + 3.128259, + 3.100188, + 3.123612, + 3.508225, + 3.256597, + 3.143851, + 3.139204, + 3.294802 + ] + }, + "python_lower_bound": { + "median_ms": 5.981145, + "samples_ms": [ + 6.481657, + 5.969357, + 5.979522, + 5.925663, + 5.981725, + 6.286281, + 5.969527, + 6.291879, + 5.973123, + 6.10815, + 5.95136, + 6.017577, + 5.969768, + 5.945963, + 6.2992, + 6.252672, + 6.065167, + 5.929859, + 5.978621, + 6.342904, + 6.066829, + 5.947305, + 6.151624, + 6.277608, + 6.038208, + 5.959964, + 6.013231, + 5.956789, + 5.981145, + 5.926935, + 5.955937 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "luapyre": { + "median_ms": 178.052044, + "samples_ms": [ + 173.887385, + 175.342122, + 178.414954, + 175.669491, + 187.134692, + 222.237892, + 175.950392, + 181.505883, + 178.052044, + 180.576905, + 174.658443, + 173.831624, + 179.667486, + 177.431026, + 187.303659, + 179.196526, + 174.021381, + 178.393283, + 174.779921, + 175.282766, + 184.934776, + 174.29411, + 173.500089, + 175.496287, + 174.630813, + 184.396094, + 201.335913, + 182.221877, + 182.034216, + 193.264921, + 174.673615 + ] + }, + "native_lua55": { + "median_ms": 0.585607, + "samples_ms": [ + 0.598496, + 0.651924, + 0.602152, + 0.594571, + 0.584776, + 0.593599, + 0.579298, + 0.583714, + 0.612196, + 0.581341, + 0.611845, + 0.579958, + 0.589492, + 0.585607, + 0.57442, + 0.600839, + 0.579628, + 0.584626, + 0.571817, + 0.589543, + 0.58755, + 0.575051, + 0.602051, + 0.574951, + 0.582202, + 0.592828, + 0.576264, + 0.588441, + 0.579327, + 0.603062, + 0.570575 + ] + }, + "python_lower_bound": { + "median_ms": 1.016398, + "samples_ms": [ + 1.009607, + 1.010298, + 1.243279, + 1.038059, + 1.012111, + 1.01134, + 1.013904, + 1.028635, + 1.012532, + 1.013913, + 1.012692, + 1.028345, + 1.010168, + 1.013183, + 1.014895, + 1.030368, + 1.009538, + 1.021946, + 1.016398, + 1.03173, + 1.017509, + 1.014354, + 1.031649, + 1.226695, + 1.020383, + 1.017198, + 1.323256, + 1.022356, + 1.015446, + 1.015936, + 1.03206 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "luapyre": { + "median_ms": 136.570972, + "samples_ms": [ + 143.320609, + 159.756515, + 150.611517, + 206.466796, + 170.07476, + 149.39657, + 135.861858, + 135.605644, + 136.554453, + 134.542788, + 133.899948, + 135.059634, + 138.606626, + 135.113322, + 136.798891, + 136.177178, + 134.249188, + 134.075353, + 134.092519, + 139.55145, + 138.37526, + 137.231428, + 135.312663, + 136.250354, + 136.570972, + 135.065731, + 135.304952, + 158.389844, + 150.108946, + 200.465789, + 156.296233 + ] + }, + "native_lua55": { + "median_ms": 0.508353, + "samples_ms": [ + 0.514242, + 0.514062, + 0.557685, + 0.553099, + 0.492451, + 0.50612, + 0.497107, + 0.510216, + 0.497537, + 0.515023, + 0.500672, + 0.537065, + 0.501424, + 0.507813, + 0.494633, + 0.508353, + 0.496296, + 0.509635, + 0.512469, + 0.512299, + 0.492219, + 0.522744, + 0.497537, + 0.510536, + 0.495084, + 0.508603, + 0.523345, + 0.506099, + 0.490407, + 0.513301, + 0.491399 + ] + }, + "python_lower_bound": { + "median_ms": 0.919664, + "samples_ms": [ + 0.917342, + 0.907146, + 0.987434, + 0.93741, + 0.907117, + 0.914818, + 0.907948, + 0.909169, + 1.331067, + 0.910291, + 1.035324, + 0.939133, + 0.975466, + 0.919664, + 0.983959, + 0.93643, + 0.899185, + 0.919004, + 0.923009, + 0.922469, + 1.203149, + 0.947545, + 0.925804, + 0.914107, + 0.938912, + 0.909519, + 0.900707, + 0.903511, + 0.924853, + 0.899015, + 0.910251 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "luapyre": { + "median_ms": 101.272538, + "samples_ms": [ + 99.147215, + 99.320619, + 99.838455, + 105.221786, + 101.833522, + 108.514791, + 106.002157, + 100.674952, + 100.204965, + 100.268878, + 99.986996, + 100.220368, + 99.051126, + 103.346193, + 99.767084, + 102.329054, + 102.72321, + 101.272538, + 99.90841, + 99.999905, + 98.940024, + 99.263147, + 99.214484, + 108.221594, + 106.5076, + 105.278712, + 123.211951, + 111.199753, + 111.83637, + 109.544056, + 105.701756 + ] + }, + "native_lua55": { + "median_ms": 0.453473, + "samples_ms": [ + 0.463218, + 0.452763, + 0.476708, + 0.452432, + 0.450169, + 0.465982, + 0.447686, + 0.465792, + 0.448045, + 0.462006, + 0.448336, + 0.47844, + 0.448206, + 0.459533, + 0.448236, + 0.463258, + 0.445672, + 0.444851, + 0.453473, + 0.460584, + 0.455436, + 0.446023, + 0.454946, + 0.446383, + 0.455637, + 0.44397, + 0.657392, + 0.509136, + 0.491189, + 0.446113, + 0.453024 + ] + }, + "python_lower_bound": { + "median_ms": 0.396741, + "samples_ms": [ + 0.386535, + 0.441937, + 0.409619, + 0.384603, + 0.399365, + 0.387717, + 0.386816, + 0.401428, + 0.388449, + 0.415378, + 0.387747, + 0.385023, + 0.400656, + 0.385985, + 0.396741, + 0.387257, + 0.38981, + 0.402329, + 0.393496, + 0.421908, + 0.446994, + 0.385694, + 0.409409, + 0.389931, + 0.401047, + 0.392675, + 0.385995, + 0.400276, + 0.3968, + 0.422558, + 0.400626 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "n_body": { + "native_lua55": { + "median_ms": 2.160572, + "samples_ms": [ + 2.593226, + 2.318414, + 2.277563, + 2.160572, + 2.252307, + 2.52738, + 2.071692, + 2.442034, + 2.288399, + 2.273387, + 2.144028, + 2.09742, + 2.151369, + 2.183967, + 2.103609, + 2.105261, + 2.493119, + 2.192439, + 2.250884, + 2.549992, + 2.12428, + 2.056921, + 2.105161, + 2.153793, + 2.069229, + 2.128295, + 2.387765, + 2.066876, + 2.111011, + 2.112742, + 2.326625 + ] + }, + "python_lower_bound": { + "median_ms": 5.452781, + "samples_ms": [ + 6.918617, + 6.882144, + 6.929273, + 6.957104, + 6.914591, + 6.889054, + 6.871078, + 6.868744, + 6.896195, + 5.377411, + 4.520988, + 4.562758, + 5.362209, + 5.476136, + 4.759016, + 4.66576, + 4.573875, + 4.756752, + 4.840014, + 4.765436, + 4.844381, + 4.676364, + 4.733098, + 4.955313, + 5.632194, + 5.452781, + 5.03539, + 7.099857, + 6.942883, + 7.032103, + 6.972907 + ] + }, + "luapyre": { + "median_ms": 762.422076, + "samples_ms": [ + 764.971767, + 782.60948, + 735.411532, + 749.132459, + 742.90179, + 732.452069, + 739.264694, + 743.891572, + 726.934128, + 733.64582, + 733.437905, + 738.329617, + 759.01212, + 763.486886, + 756.810176, + 805.586211, + 765.356639, + 786.42955, + 766.164885, + 772.56399, + 772.093313, + 749.408799, + 751.24058, + 797.523362, + 779.188304, + 777.709217, + 765.488605, + 862.469344, + 762.422076, + 775.326744, + 716.307241 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "native_lua55": { + "median_ms": 1.221779, + "samples_ms": [ + 1.219736, + 1.604149, + 1.227608, + 1.211514, + 1.276389, + 1.271001, + 1.225214, + 1.20893, + 1.224764, + 1.210983, + 1.223792, + 1.226727, + 1.20861, + 1.212085, + 1.232765, + 1.213016, + 1.219235, + 1.210582, + 1.22217, + 1.223092, + 1.210863, + 1.309548, + 1.249529, + 1.221779, + 1.501208, + 1.155722, + 1.164335, + 1.169753, + 1.315677, + 1.162762, + 1.213527 + ] + }, + "python_lower_bound": { + "median_ms": 4.229524, + "samples_ms": [ + 4.270544, + 4.205458, + 4.173992, + 4.292015, + 4.496725, + 4.216645, + 4.189145, + 4.400505, + 4.464878, + 4.190266, + 4.229524, + 4.182965, + 4.191919, + 4.204708, + 4.196676, + 4.265576, + 4.184278, + 4.206791, + 4.162014, + 4.165761, + 4.162666, + 4.174503, + 4.246608, + 6.714623, + 6.900896, + 6.748953, + 6.855359, + 7.212341, + 7.446715, + 6.337941, + 6.386152 + ] + }, + "luapyre": { + "median_ms": 436.953648, + "samples_ms": [ + 424.667536, + 425.208613, + 428.778306, + 477.309409, + 436.953648, + 418.338087, + 436.412841, + 453.436104, + 453.649895, + 420.639674, + 411.087565, + 454.404987, + 437.611391, + 423.000856, + 424.844386, + 419.738821, + 419.476206, + 439.057218, + 429.047138, + 438.678919, + 447.574367, + 453.33187, + 437.650523, + 556.629024, + 433.131393, + 445.030317, + 488.495847, + 484.309269, + 412.463967, + 417.548561, + 491.157445 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "native_lua55": { + "median_ms": 3.218722, + "samples_ms": [ + 3.195899, + 3.188798, + 3.40165, + 3.21678, + 3.259262, + 3.152075, + 3.172534, + 3.497069, + 3.171633, + 3.433938, + 3.219484, + 3.170172, + 3.33322, + 3.766033, + 3.723541, + 3.255877, + 3.484541, + 3.220125, + 3.198522, + 3.176861, + 3.325869, + 3.160337, + 3.155851, + 3.218722, + 3.156571, + 3.364605, + 3.49041, + 3.192484, + 3.109172, + 3.109382, + 3.474437 + ] + }, + "python_lower_bound": { + "median_ms": 6.335115, + "samples_ms": [ + 6.420339, + 6.396625, + 6.126259, + 6.114081, + 6.144336, + 6.375854, + 6.673099, + 6.335115, + 6.406369, + 6.767617, + 6.105789, + 6.199035, + 6.322656, + 6.301515, + 6.343357, + 6.449202, + 6.417905, + 6.447179, + 6.184404, + 6.14019, + 6.180108, + 6.282488, + 6.58419, + 6.729582, + 6.20172, + 6.271902, + 6.133149, + 6.184455, + 6.723763, + 6.559093, + 6.354563 + ] + }, + "luapyre": { + "median_ms": 726.213412, + "samples_ms": [ + 723.598901, + 793.816025, + 748.346773, + 753.93856, + 709.621578, + 689.263849, + 700.588589, + 704.92696, + 687.063267, + 715.727552, + 714.435271, + 726.213412, + 768.541353, + 768.213738, + 873.720426, + 789.954958, + 701.076487, + 688.404396, + 801.747012, + 846.409172, + 900.995708, + 737.424345, + 704.436924, + 698.699845, + 679.957548, + 709.804923, + 707.074151, + 755.018723, + 797.508269, + 863.697767, + 738.325177 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "native_lua55": { + "median_ms": 0.589253, + "samples_ms": [ + 0.586028, + 0.595071, + 0.600479, + 0.592988, + 0.586939, + 0.576394, + 0.588181, + 0.577516, + 0.58722, + 0.60121, + 0.580089, + 0.590405, + 0.580239, + 0.589583, + 0.574641, + 0.598937, + 0.612647, + 0.576824, + 0.596523, + 0.577836, + 0.594421, + 0.589253, + 0.582052, + 0.617444, + 0.574561, + 0.588692, + 0.58748, + 0.660548, + 0.60757, + 0.602733, + 0.593769 + ] + }, + "python_lower_bound": { + "median_ms": 1.018381, + "samples_ms": [ + 1.001005, + 1.005422, + 1.009508, + 1.026513, + 0.996889, + 1.016188, + 1.226155, + 1.027825, + 1.006503, + 1.135312, + 1.021866, + 1.022928, + 1.187228, + 1.059972, + 1.024379, + 0.994716, + 1.061634, + 1.023038, + 1.132769, + 1.0174, + 1.068986, + 1.007315, + 1.029677, + 1.018381, + 1.006123, + 1.006574, + 1.017841, + 0.999864, + 1.006284, + 1.004811, + 1.025371 + ] + }, + "luapyre": { + "median_ms": 180.492207, + "samples_ms": [ + 181.294861, + 208.399977, + 203.484472, + 191.229113, + 178.583813, + 176.441647, + 169.934431, + 173.021871, + 177.835372, + 176.500943, + 170.6029, + 171.137763, + 168.551296, + 178.24752, + 204.364826, + 174.938918, + 187.777049, + 174.007774, + 177.375584, + 214.197419, + 185.681803, + 209.175098, + 225.913975, + 185.21907, + 197.249666, + 192.272661, + 175.511986, + 180.492207, + 176.413855, + 246.347826, + 237.805222 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "native_lua55": { + "median_ms": 0.559208, + "samples_ms": [ + 0.676551, + 0.640918, + 0.536836, + 0.611014, + 0.530787, + 0.560841, + 0.531908, + 0.625896, + 0.811768, + 0.542504, + 0.604915, + 0.559719, + 0.537957, + 0.6762, + 0.704271, + 0.559208, + 0.548553, + 0.537897, + 0.542754, + 0.534232, + 0.605927, + 0.670903, + 0.563755, + 0.545769, + 0.528764, + 0.549645, + 0.536425, + 0.548773, + 0.613708, + 0.560431, + 0.522455 + ] + }, + "python_lower_bound": { + "median_ms": 0.964722, + "samples_ms": [ + 1.037188, + 0.940827, + 1.850528, + 0.96363, + 0.91582, + 0.982929, + 1.047543, + 1.153098, + 0.979664, + 1.07292, + 1.136153, + 0.895712, + 0.906868, + 0.905315, + 1.343326, + 0.988637, + 0.897423, + 0.919205, + 1.203221, + 0.93601, + 1.028726, + 0.907999, + 1.015787, + 0.922551, + 0.963971, + 0.918405, + 0.964722, + 0.934568, + 0.902711, + 1.142753, + 0.968928 + ] + }, + "luapyre": { + "median_ms": 138.801097, + "samples_ms": [ + 137.586861, + 141.587542, + 144.788954, + 145.420949, + 140.716649, + 141.624486, + 137.37422, + 138.442942, + 141.111626, + 139.765397, + 140.07536, + 137.863215, + 138.512555, + 138.801097, + 137.749238, + 138.749732, + 142.533145, + 143.251286, + 139.787469, + 136.429137, + 137.804079, + 138.400771, + 137.617342, + 143.560012, + 140.677792, + 160.771965, + 133.645046, + 134.886934, + 133.976782, + 134.900559, + 139.506677 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "native_lua55": { + "median_ms": 0.441867, + "samples_ms": [ + 0.486071, + 0.742747, + 0.497959, + 0.463058, + 0.436419, + 0.597525, + 0.433575, + 0.528554, + 0.42992, + 0.432713, + 0.440986, + 0.433715, + 0.441867, + 0.433184, + 0.466853, + 0.487213, + 0.456098, + 0.431391, + 0.429128, + 0.447605, + 0.42989, + 0.445032, + 0.43039, + 0.463839, + 0.429559, + 0.445963, + 0.432823, + 0.433274, + 0.65569, + 0.47144, + 0.431832 + ] + }, + "python_lower_bound": { + "median_ms": 0.390782, + "samples_ms": [ + 0.405574, + 0.388058, + 0.409259, + 0.402279, + 0.389771, + 0.390782, + 0.403281, + 0.390682, + 0.406775, + 0.389991, + 0.387037, + 0.40256, + 0.388238, + 0.413285, + 0.388549, + 0.386236, + 0.405424, + 0.391954, + 0.402109, + 0.388849, + 0.386546, + 0.398504, + 0.388409, + 0.416269, + 0.456037, + 0.388268, + 0.417922, + 0.387397, + 0.398012, + 0.385915, + 0.386035 + ] + }, + "luapyre": { + "median_ms": 106.122433, + "samples_ms": [ + 132.864002, + 101.332736, + 100.801363, + 113.086078, + 100.105575, + 148.687759, + 135.42044, + 116.297315, + 142.083675, + 132.057978, + 101.794927, + 101.745294, + 100.974486, + 101.19512, + 101.228959, + 106.122433, + 100.472842, + 105.735086, + 154.81593, + 108.678485, + 140.268353, + 112.358706, + 155.648703, + 149.734313, + 130.864675, + 133.471593, + 102.328575, + 103.643179, + 103.94376, + 100.181902, + 101.467403 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "n_body": { + "python_lower_bound": { + "median_ms": 4.881808, + "samples_ms": [ + 4.481962, + 4.452829, + 4.881808, + 4.586375, + 5.454336, + 5.727265, + 4.50093, + 4.647855, + 4.803934, + 4.647615, + 5.460475, + 5.368551, + 4.803333, + 4.727592, + 4.394495, + 4.943438, + 4.794149, + 6.124848, + 4.950418, + 5.035002, + 4.950839, + 5.353629, + 5.319318, + 4.406692, + 5.720326, + 5.140116, + 4.519217, + 4.579976, + 4.77447, + 4.985709, + 5.003145 + ] + }, + "luapyre": { + "median_ms": 787.748142, + "samples_ms": [ + 813.052802, + 792.382827, + 735.252345, + 740.674558, + 724.809776, + 722.725892, + 846.559019, + 819.81526, + 736.628529, + 788.560891, + 764.513605, + 728.054022, + 749.772772, + 807.509314, + 820.829249, + 840.412537, + 782.996033, + 836.375416, + 941.389335, + 809.126134, + 728.226152, + 749.118548, + 874.180577, + 787.748142, + 1110.781998, + 869.01747, + 770.787711, + 773.376638, + 741.530433, + 752.059321, + 869.773201 + ] + }, + "native_lua55": { + "median_ms": 2.006038, + "samples_ms": [ + 2.009533, + 1.955513, + 2.073115, + 1.980239, + 2.02107, + 1.956055, + 2.234152, + 2.027569, + 1.978988, + 2.016623, + 1.976905, + 1.937828, + 1.954502, + 1.940522, + 1.9462, + 2.13734, + 2.006038, + 1.966439, + 2.075569, + 1.975994, + 2.143669, + 2.095178, + 2.092405, + 2.042491, + 1.998476, + 1.941683, + 2.005096, + 1.924409, + 2.02142, + 2.257146, + 2.262713 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "python_lower_bound": { + "median_ms": 4.26761, + "samples_ms": [ + 5.470962, + 5.466736, + 5.450803, + 5.511613, + 5.456952, + 5.514546, + 5.493977, + 5.463872, + 5.485144, + 5.452375, + 4.615449, + 4.918253, + 4.118943, + 4.11032, + 4.141325, + 4.065755, + 4.173062, + 4.228012, + 4.122587, + 4.071333, + 4.069911, + 4.062389, + 4.057503, + 4.293198, + 5.898198, + 4.135507, + 4.133214, + 4.26761, + 4.674556, + 4.261141, + 4.205439 + ] + }, + "luapyre": { + "median_ms": 427.829748, + "samples_ms": [ + 433.208311, + 422.863672, + 498.498928, + 445.196173, + 428.319942, + 422.291462, + 446.003264, + 422.819727, + 499.43779, + 519.274803, + 467.780828, + 460.282921, + 467.456398, + 428.088188, + 445.928121, + 441.249578, + 447.804663, + 420.647043, + 427.829748, + 424.035566, + 416.883241, + 425.012818, + 447.0031, + 420.754139, + 419.835034, + 416.456786, + 412.413064, + 415.491558, + 413.095452, + 417.989123, + 416.306356 + ] + }, + "native_lua55": { + "median_ms": 1.219467, + "samples_ms": [ + 1.225115, + 1.207659, + 1.209201, + 1.220959, + 1.222451, + 1.209272, + 1.208821, + 1.224094, + 1.20756, + 1.221109, + 1.22164, + 1.210334, + 1.209403, + 1.236151, + 1.20802, + 1.216292, + 1.209252, + 1.221279, + 1.466448, + 1.219467, + 1.228911, + 1.296149, + 1.209172, + 1.22891, + 1.20838, + 1.208831, + 1.237633, + 1.207679, + 1.208351, + 1.240889, + 1.223753 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "python_lower_bound": { + "median_ms": 6.000702, + "samples_ms": [ + 5.916029, + 6.078135, + 6.325238, + 5.988324, + 6.011859, + 5.993502, + 5.988795, + 6.000543, + 5.974314, + 6.549486, + 6.389743, + 6.107489, + 5.949548, + 6.050605, + 5.953523, + 6.060781, + 5.961255, + 6.295305, + 5.962056, + 5.963288, + 6.065988, + 6.138975, + 5.945472, + 5.962276, + 5.950329, + 5.965531, + 6.188167, + 6.000702, + 6.262837, + 6.383944, + 6.404574 + ] + }, + "luapyre": { + "median_ms": 711.136892, + "samples_ms": [ + 688.106768, + 696.41277, + 686.752043, + 687.608057, + 696.278594, + 752.098481, + 746.883217, + 687.721462, + 844.518996, + 746.895928, + 783.088635, + 748.714533, + 865.741915, + 827.043216, + 733.41751, + 711.136892, + 749.309217, + 748.454994, + 737.905674, + 787.312928, + 714.538685, + 706.796211, + 687.030275, + 734.724292, + 703.981858, + 681.424661, + 672.655809, + 710.152238, + 684.69766, + 686.329962, + 670.258826 + ] + }, + "native_lua55": { + "median_ms": 3.174861, + "samples_ms": [ + 3.169042, + 3.166439, + 3.188501, + 3.172818, + 3.156334, + 3.233797, + 3.154291, + 3.151477, + 3.457886, + 3.13313, + 3.350678, + 3.219527, + 3.272294, + 3.738718, + 3.223503, + 3.328587, + 3.171115, + 3.53568, + 3.156615, + 3.148323, + 3.154121, + 3.156124, + 3.186188, + 3.140071, + 3.132049, + 3.199617, + 3.234459, + 3.147482, + 3.638561, + 3.191496, + 3.174861 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "python_lower_bound": { + "median_ms": 1.006144, + "samples_ms": [ + 1.011231, + 1.019443, + 1.000886, + 1.004422, + 1.006144, + 1.014746, + 1.004131, + 1.069006, + 1.019975, + 1.07848, + 0.997461, + 1.009559, + 1.240668, + 1.05139, + 1.006585, + 0.995198, + 1.006565, + 0.998814, + 1.001928, + 0.988338, + 1.004712, + 1.005974, + 0.990241, + 0.992774, + 1.01042, + 0.997181, + 0.996039, + 1.025282, + 1.055947, + 1.015328, + 0.999685 + ] + }, + "luapyre": { + "median_ms": 180.128886, + "samples_ms": [ + 173.9753, + 171.770299, + 170.970898, + 194.652851, + 169.847222, + 173.92102, + 180.128886, + 178.467292, + 184.833129, + 181.591249, + 178.93018, + 177.802657, + 182.844175, + 177.964205, + 181.325529, + 184.907167, + 195.368985, + 195.968229, + 186.82019, + 190.637984, + 193.988804, + 184.904563, + 188.570671, + 183.577529, + 216.318822, + 173.956573, + 172.115766, + 171.685074, + 178.325524, + 171.420096, + 170.59661 + ] + }, + "native_lua55": { + "median_ms": 0.602973, + "samples_ms": [ + 0.603464, + 0.584536, + 0.614921, + 0.602954, + 0.591787, + 0.619798, + 0.591827, + 0.598637, + 0.610595, + 0.593199, + 0.611656, + 0.609052, + 0.608481, + 0.609213, + 0.593049, + 0.602973, + 0.591707, + 0.604356, + 0.622412, + 0.591947, + 0.60724, + 0.592819, + 0.603935, + 0.597966, + 0.590756, + 0.618657, + 0.589875, + 0.605497, + 0.601782, + 0.588413, + 0.605517 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "python_lower_bound": { + "median_ms": 0.939556, + "samples_ms": [ + 0.927599, + 0.932687, + 1.045401, + 0.95556, + 0.996009, + 0.942271, + 0.946036, + 0.919637, + 0.932727, + 0.91463, + 0.942922, + 0.921149, + 0.927909, + 1.004813, + 0.945806, + 0.916282, + 0.913879, + 0.976912, + 0.967187, + 0.924654, + 0.942681, + 0.921911, + 0.934109, + 0.958274, + 0.939556, + 0.93461, + 0.929782, + 0.942351, + 0.986335, + 0.944864, + 0.923483 + ] + }, + "luapyre": { + "median_ms": 140.532259, + "samples_ms": [ + 136.611404, + 138.02145, + 161.65671, + 140.532259, + 137.938769, + 135.678387, + 139.096765, + 136.989706, + 141.567837, + 142.038357, + 147.931389, + 138.269922, + 143.625775, + 137.421718, + 137.90023, + 147.72044, + 145.364345, + 146.439431, + 165.796746, + 142.751103, + 140.342451, + 138.541059, + 141.873426, + 140.141917, + 138.724298, + 141.260787, + 143.542663, + 141.870562, + 138.515432, + 138.319064, + 156.350063 + ] + }, + "native_lua55": { + "median_ms": 0.617745, + "samples_ms": [ + 0.550307, + 0.612268, + 0.533042, + 0.683572, + 0.524199, + 0.549466, + 0.536046, + 0.796358, + 0.972526, + 0.64826, + 0.997983, + 1.15938, + 0.683592, + 0.745283, + 0.617745, + 0.677844, + 0.537908, + 0.618366, + 0.686938, + 0.685595, + 0.66153, + 0.540603, + 0.554203, + 0.543246, + 0.52547, + 0.566461, + 0.538259, + 0.540824, + 0.533873, + 0.67556, + 0.741497 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "python_lower_bound": { + "median_ms": 0.393977, + "samples_ms": [ + 0.38842, + 0.391694, + 0.402149, + 0.389652, + 0.398354, + 0.386517, + 0.408289, + 0.39585, + 0.388189, + 0.637044, + 0.387067, + 0.400537, + 0.388439, + 0.384784, + 0.394569, + 0.393977, + 0.414177, + 0.447546, + 0.392485, + 0.414719, + 0.384474, + 0.400447, + 0.385045, + 0.383873, + 0.397142, + 0.385305, + 0.413526, + 0.385875, + 0.385064, + 0.397232, + 0.454306 + ] + }, + "luapyre": { + "median_ms": 103.116786, + "samples_ms": [ + 110.06033, + 110.930354, + 104.265028, + 101.483373, + 99.869318, + 103.536051, + 105.913453, + 105.263902, + 103.155934, + 101.328195, + 106.144202, + 101.982895, + 104.149269, + 102.61432, + 101.109905, + 102.704822, + 105.337409, + 105.265555, + 103.059271, + 100.759673, + 105.445778, + 100.8606, + 100.863064, + 103.116786, + 100.657513, + 100.836726, + 107.62413, + 103.945811, + 101.396246, + 101.137486, + 106.903674 + ] + }, + "native_lua55": { + "median_ms": 0.434006, + "samples_ms": [ + 0.43043, + 0.448818, + 0.428698, + 0.445543, + 0.42942, + 0.434006, + 0.443639, + 0.451281, + 0.444721, + 0.433726, + 0.442088, + 0.432694, + 0.442057, + 0.432864, + 0.429199, + 0.446064, + 0.446785, + 0.440716, + 0.431742, + 0.446183, + 0.429399, + 0.44324, + 0.428638, + 0.429519, + 0.440395, + 0.431642, + 0.460314, + 0.43036, + 0.440495, + 0.433295, + 0.427075 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + } + ], + "workloads": { + "n_body": { + "luapyre": { + "median_ms": 762.422076, + "process_medians_ms": [ + 752.16908, + 762.422076, + 787.748142 + ] + }, + "native_lua55": { + "median_ms": 2.091349, + "process_medians_ms": [ + 2.091349, + 2.160572, + 2.006038 + ] + }, + "python_lower_bound": { + "median_ms": 4.881808, + "process_medians_ms": [ + 4.847831, + 5.452781, + 4.881808 + ] + }, + "luapyre_over_native": 364.5599448011785, + "luapyre_over_python_lower_bound": 156.17616997636938, + "native_over_python_lower_bound": 0.42839640559399306 + }, + "mandelbrot": { + "luapyre": { + "median_ms": 436.953648, + "process_medians_ms": [ + 459.138868, + 436.953648, + 427.829748 + ] + }, + "native_lua55": { + "median_ms": 1.221779, + "process_medians_ms": [ + 1.325827, + 1.221779, + 1.219467 + ] + }, + "python_lower_bound": { + "median_ms": 4.26761, + "process_medians_ms": [ + 5.442693, + 4.229524, + 4.26761 + ] + }, + "luapyre_over_native": 357.6372224436662, + "luapyre_over_python_lower_bound": 102.38837382047562, + "native_over_python_lower_bound": 0.2862911559397414 + }, + "fannkuch_redux": { + "luapyre": { + "median_ms": 711.136892, + "process_medians_ms": [ + 689.524841, + 726.213412, + 711.136892 + ] + }, + "native_lua55": { + "median_ms": 3.187255, + "process_medians_ms": [ + 3.187255, + 3.218722, + 3.174861 + ] + }, + "python_lower_bound": { + "median_ms": 6.000702, + "process_medians_ms": [ + 5.981145, + 6.335115, + 6.000702 + ] + }, + "luapyre_over_native": 223.11891957185728, + "luapyre_over_python_lower_bound": 118.50894978620833, + "native_over_python_lower_bound": 0.5311470224650382 + }, + "fasta": { + "luapyre": { + "median_ms": 180.128886, + "process_medians_ms": [ + 178.052044, + 180.492207, + 180.128886 + ] + }, + "native_lua55": { + "median_ms": 0.589253, + "process_medians_ms": [ + 0.585607, + 0.589253, + 0.602973 + ] + }, + "python_lower_bound": { + "median_ms": 1.016398, + "process_medians_ms": [ + 1.016398, + 1.018381, + 1.006144 + ] + }, + "luapyre_over_native": 305.6902315304292, + "luapyre_over_python_lower_bound": 177.2227867429885, + "native_over_python_lower_bound": 0.5797463198471465 + }, + "k_nucleotide": { + "luapyre": { + "median_ms": 138.801097, + "process_medians_ms": [ + 136.570972, + 138.801097, + 140.532259 + ] + }, + "native_lua55": { + "median_ms": 0.559208, + "process_medians_ms": [ + 0.508353, + 0.559208, + 0.617745 + ] + }, + "python_lower_bound": { + "median_ms": 0.939556, + "process_medians_ms": [ + 0.919664, + 0.964722, + 0.939556 + ] + }, + "luapyre_over_native": 248.2101418434643, + "luapyre_over_python_lower_bound": 147.7305205863195, + "native_over_python_lower_bound": 0.5951832567723478 + }, + "reverse_complement": { + "luapyre": { + "median_ms": 103.116786, + "process_medians_ms": [ + 101.272538, + 106.122433, + 103.116786 + ] + }, + "native_lua55": { + "median_ms": 0.441867, + "process_medians_ms": [ + 0.453473, + 0.441867, + 0.434006 + ] + }, + "python_lower_bound": { + "median_ms": 0.393977, + "process_medians_ms": [ + 0.396741, + 0.390782, + 0.393977 + ] + }, + "luapyre_over_native": 233.366116953744, + "luapyre_over_python_lower_bound": 261.73300979498805, + "native_over_python_lower_bound": 1.121555319219142 + } + } +} diff --git a/benchmarks/results/language_benchmarks_040_314.json b/benchmarks/results/language_benchmarks_040_314.json new file mode 100644 index 0000000..62cc00d --- /dev/null +++ b/benchmarks/results/language_benchmarks_040_314.json @@ -0,0 +1,2293 @@ +{ + "schema_version": 1, + "revision": "0.40-language", + "python": "3.14.7", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "n_body": { + "luapyre": { + "median_ms": 694.521277, + "samples_ms": [ + 683.943304, + 669.570238, + 762.7634, + 728.53282, + 773.872172, + 721.016896, + 687.547819, + 687.074825, + 688.972564, + 813.542182, + 716.438853, + 663.234924, + 672.420198, + 677.850304, + 688.982107, + 735.694897, + 753.414368, + 698.405134, + 758.986592, + 683.941062, + 732.616918, + 685.912079, + 724.0655, + 751.459098, + 712.528983, + 656.417246, + 671.315303, + 669.269962, + 694.521277, + 686.903376, + 722.622116 + ] + }, + "native_lua55": { + "median_ms": 2.150234, + "samples_ms": [ + 1.96934, + 1.974266, + 2.335065, + 2.030198, + 1.949941, + 1.966565, + 1.944983, + 1.952924, + 1.947857, + 2.080142, + 2.142023, + 2.130466, + 2.197474, + 2.117857, + 2.244192, + 2.202411, + 3.304998, + 2.032862, + 2.378679, + 2.216923, + 2.059662, + 2.150234, + 2.170104, + 2.228079, + 2.250412, + 2.2222, + 2.225064, + 2.270241, + 2.152547, + 2.204504, + 2.126009 + ] + }, + "python_lower_bound": { + "median_ms": 5.2758, + "samples_ms": [ + 5.218635, + 5.581669, + 5.537003, + 5.530212, + 5.320686, + 5.58256, + 5.317371, + 5.061204, + 4.985584, + 5.150165, + 5.086061, + 5.355587, + 5.189152, + 5.401424, + 5.594247, + 5.535591, + 5.220208, + 5.163885, + 5.530013, + 5.182943, + 5.164236, + 5.173148, + 5.113912, + 5.178928, + 5.817524, + 5.408404, + 5.290502, + 5.10618, + 5.2758, + 5.224174, + 5.30989 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "luapyre": { + "median_ms": 441.690772, + "samples_ms": [ + 477.079877, + 460.356659, + 426.599781, + 419.630555, + 503.396288, + 542.091903, + 428.122265, + 418.143728, + 474.813262, + 541.197671, + 408.766663, + 480.748221, + 403.036685, + 462.821706, + 488.080988, + 451.584251, + 499.115428, + 428.373358, + 468.533692, + 474.459351, + 386.085038, + 401.965982, + 383.874526, + 399.528271, + 481.111454, + 441.690772, + 412.78838, + 418.636293, + 463.036567, + 414.977516, + 427.493619 + ] + }, + "native_lua55": { + "median_ms": 1.249523, + "samples_ms": [ + 1.241672, + 1.22007, + 1.231146, + 1.473061, + 1.24107, + 1.388728, + 1.241191, + 1.290514, + 1.221862, + 1.308209, + 1.408286, + 1.249523, + 1.250995, + 1.261511, + 1.19309, + 1.207832, + 1.204888, + 1.190337, + 1.191919, + 1.624408, + 1.753672, + 7.866271, + 11.747865, + 1.200311, + 1.57444, + 1.289001, + 1.284975, + 1.280188, + 1.203746, + 1.191418, + 1.209274 + ] + }, + "python_lower_bound": { + "median_ms": 4.563175, + "samples_ms": [ + 4.268332, + 4.606909, + 6.05956, + 4.17144, + 4.156057, + 4.247271, + 4.194825, + 4.094927, + 4.697682, + 10.727187, + 4.902002, + 4.401727, + 4.279729, + 4.686786, + 4.699975, + 4.883845, + 4.639827, + 4.563175, + 5.73902, + 4.856425, + 4.185731, + 4.176177, + 4.40982, + 4.733154, + 4.57361, + 4.115488, + 4.889243, + 4.843045, + 4.38284, + 4.481274, + 4.329051 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "luapyre": { + "median_ms": 641.897331, + "samples_ms": [ + 623.153888, + 665.414869, + 614.446404, + 656.911483, + 605.80066, + 650.715469, + 603.050285, + 700.508454, + 627.046775, + 656.483219, + 627.75227, + 683.795131, + 616.278398, + 634.249642, + 654.960791, + 630.947507, + 602.187737, + 604.564885, + 607.828342, + 604.911572, + 620.970459, + 641.897331, + 667.274774, + 750.928867, + 659.212735, + 670.821562, + 671.217704, + 748.144372, + 641.072225, + 656.401491, + 728.846158 + ] + }, + "native_lua55": { + "median_ms": 3.340462, + "samples_ms": [ + 3.347821, + 3.350165, + 3.606882, + 3.185463, + 3.54402, + 3.294654, + 3.197141, + 3.500205, + 3.242447, + 3.147788, + 3.281624, + 3.265321, + 3.540143, + 3.876307, + 3.266582, + 3.164393, + 3.654221, + 3.377104, + 3.147107, + 3.228587, + 3.265481, + 3.226464, + 3.251881, + 3.519073, + 3.449881, + 3.340462, + 3.699157, + 3.429151, + 3.392207, + 3.297488, + 3.640141 + ] + }, + "python_lower_bound": { + "median_ms": 6.08042, + "samples_ms": [ + 6.016638, + 5.940205, + 5.981616, + 6.053451, + 5.917372, + 6.060391, + 6.435782, + 6.215649, + 5.97735, + 6.023347, + 5.861089, + 6.08042, + 6.276729, + 6.219906, + 5.785108, + 5.866627, + 6.998207, + 5.965492, + 6.439408, + 6.137384, + 5.965292, + 5.851125, + 5.784988, + 11.895442, + 10.377145, + 10.540074, + 10.415942, + 10.348323, + 10.292852, + 10.362013, + 10.29901 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "luapyre": { + "median_ms": 159.855695, + "samples_ms": [ + 168.408711, + 217.864301, + 154.727215, + 157.191101, + 156.75454, + 156.896118, + 157.847413, + 193.137738, + 156.291852, + 155.989477, + 161.066327, + 159.208601, + 160.529028, + 187.215408, + 192.364324, + 154.215443, + 158.436879, + 161.755137, + 177.266919, + 157.7293, + 158.662662, + 158.238291, + 161.335656, + 160.828612, + 174.699843, + 157.71929, + 156.796837, + 159.855695, + 165.406838, + 164.026013, + 160.412231 + ] + }, + "native_lua55": { + "median_ms": 0.59335, + "samples_ms": [ + 0.594632, + 0.592859, + 0.590687, + 0.602503, + 0.617256, + 0.581182, + 0.604837, + 0.581934, + 0.600692, + 0.582594, + 0.600361, + 0.607652, + 0.587311, + 0.589875, + 0.581793, + 0.596445, + 0.59335, + 0.576406, + 0.720838, + 0.608182, + 0.596094, + 0.604477, + 0.578729, + 0.589245, + 0.685155, + 0.60013, + 0.600822, + 0.59244, + 0.589124, + 0.59311, + 0.578149 + ] + }, + "python_lower_bound": { + "median_ms": 0.921081, + "samples_ms": [ + 0.9404, + 0.940871, + 1.091592, + 0.983442, + 0.922704, + 0.940801, + 0.921081, + 0.918247, + 0.908914, + 0.932749, + 0.924186, + 0.916504, + 0.919579, + 0.916425, + 0.936494, + 0.92011, + 0.917596, + 0.920881, + 0.934401, + 0.915623, + 0.899249, + 0.9203, + 0.934861, + 0.917396, + 0.918297, + 0.920801, + 0.917076, + 0.942954, + 1.120945, + 0.989191, + 0.922994 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "luapyre": { + "median_ms": 130.02993, + "samples_ms": [ + 128.194646, + 129.599628, + 132.574383, + 133.917022, + 179.493601, + 178.381655, + 143.209027, + 140.024134, + 143.927321, + 134.846055, + 149.276249, + 157.045538, + 131.848158, + 126.349319, + 130.022569, + 125.903565, + 126.675588, + 129.137871, + 130.02993, + 127.390546, + 130.343359, + 126.122336, + 127.393802, + 126.657301, + 127.72017, + 128.644056, + 128.46969, + 128.106918, + 133.755696, + 135.527756, + 135.309566 + ] + }, + "native_lua55": { + "median_ms": 0.550848, + "samples_ms": [ + 0.6268, + 0.552741, + 0.537829, + 0.535276, + 0.525561, + 0.512812, + 0.526783, + 0.588854, + 0.561394, + 0.649723, + 0.581483, + 0.530178, + 0.609224, + 0.54554, + 0.578999, + 0.550848, + 0.520153, + 0.578599, + 0.517199, + 0.571809, + 0.516298, + 0.593751, + 0.541935, + 0.523017, + 0.547673, + 0.51806, + 0.608192, + 0.52483, + 0.556236, + 0.588073, + 0.622914 + ] + }, + "python_lower_bound": { + "median_ms": 0.97455, + "samples_ms": [ + 1.132853, + 0.936925, + 1.002831, + 0.932228, + 1.02226, + 1.057902, + 1.078743, + 0.920781, + 0.984345, + 0.930815, + 0.916145, + 0.917166, + 0.91326, + 0.909965, + 0.910356, + 0.963483, + 0.919219, + 2.066472, + 1.26789, + 1.217596, + 0.97455, + 1.04209, + 1.116057, + 2.907466, + 1.012516, + 0.979847, + 0.950234, + 0.916234, + 0.909885, + 0.909094, + 0.996742 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "luapyre": { + "median_ms": 95.934091, + "samples_ms": [ + 94.931551, + 116.087186, + 94.218465, + 93.678242, + 94.169893, + 94.233806, + 93.505658, + 94.051549, + 94.898302, + 94.344168, + 132.990018, + 124.3566, + 115.383604, + 98.467679, + 99.496859, + 99.247423, + 95.811442, + 97.398911, + 99.521886, + 94.387312, + 101.917866, + 116.370798, + 95.934091, + 93.006332, + 92.944271, + 95.690852, + 96.693714, + 92.68988, + 96.120663, + 96.181187, + 97.621415 + ] + }, + "native_lua55": { + "median_ms": 0.445724, + "samples_ms": [ + 0.446515, + 0.437562, + 0.43574, + 0.451042, + 0.450851, + 0.455468, + 0.435609, + 0.454557, + 0.434007, + 0.460175, + 0.434588, + 0.434117, + 0.446895, + 0.451382, + 0.449619, + 0.439004, + 0.445634, + 0.437432, + 0.445724, + 0.440687, + 0.433016, + 0.451623, + 0.44972, + 0.4502, + 0.435359, + 0.449129, + 0.438233, + 0.44974, + 0.435559, + 0.439355, + 0.446886 + ] + }, + "python_lower_bound": { + "median_ms": 0.356763, + "samples_ms": [ + 0.365346, + 0.355852, + 0.367929, + 0.356763, + 0.353588, + 0.385996, + 0.555065, + 0.380438, + 0.353829, + 0.367579, + 0.366478, + 0.353539, + 0.352878, + 0.369602, + 0.417082, + 0.40185, + 0.354781, + 0.353008, + 0.36808, + 0.353158, + 0.352097, + 0.367339, + 0.354309, + 0.352477, + 0.368691, + 0.35438, + 0.38175, + 0.35426, + 0.352387, + 0.366298, + 0.352467 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "n_body": { + "native_lua55": { + "median_ms": 1.994557, + "samples_ms": [ + 2.004361, + 2.172557, + 1.982639, + 1.985624, + 2.032101, + 1.988818, + 1.99, + 2.04488, + 1.989789, + 2.271873, + 2.105529, + 2.061103, + 2.082615, + 2.101333, + 1.933337, + 1.910934, + 1.923402, + 1.978102, + 1.928179, + 2.060623, + 2.089115, + 2.0652, + 1.965384, + 1.994557, + 2.042807, + 2.323399, + 2.012433, + 1.906927, + 1.903653, + 1.926276, + 1.904013 + ] + }, + "python_lower_bound": { + "median_ms": 4.995189, + "samples_ms": [ + 4.848153, + 4.980017, + 5.048076, + 5.055067, + 5.138018, + 4.854072, + 7.319719, + 4.902833, + 4.857016, + 4.902703, + 5.151568, + 4.932197, + 4.909823, + 5.110357, + 5.315008, + 4.88657, + 5.331482, + 6.124906, + 5.315989, + 5.056088, + 5.05675, + 5.010611, + 4.952957, + 5.298523, + 4.916734, + 4.88699, + 4.901731, + 7.637375, + 4.977983, + 4.871268, + 4.995189 + ] + }, + "luapyre": { + "median_ms": 767.52489, + "samples_ms": [ + 767.52489, + 812.152796, + 691.226492, + 794.630495, + 744.538115, + 752.361653, + 753.641561, + 740.880168, + 725.827854, + 685.816183, + 674.368088, + 696.053462, + 678.155925, + 774.353497, + 792.191173, + 685.943239, + 680.63786, + 759.876466, + 780.216984, + 866.49251, + 795.680853, + 836.899027, + 866.533244, + 835.538243, + 781.072246, + 808.074331, + 809.053463, + 746.689471, + 774.360182, + 843.768891, + 735.847068 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "native_lua55": { + "median_ms": 1.19224, + "samples_ms": [ + 1.266058, + 1.223665, + 1.220501, + 1.234822, + 1.240871, + 1.233119, + 1.224276, + 1.239048, + 1.378442, + 1.19224, + 1.267069, + 1.2059, + 1.19309, + 1.218488, + 1.379574, + 1.191568, + 1.245537, + 1.163568, + 1.129026, + 1.127064, + 1.143698, + 1.128476, + 1.134134, + 1.146562, + 1.141946, + 1.127344, + 1.127715, + 1.143157, + 1.127825, + 1.127975, + 1.141204 + ] + }, + "python_lower_bound": { + "median_ms": 4.118443, + "samples_ms": [ + 4.207374, + 4.065745, + 4.0685, + 4.06143, + 4.064604, + 4.693947, + 4.142459, + 4.051985, + 4.06203, + 4.184509, + 4.183138, + 4.376121, + 4.195375, + 4.10273, + 4.228805, + 4.118443, + 4.048891, + 4.080537, + 4.049212, + 4.050183, + 4.047018, + 4.072605, + 4.051174, + 4.19138, + 4.234553, + 4.16394, + 4.118623, + 4.050403, + 4.321181, + 4.181846, + 4.344144 + ] + }, + "luapyre": { + "median_ms": 382.915302, + "samples_ms": [ + 385.425593, + 384.74237, + 392.562565, + 379.872505, + 373.74844, + 374.336563, + 377.598529, + 378.551978, + 378.729879, + 376.558894, + 381.407207, + 403.067933, + 373.376189, + 440.92356, + 429.306314, + 405.104305, + 458.118291, + 385.089662, + 389.007199, + 378.925488, + 408.309445, + 378.038067, + 388.174758, + 380.615888, + 376.087334, + 383.075056, + 377.605131, + 381.798052, + 382.915302, + 385.454187, + 384.706681 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "native_lua55": { + "median_ms": 3.227054, + "samples_ms": [ + 3.351447, + 3.227054, + 3.233334, + 3.223179, + 3.326361, + 3.301703, + 3.198643, + 3.191673, + 3.498803, + 3.301554, + 3.185914, + 3.222498, + 3.217761, + 3.354952, + 3.224241, + 3.214827, + 3.213505, + 3.228777, + 3.221978, + 3.225984, + 3.204281, + 3.215908, + 3.230379, + 3.219614, + 3.232122, + 3.239403, + 3.315024, + 3.227706, + 3.304258, + 3.206655, + 3.607332 + ] + }, + "python_lower_bound": { + "median_ms": 5.757317, + "samples_ms": [ + 5.753512, + 5.983619, + 6.018861, + 5.8617, + 5.979213, + 6.095523, + 5.847459, + 5.706332, + 5.573116, + 5.669318, + 6.179457, + 5.700223, + 5.559717, + 5.995788, + 6.075774, + 6.022526, + 5.803916, + 5.572266, + 5.57494, + 5.579115, + 5.574849, + 5.65023, + 5.769906, + 5.589861, + 5.869282, + 5.599195, + 5.558596, + 5.683599, + 6.307855, + 5.757317, + 5.796494 + ] + }, + "luapyre": { + "median_ms": 632.909543, + "samples_ms": [ + 634.326502, + 603.474148, + 614.718344, + 663.417303, + 637.916279, + 632.909543, + 623.890377, + 624.202373, + 622.655118, + 625.942443, + 636.513969, + 637.192284, + 694.622498, + 608.443659, + 604.736771, + 614.299791, + 615.384113, + 643.820143, + 617.708087, + 622.52583, + 651.28492, + 738.162223, + 658.904824, + 625.911144, + 661.233401, + 624.036333, + 663.516862, + 686.010752, + 684.276828, + 638.308929, + 602.234959 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "native_lua55": { + "median_ms": 0.596265, + "samples_ms": [ + 0.594903, + 0.57884, + 0.617676, + 0.680128, + 0.622623, + 0.596265, + 0.578158, + 0.591188, + 0.596374, + 0.642473, + 0.694268, + 0.575885, + 0.638657, + 0.611226, + 0.576846, + 0.611477, + 0.573391, + 0.623965, + 0.589245, + 0.574213, + 0.590917, + 0.59311, + 0.72943, + 0.614672, + 0.636143, + 0.604006, + 0.588192, + 0.580772, + 0.609854, + 0.567092, + 0.593651 + ] + }, + "python_lower_bound": { + "median_ms": 0.894322, + "samples_ms": [ + 0.895193, + 0.894131, + 0.896625, + 0.898839, + 0.909044, + 0.895594, + 0.981921, + 0.893701, + 0.974119, + 0.895704, + 0.964926, + 0.880341, + 0.91336, + 0.897817, + 0.889465, + 0.886621, + 0.889495, + 0.903085, + 0.888794, + 0.893822, + 0.877007, + 0.918608, + 0.891237, + 0.896195, + 0.894322, + 0.890407, + 0.913801, + 0.889745, + 0.892329, + 0.892569, + 0.893621 + ] + }, + "luapyre": { + "median_ms": 159.928711, + "samples_ms": [ + 154.237916, + 155.348196, + 157.032417, + 157.460497, + 157.727488, + 161.284921, + 159.250946, + 157.419727, + 156.375244, + 162.276597, + 162.800776, + 161.840952, + 157.738806, + 158.765233, + 159.294008, + 183.78638, + 163.37957, + 159.928711, + 159.048018, + 158.666406, + 182.83725, + 161.015426, + 202.505785, + 162.274388, + 163.002721, + 159.848031, + 165.484281, + 162.969863, + 161.368332, + 159.023181, + 168.00536 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "native_lua55": { + "median_ms": 0.529657, + "samples_ms": [ + 0.531911, + 0.615443, + 0.618678, + 0.603596, + 0.571478, + 0.514555, + 0.528355, + 0.519291, + 0.59973, + 0.527964, + 0.529657, + 0.537779, + 0.529397, + 0.514595, + 0.526392, + 0.527574, + 0.515216, + 0.525421, + 0.512401, + 0.542366, + 0.513043, + 0.525451, + 0.762229, + 0.619959, + 0.546922, + 0.517159, + 0.556297, + 0.513524, + 0.793034, + 0.529927, + 0.542025 + ] + }, + "python_lower_bound": { + "median_ms": 0.925938, + "samples_ms": [ + 0.96796, + 0.990744, + 0.961581, + 0.910666, + 0.920511, + 0.91394, + 0.939999, + 0.896365, + 0.977885, + 0.925938, + 0.928152, + 0.920831, + 0.907161, + 0.989341, + 0.907231, + 0.924817, + 0.905559, + 1.016561, + 0.959678, + 0.94049, + 0.90677, + 1.039906, + 0.999867, + 0.92673, + 0.909455, + 0.893751, + 0.921872, + 0.980819, + 0.926239, + 0.911137, + 0.909174 + ] + }, + "luapyre": { + "median_ms": 130.28077, + "samples_ms": [ + 130.28077, + 166.598345, + 129.512473, + 131.848571, + 130.912026, + 131.809133, + 127.66312, + 127.123889, + 132.624865, + 164.939813, + 139.764593, + 141.297354, + 136.243723, + 173.217406, + 165.771072, + 160.457803, + 127.658835, + 129.015969, + 132.760328, + 131.642887, + 128.449457, + 130.108443, + 125.220214, + 129.276877, + 124.934765, + 130.855391, + 127.380927, + 127.038725, + 125.055443, + 128.361698, + 128.589402 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "native_lua55": { + "median_ms": 0.453175, + "samples_ms": [ + 0.445193, + 0.459243, + 0.444953, + 0.462138, + 0.442129, + 0.461297, + 0.446545, + 0.477801, + 0.445023, + 0.459624, + 0.444252, + 0.446586, + 0.45619, + 0.447337, + 0.453175, + 0.459945, + 0.457762, + 0.444893, + 0.454968, + 0.446826, + 0.455939, + 0.445484, + 0.44284, + 0.456811, + 0.457651, + 0.460115, + 0.44287, + 0.458583, + 0.447337, + 0.459044, + 0.443601 + ] + }, + "python_lower_bound": { + "median_ms": 0.355651, + "samples_ms": [ + 0.35442, + 0.353639, + 0.377854, + 0.35407, + 0.367429, + 0.353879, + 0.352727, + 0.383262, + 0.3546, + 0.353409, + 0.365326, + 0.353288, + 0.352717, + 0.365846, + 0.397583, + 0.377424, + 0.355651, + 0.370033, + 0.366358, + 0.474396, + 0.397023, + 0.35432, + 0.352377, + 0.365066, + 0.353158, + 0.400278, + 0.438384, + 0.354771, + 0.397914, + 0.35448, + 0.352858 + ] + }, + "luapyre": { + "median_ms": 113.029968, + "samples_ms": [ + 99.321196, + 122.11914, + 135.563823, + 135.075611, + 105.16717, + 116.762629, + 113.029968, + 94.225424, + 93.143777, + 117.824252, + 106.392327, + 95.306791, + 94.244426, + 96.974528, + 95.775428, + 100.532158, + 144.65918, + 144.330381, + 139.761858, + 116.761042, + 108.093995, + 109.328365, + 116.481843, + 142.562392, + 149.860732, + 126.351961, + 111.352761, + 148.846824, + 107.261222, + 139.039613, + 91.993989 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "n_body": { + "python_lower_bound": { + "median_ms": 5.047737, + "samples_ms": [ + 5.196956, + 5.047737, + 5.137398, + 5.183977, + 5.078913, + 5.896943, + 4.885029, + 5.073876, + 4.898848, + 4.965906, + 4.91419, + 5.048648, + 4.913029, + 4.986166, + 5.191118, + 4.970784, + 4.912659, + 5.125251, + 5.226869, + 4.957564, + 4.859651, + 5.066184, + 4.907231, + 5.081526, + 5.173491, + 4.964515, + 4.950845, + 4.905808, + 5.192329, + 4.917636, + 5.202684 + ] + }, + "luapyre": { + "median_ms": 659.523094, + "samples_ms": [ + 733.616127, + 801.90564, + 862.291473, + 689.390289, + 722.651791, + 718.42638, + 681.563312, + 787.453046, + 745.125265, + 733.283654, + 775.999314, + 683.160435, + 644.227881, + 648.0094, + 641.171867, + 656.329367, + 638.50891, + 643.488116, + 654.991956, + 650.829749, + 652.065021, + 659.523094, + 656.489954, + 654.845651, + 653.748101, + 654.080659, + 654.273482, + 659.815929, + 653.748884, + 773.459797, + 702.056496 + ] + }, + "native_lua55": { + "median_ms": 2.149862, + "samples_ms": [ + 2.008526, + 1.979884, + 2.210592, + 2.145897, + 1.982937, + 2.314494, + 2.138416, + 2.183312, + 2.111646, + 2.130925, + 2.125867, + 2.303038, + 2.111706, + 2.130715, + 2.125968, + 2.528318, + 2.1343, + 2.247636, + 2.165626, + 2.128621, + 2.149862, + 2.274315, + 2.334604, + 2.273374, + 2.102513, + 2.436814, + 2.206936, + 2.25722, + 2.121661, + 2.477803, + 2.869959 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1001, + "loop_entry_executions": 1001, + "loop_iterations": 5005, + "call_ic_hits": 10010, + "table_ic_hits": 20020 + } + }, + "mandelbrot": { + "python_lower_bound": { + "median_ms": 4.152132, + "samples_ms": [ + 4.305828, + 4.472442, + 4.179973, + 4.182678, + 4.13023, + 4.173984, + 4.144381, + 4.194294, + 4.18425, + 4.169688, + 4.326649, + 4.216147, + 4.152132, + 4.174315, + 4.227624, + 4.430731, + 4.126926, + 4.168257, + 4.105063, + 4.058866, + 4.067559, + 4.129419, + 4.055671, + 4.060097, + 4.194024, + 4.147857, + 4.076811, + 4.091073, + 4.138383, + 4.091173, + 4.075039 + ] + }, + "luapyre": { + "median_ms": 401.66209, + "samples_ms": [ + 382.910675, + 401.66209, + 389.312108, + 394.313242, + 383.467067, + 392.696559, + 459.179664, + 405.056329, + 395.553955, + 391.046713, + 393.627398, + 413.710673, + 402.353657, + 399.425626, + 392.671226, + 392.222678, + 411.1427, + 418.179164, + 385.566872, + 386.152661, + 405.104699, + 414.486656, + 383.654746, + 392.158343, + 410.303727, + 404.432017, + 415.763941, + 506.077665, + 406.45396, + 404.995765, + 506.622545 + ] + }, + "native_lua55": { + "median_ms": 1.233601, + "samples_ms": [ + 1.30849, + 1.244797, + 1.337442, + 1.564406, + 1.295281, + 1.292587, + 1.221202, + 1.220891, + 1.319797, + 1.249394, + 1.225339, + 1.375799, + 1.233601, + 1.232098, + 1.608441, + 1.41725, + 1.255102, + 1.722688, + 1.126573, + 1.123929, + 1.21959, + 1.181845, + 1.142798, + 1.19255, + 1.194053, + 1.125232, + 1.126343, + 1.283213, + 1.202335, + 1.33539, + 1.205168 + ] + }, + "luapyre_warm_counters": {} + }, + "fannkuch_redux": { + "python_lower_bound": { + "median_ms": 6.471788, + "samples_ms": [ + 6.029388, + 5.670702, + 5.735968, + 5.950142, + 5.681768, + 5.864015, + 6.011592, + 6.096477, + 5.780734, + 6.036769, + 6.51437, + 6.530614, + 7.293173, + 6.389317, + 7.23608, + 7.149833, + 6.65039, + 7.742412, + 5.772921, + 7.206346, + 6.740081, + 6.51435, + 7.081753, + 6.232807, + 6.471788, + 7.110966, + 6.181321, + 5.973216, + 8.807556, + 10.299865, + 10.295449 + ] + }, + "luapyre": { + "median_ms": 636.539142, + "samples_ms": [ + 648.47795, + 655.052917, + 623.820714, + 629.672086, + 633.648164, + 673.416518, + 804.232463, + 648.408423, + 654.299111, + 625.079106, + 629.519687, + 646.232424, + 631.98441, + 626.930937, + 627.309363, + 696.597095, + 643.486208, + 663.357448, + 669.801832, + 636.539142, + 656.768267, + 698.374196, + 692.975795, + 614.660504, + 619.056909, + 625.442819, + 619.350988, + 638.421678, + 613.924633, + 616.355587, + 631.579776 + ] + }, + "native_lua55": { + "median_ms": 3.187914, + "samples_ms": [ + 3.343181, + 3.199752, + 3.16481, + 3.188584, + 3.251016, + 3.25984, + 3.191999, + 3.169366, + 3.178671, + 3.185972, + 3.19909, + 3.179461, + 3.17164, + 3.187123, + 3.174073, + 3.191809, + 3.172211, + 3.183597, + 3.189256, + 3.177438, + 3.177568, + 3.205379, + 3.16491, + 3.197999, + 3.193202, + 3.17855, + 3.207953, + 3.187364, + 3.200893, + 3.189916, + 3.187914 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 13700, + "loop_entry_executions": 13700, + "loop_iterations": 48979 + } + }, + "fasta": { + "python_lower_bound": { + "median_ms": 0.903224, + "samples_ms": [ + 0.903886, + 0.910084, + 0.919518, + 0.901652, + 0.886409, + 0.9071, + 0.917354, + 0.902664, + 0.899969, + 0.900921, + 0.901121, + 0.919338, + 0.905547, + 0.895844, + 0.905067, + 0.904335, + 0.908311, + 0.897075, + 0.907931, + 0.901502, + 0.917595, + 0.907871, + 0.902073, + 0.899138, + 0.917045, + 0.90073, + 0.896714, + 0.902984, + 0.918316, + 0.903224, + 0.898007 + ] + }, + "luapyre": { + "median_ms": 163.59445, + "samples_ms": [ + 153.028936, + 161.135912, + 158.193805, + 161.821488, + 165.971697, + 167.314838, + 161.105988, + 200.718656, + 158.018628, + 166.179482, + 164.320625, + 160.988407, + 163.59445, + 165.422541, + 172.118807, + 204.132987, + 216.780311, + 165.463862, + 163.182946, + 158.508871, + 202.489416, + 180.115846, + 184.676541, + 169.790607, + 160.919847, + 160.489125, + 156.676523, + 154.982789, + 155.607706, + 164.669036, + 154.40418 + ] + }, + "native_lua55": { + "median_ms": 0.596325, + "samples_ms": [ + 0.589755, + 0.592489, + 0.596325, + 0.659817, + 0.62028, + 0.611447, + 0.588182, + 0.685145, + 0.595533, + 0.617485, + 0.595994, + 0.602273, + 0.602453, + 0.583466, + 0.652046, + 0.625387, + 0.581694, + 0.602213, + 0.599189, + 0.597437, + 0.59338, + 0.581923, + 0.59365, + 0.578599, + 0.594622, + 0.610536, + 0.57955, + 0.59319, + 0.58638, + 0.603305, + 0.598828 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 2, + "loop_entry_executions": 2, + "loop_iterations": 19, + "call_ic_hits": 6003, + "table_ic_hits": 4002 + } + }, + "k_nucleotide": { + "python_lower_bound": { + "median_ms": 0.911637, + "samples_ms": [ + 0.904126, + 0.89981, + 0.973868, + 0.910986, + 0.911637, + 0.918357, + 0.965345, + 0.951576, + 0.922152, + 0.904787, + 0.909574, + 0.971645, + 0.922723, + 0.903886, + 0.901993, + 0.987018, + 0.925918, + 0.892208, + 0.910646, + 0.954961, + 0.936593, + 0.90683, + 0.902975, + 0.907191, + 1.250704, + 0.987308, + 0.904056, + 0.96802, + 0.966016, + 0.906189, + 0.902724 + ] + }, + "luapyre": { + "median_ms": 129.726757, + "samples_ms": [ + 128.051721, + 129.4423, + 129.919439, + 130.414505, + 133.798469, + 149.211615, + 129.370165, + 131.198576, + 130.120154, + 166.803196, + 133.271516, + 129.707479, + 142.036897, + 130.568201, + 129.798131, + 129.726757, + 131.213067, + 134.278493, + 127.146713, + 128.176003, + 125.659322, + 128.871172, + 129.354851, + 142.790227, + 126.101961, + 126.485662, + 125.951508, + 128.369041, + 126.863234, + 126.964052, + 133.863775 + ] + }, + "native_lua55": { + "median_ms": 0.534945, + "samples_ms": [ + 0.617995, + 0.627079, + 0.53196, + 0.534945, + 0.494766, + 0.535095, + 0.534954, + 0.536646, + 0.502156, + 0.526171, + 0.536306, + 0.544348, + 0.499833, + 0.541002, + 0.54549, + 0.530848, + 0.502516, + 0.587861, + 0.537859, + 0.49802, + 0.538439, + 0.545139, + 0.534393, + 0.499382, + 0.538559, + 0.495807, + 0.554813, + 0.494545, + 0.531128, + 0.518159, + 0.506722 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8656, + "table_ic_hits": 17312 + } + }, + "reverse_complement": { + "python_lower_bound": { + "median_ms": 0.625918, + "samples_ms": [ + 0.611817, + 0.625887, + 0.626258, + 0.616153, + 0.628942, + 0.633068, + 0.629933, + 0.623824, + 0.610114, + 0.6276, + 0.550917, + 0.651295, + 0.62732, + 0.612978, + 0.625256, + 0.625918, + 0.609504, + 0.626709, + 0.651555, + 0.630584, + 0.625136, + 0.610655, + 0.623444, + 0.631496, + 0.642823, + 0.626838, + 0.610154, + 0.627319, + 0.623464, + 0.610504, + 0.629533 + ] + }, + "luapyre": { + "median_ms": 96.918672, + "samples_ms": [ + 103.114123, + 120.591662, + 96.0577, + 99.47309, + 149.655872, + 96.677038, + 93.666504, + 97.729029, + 104.092757, + 97.319459, + 103.90716, + 111.002355, + 116.490594, + 92.613611, + 94.522029, + 94.12066, + 93.627317, + 96.264914, + 99.38371, + 102.785291, + 105.635889, + 108.914955, + 100.540595, + 92.522558, + 93.624423, + 95.66867, + 93.333957, + 93.109277, + 96.918672, + 92.861875, + 93.753141 + ] + }, + "native_lua55": { + "median_ms": 0.452624, + "samples_ms": [ + 0.424842, + 0.440165, + 0.425344, + 0.43003, + 0.437831, + 0.51875, + 0.459213, + 0.443731, + 0.438022, + 0.440326, + 0.439244, + 0.428989, + 0.425313, + 0.452624, + 0.738037, + 0.432714, + 0.439404, + 0.47089, + 0.435538, + 0.439074, + 0.472092, + 0.454076, + 0.480144, + 0.465412, + 0.609864, + 0.458112, + 0.659327, + 0.460525, + 0.523257, + 0.456258, + 0.470219 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 8002, + "table_ic_hits": 16004 + } + } + } + } + ], + "workloads": { + "n_body": { + "luapyre": { + "median_ms": 694.521277, + "process_medians_ms": [ + 694.521277, + 767.52489, + 659.523094 + ] + }, + "native_lua55": { + "median_ms": 2.149862, + "process_medians_ms": [ + 2.150234, + 1.994557, + 2.149862 + ] + }, + "python_lower_bound": { + "median_ms": 5.047737, + "process_medians_ms": [ + 5.2758, + 4.995189, + 5.047737 + ] + }, + "luapyre_over_native": 323.0538876448814, + "luapyre_over_python_lower_bound": 137.59062268893965, + "native_over_python_lower_bound": 0.4259061040620778 + }, + "mandelbrot": { + "luapyre": { + "median_ms": 401.66209, + "process_medians_ms": [ + 441.690772, + 382.915302, + 401.66209 + ] + }, + "native_lua55": { + "median_ms": 1.233601, + "process_medians_ms": [ + 1.249523, + 1.19224, + 1.233601 + ] + }, + "python_lower_bound": { + "median_ms": 4.152132, + "process_medians_ms": [ + 4.563175, + 4.118443, + 4.152132 + ] + }, + "luapyre_over_native": 325.6013005826033, + "luapyre_over_python_lower_bound": 96.73634894073695, + "native_over_python_lower_bound": 0.29710062204188115 + }, + "fannkuch_redux": { + "luapyre": { + "median_ms": 636.539142, + "process_medians_ms": [ + 641.897331, + 632.909543, + 636.539142 + ] + }, + "native_lua55": { + "median_ms": 3.227054, + "process_medians_ms": [ + 3.340462, + 3.227054, + 3.187914 + ] + }, + "python_lower_bound": { + "median_ms": 6.08042, + "process_medians_ms": [ + 6.08042, + 5.757317, + 6.471788 + ] + }, + "luapyre_over_native": 197.25084922657012, + "luapyre_over_python_lower_bound": 104.68670618148087, + "native_over_python_lower_bound": 0.5307287983395883 + }, + "fasta": { + "luapyre": { + "median_ms": 159.928711, + "process_medians_ms": [ + 159.855695, + 159.928711, + 163.59445 + ] + }, + "native_lua55": { + "median_ms": 0.596265, + "process_medians_ms": [ + 0.59335, + 0.596265, + 0.596325 + ] + }, + "python_lower_bound": { + "median_ms": 0.903224, + "process_medians_ms": [ + 0.921081, + 0.894322, + 0.903224 + ] + }, + "luapyre_over_native": 268.2175056392711, + "luapyre_over_python_lower_bound": 177.0642841642826, + "native_over_python_lower_bound": 0.6601518560179978 + }, + "k_nucleotide": { + "luapyre": { + "median_ms": 130.02993, + "process_medians_ms": [ + 130.02993, + 130.28077, + 129.726757 + ] + }, + "native_lua55": { + "median_ms": 0.534945, + "process_medians_ms": [ + 0.550848, + 0.529657, + 0.534945 + ] + }, + "python_lower_bound": { + "median_ms": 0.925938, + "process_medians_ms": [ + 0.97455, + 0.925938, + 0.911637 + ] + }, + "luapyre_over_native": 243.07158679864287, + "luapyre_over_python_lower_bound": 140.43049318636886, + "native_over_python_lower_bound": 0.5777330663608147 + }, + "reverse_complement": { + "luapyre": { + "median_ms": 96.918672, + "process_medians_ms": [ + 95.934091, + 113.029968, + 96.918672 + ] + }, + "native_lua55": { + "median_ms": 0.452624, + "process_medians_ms": [ + 0.445724, + 0.453175, + 0.452624 + ] + }, + "python_lower_bound": { + "median_ms": 0.356763, + "process_medians_ms": [ + 0.356763, + 0.355651, + 0.625918 + ] + }, + "luapyre_over_native": 214.1262328113401, + "luapyre_over_python_lower_bound": 271.66122047409624, + "native_over_python_lower_bound": 1.2686965856885384 + } + } +} diff --git a/benchmarks/results/native_headroom_040_313.json b/benchmarks/results/native_headroom_040_313.json new file mode 100644 index 0000000..642bd00 --- /dev/null +++ b/benchmarks/results/native_headroom_040_313.json @@ -0,0 +1,803 @@ +{ + "schema_version": 1, + "revision": "0.40.0a1", + "python": "3.13.15", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 23.057312, + "samples_ms": [ + 24.20484, + 23.371721, + 22.804353, + 23.943237, + 22.384509, + 23.057312, + 23.511816, + 23.35707, + 22.924607, + 22.608446, + 22.205317, + 22.861616, + 23.907144, + 22.908485, + 22.526736, + 22.401714, + 22.818032, + 26.397364, + 22.410837, + 22.659741, + 22.6522, + 23.195423, + 25.04618, + 23.218288, + 26.217221, + 24.644372, + 22.762611, + 23.463916, + 24.527, + 22.509572, + 24.456007 + ] + }, + "native_lua55": { + "median_ms": 1.829805, + "samples_ms": [ + 1.73094, + 1.829805, + 1.778179, + 2.03223, + 2.098668, + 2.077397, + 1.790157, + 2.039721, + 1.993404, + 2.053101, + 1.750228, + 1.859578, + 2.063776, + 1.891786, + 1.902841, + 1.843836, + 1.781124, + 1.781173, + 1.795565, + 1.765541, + 1.806891, + 1.763478, + 1.77212, + 1.812369, + 1.778409, + 2.050367, + 2.101882, + 1.912626, + 1.742257, + 1.761655, + 1.862463 + ] + }, + "python_lower_bound": { + "median_ms": 2.073441, + "samples_ms": [ + 2.034964, + 2.274344, + 2.375543, + 2.147439, + 2.076966, + 2.357215, + 2.111807, + 2.061844, + 2.058449, + 2.033553, + 2.073441, + 2.042145, + 2.049485, + 2.306502, + 2.056345, + 2.039581, + 2.056005, + 2.030228, + 2.463732, + 2.43588, + 2.614231, + 2.068564, + 2.104025, + 2.053952, + 2.328854, + 2.130183, + 2.076576, + 2.043798, + 2.110655, + 2.039892, + 2.055504 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.776719, + "samples_ms": [ + 3.840503, + 2.720767, + 2.776719, + 3.035367, + 3.305952, + 2.689972, + 2.752804, + 3.028337, + 2.822155, + 2.696662, + 2.933518, + 2.651005, + 2.666969, + 2.664916, + 2.822696, + 3.438926, + 3.633692, + 3.534957, + 3.087433, + 2.780755, + 2.622774, + 2.775577, + 2.662061, + 2.847491, + 2.712135, + 2.670023, + 3.185636, + 2.669242, + 3.02423, + 2.672636, + 2.626058 + ] + }, + "native_lua55": { + "median_ms": 0.833298, + "samples_ms": [ + 0.833298, + 0.823864, + 0.830203, + 0.840028, + 0.830884, + 0.813459, + 0.830444, + 0.830604, + 0.837023, + 0.827299, + 0.817625, + 0.825016, + 0.829082, + 0.844384, + 0.826608, + 0.830844, + 0.811216, + 0.9233, + 0.842771, + 0.943078, + 0.843052, + 0.817735, + 0.838576, + 0.915298, + 0.800129, + 0.879806, + 0.842091, + 1.297616, + 0.922809, + 1.632707, + 0.848279 + ] + }, + "python_lower_bound": { + "median_ms": 2.007985, + "samples_ms": [ + 2.20972, + 2.043798, + 1.972723, + 2.000324, + 1.97032, + 2.0113, + 2.350435, + 1.940416, + 2.007985, + 2.007424, + 2.147819, + 2.246654, + 2.340682, + 2.415981, + 1.995837, + 2.07984, + 2.017028, + 1.965663, + 2.086149, + 2.003599, + 1.999763, + 2.225933, + 2.014965, + 1.957361, + 2.081222, + 1.959784, + 2.042356, + 1.950692, + 1.927307, + 1.947006, + 1.92283 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "binary_trees": { + "native_lua55": { + "median_ms": 1.788474, + "samples_ms": [ + 1.835914, + 1.711952, + 1.829604, + 1.767112, + 1.845297, + 1.741385, + 1.788474, + 1.779731, + 1.738181, + 1.862923, + 1.737069, + 1.771039, + 1.931964, + 1.765701, + 1.786662, + 1.733764, + 1.758391, + 1.986743, + 1.871325, + 1.789896, + 1.803697, + 1.748155, + 1.74503, + 1.770147, + 1.795164, + 1.767053, + 1.815644, + 1.817056, + 1.825268, + 1.834061, + 1.802975 + ] + }, + "python_lower_bound": { + "median_ms": 2.03198, + "samples_ms": [ + 2.240244, + 2.092499, + 2.03198, + 2.03139, + 2.021985, + 2.239844, + 2.156993, + 2.144374, + 2.003999, + 2.023147, + 2.013493, + 2.031719, + 2.017128, + 2.020233, + 2.002697, + 2.015866, + 2.006332, + 2.04594, + 2.006303, + 2.024439, + 2.172125, + 2.13486, + 2.096044, + 2.158725, + 2.109704, + 2.029446, + 2.133358, + 2.064948, + 2.067622, + 2.01813, + 2.035966 + ] + }, + "luapyre": { + "median_ms": 22.708522, + "samples_ms": [ + 22.600174, + 23.323721, + 24.219932, + 22.831081, + 22.201932, + 22.182844, + 22.426761, + 22.366513, + 22.256251, + 22.417757, + 22.277813, + 22.520317, + 22.901785, + 22.248039, + 22.660022, + 22.708522, + 24.694885, + 22.703986, + 22.296981, + 22.509822, + 23.307638, + 24.941346, + 24.911893, + 23.75424, + 22.941152, + 23.619594, + 22.830681, + 23.909697, + 22.494719, + 25.343985, + 23.848729 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "native_lua55": { + "median_ms": 0.844304, + "samples_ms": [ + 0.839777, + 0.89638, + 0.850543, + 0.811837, + 0.839697, + 0.882681, + 0.8551, + 0.849752, + 1.079438, + 0.842071, + 0.839807, + 0.926424, + 0.822492, + 0.846007, + 0.830053, + 0.833929, + 1.129681, + 0.857773, + 0.838525, + 0.839147, + 0.84801, + 0.82768, + 0.834069, + 1.010677, + 0.816914, + 0.844304, + 1.061471, + 0.894958, + 0.824225, + 0.82153, + 0.889621 + ] + }, + "python_lower_bound": { + "median_ms": 1.985462, + "samples_ms": [ + 1.955928, + 2.002827, + 1.958462, + 2.028125, + 1.955247, + 1.953205, + 2.05955, + 2.03219, + 2.317858, + 2.061864, + 1.993524, + 2.015697, + 1.971963, + 1.998591, + 2.073872, + 2.019582, + 1.964201, + 1.945253, + 2.035014, + 1.971221, + 1.956199, + 1.954667, + 1.985462, + 1.937782, + 1.964992, + 1.948178, + 1.956249, + 1.996639, + 2.060542, + 1.960546, + 2.026372 + ] + }, + "luapyre": { + "median_ms": 2.693807, + "samples_ms": [ + 3.774386, + 2.70244, + 2.682511, + 2.650965, + 2.869003, + 2.735579, + 2.768587, + 2.693807, + 2.649443, + 2.673438, + 2.684674, + 2.647409, + 2.804079, + 2.780604, + 3.036208, + 2.699936, + 2.934329, + 2.805561, + 2.691504, + 2.908131, + 2.839681, + 2.670093, + 2.630755, + 2.662322, + 2.79839, + 2.633769, + 2.669812, + 2.652417, + 2.649863, + 2.657434, + 2.760906 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + }, + { + "python": "3.13.15", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "binary_trees": { + "python_lower_bound": { + "median_ms": 2.493314, + "samples_ms": [ + 3.478494, + 2.493314, + 2.438774, + 2.191032, + 4.90508, + 2.918626, + 2.72345, + 2.328503, + 2.216069, + 2.183682, + 2.347241, + 2.62716, + 2.616685, + 2.726325, + 3.065761, + 2.271861, + 2.342313, + 2.314794, + 2.563637, + 3.199036, + 3.232064, + 2.303988, + 2.499864, + 2.596445, + 2.262628, + 2.59282, + 2.248386, + 2.710382, + 2.338768, + 2.352108, + 2.4626 + ] + }, + "luapyre": { + "median_ms": 24.393839, + "samples_ms": [ + 27.032637, + 24.917819, + 22.820117, + 26.654752, + 24.735381, + 22.50199, + 25.236318, + 24.393839, + 24.474918, + 23.152886, + 25.864981, + 24.531723, + 23.417705, + 22.745488, + 23.874487, + 23.123433, + 24.485384, + 26.203088, + 23.427119, + 22.732178, + 23.584711, + 23.42087, + 24.685007, + 23.188229, + 31.67774, + 24.917289, + 25.175939, + 24.602496, + 24.293592, + 23.612852, + 24.208308 + ] + }, + "native_lua55": { + "median_ms": 1.753286, + "samples_ms": [ + 1.705426, + 1.717954, + 1.967991, + 1.711705, + 1.717543, + 1.880853, + 1.762019, + 1.717864, + 1.717063, + 1.726306, + 1.92613, + 2.04907, + 1.871169, + 1.733737, + 1.74946, + 1.744594, + 1.725605, + 1.825582, + 2.019216, + 1.877668, + 2.119674, + 2.355771, + 1.745104, + 1.939369, + 1.816098, + 1.753286, + 1.736962, + 1.761468, + 1.927752, + 1.749982, + 1.748789 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "python_lower_bound": { + "median_ms": 1.985197, + "samples_ms": [ + 1.999608, + 1.952559, + 1.936615, + 1.946821, + 1.923486, + 1.944517, + 1.936936, + 2.010884, + 2.18493, + 2.10263, + 1.929485, + 1.948352, + 2.292608, + 1.957827, + 2.206141, + 1.946901, + 2.425734, + 2.061459, + 2.12331, + 1.985197, + 1.944326, + 2.172191, + 2.036592, + 1.925869, + 1.947992, + 1.927852, + 1.94634, + 2.121006, + 2.101758, + 2.154796, + 2.056742 + ] + }, + "luapyre": { + "median_ms": 2.820033, + "samples_ms": [ + 3.760194, + 2.802747, + 2.905798, + 2.757892, + 2.783159, + 2.656824, + 2.792512, + 2.696182, + 2.742819, + 2.615573, + 2.720768, + 4.087366, + 5.544526, + 2.959277, + 2.820033, + 3.118562, + 2.977995, + 3.320269, + 2.871228, + 3.10384, + 2.853132, + 2.723752, + 2.842347, + 2.739545, + 2.761397, + 2.954952, + 3.033307, + 2.905088, + 2.731683, + 2.769459, + 2.655712 + ] + }, + "native_lua55": { + "median_ms": 0.836489, + "samples_ms": [ + 0.838031, + 0.832152, + 0.850931, + 0.822148, + 0.836489, + 0.829238, + 0.831592, + 0.851511, + 1.767076, + 0.915435, + 1.242125, + 0.781378, + 1.169508, + 0.927573, + 0.869828, + 0.816469, + 0.802388, + 0.846964, + 0.815158, + 0.819294, + 0.792775, + 0.798152, + 0.844741, + 0.814036, + 0.856268, + 0.792284, + 0.800145, + 1.010885, + 0.810231, + 0.888536, + 0.929316 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + } + ], + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 23.057312, + "process_medians_ms": [ + 23.057312, + 22.708522, + 24.393839 + ] + }, + "native_lua55": { + "median_ms": 1.788474, + "process_medians_ms": [ + 1.829805, + 1.788474, + 1.753286 + ] + }, + "python_lower_bound": { + "median_ms": 2.073441, + "process_medians_ms": [ + 2.073441, + 2.03198, + 2.493314 + ] + }, + "luapyre_over_native": 12.892170643800245, + "luapyre_over_python_lower_bound": 11.120312562546994, + "native_over_python_lower_bound": 0.8625632463137365 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.776719, + "process_medians_ms": [ + 2.776719, + 2.693807, + 2.820033 + ] + }, + "native_lua55": { + "median_ms": 0.836489, + "process_medians_ms": [ + 0.833298, + 0.844304, + 0.836489 + ] + }, + "python_lower_bound": { + "median_ms": 1.985462, + "process_medians_ms": [ + 2.007985, + 1.985462, + 1.985197 + ] + }, + "luapyre_over_native": 3.319492545628215, + "luapyre_over_python_lower_bound": 1.3985253809944487, + "native_over_python_lower_bound": 0.4213069804408244 + } + } +} diff --git a/benchmarks/results/native_headroom_040_314.json b/benchmarks/results/native_headroom_040_314.json new file mode 100644 index 0000000..5173b09 --- /dev/null +++ b/benchmarks/results/native_headroom_040_314.json @@ -0,0 +1,803 @@ +{ + "schema_version": 1, + "revision": "0.40.0a1", + "python": "3.14.7", + "native_runtime": "Lua 5.5 via lupa.lua55", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "warmups": 7, + "repeats": 31, + "processes": 3, + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "runs": [ + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "luapyre", + "native_lua55", + "python_lower_bound" + ], + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 21.245845, + "samples_ms": [ + 22.838534, + 21.083446, + 22.314054, + 22.038429, + 20.794681, + 21.458808, + 21.789504, + 23.348543, + 24.778114, + 24.701772, + 22.396856, + 24.318509, + 25.06453, + 20.88926, + 20.618192, + 20.936971, + 20.923872, + 20.879707, + 20.113731, + 20.169683, + 21.403827, + 20.680183, + 21.275729, + 20.718119, + 20.562421, + 21.245845, + 20.326744, + 25.630456, + 20.928248, + 21.072289, + 21.676207 + ] + }, + "native_lua55": { + "median_ms": 1.788217, + "samples_ms": [ + 1.8925, + 1.772013, + 1.875685, + 1.758804, + 1.762229, + 1.760347, + 1.774337, + 1.747217, + 2.100926, + 1.824741, + 1.925178, + 1.788217, + 1.792253, + 1.77642, + 1.786745, + 1.782303, + 1.779642, + 1.806782, + 1.781455, + 1.772622, + 1.799421, + 1.790248, + 1.793523, + 1.804349, + 1.782146, + 1.773763, + 1.814573, + 1.781565, + 1.848853, + 1.790298, + 2.004822 + ] + }, + "python_lower_bound": { + "median_ms": 2.009098, + "samples_ms": [ + 2.124376, + 2.003169, + 2.149193, + 2.080012, + 2.054865, + 2.02455, + 2.008948, + 2.102775, + 2.009098, + 2.079201, + 2.012273, + 1.999835, + 1.981468, + 2.005833, + 1.985954, + 2.007586, + 2.227086, + 2.056858, + 1.988137, + 1.996801, + 1.992934, + 2.005873, + 1.998232, + 2.009468, + 1.995918, + 2.020845, + 1.999044, + 2.016699, + 1.991292, + 2.01055, + 2.098829 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.837039, + "samples_ms": [ + 3.846355, + 3.035348, + 2.847073, + 2.936073, + 2.797481, + 2.779004, + 2.790801, + 2.807285, + 3.051782, + 3.036921, + 2.805363, + 2.888994, + 2.883837, + 2.793365, + 3.177577, + 2.916525, + 2.855236, + 2.795228, + 2.78344, + 3.110728, + 2.842356, + 2.837039, + 2.804581, + 2.788077, + 2.813364, + 2.799444, + 2.762349, + 2.85856, + 2.829368, + 2.791532, + 2.877097 + ] + }, + "native_lua55": { + "median_ms": 0.825276, + "samples_ms": [ + 0.904903, + 0.83447, + 0.859807, + 0.816764, + 0.989487, + 0.839127, + 0.893046, + 0.849272, + 0.817195, + 0.822824, + 0.80711, + 0.821711, + 0.844626, + 0.830685, + 0.819869, + 1.24439, + 0.899495, + 1.631165, + 0.792018, + 0.812919, + 0.914277, + 0.799078, + 0.874629, + 0.847339, + 0.810255, + 0.825276, + 0.798748, + 0.795022, + 0.79995, + 0.798628, + 0.799779 + ] + }, + "python_lower_bound": { + "median_ms": 2.364908, + "samples_ms": [ + 2.409664, + 2.332211, + 2.481138, + 2.336416, + 2.41412, + 2.416263, + 2.33235, + 2.415893, + 2.333453, + 2.35249, + 2.363255, + 2.438456, + 2.474458, + 2.450113, + 2.364768, + 2.364908, + 2.325591, + 2.347022, + 2.476842, + 2.347552, + 2.500136, + 2.341624, + 2.567585, + 2.386851, + 2.407872, + 2.553754, + 2.332831, + 2.375434, + 2.341434, + 2.359391, + 2.360982 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "native_lua55", + "python_lower_bound", + "luapyre" + ], + "workloads": { + "binary_trees": { + "native_lua55": { + "median_ms": 1.780293, + "samples_ms": [ + 1.863064, + 1.759804, + 1.825579, + 1.756379, + 1.994146, + 1.975539, + 1.754907, + 1.86704, + 1.749849, + 1.751581, + 1.780293, + 1.764831, + 1.76403, + 1.762847, + 1.794074, + 2.081394, + 1.874641, + 1.762887, + 2.021015, + 1.749458, + 1.776077, + 1.782887, + 1.767014, + 1.787914, + 1.747866, + 1.807052, + 1.775526, + 25.867066, + 1.825479, + 2.086301, + 1.766353 + ] + }, + "python_lower_bound": { + "median_ms": 1.965554, + "samples_ms": [ + 1.930963, + 1.950662, + 1.951363, + 1.951113, + 1.946286, + 2.002458, + 1.93562, + 1.948569, + 1.937312, + 1.931845, + 1.933868, + 1.952966, + 1.970281, + 1.949821, + 2.278962, + 1.959235, + 2.110446, + 2.048396, + 1.959195, + 2.025963, + 1.965554, + 1.985713, + 1.96969, + 1.992423, + 1.993235, + 1.974778, + 1.946036, + 1.975468, + 2.151777, + 1.991172, + 1.975359 + ] + }, + "luapyre": { + "median_ms": 21.28825, + "samples_ms": [ + 21.380154, + 21.060987, + 20.881645, + 20.824561, + 21.573788, + 21.094847, + 21.55506, + 20.66585, + 20.900042, + 21.777866, + 21.003863, + 21.153073, + 21.81486, + 21.680073, + 21.90379, + 21.085113, + 21.586716, + 21.031635, + 20.880623, + 20.895025, + 22.269044, + 22.619827, + 21.005907, + 21.38442, + 20.677296, + 23.195339, + 21.884772, + 21.029542, + 25.863681, + 21.322561, + 21.28825 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "native_lua55": { + "median_ms": 1.337126, + "samples_ms": [ + 1.392356, + 1.487716, + 1.346048, + 1.580862, + 1.335002, + 1.327842, + 1.328984, + 1.339749, + 1.341592, + 1.297166, + 1.337275, + 1.364486, + 1.324917, + 1.337126, + 1.308604, + 1.334312, + 1.313781, + 1.323125, + 1.391846, + 1.27176, + 1.341792, + 1.332638, + 1.344836, + 1.328943, + 1.288715, + 1.319259, + 1.560132, + 1.349553, + 1.331967, + 1.378717, + 1.414278 + ] + }, + "python_lower_bound": { + "median_ms": 2.343517, + "samples_ms": [ + 2.326062, + 2.362004, + 2.321845, + 2.350437, + 2.52306, + 2.346232, + 2.368123, + 2.321325, + 2.355154, + 2.351769, + 2.328815, + 2.343517, + 2.345169, + 2.344038, + 2.34569, + 2.335796, + 2.346491, + 2.323778, + 2.560564, + 2.588696, + 2.342466, + 2.340993, + 2.340162, + 2.337538, + 2.320353, + 2.364697, + 2.334413, + 2.323808, + 2.351969, + 2.327684, + 2.335695 + ] + }, + "luapyre": { + "median_ms": 2.950715, + "samples_ms": [ + 3.697206, + 2.802178, + 3.223844, + 4.157159, + 2.909524, + 3.007227, + 3.513528, + 3.476854, + 3.039465, + 2.907782, + 2.930185, + 2.991554, + 2.984004, + 2.813734, + 2.961791, + 2.950715, + 3.032946, + 2.823339, + 2.964655, + 3.502131, + 2.968511, + 2.788888, + 2.950295, + 2.902995, + 2.866411, + 2.82465, + 2.886641, + 2.999456, + 2.838641, + 2.769359, + 2.795238 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + }, + { + "python": "3.14.7", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "affinity_cpu": 0, + "hash_seed": "0", + "order": [ + "python_lower_bound", + "luapyre", + "native_lua55" + ], + "workloads": { + "binary_trees": { + "python_lower_bound": { + "median_ms": 1.988698, + "samples_ms": [ + 1.98996, + 1.982188, + 2.391788, + 2.057088, + 2.032002, + 1.972134, + 2.027324, + 1.978212, + 1.963201, + 1.980225, + 2.01718, + 2.181169, + 1.988698, + 2.03718, + 1.959135, + 1.981928, + 1.964192, + 1.977662, + 1.95578, + 2.05177, + 2.014496, + 1.980126, + 1.957523, + 1.979735, + 1.966275, + 1.96932, + 2.200057, + 1.99061, + 2.038151, + 2.105649, + 2.064899 + ] + }, + "luapyre": { + "median_ms": 20.878891, + "samples_ms": [ + 21.809281, + 20.708072, + 20.550571, + 20.668063, + 20.878891, + 20.853694, + 21.949527, + 20.867764, + 21.045896, + 21.516133, + 20.750203, + 20.545223, + 21.284735, + 21.135766, + 20.758956, + 21.166462, + 20.72746, + 20.632851, + 21.310763, + 20.430616, + 21.440021, + 20.749242, + 20.506236, + 21.404761, + 20.955793, + 20.6724, + 22.57407, + 21.073006, + 20.897769, + 20.954472, + 20.724585 + ] + }, + "native_lua55": { + "median_ms": 1.759923, + "samples_ms": [ + 1.705864, + 1.710251, + 1.743389, + 2.043869, + 1.799161, + 1.735588, + 1.728237, + 1.768305, + 1.754285, + 1.751011, + 1.740094, + 1.77146, + 1.745072, + 1.742478, + 1.746874, + 2.269048, + 1.777839, + 1.846711, + 1.773133, + 1.74369, + 1.773944, + 1.761916, + 1.772221, + 1.755847, + 1.760164, + 1.796337, + 1.759923, + 1.774354, + 1.732173, + 1.794334, + 1.752964 + ] + }, + "luapyre_warm_counters": { + "call_ic_hits": 60 + } + }, + "spectral_norm": { + "python_lower_bound": { + "median_ms": 2.363767, + "samples_ms": [ + 2.445786, + 2.449401, + 2.38678, + 2.946969, + 2.398137, + 2.360302, + 2.34604, + 2.328075, + 2.426999, + 2.340462, + 2.336496, + 2.554536, + 2.396494, + 2.361954, + 2.328685, + 2.333722, + 2.363767, + 2.324639, + 2.346782, + 2.335635, + 2.337419, + 2.419127, + 2.442522, + 2.430444, + 2.47612, + 2.353681, + 2.545693, + 2.37253, + 2.374001, + 2.346431, + 2.352029 + ] + }, + "luapyre": { + "median_ms": 2.808847, + "samples_ms": [ + 3.67905, + 2.813194, + 2.808847, + 2.819463, + 2.778784, + 2.812172, + 2.805392, + 2.786164, + 2.798492, + 2.808376, + 2.993307, + 2.777001, + 2.788188, + 2.89972, + 2.78991, + 3.163616, + 2.836758, + 2.801537, + 2.780495, + 2.892049, + 2.828597, + 2.902504, + 2.812152, + 2.796609, + 2.800104, + 3.062358, + 2.879351, + 2.801897, + 2.800355, + 2.808667, + 2.943584 + ] + }, + "native_lua55": { + "median_ms": 0.81413, + "samples_ms": [ + 0.956639, + 0.81413, + 0.819448, + 0.795554, + 0.821872, + 0.893426, + 0.817305, + 0.822232, + 0.795844, + 0.829793, + 0.815913, + 0.811116, + 0.817726, + 0.816003, + 0.810906, + 0.82006, + 0.811606, + 0.819048, + 0.808752, + 0.822433, + 0.81326, + 0.809013, + 0.812538, + 0.812618, + 0.822673, + 0.80035, + 0.813159, + 0.810355, + 0.808813, + 0.82762, + 0.799699 + ] + }, + "luapyre_warm_counters": { + "loop_executions": 1, + "loop_entry_executions": 1, + "loop_iterations": 30, + "call_ic_hits": 31, + "table_ic_hits": 2 + } + } + } + } + ], + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 21.245845, + "process_medians_ms": [ + 21.245845, + 21.28825, + 20.878891 + ] + }, + "native_lua55": { + "median_ms": 1.780293, + "process_medians_ms": [ + 1.788217, + 1.780293, + 1.759923 + ] + }, + "python_lower_bound": { + "median_ms": 1.988698, + "process_medians_ms": [ + 2.009098, + 1.965554, + 1.988698 + ] + }, + "luapyre_over_native": 11.933903576546108, + "luapyre_over_python_lower_bound": 10.683293793225516, + "native_over_python_lower_bound": 0.8952053051795696 + }, + "spectral_norm": { + "luapyre": { + "median_ms": 2.837039, + "process_medians_ms": [ + 2.837039, + 2.950715, + 2.808847 + ] + }, + "native_lua55": { + "median_ms": 0.825276, + "process_medians_ms": [ + 0.825276, + 1.337126, + 0.81413 + ] + }, + "python_lower_bound": { + "median_ms": 2.363767, + "process_medians_ms": [ + 2.364908, + 2.343517, + 2.363767 + ] + }, + "luapyre_over_native": 3.437685089594269, + "luapyre_over_python_lower_bound": 1.2002193955664833, + "native_over_python_lower_bound": 0.3491359342947084 + } + } +} diff --git a/benchmarks/results/profile_040_plan.json b/benchmarks/results/profile_040_plan.json new file mode 100644 index 0000000..f736a39 --- /dev/null +++ b/benchmarks/results/profile_040_plan.json @@ -0,0 +1,1502 @@ +{ + "schema_version": 1, + "purpose": "Planning evidence only; no candidate runtime implementation or speedup measurement", + "source": { + "version": "0.39.0a1", + "local_commit": "0bb256d", + "published_commit": "306428a135ec98e7ce8d1815d1bd7edc6086c3ab", + "tree": "7c78e3ac4ae5de56098eeaaae80d8cf528efde7b" + }, + "baseline_note": "Selected aggregates from the preceding full native_headroom rerun; not re-timed during this planning pass.", + "baseline": [ + { + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "processes": 3, + "python": "3.13.15", + "repeats": 31, + "revision": "0.39.0a1", + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 98.42692, + "process_medians_ms": [ + 97.584553, + 98.42692, + 100.135622 + ] + }, + "luapyre_over_native": 54.660061864486785, + "luapyre_over_python_lower_bound": 47.18117794135411, + "native_lua55": { + "median_ms": 1.80071, + "process_medians_ms": [ + 1.80071, + 1.804906, + 1.797325 + ] + }, + "native_over_python_lower_bound": 0.8631746165660346, + "python_lower_bound": { + "median_ms": 2.086148, + "process_medians_ms": [ + 2.097205, + 2.047281, + 2.086148 + ] + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 29.575543, + "process_medians_ms": [ + 30.739319, + 29.575543, + 28.623883 + ] + }, + "luapyre_over_native": 34.38412906571055, + "luapyre_over_python_lower_bound": 14.554972000340554, + "native_lua55": { + "median_ms": 0.860151, + "process_medians_ms": [ + 0.838199, + 0.860151, + 0.877386 + ] + }, + "native_over_python_lower_bound": 0.42330494899332627, + "python_lower_bound": { + "median_ms": 2.031989, + "process_medians_ms": [ + 2.065339, + 2.031989, + 1.975296 + ] + } + } + } + }, + { + "interpretation": "Native Lua uses the same untyped Lua algorithm. Python lower bounds use the same algorithm but omit Lua runtime semantics and are not attainable-speed guarantees or mathematical lower bounds.", + "method": "3 isolated processes; rotated implementation order; fixed CPU affinity and PYTHONHASHSEED=0; Python GC disabled during timing; Lua GC active; every result checked; median of process medians", + "platform": "Linux-6.18.44-x86_64-with-glibc2.39", + "processes": 3, + "python": "3.14.7", + "repeats": 31, + "revision": "0.39.0a1", + "warmups": 7, + "workloads": { + "binary_trees": { + "luapyre": { + "median_ms": 83.330924, + "process_medians_ms": [ + 85.991571, + 83.330924, + 82.358493 + ] + }, + "luapyre_over_native": 46.06099352013633, + "luapyre_over_python_lower_bound": 41.474327561433135, + "native_lua55": { + "median_ms": 1.809143, + "process_medians_ms": [ + 1.77319, + 1.809143, + 1.913095 + ] + }, + "native_over_python_lower_bound": 0.900421905647822, + "python_lower_bound": { + "median_ms": 2.009217, + "process_medians_ms": [ + 2.031298, + 1.961987, + 2.009217 + ] + } + }, + "spectral_norm": { + "luapyre": { + "median_ms": 24.75029, + "process_medians_ms": [ + 24.75029, + 23.544379, + 25.697834 + ] + }, + "luapyre_over_native": 29.038048196729005, + "luapyre_over_python_lower_bound": 10.049487563346558, + "native_lua55": { + "median_ms": 0.85234, + "process_medians_ms": [ + 0.858669, + 0.85234, + 0.821134 + ] + }, + "native_over_python_lower_bound": 0.34607999460785327, + "python_lower_bound": { + "median_ms": 2.462841, + "process_medians_ms": [ + 2.462841, + 2.356626, + 2.518843 + ] + } + } + } + } + ], + "profile_method": { + "preparation": "python_headroom.prepare; assert validate(run()) for every execution", + "executions": "Each version ran sequentially: 7 unprofiled warmups, 10 cProfile executions, 7 further unprofiled executions before adaptive disassembly.", + "accounting": "cProfile.getstats entries retained by code object; total self time is sum(entry.inlinetime). Do not aggregate by filename/name/line: generated functions can share all three.", + "caveats": "GC enabled; not CPU pinned; profiler overhead changes cost proportions. Self-time shares rank hypotheses, not production savings. Opcode counts are static including cold exits, not dynamic instruction counts.", + "specialization": "dis.get_instructions(code, adaptive=True) after unprofiled rewarm; CPython maintains bytecode and inline caches." + }, + "profiles": [ + { + "python": "3.13.15", + "warmups": 7, + "profiled_executions": 10, + "workloads": [ + { + "name": "binary_trees", + "total_profiled_seconds": 3.508428503, + "counters": { + "call_ic_hits": 600 + }, + "top_self": [ + { + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 15.600698076987433, + "self_seconds": 0.547339338 + }, + { + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 15.242768879078394, + "self_seconds": 0.534781648 + }, + { + "calls": 306000, + "file": "function_jit.py", + "function": "run", + "line": 202, + "self_percent": 12.492249838502694, + "self_seconds": 0.438281654 + }, + { + "calls": 306000, + "file": "function_jit.py", + "function": "finish", + "line": 189, + "self_percent": 7.120043198440519, + "self_seconds": 0.24980162500000003 + }, + { + "calls": 306600, + "file": "table.py", + "function": "rawset_fresh_prehashed", + "line": 245, + "self_percent": 5.318278734779735, + "self_seconds": 0.186588007 + }, + { + "calls": 1841340, + "file": "~", + "function": "", + "line": 0, + "self_percent": 5.157398300842616, + "self_seconds": 0.18094363200000002 + }, + { + "calls": 306600, + "file": "gc.py", + "function": "table_write_barrier", + "line": 447, + "self_percent": 4.336571284548135, + "self_seconds": 0.15214550300000002 + }, + { + "calls": 153300, + "file": "gcvm.py", + "function": "_new_table", + "line": 56, + "self_percent": 3.6482347264752004, + "self_seconds": 0.12799570700000001 + }, + { + "calls": 153070, + "file": "gc.py", + "function": "adopt", + "line": 387, + "self_percent": 3.6170693486125742, + "self_seconds": 0.126902292 + }, + { + "calls": 153350, + "file": "gc.py", + "function": "_object_size", + "line": 335, + "self_percent": 2.9398443466014674, + "self_seconds": 0.103142337 + }, + { + "calls": 766350, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.8410717480709056, + "self_seconds": 0.099676971 + }, + { + "calls": 926610, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.3960525610859227, + "self_seconds": 0.084063791 + }, + { + "calls": 153300, + "file": "gc.py", + "function": "_register_fresh", + "line": 380, + "self_percent": 2.383794879345158, + "self_seconds": 0.083633739 + }, + { + "calls": 459980, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.1679393761326993, + "self_seconds": 0.076060603 + }, + { + "calls": 613300, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.115549310368831, + "self_seconds": 0.074222535 + }, + { + "calls": 306350, + "file": "gc.py", + "function": "account_bytes", + "line": 371, + "self_percent": 1.800531860517723, + "self_seconds": 0.063170373 + } + ], + "generated": [ + { + "diagnostic_id": 1, + "adaptive_opcodes": { + "BINARY_OP": 8, + "BINARY_OP_ADD_INT": 23, + "BINARY_OP_SUBTRACT_INT": 12, + "BINARY_SUBSCR": 6, + "BINARY_SUBSCR_LIST_INT": 48, + "BUILD_TUPLE": 6, + "CALL": 23, + "CALL_ISINSTANCE": 7, + "CALL_KW": 2, + "CALL_PY_EXACT_ARGS": 8, + "CALL_PY_GENERAL": 2, + "CALL_TYPE_1": 8, + "COMPARE_OP": 6, + "COMPARE_OP_INT": 18, + "CONTAINS_OP": 6, + "COPY": 23, + "EXTENDED_ARG": 36, + "IS_OP": 23, + "JUMP_BACKWARD": 28, + "JUMP_FORWARD": 21, + "LOAD_ATTR": 4, + "LOAD_ATTR_INSTANCE_VALUE": 4, + "LOAD_ATTR_METHOD_NO_DICT": 4, + "LOAD_ATTR_METHOD_WITH_VALUES": 6, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 26, + "LOAD_CONST": 628, + "LOAD_FAST": 926, + "LOAD_FAST_CHECK": 2, + "LOAD_FAST_LOAD_FAST": 145, + "LOAD_GLOBAL": 56, + "LOAD_GLOBAL_BUILTIN": 21, + "LOAD_GLOBAL_MODULE": 21, + "NOP": 1, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 53, + "POP_JUMP_IF_NONE": 20, + "POP_JUMP_IF_NOT_NONE": 5, + "POP_JUMP_IF_TRUE": 11, + "POP_TOP": 16, + "PUSH_EXC_INFO": 1, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 18, + "STORE_ATTR_SLOT": 2, + "STORE_FAST": 130, + "STORE_SUBSCR": 416, + "STORE_SUBSCR_LIST_INT": 61, + "SWAP": 10, + "TO_BOOL": 14, + "TO_BOOL_BOOL": 11, + "UNARY_NOT": 1, + "UNPACK_SEQUENCE": 2, + "UNPACK_SEQUENCE_TWO_TUPLE": 2 + }, + "bytecode_bytes": 9518, + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 50, + "self_percent": 15.600698076987433, + "self_seconds": 0.547339338 + }, + { + "diagnostic_id": 2, + "adaptive_opcodes": { + "BINARY_OP": 14, + "BINARY_OP_ADD_INT": 26, + "BINARY_OP_SUBTRACT_INT": 10, + "BINARY_SUBSCR": 7, + "BINARY_SUBSCR_LIST_INT": 43, + "BINARY_SUBSCR_TUPLE_INT": 3, + "BUILD_STRING": 1, + "BUILD_TUPLE": 6, + "CALL": 17, + "CALL_ISINSTANCE": 9, + "CALL_KW": 2, + "CALL_METHOD_DESCRIPTOR_FAST": 3, + "CALL_PY_EXACT_ARGS": 4, + "CALL_TYPE_1": 14, + "COMPARE_OP": 7, + "COMPARE_OP_INT": 17, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 1, + "COPY": 23, + "EXTENDED_ARG": 32, + "FORMAT_SIMPLE": 2, + "IS_OP": 27, + "JUMP_BACKWARD": 25, + "JUMP_FORWARD": 28, + "LOAD_ATTR": 5, + "LOAD_ATTR_INSTANCE_VALUE": 4, + "LOAD_ATTR_METHOD_NO_DICT": 3, + "LOAD_ATTR_METHOD_WITH_VALUES": 4, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 28, + "LOAD_CONST": 483, + "LOAD_FAST": 687, + "LOAD_FAST_CHECK": 2, + "LOAD_FAST_LOAD_FAST": 115, + "LOAD_GLOBAL": 47, + "LOAD_GLOBAL_BUILTIN": 32, + "LOAD_GLOBAL_MODULE": 33, + "NOP": 1, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 56, + "POP_JUMP_IF_NONE": 15, + "POP_JUMP_IF_NOT_NONE": 5, + "POP_JUMP_IF_TRUE": 13, + "POP_TOP": 12, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 1, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 2, + "STORE_FAST": 133, + "STORE_SUBSCR": 317, + "STORE_SUBSCR_LIST_INT": 13, + "SWAP": 10, + "TO_BOOL": 6, + "TO_BOOL_BOOL": 14, + "TO_BOOL_INT": 2, + "UNARY_NOT": 1, + "UNPACK_SEQUENCE": 2, + "UNPACK_SEQUENCE_TWO_TUPLE": 2 + }, + "bytecode_bytes": 8230, + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 55, + "self_percent": 15.242768879078394, + "self_seconds": 0.534781648 + } + ] + }, + { + "name": "spectral_norm", + "total_profiled_seconds": 0.726284296, + "counters": { + "call_ic_hits": 310, + "compile_failures": 1, + "loop_entry_executions": 10, + "loop_executions": 10, + "loop_iterations": 300, + "table_ic_hits": 20 + }, + "top_self": [ + { + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 41.595823792946234, + "self_seconds": 0.302103936 + }, + { + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 41.303087048986676, + "self_seconds": 0.299977835 + }, + { + "calls": 180010, + "file": "opdispatch.py", + "function": "_float_divide", + "line": 231, + "self_percent": 5.2207532516991115, + "self_seconds": 0.037917511 + }, + { + "calls": 380112, + "file": "~", + "function": "", + "line": 0, + "self_percent": 4.299435382532352, + "self_seconds": 0.031226124 + }, + { + "calls": 369029, + "file": "~", + "function": "", + "line": 0, + "self_percent": 4.1850486328015, + "self_seconds": 0.030395351 + }, + { + "calls": 10, + "file": "jitvm.py", + "function": "run", + "line": 407, + "self_percent": 0.6718307179259182, + "self_seconds": 0.004879401 + }, + { + "calls": 310, + "file": "optimizing_jitvm.py", + "function": "_invoke", + "line": 163, + "self_percent": 0.23593254176598638, + "self_seconds": 0.001713541 + }, + { + "calls": 7434, + "file": "~", + "function": "", + "line": 0, + "self_percent": 0.1479193486513166, + "self_seconds": 0.0010743150000000002 + }, + { + "calls": 310, + "file": "jitvm.py", + "function": "_invoke_site", + "line": 185, + "self_percent": 0.13127146012255234, + "self_seconds": 0.000953404 + }, + { + "calls": 12517, + "file": "vm.py", + "function": "proto", + "line": 60, + "self_percent": 0.1259041404359375, + "self_seconds": 0.000914422 + }, + { + "calls": 300, + "file": "opdispatch.py", + "function": "_call", + "line": 789, + "self_percent": 0.10628097072334332, + "self_seconds": 0.0007719020000000001 + }, + { + "calls": 300, + "file": "table.py", + "function": "rawset", + "line": 166, + "self_percent": 0.09257710839998669, + "self_seconds": 0.0006723730000000001 + }, + { + "calls": 4, + "file": "range_analysis.py", + "function": "analyze_integer_ranges", + "line": 126, + "self_percent": 0.0728258896568514, + "self_seconds": 0.000528923 + }, + { + "calls": 2352, + "file": "~", + "function": "", + "line": 0, + "self_percent": 0.07243472052161788, + "self_seconds": 0.000526082 + }, + { + "calls": 200, + "file": "optimizing_jitvm.py", + "function": "_acquire_compiled_frame", + "line": 95, + "self_percent": 0.06764059235558634, + "self_seconds": 0.000491263 + }, + { + "calls": 100, + "file": "vm.py", + "function": "_new_frame", + "line": 657, + "self_percent": 0.06074164103914482, + "self_seconds": 0.000441157 + } + ], + "generated": [ + { + "diagnostic_id": 1, + "adaptive_opcodes": { + "BINARY_OP": 23, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 42, + "BINARY_OP_MULTIPLY_FLOAT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_OP_SUBTRACT_INT": 17, + "BINARY_SUBSCR": 5, + "BINARY_SUBSCR_LIST_INT": 54, + "BUILD_STRING": 4, + "BUILD_TUPLE": 3, + "CALL": 31, + "CALL_BUILTIN_CLASS": 3, + "CALL_ISINSTANCE": 3, + "CALL_LEN": 4, + "CALL_LIST_APPEND": 1, + "CALL_PY_EXACT_ARGS": 1, + "CALL_PY_GENERAL": 1, + "CALL_TYPE_1": 31, + "COMPARE_OP": 18, + "COMPARE_OP_INT": 33, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 2, + "COPY": 15, + "EXTENDED_ARG": 33, + "FORMAT_SIMPLE": 6, + "IS_OP": 24, + "JUMP_BACKWARD": 28, + "JUMP_FORWARD": 16, + "LOAD_ATTR": 11, + "LOAD_ATTR_INSTANCE_VALUE": 1, + "LOAD_ATTR_METHOD_NO_DICT": 1, + "LOAD_ATTR_METHOD_WITH_VALUES": 1, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 16, + "LOAD_CONST": 669, + "LOAD_FAST": 1107, + "LOAD_FAST_CHECK": 16, + "LOAD_FAST_LOAD_FAST": 98, + "LOAD_GLOBAL": 52, + "LOAD_GLOBAL_BUILTIN": 62, + "LOAD_GLOBAL_MODULE": 30, + "NOP": 1, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 79, + "POP_JUMP_IF_NONE": 4, + "POP_JUMP_IF_TRUE": 16, + "POP_TOP": 10, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 9, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 176, + "STORE_SUBSCR": 465, + "STORE_SUBSCR_LIST_INT": 34, + "SWAP": 8, + "TO_BOOL": 5, + "TO_BOOL_BOOL": 3, + "TO_BOOL_INT": 2 + }, + "bytecode_bytes": 10636, + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 67, + "self_percent": 41.595823792946234, + "self_seconds": 0.302103936 + }, + { + "diagnostic_id": 2, + "adaptive_opcodes": { + "BINARY_OP": 23, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 42, + "BINARY_OP_MULTIPLY_FLOAT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_OP_SUBTRACT_INT": 17, + "BINARY_SUBSCR": 5, + "BINARY_SUBSCR_LIST_INT": 54, + "BUILD_STRING": 4, + "BUILD_TUPLE": 3, + "CALL": 31, + "CALL_BUILTIN_CLASS": 3, + "CALL_ISINSTANCE": 3, + "CALL_LEN": 4, + "CALL_LIST_APPEND": 1, + "CALL_PY_EXACT_ARGS": 1, + "CALL_PY_GENERAL": 1, + "CALL_TYPE_1": 31, + "COMPARE_OP": 18, + "COMPARE_OP_INT": 33, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 2, + "COPY": 15, + "EXTENDED_ARG": 33, + "FORMAT_SIMPLE": 6, + "IS_OP": 24, + "JUMP_BACKWARD": 28, + "JUMP_FORWARD": 16, + "LOAD_ATTR": 11, + "LOAD_ATTR_INSTANCE_VALUE": 1, + "LOAD_ATTR_METHOD_NO_DICT": 1, + "LOAD_ATTR_METHOD_WITH_VALUES": 1, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 16, + "LOAD_CONST": 669, + "LOAD_FAST": 1107, + "LOAD_FAST_CHECK": 16, + "LOAD_FAST_LOAD_FAST": 98, + "LOAD_GLOBAL": 52, + "LOAD_GLOBAL_BUILTIN": 62, + "LOAD_GLOBAL_MODULE": 30, + "NOP": 1, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 79, + "POP_JUMP_IF_NONE": 4, + "POP_JUMP_IF_TRUE": 16, + "POP_TOP": 10, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 9, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 176, + "STORE_SUBSCR": 465, + "STORE_SUBSCR_LIST_INT": 34, + "SWAP": 8, + "TO_BOOL": 5, + "TO_BOOL_BOOL": 3, + "TO_BOOL_INT": 2 + }, + "bytecode_bytes": 10636, + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 67, + "self_percent": 41.303087048986676, + "self_seconds": 0.299977835 + }, + { + "diagnostic_id": 3, + "adaptive_opcodes": { + "BINARY_OP": 4, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 2, + "BINARY_OP_MULTIPLY_FLOAT": 2, + "BINARY_OP_SUBTRACT_INT": 3, + "BINARY_SUBSCR_LIST_INT": 18, + "BUILD_TUPLE": 5, + "CALL": 4, + "CALL_BUILTIN_CLASS": 4, + "CALL_ISINSTANCE": 2, + "CALL_LEN": 2, + "CALL_TYPE_1": 7, + "COMPARE_OP": 9, + "COMPARE_OP_INT": 8, + "COPY": 2, + "EXTENDED_ARG": 3, + "IS_OP": 7, + "JUMP_BACKWARD": 2, + "JUMP_FORWARD": 6, + "LOAD_ATTR": 2, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 8, + "LOAD_CONST": 154, + "LOAD_FAST": 187, + "LOAD_FAST_LOAD_FAST": 67, + "LOAD_GLOBAL": 2, + "LOAD_GLOBAL_BUILTIN": 22, + "LOAD_GLOBAL_MODULE": 4, + "POP_JUMP_IF_FALSE": 21, + "POP_JUMP_IF_NONE": 2, + "POP_JUMP_IF_TRUE": 4, + "POP_TOP": 2, + "RESUME_CHECK": 1, + "RETURN_CONST": 2, + "RETURN_VALUE": 5, + "STORE_ATTR": 6, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 42, + "STORE_SUBSCR": 96, + "STORE_SUBSCR_LIST_INT": 16, + "SWAP": 2, + "TO_BOOL_BOOL": 2 + }, + "bytecode_bytes": 2464, + "calls": 10, + "file": "", + "function": "_jit_structured_loop", + "line": 1, + "locals": 28, + "self_percent": 0.059348247287450646, + "self_seconds": 0.000431037 + } + ] + } + ] + }, + { + "python": "3.14.7", + "warmups": 7, + "profiled_executions": 10, + "workloads": [ + { + "name": "binary_trees", + "total_profiled_seconds": 3.1162029060000003, + "counters": { + "call_ic_hits": 600 + }, + "top_self": [ + { + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 14.81073334831169, + "self_seconds": 0.46153250300000004 + }, + { + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 14.674815369676699, + "self_seconds": 0.457297023 + }, + { + "calls": 306000, + "file": "function_jit.py", + "function": "run", + "line": 202, + "self_percent": 12.872416370181, + "self_seconds": 0.401130613 + }, + { + "calls": 306000, + "file": "function_jit.py", + "function": "finish", + "line": 189, + "self_percent": 7.126423942818826, + "self_seconds": 0.22207383000000003 + }, + { + "calls": 306600, + "file": "table.py", + "function": "rawset_fresh_prehashed", + "line": 245, + "self_percent": 5.646531959174035, + "self_seconds": 0.17595739300000002 + }, + { + "calls": 1841340, + "file": "~", + "function": "", + "line": 0, + "self_percent": 5.133677421710227, + "self_seconds": 0.159975805 + }, + { + "calls": 306600, + "file": "gc.py", + "function": "table_write_barrier", + "line": 447, + "self_percent": 4.294852839727119, + "self_seconds": 0.133836329 + }, + { + "calls": 153300, + "file": "gcvm.py", + "function": "_new_table", + "line": 56, + "self_percent": 3.95748998765615, + "self_seconds": 0.123323418 + }, + { + "calls": 153070, + "file": "gc.py", + "function": "adopt", + "line": 387, + "self_percent": 3.626638457412439, + "self_seconds": 0.11301341300000001 + }, + { + "calls": 153350, + "file": "gc.py", + "function": "_object_size", + "line": 335, + "self_percent": 3.213666440243028, + "self_seconds": 0.10014436700000001 + }, + { + "calls": 766350, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.837153730579314, + "self_seconds": 0.08841146700000001 + }, + { + "calls": 153300, + "file": "gc.py", + "function": "_register_fresh", + "line": 380, + "self_percent": 2.474218699031019, + "self_seconds": 0.07710167500000001 + }, + { + "calls": 926610, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.3207998381861463, + "self_seconds": 0.072320832 + }, + { + "calls": 459980, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.122475043991888, + "self_seconds": 0.066140629 + }, + { + "calls": 613300, + "file": "~", + "function": "", + "line": 0, + "self_percent": 2.017863787975044, + "self_seconds": 0.06288073000000001 + }, + { + "calls": 306350, + "file": "gc.py", + "function": "account_bytes", + "line": 371, + "self_percent": 1.8686647422053333, + "self_seconds": 0.058231385000000004 + } + ], + "generated": [ + { + "diagnostic_id": 1, + "adaptive_opcodes": { + "BINARY_OP": 14, + "BINARY_OP_ADD_INT": 23, + "BINARY_OP_SUBSCR_LIST_INT": 48, + "BINARY_OP_SUBTRACT_INT": 12, + "BUILD_TUPLE": 6, + "CALL": 23, + "CALL_ISINSTANCE": 7, + "CALL_KW_PY": 2, + "CALL_PY_EXACT_ARGS": 8, + "CALL_PY_GENERAL": 2, + "CALL_TYPE_1": 8, + "COMPARE_OP": 6, + "COMPARE_OP_INT": 18, + "CONTAINS_OP": 6, + "COPY": 23, + "EXTENDED_ARG": 36, + "IS_OP": 23, + "JUMP_BACKWARD": 23, + "JUMP_BACKWARD_NO_JIT": 5, + "JUMP_FORWARD": 21, + "LOAD_ATTR": 4, + "LOAD_ATTR_INSTANCE_VALUE": 4, + "LOAD_ATTR_METHOD_NO_DICT": 4, + "LOAD_ATTR_METHOD_WITH_VALUES": 6, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 26, + "LOAD_CONST": 11, + "LOAD_CONST_IMMORTAL": 8, + "LOAD_CONST_MORTAL": 4, + "LOAD_FAST": 9, + "LOAD_FAST_BORROW": 917, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 145, + "LOAD_FAST_CHECK": 2, + "LOAD_GLOBAL": 56, + "LOAD_GLOBAL_BUILTIN": 21, + "LOAD_GLOBAL_MODULE": 21, + "LOAD_SMALL_INT": 605, + "NOP": 1, + "NOT_TAKEN": 89, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 53, + "POP_JUMP_IF_NONE": 20, + "POP_JUMP_IF_NOT_NONE": 5, + "POP_JUMP_IF_TRUE": 11, + "POP_TOP": 16, + "PUSH_EXC_INFO": 1, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 18, + "STORE_ATTR_SLOT": 2, + "STORE_FAST": 130, + "STORE_SUBSCR": 416, + "STORE_SUBSCR_LIST_INT": 61, + "SWAP": 10, + "TO_BOOL": 14, + "TO_BOOL_BOOL": 11, + "UNARY_NOT": 1, + "UNPACK_SEQUENCE": 2, + "UNPACK_SEQUENCE_TWO_TUPLE": 2 + }, + "bytecode_bytes": 10484, + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 50, + "self_percent": 14.81073334831169, + "self_seconds": 0.46153250300000004 + }, + { + "diagnostic_id": 2, + "adaptive_opcodes": { + "BINARY_OP": 21, + "BINARY_OP_ADD_INT": 26, + "BINARY_OP_SUBSCR_LIST_INT": 43, + "BINARY_OP_SUBSCR_TUPLE_INT": 3, + "BINARY_OP_SUBTRACT_INT": 10, + "BUILD_STRING": 1, + "BUILD_TUPLE": 6, + "CALL": 17, + "CALL_ISINSTANCE": 9, + "CALL_KW_PY": 2, + "CALL_METHOD_DESCRIPTOR_FAST": 3, + "CALL_PY_EXACT_ARGS": 4, + "CALL_TYPE_1": 14, + "COMPARE_OP": 7, + "COMPARE_OP_INT": 17, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 1, + "COPY": 23, + "EXTENDED_ARG": 32, + "FORMAT_SIMPLE": 2, + "IS_OP": 27, + "JUMP_BACKWARD": 20, + "JUMP_BACKWARD_NO_JIT": 5, + "JUMP_FORWARD": 28, + "LOAD_ATTR": 5, + "LOAD_ATTR_INSTANCE_VALUE": 4, + "LOAD_ATTR_METHOD_NO_DICT": 3, + "LOAD_ATTR_METHOD_WITH_VALUES": 4, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 28, + "LOAD_CONST": 15, + "LOAD_CONST_IMMORTAL": 9, + "LOAD_CONST_MORTAL": 4, + "LOAD_FAST": 13, + "LOAD_FAST_BORROW": 674, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 115, + "LOAD_FAST_CHECK": 2, + "LOAD_GLOBAL": 47, + "LOAD_GLOBAL_BUILTIN": 32, + "LOAD_GLOBAL_MODULE": 33, + "LOAD_SMALL_INT": 455, + "NOP": 1, + "NOT_TAKEN": 89, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 55, + "POP_JUMP_IF_NONE": 15, + "POP_JUMP_IF_NOT_NONE": 5, + "POP_JUMP_IF_TRUE": 14, + "POP_TOP": 12, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 1, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 2, + "STORE_FAST": 133, + "STORE_SUBSCR": 317, + "STORE_SUBSCR_LIST_INT": 13, + "SWAP": 10, + "TO_BOOL": 6, + "TO_BOOL_BOOL": 14, + "TO_BOOL_INT": 2, + "UNARY_NOT": 1, + "UNPACK_SEQUENCE": 2, + "UNPACK_SEQUENCE_TWO_TUPLE": 2 + }, + "bytecode_bytes": 9244, + "calls": 153300, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 55, + "self_percent": 14.674815369676699, + "self_seconds": 0.457297023 + } + ] + }, + { + "name": "spectral_norm", + "total_profiled_seconds": 0.605793879, + "counters": { + "call_ic_hits": 310, + "compile_failures": 1, + "loop_entry_executions": 10, + "loop_executions": 10, + "loop_iterations": 300, + "table_ic_hits": 20 + }, + "top_self": [ + { + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 41.62509819614734, + "self_seconds": 0.25216229700000004 + }, + { + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "self_percent": 41.09026595166373, + "self_seconds": 0.248922316 + }, + { + "calls": 180010, + "file": "opdispatch.py", + "function": "_float_divide", + "line": 231, + "self_percent": 5.228853096417635, + "self_seconds": 0.031676072 + }, + { + "calls": 380112, + "file": "~", + "function": "", + "line": 0, + "self_percent": 4.209905362876735, + "self_seconds": 0.025503349 + }, + { + "calls": 369029, + "file": "~", + "function": "", + "line": 0, + "self_percent": 4.051097056396637, + "self_seconds": 0.024541298000000003 + }, + { + "calls": 10, + "file": "jitvm.py", + "function": "run", + "line": 407, + "self_percent": 0.7328766687654168, + "self_seconds": 0.004439722 + }, + { + "calls": 310, + "file": "optimizing_jitvm.py", + "function": "_invoke", + "line": 163, + "self_percent": 0.22258726057547373, + "self_seconds": 0.0013484200000000001 + }, + { + "calls": 7434, + "file": "~", + "function": "", + "line": 0, + "self_percent": 0.15555951168664747, + "self_seconds": 0.0009423700000000001 + }, + { + "calls": 12517, + "file": "vm.py", + "function": "proto", + "line": 60, + "self_percent": 0.1448657753770404, + "self_seconds": 0.000877588 + }, + { + "calls": 310, + "file": "jitvm.py", + "function": "_invoke_site", + "line": 185, + "self_percent": 0.13358784696469342, + "self_seconds": 0.000809267 + }, + { + "calls": 300, + "file": "table.py", + "function": "rawset", + "line": 166, + "self_percent": 0.11421772718175648, + "self_seconds": 0.000691924 + }, + { + "calls": 300, + "file": "opdispatch.py", + "function": "_call", + "line": 789, + "self_percent": 0.10942896965124337, + "self_seconds": 0.000662914 + }, + { + "calls": 4, + "file": "range_analysis.py", + "function": "analyze_integer_ranges", + "line": 126, + "self_percent": 0.10216049079624326, + "self_seconds": 0.0006188820000000001 + }, + { + "calls": 2352, + "file": "~", + "function": "", + "line": 0, + "self_percent": 0.07484740531688337, + "self_seconds": 0.000453421 + }, + { + "calls": 200, + "file": "optimizing_jitvm.py", + "function": "_acquire_compiled_frame", + "line": 95, + "self_percent": 0.06889092387148402, + "self_seconds": 0.00041733700000000005 + }, + { + "calls": 110, + "file": "threadvm.py", + "function": "_return", + "line": 291, + "self_percent": 0.0676631135125748, + "self_seconds": 0.00040989900000000003 + } + ], + "generated": [ + { + "diagnostic_id": 1, + "adaptive_opcodes": { + "BINARY_OP": 27, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 42, + "BINARY_OP_EXTEND": 1, + "BINARY_OP_MULTIPLY_FLOAT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_OP_SUBSCR_LIST_INT": 54, + "BINARY_OP_SUBTRACT_INT": 17, + "BUILD_STRING": 4, + "BUILD_TUPLE": 3, + "CALL": 31, + "CALL_BUILTIN_CLASS": 3, + "CALL_ISINSTANCE": 3, + "CALL_LEN": 4, + "CALL_LIST_APPEND": 1, + "CALL_PY_EXACT_ARGS": 1, + "CALL_PY_GENERAL": 1, + "CALL_TYPE_1": 31, + "COMPARE_OP": 18, + "COMPARE_OP_INT": 33, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 2, + "COPY": 15, + "EXTENDED_ARG": 33, + "FORMAT_SIMPLE": 6, + "IS_OP": 24, + "JUMP_BACKWARD": 20, + "JUMP_BACKWARD_NO_JIT": 8, + "JUMP_FORWARD": 16, + "LOAD_ATTR": 11, + "LOAD_ATTR_INSTANCE_VALUE": 1, + "LOAD_ATTR_METHOD_NO_DICT": 1, + "LOAD_ATTR_METHOD_WITH_VALUES": 1, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 16, + "LOAD_CONST": 14, + "LOAD_CONST_IMMORTAL": 1, + "LOAD_CONST_MORTAL": 1, + "LOAD_FAST": 29, + "LOAD_FAST_BORROW": 1078, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 98, + "LOAD_FAST_CHECK": 16, + "LOAD_GLOBAL": 52, + "LOAD_GLOBAL_BUILTIN": 62, + "LOAD_GLOBAL_MODULE": 30, + "LOAD_SMALL_INT": 653, + "NOP": 1, + "NOT_TAKEN": 99, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 75, + "POP_JUMP_IF_NONE": 4, + "POP_JUMP_IF_TRUE": 20, + "POP_TOP": 10, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 9, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 176, + "STORE_SUBSCR": 465, + "STORE_SUBSCR_LIST_INT": 34, + "SWAP": 8, + "TO_BOOL": 5, + "TO_BOOL_BOOL": 3, + "TO_BOOL_INT": 2 + }, + "bytecode_bytes": 11994, + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 67, + "self_percent": 41.62509819614734, + "self_seconds": 0.25216229700000004 + }, + { + "diagnostic_id": 2, + "adaptive_opcodes": { + "BINARY_OP": 27, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 42, + "BINARY_OP_EXTEND": 1, + "BINARY_OP_MULTIPLY_FLOAT": 1, + "BINARY_OP_MULTIPLY_INT": 1, + "BINARY_OP_SUBSCR_LIST_INT": 54, + "BINARY_OP_SUBTRACT_INT": 17, + "BUILD_STRING": 4, + "BUILD_TUPLE": 3, + "CALL": 31, + "CALL_BUILTIN_CLASS": 3, + "CALL_ISINSTANCE": 3, + "CALL_LEN": 4, + "CALL_LIST_APPEND": 1, + "CALL_PY_EXACT_ARGS": 1, + "CALL_PY_GENERAL": 1, + "CALL_TYPE_1": 31, + "COMPARE_OP": 18, + "COMPARE_OP_INT": 33, + "CONTAINS_OP": 10, + "CONVERT_VALUE": 2, + "COPY": 15, + "EXTENDED_ARG": 33, + "FORMAT_SIMPLE": 6, + "IS_OP": 24, + "JUMP_BACKWARD": 20, + "JUMP_BACKWARD_NO_JIT": 8, + "JUMP_FORWARD": 16, + "LOAD_ATTR": 11, + "LOAD_ATTR_INSTANCE_VALUE": 1, + "LOAD_ATTR_METHOD_NO_DICT": 1, + "LOAD_ATTR_METHOD_WITH_VALUES": 1, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 16, + "LOAD_CONST": 14, + "LOAD_CONST_IMMORTAL": 1, + "LOAD_CONST_MORTAL": 1, + "LOAD_FAST": 29, + "LOAD_FAST_BORROW": 1078, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 98, + "LOAD_FAST_CHECK": 16, + "LOAD_GLOBAL": 52, + "LOAD_GLOBAL_BUILTIN": 62, + "LOAD_GLOBAL_MODULE": 30, + "LOAD_SMALL_INT": 653, + "NOP": 1, + "NOT_TAKEN": 99, + "POP_EXCEPT": 1, + "POP_JUMP_IF_FALSE": 75, + "POP_JUMP_IF_NONE": 4, + "POP_JUMP_IF_TRUE": 20, + "POP_TOP": 10, + "PUSH_EXC_INFO": 1, + "RAISE_VARARGS": 9, + "RERAISE": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 2, + "STORE_ATTR": 15, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 176, + "STORE_SUBSCR": 465, + "STORE_SUBSCR_LIST_INT": 34, + "SWAP": 8, + "TO_BOOL": 5, + "TO_BOOL_BOOL": 3, + "TO_BOOL_INT": 2 + }, + "bytecode_bytes": 11994, + "calls": 100, + "file": "", + "function": "_jit_ast_function", + "line": 1, + "locals": 67, + "self_percent": 41.09026595166373, + "self_seconds": 0.248922316 + }, + { + "diagnostic_id": 3, + "adaptive_opcodes": { + "BINARY_OP": 3, + "BINARY_OP_ADD_FLOAT": 2, + "BINARY_OP_ADD_INT": 2, + "BINARY_OP_MULTIPLY_FLOAT": 2, + "BINARY_OP_SUBSCR_LIST_INT": 18, + "BINARY_OP_SUBTRACT_INT": 3, + "BUILD_TUPLE": 5, + "CALL": 4, + "CALL_BUILTIN_CLASS": 4, + "CALL_ISINSTANCE": 2, + "CALL_LEN": 2, + "CALL_TYPE_1": 7, + "COMPARE_OP": 8, + "COMPARE_OP_INT": 8, + "COPY": 2, + "EXTENDED_ARG": 3, + "IS_OP": 7, + "JUMP_BACKWARD": 1, + "JUMP_BACKWARD_NO_JIT": 1, + "JUMP_FORWARD": 6, + "LOAD_ATTR": 2, + "LOAD_ATTR_PROPERTY": 1, + "LOAD_ATTR_SLOT": 8, + "LOAD_CONST": 5, + "LOAD_CONST_IMMORTAL": 1, + "LOAD_FAST": 11, + "LOAD_FAST_BORROW": 176, + "LOAD_FAST_BORROW_LOAD_FAST_BORROW": 66, + "LOAD_GLOBAL": 2, + "LOAD_GLOBAL_BUILTIN": 22, + "LOAD_GLOBAL_MODULE": 4, + "LOAD_SMALL_INT": 149, + "NOT_TAKEN": 26, + "POP_JUMP_IF_FALSE": 18, + "POP_JUMP_IF_NONE": 2, + "POP_JUMP_IF_TRUE": 6, + "POP_TOP": 2, + "RESUME_CHECK": 1, + "RETURN_VALUE": 7, + "STORE_ATTR": 6, + "STORE_ATTR_SLOT": 1, + "STORE_FAST": 42, + "STORE_SUBSCR": 96, + "STORE_SUBSCR_LIST_INT": 16, + "SWAP": 2, + "TO_BOOL_BOOL": 2 + }, + "bytecode_bytes": 2744, + "calls": 10, + "file": "", + "function": "_jit_structured_loop", + "line": 1, + "locals": 28, + "self_percent": 0.06704128484599628, + "self_seconds": 0.000406132 + } + ] + } + ] + } + ], + "reproduction_python": [ + "import cProfile, dis, json, platform", + "from collections import Counter", + "from dataclasses import asdict", + "from pathlib import Path", + "from types import CodeType", + "from python_headroom import prepare", + "rows = []", + "for name in (\"binary_trees\", \"spectral_norm\"):", + " runtime, run, validate = prepare(name)", + " for _ in range(7):", + " assert validate(run())", + " before = asdict(runtime.jit_stats)", + " p = cProfile.Profile()", + " for _ in range(10):", + " assert validate(p.runcall(run))", + " after = asdict(runtime.jit_stats)", + " entries = p.getstats()", + " total = sum(e.inlinetime for e in entries)", + " # Remove profiler instrumentation and re-specialize before bytecode inspection.", + " for _ in range(7):", + " assert validate(run())", + " records = []", + " generated = []", + " for e in entries:", + " c = e.code", + " record = dict(file=Path(c.co_filename).name if isinstance(c, CodeType) else \"~\",", + " line=c.co_firstlineno if isinstance(c, CodeType) else 0,", + " function=c.co_name if isinstance(c, CodeType) else c,", + " calls=e.callcount, self_seconds=e.inlinetime,", + " self_percent=100*e.inlinetime/total)", + " records.append(record)", + " if isinstance(c, CodeType) and c.co_filename.startswith(\"= n then return 1 end return f(n-1)+f(n-1)+f(n-1) end +return f(8) +""", lambda: recursive_reference(2, 8), recursive_reference(2, 8), + "Three-branch recursion with reversed comparison"), + "record_layout_one": Case(""" +local function make(v: integer): table return {a=v} end +local total: integer=0 for i=1,5000 do local r: table=make(i) total=total+r.a end return total +""", lambda: record_reference(1), record_reference(1), "One-field record layout"), + "record_layout_three": Case(""" +local function make(v: integer): table return {a=v,b=v+1,c=v+2} end +local total: integer=0 for i=1,5000 do local r: table=make(i) total=total+r.a end return total +""", lambda: record_reference(3), record_reference(3), "Three-field record layout"), + "record_layout_five": Case(""" +local function make(v: integer): table return {e=v+4,c=v+2,a=v,b=v+1,d=v+3} end +local total: integer=0 for i=1,5000 do local r: table=make(i) total=total+r.a end return total +""", lambda: record_reference(5), record_reference(5), + "Five-field reordered record layout"), +} + +_base.CASES.update(CASES_040) +CASES = _base.CASES + +GENERALITY_GROUPS = { + "pure_scalar_expression_dag": ( + "scalar_dag_product", "scalar_dag_polynomial", "scalar_dag_split", + ), + "nested_numeric_region": ( + "nested_reduction_product", "nested_reduction_rows", "nested_reduction_modulo", + ), + "sparse_integer_boolean_region": ( + "sparse_slots", "sparse_visited", "sparse_occupancy", + ), + "scalar_call_entry": ( + "recursive_linear_general", "recursive_fibonacci_general", + "recursive_ternary_general", + ), + "fresh_record_layout": ( + "record_layout_one", "record_layout_three", "record_layout_five", + ), +} +assert all(len(set(cases)) >= 3 for cases in GENERALITY_GROUPS.values()) +FOCUSED_CASES = tuple(case for cases in GENERALITY_GROUPS.values() for case in cases) + + +if __name__ == "__main__": + if "--case" not in sys.argv: + for name in FOCUSED_CASES: + sys.argv.extend(("--case", name)) + main() diff --git a/benchmarks/vm_programs.py b/benchmarks/vm_programs.py index 2b05f16..51eb4f3 100644 --- a/benchmarks/vm_programs.py +++ b/benchmarks/vm_programs.py @@ -406,6 +406,256 @@ def validate(self, result: object) -> bool: return math.sqrt(vBv / vv) """ +_N_BODY = """ +local PI = 3.141592653589793 +local SOLAR_MASS = 4.0 * PI * PI +local DAYS_PER_YEAR = 365.24 +local bodies = { + {x = 0.0, y = 0.0, z = 0.0, vx = 0.0, vy = 0.0, vz = 0.0, mass = SOLAR_MASS}, + {x = 4.841431442464721, y = -1.1603200440274284, z = -0.10362204447112311, + vx = 0.001660076642744037 * DAYS_PER_YEAR, + vy = 0.007699011184197404 * DAYS_PER_YEAR, + vz = -0.0000690460016972063 * DAYS_PER_YEAR, + mass = 0.0009547919384243266 * SOLAR_MASS}, + {x = 8.34336671824458, y = 4.124798564124305, z = -0.4035234171143214, + vx = -0.002767425107268624 * DAYS_PER_YEAR, + vy = 0.004998528012349172 * DAYS_PER_YEAR, + vz = 0.00002304172975737639 * DAYS_PER_YEAR, + mass = 0.0002858859806661308 * SOLAR_MASS}, + {x = 12.894369562139131, y = -15.111151401698631, z = -0.22330757889265573, + vx = 0.002964601375647616 * DAYS_PER_YEAR, + vy = 0.0023784717395948095 * DAYS_PER_YEAR, + vz = -0.000029658956854023756 * DAYS_PER_YEAR, + mass = 0.00004366244043351563 * SOLAR_MASS}, + {x = 15.379697114850917, y = -25.919314609987964, z = 0.17925877295037118, + vx = 0.0026806777249038932 * DAYS_PER_YEAR, + vy = 0.001628241700382423 * DAYS_PER_YEAR, + vz = -0.00009515922545197159 * DAYS_PER_YEAR, + mass = 0.000051513890204661145 * SOLAR_MASS}, +} + +local px, py, pz = 0.0, 0.0, 0.0 +for i = 1, #bodies do + local b = bodies[i] + px = px + b.vx * b.mass + py = py + b.vy * b.mass + pz = pz + b.vz * b.mass +end +bodies[1].vx = -px / SOLAR_MASS +bodies[1].vy = -py / SOLAR_MASS +bodies[1].vz = -pz / SOLAR_MASS + +for step = 1, 1000 do + for i = 1, #bodies - 1 do + local bi = bodies[i] + for j = i + 1, #bodies do + local bj = bodies[j] + local dx = bi.x - bj.x + local dy = bi.y - bj.y + local dz = bi.z - bj.z + local distance2 = dx * dx + dy * dy + dz * dz + local magnitude = 0.01 / (distance2 * math.sqrt(distance2)) + bi.vx = bi.vx - dx * bj.mass * magnitude + bi.vy = bi.vy - dy * bj.mass * magnitude + bi.vz = bi.vz - dz * bj.mass * magnitude + bj.vx = bj.vx + dx * bi.mass * magnitude + bj.vy = bj.vy + dy * bi.mass * magnitude + bj.vz = bj.vz + dz * bi.mass * magnitude + end + end + for i = 1, #bodies do + local b = bodies[i] + b.x = b.x + 0.01 * b.vx + b.y = b.y + 0.01 * b.vy + b.z = b.z + 0.01 * b.vz + end +end + +local energy = 0.0 +for i = 1, #bodies do + local bi = bodies[i] + energy = energy + 0.5 * bi.mass * + (bi.vx * bi.vx + bi.vy * bi.vy + bi.vz * bi.vz) + for j = i + 1, #bodies do + local bj = bodies[j] + local dx = bi.x - bj.x + local dy = bi.y - bj.y + local dz = bi.z - bj.z + energy = energy - bi.mass * bj.mass / math.sqrt(dx * dx + dy * dy + dz * dz) + end +end +return energy +""" + +_MANDELBROT = """ +local size = 40 +local checksum = 0 +for y = 0, size - 1 do + local ci = 2.0 * y / size - 1.0 + local byte = 0 + local bits = 0 + for x = 0, size - 1 do + local cr = 2.0 * x / size - 1.5 + local zr, zi, tr, ti = 0.0, 0.0, 0.0, 0.0 + local iterations = 0 + while iterations < 50 and tr + ti <= 4.0 do + zi = 2.0 * zr * zi + ci + zr = tr - ti + cr + tr = zr * zr + ti = zi * zi + iterations = iterations + 1 + end + byte = byte * 2 + ((tr + ti <= 4.0) and 1 or 0) + bits = bits + 1 + if bits == 8 then + checksum = (checksum * 131 + byte) % 2147483647 + byte, bits = 0, 0 + end + end +end +return checksum +""" + +_FANNKUCH_REDUX = """ +local n = 7 +local perm1, count = {}, {} +for i = 1, n do perm1[i] = i - 1 end +local max_flips, checksum, sign, r = 0, 0, 1, n +while true do + while r ~= 1 do + count[r] = r + r = r - 1 + end + local perm = {} + for i = 1, n do perm[i] = perm1[i] end + local flips = 0 + local k = perm[1] + while k ~= 0 do + local left, right = 1, k + 1 + while left < right do + perm[left], perm[right] = perm[right], perm[left] + left, right = left + 1, right - 1 + end + flips = flips + 1 + k = perm[1] + end + checksum = checksum + sign * flips + if flips > max_flips then max_flips = flips end + + while true do + if r == n then return checksum * 100 + max_flips end + local first = perm1[1] + for i = 1, r do perm1[i] = perm1[i + 1] end + perm1[r + 1] = first + count[r + 1] = count[r + 1] - 1 + if count[r + 1] > 0 then + sign = -sign + break + end + r = r + 1 + end +end +""" + +_FASTA = """ +local alu = "GGCCGGGCGCGGTGGCTCACGCCTGTAATCCCAGCACTTTGGGAGGCCGAGGCGGGCGGATCACCTGAGGTCAGGAGTTCGAGACCAGCCTGGCCAACATGGTGAAACCCCGTCTCTACTAAAAATACAAAAATTAGCCGGGCGTGGTGGCGCGCGCCTGTAATCCCAGCTACTCGGGAGGCTGAGGCAGGAGAATCGCTTGAACCCGGGAGGCGGAGGTTGCAGTGAGCCGAGATCGCGCCACTGCACTCCAGCCTGGGCGACAGAGCGAGACTCCGTCTCAAAAA" +local iub = { + {char = "a", code = 97, p = 0.27}, {char = "c", code = 99, p = 0.12}, + {char = "g", code = 103, p = 0.12}, {char = "t", code = 116, p = 0.27}, + {char = "B", code = 66, p = 0.02}, {char = "D", code = 68, p = 0.02}, + {char = "H", code = 72, p = 0.02}, {char = "K", code = 75, p = 0.02}, + {char = "M", code = 77, p = 0.02}, {char = "N", code = 78, p = 0.02}, + {char = "R", code = 82, p = 0.02}, {char = "S", code = 83, p = 0.02}, + {char = "V", code = 86, p = 0.02}, {char = "W", code = 87, p = 0.02}, + {char = "Y", code = 89, p = 0.02}, +} +local homo = { + {char = "a", code = 97, p = 0.3029549426680}, + {char = "c", code = 99, p = 0.1979883004921}, + {char = "g", code = 103, p = 0.1975473066391}, + {char = "t", code = 116, p = 0.3015094502008}, +} +local function cumulative(dist) + local total = 0.0 + for i = 1, #dist do total = total + dist[i].p; dist[i].p = total end +end +cumulative(iub); cumulative(homo) +local seed = 42 +local function pick(dist) + seed = (seed * 3877 + 29573) % 139968 + local value = seed / 139968 + for i = 1, #dist do + if value < dist[i].p then return dist[i] end + end + return dist[#dist] +end +local chunks, checksum = {}, 0 +for i = 1, 1000 do + local at = ((i - 1) % #alu) + 1 + local ch = string.sub(alu, at, at) + chunks[#chunks + 1] = ch + checksum = checksum + string.byte(ch) +end +for i = 1, 1500 do + local item = pick(iub); chunks[#chunks + 1] = item.char; checksum = checksum + item.code +end +for i = 1, 2500 do + local item = pick(homo); chunks[#chunks + 1] = item.char; checksum = checksum + item.code +end +local output = table.concat(chunks) +return #output * 1000000 + checksum +""" + +_K_NUCLEOTIDE = """ +local motif = "GGTATTTTAATTTATAGTGGTAAGATATTAAGATAATATTTGGTGGTAGTTTTAATGTGTAA" +local sequence = string.rep(motif, 20) +local sizes = {1, 2, 3, 4, 6, 12, 18} +local checksum = 0 +for s = 1, #sizes do + local k = sizes[s] + local counts = {} + for i = 1, #sequence - k + 1 do + local key = string.sub(sequence, i, i + k - 1) + counts[key] = (counts[key] or 0) + 1 + end + checksum = (checksum * 131 + (counts[string.sub(sequence, 1, k)] or 0)) % 2147483647 + checksum = (checksum * 131 + (counts[string.sub(sequence, 7, 6 + k)] or 0)) % 2147483647 +end +return checksum +""" + +_REVERSE_COMPLEMENT = """ +local motif = "ACGTUMRWSYKVHDBN" +local sequence = string.rep(motif, 250) +local complement = { + A = "T", C = "G", G = "C", T = "A", U = "A", M = "K", R = "Y", + W = "W", S = "S", Y = "R", K = "M", V = "B", H = "D", D = "H", + B = "V", N = "N", +} +local output = {} +for i = #sequence, 1, -1 do + output[#output + 1] = complement[string.sub(sequence, i, i)] +end +local reversed = table.concat(output) +local checksum = 0 +for i = 1, #reversed do + checksum = (checksum + i * string.byte(reversed, i)) % 2147483647 +end +return checksum +""" + + +LANGUAGE_BENCHMARKS = ( + "n_body", + "mandelbrot", + "spectral_norm", + "fannkuch_redux", + "binary_trees", + "fasta", + "k_nucleotide", + "reverse_complement", +) + WORKLOADS: dict[str, Workload] = { "arithmetic": Workload(_ARITHMETIC, 450015000), @@ -492,7 +742,7 @@ def validate(self, result: object) -> bool: "binary_trees": Workload( _BINARY_TREES, 15330, - group="algorithm", + group="language", luapyre_source=_TYPED_BINARY_TREES, ), "table_mix": Workload( @@ -510,10 +760,25 @@ def validate(self, result: object) -> bool: "spectral_norm": Workload( _SPECTRAL_NORM, 1.274097380427096, - group="algorithm", + group="language", luapyre_source=_TYPED_SPECTRAL_NORM, abs_tolerance=1e-12, ), + "n_body": Workload( + _N_BODY, + -0.16908760523460614, + group="language", + abs_tolerance=1e-12, + ), + "mandelbrot": Workload(_MANDELBROT, 759728559, group="language"), + "fannkuch_redux": Workload(_FANNKUCH_REDUX, 22816, group="language"), + "fasta": Workload(_FASTA, 5000481113, group="language"), + "k_nucleotide": Workload(_K_NUCLEOTIDE, 1510947559, group="language"), + "reverse_complement": Workload( + _REVERSE_COMPLEMENT, + 607629500, + group="language", + ), } diff --git a/docs/language-benchmarks.md b/docs/language-benchmarks.md new file mode 100644 index 0000000..a03b488 --- /dev/null +++ b/docs/language-benchmarks.md @@ -0,0 +1,83 @@ +# Language benchmark corpus + +LuaPyre's permanent cross-runtime suite includes a bounded, deterministic +language-benchmark group inspired by the Computer Language Benchmarks Game. +It runs the same ordinary Lua source in LuaPyre, PUC Lua 5.5, and LuaJIT; +`native_headroom.py` also runs a reduced-contract Python implementation of the +same algorithm. + +These are microbenchmarks, not application-performance predictions. The +parameters are intentionally smaller than submission-sized Benchmarks Game +runs so the interpreter backend and routine CI remain practical. Every timed +result is checked. + +| Workload | Fixed parameter | Result checked | Primary pressure | +| --- | ---: | --- | --- | +| n-body | 1,000 steps, 5 bodies | final system energy | floating-point loops and record access | +| mandelbrot | 40 x 40, 50 iterations | packed-row checksum | scalar floating point, branches, and bit operations | +| spectral-norm | 30 x 30, 10 products | norm within `1e-12` | nested numeric loops and dense tables | +| fannkuch-redux | `n = 7` | checksum and maximum flips | permutation mutation and integer loops | +| binary-trees | 30 trees, depth 8 | aggregate node count | allocation, recursion, traversal, and GC | +| fasta | 5,000 bases | output length and byte checksum | LCG, weighted selection, strings, and concatenation | +| k-nucleotide | 1,300 bases, 7 k sizes | selected-count checksum | substring creation and hash-table updates | +| reverse-complement | 4,000 bases | position-weighted byte checksum | character mapping, reversal, and string assembly | + +## Deliberate adaptations + +The original output-oriented programs normally read or write streams. The +portable corpus instead constructs deterministic input in memory and validates +compact checksums. That keeps filesystem policy and terminal buffering out of +the language-runtime comparison while retaining the algorithms' string, +table, allocation, and numeric work. Mandelbrot is the scalar kernel; this +suite does not claim to measure threads or SIMD because neither is part of the +common Lua source. + +The Python column is an engineering lower bound rather than a semantically +equivalent runtime. It omits Lua fuel, metatables, debug state, and Lua-level +GC, but it does not substitute closed forms, memoization, NumPy, or another +algorithm. + +## Pidigits disposition + +Pidigits is not included in the portable three-runtime corpus. Its defining +workload uses arbitrary-precision integers, commonly through GMP, whereas Lua +5.5 and LuaPyre expose fixed-width signed 64-bit Lua integers. A small bounded +spigot would cease to measure the stated arbitrary-precision workload, and a +host bigint binding would measure unequal external-library interfaces. If a +portable bigint library is added to all compared runtimes later, pidigits can +be admitted as a separately labelled library benchmark. + +Run just this group with: + +```bash +python benchmarks/compare_runtimes.py --group language --require-all +``` + +For LuaPyre, native Lua 5.5, and Python lower-bound comparisons, select the same +names with repeated `--case` arguments to `benchmarks/native_headroom.py`. + +## Initial 0.40 baseline + +The initial run used three isolated processes, rotating implementation order, +with 7 warmups and 31 checked samples per implementation. Times are medians of +the three process medians. The working tree was based on commit `0bb256d`. + +| Workload | Python | LuaPyre | Native Lua 5.5 | Python lower bound | vs native | vs Python | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| n-body | 3.13 | 762.422 ms | 2.091 ms | 4.882 ms | 364.56x | 156.18x | +| n-body | 3.14 | 694.521 ms | 2.150 ms | 5.048 ms | 323.05x | 137.59x | +| Mandelbrot | 3.13 | 436.954 ms | 1.222 ms | 4.268 ms | 357.64x | 102.39x | +| Mandelbrot | 3.14 | 401.662 ms | 1.234 ms | 4.152 ms | 325.60x | 96.74x | +| fannkuch-redux | 3.13 | 711.137 ms | 3.187 ms | 6.001 ms | 223.12x | 118.51x | +| fannkuch-redux | 3.14 | 636.539 ms | 3.227 ms | 6.080 ms | 197.25x | 104.69x | +| FASTA | 3.13 | 180.129 ms | 0.589 ms | 1.016 ms | 305.69x | 177.22x | +| FASTA | 3.14 | 159.929 ms | 0.596 ms | 0.903 ms | 268.22x | 177.06x | +| k-nucleotide | 3.13 | 138.801 ms | 0.559 ms | 0.940 ms | 248.21x | 147.73x | +| k-nucleotide | 3.14 | 130.030 ms | 0.535 ms | 0.926 ms | 243.07x | 140.43x | +| reverse-complement | 3.13 | 103.117 ms | 0.442 ms | 0.394 ms | 233.37x | 261.73x | +| reverse-complement | 3.14 | 96.919 ms | 0.453 ms | 0.357 ms | 214.13x | 271.66x | + +Raw reports are retained in +[`language_benchmarks_040_313.json`](../benchmarks/results/language_benchmarks_040_313.json) +and +[`language_benchmarks_040_314.json`](../benchmarks/results/language_benchmarks_040_314.json). diff --git a/docs/performance-generality-policy.md b/docs/performance-generality-policy.md new file mode 100644 index 0000000..6f790ae --- /dev/null +++ b/docs/performance-generality-policy.md @@ -0,0 +1,33 @@ +# Performance optimization generality policy + +Every production fast path must be expressed as an instruction-, IR-, CFG-, +type-, alias-, range-, or effect-level transformation. A compiler must not +match an entire function with an `expected_ops` tuple and replace it with a +handwritten implementation of the recognized algorithm. Benchmark names, +function names, fixed field counts, and exact temporary-register allocation are +never admissible proof facts. + +Before an optimization is accepted it must: + +1. improve at least three structurally different programs that exercise the + same reusable compiler fact, by the preregistered A/B threshold (2% by + default, so timer noise does not count as acceleration); +2. retain admission and correctness under applicable semantic-preserving + mutations, including introduced temporaries, reversed comparisons, + reordered independent statements, changed record arity, and equivalent + expression trees; +3. preserve fuel, errors, metatables, debug hooks, stack limits, GC behavior, + allocation identity, and deoptimization boundaries; and +4. publish static fast-path admission counts for every benchmark workload. + +Benchmark aggregation classifies a path admitted by fewer than three distinct +workloads as `experimental`. Experimental paths may be measured and retained +behind ordinary semantic guards, but they are not evidence for a general +performance claim and must not justify release headline results. + +The 0.38 pure-base prefix remains because it skips only a proven local arm; its +matching is now def-use based for scalar comparisons and fresh nil records. +Further work should move the remaining table-prefix proof fully into CFG IR. +The 0.39 sparse integer/boolean representation remains provisional and is +covered by sieve, visited-set, slot-map, and occupancy-map shapes plus admission +telemetry. diff --git a/docs/performance-roadmap-0.40.md b/docs/performance-roadmap-0.40.md new file mode 100644 index 0000000..7e9c610 --- /dev/null +++ b/docs/performance-roadmap-0.40.md @@ -0,0 +1,36 @@ +# Performance roadmap: corrected 0.40 + +## Status + +The three whole-algorithm compilers have been removed. The production compiler +contains no full-function `expected_ops` signature or handwritten substitute +for matrix multiplication or binary-tree construction/traversal. + +## General IR path + +1. **Pure scalar expression DAGs.** Inline bounded, side-effect-free typed + callees from their instruction graph. Register copies and equivalent + temporary layouts do not define the optimization. +2. **Nested reductions.** Lower typed nested loop CFGs through the ordinary IR + region compiler. Continue replacing state dispatch with Python control flow + only from loop, dominance, type, range, and effect facts. +3. **Record layouts.** Treat fresh constructor writes using source-proven unique + keys and arbitrary layout size. Base-case analysis follows register value + provenance and accepts reordered independent loads and aliases. +4. **Recursive calls.** Reduce scalar entry, frame reset, pooling, and safe base + arms generically. Future non-tail frame elimination must be derived from call + and continuation IR; binary-tree shape is not a proof. +5. **Sparse sets.** Keep 0.39 provisionally, with unrelated sieve, visited-set, + slot-map, and occupancy-map admission coverage. + +## Release gates + +- Each claimed optimization must accelerate at least three structurally + different programs in isolated A/B measurements. +- Metamorphic source mutations must retain admission and results. +- Every benchmark report includes per-workload admission counts. +- Any path admitted by only one benchmark is labeled experimental. +- The full semantic suite and exact fuel comparisons remain mandatory. + +The detailed permanent rules are in +[performance-generality-policy.md](performance-generality-policy.md). diff --git a/docs/speed-0.39.md b/docs/speed-0.39.md new file mode 100644 index 0000000..4644b99 --- /dev/null +++ b/docs/speed-0.39.md @@ -0,0 +1,71 @@ +# LuaPyre 0.39 performance tranche + +0.39 implements the two dominant paths identified by profiling typed Sieve. +Both remain Python/AST based; no CPython bytecode or native runtime component is +generated. + +## Accepted changes + +1. **First-iteration compiled entry.** Once a numeric loop has a cached compiled + region, its next invocation enters that region immediately after `FORPREP` + normalizes the loop registers. The interpreter charges `FORPREP`; generated + code charges the body and backedge through the existing synchronized fuel + budget. Zero-trip loops, active debug hooks, cold loops and unsupported loop + shapes stay on the interpreter path. +2. **Sparse primitive-table regions.** A generated region may select direct + integer dictionary storage when the observed table is hash-only, has no + metatable or traversal state, and every access is compatible with an + integer-to-boolean map. Runtime identity, key and value guards remain at each + access. Generic mutation materializes tagged hash entries before applying + the ordinary Lua array/hash, deletion and iteration rules. +3. **Primitive barrier elimination.** Direct sparse writes skip the GC barrier + only when the runtime key is an exact integer and the value is an exact + boolean. Version increments and allocation accounting remain intact. + +## Measurements + +The baseline is merged 0.38 `162193820804911f012976aa85f72cb237375347`. +Each figure is the median of three process medians, with seven warmups and 31 +checked samples per process. The process was pinned to one CPU, Python hash seed +was fixed, and Python GC was disabled only during steady timing. Full samples +are retained in [`speed_039.json`](../benchmarks/results/speed_039.json); the +reusable harness is [`speed_039_ab.py`](../benchmarks/speed_039_ab.py). + +| Python | 0.38 | 0.39 | Improvement | Python lower bound | 0.39 / Python | +| --- | ---: | ---: | ---: | ---: | ---: | +| 3.13.15 | 22.906 ms | 8.161 ms | **64.4% faster** | 0.573 ms | 14.23× | +| 3.14.7 | 18.764 ms | 6.138 ms | **67.3% faster** | 0.512 ms | 11.98× | + +The CPython 3.14 profile for ten warmed executions fell from 2,930,560 calls +and 0.810 profiled seconds to 225,751 calls and 0.142 profiled seconds. +The steady generated path eliminates all 80,870 generic `rawset` calls, 49,990 +generic `rawget` calls and 80,870 table write-barrier calls seen in the former +ten-execution profile. The generated loop is now the clear remaining cost. + +## Correctness gates + +- Cached entry is tested against the interpreter at every fuel value through + completion, including the exact completion boundary. +- Zero-trip loops and active debug hooks do not enter compiled code. +- Sparse storage is tested for integer/integral-float reads, iteration, + versioning, generic materialization, dense growth, deletion and GC adoption. +- Generated code-shape coverage verifies direct dictionary access and the + absence of the table barrier from the primitive path. + +## Validation + +- All 606 Python tests pass on CPython 3.13.15 and 3.14.7. +- All 24 required unchanged official Lua 5.5.1 probes pass on both versions. +- Penlight passes 23/23, luatest 5/5, LuaCov scanner specs 24/24, and Are We + Fast Yet 11/11 on both versions. Established native-module and unsafe-I/O + exclusions remain unchanged. +- The 0.39.0a1 wheel builds successfully. + +## Remaining headroom + +After these changes, general VM dispatch and table methods are no longer in the +warmed Sieve profile. Most remaining time is inside the generated Python loop: +dynamic table/key guards, `isinstance` checks, dictionary lookups and allocation +accounting. Any next Sieve work should start with region-wide guard hoisting and +safe batching of allocation accounting, while preserving exact side exits and +memory-limit behavior. diff --git a/docs/speed-0.40.md b/docs/speed-0.40.md new file mode 100644 index 0000000..23fa86b --- /dev/null +++ b/docs/speed-0.40.md @@ -0,0 +1,39 @@ +# LuaPyre 0.40 corrective performance tranche + +0.40 retains the general GC/accounting, scalar-entry, record-construction, and +typed-IR work, but withdraws the earlier Binary Trees and Spectral Norm headline +results. Those results came from three whole-function algorithm replacements: +a dense matrix product, a binary-record builder, and a binary-record counter. +All three compilers have been removed. Matrix and tree programs now pass through +the ordinary typed CFG compiler. + +The retained and generalized paths are: + +- def-use analysis for integer base cases, insensitive to local temporaries and + reversed comparisons; +- field-count-independent fresh nil-record base cases, using value provenance + rather than fixed instruction positions; +- arbitrary straight-line pure scalar expression DAG inlining; +- typed CFG lowering of nested numeric regions and reductions; +- generic compiled scalar calls and pooled materialized frames; +- fresh-record setters, exact allocation accounting, cached empty-table sizing, + primitive barrier elision, and safe child ownership handling; and +- sparse integer-to-boolean regions, provisionally retained with unrelated + visited/slot/occupancy tests. + +`native_headroom.py` and `python_headroom.py` now report fast-path admissions per +workload. The native aggregate also lists every path's distinct workloads and +labels paths with fewer than three as `experimental`. See the permanent +[performance generality policy](performance-generality-policy.md). + +## Acceptance coverage + +Tests exercise three structurally different scalar DAGs, three nested +reductions, three record arities/layouts, and three unrelated sparse-set uses. +Metamorphic variants introduce temporaries, reverse comparisons, reorder record +fields, alter expression trees, and change record arity. Exact fuel boundaries, +invalid values, allocation identity, and accounting remain covered. + +The old 0.40 JSON files are retained only as historical evidence and must not be +quoted as performance of the corrected implementation. New release numbers +must be generated after the general paths satisfy the three-program speed gate. diff --git a/pyproject.toml b/pyproject.toml index a8c829f..260b861 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "luapyre" -version = "0.38.0a1" +version = "0.40.0a1" description = "A high-performance, sandboxed Lua 5.5 runtime for Python with optional gradual typing" readme = "README.md" requires-python = ">=3.13" diff --git a/src/luapyre/__init__.py b/src/luapyre/__init__.py index acfc259..3998bc4 100644 --- a/src/luapyre/__init__.py +++ b/src/luapyre/__init__.py @@ -27,4 +27,4 @@ "LuaTraceFrame", ] -__version__ = "0.38.0a1" +__version__ = "0.40.0a1" diff --git a/src/luapyre/ast_jit.py b/src/luapyre/ast_jit.py index 0d40cc8..903a124 100644 --- a/src/luapyre/ast_jit.py +++ b/src/luapyre/ast_jit.py @@ -12,7 +12,8 @@ from .opdispatch import _float_divide, _float_modulo, _float_power, _shift, _to_lua_string from .range_analysis import analyze_integer_ranges from .region_jit import RegionPythonJIT -from .table import LuaTable +from .table import LuaTable, _ABSENT, _NUM +from .typed_ir import _cfg_targets, _reads, _writes from .values import coerce_lua_integer, lua_equal, static_value_type, type_matches @@ -209,6 +210,132 @@ def _captured_registers(proto: Proto) -> set[int]: if desc.kind == "local" } + @staticmethod + def _sparse_table_registers( + frame, ir: IRLoop + ) -> dict[int, int]: + """Map safe table aliases to one sparse-integer storage owner. + + This is deliberately narrower than general escape analysis: the + observed table must already be hash-only integer-to-boolean storage, + and the region may only copy its aliases and perform compatible table + reads/writes. Runtime guards still protect every dynamic key/value. + """ + if any(item.ins.op is Op.CALL for item in ir.body.instructions): + return {} + access_regs = { + item.ins.b if item.ins.op is Op.GETTABLE else item.ins.a + for item in ir.body.instructions + if item.ins.op in (Op.GETTABLE, Op.SETTABLE) + } + groups: dict[int, set[int]] = {} + tables: dict[int, LuaTable] = {} + for reg in access_regs: + value = frame.regs[reg] + if not isinstance(value, LuaTable): + continue + identity = id(value) + groups.setdefault(identity, set()).add(reg) + tables[identity] = value + if not groups: + return {} + # Include only aliases connected to an access through explicit register + # copies. This finds the stable local behind compiler-generated table + # temporaries without treating unrelated stale registers as aliases. + for identity, aliases in groups.items(): + changed = True + while changed: + changed = False + for item in ir.body.instructions: + ins = item.ins + if ins.op not in (Op.MOVE, Op.LOCAL): + continue + if ins.a in aliases or ins.b in aliases: + for reg in (ins.a, ins.b): + if ( + reg not in aliases + and id(frame.regs[reg]) == identity + ): + aliases.add(reg) + changed = True + + result: dict[int, int] = {} + code = frame.proto.code + live = [set() for _ in range(len(code) + 1)] + changed = True + while changed: + changed = False + for pc in range(len(code) - 1, -1, -1): + outgoing: set[int] = set() + for target in _cfg_targets(code[pc], pc + 1): + if 0 <= target <= len(code): + outgoing.update(live[target]) + incoming = set(_reads(code[pc])) | ( + outgoing - set(_writes(code[pc])) + ) + if incoming != live[pc]: + live[pc] = incoming + changed = True + writes_a = frozenset( + { + Op.LOADK, Op.MOVE, Op.LOCAL, Op.NEWTABLE, Op.GETTABLE, + Op.ADD, Op.ADD_I, Op.ADD_F, Op.SUB, Op.SUB_I, Op.SUB_F, + Op.MUL, Op.MUL_I, Op.MUL_F, Op.DIV, Op.IDIV, Op.MOD, + Op.POW, Op.BAND, Op.BOR, Op.BXOR, Op.SHL, Op.SHR, + Op.CONCAT, Op.EQ, Op.LT, Op.LE, Op.NEG, Op.NOT, + Op.TOBOOL, Op.LEN, Op.BNOT, + } + ) + for identity, aliases in groups.items(): + table = tables[identity] + if ( + aliases.intersection(live[ir.exit_pc]) + or table.metatable is not None + or table.array + or table._deleted_successors is not None + or table._sparse_int is not None + or not table.hash + ): + continue + compatible = True + for token, (_key, value) in table.hash.items(): + if not ( + isinstance(token, tuple) + and len(token) == 2 + and token[0] is _NUM + and type(token[1]) is int + and token[1] >= 2 + and type(value) is bool + ): + compatible = False + break + if not compatible: + continue + for item in ir.body.instructions: + ins, op = item.ins, item.ins.op + if op is Op.GETTABLE and ins.b in aliases: + key = frame.regs[ins.c] + if type(key) is not int or key < 2: + compatible = False + break + elif op is Op.SETTABLE and ins.a in aliases: + key, value = frame.regs[ins.b], frame.regs[ins.c] + if type(key) is not int or key < 2 or type(value) is not bool: + compatible = False + break + elif op in (Op.MOVE, Op.LOCAL) and ins.a in aliases: + if ins.b not in aliases: + compatible = False + break + elif op in writes_a and ins.a in aliases: + compatible = False + break + if compatible: + owner = min(aliases) + for reg in aliases: + result[reg] = owner + return result + @staticmethod def _spill_lines(registers: tuple[int, ...], indent: str) -> list[str]: return [f"{indent}regs[{reg}] = _r{reg}" for reg in registers] @@ -433,6 +560,7 @@ def _inline_leaf_call( for index in range(ins.e): value = returned[index] if index < len(returned) else "None" lines.append(f"{indent}{self._reg(ins.a + index)} = {value}") + self.record_admission("pure_scalar_expression_dag") return lines, fn, child_cost def _emit_instruction( @@ -444,7 +572,10 @@ def _emit_instruction( captured: set[int], indent: str, expected_calls: dict[int, tuple[str, str]], + sparse_tables: dict[int, int] | None = None, ) -> list[str] | None: + if sparse_tables is None: + sparse_tables = {} ins, pc, op = item.ins, item.pc, item.ins.op a, b, c = self._reg(ins.a), self._reg(ins.b), self._reg(ins.c) out: list[str] = [] @@ -476,11 +607,49 @@ def deopt(condition: str) -> None: out.append(f"{indent}{a} = vm._new_table()") elif op is Op.GETTABLE: deopt(f"not isinstance({b}, _LuaTable) or {b}.metatable is not None") - out.extend([f"{indent}used += 1", f"{indent}{a} = {b}.rawget({c})"]) + owner = sparse_tables.get(ins.b) + if owner is None: + out.extend([f"{indent}used += 1", f"{indent}{a} = {b}.rawget({c})"]) + else: + sparse = f"_sparse_{owner}" + root = self._reg(owner) + item = f"_sparse_item_{pc}" + out.extend( + [ + f"{indent}used += 1", + f"{indent}if {sparse} is not None and {b} is {root} and type({c}) is int and {c} >= 2:", + f"{indent} {item} = {sparse}.get({c}, _ABSENT)", + f"{indent} {a} = None if {item} is _ABSENT else {item}", + f"{indent}else:", + f"{indent} {a} = {b}.rawget({c})", + ] + ) elif op is Op.SETTABLE: deopt(f"not isinstance({a}, _LuaTable) or {a}.metatable is not None") deopt(f"{b} is None or (type({b}) is float and _isnan({b}))") - out.extend([f"{indent}used += 1", f"{indent}{a}.rawset({b}, {c})"]) + owner = sparse_tables.get(ins.a) + if owner is None: + out.extend([f"{indent}used += 1", f"{indent}{a}.rawset({b}, {c})"]) + else: + sparse = f"_sparse_{owner}" + root = self._reg(owner) + collector = f"_sparse_gc_{pc}" + out.extend( + [ + f"{indent}used += 1", + f"{indent}if {sparse} is not None and {a} is {root} and type({b}) is int and {b} >= 2 and type({c}) is bool:", + f"{indent} {a}.version += 1", + f"{indent} if {b} not in {sparse}:", + f"{indent} {collector} = {a}._gc_owner", + f"{indent} if {collector} is not None:", + f"{indent} {collector}.account_bytes(32)", + f"{indent} {sparse}[{b}] = {c}", + f"{indent}else:", + f"{indent} {a}.rawset({b}, {c})", + f"{indent} if {a} is {root}:", + f"{indent} {sparse} = {root}._sparse_int", + ] + ) elif op in (Op.ADD_I, Op.SUB_I, Op.MUL_I): symbol = {Op.ADD_I: "+", Op.SUB_I: "-", Op.MUL_I: "*"}[op] out.append(f"{indent}used += 1") @@ -673,6 +842,7 @@ def _compile_ast_loop(self, frame, start_pc: int, backedge_pc: int): ir, blocks = lowered registers = self._used_registers(ir) captured = self._captured_registers(frame.proto) + sparse_tables = self._sparse_table_registers(frame, ir) if captured.intersection(registers): # Captured register replacement has subtle open-cell identity rules. # Leave those loops on the proven 0.14/Tier-0 paths for now. @@ -709,6 +879,19 @@ def state_for(pc: int) -> int: ] for reg in registers: lines.append(f" _r{reg} = regs[{reg}]") + for owner in sorted(set(sparse_tables.values())): + table = self._reg(owner) + sparse = f"_sparse_{owner}" + lines.extend( + [ + f" {sparse} = {table}._sparse_int", + f" if {sparse} is None and not {table}.array and not {table}.hash and {table}._deleted_successors is None:", + f" {sparse} = {{}}", + f" {table}._sparse_int = {sparse}", + f" if {table}._gc_owner is not None:", + f" {table}._gc_owner.account_bytes(_SPARSE_DICT_BYTES)", + ] + ) lines.append(" _state = 0") block_names = tuple(f"_jump_{index}" for index in range(len(blocks))) @@ -735,6 +918,7 @@ def state_for(pc: int) -> int: captured=captured, indent=indent, expected_calls=expected_calls, + sparse_tables=sparse_tables, ) if emitted is None: return None @@ -870,10 +1054,12 @@ def state_for(pc: int) -> int: "_INT_MAX": _INT_MAX, "_MASK64": _MASK64, "_SIGN64": _SIGN64, + "_SPARSE_DICT_BYTES": __import__("sys").getsizeof({}), "_TWO64": _TWO64, "_NUM_TYPES": (int, float), "_LuaRuntimeError": LuaRuntimeError, "_LuaTable": LuaTable, + "_ABSENT": _ABSENT, "_coerce_lua_integer": coerce_lua_integer, "_float_divide": _float_divide, "_float_modulo": _float_modulo, @@ -890,6 +1076,15 @@ def state_for(pc: int) -> int: exec(compile(tree, "", "exec"), namespace) raw_runner: FunctionType = namespace["_jit_ast_loop"] + if sparse_tables: + self.record_admission("sparse_integer_boolean_region") + if any( + item.ins.op in (Op.FORPREP, Op.FORLOOP, Op.JFORLOOP) + for block in blocks + for item in block.instructions + ): + self.record_admission("nested_numeric_region") + # The fixed value is statistical only. Exact fuel is the dynamic `used` # count returned by the runner. return CompiledLoop(ir, max(1, len(ir.body.instructions) + 1), raw_runner) diff --git a/src/luapyre/function_jit.py b/src/luapyre/function_jit.py index 2ebaa51..edf27f0 100644 --- a/src/luapyre/function_jit.py +++ b/src/luapyre/function_jit.py @@ -98,6 +98,103 @@ class PureReturnPrefix: result_constant: int | None returns_argument: bool instruction_cost: int + reversed_operands: bool = False + + +@dataclass(frozen=True, slots=True) +class PureTableNilReturnPrefix: + """A constant raw field nil-test whose taken arm returns one integer.""" + + key: bytes + token: object + result: int + instruction_cost: int + + +@dataclass(frozen=True, slots=True) +class FreshNilRecordPrefix: + """An integer base case returning a fresh record containing only nils.""" + + comparison: Op + constant: int + version: int + instruction_cost: int + reversed_operands: bool = False + + +@dataclass(frozen=True, slots=True) +class _LeadingIntegerBranch: + comparison: Op + constant: int + reversed_operands: bool + branch_pc: int + false_pc: int + values: dict[int, tuple[str, object]] + + +def _leading_integer_branch(proto: Proto) -> _LeadingIntegerBranch | None: + """Find a side-effect-free leading parameter/constant comparison. + + This operates on def-use facts rather than fixed instruction positions, so + compiler-generated temporaries, LOCAL copies, and reversed operands do not + change admission. Only a false-target branch is accepted; no observable + operation can be skipped before it. + """ + + values: dict[int, tuple[str, object]] = {0: ("argument", None)} + conditions: dict[int, tuple[Op, int, bool]] = {} + for pc, ins in enumerate(proto.code[:16]): + if ins.op is Op.LOADK: + value = proto.constants[ins.b] + values[ins.a] = ("constant", value) + elif ins.op in (Op.MOVE, Op.LOCAL) and ins.b in values: + values[ins.a] = values[ins.b] + condition = conditions.get(ins.b) + if condition is not None: + conditions[ins.a] = condition + elif ins.op in (Op.EQ, Op.LT, Op.LE): + left, right = values.get(ins.b), values.get(ins.c) + if left is None or right is None: + return None + if left[0] == "argument" and right[0] == "constant" and type(right[1]) is int: + conditions[ins.a] = (ins.op, right[1], False) + elif right[0] == "argument" and left[0] == "constant" and type(left[1]) is int: + conditions[ins.a] = (ins.op, left[1], True) + else: + return None + elif ins.op is Op.TOBOOL and ins.b in conditions: + conditions[ins.a] = conditions[ins.b] + elif ins.op is Op.JMPIFNOT and ins.b in conditions: + comparison, constant, reversed_operands = conditions[ins.b] + if not pc + 1 < ins.a <= len(proto.code): + return None + return _LeadingIntegerBranch( + comparison, + constant, + reversed_operands, + pc, + ins.a, + dict(values), + ) + else: + return None + return None + + +def _comparison_matches( + comparison: Op, + reversed_operands: bool, + argument: int, + constant: int, +) -> bool: + left, right = ( + (constant, argument) if reversed_operands else (argument, constant) + ) + if comparison is Op.EQ: + return left == right + if comparison is Op.LT: + return left < right + return left <= right def _pure_return_prefix(proto: Proto) -> PureReturnPrefix | None: @@ -113,28 +210,17 @@ def _pure_return_prefix(proto: Proto) -> PureReturnPrefix | None: or proto.param_types[0].name != "integer" or len(proto.return_types) != 1 or proto.return_types[0].name != "integer" - or len(proto.code) < 6 ): return None - load, compare, boolean, branch = proto.code[:4] - if not ( - load.op is Op.LOADK - and load.a != 0 - and type(proto.constants[load.b]) is int - and compare.op in (Op.EQ, Op.LT, Op.LE) - and compare.b == 0 - and compare.c == load.a - and compare.a != 0 - and boolean.op is Op.TOBOOL - and boolean.a == compare.a - and boolean.b == compare.a - and branch.op is Op.JMPIFNOT - and branch.b == boolean.a - and 5 < branch.a <= min(len(proto.code), 12) - ): + branch = _leading_integer_branch(proto) + if branch is None: return None - values: dict[int, tuple[bool, int | None]] = {0: (True, None)} - for pc in range(4, branch.a): + values: dict[int, tuple[bool, int | None]] = { + reg: (kind == "argument", value if type(value) is int else None) + for reg, (kind, value) in branch.values.items() + if kind == "argument" or type(value) is int + } + for pc in range(branch.branch_pc + 1, branch.false_pc): ins = proto.code[pc] if ins.op is Op.LOADK and type(proto.constants[ins.b]) is int: values[ins.a] = (False, proto.constants[ins.b]) @@ -143,50 +229,247 @@ def _pure_return_prefix(proto: Proto) -> PureReturnPrefix | None: elif ins.op is Op.RETURN and ins.b == 1 and ins.a in values: returns_argument, result_constant = values[ins.a] return PureReturnPrefix( - compare.op, - proto.constants[load.b], + branch.comparison, + branch.constant, result_constant, returns_argument, pc + 1, + branch.reversed_operands, ) else: return None return None -def _scalar_materialized_call_runner( - compiled: CompiledAstFunction, *, trusted_args: bool, arg_count: int -): - """Build a fixed-arity scalar entry for a materialized child frame.""" +def _pure_table_nil_return_prefix(proto: Proto) -> PureTableNilReturnPrefix | None: + """Find a pure leading constant-field nil branch using def-use facts.""" - if arg_count not in (1, 2): + if ( + proto.param_count != 1 + or proto.param_types[0].name != "table" + or tuple(item.name for item in proto.return_types) != ("integer",) + ): return None - proto = compiled.proto + values: dict[int, tuple[str, object]] = {0: ("table", None)} + conditions: dict[int, bytes] = {} + branch_pc = false_pc = -1 + key: bytes | None = None + for pc, ins in enumerate(proto.code[:20]): + if ins.op is Op.LOADK: + values[ins.a] = ("constant", proto.constants[ins.b]) + elif ins.op in (Op.MOVE, Op.LOCAL) and ins.b in values: + values[ins.a] = values[ins.b] + if ins.b in conditions: + conditions[ins.a] = conditions[ins.b] + elif ins.op is Op.GETTABLE: + receiver, field = values.get(ins.b), values.get(ins.c) + if receiver == ("table", None) and field is not None and isinstance(field[1], bytes): + values[ins.a] = ("field", field[1]) + else: + return None + elif ins.op is Op.EQ: + left, right = values.get(ins.b), values.get(ins.c) + if left is not None and left[0] == "field" and right == ("constant", None): + conditions[ins.a] = left[1] + elif right is not None and right[0] == "field" and left == ("constant", None): + conditions[ins.a] = right[1] + else: + return None + elif ins.op is Op.TOBOOL and ins.b in conditions: + conditions[ins.a] = conditions[ins.b] + elif ins.op is Op.JMPIFNOT and ins.b in conditions: + if not pc + 1 < ins.a <= len(proto.code): + return None + key = conditions[ins.b] + branch_pc, false_pc = pc, ins.a + break + else: + return None + if key is None: + return None + returns: dict[int, int] = {} + for pc in range(branch_pc + 1, false_pc): + ins = proto.code[pc] + if ins.op is Op.LOADK and type(proto.constants[ins.b]) is int: + returns[ins.a] = proto.constants[ins.b] + elif ins.op in (Op.MOVE, Op.LOCAL) and ins.b in returns: + returns[ins.a] = returns[ins.b] + elif ins.op is Op.RETURN and ins.b == 1 and ins.a in returns: + return PureTableNilReturnPrefix(key, _hash_key(key), returns[ins.a], pc + 1) + else: + return None + return None + + +def _fresh_nil_record_prefix(proto: Proto) -> FreshNilRecordPrefix | None: + """Recognize an allocating base arm such as ``return {a=nil,b=nil}``.""" + if ( - proto.param_count != arg_count - or len(proto.return_types) != 1 - or proto.return_types[0].name == "Any" + proto.param_count != 1 + or proto.param_types[0].name != "integer" + or tuple(item.name for item in proto.return_types) != ("table",) ): return None + branch = _leading_integer_branch(proto) + if branch is None: + return None + pc = branch.branch_pc + 1 + values = dict(branch.values) + aliases: set[int] = set() + fields = 0 + while pc < branch.false_pc: + ins = proto.code[pc] + if ins.op is Op.LOADK: + values[ins.a] = ("constant", proto.constants[ins.b]) + elif ins.op is Op.NEWTABLE: + if aliases: + return None + aliases.add(ins.a) + values[ins.a] = ("table", None) + elif ins.op in (Op.MOVE, Op.LOCAL) and ins.b in values: + values[ins.a] = values[ins.b] + if ins.b in aliases: + aliases.add(ins.a) + elif ( + ins.op is Op.SETTABLE + and ins.d + and ins.a in aliases + and values.get(ins.b, (None,))[0] == "constant" + and isinstance(values[ins.b][1], bytes) + and values.get(ins.c) == ("constant", None) + ): + fields += 1 + elif ins.op is Op.RETURN and ins.a in aliases and ins.b == 1: + if fields == 0: + return None + return FreshNilRecordPrefix( + branch.comparison, + branch.constant, + fields, + pc + 1, + branch.reversed_operands, + ) + else: + return None + pc += 1 + return None + + +def _table_nil_scalar_runner( + compiled: CompiledAstFunction, + *, + trusted_args: bool, + prefix: PureTableNilReturnPrefix, +): + proto = compiled.proto + expected = proto.param_types[0].name pool = compiled.frame_pool compiled_runner = compiled.runner env_reg = proto.env_reg - expected = tuple(item.name for item in proto.param_types) - prefix = _pure_return_prefix(proto) if arg_count == 1 else None - # A scalar recursive entry pays for a dynamic entry selection at every - # non-base call. It amortizes only when a meaningful share of calls can - # take the frame-free base arm; a single-chain recursion has one such call. - if prefix is not None and sum(ins.op is Op.CALL for ins in proto.code) < 2: - return None - def validate(index, value): - if not trusted_args and not type_matches(expected[index], value): + def run(vm, frames, closure, arg0, dest, want, budget, meter): + if len(frames) >= vm.max_frames: + raise LuaRuntimeError("stack overflow") + if not trusted_args and not type_matches(expected, arg0): raise LuaRuntimeError( - f"argument {index + 1}: expected {expected[index]}, " - f"got {static_value_type(value).name}" + f"argument 1: expected {expected}, " + f"got {static_value_type(arg0).name}" ) + if ( + not vm.debug_hooks_enabled + and budget - meter[0] >= prefix.instruction_cost + and isinstance(arg0, LuaTable) + and arg0.metatable is None + ): + item = arg0.hash.get(prefix.token, _ABSENT) + if item is _ABSENT or item[1] is None: + meter[0] += prefix.instruction_cost + vm.jit.function_executions += 1 + return _FUNC_RETURN, prefix.result + if pool: + child = pool.pop() + child.regs[0] = arg0 + if env_reg >= 0: + child.regs[env_reg] = closure.env + child.closure = closure + child.pc = 0 + child.return_reg = dest + child.return_want = want + if vm.debug_hooks_enabled: + child.hook_call_values = (arg0,) + else: + child = vm._acquire_compiled_frame( + compiled, closure, (arg0,), dest, want, validate_args=False, + ) + frames.append(child) + status, values = compiled_runner(vm, frames, child, budget, meter) + if status == _FUNC_RETURN: + if not frames or frames[-1] is not child: + raise RuntimeError("compiled function stack mismatch") + frames.pop() + vm.jit.function_executions += 1 + if len(pool) < 128 and len(pool) < vm.max_frames: + pool.append(child) + return _FUNC_RETURN, values[0] if values else None + vm.jit.function_suspends += 1 + return _FUNC_SUSPEND, None + + return run + + +def _fresh_nil_record_scalar_runner( + compiled: CompiledAstFunction, + *, + trusted_args: bool, + prefix: FreshNilRecordPrefix, +): + proto = compiled.proto + expected = proto.param_types[0].name + pool = compiled.frame_pool + compiled_runner = compiled.runner + env_reg = proto.env_reg - def finish(vm, frames, child, status, values): + def run(vm, frames, closure, arg0, dest, want, budget, meter): + if len(frames) >= vm.max_frames: + raise LuaRuntimeError("stack overflow") + if not trusted_args and not type_matches(expected, arg0): + raise LuaRuntimeError( + f"argument 1: expected {expected}, " + f"got {static_value_type(arg0).name}" + ) + if ( + not vm.debug_hooks_enabled + and budget - meter[0] >= prefix.instruction_cost + ): + matched = _comparison_matches( + prefix.comparison, + prefix.reversed_operands, + arg0, + prefix.constant, + ) + if matched: + result = vm._new_table(frames) + result.version = prefix.version + meter[0] += prefix.instruction_cost + vm.jit.function_executions += 1 + return _FUNC_RETURN, result + if pool: + child = pool.pop() + child.regs[0] = arg0 + if env_reg >= 0: + child.regs[env_reg] = closure.env + child.closure = closure + child.pc = 0 + child.return_reg = dest + child.return_want = want + if vm.debug_hooks_enabled: + child.hook_call_values = (arg0,) + else: + child = vm._acquire_compiled_frame( + compiled, closure, (arg0,), dest, want, validate_args=False, + ) + frames.append(child) + status, values = compiled_runner(vm, frames, child, budget, meter) if status == _FUNC_RETURN: if not frames or frames[-1] is not child: raise RuntimeError("compiled function stack mismatch") @@ -198,20 +481,58 @@ def finish(vm, frames, child, status, values): vm.jit.function_suspends += 1 return _FUNC_SUSPEND, None + return run + + +def _scalar_materialized_call_runner( + compiled: CompiledAstFunction, *, trusted_args: bool, arg_count: int +): + """Build a fixed-arity scalar entry for a materialized child frame.""" + + if arg_count not in (1, 2): + return None + proto = compiled.proto + if ( + proto.param_count != arg_count + or len(proto.return_types) != 1 + or proto.return_types[0].name == "Any" + ): + return None + if arg_count == 1: + table_prefix = _pure_table_nil_return_prefix(proto) + if table_prefix is not None: + return _table_nil_scalar_runner( + compiled, trusted_args=trusted_args, prefix=table_prefix + ) + fresh_prefix = _fresh_nil_record_prefix(proto) + if fresh_prefix is not None: + return _fresh_nil_record_scalar_runner( + compiled, trusted_args=trusted_args, prefix=fresh_prefix + ) + pool = compiled.frame_pool + compiled_runner = compiled.runner + env_reg = proto.env_reg + expected = tuple(item.name for item in proto.param_types) + prefix = _pure_return_prefix(proto) if arg_count == 1 else None if arg_count == 1: def run(vm, frames, closure, arg0, dest, want, budget, meter): if len(frames) >= vm.max_frames: raise LuaRuntimeError("stack overflow") - validate(0, arg0) + if not trusted_args and not type_matches(expected[0], arg0): + raise LuaRuntimeError( + f"argument 1: expected {expected[0]}, " + f"got {static_value_type(arg0).name}" + ) if ( prefix is not None and not vm.debug_hooks_enabled and budget - meter[0] >= prefix.instruction_cost ): - matched = ( - arg0 == prefix.constant if prefix.comparison is Op.EQ - else arg0 < prefix.constant if prefix.comparison is Op.LT - else arg0 <= prefix.constant + matched = _comparison_matches( + prefix.comparison, + prefix.reversed_operands, + arg0, + prefix.constant, ) if matched: meter[0] += prefix.instruction_cost @@ -238,15 +559,32 @@ def run(vm, frames, closure, arg0, dest, want, budget, meter): ) frames.append(child) status, values = compiled_runner(vm, frames, child, budget, meter) - return finish(vm, frames, child, status, values) + if status == _FUNC_RETURN: + if not frames or frames[-1] is not child: + raise RuntimeError("compiled function stack mismatch") + frames.pop() + vm.jit.function_executions += 1 + if len(pool) < 128 and len(pool) < vm.max_frames: + pool.append(child) + return _FUNC_RETURN, values[0] if values else None + vm.jit.function_suspends += 1 + return _FUNC_SUSPEND, None return run def run(vm, frames, closure, arg0, arg1, dest, want, budget, meter): if len(frames) >= vm.max_frames: raise LuaRuntimeError("stack overflow") - validate(0, arg0) - validate(1, arg1) + if not trusted_args and not type_matches(expected[0], arg0): + raise LuaRuntimeError( + f"argument 1: expected {expected[0]}, " + f"got {static_value_type(arg0).name}" + ) + if not trusted_args and not type_matches(expected[1], arg1): + raise LuaRuntimeError( + f"argument 2: expected {expected[1]}, " + f"got {static_value_type(arg1).name}" + ) if pool: child = pool.pop() child.regs[0] = arg0 @@ -266,7 +604,16 @@ def run(vm, frames, closure, arg0, arg1, dest, want, budget, meter): ) frames.append(child) status, values = compiled_runner(vm, frames, child, budget, meter) - return finish(vm, frames, child, status, values) + if status == _FUNC_RETURN: + if not frames or frames[-1] is not child: + raise RuntimeError("compiled function stack mismatch") + frames.pop() + vm.jit.function_executions += 1 + if len(pool) < 128 and len(pool) < vm.max_frames: + pool.append(child) + return _FUNC_RETURN, values[0] if values else None + vm.jit.function_suspends += 1 + return _FUNC_SUSPEND, None return run @@ -469,6 +816,13 @@ def get_compiled_function(self, closure: Closure) -> CompiledAstFunction | None: self._function_cache[ident] = (proto, compiled) if compiled is not None: self.function_compiles += 1 + self.record_admission("typed_function") + if any(ins.op is Op.NEWTABLE for ins in proto.code) and any( + ins.op is Op.SETTABLE and ins.d for ins in proto.code + ): + self.record_admission("fresh_record_layout") + if any(ins.op is Op.CALL for ins in proto.code): + self.record_admission("generic_compiled_calls") return compiled def maybe_function(self, closure: Closure) -> CompiledAstFunction | None: @@ -524,6 +878,22 @@ def get_call_entry( arg_count=arg_count, trusted_args=trusted_args, ) + if runner is not None: + self.record_admission("virtual_frame") + scalar_runner = None if runner is not None else _scalar_materialized_call_runner( + compiled, + trusted_args=trusted_args, + arg_count=arg_count if arg_count is not None else -1, + ) + if scalar_runner is not None: + self.record_admission("scalar_call_entry") + if arg_count == 1: + if _pure_table_nil_return_prefix(proto) is not None: + self.record_admission("table_nil_base") + elif _fresh_nil_record_prefix(proto) is not None: + self.record_admission("fresh_record_base") + elif _pure_return_prefix(proto) is not None: + self.record_admission("pure_scalar_base") entry = CompiledCallEntry( proto, runner if runner is not None else _materialized_call_runner( @@ -532,11 +902,7 @@ def get_call_entry( runner is not None, arg_count, trusted_args, - None if runner is not None else _scalar_materialized_call_runner( - compiled, - trusted_args=trusted_args, - arg_count=arg_count if arg_count is not None else -1, - ), + scalar_runner, ) self._call_entry_cache[key] = (proto, entry) return entry @@ -550,13 +916,14 @@ def _compile_ast_function(self, proto: Proto) -> CompiledAstFunction | None: if proto.children: return None + compiling_closure = self._compiling_closures.get(id(proto)) + registers = tuple(range(max(1, proto.register_count))) ranges = analyze_integer_ranges(proto) block_index = {block.start: index for index, block in enumerate(blocks)} typed_plan = TypedIRCompiler(proto).compile( tuple(block.instructions for block in blocks) ) - compiling_closure = self._compiling_closures.get(id(proto)) # Continuation liveness is used only at successful compiled-child # boundaries. All side exits retain the full diagnostic spill. @@ -927,14 +1294,30 @@ def emit_inline_leaf( ): token_name = f"_key_token_{pc}" constant_tokens[token_name] = token - setter = ( - "rawset_fresh_prehashed" if ins.d - else "rawset_prehashed" - ) - lines.extend([ - f"{indent}used += 1", - f"{indent}{a}.{setter}({b}, {token_name}, {c})", - ]) + if ins.d: + # The source compiler has proved this constructor + # key unique. Keep the exact table/GC operations + # visible to CPython so the record hot path avoids + # a setter call and specializes its attributes. + collector = f"_collector_{pc}" + lines.extend([ + f"{indent}used += 1", + f"{indent}if {a}._sparse_int is not None:", + f"{indent} {a}._materialize_sparse_int()", + f"{indent}{a}.version += 1", + f"{indent}{collector} = {a}._gc_owner", + f"{indent}if {collector} is not None and type({c}) not in _NON_GC_TYPES:", + f"{indent} {collector}.table_write_barrier({a}, {b}, {c})", + f"{indent}if {c} is not None:", + f"{indent} if {collector} is not None:", + f"{indent} {collector}.account_bytes(32)", + f"{indent} {a}.hash[{token_name}] = ({b}, {c})", + ]) + else: + lines.extend([ + f"{indent}used += 1", + f"{indent}{a}.rawset_prehashed({b}, {token_name}, {c})", + ]) else: lines.extend([f"{indent}used += 1", f"{indent}{a}.rawset({b}, {c})"]) elif op in (Op.ADD_I, Op.SUB_I, Op.MUL_I): @@ -1187,6 +1570,7 @@ def emit_inline_leaf( "_SIGN64": 1 << 63, "_TWO64": 1 << 64, "_NUM_TYPES": (int, float), + "_NON_GC_TYPES": (type(None), bool, int, float, bytes, str), "_Closure": Closure, "_LuaRuntimeError": LuaRuntimeError, "_LuaTable": LuaTable, diff --git a/src/luapyre/gc.py b/src/luapyre/gc.py index e370064..b4b731d 100644 --- a/src/luapyre/gc.py +++ b/src/luapyre/gc.py @@ -331,6 +331,15 @@ def __init__(self, vm): self._finalizable_ids: set[int] = set() self._finalized_ids: set[int] = set() self._liveness_cache: dict[int, tuple[Proto, tuple[frozenset[int], ...]]] = {} + # Every Lua table is registered while its Python containers are still + # empty. Their shallow sizes are stable for one CPython process, so do + # the three getsizeof calls once rather than once per table allocation. + empty_table = LuaTable() + self._empty_table_size = ( + sys.getsizeof(empty_table) + + sys.getsizeof(empty_table.array) + + sys.getsizeof(empty_table.hash) + ) @staticmethod def _object_size(value) -> int: @@ -382,7 +391,18 @@ def _register_fresh(self, value) -> None: value._gc_owner = self value._gc_age = _GC_NEW self.stats.allocations += 1 - self.account_bytes(self._object_size(value)) + if ( + type(value) is LuaTable + and not value.array + and not value.hash + and value._sparse_int is None + and value._deleted_successors is None + and value._reserved_bytes == 0 + ): + size = self._empty_table_size + else: + size = self._object_size(value) + self.account_bytes(size) def adopt(self, value) -> None: """Attach a newly reachable Lua object graph to this collector.""" @@ -450,7 +470,11 @@ def table_write_barrier(self, parent: LuaTable, key, value) -> None: for child in (key, value): if type(child) in _NON_COLLECTABLE_TYPES: continue - self.adopt(child) + # A constructor commonly stores freshly registered child tables. + # They already belong to this collector, so a graph walk cannot + # discover anything or change ownership. + if getattr(child, "_gc_owner", None) is not self: + self.adopt(child) if ( parent_is_old and self._is_collectable(child) diff --git a/src/luapyre/jit.py b/src/luapyre/jit.py index 76284c1..87909ee 100644 --- a/src/luapyre/jit.py +++ b/src/luapyre/jit.py @@ -51,6 +51,7 @@ class IRLoop: class JITStats: loop_compiles: int = 0 loop_executions: int = 0 + loop_entry_executions: int = 0 loop_iterations: int = 0 leaf_compiles: int = 0 leaf_executions: int = 0 @@ -81,6 +82,7 @@ class JITStats: coroutine_yields: int = 0 coroutine_instructions: int = 0 coroutine_deopts: int = 0 + fast_path_admissions: dict[str, int] = field(default_factory=dict) @dataclass(frozen=True, slots=True) @@ -155,6 +157,11 @@ def record_compile_failure(self, reason: str) -> None: reasons = self.stats.compile_failure_reasons reasons[reason] = reasons.get(reason, 0) + 1 + def record_admission(self, path: str) -> None: + """Record one static fast-path admission outside generated hot code.""" + admissions = self.stats.fast_path_admissions + admissions[path] = admissions.get(path, 0) + 1 + @staticmethod def _profile_binary(regs, ins: Ins) -> str | None: left, right = regs[ins.b], regs[ins.c] diff --git a/src/luapyre/jitvm.py b/src/luapyre/jitvm.py index 0011b7b..3ce98a9 100644 --- a/src/luapyre/jitvm.py +++ b/src/luapyre/jitvm.py @@ -33,7 +33,34 @@ def _jit_forloop(vm, frames, frame, ins, regs, constants): vm._jit_backedge(frame, backedge_pc, key, state) +def _jit_forprep(vm, frames, frame, ins, regs, constants): + """Enter an already-compiled numeric loop before its first body.""" + source_pc = frame.pc - 1 + result = OPCODE_HANDLERS[Op.FORPREP]( + vm, frames, frame, ins, regs, constants + ) + start_pc = source_pc + 1 + if ( + result is not None + or frame.pc != start_pc + or not vm.jit.enabled + or vm.hooks_active() + ): + return result + backedge_pc = vm.jit._loop_map(frame.proto).get(start_pc) + if backedge_pc is None: + return result + backedge = frame.proto.code[backedge_pc] + if backedge.op is not Op.JFORLOOP: + return result + state = vm._jit_loop_states.get(id(backedge)) + if isinstance(state, CompiledLoop): + vm._jit_enter_loop(frame, state) + return result + + _JIT_OPCODE_HANDLERS = dict(OPCODE_HANDLERS) +_JIT_OPCODE_HANDLERS[Op.FORPREP] = _jit_forprep _JIT_OPCODE_HANDLERS[Op.JFORLOOP] = _jit_forloop @@ -331,6 +358,17 @@ def _jit_backedge(self, frame, backedge_pc: int, key: int, state) -> None: self._jit_loop_states[key] = False self.jit.stats.retired_regions += 1 + def _jit_enter_loop(self, frame, compiled: CompiledLoop) -> None: + """Run a cached loop after FORPREP has paid and normalized its state.""" + used, progressed = compiled.runner(self, frame, self._jit_budget()) + if used: + self._jit_consume(used) + self.jit.stats.loop_executions += 1 + self.jit.stats.loop_entry_executions += 1 + self.jit.stats.loop_iterations += used // compiled.cost_per_iteration + if not progressed and used == 0 and frame.pc != compiled.ir.start_pc: + self.jit.stats.deopts += 1 + def _invoke(self, frames, parent, fn, args, dest, want, tail=False): # Synchronous stdlib callbacks intentionally stay on the interpreter: # their local fuel accounting and visible-frame prefix are specialized @@ -405,6 +443,7 @@ def wrapped(vm, active_frames, frame, ins, regs, constants): handlers[Op.CALL] = synced(handlers[Op.CALL], leaf_allowed=True) handlers[Op.CALLV] = synced(handlers[Op.CALLV], leaf_allowed=True) + handlers[Op.FORPREP] = synced(handlers[Op.FORPREP]) handlers[Op.JFORLOOP] = synced(handlers[Op.JFORLOOP]) if proto.jit_fully_typed: for branch_op in (Op.JMP, Op.JMPIF, Op.JMPIFNOT, Op.JMPIFNIL): diff --git a/src/luapyre/structured_jit.py b/src/luapyre/structured_jit.py index 5ed1578..fc511a0 100644 --- a/src/luapyre/structured_jit.py +++ b/src/luapyre/structured_jit.py @@ -1079,4 +1079,6 @@ def r(index: int) -> str: ast.fix_missing_locations(tree) exec(compile(tree, "", "exec"), namespace) runner: FunctionType = namespace["_jit_structured_loop"] + if calls: + self.record_admission("pure_scalar_expression_dag") return CompiledLoop(ir, iteration_cost, runner) diff --git a/src/luapyre/table.py b/src/luapyre/table.py index 7f17dd8..bfac3ad 100644 --- a/src/luapyre/table.py +++ b/src/luapyre/table.py @@ -9,6 +9,7 @@ _STR = object() _OBJ = object() _ITERATION_STATE = object() +_PRIMITIVE_GC_TYPES = (type(None), bool, int, float, bytes, str) def _hash_key(key): @@ -34,7 +35,7 @@ class LuaTable: __slots__ = ( "array", "hash", "metatable", "version", "_gc_owner", "_gc_age", - "_deleted_successors", "_reserved_bytes", + "_deleted_successors", "_reserved_bytes", "_sparse_int", ) def __init__(self): @@ -49,6 +50,18 @@ def __init__(self): # and dense-array allocations. self._deleted_successors: dict[object, object | None] | None = None self._reserved_bytes = 0 + # Typed generated regions may keep a proven hash-only integer/primitive + # table in an untagged dictionary. Generic mutation materializes it + # before applying the full Lua array/hash transition rules. + self._sparse_int: dict[int, object] | None = None + + def _materialize_sparse_int(self) -> None: + sparse = self._sparse_int + if sparse is None: + return + for key, value in sparse.items(): + self.hash[(_NUM, key)] = (key, value) + self._sparse_int = None def _ensure_iteration_index(self) -> None: metadata = self._deleted_successors @@ -111,6 +124,9 @@ def rawget(self, key): idx = key - 1 if idx < len(self.array): return self.array[idx] + sparse = self._sparse_int + if sparse is not None: + return sparse.get(key) item = self.hash.get((_NUM, key), _ABSENT) return None if item is _ABSENT else item[1] h = _hash_key(key) @@ -120,6 +136,9 @@ def rawget(self, key): idx = h[1] - 1 if idx < len(self.array): return self.array[idx] + sparse = self._sparse_int + if sparse is not None: + return sparse.get(h[1]) item = self.hash.get(h, _ABSENT) return None if item is _ABSENT else item[1] @@ -129,6 +148,9 @@ def rawhas(self, key) -> bool: idx = key - 1 if idx < len(self.array) and self.array[idx] is not None: return True + sparse = self._sparse_int + if sparse is not None: + return key in sparse return (_NUM, key) in self.hash h = _hash_key(key) if h is None: @@ -137,9 +159,13 @@ def rawhas(self, key) -> bool: idx = h[1] - 1 if idx < len(self.array) and self.array[idx] is not None: return True + sparse = self._sparse_int + if sparse is not None: + return h[1] in sparse return h in self.hash def rawset(self, key, value): + self._materialize_sparse_int() if type(key) is float and math.isnan(key): raise LuaRuntimeError("table index is NaN") h = _hash_key(key) @@ -192,6 +218,7 @@ def rawset(self, key, value): def rawset_prehashed(self, key, token, value): """Set a compiler-validated non-array constant key.""" + self._materialize_sparse_int() present = False if value is None: present = token in self.hash @@ -218,9 +245,18 @@ def rawset_prehashed(self, key, token, value): def rawset_fresh_prehashed(self, key, token, value): """Set a proven-new constructor field without deletion bookkeeping.""" + # Fresh record constructors normally have no sparse region. Keep the + # materialization fallback for callers that use this public internal + # helper on an already-specialized table, without paying a Python call + # for the overwhelmingly common empty-table case. + if self._sparse_int is not None: + self._materialize_sparse_int() self.version += 1 collector = self._gc_owner - if collector is not None: + if collector is not None and not ( + type(key) in _PRIMITIVE_GC_TYPES + and type(value) in _PRIMITIVE_GC_TYPES + ): collector.table_write_barrier(self, key, value) if value is not None: if collector is not None: @@ -252,6 +288,9 @@ def items(self): yield i, value for original, value in self.hash.values(): yield original, value + sparse = self._sparse_int + if sparse is not None: + yield from sparse.items() @classmethod def from_sequence(cls, values): diff --git a/src/luapyre/typed_ir_jit.py b/src/luapyre/typed_ir_jit.py index 61a7c42..c73693f 100644 --- a/src/luapyre/typed_ir_jit.py +++ b/src/luapyre/typed_ir_jit.py @@ -89,6 +89,7 @@ def _emit_typed_ir_instruction( captured: set[int], indent: str, expected_calls: dict[int, tuple[str, str]], + sparse_tables: dict[int, int] | None = None, ) -> list[str] | None: site = plan.instruction(item.pc) if site is None: @@ -100,6 +101,7 @@ def _emit_typed_ir_instruction( captured=captured, indent=indent, expected_calls=expected_calls, + sparse_tables=sparse_tables, ) ins, pc, op = item.ins, item.pc, item.ins.op @@ -268,6 +270,7 @@ def _emit_typed_ir_instruction( captured=captured, indent=indent, expected_calls=expected_calls, + sparse_tables=sparse_tables, ) def _compile_ast_loop(self, frame, start_pc: int, backedge_pc: int): @@ -288,6 +291,7 @@ def _compile_ast_loop(self, frame, start_pc: int, backedge_pc: int): register_set.add(item.ins.a) registers = tuple(sorted(register_set)) captured = self._captured_registers(frame.proto) + sparse_tables = self._sparse_table_registers(frame, ir) if captured.intersection(registers): return None @@ -337,6 +341,19 @@ def state_for(pc: int) -> int: ] for reg in registers: lines.append(f" _r{reg} = regs[{reg}]") + for owner in sorted(set(sparse_tables.values())): + table = self._reg(owner) + sparse = f"_sparse_{owner}" + lines.extend( + [ + f" {sparse} = {table}._sparse_int", + f" if {sparse} is None and not {table}.array and not {table}.hash and {table}._deleted_successors is None:", + f" {sparse} = {{}}", + f" {table}._sparse_int = {sparse}", + f" if {table}._gc_owner is not None:", + f" {table}._gc_owner.account_bytes(_SPARSE_DICT_BYTES)", + ] + ) # Loop-invariant global reads are an IR optimization, not a Python AST # peephole. The backend performs the proven access exactly once per JIT @@ -417,6 +434,7 @@ def state_for(pc: int) -> int: captured=captured, indent=indent, expected_calls=expected_calls, + sparse_tables=sparse_tables, ) if emitted is None: return None @@ -528,6 +546,7 @@ def state_for(pc: int) -> int: "_INT_MAX": _INT_MAX, "_MASK64": _MASK64, "_SIGN64": _SIGN64, + "_SPARSE_DICT_BYTES": __import__("sys").getsizeof({}), "_TWO64": _TWO64, "_NUM_TYPES": (int, float), "_LuaRuntimeError": LuaRuntimeError, @@ -547,4 +566,12 @@ def state_for(pc: int) -> int: } exec(compile(tree, "", "exec"), namespace) raw_runner: FunctionType = namespace["_jit_ast_loop"] + if sparse_tables: + self.record_admission("sparse_integer_boolean_region") + if any( + item.ins.op in (Op.FORPREP, Op.FORLOOP, Op.JFORLOOP) + for block in blocks + for item in block.instructions + ): + self.record_admission("nested_numeric_region") return CompiledLoop(ir, max(1, len(ir.body.instructions) + 1), raw_runner) diff --git a/tests/test_language_benchmarks.py b/tests/test_language_benchmarks.py new file mode 100644 index 0000000..213f564 --- /dev/null +++ b/tests/test_language_benchmarks.py @@ -0,0 +1,68 @@ +"""Correctness gates for the portable language-benchmark corpus.""" +from __future__ import annotations + +from pathlib import Path +import sys + +import pytest + +from luapyre import LuaRuntime + + +BENCHMARKS = Path(__file__).resolve().parents[1] / "benchmarks" +if str(BENCHMARKS) not in sys.path: + sys.path.insert(0, str(BENCHMARKS)) + +from python_headroom import REFERENCES # noqa: E402 +from vm_programs import LANGUAGE_BENCHMARKS, WORKLOADS # noqa: E402 + + +ADDED_CASES = ( + "n_body", + "mandelbrot", + "fannkuch_redux", + "fasta", + "k_nucleotide", + "reverse_complement", +) + + +def test_language_benchmark_inventory_is_explicit_and_portable(): + assert LANGUAGE_BENCHMARKS == ( + "n_body", + "mandelbrot", + "spectral_norm", + "fannkuch_redux", + "binary_trees", + "fasta", + "k_nucleotide", + "reverse_complement", + ) + assert "pidigits" not in WORKLOADS + for name in LANGUAGE_BENCHMARKS: + workload = WORKLOADS[name] + assert workload.group == "language" + assert "io." not in workload.source + + +@pytest.mark.parametrize("name", LANGUAGE_BENCHMARKS) +def test_python_lower_bound_matches_checked_result(name): + assert WORKLOADS[name].validate(REFERENCES[name]()) + + +@pytest.mark.parametrize("name", ADDED_CASES) +def test_added_lua_source_matches_checked_result_in_luapyre(name): + workload = WORKLOADS[name] + runtime = LuaRuntime(fuel=20_000_000) + result = runtime.execute(workload.source_for("luapyre")) + assert workload.validate(result) + + +@pytest.mark.parametrize("name", ADDED_CASES) +def test_added_lua_source_matches_checked_result_in_native_lua55(name): + lua55 = pytest.importorskip("lupa.lua55") + runtime = lua55.LuaRuntime(unpack_returned_tuples=True) + assert runtime.eval("_VERSION") == "Lua 5.5" + workload = WORKLOADS[name] + result = runtime.execute("return function()\n" + workload.source + "\nend")() + assert workload.validate(result) diff --git a/tests/test_performance_039.py b/tests/test_performance_039.py new file mode 100644 index 0000000..501ab19 --- /dev/null +++ b/tests/test_performance_039.py @@ -0,0 +1,199 @@ +"""Correctness and code-shape coverage for the 0.39 sieve paths.""" +from __future__ import annotations + +import pytest + +from luapyre import LuaQuotaError, LuaRuntime +from luapyre.jit import CompiledLoop +from luapyre.table import LuaTable + + +SMALL_SIEVE = """-- luapyre: typed +local composite: table = {} +local count: integer = 0 +for p = 2, 40 do + if not composite[p] then + count = count + 1 + local multiple: integer = p * p + while multiple <= 40 do + composite[multiple] = true + multiple = multiple + p + end + end +end +return count +""" + + +def _outcome(runtime, proto, fuel): + try: + return ("return", runtime.vm.run(proto, fuel=fuel)) + except LuaQuotaError: + return ("quota", None) + + +def test_cached_numeric_loop_enters_from_forprep(): + runtime = LuaRuntime(jit_threshold=1, fuel=2_000_000) + proto = runtime.compile(SMALL_SIEVE) + assert runtime.vm.run(proto) == 12 + before = runtime.jit_stats.loop_entry_executions + assert runtime.vm.run(proto) == 12 + assert runtime.jit_stats.loop_entry_executions == before + 1 + + +def test_forprep_entry_preserves_every_fuel_boundary(): + hot = LuaRuntime(jit_threshold=1) + hot_proto = hot.compile(SMALL_SIEVE) + for _ in range(3): + assert hot.vm.run(hot_proto, fuel=20_000) == 12 + cold = LuaRuntime(jit=False) + cold_proto = cold.compile(SMALL_SIEVE) + first_complete = next( + fuel for fuel in range(2_000) + if _outcome(cold, cold_proto, fuel)[0] == "return" + ) + for fuel in range(first_complete + 3): + assert _outcome(hot, hot_proto, fuel) == _outcome( + cold, cold_proto, fuel + ), fuel + + +def test_forprep_entry_skips_zero_trip_loop(): + runtime = LuaRuntime(jit_threshold=1) + runtime.globals.rawset(b"limit", 8) + proto = runtime.compile("""-- luapyre: typed +global limit: integer +local total: integer = 0 +for i = 1, limit do total = total + i end +return total +""") + for _ in range(2): + assert runtime.vm.run(proto) == 36 + before = runtime.jit_stats.loop_entry_executions + runtime.globals.rawset(b"limit", 0) + assert runtime.vm.run(proto) == 0 + assert runtime.jit_stats.loop_entry_executions == before + + +def test_forprep_entry_preserves_negative_step_loop(): + runtime = LuaRuntime(jit_threshold=1) + proto = runtime.compile( + "local total = 0; for i = 9, 1, -2 do total = total + i end; return total" + ) + assert runtime.vm.run(proto) == 25 + before = runtime.jit_stats.loop_entry_executions + assert runtime.vm.run(proto) == 25 + assert runtime.jit_stats.loop_entry_executions == before + 1 + + +def test_sparse_integer_boolean_region_materializes_for_generic_semantics(): + runtime = LuaRuntime(jit_threshold=1) + table = runtime.vm._new_table() + table._sparse_int = {4: True, 6: True, 8: True} + assert table.rawget(4) is True + assert table.rawget(4.0) is True + assert table.rawget(5) is None + assert dict(table.items())[4] is True + assert table.next_item() is not None + + version = table.version + child = LuaTable() + table.rawset(100, child) + assert table._sparse_int is None + assert table.version == version + 1 + assert table.rawget(4) is True and table.rawget(100) is child + assert child._gc_owner is runtime.vm.gc + + table.rawset(1, b"dense") + assert table.rawlen() == 1 + assert table.rawget(1) == b"dense" + table.rawset(4, None) + assert table.rawget(4) is None + + +def test_sparse_codegen_uses_direct_dictionary_and_primitive_barrier_path(): + runtime = LuaRuntime(jit_threshold=1) + proto = runtime.compile(SMALL_SIEVE) + for _ in range(3): + assert runtime.vm.run(proto) == 12 + runner = next( + state.runner + for state in runtime.vm._jit_loop_states.values() + if isinstance(state, CompiledLoop) and "_sparse_int" in state.runner.__code__.co_names + ) + names = set(runner.__code__.co_names) + assert "_sparse_int" in names + assert "get" in names + assert "table_write_barrier" not in names + + +@pytest.mark.parametrize( + "program,expected", + [ + ( + "local slots: table={} local hits: integer=0 " + "for i=1,60 do local k: integer=i*3 slots[k]=true " + "if slots[k] then hits=hits+1 end end return hits", + 60, + ), + ( + "local visited: table={} local first: integer=0 " + "for i=1,50 do local k: integer=i*5 if not visited[k] then " + "visited[k]=true first=first+1 end end return first", + 50, + ), + ( + "local occupied: table={} local total: integer=0 " + "for i=1,40 do local k: integer=i*7 occupied[k]=true " + "if occupied[k] then total=total+i end end return total", + 820, + ), + ], +) +def test_sparse_storage_admits_three_unrelated_regions(program, expected): + runtime = LuaRuntime(jit_threshold=1) + source = "-- luapyre: typed\n" + program + assert runtime.execute(source) == expected + assert runtime.execute(source) == expected + assert runtime.jit_stats.fast_path_admissions[ + "sparse_integer_boolean_region" + ] >= 1 + + +def test_sparse_region_requires_table_to_be_dead_after_loop(): + runtime = LuaRuntime(jit_threshold=1) + proto = runtime.compile("""-- luapyre: typed +local marked: table = {} +for p = 2, 20 do + local multiple: integer = p * p + while multiple <= 80 do + marked[multiple] = true + multiple = multiple + p + end +end +return marked[4], marked[6], marked[7] +""") + for _ in range(3): + assert runtime.vm.run(proto) == (True, True, None) + outer = next( + state + for state in runtime.vm._jit_loop_states.values() + if isinstance(state, CompiledLoop) and state.ir.start_pc == 9 + ) + assert "_sparse_int" not in outer.runner.__code__.co_names + + +def test_active_debug_hook_keeps_forprep_on_interpreter_path(): + runtime = LuaRuntime(jit_threshold=1, debug_hooks=True) + proto = runtime.compile("""-- luapyre: typed +global debug: table +local hits: integer = 0 +debug.sethook(function(): nil hits = hits + 1 end, "l") +local total: integer = 0 +for i = 1, 20 do total = total + i end +debug.sethook() +return total, hits > 0 +""") + for _ in range(2): + assert runtime.vm.run(proto) == (210, True) + assert runtime.jit_stats.loop_entry_executions == 0 diff --git a/tests/test_performance_040.py b/tests/test_performance_040.py new file mode 100644 index 0000000..6f11974 --- /dev/null +++ b/tests/test_performance_040.py @@ -0,0 +1,235 @@ +"""Generality, semantics, and policy coverage for the corrected 0.40 tranche.""" +from __future__ import annotations + +import inspect + +import pytest + +from luapyre import LuaQuotaError, LuaRuntime, LuaRuntimeError, LuaTable +from luapyre.function_jit import ( + TypedFunctionJITMixin, + _FUNC_RETURN, + _fresh_nil_record_prefix, + _pure_return_prefix, + _pure_table_nil_return_prefix, +) + + +TREE_FUNCTIONS = """-- luapyre: typed +local function make(depth: integer): table + if depth == 0 then return {left=nil, right=nil} end + return {left=make(depth-1), right=make(depth-1)} +end +local function check(node: table): integer + if node.left == nil then return 1 end + return 1 + check(node.left) + check(node.right) +end +return make, check +""" + +MATRIX_FUNCTION = """-- luapyre: typed +local function weight(i: integer, j: integer): float + return 1.0 / (((i+j)*(i+j+1)/2.0)+i+1.0) +end +local function transform(x: table, n: integer): table + local out: table = {} + for i = 1, n do + local total: float = 0.0 + for j = 1, n do + total = total + weight(i-1, j-1) * x[j] + end + out[i] = total + end + return out +end +return transform +""" + + +def _outcome(runtime, proto, fuel): + try: + return "return", runtime.vm.run(proto, fuel=fuel) + except LuaQuotaError: + return "quota", None + + +def test_whole_algorithm_replacement_compilers_are_permanently_absent(): + source = inspect.getsource(TypedFunctionJITMixin) + forbidden = ( + "expected_ops", + "_compile_dense_matrix_product", + "_compile_fresh_binary_record_builder", + "_compile_binary_record_counter", + "luapyre-dense-matrix-function", + "luapyre-fresh-binary-record-function", + "luapyre-binary-record-counter-function", + ) + assert all(token not in source for token in forbidden) + + +def test_tree_and_matrix_use_generic_typed_cfg_compiler(): + runtime = LuaRuntime(jit_threshold=1) + make, check = runtime.execute(TREE_FUNCTIONS) + transform = runtime.execute_python(MATRIX_FUNCTION) + for closure in (make, check, transform.raw): + compiled = runtime.vm.jit.get_compiled_function(closure) + assert compiled is not None + assert compiled.runner.__code__.co_filename == "" + assert transform([1.0, 1.0], 2) == pytest.approx( + [1.5, 0.5333333333333333] + ) + + +@pytest.mark.parametrize( + "body,reversed_operands", + [ + ("if n == 0 then return 1 end return n", False), + ("local z: integer=0; if n == z then local r: integer=1; return r end return n", False), + ("if 0 == n then return 1 end return n", True), + ], +) +def test_scalar_base_analysis_survives_semantic_mutations(body, reversed_operands): + runtime = LuaRuntime(jit_threshold=1) + closure = runtime.execute( + "-- luapyre: typed\nlocal function f(n: integer): integer " + + body + + " end return f" + ) + prefix = _pure_return_prefix(closure.proto) + assert prefix is not None and prefix.reversed_operands is reversed_operands + + +@pytest.mark.parametrize( + "function,argument,expected", + [ + ("local function f(n: integer): integer if n <= 0 then return 0 end return n+f(n-1) end", 8, 36), + ("local function f(n: integer): integer if n < 2 then return n end return f(n-1)+f(n-2) end", 8, 21), + ("local function f(n: integer): integer if 0 >= n then return 1 end return f(n-1)+f(n-1)+f(n-1) end", 4, 81), + ], +) +def test_generic_recursive_scalar_entry_admits_three_call_graphs(function, argument, expected): + runtime = LuaRuntime(jit_threshold=1) + closure = runtime.execute("-- luapyre: typed\n" + function + " return f") + compiled = runtime.vm.jit.get_compiled_function(closure) + entry = runtime.vm.jit.get_call_entry(compiled, arg_count=1, trusted_args=True) + assert entry.scalar_runner is not None + assert runtime.vm.call_sync(closure, (argument,)) == (expected,) + assert runtime.jit_stats.fast_path_admissions["scalar_call_entry"] >= 1 + assert runtime.jit_stats.fast_path_admissions["pure_scalar_base"] >= 1 + + +@pytest.mark.parametrize("fields", ["a=nil", "a=nil,b=nil,c=nil", "z=nil,a=nil,q=nil,b=nil"]) +def test_fresh_record_layout_is_independent_of_field_count_and_order(fields): + runtime = LuaRuntime(jit_threshold=1) + closure = runtime.execute( + "-- luapyre: typed\nlocal function f(n: integer): table " + f"if 0 == n then return {{{fields}}} end " + "return {left=f(n-1),right=f(n-1)} end return f" + ) + prefix = _fresh_nil_record_prefix(closure.proto) + assert prefix is not None + assert prefix.version == fields.count("=") + compiled = runtime.vm.jit.get_compiled_function(closure) + assert runtime.jit_stats.fast_path_admissions["fresh_record_layout"] >= 1 + entry = runtime.vm.jit.get_call_entry(compiled, arg_count=1, trusted_args=True) + status, first = entry.scalar_runner(runtime.vm, [], closure, 0, 0, 1, 100, [0]) + status2, second = entry.scalar_runner(runtime.vm, [], closure, 0, 0, 1, 100, [0]) + assert status == status2 == _FUNC_RETURN + assert isinstance(first, LuaTable) and first is not second and first.hash == {} + + +@pytest.mark.parametrize( + "body", + [ + "if node.left == nil then return 1 end return 2", + "local absent=nil; if absent == node.left then local one: integer=1; return one end return 2", + ], +) +def test_table_base_analysis_survives_temporaries_and_reversed_equality(body): + runtime = LuaRuntime(jit_threshold=1) + closure = runtime.execute( + "-- luapyre: typed\nlocal function f(node: table): integer " + + body + + " end return f" + ) + assert _pure_table_nil_return_prefix(closure.proto) is not None + + +@pytest.mark.parametrize( + "program,expected", + [ + ("local function f(x: integer): integer return (x+2)*(x-1) end", 3040), + ("local function f(x: integer): integer local y: integer=x*x; return y+3*x+1 end", 3520), + ("local function f(x: integer): integer local a: integer=x+1; local b: integer=x-1; return a*b+x end", 3060), + ], +) +def test_pure_scalar_expression_dags_admit_three_different_shapes(program, expected): + source = "-- luapyre: typed\n" + program + ( + " local total: integer=0 for i=1,20 do total=total+f(i) end return total" + ) + runtime = LuaRuntime(jit_threshold=1) + assert runtime.execute(source) == expected + assert runtime.execute(source) == expected + assert runtime.jit_stats.fast_path_admissions["pure_scalar_expression_dag"] >= 1 + + +def test_equivalent_expression_tree_and_temporary_keep_dag_admission(): + bodies = ( + "return (x+1)+(x*2)", + "local doubled: integer=x*2; local incremented: integer=x+1; return doubled+incremented", + ) + results = [] + for body in bodies: + runtime = LuaRuntime(jit_threshold=1) + source = ( + "-- luapyre: typed\nlocal function f(x: integer): integer " + + body + + " end local total: integer=0 for i=1,20 do total=total+f(i) end return total" + ) + results.append(runtime.execute(source)) + runtime.execute(source) + assert runtime.jit_stats.fast_path_admissions["pure_scalar_expression_dag"] >= 1 + assert results[0] == results[1] + + +@pytest.mark.parametrize( + "program,expected", + [ + ("local s: integer=0 for i=1,5 do for j=1,4 do s=s+i*j end end return s", 150), + ("local s: integer=1 for i=1,4 do local row: integer=0 for j=1,3 do row=row+i+j end s=s+row end return s", 55), + ("local s: float=0.0 for i=1,4 do for j=1,i do s=s+1.0/(i+j) end end return s", 2.3345238095238092), + ], +) +def test_nested_reduction_ir_admits_three_different_shapes(program, expected): + runtime = LuaRuntime(jit_threshold=1) + source = "-- luapyre: typed\n" + program + assert runtime.execute(source) == pytest.approx(expected) + assert runtime.execute(source) == pytest.approx(expected) + assert runtime.jit_stats.fast_path_admissions["nested_numeric_region"] >= 1 + + +def test_generic_paths_preserve_every_fuel_boundary(): + source = "-- luapyre: typed\n" + TREE_FUNCTIONS.split("\n", 1)[1] + "\nreturn check(make(2))" + source = source.replace("return make, check\n", "") + hot = LuaRuntime(jit_threshold=1) + cold = LuaRuntime(jit=False) + hot_proto, cold_proto = hot.compile(source), cold.compile(source) + for _ in range(3): + assert hot.vm.run(hot_proto, fuel=10_000) == 7 + first_complete = next(f for f in range(1_000) if _outcome(cold, cold_proto, f)[0] == "return") + for fuel in range(first_complete + 3): + assert _outcome(hot, hot_proto, fuel) == _outcome(cold, cold_proto, fuel) + + +def test_generic_matrix_path_preserves_invalid_argument_errors(): + runtime = LuaRuntime(jit_threshold=1) + function = runtime.execute_python(MATRIX_FUNCTION) + with pytest.raises(LuaRuntimeError): + function([1.0, b"not-a-number"], 2) + + +def test_empty_table_size_cache_keeps_exact_accounting(): + runtime = LuaRuntime() + before = runtime.vm.gc.stats.allocated_bytes + table = runtime.vm._new_table() + assert runtime.vm.gc.stats.allocated_bytes - before == runtime.vm.gc._object_size(table)