Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 6 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@

LuaPyre is a clean-slate Lua runtime written in Python. It targets **Lua 5.5.1** semantics, a sandbox-first embedding model, and optional gradual type annotations that feed runtime optimization without creating a second language/runtime.

**Python 3.13+** · **current pre-alpha: 0.37.0a1**
**Python 3.13+** · **current pre-alpha: 0.38.0a1**

LuaPyre implements Lua 5.5.1 language semantics for its supported sandboxed embedding profile. The runtime is built around a register VM and explicit Lua frames, with a guarded tiered JIT that specializes proven hot paths and deoptimizes back to the same interpreter.

Expand Down Expand Up @@ -189,6 +189,11 @@ code-shape checks, rejected experiments, and correctness gates.
strict floats and two integers. See the [0.37 performance record](docs/speed-0.37.md)
for paired Python 3.13/3.14 results.

0.38 adds frame-free pure recursive base cases, scalar compiled-call entries,
proved dense primitive-write regions, and cheaper fresh record construction.
See the [0.38 performance record](docs/speed-0.38.md) and
[implementation roadmap](docs/performance-roadmap-0.38.md).

The [0.36 performance roadmap](docs/performance-roadmap-0.36.md) adds fresh
Python-headroom measurements, 11 focused speed probes, and the next ordered
work on scalar entry, cross-block facts, nested regions, tables, and recursion.
Expand Down
10 changes: 9 additions & 1 deletion benchmarks/inspect_codegen.py
Original file line number Diff line number Diff line change
Expand Up @@ -111,7 +111,7 @@ def main():
parser.add_argument("--executions", type=int, default=3)
parser.add_argument("--top", type=int, default=3)
parser.add_argument(
"--suite", choices=("headroom", "036", "037"), default="headroom"
"--suite", choices=("headroom", "036", "037", "038"), default="headroom"
)
parser.add_argument("--case", action="append")
parser.add_argument("--include-source", action="store_true")
Expand All @@ -133,6 +133,14 @@ def prepare_case(name):

cases = CASES

def prepare_case(name):
runtime, run, _reference = prepare_probe(name)
return runtime, run, lambda value: validate(CASES[name], value)
elif args.suite == "038":
from speed_038_ab import CASES, prepare as prepare_probe, validate

cases = CASES

def prepare_case(name):
runtime, run, _reference = prepare_probe(name)
return runtime, run, lambda value: validate(CASES[name], value)
Expand Down
24 changes: 24 additions & 0 deletions benchmarks/results/speed_038.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
{
"schema_version": 1,
"baseline_revision": "10abe6061f7c8937e23ab07fc6b9ab5c6cb8d168",
"candidate": "codex/038-performance-tranche worktree",
"method": "Three paired baseline/candidate processes per Python; alternating Lua/Python timing order; seven warmups and 31 checked samples; fixed affinity; median of process medians",
"summary": [
{"python":"3.13.15","case":"recursive_balanced","baseline_ms":1.658433,"candidate_ms":1.293257,"python_ms":0.042823,"improvement_percent":22.02},
{"python":"3.14.7","case":"recursive_balanced","baseline_ms":1.335740,"candidate_ms":1.086365,"python_ms":0.032768,"improvement_percent":18.67},
{"python":"3.13.15","case":"record_continuations","baseline_ms":7.089099,"candidate_ms":6.703220,"python_ms":0.140196,"improvement_percent":5.44},
{"python":"3.14.7","case":"record_continuations","baseline_ms":5.950264,"candidate_ms":5.758843,"python_ms":0.136160,"improvement_percent":3.22},
{"python":"3.13.15","case":"dense_alias_write","baseline_ms":3.342436,"candidate_ms":3.289446,"python_ms":0.125945,"improvement_percent":1.59},
{"python":"3.14.7","case":"dense_alias_write","baseline_ms":2.691030,"candidate_ms":2.580728,"python_ms":0.105475,"improvement_percent":4.10}
],
"controls": [
{"python":"3.13.15","case":"recursive_linear","change_percent":-0.85},
{"python":"3.14.7","case":"recursive_linear","change_percent":-1.54},
{"python":"3.13.15","case":"scalar_internal_binary","change_percent":-0.32},
{"python":"3.14.7","case":"scalar_internal_binary","change_percent":0.54},
{"python":"3.13.15","case":"dense_read","change_percent":-0.08},
{"python":"3.14.7","case":"dense_read","change_percent":0.60},
{"python":"3.13.15","case":"table_mix_nested","change_percent":-0.42},
{"python":"3.14.7","case":"table_mix_nested","change_percent":-2.69}
]
}
84 changes: 84 additions & 0 deletions benchmarks/speed_038_ab.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,84 @@
"""Measure the Python/AST-only 0.38 performance tranche.

The inherited 0.36/0.37 probes remain controls. The focused default cases
cover pure recursive bases, scalar materialized calls, nested dense regions,
and record construction. Run unchanged against baseline and candidate trees.
"""
from __future__ import annotations

import sys

import speed_037_ab as _base
from speed_037_ab import * # noqa: F401,F403


def scalar_internal_binary():
total = 0
for value in range(1, 2001):
scratch = [value]
total += scratch[0] + value + 1
return total


def table_mix_nested():
values = [None]
for value in range(1, 6001):
values.append((value * 17) % 1009)
total = 0
for round_ in range(1, 5):
for index in range(1, 6001):
value = (values[index] * 33 + index + round_) % 10007
values[index] = value
total = (total + value) % 1000000007
return total


CASES_038 = {
"scalar_internal_binary": Case("""
local function combine(left: integer, right: integer): integer
local scratch: table = {left}
return scratch[1] + right
end
local total: integer = 0
for i = 1, 2000 do total = total + combine(i, i + 1) end
return total
""", scalar_internal_binary, 4004000,
"Two-scalar internal calls through a materialized child"),
"table_mix_nested": Case("""
local n: integer = 6000
local values: table = {}
for i = 1, n do values[i] = (i * 17) % 1009 end
local total: integer = 0
for round = 1, 4 do
for i = 1, n do
local value: integer = values[i]
value = (value * 33 + i + round) % 10007
values[i] = value
total = (total + value) % 1000000007
end
end
return total
""", table_mix_nested, 119882313,
"Nested-loop dense extent and primitive-write proof"),
}

_base.CASES.update(CASES_038)
CASES = _base.CASES


FOCUSED_CASES = (
"recursive_linear",
"recursive_balanced",
"scalar_internal_binary",
"record_continuations",
"dense_read",
"dense_alias_write",
"table_mix_nested",
)


if __name__ == "__main__":
if "--case" not in sys.argv:
for name in FOCUSED_CASES:
sys.argv.extend(("--case", name))
main()
46 changes: 46 additions & 0 deletions docs/performance-roadmap-0.38.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,46 @@
# Performance roadmap: 0.38

## Goal

Implement the next Python/AST-only tranche from the 0.37 roadmap in this
order: pure base-case entry, scalar internal calls, dense-region proofs, and
record-construction improvements. Direct CPython bytecode generation and
native repository runtime code remain out of scope.

## Implemented path

1. Recognize a narrow leading typed-integer comparison whose taken arm returns
only its argument or an integer constant. Charge its exact instruction cost
and return the scalar without allocating a child `Frame`. Stack, type, fuel,
debug-hook, and fallback behavior stay live.
2. Give fixed one/two-argument, one-result materialized compiled calls a scalar
entry, avoiding the parent's argument tuple and result-sequence extraction.
Reuse the existing real-frame pool for non-base calls and retain the generic
tuple path for unsupported shapes and suspension.
3. Prove stable table aliases, positive dense index ranges, and primitive
writes for a straight-line numeric-loop region. Guard table identity,
metatable absence, and dense extent once, bind the backing array once, then
emit direct array operations. Reject calls, holes, growth, nil/collectable
writes, unknown aliases, and unproved bounds.
4. Preserve constant-key facts across whole-function call continuations. Mark
unique literal constructor keys as proven fresh, lazily allocate iteration
metadata, and use a fresh pre-hashed setter that retains version increments,
GC barriers, and allocation accounting.

## Acceptance and rejected forms

Every change uses generated Python source/AST and CPython's own adaptive
specialization. The reusable `benchmarks/speed_038_ab.py` harness runs the same
checked workload against baseline and candidate checkouts in three paired
processes per CPython version, with alternating Lua/Python timing order, seven
warmups, and 31 samples.

Scalar recursion is admitted only when at least two recursive call sites make
base-case frame elision frequent enough to amortize entry selection; linear
recursion stayed on the prior path. Read-only dense loops keep their established
checked-array lowering because one-time region binding did not beat it. A
broader modulo-heavy nested-region experiment was also removed after regressing
table mix. These exclusions are part of the cost model, not semantic limits.

See [the 0.38 performance record](speed-0.38.md) for measurements and
validation.
64 changes: 64 additions & 0 deletions docs/speed-0.38.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,64 @@
# LuaPyre 0.38 performance tranche

0.38 implements four narrow Python/AST-only paths from the
[0.38 roadmap](performance-roadmap-0.38.md). It does not generate CPython
bytecode directly and adds no native runtime component.

## Accepted changes

1. **Pure recursive bases avoid frames.** A certified typed integer base arm
charges exact fuel and returns its scalar directly. Debug hooks, inadequate
fuel, invalid types, and other branches use the existing materialized path.
2. **Scalar compiled-call entries.** Fixed unary/binary, one-result compiled
calls pass Python scalars without the parent's argument tuple or generic
result-sequence extraction. Real frames and the child's internal return
representation remain for non-base execution and suspension recovery.
3. **Dense primitive-write regions.** Stable aliases and positive bounded keys
are proved once. The generated loop guards dense extent and metatable state,
binds the array once, and writes primitive values directly while incrementing
the table version.
4. **Cheaper record construction.** Constant facts now survive recursive call
continuations. Unique literal fields use a fresh pre-hashed setter; ordinary
tables allocate iteration/deletion metadata only if traversal requires it.
GC barriers and allocation accounting remain intact.

## Measurements

The baseline is merged `main` at `10abe6061f7c8937e23ab07fc6b9ab5c6cb8d168`.
Each result is the median of three paired process medians with alternating
Lua/Python timing order, seven warmups, and 31 checked samples. Python GC is
disabled only during steady timing; Lua GC remains active. Full medians are retained in
[`speed_038.json`](../benchmarks/results/speed_038.json).

| Target | Python | 0.37 | 0.38 | Change | 0.38 / Python |
| --- | --- | ---: | ---: | ---: | ---: |
| Balanced recursive calls | 3.13.15 | 1.658 ms | 1.293 ms | **22.0% faster** | 30.20× |
| Balanced recursive calls | 3.14.7 | 1.336 ms | 1.086 ms | **18.7% faster** | 33.15× |
| Recursive record continuations | 3.13.15 | 7.089 ms | 6.703 ms | **5.4% faster** | 47.81× |
| Recursive record continuations | 3.14.7 | 5.950 ms | 5.759 ms | **3.2% faster** | 42.29× |
| Dense aliased primitive writes | 3.13.15 | 3.342 ms | 3.289 ms | **1.6% faster** | 26.12× |
| Dense aliased primitive writes | 3.14.7 | 2.691 ms | 2.581 ms | **4.1% faster** | 24.47× |

Controls remained below the 5% regression gate: linear recursion was -0.9%
and -1.5%, materialized binary scalar calls -0.3%/+0.5%, read-only dense loops
-0.1%/+0.6%, and modulo-heavy table mix -0.4%/-2.7% on 3.13/3.14 respectively.

## Validation

- All 598 Python tests pass on CPython 3.13.15 and 3.14.7.
- Deterministic additions cover frame-free base entry, all tested fuel
boundaries, unary/binary scalar entry, dense code shape, fresh nil writes,
GC adoption/accounting, iteration, and duplicate/dynamic constructor keys.
- All 24 required unchanged official Lua 5.5.1 probes pass on both versions.
- Penlight passes 23/23, luatest 5/5, LuaCov scanner specs 24/24, and Are We
Fast Yet 11/11 on both versions. Established native-module and unsafe-I/O
exclusions remain unchanged.

## Deferred work

This is base-case frame elimination, not general lazy activation recovery.
Effectful, mutually recursive, yielding, and single-chain calls retain real
frames. Dense proof excludes collectable/nil writes, calls, holes, growth,
metatables, and unproved aliases. General record-layout replacement remains
deferred because identity, weak tables, finalizers, and iteration order are
observable Lua semantics.
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "hatchling.build"

[project]
name = "luapyre"
version = "0.37.0a1"
version = "0.38.0a1"
description = "A high-performance, sandboxed Lua 5.5 runtime for Python with optional gradual typing"
readme = "README.md"
requires-python = ">=3.13"
Expand Down
2 changes: 1 addition & 1 deletion src/luapyre/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,4 +27,4 @@
"LuaTraceFrame",
]

__version__ = "0.37.0a1"
__version__ = "0.38.0a1"
21 changes: 18 additions & 3 deletions src/luapyre/compiler.py
Original file line number Diff line number Diff line change
Expand Up @@ -891,6 +891,8 @@ def _expr_table_ctor(self, expr):
out = self.alloc()
self.emit(Op.NEWTABLE, out)
array_index = 1
fresh_keys: set[object] = set()
unknown_key_seen = False
for i, field in enumerate(expr.fields):
last = i == len(expr.fields) - 1
if field.key is None:
Expand All @@ -901,19 +903,32 @@ def _expr_table_ctor(self, expr):
vr, _ = self.expr(field.value)
kr = self.alloc()
self.emit(Op.LOADK, kr, self.proto.add_const(array_index))
self.emit(Op.SETTABLE, out, kr, vr)
fresh = not unknown_key_seen and array_index not in fresh_keys
self.emit(Op.SETTABLE, out, kr, vr, d=int(fresh))
fresh_keys.add(array_index)
array_index += 1
else:
if isinstance(field.key, str) and isinstance(field.value, A.FunctionExpr):
field.value.debug_name = field.key
field.value.debug_namewhat = "field"
if isinstance(field.key, str):
literal_key = field.key.encode()
kr = self.alloc()
self.emit(Op.LOADK, kr, self.proto.add_const(field.key.encode()))
self.emit(Op.LOADK, kr, self.proto.add_const(literal_key))
else:
literal_key = None
kr, _ = self.expr(field.key)
vr, _ = self.expr(field.value)
self.emit(Op.SETTABLE, out, kr, vr)
fresh = (
literal_key is not None
and not unknown_key_seen
and literal_key not in fresh_keys
)
self.emit(Op.SETTABLE, out, kr, vr, d=int(fresh))
if literal_key is None:
unknown_key_seen = True
else:
fresh_keys.add(literal_key)
return out, TABLE

def _expr_field(self, expr):
Expand Down
Loading
Loading