Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
65 commits
Select commit Hold shift + click to select a range
e8c3580
perf: count references without atomics while one thread owns the heap
owenthcarey Sep 27, 2026
f448fba
perf: add native ABCs, timsort, mimalloc, and core container ops
owenthcarey Sep 27, 2026
711cf0f
perf: iterate dicts, unpack sequences, and compare scalars in the cor…
owenthcarey Sep 27, 2026
0d8c07a
perf: trim calls, generators, JIT dispatch, random, and raises
owenthcarey Sep 28, 2026
f7e5a5b
perf: evaluate pure leaves and field updates directly at JIT call sites
owenthcarey Sep 28, 2026
305f587
perf: polymorphic call caches, keyword calls, raises, and dict probes
owenthcarey Sep 28, 2026
6b4cf87
perf: call compiled scalar functions directly from native code
owenthcarey Sep 28, 2026
6685b59
perf: cheaper method lookups, keyword calls, and datetime natives
owenthcarey Sep 28, 2026
9cc0d31
perf: allocate through mimalloc's plain entry points
owenthcarey Sep 28, 2026
faf7ae6
perf: run builtin container methods inside the core loop
owenthcarey Sep 28, 2026
ddf6d24
perf: run small mutating methods and native iterators without frames
owenthcarey Sep 28, 2026
afa8bf6
perf: cheaper JIT attribute access on slots and instance dicts
owenthcarey Sep 28, 2026
fd14f0b
build: use fat LTO for release builds
owenthcarey Sep 28, 2026
d56e063
perf: skip scalar and plain-object children in the reap cascade
owenthcarey Sep 28, 2026
8938d81
perf: store instance attributes in class-shared split layouts
owenthcarey Sep 28, 2026
0f400ff
perf: bind skipped keyword defaults in compiled calls
owenthcarey Sep 28, 2026
031ad5c
perf: cheaper short loops and short-lived containers
owenthcarey Sep 28, 2026
170930b
fix: restore the VM unit tests this branch broke
owenthcarey Sep 28, 2026
9c23bdb
perf: call native methods and subscripts in place in the core loop
owenthcarey Sep 28, 2026
169aece
perf: skip JIT compiles that cost more than they save
owenthcarey Sep 28, 2026
2025338
perf: resume generators inline for next()
owenthcarey Sep 28, 2026
acc8fc2
perf: read and write split instance fields through per-site shortcuts
owenthcarey Sep 29, 2026
09d7711
perf: run leaf bodies from a pre-translated register plan
owenthcarey Sep 29, 2026
0c0b819
perf: construct instances with leaf __init__ bodies frameless
owenthcarey Sep 29, 2026
a99ca77
perf: cheaper GC bookkeeping and frameless calls into compiled leaves
owenthcarey Sep 29, 2026
bc493fd
perf: call math functions and read module attributes in the core loop
owenthcarey Sep 29, 2026
eb3985b
perf: serve native len, truth and next on instances without generic d…
owenthcarey Sep 29, 2026
0c735b0
perf: format and join strings in the core loop, and stamp dicts witho…
owenthcarey Sep 29, 2026
d4f900c
perf: run natively served fields, arithmetic and comparisons in the c…
owenthcarey Sep 29, 2026
ba7403c
perf: serve scalar JIT attribute reads and writes on a lean path
owenthcarey Sep 29, 2026
f5a749b
perf: slice sequences and search strings without leaving the core loop
owenthcarey Sep 29, 2026
493558a
perf: reuse a dynamic call's resolved native callee
owenthcarey Sep 29, 2026
2d23ffe
perf: call read-only builtins from frameless leaves
owenthcarey Sep 29, 2026
21a63f6
perf: collect a drained generator's yields in place, and fix a leaked…
owenthcarey Sep 29, 2026
a2e94c8
perf: remember verified leaf method sites, and run **kwargs leaves fr…
owenthcarey Sep 29, 2026
8ef7e61
perf: run *args, **kwargs, keyword-only and spread calls inline
owenthcarey Sep 29, 2026
b14deb1
perf: fuse len() of native-length instances, and memoize pickled slot…
owenthcarey Sep 29, 2026
c7a2621
tools: add a profile-guided release build
owenthcarey Sep 29, 2026
81d6c6d
perf: match leaf plan ops in place, and hash collector tables by address
owenthcarey Sep 29, 2026
dbd4ec8
perf: forward f(*args, **kwargs) without building the merged dict
owenthcarey Sep 29, 2026
838dcb1
perf: read a deque's slots in one pass, and grade instance subscripts…
owenthcarey Sep 29, 2026
ed8486a
perf: retire frameless direct JIT callees that keep calling the inter…
owenthcarey Sep 29, 2026
0e11c8c
perf: cheapen the core loop's frame switch and generator resume
owenthcarey Sep 29, 2026
4071d67
perf: release still-shared JIT pins and core operands without the ful…
owenthcarey Sep 29, 2026
18ec3ec
perf: return inline frames and nested leaf calls with less bookkeeping
owenthcarey Sep 29, 2026
1587790
perf: reuse the pickle encoder's memo table and size its output
owenthcarey Sep 29, 2026
2a1c6cf
perf: fuse container method calls, and cheapen dying unpickled graphs
owenthcarey Sep 29, 2026
16fed23
ci: fix formatting, clippy lints, and the GC threshold fixture
owenthcarey Sep 30, 2026
bd076cf
perf: warm the JIT's code generator off the main thread
owenthcarey Sep 30, 2026
93320e4
perf: run simple generator bodies in fast steps, without switching
owenthcarey Sep 30, 2026
75d2023
perf: skip the fallible reservation on the pickle codec's hot pushes
owenthcarey Sep 30, 2026
f4d9b37
perf: check generator fast-step bounds once per code object
owenthcarey Sep 30, 2026
a766bfb
perf: fold int-expression arguments into fused native method calls
owenthcarey Sep 30, 2026
38f3c90
perf: run more constructors and nested leaf calls without frames
owenthcarey Sep 30, 2026
998c5c4
fix: keep tiny pure-leaf shapes on their dedicated evaluator
owenthcarey Sep 30, 2026
517deb2
perf: slice tuples, bytes, and code points without copying them whole
owenthcarey Sep 30, 2026
40bcd2c
perf: cache compiled stdlib modules in a WeavePy-native format
owenthcarey Sep 30, 2026
902ecdb
perf: keep dormant GC suspects out of the per-drop sweep
owenthcarey Sep 30, 2026
80e6378
perf: read class scalars through instances and fuse local branches
owenthcarey Sep 30, 2026
32fae0b
perf: rebind existing globals without invalidating cached global loads
owenthcarey Sep 30, 2026
eaa08a0
docs: report the branch's standing against CPython 3.14
owenthcarey Sep 30, 2026
31503af
perf: keep nested generators on fast steps past the eval breaker
owenthcarey Sep 30, 2026
2aea55a
perf: run list.append and list.pop in line from the fused method call
owenthcarey Sep 30, 2026
9e7ed7f
ci: satisfy rustfmt and clippy from Rust 1.98
owenthcarey Sep 30, 2026
3e99277
fix: release a SimpleQueue's wake gate once when puts race without th…
owenthcarey Sep 30, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

5 changes: 4 additions & 1 deletion Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -141,6 +141,7 @@ unicode_names2 = "2.0"
memmap2 = "0.9"
memchr = "2.8"
libc = "0.2"
libmimalloc-sys = { version = "0.1.49", default-features = false }
rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] }
rustls-pki-types = "1.7"
# PKCS#8 "ENCRYPTED PRIVATE KEY" (PBES2) decryption for password-protected
Expand Down Expand Up @@ -317,7 +318,9 @@ opt-level = 1

[profile.release]
opt-level = 3
lto = "thin"
# Fat LTO inlines across the VM, compiler, and JIT crates: 1-10% faster
# on the benchmark fixtures for about a third more build time.
lto = "fat"
codegen-units = 1
debug = "line-tables-only"
strip = "debuginfo"
Expand Down
12 changes: 9 additions & 3 deletions crates/weavepy-capi/src/memoryview.rs
Original file line number Diff line number Diff line change
Expand Up @@ -329,7 +329,9 @@ pub unsafe extern "C" fn PyMemoryView_FromObjectAndFlags(
Some(obj.clone())
};
PyMemoryView {
buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)),
buffer: MemoryViewBuffer::Shared(
weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer),
),
start: Cell::new(start),
len: Cell::new(view_len),
readonly: Cell::new(readonly),
Expand Down Expand Up @@ -403,7 +405,9 @@ pub unsafe extern "C" fn PyMemoryView_FromMemory(
readonly,
};
let mv = PyMemoryView::contiguous_1d(
MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)),
MemoryViewBuffer::Shared(
weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer),
),
len,
readonly,
"B".to_owned(),
Expand Down Expand Up @@ -486,7 +490,9 @@ pub unsafe extern "C" fn PyMemoryView_FromBuffer(view: *const Py_buffer) -> *mut
readonly,
};
let mv = PyMemoryView {
buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)),
buffer: MemoryViewBuffer::Shared(
weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer),
),
start: Cell::new(0),
len: Cell::new(len),
readonly: Cell::new(readonly),
Expand Down
4 changes: 4 additions & 0 deletions crates/weavepy-cli/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,7 @@ weavepy-compiler = { workspace = true }
weavepy-conformance = { workspace = true }
weavepy-version = { workspace = true }
libc = { workspace = true }
libmimalloc-sys = { workspace = true }
anyhow = { workspace = true }
clap = { workspace = true }
serde_json = { workspace = true }
Expand All @@ -59,6 +60,9 @@ windows-sys = { workspace = true }
default = ["jit"]
# RFC 0032 — build the `weavepy` binary with the tier-2 JIT compiled in.
jit = ["weavepy/jit", "weavepy-vm/jit"]
# Sample live allocations by call stack (`WEAVEPY_ALLOC_PROFILE=<file>`,
# macOS only); see `src/alloc_profile.rs`.
alloc-profile = []

[lints]
workspace = true
81 changes: 81 additions & 0 deletions crates/weavepy-cli/src/alloc.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,81 @@
//! The process allocator: mimalloc, entered through its plain allocation
//! functions whenever they already satisfy the requested alignment.
//!
//! Every mimalloc block is at least word aligned, and every block of 16
//! bytes or more is 16-byte aligned (`MI_MAX_ALIGN_SIZE`), so a layout
//! aligned to that much needs no aligned entry point. The aligned entry points
//! take their fast path only when the size class's free list happens to
//! offer a suitably aligned block, and otherwise fall to a generic path
//! that may over-allocate. Rust asks for alignment on every allocation,
//! so going through them would put every allocation on that path.

use std::alloc::{GlobalAlloc, Layout};
use std::ffi::c_void;

use libmimalloc_sys::{
mi_free, mi_malloc, mi_malloc_aligned, mi_realloc, mi_realloc_aligned, mi_zalloc,
mi_zalloc_aligned,
};

/// The largest alignment every mimalloc block already has.
const WORD_ALIGN: usize = std::mem::size_of::<usize>();

/// mimalloc's `MI_MAX_ALIGN_SIZE`: the alignment of every block at least
/// this large.
const MAX_ALIGN: usize = 16;

/// Whether the plain entry points already satisfy `align` for `size`.
#[inline(always)]
fn plain(size: usize, align: usize) -> bool {
align <= WORD_ALIGN || (align <= MAX_ALIGN && size >= align)
}

/// mimalloc as the global allocator (see the module docs).
pub(crate) struct Mimalloc;

// SAFETY: every block comes from mimalloc with at least the layout's
// alignment (word-aligned blocks, or the aligned entry points for larger
// alignments), and every block is released or resized through mimalloc.
unsafe impl GlobalAlloc for Mimalloc {
#[inline]
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
// SAFETY: plain FFI allocation calls.
unsafe {
if plain(layout.size(), layout.align()) {
mi_malloc(layout.size()).cast()
} else {
mi_malloc_aligned(layout.size(), layout.align()).cast()
}
}
}

#[inline]
unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 {
// SAFETY: as in `alloc`.
unsafe {
if plain(layout.size(), layout.align()) {
mi_zalloc(layout.size()).cast()
} else {
mi_zalloc_aligned(layout.size(), layout.align()).cast()
}
}
}

#[inline]
unsafe fn dealloc(&self, ptr: *mut u8, _layout: Layout) {
// SAFETY: `ptr` came from this allocator (the caller's contract).
unsafe { mi_free(ptr.cast::<c_void>()) }
}

#[inline]
unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 {
// SAFETY: `ptr` came from this allocator with `layout`.
unsafe {
if plain(new_size, layout.align()) {
mi_realloc(ptr.cast::<c_void>(), new_size).cast()
} else {
mi_realloc_aligned(ptr.cast::<c_void>(), new_size, layout.align()).cast()
}
}
}
}
211 changes: 211 additions & 0 deletions crates/weavepy-cli/src/alloc_profile.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,211 @@
//! Opt-in allocation-site profiler (`--features alloc-profile`, macOS).
//!
//! The global allocator samples one allocation per [`SAMPLE`] bytes and
//! records its call stack. A sampled block is charged [`SAMPLE`] bytes to
//! its stack while it lives, so at exit the table estimates the *live* heap
//! by allocation site. With `WEAVEPY_ALLOC_PROFILE=<file>` set, the live
//! samples are written to `<file>`: one line per sample, the charged bytes
//! then the slide-adjusted return addresses (resolve them with `atos`).

use std::alloc::{GlobalAlloc, Layout};
use std::cell::Cell;
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};

/// Bytes between samples.
const SAMPLE: usize = 16 * 1024;
/// Return addresses kept per sample.
const DEPTH: usize = 24;
/// Sampled blocks tracked at once (open addressing, power of two).
const SLOTS: usize = 1 << 18;

struct Sample {
ptr: usize,
stack: [usize; DEPTH],
}

static ENABLED: AtomicBool = AtomicBool::new(false);
static LOCK: AtomicBool = AtomicBool::new(false);
static TABLE: AtomicUsize = AtomicUsize::new(0);

thread_local! {
static IN_HOOK: Cell<bool> = const { Cell::new(false) };
static UNTIL_SAMPLE: Cell<usize> = const { Cell::new(SAMPLE) };
}

extern "C" {
fn _dyld_get_image_vmaddr_slide(image_index: u32) -> isize;
}

fn lock() {
while LOCK
.compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed)
.is_err()
{
std::hint::spin_loop();
}
}

fn unlock() {
LOCK.store(false, Ordering::Release);
}

fn table() -> *mut Sample {
TABLE.load(Ordering::Acquire) as *mut Sample
}

fn slot_of(ptr: usize) -> usize {
(ptr.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> 20) & (SLOTS - 1)
}

/// Start sampling (allocates the table from the system allocator).
pub(crate) fn start() {
if std::env::var_os("WEAVEPY_ALLOC_PROFILE").is_none() {
return;
}
let bytes = SLOTS * std::mem::size_of::<Sample>();
// SAFETY: a fresh zeroed mapping owned for the rest of the process.
let p = unsafe {
libc::mmap(
std::ptr::null_mut(),
bytes,
libc::PROT_READ | libc::PROT_WRITE,
libc::MAP_PRIVATE | libc::MAP_ANON,
-1,
0,
)
};
if p == libc::MAP_FAILED {
return;
}
TABLE.store(p as usize, Ordering::Release);
ENABLED.store(true, Ordering::Release);
}

fn record(ptr: usize) {
let mut stack = [0usize; DEPTH];
// SAFETY: `backtrace` fills at most `DEPTH` entries of the buffer.
let n = unsafe { libc::backtrace(stack.as_mut_ptr().cast(), DEPTH as libc::c_int) };
let _ = n;
let t = table();
lock();
let mut i = slot_of(ptr);
for _ in 0..SLOTS {
// SAFETY: `i < SLOTS`, inside the mapping.
let s = unsafe { &mut *t.add(i) };
if s.ptr == 0 || s.ptr == ptr {
s.ptr = ptr;
s.stack = stack;
break;
}
i = (i + 1) & (SLOTS - 1);
}
unlock();
}

fn forget(ptr: usize) {
let t = table();
lock();
let mut i = slot_of(ptr);
for _ in 0..SLOTS {
// SAFETY: as in `record`.
let s = unsafe { &mut *t.add(i) };
if s.ptr == 0 {
break;
}
if s.ptr == ptr {
// A tombstone keeps later probes of this chain reachable.
s.ptr = usize::MAX;
break;
}
i = (i + 1) & (SLOTS - 1);
}
unlock();
}

/// The profiling allocator: mimalloc plus the sampler.
pub(crate) struct Profiled;

// SAFETY: every allocation is mimalloc's; the sampler only records
// addresses and never touches the blocks.
unsafe impl GlobalAlloc for Profiled {
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
// SAFETY: forwarded unchanged.
let p = unsafe { crate::alloc::Mimalloc.alloc(layout) };
if ENABLED.load(Ordering::Relaxed) && !p.is_null() {
sample(p as usize, layout.size());
}
p
}

unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
if ENABLED.load(Ordering::Relaxed) {
forget(ptr as usize);
}
// SAFETY: forwarded unchanged.
unsafe { crate::alloc::Mimalloc.dealloc(ptr, layout) }
}

unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 {
if ENABLED.load(Ordering::Relaxed) {
forget(ptr as usize);
}
// SAFETY: forwarded unchanged.
let p = unsafe { crate::alloc::Mimalloc.realloc(ptr, layout, new_size) };
if ENABLED.load(Ordering::Relaxed) && !p.is_null() {
sample(p as usize, new_size);
}
p
}
}

fn sample(ptr: usize, size: usize) {
let _ = IN_HOOK.try_with(|hook| {
if hook.get() {
return;
}
let due = UNTIL_SAMPLE.with(|left| {
let l = left.get();
if size >= l {
left.set(SAMPLE);
true
} else {
left.set(l - size);
false
}
});
if due {
hook.set(true);
record(ptr);
hook.set(false);
}
});
}

/// Write the live samples (see the module docs).
pub(crate) fn finish() {
let Some(path) = std::env::var_os("WEAVEPY_ALLOC_PROFILE") else {
return;
};
if !ENABLED.swap(false, Ordering::AcqRel) {
return;
}
let t = table();
// SAFETY: the main executable is image 0.
let slide = unsafe { _dyld_get_image_vmaddr_slide(0) } as usize;
let mut out = String::new();
lock();
for i in 0..SLOTS {
// SAFETY: `i < SLOTS`.
let s = unsafe { &*t.add(i) };
if s.ptr == 0 || s.ptr == usize::MAX {
continue;
}
out.push_str(&SAMPLE.to_string());
for &a in s.stack.iter().take_while(|a| **a != 0) {
out.push_str(&format!(" {:x}", a.wrapping_sub(slide)));
}
out.push('\n');
}
unlock();
let _ = std::fs::write(path, out);
}
Loading
Loading