From e8c3580d057348b1053f9c393474fd324aee0a6f Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 12:22:29 -0700 Subject: [PATCH 01/65] perf: count references without atomics while one thread owns the heap The VM's Rc was an alias for std::sync::Arc, so every clone and drop of a heap object was a locked read-modify-write. Wrap Arc and Weak in newtypes whose strong-count updates are plain loads and stores until a second thread registers with the GIL, free-threading starts, or a Python thread is spawned. Deallocation still goes through Arc's own drop. Shared strings, bytes, and tuples use the same biased updates. Debug builds keep atomics because the unit-test binary runs interpreters on concurrent threads. --- crates/weavepy-capi/src/memoryview.rs | 12 +- crates/weavepy-vm/src/gc_trace.rs | 16 +- crates/weavepy-vm/src/lazy_arc.rs | 2 +- crates/weavepy-vm/src/lib.rs | 18 +- crates/weavepy-vm/src/pycache.rs | 11 +- crates/weavepy-vm/src/rc.rs | 472 ++++++++++++++++++ crates/weavepy-vm/src/shared_value.rs | 14 +- crates/weavepy-vm/src/stdlib/ctypes_native.rs | 11 +- .../weavepy-vm/src/stdlib/datetime_native.rs | 2 +- crates/weavepy-vm/src/stdlib/mmap_mod.rs | 2 +- crates/weavepy-vm/src/stdlib/select_mod.rs | 4 +- crates/weavepy-vm/src/stdlib/sys.rs | 4 +- crates/weavepy-vm/src/stdlib/thread_real.rs | 2 + crates/weavepy-vm/src/stdlib/weakref_real.rs | 2 +- crates/weavepy-vm/src/sync.rs | 35 +- crates/weavepy-vm/src/tuple_storage.rs | 3 +- crates/weavepy-vm/src/types.rs | 6 +- crates/weavepy-vm/src/weakref_registry.rs | 4 +- crates/weavepy/tests/fixtures.rs | 2 +- 19 files changed, 565 insertions(+), 57 deletions(-) create mode 100644 crates/weavepy-vm/src/rc.rs diff --git a/crates/weavepy-capi/src/memoryview.rs b/crates/weavepy-capi/src/memoryview.rs index 019fe566..547e6f48 100644 --- a/crates/weavepy-capi/src/memoryview.rs +++ b/crates/weavepy-capi/src/memoryview.rs @@ -329,7 +329,9 @@ pub unsafe extern "C" fn PyMemoryView_FromObjectAndFlags( Some(obj.clone()) }; PyMemoryView { - buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + buffer: MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), start: Cell::new(start), len: Cell::new(view_len), readonly: Cell::new(readonly), @@ -403,7 +405,9 @@ pub unsafe extern "C" fn PyMemoryView_FromMemory( readonly, }; let mv = PyMemoryView::contiguous_1d( - MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), len, readonly, "B".to_owned(), @@ -486,7 +490,9 @@ pub unsafe extern "C" fn PyMemoryView_FromBuffer(view: *const Py_buffer) -> *mut readonly, }; let mv = PyMemoryView { - buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + buffer: MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), start: Cell::new(0), len: Cell::new(len), readonly: Cell::new(readonly), diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index c9d0ae14..623b3274 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -2186,12 +2186,12 @@ impl GcState { let mut trash_ids: std::collections::HashSet = unreachable.iter().map(|h| h.id).collect(); let wrapper_is_trash = - |slot: &Arc, + |slot: &crate::sync::Rc, trash: &std::collections::HashSet| { slot.py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(crate::sync::Weak::upgrade) .is_none_or(|inst| { trash.contains(&(crate::sync::Rc::as_ptr(&inst) as usize as u64)) }) @@ -2217,7 +2217,7 @@ impl GcState { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(crate::sync::Weak::upgrade) .map(crate::object::Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); @@ -2520,7 +2520,7 @@ impl GcState { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(crate::sync::Weak::upgrade) .map(crate::object::Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); @@ -3548,9 +3548,9 @@ const DEFERRED_CAP: usize = 4096; /// weak `Object`, because `Object`'s payload `Arc` is what has to stay /// weak — holding the `Object` itself would pin the container alive. enum DeferredContainer { - List(std::sync::Weak>>), - Dict(std::sync::Weak>), - Set(std::sync::Weak>), + List(crate::sync::Weak>>), + Dict(crate::sync::Weak>), + Set(crate::sync::Weak>), } impl DeferredContainer { @@ -4627,7 +4627,7 @@ mod tests { let slots: Vec<_> = roots .iter() .map(|target| { - let slot = Arc::new(WeakRefSlot::new( + let slot = Rc::new(WeakRefSlot::new( id_of(target), target.clone(), false, diff --git a/crates/weavepy-vm/src/lazy_arc.rs b/crates/weavepy-vm/src/lazy_arc.rs index 9b219011..a4c07ed8 100644 --- a/crates/weavepy-vm/src/lazy_arc.rs +++ b/crates/weavepy-vm/src/lazy_arc.rs @@ -3,12 +3,12 @@ //! A published pointer is never reset through shared access. Every borrowed //! value remains live until its owner can be exclusively destroyed or replaced. +use crate::sync::Rc as Arc; use std::fmt; use std::marker::PhantomData; use std::mem::ManuallyDrop; use std::ptr; use std::sync::atomic::{AtomicPtr, Ordering}; -use std::sync::Arc; pub struct LazyArc { pointer: AtomicPtr, diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 44777f5a..bad4f4d1 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -67,6 +67,7 @@ pub mod proc_init; pub mod py_errno; pub mod pycache; pub mod rare_events; +pub mod rc; pub mod recursion; pub mod shared_value; pub mod specialize; @@ -976,7 +977,7 @@ impl Default for Interpreter { // `_Py_SetLocaleFromEnv`) so `_locale.getencoding()` and // `locale.setlocale(..., None)` observe the user's locale. crate::stdlib::locale_mod::init_from_env(); - let stdout: Stdout = Rc::new(RefCell::new(std::io::stdout())); + let stdout: Stdout = Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stdout()))); let mut builtins_dict = builtins::default_builtins(); // The `builtins` module exposes the core types/exceptions as the // real `type` objects (CPython's `builtins.int is int`), not the @@ -38082,8 +38083,8 @@ impl Interpreter { if old.solid_base_name() != new.solid_base_name() { return false; } - let oldbase = old.layout_struct_base(); - let newbase = new.layout_struct_base(); + let oldbase = TypeObject::layout_struct_base(old); + let newbase = TypeObject::layout_struct_base(new); if Rc::ptr_eq(&oldbase, &newbase) { return true; } @@ -61058,7 +61059,7 @@ fn constant_to_object(c: Constant) -> Object { } // Hand out the pool's own `Arc` so every access sees the *same* // code object (identity, like CPython's co_consts). - Constant::Code(c) => Object::Code(c), + Constant::Code(c) => Object::Code(Rc::from_arc(c)), Constant::Ellipsis => crate::vm_singletons::ellipsis(), Constant::Slice(parts) => { let (start, stop, step) = *parts; @@ -61097,7 +61098,7 @@ fn object_to_constant(o: &Object) -> Constant { Object::FrozenSet(s) => { Constant::FrozenSet(s.iter().map(|k| object_to_constant(&k.0)).collect()) } - Object::Code(c) => Constant::Code(c.clone()), + Object::Code(c) => Constant::Code(Rc::into_arc(c.clone())), // A `slice` reaches the pool from 3.14's constant-slice folding // (`a[1:2]`); its bounds must themselves be pool-representable. Object::Slice(s) => { @@ -63546,7 +63547,7 @@ next(gen) let code = compile_module(&module).expect("compile"); let mut interp = Interpreter::new(); let buf: Rc>> = Rc::new(RefCell::new(Vec::new())); - let writer: Stdout = buf.clone() as Rc>; + let writer: Stdout = crate::rc_unsize!(buf.clone() => RefCell); interp.set_stdout(writer); interp .run_module(&code) @@ -68995,7 +68996,7 @@ print(sliced('é😀z', 2)) _ => None, }) .expect("slice function"); - let function = Rc::make_mut(nested); + let function = std::sync::Arc::make_mut(nested); let pc = function .instructions .iter() @@ -69008,7 +69009,8 @@ print(sliced('é😀z', 2)) function.instructions[pc].arg = 2; let mut interp = Interpreter::new(); let buffer: Rc>> = Rc::new(RefCell::new(Vec::new())); - let writer: Stdout = buffer.clone() as Rc>; + let writer: Stdout = + crate::rc_unsize!(buffer.clone() => RefCell); interp.set_stdout(writer); interp.run_module(&code).expect("two-bound native fallback"); assert_eq!(String::from_utf8(buffer.borrow().clone()).unwrap(), "é😀\n"); diff --git a/crates/weavepy-vm/src/pycache.rs b/crates/weavepy-vm/src/pycache.rs index 0186b05e..93ba292d 100644 --- a/crates/weavepy-vm/src/pycache.rs +++ b/crates/weavepy-vm/src/pycache.rs @@ -408,6 +408,7 @@ fn mtime_seconds(meta: &fs::Metadata) -> u32 { #[cfg(test)] mod ownership_tests { use super::*; + use std::sync::Arc; use weavepy_compiler::{compile_module, Constant}; #[test] @@ -433,7 +434,7 @@ mod ownership_tests { fn collect(code: &CodeObject, rows: &mut Vec<(*const CodeObject, String)>) { for c in &code.constants { if let Constant::Code(inner) = c { - rows.push((Rc::as_ptr(inner), inner.filename.clone())); + rows.push((Arc::as_ptr(inner), inner.filename.clone())); collect(inner, rows); } } @@ -455,17 +456,17 @@ mod ownership_tests { #[test] fn relocation_preserves_shared_code_and_nested_tuple_constants() { - let child = Rc::new(CodeObject { + let child = Arc::new(CodeObject { name: "child".to_owned(), filename: "original.py".to_owned(), ..CodeObject::default() }); - let root = Rc::new(CodeObject { + let root = Arc::new(CodeObject { filename: "original.py".to_owned(), constants: vec![Constant::Tuple(vec![Constant::Code(child.clone())])], ..CodeObject::default() }); - let mut moved = own_decoded_code(root.clone()); + let mut moved = own_decoded_code(Rc::from_arc(root.clone())); rewrite_filenames(&mut moved, "relocated.py"); assert_eq!(root.filename, "original.py"); assert_eq!(child.filename, "original.py"); @@ -476,6 +477,6 @@ mod ownership_tests { panic!("expected code"); }; assert_eq!(relocated.filename, "relocated.py"); - assert!(!Rc::ptr_eq(&child, relocated)); + assert!(!Arc::ptr_eq(&child, relocated)); } } diff --git a/crates/weavepy-vm/src/rc.rs b/crates/weavepy-vm/src/rc.rs new file mode 100644 index 00000000..dc67605e --- /dev/null +++ b/crates/weavepy-vm/src/rc.rs @@ -0,0 +1,472 @@ +//! Biased reference counting for the VM heap. +//! +//! [`Rc`] and [`Weak`] wrap [`std::sync::Arc`] and [`std::sync::Weak`] +//! with the same API, but a strong-count increment or decrement is a plain +//! load and store instead of a locked read-modify-write while one thread +//! owns every object. That's the common case: a program that never starts a +//! second thread, running under the GIL. A `lock xadd` pair costs about four +//! times a plain pair on x86, and the interpreter performs roughly one pair +//! per bytecode, so atomics were a leading cost of every workload. +//! +//! The bias is the one [`crate::sync`] already maintains for cell borrows: +//! it holds until a second thread registers to run VM code or free-threading +//! starts, and it's never restored. The registering thread revokes the bias +//! while holding the GIL; the previous owner released the GIL before that, +//! and reacquires it before touching another object, so every plain update +//! happens before the revocation and every later update is atomic. Threads +//! that touch objects without the GIL announce themselves with +//! [`crate::sync::mark_cells_shared`] before they begin, and code that +//! spawns a thread which will run VM code revokes the bias first (see +//! [`revoke_refcount_bias`]). +//! +//! Deallocation always goes through `Arc`'s own drop, so the payload, +//! weak references, and the allocation are released exactly as before. +//! Only the counter arithmetic changes, and only for owners other than the +//! last one. + +use std::borrow::Borrow; +use std::fmt; +use std::hash::{Hash, Hasher}; +use std::mem::ManuallyDrop; +use std::ops::Deref; +use std::ptr; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; + +/// Whether strong counts may be updated with plain loads and stores. +/// +/// Debug builds always use atomics: the unit-test binary runs many +/// interpreters on concurrent threads with no GIL between them, and they +/// share static objects. +#[inline(always)] +pub(crate) fn refcounts_biased() -> bool { + #[cfg(debug_assertions)] + { + false + } + #[cfg(not(debug_assertions))] + { + crate::sync::bias_held() + } +} + +/// Revoke the single-thread bias before starting a thread that may touch VM +/// objects before it acquires the GIL. The caller's later updates observe +/// the revocation in program order, and the new thread observes it through +/// the spawn. +pub fn revoke_refcount_bias() { + crate::sync::revoke_bias_for_spawn(); +} + +/// The strong counter of the `Arc` allocation holding `value`. +/// +/// `ArcInner` is `#[repr(C)]` with the strong count first, then the weak +/// count, then the payload at the next offset aligned for the payload. The +/// standard library relies on this layout for `Arc::from_raw`; the +/// `strong_word_matches_arc_layout` test pins it for every payload shape +/// the VM uses. +#[inline(always)] +fn strong_word(value: *const T) -> *const AtomicUsize { + let align = std::mem::align_of_val(unsafe { &*value }); + let header = 2 * std::mem::size_of::(); + let offset = (header + align - 1) & !(align - 1); + // SAFETY: the payload lives `offset` bytes into its ArcInner, whose + // first field is the strong count. + unsafe { value.cast::().sub(offset).cast::() } +} + +/// Add one strong owner of the `Arc` payload at `value`. +/// +/// # Safety +/// +/// `value` must point at the payload of a live `Arc` allocation. +#[inline(always)] +pub(crate) unsafe fn increment_strong(value: *const T) { + if refcounts_biased() { + // SAFETY: live allocation, per the caller. + let word = unsafe { &*strong_word(value) }; + word.store(word.load(Ordering::Relaxed) + 1, Ordering::Relaxed); + } else { + // SAFETY: as above. + unsafe { Arc::increment_strong_count(value) }; + } +} + +/// Release one strong owner of the `Arc` payload at `value` without +/// destroying it. Returns `false`, having changed nothing, when this is the +/// last owner (or the bias is off); the caller must then drop an `Arc` +/// normally. +/// +/// # Safety +/// +/// `value` must point at the payload of a live `Arc` allocation, and the +/// caller must own one of its strong references. +#[inline(always)] +pub(crate) unsafe fn try_release_shared(value: *const T) -> bool { + if !refcounts_biased() { + return false; + } + // SAFETY: live allocation, per the caller. + let word = unsafe { &*strong_word(value) }; + let n = word.load(Ordering::Relaxed); + if n == 1 { + return false; + } + word.store(n - 1, Ordering::Relaxed); + true +} + +/// Convert an [`Rc`] to one of an unsized type, such as a trait object: +/// `rc_unsize!(rc => dyn Trait)`. Stable Rust only coerces its own smart +/// pointers, so the conversion passes through `Arc`. +#[macro_export] +macro_rules! rc_unsize { + ($rc:expr => $ty:ty) => { + $crate::sync::Rc::<$ty>::from_arc($crate::sync::Rc::into_arc($rc) as ::std::sync::Arc<$ty>) + }; +} + +/// A reference-counted shared pointer with biased counting. See the module +/// documentation. +#[repr(transparent)] +pub struct Rc(ManuallyDrop>); + +/// A weak reference to an [`Rc`] allocation. +#[repr(transparent)] +pub struct Weak(std::sync::Weak); + +impl Rc { + #[inline] + pub fn new(value: T) -> Self { + Self(ManuallyDrop::new(Arc::new(value))) + } + + pub fn try_unwrap(this: Self) -> Result { + Arc::try_unwrap(Self::into_arc(this)).map_err(Self::from_arc) + } + + pub fn into_inner(this: Self) -> Option { + Arc::into_inner(Self::into_arc(this)) + } + + pub fn unwrap_or_clone(this: Self) -> T + where + T: Clone, + { + Arc::unwrap_or_clone(Self::into_arc(this)) + } + + pub fn new_cyclic(data_fn: impl FnOnce(&Weak) -> T) -> Self { + Self::from_arc(Arc::new_cyclic(|weak| { + // SAFETY: Weak is a transparent wrapper over std::sync::Weak. + data_fn(unsafe { &*ptr::from_ref(weak).cast::>() }) + })) + } +} + +impl Rc { + #[inline] + pub fn from_arc(arc: Arc) -> Self { + Self(ManuallyDrop::new(arc)) + } + + #[inline] + pub fn into_arc(this: Self) -> Arc { + let this = ManuallyDrop::new(this); + // SAFETY: moves the owned Arc out; `this` is never dropped. + unsafe { ptr::read(&*this.0) } + } + + #[inline] + pub fn as_arc(this: &Self) -> &Arc { + &this.0 + } + + #[inline] + pub fn as_ptr(this: &Self) -> *const T { + Arc::as_ptr(&this.0) + } + + #[inline] + pub fn ptr_eq(this: &Self, other: &Self) -> bool { + Arc::ptr_eq(&this.0, &other.0) + } + + #[inline] + pub fn strong_count(this: &Self) -> usize { + Arc::strong_count(&this.0) + } + + #[inline] + pub fn weak_count(this: &Self) -> usize { + Arc::weak_count(&this.0) + } + + #[inline] + pub fn downgrade(this: &Self) -> Weak { + Weak(Arc::downgrade(&this.0)) + } + + #[inline] + pub fn get_mut(this: &mut Self) -> Option<&mut T> { + Arc::get_mut(&mut this.0) + } + + #[inline] + pub fn into_raw(this: Self) -> *const T { + Arc::into_raw(Self::into_arc(this)) + } + + /// # Safety + /// + /// As [`Arc::from_raw`]. + #[inline] + pub unsafe fn from_raw(ptr: *const T) -> Self { + // SAFETY: forwarded to the caller. + Self::from_arc(unsafe { Arc::from_raw(ptr) }) + } + + /// # Safety + /// + /// As [`Arc::increment_strong_count`]. + #[inline] + pub unsafe fn increment_strong_count(ptr: *const T) { + // SAFETY: forwarded to the caller. + unsafe { increment_strong(ptr) } + } + + /// # Safety + /// + /// As [`Arc::decrement_strong_count`]. + #[inline] + pub unsafe fn decrement_strong_count(ptr: *const T) { + // SAFETY: forwarded to the caller. + unsafe { drop(Self::from_raw(ptr)) } + } +} + +impl Rc { + #[inline] + pub fn make_mut(this: &mut Self) -> &mut T { + Arc::make_mut(&mut this.0) + } +} + +impl Clone for Rc { + #[inline(always)] + fn clone(&self) -> Self { + // SAFETY: `self` keeps the allocation alive, and the new owner + // accounts for the added reference. + unsafe { + increment_strong(Arc::as_ptr(&self.0)); + Self(ManuallyDrop::new(ptr::read(&*self.0))) + } + } +} + +impl Drop for Rc { + #[inline(always)] + fn drop(&mut self) { + // SAFETY: this owner holds one strong reference, released exactly + // once: either here, or by Arc's own drop. + unsafe { + if !try_release_shared(Arc::as_ptr(&self.0)) { + ManuallyDrop::drop(&mut self.0); + } + } + } +} + +impl Deref for Rc { + type Target = T; + #[inline(always)] + fn deref(&self) -> &T { + &self.0 + } +} + +impl AsRef for Rc { + fn as_ref(&self) -> &T { + self + } +} + +impl Borrow for Rc { + fn borrow(&self) -> &T { + self + } +} + +impl fmt::Debug for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Debug::fmt(&**self, f) + } +} + +impl fmt::Display for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(&**self, f) + } +} + +impl fmt::Pointer for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Pointer::fmt(&Self::as_ptr(self), f) + } +} + +impl PartialEq for Rc { + fn eq(&self, other: &Self) -> bool { + **self == **other + } +} + +impl Eq for Rc {} + +impl PartialOrd for Rc { + fn partial_cmp(&self, other: &Self) -> Option { + (**self).partial_cmp(&**other) + } +} + +impl Ord for Rc { + fn cmp(&self, other: &Self) -> std::cmp::Ordering { + (**self).cmp(&**other) + } +} + +impl Hash for Rc { + fn hash(&self, state: &mut H) { + (**self).hash(state) + } +} + +impl Default for Rc { + fn default() -> Self { + Self::new(T::default()) + } +} + +impl From for Rc { + fn from(value: T) -> Self { + Self::new(value) + } +} + +impl From> for Rc { + fn from(value: Box) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From> for Rc<[T]> { + fn from(value: Vec) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From<&str> for Rc { + fn from(value: &str) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From for Rc { + fn from(value: String) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From> for Rc { + fn from(value: Arc) -> Self { + Self::from_arc(value) + } +} + +impl Weak { + pub const fn new() -> Self { + Self(std::sync::Weak::new()) + } +} + +impl Weak { + #[inline] + pub fn upgrade(&self) -> Option> { + self.0.upgrade().map(Rc::from_arc) + } + + pub fn strong_count(&self) -> usize { + self.0.strong_count() + } + + pub fn weak_count(&self) -> usize { + self.0.weak_count() + } + + pub fn ptr_eq(&self, other: &Self) -> bool { + self.0.ptr_eq(&other.0) + } + + pub fn as_ptr(&self) -> *const T { + self.0.as_ptr() + } +} + +impl Clone for Weak { + fn clone(&self) -> Self { + Self(self.0.clone()) + } +} + +impl Default for Weak { + fn default() -> Self { + Self::new() + } +} + +impl fmt::Debug for Weak { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("(Weak)") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn check(arc: &Arc) { + let before = Arc::strong_count(arc); + // SAFETY: `arc` is live. + let word = unsafe { &*strong_word(Arc::as_ptr(arc)) }; + assert_eq!(word.load(Ordering::Relaxed), before); + let extra = arc.clone(); + assert_eq!(word.load(Ordering::Relaxed), before + 1); + drop(extra); + } + + #[test] + fn strong_word_matches_arc_layout() { + #[repr(align(64))] + struct Wide(#[allow(dead_code)] u8); + check(&Arc::new(1u8)); + check(&Arc::new(7u64)); + check(&Arc::new(Wide(3))); + check(&Arc::new(String::from("x"))); + check::(&Arc::from("text")); + check::<[u64]>(&Arc::from(vec![1u64, 2, 3])); + check::(&(Arc::new(5u32) as Arc<_>)); + } + + #[test] + fn biased_updates_keep_arc_counts() { + let rc = Rc::new(vec![1, 2, 3]); + let copies: Vec<_> = (0..10).map(|_| rc.clone()).collect(); + assert_eq!(Rc::strong_count(&rc), 11); + drop(copies); + assert_eq!(Rc::strong_count(&rc), 1); + let weak = Rc::downgrade(&rc); + assert!(weak.upgrade().is_some()); + drop(rc); + assert!(weak.upgrade().is_none()); + } +} diff --git a/crates/weavepy-vm/src/shared_value.rs b/crates/weavepy-vm/src/shared_value.rs index 7a8d88bd..7ffec9a4 100644 --- a/crates/weavepy-vm/src/shared_value.rs +++ b/crates/weavepy-vm/src/shared_value.rs @@ -113,7 +113,13 @@ impl ThinArc { impl Clone for ThinArc { #[inline] fn clone(&self) -> Self { - Self::from_arc(Arc::clone(&self.arc_view())) + // SAFETY: this owner keeps the payload alive; the copy accounts for + // the added strong reference. + unsafe { crate::rc::increment_strong(self.raw()) }; + Self { + data: self.data, + ownership: PhantomData, + } } } @@ -122,7 +128,11 @@ impl Drop for ThinArc { fn drop(&mut self) { // SAFETY: release this owner's one strong reference exactly once. // Arc drops the payload and handles remaining weak references normally. - unsafe { drop(Arc::from_raw(self.raw())) } + unsafe { + if !crate::rc::try_release_shared(self.raw()) { + drop(Arc::from_raw(self.raw())) + } + } } } diff --git a/crates/weavepy-vm/src/stdlib/ctypes_native.rs b/crates/weavepy-vm/src/stdlib/ctypes_native.rs index 288ec10c..7cf7a7b8 100644 --- a/crates/weavepy-vm/src/stdlib/ctypes_native.rs +++ b/crates/weavepy-vm/src/stdlib/ctypes_native.rs @@ -327,11 +327,12 @@ fn b_memoryview_at(args: &[Object]) -> Result { if addr == 0 && size != 0 { return Err(value_error("memoryview_at: NULL pointer access")); } - let region: Rc = Rc::new(RawRegion { - ptr: addr, - len: size as usize, - readonly, - }); + let region: Rc = + Rc::from_arc(std::sync::Arc::new(RawRegion { + ptr: addr, + len: size as usize, + readonly, + })); Ok(Object::MemoryView(Rc::new( crate::object::PyMemoryView::from_shared(region), ))) diff --git a/crates/weavepy-vm/src/stdlib/datetime_native.rs b/crates/weavepy-vm/src/stdlib/datetime_native.rs index 704c9115..b8fc8dae 100644 --- a/crates/weavepy-vm/src/stdlib/datetime_native.rs +++ b/crates/weavepy-vm/src/stdlib/datetime_native.rs @@ -1677,7 +1677,7 @@ pub(crate) fn install(args: &[Object]) -> Result { for (kind, cls) in &classes { let _ = cls .native_ext - .set(state.clone() as Rc); + .set(crate::rc_unsize!(state.clone() => dyn std::any::Any + Send + Sync)); cls.native_kind.set(*kind); } for spec in SPECS { diff --git a/crates/weavepy-vm/src/stdlib/mmap_mod.rs b/crates/weavepy-vm/src/stdlib/mmap_mod.rs index 14ab65d1..87c233c4 100644 --- a/crates/weavepy-vm/src/stdlib/mmap_mod.rs +++ b/crates/weavepy-vm/src/stdlib/mmap_mod.rs @@ -410,7 +410,7 @@ fn state_cell(inst: &Rc) -> Result>, RuntimeEr /// `None` for a closed mapping. pub fn shared_buffer(inst: &Rc) -> Option> { let cell = state_cell(inst).ok()?; - let region: Rc = cell.borrow().region.clone(); + let region = crate::rc_unsize!(cell.borrow().region.clone() => dyn SharedMemBuffer); Some(region) } diff --git a/crates/weavepy-vm/src/stdlib/select_mod.rs b/crates/weavepy-vm/src/stdlib/select_mod.rs index cf1f6e7e..707fb23b 100644 --- a/crates/weavepy-vm/src/stdlib/select_mod.rs +++ b/crates/weavepy-vm/src/stdlib/select_mod.rs @@ -1179,7 +1179,7 @@ mod kqueue_impl { /// closed, matching CPython's `kqueue_queue_traverse`/at-fork sweep). /// `Rc` is `Arc` here (the RFC 0025 shared heap), so a process-global /// registry of `Weak` is sound. - static LIVE_KQUEUES: std::sync::Mutex>> = + static LIVE_KQUEUES: std::sync::Mutex>> = std::sync::Mutex::new(Vec::new()); fn store_kqueue(inst: &Rc, fd: libc::c_int) { @@ -1205,7 +1205,7 @@ mod kqueue_impl { /// observes `kq.closed == True` and `kq.fileno()` raising. We mirror /// that here (`test_kqueue.test_fork`). pub(super) fn close_all_in_child() { - let drained: Vec> = match LIVE_KQUEUES.lock() { + let drained: Vec> = match LIVE_KQUEUES.lock() { Ok(mut reg) => reg.drain(..).collect(), // A poisoned lock can't happen across `fork` (single thread in // the child), but recover defensively rather than panic. diff --git a/crates/weavepy-vm/src/stdlib/sys.rs b/crates/weavepy-vm/src/stdlib/sys.rs index 881b3345..0181d019 100644 --- a/crates/weavepy-vm/src/stdlib/sys.rs +++ b/crates/weavepy-vm/src/stdlib/sys.rs @@ -873,9 +873,9 @@ pub fn build(cache: &ModuleCache) -> Rc { // sharing the interpreter's host sinks, so `print()` and // direct writes via `sys.stdout.write(...)` agree. let stdout_sink: Rc> = - Rc::new(RefCell::new(std::io::stdout())); + Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stdout()))); let stderr_sink: Rc> = - Rc::new(RefCell::new(std::io::stderr())); + Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stderr()))); // CPython's `init_sys_streams`: a standard stream whose fd is // closed at startup (e.g. spawned with `os.close(0)` in a // preexec hook) is `None`, not a broken file object diff --git a/crates/weavepy-vm/src/stdlib/thread_real.rs b/crates/weavepy-vm/src/stdlib/thread_real.rs index d8160362..5ddb976c 100644 --- a/crates/weavepy-vm/src/stdlib/thread_real.rs +++ b/crates/weavepy-vm/src/stdlib/thread_real.rs @@ -1345,6 +1345,8 @@ fn spawn_python_worker( // deep recursion. A 64 MiB starting reserve keeps early-startup // segment churn low while letting hundreds of workers coexist. const WORKER_STACK_BYTES: usize = 64 * 1024 * 1024; // 64 MiB + // The worker clones and drops objects before it first takes the GIL. + crate::rc::revoke_refcount_bias(); let handle = std::thread::Builder::new() .name(format!("weavepy-worker-{}", synth_id)) .stack_size(WORKER_STACK_BYTES) diff --git a/crates/weavepy-vm/src/stdlib/weakref_real.rs b/crates/weavepy-vm/src/stdlib/weakref_real.rs index c011ac2e..d312ddb3 100644 --- a/crates/weavepy-vm/src/stdlib/weakref_real.rs +++ b/crates/weavepy-vm/src/stdlib/weakref_real.rs @@ -30,8 +30,8 @@ //! `True`. use crate::sync::Rc; +use crate::sync::Rc as Arc; use crate::sync::RefCell; -use std::sync::Arc; use crate::error::{type_error, value_error, RuntimeError}; use crate::import::ModuleCache; diff --git a/crates/weavepy-vm/src/sync.rs b/crates/weavepy-vm/src/sync.rs index 894448ca..311c5ee1 100644 --- a/crates/weavepy-vm/src/sync.rs +++ b/crates/weavepy-vm/src/sync.rs @@ -43,17 +43,9 @@ use parking_lot::{Condvar, Mutex, ReentrantMutex}; // `std::cell::RefCell`, `std::cell::Cell`. // --------------------------------------------------------------------------- -/// Drop-in replacement for [`std::rc::Rc`]. Backed by -/// [`std::sync::Arc`], so it carries the same atomic refcount -/// behaviour. Every method on `Arc` (`ptr_eq`, `clone`, `as_ptr`, -/// `strong_count`, `weak_count`, `downgrade`, `try_unwrap`, -/// `get_mut`, `into_inner`) is identical to the `Rc` API the -/// workspace already calls. -pub type Rc = std::sync::Arc; - -/// Drop-in replacement for [`std::rc::Weak`]. Backed by -/// [`std::sync::Weak`]. -pub type Weak = std::sync::Weak; +/// Drop-in replacement for [`std::rc::Rc`] and [`std::rc::Weak`], backed +/// by [`std::sync::Arc`] with biased counting (see [`crate::rc`]). +pub use crate::rc::{Rc, Weak}; /// An interior-mutability cell that's `Send + Sync` (when the /// payload is `Send`) and supports both the `RefCell` and `Cell` @@ -354,6 +346,26 @@ fn cells_unguarded() -> bool { CELLS_UNGUARDED.load(Ordering::Relaxed) } +/// Whether reference counts must be updated atomically: set whenever the +/// cell bias is revoked, and also before spawning a thread that may touch +/// objects before it registers (see [`crate::rc`]). Never cleared. +static RC_SHARED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + +/// True while one thread owns every reference count (see [`crate::rc`]). +#[cfg_attr(debug_assertions, allow(dead_code))] +#[inline(always)] +pub(crate) fn bias_held() -> bool { + !RC_SHARED.load(Ordering::Relaxed) +} + +/// Make reference counting atomic before spawning a thread that will run +/// VM code. Cell borrows keep their bias until the thread registers: the +/// spawner may hold a lock-free guard now, and it can't hand the GIL to the +/// new thread until that guard is released. +pub(crate) fn revoke_bias_for_spawn() { + RC_SHARED.store(true, Ordering::SeqCst); +} + /// The first thread to run VM code (see [`note_vm_thread`]). static FIRST: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); @@ -526,6 +538,7 @@ fn retract_sole_guard() { /// including the biased thread backing out of its own fast path — never /// waits on a guard it is itself holding. fn revoke_bias() { + RC_SHARED.store(true, Ordering::SeqCst); if CELLS_SHARED.swap(true, Ordering::SeqCst) { return; } diff --git a/crates/weavepy-vm/src/tuple_storage.rs b/crates/weavepy-vm/src/tuple_storage.rs index e8f6f72f..0bc37915 100644 --- a/crates/weavepy-vm/src/tuple_storage.rs +++ b/crates/weavepy-vm/src/tuple_storage.rs @@ -5,7 +5,8 @@ use std::alloc::Layout; use std::ops::{Deref, DerefMut}; use crate::object::Object; -use crate::sync::{CachedHash, Rc}; +use crate::sync::CachedHash; +use std::sync::Arc as Rc; pub type SharedTuple = ThinArc; diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index dda93592..469b4a1f 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -1092,7 +1092,7 @@ impl TypeObject { /// CPython `best_base`: the base contributing the instance layout — /// the one whose solid base is the most derived. Ties resolve to the /// first base (matching `type_new`'s left-to-right scan). - pub fn best_base(self: &Rc) -> Option> { + pub fn best_base(&self) -> Option> { let bases = self.bases.borrow(); let mut best: Option> = None; for b in bases.iter() { @@ -1118,8 +1118,8 @@ impl TypeObject { /// CPython `compatible_for_assignment`'s `newbase`/`oldbase` walk: /// climb the `best_base` chain past every level that doesn't change /// the struct, returning the most-derived type that *does*. - pub fn layout_struct_base(self: &Rc) -> Rc { - let mut cur = self.clone(); + pub fn layout_struct_base(this: &Rc) -> Rc { + let mut cur = this.clone(); while !cur.changes_layout() { match cur.best_base() { Some(b) => cur = b, diff --git a/crates/weavepy-vm/src/weakref_registry.rs b/crates/weavepy-vm/src/weakref_registry.rs index a98189bf..71ed558c 100644 --- a/crates/weavepy-vm/src/weakref_registry.rs +++ b/crates/weavepy-vm/src/weakref_registry.rs @@ -37,9 +37,9 @@ use crate::fasthash::ObjectIdHasher; use crate::shared_value::{SharedSlice, SharedStr, ThinArc}; use crate::sync::RefCell; +use crate::sync::{Rc as Arc, Weak}; use std::hash::BuildHasherDefault; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; -use std::sync::{Arc, Weak}; use crate::object::{DictKey, Object, StrKey}; @@ -548,7 +548,7 @@ pub fn queue_callbacks(cleared: Vec<(Arc, Option)>) { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(Weak::upgrade) .map(Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); diff --git a/crates/weavepy/tests/fixtures.rs b/crates/weavepy/tests/fixtures.rs index 9651824d..e3debb84 100644 --- a/crates/weavepy/tests/fixtures.rs +++ b/crates/weavepy/tests/fixtures.rs @@ -105,7 +105,7 @@ fn run_fixture(py_path: &Path) { let mut interp = vm::Interpreter::new(); let buf: Rc>> = Rc::new(RefCell::new(Vec::new())); - let sink: vm::Stdout = buf.clone() as Rc>; + let sink: vm::Stdout = vm::rc_unsize!(buf.clone() => RefCell); interp.set_stdout(sink); // Sibling files in the fixtures directory must be importable so // multi-file tests (e.g. `import _helper`) work. From f448fba76c63886dc21cddf637af7e249a0fc51c Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 15:42:19 -0700 Subject: [PATCH 02/65] perf: add native ABCs, timsort, mimalloc, and core container ops - Port CPython's `_abc` accelerator and use CPython's own `abc.py`. Registration and ABC `isinstance` checks no longer run through Python-level `WeakSet`s, which also takes `import site` from 25 ms to 15 ms. The `io` types drop a private `register` that shadowed `ABCMeta.register`. - Port CPython's list sort (timsort with the powersort merge policy) with its type-specialized comparisons for `str`, `int`, `float`, and tuple keys. Sorting a list that mixes NaNs with other numbers no longer panics, and inconsistent comparisons behave as in CPython. - Use mimalloc as the process allocator and remove the thread-cache front end over the system allocator. - Run container subscripts, subscript stores, and comprehension appends in the core loop, out of line so the rest of the loop keeps its register allocation. Inserts into dicts of any size stay native when the probe proves no key of another kind could compare equal, and report the mutation to dict watchers and the builtins rare-event counter. - Memoize whether a type overrides `__eq__` or `__hash__` per type version instead of looking the names up for every dict operation. - Store suspended generator frames as `Box` rather than type-erased boxes that were downcast on every resume. - Deliver work queued by a dropped value, such as an unclosed file's `ResourceWarning`, at the instruction that dropped it. --- Cargo.lock | 19 + Cargo.toml | 1 + crates/weavepy-cli/Cargo.toml | 1 + crates/weavepy-cli/src/lib.rs | 9 +- crates/weavepy-vm/src/builtins.rs | 31 +- crates/weavepy-vm/src/lib.rs | 768 +++++++++++--------- crates/weavepy-vm/src/object.rs | 136 ++-- crates/weavepy-vm/src/rc.rs | 11 +- crates/weavepy-vm/src/stdlib/abc_mod.rs | 555 ++++++++++---- crates/weavepy-vm/src/stdlib/io.rs | 39 - crates/weavepy-vm/src/stdlib/python/abc.py | 253 ++----- crates/weavepy-vm/src/stdlib/thread_real.rs | 1 - crates/weavepy-vm/src/tcache.rs | 226 ------ crates/weavepy-vm/src/timsort.rs | 736 +++++++++++++++++++ crates/weavepy-vm/src/types.rs | 123 +++- 15 files changed, 1850 insertions(+), 1059 deletions(-) delete mode 100644 crates/weavepy-vm/src/tcache.rs create mode 100644 crates/weavepy-vm/src/timsort.rs diff --git a/Cargo.lock b/Cargo.lock index f5f54d62..e0d7cc38 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1262,6 +1262,15 @@ version = "0.2.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" +[[package]] +name = "libmimalloc-sys" +version = "0.1.49" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6a45a52f43e1c16f667ccfe4dd8c85b7f7c204fd5e3bf46c5b0db9a5c3c0b8e9" +dependencies = [ + "cc", +] + [[package]] name = "libredox" version = "0.1.16" @@ -1364,6 +1373,15 @@ dependencies = [ "libc", ] +[[package]] +name = "mimalloc" +version = "0.1.52" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2d4139bb28d14ad1facf21d5eb8825051b326e172d216b39f6d31df53cc97862" +dependencies = [ + "libmimalloc-sys", +] + [[package]] name = "minimal-lexical" version = "0.2.1" @@ -2819,6 +2837,7 @@ dependencies = [ "clap", "dirs", "libc", + "mimalloc", "rustyline", "serde_json", "tracing", diff --git a/Cargo.toml b/Cargo.toml index 6cb9ac9a..213ef289 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -141,6 +141,7 @@ unicode_names2 = "2.0" memmap2 = "0.9" memchr = "2.8" libc = "0.2" +mimalloc = { version = "0.1.52", default-features = false } rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] } rustls-pki-types = "1.7" # PKCS#8 "ENCRYPTED PRIVATE KEY" (PBES2) decryption for password-protected diff --git a/crates/weavepy-cli/Cargo.toml b/crates/weavepy-cli/Cargo.toml index fa3631a3..c06f3259 100644 --- a/crates/weavepy-cli/Cargo.toml +++ b/crates/weavepy-cli/Cargo.toml @@ -39,6 +39,7 @@ weavepy-compiler = { workspace = true } weavepy-conformance = { workspace = true } weavepy-version = { workspace = true } libc = { workspace = true } +mimalloc = { workspace = true } anyhow = { workspace = true } clap = { workspace = true } serde_json = { workspace = true } diff --git a/crates/weavepy-cli/src/lib.rs b/crates/weavepy-cli/src/lib.rs index ddb2edd3..89f45eb5 100644 --- a/crates/weavepy-cli/src/lib.rs +++ b/crates/weavepy-cli/src/lib.rs @@ -36,11 +36,11 @@ use tracing_subscriber::EnvFilter; use weavepy::{InterpreterFlags, RunOptions}; -/// The process allocator: the system allocator behind per-thread -/// small-block caches (see `weavepy_vm::tcache`), enabled on the VM's -/// own threads. +/// The process allocator. The interpreter allocates and frees small blocks +/// constantly; mimalloc serves them from per-thread free lists, several +/// times faster than the system allocator on macOS. #[global_allocator] -static GLOBAL_ALLOC: weavepy::vm::tcache::ThreadCacheAlloc = weavepy::vm::tcache::ThreadCacheAlloc; +static GLOBAL_ALLOC: mimalloc::MiMalloc = mimalloc::MiMalloc; const VERSION: &str = env!("CARGO_PKG_VERSION"); @@ -721,7 +721,6 @@ fn run_on_large_stack(entry: fn() -> i32) -> i32 { weavepy::vm::stdlib::signal_mod::block_async_signals_current_thread(); let vm_entry = move || -> i32 { - weavepy::vm::tcache::enable_for_current_thread(); // Opt-in (`WEAVEPY_CRASH_BT`): register the native crash handler + // per-thread sigaltstack on the VM thread itself so a stack-overflow // SIGSEGV can be caught and reported (no-op stub on Windows). diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index d46b4ca1..88ce3c5c 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -8244,17 +8244,9 @@ fn b_sorted(args: &[Object]) -> Result { while let Some(v) = it.next_value() { buf.push(v); } - let mut err: Option = None; - buf.sort_by(|a: &Object, b: &Object| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - err = Some(e); - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } + crate::timsort::sort(&mut buf, |a, b| { + crate::compare_op(a, b, weavepy_compiler::CompareKind::Lt) + })?; let obj = Object::new_list(buf); crate::gc_trace::track(obj.clone()); Ok(obj) @@ -9276,7 +9268,7 @@ pub fn ensure_hashable(obj: &Object) -> Result<(), RuntimeError> { // container lookup runs (test_import's unhashable-`__name__` // str subclass in set membership). Object::Instance(inst) => { - if matches!(inst.cls().lookup("__hash__"), Some(Object::None)) { + if inst.class_dunder(crate::types::Dunder::Hash).is_none() { return Err(type_error(format!( "unhashable type: '{}'", inst.cls().name @@ -12873,18 +12865,9 @@ fn range_count(args: &[Object]) -> Result { fn list_sort(args: &[Object]) -> Result { let l = list_self(args)?; - let mut err: Option = None; - l.borrow_mut() - .sort_by(|a: &Object, b: &Object| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - err = Some(e); - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } + crate::timsort::sort(&mut l.borrow_mut(), |a, b| { + crate::compare_op(a, b, weavepy_compiler::CompareKind::Lt) + })?; Ok(Object::None) } diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index bad4f4d1..cb29d91f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -74,12 +74,12 @@ pub mod specialize; pub mod stdlib; pub mod stdlib_tree; pub mod sync; -pub mod tcache; pub mod thread_registry; /// RFC 0032 — tier-2 Cranelift JIT integration. Present only under the /// `jit` feature; the dispatch loop calls into it behind `#[cfg]` gates. #[cfg(feature = "jit")] mod tier2; +mod timsort; pub mod trace; mod tuple_storage; pub mod type_surface; @@ -107,7 +107,10 @@ use crate::types::{PyInstance, TypeObject}; // ---------- frame ---------- -struct Frame { +/// An activation's execution state. Public only so a suspended generator +/// can own one ([`GeneratorState`]); its fields are private to the VM. +#[doc(hidden)] +pub struct Frame { code: Rc, /// Local variables, indexed by `LOAD_FAST` / `STORE_FAST`. /// @@ -248,6 +251,15 @@ struct Frame { parked_native: Option>, } +impl std::fmt::Debug for Frame { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Frame") + .field("code", &self.code.qualname) + .field("pc", &self.pc) + .finish_non_exhaustive() + } +} + impl Frame { fn push(&mut self, v: Object) { self.stack.push(v); @@ -360,7 +372,8 @@ fn generator_frame_traverse(obj: &Object, visit: &mut dyn FnMut(&Object)) { GeneratorState::Created(b) | GeneratorState::Suspended(b) => b, _ => return, }; - if let Some(frame) = boxed.downcast_ref::() { + let frame: &Frame = boxed; + { if let Ok(locals) = frame.locals.try_borrow() { for v in locals.iter() { visit(v); @@ -483,9 +496,7 @@ fn frame_reapables(g: &Rc) -> Vec { GeneratorState::Created(b) | GeneratorState::Suspended(b) => b, _ => return Vec::new(), }; - let Some(frame) = boxed.downcast_ref::() else { - return Vec::new(); - }; + let frame: &Frame = boxed; // CPython clears a generator's frame the instant it finishes/closes // (`gen_send`/`gen_close` → `_PyFrame_ClearExceptCode`), so every // local and value-stack entry is decref'd promptly. WeavePy mirrors @@ -7771,9 +7782,9 @@ impl Interpreter { return g.code.clone(); } match &*g.state.borrow() { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => boxed - .downcast_ref::() - .map_or(Object::None, |f| Object::Code(f.code.clone())), + GeneratorState::Created(frame) | GeneratorState::Suspended(frame) => { + Object::Code(frame.code.clone()) + } GeneratorState::Running | GeneratorState::Finished => Object::None, } } @@ -7787,33 +7798,28 @@ impl Interpreter { fn gen_py_frame(&self, g: &Rc) -> Object { let mut state = g.state.borrow_mut(); match &mut *state { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => { - match boxed.downcast_mut::() { - Some(frame) => { - // RFC 0073 WS4 — a Python-visible frame shares - // the locals storage; write a parked native - // activation back before exposing one (park - // refuses whenever a `PyFrame` already exists, - // so this is the only creation path to guard). - #[cfg(feature = "jit")] - crate::tier2::materialize_parked(frame); - if let Some(py) = frame.py_frame.clone() { - if py.gen_owner.borrow().is_none() { - *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); - } - return Object::Frame(py); - } - // Not yet entered: build the frame snapshot now so - // `gi_frame` is observable before the first - // `next()`. `back` is None — a created/suspended - // generator frame has no live caller. - let py = self.build_py_frame(frame, None); + GeneratorState::Created(frame) | GeneratorState::Suspended(frame) => { + // RFC 0073 WS4 — a Python-visible frame shares + // the locals storage; write a parked native + // activation back before exposing one (park + // refuses whenever a `PyFrame` already exists, + // so this is the only creation path to guard). + #[cfg(feature = "jit")] + crate::tier2::materialize_parked(frame); + if let Some(py) = frame.py_frame.clone() { + if py.gen_owner.borrow().is_none() { *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); - frame.py_frame = Some(py.clone()); - Object::Frame(py) } - None => Object::None, + return Object::Frame(py); } + // Not yet entered: build the frame snapshot now so + // `gi_frame` is observable before the first + // `next()`. `back` is None — a created/suspended + // generator frame has no live caller. + let py = self.build_py_frame(frame, None); + *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); + frame.py_frame = Some(py.clone()); + Object::Frame(py) } // Running: the frame is live on the interpreter call stack, not // in the box — find it by its generator backlink. CPython's @@ -7874,8 +7880,8 @@ impl Interpreter { /// `SEND`/`YIELD_VALUE` pair with the delegate at top-of-stack. fn gen_yieldfrom(&self, g: &Rc) -> Object { let state = g.state.borrow(); - if let GeneratorState::Suspended(boxed) = &*state { - if let Some(frame) = boxed.downcast_ref::() { + if let GeneratorState::Suspended(frame) = &*state { + { let pc = frame.pc as usize; if pc >= 2 { if let Some(send_ins) = frame.code.instructions.get(pc - 2) { @@ -9320,9 +9326,12 @@ impl Interpreter { // complete before they run. self.flush_lean(frame, shell); self.drain_if_maybe_dead(); - if crate::hot_gates::loop_gen() != snap_gen { - break 'run QuietExit::Yield; - } + } + // The drop may also have queued work of its own (an + // unclosed file's ResourceWarning), which CPython + // delivers at the instruction that dropped it. + if crate::hot_gates::loop_gen() != snap_gen { + break 'run QuietExit::Yield; } continue; } @@ -9733,7 +9742,7 @@ impl Interpreter { } _ => return None, }; - let gf = boxed.downcast_ref::()?; + let gf: &Frame = boxed; if gf.py_frame.is_some() || !gf.saved_exc_info.is_empty() || gf.pc == 0 @@ -9759,9 +9768,7 @@ impl Interpreter { else { unreachable!("checked above"); }; - let gf = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let gf: &mut Frame = &mut boxed; // A native activation parked at the yield is rebuilt into the // interpreted suspension resumed here. #[cfg(feature = "jit")] @@ -9846,9 +9853,7 @@ impl Interpreter { } Ok(FrameOutcome::Returned(_)) => { *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("a generator activation's frame"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(&gen); @@ -9863,9 +9868,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(&gen, err); - let frame = boxed - .downcast_mut::() - .expect("a generator activation's frame"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(&gen); @@ -12161,6 +12164,21 @@ impl Interpreter { last = pc; pc += 1; } + // Container reads, stores and comprehension appends run + // out of line: arms added here cost the rest of the loop + // its register allocation. + OpCode::BinarySubscr | OpCode::StoreSubscr | OpCode::ListAppend => { + // SAFETY: the `len` slots at `base` are initialized, + // and the helper touches nothing else. + match unsafe { Self::core_container_op(ins, base, len) } { + Some(n) => { + len = n; + last = pc; + pc += 1; + } + None => break None, + } + } OpCode::CompareOp => { if len < 2 { break None; @@ -13527,6 +13545,160 @@ impl Interpreter { true } + /// The core loop's container instructions: `seq[i]` and `d[key]` over + /// exact containers (an in-range index or a present `str`/`int` key), + /// `seq[i] = v` in range and `d[key] = v`, and a comprehension's + /// `LIST_APPEND` — each only when every release it makes is a scalar or + /// a plain decrement. Returns the new stack length, or `None`, having + /// touched nothing, for the full leaf arms. + /// + /// # Safety + /// + /// The `len` slots at `base` are initialized operand stack entries. + #[inline(never)] + unsafe fn core_container_op( + ins: weavepy_compiler::Instruction, + base: *mut Object, + len: usize, + ) -> Option { + fn scalar(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) + } + match ins.op { + OpCode::BinarySubscr => { + if len < 2 { + return None; + } + // SAFETY: `len >= 2`. + let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + if !Self::core_droppable(c) { + return None; + } + let r = match (c, k) { + (Object::List(xs), Object::Int(i)) => { + let xs = xs.try_borrow().ok()?; + let n = xs.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n { + return None; + } + clone_hot(&xs[i as usize]) + } + (Object::Tuple(t), Object::Int(i)) => { + let n = t.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n { + return None; + } + clone_hot(&t[i as usize]) + } + (Object::Dict(d), Object::Str(_) | Object::Int(_)) => { + let probe = crate::object::LeafProbe::new(k)?; + let d = d.try_borrow().ok()?; + clone_hot(d.get(&probe)?) + } + _ => return None, + }; + // SAFETY: both operand slots are initialized. The key is a + // scalar or string and the container a shared value + // (`core_droppable`), so neither release runs code; the + // result takes the container's slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + base.add(len - 2).write(r); + } + Some(len - 1) + } + OpCode::StoreSubscr => { + if len < 3 { + return None; + } + // SAFETY: `len >= 3`. + let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + if !Self::core_droppable(c) { + return None; + } + let displaced_ok = |old: &Object| scalar(old) || Self::core_droppable(old); + let old = match (c, k) { + (Object::List(xs), Object::Int(i)) => { + let mut xs = xs.try_borrow_mut().ok()?; + let n = xs.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n || !displaced_ok(&xs[i as usize]) { + return None; + } + // SAFETY: the value slot is initialized and leaves the + // stack into the list. + std::mem::replace(&mut xs[i as usize], unsafe { base.add(len - 3).read() }) + } + (Object::Dict(cell), Object::Str(_) | Object::Int(_)) => { + if crate::capi_watchers::dicts_active() { + return None; + } + let probe = crate::object::LeafProbe::new(k)?; + let mut d = cell.try_borrow_mut().ok()?; + let (old, changed) = match d.get_mut(&probe) { + Some(slot) => { + if !displaced_ok(slot) { + return None; + } + // SAFETY: as above. + let old = + std::mem::replace(slot, unsafe { base.add(len - 3).read() }); + let changed = !old.is_same(slot); + (old, changed) + } + None => { + if !probe.miss_is_exact() { + return None; + } + // SAFETY: as above. + d.insert(DictKey(clone_hot(k)), unsafe { + base.add(len - 3).read() + }); + (Object::None, true) + } + }; + drop(d); + if changed { + crate::object::dict_mutation_event(cell); + } + old + } + _ => return None, + }; + // SAFETY: the key and container slots are initialized (the + // value slot was moved out above); each release is a scalar + // or a plain decrement. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + drop_hot(old); + Some(len - 3) + } + OpCode::ListAppend => { + let depth = ins.arg as usize; + if len < 2 || depth == 0 || depth >= len { + return None; + } + // SAFETY: `depth < len`, so both slots are initialized. + let Object::List(lst) = (unsafe { &*base.add(len - 1 - depth) }) else { + return None; + }; + let mut l = lst.try_borrow_mut().ok()?; + // SAFETY: the value leaves the stack into the list. + l.push(unsafe { base.add(len - 1).read() }); + Some(len - 1) + } + _ => None, + } + } + /// Whether the core loop may release `v` with a plain drop: a scalar, /// a string (freeing one runs no code), or a shared heap value whose /// release is a bare decrement the collector need not hear about (the @@ -16472,16 +16644,17 @@ impl Interpreter { let v = std::mem::replace(value_slot, Object::Unbound); std::mem::replace(slot, v) } - (Object::Dict(d), key @ (Object::Str(_) | Object::Int(_))) => { + (Object::Dict(cell), key @ (Object::Str(_) | Object::Int(_))) => { if crate::capi_watchers::dicts_active() { break; } let Some(probe) = crate::object::LeafProbe::new(key) else { break; }; - let Ok(mut d) = d.try_borrow_mut() else { break }; - let d = &mut *d; - match d.get_mut(&probe) { + let Ok(mut d) = cell.try_borrow_mut() else { + break; + }; + let (old, changed) = match d.get_mut(&probe) { Some(slot) => { if Self::local_needs_prompt_reap(slot) && Self::looks_reapable_temporary(slot) @@ -16489,17 +16662,24 @@ impl Interpreter { break; } let v = std::mem::replace(value_slot, Object::Unbound); - std::mem::replace(slot, v) + let old = std::mem::replace(slot, v); + let changed = !old.is_same(slot); + (old, changed) } None => { - if d.len() > 8 || !d.keys().all(|k| k.0.is_gc_atomic()) { + if !probe.miss_is_exact() { break; } let v = std::mem::replace(value_slot, Object::Unbound); d.insert(DictKey(key.clone()), v); - Object::Unbound + (Object::Unbound, true) } + }; + drop(d); + if changed { + crate::object::dict_mutation_event(cell); } + old } _ => break, }; @@ -29800,45 +29980,9 @@ impl Interpreter { Err(e) => return Err(e), } } - // Native ABCs (e.g. the `io` `IOBase`/`RawIOBase`/`BufferedIOBase`/ - // `TextIOBase` family) record virtual subclasses in a `_abc_registry` - // set via `register()`. CPython's `ABCMeta.__subclasscheck__` honours - // that registry; the native isinstance path must too, otherwise - // `isinstance(_pyio.BufferedReader(...), io.IOBase)` — the layered `io` - // module's classes are the registered `_pyio` ones — wrongly returns - // False. Match if `real` is, or descends from, any registered class. - if Self::class_is_abc_registered(cls, &real) { - return Ok(Object::Bool(true)); - } Ok(Object::Bool(false)) } - /// Whether `candidate` is registered against the native ABC `cls` via its - /// `_abc_registry` set (directly, or as a descendant of a registered - /// virtual subclass). Mirrors one level of CPython's - /// `ABCMeta.__subclasscheck__` registry walk — sufficient for the `io` ABC - /// family, where `_pyio` registers each of `IOBase`/`RawIOBase`/ - /// `BufferedIOBase`/`TextIOBase`. - fn class_is_abc_registered(cls: &Rc, candidate: &Rc) -> bool { - let reg = cls - .dict - .borrow() - .get(&crate::object::DictKey(Object::from_static( - "_abc_registry", - ))) - .cloned(); - if let Some(Object::Set(s)) = reg { - for k in s.borrow().iter() { - if let Object::Type(r) = &k.0 { - if Rc::ptr_eq(r, candidate) || candidate.is_subclass_of(r) { - return true; - } - } - } - } - false - } - /// `issubclass(cls, classinfo)` — same protocol as /// [`do_isinstance_call`] but for class membership. fn do_issubclass_call( @@ -30237,7 +30381,7 @@ impl Interpreter { if !matches!(obj, Object::Instance(_)) { return None; } - if !crate::object::instance_has_custom_dunder(obj, "__hash__") { + if !crate::object::instance_has_custom_dunder(obj, crate::types::Dunder::Hash) { return None; } let globals = self.builtins.clone(); @@ -30701,10 +30845,10 @@ impl Interpreter { Ok(Object::None) } - /// Stable sort over `items`. With `key`, every element is mapped - /// through it once and the results are sorted alongside the - /// originals (decorate-sort-undecorate). Errors from the key - /// function propagate. + /// `list.sort`: a stable sort of `items`, by `key(item)` when a key + /// function is given (called once per item), descending when `reverse` + /// (still stable, as CPython does it: reverse, sort, reverse). On a + /// comparison error `items` holds a permutation of its input. fn sort_with_key( &mut self, items: &mut Vec, @@ -30712,112 +30856,103 @@ impl Interpreter { reverse: bool, globals: &Rc>, ) -> Result<(), RuntimeError> { - // `key=None` is CPython's spelling of "no key function" (sort by the - // elements themselves) — `sorted(xs, key=None)` and `xs.sort(key=None)` - // are identity sorts, not "call `None` on every element". Callers thread - // the raw kwarg through, so collapse the `None` sentinel here. - let key_fn = match key_fn { - Some(Object::None) => None, - other => other, - }; - // CPython's `reverse=True` is *tie-stable*: equal elements keep - // their original relative order (list.sort reverses the slice - // before and after sorting). A post-sort `.reverse()` alone would - // flip ties — observable in `heapq.nlargest`/`Counter.most_common`. - if let Some(f) = key_fn { - let mut decorated: Vec<(Object, Object)> = Vec::with_capacity(items.len()); - for item in items.iter() { - let k = self.call(f, std::slice::from_ref(item), &[], globals)?; - decorated.push((k, item.clone())); - } - if reverse { - decorated.reverse(); - } - if decorated.iter().any(|(k, _)| sort_key_needs_dunder_lt(k)) { - decorated = - merge_sort_by_pylt(self, decorated, &|p: &(Object, Object)| &p.0, globals)?; - } else { - // Uncomparable keys must surface the `TypeError` (CPython - // propagates the failed `<`), not silently compare equal. - let mut err: Option = None; - decorated.sort_by(|a, b| match a.0.cmp(&b.0) { - Ok(o) => o, - Err(e) => { - if err.is_none() { - err = Some(e); - } - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } - } + // `key=None` is CPython's spelling of "no key function". + let Some(f) = key_fn.filter(|f| !matches!(f, Object::None)) else { if reverse { - decorated.reverse(); - } - // Undecorate, then promptly reap the dying key objects: CPython - // decrefs the keys inside `list.sort` while the list is still - // detached, so a key `__del__` that mutates the list is caught - // by the caller's mutation check - // (test_sort.test_key_with_mutating_del). - let mut dead_keys: Vec = Vec::with_capacity(decorated.len()); - *items = decorated - .into_iter() - .map(|(k, v)| { - dead_keys.push(k); - v - }) - .collect(); - for k in dead_keys { - if matches!( - k, - Object::Instance(_) - | Object::Generator(_) - | Object::Coroutine(_) - | Object::AsyncGenerator(_) - ) && Self::is_refcount_dead(&k, 1) - { - self.reap_dead_subgraph(k); - } + items.reverse(); } - } else { + let result = self.sort_keyed(items, |o| o, globals); if reverse { items.reverse(); } - if items.iter().any(sort_key_needs_dunder_lt) { - // Keep a (cheap, Rc-clone) backup: `merge_sort_by_pylt` - // consumes the vec, and a raising `__lt__` must not empty - // the caller's list (CPython reattaches the partially - // sorted array on error). - let backup = items.clone(); - match merge_sort_by_pylt(self, std::mem::take(items), &|o: &Object| o, globals) { - Ok(sorted) => *items = sorted, - Err(e) => { - *items = backup; - return Err(e); + return result; + }; + let mut decorated: Vec<(Object, Object)> = Vec::with_capacity(items.len()); + for item in items.iter() { + let k = self.call(f, std::slice::from_ref(item), &[], globals)?; + decorated.push((k, item.clone())); + } + if reverse { + decorated.reverse(); + } + let result = self.sort_keyed(&mut decorated, |p| &p.0, globals); + if reverse { + decorated.reverse(); + } + // Undecorate, then promptly reap the dying key objects: CPython + // decrefs the keys inside `list.sort` while the list is still + // detached, so a key `__del__` that mutates the list is caught by + // the caller's mutation check (test_sort.test_key_with_mutating_del). + let mut dead_keys: Vec = Vec::with_capacity(decorated.len()); + *items = decorated + .into_iter() + .map(|(k, v)| { + dead_keys.push(k); + v + }) + .collect(); + for k in dead_keys { + if matches!( + k, + Object::Instance(_) + | Object::Generator(_) + | Object::Coroutine(_) + | Object::AsyncGenerator(_) + ) && Self::is_refcount_dead(&k, 1) + { + self.reap_dead_subgraph(k); + } + } + result + } + + /// Sort `v` stably by the keys `key` projects, with the comparison + /// CPython's `list.sort` pre-sort check picks for the keys' types: a + /// native one for `str`, `int` or `float` keys (and for tuples whose + /// first items are one of those), else Python's `<`. + fn sort_keyed( + &mut self, + v: &mut [T], + key: impl Fn(&T) -> &Object + Copy, + globals: &Rc>, + ) -> Result<(), RuntimeError> { + match SortKind::of(v.iter().map(key)) { + SortKind::Str => crate::timsort::sort(v, |a, b| Ok(SortKind::str_lt(key(a), key(b)))), + SortKind::Int => crate::timsort::sort(v, |a, b| Ok(SortKind::int_lt(key(a), key(b)))), + SortKind::Float => { + crate::timsort::sort(v, |a, b| Ok(SortKind::float_lt(key(a), key(b)))) + } + SortKind::Numeric => { + crate::timsort::sort(v, |a, b| compare_op(key(a), key(b), CompareKind::Lt)) + } + SortKind::Tuples(first) => crate::timsort::sort(v, |a, b| { + let (Object::Tuple(x), Object::Tuple(y)) = (key(a), key(b)) else { + return self.dispatch_compare_op(key(a), key(b), CompareKind::Lt, globals); + }; + // The first differing position decides; only it is + // compared with `<` (CPython `unsafe_tuple_compare`). + let mut i = 0; + while i < x.len() && i < y.len() { + if !x[i].is_same(&y[i]) && !self.vm_eq(&x[i], &y[i], globals)? { + break; } + i += 1; } - } else { - let mut err: Option = None; - items.sort_by(|a, b| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - if err.is_none() { - err = Some(e); - } - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); + if i >= x.len() || i >= y.len() { + return Ok(x.len() < y.len()); } - } - if reverse { - items.reverse(); - } + match (i, first) { + (0, SortKindScalar::Str) => Ok(SortKind::str_lt(&x[0], &y[0])), + (0, SortKindScalar::Int) => Ok(SortKind::int_lt(&x[0], &y[0])), + (0, SortKindScalar::Float) => Ok(SortKind::float_lt(&x[0], &y[0])), + (0, SortKindScalar::Numeric) => compare_op(&x[0], &y[0], CompareKind::Lt), + _ => self.dispatch_compare_op(&x[i], &y[i], CompareKind::Lt, globals), + } + }), + SortKind::General => crate::timsort::sort(v, |a, b| { + self.dispatch_compare_op(key(a), key(b), CompareKind::Lt, globals) + }), } - Ok(()) } /// Run `__str__` on instances, falling back to `__repr__` then @@ -32016,10 +32151,8 @@ impl Interpreter { // CPython GET_AWAITABLE: a coroutine that is suspended // inside its own `await` (it has a yield-from // sub-iterator) cannot gain a second awaiter. - let busy = matches!(&*g.state.borrow(), GeneratorState::Suspended(boxed) - if boxed - .downcast_ref::() - .is_some_and(|f| detect_yield_from_subiter(f).is_some())); + let busy = matches!(&*g.state.borrow(), GeneratorState::Suspended(frame) + if detect_yield_from_subiter(frame).is_some()); if busy { return Err(crate::error::runtime_error( "coroutine is being awaited already", @@ -33316,8 +33449,7 @@ impl Interpreter { // as "already running" in CPython's `ag_running_async` sense. let mid_await = matches!( &*g.state.borrow(), - GeneratorState::Suspended(boxed) - if boxed.downcast_ref::().is_some_and(|f| !f.agen_yielded_value) + GeneratorState::Suspended(frame) if !frame.agen_yielded_value ) || matches!(&*g.state.borrow(), GeneratorState::Running); if !mid_await { return None; @@ -33443,9 +33575,7 @@ impl Interpreter { /// (e.g. the generator already finished). fn agen_yielded_a_value(g: &Rc) -> bool { match &*g.state.borrow() { - GeneratorState::Suspended(boxed) => boxed - .downcast_ref::() - .is_none_or(|f| f.agen_yielded_value), + GeneratorState::Suspended(frame) => frame.agen_yielded_value, _ => true, } } @@ -33621,9 +33751,7 @@ impl Interpreter { ) -> Result { let prev_state = std::mem::replace(&mut *gen.state.borrow_mut(), GeneratorState::Running); let mut frame = match prev_state { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => *boxed - .downcast::() - .map_err(|_| RuntimeError::Internal("generator frame downcast".to_owned()))?, + GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => *boxed, GeneratorState::Finished => { *gen.state.borrow_mut() = GeneratorState::Finished; // bpo-25887: throwing into an exhausted coroutine is a @@ -33932,9 +34060,7 @@ impl Interpreter { if !created && crate::trace::any_observers_active() { return false; } - let Some(frame) = boxed.downcast_mut::() else { - return false; - }; + let frame: &mut Frame = boxed; // A Python-visible frame keeps showing the locals (the general // path leaves them to it). if frame.py_frame.is_some() @@ -33965,10 +34091,8 @@ impl Interpreter { else { unreachable!("checked above"); }; - if let Some(frame) = boxed.downcast_mut::() { - self.reap_dead_frame(frame); - self.recycle_frame_allocs(frame); - } + self.reap_dead_frame(&mut boxed); + self.recycle_frame_allocs(&mut boxed); Self::release_finished_gen(g); drop(boxed); true @@ -34374,7 +34498,7 @@ impl Interpreter { } _ => return None, }; - let frame = boxed.downcast_ref::()?; + let frame: &Frame = boxed; if frame.py_frame.is_some() || !frame.saved_exc_info.is_empty() || frame.pc == 0 @@ -34407,9 +34531,7 @@ impl Interpreter { unreachable!("checked above"); }; let outcome = { - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; // A native activation parked at the yield (the generator's // first run entered compiled code) is rebuilt into the // interpreted suspension this path resumes. @@ -34500,9 +34622,7 @@ impl Interpreter { } Ok(FrameOutcome::Returned(v)) => { *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -34517,9 +34637,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(gen, err); - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -34546,12 +34664,10 @@ impl Interpreter { gc_trace::untrack_id(id); } - fn park_suspended_boxed(gen: &Rc, boxed: Box) { - if let Some(frame) = boxed.downcast_ref::() { - if let Some(py) = &frame.py_frame { - if py.gen_owner.borrow().is_none() { - *py.gen_owner.borrow_mut() = Some(Rc::downgrade(gen)); - } + fn park_suspended_boxed(gen: &Rc, boxed: Box) { + if let Some(py) = &boxed.py_frame { + if py.gen_owner.borrow().is_none() { + *py.gen_owner.borrow_mut() = Some(Rc::downgrade(gen)); } } *gen.state.borrow_mut() = GeneratorState::Suspended(boxed); @@ -34590,11 +34706,6 @@ impl Interpreter { ))); } }; - if boxed.downcast_ref::().is_none() { - return Err(RuntimeError::Internal( - "generator frame downcast".to_owned(), - )); - } // On the first call, `sent` must be None (or omitted). if first_resume && !matches!(sent, Object::None) { Self::park_suspended_boxed(gen, boxed); @@ -34604,9 +34715,7 @@ impl Interpreter { ))); } let outcome = { - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; // PEP 667: writes made through the suspended frame's `f_locals` // take effect when the generator resumes. Self::apply_py_frame_locals_writes(frame); @@ -34658,9 +34767,7 @@ impl Interpreter { // string we get from `from_builtin("StopIteration", // "")`. *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); // RFC 0059 WS4: the exhausted frame's storage is dead — // donate it back to the frame pools. @@ -34677,9 +34784,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(gen, err); - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -46941,9 +47046,7 @@ impl Interpreter { /// without eager `PyFrame` construction. fn set_frame_gen_owner(gen: &Rc) { if let GeneratorState::Created(boxed) = &mut *gen.state.borrow_mut() { - if let Some(fr) = boxed.downcast_mut::() { - fr.gen_owner = Some(Rc::downgrade(gen)); - } + boxed.gen_owner = Some(Rc::downgrade(gen)); } } @@ -54890,7 +54993,7 @@ struct InlineAct { /// its own boxed frame runs instead of the slot's (which stays /// parked); `gen_frame` points into `gen_box`. gen: Option>, - gen_box: Option>, + gen_box: Option>, gen_frame: *mut Frame, /// The resuming `FOR_ITER`'s jump distance (the exhaustion exit). exhaust_arg: u32, @@ -55830,6 +55933,9 @@ static CORE_LEAF_OPS: [bool; 256] = { OpCode::StoreFast, OpCode::PopTop, OpCode::BinaryOp, + OpCode::BinarySubscr, + OpCode::StoreSubscr, + OpCode::ListAppend, OpCode::CompareOp, OpCode::PopJumpIfFalse, OpCode::PopJumpIfTrue, @@ -56648,109 +56754,95 @@ fn oserror_subclass_name(errno: i64) -> Option<&'static str> { Some(name) } -fn sort_key_needs_dunder_lt(o: &Object) -> bool { - sort_key_needs_dunder_lt_depth(o, 0) +/// The comparison `list.sort` uses for a list of keys, chosen by the +/// CPython pre-sort check: keys of one native type compare natively. +#[derive(Clone, Copy, PartialEq, Eq)] +enum SortKind { + Str, + Int, + Float, + /// Mixed `bool`, `int` and `float` keys. + Numeric, + /// Non-empty tuples whose first items have the given scalar kind. + Tuples(SortKindScalar), + General, } -/// Whether a sort key must be ordered through Python's `<` (rich-comparison -/// dispatch) rather than the fast Rust `Object::cmp` total order. True for a -/// custom instance carrying a `__lt__`, *and* — crucially — for a `tuple`/ -/// `list` that (recursively) contains one: the Rust `seq_cmp` path bottoms -/// out in `Object::cmp`, which has no case for user instances and would -/// raise a spurious "'<' not supported" (e.g. `pprint`'s `_safe_tuple` keys, -/// which wrap each dict key/value in a `__lt__`-only `_safe_key`). The depth -/// guard avoids unbounded recursion on self-referential containers. -fn sort_key_needs_dunder_lt_depth(o: &Object, depth: u32) -> bool { - if depth > 100 { - return false; - } - match o { - Object::Instance(inst) => { - // A user-defined Python `__lt__`/bound method always needs the - // Python `<` path (the historical case: `pprint`'s `_safe_key`). - if matches!( - inst.cls().lookup("__lt__"), - Some(Object::Function(_) | Object::BoundMethod(_)) - ) { - return true; - } - // CPython's `list.sort`/`sorted` *always* order through `<` - // (`PyObject_RichCompareBool(x, y, Py_LT)`); WeavePy's native - // `Object::cmp` is only a safe fast path when it yields the - // identical answer. That holds for a plain value subclass - // (`class E(int)`) — ordered by its `native_value()` payload — - // but NOT for a Cython/C extension class surfaced as an - // `Object::Instance` (pandas `Period`/`Timestamp`, - // `decimal.Decimal`) whose `__lt__` is a C slot wrapper - // (`builtin_function_or_method`) and which carries no native - // scalar for `Object::cmp` to order. Detect exactly that: an - // ordering dunder is present, yet `native_value()` is `None`, so - // the native path would raise a spurious "'<' not supported". - if o.native_value().is_none() - && (inst.cls().lookup("__lt__").is_some() || inst.cls().lookup("__gt__").is_some()) - { - return true; +/// The scalar [`SortKind`]s, for the first items of tuple keys. +#[derive(Clone, Copy, PartialEq, Eq)] +enum SortKindScalar { + Str, + Int, + Float, + Numeric, + General, +} + +impl SortKind { + fn of<'a>(keys: impl Iterator + Clone) -> Self { + let mut first = keys.clone().next(); + if let Some(Object::Tuple(t)) = first { + if t.is_empty() { + first = None; } - // A `list`/`tuple` *subclass* instance orders as its native - // payload — recurse into it so contained `__lt__`-bearing - // elements still get Python `<` dispatch - // (test_sort.test_unsafe_object_compare's WackyList1). - match inst.native.get() { - Some(n @ (Object::List(_) | Object::Tuple(_))) => { - sort_key_needs_dunder_lt_depth(n, depth + 1) + } + if matches!(first, Some(Object::Tuple(_))) { + let mut firsts = Vec::new(); + for k in keys { + match k { + Object::Tuple(t) if !t.is_empty() => firsts.push(&t[0]), + _ => return Self::General, } - _ => false, } + return Self::Tuples(SortKindScalar::of(firsts.into_iter())); + } + match SortKindScalar::of(keys) { + SortKindScalar::Str => Self::Str, + SortKindScalar::Int => Self::Int, + SortKindScalar::Float => Self::Float, + SortKindScalar::Numeric => Self::Numeric, + SortKindScalar::General => Self::General, } - // A genuinely foreign extension object (a bare numpy scalar not - // surfaced as an `Object::Instance`, …) orders through its - // `tp_richcompare` slot, which the native `Object::cmp` total order - // cannot reach — so it must sort through Python's `<`. - Object::Foreign(_) => true, - Object::Tuple(items) => items - .iter() - .any(|x| sort_key_needs_dunder_lt_depth(x, depth + 1)), - Object::List(items) => items - .borrow() - .iter() - .any(|x| sort_key_needs_dunder_lt_depth(x, depth + 1)), - _ => false, + } + + #[inline] + fn str_lt(a: &Object, b: &Object) -> bool { + // UTF-8 byte order is code point order. + matches!((a, b), (Object::Str(x), Object::Str(y)) if x.as_bytes() < y.as_bytes()) + } + + #[inline] + fn int_lt(a: &Object, b: &Object) -> bool { + matches!((a, b), (Object::Int(x), Object::Int(y)) if x < y) + } + + #[inline] + fn float_lt(a: &Object, b: &Object) -> bool { + matches!((a, b), (Object::Float(x), Object::Float(y)) if x < y) } } -/// Stable merge sort that orders by Python `<` (full rich-comparison -/// dispatch, reflected operands included). `key` projects the comparison -/// object out of each element (the decorated sort key, or the element -/// itself). Comparison errors (unorderable types, raising `__lt__`) -/// propagate exactly as CPython's `list.sort` does. -fn merge_sort_by_pylt( - interp: &mut Interpreter, - mut v: Vec, - key: &impl Fn(&T) -> &Object, - globals: &Rc>, -) -> Result, RuntimeError> { - if v.len() <= 1 { - return Ok(v); - } - let right = v.split_off(v.len() / 2); - let left = merge_sort_by_pylt(interp, v, key, globals)?; - let right = merge_sort_by_pylt(interp, right, key, globals)?; - let mut out = Vec::with_capacity(left.len() + right.len()); - let (mut li, mut ri) = (0, 0); - while li < left.len() && ri < right.len() { - // Stability: take from the right run only when strictly smaller - // (timsort's `b < a` merge test). - if interp.dispatch_compare_op(key(&right[ri]), key(&left[li]), CompareKind::Lt, globals)? { - out.push(right[ri].clone()); - ri += 1; - } else { - out.push(left[li].clone()); - li += 1; +impl SortKindScalar { + fn of<'a>(keys: impl Iterator) -> Self { + let mut kind = None; + for k in keys { + let this = match k { + Object::Str(_) => Self::Str, + Object::Int(_) => Self::Int, + Object::Float(_) => Self::Float, + Object::Long(_) | Object::Bool(_) => Self::Numeric, + _ => return Self::General, + }; + kind = Some(match kind { + None => this, + Some(k) if k == this => k, + Some(Self::Str) | Some(Self::General) => return Self::General, + Some(_) if this == Self::Str => return Self::General, + Some(_) => Self::Numeric, + }); } + kind.unwrap_or(Self::General) } - out.extend_from_slice(&left[li..]); - out.extend_from_slice(&right[ri..]); - Ok(out) } /// Convert a freshly built `set` result into a `frozenset` — used when diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index a36ab67e..feb7446a 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -3052,11 +3052,16 @@ impl indexmap::Equivalent for LeafNameProbe<'_> { /// A dict probe for the leaf burst: a `str` or `int` key that matches a /// stored key only by exact native equality. It never runs Python — an /// exotic stored key sharing the bucket (a `__eq__` that could equate) -/// reads as a miss, and the burst leaves every miss to the full path. -#[derive(Debug, Clone, Copy)] +/// reads as a miss, and the burst leaves every miss to the full path +/// unless [`LeafProbe::miss_is_exact`] proves no such key was seen. +#[derive(Debug)] pub struct LeafProbe<'a> { pub key: &'a Object, pub hash: i64, + /// Set when the table compared this probe with a stored key of another + /// kind, whose equality to the probe might need Python (`1.0`, `True`, + /// a `str` subclass, an instance with `__eq__`). + foreign: std::cell::Cell, } impl<'a> LeafProbe<'a> { @@ -3068,7 +3073,21 @@ impl<'a> LeafProbe<'a> { Object::Int(_) => py_hash_value(key)?, _ => return None, }; - Some(Self { key, hash }) + Some(Self { + key, + hash, + foreign: std::cell::Cell::new(false), + }) + } + + /// After a lookup that found nothing: whether no stored key can equal + /// this one, so inserting it natively is exact. Any key that Python + /// would compare with the probe has an equal hash, so the table + /// compared it with the probe, and a key of another kind set + /// `foreign`. + #[inline] + pub fn miss_is_exact(&self) -> bool { + !self.foreign.get() } } @@ -3085,7 +3104,10 @@ impl indexmap::Equivalent for LeafProbe<'_> { match (self.key, &key.0) { (Object::Str(a), Object::Str(b)) => a.as_bytes() == b.as_bytes(), (Object::Int(a), Object::Int(b)) => a == b, - _ => false, + _ => { + self.foreign.set(true); + false + } } } } @@ -3152,74 +3174,37 @@ fn current_interp_eq(a: &Object, b: &Object) -> Option { /// `name` dunder (a real Python `def`, not the inherited identity default). /// Used to gate the reentrant `__eq__` dispatch so plain instances keep the /// native identity fast path. -pub(crate) fn instance_has_custom_dunder(obj: &Object, name: &str) -> bool { +pub(crate) fn instance_has_custom_dunder(obj: &Object, dunder: crate::types::Dunder) -> bool { let Object::Instance(inst) = obj else { return false; }; - match inst.cls().lookup_with_owner(name) { - Some((Object::Function(_) | Object::BoundMethod(_), _)) => true, - Some((Object::None, _)) | None => false, - // A non-function dunder. A user class supplying it (e.g. - // `unittest.mock` installs `Mock` instances as `__hash__` / - // `__eq__` on per-instance subclasses) needs Python dispatch. - Some((_, owner)) if !owner.flags.is_builtin => true, - // A built-in type that *overrides* the dunder in its own dict - // rather than inheriting `object`'s identity default needs Python - // dispatch too: `weakref` compares/hashes by referent, so its - // `ref` objects can't be keyed by the native identity path - // (`test_weakref`/`test_weakset` rely on `ref(a) == ref(a)` and - // matching hashes finding the same set slot). `object`'s own - // `__eq__`/`__hash__` — where every *plain* instance resolves — - // stay native so ordinary objects keep identity semantics, and - // value-wrapping subclasses (`class C(int)`, struct sequences) - // keep their native structural comparison via `native`. - Some((_, owner)) => { - inst.native.get().is_none() - && !Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) - } - } -} - -/// [`instance_has_custom_dunder`]`(obj, "__eq__")`, memoised per class: -/// membership tests and `list.remove`/`index`/`count` ask it for every -/// element they compare. -pub(crate) fn instance_has_custom_eq(obj: &Object) -> bool { - let Object::Instance(inst) = obj else { + let info = inst.class_dunder(dunder); + if !info.present() || info.is_none() { return false; - }; - if crate::gil::free_threading_enabled() { - return custom_eq_of(&inst.cls(), inst); } - // Under the GIL nothing reassigns `__class__` while this reads it - // (the lookup below runs no Python code). - custom_eq_of(inst.cls_raw(), inst) -} - -fn custom_eq_of(cls: &crate::types::TypeObject, inst: &crate::types::PyInstance) -> bool { - let ver = cls.attr_version.get(); - let memo = cls.eq_kind.get(); - let kind = if memo != 0 && memo >> 2 == ver { - memo & 3 - } else { - let kind = match cls.lookup_with_owner("__eq__") { - Some((Object::Function(_) | Object::BoundMethod(_), _)) => 2, - Some((Object::None, _)) | None => 1, - Some((_, owner)) if !owner.flags.is_builtin => 2, - Some((_, owner)) - if Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) => - { - 1 - } - Some(_) => 3, - }; - cls.eq_kind.set(ver << 2 | kind); - kind - }; - match kind { - 1 => false, - 2 => true, - _ => inst.native.get().is_none(), + // A Python function, or any other object a user class supplies (e.g. + // `unittest.mock` installs `Mock` instances as `__hash__` / `__eq__` + // on per-instance subclasses), needs Python dispatch. + if info.is_function() || !info.builtin_owner() { + return true; } + // A built-in type that *overrides* the dunder in its own dict rather + // than inheriting `object`'s identity default needs Python dispatch + // too: `weakref` compares/hashes by referent, so its `ref` objects + // can't be keyed by the native identity path (`test_weakref`/ + // `test_weakset` rely on `ref(a) == ref(a)` and matching hashes + // finding the same set slot). `object`'s own `__eq__`/`__hash__` — + // where every *plain* instance resolves — stay native so ordinary + // objects keep identity semantics, and value-wrapping subclasses + // (`class C(int)`, struct sequences) keep their native structural + // comparison via `native`. + inst.native.get().is_none() && !info.object_owner() +} + +/// [`instance_has_custom_dunder`] for `__eq__`: membership tests and +/// `list.remove`/`index`/`count` ask it for every element they compare. +pub(crate) fn instance_has_custom_eq(obj: &Object) -> bool { + instance_has_custom_dunder(obj, crate::types::Dunder::Eq) } /// `a == b` can only be `object`'s identity default: each side is a @@ -3275,7 +3260,7 @@ pub(crate) fn key_needs_interp_eq(obj: &Object) -> bool { // `_UnionGenericAlias` — structural namespace equality can't // express either (RFC 0076 WS5). Object::SimpleNamespace(_) => crate::is_pep604_union(obj).is_some(), - _ => instance_has_custom_dunder(obj, "__eq__"), + _ => instance_has_custom_dunder(obj, crate::types::Dunder::Eq), } } @@ -3604,11 +3589,8 @@ fn key_eq_defer_active() -> bool { pub(crate) fn dict_key_is_reentrant(key: &Object) -> bool { match key { Object::Instance(inst) => { - let py_dunder = |name: &str| match inst.cls().lookup_with_owner(name) { - Some((Object::None, _)) | None => false, - Some((_, owner)) => !owner.flags.is_builtin, - }; - py_dunder("__eq__") || py_dunder("__hash__") + inst.class_dunder(crate::types::Dunder::Eq).user_defined() + || inst.class_dunder(crate::types::Dunder::Hash).user_defined() } Object::Tuple(items) => items.iter().any(dict_key_is_reentrant), _ => false, @@ -4672,7 +4654,7 @@ impl PyGenerator { qualname: impl Into, kind: CoroutineKind, code: Object, - frame: Box, + frame: Box, ) -> Self { Self { name: RefCell::new(Object::from_str(name.into())), @@ -4694,7 +4676,7 @@ impl PyGenerator { qualname: Object, kind: CoroutineKind, code: Object, - frame: Box, + frame: Box, ) -> Self { Self { name: RefCell::new(name), @@ -4812,9 +4794,9 @@ impl CoroutineKind { pub enum GeneratorState { /// Created but not yet started — body hasn't executed past the /// initial `RETURN_GENERATOR`. - Created(Box), + Created(Box), /// Paused at a `YIELD_VALUE`. - Suspended(Box), + Suspended(Box), /// Body returned (cleanly or via exception). Subsequent /// `next`/`send` raise `StopIteration`. Finished, @@ -11906,7 +11888,7 @@ pub(crate) fn py_hash_value(obj: &Object) -> Option { // A user-defined `__hash__` outranks the wrapped value's hash — // e.g. functools' `_HashedSeq(list)` caches its hash precisely so // the (unhashable) list payload is never consulted. - if instance_has_custom_dunder(obj, "__hash__") { + if instance_has_custom_dunder(obj, crate::types::Dunder::Hash) { // When the instance wraps an *immutable* builtin value (an // `int`/`str`/`tuple`/… subclass), the value can't change, so // a custom `__hash__` is genuinely constant and may be diff --git a/crates/weavepy-vm/src/rc.rs b/crates/weavepy-vm/src/rc.rs index dc67605e..34308d73 100644 --- a/crates/weavepy-vm/src/rc.rs +++ b/crates/weavepy-vm/src/rc.rs @@ -71,8 +71,11 @@ fn strong_word(value: *const T) -> *const AtomicUsize { let header = 2 * std::mem::size_of::(); let offset = (header + align - 1) & !(align - 1); // SAFETY: the payload lives `offset` bytes into its ArcInner, whose - // first field is the strong count. - unsafe { value.cast::().sub(offset).cast::() } + // first field is the strong count, aligned for `AtomicUsize`. + #[allow(clippy::cast_ptr_alignment)] + unsafe { + value.cast::().sub(offset).cast::() + } } /// Add one strong owner of the `Arc` payload at `value`. @@ -174,7 +177,7 @@ impl Rc { pub fn into_arc(this: Self) -> Arc { let this = ManuallyDrop::new(this); // SAFETY: moves the owned Arc out; `this` is never dropped. - unsafe { ptr::read(&*this.0) } + unsafe { ptr::read(&raw const *this.0) } } #[inline] @@ -259,7 +262,7 @@ impl Clone for Rc { // accounts for the added reference. unsafe { increment_strong(Arc::as_ptr(&self.0)); - Self(ManuallyDrop::new(ptr::read(&*self.0))) + Self(ManuallyDrop::new(ptr::read(&raw const *self.0))) } } } diff --git a/crates/weavepy-vm/src/stdlib/abc_mod.rs b/crates/weavepy-vm/src/stdlib/abc_mod.rs index 42774788..62f61b07 100644 --- a/crates/weavepy-vm/src/stdlib/abc_mod.rs +++ b/crates/weavepy-vm/src/stdlib/abc_mod.rs @@ -1,20 +1,106 @@ -//! The `_abc` accelerator module — RFC 0023. +//! The `_abc` accelerator module, a port of CPython's `Modules/_abc.c`. //! -//! Backs `abc.ABCMeta` with the registry of virtual subclasses and -//! the abstractmethod cache. Surface mirrors CPython's `_abc`: -//! `get_cache_token`, `_abc_init`, `_abc_register`, `_abc_instancecheck`, -//! `_abc_subclasscheck`, `_get_dump`, `_reset_registry`, -//! `_reset_caches`. +//! `abc.ABCMeta` delegates registration and the instance and subclass +//! checks here. Each ABC keeps its virtual-subclass registry, positive +//! cache, and negative cache on its [`TypeObject`]. Classes are held +//! weakly, so registering or checking a class never keeps it alive. A +//! foreign extension type has no weak form and is held strongly; such types +//! live for the whole process in practice. -use crate::sync::Rc; -use crate::sync::RefCell; +use crate::sync::{Cell, Rc, RefCell, Weak}; +use std::sync::atomic::{AtomicU64, Ordering}; -use crate::error::RuntimeError; +use crate::error::{assertion_error, runtime_error, type_error, RuntimeError}; use crate::import::ModuleCache; use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; +use crate::types::TypeObject; +use crate::weakref_registry::{id_of, ObjectId}; +use crate::Interpreter; -thread_local! { - static CACHE_TOKEN: RefCell = const { RefCell::new(1) }; +/// CPython's `abc_invalidation_counter`: bumped by every registration, so +/// each ABC's negative cache knows when it may be stale. +static INVALIDATION_COUNTER: AtomicU64 = AtomicU64::new(0); + +/// A class held by an ABC cache or registry. +enum Held { + Type(Weak), + Other(Object), +} + +impl Held { + fn new(class: &Object) -> Self { + match class { + Object::Type(t) => Self::Type(Rc::downgrade(t)), + other => Self::Other(other.clone()), + } + } + + fn get(&self) -> Option { + match self { + Self::Type(t) => t.upgrade().map(Object::Type), + Self::Other(o) => Some(o.clone()), + } + } +} + +/// A set of classes keyed by identity. A weak entry keeps its allocation, +/// so a dead class's identity is never reused while its entry remains. +#[derive(Default)] +struct ClassSet { + entries: indexmap::IndexMap, + /// Length at which dead entries are next pruned. + prune_at: usize, +} + +impl ClassSet { + fn contains(&self, class: &Object) -> bool { + self.entries.contains_key(&id_of(class)) + } + + fn insert(&mut self, class: &Object) { + self.entries + .entry(id_of(class)) + .or_insert_with(|| Held::new(class)); + if self.entries.len() >= self.prune_at.max(64) { + self.entries.retain(|_, held| held.get().is_some()); + self.prune_at = self.entries.len() * 2; + } + } + + fn clear(&mut self) { + self.entries.clear(); + } + + fn live(&self) -> Vec { + self.entries.values().filter_map(Held::get).collect() + } +} + +/// CPython's `_abc_data`: one ABC's registry and caches. +pub struct AbcState { + registry: RefCell, + cache: RefCell, + negative_cache: RefCell, + negative_cache_version: Cell, +} + +impl std::fmt::Debug for AbcState { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("AbcState") + .field("registered", &self.registry.borrow().entries.len()) + .finish_non_exhaustive() + } +} + +impl AbcState { + fn new() -> Self { + Self { + registry: RefCell::default(), + cache: RefCell::default(), + negative_cache: RefCell::default(), + negative_cache_version: Cell::new(INVALIDATION_COUNTER.load(Ordering::Relaxed)), + } + } } pub fn build(_cache: &ModuleCache) -> Rc { @@ -56,181 +142,364 @@ pub fn build(_cache: &ModuleCache) -> Rc { }) } -fn bump_cache() { - CACHE_TOKEN.with(|c| { - let mut g = c.borrow_mut(); - *g = g.wrapping_add(1); - }); +fn interpreter() -> Result<&'static mut Interpreter, RuntimeError> { + let ptr = crate::vm_singletons::current_interpreter_ptr() + .ok_or_else(|| runtime_error("_abc requires a running interpreter"))?; + // SAFETY: published by the enclosing VM frame on this thread, which + // outlives this builtin call. + Ok(unsafe { &mut *ptr }) } -fn abc_get_cache_token(_args: &[Object]) -> Result { - Ok(Object::Int(CACHE_TOKEN.with(|c| *c.borrow() as i64))) +fn args_exact(args: &[Object], name: &str) -> Result<[Object; N], RuntimeError> { + <&[Object; N]>::try_from(args).cloned().map_err(|_| { + type_error(format!( + "{name} expected {N} argument{}, got {}", + if N == 1 { "" } else { "s" }, + args.len() + )) + }) } -fn abc_init(args: &[Object]) -> Result { - // _abc_init(cls) — initialise the registry / cache on the class. - if let Some(Object::Type(cls)) = args.first() { - // CPython's `_abc_init` consumes `__abc_tpflags__`: it validates - // that a class doesn't claim both `Py_TPFLAGS_SEQUENCE` and - // `Py_TPFLAGS_MAPPING`, folds the collection bits into - // `tp_flags` (we keep them under a private dict key that - // `flags_bits` reads), and deletes the public attribute. - const COLLECTION_FLAGS: i64 = (1 << 5) | (1 << 6); - let tpflags = cls - .dict +/// The ABC state of `cls`, inherited through the MRO the way CPython's +/// `_abc_impl` attribute is. +fn state_of(cls: &Object) -> Result, RuntimeError> { + if let Object::Type(t) = cls { + if t.abc_state.get().is_some() { + return Ok(t.clone()); + } + if let Some(owner) = t + .mro .borrow() - .get(&DictKey(Object::from_static("__abc_tpflags__"))) - .cloned(); - if let Some(flags) = tpflags { - if let Some(val) = flags.as_i64() { - if (val & COLLECTION_FLAGS) == COLLECTION_FLAGS { - return Err(crate::error::type_error( - "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING", - )); - } - cls.dict.borrow_mut().insert( - DictKey(Object::from_static("_abc_collection_flags")), - Object::Int(val & COLLECTION_FLAGS), - ); + .iter() + .find(|base| base.abc_state.get().is_some()) + { + return Ok(owner.clone()); + } + } + Err(type_error("_abc_impl is set to a wrong type")) +} + +fn state(owner: &TypeObject) -> &AbcState { + owner.abc_state.get().expect("checked by state_of") +} + +fn is_class(interp: &mut Interpreter, obj: &Object) -> Result { + match obj { + Object::Type(_) => Ok(true), + Object::Foreign(_) => { + let type_type = Object::Type(crate::builtin_types::builtin_types().type_.clone()); + interp.isinstance_public(obj, &type_type) + } + _ => Ok(false), + } +} + +fn call_method( + interp: &mut Interpreter, + receiver: &Object, + name: &str, + args: &[Object], +) -> Result { + let method = interp.load_attr_public(receiver, name)?; + let globals = interp.builtins_dict(); + interp.call(&method, args, &[], &globals) +} + +/// `getattr(obj, name, None)`, propagating everything but `AttributeError`. +fn attr_or_none( + interp: &mut Interpreter, + obj: &Object, + name: &str, +) -> Result, RuntimeError> { + match interp.load_attr_public(obj, name) { + Ok(v) => Ok(Some(v)), + Err(e) if interp.is_attribute_error(&e) => Ok(None), + Err(e) => Err(e), + } +} + +/// CPython `_PyObject_IsAbstract`. +fn is_abstract(interp: &mut Interpreter, obj: &Object) -> Result { + match attr_or_none(interp, obj, "__isabstractmethod__")? { + Some(flag) => interp.op_truth(&flag), + None => Ok(false), + } +} + +fn abc_get_cache_token(args: &[Object]) -> Result { + args_exact::<0>(args, "get_cache_token")?; + Ok(Object::Int( + INVALIDATION_COUNTER.load(Ordering::Relaxed) as i64 + )) +} + +/// `_abc_init(cls)`: compute `__abstractmethods__`, fold +/// `__abc_tpflags__`, and give the class fresh caches. +fn abc_init(args: &[Object]) -> Result { + let [cls] = args_exact::<1>(args, "_abc_init")?; + let interp = interpreter()?; + compute_abstract_methods(interp, &cls)?; + let Object::Type(t) = &cls else { + return Ok(Object::None); + }; + // CPython's `_abc_init` consumes `__abc_tpflags__`: a class may not + // claim both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING. The collection + // bits are kept under a private key that `flags_bits` and pattern + // matching read. + const COLLECTION_FLAGS: i64 = (1 << 5) | (1 << 6); + let key = DictKey(Object::from_static("__abc_tpflags__")); + let tpflags = t.dict.borrow().get(&key).cloned(); + if let Some(flags) = tpflags { + if let Some(val) = flags.as_i64() { + if (val & COLLECTION_FLAGS) == COLLECTION_FLAGS { + return Err(type_error( + "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING", + )); } - cls.dict - .borrow_mut() - .shift_remove(&DictKey(Object::from_static("__abc_tpflags__"))); - } - let mut td = cls.dict.borrow_mut(); - td.insert( - DictKey(Object::from_static("_abc_registry")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_cache")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_negative_cache")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_negative_cache_version")), - Object::Int(CACHE_TOKEN.with(|c| *c.borrow() as i64)), - ); + t.dict.borrow_mut().insert( + DictKey(Object::from_static("_abc_collection_flags")), + Object::Int(val & COLLECTION_FLAGS), + ); + } + t.dict.borrow_mut().shift_remove(&key); + } + match t.abc_state.get() { + Some(existing) => { + existing.registry.borrow_mut().clear(); + existing.cache.borrow_mut().clear(); + existing.negative_cache.borrow_mut().clear(); + existing + .negative_cache_version + .set(INVALIDATION_COUNTER.load(Ordering::Relaxed)); + } + None => { + let _ = t.abc_state.set(Box::new(AbcState::new())); + } } Ok(Object::None) } -fn abc_register(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let Object::Type(t) = &cls { - if let Some(Object::Set(reg)) = t +/// CPython `compute_abstract_methods`. +fn compute_abstract_methods(interp: &mut Interpreter, cls: &Object) -> Result<(), RuntimeError> { + let mut abstracts: Vec = Vec::new(); + // Stage 1: the class's own abstract methods. + let own: Vec<(Object, Object)> = match cls { + Object::Type(t) => t .dict .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - { - reg.borrow_mut().insert(DictKey(sub.clone())); + .iter() + .map(|(k, v)| (k.0.clone(), v.clone())) + .collect(), + _ => Vec::new(), + }; + for (name, value) in own { + if is_abstract(interp, &value)? { + abstracts.push(name); + } + } + // Stage 2: inherited abstract methods the class didn't override. + let bases = interp.load_attr_public(cls, "__bases__")?; + let Object::Tuple(bases) = bases else { + return Err(type_error("__bases__ is not tuple")); + }; + let globals = interp.builtins_dict(); + for base in bases.iter() { + let Some(inherited) = attr_or_none(interp, base, "__abstractmethods__")? else { + continue; + }; + for name in interp.collect_iterable(&inherited, &globals)? { + let Object::Str(s) = &name else { + continue; + }; + let Some(value) = attr_or_none(interp, cls, s)? else { + continue; + }; + if is_abstract(interp, &value)? && !abstracts.iter().any(|a| a.is_same(&name)) { + abstracts.push(name); + } } } - // CPython's `_abc_register` copies the ABC's collection flag - // (Py_TPFLAGS_SEQUENCE / Py_TPFLAGS_MAPPING) onto the registered - // class so late registration makes match patterns work (PEP 634). - if let (Some(Object::Type(t)), Object::Type(s)) = (args.first(), &sub) { - let flag = t.collection_flags(); - if flag != 0 { - s.dict.borrow_mut().insert( + interp.store_attr_public( + cls, + "__abstractmethods__", + Object::new_frozenset_from(abstracts), + ) +} + +/// `_abc_register(cls, subclass)`. +fn abc_register(args: &[Object]) -> Result { + let [cls, subclass] = args_exact::<2>(args, "_abc_register")?; + let interp = interpreter()?; + if !is_class(interp, &subclass)? { + return Err(type_error("Can only register classes")); + } + if interp.issubclass_public(&subclass, &cls)? { + return Ok(subclass); // Already a subclass. + } + // Test for cycles *after* testing for "already a subclass", so that + // `X.register(X)` is a no-op. + if interp.issubclass_public(&cls, &subclass)? { + return Err(runtime_error("Refusing to create an inheritance cycle")); + } + let owner = state_of(&cls)?; + state(&owner).registry.borrow_mut().insert(&subclass); + INVALIDATION_COUNTER.fetch_add(1, Ordering::Relaxed); + // Late registration on a Sequence or Mapping ABC must make pattern + // matching treat the class accordingly (CPython sets the type flag + // recursively; the VM reads this marker through the MRO). + if let (Object::Type(abc), Object::Type(sub)) = (&cls, &subclass) { + let flag = abc.collection_flags(); + if flag != 0 && !sub.flags.is_builtin { + sub.dict.borrow_mut().insert( DictKey(Object::from_static("_abc_collection_flags")), Object::Int(flag), ); } } - bump_cache(); - Ok(sub) + Ok(subclass) } +/// `_abc_instancecheck(cls, instance)`. fn abc_instancecheck(args: &[Object]) -> Result { - // Delegates to issubclass(type(obj), cls) — the Python wrapper - // dispatches the full protocol. - let cls = args.first().cloned().unwrap_or(Object::None); - let inst = args.get(1).cloned().unwrap_or(Object::None); - if let (Object::Type(t), Object::Instance(i)) = (&cls, &inst) { - if i.cls().is_subclass_of(t) { - return Ok(Object::Bool(true)); - } - if let Some(Object::Set(reg)) = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() + let [cls, instance] = args_exact::<2>(args, "_abc_instancecheck")?; + let interp = interpreter()?; + let owner = state_of(&cls)?; + let subclass = interp.load_attr_public(&instance, "__class__")?; + if state(&owner).cache.borrow().contains(&subclass) { + return Ok(Object::Bool(true)); + } + let subtype = Object::Type(crate::builtins::class_of(&instance)); + if subtype.is_same(&subclass) { + let data = state(&owner); + if data.negative_cache_version.get() == INVALIDATION_COUNTER.load(Ordering::Relaxed) + && data.negative_cache.borrow().contains(&subclass) { - for entry in reg.borrow().iter() { - if let Object::Type(et) = &entry.0 { - if i.cls().is_subclass_of(et) { - return Ok(Object::Bool(true)); - } - } - } + return Ok(Object::Bool(false)); } + return call_method(interp, &cls, "__subclasscheck__", &[subclass]); } - Ok(Object::Bool(false)) + let result = call_method(interp, &cls, "__subclasscheck__", &[subclass])?; + if interp.op_truth(&result)? { + return Ok(result); + } + call_method(interp, &cls, "__subclasscheck__", &[subtype]) } +/// `_abc_subclasscheck(cls, subclass)`. fn abc_subclasscheck(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let (Object::Type(t), Object::Type(st)) = (&cls, &sub) { - if st.is_subclass_of(t) { - return Ok(Object::Bool(true)); + let [cls, subclass] = args_exact::<2>(args, "_abc_subclasscheck")?; + let interp = interpreter()?; + if !is_class(interp, &subclass)? { + return Err(type_error("issubclass() arg 1 must be a class")); + } + let owner = state_of(&cls)?; + let data = state(&owner); + // 1. The positive cache. + if data.cache.borrow().contains(&subclass) { + return Ok(Object::Bool(true)); + } + // 2. The negative cache, invalidated by any registration since. + let counter = INVALIDATION_COUNTER.load(Ordering::Relaxed); + if data.negative_cache_version.get() < counter { + data.negative_cache.borrow_mut().clear(); + data.negative_cache_version.set(counter); + } else if data.negative_cache.borrow().contains(&subclass) { + return Ok(Object::Bool(false)); + } + let found = |data: &AbcState| { + data.cache.borrow_mut().insert(&subclass); + Ok(Object::Bool(true)) + }; + // 3. The subclass hook. + let ok = call_method( + interp, + &cls, + "__subclasshook__", + std::slice::from_ref(&subclass), + )?; + match ok { + Object::Bool(true) => return found(data), + Object::Bool(false) => { + data.negative_cache.borrow_mut().insert(&subclass); + return Ok(Object::Bool(false)); } - if let Some(Object::Set(reg)) = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - { - for entry in reg.borrow().iter() { - if let Object::Type(et) = &entry.0 { - if st.is_subclass_of(et) { - return Ok(Object::Bool(true)); - } - } - } + ref other if crate::vm_singletons::is_not_implemented(other) => {} + _ => { + return Err(assertion_error( + "__subclasshook__ must return either False, True, or NotImplemented", + )) + } + } + // 4. A direct subclass. + let direct = match (&subclass, &cls) { + (Object::Type(sub), Object::Type(abc)) => sub.is_subclass_of(abc), + _ => match attr_or_none(interp, &subclass, "__mro__")? { + Some(Object::Tuple(mro)) => mro.iter().any(|c| c.is_same(&cls)), + _ => false, + }, + }; + if direct { + return found(data); + } + // 5. A subclass of a registered class (recursive). + let registered = data.registry.borrow().live(); + for rcls in registered { + if interp.issubclass_public(&subclass, &rcls)? { + return found(data); + } + } + // 6. A subclass of a subclass (recursive). + let subclasses = call_method(interp, &cls, "__subclasses__", &[])?; + let globals = interp.builtins_dict(); + for scls in interp.collect_iterable(&subclasses, &globals)? { + if interp.issubclass_public(&subclass, &scls)? { + return found(data); } } + data.negative_cache.borrow_mut().insert(&subclass); Ok(Object::Bool(false)) } +/// `_get_dump(cls)`: weak references to the registry and caches, plus the +/// negative-cache version (used by the refleak hunter and `_dump_registry`). fn abc_get_dump(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - if let Object::Type(t) = cls { - let reg = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - .unwrap_or(Object::new_set()); - return Ok(Object::new_tuple_array([ - reg, - Object::new_set(), - Object::new_set(), - Object::Int(0), - ])); - } + let [cls] = args_exact::<1>(args, "_get_dump")?; + let interp = interpreter()?; + let owner = state_of(&cls)?; + let data = state(&owner); + let globals = interp.builtins_dict(); + let weakref = interp.do_import("_weakref", &Object::None, 0, &globals)?; + let make_ref = interp.load_attr_public(&weakref, "ref")?; + let mut weak_set = |classes: Vec| -> Result { + let mut refs = Vec::with_capacity(classes.len()); + for class in classes { + refs.push(interp.call(&make_ref, &[class], &[], &globals)?); + } + Ok(Object::new_set_from(refs)) + }; + let registry = weak_set(data.registry.borrow().live())?; + let cache = weak_set(data.cache.borrow().live())?; + let negative = weak_set(data.negative_cache.borrow().live())?; Ok(Object::new_tuple_array([ - Object::new_set(), - Object::new_set(), - Object::new_set(), - Object::Int(0), + registry, + cache, + negative, + Object::Int(data.negative_cache_version.get() as i64), ])) } fn abc_reset_registry(args: &[Object]) -> Result { - let _ = args; - bump_cache(); + let [cls] = args_exact::<1>(args, "_reset_registry")?; + let owner = state_of(&cls)?; + state(&owner).registry.borrow_mut().clear(); Ok(Object::None) } fn abc_reset_caches(args: &[Object]) -> Result { - let _ = args; - bump_cache(); + let [cls] = args_exact::<1>(args, "_reset_caches")?; + let owner = state_of(&cls)?; + let data = state(&owner); + data.cache.borrow_mut().clear(); + data.negative_cache.borrow_mut().clear(); Ok(Object::None) } diff --git a/crates/weavepy-vm/src/stdlib/io.rs b/crates/weavepy-vm/src/stdlib/io.rs index e0077afc..01e39199 100644 --- a/crates/weavepy-vm/src/stdlib/io.rs +++ b/crates/weavepy-vm/src/stdlib/io.rs @@ -182,34 +182,6 @@ pub fn build(_cache: &ModuleCache) -> Rc { }) } -/// `IOBase.register(subclass)` — ABC virtual-subclass registration. The -/// class binds first (classmethod); we record the subclass in the class's -/// `_abc_registry` set and return it so `register` also works as a -/// decorator, mirroring `_abc._abc_register`. -fn io_abc_register(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let Object::Type(t) = &cls { - let key = DictKey(Object::from_static("_abc_registry")); - let reg = { - let existing = t.dict.borrow().get(&key).cloned(); - match existing { - Some(Object::Set(s)) => s, - _ => { - let s = Object::new_set(); - t.dict.borrow_mut().insert(key, s.clone()); - match s { - Object::Set(s) => s, - _ => return Ok(sub), - } - } - } - }; - reg.borrow_mut().insert(DictKey(sub.clone())); - } - Ok(sub) -} - fn builtin(name: &'static str, body: fn(&[Object]) -> Result) -> Object { Object::Builtin(Rc::new(BuiltinFn { name, @@ -1861,21 +1833,10 @@ pub(crate) fn file_io_abc_match( /// Build the `IOBase → {RawIOBase, BufferedIOBase, TextIOBase} → FileIO` /// hierarchy with the CPython mixin methods installed on the root. fn build_iobase_family_inner() -> IoFamily { - use crate::object::MethodWrapper; use crate::types::{TypeFlags, TypeObject}; let bt = crate::builtin_types::builtin_types(); let mut dict = DictData::default(); install_iobase_mixins(&mut dict); - // `io.IOBase.register(...)` — ABC virtual-subclass registration. - dict.insert( - DictKey(Object::from_static("register")), - Object::ClassMethod(MethodWrapper::new(Object::Builtin(Rc::new(BuiltinFn { - name: "register", - binds_instance: true, - call: Box::new(io_abc_register), - call_kw: None, - })))), - ); let flags = || TypeFlags { is_exception: false, is_builtin: true, diff --git a/crates/weavepy-vm/src/stdlib/python/abc.py b/crates/weavepy-vm/src/stdlib/python/abc.py index cf212823..f8a4e11c 100644 --- a/crates/weavepy-vm/src/stdlib/python/abc.py +++ b/crates/weavepy-vm/src/stdlib/python/abc.py @@ -81,201 +81,66 @@ def my_abstract_property(self): __isabstractmethod__ = True -# WeavePy has no `_abc` C accelerator, so `ABCMeta` is implemented in -# pure Python (mirroring `_py_abc`, which stays a separate module that -# `test_abc` imports to exercise the reference implementation). It is -# defined *here* rather than imported so that its frames' module is -# `'abc'`: `typing._allow_reckless_class_checks()` walks the frame stack -# and only skips the runtime-protocol restrictions when the caller -# module is `'abc'` or `'functools'` (e.g. `isinstance(x, Traversable)` -# reaches `_ProtocolMeta.__subclasscheck__` *via* -# `ABCMeta.__instancecheck__`, which CPython sees as module `'abc'` -# because that's where the delegating method lives). -from _weakrefset import WeakSet - - -def get_cache_token(): - """Returns the current ABC cache token. - - The token is an opaque object (supporting equality testing) identifying the - current version of the ABC cache for virtual subclasses. The token changes - with every call to ``register()`` on any ABC. - """ - return ABCMeta._abc_invalidation_counter - - -class ABCMeta(type): - """Metaclass for defining Abstract Base Classes (ABCs). - - Use this metaclass to create an ABC. An ABC can be subclassed - directly, and then acts as a mix-in class. You can also register - unrelated concrete classes (even built-in classes) and unrelated - ABCs as 'virtual subclasses' -- these and their descendants will - be considered subclasses of the registering ABC by the built-in - issubclass() function, but the registering ABC won't show up in - their MRO (Method Resolution Order) nor will method - implementations defined by the registering ABC be callable (not - even via super()). - """ - - # A global counter that is incremented each time a class is - # registered as a virtual subclass of anything. It forces the - # negative cache to be cleared before its next use. - # Note: this counter is private. Use `abc.get_cache_token()` for - # external code. - _abc_invalidation_counter = 0 - - def __new__(mcls, name, bases, namespace, /, **kwargs): - cls = super().__new__(mcls, name, bases, namespace, **kwargs) - # Compute set of abstract method names - abstracts = {name - for name, value in namespace.items() - if getattr(value, "__isabstractmethod__", False)} - for base in bases: - for name in getattr(base, "__abstractmethods__", set()): - value = getattr(cls, name, None) - if getattr(value, "__isabstractmethod__", False): - abstracts.add(name) - cls.__abstractmethods__ = frozenset(abstracts) - # CPython's C `_abc_init` consumes `__abc_tpflags__` here: reject - # a class claiming both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING, - # fold the collection bits into the type's flags (stored under a - # private name the VM's `__flags__` getter folds in), and delete - # the public attribute. - tpflags = namespace.get("__abc_tpflags__") - if isinstance(tpflags, int): - COLLECTION_FLAGS = (1 << 5) | (1 << 6) - if tpflags & COLLECTION_FLAGS == COLLECTION_FLAGS: - raise TypeError( - "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE" - " and Py_TPFLAGS_MAPPING" - ) - cls._abc_collection_flags = tpflags & COLLECTION_FLAGS - if "__abc_tpflags__" in namespace: - del cls.__abc_tpflags__ - # Set up inheritance registry - cls._abc_registry = WeakSet() - cls._abc_cache = WeakSet() - cls._abc_negative_cache = WeakSet() - cls._abc_negative_cache_version = ABCMeta._abc_invalidation_counter - return cls - - def register(cls, subclass): - """Register a virtual subclass of an ABC. - - Returns the subclass, to allow usage as a class decorator. +try: + from _abc import (get_cache_token, _abc_init, _abc_register, + _abc_instancecheck, _abc_subclasscheck, _get_dump, + _reset_registry, _reset_caches) +except ImportError: + from _py_abc import ABCMeta, get_cache_token + ABCMeta.__module__ = 'abc' +else: + class ABCMeta(type): + """Metaclass for defining Abstract Base Classes (ABCs). + + Use this metaclass to create an ABC. An ABC can be subclassed + directly, and then acts as a mix-in class. You can also register + unrelated concrete classes (even built-in classes) and unrelated + ABCs as 'virtual subclasses' -- these and their descendants will + be considered subclasses of the registering ABC by the built-in + issubclass() function, but the registering ABC won't show up in + their MRO (Method Resolution Order) nor will method + implementations defined by the registering ABC be callable (not + even via super()). """ - if not isinstance(subclass, type): - raise TypeError("Can only register classes") - if issubclass(subclass, cls): - return subclass # Already a subclass - # Subtle: test for cycles *after* testing for "already a subclass"; - # this means we allow X.register(X) and interpret it as a no-op. - if issubclass(cls, subclass): - # This would create a cycle, which is bad for the algorithm below - raise RuntimeError("Refusing to create an inheritance cycle") - cls._abc_registry.add(subclass) - ABCMeta._abc_invalidation_counter += 1 # Invalidate negative cache - # CPython's C `_abc_register` copies the ABC's collection flag - # (Py_TPFLAGS_SEQUENCE / Py_TPFLAGS_MAPPING) onto the registered - # class (and recursively its subclasses), so late registration on - # Sequence/Mapping makes match patterns work (test_patma). The VM - # walks the MRO for this marker, so stamping the registered class - # also covers its subclasses. - collection_flag = getattr(cls, "_abc_collection_flags", 0) - if collection_flag: - try: - subclass._abc_collection_flags = collection_flag - except TypeError: - pass # immutable (builtin) type - return subclass - - def _dump_registry(cls, file=None): - """Debug helper to print the ABC registry.""" - print(f"Class: {cls.__module__}.{cls.__qualname__}", file=file) - print(f"Inv. counter: {get_cache_token()}", file=file) - for name in cls.__dict__: - if name.startswith("_abc_"): - value = getattr(cls, name) - if isinstance(value, WeakSet): - value = set(value) - print(f"{name}: {value!r}", file=file) - - def _abc_registry_clear(cls): - """Clear the registry (for debugging or testing).""" - cls._abc_registry.clear() - - def _abc_caches_clear(cls): - """Clear the caches (for debugging or testing).""" - cls._abc_cache.clear() - cls._abc_negative_cache.clear() - - def __instancecheck__(cls, instance): - """Override for isinstance(instance, cls).""" - # Inline the cache checking - subclass = instance.__class__ - if subclass in cls._abc_cache: - return True - subtype = type(instance) - if subtype is subclass: - if (cls._abc_negative_cache_version == - ABCMeta._abc_invalidation_counter and - subclass in cls._abc_negative_cache): - return False - # Fall back to the subclass check. - return cls.__subclasscheck__(subclass) - return any(cls.__subclasscheck__(c) for c in (subclass, subtype)) - - def __subclasscheck__(cls, subclass): - """Override for issubclass(subclass, cls).""" - if not isinstance(subclass, type): - raise TypeError('issubclass() arg 1 must be a class') - # Check cache - if subclass in cls._abc_cache: - return True - # Check negative cache; may have to invalidate - if cls._abc_negative_cache_version < ABCMeta._abc_invalidation_counter: - # Invalidate the negative cache - cls._abc_negative_cache = WeakSet() - cls._abc_negative_cache_version = ABCMeta._abc_invalidation_counter - elif subclass in cls._abc_negative_cache: - return False - # Check the subclass hook - ok = cls.__subclasshook__(subclass) - if ok is not NotImplemented: - assert isinstance(ok, bool) - if ok: - cls._abc_cache.add(subclass) - else: - cls._abc_negative_cache.add(subclass) - return ok - # Check if it's a direct subclass - if cls in getattr(subclass, '__mro__', ()): - cls._abc_cache.add(subclass) - return True - # Check if it's a subclass of a registered class (recursive) - for rcls in cls._abc_registry: - if issubclass(subclass, rcls): - cls._abc_cache.add(subclass) - return True - # Check if it's a subclass of a subclass (recursive) - for scls in cls.__subclasses__(): - if issubclass(subclass, scls): - cls._abc_cache.add(subclass) - return True - # No dice; update negative cache - cls._abc_negative_cache.add(subclass) - return False - - -# CPython's pure-Python path does `from _py_abc import ABCMeta` and then -# `ABCMeta.__module__ = 'abc'` — and *that assignment* drops the class's -# `__firstlineno__` (type_set_module invalidates stale source info), so -# `inspect.getsource(abc.ABCMeta)` reports "source code not available" -# whenever the C accelerator is absent (test_inspect -# test_getsource_stdlib_abc). Mirror the assignment — a no-op for the -# module name itself, but with the same firstlineno-dropping effect. -ABCMeta.__module__ = 'abc' + def __new__(mcls, name, bases, namespace, /, **kwargs): + cls = super().__new__(mcls, name, bases, namespace, **kwargs) + _abc_init(cls) + return cls + + def register(cls, subclass): + """Register a virtual subclass of an ABC. + + Returns the subclass, to allow usage as a class decorator. + """ + return _abc_register(cls, subclass) + + def __instancecheck__(cls, instance): + """Override for isinstance(instance, cls).""" + return _abc_instancecheck(cls, instance) + + def __subclasscheck__(cls, subclass): + """Override for issubclass(subclass, cls).""" + return _abc_subclasscheck(cls, subclass) + + def _dump_registry(cls, file=None): + """Debug helper to print the ABC registry.""" + print(f"Class: {cls.__module__}.{cls.__qualname__}", file=file) + print(f"Inv. counter: {get_cache_token()}", file=file) + (_abc_registry, _abc_cache, _abc_negative_cache, + _abc_negative_cache_version) = _get_dump(cls) + print(f"_abc_registry: {_abc_registry!r}", file=file) + print(f"_abc_cache: {_abc_cache!r}", file=file) + print(f"_abc_negative_cache: {_abc_negative_cache!r}", file=file) + print(f"_abc_negative_cache_version: {_abc_negative_cache_version!r}", + file=file) + + def _abc_registry_clear(cls): + """Clear the registry (for debugging or testing).""" + _reset_registry(cls) + + def _abc_caches_clear(cls): + """Clear the caches (for debugging or testing).""" + _reset_caches(cls) def update_abstractmethods(cls): diff --git a/crates/weavepy-vm/src/stdlib/thread_real.rs b/crates/weavepy-vm/src/stdlib/thread_real.rs index 5ddb976c..e67d8690 100644 --- a/crates/weavepy-vm/src/stdlib/thread_real.rs +++ b/crates/weavepy-vm/src/stdlib/thread_real.rs @@ -1351,7 +1351,6 @@ fn spawn_python_worker( .name(format!("weavepy-worker-{}", synth_id)) .stack_size(WORKER_STACK_BYTES) .spawn(move || { - crate::tcache::enable_for_current_thread(); crate::vm_singletons::install_worker_thread_id(synth_id); // RFC 0040 WS4: record this worker's pthread_t so // `signal.pthread_kill(ident, sig)` can target it. diff --git a/crates/weavepy-vm/src/tcache.rs b/crates/weavepy-vm/src/tcache.rs deleted file mode 100644 index e7c33148..00000000 --- a/crates/weavepy-vm/src/tcache.rs +++ /dev/null @@ -1,226 +0,0 @@ -//! A thread-caching front end for the system allocator. -//! -//! The interpreter allocates and frees small blocks constantly — iterators, -//! tuples, list and dict storage, strings, boxed payloads — and a system -//! `malloc`/`free` pair costs ~20ns on macOS. [`ThreadCacheAlloc`] keeps -//! per-thread free lists of recently freed small blocks (16-byte size -//! classes up to [`MAX_SMALL`] bytes) and serves allocations of the same -//! class from them, which turns the common alloc/free pair into a handful -//! of loads and stores. -//! -//! Every block, cached or not, is a genuine system `malloc` block whose -//! usable size is at least its class size: a small request is rounded up -//! to its class before it reaches the system, and a block only enters a -//! class list when its layout maps to that class. So a block may be -//! handed back to the system (`free`, `realloc`) at any time, from any -//! thread, whatever list it last sat on. -//! -//! Caching is opt-in per thread ([`enable_for_current_thread`]): enabling -//! registers a thread-exit guard that returns the thread's cached blocks -//! to the system, so short-lived threads cannot strand memory. Threads -//! that never opt in (and a thread past its exit flush) go straight to the -//! system allocator. - -// Every cached block is at least `QUANTUM`-aligned (16 bytes, the class -// granularity), so threading the free-list link through a `*mut *mut u8` -// is aligned by construction. -#![allow(clippy::cast_ptr_alignment)] - -use std::alloc::{GlobalAlloc, Layout, System}; -use std::cell::UnsafeCell; -use std::ptr; - -/// Largest request served from the class lists. -const MAX_SMALL: usize = 512; -/// Size-class granularity (also the guaranteed alignment of a class block). -const QUANTUM: usize = 16; -const NCLASSES: usize = MAX_SMALL / QUANTUM; -/// Bytes one class list may retain. -const CLASS_BUDGET: usize = 8 * 1024; - -struct Lists { - /// Whether this thread caches (see [`enable_for_current_thread`]). - enabled: bool, - heads: [*mut u8; NCLASSES], - counts: [u32; NCLASSES], -} - -thread_local! { - /// This thread's class lists (intrusive: a free block's first word - /// links to the next). - static LISTS: UnsafeCell = const { - UnsafeCell::new(Lists { - enabled: false, - heads: [ptr::null_mut(); NCLASSES], - counts: [0; NCLASSES], - }) - }; - /// Flushes the lists when the thread exits. - static EXIT_GUARD: ExitGuard = const { ExitGuard }; -} - -struct ExitGuard; - -impl Drop for ExitGuard { - fn drop(&mut self) { - let _ = LISTS.try_with(|l| { - // SAFETY: this thread's own lists; caching goes off first, so - // the frees below go straight to the system. - let lists = unsafe { &mut *l.get() }; - lists.enabled = false; - for class in 0..NCLASSES { - let mut p = lists.heads[class]; - while !p.is_null() { - // SAFETY: every listed block is a live system block - // whose first word holds the next link. - let next = unsafe { *p.cast::<*mut u8>() }; - unsafe { libc_free(p) }; - p = next; - } - lists.heads[class] = ptr::null_mut(); - lists.counts[class] = 0; - } - }); - } -} - -extern "C" { - #[link_name = "free"] - fn libc_free(p: *mut u8); -} - -/// Turn on block caching for the calling thread (idempotent). Registers -/// the thread-exit flush first, while caching is still off, so the -/// registration's own allocations go straight to the system. -pub fn enable_for_current_thread() { - // SAFETY (both reads/writes): this thread's own lists, outside any - // allocator call. - let on = LISTS - .try_with(|l| unsafe { (*l.get()).enabled }) - .unwrap_or(true); - if on { - return; - } - let _ = EXIT_GUARD.try_with(|_| ()); - let _ = LISTS.try_with(|l| unsafe { (*l.get()).enabled = true }); -} - -/// The class of a small layout, or `None` for one the lists never hold. -#[inline] -fn class_of(layout: &Layout) -> Option { - let size = layout.size(); - if size == 0 || size > MAX_SMALL || layout.align() > QUANTUM { - return None; - } - Some((size - 1) / QUANTUM) -} - -#[inline] -fn class_layout(class: usize) -> Layout { - // SAFETY: a multiple of 16 no larger than MAX_SMALL, aligned to 16. - unsafe { Layout::from_size_align_unchecked((class + 1) * QUANTUM, QUANTUM) } -} - -/// The global allocator: [`System`] behind per-thread class lists. -#[derive(Debug, Default, Clone, Copy)] -pub struct ThreadCacheAlloc; - -unsafe impl GlobalAlloc for ThreadCacheAlloc { - #[inline] - unsafe fn alloc(&self, layout: Layout) -> *mut u8 { - if let Some(class) = class_of(&layout) { - let got = LISTS - .try_with(|l| { - // SAFETY: this thread's own lists; the allocator is - // never re-entered while they are borrowed. - let lists = unsafe { &mut *l.get() }; - if !lists.enabled { - return ptr::null_mut(); - } - let head = lists.heads[class]; - if !head.is_null() { - // SAFETY: a listed block's first word links on. - lists.heads[class] = unsafe { *head.cast::<*mut u8>() }; - lists.counts[class] -= 1; - } - head - }) - .unwrap_or(ptr::null_mut()); - if !got.is_null() { - return got; - } - // SAFETY: a valid non-zero layout. - return unsafe { System.alloc(class_layout(class)) }; - } - // SAFETY: forwarded unchanged. - unsafe { System.alloc(layout) } - } - - #[inline] - unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { - if let Some(class) = class_of(&layout) { - let kept = LISTS - .try_with(|l| { - // SAFETY: as in `alloc`. - let lists = unsafe { &mut *l.get() }; - if !lists.enabled { - return false; - } - let cap = (CLASS_BUDGET / ((class + 1) * QUANTUM)).max(8) as u32; - if lists.counts[class] >= cap { - return false; - } - // SAFETY: the block is ours now and at least 16 bytes. - unsafe { *ptr.cast::<*mut u8>() = lists.heads[class] }; - lists.heads[class] = ptr; - lists.counts[class] += 1; - true - }) - .unwrap_or(false); - if !kept { - // SAFETY: a system block allocated at its class layout. - unsafe { System.dealloc(ptr, class_layout(class)) }; - } - return; - } - // SAFETY: forwarded unchanged. - unsafe { System.dealloc(ptr, layout) } - } - - #[inline] - unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { - if layout.align() > QUANTUM { - // The generic path: fresh block, copy, release. - // SAFETY: the caller's layout invariants hold for `new_layout`. - let new_layout = unsafe { Layout::from_size_align_unchecked(new_size, layout.align()) }; - let new = unsafe { self.alloc(new_layout) }; - if !new.is_null() { - unsafe { - ptr::copy_nonoverlapping(ptr, new, layout.size().min(new_size)); - self.dealloc(ptr, layout); - } - } - return new; - } - let old_class = class_of(&layout); - // SAFETY: `new_size` is non-zero and fits the caller's layout rules. - let new_layout = unsafe { Layout::from_size_align_unchecked(new_size, layout.align()) }; - let new_class = class_of(&new_layout); - if old_class.is_some() && old_class == new_class { - // Same class: the block already has the room. - return ptr; - } - // Resize the underlying system block to what the new layout's - // eventual `dealloc` will assume (its class size, when small). - let target = match new_class { - Some(class) => (class + 1) * QUANTUM, - None => new_size, - }; - let old_system = match old_class { - Some(class) => class_layout(class), - None => layout, - }; - // SAFETY: `ptr` is a live system block of `old_system`. - unsafe { System.realloc(ptr, old_system, target) } - } -} diff --git a/crates/weavepy-vm/src/timsort.rs b/crates/weavepy-vm/src/timsort.rs new file mode 100644 index 00000000..3189029a --- /dev/null +++ b/crates/weavepy-vm/src/timsort.rs @@ -0,0 +1,736 @@ +//! CPython's list sort: an adaptive, stable, natural merge sort (timsort with +//! the powersort merge policy), ported from `Objects/listobject.c` in 3.14. +//! +//! The comparison sequence follows CPython's, so a comparator that is +//! inconsistent (NaNs, a random `__lt__`) or that raises produces the same +//! observable behavior: no panic, and on error the slice is left as a +//! permutation of its input. Every comparison is a strict "less than". + +use std::ptr; + +/// Once a merge is galloping, it stays there until both runs win fewer +/// than this many consecutive times. +const MIN_GALLOP: usize = 7; + +/// The largest minimum run length; a power of 2. +const MAX_MINRUN: usize = 64; + +/// A run pending a merge: `len` elements starting at index `base`. +#[derive(Clone, Copy)] +struct Run { + base: usize, + len: usize, + /// Depth in the conceptual binary merge tree (powersort). + power: u32, +} + +struct MergeState { + base: *mut T, + len: usize, + lt: F, + min_gallop: usize, + /// Scratch storage for merges. Its length is always 0: elements are + /// moved in and out bitwise, and never dropped from here. + tmp: Vec, + pending: Vec, +} + +/// Sort `v` in place, stably, by `lt`, the strict ordering. On error, `v` +/// holds a permutation of its input. +pub(crate) fn sort(v: &mut [T], lt: F) -> Result<(), E> +where + F: FnMut(&T, &T) -> Result, +{ + let n = v.len(); + if n < 2 { + return Ok(()); + } + let mut ms = MergeState { + base: v.as_mut_ptr(), + len: n, + lt, + min_gallop: MIN_GALLOP, + tmp: Vec::new(), + pending: Vec::new(), + }; + let minrun = compute_minrun(n); + let mut lo = 0; + let mut nremaining = n; + // SAFETY: every index handed to the helpers below lies in `v`, which is + // borrowed mutably for the whole sort. + unsafe { + while nremaining > 0 { + let mut run = ms.count_run(lo, nremaining)?; + if run < minrun { + let force = nremaining.min(minrun); + ms.binarysort(lo, force, run)?; + run = force; + } + ms.found_new_run(run)?; + ms.pending.push(Run { + base: lo, + len: run, + power: 0, + }); + lo += run; + nremaining -= run; + } + ms.merge_force_collapse() + } +} + +/// A good minimum run length: `n` itself below `MAX_MINRUN`, else a value in +/// `MAX_MINRUN / 2 ..= MAX_MINRUN` such that `n / minrun` is close to, but +/// strictly less than, a power of 2. +fn compute_minrun(mut n: usize) -> usize { + let mut r = 0; + while n >= MAX_MINRUN { + r |= n & 1; + n >>= 1; + } + n + r +} + +/// The powersort "power" of the run at `s1` (length `n1`) followed by one of +/// length `n2`, in a list of length `n`. +fn powerloop(s1: usize, n1: usize, n2: usize, n: usize) -> u32 { + let mut result = 0; + // Twice the two runs' midpoints, so that both are integers. + let mut a = 2 * s1 + n1; + let mut b = a + n1 + n2; + loop { + result += 1; + if a >= n { + a -= n; + b -= n; + } else if b >= n { + break; + } + a <<= 1; + b <<= 1; + } + result +} + +/// Unmerged elements of a merge, parked in scratch storage: `len` of them +/// from `src` belong at `dest`. Dropping the hole moves them there, which +/// also restores a complete permutation if a comparison fails or panics. +struct Hole { + src: *const T, + dest: *mut T, + len: usize, +} + +impl Drop for Hole { + fn drop(&mut self) { + // SAFETY: the merge maintains that `dest` is exactly the vacated + // range the parked elements fill, and scratch never overlaps the + // list. + unsafe { ptr::copy_nonoverlapping(self.src, self.dest, self.len) }; + } +} + +impl MergeState +where + F: FnMut(&T, &T) -> Result, +{ + #[inline] + unsafe fn at(&self, i: usize) -> *mut T { + // SAFETY: callers pass indices within the list. + unsafe { self.base.add(i) } + } + + #[inline] + fn lt(&mut self, a: *const T, b: *const T) -> Result { + // SAFETY: both point at initialized elements, in the list or + // parked in scratch, and no element moves during a comparison. + unsafe { (self.lt)(&*a, &*b) } + } + + /// Scratch storage for `need` elements. + fn scratch(&mut self, need: usize) -> *mut T { + if self.tmp.capacity() < need { + self.tmp = Vec::with_capacity(need); + } + self.tmp.as_mut_ptr() + } + + unsafe fn reverse(&mut self, lo: usize, n: usize) { + // SAFETY: `lo..lo + n` lies in the list. + unsafe { std::slice::from_raw_parts_mut(self.at(lo), n).reverse() }; + } + + /// Stable binary insertion sort of `lo..lo + n`, whose first `ok` + /// elements are already sorted. + unsafe fn binarysort(&mut self, lo: usize, n: usize, ok: usize) -> Result<(), E> { + let a = unsafe { self.at(lo) }; + let mut ok = ok.max(1); + while ok < n { + // Find where a[ok] belongs: a[..l] <= pivot < a[r..ok]. + let (mut l, mut r) = (0, ok); + let pivot = unsafe { a.add(ok) }; + while l < r { + let m = (l + r) >> 1; + if self.lt(pivot, unsafe { a.add(m) })? { + r = m; + } else { + l = m + 1; + } + } + // SAFETY: rotate a[l..=ok] right by one; nothing compares + // while the pivot is out. + unsafe { + let p = ptr::read(pivot); + ptr::copy(a.add(l), a.add(l + 1), ok - l); + ptr::write(a.add(l), p); + } + ok += 1; + } + Ok(()) + } + + /// The length of the run starting at `lo`, no longer than `nremaining`, + /// made ascending in place. + unsafe fn count_run(&mut self, lo: usize, nremaining: usize) -> Result { + let a = unsafe { self.at(lo) }; + let next_smaller = + |ms: &mut Self, n: usize| ms.lt(unsafe { a.add(n) }, unsafe { a.add(n - 1) }); + let next_larger = + |ms: &mut Self, n: usize| ms.lt(unsafe { a.add(n - 1) }, unsafe { a.add(n) }); + // Try an ascending run first. + let mut n = 1; + while n < nremaining { + if next_smaller(self, n)? { + break; + } + n += 1; + } + if n == nremaining { + return Ok(n); + } + // a[n] is strictly less. With a longer ascending prefix, it either + // rose somewhere (done), or is all equal and can start a + // descending run, reversed in place. + if n > 1 { + if self.lt(a, unsafe { a.add(n - 1) })? { + return Ok(n); + } + unsafe { self.reverse(lo, n) }; + } + n += 1; + // Finish the descending run, reversing all-equal subruns on the fly + // so the final whole-run reversal restores their order. + let mut neq = 0; + while n < nremaining { + if next_smaller(self, n)? { + if neq > 0 { + neq += 1; + unsafe { self.reverse(lo + n - neq, neq) }; + neq = 0; + } + } else if next_larger(self, n)? { + break; + } else { + neq += 1; + } + n += 1; + } + if neq > 0 { + neq += 1; + unsafe { self.reverse(lo + n - neq, neq) }; + } + unsafe { self.reverse(lo, n) }; + // The reversed run may extend with a naturally increasing suffix. + while n < nremaining { + if next_smaller(self, n)? { + break; + } + n += 1; + } + Ok(n) + } + + /// The index `k` in `0..=n` where `key` belongs in the sorted `a[..n]`, + /// left of any equal elements: `a[k - 1] < key <= a[k]`. The search + /// starts at `hint`. + unsafe fn gallop_left( + &mut self, + key: *const T, + a: *const T, + n: usize, + hint: usize, + ) -> Result { + let at = |i: usize| unsafe { a.add(i) }; + let (mut lastofs, mut ofs); + if self.lt(at(hint), key)? { + // Gallop right until a[hint + lastofs] < key <= a[hint + ofs]. + let maxofs = n - hint; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(at(hint + ofs), key)? { + lastofs = ofs; + ofs = (ofs << 1) + 1; + } else { + break; + } + } + ofs = ofs.min(maxofs); + lastofs += hint + 1; + ofs += hint; + } else { + // Gallop left until a[hint - ofs] < key <= a[hint - lastofs]. + let maxofs = hint + 1; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(at(hint - ofs), key)? { + break; + } + lastofs = ofs; + ofs = (ofs << 1) + 1; + } + ofs = ofs.min(maxofs); + let k = lastofs; + // `hint - ofs` may be -1; the binary search starts one past it. + lastofs = hint + 1 - ofs; + ofs = hint - k; + } + // Binary search with a[lastofs - 1] < key <= a[ofs]. + while lastofs < ofs { + let m = lastofs + ((ofs - lastofs) >> 1); + if self.lt(at(m), key)? { + lastofs = m + 1; + } else { + ofs = m; + } + } + Ok(ofs) + } + + /// Like [`Self::gallop_left`], but right of any equal elements: + /// `a[k - 1] <= key < a[k]`. + unsafe fn gallop_right( + &mut self, + key: *const T, + a: *const T, + n: usize, + hint: usize, + ) -> Result { + let at = |i: usize| unsafe { a.add(i) }; + let (mut lastofs, mut ofs); + if self.lt(key, at(hint))? { + // Gallop left until a[hint - ofs] <= key < a[hint - lastofs]. + let maxofs = hint + 1; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(key, at(hint - ofs))? { + lastofs = ofs; + ofs = (ofs << 1) + 1; + } else { + break; + } + } + ofs = ofs.min(maxofs); + let k = lastofs; + lastofs = hint + 1 - ofs; + ofs = hint - k; + } else { + // Gallop right until a[hint + lastofs] <= key < a[hint + ofs]. + let maxofs = n - hint; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(key, at(hint + ofs))? { + break; + } + lastofs = ofs; + ofs = (ofs << 1) + 1; + } + ofs = ofs.min(maxofs); + lastofs += hint + 1; + ofs += hint; + } + // Binary search with a[lastofs - 1] <= key < a[ofs]. + while lastofs < ofs { + let m = lastofs + ((ofs - lastofs) >> 1); + if self.lt(key, at(m))? { + ofs = m; + } else { + lastofs = m + 1; + } + } + Ok(ofs) + } + + /// Merge the adjacent runs `sa..sa + na` and `sa + na..sa + na + nb`, + /// with `na <= nb`, through scratch space for the first. The last + /// element of the first run belongs at the end of the merge. + // A move's bookkeeping is dead when the merge returns right after it. + #[allow(unused_assignments)] + unsafe fn merge_lo(&mut self, sa: usize, na: usize, nb: usize) -> Result<(), E> { + let tmp = self.scratch(na); + let mut dest = unsafe { self.at(sa) }; + let mut b = unsafe { self.at(sa + na) }; + unsafe { ptr::copy_nonoverlapping(dest, tmp, na) }; + // The unmerged part of the first run, parked in scratch. + let mut hole = Hole { + src: tmp, + dest, + len: na, + }; + let mut nb = nb; + // SAFETY (throughout): the merged prefix, the parked first-run + // elements, and the rest of the second run always partition the + // two runs' span, so `hole.dest` is where the parked ones belong. + macro_rules! take_b { + ($k:expr) => {{ + let k = $k; + unsafe { ptr::copy(b, dest, k) }; + dest = unsafe { dest.add(k) }; + b = unsafe { b.add(k) }; + nb -= k; + hole.dest = dest; + }}; + } + macro_rules! take_a { + ($k:expr) => {{ + let k = $k; + unsafe { ptr::copy_nonoverlapping(hole.src, dest, k) }; + dest = unsafe { dest.add(k) }; + hole.src = unsafe { hole.src.add(k) }; + hole.len -= k; + hole.dest = dest; + }}; + } + take_b!(1); + if nb == 0 { + return Ok(()); + } + if hole.len == 1 { + // The last element of the first run goes after the second. + take_b!(nb); + return Ok(()); + } + let mut min_gallop = self.min_gallop; + loop { + let mut acount = 0; + let mut bcount = 0; + // One element at a time, until one run keeps winning. + loop { + if self.lt(b, hole.src)? { + take_b!(1); + bcount += 1; + acount = 0; + if nb == 0 { + return Ok(()); + } + if bcount >= min_gallop { + break; + } + } else { + take_a!(1); + acount += 1; + bcount = 0; + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + if acount >= min_gallop { + break; + } + } + } + // Gallop until neither run is winning consistently. + min_gallop += 1; + loop { + min_gallop -= usize::from(min_gallop > 1); + self.min_gallop = min_gallop; + let k = unsafe { self.gallop_right(b, hole.src, hole.len, 0)? }; + acount = k; + if k > 0 { + take_a!(k); + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + // Impossible for a consistent comparison, but possible. + if hole.len == 0 { + return Ok(()); + } + } + take_b!(1); + if nb == 0 { + return Ok(()); + } + let k = unsafe { self.gallop_left(hole.src, b, nb, 0)? }; + bcount = k; + if k > 0 { + take_b!(k); + if nb == 0 { + return Ok(()); + } + } + take_a!(1); + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + if acount < MIN_GALLOP && bcount < MIN_GALLOP { + break; + } + } + // Penalize leaving galloping mode. + min_gallop += 1; + self.min_gallop = min_gallop; + } + } + + /// Merge the adjacent runs `sa..sa + na` and `sa + na..sa + na + nb`, + /// with `na >= nb`, from the top, through scratch space for the second. + /// The first element of the second run belongs at the front. + #[allow(unused_assignments)] + unsafe fn merge_hi(&mut self, sa: usize, na: usize, nb: usize) -> Result<(), E> { + let tmp = self.scratch(nb); + let a_base = unsafe { self.at(sa) }; + unsafe { ptr::copy_nonoverlapping(self.at(sa + na), tmp, nb) }; + // The unmerged part of the second run, parked in scratch; it always + // belongs just above the unmerged part of the first. + let mut hole = Hole { + src: tmp, + dest: unsafe { a_base.add(na) }, + len: nb, + }; + let mut na = na; + // `dest` is the highest unfilled slot: a_base[na + hole.len - 1]. + // SAFETY (throughout): as in `merge_lo`, mirrored. + macro_rules! take_a { + ($k:expr) => {{ + let k = $k; + // Move a_base[na - k..na] up to end at the top slot. + unsafe { ptr::copy(a_base.add(na - k), a_base.add(na - k + hole.len), k) }; + na -= k; + hole.dest = unsafe { a_base.add(na) }; + }}; + } + macro_rules! take_b { + ($k:expr) => {{ + let k = $k; + let len = hole.len; + unsafe { ptr::copy_nonoverlapping(tmp.add(len - k), a_base.add(na + len - k), k) }; + hole.len -= k; + }}; + } + take_a!(1); + if na == 0 { + return Ok(()); + } + if hole.len == 1 { + // The first element of the second run goes before the first. + take_a!(na); + return Ok(()); + } + let mut min_gallop = self.min_gallop; + loop { + let mut acount = 0; + let mut bcount = 0; + loop { + let (a_top, b_top) = unsafe { (a_base.add(na - 1), tmp.add(hole.len - 1)) }; + if self.lt(b_top, a_top)? { + take_a!(1); + acount += 1; + bcount = 0; + if na == 0 { + return Ok(()); + } + if acount >= min_gallop { + break; + } + } else { + take_b!(1); + bcount += 1; + acount = 0; + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + if bcount >= min_gallop { + break; + } + } + } + min_gallop += 1; + loop { + min_gallop -= usize::from(min_gallop > 1); + self.min_gallop = min_gallop; + let b_top = unsafe { tmp.add(hole.len - 1) }; + let k = na - unsafe { self.gallop_right(b_top, a_base, na, na - 1)? }; + acount = k; + if k > 0 { + take_a!(k); + if na == 0 { + return Ok(()); + } + } + take_b!(1); + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + let a_top = unsafe { a_base.add(na - 1) }; + let k = hole.len - unsafe { self.gallop_left(a_top, tmp, hole.len, hole.len - 1)? }; + bcount = k; + if k > 0 { + take_b!(k); + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + // Impossible for a consistent comparison, but possible. + if hole.len == 0 { + return Ok(()); + } + } + take_a!(1); + if na == 0 { + return Ok(()); + } + if acount < MIN_GALLOP && bcount < MIN_GALLOP { + break; + } + } + min_gallop += 1; + self.min_gallop = min_gallop; + } + } + + /// Merge the pending runs at stack indices `i` and `i + 1`. + unsafe fn merge_at(&mut self, i: usize) -> Result<(), E> { + let Run { + base: sa, len: na, .. + } = self.pending[i]; + let Run { + base: sb, len: nb, .. + } = self.pending[i + 1]; + self.pending[i].len = na + nb; + self.pending.remove(i + 1); + // Elements of the first run before where the second starts are + // already in place. + let k = unsafe { self.gallop_right(self.at(sb), self.at(sa), na, 0)? }; + let (sa, na) = (sa + k, na - k); + if na == 0 { + return Ok(()); + } + // Elements of the second run after where the first ends are too. + let nb = unsafe { self.gallop_left(self.at(sa + na - 1), self.at(sb), nb, nb - 1)? }; + if nb == 0 { + return Ok(()); + } + if na <= nb { + unsafe { self.merge_lo(sa, na, nb) } + } else { + unsafe { self.merge_hi(sa, na, nb) } + } + } + + /// A run of length `n2` follows the pending ones: merge runs deeper in + /// the powersort tree than the top one. + unsafe fn found_new_run(&mut self, n2: usize) -> Result<(), E> { + let Some(top) = self.pending.last() else { + return Ok(()); + }; + let power = powerloop(top.base, top.len, n2, self.len); + while self.pending.len() > 1 && self.pending[self.pending.len() - 2].power > power { + unsafe { self.merge_at(self.pending.len() - 2)? }; + } + let last = self.pending.len() - 1; + self.pending[last].power = power; + Ok(()) + } + + /// Merge every pending run into one. + unsafe fn merge_force_collapse(&mut self) -> Result<(), E> { + while self.pending.len() > 1 { + let mut n = self.pending.len() - 2; + if n > 0 && self.pending[n - 1].len < self.pending[n + 1].len { + n -= 1; + } + unsafe { self.merge_at(n)? }; + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn check(mut v: Vec<(i32, usize)>) { + let mut expected = v.clone(); + expected.sort_by_key(|p| p.0); + sort(&mut v, |a, b| Ok::(a.0 < b.0)).unwrap(); + assert_eq!(v, expected); + } + + #[test] + fn sorts_stably() { + let mut seed = 12345u64; + let mut rand = move || { + seed ^= seed << 13; + seed ^= seed >> 7; + seed ^= seed << 17; + seed + }; + for n in [0, 1, 2, 3, 10, 63, 64, 65, 100, 257, 1000, 5000] { + for modulus in [2, 10, 1000, u64::MAX] { + let v: Vec<(i32, usize)> = (0..n).map(|i| ((rand() % modulus) as i32, i)).collect(); + check(v.clone()); + let mut sorted = v.clone(); + sorted.sort_unstable(); + check(sorted.clone()); + sorted.reverse(); + check(sorted); + } + } + } + + #[test] + fn inconsistent_order_keeps_every_element() { + let mut seed = 99u64; + let v: Vec = (0..3000).map(|i| i.to_string()).collect(); + let mut w = v.clone(); + sort(&mut w, |_, _| { + seed = seed.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1); + Ok::(seed >> 63 == 1) + }) + .unwrap(); + w.sort(); + let mut v = v; + v.sort(); + assert_eq!(v, w); + } + + #[test] + fn error_keeps_every_element() { + let v: Vec = (0..2000).rev().map(|i| i.to_string()).collect(); + for limit in [0, 1, 10, 500, 5000] { + let mut w = v.clone(); + let mut count = 0; + let r = sort(&mut w, |a, b| { + count += 1; + if count > limit { + Err(()) + } else { + Ok(a.len() < b.len() || (a.len() == b.len() && a < b)) + } + }); + assert!(r.is_err() || limit >= 5000); + let mut sorted = w.clone(); + sorted.sort(); + let mut expected = v.clone(); + expected.sort(); + assert_eq!(sorted, expected); + } + } +} diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index 469b4a1f..1ffe1e8e 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -37,6 +37,86 @@ static NEXT_TYPE_VERSION: std::sync::atomic::AtomicU64 = std::sync::atomic::Atom /// cached entry (RFC 0077 WS4, [`type_cache`]). Reads are relaxed atomic /// loads — the previous `Cell` paid a `GilCell` lock round-trip on /// every inline-cache guard. +/// Dunders whose resolution [`TypeObject::dunder`] memoises. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Dunder { + Eq, + Hash, +} + +impl Dunder { + pub const COUNT: usize = 2; + + pub const fn name(self) -> &'static str { + match self { + Self::Eq => "__eq__", + Self::Hash => "__hash__", + } + } +} + +/// How a type resolves one dunder, as a set of flags. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DunderInfo(u8); + +impl DunderInfo { + const VALID: u8 = 1; + const PRESENT: u8 = 1 << 1; + const NONE: u8 = 1 << 2; + const FUNCTION: u8 = 1 << 3; + const BUILTIN_OWNER: u8 = 1 << 4; + const OBJECT_OWNER: u8 = 1 << 5; + + fn resolve(found: Option<(Object, Rc)>) -> Self { + let Some((value, owner)) = found else { + return Self(Self::VALID); + }; + let mut bits = Self::VALID | Self::PRESENT; + match value { + Object::None => bits |= Self::NONE, + Object::Function(_) | Object::BoundMethod(_) => bits |= Self::FUNCTION, + _ => {} + } + if owner.flags.is_builtin { + bits |= Self::BUILTIN_OWNER; + } + if Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) { + bits |= Self::OBJECT_OWNER; + } + Self(bits) + } + + /// The MRO defines the dunder (possibly as `None`). + pub fn present(self) -> bool { + self.0 & Self::PRESENT != 0 + } + + /// The dunder is set to `None` (the unhashable marker for `__hash__`). + pub fn is_none(self) -> bool { + self.0 & Self::NONE != 0 + } + + /// The dunder is a Python function or bound method. + pub fn is_function(self) -> bool { + self.0 & Self::FUNCTION != 0 + } + + /// The defining class is a built-in type. + pub fn builtin_owner(self) -> bool { + self.0 & Self::BUILTIN_OWNER != 0 + } + + /// The defining class is `object`. + pub fn object_owner(self) -> bool { + self.0 & Self::OBJECT_OWNER != 0 + } + + /// A non-`None` definition supplied by a class written in Python. + pub fn user_defined(self) -> bool { + self.present() && !self.is_none() && !self.builtin_owner() + } +} + pub struct AttrVersion(std::sync::atomic::AtomicU64); impl AttrVersion { @@ -582,6 +662,9 @@ pub struct TypeObject { /// Native-implementation state for such a class (see /// `stdlib::datetime_native`), set once. pub native_ext: std::sync::OnceLock>, + /// An `abc.ABCMeta` class's registry and caches (see + /// [`crate::stdlib::abc_mod`]). + pub abc_state: std::sync::OnceLock>, /// Cached "do instances of this type carry a `__del__` finalizer /// anywhere in their MRO?" answer, so [`crate::object::PyInstance`]'s /// `Drop` safety net can skip an MRO walk on the hot per-instance drop @@ -590,13 +673,9 @@ pub struct TypeObject { /// `__del__` is assigned to / deleted from a type's dict or the MRO is /// recomputed (`__bases__` assignment). pub has_del: Cell, - /// Memoised `__eq__` resolution for instances of this type, packed - /// as `attr_version << 2 | kind` (`0` = not yet computed): kind `1` = - /// `object`'s identity default (or no `__eq__`), `2` = a Python-level - /// override, `3` = a built-in type's own override (Python dispatch - /// only for instances without a native payload). A stale version - /// recomputes. See `object::instance_has_custom_eq`. - pub eq_kind: Cell, + /// Memoised resolution of the dunders in [`Dunder`], one slot each, + /// packed as `attr_version << 8 | DunderInfo` (see [`Self::dunder`]). + pub dunder_memo: [Cell; Dunder::COUNT], /// Memoised instantiation plan (`type(…)` call protocol resolution: /// `__new__`/`__init__`/native-payload classification), stamped with /// the [`Self::attr_version`] observed when it was built. Rebuilt @@ -981,6 +1060,7 @@ impl TypeObject { inst_dict_hint: std::sync::atomic::AtomicU32::new(0), native_kind: Cell::new(0), native_ext: std::sync::OnceLock::new(), + abc_state: std::sync::OnceLock::new(), slot_names: RefCell::new(Vec::new()), declares_slots: Cell::new(false), forbids_dict: false, @@ -990,7 +1070,7 @@ impl TypeObject { mro_kind: std::sync::atomic::AtomicU8::new(0), attr_version: AttrVersion::fresh(), has_del: Cell::new(0), - eq_kind: Cell::new(0), + dunder_memo: Default::default(), instance_plan: RefCell::new(None), c_tp_name: crate::sync::RefCell::new(None), c_sq_item: Cell::new(false), @@ -1578,6 +1658,22 @@ impl TypeObject { /// the attribute. Lets callers distinguish a dunder *supplied by a /// user class* from one inherited off a built-in (e.g. `object`'s /// identity `__hash__`). + /// How this type resolves `dunder`, memoised until the type or a base + /// changes. Hot paths that only need to know whether a dunder is + /// overridden, and by whom, use this instead of an MRO lookup by name. + #[inline] + pub fn dunder(&self, dunder: Dunder) -> DunderInfo { + let version = self.attr_version.get(); + let slot = &self.dunder_memo[dunder as usize]; + let memo = slot.get(); + if memo & u64::from(DunderInfo::VALID) != 0 && memo >> 8 == version { + return DunderInfo(memo as u8); + } + let info = DunderInfo::resolve(self.lookup_with_owner(dunder.name())); + slot.set(version << 8 | u64::from(info.0)); + info + } + pub fn lookup_with_owner(&self, name: &str) -> Option<(Object, Rc)> { // Fast pass — see `lookup` for the gate rationale. if !crate::object::exotic_str_keys_possible() { @@ -2658,6 +2754,17 @@ impl PyInstance { unsafe { &*self.class.as_ptr() } } + /// How this instance's class resolves `dunder` (see + /// [`TypeObject::dunder`]). + #[inline] + pub fn class_dunder(&self, dunder: Dunder) -> DunderInfo { + if crate::gil::free_threading_enabled() { + self.cls().dunder(dunder) + } else { + self.cls_raw().dunder(dunder) + } + } + /// Re-point the instance at a new class (`obj.__class__ = C`). pub fn set_cls(&self, class: Rc) { *self.class.borrow_mut() = class; From 711cf0f29c5438804e175239ff4eb45dd7cd1520 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 16:16:40 -0700 Subject: [PATCH 03/65] perf: iterate dicts, unpack sequences, and compare scalars in the core loop - Step dict, key, value, and item iterators in the core loop, including their first step and exhaustion. An item unpacked by the following `UNPACK_SEQUENCE 2` goes straight to the stack, so no tuple is built. Iterating a 1,000-key dict's items is 8 times as fast. - Unpack exact tuples and lists of the expected length, and get iterators for dicts and dict views, without leaving the core loop. - Compare pairs of ints, floats, or strings natively before consulting any comparison protocol. --- crates/weavepy-vm/src/lib.rs | 221 ++++++++++++++++++++++++++++++++++- 1 file changed, 217 insertions(+), 4 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index cb29d91f..ff50b8c7 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12167,10 +12167,13 @@ impl Interpreter { // Container reads, stores and comprehension appends run // out of line: arms added here cost the rest of the loop // its register allocation. - OpCode::BinarySubscr | OpCode::StoreSubscr | OpCode::ListAppend => { + OpCode::BinarySubscr + | OpCode::StoreSubscr + | OpCode::ListAppend + | OpCode::UnpackSequence => { // SAFETY: the `len` slots at `base` are initialized, // and the helper touches nothing else. - match unsafe { Self::core_container_op(ins, base, len) } { + match unsafe { Self::core_container_op(ins, base, len, cap) } { Some(n) => { len = n; last = pc; @@ -12298,6 +12301,7 @@ impl Interpreter { } _ => break None, }; + let unique = Rc::strong_count(it) == 1; // SAFETY: nothing below runs code until `it`'s last use // (the guard-free reads of `GilCell::peek`). let Some(it) = (unsafe { it.peek_mut() }) else { @@ -12363,6 +12367,54 @@ impl Interpreter { *index += ch.len_utf8(); Object::from_char(ch) } + // A dict or dict view's next key, value, or item; an + // item unpacked by the `UNPACK_SEQUENCE 2` that + // follows goes straight to the stack, as below. + it @ crate::object::PyIterator::DictKeys { .. } => { + let fuse = len + 2 <= cap + && pc + 1 < ninstrs + // SAFETY: `pc + 1 < ninstrs`. + && unsafe { (*instrs.add(pc + 1)).op } + == OpCode::UnpackSequence + // SAFETY: as above. + && unsafe { (*instrs.add(pc + 1)).arg } == 2; + match Self::core_dict_next(it, fuse, unique) { + DictStep::Item(v, Some(k)) => { + // SAFETY: `len + 2 <= cap`. + unsafe { base.add(len).write(v) }; + len += 1; + pc += 1; + k + } + DictStep::Item(v, None) => v, + DictStep::Exhausted => { + // The iterator (its sole owner is the + // stack) leaves it, and the loop exits + // past its `END_FOR`/`POP_ITER` pair. + len -= 1; + // SAFETY: the slot is initialized; its + // release frees only the iterator. + unsafe { drop_hot(base.add(len).read()) }; + last = pc; + pc += 1 + ins.arg as usize; + let op_at = |pc: usize| { + // SAFETY: `pc < ninstrs` is checked first. + (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }) + }; + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!( + op_at(pc), + Some(OpCode::PopIter | OpCode::PopTop) + ) { + pc += 1; + } + } + continue; + } + DictStep::Decline => break None, + } + } // `for i, x in enumerate(xs)`: the pair goes straight // to the `UNPACK_SEQUENCE 2` that follows, which is // skipped, so no tuple is ever built. @@ -13322,8 +13374,14 @@ impl Interpreter { // SAFETY: `len > 0`. let top = unsafe { base.add(len - 1) }; let v = unsafe { &*top }; - if !matches!(v, Object::List(_) | Object::Tuple(_) | Object::Range(_)) - || !Self::core_droppable(v) + if !matches!( + v, + Object::List(_) + | Object::Tuple(_) + | Object::Range(_) + | Object::Dict(_) + | Object::DictView(_) + ) || !Self::core_droppable(v) { break Some(CoreExit::Helper); } @@ -13560,6 +13618,7 @@ impl Interpreter { ins: weavepy_compiler::Instruction, base: *mut Object, len: usize, + cap: usize, ) -> Option { fn scalar(v: &Object) -> bool { matches!( @@ -13681,6 +13740,57 @@ impl Interpreter { drop_hot(old); Some(len - 3) } + OpCode::UnpackSequence => { + let n = ins.arg as usize; + if len == 0 || n == 0 || len - 1 + n > cap { + return None; + } + // SAFETY: `len > 0`. + let seq = unsafe { &*base.add(len - 1) }; + // The sequence's release frees nothing that could finalize: + // its items outlive it on the stack, and a sole owner here + // means neither the collector nor a weakref holds it. + let unique = match seq { + Object::Tuple(t) if t.len() == n => ThinArc::strong_count(t) == 1, + Object::List(l) if Rc::strong_count(l) == 1 => true, + _ => false, + }; + if !unique && !Self::core_droppable(seq) { + return None; + } + // SAFETY: the sequence leaves its slot, which the last item + // (pushed first) takes; `len - 1 + n <= cap`. + unsafe { + let seq = base.add(len - 1).read(); + match &seq { + Object::Tuple(t) if t.len() == n => { + for (k, item) in t.iter().rev().enumerate() { + base.add(len - 1 + k).write(clone_hot(item)); + } + } + Object::List(l) => { + let Ok(items) = l.try_borrow() else { + base.add(len - 1).write(seq); + return None; + }; + if items.len() != n { + drop(items); + base.add(len - 1).write(seq); + return None; + } + for (k, item) in items.iter().rev().enumerate() { + base.add(len - 1 + k).write(clone_hot(item)); + } + } + _ => { + base.add(len - 1).write(seq); + return None; + } + } + drop_hot(seq); + } + Some(len - 1 + n) + } OpCode::ListAppend => { let depth = ins.arg as usize; if len < 2 || depth == 0 || depth >= len { @@ -13699,6 +13809,65 @@ impl Interpreter { } } + /// The next step of a dict or dict-view iterator, for the core loop: + /// the key, the value, or the item (as `(value, Some(key))` when + /// `fuse`, for an unpacking `FOR_ITER`), or exhaustion, detaching the + /// iterator from its dict. `Decline`, having changed nothing, for a + /// size or key change the checked step reports, an exhaustion that + /// could release anything (a shared iterator, the dict's last + /// reference, a subclass keepalive), or a reverse iterator. + #[inline(never)] + fn core_dict_next(it: &mut crate::object::PyIterator, fuse: bool, unique: bool) -> DictStep { + use crate::object::DictViewKind; + let crate::object::PyIterator::DictKeys { + kind, + index, + dict: dict @ Some(_), + len, + watch, + reverse: false, + owner, + } = it + else { + return DictStep::Decline; + }; + let d = dict.as_ref().expect("matched above"); + // SAFETY: a read with nothing running (see `peek`). + let Some(data) = (unsafe { d.peek() }) else { + return DictStep::Decline; + }; + if data.len() != *len + || watch + .as_ref() + .is_some_and(crate::object::DictWatch::changed) + { + return DictStep::Decline; + } + let Some((k, v)) = data.get_index(*index) else { + if !unique || owner.is_some() || Rc::strong_count(d) == 1 { + return DictStep::Decline; + } + // CPython clears `di_dict` on the first StopIteration. + *dict = None; + *watch = None; + return DictStep::Exhausted; + }; + let item = match kind { + DictViewKind::Keys => (clone_hot(&k.0), None), + DictViewKind::Values => (clone_hot(v), None), + DictViewKind::Items if fuse => (clone_hot(v), Some(clone_hot(&k.0))), + DictViewKind::Items => ( + Object::new_tuple_array([clone_hot(&k.0), clone_hot(v)]), + None, + ), + }; + if watch.is_none() { + *watch = Some(crate::object::DictWatch::new(d)); + } + *index += 1; + DictStep::Item(item.0, item.1) + } + /// Whether the core loop may release `v` with a plain drop: a scalar, /// a string (freeing one runs no code), or a shared heap value whose /// release is a bare decrement the collector need not hear about (the @@ -35778,6 +35947,9 @@ impl Interpreter { op: CompareKind, globals: &Rc>, ) -> Result { + if let Some(r) = native_scalar_compare(a, b, op) { + return Ok(r); + } // CPython's `PyObject_RichCompareBool` truth-tests the comparison // *result* with `PyObject_IsTrue`, so `arr == x` yielding a // multi-element numpy bool array raises "truth value ... ambiguous" @@ -35885,6 +36057,9 @@ impl Interpreter { op: CompareKind, globals: &Rc>, ) -> Result { + if let Some(r) = native_scalar_compare(a, b, op) { + return Ok(Object::Bool(r)); + } // An object-backed `mappingproxy` compares as the wrapped mapping // (CPython `mappingproxy_richcompare` delegates unconditionally). if matches!(a, Object::MappingProxyObj(_)) || matches!(b, Object::MappingProxyObj(_)) { @@ -55835,6 +56010,14 @@ enum LeafStop { Core, } +/// A step of a dict iterator in the core loop (see +/// [`Interpreter::core_dict_next`]). +enum DictStep { + Item(Object, Option), + Exhausted, + Decline, +} + /// Why [`Interpreter::leaf_core`] handed control back. enum CoreExit { /// End the burst. @@ -55936,6 +56119,7 @@ static CORE_LEAF_OPS: [bool; 256] = { OpCode::BinarySubscr, OpCode::StoreSubscr, OpCode::ListAppend, + OpCode::UnpackSequence, OpCode::CompareOp, OpCode::PopJumpIfFalse, OpCode::PopJumpIfTrue, @@ -63289,6 +63473,35 @@ pub(crate) fn coerce_len_result(r: Object) -> Result { } } +/// `a b` for a pair of native `int`s, `float`s or `str`s, whose +/// comparison consults no protocol; `None` for anything else. +#[inline] +fn native_scalar_compare(a: &Object, b: &Object, op: CompareKind) -> Option { + use std::cmp::Ordering; + let ord = match (a, b) { + (Object::Int(x), Object::Int(y)) => Some(x.cmp(y)), + (Object::Float(x), Object::Float(y)) => x.partial_cmp(y), + (Object::Str(x), Object::Str(y)) => { + if matches!(op, CompareKind::Eq | CompareKind::NotEq) { + let eq = SharedStr::ptr_eq(x, y) || x.as_bytes() == y.as_bytes(); + return Some(eq == matches!(op, CompareKind::Eq)); + } + // UTF-8 byte order is code point order. + Some(x.as_bytes().cmp(y.as_bytes())) + } + _ => return None, + }; + // An unordered pair (a NaN) is unequal and not ordered either way. + Some(match op { + CompareKind::Eq => ord == Some(Ordering::Equal), + CompareKind::NotEq => ord != Some(Ordering::Equal), + CompareKind::Lt => ord == Some(Ordering::Less), + CompareKind::LtE => matches!(ord, Some(Ordering::Less | Ordering::Equal)), + CompareKind::Gt => ord == Some(Ordering::Greater), + CompareKind::GtE => matches!(ord, Some(Ordering::Greater | Ordering::Equal)), + }) +} + pub(crate) fn compare_op(a: &Object, b: &Object, op: CompareKind) -> Result { // CPython lifts ``<``, ``<=``, ``>``, ``>=`` to subset/superset // tests on the set family. They are *not* total orderings, so we From 0d8c07ab4addb7ede1ebe5101fcd0e096332b4c4 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 20:18:51 -0700 Subject: [PATCH 04/65] perf: trim calls, generators, JIT dispatch, random, and raises - Make `builtin_types()` return a leaked `&'static` registry instead of cloning an `Rc` through a thread-local `RefCell` at every call site. - Run `isinstance` in the core loop's in-place builtin arm and grade dropped type operands inline. - Serve the JIT's dict reads with the native `LeafProbe` lookup. - Retire compiled callees that keep calling back into the interpreter from native-to-native entries (deltablue's method pattern ran 4x slower compiled than interpreted). - Run `random.Random` on its state bytearray in place instead of copying and reallocating 2.5 KB per call (`random()` was 14x CPython). - Derive the core loop's cold per-activation state lazily, so calls, returns and helper handoffs reload less. - Park yielding generators and deliver clean returns directly, without the generic exit plumbing; pass the recursion depth cell through. - Construct built-in exceptions before the conversion-constructor chain, share code objects in traceback entries, and intern the slot keys every raise sets. - Allocate function `__dict__`s on first use and stop mirroring every disk-cached frozen module's code in memory. - Add an opt-in allocation-site profiler (`--features alloc-profile`). Also includes the getattr/hasattr miss precheck, the slot wrapper miss cache, single-pass exception class matching, and deferring JIT warm compiles during startup. --- crates/weavepy-capi/src/pep741.rs | 4 +- crates/weavepy-cli/Cargo.toml | 3 + crates/weavepy-cli/src/alloc_profile.rs | 211 +++++++ crates/weavepy-cli/src/lib.rs | 13 + crates/weavepy-vm/src/builtin_types.rs | 57 +- crates/weavepy-vm/src/builtins.rs | 17 +- crates/weavepy-vm/src/error.rs | 24 +- crates/weavepy-vm/src/frozen_code_cache.rs | 5 +- crates/weavepy-vm/src/gc_trace.rs | 8 +- crates/weavepy-vm/src/lib.rs | 574 ++++++++++++-------- crates/weavepy-vm/src/object.rs | 20 +- crates/weavepy-vm/src/stdlib/abc_mod.rs | 17 +- crates/weavepy-vm/src/stdlib/random_core.rs | 180 +++--- crates/weavepy-vm/src/tier2.rs | 120 +++- crates/weavepy-vm/src/types.rs | 50 +- crates/weavepy-vm/src/vm_singletons.rs | 2 +- crates/weavepy/src/lib.rs | 4 +- 17 files changed, 959 insertions(+), 350 deletions(-) create mode 100644 crates/weavepy-cli/src/alloc_profile.rs diff --git a/crates/weavepy-capi/src/pep741.rs b/crates/weavepy-capi/src/pep741.rs index 2d5b096e..06d90417 100644 --- a/crates/weavepy-capi/src/pep741.rs +++ b/crates/weavepy-capi/src/pep741.rs @@ -29,8 +29,8 @@ use weavepy_vm::object::Object; use crate::embed::PyInitFn; use crate::initconfig::{ - self, _PyStatus_TYPE_ERROR, _PyStatus_TYPE_EXIT, EmbedConfig, PyConfig, - PyConfig_InitIsolatedConfig, PyStatus, PyWideStringList, + self, EmbedConfig, PyConfig, PyConfig_InitIsolatedConfig, PyStatus, PyWideStringList, + _PyStatus_TYPE_ERROR, _PyStatus_TYPE_EXIT, }; use crate::object::PyObject; diff --git a/crates/weavepy-cli/Cargo.toml b/crates/weavepy-cli/Cargo.toml index c06f3259..9d1c0124 100644 --- a/crates/weavepy-cli/Cargo.toml +++ b/crates/weavepy-cli/Cargo.toml @@ -60,6 +60,9 @@ windows-sys = { workspace = true } default = ["jit"] # RFC 0032 — build the `weavepy` binary with the tier-2 JIT compiled in. jit = ["weavepy/jit", "weavepy-vm/jit"] +# Sample live allocations by call stack (`WEAVEPY_ALLOC_PROFILE=`, +# macOS only); see `src/alloc_profile.rs`. +alloc-profile = [] [lints] workspace = true diff --git a/crates/weavepy-cli/src/alloc_profile.rs b/crates/weavepy-cli/src/alloc_profile.rs new file mode 100644 index 00000000..b3e48335 --- /dev/null +++ b/crates/weavepy-cli/src/alloc_profile.rs @@ -0,0 +1,211 @@ +//! Opt-in allocation-site profiler (`--features alloc-profile`, macOS). +//! +//! The global allocator samples one allocation per [`SAMPLE`] bytes and +//! records its call stack. A sampled block is charged [`SAMPLE`] bytes to +//! its stack while it lives, so at exit the table estimates the *live* heap +//! by allocation site. With `WEAVEPY_ALLOC_PROFILE=` set, the live +//! samples are written to ``: one line per sample, the charged bytes +//! then the slide-adjusted return addresses (resolve them with `atos`). + +use std::alloc::{GlobalAlloc, Layout}; +use std::cell::Cell; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +/// Bytes between samples. +const SAMPLE: usize = 16 * 1024; +/// Return addresses kept per sample. +const DEPTH: usize = 24; +/// Sampled blocks tracked at once (open addressing, power of two). +const SLOTS: usize = 1 << 18; + +struct Sample { + ptr: usize, + stack: [usize; DEPTH], +} + +static ENABLED: AtomicBool = AtomicBool::new(false); +static LOCK: AtomicBool = AtomicBool::new(false); +static TABLE: AtomicUsize = AtomicUsize::new(0); + +thread_local! { + static IN_HOOK: Cell = const { Cell::new(false) }; + static UNTIL_SAMPLE: Cell = const { Cell::new(SAMPLE) }; +} + +extern "C" { + fn _dyld_get_image_vmaddr_slide(image_index: u32) -> isize; +} + +fn lock() { + while LOCK + .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) + .is_err() + { + std::hint::spin_loop(); + } +} + +fn unlock() { + LOCK.store(false, Ordering::Release); +} + +fn table() -> *mut Sample { + TABLE.load(Ordering::Acquire) as *mut Sample +} + +fn slot_of(ptr: usize) -> usize { + (ptr.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> 20) & (SLOTS - 1) +} + +/// Start sampling (allocates the table from the system allocator). +pub(crate) fn start() { + if std::env::var_os("WEAVEPY_ALLOC_PROFILE").is_none() { + return; + } + let bytes = SLOTS * std::mem::size_of::(); + // SAFETY: a fresh zeroed mapping owned for the rest of the process. + let p = unsafe { + libc::mmap( + std::ptr::null_mut(), + bytes, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANON, + -1, + 0, + ) + }; + if p == libc::MAP_FAILED { + return; + } + TABLE.store(p as usize, Ordering::Release); + ENABLED.store(true, Ordering::Release); +} + +fn record(ptr: usize) { + let mut stack = [0usize; DEPTH]; + // SAFETY: `backtrace` fills at most `DEPTH` entries of the buffer. + let n = unsafe { libc::backtrace(stack.as_mut_ptr().cast(), DEPTH as libc::c_int) }; + let _ = n; + let t = table(); + lock(); + let mut i = slot_of(ptr); + for _ in 0..SLOTS { + // SAFETY: `i < SLOTS`, inside the mapping. + let s = unsafe { &mut *t.add(i) }; + if s.ptr == 0 || s.ptr == ptr { + s.ptr = ptr; + s.stack = stack; + break; + } + i = (i + 1) & (SLOTS - 1); + } + unlock(); +} + +fn forget(ptr: usize) { + let t = table(); + lock(); + let mut i = slot_of(ptr); + for _ in 0..SLOTS { + // SAFETY: as in `record`. + let s = unsafe { &mut *t.add(i) }; + if s.ptr == 0 { + break; + } + if s.ptr == ptr { + // A tombstone keeps later probes of this chain reachable. + s.ptr = usize::MAX; + break; + } + i = (i + 1) & (SLOTS - 1); + } + unlock(); +} + +/// The profiling allocator: mimalloc plus the sampler. +pub(crate) struct Profiled; + +// SAFETY: every allocation is mimalloc's; the sampler only records +// addresses and never touches the blocks. +unsafe impl GlobalAlloc for Profiled { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + // SAFETY: forwarded unchanged. + let p = unsafe { mimalloc::MiMalloc.alloc(layout) }; + if ENABLED.load(Ordering::Relaxed) && !p.is_null() { + sample(p as usize, layout.size()); + } + p + } + + unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { + if ENABLED.load(Ordering::Relaxed) { + forget(ptr as usize); + } + // SAFETY: forwarded unchanged. + unsafe { mimalloc::MiMalloc.dealloc(ptr, layout) } + } + + unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { + if ENABLED.load(Ordering::Relaxed) { + forget(ptr as usize); + } + // SAFETY: forwarded unchanged. + let p = unsafe { mimalloc::MiMalloc.realloc(ptr, layout, new_size) }; + if ENABLED.load(Ordering::Relaxed) && !p.is_null() { + sample(p as usize, new_size); + } + p + } +} + +fn sample(ptr: usize, size: usize) { + let _ = IN_HOOK.try_with(|hook| { + if hook.get() { + return; + } + let due = UNTIL_SAMPLE.with(|left| { + let l = left.get(); + if size >= l { + left.set(SAMPLE); + true + } else { + left.set(l - size); + false + } + }); + if due { + hook.set(true); + record(ptr); + hook.set(false); + } + }); +} + +/// Write the live samples (see the module docs). +pub(crate) fn finish() { + let Some(path) = std::env::var_os("WEAVEPY_ALLOC_PROFILE") else { + return; + }; + if !ENABLED.swap(false, Ordering::AcqRel) { + return; + } + let t = table(); + // SAFETY: the main executable is image 0. + let slide = unsafe { _dyld_get_image_vmaddr_slide(0) } as usize; + let mut out = String::new(); + lock(); + for i in 0..SLOTS { + // SAFETY: `i < SLOTS`. + let s = unsafe { &*t.add(i) }; + if s.ptr == 0 || s.ptr == usize::MAX { + continue; + } + out.push_str(&SAMPLE.to_string()); + for &a in s.stack.iter().take_while(|a| **a != 0) { + out.push_str(&format!(" {:x}", a.wrapping_sub(slide))); + } + out.push('\n'); + } + unlock(); + let _ = std::fs::write(path, out); +} diff --git a/crates/weavepy-cli/src/lib.rs b/crates/weavepy-cli/src/lib.rs index 89f45eb5..4ae44b6a 100644 --- a/crates/weavepy-cli/src/lib.rs +++ b/crates/weavepy-cli/src/lib.rs @@ -39,9 +39,18 @@ use weavepy::{InterpreterFlags, RunOptions}; /// The process allocator. The interpreter allocates and frees small blocks /// constantly; mimalloc serves them from per-thread free lists, several /// times faster than the system allocator on macOS. +#[cfg(not(all(feature = "alloc-profile", target_os = "macos")))] #[global_allocator] static GLOBAL_ALLOC: mimalloc::MiMalloc = mimalloc::MiMalloc; +#[cfg(all(feature = "alloc-profile", target_os = "macos"))] +mod alloc_profile; + +/// The allocation-site profiler's allocator (see [`alloc_profile`]). +#[cfg(all(feature = "alloc-profile", target_os = "macos"))] +#[global_allocator] +static GLOBAL_ALLOC: alloc_profile::Profiled = alloc_profile::Profiled; + const VERSION: &str = env!("CARGO_PKG_VERSION"); /// Opt-in in-process PC sampler (`WEAVEPY_PCPROF=`): a `SIGPROF` @@ -744,8 +753,12 @@ fn run_on_large_stack(entry: fn() -> i32) -> i32 { // fork-warning check measures additional threads against. weavepy::vm::stdlib::os_process::capture_thread_baseline(); pcprof::start(); + #[cfg(all(feature = "alloc-profile", target_os = "macos"))] + alloc_profile::start(); let code = entry(); pcprof::finish(); + #[cfg(all(feature = "alloc-profile", target_os = "macos"))] + alloc_profile::finish(); code }; diff --git a/crates/weavepy-vm/src/builtin_types.rs b/crates/weavepy-vm/src/builtin_types.rs index 8fcba4d6..46ae4e32 100644 --- a/crates/weavepy-vm/src/builtin_types.rs +++ b/crates/weavepy-vm/src/builtin_types.rs @@ -1340,6 +1340,9 @@ thread_local! { const { std::cell::RefCell::new(None) }; static PROPERTY_CLASS: std::cell::RefCell>> = const { std::cell::RefCell::new(None) }; + /// The adopted registry behind [`builtin_types`]'s fast path. + static BUILTIN_TYPES_PTR: std::cell::Cell<*const BuiltinTypes> = + const { std::cell::Cell::new(std::ptr::null()) }; } /// Drop this thread's lazily-built type registry (and the derived @@ -1352,6 +1355,7 @@ thread_local! { pub fn clear_thread_type_registry() { let _ = PROPERTY_CLASS.try_with(|slot| slot.borrow_mut().take()); let _ = BUILTIN_TYPES.try_with(|slot| slot.borrow_mut().take()); + let _ = BUILTIN_TYPES_PTR.try_with(|p| p.set(std::ptr::null())); } /// Per-thread accessor. The registry is constructed lazily on first @@ -1370,7 +1374,30 @@ pub fn property_class() -> Rc { }) } -pub fn builtin_types() -> Rc { +/// This thread's type registry. A registry is never freed once a thread +/// has adopted it (see [`adopt_registry`]), so the reference is valid for +/// the rest of the process: the hot accessor is one thread-local load, with +/// no borrow flag and no reference-count traffic. +#[inline] +pub fn builtin_types() -> &'static BuiltinTypes { + let p = BUILTIN_TYPES_PTR.with(std::cell::Cell::get); + if p.is_null() { + return builtin_types_init(); + } + // SAFETY: `adopt_registry` leaked a strong count before publishing the + // pointer, so the registry outlives every thread that can read it. + unsafe { &*p } +} + +/// [`builtin_types`] as an owned handle (for publishing to other threads). +pub fn builtin_types_rc() -> Rc { + builtin_types(); + BUILTIN_TYPES.with(|cell| cell.borrow().clone().expect("registry installed above")) +} + +#[cold] +#[inline(never)] +fn builtin_types_init() -> &'static BuiltinTypes { let (bt, fresh) = BUILTIN_TYPES.with(|cell| { if let Some(bt) = cell.borrow().as_ref() { return (bt.clone(), false); @@ -1379,15 +1406,29 @@ pub fn builtin_types() -> Rc { *cell.borrow_mut() = Some(bt.clone()); (bt, true) }); + let bt = adopt_registry(&bt); if fresh { // Deferred surface pass (RFC 0056 WS4): synthesizing descriptor- // type members re-enters `builtin_types()`, which must resolve to // the just-published cell rather than recursively rebuild. - crate::type_surface::install_docs_table_surface(&bt); + crate::type_surface::install_docs_table_surface(bt); } bt } +/// Publish `bt` as this thread's registry for [`builtin_types`], leaking +/// one strong count so the reference it hands out stays valid. The type +/// objects inside already live for the process (each one's MRO holds +/// itself), so the leak is the registry struct alone, once per thread +/// that adopts one. +fn adopt_registry(bt: &Rc) -> &'static BuiltinTypes { + let p = Rc::as_ptr(bt); + std::mem::forget(bt.clone()); + BUILTIN_TYPES_PTR.with(|c| c.set(p)); + // SAFETY: the leaked count above keeps the registry alive. + unsafe { &*p } +} + /// Resolve `__objclass__` for a built-in method/slot-wrapper object by /// locating the built-in type whose dict holds this exact descriptor /// (CPython stores the owner in the descriptor itself; we recover it @@ -4818,9 +4859,15 @@ fn install_exception_str_repr(base_exception: &Rc) { pub fn make_exception_with_class(class: Rc, message: impl Into) -> Object { use crate::types::PyInstance; - let is_syntax = is_subclass_by_name(&class, "SyntaxError"); - let is_stop_iteration = is_subclass_by_name(&class, "StopIteration"); - let is_import = is_subclass_by_name(&class, "ImportError"); + let (mut is_syntax, mut is_stop_iteration, mut is_import) = (false, false, false); + for t in class.mro.borrow().iter() { + match t.name.as_str() { + "SyntaxError" => is_syntax = true, + "StopIteration" => is_stop_iteration = true, + "ImportError" => is_import = true, + _ => {} + } + } let inst = PyInstance::new(class); let msg = Object::from_str(message); // A messageless raise (`StopIteration()`, `GeneratorExit()`, …) diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index 88ce3c5c..2c9967bb 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -4117,12 +4117,7 @@ fn attr_get(obj: &Object, name: &str) -> Option { if let Some(v) = f.slot(name) { return Some(v); } - } else if let Some(v) = f - .attrs() - .borrow() - .get(&crate::object::DictKey(Object::from_str(name))) - .cloned() - { + } else if let Some(v) = f.attr_get(name) { return Some(v); } // Synthetic dunders. Mirror `Vm::load_attr`'s function @@ -9547,9 +9542,11 @@ pub fn b_dir(args: &[Object]) -> Result { // transplants pytest marks onto its wrapper that way, so a dir // that hid `f.pytestmark` silently dropped every // `@pytest.mark.parametrize` stacked under `@given`. - for k in f.attrs().borrow().keys() { - if let Object::Str(s) = &k.0 { - names.insert(s.to_string()); + if let Some(attrs) = f.attrs.borrow().as_ref() { + for k in attrs.borrow().keys() { + if let Object::Str(s) = &k.0 { + names.insert(s.to_string()); + } } } for n in [ @@ -10426,7 +10423,7 @@ fn b_mark_iterable_coroutine(args: &[Object]) -> Result { closure: f.closure.clone(), // Shared, not copied: `func.__dict__` mutations stay visible on // both, matching CPython where the function object is the same. - attrs: RefCell::new(f.attrs()), + attrs: RefCell::new(Some(f.attrs())), slots: RefCell::new(f.slots.borrow().clone()), closure_cells: std::sync::OnceLock::new(), // The copied slot store carries any override along. diff --git a/crates/weavepy-vm/src/error.rs b/crates/weavepy-vm/src/error.rs index f0413556..2a1a5c75 100644 --- a/crates/weavepy-vm/src/error.rs +++ b/crates/weavepy-vm/src/error.rs @@ -11,14 +11,32 @@ use thiserror::Error; use crate::object::Object; -/// A traceback frame captured as the exception unwinds. +/// A traceback frame captured as the exception unwinds: the frame's code +/// object (shared, so a raise copies no strings) and its line. #[derive(Debug, Clone)] pub struct TracebackEntry { - pub filename: String, - pub funcname: String, + pub code: crate::sync::Rc, pub lineno: u32, } +impl TracebackEntry { + pub fn filename(&self) -> &str { + &self.code.filename + } + + pub fn funcname(&self) -> &str { + &self.code.name + } +} + +impl PartialEq for TracebackEntry { + fn eq(&self, other: &Self) -> bool { + self.lineno == other.lineno + && self.filename() == other.filename() + && self.funcname() == other.funcname() + } +} + /// A Python-visible exception. The wrapped [`Object`] is always an /// `Object::Instance` whose class's MRO contains `BaseException`. #[derive(Debug, Clone)] diff --git a/crates/weavepy-vm/src/frozen_code_cache.rs b/crates/weavepy-vm/src/frozen_code_cache.rs index af0c2f1e..749f86e7 100644 --- a/crates/weavepy-vm/src/frozen_code_cache.rs +++ b/crates/weavepy-vm/src/frozen_code_cache.rs @@ -178,7 +178,10 @@ pub fn get_disk(name: &str, source: &str, filename: &str) -> Option if code.filename != filename { crate::pycache::rewrite_filenames(&mut code, filename); } - insert(name, &code); + // Not mirrored into the in-memory cache: a later interpreter + // reads the same artifact again, and a resident clone of every + // loaded module's code would double its footprint in the + // (usual) single-interpreter process. Some(code) } _ => None, diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 623b3274..c43e4323 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -2954,7 +2954,7 @@ pub fn traverse_object(obj: &Object, visit: &mut dyn FnMut(&Object)) { } } } - if let Ok(attrs_rc) = f.attrs.try_borrow() { + if let Some(attrs_rc) = f.attrs.try_borrow().ok().and_then(|a| a.clone()) { if let Ok(attrs) = attrs_rc.try_borrow() { for (k, v) in attrs.iter() { visit(&k.0); @@ -3159,7 +3159,7 @@ pub fn clear_object_fields(obj: &Object) -> bool { // dict (a module's `__dict__` or the `exec` target), reclaimed as // its own candidate if it too is unreachable — clearing it here // could wipe a live module. - if let Ok(attrs_rc) = f.attrs.try_borrow() { + if let Some(attrs_rc) = f.attrs.try_borrow().ok().and_then(|a| a.clone()) { if let Ok(mut attrs) = attrs_rc.try_borrow_mut() { attrs.clear(); } @@ -4378,6 +4378,10 @@ pub fn note_dropped_marks(obj: &crate::object::Object) -> bool { crate::sync::Rc::strong_count(f), crate::sync::Rc::as_ptr(f) as usize as u64, ), + O::Type(t) => note_dropped_counted( + crate::sync::Rc::strong_count(t), + crate::sync::Rc::as_ptr(t) as usize as u64, + ), O::Tuple(t) => note_dropped_counted( ThinArc::strong_count(t), ThinArc::as_ptr(t).cast::<()>() as usize as u64, diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index ff50b8c7..dd8fccd6 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -2738,8 +2738,10 @@ impl Interpreter { let Ok(attrs) = f.attrs.try_borrow() else { return false; }; - if Rc::strong_count(&attrs) == 1 && !attrs.try_borrow().is_ok_and(|d| d.is_empty()) { - return false; + if let Some(attrs) = attrs.as_ref() { + if Rc::strong_count(attrs) == 1 && !attrs.try_borrow().is_ok_and(|d| d.is_empty()) { + return false; + } } let Ok(slots) = f.slots.try_borrow() else { return false; @@ -4905,8 +4907,7 @@ impl Interpreter { let mut cur = Some(tb); while let Some(node) = cur { entries.push(crate::error::TracebackEntry { - filename: node.frame.code.filename.clone(), - funcname: node.frame.code.name.clone(), + code: node.frame.code.clone(), lineno: node.lineno, }); cur = node.next.borrow().clone(); @@ -4960,7 +4961,9 @@ impl Interpreter { for e in traceback { s.push_str(&format!( " File \"{}\", line {}, in {}\n", - e.filename, e.lineno, e.funcname + e.filename(), + e.lineno, + e.funcname() )); } } @@ -9430,7 +9433,9 @@ impl Interpreter { && matches!(frame.stack.last(), Some(Object::Generator(_))) && self.inline_calls_ok() { - if let Some(act) = self.try_inline_gen(frame, shell, cur_pc) { + if let Some(act) = + self.try_inline_gen(frame, shell, cur_pc, crate::recursion::depth_cell()) + { return FrameEv::Call(act); } } @@ -9643,7 +9648,6 @@ impl Interpreter { /// Park a finished inline slot (its callable already taken): release /// what the activation owned and return the slot to the pool. fn inline_park(&mut self, mut act: Box) { - const POOL_CAP: usize = 64; debug_assert!(!act.parked); let fr: &mut Frame = &mut act.frame; // Leftover operands drop before anything is reused (their drop @@ -9707,7 +9711,7 @@ impl Interpreter { act.guard = None; act.caller_pending = None; act.init_inst = None; - if self.inline_pool.len() < POOL_CAP { + if self.inline_pool.len() < INLINE_POOL_CAP { self.inline_pool.push(act); } } @@ -9721,6 +9725,7 @@ impl Interpreter { frame: &mut Frame, shell: &mut QuietShell<'_>, pc: usize, + depth_cell: *const std::cell::Cell, ) -> Option> { let arg = frame.code.instructions.get(pc)?.arg; let Some(Object::Generator(g)) = frame.stack.last() else { @@ -9756,7 +9761,7 @@ impl Interpreter { } } // Past the recursion limit the lean path raises. - let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter_with(depth_cell) else { return None; }; // Committed. @@ -9887,7 +9892,6 @@ impl Interpreter { mut done: Box, result: Result, ) -> QuietEntry { - const POOL_CAP: usize = 64; let call_pc = done.call_pc; let arg = done.exhaust_arg; self.lean_pending_exit(done.caller_pending); @@ -9896,7 +9900,7 @@ impl Interpreter { done.act.shell = None; done.act.frame = std::ptr::from_mut::(&mut done.frame); done.caller_pending = None; - if self.inline_pool.len() < POOL_CAP { + if self.inline_pool.len() < INLINE_POOL_CAP { self.inline_pool.push(done); } match result { @@ -10075,7 +10079,7 @@ impl Interpreter { /// code); a bound one is just graded. The usual callee — a function /// its namespace still holds (our clone, that binding, its collector /// handle: three or more) with no weakref — needs neither. - #[inline] + #[inline(always)] fn drop_lean_callable( &mut self, frame: &mut Frame, @@ -10091,7 +10095,20 @@ impl Interpreter { }; if alive_elsewhere { drop_hot(callable); - } else if Self::looks_reapable_temporary(&callable) { + } else { + self.drop_lean_callable_slow(frame, shell, callable); + } + } + + /// [`Self::drop_lean_callable`] for a callable that may die here. + #[inline(never)] + fn drop_lean_callable_slow( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + callable: Object, + ) { + if Self::looks_reapable_temporary(&callable) { self.flush_lean(frame, shell); self.prompt_reap_dropped(callable); } else { @@ -11550,6 +11567,28 @@ impl Interpreter { Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None ) } + // The colder per-activation state (see the prologue note), read at + // its uses from the running activation's frame and code extension. + macro_rules! stamps { + ($cache:ident, $frame:expr, $ext:expr) => { + *$cache.get_or_insert_with(|| { + if $frame.builtins_obj.is_none() { + $ext.and_then(|e| e.stamp_slots.get()) + .map_or(&[][..], |s| &s[..]) + } else { + &[][..] + } + }) + }; + } + macro_rules! mslots { + ($cache:ident, $ext:expr) => { + *$cache.get_or_insert_with(|| { + $ext.and_then(|e| e.method_slots.get()) + .map_or(&[][..], |s| &s[..]) + }) + }; + } 'reload: loop { // (A discriminant test first: `take` would copy the whole exit.) if sw.pending.is_some() { @@ -11571,26 +11610,18 @@ impl Interpreter { // during an activation, and nothing here runs Python code). let locals: &mut Vec = unsafe { &mut *frame.locals.as_ptr() }; let (lbase, nlocals) = (locals.as_mut_ptr(), locals.len()); - // The `LOAD_GLOBAL` arm's cache-hit state (see `leaf_global`): the - // site stamps exist once any global hit filled one, and a custom - // `__builtins__` mapping keeps every load on the full path. - let stamps: &[StampSlot] = if frame.builtins_obj.is_none() { - ext.and_then(|e| e.stamp_slots.get()) - .map_or(&[], |s| &s[..]) - } else { - &[] - }; - // The method-form `LOAD_ATTR` arm's per-site functions. - let mslots: &[MethodSlot] = ext - .and_then(|e| e.method_slots.get()) - .map_or(&[], |s| &s[..]); let (gdict, bdict) = (frame.globals.as_ptr(), frame.builtins.as_ptr()); let (gid, bid) = ( specialize::rc_id(&frame.globals), specialize::rc_id(&frame.builtins), ); - // Module scope (names resolve in the globals, then the builtins, - // both exact dicts): the `LOAD_NAME` / `STORE_NAME` arms below. + // The arms' colder per-activation state (the global and class + // attribute stamps, the method slots) is derived at its first + // use: every call, return and helper handoff runs this prologue + // again, and most activations never read it. + let mut cold_stamps: Option<&[StampSlot]> = None; + let mut cold_mslots: Option<&[MethodSlot]> = None; + // // The fused local pairs below, one byte per instruction (empty // for code that has none). let fast_pairs: &[u8] = if crate::hot_gates::env_flags::no_pairs() { @@ -11598,11 +11629,6 @@ impl Interpreter { } else { code_fast_pairs(code, ext) }; - let name_scope = frame.class_namespace.is_none() - && frame.class_namespace_obj.is_none() - && frame.builtins_obj.is_none() - && !frame.code.is_class_body - && !self.globals_missing_any.get(); let stack = &mut frame.stack; let base = stack.as_mut_ptr(); let cap = stack.capacity(); @@ -11731,7 +11757,10 @@ impl Interpreter { && len < cap && simple_args_prefix(&code.instructions, pc + 2) { - let pure_site = match (other, mslots.get(pc + 1)) { + let pure_site = match ( + other, + mslots!(cold_mslots, ext).get(pc + 1), + ) { (Object::Instance(i), Some(ms)) => ms .peek_fn(i.cls_raw().attr_version.get()) .is_some_and(|fp| fn_is_pure_leaf(&*fp)), @@ -11743,7 +11772,7 @@ impl Interpreter { other, pc + 1, next.arg, - mslots, + mslots!(cold_mslots, ext), lbase, nlocals, consts, @@ -12611,7 +12640,7 @@ impl Interpreter { // builtin or type callee has leaf arms. // Module-scope names: the globals, then the builtins, probed // with the interned name's hash (no per-read name object). - OpCode::LoadName if name_scope => { + OpCode::LoadName if self.core_name_scope(frame) => { if len == cap { break None; } @@ -12642,7 +12671,7 @@ impl Interpreter { // Rebinding an existing module-scope name whose old value // leaves by a plain drop; a new name, a finalizer's // candidate or a watched dict takes the full handler. - OpCode::StoreName if name_scope => { + OpCode::StoreName if self.core_name_scope(frame) => { if len == 0 || crate::capi_watchers::dicts_active() { break None; } @@ -12766,6 +12795,8 @@ impl Interpreter { } break Some(CoreExit::Reload); } + // The site's method slot (its leaf kind), read once. + let site_slot = mslots!(cold_mslots, ext).get(pc); // SAFETY: `len >= argc + 2`. match unsafe { &*base.add(len - argc - 2) } { // `lst.append(x)` / `lst.pop()` on an exact list (the @@ -12778,7 +12809,7 @@ impl Interpreter { // SAFETY: `len >= argc + 2`: the self slot. && matches!(unsafe { &*base.add(len - argc - 1) }, Object::List(_)) => { - let kind = mslots.get(pc).and_then(|s| s.get_leaf(b)); + let kind = site_slot.and_then(|s| s.get_leaf(b)); // SAFETY: `len >= argc + 2`: the self slot. let recv = unsafe { &*base.add(len - argc - 1) }; let Object::List(l) = recv else { @@ -12828,11 +12859,15 @@ impl Interpreter { Object::Builtin(b) if Rc::strong_count(b) > 1 && matches!( - mslots.get(pc).and_then(|s| s.get_leaf(b)), - Some(LeafKind::Opaque | LeafKind::Fast(_)) + site_slot.and_then(|s| s.get_leaf(b)), + Some( + LeafKind::Opaque + | LeafKind::Fast(_) + | LeafKind::Isinstance + ) ) => { - let kind = mslots.get(pc).and_then(|s| s.get_leaf(b)); + let kind = site_slot.and_then(|s| s.get_leaf(b)); let callee_at = len - argc - 2; // SAFETY: `len >= argc + 2`: the self slot. let first = if matches!( @@ -12852,6 +12887,7 @@ impl Interpreter { } let r = match kind { Some(LeafKind::Fast(f)) => f(ops), + Some(LeafKind::Isinstance) => Self::core_isinstance(ops), _ => Some(match b.call_kw.as_ref() { Some(ckw) => ckw(ops, &[]), None => (b.call)(ops), @@ -12893,7 +12929,7 @@ impl Interpreter { Object::BoundMethod(bm) if argc == 0 && matches!(&bm.function, Object::Builtin(_)) => { - if let Some(slot) = mslots.get(pc) { + if let Some(slot) = site_slot { if slot.is_non_leaf(bm) { // A prior body/receiver rejection still // applies to this immutable bound method. @@ -12921,7 +12957,7 @@ impl Interpreter { // with `len(local)` fused as `leaf_fused_len_at` does. OpCode::LoadGlobal => { use weavepy_compiler::InlineCache as IC; - let Some(slot) = stamps.get(pc) else { + let Some(slot) = stamps!(cold_stamps, frame, ext).get(pc) else { break Some(CoreExit::Helper); }; if len == cap { @@ -13040,12 +13076,12 @@ impl Interpreter { // SAFETY: `pc + 1 < ninstrs`. && unsafe { (*instrs.add(pc + 1)).op } == OpCode::LoadMethodAttr && simple_args_prefix(&code.instructions, pc + 2) - && mslots + && mslots!(cold_mslots, ext) .get(pc + 1) .and_then(|ms| ms.peek_unbound(cls.attr_version.get())) .is_some_and(|fp| fn_is_pure_leaf(unsafe { &*fp })) => { - let fp = mslots + let fp = mslots!(cold_mslots, ext) .get(pc + 1) .and_then(|ms| ms.peek_unbound(cls.attr_version.get())) .expect("checked by the guard"); @@ -13077,7 +13113,10 @@ impl Interpreter { // SAFETY: `pc + 1 < ninstrs`. && unsafe { (*instrs.add(pc + 1)).op } == OpCode::LoadAttr => { - match stamps.get(pc + 1).and_then(|s| class_attr_hit(s, cls)) { + match stamps!(cold_stamps, frame, ext) + .get(pc + 1) + .and_then(|s| class_attr_hit(s, cls)) + { Some(c) => { // SAFETY: `len < cap`. unsafe { base.add(len).write(c) }; @@ -13178,7 +13217,7 @@ impl Interpreter { // receiver moves up into the self slot under the function; // a class receiver leaves an empty self slot. OpCode::LoadMethodAttr => { - let Some(ms) = mslots.get(pc) else { + let Some(ms) = mslots!(cold_mslots, ext).get(pc) else { break Some(CoreExit::Helper); }; if len == 0 || len == cap { @@ -13300,7 +13339,10 @@ impl Interpreter { // stamp); the class (count above one) leaves by a // plain decrement. Object::Type(cls) => { - match stamps.get(pc).and_then(|s| class_attr_hit(s, cls)) { + match stamps!(cold_stamps, frame, ext) + .get(pc) + .and_then(|s| class_attr_hit(s, cls)) + { Some(v) if Rc::strong_count(cls) > 1 => { // SAFETY: the receiver is replaced in place. unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; @@ -13895,6 +13937,40 @@ impl Interpreter { } } + /// Module scope for the core loop's `LOAD_NAME` / `STORE_NAME` arms: + /// names resolve in the globals, then the builtins, both exact dicts. + #[inline(always)] + fn core_name_scope(&self, frame: &Frame) -> bool { + frame.class_namespace.is_none() + && frame.class_namespace_obj.is_none() + && frame.builtins_obj.is_none() + && !frame.code.is_class_body + && !self.globals_missing_any.get() + } + + /// The core loop's `isinstance(obj, cls)` (the leaf kind's settled + /// shapes, see `leaf_builtin_call`); `None` declines untouched. + #[inline(never)] + fn core_isinstance(ops: &[Object]) -> Option> { + let [obj, Object::Type(cls)] = ops else { + return None; + }; + if !cls.metaclass_is_type() { + return None; + } + let r = match obj { + Object::Instance(inst) => { + if !inst.cls_raw().is_subclass_of(cls) { + return None; + } + true + } + Object::File(_) => return None, + obj => builtins::class_of(obj).is_subclass_of(cls), + }; + Some(Ok(Object::Bool(r))) + } + /// Whether a returning frame's `locals` owe the exit reap /// (`reap_frame_locals_on_exit`) nothing: scalars and strings, and /// instances or containers held beyond every slot of the frame that @@ -14972,7 +15048,12 @@ impl Interpreter { let act = unsafe { let depth = (*sw.inl).len(); let (frame, _, shell) = sw.activation(depth, &mut tmp); - self.try_inline_gen(&mut *frame, &mut *shell.cast::>(), pc) + self.try_inline_gen( + &mut *frame, + &mut *shell.cast::>(), + pc, + sw.depth_cell, + ) }; let Some(act) = act else { return false; @@ -15019,25 +15100,28 @@ impl Interpreter { if depth == sw.entry_depth { sw.entry_dead = true; } - let result = self.inline_gen_finish( - &mut done, - QuietExit::Outcome { - stepped: Ok(StepOutcome::Yield(v)), - cur_pc, - }, - ); + // `inline_gen_finish`'s shell-less yield (the generator parks + // again) and `inline_gen_deliver`'s yielded value, directly. + drop(done.guard.take()); + let gen = done.gen.take().expect("a generator activation"); + let boxed = done.gen_box.take().expect("a generator activation"); + done.gen_frame = std::ptr::null_mut(); + Self::park_suspended_boxed(&gen, boxed); + drop(gen); + let call_pc = done.call_pc; + self.lean_pending_exit(done.caller_pending); + done.act.shell = None; + done.act.frame = std::ptr::from_mut::(&mut done.frame); + done.caller_pending = None; + if self.inline_pool.len() < INLINE_POOL_CAP { + self.inline_pool.push(done); + } let mut tmp = None; // SAFETY: the consumer is the innermost remaining activation. let (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; // SAFETY: as above. - let entry = unsafe { - self.inline_gen_deliver( - &mut *cframe, - &mut *cshell.cast::>(), - done, - result, - ) - }; + unsafe { (*cframe).stack.push(v) }; + let entry = QuietEntry::Returned { cur_pc: call_pc }; sw.cur = cframe; sw.last = if clast == &raw mut sw.scratch { sw.scratch = usize::MAX; @@ -15120,12 +15204,14 @@ impl Interpreter { // leftover operands or cells, and locals that die here with no drop // glue (scalars) or whose release the exit reap would pass over // (see `core_escaped_locals`). - let result = if frame.stack.is_empty() + let clean = frame.stack.is_empty() && frame.code.cellvars.is_empty() && Rc::strong_count(&frame.locals) == 1 // SAFETY: sole owner; nothing else reaches the vector. - && Self::core_escaped_locals(unsafe { &*frame.locals.as_ptr() }) - { + && Self::core_escaped_locals(unsafe { &*frame.locals.as_ptr() }); + let mut tmp = None; + let (cframe, clast, cshell, entry); + if clean && done.init_inst.is_none() { done.clean = true; // SAFETY: as above; the values drop by plain decrements (or // own nothing). @@ -15133,28 +15219,55 @@ impl Interpreter { drop_hot(v); } drop(done.guard.take()); - Ok(v) + // SAFETY: the caller is the innermost remaining activation. + (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; + // `inline_deliver` for a plain function's return, directly. + let call_pc = done.call_pc; + self.lean_pending_exit(done.caller_pending); + // SAFETY: the callable is moved out exactly once; the parked + // slot treats the field as stale (see `inline_deliver`). + let callable = unsafe { std::ptr::read(&raw const done.callable) }; + self.inline_park(done); + // SAFETY: as above. + unsafe { + (*cframe).stack.push(v); + self.drop_lean_callable( + &mut *cframe, + &mut *cshell.cast::>(), + callable, + ); + } + entry = QuietEntry::Returned { cur_pc: call_pc }; } else { - self.inline_finish( - &mut done, - QuietExit::Outcome { - stepped: Ok(StepOutcome::Return(v)), - cur_pc, - }, - ) - }; - let mut tmp = None; - // SAFETY: the caller is the innermost remaining activation. - let (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; - // SAFETY: as above. - let entry = unsafe { - self.inline_deliver( - &mut *cframe, - &mut *cshell.cast::>(), - done, - result, - ) - }; + let result = if clean { + done.clean = true; + // SAFETY: as above. + for v in unsafe { (*frame.locals.as_ptr()).drain(..) } { + drop_hot(v); + } + drop(done.guard.take()); + Ok(v) + } else { + self.inline_finish( + &mut done, + QuietExit::Outcome { + stepped: Ok(StepOutcome::Return(v)), + cur_pc, + }, + ) + }; + // SAFETY: the caller is the innermost remaining activation. + (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; + // SAFETY: as above. + entry = unsafe { + self.inline_deliver( + &mut *cframe, + &mut *cshell.cast::>(), + done, + result, + ) + }; + } sw.cur = cframe; sw.last = if clast == &raw mut sw.scratch { sw.scratch = usize::MAX; @@ -17988,33 +18101,7 @@ impl Interpreter { | O::FrozenSet(_) ) } - K::Isinstance => { - if args.len() != 2 { - return None; - } - let O::Type(cls) = &args[1] else { - return None; - }; - // A plain metaclass: no `__instancecheck__` hook. - if !Rc::ptr_eq(&cls.metaclass_or_type(), &builtin_types().type_) { - return None; - } - // Only a positive MRO answer settles an instance without - // consulting `__class__` (see `recursive_isinstance_type`); - // every other object's type is its `__class__`. - let r = match &args[0] { - O::Instance(inst) => { - if inst.cls_raw().is_subclass_of(cls) { - true - } else { - return None; - } - } - O::File(_) => return None, - obj => builtins::class_of(obj).is_subclass_of(cls), - }; - return Some(Ok(O::Bool(r))); - } + K::Isinstance => return Self::core_isinstance(args), K::ListAppend => args.len() == 2 && matches!(args[0], O::List(_)), K::ListPop => { (args.len() == 1 || (args.len() == 2 && leaf_int(&args[1]))) @@ -20764,12 +20851,7 @@ impl Interpreter { // dict intact. if let Object::BoundMethod(bm) = &callable { let raw = match &bm.function { - Object::Function(f) => f - .attrs - .borrow() - .borrow() - .get(&crate::object::StrKey("__weave_raw_call__")) - .cloned(), + Object::Function(f) => f.attr_get("__weave_raw_call__"), _ => None, }; if let Some(raw) = raw { @@ -21775,7 +21857,7 @@ impl Interpreter { defaults, kw_defaults, closure, - attrs: RefCell::new(Rc::new(RefCell::new(DictData::default()))), + attrs: RefCell::new(None), slots, closure_cells: std::sync::OnceLock::new(), defaults_override: crate::object::OverrideFlag::new(false), @@ -23205,8 +23287,7 @@ impl Interpreter { /// can walk `exc.__traceback__`. fn append_traceback(&self, exc: &mut PyException, frame: &mut Frame, lasti: u32, lineno: u32) { exc.push_traceback(TracebackEntry { - filename: frame.code.filename.clone(), - funcname: frame.code.name.clone(), + code: frame.code.clone(), lineno, }); // The Python-visible frame for this entry must be *this* frame's @@ -23566,9 +23647,9 @@ impl Interpreter { // matching happens — `except (ValueError, 42):` is a TypeError // even when the raised exception would match ValueError // (CPython `check_except_type` over the whole tuple). + let base_exception = &builtin_types().base_exception; let check = |t: &Object| -> Result<(), RuntimeError> { - let ok = matches!(t, Object::Type(c) - if c.mro.borrow().iter().any(|m| m.name == "BaseException")); + let ok = matches!(t, Object::Type(c) if c.is_subclass_of(base_exception)); if ok { Ok(()) } else { @@ -24646,8 +24727,8 @@ impl Interpreter { if let Some(v) = f.slot(name) { return Ok(v); } - } else if let Some(v) = f.attrs().borrow().get(&crate::object::StrKey(name)) { - return Ok(v.clone()); + } else if let Some(v) = f.attr_get(name) { + return Ok(v); } match name { // Stash the computed value on the slot so repeated @@ -30634,6 +30715,9 @@ impl Interpreter { ))), }; } + if args.len() >= 3 && attr_certainly_missing(&args[0], &name) { + return Ok(args[2].clone()); + } match self.load_attr(&args[0], &name) { Ok(v) => Ok(v), Err(e) if args.len() >= 3 && self.is_attribute_error(&e) => Ok(args[2].clone()), @@ -30664,6 +30748,9 @@ impl Interpreter { crate::builtins::wstr_attr_get(&args[0], &args[1]).is_some(), )); } + if attr_certainly_missing(&args[0], &name) { + return Ok(Object::Bool(false)); + } match self.load_attr(&args[0], &name) { Ok(_) => Ok(Object::Bool(true)), Err(e) if self.is_attribute_error(&e) => Ok(Object::Bool(false)), @@ -30676,12 +30763,10 @@ impl Interpreter { /// attribute-lookup failures. fn is_attribute_error(&self, err: &RuntimeError) -> bool { match err { - RuntimeError::PyException(pe) => self - .exception_matches( - &pe.instance, - &Object::Type(builtin_types().attribute_error.clone()), - ) - .unwrap_or(false), + RuntimeError::PyException(pe) => crate::builtin_types::instance_is_subclass( + &pe.instance, + &builtin_types().attribute_error, + ), RuntimeError::Internal(_) => false, } } @@ -38933,7 +39018,7 @@ impl Interpreter { value.type_name() ))); }; - *f.attrs.borrow_mut() = d; + *f.attrs.borrow_mut() = Some(d); return Ok(()); } "__defaults__" if !matches!(value, Object::Tuple(_) | Object::None) => { @@ -43143,7 +43228,7 @@ impl Interpreter { defaults, kw_defaults, closure, - attrs: RefCell::new(Rc::new(RefCell::new(DictData::default()))), + attrs: RefCell::new(None), slots: RefCell::new(DictData::default()), closure_cells: std::sync::OnceLock::new(), defaults_override: crate::object::OverrideFlag::new(false), @@ -45618,6 +45703,78 @@ impl Interpreter { // Built-in conversion types route to the underlying builtin // function so `int("3")`, `range(5)`, `list(xs)` keep working. if cls.flags.is_builtin { + // Exception classes first: a raise constructs one, and none of + // the conversion or descriptor constructors below is one. + if cls.flags.is_exception { + // PEP 654: `BaseExceptionGroup(msg, excs)` goes through + // the full `BaseExceptionGroup.__new__` — argument + // validation, class lowering, `exceptions` freezing. + let instance = if cls + .is_subclass_of(&crate::builtin_types::builtin_types().base_exception_group) + { + let _interp_guard = crate::vm_singletons::publish_interpreter_ptr( + std::ptr::from_mut::(self), + ); + crate::builtin_types::exception_group_new(&cls, args)? + } else { + self.build_exception_instance(cls.clone(), args) + }; + // Keyword fields accepted by the builtin constructors: + // `AttributeError(name=, obj=)`, `NameError(name=)`, + // `ImportError(name=, path=, name_from=)` — mirroring + // CPython's `*_init` C slots. Only consumed when no user + // `__init__` overrides the construction protocol. + if !kwargs.is_empty() && lookup_exception_init(&cls).is_none() { + let bt = crate::builtin_types::builtin_types(); + let allowed: &[&str] = if cls.is_subclass_of(&bt.import_error) { + &["name", "path", "name_from"] + } else if cls.is_subclass_of(&bt.attribute_error) { + &["name", "obj"] + } else if cls.is_subclass_of(&bt.name_error) { + &["name"] + } else { + &[] + }; + if let Object::Instance(inst) = &instance { + for (k, v) in kwargs { + if !allowed.contains(&k.as_str()) { + return Err(type_error(format!( + "{}() got an unexpected keyword argument '{}'", + cls.name, k + ))); + } + // These fields are exception pseudo-slots + // (class-level slot descriptors), so the value + // must land in the slot table — an instance-dict + // entry would be shadowed by the descriptor. + inst.slot_set(k, v.clone()); + } + } + } + // If a class anywhere between `cls` and `BaseException` + // (exclusive) defines its own `__init__`, run it so + // subclasses such as `BaseExceptionGroup` get to stitch + // `exceptions` onto the instance. We stop at + // `BaseException` because its default `__init__` only + // populates `args` — which the fast path already did. + if let Some(init) = lookup_exception_init(&cls) { + let bound = + Object::BoundMethod(Rc::new(BoundMethod::new(instance.clone(), init))); + let result = self.call( + &bound, + args, + kwargs, + &Rc::new(RefCell::new(DictData::default())), + )?; + if !matches!(result, Object::None) { + return Err(type_error(format!( + "__init__() should return None, not '{}'", + result.type_name() + ))); + } + } + return Ok(instance); + } // Descriptor wrapper classes (property/staticmethod/ // classmethod) — route to dedicated constructors. match cls.name.as_str() { @@ -45977,76 +46134,6 @@ impl Interpreter { } return (builtin.call)(args); } - if cls.flags.is_exception { - // PEP 654: `BaseExceptionGroup(msg, excs)` goes through - // the full `BaseExceptionGroup.__new__` — argument - // validation, class lowering, `exceptions` freezing. - let instance = if cls - .is_subclass_of(&crate::builtin_types::builtin_types().base_exception_group) - { - let _interp_guard = crate::vm_singletons::publish_interpreter_ptr( - std::ptr::from_mut::(self), - ); - crate::builtin_types::exception_group_new(&cls, args)? - } else { - self.build_exception_instance(cls.clone(), args) - }; - // Keyword fields accepted by the builtin constructors: - // `AttributeError(name=, obj=)`, `NameError(name=)`, - // `ImportError(name=, path=, name_from=)` — mirroring - // CPython's `*_init` C slots. Only consumed when no user - // `__init__` overrides the construction protocol. - if !kwargs.is_empty() && lookup_exception_init(&cls).is_none() { - let bt = crate::builtin_types::builtin_types(); - let allowed: &[&str] = if cls.is_subclass_of(&bt.import_error) { - &["name", "path", "name_from"] - } else if cls.is_subclass_of(&bt.attribute_error) { - &["name", "obj"] - } else if cls.is_subclass_of(&bt.name_error) { - &["name"] - } else { - &[] - }; - if let Object::Instance(inst) = &instance { - for (k, v) in kwargs { - if !allowed.contains(&k.as_str()) { - return Err(type_error(format!( - "{}() got an unexpected keyword argument '{}'", - cls.name, k - ))); - } - // These fields are exception pseudo-slots - // (class-level slot descriptors), so the value - // must land in the slot table — an instance-dict - // entry would be shadowed by the descriptor. - inst.slot_set(k, v.clone()); - } - } - } - // If a class anywhere between `cls` and `BaseException` - // (exclusive) defines its own `__init__`, run it so - // subclasses such as `BaseExceptionGroup` get to stitch - // `exceptions` onto the instance. We stop at - // `BaseException` because its default `__init__` only - // populates `args` — which the fast path already did. - if let Some(init) = lookup_exception_init(&cls) { - let bound = - Object::BoundMethod(Rc::new(BoundMethod::new(instance.clone(), init))); - let result = self.call( - &bound, - args, - kwargs, - &Rc::new(RefCell::new(DictData::default())), - )?; - if !matches!(result, Object::None) { - return Err(type_error(format!( - "__init__() should return None, not '{}'", - result.type_name() - ))); - } - } - return Ok(instance); - } } // Everything below that is a pure function of the class dict / @@ -55131,6 +55218,9 @@ struct LeanAct { shell: Option>, } +/// How many parked [`InlineAct`] slots `Interpreter::inline_pool` keeps. +const INLINE_POOL_CAP: usize = 64; + /// A lean activation the quiet loop runs *inline*: its frame lives in /// this box instead of on a nested native activation, and /// `Interpreter::quiet_run` switches to it at the caller's `CALL` and @@ -56328,8 +56418,41 @@ thread_local! { /// identity `enum`'s bootstrap (`found in (data_type_method, object_method)`) /// depends on. Only built-in types are keyed, so the entries live as long /// as the type singletons themselves. - static SLOT_WRAPPER_CACHE: std::cell::RefCell> = - std::cell::RefCell::new(std::collections::HashMap::new()); + /// `builtin_type_dunder` results (misses included) by built-in type + /// address, then name. + #[allow(clippy::type_complexity)] + static SLOT_WRAPPER_CACHE: std::cell::RefCell< + crate::fasthash::FxHashMap, Option>>, + > = std::cell::RefCell::new(crate::fasthash::FxHashMap::default()); +} + +/// Whether `obj.name` certainly raises `AttributeError` without running +/// any code: an ordinary instance under the default attribute protocol, +/// with no class attribute of that name and none in its `__dict__`. A +/// `getattr` default or `hasattr` then needs no exception at all. Dunder +/// names, which the lookup special-cases, never qualify. +fn attr_certainly_missing(obj: &Object, name: &str) -> bool { + use crate::types::Dunder; + let Object::Instance(inst) = obj else { + return false; + }; + if name.starts_with("__") || inst.native.get().is_some() || inst.c_body.get() != 0 { + return false; + } + let cls = inst.cls(); + if cls.native_kind.get() != 0 + || !cls.dunder(Dunder::GetAttribute).object_owner() + || cls.dunder(Dunder::GetAttr).present() + || cls.lookup(name).is_some() + { + return false; + } + match inst.dict.get() { + Some(d) => d + .try_borrow() + .is_ok_and(|d| !d.contains_key(&crate::object::StrKey(name))), + None => true, + } } /// Resolve a built-in slot-wrapper dunder reached via *type-level* attribute @@ -56355,22 +56478,31 @@ pub(crate) fn builtin_slot_wrapper(ty: &Rc, name: &str) -> Option Object::None, }); } - let mro: Vec> = ty.mro.borrow().iter().cloned().collect(); - for base in mro { + for base in ty.mro.borrow().iter() { if !base.flags.is_builtin { continue; } - let ptr = Rc::as_ptr(&base) as usize; - if let Some(o) = - SLOT_WRAPPER_CACHE.with(|c| c.borrow().get(&(ptr, name.to_owned())).cloned()) - { - return Some(o); - } - if let Some(o) = crate::builtins::builtin_type_dunder(&base.name, name) { - SLOT_WRAPPER_CACHE.with(|c| { - c.borrow_mut().insert((ptr, name.to_owned()), o.clone()); - }); - return Some(o); + let ptr = Rc::as_ptr(base) as usize; + let cached = SLOT_WRAPPER_CACHE.with(|c| { + c.borrow() + .get(&ptr) + .and_then(|names| names.get(name).cloned()) + }); + let dunder = match cached { + Some(hit) => hit, + None => { + let found = crate::builtins::builtin_type_dunder(&base.name, name); + SLOT_WRAPPER_CACHE.with(|c| { + c.borrow_mut() + .entry(ptr) + .or_default() + .insert(name.into(), found.clone()); + }); + found + } + }; + if dunder.is_some() { + return dunder; } // The base's own type dict — where `object.__reduce_ex__` / // `__reduce__` / `__getattribute__` sentinels live. Already diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index feb7446a..5b1df89c 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -4284,7 +4284,8 @@ pub struct PyFunction { /// CPython's `func_set_dict` *aliases* the assigned dict /// (`f.__dict__ = d; f.__dict__ is d` — test_funcattrs), so the /// whole payload must be swappable, not just its contents. - pub attrs: RefCell>>, + /// Allocated on first use: most functions never get a `__dict__`. + pub attrs: RefCell>>>, /// CPython function *getset/member slots* (`__name__`, /// `__qualname__`, `__doc__`, `__module__`, `__annotations__`, /// `__type_params__`, …). These live outside `__dict__`: they're @@ -4403,7 +4404,22 @@ impl PyFunction { /// The live `__dict__` payload (honours `f.__dict__ = d` swapping). pub fn attrs(&self) -> Rc> { - self.attrs.borrow().clone() + if let Some(d) = self.attrs.borrow().as_ref() { + return d.clone(); + } + let d = Rc::new(RefCell::new(DictData::default())); + *self.attrs.borrow_mut() = Some(d.clone()); + d + } + + /// A `__dict__` entry, without allocating an empty dictionary. + pub fn attr_get(&self, name: &str) -> Option { + self.attrs + .borrow() + .as_ref()? + .borrow() + .get(&StrKey(name)) + .cloned() } /// Read a slot value if one has been stored (explicitly assigned or diff --git a/crates/weavepy-vm/src/stdlib/abc_mod.rs b/crates/weavepy-vm/src/stdlib/abc_mod.rs index 62f61b07..5c38d77a 100644 --- a/crates/weavepy-vm/src/stdlib/abc_mod.rs +++ b/crates/weavepy-vm/src/stdlib/abc_mod.rs @@ -220,7 +220,22 @@ fn attr_or_none( /// CPython `_PyObject_IsAbstract`. fn is_abstract(interp: &mut Interpreter, obj: &Object) -> Result { - match attr_or_none(interp, obj, "__isabstractmethod__")? { + let flag = match obj { + // A plain function carries the flag only in its `__dict__`, and + // built-in data values never do: answer both without an + // attribute lookup that would build an `AttributeError`. + Object::Function(f) => f.attr_get("__isabstractmethod__"), + Object::None + | Object::Bool(_) + | Object::Int(_) + | Object::Long(_) + | Object::Float(_) + | Object::Str(_) + | Object::Bytes(_) + | Object::Tuple(_) => None, + _ => attr_or_none(interp, obj, "__isabstractmethod__")?, + }; + match flag { Some(flag) => interp.op_truth(&flag), None => Ok(false), } diff --git a/crates/weavepy-vm/src/stdlib/random_core.rs b/crates/weavepy-vm/src/stdlib/random_core.rs index 919de177..c270fd30 100644 --- a/crates/weavepy-vm/src/stdlib/random_core.rs +++ b/crates/weavepy-vm/src/stdlib/random_core.rs @@ -139,40 +139,91 @@ impl Mt { mt } + /// Regenerate the whole block of state words (`genrand_uint32`'s + /// refill). + fn regenerate(&mut self) { + for kk in 0..(N - M) { + let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); + self.key[kk] = self.key[kk + M] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + } + for kk in (N - M)..(N - 1) { + let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); + self.key[kk] = self.key[kk + M - N] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + } + let y = (self.key[N - 1] & UPPER_MASK) | (self.key[0] & LOWER_MASK); + self.key[N - 1] = self.key[M - 1] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + self.pos = 0; + } +} + +/// `genrand_uint32`'s output tempering. +#[inline] +fn temper(mut y: u32) -> u32 { + y ^= y >> 11; + y ^= (y << 7) & 0x9d2c_5680; + y ^= (y << 15) & 0xefc6_0000; + y ^ (y >> 18) +} + +/// The persisted state, borrowed in place: the 624 key words, then the +/// cursor, each little-endian (see [`STATE_LEN`]). +struct MtBytes<'a>(&'a mut [u8]); + +impl MtBytes<'_> { + #[inline] + fn word(&self, i: usize) -> u32 { + let mut w = [0u8; 4]; + w.copy_from_slice(&self.0[4 * i..4 * i + 4]); + u32::from_le_bytes(w) + } + + #[inline] + fn set_word(&mut self, i: usize, v: u32) { + self.0[4 * i..4 * i + 4].copy_from_slice(&v.to_le_bytes()); + } + + fn to_mt(&self) -> Mt { + let mut key = [0u32; N]; + for (i, k) in key.iter_mut().enumerate() { + *k = self.word(i); + } + Mt { + key, + pos: (self.word(N) as usize).min(N), + } + } + + fn write(&mut self, mt: &Mt) { + for (i, k) in mt.key.iter().enumerate() { + self.set_word(i, *k); + } + self.set_word(N, mt.pos as u32); + } + /// `genrand_uint32` — the raw 32-bit output stream. fn genrand_u32(&mut self) -> u32 { - if self.pos >= N { - // Regenerate the whole block. - for kk in 0..(N - M) { - let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); - self.key[kk] = self.key[kk + M] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - } - for kk in (N - M)..(N - 1) { - let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); - self.key[kk] = - self.key[kk + M - N] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - } - let y = (self.key[N - 1] & UPPER_MASK) | (self.key[0] & LOWER_MASK); - self.key[N - 1] = self.key[M - 1] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - self.pos = 0; + let mut pos = self.word(N) as usize; + if pos >= N { + let mut mt = self.to_mt(); + mt.regenerate(); + self.write(&mt); + pos = 0; } - let mut y = self.key[self.pos]; - self.pos += 1; - y ^= y >> 11; - y ^= (y << 7) & 0x9d2c_5680; - y ^= (y << 15) & 0xefc6_0000; - y ^ (y >> 18) + let y = self.word(pos); + self.set_word(N, pos as u32 + 1); + temper(y) } } // =================================================================== -// Instance-state plumbing. The 624-word state lives in a bytearray in -// the instance dict (so Python-level subclasses share it), the cursor -// in an int. +// Instance-state plumbing. The state lives in a bytearray in the +// instance dict (so Python-level subclasses share it): the 624 key +// words, then the cursor. The generator runs on it in place. // =================================================================== const STATE_KEY: &str = "_mt_state"; -const POS_KEY: &str = "_mt_pos"; +/// The state bytearray's length: the key words and the cursor. +const STATE_LEN: usize = (N + 1) * 4; fn self_instance(args: &[Object], what: &str) -> Result, RuntimeError> { match args.first() { @@ -181,54 +232,43 @@ fn self_instance(args: &[Object], what: &str) -> Result, RuntimeE } } -fn load_mt(inst: &Rc) -> Result { - let dict = inst.dict_cell().borrow(); - let bytes = match dict.get(&DictKey(Object::from_static(STATE_KEY))) { - Some(Object::ByteArray(b)) => b.clone(), - _ => { - drop(dict); - // Unseeded use (e.g. subclass skipping __init__): seed from - // system entropy, as CPython does at allocation time. - let mt = seed_from_entropy(); - store_mt(inst, &mt); - return Ok(mt); - } - }; - let pos = match dict.get(&DictKey(Object::from_static(POS_KEY))) { - Some(Object::Int(i)) => *i as usize, - _ => N, +/// The instance's state buffer, seeding it from system entropy when it +/// is missing (a subclass skipping `__init__`), as CPython does at +/// allocation time. +fn state_buffer(inst: &Rc) -> Rc>> { + let found = match inst + .dict_cell() + .borrow() + .get(&crate::object::StrKey(STATE_KEY)) + { + Some(Object::ByteArray(b)) if b.borrow().len() == STATE_LEN => Some(b.clone()), + _ => None, }; - let buf = bytes.borrow(); - let mut key = [0u32; N]; - for (i, chunk) in buf.as_chunks::<4>().0.iter().enumerate().take(N) { - key[i] = u32::from_le_bytes(*chunk); - } - Ok(Mt { key, pos }) + found.unwrap_or_else(|| store_mt(inst, &seed_from_entropy())) } -fn store_mt(inst: &Rc, mt: &Mt) { - let mut buf = Vec::with_capacity(N * 4); - for w in &mt.key { - buf.extend_from_slice(&w.to_le_bytes()); - } - let mut dict = inst.dict_cell().borrow_mut(); - dict.insert( +fn load_mt(inst: &Rc) -> Mt { + let buf = state_buffer(inst); + let mut bytes = buf.borrow_mut(); + MtBytes(&mut bytes).to_mt() +} + +fn store_mt(inst: &Rc, mt: &Mt) -> Rc>> { + let mut bytes = vec![0u8; STATE_LEN]; + MtBytes(&mut bytes).write(mt); + let buf = Rc::new(RefCell::new(bytes)); + inst.dict_cell().borrow_mut().insert( DictKey(Object::from_static(STATE_KEY)), - Object::ByteArray(Rc::new(RefCell::new(buf))), - ); - dict.insert( - DictKey(Object::from_static(POS_KEY)), - Object::Int(mt.pos as i64), + Object::ByteArray(buf.clone()), ); + buf } -/// Mutate-in-place fast path: run `f` against the deserialized state, -/// then persist the (changed) words back into the bytearray buffer. -fn with_mt(inst: &Rc, f: impl FnOnce(&mut Mt) -> R) -> Result { - let mut mt = load_mt(inst)?; - let r = f(&mut mt); - store_mt(inst, &mt); - Ok(r) +/// Run `f` against the instance's state, in place. +fn with_mt(inst: &Rc, f: impl FnOnce(&mut MtBytes<'_>) -> R) -> R { + let buf = state_buffer(inst); + let mut bytes = buf.borrow_mut(); + f(&mut MtBytes(&mut bytes)) } fn seed_from_entropy() -> Mt { @@ -317,7 +357,7 @@ fn random_random(args: &[Object]) -> Result { let a = mt.genrand_u32() >> 5; let b = mt.genrand_u32() >> 6; (f64::from(a) * 67_108_864.0 + f64::from(b)) * (1.0 / 9_007_199_254_740_992.0) - })?; + }); Ok(Object::Float(v)) } @@ -363,7 +403,7 @@ fn random_getrandbits(args: &[Object]) -> Result { return Ok(Object::Int(0)); } if k <= 32 { - let v = with_mt(&inst, |mt| mt.genrand_u32())? >> (32 - k as u32); + let v = with_mt(&inst, |mt| mt.genrand_u32()) >> (32 - k as u32); return Ok(Object::Int(i64::from(v))); } if (k - 1) / 32 + 1 > (isize::MAX as u64) / 4 { @@ -386,7 +426,7 @@ fn random_getrandbits(args: &[Object]) -> Result { remaining = remaining.saturating_sub(32); } out - })?; + }); let mut bytes = Vec::with_capacity(words * 4); for d in &digits { bytes.extend_from_slice(&d.to_le_bytes()); @@ -421,14 +461,14 @@ fn random_randbytes(args: &[Object]) -> Result { buf.extend_from_slice(&w[..take]); } buf - })?; + }); Ok(Object::new_bytes(out)) } /// `getstate()` → 625-tuple: the 624 state words plus the cursor. fn random_getstate(args: &[Object]) -> Result { let inst = self_instance(args, "getstate()")?; - let mt = load_mt(&inst)?; + let mt = load_mt(&inst); let mut items: Vec = mt.key.iter().map(|w| Object::Int(i64::from(*w))).collect(); items.push(Object::Int(mt.pos as i64)); Ok(Object::new_tuple(items)) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index ef1086df..73a0c481 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -327,6 +327,10 @@ struct Artifacts { /// generator — possibly to another thread — so buffer-layout /// identity needs a process-wide id. compile_id: u64, + /// Native-to-native entries of this compilation, and the interpreter + /// round-trips they made (see [`note_callee_exit`]). + callee_entries: Cell, + callee_roundtrips: Cell, } /// RFC 0073 WS4 — source of [`Artifacts::compile_id`]. @@ -422,6 +426,13 @@ pub(crate) const DEOPT_BUDGET: u32 = 64; /// Retire it to tier-1. pub(crate) const GENERIC_CALL_RETIRE_RATIO: u32 = 4; +/// [`GENERIC_CALL_RETIRE_RATIO`] for native-to-native entries. Such a +/// callee is usually loop-free, so native code saves it a few dozen +/// nanoseconds per activation while one interpreter call from it costs +/// several hundred more than the interpreter's inline call (measured on +/// deltablue's `execute` / `input` / `output` methods: 4x slower compiled). +pub(crate) const CALLEE_ROUNDTRIP_RETIRE_RATIO: u32 = 1; + /// Retire a compiled code object — and deopt the running activation — /// once one activation has made this many interpreter round-trips /// through the call helpers. Each such call pays activation-shell @@ -1228,6 +1239,8 @@ impl JitState { math: StdRc::new(math_tbl), compile_id: NEXT_COMPILE_ID .fetch_add(1, std::sync::atomic::Ordering::Relaxed), + callee_entries: Cell::new(0), + callee_roundtrips: Cell::new(0), }); // RFC 0067 WS1 — a fresh compile can flip a // `None` native-callee slot in *other* frames' @@ -1743,10 +1756,7 @@ fn probe_class_ctor_shape( let bt = crate::builtin_types::builtin_types(); // `type` subclasses (metaclasses) construct *classes* through the // three-argument form, never plain instances. - if cls.flags.is_builtin - || cls.is_subclass_of(&bt.type_) - || !Rc::ptr_eq(&cls.metaclass_or_type(), &bt.type_) - { + if cls.flags.is_builtin || cls.is_subclass_of(&bt.type_) || !cls.metaclass_is_type() { return None; } let plan = interp.instance_plan(cls); @@ -2858,6 +2868,15 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F if frame.code.jit_hint.is_not_jitable() || jit_off_for_process() { return; } + // A loop-free body gains nothing from native code until compiled + // callers exist to take its direct lanes, while compiling one during + // start-up or an import costs time and memory the program may never + // recover (`ABCMeta.register` while `_collections_abc` loads). Count + // again from zero; a body that stays hot compiles afterwards. + if phase != CompilationPhase::Normal { + frame.code.jit_hint.defer_lean_compile(); + return; + } JIT.with(|cell| { let mut st = cell.borrow_mut(); if !st.enabled { @@ -2865,10 +2884,9 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F } let key = Rc::as_ptr(&frame.code).cast::(); let threshold = st.threshold; - let importing = phase == CompilationPhase::Import; - let warm = if phase == CompilationPhase::Normal - && STARTUP_DONE.load(std::sync::atomic::Ordering::Relaxed) - { + // Embedders that never report start-up finished still compile, + // after sustained work. + let warm = if STARTUP_DONE.load(std::sync::atomic::Ordering::Relaxed) { threshold } else { threshold.saturating_mul(16) @@ -2888,20 +2906,8 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F code: frame.code.clone(), }); if matches!(entry.tier, Tier::Cold) { - if importing { - // The lean path has no ordinary frame-entry counter. Account - // for the interval just completed, then permit another one. - // Resetting keeps the equality checkpoint reachable after the - // import exits, including when pure-leaf calls can skip frames. - let interval = lean_warm_at(); - entry.counter = entry.counter.saturating_add(interval); - if entry.counter < interval.saturating_mul(16) { - frame.code.jit_hint.defer_lean_compile(); - return; - } - } // Preserve the earlier lean warm point relative to frame/loop - // hotness, including the escape hatch for unreported startup. + // hotness. entry.counter = entry.counter.max(warm); } let interp_ref: &super::Interpreter = interp; @@ -4388,6 +4394,7 @@ unsafe fn try_native_call( } } }; + note_callee_exit(nc, nctx); if !inline_bufs { put_u64(u64_buf); put_u32(u32_buf); @@ -4941,6 +4948,41 @@ fn finish_interp_call( } } +/// The generic-call backoff for native-to-native entries (the framed +/// entries' twin lives in [`note_native_exit`]): a compiled callee whose +/// activations average [`CALLEE_ROUNDTRIP_RETIRE_RATIO`] or more +/// interpreter calls is a thin native driver around them. Each such call pays pin +/// traffic, an activation shell and a generic call that the interpreter's +/// inline call path avoids, so the callee retires to tier-1. +#[inline] +fn note_callee_exit(nc: &NativeCallee, child: &CallCtx) { + let entries = nc.art.callee_entries.get().saturating_add(1); + nc.art.callee_entries.set(entries); + if child.dyn_py_calls == 0 { + return; + } + let trips = nc + .art + .callee_roundtrips + .get() + .saturating_add(child.dyn_py_calls); + nc.art.callee_roundtrips.set(trips); + if entries >= GENERIC_RETIRE_MIN_ENTRIES + && trips / entries >= CALLEE_ROUNDTRIP_RETIRE_RATIO + && !nc.code.jit_hint.is_not_jitable() + { + let key = Rc::as_ptr(&nc.code).cast::(); + JIT.with(|cell| { + let mut st = cell.borrow_mut(); + if let Some(ce) = st.cache.get_mut(&key) { + ce.tier = Tier::NotJitable; + } + st.stats.generic_retires += 1; + }); + nc.code.jit_hint.mark_not_jitable(); + } +} + /// Charge one expensive round-trip (an interpreter call, a generic /// attribute access, or a heavy native-to-native call) to the running /// activation. Returns true once the activation has spent @@ -5838,6 +5880,17 @@ fn dict_pin_and_key( /// natively found) reports `Err(())` so the caller deopts and the /// interpreter runs the comparison with full semantics. fn dict_probe_native(d: &Rc>, key: &Object) -> Result, ()> { + // A `str` or `int` key settles by native equality unless the table + // compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(key) { + if let Ok(m) = d.try_borrow() { + match m.get(&probe) { + Some(v) => return Ok(Some(v.clone())), + None if probe.miss_is_exact() => return Ok(None), + None => {} + } + } + } let (found, deferred) = crate::object::with_key_eq_deferred(|| { crate::object::key_cmp_scope(|| d.borrow().get(&DictKey(key.clone())).cloned()) }); @@ -5871,15 +5924,25 @@ unsafe extern "C" fn wpjit_dict_get( let jf = unsafe { &mut *frame }; #[allow(clippy::cast_ptr_alignment)] let ctx = unsafe { &mut *jf.ctx.cast::() }; - let Some((d, key)) = dict_pin_and_key(ctx, pin, key_bits, key_tag) else { + let Some(Pin::Obj(Object::Dict(d))) = ctx.pins.get(pin as usize) else { return 1; }; - let found = match dict_probe_native(&d, &key) { + let int_key; + let key: &Object = if key_tag == weavepy_jit::DICT_KEY_STR { + match ctx.pins.get(key_bits as usize) { + Some(Pin::Obj(o @ Object::Str(_))) => o, + _ => return 1, + } + } else { + int_key = Object::Int(key_bits); + &int_key + }; + let found = match dict_probe_native(d, key) { Ok(f) => f, Err(()) => return 1, }; let Some(v) = found else { - ctx.raised = Some(crate::error::key_error_object(key)); + ctx.raised = Some(crate::error::key_error_object(key.clone())); return 2; }; match (val_tag, &v) { @@ -10943,7 +11006,16 @@ fn park_plan(frame: &super::Frame, entry: &CompiledEntry, jf: &JitFrame) -> Opti /// a resume's sent value). Afterwards the frame is indistinguishable /// from an interpreted suspension. No-op without a parked box; never /// needs an interpreter (park refused any shape whose rebuild would). +#[inline] pub(crate) fn materialize_parked(frame: &mut super::Frame) { + if frame.parked_native.is_some() { + materialize_parked_native(frame); + } +} + +#[cold] +#[inline(never)] +fn materialize_parked_native(frame: &mut super::Frame) { let Some(mut act) = frame.parked_native.take() else { return; }; diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index 1ffe1e8e..52df186a 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -42,15 +42,19 @@ static NEXT_TYPE_VERSION: std::sync::atomic::AtomicU64 = std::sync::atomic::Atom pub enum Dunder { Eq, Hash, + GetAttr, + GetAttribute, } impl Dunder { - pub const COUNT: usize = 2; + pub const COUNT: usize = 4; pub const fn name(self) -> &'static str { match self { Self::Eq => "__eq__", Self::Hash => "__hash__", + Self::GetAttr => "__getattr__", + Self::GetAttribute => "__getattribute__", } } } @@ -1617,6 +1621,22 @@ impl TypeObject { crate::builtin_types::builtin_types().type_.clone() } + /// Whether [`Self::metaclass_or_type`] is `type` itself, without + /// cloning the metaclass handle. + pub fn metaclass_is_type(&self) -> bool { + let ext = self.c_ext_ptr.get(); + if ext != 0 { + if let Some(h) = METACLASS_DRIFT_HOOK.get() { + h(ext, self); + } + } + let ty = &crate::builtin_types::builtin_types().type_; + match self.metaclass.borrow().as_ref() { + Some(m) => Rc::ptr_eq(m, ty), + None => true, + } + } + /// `True` when `self` is a subclass of `other` (including itself). pub fn is_subclass_of(&self, other: &TypeObject) -> bool { let other_ptr = std::ptr::from_ref::(other); @@ -1994,6 +2014,26 @@ fn slot_name_eq(stored: &str, name: &str) -> bool { true } +/// The key for a newly populated slot `name`. The slots every raise +/// populates (`args`, `__traceback__`, the chaining links) share one +/// interned key each instead of allocating a string per exception. +fn slot_key(name: &str) -> DictKey { + const COMMON: [&str; 6] = [ + "args", + "__traceback__", + "__context__", + "__cause__", + "__suppress_context__", + "message", + ]; + static KEYS: std::sync::OnceLock<[Object; 6]> = std::sync::OnceLock::new(); + if let Some(i) = COMMON.iter().position(|c| *c == name) { + let keys = KEYS.get_or_init(|| COMMON.map(crate::stdlib::sys::intern_name)); + return DictKey(keys[i].clone()); + } + DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) +} + impl SlotStorage { /// Storage holding exactly `entries` (distinct `str` keys, in slot /// order), built in one step. @@ -2226,9 +2266,7 @@ impl SlotStorage { } pub fn insert(&mut self, name: &str, value: Object) -> Option { - self.insert_with_key(name, value, || { - DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) - }) + self.insert_with_key(name, value, || slot_key(name)) } /// Reuse an existing name allocation when populating a new slot. @@ -2380,9 +2418,7 @@ impl SlotStorage { } pub fn insert(&mut self, name: &str, value: Object) -> Option { - self.insert_with_key(name, value, || { - DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) - }) + self.insert_with_key(name, value, || slot_key(name)) } /// Share the name allocation only for a newly populated slot. diff --git a/crates/weavepy-vm/src/vm_singletons.rs b/crates/weavepy-vm/src/vm_singletons.rs index ef5f8247..f78b67c5 100644 --- a/crates/weavepy-vm/src/vm_singletons.rs +++ b/crates/weavepy-vm/src/vm_singletons.rs @@ -295,7 +295,7 @@ pub fn publish_interpreter_seed(interp: &crate::Interpreter) { let mut slot = seed_slot().lock(); *slot = Some(interp.fork_for_thread()); drop(slot); - *seed_types_slot().lock() = Some(crate::builtin_types::builtin_types()); + *seed_types_slot().lock() = Some(crate::builtin_types::builtin_types_rc()); } /// Run `f` (typically a `Interpreter::new()` for a *sub*-interpreter) diff --git a/crates/weavepy/src/lib.rs b/crates/weavepy/src/lib.rs index 91be56d7..d2835c35 100644 --- a/crates/weavepy/src/lib.rs +++ b/crates/weavepy/src/lib.rs @@ -87,7 +87,9 @@ impl Error { let _ = writeln!( s, " File \"{}\", line {}, in {}", - entry.filename, entry.lineno, entry.funcname + entry.filename(), + entry.lineno, + entry.funcname() ); } } From f7e5a5bf2b42c98894a2d59dce92a50fcde32156 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 21:38:11 -0700 Subject: [PATCH 05/65] perf: evaluate pure leaves and field updates directly at JIT call sites Compiled code called pure-leaf functions and methods (getters, predicates, small arithmetic) through a full native activation and then counted each one as a reason to retire the calling loop to the interpreter. Call sites now evaluate them with the interpreter's frameless pure-leaf evaluator right after the site's guards, so such loops stay native; the verdict is mirrored on the code's `JitHint` so other callees pay one relaxed load. A method with a callback-free scalar field update plan (`self.n += k; return self.n`) likewise runs the update directly after the method guard instead of through the native-call preflight. Instructions retired, JIT on: a `c.m()` getter loop -32%, a two-argument function call loop -25%, richards -20%, attr_access -4%. --- crates/weavepy-compiler/src/lib.rs | 22 +++ crates/weavepy-vm/src/lib.rs | 1 + crates/weavepy-vm/src/tier2.rs | 240 +++++++++++++++++++++++++---- 3 files changed, 232 insertions(+), 31 deletions(-) diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index b5f3cb84..aa426d2f 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -216,6 +216,10 @@ pub struct JitHint { /// The tier-2 state has no further interest in this code's back /// edges (compiled, OSR budget spent). backedge_quiet: std::sync::atomic::AtomicBool, + /// The VM's pure-leaf verdict (0 not yet decided, 1 no, 2 yes), + /// mirrored here so native call sites read it without the VM + /// extension lookup. + pure_leaf: std::sync::atomic::AtomicU8, } impl JitHint { @@ -269,6 +273,24 @@ impl JitHint { .store(true, std::sync::atomic::Ordering::Relaxed); } + /// The recorded pure-leaf verdict, if the VM has decided one. + #[must_use] + pub fn pure_leaf(&self) -> Option { + match self.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) { + 1 => Some(false), + 2 => Some(true), + _ => None, + } + } + + /// Record the VM's pure-leaf verdict (see [`Self::pure_leaf`]). + pub fn set_pure_leaf(&self, yes: bool) { + self.pure_leaf.store( + if yes { 2 } else { 1 }, + std::sync::atomic::Ordering::Relaxed, + ); + } + #[must_use] pub fn is_compiled(&self) -> bool { self.compiled.load(std::sync::atomic::Ordering::Relaxed) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index dd8fccd6..6453a2ef 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -61163,6 +61163,7 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { 1 }; ext.pure_leaf.store(shape, Relaxed); + code.jit_hint.set_pure_leaf(ok); ok } diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 73a0c481..bb6b572c 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -4756,6 +4756,31 @@ unsafe extern "C" fn wpjit_call_py( // while the helper runs; this is the only live path to it. let interp = unsafe { &mut *ctx.interp }; + // A pure-leaf callee evaluates frameless (see `try_pure_leaf_call`). + let maybe_pure = ctx + .callees + .get(token as usize) + .is_some_and(|(_, code)| code.jit_hint.pure_leaf() != Some(false)); + let callees = if maybe_pure { + Some(StdRc::clone(&ctx.callees)) + } else { + None + }; + if let Some((Object::Function(f), code)) = callees.as_ref().and_then(|c| c.get(token as usize)) + { + // SAFETY: GIL-serialized raw read of the function's code cell; + // only compared. + if std::ptr::eq(unsafe { Rc::as_ptr(&*f.code.as_ptr()) }, Rc::as_ptr(code)) { + // SAFETY: `argc` marshaled entries are live (the function + // contract). + if let Some(status) = + unsafe { try_pure_leaf_call(jf, ctx, interp, f, code, None, argc, expect_tag) } + { + return status; + } + } + } + // RFC 0067 WS1 — the native-to-native fast path: a compiled, // shape-eligible callee is entered directly with the marshaled // scalars, skipping the interpreter frame entirely. (The table @@ -4910,37 +4935,7 @@ fn finish_interp_call( &ctx.math, ); if still_valid { - match SlotTag::from_raw(expect_tag) { - // The procedure lane: nothing to write back, the - // compiled code pushes no result. - SlotTag::None => { - if matches!(v, Object::None) { - return CallStatus::Ok as i64; - } - } - SlotTag::Int | SlotTag::Float | SlotTag::Bool => { - let expect = match SlotTag::from_raw(expect_tag) { - SlotTag::Int => JitType::Int, - SlotTag::Float => JitType::Float, - _ => JitType::Bool, - }; - if let Some(bits) = pack(&v, expect) { - jf.ret_bits = bits; - jf.ret_tag = expect_tag; - return CallStatus::Ok as i64; - } - } - // RFC 0071 WS1 — an object-lane result pins into - // this activation's table. - SlotTag::ObjPin => { - if let Some(bits) = obj_ret_bits(&v, &mut ctx.pins) { - jf.ret_bits = bits; - jf.ret_tag = expect_tag; - return CallStatus::Ok as i64; - } - } - SlotTag::Boxed | SlotTag::ListPin => {} - } + return deliver_call_result(jf, ctx, v, expect_tag); } ctx.parked = Some(v); CallStatus::Boxed as i64 @@ -4948,6 +4943,140 @@ fn finish_interp_call( } } +/// Hand a completed call's result `v` to the compiled caller in its +/// `expect_tag` lane, or park it (`Boxed`: the caller deopts after the +/// call) when the lane cannot carry it. +fn deliver_call_result(jf: &mut JitFrame, ctx: &mut CallCtx, v: Object, expect_tag: u32) -> i64 { + match SlotTag::from_raw(expect_tag) { + // The procedure lane: nothing to write back, the compiled code + // pushes no result. + SlotTag::None => { + if matches!(v, Object::None) { + return CallStatus::Ok as i64; + } + } + SlotTag::Int | SlotTag::Float | SlotTag::Bool => { + let expect = match SlotTag::from_raw(expect_tag) { + SlotTag::Int => JitType::Int, + SlotTag::Float => JitType::Float, + _ => JitType::Bool, + }; + if let Some(bits) = pack(&v, expect) { + jf.ret_bits = bits; + jf.ret_tag = expect_tag; + return CallStatus::Ok as i64; + } + } + // RFC 0071 WS1 — an object-lane result pins into this + // activation's table. + SlotTag::ObjPin => { + if let Some(bits) = obj_ret_bits(&v, &mut ctx.pins) { + jf.ret_bits = bits; + jf.ret_tag = expect_tag; + return CallStatus::Ok as i64; + } + } + SlotTag::Boxed | SlotTag::ListPin => {} + } + ctx.parked = Some(v); + CallStatus::Boxed as i64 +} + +/// Evaluate a pure-leaf callee (see `code_is_pure_leaf`) frameless, as +/// the interpreter's core loop does: `recv` (a method's receiver) then +/// the `argc` marshaled arguments bind its parameters exactly. A native +/// activation would cost several times the body; `None` (nothing ran) +/// leaves the call to the ordinary paths. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]: `argc` marshal entries are live. +#[inline(always)] +#[allow(clippy::too_many_arguments)] +unsafe fn try_pure_leaf_call( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &super::Interpreter, + func: &crate::object::PyFunction, + code: &CodeObject, + recv: Option<*const Object>, + argc: u32, + expect_tag: u32, +) -> Option { + // The common native callee is no pure leaf: one relaxed load decides. + if code.jit_hint.pure_leaf() == Some(false) + || code.arg_count != argc + u32::from(recv.is_some()) + { + return None; + } + // SAFETY: the caller's contract. + unsafe { pure_leaf_call(jf, ctx, interp, func, code, recv, argc, expect_tag) } +} + +/// [`try_pure_leaf_call`]'s evaluation, out of line (its argument +/// buffers would otherwise widen every native call helper's frame). +/// +/// # Safety +/// +/// As [`try_pure_leaf_call`]. +#[inline(never)] +#[allow(clippy::too_many_arguments)] +unsafe fn pure_leaf_call( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &super::Interpreter, + func: &crate::object::PyFunction, + code: &CodeObject, + recv: Option<*const Object>, + argc: u32, + expect_tag: u32, +) -> Option { + const MAX: usize = 8; + let offset = usize::from(recv.is_some()); + let n = argc as usize + offset; + if n > MAX + || code.arg_count as usize != n + || !code + .jit_hint + .pure_leaf() + .unwrap_or_else(|| crate::code_is_pure_leaf_pub(code)) + || crate::hot_gates::load() != 0 + || crate::trace::any_observers_active() + { + return None; + } + // Only the marshaled arguments need owned values (a scalar lane + // becomes its `Object`, a pin names the pinned one); the receiver is + // borrowed where it lives. + let mut owned = [const { std::mem::MaybeUninit::::uninit() }; MAX]; + /// Drops the first `.1` values at `.0` (the written arguments). + struct Owned(*mut Object, usize); + impl Drop for Owned { + fn drop(&mut self) { + for j in 0..self.1 { + // SAFETY: the first `self.1` values were written below. + unsafe { std::ptr::drop_in_place(self.0.add(j)) }; + } + } + } + let mut written = Owned(owned.as_mut_ptr().cast::(), 0); + let mut ptrs: [*const Object; MAX] = [std::ptr::null(); MAX]; + if let Some(r) = recv { + ptrs[0] = r; + } + for j in 0..argc as usize { + // SAFETY: the caller's contract — `argc` marshaled entries. + let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; + // SAFETY: `j < argc <= MAX`; the slot is uninitialized until now. + unsafe { written.0.add(j).write(unpack_pins(bits, tag, &ctx.pins)) }; + written.1 = j + 1; + // SAFETY: as above. + ptrs[offset + j] = unsafe { written.0.add(j) }; + } + let v = interp.pure_leaf_eval::(code, func, &ptrs[..n])?; + Some(deliver_call_result(jf, ctx, v, expect_tag)) +} + /// The generic-call backoff for native-to-native entries (the framed /// entries' twin lives in [`note_native_exit`]): a compiled callee whose /// activations average [`CALLEE_ROUNDTRIP_RETIRE_RATIO`] or more @@ -5178,6 +5307,55 @@ unsafe extern "C" fn wpjit_call_method( return finish_interp_call(jf, ctx, interp, res, expect_tag); } + // A pure-leaf method (a getter, a predicate) evaluates frameless. + // SAFETY: `argc` marshaled entries are live (the function contract), + // and `recv` outlives the evaluation. + if let Some(status) = unsafe { + try_pure_leaf_call( + jf, + ctx, + interp, + &entry.func, + &entry.code, + Some(&raw const recv), + argc, + expect_tag, + ) + } { + return status; + } + + // A callback-free field update (`self.n += k; return self.n`) bound + // exactly: the guards above and the update's own checks are all the + // native activation would validate for it. + if let Some(nc) = ctx + .method_native + .as_deref() + .and_then(|t| t.get(token as usize)) + .and_then(Option::as_ref) + { + if nc.scalar_update.is_some() + && entry.code.arg_count == argc + 1 + && Rc::ptr_eq(&nc.func, &entry.func) + && Rc::ptr_eq(&nc.code, &entry.code) + { + // SAFETY: the method guard above pinned the binding; the update + // checks its receiver, argument lane, and observers itself. + if let Some(value) = + unsafe { native_scalar_field_update(jf, ctx, nc, &recv, argc as usize) } + { + if expect_tag == SlotTag::Int as u32 { + jf.ret_bits = value as u64; + jf.ret_tag = SlotTag::Int as u32; + return CallStatus::Ok as i64; + } + // The store is complete: never repeat it. + ctx.parked = Some(Object::Int(value)); + return CallStatus::Boxed as i64; + } + } + } + // RFC 0069 WS1 — the native fast path: the guarded method's own // body is compiled and shape-eligible, so enter it directly with // the receiver seeded as its pin 0. The table is parallel to From 305f5873fad1176b3902ca8ae4bc1e3b62f436b4 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:54:57 -0700 Subject: [PATCH 06/65] perf: polymorphic call caches, keyword calls, raises, and dict probes - Keep up to three earlier resolutions in method slots and call slots, so call sites over several receiver classes stop re-resolving on every class change (a four-class `o.f()` site: -48% instructions), and widen the instance-attribute polymorphic cache to four classes. - Switch to `CALL_KW` callees inside the core loop, and cache each keyword site's validated binding instead of re-checking the names on every call; unpack bound-method callees in place like CPython's `CALL_BOUND_METHOD_EXACT_ARGS`. - Serve `dict.get`, `in`, and the generic dict lookup with the native `LeafProbe` for `str` and `int` keys, and format `str % scalars` in the leaf loop (dict_ops: -25% instructions, now at parity). - Raises: memoize whether an exception class has its own `__init__`, recognize an inert handled exception without the reap walks, and re-probe whether the frame object is observed right after `POP_EXCEPT` so the frame returns to the fast loop (raise loop: -22%). - Charge JIT-driven generator resumes like interpreter calls so such loops retire to the interpreter's cheaper inline resume (-21%), mirror the pure-leaf verdict on `JitHint`, and fill fresh locals in line. --- crates/weavepy-vm/src/builtins.rs | 11 + crates/weavepy-vm/src/lib.rs | 488 +++++++++++++++++++++++++++--- crates/weavepy-vm/src/tier2.rs | 10 + 3 files changed, 474 insertions(+), 35 deletions(-) diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index 2c9967bb..58dac7d8 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -12949,6 +12949,17 @@ pub(crate) fn dict_lookup( d: &Rc>, key: &Object, ) -> Result, RuntimeError> { + // A `str` or `int` key settles by native equality unless the table + // compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(key) { + if let Ok(m) = d.try_borrow() { + match m.get(&probe) { + Some(v) => return Ok(Some(v.clone())), + None if probe.miss_is_exact() => return Ok(None), + None => {} + } + } + } if crate::object::dict_key_is_reentrant(key) { return crate::object::dict_reentrant_get(d, key); } diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 6453a2ef..deb4272f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -940,6 +940,11 @@ pub struct Interpreter { /// exact per-instruction event stream. Refreshed from the dispatch /// loop's [`crate::trace::ObserverSnapshot`] every iteration. fuse_off: bool, + /// A handled exception was just released (`POP_EXCEPT`): the running + /// frame's object may have lost its last outside holder (the + /// traceback), so the dispatch loop re-probes it on the next + /// instruction instead of waiting for its stride. + recheck_frame_observed: bool, /// `WEAVEPY_NO_QUIET`: pin every dispatch-loop iteration to the full /// prologue (RFC 0065 bisection aid). Read once at construction so a /// frame entry does not pay a `OnceLock` probe for it (RFC 0077 WS3). @@ -1182,6 +1187,7 @@ impl Default for Interpreter { sum_fold_acc: None, sum_fold: None, fuse_off: false, + recheck_frame_observed: false, quiet_off: crate::hot_gates::env_flags::no_quiet(), burst_on: !crate::hot_gates::env_flags::no_burst() && !crate::hot_gates::env_flags::no_quiet() @@ -1307,6 +1313,7 @@ impl Interpreter { sum_fold_acc: None, sum_fold: None, fuse_off: false, + recheck_frame_observed: false, quiet_off: crate::hot_gates::env_flags::no_quiet(), burst_on: !crate::hot_gates::env_flags::no_burst() && !crate::hot_gates::env_flags::no_quiet() @@ -3879,6 +3886,45 @@ impl Interpreter { } } + /// The common handled exception, recognized without the reap walks: a + /// class with no finalizer, no instance dict, and slots holding only + /// atomic values, a tuple of them (`args`), or a traceback whose frames + /// are all still executing. Neither `exc_has_finalizable` nor + /// `anchors_tracked_child` can find anything in it. + fn exc_plainly_inert(obj: &Object) -> bool { + let Object::Instance(inst) = obj else { + return false; + }; + if inst.cls().instances_need_finalize() + || inst + .dict + .get() + .is_some_and(|d| d.try_borrow().map_or(true, |d| !d.is_empty())) + { + return false; + } + let Ok(slots) = inst.slots.try_borrow() else { + return false; + }; + // (Bound, not returned directly: the iterator borrows `slots`.) + #[allow(clippy::let_and_return)] + let inert = slots.iter().all(|(_, v)| match v { + Object::Tuple(t) => t.iter().all(Object::is_gc_atomic), + Object::Traceback(tb) => { + let mut cur = Some(tb.clone()); + while let Some(node) = cur { + if node.frame.on_stack.get() == 0 { + return false; + } + cur = node.next.borrow().clone(); + } + true + } + other => other.is_gc_atomic(), + }); + inert + } + /// Cheap, allocation-free pre-check for the `POP_EXCEPT` reap: does the /// acyclic subgraph rooted at `obj` contain a finalizable object /// (`__del__` / an unfinished generator) reachable through value @@ -6039,7 +6085,8 @@ impl Interpreter { // ran. A few extra instructions on the full path after the // holder goes away cost far less. let rederive = frame_blocks_quiet - && self.gil_countdown.trailing_zeros() >= 4 + && (std::mem::take(&mut self.recheck_frame_observed) + || self.gil_countdown.trailing_zeros() >= 4) && !Self::frame_object_observed(&shell, py_frame_slot.as_ref(), frame); if lgen != loop_snap_gen || rederive { loop_snap_gen = lgen; @@ -10438,7 +10485,7 @@ impl Interpreter { // missing-tail fill. v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); } - v.resize(nlocals, Object::Unbound); + fill_unbound(v, nlocals); if !has_self { frame.stack.pop(); // the NULL self slot } @@ -10593,6 +10640,42 @@ impl Interpreter { Some((code, covered)) } + /// [`Self::kw_names_bind_check`] for the `CALL_KW` at `pc` of + /// `caller`, remembered in the site's call slot: the names tuple is the + /// site's constant, so a function and code already verified there + /// (under the site's `func_id`) bind the same way again, unless their + /// defaults may have been replaced since. + #[allow(clippy::too_many_arguments)] + fn kw_names_bind_cached( + caller: &CodeObject, + pc: usize, + f: &Rc, + func_id: u64, + perm: u32, + name_items: &[Object], + eff_argc: usize, + ) -> Option<(Rc, u32)> { + // SAFETY: GIL-serialized raw read of the function's code cell; + // only compared, then cloned. + let live: &Rc = unsafe { &*f.code.as_ptr() }; + let slot = code_call_slot(caller, pc); + if specialize::rc_id(f) == func_id && !f.defaults_maybe_overridden() { + if let Some((covered, _)) = slot.and_then(|s| s.hit(Rc::as_ptr(f), Rc::as_ptr(live))) { + return Some((live.clone(), covered)); + } + } + let (code, covered) = Self::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; + if let Some(slot) = slot { + slot.set(CallShape { + func: Rc::downgrade(f), + code: Rc::downgrade(&code), + missing: covered, + has_self: false, + }); + } + Some((code, covered)) + } + /// Move a verified `CallPyKwNames` call's operands off `stack` into /// `locals` (cleared and sized here): keyword values to their /// permuted slots, the positional run (with a real self as slot 0) @@ -10731,7 +10814,8 @@ impl Interpreter { let Object::Function(f) = &frame.stack[self_at - 1] else { return None; }; - let (code, covered) = Self::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; + let (code, covered) = + Self::kw_names_bind_cached(&frame.code, pc, f, func_id, perm, name_items, eff_argc)?; if !Self::lean_code_ok(&code) { return None; } @@ -12726,7 +12810,19 @@ impl Interpreter { unsafe { std::slice::from_raw_parts(base.add(start), len - start) }; let Some(r) = self.core_pure_kw_call(code, pc, ops, argc, sw.depth_cell) else { - break None; + // A plain Python callee switches in place, as + // `CALL`'s does (see `core_call_kw`). + if !matches!(&ops[0], Object::Function(_)) { + break None; + } + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_kw(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); }; // SAFETY: every operand was checked to leave by a plain // decrement; the result takes the callee's slot. @@ -13467,6 +13563,9 @@ impl Interpreter { if !self.inline_calls_ok() { return false; } + // SAFETY: see `CoreSwitch`: the running activation is synced and + // unborrowed here. + Self::unpack_bound_callee(unsafe { &mut *sw.cur }, pc); let mut tmp = None; // SAFETY: see `CoreSwitch`: the running activation is the // innermost; its handles are live and unborrowed here. @@ -13508,6 +13607,71 @@ impl Interpreter { true } + /// A bound method over a plain function with an empty self slot, + /// called at `pc` of `frame`, becomes that function with the receiver + /// as self (CPython's `CALL_BOUND_METHOD_EXACT_ARGS`), so the cached + /// and pure-leaf call paths apply to it. + #[inline] + fn unpack_bound_callee(frame: &mut Frame, pc: usize) { + let Some(argc) = frame.code.instructions.get(pc).map(|i| i.arg as usize) else { + return; + }; + let n = frame.stack.len(); + let Some(self_slot) = n.checked_sub(argc + 1) else { + return; + }; + let Some(callee_slot) = self_slot.checked_sub(1) else { + return; + }; + let (f, receiver) = match &frame.stack[callee_slot] { + Object::BoundMethod(bm) + if !bm.redispatch_descriptor + && matches!(frame.stack[self_slot], Object::Unbound) => + { + match &bm.function { + Object::Function(f) => (f.clone(), bm.receiver.clone()), + _ => return, + } + } + _ => return, + }; + let bm = std::mem::replace(&mut frame.stack[callee_slot], Object::Function(f)); + frame.stack[self_slot] = receiver; + // The bound method was a call temporary (or is still held + // elsewhere): grade its release like any dropped operand. + if gc_trace::note_dropped_marks(&bm) { + gc_trace::mark_maybe_dead(); + } + drop(bm); + } + + /// [`Self::core_call`] for the `CALL_KW` at `pc`: the quiet loop's + /// inline keyword call ([`Self::try_inline_call_kw`]), pushed and made + /// the running activation. `false` touches nothing. + #[inline(never)] + fn core_call_kw(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + self.try_inline_call_kw(&mut *frame, &mut *shell.cast::>(), pc) + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + /// The core loop's `CALL` of a plain class at `pc` of `sw`'s (synced) /// running activation: `try_lean_call`'s construction shape, with the /// `__init__` activation run inline and switched to here (the caller @@ -13612,9 +13776,7 @@ impl Interpreter { locals.reserve(nlocals.max(argc + 1)); locals.push(inst.clone()); locals.extend(frame.stack.drain(self_slot + 1..)); - if locals.len() < nlocals { - locals.resize(nlocals, Object::Unbound); - } + fill_unbound(locals, nlocals); // The NULL self slot and the class. frame.stack.truncate(callee_slot); drop(ty); @@ -14120,9 +14282,7 @@ impl Interpreter { }; locals.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); } - if locals.len() < nlocals { - locals.resize(nlocals, Object::Unbound); - } + fill_unbound(locals, nlocals); frame.pc = pc as u32 + 1; // The cells handle is `f`'s (now `callable`'s) or the shared empty // vector: moving the callable leaves it where it was. @@ -14490,7 +14650,7 @@ impl Interpreter { if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) || !Self::lean_code_ok(code_rc) { return None; } - let (_, covered) = Self::kw_names_bind_check(f, func_id, perm, names, eff_argc)?; + let (_, covered) = Self::kw_names_bind_cached(code, pc, f, func_id, perm, names, eff_argc)?; // The ordinary call's `RecursionError` check. // SAFETY: this thread's own depth cell. if unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() { @@ -16272,9 +16432,23 @@ impl Interpreter { Object::Str(SharedStr::repeat(a, times)) } } + // `"k%d" % i`: formatting scalars and strings runs no + // Python code; any error is the full handler's to + // raise. + (Object::Str(a), args) + if kind == BinOpKind::Mod + && percent_leaf_args(args) + && !percent_args_need_bridge(args) => + { + match percent_format(a, args) { + Ok(s) => Object::from_str(s), + Err(_) => break, + } + } _ => break, }; - // Operands are scalars or strings: nothing to grade. + // Operands are scalars or strings (or a tuple of them): + // nothing to grade. drop_operand(stack.pop().expect("length checked above")); drop_operand(std::mem::replace(&mut stack[n - 2], r)); last = pc; @@ -17541,6 +17715,18 @@ impl Interpreter { Object::Str(_) | Object::Int(_) | Object::Bool(_) | Object::None ) => { + // A `str` or `int` key settles natively (see `LeafProbe`). + if let (Object::Dict(d), Some(probe)) = + (container, crate::object::LeafProbe::new(item)) + { + let m = d.try_borrow().ok()?; + if m.get(&probe).is_some() { + return Some(true); + } + if probe.miss_is_exact() { + return Some(false); + } + } let key = DictKey(item.clone()); let (found, deferred) = crate::object::with_key_eq_deferred(|| { crate::object::key_cmp_scope(|| match container { @@ -18131,6 +18317,18 @@ impl Interpreter { return None; } let O::Dict(d) = &args[0] else { return None }; + // A `str` or `int` key settles by native equality unless the + // table compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(&args[1]) { + let m = d.try_borrow().ok()?; + match m.get(&probe) { + Some(v) => return Some(Ok(v.clone())), + None if probe.miss_is_exact() => { + return Some(Ok(args.get(2).cloned().unwrap_or(O::None))); + } + None => {} + } + } // The probe may need a Python comparison against a stored // key: deferred, and settled by the full path. let (found, deferred) = crate::object::with_key_eq_deferred(|| { @@ -22326,6 +22524,7 @@ impl Interpreter { // timing rather than waiting for the next cyclic collection // (RFC 0040 deterministic-finalization arc: `test_io` / // `test_subprocess` destructor-timing cases). + self.recheck_frame_observed = true; if let Some((_, pe)) = popped { // Deconstruct the handler's `PyException` *first*: its // `context`/`cause` boxes and traceback entries hold @@ -22404,9 +22603,10 @@ impl Interpreter { // so an `except E as e: saved = e` pays nothing. if weakly_observed || gc_trace::is_tracked(id) - || Self::exc_has_finalizable(&pe_instance, 6) - || (Self::is_refcount_dead(&pe_instance, 1) - && Self::anchors_tracked_child(&pe_instance, 5)) + || (!Self::exc_plainly_inert(&pe_instance) + && (Self::exc_has_finalizable(&pe_instance, 6) + || (Self::is_refcount_dead(&pe_instance, 1) + && Self::anchors_tracked_child(&pe_instance, 5)))) { self.reap_dead_subgraph(pe_instance); } @@ -56071,6 +56271,27 @@ fn is_object_new(b: &Rc) -> bool { Rc::as_ptr(b) as usize == want } +/// Extend `v` to `n` slots with `Unbound` (the fresh locals of an +/// activation), in line: a call's handful of slots isn't worth +/// `Vec::resize`'s out-of-line loop. +#[inline(always)] +fn fill_unbound(v: &mut Vec, n: usize) { + let len = v.len(); + if len >= n { + return; + } + v.reserve(n - len); + // SAFETY: capacity reserved above; `Unbound` owns nothing, and every + // slot below `n` is written before the length covers it. + unsafe { + let p = v.as_mut_ptr(); + for k in len..n { + p.add(k).write(Object::Unbound); + } + v.set_len(n); + } +} + /// Release an operand the burst is done with: an unboxed scalar has no /// heap and `Object` has no `Drop` of its own, so it needs no drop glue. #[inline(always)] @@ -56673,6 +56894,19 @@ fn dunder_key(which: usize) -> &'static Object { /// class in that prefix carries its own `__init__`, return it. /// Otherwise the caller can stick with the cheap `args`-only setup. fn lookup_exception_init(cls: &Rc) -> Option { + if exc_family_flags(cls) & EXC_CUSTOM_INIT == 0 { + return None; + } + exception_init_uncached(cls) +} + +/// [`exc_family_flags`]'s pseudo-family: a class between `cls` and +/// `BaseException` defines its own `__init__` (see +/// [`lookup_exception_init`]). +const EXC_CUSTOM_INIT: u16 = 1 << 9; + +/// [`lookup_exception_init`]'s MRO walk. +fn exception_init_uncached(cls: &TypeObject) -> Option { let mro = cls.mro.borrow(); for ty in mro.iter() { if ty.name == "BaseException" || ty.name == "object" { @@ -57851,6 +58085,21 @@ pub(crate) enum PercentMode { Bytes, } +/// A `%`-format argument (or argument tuple) of scalars and exact +/// strings: its conversions run no Python code. +fn percent_leaf_args(args: &Object) -> bool { + let leaf = |o: &Object| { + matches!( + o, + Object::Int(_) | Object::Float(_) | Object::Str(_) | Object::Bool(_) | Object::None + ) + }; + match args { + Object::Tuple(items) => items.iter().all(leaf), + other => leaf(other), + } +} + pub(crate) fn percent_format(template: &str, value: &Object) -> Result { let mut noop = |_: &Object, _: char| Ok(None); percent_format_with(template, value, PercentMode::Str, &mut noop) @@ -60483,7 +60732,22 @@ struct CodeConstObjects { /// that can change under a fixed code object — the defaults, the /// closure's cells, the JIT's claim on the code — is re-validated per /// call. -struct CallSlot(std::cell::UnsafeCell>); +/// A `CALL` site's inline-call shape, plus up to [`POLY_CALLS`] earlier +/// ones for a site that calls several functions (a method call over +/// receivers of several classes). +struct CallSlot( + std::cell::UnsafeCell>, + std::cell::UnsafeCell>>, +); + +/// Earlier shapes a polymorphic [`CallSlot`] keeps. +const POLY_CALLS: usize = 3; + +/// A [`CallSlot`]'s earlier shapes, replaced round-robin. +struct PolyCalls { + shapes: [Option; POLY_CALLS], + next: usize, +} struct CallShape { func: crate::sync::Weak, @@ -60500,9 +60764,24 @@ struct CallShape { unsafe impl Send for CallSlot {} unsafe impl Sync for CallSlot {} +impl CallShape { + #[inline] + fn matches( + &self, + func: *const crate::object::PyFunction, + code: *const CodeObject, + ) -> Option<(u32, bool)> { + (std::ptr::eq(self.func.as_ptr(), func) && std::ptr::eq(self.code.as_ptr(), code)) + .then_some((self.missing, self.has_self)) + } +} + impl CallSlot { const fn empty() -> Self { - Self(std::cell::UnsafeCell::new(None)) + Self( + std::cell::UnsafeCell::new(None), + std::cell::UnsafeCell::new(None), + ) } /// The shape recorded for `func` running `code` (by identity). @@ -60514,16 +60793,53 @@ impl CallSlot { ) -> Option<(u32, bool)> { // SAFETY: see the type docs. let shape = unsafe { &*self.0.get() }.as_ref()?; - (std::ptr::eq(shape.func.as_ptr(), func) && std::ptr::eq(shape.code.as_ptr(), code)) - .then_some((shape.missing, shape.has_self)) + shape + .matches(func, code) + .or_else(|| self.poly_hit(func, code)) + } + + /// [`Self::hit`] among the earlier shapes. + #[cold] + #[inline(never)] + fn poly_hit( + &self, + func: *const crate::object::PyFunction, + code: *const CodeObject, + ) -> Option<(u32, bool)> { + // SAFETY: see the type docs. + let poly = unsafe { &*self.1.get() }.as_deref()?; + poly.shapes + .iter() + .flatten() + .find_map(|shape| shape.matches(func, code)) } #[inline] fn set(&self, shape: CallShape) { + let (func, code) = (shape.func.as_ptr(), shape.code.as_ptr()); // SAFETY: see the type docs (the old handles drop after the // store; dropping a weak handle runs no code). let old = unsafe { (*self.0.get()).replace(shape) }; - drop(old); + // A different live callee keeps its shape among the earlier ones. + if let Some(old) = old.filter(|o| { + o.func.strong_count() > 0 && !(o.func.as_ptr() == func && o.code.as_ptr() == code) + }) { + // SAFETY: as above. + let poly = unsafe { &mut *self.1.get() }.get_or_insert_with(|| { + Box::new(PolyCalls { + shapes: std::array::from_fn(|_| None), + next: 0, + }) + }); + let seen = poly.shapes.iter().flatten().any(|s| { + s.func.as_ptr() == old.func.as_ptr() && s.code.as_ptr() == old.code.as_ptr() + }); + if !seen { + let i = poly.next; + poly.shapes[i] = Some(old); + poly.next = (i + 1) % POLY_CALLS; + } + } } } @@ -60540,13 +60856,16 @@ fn code_call_slot(code: &CodeObject, pc: usize) -> Option<&CallSlot> { .get(pc) } -/// A `LOAD_ATTR` site's polymorphic instance-attribute cache: up to two -/// `(class attr_version, instance-dict index)` pairs. Versions are +/// A `LOAD_ATTR` site's polymorphic instance-attribute cache: up to +/// [`ATTR_POLY`] `(class attr_version, instance-dict index)` pairs. Versions are /// process-unique and never reused, so a match names exactly one class /// in one state — the state in which the class cache resolved the name /// to the instance dict (no data descriptor, default /// `__getattribute__`); the index is a hint the name check validates. -struct AttrPoly(std::cell::UnsafeCell<[(u64, u32); 2]>); +struct AttrPoly(std::cell::UnsafeCell<[(u64, u32); ATTR_POLY]>); + +/// Receiver classes an [`AttrPoly`] remembers. +const ATTR_POLY: usize = 4; // SAFETY: read and written only from the dispatch loop with the GIL held // (the `MethodSlot` invariant). @@ -60555,7 +60874,7 @@ unsafe impl Sync for AttrPoly {} impl AttrPoly { const fn empty() -> Self { - Self(std::cell::UnsafeCell::new([(0, 0); 2])) + Self(std::cell::UnsafeCell::new([(0, 0); ATTR_POLY])) } /// The instance-dict value an entry for `ver` points at, when its @@ -60578,7 +60897,7 @@ impl AttrPoly { entries.iter().find(|e| e.0 == ver && ver != 0).map(|e| e.1) } - /// Remember `ver` → `ix` (the older of two full entries yields). + /// Remember `ver` → `ix` (the oldest of full entries yields). #[inline] fn record(&self, ver: u64, ix: u32) { // SAFETY: see the type docs. @@ -60586,8 +60905,8 @@ impl AttrPoly { if let Some(e) = entries.iter_mut().find(|e| e.0 == ver || e.0 == 0) { *e = (ver, ix); } else { - entries[0] = entries[1]; - entries[1] = (ver, ix); + entries.rotate_left(1); + entries[ATTR_POLY - 1] = (ver, ix); } } } @@ -60739,7 +61058,25 @@ fn class_attr_fill(code: &CodeObject, cls: &crate::types::TypeObject, pc: usize, /// the name compare that the `InlineCache` shape needs. The reference /// is weak (CPython's `_PyType_Lookup` cache is a borrowed pointer for /// the same reason): a site must not keep a class or module alive. -struct MethodSlot(std::cell::UnsafeCell<(u64, MethodSlotFn)>); +/// +/// A site that sees several receiver classes (`c.execute()` over a list +/// of constraint subclasses) keeps up to [`POLY_METHODS`] earlier +/// resolutions beside the current one, so it stops re-resolving on every +/// class change. +struct MethodSlot( + std::cell::UnsafeCell<(u64, MethodSlotFn)>, + std::cell::UnsafeCell>>, +); + +/// Earlier resolutions a polymorphic [`MethodSlot`] keeps. +const POLY_METHODS: usize = 3; + +/// A [`MethodSlot`]'s earlier plain or static functions, by class +/// version (`0` marks an empty entry), replaced round-robin. +struct PolyMethods { + entries: [(u64, crate::sync::Weak, bool); POLY_METHODS], + next: usize, +} /// What a [`MethodSlot`] resolved: a Python function off a class (keyed /// by the class's attribute version), or a builtin type's method body @@ -60777,10 +61114,68 @@ impl MethodSlot { const BUILTIN_TAG: u64 = 1 << 63; const fn empty() -> Self { - Self(std::cell::UnsafeCell::new(( - 0, - MethodSlotFn::Py(crate::sync::Weak::new()), - ))) + Self( + std::cell::UnsafeCell::new((0, MethodSlotFn::Py(crate::sync::Weak::new()))), + std::cell::UnsafeCell::new(None), + ) + } + + /// An earlier resolution under `ver` (a static method's only when + /// `unbound`), without a reference (see `peek_fn`). + #[cold] + #[inline(never)] + fn poly_peek(&self, ver: u64, unbound: bool) -> Option<*const crate::object::PyFunction> { + // SAFETY: GIL-serialized; no `&mut` escapes `poly_remember`. + let poly = unsafe { &*self.1.get() }.as_deref()?; + poly.entries + .iter() + .find(|(v, w, is_static)| { + *v == ver && ver != 0 && (unbound || !*is_static) && w.strong_count() > 0 + }) + .map(|(_, w, _)| w.as_ptr()) + } + + /// Keep the resolution the slot is about to replace, if it names a + /// live plain or static function under another version. + fn poly_remember(&self, ver: u64) { + // SAFETY: GIL-serialized; the references end before the slot is + // rewritten. + let (old_ver, old) = unsafe { &*self.0.get() }; + let (w, is_static) = match old { + MethodSlotFn::Py(w) => (w, false), + MethodSlotFn::Static(w) => (w, true), + _ => return, + }; + if *old_ver == 0 || *old_ver == ver || w.strong_count() == 0 { + return; + } + // SAFETY: as above. + let poly = unsafe { &mut *self.1.get() }.get_or_insert_with(|| { + Box::new(PolyMethods { + entries: std::array::from_fn(|_| (0, crate::sync::Weak::new(), false)), + next: 0, + }) + }); + if poly.entries.iter().any(|(v, _, _)| v == old_ver) { + return; + } + let i = poly.next; + poly.entries[i] = (*old_ver, w.clone(), is_static); + poly.next = (i + 1) % POLY_METHODS; + } + + /// A strong handle to the function at `p` (see `get_held`). + /// + /// # Safety + /// + /// `p` names a live function its class holds. + #[inline] + unsafe fn held(p: *const crate::object::PyFunction) -> Rc { + // SAFETY: the caller's contract. + unsafe { + Rc::increment_strong_count(p); + Rc::from_raw(p) + } } /// The cached function if the slot was filled under `ver`. @@ -60789,6 +61184,10 @@ impl MethodSlot { // SAFETY: GIL-serialized; no `&mut` escapes `set`. match unsafe { &*self.0.get() } { (v, MethodSlotFn::Py(w)) if *v == ver => w.upgrade(), + // SAFETY: the class holds a function it resolved at `ver`. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, false).map(|p| unsafe { Self::held(p) }) + } _ => None, } } @@ -60811,12 +61210,17 @@ impl MethodSlot { Some(Rc::from_raw(p)) } } + // SAFETY: as above. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, false).map(|p| unsafe { Self::held(p) }) + } _ => None, } } #[inline] fn set(&self, ver: u64, f: &Rc) { + self.poly_remember(ver); // SAFETY: GIL-serialized; the exclusive reference lives only for // the assignment. unsafe { *self.0.get() = (ver, MethodSlotFn::Py(Rc::downgrade(f))) }; @@ -60840,6 +61244,10 @@ impl MethodSlot { Some(Rc::from_raw(p)) } } + // SAFETY: as above. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, true).map(|p| unsafe { Self::held(p) }) + } _ => None, } } @@ -60852,6 +61260,7 @@ impl MethodSlot { // SAFETY: GIL-serialized; no `&mut` escapes `set`. match unsafe { &*self.0.get() } { (v, MethodSlotFn::Py(w)) if *v == ver && w.strong_count() > 0 => Some(w.as_ptr()), + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => self.poly_peek(ver, false), _ => None, } } @@ -60867,6 +61276,7 @@ impl MethodSlot { { Some(w.as_ptr()) } + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => self.poly_peek(ver, true), _ => None, } } @@ -60874,6 +61284,7 @@ impl MethodSlot { /// Remember a static method's function read through a class at `ver`. #[inline] fn set_static(&self, ver: u64, f: &Rc) { + self.poly_remember(ver); // SAFETY: as `set`. unsafe { *self.0.get() = (ver, MethodSlotFn::Static(Rc::downgrade(f))) }; } @@ -61060,6 +61471,10 @@ fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> } fn code_is_pure_leaf(code: &CodeObject) -> bool { + // The recorded verdict (see `code_pure_leaf_decide`): one load. + if let Some(yes) = code.jit_hint.pure_leaf() { + return yes; + } let Some(ext) = code_vm_ext(code) else { return false; }; @@ -61378,14 +61793,17 @@ fn exc_family_bit(name: &str) -> u16 { fn exc_family_flags(cls: &crate::types::TypeObject) -> u16 { let ver = cls.attr_version.get(); let memo = cls.exc_families.get(); - if memo & 1 != 0 && memo >> 10 == ver { - return ((memo >> 1) & 0x1FF) as u16; + if memo & 1 != 0 && memo >> 11 == ver { + return ((memo >> 1) & 0x3FF) as u16; } let mut flags = 0u16; for t in cls.mro.borrow().iter() { flags |= exc_family_bit(&t.name); } - cls.exc_families.set(ver << 10 | u64::from(flags) << 1 | 1); + if exception_init_uncached(cls).is_some() { + flags |= EXC_CUSTOM_INIT; + } + cls.exc_families.set(ver << 11 | u64::from(flags) << 1 | 1); flags } diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index bb6b572c..890d8910 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -6603,6 +6603,11 @@ unsafe extern "C" fn wpjit_iter_next(frame: *mut JitFrame, pin: i64, elem_tag: i let runs_python = !matches!(it, Object::Iter(_)); if runs_python { ctx.dirty = true; + // A generator resume from native code rebuilds a whole interpreter + // activation, several times what the interpreter's own inline + // resume costs: charged like an interpreter call, so a loop that + // drives a generator retires at the next poll (see `wpjit_poll`). + ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } match interp.iter_next(&it, &ctx.globals) { Err(err) => { @@ -9329,6 +9334,11 @@ unsafe extern "C" fn wpjit_iter_next_pair( let runs_python = !matches!(it, Object::Iter(_)); if runs_python { ctx.dirty = true; + // A generator resume from native code rebuilds a whole interpreter + // activation, several times what the interpreter's own inline + // resume costs: charged like an interpreter call, so a loop that + // drives a generator retires at the next poll (see `wpjit_poll`). + ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } match interp.iter_next(&it, &ctx.globals) { Err(err) => { From 6b4cf8756dd886a70018dfa148a760e0397e74a0 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 02:20:41 -0700 Subject: [PATCH 07/65] perf: call compiled scalar functions directly from native code A compiled self-recursive scalar frame (fib's shape) now calls itself with a native call: the callee's JitFrame and buffers live on the caller's stack frame, the activation is charged through small enter and exit helpers (GIL countdown, observers, recursion depth), and a deopt or raise finishes through a slow helper that rebuilds the callee's interpreter frame. The running compilation's layout rides on the call context so that rebuild never depends on the tier cache, which may retire the code mid-recursion. A call site whose callee is an already compiled, guard-free scalar leaf (spectral_norm's `_eval_a`) enters that leaf's code directly, falling back to the ordinary call helper when the enter helper declines or the leaf deopts (the leaf only computes, so it restarts from the top). Keyword calls no longer count toward retiring a native driver loop: the interpreter binds them through the same permutation, so tier 1 would not run them any cheaper (call_overhead's loop now stays compiled). fib: 445M -> 96M instructions; spectral_norm: 657M -> 228M. --- crates/weavepy-jit/src/analyze.rs | 1 + crates/weavepy-jit/src/engine.rs | 183 ++++++++++++++- crates/weavepy-jit/src/ir.rs | 9 +- crates/weavepy-jit/src/lib.rs | 24 +- crates/weavepy-jit/src/lower.rs | 365 +++++++++++++++++++++++++++++- crates/weavepy-jit/src/runtime.rs | 51 +++++ crates/weavepy-vm/src/tier2.rs | 267 +++++++++++++++++++--- 7 files changed, 852 insertions(+), 48 deletions(-) diff --git a/crates/weavepy-jit/src/analyze.rs b/crates/weavepy-jit/src/analyze.rs index fa661d61..b5ac4c45 100644 --- a/crates/weavepy-jit/src/analyze.rs +++ b/crates/weavepy-jit/src/analyze.rs @@ -7467,6 +7467,7 @@ fn emit_instr( token: mark.token, argc: argc as u8, ret, + is_self: mark.is_self, }, Some(ret), stack, diff --git a/crates/weavepy-jit/src/engine.rs b/crates/weavepy-jit/src/engine.rs index 272092b2..e0f28a63 100644 --- a/crates/weavepy-jit/src/engine.rs +++ b/crates/weavepy-jit/src/engine.rs @@ -9,12 +9,12 @@ use std::mem::{self, ManuallyDrop}; -use cranelift_codegen::ir::{types, AbiParam, Type}; +use cranelift_codegen::ir::{types, AbiParam, FuncRef, Type}; use cranelift_codegen::settings::{self, Configurable}; use cranelift_codegen::Context; use cranelift_frontend::FunctionBuilderContext; use cranelift_jit::{JITBuilder, JITModule}; -use cranelift_module::{Linkage, Module}; +use cranelift_module::{FuncId, Linkage, Module}; use crate::analyze::{JitVerdict, Probes}; use crate::ir::{ @@ -118,6 +118,21 @@ pub struct CompiledFrame { pub ret_lane: Option, scalar_leaf: bool, op_mix: OpMix, + /// The engine's handle for this function, for direct calls from + /// frames compiled later (see [`Self::direct_leaf`]). + func_id: FuncId, +} + +/// A compiled scalar leaf another frame may call directly (see +/// [`CompiledFrame::direct_leaf`]): its function and frame layout, and +/// the lanes of its parameters and result. +#[derive(Clone, Debug)] +pub struct DirectLeaf { + func_id: FuncId, + n_locals: u32, + max_stack: u32, + params: Vec, + ret: JitType, } impl CompiledFrame { @@ -131,6 +146,33 @@ impl CompiledFrame { self.scalar_leaf } + /// This frame as a direct-call target taking `arg_count` arguments: + /// a scalar leaf with no global or math guards (so nothing about + /// the namespace can invalidate it) and scalar parameter lanes. A + /// caller compiled on the same engine calls it in native code; its + /// numeric deopt restarts the call through the ordinary path. + #[must_use] + pub fn direct_leaf(&self, arg_count: u32) -> Option { + let scalar = |t: JitType| matches!(t, JitType::Int | JitType::Float | JitType::Bool); + if !self.scalar_leaf || !self.global_guards.is_empty() || !self.math_guards.is_empty() { + return None; + } + let ret = self.ret_lane.filter(|&t| scalar(t))?; + let params = self + .local_types + .get(..arg_count as usize)? + .iter() + .map(|t| t.filter(|&t| scalar(t))) + .collect::>>()?; + Some(DirectLeaf { + func_id: self.func_id, + n_locals: self.n_locals, + max_stack: self.max_stack, + params, + ret, + }) + } + /// `(generic, total)` statement counts: how many of the compiled /// statements go through the interpreter's generic object protocol /// (dynamic calls and attribute accesses). @@ -279,13 +321,37 @@ impl JitEngine { code: &CodeObject, resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, probes: &mut Probes<'_>, + ) -> Result { + self.compile_frame_direct(code, resolve, probes, &mut |_| None) + } + + /// [`Self::compile_frame`] with the direct-call targets of the callee + /// tokens: `direct(token)` names a scalar leaf compiled on this + /// engine (see [`CompiledFrame::direct_leaf`]) that a matching call + /// site enters in native code. + pub fn compile_frame_direct( + &mut self, + code: &CodeObject, + resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, + probes: &mut Probes<'_>, + direct: &mut dyn FnMut(u32) -> Option, ) -> Result { let tfunc = crate::analyze::analyze_frame(code, resolve, probes)?; - self.compile_tfunc(&tfunc) + self.compile_tfunc_direct(&tfunc, direct) } /// Compile an already-analyzed [`TFunc`] (also the unit-test entry). pub fn compile_tfunc(&mut self, tfunc: &TFunc) -> Result { + self.compile_tfunc_direct(tfunc, &mut |_| None) + } + + /// [`Self::compile_tfunc`] with direct-call targets (see + /// [`Self::compile_frame_direct`]). + pub fn compile_tfunc_direct( + &mut self, + tfunc: &TFunc, + direct: &mut dyn FnMut(u32) -> Option, + ) -> Result { // These operations each lower to a dedicated embedder helper. // Reject missing registrations before embedding an absolute address. for stmt in tfunc.blocks.iter().flat_map(|block| &block.stmts) { @@ -440,14 +506,64 @@ impl JitEngine { .returns .push(AbiParam::new(types::I64)); - build_function(&mut self.ctx.func, &mut self.fbctx, tfunc, self.ptr_ty); - let name = format!("wpjit_{}", self.next_id); self.next_id += 1; let id = self .module .declare_function(&name, Linkage::Local, &self.ctx.func.signature) .map_err(|_| JitVerdict::NotConverged)?; + // A scalar frame calls itself directly (declared above so the body + // can name its own function). + let self_func = (runtime::self_call_helper_addrs().is_some() + && self_direct_eligible(tfunc)) + .then(|| self.module.declare_func_in_func(id, &mut self.ctx.func)); + // Direct leaf targets, per call token whose sites agree with the + // leaf's arity and result lane (the lowering checks the argument + // lanes at each site). + let mut leaves: Vec<(u32, FuncRef, DirectLeaf)> = Vec::new(); + if runtime::self_call_helper_addrs().is_some() { + for stmt in tfunc.blocks.iter().flat_map(|b| &b.stmts) { + let TOp::CallPy { + token, + argc, + ret, + is_self: false, + } = stmt.op + else { + continue; + }; + if leaves.iter().any(|(t, ..)| *t == token) { + continue; + } + if let Some(leaf) = direct(token) { + if leaf.params.len() == argc as usize && leaf.ret == ret { + let fref = self + .module + .declare_func_in_func(leaf.func_id, &mut self.ctx.func); + leaves.push((token, fref, leaf)); + } + } + } + } + + build_function( + &mut self.ctx.func, + &mut self.fbctx, + tfunc, + self.ptr_ty, + self_func, + leaves + .into_iter() + .map(|(token, fref, leaf)| crate::lower::LeafTarget { + token, + func: fref, + n_locals: leaf.n_locals, + max_stack: leaf.max_stack, + params: leaf.params, + }) + .collect(), + ); + self.module .define_function(id, &mut self.ctx) .map_err(|_| JitVerdict::NotConverged)?; @@ -519,6 +635,7 @@ impl JitEngine { ret_none: tfunc.ret_none, scalar_leaf: is_scalar_leaf(tfunc), op_mix: op_mix(tfunc), + func_id: id, }) } } @@ -649,5 +766,61 @@ fn is_scalar_leaf(tfunc: &TFunc) -> bool { }) } +/// Whether `tfunc` may call itself directly (see +/// [`crate::runtime::SelfEnterHelper`]): it makes a self call, and every +/// local, operand and result is a scalar and every operation pure scalar +/// work or a self call, so no activation of it ever pins an object. Its +/// activations can then share one embedder context (the pin table, the +/// parked result) and live on the native stack. +fn self_direct_eligible(tfunc: &TFunc) -> bool { + let scalar = |t: JitType| matches!(t, JitType::Int | JitType::Float | JitType::Bool); + let mut any_self = false; + let ops_ok = tfunc.blocks.iter().all(|b| { + let term_ok = matches!( + b.term, + TTerm::Return + | TTerm::Jump(_) + | TTerm::BranchFalse { .. } + | TTerm::BranchTrue { .. } + | TTerm::Deopt { .. } + ); + term_ok + && b.stmts.iter().all(|st| match st.op { + TOp::CallPy { + is_self: true, ret, .. + } => { + any_self = true; + scalar(ret) + } + TOp::PushConstInt(_) + | TOp::PushConstFloat(_) + | TOp::PushConstBool(_) + | TOp::LoadLocal(_) + | TOp::StoreLocal(_) + | TOp::IntArith(_) + | TOp::FloatArith(_) + | TOp::IntTrueDiv + | TOp::IntCmp(_) + | TOp::FloatCmp(_) + | TOp::IntNeg + | TOp::FloatNeg + | TOp::IntInvert + | TOp::IntNot + | TOp::FloatNot + | TOp::Pop + | TOp::Dup { .. } + | TOp::Swap2 + | TOp::SwapN { .. } + | TOp::IntToFloatTos { .. } + | TOp::IntToFloatSecond { .. } => true, + _ => false, + }) + }); + any_self + && ops_ok + && tfunc.ret_lane.is_some_and(scalar) + && tfunc.local_types.iter().flatten().all(|t| scalar(*t)) +} + #[cfg(test)] mod tests; diff --git a/crates/weavepy-jit/src/ir.rs b/crates/weavepy-jit/src/ir.rs index 289a4cac..d98a01bc 100644 --- a/crates/weavepy-jit/src/ir.rs +++ b/crates/weavepy-jit/src/ir.rs @@ -173,7 +173,14 @@ pub enum TOp { /// callee takes the `Raised` exit at this pc; a result outside the /// `ret` lane (or a caller guard invalidated by the callee's side /// effects) deopts *after* the call with the result spilled. - CallPy { token: u32, argc: u8, ret: JitType }, + /// `is_self`: the callee is this very code object, which a scalar + /// frame calls directly (see `engine::self_direct_eligible`). + CallPy { + token: u32, + argc: u8, + ret: JitType, + is_self: bool, + }, /// RFC 0073 WS5 — a Python-to-Python *keyword* call (`CALL_KW`) /// through the same `wpjit_call_py` helper. Pops `argc + kwc` /// values (positionals below, keyword values above, interpreter diff --git a/crates/weavepy-jit/src/lib.rs b/crates/weavepy-jit/src/lib.rs index 4735c481..f846c037 100644 --- a/crates/weavepy-jit/src/lib.rs +++ b/crates/weavepy-jit/src/lib.rs @@ -34,7 +34,7 @@ pub use analyze::{ returns_none_syntactically, returns_self_syntactically, JitVerdict, MethodResolution, PathArena, Probes, ELEM_SENTINEL, }; -pub use engine::{CompiledFrame, JitEngine, OpMix}; +pub use engine::{CompiledFrame, DirectLeaf, JitEngine, OpMix}; pub use ir::{ ArithKind, AttrSiteMeta, BlockId, CalleeSpanMeta, CmpKind, CompSavedMeta, CtorFieldSrc, GlobalGuard, IterLoopMeta, ListLoopMeta, MathFunc, MathGuardMeta, MethodRet, MethodSiteMeta, @@ -51,17 +51,17 @@ pub use runtime::{ register_iter_helpers, register_iter_new_helper, register_iter_next_pair_helper, register_list_extra_helpers, register_list_from_range_helper, register_list_helpers, register_list_next_helper, register_math_helpers, register_poll_helper, - register_str_format_helpers, register_str_helpers, register_str_method_helper, - register_str_write_helpers, register_truth_helper, register_tuple_read_helpers, - register_unbox_int_helper, AttrGetChainHelper, AttrGetHelper, AttrSetHelper, BuildListHelper, - BuildTupleHelper, BytesGetHelper, CachedAttrChainHelper, CallDynHelper, CallMethodHelper, - CallPyHelper, CallStatus, CellGetHelper, CellSetHelper, DictAccessHelper, DynAttrHelper, - GetIterHelper, IterNextHelper, IterNextPairHelper, JitFrame, JitStatus, ListAppendHelper, - ListFromRangeHelper, ListGetHelper, ListLenHelper, ListNextHelper, ListRepeatHelper, - ListSetHelper, ListSliceHelper, MathBinaryHelper, MathUnaryHelper, PollHelper, SlotTag, - StrEqHelper, StrLenHelper, StrModHelper, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, - DICT_VAL_INT, DICT_VAL_OBJ, ITER_ELEM_STR, JIT_POLL_STRIDE, MAX_ATTR_CHAIN_LEN, - MAX_CACHED_ATTR_CHAIN_LEN, + register_self_call_helpers, register_str_format_helpers, register_str_helpers, + register_str_method_helper, register_str_write_helpers, register_truth_helper, + register_tuple_read_helpers, register_unbox_int_helper, AttrGetChainHelper, AttrGetHelper, + AttrSetHelper, BuildListHelper, BuildTupleHelper, BytesGetHelper, CachedAttrChainHelper, + CallDynHelper, CallMethodHelper, CallPyHelper, CallStatus, CellGetHelper, CellSetHelper, + DictAccessHelper, DynAttrHelper, GetIterHelper, IterNextHelper, IterNextPairHelper, JitFrame, + JitStatus, ListAppendHelper, ListFromRangeHelper, ListGetHelper, ListLenHelper, ListNextHelper, + ListRepeatHelper, ListSetHelper, ListSliceHelper, MathBinaryHelper, MathUnaryHelper, + PollHelper, SelfEnterHelper, SelfExitHelper, SelfSlowHelper, SlotTag, StrEqHelper, + StrLenHelper, StrModHelper, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, DICT_VAL_INT, + DICT_VAL_OBJ, ITER_ELEM_STR, JIT_POLL_STRIDE, MAX_ATTR_CHAIN_LEN, MAX_CACHED_ATTR_CHAIN_LEN, }; pub use value::JitType; diff --git a/crates/weavepy-jit/src/lower.rs b/crates/weavepy-jit/src/lower.rs index 1ad84fe7..9651cef4 100644 --- a/crates/weavepy-jit/src/lower.rs +++ b/crates/weavepy-jit/src/lower.rs @@ -13,8 +13,8 @@ use cranelift_codegen::ir::condcodes::{FloatCC, IntCC}; use cranelift_codegen::ir::{ - types, AbiParam, Block, BlockArg, Function, InstBuilder, MemFlags, SigRef, Signature, Type, - Value, + types, AbiParam, Block, BlockArg, FuncRef, Function, InstBuilder, MemFlags, SigRef, Signature, + StackSlot, StackSlotData, StackSlotKind, Type, Value, }; use cranelift_frontend::{FunctionBuilder, FunctionBuilderContext, Variable}; @@ -32,6 +32,20 @@ const OFF_STACK_TAGS: i32 = core::mem::offset_of!(JitFrame, stack_tags) as i32; const OFF_STACK_LEN: i32 = core::mem::offset_of!(JitFrame, stack_len) as i32; const OFF_CALL_ARGS: i32 = core::mem::offset_of!(JitFrame, call_args) as i32; const OFF_CALL_TAGS: i32 = core::mem::offset_of!(JitFrame, call_tags) as i32; +const OFF_N_LOCALS: i32 = core::mem::offset_of!(JitFrame, n_locals) as i32; +const OFF_STACK_CAP: i32 = core::mem::offset_of!(JitFrame, stack_cap) as i32; +const OFF_CTX: i32 = core::mem::offset_of!(JitFrame, ctx) as i32; + +/// A call token whose sites may enter a compiled scalar leaf directly +/// (see `engine::CompiledFrame::direct_leaf`): the leaf's function in +/// this module, its frame layout, and its parameter lanes. +pub(crate) struct LeafTarget { + pub token: u32, + pub func: FuncRef, + pub n_locals: u32, + pub max_stack: u32, + pub params: Vec, +} /// Build the Cranelift function body for `tfunc` into `func`. pub(crate) fn build_function( @@ -39,9 +53,13 @@ pub(crate) fn build_function( fbctx: &mut FunctionBuilderContext, tfunc: &TFunc, ptr_ty: Type, + self_func: Option, + leaves: Vec, ) { let mut builder = FunctionBuilder::new(func, fbctx); let mut lc = Lowerer::new(&mut builder, tfunc, ptr_ty); + lc.self_func = self_func; + lc.leaves = leaves; lc.build(); builder.seal_all_blocks(); builder.finalize(); @@ -100,6 +118,19 @@ struct Lowerer<'a, 'b> { poll_countdown: Option, /// The abstract operand stack: SSA value + lane. vstack: Vec<(Value, JitType)>, + /// This function itself, for direct self calls (see + /// `engine::self_direct_eligible`); `None` when it makes none. + self_func: Option, + /// The callee frame of a direct self call (lazy): its `JitFrame` + /// then its buffers. + self_slot: Option, + /// Imported signatures of the self-call helpers (lazy): `(frame) -> + /// i64` and the slow path's `(frame, callee, status, token, tag) -> + /// i64`. + self_sig: Option, + self_slow_sig: Option, + /// Direct scalar-leaf call targets by token. + leaves: Vec, } impl<'a, 'b> Lowerer<'a, 'b> { @@ -130,6 +161,11 @@ impl<'a, 'b> Lowerer<'a, 'b> { poll_sig: None, poll_countdown: None, vstack: Vec::new(), + self_func: None, + self_slot: None, + self_sig: None, + self_slow_sig: None, + leaves: Vec::new(), } } @@ -948,7 +984,20 @@ impl<'a, 'b> Lowerer<'a, 'b> { let depth = self.vstack.len() - 2; self.emit_int_to_float(depth, guarded, stmt.pc); } - TOp::CallPy { token, argc, ret } => self.emit_call_py(token, argc, 0, 0, ret, stmt.pc), + TOp::CallPy { + token, + argc, + ret, + is_self, + } => { + if is_self && self.self_func.is_some() { + self.emit_call_self(token, argc, ret, stmt.pc); + } else if let Some(ix) = self.leaf_for(token, argc) { + self.emit_call_leaf(ix, token, argc, ret, stmt.pc); + } else { + self.emit_call_py(token, argc, 0, 0, ret, stmt.pc); + } + } TOp::CallPyKw { token, argc, @@ -3089,6 +3138,316 @@ impl<'a, 'b> Lowerer<'a, 'b> { /// (4 bits each, tier-1's `CallPyKwNames` packing). The analyzer /// validated the filled set to be exactly `0..argc+kwc`, so the /// helper still sees a plain positional prefix. + /// Lower a direct self call (see `engine::self_direct_eligible`): + /// charge the activation through the enter helper, fill a callee + /// `JitFrame` on this function's native stack frame (the arguments + /// in its first locals, the caller's embedder context shared), call + /// this function itself, and release the charge. A callee that + /// deopts or raises finishes through the slow helper, whose + /// [`crate::runtime::CallStatus`] the caller handles as it does the + /// call helper's. When the enter helper declines (recursion limit, + /// pending interpreter work, observers), the ordinary call runs. + fn emit_call_self(&mut self, token: u32, argc: u8, ret: JitType, pc: u32) { + let trusted = MemFlags::trusted(); + let (enter_addr, exit_addr, slow_addr) = + runtime::self_call_helper_addrs().expect("checked by the engine"); + let n = argc as usize; + let base = self.vstack.len() - n; + let args: Vec<(Value, JitType)> = self.vstack[base..].to_vec(); + self.vstack.truncate(base); + let snapshot = self.vstack.clone(); + self.writeback_locals(); + self.store_call_site_pc(pc); + + let sig = self.self_sig(); + let enter = self.b.ins().iconst(self.ptr_ty, enter_addr as i64); + let call = self.b.ins().call_indirect(sig, enter, &[self.frame_ptr]); + let declined = self.b.inst_results(call)[0]; + + let direct_b = self.b.create_block(); + let generic_b = self.b.create_block(); + let join_b = self.b.create_block(); + self.b.append_block_param(join_b, Self::cl_ty(ret)); + let go = self.b.ins().icmp_imm(IntCC::Equal, declined, 0); + self.b.ins().brif(go, direct_b, &[], generic_b, &[]); + + // Declined: the ordinary call helper. + self.b.switch_to_block(generic_b); + self.vstack.extend(args.iter().copied()); + self.emit_call_py(token, argc, 0, 0, ret, pc); + let (v, _) = self.vstack.pop().expect("the call's result"); + self.b.ins().jump(join_b, &[v.into()]); + + // Direct: the callee frame and buffers, laid out in one slot. + self.b.switch_to_block(direct_b); + let n_locals = self.tfunc.n_locals.max(1) as i32; + let cap = self.tfunc.max_stack as i32 + 1; + let call_cap = self.tfunc.max_call_args.max(1) as i32; + let frame_size = core::mem::size_of::() as i32; + let off_locals = frame_size; + let off_spill = off_locals + n_locals * 8; + let off_tags = off_spill + cap * 8; + let off_call_args = (off_tags + cap * 4 + 7) & !7; + let off_call_tags = off_call_args + call_cap * 8; + let size = off_call_tags + call_cap * 4; + let slot = match self.self_slot { + Some(slot) => slot, + None => { + let slot = self.b.create_sized_stack_slot(StackSlotData::new( + StackSlotKind::ExplicitSlot, + size as u32, + 3, + )); + self.self_slot = Some(slot); + slot + } + }; + let fp = self.b.ins().stack_addr(self.ptr_ty, slot, 0); + let at = |b: &mut FunctionBuilder<'_>, ptr_ty: Type, off: i32| { + b.ins().stack_addr(ptr_ty, slot, off) + }; + let locals = at(self.b, self.ptr_ty, off_locals); + let spill = at(self.b, self.ptr_ty, off_spill); + let tags = at(self.b, self.ptr_ty, off_tags); + let cargs = at(self.b, self.ptr_ty, off_call_args); + let ctags = at(self.b, self.ptr_ty, off_call_tags); + let ctx = self + .b + .ins() + .load(self.ptr_ty, trusted, self.frame_ptr, OFF_CTX); + let zero64 = self.b.ins().iconst(types::I64, 0); + let zero32 = self.b.ins().iconst(types::I32, 0); + self.b.ins().store(trusted, locals, fp, OFF_LOCALS); + let nl = self.b.ins().iconst(types::I32, i64::from(n_locals)); + self.b.ins().store(trusted, nl, fp, OFF_N_LOCALS); + self.b.ins().store(trusted, zero32, fp, OFF_ENTRY_PC); + self.b.ins().store(trusted, zero64, fp, OFF_RET_BITS); + self.b.ins().store(trusted, zero32, fp, OFF_RET_TAG); + self.b.ins().store(trusted, zero32, fp, OFF_DEOPT_PC); + self.b.ins().store(trusted, spill, fp, OFF_STACK_SPILL); + self.b.ins().store(trusted, tags, fp, OFF_STACK_TAGS); + self.b.ins().store(trusted, zero32, fp, OFF_STACK_LEN); + let capv = self.b.ins().iconst(types::I32, i64::from(cap)); + self.b.ins().store(trusted, capv, fp, OFF_STACK_CAP); + self.b.ins().store(trusted, ctx, fp, OFF_CTX); + self.b.ins().store(trusted, cargs, fp, OFF_CALL_ARGS); + self.b.ins().store(trusted, ctags, fp, OFF_CALL_TAGS); + // The arguments bind the first locals (their lanes are the + // parameters' own); every other local starts zeroed, as the + // framed entries leave it. + for slot_ix in 0..n_locals { + let off = slot_ix * 8; + match args.get(slot_ix as usize) { + Some(&(v, _)) => { + self.b.ins().store(trusted, v, locals, off); + } + None => { + self.b.ins().store(trusted, zero64, locals, off); + } + } + } + let self_func = self.self_func.expect("checked by the caller"); + let call = self.b.ins().call(self_func, &[fp]); + let status = self.b.inst_results(call)[0]; + let exit = self.b.ins().iconst(self.ptr_ty, exit_addr as i64); + self.b.ins().call_indirect(sig, exit, &[self.frame_ptr]); + + let returned_b = self.b.create_block(); + let slow_b = self.b.create_block(); + let is_ret = self + .b + .ins() + .icmp_imm(IntCC::Equal, status, JitStatus::Returned as i64); + self.b.ins().brif(is_ret, returned_b, &[], slow_b, &[]); + + self.b.switch_to_block(returned_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, fp, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + // Deopted or raised: finished by the slow helper. + self.b.switch_to_block(slow_b); + let slow_sig = self.self_slow_sig(); + let slow = self.b.ins().iconst(self.ptr_ty, slow_addr as i64); + let tokenv = self.b.ins().iconst(types::I64, i64::from(token)); + let tagv = self.b.ins().iconst(types::I64, Self::tag(ret)); + let call = + self.b + .ins() + .call_indirect(slow_sig, slow, &[self.frame_ptr, fp, status, tokenv, tagv]); + let cstatus = self.b.inst_results(call)[0]; + let ok_b = self.b.create_block(); + let bad_b = self.b.create_block(); + let is_ok = self.b.ins().icmp_imm(IntCC::Equal, cstatus, 0); + self.b.ins().brif(is_ok, ok_b, &[], bad_b, &[]); + self.b.switch_to_block(bad_b); + let raised_b = self.b.create_block(); + let boxed_b = self.b.create_block(); + let is_raised = self.b.ins().icmp_imm(IntCC::Equal, cstatus, 1); + self.b.ins().brif(is_raised, raised_b, &[], boxed_b, &[]); + self.b.switch_to_block(raised_b); + self.emit_exit(pc, &snapshot, JitStatus::Raised); + self.b.switch_to_block(boxed_b); + self.emit_exit(pc + 1, &snapshot, JitStatus::Deopt); + self.b.switch_to_block(ok_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, self.frame_ptr, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + self.b.switch_to_block(join_b); + let v = self.b.block_params(join_b)[0]; + self.vstack.push((v, ret)); + } + + /// The direct leaf target of `token` when this site's argument lanes + /// are exactly the leaf's parameter lanes. + fn leaf_for(&self, token: u32, argc: u8) -> Option { + let ix = self.leaves.iter().position(|l| l.token == token)?; + let base = self.vstack.len().checked_sub(argc as usize)?; + let lanes = self.vstack[base..].iter().map(|&(_, ty)| ty); + lanes + .eq(self.leaves[ix].params.iter().copied()) + .then_some(ix) + } + + /// Lower a direct call of a compiled scalar leaf (see [`LeafTarget`]): + /// charge the activation through the self-call enter helper, fill the + /// leaf's `JitFrame` on this function's native stack frame, call it, + /// and release the charge. The leaf only computes, so when the enter + /// helper declines or the leaf deopts (an overflow, a zero divisor), + /// the ordinary call helper runs the call from the start. + fn emit_call_leaf(&mut self, ix: usize, token: u32, argc: u8, ret: JitType, pc: u32) { + let trusted = MemFlags::trusted(); + let (enter_addr, exit_addr, _) = + runtime::self_call_helper_addrs().expect("checked by the engine"); + let n = argc as usize; + let base = self.vstack.len() - n; + let args: Vec<(Value, JitType)> = self.vstack[base..].to_vec(); + self.vstack.truncate(base); + self.writeback_locals(); + self.store_call_site_pc(pc); + + let sig = self.self_sig(); + let enter = self.b.ins().iconst(self.ptr_ty, enter_addr as i64); + let call = self.b.ins().call_indirect(sig, enter, &[self.frame_ptr]); + let declined = self.b.inst_results(call)[0]; + + let direct_b = self.b.create_block(); + let generic_b = self.b.create_block(); + let join_b = self.b.create_block(); + self.b.append_block_param(join_b, Self::cl_ty(ret)); + let go = self.b.ins().icmp_imm(IntCC::Equal, declined, 0); + self.b.ins().brif(go, direct_b, &[], generic_b, &[]); + + // The leaf's frame and buffers, in one slot per site. + self.b.switch_to_block(direct_b); + let leaf = &self.leaves[ix]; + let (func, n_locals, cap) = ( + leaf.func, + leaf.n_locals.max(1) as i32, + leaf.max_stack as i32 + 1, + ); + let frame_size = core::mem::size_of::() as i32; + let off_locals = frame_size; + let off_spill = off_locals + n_locals * 8; + let off_tags = off_spill + cap * 8; + let size = off_tags + cap * 4; + let slot = self.b.create_sized_stack_slot(StackSlotData::new( + StackSlotKind::ExplicitSlot, + size as u32, + 3, + )); + let fp = self.b.ins().stack_addr(self.ptr_ty, slot, 0); + let locals = self.b.ins().stack_addr(self.ptr_ty, slot, off_locals); + let spill = self.b.ins().stack_addr(self.ptr_ty, slot, off_spill); + let tags = self.b.ins().stack_addr(self.ptr_ty, slot, off_tags); + let null = self.b.ins().iconst(self.ptr_ty, 0); + let zero64 = self.b.ins().iconst(types::I64, 0); + let zero32 = self.b.ins().iconst(types::I32, 0); + self.b.ins().store(trusted, locals, fp, OFF_LOCALS); + let nl = self.b.ins().iconst(types::I32, i64::from(n_locals)); + self.b.ins().store(trusted, nl, fp, OFF_N_LOCALS); + self.b.ins().store(trusted, zero32, fp, OFF_ENTRY_PC); + self.b.ins().store(trusted, zero64, fp, OFF_RET_BITS); + self.b.ins().store(trusted, zero32, fp, OFF_RET_TAG); + self.b.ins().store(trusted, zero32, fp, OFF_DEOPT_PC); + self.b.ins().store(trusted, spill, fp, OFF_STACK_SPILL); + self.b.ins().store(trusted, tags, fp, OFF_STACK_TAGS); + self.b.ins().store(trusted, zero32, fp, OFF_STACK_LEN); + let capv = self.b.ins().iconst(types::I32, i64::from(cap)); + self.b.ins().store(trusted, capv, fp, OFF_STACK_CAP); + // A scalar leaf reads no embedder context and marshals no calls. + self.b.ins().store(trusted, null, fp, OFF_CTX); + self.b.ins().store(trusted, null, fp, OFF_CALL_ARGS); + self.b.ins().store(trusted, null, fp, OFF_CALL_TAGS); + for slot_ix in 0..n_locals { + let v = args.get(slot_ix as usize).map_or(zero64, |&(v, _)| v); + self.b.ins().store(trusted, v, locals, slot_ix * 8); + } + let call = self.b.ins().call(func, &[fp]); + let status = self.b.inst_results(call)[0]; + let exit = self.b.ins().iconst(self.ptr_ty, exit_addr as i64); + self.b.ins().call_indirect(sig, exit, &[self.frame_ptr]); + let returned_b = self.b.create_block(); + let is_ret = self + .b + .ins() + .icmp_imm(IntCC::Equal, status, JitStatus::Returned as i64); + self.b.ins().brif(is_ret, returned_b, &[], generic_b, &[]); + self.b.switch_to_block(returned_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, fp, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + // Declined or deopted: the ordinary call, from the start. + self.b.switch_to_block(generic_b); + self.vstack.extend(args.iter().copied()); + self.emit_call_py(token, argc, 0, 0, ret, pc); + let (v, _) = self.vstack.pop().expect("the call's result"); + self.b.ins().jump(join_b, &[v.into()]); + + self.b.switch_to_block(join_b); + let v = self.b.block_params(join_b)[0]; + self.vstack.push((v, ret)); + } + + /// The `(frame) -> i64` signature of the self-call enter/exit + /// helpers (lazy). + fn self_sig(&mut self) -> SigRef { + if let Some(sig) = self.self_sig { + return sig; + } + let mut sig = Signature::new(self.b.func.signature.call_conv); + sig.params.push(AbiParam::new(self.ptr_ty)); + sig.returns.push(AbiParam::new(types::I64)); + let r = self.b.import_signature(sig); + self.self_sig = Some(r); + r + } + + /// The slow self-call helper's signature (lazy). + fn self_slow_sig(&mut self) -> SigRef { + if let Some(sig) = self.self_slow_sig { + return sig; + } + let mut sig = Signature::new(self.b.func.signature.call_conv); + sig.params.push(AbiParam::new(self.ptr_ty)); // frame + sig.params.push(AbiParam::new(self.ptr_ty)); // callee frame + sig.params.push(AbiParam::new(types::I64)); // callee status + sig.params.push(AbiParam::new(types::I64)); // token + sig.params.push(AbiParam::new(types::I64)); // expected tag + sig.returns.push(AbiParam::new(types::I64)); // call status + let r = self.b.import_signature(sig); + self.self_slow_sig = Some(r); + r + } + fn emit_call_py(&mut self, token: u32, argc: u8, kwc: u8, perm: u32, ret: JitType, pc: u32) { let trusted = MemFlags::trusted(); let n = argc as usize + kwc as usize; diff --git a/crates/weavepy-jit/src/runtime.rs b/crates/weavepy-jit/src/runtime.rs index 67d74be6..41675dcb 100644 --- a/crates/weavepy-jit/src/runtime.rs +++ b/crates/weavepy-jit/src/runtime.rs @@ -241,6 +241,57 @@ pub(crate) fn call_py_helper_addr() -> usize { CALL_PY_HELPER.load(std::sync::atomic::Ordering::Acquire) } +/// The embedder's direct self-call helpers (see +/// `engine::self_direct_eligible`): a scalar frame calls itself natively, +/// its callee's [`JitFrame`] on the native stack and the caller's +/// embedder context shared. +/// +/// - [`SelfEnterHelper`] charges one activation (recursion depth, GIL +/// countdown) before the call: `0` to call directly, non-zero to take +/// the ordinary `wpjit_call_py` path instead (nothing charged). +/// - [`SelfExitHelper`] releases the charge after the callee returns. +/// - [`SelfSlowHelper`] finishes a callee that did not return: its +/// [`JitStatus`] (`Deopt` or `Raised`) with its frame still live. +/// Returns a [`CallStatus`] for the caller, as the call helper does. +pub type SelfEnterHelper = unsafe extern "C" fn(frame: *mut JitFrame) -> i64; +/// See [`SelfEnterHelper`]. +pub type SelfExitHelper = unsafe extern "C" fn(frame: *mut JitFrame) -> i64; +/// See [`SelfEnterHelper`]. +pub type SelfSlowHelper = unsafe extern "C" fn( + frame: *mut JitFrame, + callee: *mut JitFrame, + status: i64, + token: i64, + expect_tag: i64, +) -> i64; + +static SELF_ENTER_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static SELF_EXIT_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static SELF_SLOW_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + +/// Register the direct self-call helpers (see [`SelfEnterHelper`]). +pub fn register_self_call_helpers( + enter: SelfEnterHelper, + exit: SelfExitHelper, + slow: SelfSlowHelper, +) { + use std::sync::atomic::Ordering::Release; + SELF_ENTER_HELPER.store(enter as usize, Release); + SELF_EXIT_HELPER.store(exit as usize, Release); + SELF_SLOW_HELPER.store(slow as usize, Release); +} + +/// The registered self-call helpers' addresses (enter, exit, slow), or +/// `None` when any is absent. +#[must_use] +pub(crate) fn self_call_helper_addrs() -> Option<(usize, usize, usize)> { + use std::sync::atomic::Ordering::Acquire; + let a = SELF_ENTER_HELPER.load(Acquire); + let b = SELF_EXIT_HELPER.load(Acquire); + let c = SELF_SLOW_HELPER.load(Acquire); + (a != 0 && b != 0 && c != 0).then_some((a, b, c)) +} + /// RFC 0061 WS5 — the embedder's pinned-list *read* helper. `pin` /// indexes the per-entry pinned-object table on the embedder context; /// `idx` is the (possibly negative) Python index. Returns `0` (Ok) with diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 890d8910..e7228082 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -784,6 +784,7 @@ impl JitState { ); // RFC 0067 WS2 — the eval-breaker poll for native loop headers. weavepy_jit::register_poll_helper(wpjit_poll); + weavepy_jit::register_self_call_helpers(wpjit_self_enter, wpjit_self_exit, wpjit_self_slow); // RFC 0069 WS1 — the guarded method-call lane. weavepy_jit::register_call_method_helper(wpjit_call_method); // RFC 0073 WS3 — the native `str`-method lane. @@ -1133,7 +1134,21 @@ impl JitState { let t0 = std::env::var_os("WEAVEPY_JIT_TRACE") .is_some() .then(std::time::Instant::now); - let r = engine.compile_frame(code, &mut classify, &mut jit_probes); + // A callee already compiled as a guard-free scalar leaf is + // entered directly by native code (its identity is guarded + // with the callee table like any burned-in callee). + let mut direct = |token: u32| { + let callees = callees.borrow(); + let (Object::Function(_), fcode) = callees.get(token as usize)? else { + return None; + }; + let k = Rc::as_ptr(fcode).cast::(); + match &cache_ref.get(&k)?.tier { + Tier::Compiled(a) => a.cf.direct_leaf(fcode.arg_count), + _ => None, + } + }; + let r = engine.compile_frame_direct(code, &mut classify, &mut jit_probes, &mut direct); if let Some(t0) = t0 { eprintln!("jit compile-time {:?} {:?}", code.name, t0.elapsed()); } @@ -3047,6 +3062,11 @@ struct CallCtx { /// re-entrancy pattern as `vm_singletons::publish_interpreter_ptr`). interp: *mut super::Interpreter, callees: StdRc, + /// The running compilation's frame layout, owned by whoever entered + /// this activation. A direct self call shares its caller's context, + /// so its deopt rebuild reads the layout here rather than through + /// the tier cache, which may retire the code mid-recursion. + cf: *const CompiledFrame, guard_snapshot: StdRc, /// The caller frame's namespaces, for post-call guard revalidation /// (the caller `Frame` itself is mutably borrowed across the native @@ -4230,6 +4250,7 @@ unsafe fn try_native_call( if !c.obj_global_pins.is_empty() { c.obj_global_pins.clear(); } + c.cf = StdRc::as_ptr(&nc.cf); c.dirty = false; c.interp_calls = 0; c.dyn_py_calls = 0; @@ -4265,6 +4286,7 @@ unsafe fn try_native_call( Box::new(CallCtx { interp: ctx.interp, callees: nc.callees.clone(), + cf: StdRc::as_ptr(&nc.cf), guard_snapshot: nc.snap.clone(), globals: nc.func.globals.clone(), builtins: nc.func.builtins.clone(), @@ -4609,19 +4631,6 @@ fn finish_deopted_callee( njf: &JitFrame, raised: Option, ) -> Result { - let code = &nc.code; - let n_real = code.varnames.len(); - let mut locals_v: Vec = Vec::with_capacity(n_real); - for slot in 0..n_real { - match nc.cf.local_types.get(slot).copied().flatten() { - Some(ty) => locals_v.push(unpack_ty( - locals_buf.get(slot).copied().unwrap_or(0), - ty, - &nctx.pins, - )), - None => locals_v.push(Object::Unbound), - } - } let entry = CompiledEntry { cf: nc.cf.clone(), guard_snapshot: nc.snap.clone(), @@ -4632,17 +4641,49 @@ fn finish_deopted_callee( math: nc.math.clone(), native: None, method_native: None, - // Synthetic entry, only for the stack rebuild below; `0` is - // never a real compile id, so nothing can park against it. + // Synthetic entry, only for the stack rebuild; `0` is never a + // real compile id, so nothing can park against it. compile_id: 0, }; + finish_deopted( + interp, &nc.code, &nc.func, &entry, nctx, locals_buf, spill, tags, njf, raised, + ) +} + +/// [`finish_deopted_callee`] for an activation described by `entry` +/// (the callee's own tables), `code` and `func`. +#[allow(clippy::too_many_arguments)] +fn finish_deopted( + interp: &mut super::Interpreter, + code: &Rc, + func: &PyFunction, + entry: &CompiledEntry, + nctx: &mut CallCtx, + locals_buf: &[u64], + spill: &[u64], + tags: &[u32], + njf: &JitFrame, + raised: Option, +) -> Result { + let n_real = code.varnames.len(); + let mut locals_v: Vec = Vec::with_capacity(n_real); + for slot in 0..n_real { + match entry.cf.local_types.get(slot).copied().flatten() { + Some(ty) => locals_v.push(unpack_ty( + locals_buf.get(slot).copied().unwrap_or(0), + ty, + &nctx.pins, + )), + None => locals_v.push(Object::Unbound), + } + } let mut frame = super::Frame { code: code.clone(), locals: Rc::new(GilRefCell::new(locals_v)), cells: crate::object::empty_cells(), stack: Vec::new(), - globals: nc.func.globals.clone(), - builtins: nc.func.builtins.clone(), + globals: func.globals.clone(), + builtins: func.builtins.clone(), builtins_obj: None, class_namespace: None, class_namespace_obj: None, @@ -4667,7 +4708,7 @@ fn finish_deopted_callee( nctx.parked.take() }; rebuild_stack( - interp, &mut frame, &entry, locals_buf, spill, tags, njf, &nctx.pins, parked, + interp, &mut frame, entry, locals_buf, spill, tags, njf, &nctx.pins, parked, ); if raised.is_some() { // As though the raising CALL just executed: pc points past it @@ -5077,6 +5118,165 @@ unsafe fn pure_leaf_call( Some(deliver_call_result(jf, ctx, v, expect_tag)) } +/// The direct self-call enter helper (see `weavepy_jit::SelfEnterHelper`): +/// the per-call work `try_native_call` does for a callee, reduced to what +/// a pin-free activation of the caller's own code needs — a GIL +/// checkpoint, the observer gate, and the recursion tick (checked +/// against the limit, and against the native stack's headroom every few +/// levels: the ordinary path grows the stack, this one cannot). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_self_enter(frame: *mut JitFrame) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + // SAFETY: the `&mut Interpreter` that entered native code is dormant + // while the helper runs. + let interp = unsafe { &mut *ctx.interp }; + interp.gil_countdown = interp.gil_countdown.wrapping_sub(1); + if interp.gil_countdown == 0 { + interp.gil_countdown = crate::gil::GIL_CHECK_INTERVAL; + crate::gil::yield_checkpoint(); + } + if crate::hot_gates::load() != 0 || crate::trace::any_observers_active() { + return 1; + } + // SAFETY: this thread's own depth cell (see `CallCtx::depth_cell`). + let depth = unsafe { &*ctx.depth_cell }; + let n = depth.get() + 1; + if n > crate::recursion::recursion_limit() + || (n % 8 == 0 && stacker::remaining_stack().is_some_and(|r| r < 256 * 1024)) + { + return 1; + } + depth.set(n); + 0 +} + +/// Release [`wpjit_self_enter`]'s recursion tick. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_self_exit(frame: *mut JitFrame) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &*frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &*jf.ctx.cast::() }; + // SAFETY: as in `wpjit_self_enter`. + let depth = unsafe { &*ctx.depth_cell }; + depth.set(depth.get().saturating_sub(1)); + 0 +} + +/// Finish a direct self call whose callee did not return (see +/// `weavepy_jit::SelfSlowHelper`): exactly `try_native_call`'s deopt and +/// raise handling, the callee's frame and buffers being the ones on the +/// caller's native stack and its context the caller's own (a pin-free +/// activation leaves the shared table empty). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]; `callee` is the live callee frame. +unsafe extern "C" fn wpjit_self_slow( + frame: *mut JitFrame, + callee: *mut JitFrame, + status: i64, + token: i64, + expect_tag: i64, +) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + // SAFETY: as in `wpjit_self_enter`. + let interp = unsafe { &mut *ctx.interp }; + // SAFETY: the caller's contract. + let cjf = unsafe { &*callee }; + let raised = (status == JitStatus::Raised as i64).then(|| { + ctx.raised.take().unwrap_or_else(|| { + RuntimeError::Internal("JIT Raised exit without a parked exception".to_owned()) + }) + }); + let Some((Object::Function(pf), code)) = ctx.callees.get(token as usize).cloned() else { + // Unreachable: the lowering only emits direct calls for tokens + // naming this very function. + ctx.raised = Some(raised.unwrap_or_else(|| { + RuntimeError::Internal("direct self call without its callee".to_owned()) + })); + return CallStatus::Raised as i64; + }; + if status == JitStatus::Deopt as i64 { + native_stat(|s| s.deopts.set(s.deopts.get() + 1)); + let key = Rc::as_ptr(&code).cast::(); + JIT.with(|cell| { + if let Some(ce) = cell.borrow_mut().cache.get_mut(&key) { + ce.deopts += 1; + if ce.deopts >= DEOPT_BUDGET { + ce.tier = Tier::NotJitable; + code.jit_hint.mark_not_jitable(); + } + } + }); + } + // The callee runs the caller's own compilation, whose layout the + // shared context holds (the tier cache may have just retired it). + // SAFETY: `ctx.cf` came from a live `StdRc` its entry still owns. + let cf = unsafe { + StdRc::increment_strong_count(ctx.cf); + StdRc::from_raw(ctx.cf) + }; + let entry = CompiledEntry { + cf, + guard_snapshot: ctx.guard_snapshot.clone(), + callees: ctx.callees.clone(), + obj_globals: ctx.obj_globals.clone(), + attr_guards: ctx.attr_guards.clone(), + methods: ctx.methods.clone(), + math: ctx.math.clone(), + native: None, + method_native: None, + compile_id: 0, + }; + ctx.dirty = true; + // SAFETY: the callee frame's buffers are live on the caller's stack, + // sized by its own compiled frame. + let (locals, spill, tags) = unsafe { + ( + std::slice::from_raw_parts(cjf.locals, cjf.n_locals as usize), + std::slice::from_raw_parts(cjf.stack_spill, cjf.stack_cap as usize), + std::slice::from_raw_parts(cjf.stack_tags, cjf.stack_cap as usize), + ) + }; + match finish_deopted( + interp, &code, &pf, &entry, ctx, locals, spill, tags, cjf, raised, + ) { + Err(e) => { + ctx.raised = Some(e); + CallStatus::Raised as i64 + } + Ok(v) => { + // Python ran for the continuation: the caller's burned-in + // resolutions must still hold for it to continue natively. + if !guards_hold( + interp, + &ctx.globals, + &ctx.builtins, + &ctx.guard_snapshot, + &ctx.callees, + &ctx.math, + ) { + ctx.parked = Some(v); + return CallStatus::Boxed as i64; + } + deliver_call_result(jf, ctx, v, expect_tag as u32) + } + } +} + /// The generic-call backoff for native-to-native entries (the framed /// entries' twin lives in [`note_native_exit`]): a compiled callee whose /// activations average [`CALLEE_ROUNDTRIP_RETIRE_RATIO`] or more @@ -8461,12 +8661,14 @@ unsafe fn call_dyn_impl( .then(|| unsafe { dyn_kw_site_bind(jf, ctx, &callee, argc, kwc, names) }) .flatten() { - note_generic_dyn_call(ctx); + // Not charged against the native driver: the interpreter's own + // `CALL_KW` binds through this same permutation and activation, + // so tier-1 would not run the call any cheaper. ctx.dirty = true; let called = call_with_activation_shell(interp, ctx, jf, |i| i.run_py_exact_nofree(&f, locals)); // SAFETY: as above. - return unsafe { dyn_call_result(jf, ctx, called, false, int_result) }; + return unsafe { dyn_call_result(jf, ctx, called, false, false, int_result) }; } let n = (argc + kwc) as usize; let mut args: Vec = Vec::with_capacity(n); @@ -8498,8 +8700,13 @@ unsafe fn call_dyn_impl( } } // Arbitrary Python runs on behalf of this activation (RFC 0067 - // WS1's dirtiness discipline). - note_generic_dyn_call(ctx); + // WS1's dirtiness discipline). A keyword call pays the generic + // binder in tier-1 too, so only positional calls count against the + // native driver. + let charged = kwc == 0; + if charged { + note_generic_dyn_call(ctx); + } ctx.dirty = true; let native_callee = matches!(&callee, Object::Builtin(_)) || matches!(&callee, Object::BoundMethod(bm) if matches!(bm.function, Object::Builtin(_))); @@ -8507,12 +8714,14 @@ unsafe fn call_dyn_impl( i.call_object_with_globals(&callee, &args, &kwargs, &ctx.globals) }); // SAFETY: as above. - unsafe { dyn_call_result(jf, ctx, called, native_callee, int_result) } + unsafe { dyn_call_result(jf, ctx, called, native_callee, charged, int_result) } } /// `call_dyn_impl`'s result protocol: park a raise, or deliver the /// result unboxed (`int_result`) or pinned, parking it (`Boxed`) when a -/// round-trip charge or an invalidated guard ends the activation. +/// round-trip charge or an invalidated guard ends the activation. An +/// uncharged call (one tier-1 would not run any cheaper) leaves the +/// native driver's call density alone. /// /// # Safety /// @@ -8522,6 +8731,7 @@ unsafe fn dyn_call_result( ctx: &mut CallCtx, called: Result, native_callee: bool, + charged: bool, int_result: bool, ) -> i64 { // SAFETY: the `&mut Interpreter` that entered native code is @@ -8533,10 +8743,10 @@ unsafe fn dyn_call_result( CallStatus::Raised as i64 } Ok(v) => { - if !native_callee { + if charged && !native_callee { ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } - if charge_roundtrip(ctx) || (native_callee && charge_native_roundtrip(ctx)) { + if charge_roundtrip(ctx) || (charged && native_callee && charge_native_roundtrip(ctx)) { ctx.parked = Some(v); return CallStatus::Boxed as i64; } @@ -9659,6 +9869,7 @@ pub(crate) fn try_call_native_direct( let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.art.callees.clone(), + cf: StdRc::as_ptr(&entry.art.cf), guard_snapshot: entry.art.snap.clone(), globals: f.globals.clone(), builtins: f.builtins.clone(), @@ -10461,6 +10672,7 @@ fn enter_compiled( let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.callees.clone(), + cf: StdRc::as_ptr(&entry.cf), guard_snapshot: entry.guard_snapshot.clone(), globals: frame.globals.clone(), builtins: frame.builtins.clone(), @@ -11348,6 +11560,7 @@ fn resume_parked(interp: &mut super::Interpreter, frame: &mut super::Frame) -> J let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.callees.clone(), + cf: StdRc::as_ptr(&entry.cf), guard_snapshot: entry.guard_snapshot.clone(), globals: frame.globals.clone(), builtins: frame.builtins.clone(), From 6685b59206206ac9a80eb85e4c4479b70a228baa Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 03:28:01 -0700 Subject: [PATCH 08/65] perf: cheaper method lookups, keyword calls, and datetime natives - Inline-cache reads on x86_64 without AVX use the epoch snapshot's plain loads instead of portable-atomic's run-time dispatched 128-bit load (an out-of-line indirect call on every cache read). - Dicts keep a one-word summary of their `str` keys' hash classes, built once per key layout, so the per-call check that an instance dict doesn't shadow a method usually skips the probe. - A frame's small, untracked, uniquely held list, tuple or dict of atomic values (a `**kwargs` dict) is released by a plain drop instead of the prompt-reap cascade. - Keyword calls from native code bind into pooled vectors and run the callee as a lean activation (a pure leaf evaluates frameless), as the interpreter's own `CALL_KW` does, instead of a full framed run. - `datetime` natives read fields straight from the shared slot layout, shift with 64-bit arithmetic, and build exact `date`/`datetime` instances from int fields without running the Python `__new__`. call_overhead: 2280M -> 1805M instructions; spectral_norm, dict_ops, deltablue and datetime_ops improve 1-10%. --- crates/weavepy-compiler/src/cache_snapshot.rs | 11 +- crates/weavepy-vm/src/lib.rs | 108 ++++++++- crates/weavepy-vm/src/object.rs | 45 ++++ .../weavepy-vm/src/stdlib/datetime_native.rs | 227 ++++++++++++++---- crates/weavepy-vm/src/tier2.rs | 11 +- tests/regrtest/test_datetime_fields.py | 5 +- 6 files changed, 345 insertions(+), 62 deletions(-) diff --git a/crates/weavepy-compiler/src/cache_snapshot.rs b/crates/weavepy-compiler/src/cache_snapshot.rs index 5b4c74fc..45966360 100644 --- a/crates/weavepy-compiler/src/cache_snapshot.rs +++ b/crates/weavepy-compiler/src/cache_snapshot.rs @@ -97,8 +97,15 @@ impl SelectStorage for Select { // Avoid portable-atomic's global-lock fallback. An inherited global lock can // belong to a vanished writer after fork; advisory caches can simply miss. -pub(crate) type CacheSnapshot = - as SelectStorage>::Storage; +// +// On x86_64 without AVX enabled at compile time, portable-atomic picks its +// 128-bit load at run time, so every cache read is an out-of-line indirect +// call. The epoch snapshot's loads are plain inline moves there, and cache +// reads vastly outnumber cache writes. +const NATIVE_SNAPSHOT: bool = portable_atomic::AtomicU128::is_always_lock_free() + && !(cfg!(target_arch = "x86_64") && !cfg!(target_feature = "avx")); + +pub(crate) type CacheSnapshot = as SelectStorage>::Storage; #[cfg(test)] mod tests { diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index deb4272f..32c0388f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -3312,6 +3312,7 @@ impl Interpreter { !(Self::local_needs_prompt_reap(v) || matches!(v, Object::Function(_))) || escaped(v) || matches!(v, Object::Instance(i) if i.dies_by_plain_drop()) + || Self::atomic_container_dies_plainly(v) }) { for v in locals { gc_trace::note_dropped(v); @@ -3792,6 +3793,42 @@ impl Interpreter { /// generators/coroutines/async-generators (their `close()` delivers /// `GeneratorExit`). Kept deliberately narrow so the hot return path of /// scalar-only frames pays a single cheap `matches!` per local. + /// A small untracked list, tuple or dict of atomic values held only + /// here: its release frees nothing a reap would visit (no finalizer, + /// weakref or collector handle can hang off it or its values). + fn atomic_container_dies_plainly(o: &Object) -> bool { + const MAX: usize = 16; + fn atomic(o: &Object) -> bool { + matches!( + o, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Str(_) + ) + } + let (sc, id) = match o { + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + Object::Tuple(t) => ( + ThinArc::strong_count(t), + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ), + _ => return false, + }; + if sc != 1 || gc_trace::maybe_tracked(id) || crate::weakref_registry::may_have_weakrefs(id) + { + return false; + } + match o { + Object::List(l) => l + .try_borrow() + .is_ok_and(|v| v.len() <= MAX && v.iter().all(atomic)), + Object::Dict(d) => d + .try_borrow() + .is_ok_and(|d| d.len() <= MAX && d.iter().all(|(k, v)| atomic(&k.0) && atomic(v))), + Object::Tuple(t) => t.len() <= MAX && t.iter().all(atomic), + _ => false, + } + } + fn local_needs_prompt_reap(o: &Object) -> bool { matches!( o, @@ -13344,7 +13381,9 @@ impl Interpreter { else { break Some(CoreExit::Helper); }; - if d.contains_key(&probe) || probe.saw_exotic() { + if d.may_hold_str_hash(probe.hash) + && (d.contains_key(&probe) || probe.saw_exotic()) + { break Some(CoreExit::Helper); } } @@ -14452,7 +14491,8 @@ impl Interpreter { let d = unsafe { dict.peek() }?; if !d.is_empty() { let probe = code_name_leaf_probe(code, name_idx)?; - if d.contains_key(&probe) || probe.saw_exotic() { + if d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) + { return None; } } @@ -19022,7 +19062,8 @@ impl Interpreter { let d = dict.try_borrow().ok()?; if !d.is_empty() { let probe = code_name_leaf_probe(code, name_idx)?; - if d.contains_key(&probe) || probe.saw_exotic() { + if d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) + { return None; } } @@ -37663,8 +37704,9 @@ impl Interpreter { let shadowed = inst.dict.get().is_some_and(|dict| { let d = dict.borrow(); !d.is_empty() - && code_name_key(&frame.code, name_idx) - .is_none_or(|k| d.contains_key(&k)) + && code_name_key(&frame.code, name_idx).is_none_or(|k| { + d.may_hold_str_hash(k.hash) && d.contains_key(&k) + }) }); if !shadowed { let slot = code_method_slot(&frame.code, cache_pc); @@ -37718,7 +37760,9 @@ impl Interpreter { match code_name_key(&frame.code, name_idx) { Some(key) => { let d = dict.borrow(); - !d.is_empty() && d.contains_key(&key) + !d.is_empty() + && d.may_hold_str_hash(key.hash) + && d.contains_key(&key) } None => true, } @@ -45876,6 +45920,12 @@ impl Interpreter { args: &[Object], kwargs: &[(String, Object)], ) -> Result { + // A natively served class's common constructor shapes. + if cls.native_kind.get() != 0 { + if let Some(r) = crate::stdlib::datetime_native::construct(&cls, args, kwargs) { + return r; + } + } if kwargs.is_empty() && !cls.flags.is_builtin { if let Some(r) = self.instantiate_plain_lean(&cls, args) { return r; @@ -48142,6 +48192,52 @@ impl Interpreter { self.run_py_exact_nofree_with(f, f.code(), args) } + /// Run plain function `f` on `locals` already bound to its parameter + /// slots (sized to its locals): as a lean activation when the + /// dispatch loop is quiet and the code allows one, as + /// [`Self::run_py_exact_nofree`] otherwise. + pub(crate) fn run_py_bound( + &mut self, + f: &Rc, + mut locals: Vec, + ) -> Result { + let code = f.code(); + if let Some(snap_gen) = self.lean_snapshot() { + // A pure leaf evaluates frameless, as the core loop's call does. + let nargs = code.arg_count as usize; + if nargs <= 8 + && locals.len() >= nargs + && code_is_pure_leaf(&code) + && pure_leaf_warm(&code) + && crate::recursion::current_depth() < crate::recursion::recursion_limit() + { + let mut ptrs = [std::ptr::null::(); 8]; + for (p, v) in ptrs.iter_mut().zip(&locals[..nargs]) { + *p = v; + } + if let Some(v) = self.pure_leaf_eval::(&code, f, &ptrs[..nargs]) { + self.recycle_scratch(locals); + return Ok(v); + } + } + if Self::lean_code_ok(&code) && locals.len() <= code.varnames.len() { + if let Some(cells) = f.lean_cells_ref(&code) { + let cells: *const Rc>>> = cells; + let n = code.varnames.len(); + let locals_rc = self.pooled_locals_from_args(&mut locals, n); + self.recycle_scratch(locals); + // SAFETY: `f` (borrowed for this whole call) owns the + // cells handle and outlives the frame. + let mut callee = unsafe { self.lean_frame(f, code, locals_rc, &*cells) }; + let result = self.run_frame_lean(&mut callee, snap_gen); + self.retire_lean_frame(callee); + return result; + } + } + } + self.run_py_exact_nofree_with(f, code, locals) + } + /// [`Self::run_py_exact_nofree`] for a caller that already holds the /// function's current code object (its inline-cache guard read it). fn run_py_exact_nofree_with( diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index 5b1df89c..db103ba6 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -4059,6 +4059,10 @@ pub struct DictData { /// value (every `DerefMut`) tracks the owner first; the owner clears /// it when it is tracked by other means or dies. deferred_owner: std::sync::atomic::AtomicUsize, + /// A one-bit-per-hash-class summary of the `str` keys (see + /// [`Self::may_hold_str_hash`]); `0` until built, and reset by every + /// access that stamps. + key_filter: std::sync::atomic::AtomicU64, } impl Clone for DictData { @@ -4067,6 +4071,7 @@ impl Clone for DictData { map: self.map.clone(), stamp: self.stamp, deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } @@ -4078,6 +4083,7 @@ impl DictData { map: DictMap::with_capacity_and_hasher(n, h), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4088,6 +4094,7 @@ impl DictData { map: DictMap::default(), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(owner), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4100,6 +4107,7 @@ impl DictData { map: DictMap::with_capacity_and_hasher(n, crate::fasthash::FxBuildHasher), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(owner), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4133,6 +4141,7 @@ impl DictData { #[inline] pub fn map_mut_atomic_store(&mut self) -> &mut DictMap { self.stamp = next_dict_stamp(); + *self.key_filter.get_mut() = 0; &mut self.map } @@ -4149,6 +4158,39 @@ impl DictData { self.stamp } + /// Whether a `str` key with Python hash `hash` may be present: `false` + /// proves it absent. The summary is built once per key layout (value + /// stores in place leave it standing, so an instance dict's summary + /// survives its attribute updates). Bit 63 marks a built summary, so + /// the hash class it shares always answers "maybe"; a non-`str` key + /// makes every hash possible. + #[inline] + pub fn may_hold_str_hash(&self, hash: i64) -> bool { + let mut bits = self.key_filter.load(std::sync::atomic::Ordering::Relaxed); + if bits == 0 { + bits = self.build_key_filter(); + } + bits & (1u64 << (hash as u64 & 63)) != 0 + } + + #[cold] + #[inline(never)] + fn build_key_filter(&self) -> u64 { + let mut bits = 1u64 << 63; + for k in self.map.keys() { + match &k.0 { + Object::Str(s) => bits |= 1u64 << (SharedStr::hash_cached(s) as u64 & 63), + _ => { + bits = u64::MAX; + break; + } + } + } + self.key_filter + .store(bits, std::sync::atomic::Ordering::Relaxed); + bits + } + /// Mutable access for replacing an existing key's value in place: the /// keys do not move, and only key layout is what the stamp's readers /// (the global/builtin load caches) watch, so it stays put. A @@ -4178,6 +4220,7 @@ impl Default for DictData { map: DictMap::default(), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } @@ -4198,6 +4241,7 @@ impl std::ops::DerefMut for DictData { self.track_deferred_owner(); } self.stamp = next_dict_stamp(); + *self.key_filter.get_mut() = 0; &mut self.map } } @@ -4214,6 +4258,7 @@ impl From for DictData { map, stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } diff --git a/crates/weavepy-vm/src/stdlib/datetime_native.rs b/crates/weavepy-vm/src/stdlib/datetime_native.rs index b8fc8dae..317014ab 100644 --- a/crates/weavepy-vm/src/stdlib/datetime_native.rs +++ b/crates/weavepy-vm/src/stdlib/datetime_native.rs @@ -182,6 +182,8 @@ const SHORTCUT_NAMES: &[(u8, &str)] = &[ (KIND_DATE, "__le__"), (KIND_DATE, "__gt__"), (KIND_DATE, "__ge__"), + (KIND_DATE, "__new__"), + (KIND_DATE, "__init__"), (KIND_DATE, "year"), (KIND_DATE, "month"), (KIND_DATE, "day"), @@ -193,6 +195,8 @@ const SHORTCUT_NAMES: &[(u8, &str)] = &[ (KIND_DATETIME, "__le__"), (KIND_DATETIME, "__gt__"), (KIND_DATETIME, "__ge__"), + (KIND_DATETIME, "__new__"), + (KIND_DATETIME, "__init__"), (KIND_DATETIME, "year"), (KIND_DATETIME, "month"), (KIND_DATETIME, "day"), @@ -273,8 +277,19 @@ fn as_int(o: &Object) -> Option { } } -fn td_fields(i: &PyInstance, n: &Names) -> Option<(i64, i64, i64)> { +// The field readers take a natively built instance's values straight +// from the shared layout; any other storage is read by name. + +fn td_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s.values_for_layout(&st.td_layout) { + return Some(( + as_int(&v[TD_DAYS])?, + as_int(&v[TD_SECONDS])?, + as_int(&v[TD_US])?, + )); + } Some(( as_int(s.get_hinted(TD_DAYS, &n.days)?)?, as_int(s.get_hinted(TD_SECONDS, &n.seconds)?)?, @@ -286,8 +301,19 @@ fn td_us(f: (i64, i64, i64)) -> i128 { (i128::from(f.0) * 86_400 + i128::from(f.1)) * 1_000_000 + i128::from(f.2) } -fn date_fields(i: &PyInstance, n: &Names) -> Option<(i64, i64, i64)> { +fn date_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s + .values_for_layout(&st.date_layout) + .or_else(|| s.values_for_layout(&st.dt_layout)) + { + return Some(( + as_int(&v[D_YEAR])?, + as_int(&v[D_MONTH])?, + as_int(&v[D_DAY])?, + )); + } Some(( as_int(s.get_hinted(D_YEAR, &n.year)?)?, as_int(s.get_hinted(D_MONTH, &n.month)?)?, @@ -307,8 +333,22 @@ struct Dt { fold: i64, } -fn dt_fields(i: &PyInstance, n: &Names) -> Option
{ +fn dt_fields(i: &PyInstance, st: &State) -> Option
{ + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s.values_for_layout(&st.dt_layout) { + return Some(Dt { + y: as_int(&v[D_YEAR])?, + m: as_int(&v[D_MONTH])?, + d: as_int(&v[D_DAY])?, + hh: as_int(&v[DT_HOUR])?, + mm: as_int(&v[DT_MINUTE])?, + ss: as_int(&v[DT_SECOND])?, + us: as_int(&v[DT_US])?, + tz: v[DT_TZINFO].clone(), + fold: as_int(&v[DT_FOLD])?, + }); + } Some(Dt { y: as_int(s.get_hinted(D_YEAR, &n.year)?)?, m: as_int(s.get_hinted(D_MONTH, &n.month)?)?, @@ -329,7 +369,8 @@ enum Tz { Fixed(i128), } -fn tz_of(tz: &Object, n: &Names) -> Option { +fn tz_of(tz: &Object, st: &State) -> Option { + let n = &st.names; match tz { Object::None => Some(Tz::Naive), o => { @@ -339,7 +380,7 @@ fn tz_of(tz: &Object, n: &Names) -> Option { s.get_hinted(TZ_OFFSET, &n.offset)?.clone() }; let td = inst(&off, KIND_TIMEDELTA)?; - Some(Tz::Fixed(td_us(td_fields(td, n)?))) + Some(Tz::Fixed(td_us(td_fields(td, st)?))) } } } @@ -423,11 +464,88 @@ fn instance_fixed( Object::Instance(Rc::new(i)) } +/// `date(y, m, d)` and `datetime(y, m, d[, hh[, mm[, ss[, us[, tz]]]]])` +/// (`tzinfo=` by keyword too) of the exact classes, with in-range `int` +/// fields and a naive or fixed-offset zone: the instance the replaced +/// `__new__` builds, built natively. `None` for every other shape, which +/// the Python constructor serves (and diagnoses). +pub(crate) fn construct( + cls: &Rc, + args: &[Object], + kwargs: &[(String, Object)], +) -> Option> { + let st = state_of_cls(cls)?; + let kind = cls.native_kind.get(); + let exact = match kind { + KIND_DATE => st.date.upgrade(), + KIND_DATETIME => st.datetime.upgrade(), + _ => None, + }; + if !exact.is_some_and(|c| Rc::ptr_eq(&c, cls)) || !verified(st, cls, kind) { + return None; + } + let int = |o: &Object| match o { + Object::Int(v) => Some(*v), + _ => None, + }; + if kind == KIND_DATE { + let ([y, m, d], true) = (args, kwargs.is_empty()) else { + return None; + }; + let (y, m, d) = (int(y)?, int(m)?, int(d)?); + return valid_date(y, m, d) + .then(|| new_date(st, y, m, d)) + .flatten() + .map(Ok); + } + if !(3..=8).contains(&args.len()) { + return None; + } + let mut tz = args.get(7).cloned().unwrap_or(Object::None); + for (name, v) in kwargs { + if name != "tzinfo" || args.len() == 8 { + return None; + } + tz = v.clone(); + } + let field = |ix: usize| args.get(ix).map_or(Some(0), int); + let f = Dt { + y: int(&args[0])?, + m: int(&args[1])?, + d: int(&args[2])?, + hh: field(3)?, + mm: field(4)?, + ss: field(5)?, + us: field(6)?, + tz, + fold: 0, + }; + let ok = valid_date(f.y, f.m, f.d) + && (0..24).contains(&f.hh) + && (0..60).contains(&f.mm) + && (0..60).contains(&f.ss) + && (0..1_000_000).contains(&f.us); + if !ok { + return None; + } + tz_of(&f.tz, st)?; + new_dt(st, &f).map(Ok) +} + /// A normalized exact `timedelta` from unnormalized components. fn new_td(st: &State, d: i128, s: i128, us: i128) -> Option> { let total = d * US_PER_DAY + s * 1_000_000 + us; - let days = total.div_euclid(US_PER_DAY); - let rest = total.rem_euclid(US_PER_DAY); + // 64-bit division when it fits (128-bit division is a library call). + let (days, rest) = match i64::try_from(total) { + Ok(t) => { + let per_day = US_PER_DAY as i64; + ( + i128::from(t.div_euclid(per_day)), + i128::from(t.rem_euclid(per_day)), + ) + } + Err(_) => (total.div_euclid(US_PER_DAY), total.rem_euclid(US_PER_DAY)), + }; if days.abs() > MAX_DAYS { return Some(Err(overflow_error(format!( "days={days}; must have magnitude <= 999999999" @@ -492,17 +610,28 @@ fn new_dt(st: &State, f: &Dt) -> Option { /// `datetime` fields shifted by `delta_us` (fold cleared, `tzinfo` /// kept), or `None` past the representable range. fn dt_shift(f: &Dt, delta_us: i128) -> Option
{ - let base = (i128::from(ymd2ord(f.y, f.m, f.d)) * 86_400 - + i128::from(f.hh * 3600 + f.mm * 60 + f.ss)) - * 1_000_000 - + i128::from(f.us); - let total = base + delta_us; - let ord = total.div_euclid(US_PER_DAY); - if !(1..=i128::from(MAX_ORDINAL)).contains(&ord) { + // Every representable datetime is under 2^59 microseconds, so a + // delta that fits no `i64` lands out of range (the Python path + // raises); the rest is 64-bit arithmetic over in-range fields. + let in_range = (1..=9999).contains(&f.y) + && valid_date(f.y, f.m, f.d) + && (0..24).contains(&f.hh) + && (0..60).contains(&f.mm) + && (0..60).contains(&f.ss) + && (0..1_000_000).contains(&f.us); + if !in_range { + return None; + } + let base = + (ymd2ord(f.y, f.m, f.d) * 86_400 + f.hh * 3600 + f.mm * 60 + f.ss) * 1_000_000 + f.us; + let total = base.checked_add(i64::try_from(delta_us).ok()?)?; + let per_day = US_PER_DAY as i64; + let ord = total.div_euclid(per_day); + if !(1..=MAX_ORDINAL).contains(&ord) { return None; } - let rest = total.rem_euclid(US_PER_DAY); - let (y, m, d) = ord2ymd(ord as i64); + let rest = total.rem_euclid(per_day); + let (y, m, d) = ord2ymd(ord); let secs = (rest / 1_000_000) as i64; Some(Dt { y, @@ -533,8 +662,8 @@ type Fast = fn(&[Object]) -> Option>; fn td_add(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) + i128::from(q.0), @@ -546,8 +675,8 @@ fn td_add(a: &[Object]) -> Option> { fn td_sub(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) - i128::from(q.0), @@ -559,7 +688,7 @@ fn td_sub(a: &[Object]) -> Option> { fn td_neg(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; new_td(st, -i128::from(p.0), -i128::from(p.1), -i128::from(p.2)) } @@ -567,7 +696,7 @@ fn td_mul(a: &[Object]) -> Option> { let [x, k] = a else { return None }; let k = i128::from(as_int(k)?); let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) * k, @@ -579,15 +708,15 @@ fn td_mul(a: &[Object]) -> Option> { fn td_bool(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; Some(Ok(Object::Bool(p != (0, 0, 0)))) } fn td_cmp(a: &[Object]) -> Option { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; Some(p.cmp(&q)) } @@ -617,7 +746,7 @@ fn date_toordinal(a: &[Object]) -> Option> { }, _ => return None, }; - let (y, m, d) = date_fields(i, &st.names)?; + let (y, m, d) = date_fields(i, st)?; if !valid_date(y, m, d) { return None; } @@ -645,8 +774,8 @@ fn date_isoweekday(a: &[Object]) -> Option> { fn date_shift(a: &[Object], sign: i64) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let (yy, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let (days, _, _) = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let (yy, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; + let (days, _, _) = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; if !valid_date(yy, m, d) { return None; } @@ -668,8 +797,8 @@ fn date_sub(a: &[Object]) -> Option> { return date_shift(a, -1); } let st = state_of(x)?; - let (y1, m1, d1) = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let (y2, m2, d2) = date_fields(inst(y, KIND_DATE)?, &st.names)?; + let (y1, m1, d1) = date_fields(inst(x, KIND_DATE)?, st)?; + let (y2, m2, d2) = date_fields(inst(y, KIND_DATE)?, st)?; if !valid_date(y1, m1, d1) || !valid_date(y2, m2, d2) { return None; } @@ -684,8 +813,8 @@ fn date_sub(a: &[Object]) -> Option> { fn date_cmp(a: &[Object]) -> Option { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let q = date_fields(inst(y, KIND_DATE)?, &st.names)?; + let p = date_fields(inst(x, KIND_DATE)?, st)?; + let q = date_fields(inst(y, KIND_DATE)?, st)?; Some(p.cmp(&q)) } @@ -698,7 +827,7 @@ cmp_fast!(date_ge, date_cmp, |o| o.is_ge()); fn date_isoformat(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; if !valid_date(y, m, d) { return None; } @@ -709,7 +838,7 @@ fn date_isoformat(a: &[Object]) -> Option> { fn date_replace(a: &[Object]) -> Option> { let (x, rest) = a.split_first()?; let st = state_of(x)?; - let (mut y, mut m, mut d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (mut y, mut m, mut d) = date_fields(inst(x, KIND_DATE)?, st)?; if rest.len() > 3 { return None; } @@ -733,8 +862,8 @@ fn date_replace(a: &[Object]) -> Option> { fn dt_add(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; - let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?); + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; + let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, st)?); if !valid_date(f.y, f.m, f.d) { return None; } @@ -747,26 +876,26 @@ fn dt_add(a: &[Object]) -> Option> { fn dt_sub(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } match kind_of(y) { KIND_TIMEDELTA => { - let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?); + let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, st)?); match dt_shift(&f, -delta) { Some(g) => Some(Ok(new_dt(st, &g)?)), None => Some(Err(overflow_error("date value out of range"))), } } KIND_DATETIME => { - let g = dt_fields(inst(y, KIND_DATETIME)?, &st.names)?; + let g = dt_fields(inst(y, KIND_DATETIME)?, st)?; if !valid_date(g.y, g.m, g.d) { return None; } let mut diff = dt_us(&f) - dt_us(&g); if !f.tz.is_same(&g.tz) { - match (tz_of(&f.tz, &st.names)?, tz_of(&g.tz, &st.names)?) { + match (tz_of(&f.tz, st)?, tz_of(&g.tz, st)?) { (Tz::Naive, Tz::Naive) => {} (Tz::Fixed(a), Tz::Fixed(b)) => diff += b - a, // The mixed case raises in the Python code. @@ -784,15 +913,15 @@ fn dt_sub(a: &[Object]) -> Option> { fn dt_cmp_raw(a: &[Object]) -> Option<(std::cmp::Ordering, bool)> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; - let g = dt_fields(inst(y, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; + let g = dt_fields(inst(y, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) || !valid_date(g.y, g.m, g.d) { return None; } if f.tz.is_same(&g.tz) { return Some((dt_us(&f).cmp(&dt_us(&g)), true)); } - match (tz_of(&f.tz, &st.names)?, tz_of(&g.tz, &st.names)?) { + match (tz_of(&f.tz, st)?, tz_of(&g.tz, st)?) { (Tz::Naive, Tz::Naive) => Some((dt_us(&f).cmp(&dt_us(&g)), true)), (Tz::Fixed(a), Tz::Fixed(b)) => Some(((dt_us(&f) - a).cmp(&(dt_us(&g) - b)), true)), _ => Some((std::cmp::Ordering::Equal, false)), @@ -826,7 +955,7 @@ dt_order!(dt_ge, |o| o.is_ge()); fn dt_date(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(y, m, d) { return None; } @@ -881,11 +1010,11 @@ fn dt_isoformat(a: &[Object]) -> Option> { return None; } let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } - let tz = tz_of(&f.tz, &st.names)?; + let tz = tz_of(&f.tz, st)?; let mut out = String::with_capacity(32); let _ = write!(out, "{:04}-{:02}-{:02}{sep}", f.y, f.m, f.d); match spec { @@ -1010,11 +1139,11 @@ fn dt_strftime(a: &[Object]) -> Option> { return None; }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } - let tz = tz_of(&f.tz, &st.names)?; + let tz = tz_of(&f.tz, st)?; let s = strftime_numeric( f.y, f.m, @@ -1031,7 +1160,7 @@ fn date_strftime(a: &[Object]) -> Option> { return None; }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; if !valid_date(y, m, d) { return None; } diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index e7228082..57083910 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -8665,8 +8665,7 @@ unsafe fn call_dyn_impl( // `CALL_KW` binds through this same permutation and activation, // so tier-1 would not run the call any cheaper. ctx.dirty = true; - let called = - call_with_activation_shell(interp, ctx, jf, |i| i.run_py_exact_nofree(&f, locals)); + let called = call_with_activation_shell(interp, ctx, jf, |i| i.run_py_bound(&f, locals)); // SAFETY: as above. return unsafe { dyn_call_result(jf, ctx, called, false, false, int_result) }; } @@ -8827,7 +8826,10 @@ unsafe fn dyn_kw_site_bind( } let (fcode, covered) = crate::Interpreter::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; - let mut staged: Vec = Vec::with_capacity(eff_argc + kwc); + // SAFETY: the `&mut Interpreter` that entered native code is dormant + // while the helper runs; only its vector pools are used here. + let interp = unsafe { &*ctx.interp }; + let mut staged = interp.pooled_scratch(); staged.extend(recv.cloned()); for j in 0..argc + kwc { // SAFETY: native code wrote `argc + kwc` entries, and the @@ -8835,7 +8837,7 @@ unsafe fn dyn_kw_site_bind( let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; staged.push(unpack_pins(bits, tag, &ctx.pins)); } - let mut locals = Vec::new(); + let mut locals = interp.pooled_scratch(); crate::Interpreter::kw_names_fill_locals( &mut locals, &mut staged, @@ -8847,6 +8849,7 @@ unsafe fn dyn_kw_site_bind( kwc, eff_argc, ); + interp.recycle_scratch(staged); Some((f.clone(), locals)) } diff --git a/tests/regrtest/test_datetime_fields.py b/tests/regrtest/test_datetime_fields.py index e69fa838..9215bd36 100644 --- a/tests/regrtest/test_datetime_fields.py +++ b/tests/regrtest/test_datetime_fields.py @@ -89,7 +89,10 @@ def record_time(*args): return result try: reference._date_fields, reference._time_fields = record_date, record_time - datetime.datetime(2024, 2, 29, 12, 34, 56, 123456) + # The exact class with int fields is built natively, before + # `__new__`; a subclass runs `__new__` and its helpers. + value = datetime.datetime(2024, 2, 29, 12, 34, 56, 123456) + assert calls == [] and value == Moment(2024, 2, 29, 12, 34, 56, 123456) assert calls == [('date', True), ('time', True)] calls.clear() datetime.date(Index('year', 2024), 2, 29) From 9cc0d31a9f16bd20e0359e902f643bec535311ab Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 04:10:10 -0700 Subject: [PATCH 09/65] perf: allocate through mimalloc's plain entry points The `mimalloc` crate's `GlobalAlloc` sends every Rust allocation through `mi_malloc_aligned`, whose fast path needs the size class's free list to offer a suitably aligned block; otherwise it takes a generic path that can over-allocate (mimalloc v3 treats only power-of-two block sizes as naturally aligned). Every mimalloc block is word aligned, so a small allocator shim over `libmimalloc-sys` uses `mi_malloc`/`mi_zalloc`/ `mi_realloc` for alignments up to a word and the aligned entry points only above that. 1-4% fewer instructions on allocation-heavy fixtures, and float_math's peak RSS drops 1.5 MB. --- Cargo.lock | 11 +--- Cargo.toml | 2 +- crates/weavepy-cli/Cargo.toml | 2 +- crates/weavepy-cli/src/alloc.rs | 70 +++++++++++++++++++++++++ crates/weavepy-cli/src/alloc_profile.rs | 6 +-- crates/weavepy-cli/src/lib.rs | 4 +- 6 files changed, 79 insertions(+), 16 deletions(-) create mode 100644 crates/weavepy-cli/src/alloc.rs diff --git a/Cargo.lock b/Cargo.lock index e0d7cc38..1036cdee 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1373,15 +1373,6 @@ dependencies = [ "libc", ] -[[package]] -name = "mimalloc" -version = "0.1.52" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2d4139bb28d14ad1facf21d5eb8825051b326e172d216b39f6d31df53cc97862" -dependencies = [ - "libmimalloc-sys", -] - [[package]] name = "minimal-lexical" version = "0.2.1" @@ -2837,7 +2828,7 @@ dependencies = [ "clap", "dirs", "libc", - "mimalloc", + "libmimalloc-sys", "rustyline", "serde_json", "tracing", diff --git a/Cargo.toml b/Cargo.toml index 213ef289..05e4d0d2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -141,7 +141,7 @@ unicode_names2 = "2.0" memmap2 = "0.9" memchr = "2.8" libc = "0.2" -mimalloc = { version = "0.1.52", default-features = false } +libmimalloc-sys = { version = "0.1.49", default-features = false } rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] } rustls-pki-types = "1.7" # PKCS#8 "ENCRYPTED PRIVATE KEY" (PBES2) decryption for password-protected diff --git a/crates/weavepy-cli/Cargo.toml b/crates/weavepy-cli/Cargo.toml index 9d1c0124..25c2a067 100644 --- a/crates/weavepy-cli/Cargo.toml +++ b/crates/weavepy-cli/Cargo.toml @@ -39,7 +39,7 @@ weavepy-compiler = { workspace = true } weavepy-conformance = { workspace = true } weavepy-version = { workspace = true } libc = { workspace = true } -mimalloc = { workspace = true } +libmimalloc-sys = { workspace = true } anyhow = { workspace = true } clap = { workspace = true } serde_json = { workspace = true } diff --git a/crates/weavepy-cli/src/alloc.rs b/crates/weavepy-cli/src/alloc.rs new file mode 100644 index 00000000..758a0485 --- /dev/null +++ b/crates/weavepy-cli/src/alloc.rs @@ -0,0 +1,70 @@ +//! The process allocator: mimalloc, entered through its plain allocation +//! functions whenever they already satisfy the requested alignment. +//! +//! Every mimalloc block is at least word aligned, so a layout aligned to +//! 8 bytes or less needs no aligned entry point. The aligned entry points +//! take their fast path only when the size class's free list happens to +//! offer a suitably aligned block, and otherwise fall to a generic path +//! that may over-allocate. Rust asks for alignment on every allocation, +//! so going through them would put every allocation on that path. + +use std::alloc::{GlobalAlloc, Layout}; +use std::ffi::c_void; + +use libmimalloc_sys::{ + mi_free, mi_malloc, mi_malloc_aligned, mi_realloc, mi_realloc_aligned, mi_zalloc, + mi_zalloc_aligned, +}; + +/// The largest alignment every mimalloc block already has. +const WORD_ALIGN: usize = std::mem::size_of::(); + +/// mimalloc as the global allocator (see the module docs). +pub(crate) struct Mimalloc; + +// SAFETY: every block comes from mimalloc with at least the layout's +// alignment (word-aligned blocks, or the aligned entry points for larger +// alignments), and every block is released or resized through mimalloc. +unsafe impl GlobalAlloc for Mimalloc { + #[inline] + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + // SAFETY: plain FFI allocation calls. + unsafe { + if layout.align() <= WORD_ALIGN { + mi_malloc(layout.size()).cast() + } else { + mi_malloc_aligned(layout.size(), layout.align()).cast() + } + } + } + + #[inline] + unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { + // SAFETY: as in `alloc`. + unsafe { + if layout.align() <= WORD_ALIGN { + mi_zalloc(layout.size()).cast() + } else { + mi_zalloc_aligned(layout.size(), layout.align()).cast() + } + } + } + + #[inline] + unsafe fn dealloc(&self, ptr: *mut u8, _layout: Layout) { + // SAFETY: `ptr` came from this allocator (the caller's contract). + unsafe { mi_free(ptr.cast::()) } + } + + #[inline] + unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { + // SAFETY: `ptr` came from this allocator with `layout`. + unsafe { + if layout.align() <= WORD_ALIGN { + mi_realloc(ptr.cast::(), new_size).cast() + } else { + mi_realloc_aligned(ptr.cast::(), new_size, layout.align()).cast() + } + } + } +} diff --git a/crates/weavepy-cli/src/alloc_profile.rs b/crates/weavepy-cli/src/alloc_profile.rs index b3e48335..13bfb8a0 100644 --- a/crates/weavepy-cli/src/alloc_profile.rs +++ b/crates/weavepy-cli/src/alloc_profile.rs @@ -130,7 +130,7 @@ pub(crate) struct Profiled; unsafe impl GlobalAlloc for Profiled { unsafe fn alloc(&self, layout: Layout) -> *mut u8 { // SAFETY: forwarded unchanged. - let p = unsafe { mimalloc::MiMalloc.alloc(layout) }; + let p = unsafe { crate::alloc::Mimalloc.alloc(layout) }; if ENABLED.load(Ordering::Relaxed) && !p.is_null() { sample(p as usize, layout.size()); } @@ -142,7 +142,7 @@ unsafe impl GlobalAlloc for Profiled { forget(ptr as usize); } // SAFETY: forwarded unchanged. - unsafe { mimalloc::MiMalloc.dealloc(ptr, layout) } + unsafe { crate::alloc::Mimalloc.dealloc(ptr, layout) } } unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { @@ -150,7 +150,7 @@ unsafe impl GlobalAlloc for Profiled { forget(ptr as usize); } // SAFETY: forwarded unchanged. - let p = unsafe { mimalloc::MiMalloc.realloc(ptr, layout, new_size) }; + let p = unsafe { crate::alloc::Mimalloc.realloc(ptr, layout, new_size) }; if ENABLED.load(Ordering::Relaxed) && !p.is_null() { sample(p as usize, new_size); } diff --git a/crates/weavepy-cli/src/lib.rs b/crates/weavepy-cli/src/lib.rs index 4ae44b6a..6fa2bdec 100644 --- a/crates/weavepy-cli/src/lib.rs +++ b/crates/weavepy-cli/src/lib.rs @@ -36,12 +36,14 @@ use tracing_subscriber::EnvFilter; use weavepy::{InterpreterFlags, RunOptions}; +mod alloc; + /// The process allocator. The interpreter allocates and frees small blocks /// constantly; mimalloc serves them from per-thread free lists, several /// times faster than the system allocator on macOS. #[cfg(not(all(feature = "alloc-profile", target_os = "macos")))] #[global_allocator] -static GLOBAL_ALLOC: mimalloc::MiMalloc = mimalloc::MiMalloc; +static GLOBAL_ALLOC: alloc::Mimalloc = alloc::Mimalloc; #[cfg(all(feature = "alloc-profile", target_os = "macos"))] mod alloc_profile; From faf7ae6ff01073f847d03dc022b548190ecc0c88 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 04:10:10 -0700 Subject: [PATCH 10/65] perf: run builtin container methods inside the core loop The core loop now loads a `dict`, `set` or `str` receiver's site-cached native method itself (it already did for lists), and calls every leaf builtin kind whose body releases no reference beyond its operands (the `str` methods, `len`, `dict.get` and views, `list.append`/`pop`/ `insert`/`reverse`/`copy`, `set.pop`) through `leaf_builtin_call` right there. Each such call used to leave the loop twice, once for the method load and once for the call, and rerun the loop's prologue both times. About 800 fewer instructions per call: str_methods 955M -> 828M, call_overhead and datetime_ops improve as well. --- crates/weavepy-vm/src/lib.rs | 80 ++++++++++++++++++++++++++++-------- 1 file changed, 64 insertions(+), 16 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 32c0388f..0548b21e 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12991,14 +12991,9 @@ impl Interpreter { // `leaf_builtin_call` for these kinds). Object::Builtin(b) if Rc::strong_count(b) > 1 - && matches!( - site_slot.and_then(|s| s.get_leaf(b)), - Some( - LeafKind::Opaque - | LeafKind::Fast(_) - | LeafKind::Isinstance - ) - ) => + && site_slot + .and_then(|s| s.get_leaf(b)) + .is_some_and(LeafKind::runs_in_core) => { let kind = site_slot.and_then(|s| s.get_leaf(b)); let callee_at = len - argc - 2; @@ -13021,10 +13016,13 @@ impl Interpreter { let r = match kind { Some(LeafKind::Fast(f)) => f(ops), Some(LeafKind::Isinstance) => Self::core_isinstance(ops), - _ => Some(match b.call_kw.as_ref() { - Some(ckw) => ckw(ops, &[]), - None => (b.call)(ops), - }), + Some(LeafKind::Opaque) | None => { + Some(match b.call_kw.as_ref() { + Some(ckw) => ckw(ops, &[]), + None => (b.call)(ops), + }) + } + Some(k) => self.leaf_builtin_call(k, b, ops), }; match r { // A fast half declined, untouched. @@ -13428,10 +13426,20 @@ impl Interpreter { None => break Some(CoreExit::Helper), } } - // A list's site-cached native method (the helper's - // builtin-receiver case, which fills the slot). - Object::List(_) => { - let Some(b) = ms.get_builtin(1) else { + // A native container's site-cached method (the + // helper's builtin-receiver case, which fills the + // slot under the receiver's tag). + recv @ (Object::List(_) + | Object::Dict(_) + | Object::Set(_) + | Object::Str(_)) => { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + _ => 4, + }; + let Some(b) = ms.get_builtin(tag) else { break Some(CoreExit::Helper); }; // SAFETY: `len < cap`; the receiver moves up. @@ -55440,6 +55448,46 @@ enum LeafKind { } impl LeafKind { + /// Whether the core loop runs this kind's call itself (operands that + /// all leave by plain decrements): the registered leaf bodies, and the + /// native methods that release no reference beyond their operands (a + /// removal, which drops an element, goes through the helper, whose + /// release grading covers it). + fn runs_in_core(self) -> bool { + matches!( + self, + Self::Opaque + | Self::Fast(_) + | Self::Isinstance + | Self::Len + | Self::ListAppend + | Self::ListPop + | Self::ListInsert + | Self::ListReverse + | Self::ListCopy + | Self::DictGet + | Self::DictKeys + | Self::DictValues + | Self::DictItems + | Self::SetPop + | Self::StrStartswith + | Self::StrEndswith + | Self::StrLower + | Self::StrUpper + | Self::StrStrip + | Self::StrLstrip + | Self::StrRstrip + | Self::StrFind + | Self::StrIsdigit + | Self::StrIsalpha + | Self::StrIsspace + | Self::StrSplit + | Self::StrJoin + | Self::StrReplace + | Self::StrFormat + ) + } + /// These self-only operations neither invoke Python nor discard any /// existing reference held by the receiver. Argument-shape admission /// still belongs to `leaf_builtin_call`; subclasses must fall back. From ddf6d240fc6d8dc7faaec1e6e9a1f60e11df46e3 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 08:07:18 -0700 Subject: [PATCH 11/65] perf: run small mutating methods and native iterators without frames The frameless leaf evaluator, which ran pure getters and predicates in place, now also covers: - Effect leaves: bodies whose only side effects are attribute stores. Stores are buffered as owned values (later loads of the same attribute read them back) and committed at the return, after every target is re-validated, so a bail-out anywhere leaves nothing behind and the ordinary call reruns the body. Setters get a dedicated shape. - Local variables (STORE_FAST and its fused forms). - Nested calls to pure leaves (methods and functions), evaluated in place up to a small depth while no store is buffered. - A code that keeps bailing (64 consecutive misses) loses its leaf verdict, so calls stop paying for the attempt. Pure method calls also take the interned names from the loaded constant table instead of re-deriving it per attribute check. `len()`, truth tests, and iteration over an instance whose class resolves the dunder to a native builtin call that builtin directly, and the core loop runs a native iterator's `__next__` in place. Deque iteration reads its slots by position. A setter call: 2001 -> 924 instructions; deltablue's `execute` kernel: 3976 -> 3313; deque iteration: 3278 -> 1562 per item. --- crates/weavepy-compiler/src/lib.rs | 46 +- crates/weavepy-vm/src/lib.rs | 843 ++++++++++++++++-- .../src/stdlib/collections_native.rs | 69 +- 3 files changed, 854 insertions(+), 104 deletions(-) diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index aa426d2f..71bba165 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -216,10 +216,13 @@ pub struct JitHint { /// The tier-2 state has no further interest in this code's back /// edges (compiled, OSR budget spent). backedge_quiet: std::sync::atomic::AtomicBool, - /// The VM's pure-leaf verdict (0 not yet decided, 1 no, 2 yes), - /// mirrored here so native call sites read it without the VM - /// extension lookup. + /// The VM's pure-leaf verdict (0 not yet decided, 1 no, 2 yes, 3 an + /// effect leaf), mirrored here so native call sites read it without + /// the VM extension lookup. pure_leaf: std::sync::atomic::AtomicU8, + /// Consecutive leaf evaluations the VM abandoned (see + /// [`Self::note_leaf_miss`]). + leaf_misses: std::sync::atomic::AtomicU8, } impl JitHint { @@ -277,12 +280,47 @@ impl JitHint { #[must_use] pub fn pure_leaf(&self) -> Option { match self.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) { - 1 => Some(false), + 1 | 3 => Some(false), 2 => Some(true), _ => None, } } + /// Whether the VM decided the code is an *effect leaf*: a pure leaf + /// but for attribute stores (never a pure leaf itself). + #[must_use] + pub fn effect_leaf(&self) -> bool { + self.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) == 3 + } + + /// Record an effect-leaf verdict (see [`Self::effect_leaf`]). + pub fn set_effect_leaf(&self) { + self.pure_leaf + .store(3, std::sync::atomic::Ordering::Relaxed); + } + + /// Count one abandoned leaf evaluation. A long enough run of them + /// (a callee the evaluator never settles, a store that always + /// declines) withdraws the leaf verdict for good, so calls stop + /// paying for an attempt before their ordinary path. + pub fn note_leaf_miss(&self) { + use std::sync::atomic::Ordering::Relaxed; + let n = self.leaf_misses.load(Relaxed).saturating_add(1); + self.leaf_misses.store(n, Relaxed); + if n >= 64 { + self.pure_leaf.store(1, Relaxed); + } + } + + /// A completed leaf evaluation ends a run of misses. + #[inline] + pub fn note_leaf_hit(&self) { + use std::sync::atomic::Ordering::Relaxed; + if self.leaf_misses.load(Relaxed) != 0 { + self.leaf_misses.store(0, Relaxed); + } + } + /// Record the VM's pure-leaf verdict (see [`Self::pure_leaf`]). pub fn set_pure_leaf(&self, yes: bool) { self.pure_leaf.store( diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 0548b21e..4ffa4326 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -966,6 +966,10 @@ pub struct Interpreter { /// The registered leaf builtins (see [`leaf_builtins`]) as last /// snapshotted, with the registry generation they reflect. leaf_opaque: ThreadCell<(u64, leaf_builtins::LeafMap)>, + /// The core loop's last native iterator `__next__`, keyed by the + /// (process-unique) attribute version of the class that resolved it + /// (see [`Interpreter::core_leaf_next`]). + core_next: ThreadCell)>>, /// `WP_DBG_SAMPLE`: periodic frame-entry sampling to stderr. dbg_sample: bool, } @@ -1195,6 +1199,7 @@ impl Default for Interpreter { lean_pending: Vec::new(), leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), + core_next: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), }; // RFC 0025: publish the shared parts of this interpreter @@ -1321,6 +1326,7 @@ impl Interpreter { lean_pending: Vec::new(), leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), + core_next: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), } } @@ -11884,11 +11890,12 @@ impl Interpreter { ) { (Object::Instance(i), Some(ms)) => ms .peek_fn(i.cls_raw().attr_version.get()) - .is_some_and(|fp| fn_is_pure_leaf(&*fp)), + .is_some_and(|fp| fn_is_leaf(&*fp)), _ => false, }; if pure_site { if let Some((v, call_pc)) = self.core_pure_method( + ext, code, other, pc + 1, @@ -11923,9 +11930,13 @@ impl Interpreter { continue; } } - if let Some(v) = - Self::core_local_attr(code, other, pc + 1, next.arg) - { + if let Some(v) = Self::core_local_attr( + ext.map_or(&[], |t| &t.name_objs), + code, + other, + pc + 1, + next.arg, + ) { base.add(len).write(v); len += 1; last = pc + 1; @@ -12437,6 +12448,35 @@ impl Interpreter { // SAFETY: `len > 0`. let it = match unsafe { &*base.add(len - 1) } { Object::Iter(it) => it, + // A native iterator class's `__next__` (a registered + // leaf builtin), called in place. Exhaustion goes to + // the full arm, which asks again (a leaf iterator + // stays exhausted) and ends the loop. + obj @ Object::Instance(inst) => { + let Some(b) = self.core_leaf_next(inst) else { + break None; + }; + match (b.call)(std::slice::from_ref(obj)) { + Ok(v) => { + // SAFETY: `len < cap`. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + break None; + } + // As a raising call: the pc moves past it. + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } // A generator resumes inline, switched to in place // (the quiet loop's lean path when it declines). Object::Generator(_) => { @@ -12894,7 +12934,7 @@ impl Interpreter { let ops = unsafe { std::slice::from_raw_parts(base.add(len - argc - 2), argc + 2) }; - if matches!(&ops[0], Object::Function(f) if fn_is_pure_leaf(f)) { + if matches!(&ops[0], Object::Function(f) if fn_is_leaf(f)) { if let Some(r) = self.core_pure_call(code, pc, ops, sw.depth_cell) { let start = len - argc - 2; // SAFETY: every operand was checked to leave @@ -14418,7 +14458,9 @@ impl Interpreter { let f = unsafe { &*fp }; // SAFETY: GIL-serialized raw read of the function's code cell. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) { + let pure = code_is_pure_leaf(code_rc); + let effect = !pure && code_is_effect_leaf(code_rc); + if !(pure || effect) || !pure_leaf_warm(code_rc) { return None; } let (missing, slot_self) = code_call_slot(code, call_pc)?.hit(fp, Rc::as_ptr(code_rc))?; @@ -14444,7 +14486,11 @@ impl Interpreter { for (k, o) in f.defaults[f.defaults.len() - missing..].iter().enumerate() { args[nargs + k] = o; } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } } /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); , code: &CodeObject, recv: &Object, attr_pc: usize, @@ -14498,10 +14545,25 @@ impl Interpreter { // SAFETY: a read with nothing running (see `GilCell::peek`). let d = unsafe { dict.peek() }?; if !d.is_empty() { - let probe = code_name_leaf_probe(code, name_idx)?; - if d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) - { - return None; + // The interned name and its hash, from the loaded table. + let idx = name_idx as usize; + let named = + ext.and_then(|t| match (t.name_objs.get(idx), t.name_hashes.get(idx)) { + (Some(Object::Str(n)), Some(h)) => Some((&**n, *h)), + _ => None, + }); + let (name, hash) = match named { + Some(nh) => nh, + None => { + let k = code_name_key(code, name_idx)?; + (k.s, k.hash) + } + }; + if d.may_hold_str_hash(hash) { + let probe = crate::object::LeafNameProbe::new(name, hash); + if d.contains_key(&probe) || probe.saw_exotic() { + return None; + } } } } @@ -14575,7 +14637,9 @@ impl Interpreter { let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; #[cfg(test)] note_predicate_stage(code_rc, 0); - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) { + let pure = code_is_pure_leaf(code_rc); + let effect = !pure && code_is_effect_leaf(code_rc); + if !(pure || effect) || !pure_leaf_warm(code_rc) { return None; } #[cfg(test)] @@ -14656,7 +14720,11 @@ impl Interpreter { for (k, o) in f.defaults[f.defaults.len() - missing..].iter().enumerate() { args[nargs + k] = o; } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } } /// [`Self::core_pure_call`] for a `CALL_KW` at `pc`: `ops` is the @@ -14735,10 +14803,11 @@ impl Interpreter { .get(f.defaults.len().checked_sub(total - slot)?)?; } } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + self.pure_leaf_eval::(code_rc, f, &args[..total]) } - /// Borrow a guarded instance-dictionary or slot cache hit. + /// Borrow a guarded instance-dictionary or slot cache hit. `names` is + /// the code's interned name objects (see [`slot_name_matches_in`]). /// /// # Safety /// @@ -14747,6 +14816,7 @@ impl Interpreter { /// shared storage and conflicting mutable borrows. #[inline(always)] unsafe fn leaf_cached_instance_field<'a>( + names: &[Object], code: &CodeObject, inst: &'a PyInstance, cache_pc: u32, @@ -14766,13 +14836,13 @@ impl Interpreter { // SAFETY: the caller keeps this rooted read callback-free. let dict = unsafe { inst.dict.get()?.peek() }?; let (key, value) = dict.get_index(key_idx as usize)?; - slot_name_matches(code, name_idx, key).then_some(value) + slot_name_matches_in(names, code, name_idx, key).then_some(value) } else { // SAFETY: the same rooted read as the dictionary path. let slots = unsafe { inst.slots.peek() }?; let indexed = slots .get_index(key_idx as usize) - .filter(|(key, _)| slot_name_matches(code, name_idx, key)) + .filter(|(key, _)| slot_name_matches_in(names, code, name_idx, key)) .map(|(_, value)| value); // Slot order can vary by instance or after deletion. let value = indexed.or_else(|| slots.get(code.names.get(name_idx as usize)?))?; @@ -14795,22 +14865,48 @@ impl Interpreter { /// Evaluate a pure leaf's body (see [`code_is_pure_leaf`]) on borrowed /// arguments: operands are scalars or pointers to objects that stay - /// put (nothing here stores, calls, or runs Python code), so no - /// reference is taken until the returned value's. Loads take the core - /// loop's cache-hit paths; anything else — a miss, an operand shape - /// the scalar arms don't settle, an operation that could raise — - /// abandons the evaluation with `None`, having done nothing - /// observable. + /// put (nothing here runs Python code, and an effect leaf's stores + /// wait for its return), so no reference is taken until the returned + /// value's. Loads take the core loop's cache-hit paths, and a call + /// evaluates a pure-leaf callee the same way; anything else — a miss, + /// an operand shape the scalar arms don't settle, an operation that + /// could raise — abandons the evaluation with `None`, having done + /// nothing observable. + #[inline(always)] + fn pure_leaf_eval( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + ) -> Option { + let r = self.leaf_eval::(code, f, args, 0); + if r.is_some() { + code.jit_hint.note_leaf_hit(); + } else { + code.jit_hint.note_leaf_miss(); + } + r + } + + /// [`Self::pure_leaf_eval`] at call-nesting depth `nest` (a pure-leaf + /// callee of a leaf is evaluated one level down, to a small bound). #[inline(never)] - fn pure_leaf_eval( + fn leaf_eval( &self, code: &CodeObject, f: &crate::object::PyFunction, args: &[*const Object], + nest: u8, ) -> Option { use weavepy_compiler::InlineCache as IC; + /// Nested leaf calls evaluated in place at most this deep. + const NEST: u8 = 3; #[derive(Clone, Copy)] enum V { + /// A resolved callee: a function the class or namespace holds. + Fn(*const crate::object::PyFunction), + /// A call's empty self slot. + Null, /// A heap object, borrowed. R(*const Object), I(i64), @@ -14845,6 +14941,7 @@ impl Interpreter { Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), _ => return None, }, + V::Fn(_) | V::Null => return None, }) } // Both the decoded field shape and the general evaluator use @@ -14893,6 +14990,39 @@ impl Interpreter { let start = usize::from(instrs.first()?.op == OpCode::Resume); let load = instrs.get(start)?; match shape { + 16 if EFFECT => { + // The setter: store, and return `None`. A declined + // store touched nothing (the ordinary call runs it). + // SAFETY (argument reads): the arguments stay live for + // this evaluation. + let arg = |k: u32| args.get(k as usize).map(|&p| unsafe { &*p }); + let (value, recv, store_pc) = match load.op { + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => ( + clone_hot(arg(load.arg >> 4)?), + arg(load.arg & 15)?, + start + 1, + ), + op => { + let value = match op { + OpCode::LoadConst => clone_hot(consts.get(load.arg as usize)?), + OpCode::LoadSmallInt => Object::Int(i64::from(load.arg)), + _ => clone_hot(arg(load.arg)?), + }; + (value, arg(instrs.get(start + 1)?.arg)?, start + 2) + } + }; + let Object::Instance(inst) = recv else { + return None; + }; + let store = instrs.get(store_pc)?; + let value = std::mem::ManuallyDrop::new(value); + // On `true` the value moved into the dict. + if Self::core_store_attr(code, inst, store_pc, store.arg, &value) { + return Some(Object::None); + } + drop(std::mem::ManuallyDrop::into_inner(value)); + return None; + } 3 => { // SAFETY: the argument remains live for this evaluation. return Some(clone_hot(unsafe { &**args.get(load.arg as usize)? })); @@ -14908,7 +15038,13 @@ impl Interpreter { // SAFETY: the argument roots the receiver until the // result is retained; nothing here invokes Python. if let Some(value) = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, attr.arg) + Self::leaf_cached_instance_field( + &ext.name_objs, + code, + inst, + pc as u32, + attr.arg, + ) } { return Some(clone_hot(value)); } @@ -14932,7 +15068,13 @@ impl Interpreter { let attr = instrs.get(pc)?; // SAFETY: the same rooted, callback-free read. let value = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, attr.arg) + Self::leaf_cached_instance_field( + &ext.name_objs, + code, + inst, + pc as u32, + attr.arg, + ) }?; #[cfg(test)] note_predicate_stage(code, if load_pc == start { 6 } else { 7 }); @@ -14986,6 +15128,55 @@ impl Interpreter { n: 0, }; let op: *mut Owned = &raw mut owned; + // An effect leaf's attribute stores, in order, until the return + // commits them: receiver, store pc, name index, value (all owned). + // Entries are only appended, so a value read back from one stays + // put while the evaluation runs. + const PENDING: usize = 6; + // The receiver is a stable location for the whole evaluation: an + // argument (rooted by the caller) or an owned-scratch clone. + type Store = (*const Object, u32, u32, Object); + struct Pending { + buf: [std::mem::MaybeUninit; PENDING], + n: usize, + /// Entries whose value the return moved into a dict. + moved: u8, + } + impl Pending { + fn get(&self, k: usize) -> &Store { + debug_assert!(k < self.n); + // SAFETY: the first `n` entries are initialized. + unsafe { self.buf[k].assume_init_ref() } + } + /// Whether entry `k` is the last store to its attribute. + fn latest(&self, k: usize) -> bool { + let (r0, _, n0, _) = self.get(k); + !(k + 1..self.n).any(|j| { + let (r1, _, n1, _) = self.get(j); + // SAFETY: stable receivers (see `Store`). + n1 == n0 && unsafe { (**r1).is_same(&**r0) } + }) + } + } + impl Drop for Pending { + fn drop(&mut self) { + for k in 0..self.n { + // SAFETY: the first `n` entries are initialized; a + // moved value is left in place, not dropped again. + let (_, _, _, value) = unsafe { self.buf[k].assume_init_read() }; + if self.moved & (1 << k) == 0 { + drop_hot(value); + } else { + std::mem::forget(value); + } + } + } + } + let mut pend = Pending { + buf: [const { std::mem::MaybeUninit::uninit() }; PENDING], + n: 0, + moved: 0, + }; let own = |v: Object| -> Option { match v { Object::Int(i) => Some(V::I(i)), @@ -15024,17 +15215,57 @@ impl Interpreter { st[sp] }}; } + // Locals: an argument reads through its pointer until the body + // assigns it; any other local must be assigned before it's read. + const LOCALS: usize = 16; + let mut locs = [V::N; LOCALS]; + let mut bound: u32 = 0; + macro_rules! local { + ($i:expr) => {{ + let i = $i as usize; + if i < LOCALS && bound & (1 << i) != 0 { + locs[i] + } else { + norm(*args.get(i)?) + } + }}; + } + macro_rules! set_local { + ($i:expr, $v:expr) => {{ + let i = $i as usize; + if i >= LOCALS { + return None; + } + locs[i] = $v; + bound |= 1 << i; + }}; + } let mut pc = 0usize; loop { let ins = *instrs.get(pc)?; match ins.op { OpCode::Resume | OpCode::Nop | OpCode::NotTaken => {} OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { - push!(norm(*args.get(ins.arg as usize)?)); + push!(local!(ins.arg)); } OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { - push!(norm(*args.get((ins.arg >> 4) as usize)?)); - push!(norm(*args.get((ins.arg & 15) as usize)?)); + push!(local!(ins.arg >> 4)); + push!(local!(ins.arg & 15)); + } + OpCode::StoreFast => { + let v = pop!(); + set_local!(ins.arg, v); + } + OpCode::StoreFastLoadFast => { + let v = pop!(); + set_local!(ins.arg >> 4, v); + push!(local!(ins.arg & 15)); + } + OpCode::StoreFastStoreFast => { + let v = pop!(); + set_local!(ins.arg >> 4, v); + let w = pop!(); + set_local!(ins.arg & 15, w); } OpCode::LoadConst => push!(norm(consts.get(ins.arg as usize)?)), OpCode::LoadSmallInt => push!(V::I(i64::from(ins.arg))), @@ -15071,6 +15302,19 @@ impl Interpreter { }; // SAFETY: as `norm`. let recv = unsafe { &*p }; + if EFFECT && pend.n > 0 { + // A buffered store to this attribute, the latest. + let hit = (0..pend.n) + .rev() + .map(|k| pend.get(k)) + // SAFETY: stable receivers (see `Store`). + .find(|(r, _, name, _)| *name == ins.arg && unsafe { (**r).is_same(recv) }); + if let Some((_, _, _, v)) = hit { + push!(norm(v)); + pc += 1; + continue; + } + } let v = match recv { Object::Instance(inst) => { let cls = inst.cls_raw(); @@ -15080,7 +15324,13 @@ impl Interpreter { // SAFETY: the receiver remains rooted by an // argument or owned scratch; no Python runs. let hit = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, ins.arg) + Self::leaf_cached_instance_field( + &ext.name_objs, + code, + inst, + pc as u32, + ins.arg, + ) } .map(std::ptr::from_ref); match hit { @@ -15225,8 +15475,129 @@ impl Interpreter { OpCode::PopTop => { let _ = pop!(); } + OpCode::PushNull => push!(V::Null), + OpCode::LoadGlobalPushNull => { + // As `LOAD_GLOBAL`, then the call's empty self slot. + let slot = stamps.get(pc)?; + let gdict = f.globals.as_ptr(); + let gid = specialize::rc_id(&f.globals); + // SAFETY (raw dict reads): nothing runs code here. + let g_stamp = unsafe { (*gdict).mutation_stamp() }; + let hit = match code.caches.get(pc as u32) { + IC::LoadGlobalModule { + globals_id, + key_idx, + } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { + (*gdict).get_index(key_idx as usize) + }, + _ => return None, + }; + push!(norm(hit?.1)); + push!(V::Null); + } + OpCode::LoadMethodAttr => { + // A method off the site's slot: the function under + // the receiver, or a class's function with an empty + // self slot (as the core loop's arm). + let V::R(p) = pop!() else { + return None; + }; + let ms = code_method_slot(code, pc as u32)?; + // SAFETY: as `norm`. + match unsafe { &*p } { + Object::Instance(inst) => { + let cls = inst.cls_raw(); + if !Self::default_getattribute(cls) { + return None; + } + let fp = ms.peek_fn(cls.attr_version.get())?; + // The instance dict must not shadow the method. + if let Some(dict) = inst.dict.get() { + // SAFETY: a read with nothing running. + let d = unsafe { dict.peek() }?; + let idx = ins.arg as usize; + let (Some(Object::Str(name)), Some(&hash)) = + (ext.name_objs.get(idx), ext.name_hashes.get(idx)) + else { + return None; + }; + if !d.is_empty() && d.may_hold_str_hash(hash) { + let probe = crate::object::LeafNameProbe::new(name, hash); + if d.contains_key(&probe) || probe.saw_exotic() { + return None; + } + } + } + push!(V::Fn(fp)); + push!(V::R(p)); + } + Object::Type(cls) => { + push!(V::Fn(ms.peek_unbound(cls.attr_version.get())?)); + push!(V::Null); + } + _ => return None, + } + } + OpCode::Call => { + // A pure-leaf callee, evaluated in place: only while + // no store is buffered (it would not see one). + let argc = ins.arg as usize; + if (EFFECT && pend.n > 0) || nest >= NEST || sp < argc + 2 { + return None; + } + let at = sp - argc - 2; + let fp = match st[at] { + V::Fn(fp) => fp, + // SAFETY: as `norm`. + V::R(p) => match unsafe { &*p } { + Object::Function(func) => Rc::as_ptr(func), + _ => return None, + }, + _ => return None, + }; + // SAFETY: the class or the namespace holds the callee, + // and nothing here runs code that could release it. + let callee = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the code cell. + let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; + let first = if matches!(st[at + 1], V::Null) { + at + 2 + } else { + at + 1 + }; + let n = sp - first; + if !code_is_pure_leaf(ccode) + || !Self::lean_code_ok(ccode) + || n != ccode.arg_count as usize + || n > 8 + || crate::recursion::current_depth() + usize::from(nest) + 1 + >= crate::recursion::recursion_limit() + { + return None; + } + // Scalar arguments are staged as objects (no drop glue). + let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut ptrs = [std::ptr::null::(); 8]; + for k in 0..n { + let o = match st[first + k] { + V::R(p) => { + ptrs[k] = p; + continue; + } + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Null => return None, + }; + ptrs[k] = staged[k].write(o); + } + let r = self.leaf_eval::(ccode, callee, &ptrs[..n], nest + 1)?; + sp = at; + push!(own(r)?); + } OpCode::ReturnValue => { - return Some(match pop!() { + let r = match pop!() { // SAFETY: as `norm` (an owned value is cloned before // its holder drops). V::R(p) => clone_hot(unsafe { &*p }), @@ -15234,7 +15605,82 @@ impl Interpreter { V::F(x) => Object::Float(x), V::B(b) => Object::Bool(b), V::N => Object::None, - }); + V::Fn(_) | V::Null => return None, + }; + if EFFECT && pend.n > 0 { + // The latest store to each attribute is the one + // that lands; every one of them must go through + // before any does (a decline touched nothing, and + // the ordinary call runs the body instead). A + // lone store is its own check: it declines whole. + if pend.n > 1 { + for k in 0..pend.n { + if !pend.latest(k) { + continue; + } + let (rp, spc, name, _) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + if !Self::core_store_attr_ready(code, inst, *spc as usize, *name) { + return None; + } + } + } + let n = pend.n; + for k in 0..n { + if !pend.latest(k) { + continue; + } + let (rp, spc, name, value) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + // On `true` the value moved into the dict. + if Self::core_store_attr(code, inst, *spc as usize, *name, value) { + pend.moved |= 1 << k; + } else if n == 1 { + // The lone store declined, untouched. + return None; + } else { + debug_assert!(false, "a ready store declined"); + } + } + } + return Some(r); + } + OpCode::StoreAttr if EFFECT => { + let V::R(rp) = pop!() else { + return None; + }; + // SAFETY: as `norm`. + let recv = unsafe { &*rp }; + if !matches!(recv, Object::Instance(_)) || pend.n == PENDING { + return None; + } + let value = match pop!() { + // SAFETY: as `norm`. + V::R(p) => clone_hot(unsafe { &*p }), + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Null => return None, + }; + // An argument receiver is used in place; any other + // (a field's value) is held by the owned scratch. + let rp = if args.iter().any(|&a| std::ptr::eq(a, rp)) { + rp + } else { + let V::R(held) = own(clone_hot(recv))? else { + return None; + }; + held + }; + pend.buf[pend.n].write((rp, pc as u32, ins.arg, value)); + pend.n += 1; } _ => return None, } @@ -15804,6 +16250,7 @@ impl Interpreter { /// code), anything else through [`Self::leaf_fused_local_attr`]. #[inline(never)] fn core_local_attr( + names: &[Object], code: &CodeObject, local: &Object, attr_pc: usize, @@ -15820,7 +16267,7 @@ impl Interpreter { // SAFETY: a read between two instructions (see `peek`). let d = unsafe { inst.dict.get()?.peek() }?; let (k, v) = d.get_index(key_idx as usize)?; - return slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)); + return slot_name_matches_in(names, code, name_idx, k).then(|| Self::clone_operand(v)); } Self::leaf_fused_local_attr(code, local, attr_pc, name_idx) } @@ -15987,6 +16434,62 @@ impl Interpreter { } } + /// Whether [`Self::core_store_attr`] would perform this store right + /// now (it declines exactly when this is `false`), touching nothing: + /// an effect leaf validates every buffered store before committing + /// any of them. + fn core_store_attr_ready( + code: &CodeObject, + inst: &PyInstance, + attr_pc: usize, + name_idx: u32, + ) -> bool { + use weavepy_compiler::InlineCache as IC; + let cls = inst.cls_raw(); + let ver = match code.caches.get(attr_pc as u32) { + IC::StoreAttrInstance { ver, .. } | IC::StoreAttrNewKey { ver } => ver, + _ => return false, + }; + if cls.native_kind.get() != 0 + || cls.attr_version.get() != ver + || crate::capi_watchers::dicts_active() + { + return false; + } + match code.caches.get(attr_pc as u32) { + IC::StoreAttrInstance { key_idx, .. } => { + // SAFETY: an unused view with nothing running (the store's + // own `peek_mut` condition: no borrow at all). + let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek_mut() }) else { + return false; + }; + matches!(d.get_index(key_idx as usize), Some((k, old)) + if slot_name_matches(code, name_idx, k) && Self::core_droppable(old)) + } + _ => { + // No dictionary yet: the store creates it and inserts. + let Some(dict) = inst.dict.get() else { + return true; + }; + let Some(probe) = code_name_leaf_probe(code, name_idx) else { + return false; + }; + // SAFETY: as above. + let Some(d) = (unsafe { dict.peek_mut() }) else { + return false; + }; + let found = d.get_index_of(&probe); + if probe.saw_exotic() { + return false; + } + found.is_none_or(|i| { + d.get_index(i) + .is_some_and(|(_, old)| Self::core_droppable(old)) + }) + } + } + } + /// The constructor shape of [`Self::core_store_attr`]: a /// `StoreAttrNewKey` site's insert-or-overwrite, in one probe. Out of /// line so the warm indexed store stays small. @@ -18227,6 +18730,27 @@ impl Interpreter { /// Which leaf builtin `b` is, if any (pointer identity). #[inline] + /// `inst`'s class's `__next__` when it is a registered leaf builtin + /// that binds its instance (the core loop's native iterators). + fn core_leaf_next(&self, inst: &PyInstance) -> Option> { + let ver = inst.cls_raw().attr_version.get(); + if let Some((v, b)) = &*self.core_next.borrow() { + if *v == ver { + return Some(b.clone()); + } + } + match inst.cls().lookup("__next__")? { + Object::Builtin(b) + if b.binds_instance + && matches!(self.leaf_call_kind(&b), Some(LeafKind::Opaque)) => + { + *self.core_next.borrow_mut() = Some((ver, b.clone())); + Some(b) + } + _ => None, + } + } + fn leaf_call_kind(&self, b: &Rc) -> Option { let p = Rc::as_ptr(b) as usize; if let Some(k) = self.leaf_fns().calls.get(&p) { @@ -18683,7 +19207,7 @@ impl Interpreter { // Getter mode additionally validates metaclasses and module overrides; // ordinary pure-call evaluation avoids these extra checks. let _ = code_is_pure_leaf(&fcode); - self.pure_leaf_eval::(&fcode, &f, &args[..nargs]) + self.pure_leaf_eval::(&fcode, &f, &args[..nargs]) } /// The cache-free half of [`Self::leaf_load_attr_recv`]: the site's @@ -21387,8 +21911,13 @@ impl Interpreter { }, Object::Instance(_) => { // Call __next__; treat StopIteration as exhaustion. - match instance_method(&it_obj, "__next__") { - Some(m) => match self.call(&m, &[], &[], &frame.globals) { + let next = match instance_native_dunder(&it_obj, "__next__", None) { + Some(r) => Some(r), + None => instance_method(&it_obj, "__next__") + .map(|m| self.call(&m, &[], &[], &frame.globals)), + }; + match next { + Some(r) => match r { Ok(v) => Some(v), Err(RuntimeError::PyException(exc)) if exc.type_name() == "StopIteration" => @@ -28201,6 +28730,9 @@ impl Interpreter { // `len(x)` is `type(x).__len__(x)` — for an *instance* that's the // class's method; for a *class* it's the metaclass's // (`len(SomeEnum)` → `EnumType.__len__`). + if let Some(r) = instance_native_dunder(v, "__len__", None) { + return coerce_len_result(r?).map(Object::Int); + } let method = instance_method(v, "__len__").or_else(|| metaclass_method(v, "__len__")); if let Some(method) = method { let r = self.call(&method, &[], &[], globals)?; @@ -29110,8 +29642,18 @@ impl Interpreter { "NotImplemented should not be used in a boolean context", )); } - if let Some(method) = instance_method(v, "__bool__") { - let r = self.call(&method, &[], &[], globals)?; + let native = instance_native_dunder(v, "__bool__", None); + let method = if native.is_none() { + instance_method(v, "__bool__") + } else { + None + }; + if native.is_some() || method.is_some() { + let r = match (native, method) { + (Some(r), _) => r?, + (None, Some(method)) => self.call(&method, &[], &[], globals)?, + (None, None) => unreachable!("checked above"), + }; // CPython's `slot_nb_bool` is strict: anything but an exact // bool — even an int — raises (`__bool__` returning `1` is a // TypeError, test_bool `test_convert_to_bool`). @@ -29123,6 +29665,9 @@ impl Interpreter { ))), }; } + if let Some(r) = instance_native_dunder(v, "__len__", None) { + return coerce_len_result(r?).map(|n| n != 0); + } if let Some(method) = instance_method(v, "__len__") { let r = self.call(&method, &[], &[], globals)?; // `PyObject_Size` semantics; errors must match `len()`'s @@ -32827,6 +33372,17 @@ impl Interpreter { Err(e) => Err(e), }, Object::Instance(inst) => { + if let Some(r) = instance_native_dunder(iter, "__next__", None) { + return match r { + Ok(v) => Ok(Some(v)), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + Ok(None) + } + Err(e) => Err(e), + }; + } if let Some(method) = instance_method(iter, "__next__") { match self.call(&method, &[], &[], globals) { Ok(v) => Ok(Some(v)), @@ -48223,7 +48779,7 @@ impl Interpreter { for (p, v) in ptrs.iter_mut().zip(&locals[..nargs]) { *p = v; } - if let Some(v) = self.pure_leaf_eval::(&code, f, &ptrs[..nargs]) { + if let Some(v) = self.pure_leaf_eval::(&code, f, &ptrs[..nargs]) { self.recycle_scratch(locals); return Ok(v); } @@ -56736,6 +57292,43 @@ pub(crate) fn callable_function_str(callable: &Object) -> Option { } } +/// `type(obj).(obj, arg?)` for an instance whose class resolves the +/// dunder `name` to a native builtin that binds its instance: the +/// builtin's body, called directly (no bound method, no general call). +/// `None`, with nothing done, for any other resolution (a Python method, a +/// descriptor, an interpreter-aware builtin) or while a profiler or tracer +/// could observe the call; the caller then takes its ordinary path. +fn instance_native_dunder( + obj: &Object, + name: &str, + arg: Option<&Object>, +) -> Option> { + let Object::Instance(inst) = obj else { + return None; + }; + if crate::trace::any_observers_active() { + return None; + } + let Object::Builtin(b) = inst.cls().lookup(name)? else { + return None; + }; + if !b.binds_instance || builtin_needs_interp(b.name) { + return None; + } + let pair; + let args: &[Object] = match arg { + Some(a) => { + pair = [obj.clone(), a.clone()]; + &pair + } + None => std::slice::from_ref(obj), + }; + Some(match b.call_kw.as_ref() { + Some(ckw) => ckw(args, &[]), + None => (b.call)(args), + }) +} + pub(crate) fn instance_method(obj: &Object, name: &str) -> Option { let inst = match obj { Object::Instance(i) => i.clone(), @@ -61623,7 +62216,7 @@ fn code_is_pure_leaf(code: &CodeObject) -> bool { return false; }; match ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) { - 1 => false, + 1 | 16 => false, 2..=7 => true, _ => code_pure_leaf_decide(code, ext), } @@ -61634,7 +62227,7 @@ fn code_is_pure_leaf(code: &CodeObject) -> bool { #[inline(never)] fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { use std::sync::atomic::Ordering::Relaxed; - let ok = !code.is_generator + let structural = !code.is_generator && !code.is_coroutine && !code.is_async_generator && !code.is_iterable_coroutine @@ -61645,40 +62238,58 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { && !code.has_varkeywords && code.kwonly_count == 0 && code.arg_count <= 8 - && code.varnames.len() == code.arg_count as usize + && code.varnames.len() <= 16 && code.exception_table.is_empty() - && code.instructions.len() <= 64 - && code.instructions.iter().all(|i| { - matches!( - i.op, - OpCode::Resume - | OpCode::Nop - | OpCode::NotTaken - | OpCode::LoadFast - | OpCode::LoadFastBorrow - | OpCode::LoadFastCheck - | OpCode::LoadFastLoadFast - | OpCode::LoadFastBorrowLoadFastBorrow - | OpCode::LoadConst - | OpCode::LoadSmallInt - | OpCode::LoadGlobal - | OpCode::LoadAttr - | OpCode::CompareOp - | OpCode::IsOp - | OpCode::ToBool - | OpCode::UnaryOp - | OpCode::PopJumpIfFalse - | OpCode::PopJumpIfTrue - | OpCode::PopJumpIfNone - | OpCode::PopJumpIfNotNone - | OpCode::JumpForward - | OpCode::BinaryOp - | OpCode::CopyTop - | OpCode::Swap - | OpCode::PopTop - | OpCode::ReturnValue - ) - }); + && code.instructions.len() <= 64; + let pure_op = |op: OpCode| { + matches!( + op, + OpCode::Resume + | OpCode::Nop + | OpCode::NotTaken + | OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::StoreFast + | OpCode::StoreFastLoadFast + | OpCode::StoreFastStoreFast + | OpCode::LoadGlobalPushNull + | OpCode::PushNull + | OpCode::LoadMethodAttr + | OpCode::Call + | OpCode::LoadFastLoadFast + | OpCode::LoadFastBorrowLoadFastBorrow + | OpCode::LoadConst + | OpCode::LoadSmallInt + | OpCode::LoadGlobal + | OpCode::LoadAttr + | OpCode::CompareOp + | OpCode::IsOp + | OpCode::ToBool + | OpCode::UnaryOp + | OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::PopJumpIfNone + | OpCode::PopJumpIfNotNone + | OpCode::JumpForward + | OpCode::BinaryOp + | OpCode::CopyTop + | OpCode::Swap + | OpCode::PopTop + | OpCode::ReturnValue + ) + }; + let ok = structural && code.instructions.iter().all(|i| pure_op(i.op)); + // An effect leaf: a pure leaf but for attribute stores, which its + // evaluation buffers and commits at the return (see + // `Interpreter::pure_leaf_eval`). + let effect = !ok + && structural + && code + .instructions + .iter() + .all(|i| pure_op(i.op) || i.op == OpCode::StoreAttr) + && code.instructions.iter().any(|i| i.op == OpCode::StoreAttr); let shape = if ok { let body = code.instructions.as_slice(); let body = if body.first().is_some_and(|i| i.op == OpCode::Resume) { @@ -61721,11 +62332,72 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { } else { 1 }; + // An effect leaf's one fast shape: the setter, `self.x = ; return None`. + let shape = if effect && effect_setter_shape(code) { + 16 + } else { + shape + }; ext.pure_leaf.store(shape, Relaxed); - code.jit_hint.set_pure_leaf(ok); + if effect { + code.jit_hint.set_effect_leaf(); + } else { + code.jit_hint.set_pure_leaf(ok); + } ok } +/// Whether an effect leaf's body is the setter shape (`16`, see +/// `Interpreter::pure_leaf_eval`): one attribute store of an argument, +/// constant or small int into an argument, then `return None`. +fn effect_setter_shape(code: &CodeObject) -> bool { + let body = code.instructions.as_slice(); + let body = match body.first() { + Some(i) if i.op == OpCode::Resume => &body[1..], + _ => body, + }; + let local = |op| { + matches!( + op, + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck + ) + }; + let tail = match body { + [pair, rest @ ..] + if matches!( + pair.op, + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow + ) => + { + rest + } + [value, recv, rest @ ..] + if (local(value.op) + || matches!(value.op, OpCode::LoadConst | OpCode::LoadSmallInt)) + && local(recv.op) => + { + rest + } + _ => return false, + }; + matches!(tail, [store, none, ret] + if store.op == OpCode::StoreAttr + && none.op == OpCode::LoadConst + && matches!(code.constants.get(none.arg as usize), Some(Constant::None)) + && ret.op == OpCode::ReturnValue) +} + +/// Whether `code` is an *effect leaf* (see `code_pure_leaf_decide`), +/// deciding its leaf verdicts on first use. +#[inline] +fn code_is_effect_leaf(code: &CodeObject) -> bool { + if code.jit_hint.pure_leaf().is_none() { + code_is_pure_leaf(code); + } + code.jit_hint.effect_leaf() +} + /// Whether the instructions from `pc` could open a simple call's /// argument run (see `Interpreter::core_simple_args`): each of the first /// two is a plain operand load or the `CALL` itself. A cheap filter for @@ -61761,6 +62433,15 @@ fn pure_leaf_warm(code: &CodeObject) -> bool { true } +/// Whether `f`'s code is a pure or an effect leaf (the core loop's +/// frameless call candidates). +#[inline(always)] +fn fn_is_leaf(f: &crate::object::PyFunction) -> bool { + // SAFETY: as in `fn_is_pure_leaf`. + let code = unsafe { &*f.code.as_ptr() }; + code_is_pure_leaf(code) || code.jit_hint.effect_leaf() +} + /// Whether function `f`'s current code is a pure leaf (see /// [`code_is_pure_leaf`]). #[inline(always)] @@ -61863,10 +62544,18 @@ fn code_const_objects(code: &CodeObject) -> &[Object] { /// needed the interpreter): does dict key `key` name `co_names[name_idx]`? #[inline] fn slot_name_matches(code: &CodeObject, name_idx: u32, key: &DictKey) -> bool { + let names: &[Object] = code_vm_ext(code).map_or(&[], |t| &t.name_objs); + slot_name_matches_in(names, code, name_idx, key) +} + +/// [`slot_name_matches`] with the code's interned name objects already +/// in hand (`code_vm_ext(code).name_objs`, or empty). +#[inline(always)] +fn slot_name_matches_in(names: &[Object], code: &CodeObject, name_idx: u32, key: &DictKey) -> bool { let Object::Str(s) = &key.0 else { return false; }; - if let Some(Object::Str(n)) = code_name_obj(code, name_idx) { + if let Some(Object::Str(n)) = names.get(name_idx as usize) { if SharedStr::ptr_eq(n, s) { return true; } @@ -65239,7 +65928,7 @@ assert loop(2000) == 1999000 let args = [std::ptr::from_ref(receiver); 2]; eprintln!( " direct evaluator: {:?}, class_version={}, native_kind={}", - interp.pure_leaf_eval::(&code, function, &args), + interp.pure_leaf_eval::(&code, function, &args), inst.cls_raw().attr_version.get(), inst.cls_raw().native_kind.get() ); diff --git a/crates/weavepy-vm/src/stdlib/collections_native.rs b/crates/weavepy-vm/src/stdlib/collections_native.rs index 64b7227d..e8b4f089 100644 --- a/crates/weavepy-vm/src/stdlib/collections_native.rs +++ b/crates/weavepy-vm/src/stdlib/collections_native.rs @@ -35,7 +35,6 @@ use crate::sync::RefCell; use crate::error::{index_error, runtime_error, stop_iteration, type_error, RuntimeError}; use crate::import::ModuleCache; use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; -use crate::types::PyInstance; /// The receiver must be a deque instance (any class whose `_data` slot /// is the backing list). CPython's method descriptor rejects a foreign @@ -172,13 +171,6 @@ fn receiver<'a>(args: &'a [Object], method: &str) -> Result, Runt /// Read-only helper for the iterator paths that still address the /// instance directly. -fn head(inst: &PyInstance) -> usize { - match inst.slot_get("_head") { - Some(Object::Int(h)) if h >= 0 => h as usize, - _ => 0, - } -} - fn popleft_locked(st: &mut DequeState<'_>, d: &mut Vec) -> Result { let mut h = st.head(); if h >= d.len() { @@ -464,39 +456,70 @@ fn deque_reverse_iterator_next(args: &[Object]) -> Result deque_next(args, true) } +// The iterator classes' slot positions as their `__init__` assigns them +// (hints, like the deque's own). +const IT_DEQ: usize = 0; +const IT_INDEX: usize = 1; +const IT_STATE: usize = 2; + fn deque_next(args: &[Object], reverse: bool) -> Result { let [Object::Instance(iterator)] = args else { return Err(type_error("deque iterator __next__ requires one iterator")); }; - let deque = match iterator.slot_get("_deq") { - Some(Object::Instance(deque)) => deque, - Some(Object::None) => return Err(stop_iteration()), - _ => return Err(type_error("deque iterator expected")), + let (deque, index, it_state) = { + let s = iterator.slots.borrow(); + let deque = match s.get_hinted(IT_DEQ, "_deq") { + Some(Object::Instance(deque)) => deque.clone(), + Some(Object::None) => return Err(stop_iteration()), + _ => return Err(type_error("deque iterator expected")), + }; + ( + deque, + s.get_hinted(IT_INDEX, "_index").and_then(Object::as_i64), + s.get_hinted(IT_STATE, "_deq_state") + .and_then(Object::as_i64), + ) }; - let Some(Object::List(data)) = deque.slot_get("_data") else { - return Err(type_error("deque expected")); + let (data, h, state) = { + let s = deque.slots.borrow(); + let Some(Object::List(data)) = s.get_hinted(SLOT_DATA, "_data") else { + return Err(type_error("deque expected")); + }; + ( + data.clone(), + head_of(&s), + s.get_hinted(SLOT_STATE, "_state").and_then(Object::as_i64), + ) }; // Serialize the state check, cursor advance, and item read with deque // end operations, including simultaneous next() calls under gil=0. // The local deque reference outlives the guard, so clearing _deq can't // finalize its items while the backing list is borrowed. let d = data.borrow(); - if deque.slot_get("_state").and_then(|s| s.as_i64()) - != iterator.slot_get("_deq_state").and_then(|s| s.as_i64()) - { + if state != it_state { iterator.slot_set("_deq", Object::None); return Err(runtime_error("deque mutated during iteration")); } - let h = head(&deque).min(d.len()); - let index = iterator - .slot_get("_index") - .and_then(|i| i.as_i64()) - .ok_or_else(|| type_error("invalid deque iterator index"))?; + let h = h.min(d.len()); + let index = index.ok_or_else(|| type_error("invalid deque iterator index"))?; if index < 0 || index as usize >= d.len() - h { iterator.slot_set("_deq", Object::None); return Err(stop_iteration()); } - iterator.slot_set("_index", Object::Int(index + 1)); + let advanced = match iterator + .slots + .borrow_mut() + .get_hinted_mut(IT_INDEX, "_index") + { + Some(slot) => { + *slot = Object::Int(index + 1); + true + } + None => false, + }; + if !advanced { + iterator.slot_set("_index", Object::Int(index + 1)); + } let slot = if reverse { d.len() - 1 - index as usize } else { From afa8bf644f36b9261db095a2a6b76f7218182abe Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 08:07:19 -0700 Subject: [PATCH 12/65] perf: cheaper JIT attribute access on slots and instance dicts - The JIT's attribute helpers read `__slots__` storage with an unguarded peek, as they already did for instance dicts. - An indexed store replacing an existing key's value no longer bumps the dict's stamp (the interpreter's store already used the unstamped and value-store accessors), so stamp-keyed caches survive it. - The callback-free field update (`self.n += k; return self.n`) also serves `__slots__` classes. - Slot keys are interned, so guards settle them by identity and no instance allocates its own copy of each slot name. attr_access: 922M -> 730M instructions. --- crates/weavepy-vm/src/tier2.rs | 43 +++++++++++++++++++++++++++------- crates/weavepy-vm/src/types.rs | 5 +++- 2 files changed, 38 insertions(+), 10 deletions(-) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 57083910..b26d21ea 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -3702,10 +3702,11 @@ fn scalar_field_update_plan( { return None; } - let AttrStorage::Indexed(index) = guard.storage else { - return None; + let current = match guard.storage { + AttrStorage::Indexed(index) => (guard.ver, index, false), + AttrStorage::Slot(index) => (guard.ver, index, true), + AttrStorage::NewKey => return None, }; - let current = (guard.ver, index); if fingerprint.is_some_and(|old| old != current) { return None; } @@ -3816,8 +3817,10 @@ unsafe fn native_scalar_field_update( } let plan = nc.scalar_update.as_deref()?; let guard = nc.attr_guards.get(plan.store_token)?; - let AttrStorage::Indexed(index) = guard.storage else { - return None; + let (index, slot_storage) = match guard.storage { + AttrStorage::Indexed(index) => (index, false), + AttrStorage::Slot(index) => (index, true), + AttrStorage::NewKey => return None, }; let Object::Instance(inst) = receiver else { return None; @@ -3850,6 +3853,22 @@ unsafe fn native_scalar_field_update( return None; } }; + if slot_storage { + // SAFETY: as for the dictionary below. + let slots = unsafe { inst.slots.peek_mut() }?; + let (key, old) = slots.get_index(index as usize)?; + if !key_is(key, &guard.name) { + return None; + } + let Object::Int(old) = old else { + return None; + }; + let value = old.checked_add(increment)?; + let (_, slot) = slots.get_index_mut(index as usize)?; + // An exact integer owns no destructor or GC edge. + *slot = Object::Int(value); + return Some(value); + } // SAFETY: no callback, allocation, or Python execution can overlap this // exclusive view. A shared cell is rejected by peek_mut. let dict = unsafe { inst.dict.get()?.peek_mut() }?; @@ -5114,7 +5133,7 @@ unsafe fn pure_leaf_call( // SAFETY: as above. ptrs[offset + j] = unsafe { written.0.add(j) }; } - let v = interp.pure_leaf_eval::(code, func, &ptrs[..n])?; + let v = interp.pure_leaf_eval::(code, func, &ptrs[..n])?; Some(deliver_call_result(jf, ctx, v, expect_tag)) } @@ -7527,7 +7546,11 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - }; match g.storage { AttrStorage::Slot(key_idx) => { - let slots = inst.slots.borrow(); + // SAFETY: a read between two native ops; nothing here + // runs code (see `GilCell::peek`). + let Some(slots) = (unsafe { inst.slots.peek() }) else { + return 1; + }; let indexed = slots .get_index(key_idx as usize) .filter(|(key, _)| key_is(key, &g.name)) @@ -8140,10 +8163,12 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - let mut dict = inst.dict_cell().borrow_mut(); let atomic = v.is_gc_atomic(); let dict = &mut *dict; + // Replacing an existing key's value leaves the key layout (and + // so the stamp) alone, as the interpreter's indexed store does. let dict = if atomic { - dict.map_mut_atomic_store() + dict.map_mut_unstamped() } else { - &mut **dict + dict.map_mut_value_store() }; let Some((k, dst)) = dict.get_index_mut(key_idx as usize) else { return 1; diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index 52df186a..d20646a1 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -2031,7 +2031,10 @@ fn slot_key(name: &str) -> DictKey { let keys = KEYS.get_or_init(|| COMMON.map(crate::stdlib::sys::intern_name)); return DictKey(keys[i].clone()); } - DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) + // Interned, as instance-dict keys are: a guard holding the interned + // name settles a slot's key by identity, and no instance allocates + // its own copy of the name. + DictKey(crate::stdlib::sys::intern_name(name)) } impl SlotStorage { From fd14f0b3d9032fb19c803e3dbcefd81630c2ad24 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 08:20:01 -0700 Subject: [PATCH 13/65] build: use fat LTO for release builds Cross-crate inlining between the VM, compiler, and JIT crates makes the benchmark fixtures 1-10% faster in wall time (pickle_bench 10%, float_math 6%, deltablue 3%) for about a third more build time. --- Cargo.toml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Cargo.toml b/Cargo.toml index 05e4d0d2..bce0707c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -318,7 +318,9 @@ opt-level = 1 [profile.release] opt-level = 3 -lto = "thin" +# Fat LTO inlines across the VM, compiler, and JIT crates: 1-10% faster +# on the benchmark fixtures for about a third more build time. +lto = "fat" codegen-units = 1 debug = "line-tables-only" strip = "debuginfo" From d56e063deb839fad6250fb46df6cdeaa9cc862d0 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 09:01:19 -0700 Subject: [PATCH 14/65] perf: skip scalar and plain-object children in the reap cascade A dying list of plain objects no longer walks each element's fields: scalars and deferred-tracking instances can't anchor anything the cascade must visit. The scan also reuses the handles it finds instead of looking each child up twice, and hashes ids with the VM's id hasher. --- crates/weavepy-vm/src/lib.rs | 37 ++++++++++++++++++++++++++++-------- 1 file changed, 29 insertions(+), 8 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 4ffa4326..d06f42b3 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -3101,7 +3101,7 @@ impl Interpreter { // unpickler temporary holding the memo) cascades through the // untracked `memo` dict to the tracked argument. The refcount guard // below still filters anything that stays externally reachable. - let mut child_ids: Vec = Vec::new(); + let mut children: Vec> = Vec::new(); // Untracked descendants with live weakrefs: CPython clears an // object's weakrefs at refcount zero whether or not the GC ever // tracked it, but this cascade's "next link" set is (otherwise) @@ -3124,10 +3124,30 @@ impl Interpreter { // (the argument tuple is anchored by a dead list/frame local). let mut pool_candidates: Vec = Vec::new(); let mut scan_through: Vec = vec![obj.clone()]; - let mut scanned: std::collections::HashSet = - std::collections::HashSet::new(); + let mut scanned: std::collections::HashSet< + crate::weakref_registry::ObjectId, + std::hash::BuildHasherDefault, + > = std::collections::HashSet::default(); while let Some(parent) = scan_through.pop() { gc_trace::traverse_object(&parent, &mut |c| { + // Scalars are never tracked, weakly referenced, or + // anchors of anything; a deferred-tracking instance + // holds only scalars and has no collector handle. + // Neither can lead the cascade anywhere, which matters + // for the commonest large death: a list of plain + // objects falling out of scope. + if gc_trace::is_atomic(c) { + return; + } + if let Object::Instance(i) = c { + if i.is_gc_deferred() && i.c_body.get() == 0 { + let cid = crate::weakref_registry::id_of(c); + if crate::weakref_registry::count_for(cid) > 0 && scanned.insert(cid) { + weakref_candidates.push(c.clone()); + } + return; + } + } let cid = crate::weakref_registry::id_of(c); if let Object::Tuple(t) = c { if !t.is_empty() @@ -3138,8 +3158,8 @@ impl Interpreter { pool_candidates.push(c.clone()); } } - if gc_trace::is_tracked(cid) { - child_ids.push(cid); + if let Some(h) = gc_trace::find_handle(cid) { + children.push(h); } else if matches!( c, Object::Tuple(_) @@ -3207,9 +3227,9 @@ impl Interpreter { } // Any tracked child that just lost its last program reference // is the next link in the chain. - for cid in child_ids { - if let Some(h) = gc_trace::find_handle(cid) { - let weak = crate::weakref_registry::strong_clone_count(cid); + for h in children { + if !h.untracked.load(std::sync::atomic::Ordering::Acquire) { + let weak = crate::weakref_registry::strong_clone_count(h.id); // `h.object` is the GC's own strong reference; the // child is dead iff nothing beyond that handle and its // weakref clones still points at it. @@ -4038,6 +4058,7 @@ impl Interpreter { || *budget == 0 || c.is_gc_atomic() || matches!(c, Object::Frame(f) if f.on_stack.get() > 0) + || matches!(c, Object::Instance(i) if i.is_gc_deferred() && i.c_body.get() == 0) { return; } From 8938d8133540457d956b975fd6f1527e5a08487f Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:53:51 -0700 Subject: [PATCH 15/65] perf: store instance attributes in class-shared split layouts An instance that assigns its attributes in its class's usual order keeps them as a plain vector of values whose names live once per class, like CPython's shared keys. Constructing such an instance allocates no hash table, and an attribute keeps the position a __dict__ entry would have, so the indexed attribute caches serve both layouts. Anything that needs the real dictionary (vars(), del, an out-of-order attribute, the C API) materializes it, and the instance keeps it from then on. A three-attribute instance drops from 490 to 249 bytes, and constructors run 15-20% fewer instructions (float_math: -11% instructions, -20% wall). The rarely set native-value slot becomes a pointer-sized write-once cell, so PyInstance grows by only one word. --- crates/weavepy-vm/src/builtins.rs | 6 +- crates/weavepy-vm/src/gc_trace.rs | 21 +- crates/weavepy-vm/src/inst_dict.rs | 989 ++++++++++++++++++ crates/weavepy-vm/src/lazy_arc.rs | 90 ++ crates/weavepy-vm/src/lib.rs | 489 +++++---- crates/weavepy-vm/src/specialize.rs | 12 +- .../src/stdlib/multiprocessing_mod.rs | 4 +- crates/weavepy-vm/src/stdlib/thread_real.rs | 6 +- crates/weavepy-vm/src/stdlib/weakref_real.rs | 4 +- crates/weavepy-vm/src/sync.rs | 2 +- crates/weavepy-vm/src/tier2.rs | 99 +- crates/weavepy-vm/src/types.rs | 37 +- 12 files changed, 1454 insertions(+), 305 deletions(-) create mode 100644 crates/weavepy-vm/src/inst_dict.rs diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index 58dac7d8..a2900c13 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -2655,7 +2655,7 @@ fn slot_sizeof(args: &[Object]) -> Result { return Ok(Object::Int(28 + 4 * ndigits)); } } - 16 + 8 * inst.dict.get().map_or(0, |dict| dict.borrow().len()) as i64 + 16 + 8 * inst.attr_count() as i64 } // CPython's compact-unicode layout (test_str.test_raiseMemError): // ASCII is a 40-byte struct + len+1 one-byte units; anything wider @@ -8549,7 +8549,7 @@ pub(crate) fn make_unbound_super(class: Rc) -> Object d })) .into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -8628,7 +8628,7 @@ pub(crate) fn build_super_proxy( d })) .into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index c43e4323..e0a620a3 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -2749,7 +2749,15 @@ pub fn traverse_object(obj: &Object, visit: &mut dyn FnMut(&Object)) { // `dict -> instance -> class -> method -> __globals__` cycle // in a dead ModuleType namespace would be immortal // (test_module.test_clear_dict_in_ref_cycle). - if let Some(dict) = i.dict.get_shared() { + if i.dict.published().is_none() { + // Split values: the instance's own children. + if let Ok(split) = i.dict.split_cell().try_borrow() { + for (k, v) in split.iter() { + visit(&k.0); + visit(v); + } + } + } else if let Some(dict) = i.dict.get_shared() { let dict_obj = Object::Dict(dict); if is_tracked(id_of(&dict_obj)) { visit(&dict_obj); @@ -3116,12 +3124,21 @@ pub fn clear_object_fields(obj: &Object) -> bool { if let Ok(mut slots) = i.slots.try_borrow_mut() { *slots = crate::types::SlotStorage::default(); } + if i.dict.published().is_none() { + let values = i + .dict + .split_cell() + .try_borrow_mut() + .map(|mut s| s.take()) + .unwrap_or_default(); + drop(values); + } if i.dict.strong_count() > 1 { // Shared `__dict__`: leave its contents to the other // holder (see the doc comment). return false; } - if let Some(dict) = i.dict.get() { + if let Some(dict) = i.dict.published() { if let Ok(mut m) = dict.try_borrow_mut() { m.clear(); } diff --git a/crates/weavepy-vm/src/inst_dict.rs b/crates/weavepy-vm/src/inst_dict.rs new file mode 100644 index 00000000..2ee8e887 --- /dev/null +++ b/crates/weavepy-vm/src/inst_dict.rs @@ -0,0 +1,989 @@ +//! Split instance dictionaries: CPython's shared keys and inline values. +//! +//! An ordinary instance stores its attributes as a plain vector of values +//! whose names live once per class in a [`SharedKeys`] table, as long as +//! it assigns them in the order the class's first instances did (the +//! usual `__init__` shape). Constructing such an instance allocates no +//! hash table, and an attribute's position is the same in every instance, +//! so the indexed caches that address `__dict__` entries by insertion +//! order serve both layouts unchanged. +//! +//! Anything that needs the real `__dict__` (`vars(obj)`, a `del`, an +//! out-of-order or thirty-first attribute, the C API, pickling) goes +//! through [`InstDict::get`] or one of its siblings, which *materializes* +//! it: the values move into an ordinary [`DictData`] published in the +//! instance, and the instance keeps that dictionary from then on, as +//! CPython does. Only the hot paths that know about the split layout +//! avoid that step, so every other path sees exactly the dictionary it +//! always did. + +use crate::object::{DictData, DictKey, Object}; +use crate::shared_value::SharedStr; +use crate::sync::{LazyArc, Rc, RefCell}; +use std::cell::UnsafeCell; +use std::mem::MaybeUninit; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; + +/// The most attribute names a class shares (CPython's `SHARED_KEYS_MAX_SIZE`). +pub const SHARED_KEYS_CAP: usize = 30; + +/// A class's attribute names for its split instance dictionaries, in the +/// order its instances first assigned them. +/// +/// Append-only: a published name never moves or changes, and the table +/// never reallocates, so a reader needs only the published length. +/// Appends happen under the GIL (never in free-threaded mode). +pub struct SharedKeys { + len: AtomicUsize, + /// One bit per published name's Python hash (`hash & 63`): a clear + /// bit proves a name absent without comparing any. + filter: AtomicU64, + keys: [UnsafeCell>; SHARED_KEYS_CAP], + /// Each published name's Python hash. + hashes: [UnsafeCell; SHARED_KEYS_CAP], +} + +// SAFETY: names are published with release/acquire ordering and are +// immutable afterwards; the single writer is serialized by the GIL. +unsafe impl Send for SharedKeys {} +// SAFETY: as above. +unsafe impl Sync for SharedKeys {} + +impl std::fmt::Debug for SharedKeys { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_list() + .entries((0..self.len()).filter_map(|i| self.get(i))) + .finish() + } +} + +impl Default for SharedKeys { + fn default() -> Self { + Self { + len: AtomicUsize::new(0), + filter: AtomicU64::new(0), + keys: [const { UnsafeCell::new(MaybeUninit::uninit()) }; SHARED_KEYS_CAP], + hashes: [const { UnsafeCell::new(0) }; SHARED_KEYS_CAP], + } + } +} + +impl SharedKeys { + /// How many names are published. + #[inline] + pub fn len(&self) -> usize { + self.len.load(Ordering::Acquire) + } + + /// Whether no name is published yet. + #[inline] + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// The `i`th name. + #[inline] + pub fn get(&self, i: usize) -> Option<&DictKey> { + if i < self.len() { + // SAFETY: slots below the published length are initialized + // and never written again. + Some(unsafe { (*self.keys[i].get()).assume_init_ref() }) + } else { + None + } + } + + /// The first of the first `n` names equal to `name`, whose Python + /// hash is `hash`. + #[inline(always)] + fn position_hashed(&self, n: usize, name: &str, hash: i64) -> Option { + if self.filter.load(Ordering::Relaxed) & (1 << (hash & 63)) == 0 { + return None; + } + (0..n.min(self.len())).find(|&i| { + // SAFETY: slot `i` is published (below the length). + let h = unsafe { *self.hashes[i].get() }; + h == hash && self.get(i).is_some_and(|k| key_names_str(k, name)) + }) + } + + /// Publish the `str` name `name` as the next name; `None` when the + /// table is full. The caller holds the GIL outside free-threaded mode. + fn push(&self, name: &SharedStr) -> Option { + let n = self.len.load(Ordering::Relaxed); + if n >= SHARED_KEYS_CAP { + return None; + } + let hash = crate::object::py_str_hash(name); + // SAFETY: slot `n` is unpublished, so no reader can see it, and + // the GIL serializes writers. + unsafe { + (*self.keys[n].get()).write(DictKey(Object::Str(name.clone()))); + *self.hashes[n].get() = hash; + } + self.filter.fetch_or(1 << (hash & 63), Ordering::Relaxed); + self.len.store(n + 1, Ordering::Release); + Some(n) + } +} + +impl Drop for SharedKeys { + fn drop(&mut self) { + let n = *self.len.get_mut(); + for slot in &mut self.keys[..n] { + // SAFETY: the first `n` slots are initialized, and each is + // dropped once here. + unsafe { slot.get_mut().assume_init_drop() }; + } + } +} + +/// `a == b` for attribute names without a call into `memcmp`: they are +/// short and almost always differ in length or the first byte. +#[inline(always)] +fn name_eq(a: &str, b: &str) -> bool { + let (a, b) = (a.as_bytes(), b.as_bytes()); + if a.len() != b.len() { + return false; + } + if a.len() > 16 { + return a == b; + } + let mut i = 0; + while i < a.len() { + if a[i] != b[i] { + return false; + } + i += 1; + } + true +} + +/// Whether the `str` key `key` names `name`: identity first (names are +/// interned on both sides), then contents. +#[inline] +fn key_names(key: &DictKey, name: &SharedStr) -> bool { + matches!(&key.0, Object::Str(s) if SharedStr::ptr_eq(s, name) || name_eq(s, name)) +} + +/// [`key_names`] for a borrowed name. +#[inline] +fn key_names_str(key: &DictKey, name: &str) -> bool { + matches!(&key.0, Object::Str(s) if name_eq(s, name)) +} + +/// The header of an instance's split values allocation; the values +/// follow it. +#[repr(C)] +struct SplitHeader { + /// An owned strong reference to the class's shared names. + keys: *const SharedKeys, + len: u32, + cap: u32, +} + +/// An instance's split attribute values: `values[i]` belongs to +/// `keys[i]`, and the instance holds exactly the first `len` shared +/// names. One pointer wide; the names, length and capacity live in the +/// values allocation's header. +#[derive(Default)] +pub struct SplitValues { + block: Option>, +} + +// SAFETY: the block is owned exclusively; its contents are `Send`/`Sync` +// objects and a shared-keys reference. +unsafe impl Send for SplitValues {} +// SAFETY: as above; shared access is read-only. +unsafe impl Sync for SplitValues {} + +impl std::fmt::Debug for SplitValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_list().entries(self.iter()).finish() + } +} + +impl Drop for SplitValues { + fn drop(&mut self) { + let Some(block) = self.block.take() else { + return; + }; + // SAFETY: the block is ours; its first `len` values are + // initialized and its `keys` holds one strong count (or is null). + unsafe { + let h = block.as_ptr(); + std::ptr::drop_in_place(std::ptr::slice_from_raw_parts_mut( + Self::values_ptr(h), + (*h).len as usize, + )); + if !(*h).keys.is_null() { + drop(Rc::from_raw((*h).keys)); + } + std::alloc::dealloc(h.cast(), Self::layout((*h).cap as usize)); + } + } +} + +impl SplitValues { + #[inline] + fn layout(cap: usize) -> std::alloc::Layout { + std::alloc::Layout::new::() + .extend(std::alloc::Layout::array::(cap).expect("split capacity")) + .expect("split layout") + .0 + .pad_to_align() + } + + /// The first value slot of the block at `h`. + #[inline(always)] + fn values_ptr(h: *mut SplitHeader) -> *mut Object { + // SAFETY: the values start right after the header (both are + // word-aligned), inside the block's allocation. + unsafe { h.add(1).cast::() } + } + + /// The shared names, while any value is (or was) set. + #[inline(always)] + fn keys(&self) -> Option<&SharedKeys> { + let h = self.block?.as_ptr(); + // SAFETY: a non-null `keys` is a live strong reference owned by + // the block, which outlives `&self`. + unsafe { (*h).keys.as_ref() } + } + + /// How many attributes are set. + #[inline(always)] + pub fn len(&self) -> usize { + match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { (*b.as_ptr()).len as usize }, + None => 0, + } + } + + /// Whether no attribute is set. + #[inline(always)] + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// Every value, in assignment order. + #[inline(always)] + pub fn values(&self) -> &[Object] { + match self.block { + // SAFETY: the first `len` values are initialized. + Some(b) => unsafe { + std::slice::from_raw_parts(Self::values_ptr(b.as_ptr()), (*b.as_ptr()).len as usize) + }, + None => &[], + } + } + + /// [`Self::values`], writable. + #[inline(always)] + fn values_mut(&mut self) -> &mut [Object] { + match self.block { + // SAFETY: as above, and `&mut self` is exclusive. + Some(b) => unsafe { + std::slice::from_raw_parts_mut( + Self::values_ptr(b.as_ptr()), + (*b.as_ptr()).len as usize, + ) + }, + None => &mut [], + } + } + + /// Append `value`, growing the block to at least `want` slots; the + /// first append adopts `keys`. + fn push(&mut self, keys: impl FnOnce() -> Rc, want: usize, value: Object) { + let (len, cap) = match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { ((*b.as_ptr()).len as usize, (*b.as_ptr()).cap as usize) }, + None => (0, 0), + }; + if len == cap { + let new_cap = want.max(len + 1).max(cap * 2).max(2); + let new_layout = Self::layout(new_cap); + // SAFETY: a nonzero-size layout; a grown block keeps its + // header and initialized prefix (realloc copies them). + let h = unsafe { + match self.block { + Some(b) => { + std::alloc::realloc(b.as_ptr().cast(), Self::layout(cap), new_layout.size()) + } + None => std::alloc::alloc(new_layout), + } + } + .cast::(); + let Some(h) = std::ptr::NonNull::new(h) else { + std::alloc::handle_alloc_error(new_layout); + }; + // SAFETY: `h` is a live block of `new_cap` slots. + unsafe { + if self.block.is_none() { + h.as_ptr().write(SplitHeader { + keys: std::ptr::null(), + len: 0, + cap: 0, + }); + } + (*h.as_ptr()).cap = new_cap as u32; + } + self.block = Some(h); + } + let h = self.block.expect("allocated above").as_ptr(); + // SAFETY: slot `len` is inside the capacity and uninitialized. + unsafe { + if (*h).keys.is_null() { + (*h).keys = Rc::into_raw(keys()); + } + Self::values_ptr(h).add(len).write(value); + (*h).len = len as u32 + 1; + } + } + + /// The `i`th attribute in assignment order (the index a `__dict__` + /// entry of the same instance would have). + #[inline(always)] + pub fn get_index(&self, i: usize) -> Option<(&DictKey, &Object)> { + let v = self.values().get(i)?; + let k = self.keys()?.get(i)?; + Some((k, v)) + } + + /// [`Self::get_index`] with the value writable in place. + #[inline(always)] + pub fn get_index_mut(&mut self, i: usize) -> Option<(&DictKey, &mut Object)> { + let keys: *const SharedKeys = self.keys()?; + let v = self.values_mut().get_mut(i)?; + // SAFETY: the keys outlive `&mut self` (the block owns them), and + // they are disjoint from the values. + let k = unsafe { &*keys }.get(i)?; + Some((k, v)) + } + + /// The position of attribute `name`. + #[inline] + pub fn position(&self, name: &SharedStr) -> Option { + let keys = self.keys()?; + let n = self.len(); + // The constructor shape: the class's next name is unset here, and + // names are unique, so none of the set ones can match. + if keys.get(n).is_some_and(|k| key_names(k, name)) { + return None; + } + (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names(k, name))) + } + + /// The position of attribute `name` whose Python hash is `hash`: a + /// name no instance of the class ever set answers without comparing. + #[inline(always)] + pub fn position_hashed(&self, name: &str, hash: i64) -> Option { + self.keys()?.position_hashed(self.len(), name, hash) + } + + /// [`Self::position`] for a borrowed name (identity of the bytes + /// first: a name read off an interned string settles without a + /// comparison). + #[inline] + pub fn position_str(&self, name: &str) -> Option { + let keys = self.keys()?; + let n = self.len(); + for i in 0..n { + if let Some(DictKey(Object::Str(s))) = keys.get(i) { + if std::ptr::eq(s.as_ptr(), name.as_ptr()) && s.len() == name.len() { + return Some(i); + } + } + } + (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names_str(k, name))) + } + + /// The value of attribute `name`. + #[inline] + pub fn get(&self, name: &SharedStr) -> Option<&Object> { + self.position(name).map(|i| &self.values()[i]) + } + + /// [`Self::get`] for a borrowed name. + pub fn get_str(&self, name: &str) -> Option<&Object> { + self.position_str(name).map(|i| &self.values()[i]) + } + + /// Every attribute, in assignment order. + pub fn iter(&self) -> impl Iterator { + let keys = self.keys(); + self.values() + .iter() + .enumerate() + .filter_map(move |(i, v)| Some((keys?.get(i)?, v))) + } + + /// A copy of the names and values (for a materialization that can't + /// move them). + fn snapshot(&self) -> (Option>, Vec) { + let keys = self.block.and_then(|b| { + // SAFETY: a non-null `keys` is a live strong reference. + let k = unsafe { (*b.as_ptr()).keys }; + (!k.is_null()).then(|| unsafe { + Rc::increment_strong_count(k); + Rc::from_raw(k) + }) + }); + (keys, self.values().to_vec()) + } + + /// Store `value` under the interned name `name` if the split layout + /// can hold it: an existing attribute is overwritten in place + /// (returning the old value), and a new one is appended when it is + /// the next shared name (or becomes it). `Err` hands the value back + /// when the instance needs a real dictionary instead. + pub fn store( + &mut self, + class_keys: impl FnOnce() -> Rc, + name: &SharedStr, + value: Object, + ) -> Result, Object> { + let n = self.len(); + let adopted; + let keys: &SharedKeys = match self.keys() { + Some(k) => k, + None => { + adopted = class_keys(); + &adopted + } + }; + // The constructor shape first: the class's next name (names are + // unique, so it can't also be among the set ones). + let next = keys.get(n); + if !next.is_some_and(|k| key_names(k, name)) { + // An existing attribute. + if let Some(i) = (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names(k, name))) { + return Ok(Some(std::mem::replace(&mut self.values_mut()[i], value))); + } + // A new name at the end of the table. + if next.is_some() || keys.len() != n || keys.push(name).is_none() { + return Err(value); + } + } + let want = keys.len(); + let keys_ptr: *const SharedKeys = keys; + self.push( + // SAFETY: `keys_ptr` is either the block's own reference or + // `adopted`, both live here; the new strong count is the + // block's. + || unsafe { + Rc::increment_strong_count(keys_ptr); + Rc::from_raw(keys_ptr) + }, + want, + value, + ); + Ok(None) + } + + /// Move every value out (the caller drops them after releasing the + /// cell) and forget the names. + pub fn take(&mut self) -> Vec { + let Some(b) = self.block else { + return Vec::new(); + }; + let h = b.as_ptr(); + // SAFETY: the first `len` values move out exactly once (the length + // is zeroed before anything can observe it), and the names' + // strong count is released once. + unsafe { + let n = (*h).len as usize; + let mut out = Vec::with_capacity(n); + std::ptr::copy_nonoverlapping(Self::values_ptr(h), out.as_mut_ptr(), n); + out.set_len(n); + (*h).len = 0; + let keys = std::mem::replace(&mut (*h).keys, std::ptr::null()); + if !keys.is_null() { + drop(Rc::from_raw(keys)); + } + out + } + } + + /// Drop every value, keeping the allocation, and forget the names. + pub fn reset(&mut self) { + drop(self.take()); + } +} + +/// An instance's `__dict__`: split values until something needs the +/// dictionary itself, then that dictionary (see the module docs). +/// +/// The accessors mirror [`LazyArc`]'s and materialize a split layout +/// first, so a caller that never heard of the split layout sees the same +/// dictionary it always did. +pub struct InstDict { + lazy: LazyArc>, + split: RefCell, +} + +impl Default for InstDict { + fn default() -> Self { + Self::new() + } +} + +impl From>> for InstDict { + fn from(d: Rc>) -> Self { + Self { + lazy: LazyArc::from(d), + split: RefCell::new(SplitValues::default()), + } + } +} + +impl From>> for InstDict { + fn from(lazy: LazyArc>) -> Self { + Self { + lazy, + split: RefCell::new(SplitValues::default()), + } + } +} + +impl Clone for InstDict { + #[track_caller] + fn clone(&self) -> Self { + // A shallow instance copy shares the dictionary itself. + self.get(); + Self::from(self.lazy.clone()) + } +} + +impl std::fmt::Debug for InstDict { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_tuple("InstDict").field(&self.lazy).finish() + } +} + +impl InstDict { + pub fn new() -> Self { + Self { + lazy: LazyArc::new(), + split: RefCell::new(SplitValues::default()), + } + } + + /// The instance that owns this field. + #[inline] + fn owner(&self) -> &crate::types::PyInstance { + let off = std::mem::offset_of!(crate::types::PyInstance, dict); + // SAFETY: an `InstDict` exists only as `PyInstance::dict`, so the + // containing instance starts `off` bytes before it and outlives + // this borrow. + unsafe { + &*std::ptr::from_ref(self) + .cast::() + .sub(off) + .cast::() + } + } + + /// The published dictionary, if any, without materializing. + #[inline] + pub fn published(&self) -> Option<&RefCell> { + self.lazy.get() + } + + /// The split values, exclusively. + #[inline] + pub fn split_mut(&mut self) -> &mut SplitValues { + self.split.get_mut() + } + + /// The split values cell (meaningful while nothing is published). + #[inline] + pub fn split_cell(&self) -> &RefCell { + &self.split + } + + /// The split values, read without borrow bookkeeping: `None` when a + /// dictionary is published, the cell is mutably borrowed, or cells are + /// shared across threads. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]: the view must not outlive + /// anything that could store an attribute or materialize the dict. + #[inline] + pub unsafe fn split_peek(&self) -> Option<&SplitValues> { + if self.lazy.get().is_some() { + return None; + } + // SAFETY: forwarded contract. + unsafe { self.split.peek() } + } + + /// [`Self::split_peek`]'s exclusive form. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline] + #[allow(clippy::mut_from_ref)] + pub unsafe fn split_peek_mut(&self) -> Option<&mut SplitValues> { + if self.lazy.get().is_some() { + return None; + } + // SAFETY: forwarded contract. + unsafe { self.split.peek_mut() } + } + + /// Whether the split layout holds any attribute. + #[inline] + fn has_split(&self) -> bool { + // SAFETY: a length read, finished before anything else runs. + match unsafe { self.split.peek() } { + Some(s) => !s.is_empty(), + None => self.split.try_borrow().map_or(true, |s| !s.is_empty()), + } + } + + /// The dictionary, materializing a split layout; `None` when the + /// instance has no attributes and never had a dictionary. + #[inline] + #[track_caller] + pub fn get(&self) -> Option<&RefCell> { + if let Some(d) = self.lazy.get() { + return Some(d); + } + if !self.has_split() { + return None; + } + Some(self.materialize()) + } + + /// The dictionary, created by `make` when there is none (after + /// materializing a split layout). + #[inline] + #[track_caller] + pub fn get_or_init(&self, make: impl FnOnce() -> Rc>) -> &RefCell { + if let Some(d) = self.get() { + return d; + } + self.lazy.get_or_init(make) + } + + /// Another owner of the dictionary, if one exists (materializing). + #[track_caller] + pub fn get_shared(&self) -> Option>> { + self.get(); + self.lazy.get_shared() + } + + /// Another owner of the dictionary, created with the owner's + /// deferred-tracking record when there is none. + #[track_caller] + pub fn share(&self) -> Rc> { + self.owner().dict_cell(); + self.lazy.share() + } + + /// The published dictionary's strong count (`0` when none is). + pub fn strong_count(&self) -> usize { + self.lazy.strong_count() + } + + /// Move the split values into a published dictionary. + #[cold] + #[inline(never)] + #[track_caller] + fn materialize(&self) -> &RefCell { + note_materialize(std::panic::Location::caller()); + let owner = self.owner(); + let (keys, values) = match self.split.try_borrow_mut() { + Ok(mut s) => { + let keys = s.snapshot().0; + (keys, s.take()) + } + // A live view of the values (nothing in the VM holds one across + // a materializing call): copy them, and leave the stale split + // to the next exclusive reset. Readers check the dictionary + // first. + Err(_) => { + // SAFETY: only shared borrows can be live here; the copy + // reads through the cell without creating a `&mut`. + let s = unsafe { &*self.split.as_ptr() }; + s.snapshot() + } + }; + let n = values.len(); + let mut d = if owner.deferred.get() { + DictData::deferred_for_capacity(std::ptr::from_ref(owner) as usize, n) + } else { + DictData::with_capacity_and_hasher(n, crate::fasthash::FxBuildHasher) + }; + if let Some(keys) = keys { + // A deferred owner holds only atomic values, and a tracked + // one needs no barrier: insert without either. + let map = d.map_mut_atomic_store(); + for (i, v) in values.into_iter().enumerate() { + if let Some(k) = keys.get(i) { + map.insert(k.clone(), v); + } + } + } + self.lazy.get_or_init(|| Rc::new(RefCell::new(d))) + } +} + +/// `WEAVEPY_SPLIT_TRACE=1`: count materializations by call site and +/// report them at exit (to find hot paths that should read the split +/// layout directly). +fn note_materialize(at: &'static std::panic::Location<'static>) { + use std::collections::HashMap; + use std::sync::Mutex; + static ON: std::sync::OnceLock = std::sync::OnceLock::new(); + static COUNTS: Mutex>> = Mutex::new(None); + if !*ON.get_or_init(|| { + let on = std::env::var_os("WEAVEPY_SPLIT_TRACE").is_some(); + if on { + extern "C" fn report() { + let Ok(g) = COUNTS.lock() else { return }; + let Some(m) = g.as_ref() else { return }; + let mut v: Vec<_> = m.iter().collect(); + v.sort_by(|a, b| b.1.cmp(a.1)); + eprintln!("## split dict materializations"); + for (site, n) in v.into_iter().take(40) { + eprintln!("{n:10} {site}"); + } + } + // SAFETY: registering a plain `extern "C"` exit handler. + unsafe { libc::atexit(report) }; + } + on + }) { + return; + } + if let Ok(mut g) = COUNTS.lock() { + *g.get_or_insert_with(HashMap::new) + .entry(format!("{}:{}", at.file(), at.line())) + .or_default() += 1; + } +} + +impl crate::types::PyInstance { + /// The `i`th attribute in assignment order, from whichever layout the + /// instance uses, read without borrow bookkeeping. `None` when it is + /// absent or the storage is borrowed (take the general path). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]: the view ends before anything + /// that could store an attribute runs. + #[inline(always)] + pub unsafe fn attr_peek_index(&self, i: usize) -> Option<(&DictKey, &Object)> { + match self.dict.published() { + // SAFETY: forwarded contract. + Some(d) => unsafe { d.peek() }?.get_index(i), + // SAFETY: forwarded contract. + None => unsafe { self.dict.split_cell().peek() }?.get_index(i), + } + } + + /// [`Self::attr_peek_index`] for an in-place value store: the write + /// barrier for a value of that atomicity runs first (a non-atomic + /// value starts tracking a deferred instance), and the key layout is + /// left alone. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + pub unsafe fn attr_peek_index_mut( + &self, + i: usize, + atomic: bool, + ) -> Option<(&DictKey, &mut Object)> { + match self.dict.published() { + Some(d) => { + // SAFETY: forwarded contract. + let d = unsafe { d.peek_mut() }?; + let map = if atomic { + d.map_mut_unstamped() + } else { + d.map_mut_value_store() + }; + map.get_index_mut(i).map(|(k, v)| (&*k, v)) + } + None => { + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + unsafe { self.dict.split_cell().peek_mut() }?.get_index_mut(i) + } + } + } + + /// `f` of the `i`th attribute (either layout, borrowed safely); `None` + /// when it is absent or the storage is mutably borrowed. + #[inline] + pub fn attr_index_map(&self, i: usize, f: impl FnOnce(&DictKey, &Object) -> R) -> Option { + match self.dict.published() { + Some(d) => { + let d = d.try_borrow().ok()?; + let (k, v) = d.get_index(i)?; + Some(f(k, v)) + } + None => { + let s = self.dict.split_cell().try_borrow().ok()?; + let (k, v) = s.get_index(i)?; + Some(f(k, v)) + } + } + } + + /// Replace the value of the `i`th attribute (either layout, borrowed + /// safely) with `value`, returning the displaced value; `Err` hands + /// `value` back when there is no such attribute or the storage is + /// borrowed. The write barrier for `value` runs first. + pub fn attr_replace_index(&self, i: usize, value: Object) -> Result { + let atomic = value.is_gc_atomic(); + match self.dict.published() { + Some(d) => { + let Ok(mut d) = d.try_borrow_mut() else { + return Err(value); + }; + let map = if atomic { + d.map_mut_unstamped() + } else { + d.map_mut_value_store() + }; + match map.get_index_mut(i) { + Some((_, slot)) => Ok(std::mem::replace(slot, value)), + None => Err(value), + } + } + None => { + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + let Ok(mut s) = self.dict.split_cell().try_borrow_mut() else { + return Err(value); + }; + match s.get_index_mut(i) { + Some((_, slot)) => Ok(std::mem::replace(slot, value)), + None => Err(value), + } + } + } + } + + /// Whether attribute `name` (Python hash `hash`) is set, read without + /// borrow bookkeeping; `None` when that can't be told natively. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline] + pub unsafe fn attr_peek_has(&self, name: &str, hash: i64) -> Option { + match self.dict.published() { + Some(d) => { + // SAFETY: forwarded contract. + let d = unsafe { d.peek() }?; + if d.is_empty() || !d.may_hold_str_hash(hash) { + return Some(false); + } + let probe = crate::object::LeafNameProbe::new(name, hash); + let hit = d.contains_key(&probe); + (!probe.saw_exotic()).then_some(hit) + } + None => { + // SAFETY: forwarded contract. + let s = unsafe { self.dict.split_cell().peek() }?; + Some(s.position_hashed(name, hash).is_some()) + } + } + } + + /// Attribute `name`'s value (either layout), without materializing + /// a split layout and without running code. + pub fn attr_get_str(&self, name: &str) -> Option { + match self.dict.published() { + Some(d) => d.borrow().get(&crate::object::StrKey(name)).cloned(), + None => self.dict.split_cell().borrow().get_str(name).cloned(), + } + } + + /// The position of attribute `name` in assignment order (either + /// layout, no materializing). + pub fn attr_position_str(&self, name: &str) -> Option { + let i = match self.dict.published() { + Some(d) => d + .try_borrow() + .ok()? + .get_index_of(&crate::object::StrKey(name))?, + None => self + .dict + .split_cell() + .try_borrow() + .ok()? + .position_str(name)?, + }; + u32::try_from(i).ok() + } + + /// How many attributes are set (either layout, no materializing). + pub fn attr_count(&self) -> usize { + match self.dict.published() { + Some(d) => d.try_borrow().map_or(0, |d| d.len()), + None => self.dict.split_cell().try_borrow().map_or(0, |s| s.len()), + } + } + + /// Visit every attribute (either layout, no materializing); a + /// borrowed storage is skipped. + pub fn for_each_attr(&self, mut f: impl FnMut(&DictKey, &Object)) { + match self.dict.published() { + Some(d) => { + if let Ok(d) = d.try_borrow() { + for (k, v) in d.iter() { + f(k, v); + } + } + } + None => { + if let Ok(s) = self.dict.split_cell().try_borrow() { + for (k, v) in s.iter() { + f(k, v); + } + } + } + } + } + + /// Store `value` under the interned name `name` through the split + /// layout: the displaced value on an overwrite, or `Err` handing the + /// value back when the instance needs (or has) a real dictionary. + /// Runs no code. The caller has established that a plain `__dict__` + /// store is what this assignment means. + #[inline] + pub fn split_store(&self, name: &SharedStr, value: Object) -> Result, Object> { + if self.dict.published().is_some() + || self.c_body.get() != 0 + || crate::gil::free_threading_enabled() + { + return Err(value); + } + let cls = self.cls_raw(); + if cls.native_kind.get() != 0 { + return Err(value); + } + if self.deferred.get() && !value.is_gc_atomic() { + self.ensure_gc_tracked(); + } + // SAFETY: nothing below runs code or reaches this cell again. + let Some(split) = (unsafe { self.dict.split_cell().peek_mut() }) else { + return Err(value); + }; + split.store(|| cls.shared_keys.share(), name, value) + } +} diff --git a/crates/weavepy-vm/src/lazy_arc.rs b/crates/weavepy-vm/src/lazy_arc.rs index a4c07ed8..728983c2 100644 --- a/crates/weavepy-vm/src/lazy_arc.rs +++ b/crates/weavepy-vm/src/lazy_arc.rs @@ -160,6 +160,96 @@ impl Drop for LazyArc { } } +/// A write-once value behind one pointer: `OnceLock` for a field that +/// is rarely set, where the lock's state word and an inline `T` would +/// enlarge every owner. +pub struct OnceBox { + pointer: AtomicPtr, + owner: PhantomData>, +} + +// SAFETY: publication is a single release compare-exchange of an owned +// box, and readers only take shared references; the published value is +// never replaced through shared access. +unsafe impl Sync for OnceBox {} +// SAFETY: as above. +unsafe impl Send for OnceBox {} + +impl OnceBox { + pub const fn new() -> Self { + Self { + pointer: AtomicPtr::new(ptr::null_mut()), + owner: PhantomData, + } + } + + /// The value, if one was set. + #[inline] + pub fn get(&self) -> Option<&T> { + // SAFETY: a published pointer owns a box that lives until `&mut` + // access destroys it. + unsafe { self.pointer.load(Ordering::Acquire).as_ref() } + } + + /// Set the value once; a second set hands the value back. + pub fn set(&self, value: T) -> Result<(), T> { + let fresh = Box::into_raw(Box::new(value)); + match self.pointer.compare_exchange( + ptr::null_mut(), + fresh, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => Ok(()), + // SAFETY: the unpublished box is still ours alone. + Err(_) => Err(*unsafe { Box::from_raw(fresh) }), + } + } + + /// Remove the value. + pub fn take(&mut self) -> Option { + let pointer = std::mem::replace(self.pointer.get_mut(), ptr::null_mut()); + // SAFETY: exclusive access; the published box is released once. + (!pointer.is_null()).then(|| *unsafe { Box::from_raw(pointer) }) + } +} + +impl Default for OnceBox { + fn default() -> Self { + Self::new() + } +} + +impl From for OnceBox { + fn from(value: T) -> Self { + Self { + pointer: AtomicPtr::new(Box::into_raw(Box::new(value))), + owner: PhantomData, + } + } +} + +impl Clone for OnceBox { + fn clone(&self) -> Self { + match self.get() { + Some(v) => Self::from(v.clone()), + None => Self::new(), + } + } +} + +impl fmt::Debug for OnceBox { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_tuple("OnceBox").field(&self.get()).finish() + } +} + +impl Drop for OnceBox { + fn drop(&mut self) { + drop(self.take()); + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index d06f42b3..4de1c60b 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -59,6 +59,7 @@ pub mod hot_filter; pub mod hot_gates; pub mod import; pub mod import_time; +pub mod inst_dict; mod lazy_arc; pub mod linejump; pub mod malloc_stats; @@ -2796,11 +2797,20 @@ impl Interpreter { let mut n = 0; // A dict shared through `vars(obj)` outlives the instance, and // with it every value. - let Some(dict) = inst.dict.get().filter(|_| inst.dict.strong_count() == 1) else { - return Some((leaves, n)); + let split; + let d; + let attrs: &mut dyn Iterator = match inst.dict.published() { + Some(dict) if inst.dict.strong_count() == 1 => { + d = dict.try_borrow().ok()?; + &mut d.iter() + } + Some(_) => return Some((leaves, n)), + None => { + split = inst.dict.split_cell().try_borrow().ok()?; + &mut split.iter() + } }; - let d = dict.try_borrow().ok()?; - for (k, v) in d.iter() { + for (k, v) in attrs { if !gc_trace::is_atomic(&k.0) { return None; } @@ -11916,7 +11926,6 @@ impl Interpreter { }; if pure_site { if let Some((v, call_pc)) = self.core_pure_method( - ext, code, other, pc + 1, @@ -13428,24 +13437,10 @@ impl Interpreter { if !Self::default_getattribute(cls) { break Some(CoreExit::Helper); } - // The instance dict must not shadow the method. - if let Some(dict) = inst.dict.get() { - // SAFETY: a read between two instructions - // (see `GilCell::peek`). - let Some(d) = (unsafe { dict.peek() }) else { - break Some(CoreExit::Helper); - }; - if !d.is_empty() { - let Some(probe) = code_name_leaf_probe(code, ins.arg) - else { - break Some(CoreExit::Helper); - }; - if d.may_hold_str_hash(probe.hash) - && (d.contains_key(&probe) || probe.saw_exotic()) - { - break Some(CoreExit::Helper); - } - } + // The instance's attributes must not shadow + // the method. + if inst_may_shadow(inst, code, ins.arg) { + break Some(CoreExit::Helper); } let f = match ms.get_held(ver) { Some(f) => Object::Function(f), @@ -13572,10 +13567,8 @@ impl Interpreter { } // SAFETY: a read between two instructions (see // `GilCell::peek`). - let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek() }) else { - break Some(CoreExit::Helper); - }; - let Some((k, v)) = d.get_index(key_idx as usize) else { + let Some((k, v)) = (unsafe { inst.attr_peek_index(key_idx as usize) }) + else { break Some(CoreExit::Helper); }; if !slot_name_matches(code, ins.arg, k) { @@ -14525,7 +14518,6 @@ impl Interpreter { #[allow(clippy::too_many_arguments)] fn core_pure_method( &self, - ext: Option<&CodeConstObjects>, code: &CodeObject, recv: &Object, attr_pc: usize, @@ -14561,32 +14553,9 @@ impl Interpreter { 1, ) }?; - // The instance dict must not shadow the method. - if let Some(dict) = inst.dict.get() { - // SAFETY: a read with nothing running (see `GilCell::peek`). - let d = unsafe { dict.peek() }?; - if !d.is_empty() { - // The interned name and its hash, from the loaded table. - let idx = name_idx as usize; - let named = - ext.and_then(|t| match (t.name_objs.get(idx), t.name_hashes.get(idx)) { - (Some(Object::Str(n)), Some(h)) => Some((&**n, *h)), - _ => None, - }); - let (name, hash) = match named { - Some(nh) => nh, - None => { - let k = code_name_key(code, name_idx)?; - (k.s, k.hash) - } - }; - if d.may_hold_str_hash(hash) { - let probe = crate::object::LeafNameProbe::new(name, hash); - if d.contains_key(&probe) || probe.saw_exotic() { - return None; - } - } - } + // The instance's attributes must not shadow the method. + if inst_may_shadow(inst, code, name_idx) { + return None; } let r = self.core_pure_fused_eval(code, fp, call_pc, true, &mut args, nargs + 1, depth_cell)?; @@ -14855,8 +14824,7 @@ impl Interpreter { } if !is_slot { // SAFETY: the caller keeps this rooted read callback-free. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(key_idx as usize)?; + let (key, value) = unsafe { inst.attr_peek_index(key_idx as usize) }?; slot_name_matches_in(names, code, name_idx, key).then_some(value) } else { // SAFETY: the same rooted read as the dictionary path. @@ -15532,22 +15500,10 @@ impl Interpreter { return None; } let fp = ms.peek_fn(cls.attr_version.get())?; - // The instance dict must not shadow the method. - if let Some(dict) = inst.dict.get() { - // SAFETY: a read with nothing running. - let d = unsafe { dict.peek() }?; - let idx = ins.arg as usize; - let (Some(Object::Str(name)), Some(&hash)) = - (ext.name_objs.get(idx), ext.name_hashes.get(idx)) - else { - return None; - }; - if !d.is_empty() && d.may_hold_str_hash(hash) { - let probe = crate::object::LeafNameProbe::new(name, hash); - if d.contains_key(&probe) || probe.saw_exotic() { - return None; - } - } + // The instance's attributes must not shadow + // the method. + if inst_may_shadow(inst, code, ins.arg) { + return None; } push!(V::Fn(fp)); push!(V::R(p)); @@ -16286,8 +16242,7 @@ impl Interpreter { return None; } // SAFETY: a read between two instructions (see `peek`). - let d = unsafe { inst.dict.get()?.peek() }?; - let (k, v) = d.get_index(key_idx as usize)?; + let (k, v) = unsafe { inst.attr_peek_index(key_idx as usize) }?; return slot_name_matches_in(names, code, name_idx, k).then(|| Self::clone_operand(v)); } Self::leaf_fused_local_attr(code, local, attr_pc, name_idx) @@ -16345,11 +16300,9 @@ impl Interpreter { } _ => { // SAFETY: the same rooted, callback-free walk as above. - let Some(dict) = inst.dict.get().and_then(|d| unsafe { d.peek() }) else { - break; - }; let indexed = |index| { - dict.get_index(index as usize) + // SAFETY: as above. + unsafe { inst.attr_peek_index(index as usize) } .filter(|(key, _)| slot_name_matches(code, ins.arg, key)) .map(|(_, value)| value) }; @@ -16423,24 +16376,16 @@ impl Interpreter { { return false; } + // Replacing a value leaves the keys (and so the stamp) alone. // SAFETY: a read between two instructions (see `peek`). - let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek_mut() }) else { + let Some((k, slot)) = + (unsafe { inst.attr_peek_index_mut(key_idx as usize, value.is_gc_atomic()) }) + else { return false; }; - match d.get_index(key_idx as usize) { - Some((k, old)) - if slot_name_matches(code, name_idx, k) && Self::core_droppable(old) => {} - _ => return false, - } - // Replacing a value leaves the keys (and so the stamp) alone. - let map = if value.is_gc_atomic() { - d.map_mut_unstamped() - } else { - d.map_mut_value_store() - }; - let Some((_, slot)) = map.get_index_mut(key_idx as usize) else { + if !slot_name_matches(code, name_idx, k) || !Self::core_droppable(slot) { return false; - }; + } // SAFETY: the value moves out of the caller's stack slot // (the caller drops the slot without dropping the value) // and the displaced value was checked droppable above. @@ -16479,18 +16424,24 @@ impl Interpreter { } match code.caches.get(attr_pc as u32) { IC::StoreAttrInstance { key_idx, .. } => { - // SAFETY: an unused view with nothing running (the store's - // own `peek_mut` condition: no borrow at all). - let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek_mut() }) else { - return false; - }; - matches!(d.get_index(key_idx as usize), Some((k, old)) + // SAFETY: an unused view with nothing running. + matches!(unsafe { inst.attr_peek_index(key_idx as usize) }, Some((k, old)) if slot_name_matches(code, name_idx, k) && Self::core_droppable(old)) } _ => { - // No dictionary yet: the store creates it and inserts. - let Some(dict) = inst.dict.get() else { - return true; + let Some(dict) = inst.dict.published() else { + // Split storage: the store overwrites a droppable + // value or appends (or declines, touching nothing). + // SAFETY: as above. + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return false; + }; + let Some(Object::Str(name)) = code_name_obj(code, name_idx) else { + return false; + }; + return split + .position(name) + .is_none_or(|i| Self::core_droppable(&split.values()[i])); }; let Some(probe) = code_name_leaf_probe(code, name_idx) else { return false; @@ -16531,6 +16482,33 @@ impl Interpreter { { return false; } + // The split layout: no hash table at all while the instance + // assigns its attributes in its class's usual order. + if inst.dict.published().is_none() { + let Some(Object::Str(name)) = code_name_obj(code, name_idx) else { + return false; + }; + // SAFETY: a read between two instructions (see `peek`). + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return false; + }; + if split + .position(name) + .is_some_and(|i| !Self::core_droppable(&split.values()[i])) + { + return false; + } + // SAFETY: the value moves out of the caller's stack slot (the + // caller drops the slot without dropping the value); a + // declined store hands the bits back, still owned there. + match inst.split_store(name, unsafe { std::ptr::read(value) }) { + Ok(old) => { + drop(old); + return true; + } + Err(v) => std::mem::forget(v), + } + } let atomic = value.is_gc_atomic(); // The class facts were validated when the site specialized and // `ver` guards them since. @@ -19114,9 +19092,9 @@ impl Interpreter { if inst.cls_raw().attr_version.get() != ver { return None; } - let dict = inst.dict.get()?.try_borrow().ok()?; - let (k, v) = dict.get_index(key_idx as usize)?; - slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + inst.attr_index_map(key_idx as usize, |k, v| { + slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + })? } (IC::LoadAttrSlot { key_idx, ver }, Object::Instance(inst)) => { if inst.cls_raw().attr_version.get() != ver { @@ -19180,12 +19158,8 @@ impl Interpreter { LeafAttr::InstanceOnly => { // A data descriptor or a real field wins over __getattr__. // This probe must prove a miss without comparing Python keys. - if let Some(dict) = inst.dict.get() { - let probe = code_name_leaf_probe(code, name_idx)?; - let dict = dict.try_borrow().ok()?; - if dict.contains_key(&probe) || probe.saw_exotic() { - return None; - } + if inst_may_shadow(inst, code, name_idx) { + return None; } let name = code_name_obj(code, name_idx)?; (cls.lookup("__getattr__")?, Some(name)) @@ -19321,18 +19295,33 @@ impl Interpreter { prop @ LeafAttr::Property(_) => return Some((prop, None)), other => Some(other), }; - if let Some(dict) = inst.dict.get() { - let probe = code_name_leaf_probe(code, name_idx)?; - let d = dict.try_borrow().ok()?; - if let Some((ix, _, v)) = d.get_full(&probe) { - return Some(( - LeafAttr::Value(Self::clone_operand(v)), - u32::try_from(ix).ok(), - )); + match inst.dict.published() { + Some(dict) => { + let probe = code_name_leaf_probe(code, name_idx)?; + let d = dict.try_borrow().ok()?; + if let Some((ix, _, v)) = d.get_full(&probe) { + return Some(( + LeafAttr::Value(Self::clone_operand(v)), + u32::try_from(ix).ok(), + )); + } + // A key only a user `__eq__` could compare: the full path. + if probe.saw_exotic() { + return None; + } } - // A key only a user `__eq__` could compare: the full path. - if probe.saw_exotic() { - return None; + None => { + let split = inst.dict.split_cell().try_borrow().ok()?; + let ix = match code_name_obj(code, name_idx) { + Some(Object::Str(n)) => split.position(n), + _ => split.position_str(name.as_str()), + }; + if let Some(ix) = ix { + return Some(( + LeafAttr::Value(Self::clone_operand(&split.values()[ix])), + u32::try_from(ix).ok(), + )); + } } } on_class.map(|a| (a, None)) @@ -19611,15 +19600,8 @@ impl Interpreter { _ => None, }; } - if let Some(dict) = inst.dict.get() { - let d = dict.try_borrow().ok()?; - if !d.is_empty() { - let probe = code_name_leaf_probe(code, name_idx)?; - if d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) - { - return None; - } - } + if inst_may_shadow(inst, code, name_idx) { + return None; } let slot = code_method_slot(code, cache_pc); if let Some(f) = slot.and_then(|s| s.get(ver)) { @@ -19717,14 +19699,7 @@ impl Interpreter { // because the site never reaches it). if matches!(cache, IC::Empty) { let ver = cls.attr_version.get(); - let indexed = inst - .dict - .get() - .and_then(|d| d.try_borrow().ok()) - .and_then(|d| { - use crate::specialize::DictDataExt; - d.index_of_key_str(name) - }); + let indexed = inst.attr_position_str(name); code.caches.set( cache_pc, match indexed { @@ -19738,10 +19713,10 @@ impl Interpreter { }; match cache { IC::StoreAttrInstance { key_idx, .. } => { - let dict = inst.dict.get()?; { - let d = dict.try_borrow().ok()?; - let (k, old) = d.get_index(key_idx as usize)?; + // SAFETY: a leaf store runs no code while the + // views below are live. + let (k, old) = unsafe { inst.attr_peek_index(key_idx as usize) }?; if !slot_name_matches(code, name_idx, k) { return None; } @@ -19751,18 +19726,48 @@ impl Interpreter { return None; } } - let mut d = dict.try_borrow_mut().ok()?; - let d = &mut *d; // As the core loop's arm: an in-place value store. - let d = if value_slot.is_gc_atomic() { - d.map_mut_unstamped() - } else { - d.map_mut_value_store() - }; - let (_, slot) = d.get_index_mut(key_idx as usize)?; + // SAFETY: as above. + let (_, slot) = unsafe { + inst.attr_peek_index_mut(key_idx as usize, value_slot.is_gc_atomic()) + }?; let val = std::mem::replace(value_slot, Object::Unbound); Some(std::mem::replace(slot, val)) } + IC::StoreAttrNewKey { .. } if inst.dict.published().is_none() => { + // The split layout (see `core_store_new_attr`). + let Some(Object::Str(shared)) = code_name_obj(code, name_idx) else { + return None; + }; + { + // SAFETY: as above. + let split = unsafe { inst.dict.split_cell().peek() }?; + if let Some(i) = split.position(shared) { + let old = &split.values()[i]; + if Self::local_needs_prompt_reap(old) + && Self::looks_reapable_temporary(old) + { + return None; + } + } + } + let val = std::mem::replace(value_slot, Object::Unbound); + match inst.split_store(shared, val) { + Ok(old) => old, + Err(val) => { + // Needs the real dictionary: store there. + let dict = inst.dict_cell(); + let mut d = dict.try_borrow_mut().ok()?; + let d = &mut *d; + let d = if val.is_gc_atomic() { + d.map_mut_atomic_store() + } else { + &mut **d + }; + d.insert(DictKey(Object::Str(shared.clone())), val) + } + } + } IC::StoreAttrNewKey { .. } => { let probe = code_name_leaf_probe(code, name_idx)?; // `dict_cell`, not a bare `get_or_init`: see @@ -20473,21 +20478,15 @@ impl Interpreter { cls.attr_version.get() == ver }; if guard_ok { - inst.dict.get().and_then(|dict| { - let dict = dict.borrow(); - match dict.get_index(key_idx as usize) { - Some((k, v)) - if self.cached_slot_name_matches( - &frame.code, - attr_ins.arg, - k, - ) => - { - Some(Self::clone_operand(v)) - } - _ => None, - } + inst.attr_index_map(key_idx as usize, |k, v| { + self.cached_slot_name_matches( + &frame.code, + attr_ins.arg, + k, + ) + .then(|| Self::clone_operand(v)) }) + .flatten() } else { None } @@ -26986,10 +26985,8 @@ impl Interpreter { } // (2) Instance dict. - if let Some(dict) = inst.dict.get() { - if let Some(v) = dict.borrow().get(&crate::object::StrKey(name)) { - return Ok(v.clone()); - } + if let Some(v) = inst.attr_get_str(name) { + return Ok(v); } // (3) Non-data descriptor / function on class. @@ -38124,11 +38121,8 @@ impl Interpreter { if cls.attr_version.get() != ver { return None; } - if let Some(dict) = inst.dict.get() { - let d = dict.borrow(); - if !d.is_empty() && d.contains_key(&code_name_key(&frame.code, name_idx)?) { - return None; - } + if inst_may_shadow(inst, &frame.code, name_idx) { + return None; } let slot = code_method_slot(&frame.code, cache_pc); if let Some(f) = slot.and_then(|s| s.get(ver)) { @@ -38194,17 +38188,12 @@ impl Interpreter { cls.attr_version.get() == ver }; if guard_ok { - let hit = inst.dict.get().and_then(|dict| { - let dict = dict.borrow(); - match dict.get_index(key_idx as usize) { - Some((k, v)) - if self.cached_slot_name_matches(&frame.code, name_idx, k) => - { - Some(Self::clone_operand(v)) - } - _ => None, - } - }); + let hit = inst + .attr_index_map(key_idx as usize, |k, v| { + self.cached_slot_name_matches(&frame.code, name_idx, k) + .then(|| Self::clone_operand(v)) + }) + .flatten(); if let Some(v) = hit { specialize::record_hit(op_idx); return Ok(v); @@ -38286,13 +38275,7 @@ impl Interpreter { // differ, so probe when non-empty (with the // name's memoised hash — no siphash, no // allocation). - let shadowed = inst.dict.get().is_some_and(|dict| { - let d = dict.borrow(); - !d.is_empty() - && code_name_key(&frame.code, name_idx).is_none_or(|k| { - d.may_hold_str_hash(k.hash) && d.contains_key(&k) - }) - }); + let shadowed = inst_may_shadow(inst, &frame.code, name_idx); if !shadowed { let slot = code_method_slot(&frame.code, cache_pc); match slot.and_then(|s| s.get(ver)) { @@ -38341,17 +38324,7 @@ impl Interpreter { // it (pandas `MultiIndex.__new__` writes `_names` // over a class-level default), so probe when // non-empty — one hash of the interned name. - let shadowed = inst.dict.get().is_some_and(|dict| { - match code_name_key(&frame.code, name_idx) { - Some(key) => { - let d = dict.borrow(); - !d.is_empty() - && d.may_hold_str_hash(key.hash) - && d.contains_key(&key) - } - None => true, - } - }); + let shadowed = inst_may_shadow(inst, &frame.code, name_idx); if shadowed { None } else { @@ -38496,12 +38469,11 @@ impl Interpreter { // Validate the cached index still holds *this* name: // a `del` on an earlier attribute shift-renumbers // every later slot (same guard as LOAD_ATTR). - let name_ok = { - let dict = inst.dict_cell().borrow(); - dict.get_index(key_idx as usize).is_some_and(|(k, _)| { + let name_ok = inst + .attr_index_map(key_idx as usize, |k, _| { self.cached_slot_name_matches(&frame.code, name_idx, k) }) - }; + .unwrap_or(false); if name_ok { let val = frame.pop()?; // Mirror the slow path: a bound method stored @@ -38516,11 +38488,13 @@ impl Interpreter { // earlier read-only check has been dropped, and // scope it tightly so it is released before the // prompt-reap cascade (which can run `__del__`). - let old = inst - .dict_cell() - .borrow_mut() - .get_index_mut(key_idx as usize) - .map(|(_, slot)| std::mem::replace(slot, val)); + let old = match inst.attr_replace_index(key_idx as usize, val) { + Ok(old) => Some(old), + Err(val) => { + frame.push(val); + None + } + }; if let Some(old) = old { specialize::record_hit(op_idx); // CPython decrefs the overwritten value now; @@ -38570,7 +38544,19 @@ impl Interpreter { // the site takes the indexed shape from here on // (the core loop's `STORE_ATTR` arm serves it). let mut upgrade_idx = false; - let old = { + let mut val = Some(val); + let old = 'store: { + // The split layout (see `core_store_new_attr`). + if watch_value.is_none() && inst.dict.published().is_none() { + if let Some(Object::Str(n)) = code_name_obj(code, name_idx) { + let v = val.take().expect("value not consumed"); + match inst.split_store(n, v) { + Ok(old) => break 'store old, + Err(v) => val = Some(v), + } + } + } + let val = val.take().expect("value not consumed"); let mut dict = inst.dict_cell().borrow_mut(); let dict = &mut *dict; let dict = if val.is_gc_atomic() { @@ -40576,7 +40562,21 @@ impl Interpreter { } else { None }; - let old = { + let key = crate::stdlib::sys::intern_name(name); + let mut value = Some(value); + let old = 'store: { + // The split layout, while the instance keeps one (and no + // watcher needs to see a real dictionary change). + if watch_value.is_none() && inst.dict.published().is_none() { + if let Object::Str(n) = &key { + let v = value.take().expect("value not consumed"); + match inst.split_store(n, v) { + Ok(old) => break 'store old, + Err(v) => value = Some(v), + } + } + } + let value = value.take().expect("value not consumed"); let mut dict = inst.dict_cell().borrow_mut(); let dict = &mut *dict; let dict = if value.is_gc_atomic() { @@ -40584,7 +40584,7 @@ impl Interpreter { } else { &mut **dict }; - dict.insert(DictKey(crate::stdlib::sys::intern_name(name)), value) + dict.insert(DictKey(key), value) }; // Instance `__dict__` is a real dict; a watched one observes // attribute stores as ADDED/MODIFIED (test_watchers @@ -57426,11 +57426,15 @@ fn attr_certainly_missing(obj: &Object, name: &str) -> bool { { return false; } - match inst.dict.get() { + match inst.dict.published() { Some(d) => d .try_borrow() .is_ok_and(|d| !d.contains_key(&crate::object::StrKey(name))), - None => true, + None => inst + .dict + .split_cell() + .try_borrow() + .is_ok_and(|s| s.position_str(name).is_none()), } } @@ -57978,8 +57982,12 @@ fn exception_holds_nonatomic(obj: &Object) -> bool { return true; } } - let Some(dict) = inst.dict.get() else { - return false; + let Some(dict) = inst.dict.published() else { + return inst + .dict + .split_cell() + .try_borrow() + .map_or(true, |s| s.values().iter().any(nonatomic)); }; let Ok(dict) = dict.try_borrow() else { // Borrowed elsewhere (shouldn't happen on a just-built instance); err @@ -61640,10 +61648,8 @@ impl AttrPoly { #[inline] fn hit(&self, code: &CodeObject, inst: &PyInstance, ver: u64, name_idx: u32) -> Option { let ix = self.index(ver)?; - let dict = inst.dict.get()?; // SAFETY: a read between two instructions (see `GilCell::peek`). - let d = unsafe { dict.peek() }?; - let (k, v) = d.get_index(ix as usize)?; + let (k, v) = unsafe { inst.attr_peek_index(ix as usize) }?; slot_name_matches(code, name_idx, k).then(|| Interpreter::clone_operand(v)) } @@ -62669,6 +62675,49 @@ fn code_name_obj(code: &CodeObject, name_idx: u32) -> Option<&Object> { code_vm_ext(code).and_then(|t| t.name_objs.get(name_idx as usize)) } +/// Whether `inst`'s own attributes may shadow `co_names[name_idx]` (a +/// method cache's guard): `false` proves the name absent, and `true` also +/// answers when that can't be told without borrowing. +#[inline(always)] +fn inst_may_shadow(inst: &PyInstance, code: &CodeObject, name_idx: u32) -> bool { + match inst.dict.published() { + Some(dict) => dict_may_shadow(dict, code, name_idx), + None => { + // SAFETY: a read between two instructions (see `GilCell::peek`). + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return true; + }; + if split.is_empty() { + return false; + } + let i = name_idx as usize; + match code_vm_ext(code).map(|t| (t.name_objs.get(i), t.name_hashes.get(i))) { + Some((Some(Object::Str(n)), Some(&hash))) => { + split.position_hashed(n, hash).is_some() + } + _ => code_name_key(code, name_idx) + .is_none_or(|k| split.position_hashed(k.s, k.hash).is_some()), + } + } + } +} + +/// [`inst_may_shadow`] for an instance with a real dictionary. +#[inline(never)] +fn dict_may_shadow(dict: &RefCell, code: &CodeObject, name_idx: u32) -> bool { + // SAFETY: a read between two instructions (see `GilCell::peek`). + let Some(d) = (unsafe { dict.peek() }) else { + return true; + }; + if d.is_empty() { + return false; + } + let Some(probe) = code_name_leaf_probe(code, name_idx) else { + return true; + }; + d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) +} + /// A pre-hashed, Python-free probe for `co_names[name_idx]` (see /// [`crate::object::LeafNameProbe`]). #[inline] diff --git a/crates/weavepy-vm/src/specialize.rs b/crates/weavepy-vm/src/specialize.rs index bb161da9..de5bd4fd 100644 --- a/crates/weavepy-vm/src/specialize.rs +++ b/crates/weavepy-vm/src/specialize.rs @@ -171,10 +171,8 @@ pub fn attempt_specialize_load_attr(obj: &Object, name: &str) -> InlineCache { } // First check the instance dict — that's the // `LoadAttrInstance` shape. - if let Some(dict) = inst.dict.get() { - if let Some(idx) = dict.borrow().index_of_key_str(name) { - return InlineCache::LoadAttrInstance { key_idx: idx, ver }; - } + if let Some(idx) = inst.attr_position_str(name) { + return InlineCache::LoadAttrInstance { key_idx: idx, ver }; } // Not on the instance: resolve through the MRO. A plain // Python function anywhere on it is the *method* shape — @@ -322,10 +320,8 @@ pub fn attempt_specialize_store_attr(obj: &Object, name: &str) -> InlineCache { ) { return InlineCache::Cooldown(COOLDOWN); } - if let Some(dict) = inst.dict.get() { - if let Some(idx) = dict.borrow().index_of_key_str(name) { - return InlineCache::StoreAttrInstance { key_idx: idx, ver }; - } + if let Some(idx) = inst.attr_position_str(name) { + return InlineCache::StoreAttrInstance { key_idx: idx, ver }; } // Key not present: the constructor pattern (`self.x = …` on a // fresh instance). Specialize to a single-probe insert when diff --git a/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs b/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs index 7db1b355..8330be53 100644 --- a/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs +++ b/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs @@ -592,7 +592,7 @@ fn make_semlock_instance(inner: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(semlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -1225,7 +1225,7 @@ fn nt_make_semlock_instance(inner: &Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(nt_semlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/stdlib/thread_real.rs b/crates/weavepy-vm/src/stdlib/thread_real.rs index e67d8690..06eebcda 100644 --- a/crates/weavepy-vm/src/stdlib/thread_real.rs +++ b/crates/weavepy-vm/src/stdlib/thread_real.rs @@ -708,7 +708,7 @@ fn make_lock_object(lock: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(lock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -893,7 +893,7 @@ fn make_rlock_object(rlock: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(rlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -1831,7 +1831,7 @@ fn make_thread_handle_object(state: Arc, ident: Object) -> Ob let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(thread_handle_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/stdlib/weakref_real.rs b/crates/weavepy-vm/src/stdlib/weakref_real.rs index d312ddb3..3eeda5a7 100644 --- a/crates/weavepy-vm/src/stdlib/weakref_real.rs +++ b/crates/weavepy-vm/src/stdlib/weakref_real.rs @@ -1202,8 +1202,8 @@ fn make_ref_object_with_class( let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(class), - dict, - native: std::sync::OnceLock::new(), + dict: dict.into(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(!fixed_wrapper), slots: crate::sync::RefCell::new(slots), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/sync.rs b/crates/weavepy-vm/src/sync.rs index 311c5ee1..1d8b2ef0 100644 --- a/crates/weavepy-vm/src/sync.rs +++ b/crates/weavepy-vm/src/sync.rs @@ -26,7 +26,7 @@ //! The RFC 0024 surface (real lock / event / barrier primitives //! that back `threading.Lock` etc.) lives below the new aliases. -pub use crate::lazy_arc::LazyArc; +pub use crate::lazy_arc::{LazyArc, OnceBox}; use std::cell::UnsafeCell; use std::fmt; diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index b26d21ea..8d0b7625 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -2770,20 +2770,12 @@ fn attr_fingerprint_obj( // RFC 0071 WS2 — a new-key store has no current value by // definition: the `Unknown` lane tells the analyzer to type the // site from the stored value instead. - let slot_val; - let dict; - let v: &Object = match storage { + let current = match storage { AttrStorage::NewKey => return Some((JitType::Unknown, ver, storage)), - AttrStorage::Slot(_) => { - slot_val = inst.slot_get(name)?; - &slot_val - } - AttrStorage::Indexed(key_idx) => { - dict = inst.dict.get()?.borrow(); - let (_, v) = dict.get_index(key_idx as usize)?; - v - } + AttrStorage::Slot(_) => inst.slot_get(name)?, + AttrStorage::Indexed(key_idx) => inst.attr_index_map(key_idx as usize, |_, v| v.clone())?, }; + let v = ¤t; // RFC 0070 WS1 — instance- or `None`-valued attributes take the // nullable object lane (loads pin the value at runtime; stores // resolve the staged pin); RFC 0071 WS6 — exact `str`/`bytes` @@ -3871,16 +3863,14 @@ unsafe fn native_scalar_field_update( } // SAFETY: no callback, allocation, or Python execution can overlap this // exclusive view. A shared cell is rejected by peek_mut. - let dict = unsafe { inst.dict.get()?.peek_mut() }?; - let (key, old) = dict.get_index(index as usize)?; + let (key, slot) = unsafe { inst.attr_peek_index_mut(index as usize, true) }?; if !key_is(key, &guard.name) { return None; } - let Object::Int(old) = old else { + let Object::Int(old) = slot else { return None; }; let value = old.checked_add(increment)?; - let (_, slot) = dict.map_mut_unstamped().get_index_mut(index as usize)?; // Exact integers own no destructor or GC edge. Existing-key replacement // preserves key stamps. No failing operation follows the completed store. *slot = Object::Int(value); @@ -5485,16 +5475,14 @@ unsafe extern "C" fn wpjit_call_method( s: &entry.name, hash: entry.name_hash, }; + // SAFETY: a read between two native ops; nothing here runs + // code (see `GilCell::peek`). attr_class_ok(inst, entry.ver) - && inst.dict.get().is_none_or(|dict| { - // SAFETY: a read between two native ops; nothing - // here runs code (see `GilCell::peek`). - match unsafe { dict.peek() } { - Some(d) => d.get(&probe).is_none(), - None => dict.borrow().get(&probe).is_none(), - } - }) - && Rc::ptr_eq(&entry.func.code.borrow(), &entry.code) + && unsafe { inst.attr_peek_has(probe.s, probe.hash) } == Some(false) + && match unsafe { entry.func.code.peek() } { + Some(code) => Rc::ptr_eq(code, &entry.code), + None => Rc::ptr_eq(&entry.func.code.borrow(), &entry.code), + } } _ => false, }; @@ -7566,15 +7554,9 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - } } AttrStorage::Indexed(key_idx) => { - let Some(dict) = inst.dict.get() else { - return 1; - }; // SAFETY: a read between two native ops; nothing here // runs code (see `GilCell::peek`). - let Some(dict) = (unsafe { dict.peek() }) else { - return 1; - }; - match dict.get_index(key_idx as usize) { + match unsafe { inst.attr_peek_index(key_idx as usize) } { Some((k, v)) if key_is(k, &g.name) => match classify(v) { Some(o) => o, None => return 1, @@ -7640,9 +7622,8 @@ unsafe fn chain_attr_peek<'a>( } AttrStorage::Indexed(index) => { // SAFETY: the same callback-free interval as the slot read. - let dict = unsafe { inst.dict.get().ok_or(AttrChainMiss::Guard)?.peek() } - .ok_or(AttrChainMiss::Guard)?; - let (name, value) = dict.get_index(index as usize).ok_or(AttrChainMiss::Guard)?; + let (name, value) = + unsafe { inst.attr_peek_index(index as usize) }.ok_or(AttrChainMiss::Guard)?; if !key_is(name, &guard.name) { return Err(AttrChainMiss::Guard); } @@ -7717,8 +7698,7 @@ fn chain_attr_read( } AttrStorage::Indexed(index) => { // SAFETY: every chain step is a read without Python callbacks. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (name, value) = dict.get_index(index as usize)?; + let (name, value) = unsafe { inst.attr_peek_index(index as usize) }?; key_is(name, &guard.name).then(|| read(value)) } AttrStorage::NewKey => None, @@ -7866,8 +7846,7 @@ unsafe fn cached_chain_peek<'a>( match cache { IC::LoadAttrInstance { key_idx, ver } if ver == version => { // SAFETY: the rooted walk is read-only and callback-free. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(key_idx as usize)?; + let (key, value) = unsafe { inst.attr_peek_index(key_idx as usize) }?; key_is(key, name).then_some(value) } IC::LoadAttrSlot { key_idx, ver } if ver == version => { @@ -7887,8 +7866,7 @@ unsafe fn cached_chain_peek<'a>( .get(pc as usize)? .index(version)?; // SAFETY: the same callback-free, rooted interval. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(index as usize)?; + let (key, value) = unsafe { inst.attr_peek_index(index as usize) }?; key_is(key, name).then_some(value) } } @@ -8160,17 +8138,12 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - 0 } AttrStorage::Indexed(key_idx) => { - let mut dict = inst.dict_cell().borrow_mut(); - let atomic = v.is_gc_atomic(); - let dict = &mut *dict; // Replacing an existing key's value leaves the key layout (and // so the stamp) alone, as the interpreter's indexed store does. - let dict = if atomic { - dict.map_mut_unstamped() - } else { - dict.map_mut_value_store() - }; - let Some((k, dst)) = dict.get_index_mut(key_idx as usize) else { + // SAFETY: nothing below runs code while the view is live. + let Some((k, dst)) = + (unsafe { inst.attr_peek_index_mut(key_idx as usize, v.is_gc_atomic()) }) + else { return 1; }; if !key_is(k, &g.name) { @@ -8206,6 +8179,32 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - if crate::capi_watchers::dicts_active() { return 1; } + // The split layout (see `Interpreter::core_store_new_attr`). + let v = if inst.dict.published().is_none() { + // SAFETY: a read between two native ops. + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return 1; + }; + if let Some(dst) = split.get(&g.name) { + if !matches!( + dst, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) && super::Interpreter::local_needs_prompt_reap(dst) + && super::Interpreter::looks_reapable_temporary(dst) + { + return 1; + } + } + match inst.split_store(&g.name, v) { + Ok(old) => { + drop(old); + return 0; + } + Err(v) => v, + } + } else { + v + }; let mut dict = inst.dict_cell().borrow_mut(); let atomic = v.is_gc_atomic(); let dict = &mut *dict; diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index d20646a1..859c2743 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -658,6 +658,9 @@ pub struct TypeObject { /// (capped): new instances' dicts are presized to it, so `__init__`'s /// stores do not regrow them. pub inst_dict_hint: std::sync::atomic::AtomicU32, + /// The attribute names this class's split instance dictionaries share + /// (see [`crate::inst_dict`]), created at the first split store. + pub shared_keys: crate::sync::LazyArc, /// Non-zero for an exact class whose hot methods have native /// implementations (see `stdlib::datetime_native`): its instances /// are never cycle-collector tracked (like CPython's C types without @@ -1062,6 +1065,7 @@ impl TypeObject { metaclass: RefCell::new(None), leaf_attrs: LeafAttrCache::new(), inst_dict_hint: std::sync::atomic::AtomicU32::new(0), + shared_keys: crate::sync::LazyArc::new(), native_kind: Cell::new(0), native_ext: std::sync::OnceLock::new(), abc_state: std::sync::OnceLock::new(), @@ -2467,7 +2471,7 @@ pub struct PyInstance { pub class: RefCell>, /// Instance attributes, allocated on the first write or exported handle. /// Reads through `get()` preserve the absence of an unused dictionary. - pub dict: crate::sync::LazyArc>, + pub dict: crate::inst_dict::InstDict, /// For instances of a subclass of an immutable built-in /// (`int`, `str`, `float`, `bytes`, `tuple`, …) this holds the /// underlying primitive value the instance *is* — the moral @@ -2479,7 +2483,7 @@ pub struct PyInstance { /// inline body by the extension's `tp_new` chain after allocation). /// Unwrapped by the numeric / comparison / hashing / conversion /// fast paths so e.g. `class C(int)` instances behave like real ints. - pub native: std::sync::OnceLock, + pub native: crate::sync::OnceBox, /// Mirrors CPython 3.13's "inline values" state observable through /// `_testinternalcapi.has_inline_values`: starts `true` for ordinary /// instances (native fixed weakrefs start `false`) and is @@ -2552,8 +2556,8 @@ impl PyInstance { pub fn new(class: Rc) -> Self { Self { class: RefCell::new(class), - dict: crate::sync::LazyArc::new(), - native: std::sync::OnceLock::new(), + dict: crate::inst_dict::InstDict::new(), + native: crate::sync::OnceBox::new(), inline_values: Cell::new(true), slots: RefCell::new(SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -2568,8 +2572,8 @@ impl PyInstance { pub fn with_native(class: Rc, native: Object) -> Self { Self { class: RefCell::new(class), - dict: crate::sync::LazyArc::new(), - native: std::sync::OnceLock::from(native), + dict: crate::inst_dict::InstDict::new(), + native: crate::sync::OnceBox::from(native), inline_values: Cell::new(true), slots: RefCell::new(SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -2599,7 +2603,7 @@ impl PyInstance { *m.class.get_mut() = class; m.deferred.set(true); if hint > 0 { - if let Some(dict) = m.dict.get() { + if let Some(dict) = m.dict.published() { if let Ok(mut d) = dict.try_borrow_mut() { if d.capacity() < hint { d.map_mut_atomic_store().reserve(hint); @@ -2668,10 +2672,13 @@ impl PyInstance { if m.native.get().is_some() || m.finalize_ran.get() || m.c_body.get() != 0 { return; } + // Split values are atomic (the instance is deferred): clearing + // them runs no code. The allocation stays for the next tenant. + m.dict.split_mut().reset(); // An instance that never grew a `__dict__` is the common case // now and needs no reset; one that did keeps it for the next // tenant, cleared and carrying a fresh owner record. - if let Some(dict) = m.dict.get() { + if let Some(dict) = m.dict.published() { // The dict must be private too: `vars(obj)` / `obj.__dict__` // hand out the same `Arc`, and a holder must keep seeing the // dead instance's attributes, not the next tenant's. @@ -2737,7 +2744,7 @@ impl PyInstance { #[inline] pub(crate) fn clear_deferred_tracking(&self) { self.deferred.set(false); - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { match d.try_borrow() { Ok(d) => { d.take_deferred_owner(); @@ -2757,7 +2764,7 @@ impl PyInstance { // Retire the dict's copy of the record so the write barrier does // not track a second time, then track from the instance itself — // which works whether or not a `__dict__` was ever created. - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { if let Ok(d) = d.try_borrow() { d.take_deferred_owner(); } @@ -2858,7 +2865,7 @@ impl Drop for PyInstance { fn drop(&mut self) { // A deferred-tracking record names this instance; the dict may // outlive it (`d = obj.__dict__`), so retire the record first. - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { match d.try_borrow() { Ok(d) => { d.take_deferred_owner(); @@ -2914,8 +2921,8 @@ impl Drop for PyInstance { class: RefCell::new(self.cls()), dict: self.dict.clone(), native: match self.native.get() { - Some(v) => std::sync::OnceLock::from(v.clone()), - None => std::sync::OnceLock::new(), + Some(v) => crate::sync::OnceBox::from(v.clone()), + None => crate::sync::OnceBox::new(), }, inline_values: Cell::new(self.inline_values.get()), slots: RefCell::new(self.slots.borrow().clone()), @@ -2940,7 +2947,9 @@ mod slot_storage_tests { #[test] fn shared_layout_keeps_slot_storage_compact() { assert_eq!(std::mem::size_of::(), 32); - assert_eq!(std::mem::size_of::(), 128); + // The split `__dict__` values pointer and its cell: with the + // allocator's header the instance stays in the 160-byte class. + assert_eq!(std::mem::size_of::(), 136); } #[test] From 0f400ff53d62214c0fc25374bb6fdc7ee0a32d69 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 10:54:29 -0700 Subject: [PATCH 16/65] perf: bind skipped keyword defaults in compiled calls A keyword call that skips a defaulted parameter (f(x, c=1) over def f(a, b=0, c=0)) used to fall back to the generic dynamic-call path, which re-verified the keyword names and bound a fresh frame every call. The JIT now marshals it like any keyword call and tags the skipped slots; the call helper fills them from the callee's current scalar defaults (or, for anything else, binds the call generically by name), so the compiled callee runs natively. Compiled scalar leaves are also preferred over the frameless evaluator when both apply. call_overhead's f(i, c=5) shape: 256 ns -> 52 ns per call (CPython: 99). --- crates/weavepy-jit/src/analyze.rs | 30 +++--- crates/weavepy-jit/src/ir.rs | 10 +- crates/weavepy-jit/src/lib.rs | 5 +- crates/weavepy-jit/src/lower.rs | 47 +++++++-- crates/weavepy-jit/src/runtime.rs | 9 ++ crates/weavepy-vm/src/tier2.rs | 153 +++++++++++++++++++++++++++--- 6 files changed, 215 insertions(+), 39 deletions(-) diff --git a/crates/weavepy-jit/src/analyze.rs b/crates/weavepy-jit/src/analyze.rs index b5ac4c45..45ac5419 100644 --- a/crates/weavepy-jit/src/analyze.rs +++ b/crates/weavepy-jit/src/analyze.rs @@ -2941,18 +2941,20 @@ fn kw_call_names(code: &CodeObject, cidx: u32) -> Vec<&str> { /// RFC 0073 WS5 — resolve a `CALL_KW` site's keyword permutation /// against the burned callee: keyword value `j` binds parameter slot -/// `(perm >> 4j) & 0xF` (tier-1's `CallPyKwNames` packing). Admitted -/// only when the filled set — the positional prefix plus the keyword -/// slots — is exactly `0..argc+kwc`: the marshaled call is then a -/// plain positional prefix through the unchanged `wpjit_call_py` -/// helper, and the trailing-defaults window binds the remaining tail. -/// Returns `(perm, filled count)`. +/// `(perm >> 4j) & 0xF` (tier-1's `CallPyKwNames` packing). The filled +/// set — the positional prefix plus the keyword slots — must reach +/// every parameter without a default; a defaulted parameter it skips +/// (`f(x, c=1)` over `def f(a, b=0, c=0)`) is a *gap* the call helper +/// fills from the callee's current defaults, so the marshaled call is +/// a positional prefix of `k` slots through `wpjit_call_py`, whose +/// trailing-defaults window binds any remaining tail. Returns +/// `(perm, k, gap mask)`. fn resolve_kw_perm( mark: &CalleeMark, names: &[&str], argc: usize, kw_slot: &mut dyn FnMut(u32, &str) -> Option, -) -> Result<(u32, usize), JitVerdict> { +) -> Result<(u32, usize, u32), JitVerdict> { if mark.kind != MarkKind::Py || mark.ctor { return Err(JitVerdict::UnsupportedOpcode("CALL_KW (callee kind)")); } @@ -2974,7 +2976,11 @@ fn resolve_kw_perm( perm |= slot << (4 * j); covered |= 1 << slot; } - if covered != (1u32 << k) - 1 { + // The highest filled slot bounds the marshaled prefix; the slots + // it skips must all have defaults. + let k = (32 - covered.leading_zeros()) as usize; + let gaps = ((1u32 << k) - 1) & !covered; + if gaps & ((1u32 << mark.min_args.min(16)) - 1) != 0 { return Err(JitVerdict::UnsupportedOpcode("CALL_KW (keyword gap)")); } // The uncovered tail binds trailing defaults, exactly like the @@ -2982,7 +2988,7 @@ fn resolve_kw_perm( if k < mark.min_args as usize || k > mark.arg_count as usize { return Err(JitVerdict::UnsupportedOpcode("CALL (arity)")); } - Ok((perm, k)) + Ok((perm, k, gaps)) } /// Map a representable [`Constant`] to its lane, or `None`. @@ -7546,7 +7552,7 @@ fn emit_instr( *max_stack = (*max_stack).max(stack.len() as u32); return Ok(()); }; - let (perm, _) = resolve_kw_perm(&mark, &names, argc, probes.kw_slot)?; + let (perm, k, gaps) = resolve_kw_perm(&mark, &names, argc, probes.kw_slot)?; for &ty in &arg_tys { if !ty.is_representable() { return Err(JitVerdict::TypeUnknown); @@ -7573,13 +7579,15 @@ fn emit_instr( live_to: pc + 1, interp_depth: mark.interp_depth, }); - *max_call_args = (*max_call_args).max(n as u32); + // The marshaled prefix, skipped defaulted slots included. + *max_call_args = (*max_call_args).max(k as u32); push( TOp::CallPyKw { token: mark.token, argc: argc as u8, kwc: kwc as u8, perm, + gaps, ret, }, Some(ret), diff --git a/crates/weavepy-jit/src/ir.rs b/crates/weavepy-jit/src/ir.rs index d98a01bc..66f7be00 100644 --- a/crates/weavepy-jit/src/ir.rs +++ b/crates/weavepy-jit/src/ir.rs @@ -187,10 +187,11 @@ pub enum TOp { /// stack order); the analyzer resolved each keyword to its /// parameter slot at compile time, packed 4 bits per keyword in /// `perm` (keyword value `j` → slot `(perm >> 4j) & 0xF`, tier-1's - /// `CallPyKwNames` encoding). The filled slots are validated to be - /// exactly `0..argc+kwc`, so lowering marshals a plain positional - /// prefix and the call helper needs no keyword awareness (the - /// trailing-defaults window binds any remaining tail). The names + /// `CallPyKwNames` encoding). Lowering marshals a positional prefix + /// up to the highest filled slot; the defaulted slots it skips + /// (`gaps`, one bit per slot) are tagged for the call helper to + /// bind from the callee's current defaults, and the + /// trailing-defaults window binds any remaining tail. The names /// tuple's `LOAD_CONST` is erased from the trace; it never exists /// on the native stack. Exits mirror [`TOp::CallPy`]. CallPyKw { @@ -198,6 +199,7 @@ pub enum TOp { argc: u8, kwc: u8, perm: u32, + gaps: u32, ret: JitType, }, /// RFC 0061 WS5 — `BINARY_SUBSCR` on a pinned list: pops the `int` diff --git a/crates/weavepy-jit/src/lib.rs b/crates/weavepy-jit/src/lib.rs index f846c037..5e04cb84 100644 --- a/crates/weavepy-jit/src/lib.rs +++ b/crates/weavepy-jit/src/lib.rs @@ -60,8 +60,9 @@ pub use runtime::{ JitStatus, ListAppendHelper, ListFromRangeHelper, ListGetHelper, ListLenHelper, ListNextHelper, ListRepeatHelper, ListSetHelper, ListSliceHelper, MathBinaryHelper, MathUnaryHelper, PollHelper, SelfEnterHelper, SelfExitHelper, SelfSlowHelper, SlotTag, StrEqHelper, - StrLenHelper, StrModHelper, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, DICT_VAL_INT, - DICT_VAL_OBJ, ITER_ELEM_STR, JIT_POLL_STRIDE, MAX_ATTR_CHAIN_LEN, MAX_CACHED_ATTR_CHAIN_LEN, + StrLenHelper, StrModHelper, CALL_GAPS, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, + DICT_VAL_INT, DICT_VAL_OBJ, ITER_ELEM_STR, JIT_POLL_STRIDE, MAX_ATTR_CHAIN_LEN, + MAX_CACHED_ATTR_CHAIN_LEN, }; pub use value::JitType; diff --git a/crates/weavepy-jit/src/lower.rs b/crates/weavepy-jit/src/lower.rs index 9651cef4..9f327aac 100644 --- a/crates/weavepy-jit/src/lower.rs +++ b/crates/weavepy-jit/src/lower.rs @@ -995,7 +995,7 @@ impl<'a, 'b> Lowerer<'a, 'b> { } else if let Some(ix) = self.leaf_for(token, argc) { self.emit_call_leaf(ix, token, argc, ret, stmt.pc); } else { - self.emit_call_py(token, argc, 0, 0, ret, stmt.pc); + self.emit_call_py(token, argc, 0, 0, 0, ret, stmt.pc); } } TOp::CallPyKw { @@ -1003,8 +1003,9 @@ impl<'a, 'b> Lowerer<'a, 'b> { argc, kwc, perm, + gaps, ret, - } => self.emit_call_py(token, argc, kwc, perm, ret, stmt.pc), + } => self.emit_call_py(token, argc, kwc, perm, gaps, ret, stmt.pc), TOp::ListGet { elem } => self.emit_list_get(elem, stmt.pc), TOp::ListSet => self.emit_list_set(stmt.pc), TOp::CellGet { idx, lane } => self.emit_cell_get(idx, lane, stmt.pc), @@ -3174,7 +3175,7 @@ impl<'a, 'b> Lowerer<'a, 'b> { // Declined: the ordinary call helper. self.b.switch_to_block(generic_b); self.vstack.extend(args.iter().copied()); - self.emit_call_py(token, argc, 0, 0, ret, pc); + self.emit_call_py(token, argc, 0, 0, 0, ret, pc); let (v, _) = self.vstack.pop().expect("the call's result"); self.b.ins().jump(join_b, &[v.into()]); @@ -3408,7 +3409,7 @@ impl<'a, 'b> Lowerer<'a, 'b> { // Declined or deopted: the ordinary call, from the start. self.b.switch_to_block(generic_b); self.vstack.extend(args.iter().copied()); - self.emit_call_py(token, argc, 0, 0, ret, pc); + self.emit_call_py(token, argc, 0, 0, 0, ret, pc); let (v, _) = self.vstack.pop().expect("the call's result"); self.b.ins().jump(join_b, &[v.into()]); @@ -3448,9 +3449,33 @@ impl<'a, 'b> Lowerer<'a, 'b> { r } - fn emit_call_py(&mut self, token: u32, argc: u8, kwc: u8, perm: u32, ret: JitType, pc: u32) { + #[allow(clippy::too_many_arguments)] + fn emit_call_py( + &mut self, + token: u32, + argc: u8, + kwc: u8, + perm: u32, + gaps: u32, + ret: JitType, + pc: u32, + ) { let trusted = MemFlags::trusted(); let n = argc as usize + kwc as usize; + // Skipped defaulted slots: the helper binds them (see + // `SlotTag::Default`). + let mut g = gaps; + while g != 0 { + let slot = g.trailing_zeros() as i32; + g &= g - 1; + let tagv = self + .b + .ins() + .iconst(types::I32, runtime::SlotTag::Default as i64); + self.b + .ins() + .store(trusted, tagv, self.call_tags_base, slot * 4); + } let base = self.vstack.len() - n; for (j, &(v, ty)) in self.vstack[base..].iter().enumerate() { let dst = if j < argc as usize { @@ -3480,9 +3505,15 @@ impl<'a, 'b> Lowerer<'a, 'b> { .ins() .iconst(self.ptr_ty, runtime::call_py_helper_addr() as i64); let tokenv = self.b.ins().iconst(types::I32, i64::from(token)); - // The helper receives the *filled* count — keyword values were - // shuffled into a contiguous positional prefix above. - let argcv = self.b.ins().iconst(types::I32, n as i64); + // The helper receives the prefix length — keyword values were + // shuffled into parameter slots above, with any skipped + // defaulted slots flagged. + let filled = if gaps == 0 { + n as i64 + } else { + i64::from((n as u32 + gaps.count_ones()) | runtime::CALL_GAPS) + }; + let argcv = self.b.ins().iconst(types::I32, filled); let expect = self.b.ins().iconst(types::I32, Self::tag(ret)); let call = self.b diff --git a/crates/weavepy-jit/src/runtime.rs b/crates/weavepy-jit/src/runtime.rs index 41675dcb..bd9e6b57 100644 --- a/crates/weavepy-jit/src/runtime.rs +++ b/crates/weavepy-jit/src/runtime.rs @@ -122,8 +122,16 @@ pub enum SlotTag { /// exit, or a provably-`None` method-call result). The bits are /// ignored; the embedder rebuilds `Object::None`. None = 6, + /// A call argument slot a keyword call skipped: the call helper + /// binds the callee's default there before anything reads it. Only + /// ever appears in a call marshal buffer under [`CALL_GAPS`]. + Default = 7, } +/// Set in a `wpjit_call_py` argument count when some marshaled slots +/// are tagged [`SlotTag::Default`]. +pub const CALL_GAPS: u32 = 1 << 31; + impl SlotTag { /// Decode a raw tag written by native code. #[inline] @@ -136,6 +144,7 @@ impl SlotTag { 4 => SlotTag::ListPin, 5 => SlotTag::ObjPin, 6 => SlotTag::None, + 7 => SlotTag::Default, _ => SlotTag::Int, } } diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 8d0b7625..4566af5e 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -2175,7 +2175,7 @@ fn unpack(bits: u64, tag: u32) -> Object { SlotTag::Bool => Object::Bool(bits != 0), // RFC 0069 WS1 — the `None` singleton (a `ReturnNone` exit). SlotTag::None => Object::None, - SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => Object::None, + SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin | SlotTag::Default => Object::None, } } @@ -3879,6 +3879,110 @@ unsafe fn native_scalar_field_update( Some(value) } +/// Bind the slots a keyword call skipped (tagged [`SlotTag::Default`] +/// under [`weavepy_jit::CALL_GAPS`]) to the callee's current +/// `__defaults__` when those are scalars, so the call is a positional +/// prefix again (whose trailing window the lanes below bind). Returns +/// the argument count to call with and whether every skipped slot is +/// bound (`false` leaves the call to [`call_py_with_gaps`]). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]; the buffers are the compiled +/// frame's `max_call_args` wide. +unsafe fn bind_call_defaults( + jf: &mut JitFrame, + ctx: &CallCtx, + token: u32, + raw: u32, +) -> (u32, bool) { + let gapped = raw & weavepy_jit::CALL_GAPS != 0; + let argc = raw & !weavepy_jit::CALL_GAPS; + let Some((Object::Function(f), code)) = ctx.callees.get(token as usize) else { + return (argc, !gapped); + }; + if !gapped { + return (argc, true); + } + // A reassigned `__defaults__` lives in the function's slots: the + // generic binder reads it. + if f.defaults_maybe_overridden() { + return (argc, false); + } + let npos = code.arg_count as usize; + // SAFETY: the activation's compiled frame outlives its calls. + let cap = unsafe { ctx.cf.as_ref() }.map_or(0, |cf| cf.max_call_args as usize); + let first = npos.saturating_sub(f.defaults.len()); + let scalar = |j: usize| -> Option<(u64, SlotTag)> { + let v = f.defaults.get(j.checked_sub(first)?)?; + match v { + Object::Int(i) => Some((*i as u64, SlotTag::Int)), + Object::Float(x) => Some((x.to_bits(), SlotTag::Float)), + Object::Bool(b) => Some((u64::from(*b), SlotTag::Bool)), + Object::None => Some((u64::MAX, SlotTag::ObjPin)), + _ => None, + } + }; + let write = |jf: &mut JitFrame, j: usize, (bits, tag): (u64, SlotTag)| { + // SAFETY: `j` is below the buffers' width (checked by callers). + unsafe { + *jf.call_args.add(j) = bits; + *jf.call_tags.add(j) = tag as u32; + } + }; + let mut bound = true; + for j in 0..(argc as usize).min(cap).min(npos) { + // SAFETY: native code wrote `argc` tags. + if unsafe { *jf.call_tags.add(j) } == SlotTag::Default as u32 { + match scalar(j) { + Some(v) => write(jf, j, v), + None => bound = false, + } + } + } + (argc, bound) +} + +/// [`wpjit_call_py`] for a keyword call whose skipped defaulted slot +/// could not be bound natively (a non-scalar default, or none any +/// more): the generic call, with the prefix before the first skipped +/// slot positional and the rest by name, binds (or rejects) it exactly +/// as the interpreter would. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe fn call_py_with_gaps( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &mut super::Interpreter, + token: u32, + argc: u32, + expect_tag: u32, +) -> i64 { + ctx.dirty = true; + let (callee, code) = ctx.callees[token as usize].clone(); + let mut args: Vec = Vec::new(); + let mut kwargs: Vec<(String, Object)> = Vec::new(); + for j in 0..argc as usize { + // SAFETY: native code wrote `argc` entries. + let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; + if tag == SlotTag::Default as u32 { + continue; + } + let v = unpack_pins(bits, tag, &ctx.pins); + if kwargs.is_empty() && args.len() == j { + args.push(v); + } else if let Some(name) = code.varnames.get(j) { + kwargs.push((name.to_string(), v)); + } + } + let res = call_with_activation_shell(interp, ctx, jf, |i| { + i.call(&callee, &args, &kwargs, &ctx.globals) + }); + finish_interp_call(jf, ctx, interp, res, expect_tag) +} + /// RFC 0067 WS1 — attempt a native-to-native call for one marshaled /// `CallPy` site. Returns `Some(CallStatus as i64)` when the call /// completed through the native path (including via a materialized @@ -4523,9 +4627,11 @@ unsafe fn try_native_call( SlotTag::Int => JitType::Int, SlotTag::Float => JitType::Float, SlotTag::Bool => JitType::Bool, - SlotTag::None | SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => { - JitType::Unknown - } + SlotTag::None + | SlotTag::Boxed + | SlotTag::ListPin + | SlotTag::ObjPin + | SlotTag::Default => JitType::Unknown, }; match pack(&v, expect) { Some(bits) if guards_ok => { @@ -4806,11 +4912,28 @@ unsafe extern "C" fn wpjit_call_py( // while the helper runs; this is the only live path to it. let interp = unsafe { &mut *ctx.interp }; - // A pure-leaf callee evaluates frameless (see `try_pure_leaf_call`). - let maybe_pure = ctx - .callees - .get(token as usize) - .is_some_and(|(_, code)| code.jit_hint.pure_leaf() != Some(false)); + // Defaulted parameters the site didn't pass: bound here, so every + // lane below sees a full-arity call. + // SAFETY: per the function contract. + let (argc, bound) = unsafe { bind_call_defaults(jf, ctx, token, argc) }; + if !bound { + // SAFETY: as above. + return unsafe { call_py_with_gaps(jf, ctx, interp, token, argc, expect_tag) }; + } + + // A pure-leaf callee evaluates frameless (see `try_pure_leaf_call`), + // unless its compiled scalar leaf runs it natively (cheaper still). + let native_scalar = ctx + .native + .as_deref() + .and_then(|t| t.get(token as usize)) + .and_then(Option::as_ref) + .is_some_and(|nc| nc.ctor.is_none() && nc.cf.is_scalar_leaf()); + let maybe_pure = !native_scalar + && ctx + .callees + .get(token as usize) + .is_some_and(|(_, code)| code.jit_hint.pure_leaf() != Some(false)); let callees = if maybe_pure { Some(StdRc::clone(&ctx.callees)) } else { @@ -4933,9 +5056,11 @@ unsafe extern "C" fn wpjit_call_py( // Other pin-lane call results are rejected at // emission; `Unknown` never packs, forcing the // boxed path. - SlotTag::None | SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => { - JitType::Unknown - } + SlotTag::None + | SlotTag::Boxed + | SlotTag::ListPin + | SlotTag::ObjPin + | SlotTag::Default => JitType::Unknown, }; if let Some(bits) = pack(&v, expect) { jf.ret_bits = bits; @@ -5026,7 +5151,7 @@ fn deliver_call_result(jf: &mut JitFrame, ctx: &mut CallCtx, v: Object, expect_t return CallStatus::Ok as i64; } } - SlotTag::Boxed | SlotTag::ListPin => {} + SlotTag::Boxed | SlotTag::ListPin | SlotTag::Default => {} } ctx.parked = Some(v); CallStatus::Boxed as i64 @@ -5798,7 +5923,7 @@ unsafe extern "C" fn wpjit_str_method( } } } - SlotTag::None | SlotTag::Float | SlotTag::Boxed => {} + SlotTag::None | SlotTag::Float | SlotTag::Boxed | SlotTag::Default => {} } // Lane surprise (`WStr` result, huge `int`, pin-cap // pressure): park the exact result and deopt after the From 031ad5c6d6580236cfd7faa2ad490659f852c184 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:23:49 -0700 Subject: [PATCH 17/65] perf: cheaper short loops and short-lived containers - Retire an exhausted range or list iterator in the core loop instead of handing the loop exit to the full handler, as the leaf arm does. An interpreted loop over a short list costs 36% fewer instructions. - Build list literals in the core loop, and release the last reference to an untracked list or dict of scalars there as a plain drop instead of taking the prompt-reap path. - Reclaim dead deferred containers at the tail of the deferral list as new ones arrive, so container churn neither grows the list nor parks freed allocations until a sweep. - Allocate 16-byte-aligned layouts through mimalloc's plain entry points: every block of 16 bytes or more already has that alignment, and the aligned entry points fell to their generic path for iterators. --- crates/weavepy-cli/src/alloc.rs | 21 +++-- crates/weavepy-vm/src/gc_trace.rs | 20 +++++ crates/weavepy-vm/src/lib.rs | 136 +++++++++++++++++++++++++++--- 3 files changed, 159 insertions(+), 18 deletions(-) diff --git a/crates/weavepy-cli/src/alloc.rs b/crates/weavepy-cli/src/alloc.rs index 758a0485..7c78cd0b 100644 --- a/crates/weavepy-cli/src/alloc.rs +++ b/crates/weavepy-cli/src/alloc.rs @@ -1,8 +1,9 @@ //! The process allocator: mimalloc, entered through its plain allocation //! functions whenever they already satisfy the requested alignment. //! -//! Every mimalloc block is at least word aligned, so a layout aligned to -//! 8 bytes or less needs no aligned entry point. The aligned entry points +//! Every mimalloc block is at least word aligned, and every block of 16 +//! bytes or more is 16-byte aligned (`MI_MAX_ALIGN_SIZE`), so a layout +//! aligned to that much needs no aligned entry point. The aligned entry points //! take their fast path only when the size class's free list happens to //! offer a suitably aligned block, and otherwise fall to a generic path //! that may over-allocate. Rust asks for alignment on every allocation, @@ -19,6 +20,16 @@ use libmimalloc_sys::{ /// The largest alignment every mimalloc block already has. const WORD_ALIGN: usize = std::mem::size_of::(); +/// mimalloc's `MI_MAX_ALIGN_SIZE`: the alignment of every block at least +/// this large. +const MAX_ALIGN: usize = 16; + +/// Whether the plain entry points already satisfy `align` for `size`. +#[inline(always)] +fn plain(size: usize, align: usize) -> bool { + align <= WORD_ALIGN || (align <= MAX_ALIGN && size >= align) +} + /// mimalloc as the global allocator (see the module docs). pub(crate) struct Mimalloc; @@ -30,7 +41,7 @@ unsafe impl GlobalAlloc for Mimalloc { unsafe fn alloc(&self, layout: Layout) -> *mut u8 { // SAFETY: plain FFI allocation calls. unsafe { - if layout.align() <= WORD_ALIGN { + if plain(layout.size(), layout.align()) { mi_malloc(layout.size()).cast() } else { mi_malloc_aligned(layout.size(), layout.align()).cast() @@ -42,7 +53,7 @@ unsafe impl GlobalAlloc for Mimalloc { unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { // SAFETY: as in `alloc`. unsafe { - if layout.align() <= WORD_ALIGN { + if plain(layout.size(), layout.align()) { mi_zalloc(layout.size()).cast() } else { mi_zalloc_aligned(layout.size(), layout.align()).cast() @@ -60,7 +71,7 @@ unsafe impl GlobalAlloc for Mimalloc { unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { // SAFETY: `ptr` came from this allocator with `layout`. unsafe { - if layout.align() <= WORD_ALIGN { + if plain(new_size, layout.align()) { mi_realloc(ptr.cast::(), new_size).cast() } else { mi_realloc_aligned(ptr.cast::(), new_size, layout.align()).cast() diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index e0a620a3..a0f2c486 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -760,6 +760,16 @@ impl GcState { // queue onto a list we cannot touch. return false; }; + // The churn shape — a loop that builds a container and drops + // the previous one — leaves its dead predecessors just below + // the tail: reclaim them here, so steady churn neither grows + // the list nor parks dead allocations until a sweep. + let n = deferred.len(); + for i in (n.saturating_sub(2)..n).rev() { + if deferred[i].is_dead() { + deferred.swap_remove(i); + } + } deferred.push(weak); deferred.len() >= self.deferred_limit.load(Ordering::Relaxed) }; @@ -3581,6 +3591,16 @@ impl DeferredContainer { } } + /// Whether the container has died. + #[inline] + fn is_dead(&self) -> bool { + match self { + Self::List(w) => w.strong_count() == 0, + Self::Dict(w) => w.strong_count() == 0, + Self::Set(w) => w.strong_count() == 0, + } + } + /// The container, if it is still alive. fn upgrade(&self) -> Option { match self { diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 4de1c60b..37df1d86 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12103,6 +12103,20 @@ impl Interpreter { pc += 1; continue; } + // An untracked container of scalars: + // a plain drop. + Object::List(_) | Object::Dict(_) + if Self::core_plain_last_container(&*slot) => + { + len -= 1; + drop(std::mem::replace( + &mut *slot, + base.add(len).read(), + )); + last = pc; + pc += 1; + continue; + } _ => break None, } } @@ -12121,6 +12135,32 @@ impl Interpreter { last = pc; pc += 1; } + // A list literal, tracked like the full handler's (an + // allocation due to trigger a collection is left to it). + OpCode::BuildList => { + let n = ins.arg as usize; + if n > len + || (n == 0 && len == cap) + || crate::stdlib::tracemalloc_real::is_tracking() + || crate::stdlib::testinternalcapi_mod::reftrace_print_active() + || gc_trace::auto_collect_due() + { + break None; + } + // SAFETY: the top `n` slots are initialized; they move + // into the list and leave the stack. + let items: Vec = (len - n..len) + .map(|j| unsafe { base.add(j).read() }) + .collect(); + len -= n; + let obj = Object::new_list(items); + gc_trace::track(obj.clone()); + // SAFETY: `len < cap` (the operands' slots were freed). + unsafe { base.add(len).write(obj) }; + len += 1; + last = pc; + pc += 1; + } OpCode::PopTop => { // A scalar needs no drop; a shared heap value is a // plain decrement (see `core_droppable`). @@ -12527,6 +12567,10 @@ impl Interpreter { let Some(it) = (unsafe { it.peek_mut() }) else { break None; }; + // An exhausted range or list iterator the loop holds + // alone, whose death frees nothing else (see the leaf + // arm): retired here instead of by the full handler. + let mut retire = false; let v = match it { crate::object::PyIterator::Range { current, @@ -12538,22 +12582,42 @@ impl Interpreter { } else { *step < 0 && *current > *stop }; - if !live { + if live { + let v = *current; + *current = current.wrapping_add(*step); + Object::Int(v) + } else if unique { + retire = true; + Object::None + } else { break None; } - let v = *current; - *current = current.wrapping_add(*step); - Object::Int(v) } - crate::object::PyIterator::List { items, index, .. } => { + crate::object::PyIterator::List { + items, + index, + owner, + } => { // SAFETY: as above. - let v = match unsafe { items.peek() } { - Some(xs) => xs.get(*index).map(Self::clone_operand), - None => None, + let Some(xs) = (unsafe { items.peek() }) else { + break None; }; - let Some(v) = v else { break None }; - *index += 1; - v + match xs.get(*index) { + Some(v) => { + let v = Self::clone_operand(v); + *index += 1; + v + } + None if unique + && owner.is_none() + && Rc::strong_count(items) >= 2 + && !Self::iter_backing_list_dead(items) => + { + retire = true; + Object::None + } + None => break None, + } } crate::object::PyIterator::Tuple { items, index } => { let Some(v) = items.get(*index).cloned() else { @@ -12734,6 +12798,34 @@ impl Interpreter { } _ => break None, }; + if retire { + // Popped, and the loop exit skips the `END_FOR` / + // `POP_ITER` pair (the full handler's shape). + len -= 1; + // SAFETY: the iterator slot leaves the stack. + let it = unsafe { base.add(len).read() }; + let marked = + gc_trace::maybe_tracked(crate::weakref_registry::id_of(&it)) + && gc_trace::note_dropped_marks(&it); + drop(it); + last = pc; + pc += 1 + ins.arg as usize; + let op_at = |pc: usize| { + // SAFETY: `pc < ninstrs` is checked first. + (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }) + }; + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!(op_at(pc), Some(OpCode::PopIter | OpCode::PopTop)) { + pc += 1; + } + } + if marked { + gc_trace::mark_maybe_dead(); + break Some(CoreExit::Stop(LeafStop::Marked)); + } + continue; + } // SAFETY: `len < cap`. unsafe { base.add(len).write(v) }; len += 1; @@ -14191,8 +14283,13 @@ impl Interpreter { // weakref operation that requires tracking revokes the flag. Rc::strong_count(i) > 1 && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) } - Object::List(l) => Rc::strong_count(l) > 1 && !gc_trace::note_dropped_marks(v), - Object::Dict(d) => Rc::strong_count(d) > 1 && !gc_trace::note_dropped_marks(v), + Object::List(l) if Rc::strong_count(l) > 1 => !gc_trace::note_dropped_marks(v), + Object::Dict(d) if Rc::strong_count(d) > 1 => !gc_trace::note_dropped_marks(v), + // The last owner of a container the collector never took, and + // no weakref watches, holding only scalars: freeing it runs no + // code (`prompt_reap_dropped` reaches the same plain drop the + // long way round). + Object::List(_) | Object::Dict(_) => Self::core_plain_last_container(v), Object::Tuple(t) => ThinArc::strong_count(t) > 1 && !gc_trace::note_dropped_marks(v), Object::Function(f) => Rc::strong_count(f) > 1 && !gc_trace::note_dropped_marks(v), Object::Type(t) => Rc::strong_count(t) > 1 && !gc_trace::note_dropped_marks(v), @@ -14200,6 +14297,19 @@ impl Interpreter { } } + /// [`Self::core_droppable`] for the last owner of an exact `list` or + /// `dict`: untracked (the miss filter proves it), unwatched, and + /// holding only scalars. + #[inline(never)] + fn core_plain_last_container(v: &Object) -> bool { + let id = crate::weakref_registry::id_of(v); + !gc_trace::maybe_tracked(id) + && !crate::weakref_registry::may_have_weakrefs(id) + && !crate::capi_watchers::dicts_active() + && !crate::stdlib::testinternalcapi_mod::reftrace_print_active() + && Self::is_scalar_leaf_container(v) + } + /// Module scope for the core loop's `LOAD_NAME` / `STORE_NAME` arms: /// names resolve in the globals, then the builtins, both exact dicts. #[inline(always)] From 170930b175d2a0087cfe49c9bd812de6e6d73895 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:24:06 -0700 Subject: [PATCH 18/65] fix: restore the VM unit tests this branch broke - Recognize the process-wide NotImplemented and Ellipsis singletons by identity: they carry the type registry of the thread that built them, so a class check against the current thread's registry failed for every later thread (the native _abc raised AssertionError). - Restore compiling sustained start-up and import work: warm compiles during those phases count lean intervals again, rather than deferring every one, and frameless leaf calls count toward the warm-up, handing the call at which a compile falls due to the framed path. - Count direct self and leaf calls and frameless call-site evaluations in the native-call statistics, and scalar field updates' boxed returns on the method path, which the tests assert on. All 419 VM unit tests pass with the CI stack size. --- crates/weavepy-vm/src/lib.rs | 32 ++++++++++++-- crates/weavepy-vm/src/tier2.rs | 60 +++++++++++++++++++++----- crates/weavepy-vm/src/vm_singletons.rs | 49 +++++++++++++++------ 3 files changed, 113 insertions(+), 28 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 37df1d86..51be7dd4 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -14978,6 +14978,29 @@ impl Interpreter { f: &crate::object::PyFunction, args: &[*const Object], ) -> Option { + // A frameless call is an activation the JIT's warm-up never sees: + // count it as a lean one, and credit each interval to the tier-2 + // counter. The call at which a compile falls due goes to the + // framed path, which compiles the hot leaf so native callers can + // take its direct lanes. + #[cfg(feature = "jit")] + { + let hint = &code.jit_hint; + let n = hint.lean_entries(); + if n <= crate::tier2::LEAN_WARM_COMPILE_THRESHOLD_CAP { + if n + 1 == crate::tier2::lean_warm_at() + && !hint.is_not_jitable() + && !crate::tier2::jit_off_for_process() + { + if crate::tier2::note_frameless_calls(code, n + 1) { + return None; + } + hint.defer_lean_compile(); + } else { + hint.bump_lean_entries(); + } + } + } let r = self.leaf_eval::(code, f, args, 0); if r.is_some() { code.jit_hint.note_leaf_hit(); @@ -71728,8 +71751,9 @@ print("native pickle coverage: ok") #[test] fn jit_native_lane_mismatch_falls_back() { // `dbl` compiled for int arguments; the native caller passes a - // float — the argument-lane check rejects the fast path and - // the interpreter call stays exact. + // float — the argument-lane check rejects the native fast path, + // and the call (frameless or through the interpreter) stays + // exact. let src = "def dbl(x):\n if x < 0:\n return 0\n\ \x20 return x + x\n\ k = 0\nwhile k < 10:\n dbl(3)\n k = k + 1\n\ @@ -71739,8 +71763,8 @@ print("native pickle coverage: ok") r = 0.0\nk = 0\n\ while k < 10:\n r = spin(20)\n k = k + 1\n\ print(r)\n"; - let (out, _calls, fallbacks, _deopts) = run_jit_native(src); - assert!(fallbacks >= 1, "float-for-int argument must fall back"); + let (out, _calls, _fallbacks, deopts) = run_jit_native(src); + assert_eq!(deopts, 0, "a lane mismatch never enters the native callee"); assert_eq!(out, "20.0\n"); assert_eq!(out, run(src)); } diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 4566af5e..fb2f474c 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -632,6 +632,31 @@ thread_local! { pub(crate) const LEAN_WARM_COMPILE_THRESHOLD_CAP: u32 = 24; #[inline] +/// Credit `n` activations a frameless path ran for `code` to its tier-2 +/// warm-up counter (the framed entries that count otherwise never happen +/// for them). Returns whether the next framed or lean entry should +/// compile it: a compile is due, or no entry exists yet to count in. +pub(crate) fn note_frameless_calls(code: &CodeObject, n: u32) -> bool { + JIT.with(|cell| { + let Ok(mut st) = cell.try_borrow_mut() else { + return false; + }; + if !st.enabled { + return false; + } + let threshold = st.threshold; + match st.cache.get_mut(&std::ptr::from_ref(code)) { + None => true, + Some(entry) if matches!(entry.tier, Tier::Cold) => { + entry.counter = entry.counter.saturating_add(n); + let next = entry.counter.saturating_add(1); + next >= threshold && compile_allowed(next, threshold) + } + Some(_) => false, + } + }) +} + pub(crate) fn lean_warm_at() -> u32 { LEAN_WARM_AT .try_with(std::cell::Cell::get) @@ -2875,15 +2900,6 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F if frame.code.jit_hint.is_not_jitable() || jit_off_for_process() { return; } - // A loop-free body gains nothing from native code until compiled - // callers exist to take its direct lanes, while compiling one during - // start-up or an import costs time and memory the program may never - // recover (`ABCMeta.register` while `_collections_abc` loads). Count - // again from zero; a body that stays hot compiles afterwards. - if phase != CompilationPhase::Normal { - frame.code.jit_hint.defer_lean_compile(); - return; - } JIT.with(|cell| { let mut st = cell.borrow_mut(); if !st.enabled { @@ -2893,7 +2909,9 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F let threshold = st.threshold; // Embedders that never report start-up finished still compile, // after sustained work. - let warm = if STARTUP_DONE.load(std::sync::atomic::Ordering::Relaxed) { + let warm = if phase == CompilationPhase::Normal + && STARTUP_DONE.load(std::sync::atomic::Ordering::Relaxed) + { threshold } else { threshold.saturating_mul(16) @@ -2913,6 +2931,23 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F code: frame.code.clone(), }); if matches!(entry.tier, Tier::Cold) { + if phase != CompilationPhase::Normal { + // A loop-free body gains nothing from native code until + // compiled callers exist to take its direct lanes, while + // compiling one during start-up or an import costs time and + // memory the program may never recover (`ABCMeta.register` + // while `_collections_abc` loads). The lean path has no + // ordinary frame-entry counter: account for the interval + // just completed and count another, so only sustained work + // compiles (the checkpoint stays reachable afterwards, even + // when pure-leaf calls skip frames). + let interval = lean_warm_at(); + entry.counter = entry.counter.saturating_add(interval); + if entry.counter < interval.saturating_mul(16) { + frame.code.jit_hint.defer_lean_compile(); + return; + } + } // Preserve the earlier lean warm point relative to frame/loop // hotness. entry.counter = entry.counter.max(warm); @@ -5249,6 +5284,8 @@ unsafe fn pure_leaf_call( ptrs[offset + j] = unsafe { written.0.add(j) }; } let v = interp.pure_leaf_eval::(code, func, &ptrs[..n])?; + // A call served without an interpreter frame, like a native one. + native_stat(|s| s.calls.set(s.calls.get() + 1)); Some(deliver_call_result(jf, ctx, v, expect_tag)) } @@ -5287,6 +5324,7 @@ unsafe extern "C" fn wpjit_self_enter(frame: *mut JitFrame) -> i64 { return 1; } depth.set(n); + native_stat(|s| s.calls.set(s.calls.get() + 1)); 0 } @@ -5682,6 +5720,8 @@ unsafe extern "C" fn wpjit_call_method( return CallStatus::Ok as i64; } // The store is complete: never repeat it. + #[cfg(test)] + crate::SCALAR_FIELD_UPDATE_BOXED_RETURNS.with(|hits| hits.set(hits.get() + 1)); ctx.parked = Some(Object::Int(value)); return CallStatus::Boxed as i64; } diff --git a/crates/weavepy-vm/src/vm_singletons.rs b/crates/weavepy-vm/src/vm_singletons.rs index f78b67c5..e74ab8c8 100644 --- a/crates/weavepy-vm/src/vm_singletons.rs +++ b/crates/weavepy-vm/src/vm_singletons.rs @@ -174,6 +174,12 @@ fn make_singleton(cls: Rc) -> Object { /// `NotImplementedType` (an `object` subclass), so `type(NotImplemented)` /// and the MRO match CPython. pub fn not_implemented() -> Object { + not_implemented_ref().clone() +} + +/// [`not_implemented`] by reference. The singleton is process-wide, built +/// on the first thread to ask, whose type registry supplies its class. +fn not_implemented_ref() -> &'static Object { static SLOT: OnceLock = OnceLock::new(); SLOT.get_or_init(|| { let cls = crate::builtin_types::builtin_types() @@ -181,18 +187,35 @@ pub fn not_implemented() -> Object { .clone(); make_singleton(cls) }) - .clone() } /// Same idea for `Ellipsis` (the value of `...`); its class is the /// registry's `ellipsis` type. pub fn ellipsis() -> Object { + ellipsis_ref().clone() +} + +/// [`ellipsis`] by reference (see [`not_implemented_ref`]). +fn ellipsis_ref() -> &'static Object { static SLOT: OnceLock = OnceLock::new(); SLOT.get_or_init(|| { let cls = crate::builtin_types::builtin_types().ellipsis_.clone(); make_singleton(cls) }) - .clone() +} + +/// Whether `obj` is the process-wide singleton `single`, or another +/// instance of its type or of this thread's registry type `cls` (a thread +/// with its own registry, or a C-API proxy, sees the same singleton). +fn is_singleton_of(obj: &Object, single: &Object, cls: &Rc) -> bool { + let (Object::Instance(inst), Object::Instance(one)) = (obj, single) else { + return false; + }; + if Rc::ptr_eq(inst, one) { + return true; + } + let ty = inst.cls(); + Rc::ptr_eq(&ty, cls) || Rc::ptr_eq(&ty, &one.cls()) } /// `True` if `obj` is the canonical `Ellipsis` singleton — an instance of @@ -203,26 +226,24 @@ pub fn ellipsis() -> Object { /// (numpy's `prepare_index`) takes the right branch rather than rejecting a /// freshly-boxed proxy with "only integers, slices … are valid indices". pub fn is_ellipsis(obj: &Object) -> bool { - if let Object::Instance(inst) = obj { - return Rc::ptr_eq( - &inst.cls(), + matches!(obj, Object::Instance(_)) + && is_singleton_of( + obj, + ellipsis_ref(), &crate::builtin_types::builtin_types().ellipsis_, - ); - } - false + ) } /// `True` if `obj` is the canonical `NotImplemented` singleton. The C-API /// bridge maps it to the static `_Py_NotImplementedStruct` so extensions /// that compare against `Py_NotImplemented` by pointer behave correctly. pub fn is_not_implemented(obj: &Object) -> bool { - if let Object::Instance(inst) = obj { - return Rc::ptr_eq( - &inst.cls(), + matches!(obj, Object::Instance(_)) + && is_singleton_of( + obj, + not_implemented_ref(), &crate::builtin_types::builtin_types().not_implemented_type_, - ); - } - false + ) } /// CPython's `help`/`copyright`/`license`/`credits` builtins are From 9c23bdb8009f6a7d1237d262ac69606645960abe Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:19:57 -0700 Subject: [PATCH 19/65] perf: call native methods and subscripts in place in the core loop - Fuse x.m(simple args) for a class's registered native method (a deque's append, popleft, pop, ...): the builtin runs on bitwise views of the borrowed receiver and arguments, with no method object, no receiver or argument references taken, and no operand stack traffic. - Serve a subscript whose class has a cached native fast __getitem__ in the core loop instead of handing it to the full arm. - Cache how each class answers a boolean context (native __bool__ or __len__, always true, or the full path) by attribute version, instead of looking both names up for every test. - Give deque iteration, integer indexing and appendleft guard-free fast paths, and keep an emptied deque's free prefix for the next appendleft. deque_ops: 2.58x -> 2.03x CPython in instructions; append/popleft pairs -29%, indexing -44%, truth tests -39%. --- crates/weavepy-vm/src/lib.rs | 308 ++++++++++++++++-- .../src/stdlib/collections_native.rs | 84 ++++- 2 files changed, 367 insertions(+), 25 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 51be7dd4..66f4499c 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -971,6 +971,9 @@ pub struct Interpreter { /// (process-unique) attribute version of the class that resolved it /// (see [`Interpreter::core_leaf_next`]). core_next: ThreadCell)>>, + /// How the last class seen in a boolean context answers it (see + /// [`Interpreter::leaf_instance_truth`]), by attribute version. + core_truth: ThreadCell>, /// `WP_DBG_SAMPLE`: periodic frame-entry sampling to stderr. dbg_sample: bool, } @@ -1201,6 +1204,7 @@ impl Default for Interpreter { leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), core_next: ThreadCell::new(None), + core_truth: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), }; // RFC 0025: publish the shared parts of this interpreter @@ -1328,6 +1332,7 @@ impl Interpreter { leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), core_next: ThreadCell::new(None), + core_truth: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), } } @@ -11942,6 +11947,32 @@ impl Interpreter { pc = call_pc + 1; continue; } + } else if let Some((r, call_pc)) = self + .core_native_method( + code, + other, + pc + 1, + next.arg, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) + { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop( + LeafStop::Raised(e), + )); + } + } } } if next.op == OpCode::LoadAttr { @@ -12410,6 +12441,45 @@ impl Interpreter { last = pc; pc += 1; } + // An instance whose class's `__getitem__` is a + // native fast subscript the site cached. + None if ins.op == OpCode::BinarySubscr && len >= 2 => { + let Some(fast) = self.core_native_subscript( + code, + pc, + // SAFETY: `len >= 2`. + unsafe { std::slice::from_raw_parts(base.add(len - 2), 2) }, + ) else { + break None; + }; + // SAFETY: both operands leave by plain + // decrements (checked); the result takes the + // container's slot. + let r = fast(unsafe { + std::slice::from_raw_parts(base.add(len - 2), 2) + }); + let Some(r) = r else { break None }; + #[cfg(test)] + NATIVE_FAST_SUBSCRIPTS.with(|calls| calls.set(calls.get() + 1)); + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 2; + match r { + Ok(v) => { + // SAFETY: the container's slot is free. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + } + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } None => break None, } } @@ -12523,10 +12593,12 @@ impl Interpreter { // the full arm, which asks again (a leaf iterator // stays exhausted) and ends the loop. obj @ Object::Instance(inst) => { - let Some(b) = self.core_leaf_next(inst) else { + let Some(b) = self.core_leaf_next_ptr(inst) else { break None; }; - match (b.call)(std::slice::from_ref(obj)) { + // SAFETY: the cache and the class keep the + // builtin alive; its body runs no Python. + match unsafe { ((*b).call)(std::slice::from_ref(obj)) } { Ok(v) => { // SAFETY: `len < cap`. unsafe { base.add(len).write(v) }; @@ -14561,6 +14633,113 @@ impl Interpreter { } } + /// The native fast `__getitem__` the `BINARY_SUBSCR` at `pc` cached for + /// `ops[0]`'s class (see [`Self::leaf_instance_subscript`]), when both + /// operands leave by plain decrements. + #[inline(never)] + fn core_native_subscript( + &self, + code: &CodeObject, + pc: usize, + ops: &[Object], + ) -> Option { + let [Object::Instance(inst), key] = ops else { + return None; + }; + let cls = inst.cls_raw(); + if !Self::core_droppable(&ops[0]) + || !Self::core_droppable(key) + || !Self::default_getattribute(cls) + || crate::object::exotic_str_keys_possible() + { + return None; + } + let fast = code_vm_ext(code)? + .method_slots + .get()? + .get(pc)? + .get_native_subscript(cls.attr_version.get(), leaf_builtins::generation())?; + #[cfg(test)] + NATIVE_SUBSCRIPT_CACHE_HITS.with(|hits| hits.set(hits.get() + 1)); + Some(fast) + } + + /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); ; CALL k` on a local instance `recv` whose class's `m` + /// is a registered native builtin that runs no Python code (the call + /// site's leaf kind): called on bitwise views of the borrowed operands, + /// so no reference to the receiver, the method or an argument is taken + /// or released. Returns the call's outcome and its pc; `None` (nothing + /// touched) runs the instructions one by one. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_native_method( + &self, + code: &CodeObject, + recv: &Object, + attr_pc: usize, + name_idx: u32, + mslots: &[MethodSlot], + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ) -> Option<(Result, usize)> { + let Object::Instance(inst) = recv else { + return None; + }; + let cls = inst.cls_raw(); + let b = mslots + .get(attr_pc)? + .peek_inst_builtin(cls.attr_version.get())?; + if !Self::default_getattribute(cls) { + return None; + } + let mut scratch = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + args[0] = recv; + // SAFETY: the core loop's own locals and constants. + let (nargs, call_pc) = unsafe { + Self::core_simple_args( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut scratch, + &mut args, + 1, + ) + }?; + let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; + // SAFETY: the builtin lives in the slot (and its class) for the + // whole call, which runs no Python code. + let b = unsafe { &*b }; + if !matches!(kind, LeafKind::Opaque | LeafKind::Fast(_)) + || inst_may_shadow(inst, code, name_idx) + { + return None; + } + // Bitwise views of the operands, never dropped: the callee reads + // them (and clones what it keeps), and nothing it runs can reach + // the locals or constants they alias. + let n = nargs + 1; + let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; + for (slot, &p) in ops.iter_mut().zip(&args[..n]) { + // SAFETY: each pointer names a live operand (see above). + slot.write(unsafe { std::ptr::read(p) }); + } + // SAFETY: the first `n` entries were just written. + let ops = unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), n) }; + let r = match kind { + LeafKind::Fast(f) => f(ops)?, + _ => match b.call_kw.as_ref() { + Some(ckw) => ckw(ops, &[]), + None => (b.call)(ops), + }, + }; + Some((r, call_pc)) + } + /// A fused simple call of pure leaf `fp` (see [`code_is_pure_leaf`]): /// `args[..nargs]` are the borrowed operands (the receiver first for a /// bound call, `has_self`), and `call_pc` the `CALL` whose site slot @@ -18460,35 +18639,64 @@ impl Interpreter { let Object::Instance(inst) = v else { return None; }; - // `NotImplemented` has neither method yet refuses a boolean - // context (a TypeError since 3.14): the full path raises it. - if v.is_same(&crate::vm_singletons::not_implemented()) { - return None; - } let cls = inst.cls_raw(); - if !Self::default_getattribute(cls) || crate::object::exotic_str_keys_possible() { + if crate::object::exotic_str_keys_possible() { return None; } - for name in ["__bool__", "__len__"] { + let ver = cls.attr_version.get(); + let fresh = !matches!(&*self.core_truth.borrow(), Some((v, _)) if *v == ver); + if fresh { + let t = self.native_truth_of(v, cls); + *self.core_truth.borrow_mut() = Some((ver, t)); + } + // The cache (and the class) keep the builtin alive through the + // call, whose body runs no Python. + let (b, is_len): (*const crate::object::BuiltinFn, bool) = match &*self.core_truth.borrow() + { + Some((_, NativeTruth::AlwaysTrue)) => return Some(true), + Some((_, NativeTruth::Bool(b))) => (Rc::as_ptr(b), false), + Some((_, NativeTruth::Len(b))) => (Rc::as_ptr(b), true), + _ => return None, + }; + // SAFETY: see above. + let b = unsafe { &*b }; + let r = match b.call_kw.as_ref() { + Some(ckw) => ckw(std::slice::from_ref(v), &[]), + None => (b.call)(std::slice::from_ref(v)), + }; + match r.ok()? { + Object::Bool(b) => Some(b), + Object::Int(n) if is_len && n >= 0 => Some(n != 0), + _ => None, + } + } + + /// How instances of `cls` answer a boolean context natively (keyed by + /// the class's attribute version in `core_truth`). + #[cold] + #[inline(never)] + fn native_truth_of(&self, v: &Object, cls: &TypeObject) -> NativeTruth { + // `NotImplemented` has neither method yet refuses a boolean + // context (a TypeError since 3.14): the full path raises it. + if v.is_same(&crate::vm_singletons::not_implemented()) || !Self::default_getattribute(cls) { + return NativeTruth::Decline; + } + for (name, is_len) in [("__bool__", false), ("__len__", true)] { match cls.lookup(name) { None => continue, Some(Object::Builtin(b)) if b.binds_instance && self.leaf_call_kind(&b) == Some(LeafKind::Opaque) => { - let r = match b.call_kw.as_ref() { - Some(ckw) => ckw(std::slice::from_ref(v), &[]), - None => (b.call)(std::slice::from_ref(v)), - }; - return match r.ok()? { - Object::Bool(b) => Some(b), - Object::Int(n) if name == "__len__" && n >= 0 => Some(n != 0), - _ => None, + return if is_len { + NativeTruth::Len(b) + } else { + NativeTruth::Bool(b) }; } - Some(_) => return None, + Some(_) => return NativeTruth::Decline, } } - Some(true) + NativeTruth::AlwaysTrue } /// Offer borrowed subscript operands to a registered native body or its @@ -18860,17 +19068,29 @@ impl Interpreter { .map(|(_, _, f)| f.clone()) } - /// Which leaf builtin `b` is, if any (pointer identity). - #[inline] /// `inst`'s class's `__next__` when it is a registered leaf builtin - /// that binds its instance (the core loop's native iterators). - fn core_leaf_next(&self, inst: &PyInstance) -> Option> { + /// that binds its instance (the core loop's native iterators), + /// uncounted: the cache (and the class) keep it alive until the next + /// lookup. + #[inline] + fn core_leaf_next_ptr(&self, inst: &PyInstance) -> Option<*const crate::object::BuiltinFn> { let ver = inst.cls_raw().attr_version.get(); if let Some((v, b)) = &*self.core_next.borrow() { if *v == ver { - return Some(b.clone()); + return Some(Rc::as_ptr(b)); } } + self.core_leaf_next_resolve(inst, ver) + .map(|b| Rc::as_ptr(&b)) + } + + #[cold] + #[inline(never)] + fn core_leaf_next_resolve( + &self, + inst: &PyInstance, + ver: u64, + ) -> Option> { match inst.cls().lookup("__next__")? { Object::Builtin(b) if b.binds_instance @@ -18883,6 +19103,7 @@ impl Interpreter { } } + /// Which leaf builtin `b` is, if any (pointer identity). fn leaf_call_kind(&self, b: &Rc) -> Option { let p = Rc::as_ptr(b) as usize; if let Some(k) = self.leaf_fns().calls.get(&p) { @@ -62200,6 +62421,19 @@ impl MethodSlot { } } + /// [`Self::get_inst_builtin`] without taking a reference: the slot + /// (and the class) keep the builtin alive while it is used. + #[inline] + fn peek_inst_builtin(&self, ver: u64) -> Option<*const crate::object::BuiltinFn> { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (v, MethodSlotFn::Builtin(f)) if *v == ver && ver & Self::BUILTIN_TAG == 0 => { + Some(Rc::as_ptr(f)) + } + _ => None, + } + } + #[inline] fn set_inst_builtin(&self, ver: u64, f: &Rc) { // SAFETY: as `set`. @@ -62232,6 +62466,18 @@ impl MethodSlot { } } + /// [`Self::get_leaf`] by pointer identity. + #[inline] + fn get_leaf_ptr(&self, f: *const crate::object::BuiltinFn) -> Option { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (_, MethodSlotFn::Leaf(cached, kind)) if std::ptr::eq(Rc::as_ptr(cached), f) => { + Some(*kind) + } + _ => None, + } + } + #[inline] fn set_leaf(&self, f: &Rc, kind: LeafKind) { // SAFETY: as `set`. @@ -62808,6 +63054,20 @@ fn code_name_obj(code: &CodeObject, name_idx: u32) -> Option<&Object> { code_vm_ext(code).and_then(|t| t.name_objs.get(name_idx as usize)) } +/// How a class's instances answer a boolean context without running +/// Python (see [`Interpreter::leaf_instance_truth`]). +#[derive(Clone)] +enum NativeTruth { + /// Neither `__bool__` nor `__len__`: always true. + AlwaysTrue, + /// A registered native `__bool__`. + Bool(Rc), + /// A registered native `__len__`. + Len(Rc), + /// Anything else: the full path decides. + Decline, +} + /// Whether `inst`'s own attributes may shadow `co_names[name_idx]` (a /// method cache's guard): `false` proves the name absent, and `true` also /// answers when that can't be told without borrowing. diff --git a/crates/weavepy-vm/src/stdlib/collections_native.rs b/crates/weavepy-vm/src/stdlib/collections_native.rs index e8b4f089..40f773db 100644 --- a/crates/weavepy-vm/src/stdlib/collections_native.rs +++ b/crates/weavepy-vm/src/stdlib/collections_native.rs @@ -242,6 +242,28 @@ fn deque_append(args: &[Object]) -> Result { } fn deque_appendleft(args: &[Object]) -> Result { + if let [_, x] = args { + if let Some((slots, d)) = fast_parts(args) { + bump_state_of(slots); + let mut h = head_of(slots).min(d.len()); + if h == 0 { + h = std::cmp::max(8, d.len() / 2); + d.splice(0..0, std::iter::repeat_n(Object::None, h)); + } + h -= 1; + d[h] = x.clone(); + set_head_of(slots, h); + let trimmed = match maxlen_of(slots) { + Some(m) if d.len() - h > m => d.pop(), + _ => None, + }; + if trimmed.is_some() { + crate::gc_trace::mark_maybe_dead(); + } + drop(trimmed); + return Ok(Object::None); + } + } let mut st = receiver(args, "appendleft")?; let x = match args { [_, x] => x.clone(), @@ -285,7 +307,9 @@ fn deque_pop(args: &[Object]) -> Result { if d.len() > h { bump_state_of(slots); let x = d.pop().expect("len checked"); - if d.len() == h && h != 0 { + // An emptied deque keeps a short free prefix for the next + // `appendleft` instead of splicing a new one. + if d.len() == h && h > 32 { d.clear(); set_head_of(slots, 0); } @@ -364,6 +388,16 @@ fn deque_bool(args: &[Object]) -> Result { } fn deque_getitem(args: &[Object]) -> Result { + if let [_, Object::Int(i)] = args { + if let Some((slots, d)) = fast_parts(args) { + let h = head_of(slots); + let n = d.len().saturating_sub(h) as i64; + let index = if *i < 0 { *i + n } else { *i }; + if (0..n).contains(&index) { + return Ok(d[h + index as usize].clone()); + } + } + } let [_, index] = args else { receiver(args, "__getitem__")?; return Err(type_error("deque.__getitem__() takes one argument")); @@ -462,10 +496,58 @@ const IT_DEQ: usize = 0; const IT_INDEX: usize = 1; const IT_STATE: usize = 2; +/// [`deque_next`]'s common case over unguarded views (see +/// [`fast_parts`]): a live iterator over an unmutated deque with an item +/// left. `None` (nothing touched) leaves every other case to the full +/// body. +fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option { + // SAFETY: no guard is live on the cells (`peek`/`peek_mut` check), no + // Python runs before the last use, and the iterator and its deque are + // distinct objects. + let its = unsafe { iterator.slots.peek_mut() }?; + let Some(Object::Instance(deque)) = its.get_hinted(IT_DEQ, "_deq") else { + return None; + }; + // The deque lives while the iterator's slot holds it (unchanged here). + let deque: *const crate::types::PyInstance = Rc::as_ptr(deque); + let index = its + .get_hinted(IT_INDEX, "_index") + .and_then(Object::as_i64)?; + let it_state = its + .get_hinted(IT_STATE, "_deq_state") + .and_then(Object::as_i64); + // SAFETY: as above. + let ds = unsafe { (*deque).slots.peek() }?; + let Some(Object::List(data)) = ds.get_hinted(SLOT_DATA, "_data") else { + return None; + }; + if ds.get_hinted(SLOT_STATE, "_state").and_then(Object::as_i64) != it_state { + return None; + } + let h = head_of(ds); + // SAFETY: as above. + let d = unsafe { data.peek() }?; + let h = h.min(d.len()); + if index < 0 || index as usize >= d.len() - h { + return None; + } + let slot = if reverse { + d.len() - 1 - index as usize + } else { + h + index as usize + }; + let v = d[slot].clone(); + *its.get_hinted_mut(IT_INDEX, "_index")? = Object::Int(index + 1); + Some(v) +} + fn deque_next(args: &[Object], reverse: bool) -> Result { let [Object::Instance(iterator)] = args else { return Err(type_error("deque iterator __next__ requires one iterator")); }; + if let Some(v) = deque_next_fast(iterator, reverse) { + return Ok(v); + } let (deque, index, it_state) = { let s = iterator.slots.borrow(); let deque = match s.get_hinted(IT_DEQ, "_deq") { From 169aece7d4e73ff69af9125779c6d4499a8241ad Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 15:54:57 -0700 Subject: [PATCH 20/65] perf: skip JIT compiles that cost more than they save - Don't compile a loop-free function that makes a dynamic Python call: from native code the call takes the interpreter's generic path (a nested run rather than the inline activation an interpreted caller uses), and compiling was the largest cost a short-lived method paid. deltablue (default size) drops from 97 ms to 81 ms, with JIT compile time cut from 17 ms to about 10 ms. - Switch generator state in and out of its cell without guard atomics on inline resumes and yields (about 5% per resume/yield pair). --- crates/weavepy-jit/src/engine.rs | 33 ++++++++++++++++++++++++++++++++ crates/weavepy-vm/src/lib.rs | 19 ++++++++++++------ 2 files changed, 46 insertions(+), 6 deletions(-) diff --git a/crates/weavepy-jit/src/engine.rs b/crates/weavepy-jit/src/engine.rs index e0f28a63..e5408456 100644 --- a/crates/weavepy-jit/src/engine.rs +++ b/crates/weavepy-jit/src/engine.rs @@ -337,6 +337,9 @@ impl JitEngine { direct: &mut dyn FnMut(u32) -> Option, ) -> Result { let tfunc = crate::analyze::analyze_frame(code, resolve, probes)?; + if calls_dynamically(&tfunc) { + return Err(JitVerdict::UnsupportedOpcode("dynamic call (loop-free)")); + } self.compile_tfunc_direct(&tfunc, direct) } @@ -640,6 +643,36 @@ impl JitEngine { } } +/// Whether native code would cost a loop-free body more than it saves: +/// it makes a dynamic Python call, which from native code takes the +/// interpreter's generic call path (a nested run instead of the inline +/// activation an interpreted caller uses), while the compile itself is +/// the largest cost a short-lived method ever pays. +fn calls_dynamically(tfunc: &TFunc) -> bool { + let has_loop = !tfunc.range_loops.is_empty() + || !tfunc.list_loops.is_empty() + || !tfunc.iter_loops.is_empty() + || tfunc.blocks.iter().enumerate().any(|(i, b)| { + use crate::ir::TTerm; + match b.term { + TTerm::Jump(t) => t as usize <= i, + TTerm::BranchFalse { + target, + fallthrough, + } + | TTerm::BranchTrue { + target, + fallthrough, + } => target as usize <= i || fallthrough as usize <= i, + _ => false, + } + }); + if has_loop { + return false; + } + op_mix(tfunc).dyn_calls > 0 +} + /// `(generic, total)`: statements that hand an operation to the /// interpreter's generic object protocol (dynamic calls and attribute /// accesses) against all statements (see [`CompiledFrame::op_mix`]). diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 66f4499c..502f5a15 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -9857,9 +9857,11 @@ impl Interpreter { let Some(Object::Generator(g)) = frame.stack.last() else { return None; }; - // Validate and take the frame under one borrow. No Python runs - // while it is held; release it before resuming the activation. - let mut state = g.state.try_borrow_mut().ok()?; + // Validate and take the frame under one exclusive view. No Python + // runs while it is held; it ends before the activation resumes. + // SAFETY: nothing below reaches the cell again until `state`'s last + // use (`peek_mut` rejects a live guard or shared cells). + let state = unsafe { g.state.peek_mut() }?; let first_resume; { let boxed = match &*state { @@ -9891,8 +9893,7 @@ impl Interpreter { return None; }; // Committed. - let prev_state = std::mem::replace(&mut *state, GeneratorState::Running); - drop(state); + let prev_state = std::mem::replace(state, GeneratorState::Running); let g = g.clone(); let (GeneratorState::Suspended(mut boxed) | GeneratorState::Created(mut boxed)) = prev_state @@ -36101,7 +36102,13 @@ impl Interpreter { *py.gen_owner.borrow_mut() = Some(Rc::downgrade(gen)); } } - *gen.state.borrow_mut() = GeneratorState::Suspended(boxed); + // SAFETY: the store runs no code (the displaced state is + // `Running`, which owns nothing) and no guard is live on the cell + // (`peek_mut` checks). + match unsafe { gen.state.peek_mut() } { + Some(state) => *state = GeneratorState::Suspended(boxed), + None => *gen.state.borrow_mut() = GeneratorState::Suspended(boxed), + } } fn generator_send( From 20253389d4af433f62e00dbd2d04f2a991697bc6 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 16:54:57 -0700 Subject: [PATCH 21/65] perf: resume generators inline for next() --- crates/weavepy-vm/src/lib.rs | 78 ++++++++++++++++++++++++++++++------ 1 file changed, 66 insertions(+), 12 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 502f5a15..8e60e4c0 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -9559,9 +9559,13 @@ impl Interpreter { && matches!(frame.stack.last(), Some(Object::Generator(_))) && self.inline_calls_ok() { - if let Some(act) = - self.try_inline_gen(frame, shell, cur_pc, crate::recursion::depth_cell()) - { + if let Some(act) = self.try_inline_gen( + frame, + shell, + cur_pc, + crate::recursion::depth_cell(), + false, + ) { return FrameEv::Call(act); } } @@ -9845,6 +9849,9 @@ impl Interpreter { /// `FOR_ITER` over a suspended generator at `pc` of `frame`, as an /// inline activation: the generator's boxed frame runs in place (the /// `generator_send_lean` shape, without a nested native activation). + /// With `next_call`, the instruction is instead the `CALL` of builtin + /// `next` on the generator: the call's operands leave the stack, the + /// yielded value is its result, and exhaustion raises `StopIteration`. /// `None` leaves everything untouched. fn try_inline_gen( &mut self, @@ -9852,8 +9859,13 @@ impl Interpreter { shell: &mut QuietShell<'_>, pc: usize, depth_cell: *const std::cell::Cell, + next_call: bool, ) -> Option> { - let arg = frame.code.instructions.get(pc)?.arg; + let arg = if next_call { + GEN_NEXT_CALL + } else { + frame.code.instructions.get(pc)?.arg + }; let Some(Object::Generator(g)) = frame.stack.last() else { return None; }; @@ -9909,6 +9921,11 @@ impl Interpreter { gf.push(Object::None); let gen_frame: *mut Frame = gf; frame.pc = pc as u32 + 1; + if next_call { + // `next`, its empty self slot and the generator (held above). + let n = frame.stack.len(); + drop(frame.stack.drain(n - 3..)); + } let mut act = self.inline_slot(); act.gen = Some(g); act.gen_box = Some(boxed); @@ -9983,13 +10000,13 @@ impl Interpreter { Self::park_suspended_boxed(&gen, boxed); Ok(GenStep::Yielded(v)) } - Ok(FrameOutcome::Returned(_)) => { + Ok(FrameOutcome::Returned(v)) => { *gen.state.borrow_mut() = GeneratorState::Finished; let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(&gen); - Ok(GenStep::Exhausted) + Ok(GenStep::Exhausted(v)) } Ok(FrameOutcome::StartGenerator) => { *gen.state.borrow_mut() = GeneratorState::Finished; @@ -10035,7 +10052,12 @@ impl Interpreter { frame.stack.push(v); QuietEntry::Returned { cur_pc: call_pc } } - Ok(GenStep::Exhausted) => { + // `next(gen)`: the return value rides the StopIteration. + Ok(GenStep::Exhausted(v)) if arg == GEN_NEXT_CALL => QuietEntry::Raised { + err: crate::error::stop_iteration_with(v), + cur_pc: call_pc, + }, + Ok(GenStep::Exhausted(_)) => { let it = frame.stack.pop(); frame.pc += arg; frame.skip_end_for(); @@ -12627,7 +12649,7 @@ impl Interpreter { unsafe { frame.stack.set_len(len) }; frame.pc = pc as u32; *last_pc = last; - if !self.core_gen_resume(sw, pc) { + if !self.core_gen_resume(sw, pc, false) { sw.pending = Some(CoreExit::Stop(LeafStop::Step)); } break Some(CoreExit::Reload); @@ -13113,6 +13135,25 @@ impl Interpreter { if len < argc + 2 { break None; } + // `next(gen)`: the generator resumes inline, as for + // `FOR_ITER`, with the yield as the call's result. + // SAFETY: `len >= argc + 2 == 3`. + if argc == 1 + && matches!( + unsafe { (&*base.add(len - 3), &*base.add(len - 2), &*base.add(len - 1)) }, + (Object::Builtin(b), Object::Unbound, Object::Generator(_)) + if Rc::as_ptr(b) as usize == self.leaf_fns().next_ptr + ) + { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_gen_resume(sw, pc, true) { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } // SAFETY: `len >= argc + 2`. let python = match unsafe { &*base.add(len - argc - 2) } { Object::Function(_) => 1, @@ -15983,7 +16024,7 @@ impl Interpreter { /// ([`Self::try_inline_gen`]), with the generator's activation pushed /// and made the running one here. `false` touches nothing. #[inline(never)] - fn core_gen_resume(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + fn core_gen_resume(&mut self, sw: &mut CoreSwitch, pc: usize, next_call: bool) -> bool { if !self.inline_calls_ok() { return false; } @@ -15997,6 +16038,7 @@ impl Interpreter { &mut *shell.cast::>(), pc, sw.depth_cell, + next_call, ) }; let Some(act) = act else { @@ -18836,8 +18878,12 @@ impl Interpreter { self.leaf_fns.get_or_init(|| { let mut calls = std::collections::HashMap::default(); let mut len_ptr = 0usize; + let mut next_ptr = 0usize; { let b = self.builtins.borrow(); + if let Some(Object::Builtin(f)) = b.get(&crate::object::StrKey("next")) { + next_ptr = Rc::as_ptr(f) as usize; + } for (name, kind) in [("len", LeafKind::Len), ("isinstance", LeafKind::Isinstance)] { if let Some(Object::Builtin(f)) = b.get(&crate::object::StrKey(name)) { calls.insert(Rc::as_ptr(f) as usize, kind); @@ -18954,6 +19000,7 @@ impl Interpreter { LeafFns { calls, len_ptr, + next_ptr, methods, range_ty: bt.range_.clone(), list_ty: bt.list_.clone(), @@ -56470,6 +56517,8 @@ struct LeafFns { calls: std::collections::HashMap, /// The builtin `len`'s address (for the fused `len(x)`), or 0. len_ptr: usize, + /// The builtin `next`'s address (an inline generator resume), or 0. + next_ptr: usize, /// The builtin `range` and `list` types, whose calls with leaf /// argument shapes construct directly. range_ty: Rc, @@ -56542,7 +56591,8 @@ struct InlineAct { gen: Option>, gen_box: Option>, gen_frame: *mut Frame, - /// The resuming `FOR_ITER`'s jump distance (the exhaustion exit). + /// The resuming `FOR_ITER`'s jump distance (the exhaustion exit), or + /// [`GEN_NEXT_CALL`] for a resume by builtin `next`. exhaust_arg: u32, /// A constructor call's fresh instance (see `Interpreter::core_new`): /// the activation runs its `__init__`, and the caller receives the @@ -56652,11 +56702,15 @@ enum FrameEv { } /// How an inline generator resume concluded. +/// [`InlineAct::exhaust_arg`] of a generator resumed by builtin `next` +/// (no `FOR_ITER` jump is that long). +const GEN_NEXT_CALL: u32 = u32::MAX; + enum GenStep { /// The generator yielded this value. Yielded(Object), - /// The generator returned: the loop is exhausted. - Exhausted, + /// The generator returned this value: the loop is exhausted. + Exhausted(Object), } /// Where an activation's stretch of the quiet loop starts. From acc8fc2414ab3c0d90ddfa2ce95580f214bc77a5 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 18:08:33 -0700 Subject: [PATCH 22/65] perf: read and write split instance fields through per-site shortcuts A LOAD_ATTR or STORE_ATTR site that proved a split-layout field records the class version and shared-key position, so later hits skip the inline cache decode and the key-name check. The leaf evaluator's operand values also get a primitive layout, so copying one is two words. --- crates/weavepy-vm/src/inst_dict.rs | 61 +++++++++ crates/weavepy-vm/src/lib.rs | 204 ++++++++++++++++++++++++----- 2 files changed, 235 insertions(+), 30 deletions(-) diff --git a/crates/weavepy-vm/src/inst_dict.rs b/crates/weavepy-vm/src/inst_dict.rs index 2ee8e887..159079ec 100644 --- a/crates/weavepy-vm/src/inst_dict.rs +++ b/crates/weavepy-vm/src/inst_dict.rs @@ -352,6 +352,34 @@ impl SplitValues { Some((k, v)) } + /// The `i`th value, when these values are laid out over `keys` (so the + /// name at `i` is whatever `keys` published there). + #[inline(always)] + pub fn get_over(&self, keys: *const SharedKeys, i: usize) -> Option<&Object> { + let h = self.block?.as_ptr(); + // SAFETY: the block is live while owned, and its first `len` values + // are initialized. + unsafe { + if !std::ptr::eq((*h).keys, keys) || i >= (*h).len as usize { + return None; + } + Some(&*Self::values_ptr(h).add(i)) + } + } + + /// [`Self::get_over`], writable in place. + #[inline(always)] + pub fn get_over_mut(&mut self, keys: *const SharedKeys, i: usize) -> Option<&mut Object> { + let h = self.block?.as_ptr(); + // SAFETY: as `get_over`, and `&mut self` is exclusive. + unsafe { + if !std::ptr::eq((*h).keys, keys) || i >= (*h).len as usize { + return None; + } + Some(&mut *Self::values_ptr(h).add(i)) + } + } + /// [`Self::get_index`] with the value writable in place. #[inline(always)] pub fn get_index_mut(&mut self, i: usize) -> Option<(&DictKey, &mut Object)> { @@ -770,6 +798,39 @@ fn note_materialize(at: &'static std::panic::Location<'static>) { } impl crate::types::PyInstance { + /// The attribute at position `i` of the class's shared names, while + /// this instance's values are still split over its own class's names: + /// the name there is fixed for as long as the class lives, so a + /// caller that proved it once needs no name check again. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline(always)] + pub unsafe fn split_field(&self, i: usize) -> Option<&Object> { + let keys: *const SharedKeys = self.cls_raw().shared_keys.get()?; + // SAFETY: forwarded contract. + unsafe { self.dict.split_peek() }?.get_over(keys, i) + } + + /// [`Self::split_field`] for an in-place overwrite by a value of + /// atomicity `atomic`, whose write barrier runs first (a non-atomic + /// value starts tracking a deferred instance). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + #[allow(clippy::mut_from_ref)] + pub unsafe fn split_field_mut(&self, i: usize, atomic: bool) -> Option<&mut Object> { + let keys: *const SharedKeys = self.cls_raw().shared_keys.get()?; + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + unsafe { self.dict.split_peek_mut() }?.get_over_mut(keys, i) + } + /// The `i`th attribute in assignment order, from whichever layout the /// instance uses, read without borrow bookkeeping. `None` when it is /// absent or the storage is borrowed (take the general path). diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 8e60e4c0..981d68eb 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12015,7 +12015,7 @@ impl Interpreter { } } if let Some(v) = Self::core_local_attr( - ext.map_or(&[], |t| &t.name_objs), + ext, code, other, pc + 1, @@ -13760,6 +13760,19 @@ impl Interpreter { } _ => break Some(CoreExit::Helper), }; + // SAFETY: a read between two instructions (see + // `GilCell::peek`). + if let Some(v) = ext.and_then(|e| unsafe { field_slot_hit(e, pc, inst) }) { + if Self::core_droppable(unsafe { &*top }) { + let v = Self::clone_operand(v); + // SAFETY: the receiver (droppable) is replaced + // in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + } let IC::LoadAttrInstance { key_idx, ver } = code.caches.get(pc as u32) else { break Some(CoreExit::Helper); @@ -13780,6 +13793,9 @@ impl Interpreter { if !slot_name_matches(code, ins.arg, k) { break Some(CoreExit::Helper); } + if let Some(e) = ext { + field_slot_note(e, ninstrs, pc, inst, key_idx); + } let v = Self::clone_operand(v); // SAFETY: the receiver (droppable) is replaced in place. unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; @@ -15137,13 +15153,18 @@ impl Interpreter { /// shared storage and conflicting mutable borrows. #[inline(always)] unsafe fn leaf_cached_instance_field<'a>( - names: &[Object], + ext: &CodeConstObjects, code: &CodeObject, inst: &'a PyInstance, cache_pc: u32, name_idx: u32, ) -> Option<&'a Object> { use weavepy_compiler::InlineCache as IC; + // SAFETY: forwarded contract. + if let Some(v) = unsafe { field_slot_hit(ext, cache_pc as usize, inst) } { + return Some(v); + } + let names: &[Object] = &ext.name_objs; let (key_idx, cached, is_slot) = match code.caches.get(cache_pc) { IC::LoadAttrInstance { key_idx, ver } => (key_idx, ver, false), IC::LoadAttrSlot { key_idx, ver } => (key_idx, ver, true), @@ -15156,7 +15177,17 @@ impl Interpreter { if !is_slot { // SAFETY: the caller keeps this rooted read callback-free. let (key, value) = unsafe { inst.attr_peek_index(key_idx as usize) }?; - slot_name_matches_in(names, code, name_idx, key).then_some(value) + if !slot_name_matches_in(names, code, name_idx, key) { + return None; + } + field_slot_note( + ext, + code.instructions.len(), + cache_pc as usize, + inst, + key_idx, + ); + Some(value) } else { // SAFETY: the same rooted read as the dictionary path. let slots = unsafe { inst.slots.peek() }?; @@ -15244,7 +15275,10 @@ impl Interpreter { use weavepy_compiler::InlineCache as IC; /// Nested leaf calls evaluated in place at most this deep. const NEST: u8 = 3; + // Every payload sits at offset 8 (a primitive representation), so + // a copy is two words rather than a byte-wise shuffle. #[derive(Clone, Copy)] + #[repr(u64)] enum V { /// A resolved callee: a function the class or namespace holds. Fn(*const crate::object::PyFunction), @@ -15324,7 +15358,7 @@ impl Interpreter { let ext = code_vm_ext(code)?; let consts: &[Object] = &ext.objects; let stamps: &[StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); - let instrs = &code.instructions; + let instrs: &[weavepy_compiler::Instruction] = &code.instructions; // Tiny return bodies need no operand stack or owned-value scratch. // The shape is certified once, alongside the pure-leaf decision; // call, observer, and recursion guards still belong to the caller. @@ -15381,13 +15415,7 @@ impl Interpreter { // SAFETY: the argument roots the receiver until the // result is retained; nothing here invokes Python. if let Some(value) = unsafe { - Self::leaf_cached_instance_field( - &ext.name_objs, - code, - inst, - pc as u32, - attr.arg, - ) + Self::leaf_cached_instance_field(ext, code, inst, pc as u32, attr.arg) } { return Some(clone_hot(value)); } @@ -15411,13 +15439,7 @@ impl Interpreter { let attr = instrs.get(pc)?; // SAFETY: the same rooted, callback-free read. let value = unsafe { - Self::leaf_cached_instance_field( - &ext.name_objs, - code, - inst, - pc as u32, - attr.arg, - ) + Self::leaf_cached_instance_field(ext, code, inst, pc as u32, attr.arg) }?; #[cfg(test)] note_predicate_stage(code, if load_pc == start { 6 } else { 7 }); @@ -15542,20 +15564,23 @@ impl Interpreter { }; macro_rules! push { ($v:expr) => {{ - if sp == 8 { + let v = $v; + if sp >= st.len() { return None; } - st[sp] = $v; + // SAFETY: `sp < st.len()`, checked just above. + unsafe { *st.get_unchecked_mut(sp) = v }; sp += 1; }}; } macro_rules! pop { () => {{ - if sp == 0 { + if sp == 0 || sp > st.len() { return None; } sp -= 1; - st[sp] + // SAFETY: `sp < st.len()`, checked just above. + unsafe { *st.get_unchecked(sp) } }}; } // Locals: an argument reads through its pointer until the body @@ -15668,11 +15693,7 @@ impl Interpreter { // argument or owned scratch; no Python runs. let hit = unsafe { Self::leaf_cached_instance_field( - &ext.name_objs, - code, - inst, - pc as u32, - ins.arg, + ext, code, inst, pc as u32, ins.arg, ) } .map(std::ptr::from_ref); @@ -16582,13 +16603,19 @@ impl Interpreter { /// code), anything else through [`Self::leaf_fused_local_attr`]. #[inline(never)] fn core_local_attr( - names: &[Object], + ext: Option<&CodeConstObjects>, code: &CodeObject, local: &Object, attr_pc: usize, name_idx: u32, ) -> Option { use weavepy_compiler::InlineCache as IC; + if let (Some(ext), Object::Instance(inst)) = (ext, local) { + // SAFETY: a read between two instructions (see `peek`). + if let Some(v) = unsafe { field_slot_hit(ext, attr_pc, inst) } { + return Some(Self::clone_operand(v)); + } + } if let (IC::LoadAttrInstance { key_idx, ver }, Object::Instance(inst)) = (code.caches.get(attr_pc as u32), local) { @@ -16596,9 +16623,16 @@ impl Interpreter { if cls.native_kind.get() != 0 || cls.attr_version.get() != ver { return None; } + let names: &[Object] = ext.map_or(&[], |t| &t.name_objs); // SAFETY: a read between two instructions (see `peek`). let (k, v) = unsafe { inst.attr_peek_index(key_idx as usize) }?; - return slot_name_matches_in(names, code, name_idx, k).then(|| Self::clone_operand(v)); + if !slot_name_matches_in(names, code, name_idx, k) { + return None; + } + if let Some(ext) = ext { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, key_idx); + } + return Some(Self::clone_operand(v)); } Self::leaf_fused_local_attr(code, local, attr_pc, name_idx) } @@ -16615,6 +16649,7 @@ impl Interpreter { attr_pc: usize, ) -> Option<(Object, usize)> { use weavepy_compiler::InlineCache as IC; + let ext = code_vm_ext(code); let mut value = root; let mut count = 0; for pc in attr_pc..attr_pc.saturating_add(8) { @@ -16627,6 +16662,12 @@ impl Interpreter { let Object::Instance(inst) = value else { break; }; + // SAFETY: the rooted, callback-free walk described below. + if let Some(next) = ext.and_then(|ext| unsafe { field_slot_hit(ext, pc, inst) }) { + value = next; + count += 1; + continue; + } let cls = inst.cls_raw(); if cls.native_kind.get() != 0 { break; @@ -16665,7 +16706,13 @@ impl Interpreter { IC::LoadAttrInstance { key_idx, ver: cached, - } if ver == cached => indexed(key_idx), + } if ver == cached => { + let hit = indexed(key_idx); + if let (Some(_), Some(ext)) = (hit, ext) { + field_slot_note(ext, code.instructions.len(), pc, inst, key_idx); + } + hit + } _ => None, }; primary.or_else(|| { @@ -16723,6 +16770,26 @@ impl Interpreter { ) -> bool { use weavepy_compiler::InlineCache as IC; let cls = inst.cls_raw(); + let ext = code_vm_ext(code); + if let Some((ver, idx)) = ext + .and_then(|e| e.field_slots.get()) + .and_then(|slots| slots.get(attr_pc)) + .map(FieldSlot::get) + { + if cls.attr_version.get() == ver && !crate::capi_watchers::dicts_active() { + // SAFETY: a store between two instructions (see `peek_mut`). + if let Some(slot) = + unsafe { inst.split_field_mut(idx as usize, value.is_gc_atomic()) } + { + if Self::core_droppable(slot) { + // SAFETY: as the indexed store below. + drop(std::mem::replace(slot, unsafe { std::ptr::read(value) })); + return true; + } + return false; + } + } + } match code.caches.get(attr_pc as u32) { IC::StoreAttrInstance { key_idx, ver } => { if cls.native_kind.get() != 0 @@ -16746,6 +16813,9 @@ impl Interpreter { // and the displaced value was checked droppable above. let old = std::mem::replace(slot, unsafe { std::ptr::read(value) }); drop(old); + if let Some(ext) = ext { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, key_idx); + } true } IC::StoreAttrNewKey { ver } => { @@ -61903,6 +61973,9 @@ struct CodeConstObjects { /// splits them back apart; the core loop reads one byte to run the /// pair as a single dispatch. Empty when the code has no pair. fast_pairs: std::sync::OnceLock>, + /// Split-layout attribute shortcuts per `LOAD_ATTR` site (see + /// [`FieldSlot`]); allocated on the first recorded one. + field_slots: std::sync::OnceLock>, } /// A `CALL` site's inline-call shape (see `Interpreter::core_call`): the @@ -62134,6 +62207,76 @@ impl StampSlot { } } +/// A `LOAD_ATTR` site's split-layout shortcut: the receiver class's +/// attribute version and the attribute's position among the class's +/// shared names, recorded after a full site read proved the name there +/// (see [`field_slot_note`]). The names are append-only and the version +/// is process-unique, so while both match, the value at that position of +/// a split instance *is* the attribute: no inline-cache decode and no +/// name comparison. +struct FieldSlot(std::cell::UnsafeCell<(u64, u32)>); + +// SAFETY: as `StampSlot`. +unsafe impl Send for FieldSlot {} +unsafe impl Sync for FieldSlot {} + +impl FieldSlot { + const fn empty() -> Self { + Self(std::cell::UnsafeCell::new((0, 0))) + } + + #[inline(always)] + fn get(&self) -> (u64, u32) { + // SAFETY: GIL-serialized; no `&mut` escapes `set`. + unsafe { *self.0.get() } + } + + #[inline] + fn set(&self, v: (u64, u32)) { + // SAFETY: as `StampSlot::set`. + unsafe { *self.0.get() = v }; + } +} + +/// The attribute `inst` holds for the `LOAD_ATTR` at `pc`, through the +/// site's [`FieldSlot`] (never recorded for a slot or native class). +/// +/// # Safety +/// +/// As [`crate::sync::GilCell::peek`]: the view must not outlive anything +/// that could store an attribute or release the receiver. +#[inline(always)] +unsafe fn field_slot_hit<'a>( + ext: &CodeConstObjects, + pc: usize, + inst: &'a PyInstance, +) -> Option<&'a Object> { + let (ver, idx) = ext.field_slots.get()?.get(pc)?.get(); + if inst.cls_raw().attr_version.get() != ver { + return None; + } + // SAFETY: forwarded contract. + unsafe { inst.split_field(idx as usize) } +} + +/// Record the [`FieldSlot`] shortcut for the `LOAD_ATTR` at `pc`, after +/// a guarded read found its name at dictionary position `idx` of `inst` +/// (whose class passed the site's version check): kept only when that +/// position is the split layout's, over the class's own names. +#[inline(never)] +fn field_slot_note(ext: &CodeConstObjects, ninstrs: usize, pc: usize, inst: &PyInstance, idx: u32) { + // SAFETY: a read with nothing running (the caller's own guarded read). + if unsafe { inst.split_field(idx as usize) }.is_none() { + return; + } + let slots = ext + .field_slots + .get_or_init(|| (0..ninstrs).map(|_| FieldSlot::empty()).collect()); + if let Some(slot) = slots.get(pc) { + slot.set((inst.cls_raw().attr_version.get(), idx)); + } +} + /// `v.clone()` with the scalars copied and the common heap variants' /// increments in line (the enum's `Clone` is an out-of-line call and a /// variant switch per clone). @@ -62995,6 +63138,7 @@ fn code_vm_ext_init( call_slots: std::sync::OnceLock::new(), pure_leaf: std::sync::atomic::AtomicU8::new(0), fast_pairs: std::sync::OnceLock::new(), + field_slots: std::sync::OnceLock::new(), }) }) } From 09d7711c7a404e70f2c28a2b6186670bbd548de9 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 18:46:29 -0700 Subject: [PATCH 23/65] perf: run leaf bodies from a pre-translated register plan The frameless leaf evaluator now translates each pure or effect leaf once into register-addressed ops (locals and stack positions get fixed registers, jumps name their target op, and local loads, POP_TOP, COPY and SWAP mostly disappear) and runs them with one unchecked dispatch per operation. A path that reaches an instruction the plan can't express still declines only when it runs. --- crates/weavepy-vm/src/leaf_plan.rs | 1187 ++++++++++++++++++++++++++++ crates/weavepy-vm/src/lib.rs | 663 +--------------- 2 files changed, 1200 insertions(+), 650 deletions(-) create mode 100644 crates/weavepy-vm/src/leaf_plan.rs diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs new file mode 100644 index 00000000..7c427151 --- /dev/null +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -0,0 +1,1187 @@ +//! Pre-decoded leaf bodies for the frameless leaf evaluator. +//! +//! A pure or effect leaf (see `code_pure_leaf_decide`) runs without an +//! activation: its operands are scalars or borrowed pointers, and any +//! miss abandons the evaluation having done nothing observable. This +//! module translates such a body once into a [`LeafPlan`]: every local +//! and every operand-stack position gets a fixed register, jumps name +//! their target op directly, and local loads, `POP_TOP`, `COPY` and +//! `SWAP` mostly disappear into operand addressing. Running a plan is +//! then one dispatch per real operation, with no stack pointer, no +//! bounds checks, and no per-read "was this local assigned" test. + +use weavepy_compiler::{BinOpKind, CodeObject, CompareKind, OpCode, UnaryKind}; + +use crate::object::Object; +use crate::sync::Rc; +use crate::{CodeConstObjects, Interpreter}; + +/// A leaf operand: a scalar by value, or a borrowed heap object. +#[derive(Clone, Copy)] +// Every payload sits at offset 8 (a primitive representation), so a copy +// is two words rather than a byte-wise shuffle. +#[repr(u64)] +pub(crate) enum V { + /// A resolved callee: a function the class or namespace holds. + Fn(*const crate::object::PyFunction), + /// A call's empty self slot. + Null, + /// A heap object, borrowed. + R(*const Object), + I(i64), + F(f64), + B(bool), + N, +} + +#[inline(always)] +pub(crate) fn norm(p: *const Object) -> V { + // SAFETY: `p` names a live object (see `Interpreter::leaf_eval`). + match unsafe { &*p } { + Object::Int(i) => V::I(*i), + Object::Float(x) => V::F(*x), + Object::Bool(b) => V::B(*b), + Object::None => V::N, + _ => V::R(p), + } +} + +#[inline(always)] +pub(crate) fn truth(v: V) -> Option { + Some(match v { + V::B(b) => b, + V::I(i) => i != 0, + V::F(x) => x != 0.0, + V::N => false, + // SAFETY: as `norm`. + V::R(p) => match unsafe { &*p } { + Object::Str(s) => !s.is_empty(), + Object::Tuple(t) => !t.is_empty(), + // SAFETY: a read with nothing running (see `peek`). + Object::List(l) => !unsafe { l.peek() }?.is_empty(), + Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), + _ => return None, + }, + V::Fn(_) | V::Null => return None, + }) +} + +/// The callback-free comparison rules both the decoded field shapes and +/// the plan runner use. NaNs and unsupported operands fall back to the +/// interpreter. +#[inline(always)] +pub(crate) fn compare(a: V, b: V, kind: CompareKind) -> Option { + const EXACT: u64 = 1 << 53; + let ord = match (a, b) { + (V::I(x), V::I(y)) => x.cmp(&y), + (V::F(x), V::F(y)) => x.partial_cmp(&y)?, + (V::I(x), V::F(y)) if x.unsigned_abs() < EXACT => (x as f64).partial_cmp(&y)?, + (V::F(x), V::I(y)) if y.unsigned_abs() < EXACT => x.partial_cmp(&(y as f64))?, + (V::B(x), V::B(y)) => x.cmp(&y), + (V::B(x), V::I(y)) => i64::from(x).cmp(&y), + (V::I(x), V::B(y)) => x.cmp(&i64::from(y)), + (V::N, V::N) if matches!(kind, CompareKind::Eq | CompareKind::NotEq) => { + std::cmp::Ordering::Equal + } + // SAFETY: these pointers name values owned by live arguments or + // the evaluator's scratch; no Python runs. + (V::R(p), V::R(q)) => match (unsafe { &*p }, unsafe { &*q }) { + (Object::Str(s), Object::Str(t)) => (**s).cmp(&**t), + _ => return None, + }, + _ => return None, + }; + Some(match kind { + CompareKind::Lt => ord.is_lt(), + CompareKind::LtE => ord.is_le(), + CompareKind::Eq => ord.is_eq(), + CompareKind::NotEq => ord.is_ne(), + CompareKind::Gt => ord.is_gt(), + CompareKind::GtE => ord.is_ge(), + }) +} + +/// Registers a plan may address: the locals, then the operand stack. +const REGS: usize = 32; + +/// One plan operation. Registers are `u8` indices below [`REGS`]; `pc` +/// is the source instruction (its inline caches and site slots), and +/// `name` its `co_names` index. +#[derive(Clone, Copy, Debug)] +enum Op { + /// `regs[dst] = regs[src]`. + Move { + dst: u8, + src: u8, + }, + /// `regs[dst] = consts[k]`. + Const { + dst: u8, + k: u16, + }, + Global { + dst: u8, + pc: u16, + }, + Attr { + dst: u8, + src: u8, + pc: u16, + name: u16, + }, + /// The method-form load: `regs[dst]` the function, `regs[dst + 1]` + /// the receiver (or the empty self slot for a class receiver). + Method { + dst: u8, + src: u8, + pc: u16, + name: u16, + }, + Compare { + dst: u8, + a: u8, + b: u8, + kind: u8, + }, + Is { + dst: u8, + a: u8, + b: u8, + invert: bool, + }, + Truth { + dst: u8, + src: u8, + }, + Unary { + dst: u8, + src: u8, + kind: u8, + }, + Binary { + dst: u8, + a: u8, + b: u8, + kind: u8, + }, + /// Exchange two registers. + Swap { + a: u8, + b: u8, + }, + /// Jump to op `target` when `regs[src]`'s truth is `when`. + BranchIf { + src: u8, + when: bool, + target: u16, + }, + /// Jump to op `target` when `regs[src]` `is None` is `when`. + BranchNone { + src: u8, + when: bool, + target: u16, + }, + Jump { + target: u16, + }, + /// `regs[at]` the callee, `regs[at + 1]` its self slot, then `argc` + /// arguments; the result lands in `regs[at]`. + Call { + at: u8, + argc: u8, + }, + Return { + src: u8, + }, + /// Abandon the evaluation: the path reached an instruction the plan + /// can't run. + Decline, + /// An effect leaf's buffered `recv.name = val`. + StoreAttr { + recv: u8, + val: u8, + pc: u16, + name: u16, + }, +} + +/// A leaf body, translated (see the module docs). +pub(crate) struct LeafPlan { + ops: Box<[Op]>, + /// The loaded constants, normalized. A heap constant points into the + /// code extension's materialized constants, which outlive the plan. + consts: Box<[V]>, + /// Arguments, copied into the first registers on entry. + nargs: u8, +} + +// SAFETY: the constants' pointers name the code extension's immutable +// materialized constants (`Object`s shared the way the extension itself +// shares them); a plan is only read. +unsafe impl Send for LeafPlan {} +// SAFETY: as above. +unsafe impl Sync for LeafPlan {} + +/// A value on the translator's abstract operand stack: the register +/// that holds it. A stack position's own register is `nl + position`; +/// a local's value is read in place from the local's register until +/// something would overwrite it. +type Opnd = u8; + +struct Builder<'a> { + code: &'a CodeObject, + ext: &'a CodeConstObjects, + ops: Vec, + consts: Vec, + /// Number of locals (registers below it). + nl: usize, + stack: Vec, + /// Locals definitely assigned on the current path (bit per local). + assigned: u32, + /// Per bytecode target: the stack depth and assigned set on the + /// edges seen so far. + targets: std::collections::HashMap, + /// `(op index, bytecode target)` of every emitted jump. + fixups: Vec<(usize, usize)>, + /// The op index each bytecode instruction starts at. + starts: Vec, +} + +impl Builder<'_> { + fn slot(&self, pos: usize) -> Option { + let r = self.nl + pos; + (r < REGS).then_some(r as u8) + } + + fn push_new(&mut self) -> Option { + let r = self.slot(self.stack.len())?; + self.stack.push(r); + Some(r) + } + + fn pop(&mut self) -> Option { + self.stack.pop() + } + + /// Put every stack entry in its own position's register. + fn canonicalize(&mut self) -> Option<()> { + for pos in 0..self.stack.len() { + let own = self.slot(pos)?; + if self.stack[pos] != own { + // Entries only ever name locals or their own position + // (see `swap` and `copy`), so this write clobbers none. + self.ops.push(Op::Move { + dst: own, + src: self.stack[pos], + }); + self.stack[pos] = own; + } + } + Some(()) + } + + /// Before local `i` is overwritten: every stack entry still reading + /// it moves into its own position's register. + fn release_local(&mut self, i: u8) -> Option<()> { + for pos in 0..self.stack.len() { + if self.stack[pos] == i { + let own = self.slot(pos)?; + self.ops.push(Op::Move { dst: own, src: i }); + self.stack[pos] = own; + } + } + Some(()) + } + + fn load_local(&mut self, i: u32) -> Option<()> { + let i = i as usize; + if i >= self.nl || self.assigned & (1 << i) == 0 { + return None; + } + self.stack.push(i as u8); + Some(()) + } + + fn store_local(&mut self, i: u32) -> Option<()> { + let i = i as usize; + if i >= self.nl { + return None; + } + let src = self.pop()?; + if src != i as u8 { + self.release_local(i as u8)?; + self.ops.push(Op::Move { dst: i as u8, src }); + } + self.assigned |= 1 << i; + Some(()) + } + + fn constant(&mut self, v: V) -> Option<()> { + let k = u16::try_from(self.consts.len()).ok()?; + self.consts.push(v); + let dst = self.push_new()?; + self.ops.push(Op::Const { dst, k }); + Some(()) + } + + /// Record a jump edge to bytecode `target` with the current state + /// (already canonical), and emit its fixup. + fn edge(&mut self, target: usize) -> Option<()> { + let state = (self.stack.len(), self.assigned); + match self.targets.get_mut(&target) { + Some((depth, assigned)) => { + if *depth != state.0 { + return None; + } + *assigned &= state.1; + } + None => { + self.targets.insert(target, state); + } + } + self.fixups.push((self.ops.len() - 1, target)); + Some(()) + } + + /// Translate the instruction at `pc`: whether control falls through + /// to the next one. `None` for one the plan can't express (the + /// caller rolls back what it emitted and declines on that path). + fn step(&mut self, pc: usize, ins: weavepy_compiler::Instruction) -> Option { + let arg = ins.arg; + match ins.op { + OpCode::Resume | OpCode::Nop | OpCode::NotTaken => {} + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { + self.load_local(arg)?; + } + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + self.load_local(arg >> 4)?; + self.load_local(arg & 15)?; + } + OpCode::StoreFast => self.store_local(arg)?, + OpCode::StoreFastLoadFast => { + self.store_local(arg >> 4)?; + self.load_local(arg & 15)?; + } + OpCode::StoreFastStoreFast => { + self.store_local(arg >> 4)?; + self.store_local(arg & 15)?; + } + OpCode::LoadConst => { + let v = norm(self.ext.objects.get(arg as usize)?); + self.constant(v)?; + } + OpCode::LoadSmallInt => self.constant(V::I(i64::from(arg)))?, + OpCode::PushNull => self.constant(V::Null)?, + OpCode::LoadGlobal => { + let dst = self.push_new()?; + self.ops.push(Op::Global { + dst, + pc: u16::try_from(pc).ok()?, + }); + } + OpCode::LoadGlobalPushNull => { + let dst = self.push_new()?; + self.ops.push(Op::Global { + dst, + pc: u16::try_from(pc).ok()?, + }); + self.constant(V::Null)?; + } + OpCode::LoadAttr => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Attr { + dst, + src, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + OpCode::LoadMethodAttr => { + let src = self.pop()?; + let dst = self.push_new()?; + self.push_new()?; + self.ops.push(Op::Method { + dst, + src, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + OpCode::CompareOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + let kind = (arg & !weavepy_compiler::COMPARE_OP_TO_BOOL_FLAG) as u8; + if kind > CompareKind::GtE as u8 { + return None; + } + self.ops.push(Op::Compare { dst, a, b, kind }); + } + OpCode::IsOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Is { + dst, + a, + b, + invert: arg == 1, + }); + } + OpCode::ToBool => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Truth { dst, src }); + } + OpCode::UnaryOp => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Unary { + dst, + src, + kind: arg as u8, + }); + } + OpCode::BinaryOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + // (The in-place flag sits above the kind's byte, and + // means nothing for the scalars this runs.) + self.ops.push(Op::Binary { + dst, + a, + b, + kind: arg as u8, + }); + } + OpCode::CopyTop => { + let n = (arg as usize).max(1); + let depth = self.stack.len(); + if n > depth { + return None; + } + let src = self.stack[depth - n]; + let dst = self.push_new()?; + self.ops.push(Op::Move { dst, src }); + } + OpCode::Swap => { + let n = arg as usize; + let depth = self.stack.len(); + if n < 2 || n > depth { + return None; + } + // Both entries move into their own registers first, so + // the exchange keeps every entry in its own position. + let (lo, hi) = (depth - n, depth - 1); + for pos in [lo, hi] { + let own = self.slot(pos)?; + if self.stack[pos] != own { + self.ops.push(Op::Move { + dst: own, + src: self.stack[pos], + }); + self.stack[pos] = own; + } + } + self.ops.push(Op::Swap { + a: self.stack[lo], + b: self.stack[hi], + }); + } + OpCode::PopTop => { + self.pop()?; + } + OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::PopJumpIfNone + | OpCode::PopJumpIfNotNone => { + // The condition is a local or its own position's + // register, which lies above the remaining entries: + // canonicalizing them writes neither. + let src = self.pop()?; + self.canonicalize()?; + let target = pc + 1 + arg as usize; + self.ops.push(match ins.op { + OpCode::PopJumpIfFalse => Op::BranchIf { + src, + when: false, + target: 0, + }, + OpCode::PopJumpIfTrue => Op::BranchIf { + src, + when: true, + target: 0, + }, + OpCode::PopJumpIfNone => Op::BranchNone { + src, + when: true, + target: 0, + }, + _ => Op::BranchNone { + src, + when: false, + target: 0, + }, + }); + self.edge(target)?; + } + OpCode::JumpForward => { + self.canonicalize()?; + self.ops.push(Op::Jump { target: 0 }); + self.edge(pc + 1 + arg as usize)?; + return Some(false); + } + OpCode::Call => { + let argc = arg as usize; + let depth = self.stack.len(); + if depth < argc + 2 { + return None; + } + self.canonicalize()?; + let at = depth - argc - 2; + self.stack.truncate(at); + let r = self.push_new()?; + self.ops.push(Op::Call { + at: r, + argc: u8::try_from(argc).ok()?, + }); + } + OpCode::ReturnValue => { + let src = self.pop()?; + self.ops.push(Op::Return { src }); + return Some(false); + } + OpCode::StoreAttr => { + let recv = self.pop()?; + let val = self.pop()?; + self.ops.push(Op::StoreAttr { + recv, + val, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + _ => return None, + } + Some(true) + } + + fn build(mut self) -> Option { + let instrs = &self.code.instructions; + // Whether the previous instruction falls through to this one. + let mut live = true; + for pc in 0..instrs.len() { + if let Some(&(depth, assigned)) = self.targets.get(&pc) { + if live { + // A fallthrough into a merge point arrives canonical + // (its moves run before the point the jumps land on). + self.canonicalize()?; + if self.stack.len() != depth { + return None; + } + self.assigned &= assigned; + } else { + self.assigned = assigned; + } + self.stack.clear(); + for pos in 0..depth { + let r = self.slot(pos)?; + self.stack.push(r); + } + live = true; + } + self.starts.push(u16::try_from(self.ops.len()).ok()?); + if !live { + // Unreachable code: nothing jumps here. + continue; + } + let mark = (self.ops.len(), self.fixups.len()); + match self.step(pc, instrs[pc]) { + Some(next) => live = next, + None => { + // The path that reaches this instruction declines + // (the old evaluator's behaviour: only a path that + // runs an unsupported instruction abandons). + self.ops.truncate(mark.0); + self.fixups.truncate(mark.1); + self.ops.push(Op::Decline); + live = false; + } + } + } + // Every path ends in a return or a jump: the runner never steps + // past the last op. + if live || self.ops.is_empty() { + return None; + } + for &(at, target) in &self.fixups { + let to = *self.starts.get(target)?; + if usize::from(to) >= self.ops.len() { + return None; + } + match &mut self.ops[at] { + Op::BranchIf { target, .. } + | Op::BranchNone { target, .. } + | Op::Jump { target } => { + *target = to; + } + _ => return None, + } + } + Some(LeafPlan { + ops: self.ops.into_boxed_slice(), + consts: self.consts.into_boxed_slice(), + nargs: u8::try_from(self.code.arg_count).ok()?, + }) + } +} + +/// Translate `code` (a pure or effect leaf) into a plan; `None` for a +/// body the plan can't express (the ordinary call runs it instead). +pub(crate) fn build(code: &CodeObject, ext: &CodeConstObjects) -> Option { + let nl = code.varnames.len(); + let nargs = code.arg_count as usize; + if nl > 16 || nargs > nl || nargs > 8 || code.instructions.len() > u16::MAX as usize { + return None; + } + Builder { + code, + ext, + ops: Vec::new(), + consts: Vec::new(), + nl, + stack: Vec::new(), + assigned: (1u32 << nargs) - 1, + targets: std::collections::HashMap::new(), + fixups: Vec::new(), + starts: Vec::with_capacity(code.instructions.len()), + } + .build() +} + +/// Values a leaf path hands back owned (a polymorphic or class read, a +/// nested call's result) stay here until the evaluation ends; registers +/// point in. +struct Owned { + buf: [std::mem::MaybeUninit; 4], + n: usize, +} + +impl Drop for Owned { + fn drop(&mut self) { + for k in 0..self.n { + // SAFETY: the first `n` entries are initialized. + crate::drop_hot(unsafe { self.buf[k].assume_init_read() }); + } + } +} + +impl Owned { + /// Hold `v` (a scalar needs no holding) and name it. + #[inline] + fn own(&mut self, v: Object) -> Option { + Some(match v { + Object::Int(i) => V::I(i), + Object::Float(x) => V::F(x), + Object::Bool(b) => V::B(b), + Object::None => V::N, + v => { + if self.n == self.buf.len() { + return None; + } + let slot = &mut self.buf[self.n]; + slot.write(v); + self.n += 1; + V::R(slot.as_ptr()) + } + }) + } +} + +/// An effect leaf's attribute stores, in order, until the return commits +/// them: receiver, store pc, name index, value (all owned). Entries are +/// only appended, so a value read back from one stays put while the +/// evaluation runs. +const PENDING: usize = 6; + +/// The receiver is a stable location for the whole evaluation: an +/// argument (rooted by the caller) or an owned-scratch clone. +type Store = (*const Object, u32, u32, Object); + +struct Pending { + buf: [std::mem::MaybeUninit; PENDING], + n: usize, + /// Entries whose value the return moved into a dict. + moved: u8, +} + +impl Pending { + fn get(&self, k: usize) -> &Store { + debug_assert!(k < self.n); + // SAFETY: the first `n` entries are initialized. + unsafe { self.buf[k].assume_init_ref() } + } + + /// Whether entry `k` is the last store to its attribute. + fn latest(&self, k: usize) -> bool { + let (r0, _, n0, _) = self.get(k); + !(k + 1..self.n).any(|j| { + let (r1, _, n1, _) = self.get(j); + // SAFETY: stable receivers (see `Store`). + n1 == n0 && unsafe { (**r1).is_same(&**r0) } + }) + } +} + +impl Drop for Pending { + fn drop(&mut self) { + for k in 0..self.n { + // SAFETY: the first `n` entries are initialized; a moved value + // is left in place, not dropped again. + let (_, _, _, value) = unsafe { self.buf[k].assume_init_read() }; + if self.moved & (1 << k) == 0 { + crate::drop_hot(value); + } else { + std::mem::forget(value); + } + } + } +} + +/// An owned object for a leaf value (the return value, a buffered +/// store's value); `None` for the markers that are never values. +#[inline(always)] +fn to_object(v: V) -> Option { + Some(match v { + // SAFETY: as `norm` (an owned value is cloned before its holder + // drops). + V::R(p) => crate::clone_hot(unsafe { &*p }), + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Null => return None, + }) +} + +impl Interpreter { + /// Run `plan` (the translation of `code`, whose namespaces are `f`'s) + /// on borrowed `args`, at call-nesting depth `nest`: the leaf's + /// result, or `None` having done nothing observable (see + /// `Interpreter::leaf_eval`). + #[inline(never)] + pub(crate) fn leaf_run( + &self, + code: &CodeObject, + ext: &CodeConstObjects, + plan: &LeafPlan, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + use weavepy_compiler::InlineCache as IC; + /// Nested leaf calls evaluated in place at most this deep. + const NEST: u8 = 3; + let nargs = usize::from(plan.nargs); + if args.len() != nargs { + return None; + } + let mut regs = [const { std::mem::MaybeUninit::::uninit() }; REGS]; + for (k, &a) in args.iter().enumerate() { + regs[k].write(norm(a)); + } + // The translation proves every register is written before it's + // read, and that every index is below `REGS`. + macro_rules! get { + ($r:expr) => { + // SAFETY: see above. + unsafe { regs.get_unchecked(usize::from($r)).assume_init() } + }; + } + macro_rules! set { + ($r:expr, $v:expr) => {{ + let v = $v; + // SAFETY: see above. + unsafe { regs.get_unchecked_mut(usize::from($r)).write(v) }; + }}; + } + let consts: &[V] = &plan.consts; + let stamps: &[crate::StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + let mut owned = Owned { + buf: [const { std::mem::MaybeUninit::uninit() }; 4], + n: 0, + }; + let mut pend = Pending { + buf: [const { std::mem::MaybeUninit::uninit() }; PENDING], + n: 0, + moved: 0, + }; + let ops: &[Op] = &plan.ops; + let mut ip = 0usize; + loop { + // SAFETY: every path ends in a return or a jump, and every + // jump target is an op (checked when the plan was built). + let op = unsafe { *ops.get_unchecked(ip) }; + ip += 1; + match op { + Op::Move { dst, src } => set!(dst, get!(src)), + // SAFETY: `k` names a plan constant (the translation). + Op::Const { dst, k } => set!(dst, unsafe { *consts.get_unchecked(usize::from(k)) }), + Op::Global { dst, pc } => { + // The callee's own namespaces, stamp-validated as the + // core loop's `LOAD_GLOBAL` arm. + let pc = usize::from(pc); + let slot = stamps.get(pc)?; + let (gdict, bdict) = (f.globals.as_ptr(), f.builtins.as_ptr()); + let gid = crate::specialize::rc_id(&f.globals); + // SAFETY (raw dict reads): nothing runs code here. + let g_stamp = unsafe { (*gdict).mutation_stamp() }; + let hit = match code.caches.get(pc as u32) { + IC::LoadGlobalModule { + globals_id, + key_idx, + } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { + (*gdict).get_index(key_idx as usize) + }, + IC::LoadGlobalBuiltin { + builtins_id, + key_idx, + } if builtins_id == crate::specialize::rc_id(&f.builtins) + && !self.globals_missing_any.get() + && slot.get() + == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => + unsafe { (*bdict).get_index(key_idx as usize) }, + _ => return None, + }; + set!(dst, norm(hit?.1)); + } + Op::Attr { dst, src, pc, name } => { + let V::R(p) = get!(src) else { + return None; + }; + let (pc, name) = (u32::from(pc), u32::from(name)); + // SAFETY: as `norm`. + let recv = unsafe { &*p }; + if EFFECT && pend.n > 0 { + // A buffered store to this attribute, the latest. + let hit = (0..pend.n) + .rev() + .map(|k| pend.get(k)) + // SAFETY: stable receivers (see `Store`). + .find(|(r, _, n, _)| *n == name && unsafe { (**r).is_same(recv) }); + if let Some((_, _, _, v)) = hit { + set!(dst, norm(v)); + continue; + } + } + let v = match recv { + Object::Instance(inst) => { + if GETTER && inst.cls_raw().native_kind.get() != 0 { + return None; + } + // SAFETY: the receiver remains rooted by an + // argument or owned scratch; no Python runs. + let hit = unsafe { + Self::leaf_cached_instance_field(ext, code, inst, pc, name) + } + .map(std::ptr::from_ref); + match hit { + Some(v) => norm(v), + // A stale or absent site cache resolves + // through the site's own entries. + None => owned.own(Self::leaf_attr_resolve_site( + code, inst, recv, pc, name, + )?)?, + } + } + Object::Type(cls) => { + if GETTER && !Self::plain_metaclass(cls) { + return None; + } + match stamps + .get(pc as usize) + .and_then(|s| crate::class_attr_hit(s, cls)) + { + Some(v) => owned.own(v)?, + None => { + owned.own(Self::leaf_load_type_attr(code, cls, pc, name)?)? + } + } + } + Object::Module(module) => { + if GETTER + && (crate::object::module_class(module).is_some() + || code.names.get(name as usize)?.starts_with("__")) + { + return None; + } + owned.own(Self::leaf_load_attr_recv(code, recv, pc, name)?)? + } + _ => return None, + }; + set!(dst, v); + } + Op::Method { dst, src, pc, name } => { + // A method off the site's slot: the function under + // the receiver, or a class's function with an empty + // self slot (as the core loop's arm). + let V::R(p) = get!(src) else { + return None; + }; + let ms = crate::code_method_slot(code, u32::from(pc))?; + // SAFETY: as `norm`. + match unsafe { &*p } { + Object::Instance(inst) => { + let cls = inst.cls_raw(); + if !Self::default_getattribute(cls) { + return None; + } + let fp = ms.peek_fn(cls.attr_version.get())?; + // The instance's attributes must not shadow + // the method. + if crate::inst_may_shadow(inst, code, u32::from(name)) { + return None; + } + set!(dst, V::Fn(fp)); + set!(dst + 1, V::R(p)); + } + Object::Type(cls) => { + set!(dst, V::Fn(ms.peek_unbound(cls.attr_version.get())?)); + set!(dst + 1, V::Null); + } + _ => return None, + } + } + Op::Compare { dst, a, b, kind } => { + // SAFETY: the translation checked `kind` names a + // comparison. + let kind: CompareKind = unsafe { std::mem::transmute(kind) }; + set!(dst, V::B(compare(get!(a), get!(b), kind)?)); + } + Op::Is { dst, a, b, invert } => { + let same = match (get!(a), get!(b)) { + (V::N, V::N) => true, + (V::B(x), V::B(y)) => x == y, + (V::I(x), V::I(y)) => Object::Int(x).is_same(&Object::Int(y)), + (V::F(x), V::F(y)) => Object::Float(x).is_same(&Object::Float(y)), + // SAFETY: as `norm`. + (V::R(p), V::R(q)) => unsafe { (*p).is_same(&*q) }, + _ => false, + }; + set!(dst, V::B(same != invert)); + } + Op::Truth { dst, src } => set!(dst, V::B(truth(get!(src))?)), + Op::Unary { dst, src, kind } => { + let v = get!(src); + let r = match kind { + k if k == UnaryKind::Not as u8 => V::B(!truth(v)?), + k if k == UnaryKind::Neg as u8 => match v { + V::I(i) => V::I(i.checked_neg()?), + V::F(x) => V::F(-x), + _ => return None, + }, + k if k == UnaryKind::Pos as u8 => match v { + V::I(_) | V::F(_) => v, + _ => return None, + }, + k if k == UnaryKind::Invert as u8 => match v { + V::I(i) => V::I(!i), + _ => return None, + }, + _ => return None, + }; + set!(dst, r); + } + Op::Binary { dst, a, b, kind } => { + let r = match (get!(a), get!(b)) { + (V::I(a), V::I(b)) => V::I(match kind { + k if k == BinOpKind::Add as u8 => a.checked_add(b)?, + k if k == BinOpKind::Sub as u8 => a.checked_sub(b)?, + k if k == BinOpKind::Mult as u8 => a.checked_mul(b)?, + k if k == BinOpKind::BitAnd as u8 => a & b, + k if k == BinOpKind::BitOr as u8 => a | b, + k if k == BinOpKind::BitXor as u8 => a ^ b, + k if k == BinOpKind::RShift as u8 && (0..64).contains(&b) => a >> b, + k if k == BinOpKind::LShift as u8 + && (0..63).contains(&b) + && ((a << b) >> b) == a => + { + a << b + } + k if k == BinOpKind::FloorDiv as u8 && b > 0 && a >= 0 => a / b, + k if k == BinOpKind::Mod as u8 && b > 0 && a >= 0 => a % b, + _ => return None, + }), + (x, y) => { + let (x, y) = match (x, y) { + (V::F(x), V::F(y)) => (x, y), + (V::I(x), V::F(y)) => (x as f64, y), + (V::F(x), V::I(y)) => (x, y as f64), + _ => return None, + }; + // SAFETY: as the core loop's `BINARY_OP` arm + // (the compiler emits only valid kinds). + let kind: BinOpKind = unsafe { std::mem::transmute(kind) }; + match Self::leaf_float_op(x, y, kind)? { + Object::Float(x) => V::F(x), + _ => return None, + } + } + }; + set!(dst, r); + } + Op::Swap { a, b } => { + let (x, y) = (get!(a), get!(b)); + set!(a, y); + set!(b, x); + } + Op::BranchIf { src, when, target } => { + if truth(get!(src))? == when { + ip = usize::from(target); + } + } + Op::BranchNone { src, when, target } => { + if matches!(get!(src), V::N) == when { + ip = usize::from(target); + } + } + Op::Jump { target } => ip = usize::from(target), + Op::Decline => return None, + Op::Call { at, argc } => { + // A pure-leaf callee, evaluated in place: only while + // no store is buffered (it would not see one). + if (EFFECT && pend.n > 0) || nest >= NEST { + return None; + } + let fp = match get!(at) { + V::Fn(fp) => fp, + // SAFETY: as `norm`. + V::R(p) => match unsafe { &*p } { + Object::Function(func) => Rc::as_ptr(func), + _ => return None, + }, + _ => return None, + }; + // SAFETY: the class or the namespace holds the callee, + // and nothing here runs code that could release it. + let callee = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the code cell. + let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; + let first = if matches!(get!(at + 1), V::Null) { + at + 2 + } else { + at + 1 + }; + let n = usize::from(at + 2 + argc - first); + if !crate::code_is_pure_leaf(ccode) + || !Self::lean_code_ok(ccode) + || n != ccode.arg_count as usize + || n > 8 + || crate::recursion::current_depth() + usize::from(nest) + 1 + >= crate::recursion::recursion_limit() + { + return None; + } + // Scalar arguments are staged as objects (no drop glue). + let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut ptrs = [std::ptr::null::(); 8]; + for k in 0..n { + let o = match get!(first + k as u8) { + V::R(p) => { + ptrs[k] = p; + continue; + } + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Null => return None, + }; + ptrs[k] = staged[k].write(o); + } + let r = self.leaf_eval::(ccode, callee, &ptrs[..n], nest + 1)?; + set!(at, owned.own(r)?); + } + Op::Return { src } => { + let r = to_object(get!(src))?; + if EFFECT && pend.n > 0 { + // The latest store to each attribute is the one + // that lands; every one of them must go through + // before any does (a decline touched nothing, and + // the ordinary call runs the body instead). A + // lone store is its own check: it declines whole. + if pend.n > 1 { + for k in 0..pend.n { + if !pend.latest(k) { + continue; + } + let (rp, spc, name, _) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + if !Self::core_store_attr_ready(code, inst, *spc as usize, *name) { + return None; + } + } + } + let n = pend.n; + for k in 0..n { + if !pend.latest(k) { + continue; + } + let (rp, spc, name, value) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + // On `true` the value moved into the dict. + if Self::core_store_attr(code, inst, *spc as usize, *name, value) { + pend.moved |= 1 << k; + } else if n == 1 { + // The lone store declined, untouched. + return None; + } else { + debug_assert!(false, "a ready store declined"); + } + } + } + return Some(r); + } + Op::StoreAttr { + recv, + val, + pc, + name, + } => { + if !EFFECT { + return None; + } + let V::R(rp) = get!(recv) else { + return None; + }; + // SAFETY: as `norm`. + let recv = unsafe { &*rp }; + if !matches!(recv, Object::Instance(_)) || pend.n == PENDING { + return None; + } + let value = to_object(get!(val))?; + // An argument receiver is used in place; any other + // (a field's value) is held by the owned scratch. + let rp = if args.iter().any(|&a| std::ptr::eq(a, rp)) { + rp + } else { + let V::R(held) = owned.own(crate::clone_hot(recv))? else { + return None; + }; + held + }; + pend.buf[pend.n].write((rp, u32::from(pc), u32::from(name), value)); + pend.n += 1; + } + } + } + } +} diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 981d68eb..904b7416 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -61,6 +61,7 @@ pub mod import; pub mod import_time; pub mod inst_dict; mod lazy_arc; +mod leaf_plan; pub mod linejump; pub mod malloc_stats; pub mod object; @@ -15272,92 +15273,9 @@ impl Interpreter { args: &[*const Object], nest: u8, ) -> Option { - use weavepy_compiler::InlineCache as IC; - /// Nested leaf calls evaluated in place at most this deep. - const NEST: u8 = 3; - // Every payload sits at offset 8 (a primitive representation), so - // a copy is two words rather than a byte-wise shuffle. - #[derive(Clone, Copy)] - #[repr(u64)] - enum V { - /// A resolved callee: a function the class or namespace holds. - Fn(*const crate::object::PyFunction), - /// A call's empty self slot. - Null, - /// A heap object, borrowed. - R(*const Object), - I(i64), - F(f64), - B(bool), - N, - } - #[inline(always)] - fn norm(p: *const Object) -> V { - // SAFETY: `p` names a live object (see the method docs). - match unsafe { &*p } { - Object::Int(i) => V::I(*i), - Object::Float(x) => V::F(*x), - Object::Bool(b) => V::B(*b), - Object::None => V::N, - _ => V::R(p), - } - } - #[inline(always)] - fn truth(v: V) -> Option { - Some(match v { - V::B(b) => b, - V::I(i) => i != 0, - V::F(x) => x != 0.0, - V::N => false, - // SAFETY: as `norm`. - V::R(p) => match unsafe { &*p } { - Object::Str(s) => !s.is_empty(), - Object::Tuple(t) => !t.is_empty(), - // SAFETY: a read with nothing running (see `peek`). - Object::List(l) => !unsafe { l.peek() }?.is_empty(), - Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), - _ => return None, - }, - V::Fn(_) | V::Null => return None, - }) - } - // Both the decoded field shape and the general evaluator use - // the same callback-free comparison rules. NaNs and unsupported - // operands still fall back to the interpreter. - #[inline(always)] - fn compare(a: V, b: V, kind: CompareKind) -> Option { - const EXACT: u64 = 1 << 53; - let ord = match (a, b) { - (V::I(x), V::I(y)) => x.cmp(&y), - (V::F(x), V::F(y)) => x.partial_cmp(&y)?, - (V::I(x), V::F(y)) if x.unsigned_abs() < EXACT => (x as f64).partial_cmp(&y)?, - (V::F(x), V::I(y)) if y.unsigned_abs() < EXACT => x.partial_cmp(&(y as f64))?, - (V::B(x), V::B(y)) => x.cmp(&y), - (V::B(x), V::I(y)) => i64::from(x).cmp(&y), - (V::I(x), V::B(y)) => x.cmp(&i64::from(y)), - (V::N, V::N) if matches!(kind, CompareKind::Eq | CompareKind::NotEq) => { - std::cmp::Ordering::Equal - } - // SAFETY: these pointers name values owned by live - // arguments or the evaluator's scratch; no Python runs. - (V::R(p), V::R(q)) => match (unsafe { &*p }, unsafe { &*q }) { - (Object::Str(s), Object::Str(t)) => (**s).cmp(&**t), - _ => return None, - }, - _ => return None, - }; - Some(match kind { - CompareKind::Lt => ord.is_lt(), - CompareKind::LtE => ord.is_le(), - CompareKind::Eq => ord.is_eq(), - CompareKind::NotEq => ord.is_ne(), - CompareKind::Gt => ord.is_gt(), - CompareKind::GtE => ord.is_ge(), - }) - } + use leaf_plan::{compare, norm, V}; let ext = code_vm_ext(code)?; let consts: &[Object] = &ext.objects; - let stamps: &[StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); let instrs: &[weavepy_compiler::Instruction] = &code.instructions; // Tiny return bodies need no operand stack or owned-value scratch. // The shape is certified once, alongside the pure-leaf decision; @@ -15472,572 +15390,13 @@ impl Interpreter { _ => {} } } - let mut st = [V::N; 8]; - let mut sp = 0usize; - // Values a leaf path hands back owned (a polymorphic or class - // read) stay here until the evaluation ends; the stack points in. - struct Owned { - buf: [std::mem::MaybeUninit; 4], - n: usize, - } - impl Drop for Owned { - fn drop(&mut self) { - for k in 0..self.n { - // SAFETY: the first `n` entries are initialized. - drop_hot(unsafe { self.buf[k].assume_init_read() }); - } - } - } - let mut owned = Owned { - buf: [const { std::mem::MaybeUninit::uninit() }; 4], - n: 0, - }; - let op: *mut Owned = &raw mut owned; - // An effect leaf's attribute stores, in order, until the return - // commits them: receiver, store pc, name index, value (all owned). - // Entries are only appended, so a value read back from one stays - // put while the evaluation runs. - const PENDING: usize = 6; - // The receiver is a stable location for the whole evaluation: an - // argument (rooted by the caller) or an owned-scratch clone. - type Store = (*const Object, u32, u32, Object); - struct Pending { - buf: [std::mem::MaybeUninit; PENDING], - n: usize, - /// Entries whose value the return moved into a dict. - moved: u8, - } - impl Pending { - fn get(&self, k: usize) -> &Store { - debug_assert!(k < self.n); - // SAFETY: the first `n` entries are initialized. - unsafe { self.buf[k].assume_init_ref() } - } - /// Whether entry `k` is the last store to its attribute. - fn latest(&self, k: usize) -> bool { - let (r0, _, n0, _) = self.get(k); - !(k + 1..self.n).any(|j| { - let (r1, _, n1, _) = self.get(j); - // SAFETY: stable receivers (see `Store`). - n1 == n0 && unsafe { (**r1).is_same(&**r0) } - }) - } - } - impl Drop for Pending { - fn drop(&mut self) { - for k in 0..self.n { - // SAFETY: the first `n` entries are initialized; a - // moved value is left in place, not dropped again. - let (_, _, _, value) = unsafe { self.buf[k].assume_init_read() }; - if self.moved & (1 << k) == 0 { - drop_hot(value); - } else { - std::mem::forget(value); - } - } - } - } - let mut pend = Pending { - buf: [const { std::mem::MaybeUninit::uninit() }; PENDING], - n: 0, - moved: 0, - }; - let own = |v: Object| -> Option { - match v { - Object::Int(i) => Some(V::I(i)), - Object::Float(x) => Some(V::F(x)), - Object::Bool(b) => Some(V::B(b)), - Object::None => Some(V::N), - v => { - // SAFETY: `owned` outlives every use of `op`, and is - // not otherwise touched while the evaluation runs. - let o = unsafe { &mut *op }; - if o.n == 4 { - return None; - } - let slot = &mut o.buf[o.n]; - slot.write(v); - o.n += 1; - Some(V::R(slot.as_ptr())) - } - } - }; - macro_rules! push { - ($v:expr) => {{ - let v = $v; - if sp >= st.len() { - return None; - } - // SAFETY: `sp < st.len()`, checked just above. - unsafe { *st.get_unchecked_mut(sp) = v }; - sp += 1; - }}; - } - macro_rules! pop { - () => {{ - if sp == 0 || sp > st.len() { - return None; - } - sp -= 1; - // SAFETY: `sp < st.len()`, checked just above. - unsafe { *st.get_unchecked(sp) } - }}; - } - // Locals: an argument reads through its pointer until the body - // assigns it; any other local must be assigned before it's read. - const LOCALS: usize = 16; - let mut locs = [V::N; LOCALS]; - let mut bound: u32 = 0; - macro_rules! local { - ($i:expr) => {{ - let i = $i as usize; - if i < LOCALS && bound & (1 << i) != 0 { - locs[i] - } else { - norm(*args.get(i)?) - } - }}; - } - macro_rules! set_local { - ($i:expr, $v:expr) => {{ - let i = $i as usize; - if i >= LOCALS { - return None; - } - locs[i] = $v; - bound |= 1 << i; - }}; - } - let mut pc = 0usize; - loop { - let ins = *instrs.get(pc)?; - match ins.op { - OpCode::Resume | OpCode::Nop | OpCode::NotTaken => {} - OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { - push!(local!(ins.arg)); - } - OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { - push!(local!(ins.arg >> 4)); - push!(local!(ins.arg & 15)); - } - OpCode::StoreFast => { - let v = pop!(); - set_local!(ins.arg, v); - } - OpCode::StoreFastLoadFast => { - let v = pop!(); - set_local!(ins.arg >> 4, v); - push!(local!(ins.arg & 15)); - } - OpCode::StoreFastStoreFast => { - let v = pop!(); - set_local!(ins.arg >> 4, v); - let w = pop!(); - set_local!(ins.arg & 15, w); - } - OpCode::LoadConst => push!(norm(consts.get(ins.arg as usize)?)), - OpCode::LoadSmallInt => push!(V::I(i64::from(ins.arg))), - OpCode::LoadGlobal => { - // The callee's own namespaces, stamp-validated as the - // core loop's `LOAD_GLOBAL` arm. - let slot = stamps.get(pc)?; - let (gdict, bdict) = (f.globals.as_ptr(), f.builtins.as_ptr()); - let gid = specialize::rc_id(&f.globals); - // SAFETY (raw dict reads): nothing runs code here. - let g_stamp = unsafe { (*gdict).mutation_stamp() }; - let hit = match code.caches.get(pc as u32) { - IC::LoadGlobalModule { - globals_id, - key_idx, - } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { - (*gdict).get_index(key_idx as usize) - }, - IC::LoadGlobalBuiltin { - builtins_id, - key_idx, - } if builtins_id == specialize::rc_id(&f.builtins) - && !self.globals_missing_any.get() - && slot.get() - == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => - unsafe { (*bdict).get_index(key_idx as usize) }, - _ => return None, - }; - push!(norm(hit?.1)); - } - OpCode::LoadAttr => { - let V::R(p) = pop!() else { - return None; - }; - // SAFETY: as `norm`. - let recv = unsafe { &*p }; - if EFFECT && pend.n > 0 { - // A buffered store to this attribute, the latest. - let hit = (0..pend.n) - .rev() - .map(|k| pend.get(k)) - // SAFETY: stable receivers (see `Store`). - .find(|(r, _, name, _)| *name == ins.arg && unsafe { (**r).is_same(recv) }); - if let Some((_, _, _, v)) = hit { - push!(norm(v)); - pc += 1; - continue; - } - } - let v = match recv { - Object::Instance(inst) => { - let cls = inst.cls_raw(); - if GETTER && cls.native_kind.get() != 0 { - return None; - } - // SAFETY: the receiver remains rooted by an - // argument or owned scratch; no Python runs. - let hit = unsafe { - Self::leaf_cached_instance_field( - ext, code, inst, pc as u32, ins.arg, - ) - } - .map(std::ptr::from_ref); - match hit { - Some(v) => norm(v), - // A stale or absent site cache resolves - // through the site's own entries — the - // full handler never sees this body, so - // there is nothing to deopt to. - None => own(Self::leaf_attr_resolve_site( - code, inst, recv, pc as u32, ins.arg, - )?)?, - } - } - Object::Type(cls) => { - if GETTER && !Self::plain_metaclass(cls) { - return None; - } - match stamps.get(pc).and_then(|s| class_attr_hit(s, cls)) { - Some(v) => own(v)?, - None => { - own(Self::leaf_load_type_attr(code, cls, pc as u32, ins.arg)?)? - } - } - } - Object::Module(module) => { - if GETTER - && (crate::object::module_class(module).is_some() - || code.names.get(ins.arg as usize)?.starts_with("__")) - { - return None; - } - own(Self::leaf_load_attr_recv(code, recv, pc as u32, ins.arg)?)? - } - _ => return None, - }; - push!(v); - } - OpCode::CompareOp => { - let b = pop!(); - let a = pop!(); - // SAFETY: as in `compare_op_step`. - let kind: CompareKind = - unsafe { std::mem::transmute((ins.arg & !COMPARE_OP_TO_BOOL_FLAG) as u8) }; - push!(V::B(compare(a, b, kind)?)); - } - OpCode::IsOp => { - let b = pop!(); - let a = pop!(); - let same = match (a, b) { - (V::N, V::N) => true, - (V::B(x), V::B(y)) => x == y, - (V::I(x), V::I(y)) => Object::Int(x).is_same(&Object::Int(y)), - (V::F(x), V::F(y)) => Object::Float(x).is_same(&Object::Float(y)), - // SAFETY: as `norm`. - (V::R(p), V::R(q)) => unsafe { (*p).is_same(&*q) }, - _ => false, - }; - push!(V::B(same != (ins.arg == 1))); - } - OpCode::ToBool => { - let v = pop!(); - push!(V::B(truth(v)?)); - } - OpCode::UnaryOp => { - let v = pop!(); - // SAFETY: as in the full handler. - let kind: UnaryKind = unsafe { std::mem::transmute(ins.arg as u8) }; - push!(match (v, kind) { - (v, UnaryKind::Not) => V::B(!truth(v)?), - (V::I(i), UnaryKind::Neg) => V::I(i.checked_neg()?), - (V::F(x), UnaryKind::Neg) => V::F(-x), - (V::I(i), UnaryKind::Pos) => V::I(i), - (V::F(x), UnaryKind::Pos) => V::F(x), - (V::I(i), UnaryKind::Invert) => V::I(!i), - _ => return None, - }); - } - OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue => { - let v = pop!(); - if truth(v)? == (ins.op == OpCode::PopJumpIfTrue) { - pc += ins.arg as usize; - } - } - OpCode::PopJumpIfNone | OpCode::PopJumpIfNotNone => { - let v = pop!(); - if matches!(v, V::N) == (ins.op == OpCode::PopJumpIfNone) { - pc += ins.arg as usize; - } - } - OpCode::JumpForward => pc += ins.arg as usize, - OpCode::BinaryOp => { - let b = pop!(); - let a = pop!(); - // SAFETY: as the core loop's `BINARY_OP` arm. - let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; - let r = match (a, b) { - (V::I(a), V::I(b)) => V::I(match kind { - BinOpKind::Add => a.checked_add(b)?, - BinOpKind::Sub => a.checked_sub(b)?, - BinOpKind::Mult => a.checked_mul(b)?, - BinOpKind::BitAnd => a & b, - BinOpKind::BitOr => a | b, - BinOpKind::BitXor => a ^ b, - BinOpKind::RShift if (0..64).contains(&b) => a >> b, - BinOpKind::LShift if (0..63).contains(&b) && ((a << b) >> b) == a => { - a << b - } - BinOpKind::FloorDiv if b > 0 && a >= 0 => a / b, - BinOpKind::Mod if b > 0 && a >= 0 => a % b, - _ => return None, - }), - (V::F(a), V::F(b)) => match Self::leaf_float_op(a, b, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - (V::I(a), V::F(b)) => match Self::leaf_float_op(a as f64, b, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - (V::F(a), V::I(b)) => match Self::leaf_float_op(a, b as f64, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - _ => return None, - }; - push!(r); - } - OpCode::CopyTop => { - let n = (ins.arg as usize).max(1); - if n > sp { - return None; - } - push!(st[sp - n]); - } - OpCode::Swap => { - let n = ins.arg as usize; - if n < 2 || n > sp { - return None; - } - st.swap(sp - 1, sp - n); - } - OpCode::PopTop => { - let _ = pop!(); - } - OpCode::PushNull => push!(V::Null), - OpCode::LoadGlobalPushNull => { - // As `LOAD_GLOBAL`, then the call's empty self slot. - let slot = stamps.get(pc)?; - let gdict = f.globals.as_ptr(); - let gid = specialize::rc_id(&f.globals); - // SAFETY (raw dict reads): nothing runs code here. - let g_stamp = unsafe { (*gdict).mutation_stamp() }; - let hit = match code.caches.get(pc as u32) { - IC::LoadGlobalModule { - globals_id, - key_idx, - } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { - (*gdict).get_index(key_idx as usize) - }, - _ => return None, - }; - push!(norm(hit?.1)); - push!(V::Null); - } - OpCode::LoadMethodAttr => { - // A method off the site's slot: the function under - // the receiver, or a class's function with an empty - // self slot (as the core loop's arm). - let V::R(p) = pop!() else { - return None; - }; - let ms = code_method_slot(code, pc as u32)?; - // SAFETY: as `norm`. - match unsafe { &*p } { - Object::Instance(inst) => { - let cls = inst.cls_raw(); - if !Self::default_getattribute(cls) { - return None; - } - let fp = ms.peek_fn(cls.attr_version.get())?; - // The instance's attributes must not shadow - // the method. - if inst_may_shadow(inst, code, ins.arg) { - return None; - } - push!(V::Fn(fp)); - push!(V::R(p)); - } - Object::Type(cls) => { - push!(V::Fn(ms.peek_unbound(cls.attr_version.get())?)); - push!(V::Null); - } - _ => return None, - } - } - OpCode::Call => { - // A pure-leaf callee, evaluated in place: only while - // no store is buffered (it would not see one). - let argc = ins.arg as usize; - if (EFFECT && pend.n > 0) || nest >= NEST || sp < argc + 2 { - return None; - } - let at = sp - argc - 2; - let fp = match st[at] { - V::Fn(fp) => fp, - // SAFETY: as `norm`. - V::R(p) => match unsafe { &*p } { - Object::Function(func) => Rc::as_ptr(func), - _ => return None, - }, - _ => return None, - }; - // SAFETY: the class or the namespace holds the callee, - // and nothing here runs code that could release it. - let callee = unsafe { &*fp }; - // SAFETY: GIL-serialized raw read of the code cell. - let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; - let first = if matches!(st[at + 1], V::Null) { - at + 2 - } else { - at + 1 - }; - let n = sp - first; - if !code_is_pure_leaf(ccode) - || !Self::lean_code_ok(ccode) - || n != ccode.arg_count as usize - || n > 8 - || crate::recursion::current_depth() + usize::from(nest) + 1 - >= crate::recursion::recursion_limit() - { - return None; - } - // Scalar arguments are staged as objects (no drop glue). - let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; - let mut ptrs = [std::ptr::null::(); 8]; - for k in 0..n { - let o = match st[first + k] { - V::R(p) => { - ptrs[k] = p; - continue; - } - V::I(i) => Object::Int(i), - V::F(x) => Object::Float(x), - V::B(b) => Object::Bool(b), - V::N => Object::None, - V::Fn(_) | V::Null => return None, - }; - ptrs[k] = staged[k].write(o); - } - let r = self.leaf_eval::(ccode, callee, &ptrs[..n], nest + 1)?; - sp = at; - push!(own(r)?); - } - OpCode::ReturnValue => { - let r = match pop!() { - // SAFETY: as `norm` (an owned value is cloned before - // its holder drops). - V::R(p) => clone_hot(unsafe { &*p }), - V::I(i) => Object::Int(i), - V::F(x) => Object::Float(x), - V::B(b) => Object::Bool(b), - V::N => Object::None, - V::Fn(_) | V::Null => return None, - }; - if EFFECT && pend.n > 0 { - // The latest store to each attribute is the one - // that lands; every one of them must go through - // before any does (a decline touched nothing, and - // the ordinary call runs the body instead). A - // lone store is its own check: it declines whole. - if pend.n > 1 { - for k in 0..pend.n { - if !pend.latest(k) { - continue; - } - let (rp, spc, name, _) = pend.get(k); - // SAFETY: stable receivers (see `Store`). - let Object::Instance(inst) = (unsafe { &**rp }) else { - return None; - }; - if !Self::core_store_attr_ready(code, inst, *spc as usize, *name) { - return None; - } - } - } - let n = pend.n; - for k in 0..n { - if !pend.latest(k) { - continue; - } - let (rp, spc, name, value) = pend.get(k); - // SAFETY: stable receivers (see `Store`). - let Object::Instance(inst) = (unsafe { &**rp }) else { - return None; - }; - // On `true` the value moved into the dict. - if Self::core_store_attr(code, inst, *spc as usize, *name, value) { - pend.moved |= 1 << k; - } else if n == 1 { - // The lone store declined, untouched. - return None; - } else { - debug_assert!(false, "a ready store declined"); - } - } - } - return Some(r); - } - OpCode::StoreAttr if EFFECT => { - let V::R(rp) = pop!() else { - return None; - }; - // SAFETY: as `norm`. - let recv = unsafe { &*rp }; - if !matches!(recv, Object::Instance(_)) || pend.n == PENDING { - return None; - } - let value = match pop!() { - // SAFETY: as `norm`. - V::R(p) => clone_hot(unsafe { &*p }), - V::I(i) => Object::Int(i), - V::F(x) => Object::Float(x), - V::B(b) => Object::Bool(b), - V::N => Object::None, - V::Fn(_) | V::Null => return None, - }; - // An argument receiver is used in place; any other - // (a field's value) is held by the owned scratch. - let rp = if args.iter().any(|&a| std::ptr::eq(a, rp)) { - rp - } else { - let V::R(held) = own(clone_hot(recv))? else { - return None; - }; - held - }; - pend.buf[pend.n].write((rp, pc as u32, ins.arg, value)); - pend.n += 1; - } - _ => return None, - } - pc += 1; - } + // The general body: its translated plan (see `leaf_plan`), built + // on the first evaluation that reaches it. + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run::(code, ext, plan, f, args, nest) } /// The core loop's `FOR_ITER` over a generator at `pc` of `sw`'s @@ -61976,6 +61335,9 @@ struct CodeConstObjects { /// Split-layout attribute shortcuts per `LOAD_ATTR` site (see /// [`FieldSlot`]); allocated on the first recorded one. field_slots: std::sync::OnceLock>, + /// A leaf body's translation for the frameless evaluator (see + /// [`leaf_plan`]), or `None` when the body has none. + leaf_plan: std::sync::OnceLock>>, } /// A `CALL` site's inline-call shape (see `Interpreter::core_call`): the @@ -63139,6 +62501,7 @@ fn code_vm_ext_init( pure_leaf: std::sync::atomic::AtomicU8::new(0), fast_pairs: std::sync::OnceLock::new(), field_slots: std::sync::OnceLock::new(), + leaf_plan: std::sync::OnceLock::new(), }) }) } From 0c0b819acf3b45d05660d00e0b76379cca58db0a Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 20:37:54 -0700 Subject: [PATCH 24/65] perf: construct instances with leaf __init__ bodies frameless A constructor whose __init__ is a pure or effect leaf returning None now runs it through the leaf evaluator on the fresh instance, storing straight into it (nothing else can see it, so a decline just discards it). New-attribute stores take the split-layout shortcut by appending the class's next shared key, and an effect leaf's multi-store commit skips the latest-store scan when every store goes to self under a distinct name. --- crates/weavepy-vm/src/inst_dict.rs | 146 ++++++++++++++++++++ crates/weavepy-vm/src/leaf_plan.rs | 68 +++++++++- crates/weavepy-vm/src/lib.rs | 207 ++++++++++++++++++++++++++++- 3 files changed, 412 insertions(+), 9 deletions(-) diff --git a/crates/weavepy-vm/src/inst_dict.rs b/crates/weavepy-vm/src/inst_dict.rs index 159079ec..442a6013 100644 --- a/crates/weavepy-vm/src/inst_dict.rs +++ b/crates/weavepy-vm/src/inst_dict.rs @@ -380,6 +380,87 @@ impl SplitValues { } } + /// Append `value` as the value of position `i` of `keys`, when that + /// is the next unset position of values laid out over `keys` (or the + /// first value of an instance with none, which adopts `keys` through + /// `share`); `Err` hands the value back, touching nothing. + #[inline(always)] + pub fn append_over( + &mut self, + keys: &SharedKeys, + share: impl FnOnce() -> Rc, + i: usize, + value: Object, + ) -> Result<(), Object> { + if let Some(b) = self.block { + let h = b.as_ptr(); + // SAFETY: the block is live while owned; slot `len` is inside + // the capacity (checked) and uninitialized. + unsafe { + let len = (*h).len as usize; + if std::ptr::eq((*h).keys, keys) && i == len && len < (*h).cap as usize { + Self::values_ptr(h).add(len).write(value); + (*h).len = len as u32 + 1; + return Ok(()); + } + if !(*h).keys.is_null() || len != 0 { + return Err(value); + } + } + } + // No value yet (a fresh or recycled instance): adopt the names. + if i != 0 || keys.is_empty() { + return Err(value); + } + self.first_push(share, keys.len(), value); + Ok(()) + } + + /// [`Self::push`] for an instance's first value, out of line. + #[inline(never)] + fn first_push(&mut self, keys: impl FnOnce() -> Rc, want: usize, value: Object) { + // A recycled block too small for every shared name goes: the + // appends after this one then always find room (see + // `can_append_at`). + if let Some(b) = self.block { + // SAFETY: the block is live while owned; with no value and no + // names (the caller's state) it owns nothing else. + unsafe { + let h = b.as_ptr(); + if ((*h).cap as usize) < want { + debug_assert!((*h).len == 0 && (*h).keys.is_null()); + std::alloc::dealloc(h.cast(), Self::layout((*h).cap as usize)); + self.block = None; + } + } + } + self.push(keys, want, value); + } + + /// Whether [`Self::append_over`] of position `i` of `keys` succeeds + /// once the positions from the current length up to `i` have been + /// appended first (the next of a run of in-order appends). + #[inline] + pub fn can_append_at(&self, keys: &SharedKeys, i: usize) -> bool { + if i >= keys.len() { + return false; + } + match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { + let h = b.as_ptr(); + if std::ptr::eq((*h).keys, keys) { + i >= (*h).len as usize && i < (*h).cap as usize + } else { + // No values yet: the first append sizes the block for + // every name. + (*h).keys.is_null() && (*h).len == 0 + } + }, + None => true, + } + } + /// [`Self::get_index`] with the value writable in place. #[inline(always)] pub fn get_index_mut(&mut self, i: usize) -> Option<(&DictKey, &mut Object)> { @@ -813,6 +894,71 @@ impl crate::types::PyInstance { unsafe { self.dict.split_peek() }?.get_over(keys, i) } + /// Set the attribute at position `i` of the class's shared names by + /// appending `value`, when the instance's split values stop just + /// before `i` and their block has room (the constructor shape, after + /// its first store); `Err` hands the value back, touching nothing. + /// The caller proved the name at `i` and that a plain `__dict__` + /// store is what the assignment means (see `split_store`). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + pub unsafe fn split_append(&self, i: usize, value: Object) -> Result<(), Object> { + if self.c_body.get() != 0 || crate::gil::free_threading_enabled() { + return Err(value); + } + let cls = self.cls_raw(); + let Some(keys) = cls.shared_keys.get() else { + return Err(value); + }; + if !value.is_gc_atomic() && self.deferred.get() { + // The write barrier, before the values are borrowed (tracking + // early is always sound). + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + let Some(split) = (unsafe { self.dict.split_peek_mut() }) else { + return Err(value); + }; + split.append_over(keys, || cls.shared_keys.share(), i, value) + } + + /// The split length and whether position `i` of the class's shared + /// names is ready for a store: an overwrite of a set value (which + /// `droppable` must accept), or the next of a run of in-order appends + /// starting at `*cursor` (the split length once the earlier appends + /// land; `None` starts it at the current length). `None` when the + /// split layout can't say. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline] + pub unsafe fn split_store_ready( + &self, + i: usize, + cursor: &mut Option, + droppable: impl FnOnce(&Object) -> bool, + ) -> Option { + if self.c_body.get() != 0 || crate::gil::free_threading_enabled() { + return None; + } + let keys = self.cls_raw().shared_keys.get()?; + // SAFETY: forwarded contract. + let split = unsafe { self.dict.split_peek() }?; + let at = cursor.get_or_insert(split.len()); + if let Some(v) = split.get_over(keys, i) { + return Some(droppable(v)); + } + if i == *at && split.can_append_at(keys, i) { + *at += 1; + return Some(true); + } + None + } + /// [`Self::split_field`] for an in-place overwrite by a value of /// atomicity `atomic`, whose write barrier runs first (a non-atomic /// value starts tracking a deferred instance). diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index 7c427151..d020bc24 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -213,6 +213,10 @@ pub(crate) struct LeafPlan { consts: Box<[V]>, /// Arguments, copied into the first registers on entry. nargs: u8, + /// Every attribute store goes to the first argument (never + /// reassigned), each to a different name: the buffered stores share + /// one receiver and each is the latest to its attribute. + unique_stores: bool, } // SAFETY: the constants' pointers name the code extension's immutable @@ -236,6 +240,10 @@ struct Builder<'a> { /// Number of locals (registers below it). nl: usize, stack: Vec, + /// The names stored into so far, while every store goes to the + /// first argument (see [`LeafPlan::unique_stores`]); `None` once one + /// doesn't. + stored: Option>, /// Locals definitely assigned on the current path (bit per local). assigned: u32, /// Per bytecode target: the stack depth and assigned set on the @@ -312,6 +320,10 @@ impl Builder<'_> { self.release_local(i as u8)?; self.ops.push(Op::Move { dst: i as u8, src }); } + if i == 0 { + // The first argument no longer names one receiver. + self.stored = None; + } self.assigned |= 1 << i; Some(()) } @@ -556,6 +568,13 @@ impl Builder<'_> { OpCode::StoreAttr => { let recv = self.pop()?; let val = self.pop()?; + if let Some(names) = &mut self.stored { + if recv != 0 || names.contains(&arg) { + self.stored = None; + } else { + names.push(arg); + } + } self.ops.push(Op::StoreAttr { recv, val, @@ -634,6 +653,7 @@ impl Builder<'_> { ops: self.ops.into_boxed_slice(), consts: self.consts.into_boxed_slice(), nargs: u8::try_from(self.code.arg_count).ok()?, + unique_stores: self.stored.is_some() && self.code.arg_count > 0, }) } } @@ -653,6 +673,7 @@ pub(crate) fn build(code: &CodeObject, ext: &CodeConstObjects) -> Option( + /// + /// `FRESH`: the first argument is an instance nothing else has seen + /// (a constructor's `self`), so a store into it lands at once — a + /// later decline leaves it half-built, and the caller discards it. + pub(crate) fn leaf_run( &self, code: &CodeObject, ext: &CodeConstObjects, @@ -1112,9 +1137,17 @@ impl Interpreter { // before any does (a decline touched nothing, and // the ordinary call runs the body instead). A // lone store is its own check: it declines whole. + let unique = plan.unique_stores; if pend.n > 1 { + // With one receiver, appends of new attributes + // land in order: `cursor` is its split length + // once the earlier ones have. + let mut cursor = None; + // A store the shortcut can't vouch for may + // move the layout: the rest check in full. + let mut split_ok = unique; for k in 0..pend.n { - if !pend.latest(k) { + if !unique && !pend.latest(k) { continue; } let (rp, spc, name, _) = pend.get(k); @@ -1122,14 +1155,23 @@ impl Interpreter { let Object::Instance(inst) = (unsafe { &**rp }) else { return None; }; - if !Self::core_store_attr_ready(code, inst, *spc as usize, *name) { + let (spc, name) = (*spc as usize, *name); + let ready = if split_ok { + Self::core_store_attr_ready_split(ext, inst, spc, &mut cursor) + } else { + None + }; + split_ok &= ready.is_some(); + if !ready.unwrap_or_else(|| { + Self::core_store_attr_ready(code, inst, spc, name) + }) { return None; } } } let n = pend.n; for k in 0..n { - if !pend.latest(k) { + if !unique && !pend.latest(k) { continue; } let (rp, spc, name, value) = pend.get(k); @@ -1164,6 +1206,24 @@ impl Interpreter { }; // SAFETY: as `norm`. let recv = unsafe { &*rp }; + if FRESH && args.first().is_some_and(|&a| std::ptr::eq(a, rp)) { + let Object::Instance(inst) = recv else { + return None; + }; + let value = std::mem::ManuallyDrop::new(to_object(get!(val))?); + // On `true` the value moved into the instance. + if !Self::core_store_attr( + code, + inst, + usize::from(pc), + u32::from(name), + &value, + ) { + drop(std::mem::ManuallyDrop::into_inner(value)); + return None; + } + continue; + } if !matches!(recv, Object::Instance(_)) || pend.n == PENDING { return None; } diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 904b7416..fd4d3365 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -14077,6 +14077,21 @@ impl Interpreter { { return false; } + // A leaf `__init__` (plain stores of its arguments into `self`) + // runs frameless, as a leaf method call does. + if self.core_leaf_init( + frame, + &ty, + init, + code, + argc, + self_slot, + callee_slot, + sw.depth_cell, + ) { + frame.pc = pc as u32 + 1; + return true; + } let Some(cells) = init.lean_cells_ref(code) else { return false; }; @@ -14131,6 +14146,77 @@ impl Interpreter { true } + /// [`Self::core_new`] for an `__init__` that is a pure or effect leaf + /// whose every return is `return None`: the new instance, evaluated + /// into by [`Self::pure_leaf_eval`] with no activation, replaces the + /// call's operands (`frame.stack[callee_slot..]`). `false` touches + /// nothing observable (a declined evaluation stores nothing, and the + /// unused instance was never seen). + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_leaf_init( + &self, + frame: &mut Frame, + ty: &Rc, + init: &Rc, + code: &Rc, + argc: usize, + self_slot: usize, + callee_slot: usize, + depth_cell: *const std::cell::Cell, + ) -> bool { + let pure = code_is_pure_leaf(code); + if !(pure || code_is_effect_leaf(code)) + || argc >= 8 + || !pure_leaf_warm(code) + || !code_returns_only_none(code) + || ty.flags.is_builtin + || ty.native_kind.get() != 0 + || ty.instances_need_finalize() + // The ordinary call's `RecursionError` check. + // SAFETY: this thread's own depth cell. + || unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() + { + return false; + } + // The arguments leave by plain decrements (whatever `__init__` + // stored holds its own reference). + if !frame.stack[self_slot + 1..] + .iter() + .all(Self::core_droppable) + { + return false; + } + let (inst, _) = self.alloc_plain_instance_obj(ty); + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + args[0] = &inst; + for (k, a) in frame.stack[self_slot + 1..].iter().enumerate() { + args[k + 1] = a; + } + let r = if pure { + self.pure_leaf_eval::(code, init, &args[..=argc]) + } else { + self.leaf_init_eval(code, init, &args[..=argc]) + }; + match r { + Some(done) => { + // Every return is `return None` (checked above). + drop(done); + frame.stack.truncate(callee_slot); + frame.stack.push(inst); + true + } + None => { + // A declined `__init__` may have stored into the instance + // before it stopped; nothing else ever saw it, so it goes + // (the framed call starts over on a fresh one). + gc_trace::note_dropped(&inst); + drop(inst); + false + } + } + } + /// The core loop's container instructions: `seq[i]` and `d[key]` over /// exact containers (an in-range index or a present `str`/`int` key), /// `seq[i] = v` in range and `d[key] = v`, and a comprehension's @@ -15231,6 +15317,45 @@ impl Interpreter { f: &crate::object::PyFunction, args: &[*const Object], ) -> Option { + if !Self::leaf_call_entry(code) { + return None; + } + let r = self.leaf_eval::(code, f, args, 0); + Self::leaf_call_exit(code, r.is_some()); + r + } + + /// [`Self::pure_leaf_eval`] for a constructor's leaf `__init__` on the + /// fresh instance `args[0]` (see `leaf_run`'s `FRESH`): `None` may + /// leave that instance half-initialized, for the caller to discard. + fn leaf_init_eval( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + ) -> Option { + if !Self::leaf_call_entry(code) { + return None; + } + let ext = code_vm_ext(code)?; + let r = if ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) == 16 { + // The setter shape stores at once anyway. + self.leaf_eval::(code, f, args, 0) + } else { + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run::(code, ext, plan, f, args, 0) + }; + Self::leaf_call_exit(code, r.is_some()); + r + } + + /// A frameless call's JIT bookkeeping, before it runs: `false` sends + /// it to the framed path instead. + #[inline(always)] + fn leaf_call_entry(code: &CodeObject) -> bool { // A frameless call is an activation the JIT's warm-up never sees: // count it as a lean one, and credit each interval to the tier-2 // counter. The call at which a compile falls due goes to the @@ -15246,7 +15371,7 @@ impl Interpreter { && !crate::tier2::jit_off_for_process() { if crate::tier2::note_frameless_calls(code, n + 1) { - return None; + return false; } hint.defer_lean_compile(); } else { @@ -15254,13 +15379,18 @@ impl Interpreter { } } } - let r = self.leaf_eval::(code, f, args, 0); - if r.is_some() { + let _ = code; + true + } + + /// A frameless call's bookkeeping after it ran (`hit`: it finished). + #[inline(always)] + fn leaf_call_exit(code: &CodeObject, hit: bool) { + if hit { code.jit_hint.note_leaf_hit(); } else { code.jit_hint.note_leaf_miss(); } - r } /// [`Self::pure_leaf_eval`] at call-nesting depth `nest` (a pure-leaf @@ -15396,7 +15526,7 @@ impl Interpreter { .leaf_plan .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) .as_deref()?; - self.leaf_run::(code, ext, plan, f, args, nest) + self.leaf_run::(code, ext, plan, f, args, nest) } /// The core loop's `FOR_ITER` over a generator at `pc` of `sw`'s @@ -16147,6 +16277,14 @@ impl Interpreter { } return false; } + // The constructor shape: the attribute is the instance's + // next one. + // SAFETY: as above; the value moves out of the caller's + // stack slot, and a declined append hands the bits back. + match unsafe { inst.split_append(idx as usize, std::ptr::read(value)) } { + Ok(()) => return true, + Err(v) => std::mem::forget(v), + } } } match code.caches.get(attr_pc as u32) { @@ -16184,6 +16322,25 @@ impl Interpreter { } } + /// [`Self::core_store_attr_ready`] through the site's [`FieldSlot`], + /// for the next of an effect leaf's stores to one receiver in order + /// (`cursor` as [`PyInstance::split_store_ready`]): `None` when the + /// shortcut can't answer. A `true` here is a store the shortcut in + /// [`Self::core_store_attr`] then performs. + fn core_store_attr_ready_split( + ext: &CodeConstObjects, + inst: &PyInstance, + attr_pc: usize, + cursor: &mut Option, + ) -> Option { + let (ver, idx) = ext.field_slots.get()?.get(attr_pc)?.get(); + if inst.cls_raw().attr_version.get() != ver || crate::capi_watchers::dicts_active() { + return None; + } + // SAFETY: a read with nothing running (the commit's check pass). + unsafe { inst.split_store_ready(idx as usize, cursor, Self::core_droppable) } + } + /// Whether [`Self::core_store_attr`] would perform this store right /// now (it declines exactly when this is `false`), touching nothing: /// an effect leaf validates every buffered store before committing @@ -16288,6 +16445,15 @@ impl Interpreter { match inst.split_store(name, unsafe { std::ptr::read(value) }) { Ok(old) => { drop(old); + // Later stores at this site (to the same position of + // later instances) take the split shortcut. + if let Some(ext) = code_vm_ext(code) { + // SAFETY: a read between two instructions. + let pos = unsafe { inst.dict.split_peek() }.and_then(|s| s.position(name)); + if let Some(pos) = pos.and_then(|p| u32::try_from(p).ok()) { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, pos); + } + } return true; } Err(v) => std::mem::forget(v), @@ -61338,6 +61504,9 @@ struct CodeConstObjects { /// A leaf body's translation for the frameless evaluator (see /// [`leaf_plan`]), or `None` when the body has none. leaf_plan: std::sync::OnceLock>>, + /// Whether every return is `return None` (see + /// [`code_returns_only_none`]): `0` not yet decided, `1` no, `2` yes. + returns_none: std::sync::atomic::AtomicU8, } /// A `CALL` site's inline-call shape (see `Interpreter::core_call`): the @@ -62360,6 +62529,33 @@ fn effect_setter_shape(code: &CodeObject) -> bool { && ret.op == OpCode::ReturnValue) } +/// Whether every `RETURN_VALUE` in `code` returns the constant `None` +/// (what a constructor's `__init__` must return), decided once. +fn code_returns_only_none(code: &CodeObject) -> bool { + use std::sync::atomic::Ordering::Relaxed; + let Some(ext) = code_vm_ext(code) else { + return false; + }; + match ext.returns_none.load(Relaxed) { + 1 => return false, + 2 => return true, + _ => {} + } + let instrs = &code.instructions; + let yes = instrs.iter().enumerate().all(|(pc, ins)| { + ins.op != OpCode::ReturnValue + || pc + .checked_sub(1) + .and_then(|p| instrs.get(p)) + .is_some_and(|prev| { + prev.op == OpCode::LoadConst + && matches!(code.constants.get(prev.arg as usize), Some(Constant::None)) + }) + }); + ext.returns_none.store(if yes { 2 } else { 1 }, Relaxed); + yes +} + /// Whether `code` is an *effect leaf* (see `code_pure_leaf_decide`), /// deciding its leaf verdicts on first use. #[inline] @@ -62502,6 +62698,7 @@ fn code_vm_ext_init( fast_pairs: std::sync::OnceLock::new(), field_slots: std::sync::OnceLock::new(), leaf_plan: std::sync::OnceLock::new(), + returns_none: std::sync::atomic::AtomicU8::new(0), }) }) } From a99ca77ad628f948d7e72a3eb6feea973d46889c Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:57:45 -0700 Subject: [PATCH 25/65] perf: cheaper GC bookkeeping and frameless calls into compiled leaves Interpreter call sites keep evaluating a pure or effect leaf frameless after the JIT compiles it (native callers still enter the compiled code). The collector's tracked-handle references use the biased Rc, its Bloom-filter inserts skip already-set bits, and its population counters update without locked read-modify-writes while the GIL serializes them. The generation-0 threshold now matches CPython 3.14's (2000). --- crates/weavepy-vm/src/gc_trace.rs | 132 +++++++++++++++--------- crates/weavepy-vm/src/hot_filter.rs | 13 ++- crates/weavepy-vm/src/leaf_plan.rs | 2 +- crates/weavepy-vm/src/lib.rs | 21 ++-- crates/weavepy-vm/src/stdlib/gc_mod.rs | 4 +- crates/weavepy-vm/src/stdlib/gc_real.rs | 2 +- tests/regrtest/test_ws4_gc_cascade.py | 2 +- 7 files changed, 111 insertions(+), 65 deletions(-) diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index a0f2c486..0a24efbb 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -32,7 +32,7 @@ //! A type's flags decide whether tracking is needed at //! construction time. //! - The **eval breaker** triggers a collection when the -//! generation-0 counter exceeds the threshold (default 700). +//! generation-0 counter exceeds the threshold (default 2000, as in CPython 3.14). //! Collections also happen on explicit `gc.collect()`. //! //! Today's implementation is *non-incremental*: a full @@ -73,28 +73,31 @@ use crate::fasthash::ObjectIdHasher; use crate::shared_value::ThinArc; +use crate::sync::Rc as HandleRc; use crate::sync::RefCell; use std::hash::BuildHasherDefault; use std::sync::atomic::{ AtomicBool, AtomicI64, AtomicU32, AtomicU64, AtomicU8, AtomicUsize, Ordering, }; -use std::sync::Arc; use crate::object::Object; use crate::weakref_registry::{id_of, ObjectId}; -type GcIndex = - std::collections::HashMap, BuildHasherDefault>; +type GcIndex = std::collections::HashMap< + ObjectId, + HandleRc, + BuildHasherDefault, +>; /// The standard CPython generation count (3) and default -/// thresholds: gen 0 collects when 700 untracked allocations +/// thresholds (CPython 3.14's): gen 0 collects when 2000 net tracked allocations /// have happened; gen 1 every 10 gen 0 collections; gen 2 /// every 10 gen 1 collections. pub const N_GENERATIONS: usize = 3; // Color and generation are bounded states; reference counts remain full-width. const _: () = assert!(N_GENERATIONS > 0 && N_GENERATIONS <= u8::MAX as usize + 1); const MAX_GENERATION: u8 = (N_GENERATIONS - 1) as u8; -pub const DEFAULT_THRESHOLDS: [usize; N_GENERATIONS] = [700, 10, 10]; +pub const DEFAULT_THRESHOLDS: [usize; N_GENERATIONS] = [2000, 10, 10]; /// Upper bound on the number of mark-sweep passes a single /// [`GcState::collect`] runs to reach a fixpoint. Convergence is normally 2–3 @@ -316,7 +319,7 @@ impl TrackedHandle { /// in from the end, and its `slot` field is corrected here so the /// per-handle position invariant holds after the call. #[inline] -fn swap_remove_handle(vec: &mut Vec>, slot: usize) { +fn swap_remove_handle(vec: &mut Vec>, slot: usize) { if slot >= vec.len() { return; } @@ -333,8 +336,11 @@ fn swap_remove_handle(vec: &mut Vec>, slot: usize) { /// handle was found and removed. O(n) in the generation length, but only ever /// taken on the rare stale-cache path — the common case stays O(1). #[inline] -fn remove_handle_by_ptr(vec: &mut Vec>, handle: &Arc) -> bool { - if let Some(pos) = vec.iter().position(|h| Arc::ptr_eq(h, handle)) { +fn remove_handle_by_ptr( + vec: &mut Vec>, + handle: &HandleRc, +) -> bool { + if let Some(pos) = vec.iter().position(|h| HandleRc::ptr_eq(h, handle)) { swap_remove_handle(vec, pos); true } else { @@ -347,7 +353,7 @@ struct Generation { /// All tracked handles in this generation. Append-only /// during normal allocation; rewritten in place when /// objects are promoted or moved to the unreachable list. - handles: Vec>, + handles: Vec>, } #[derive(Debug, Default, Clone, Copy)] @@ -430,7 +436,7 @@ pub struct GcState { /// Frozen handles. `gc.freeze()` moves all tracked objects /// here; they are skipped by future collections until /// `gc.unfreeze()` runs. - frozen: RefCell>>, + frozen: RefCell>>, /// `gc.garbage` — uncollectable objects (cycles whose /// finalisers refused to release). pub garbage: RefCell>, @@ -461,7 +467,7 @@ pub struct GcState { /// path by scanning *this* small set (not the whole tracked /// population) at the interpreter's reference-drop safe points. Keyed /// by id like `index`; an object is in both while finalizable. - finalizable: RefCell>>, + finalizable: RefCell>>, /// Live population of [`Self::finalizable`]. A relaxed load of this /// atomic is the gate the interpreter checks before every prompt- /// finalization sweep: when it is zero (the overwhelmingly common @@ -484,7 +490,7 @@ pub struct GcState { /// [`Self::finalizable`], so the per-drop scan walks only entries /// that could plausibly die this safe point. Membership is mirrored /// by each handle's `fin_cold` flag and `fin_hot_slot` index. - finalizable_hot: RefCell>>, + finalizable_hot: RefCell>>, /// `finalizable_hot.len()`, published for the lock-free gate: when /// zero, a scan that isn't on the cold stride returns without /// borrowing anything — the steady state of a program whose @@ -511,7 +517,7 @@ impl Drop for GcState { // first so the chains are already severed when the handle // vectors drop. Safe at this point: the thread is exiting, no // Python code will observe the cleared objects. - let mut handles: Vec> = Vec::new(); + let mut handles: Vec> = Vec::new(); if let Ok(gens) = self.generations.try_borrow() { for g in gens.iter() { handles.extend(g.handles.iter().cloned()); @@ -582,7 +588,7 @@ impl GcState { /// Insert `h` into the finalizable index (RFC 0077 WS2: one place /// for the population and hot-set bookkeeping). New entries start /// hot; the next scan grades them. Returns whether it was new. - fn fin_insert(&self, id: ObjectId, h: Arc) -> bool { + fn fin_insert(&self, id: ObjectId, h: HandleRc) -> bool { let mut fin = self.finalizable.borrow_mut(); if fin.contains_key(&id) { return false; @@ -601,6 +607,12 @@ impl GcState { /// Remove `id` from the finalizable index, if present, keeping the /// population and hot set exact. fn fin_remove(&self, id: ObjectId) -> bool { + // The usual case: nothing finalizable is tracked at all. (Inserts + // count after they land, and the same thread removes an id it + // inserted.) + if self.finalizable_count.load(Ordering::Acquire) == 0 { + return false; + } let removed = self.finalizable.borrow_mut().remove(&id); let Some(h) = removed else { return false; @@ -611,7 +623,7 @@ impl GcState { /// The hot-set and population half of [`Self::fin_remove`], for the /// one caller that already holds the finalizable lock. - fn fin_remove_bookkeeping(&self, h: &Arc) { + fn fin_remove_bookkeeping(&self, h: &HandleRc) { self.fin_make_cold(h); if self.finalizable_count.fetch_sub(1, Ordering::AcqRel) == 1 { // RFC 0065 (WS1): population reached zero — dispatch loops @@ -621,7 +633,7 @@ impl GcState { } /// Move `h` into the hot set (no-op if already hot). - fn fin_make_hot(&self, h: &Arc) { + fn fin_make_hot(&self, h: &HandleRc) { let mut hot = self.finalizable_hot.borrow_mut(); if !h.fin_cold.swap(false, Ordering::AcqRel) && h.fin_hot_slot.load(Ordering::Relaxed) != usize::MAX @@ -636,19 +648,19 @@ impl GcState { /// Move `h` out of the hot set (no-op if already cold). O(1) via /// the handle's cached slot, with the swap-remove fixup. - fn fin_make_cold(&self, h: &Arc) { + fn fin_make_cold(&self, h: &HandleRc) { let mut hot = self.finalizable_hot.borrow_mut(); h.fin_cold.store(true, Ordering::Release); let slot = h.fin_hot_slot.swap(usize::MAX, Ordering::AcqRel); if slot == usize::MAX { return; } - if slot < hot.len() && Arc::ptr_eq(&hot[slot], h) { + if slot < hot.len() && HandleRc::ptr_eq(&hot[slot], h) { hot.swap_remove(slot); if let Some(moved) = hot.get(slot) { moved.fin_hot_slot.store(slot, Ordering::Relaxed); } - } else if let Some(pos) = hot.iter().position(|x| Arc::ptr_eq(x, h)) { + } else if let Some(pos) = hot.iter().position(|x| HandleRc::ptr_eq(x, h)) { hot.swap_remove(pos); if let Some(moved) = hot.get(pos) { moved.fin_hot_slot.store(pos, Ordering::Relaxed); @@ -845,7 +857,7 @@ impl GcState { // insert becomes observable (we hold the index borrow, so // no prober can race past a fresh registration). self.tracked_filter.insert(new_id); - let handle = Arc::new(TrackedHandle::new(obj, 0)); + let handle = HandleRc::new(TrackedHandle::new(obj, 0)); entry.insert(handle.clone()); // Enroll finalizable objects in the dedicated prompt-finalization // index so the per-safe-point sweep scans only them, not the whole @@ -877,8 +889,8 @@ impl GcState { if !self.finalized_ids.borrow().is_empty() { self.finalized_ids.borrow_mut().remove(&new_id); } - self.tracked_count.fetch_add(1, Ordering::AcqRel); - self.tracked_version.fetch_add(1, Ordering::AcqRel); + serial_add(&self.tracked_count, 1); + serial_add(&self.tracked_version, 1); self.note_gen0_alloc(); } @@ -937,7 +949,10 @@ impl GcState { if handle.color.load(Ordering::Acquire) == color::Frozen { let mut frozen = self.frozen.borrow_mut(); let slot = handle.slot.load(Ordering::Acquire); - if frozen.get(slot).is_some_and(|h| Arc::ptr_eq(h, &handle)) { + if frozen + .get(slot) + .is_some_and(|h| HandleRc::ptr_eq(h, &handle)) + { swap_remove_handle(&mut frozen, slot); } else { remove_handle_by_ptr(&mut frozen, &handle); @@ -954,7 +969,7 @@ impl GcState { if gens[g] .handles .get(slot) - .is_some_and(|h| Arc::ptr_eq(h, &handle)) + .is_some_and(|h| HandleRc::ptr_eq(h, &handle)) { swap_remove_handle(&mut gens[g].handles, slot); } else if !remove_handle_by_ptr(&mut gens[g].handles, &handle) { @@ -968,8 +983,8 @@ impl GcState { } } } - self.tracked_count.fetch_sub(1, Ordering::AcqRel); - self.tracked_version.fetch_add(1, Ordering::AcqRel); + serial_add(&self.tracked_count, usize::MAX); + serial_add(&self.tracked_version, 1); } /// Reclaim every tracked object on this thread whose only remaining @@ -1231,8 +1246,10 @@ impl GcState { Hot, Cold, } - let mut out: Vec> = Vec::new(); - let mut probe = |h: &Arc, out: &mut Vec>| -> Grade { + let mut out: Vec> = Vec::new(); + let mut probe = |h: &HandleRc, + out: &mut Vec>| + -> Grade { probed += 1; let sc = strong_count_for(&h.object); let cached = h.weak_clones.load(Ordering::Acquire); @@ -1300,11 +1317,11 @@ impl GcState { // Walk the whole index (or a rotating window of it) and // re-grade every entry; collect the transitions and apply // them after the index borrow ends. - let mut to_cold: Vec> = Vec::new(); - let mut to_hot: Vec> = Vec::new(); + let mut to_cold: Vec> = Vec::new(); + let mut to_hot: Vec> = Vec::new(); { let fin = self.finalizable.borrow(); - let mut grade = |h: &Arc| { + let mut grade = |h: &HandleRc| { let was_cold = h.fin_cold.load(Ordering::Relaxed); match probe(h, &mut out) { Grade::Cold if !was_cold => to_cold.push(h.clone()), @@ -1438,7 +1455,7 @@ impl GcState { } /// O(1) handle lookup by object id (any generation or frozen). - pub fn handle_for(&self, id: ObjectId) -> Option> { + pub fn handle_for(&self, id: ObjectId) -> Option> { // RFC 0065 (WS4): see `is_tracked`. if !self.tracked_filter.may_contain(id) { return None; @@ -1452,9 +1469,9 @@ impl GcState { /// finalizers for everything during interpreter teardown, not just /// for cyclic garbage. The per-handle `finalized` flag (shared with /// the cycle collector) guarantees each `__del__` runs at most once. - pub fn finalization_candidates(&self) -> Vec> { + pub fn finalization_candidates(&self) -> Vec> { let mut out = Vec::new(); - let pending = |h: &Arc| { + let pending = |h: &HandleRc| { !h.finalized.load(Ordering::Acquire) // A finalizer already queued by a collection (but not yet // drained) must not be listed again — the pending queue owns @@ -1911,7 +1928,7 @@ impl GcState { // are not counted as collected — they're dropped when this pass ends, // and the underlying iterator is freed by refcount once the real // objects in its (dead) cycle are cleared. - let mut temp_handles: Vec> = Vec::new(); + let mut temp_handles: Vec> = Vec::new(); { // Scan the existing candidate list, then the growing temporary // list, preserving discovery order without copying either list. @@ -2039,7 +2056,7 @@ impl GcState { if by_id.contains_key(&cid) { return; } - let handle = Arc::new(TrackedHandle::new(child.clone(), 0)); + let handle = HandleRc::new(TrackedHandle::new(child.clone(), 0)); by_id.insert(cid, handle.clone()); temp_handles.push(handle); }); @@ -2127,7 +2144,7 @@ impl GcState { } // Phase 5: white objects are unreachable cyclic garbage. - let unreachable: Vec> = candidate_set + let unreachable: Vec> = candidate_set .iter() .filter(|h| h.color.load(Ordering::Acquire) == color::White) .cloned() @@ -2271,8 +2288,8 @@ impl GcState { // wasn't resurrected falls into `dead` and is reclaimed (its weakrefs // cleared in that second pass, so single-`collect()` weakref tests // still observe `ref() is None`). - let mut deferred: Vec> = Vec::new(); - let mut maybe_dead: Vec> = Vec::new(); + let mut deferred: Vec> = Vec::new(); + let mut maybe_dead: Vec> = Vec::new(); for h in &unreachable { let pending_finalizer = has_finalizer(&h.object) && !h.finalized.load(Ordering::Acquire); @@ -2372,7 +2389,7 @@ impl GcState { // Whatever stayed White after the resurrection re-mark and the finalizer // subgraph protection is genuinely dead this pass. - let dead: Vec> = maybe_dead + let dead: Vec> = maybe_dead .into_iter() .filter(|h| h.color.load(Ordering::Acquire) == color::White) .collect(); @@ -2431,7 +2448,7 @@ impl GcState { // every other dead object has released its references, a // dict held only by dead holders is down to one owner and // the retry clears it (a live holder keeps it intact). - let mut shared_dict_holders: Vec<&Arc> = Vec::new(); + let mut shared_dict_holders: Vec<&HandleRc> = Vec::new(); for h in &dead { if !clear_object_fields(&h.object) { shared_dict_holders.push(h); @@ -2565,7 +2582,7 @@ impl GcState { reported } - fn snapshot_for_collection(&self, upto: usize) -> Vec> { + fn snapshot_for_collection(&self, upto: usize) -> Vec> { let gens = self.generations.borrow(); let selected = &gens[..=upto.min(N_GENERATIONS - 1)]; let mut out = Vec::with_capacity(selected.iter().map(|g| g.handles.len()).sum()); @@ -2575,7 +2592,7 @@ impl GcState { out } - fn rebuild_generations(&self, upto: usize, candidates: &[Arc]) { + fn rebuild_generations(&self, upto: usize, candidates: &[HandleRc]) { // Lock order MUST match `track` (index before generations): the // collector and a mutator thread can both reach the GC under the // shared, process-global state, and acquiring these two cells in @@ -3310,7 +3327,7 @@ pub fn with_state(f: impl FnOnce(&GcState) -> R) -> R { /// well away from `getrefcount`-hot paths like pandas'. pub fn zombie_memoryview_refs_to(target: ObjectId) -> usize { with_state(|s| { - let mut handles: Vec> = Vec::new(); + let mut handles: Vec> = Vec::new(); { let Ok(gens) = s.generations.try_borrow() else { return 0; @@ -3414,6 +3431,19 @@ pub fn track(obj: Object) { with_state(|s| s.track(obj)); } +/// `counter += n` (wrapping) for a collector counter: a plain load and +/// store while the GIL serializes every writer, a locked read-modify-write +/// only in free-threaded mode. +#[inline(always)] +fn serial_add(counter: &AtomicUsize, n: usize) { + if crate::gil::free_threading_enabled() { + counter.fetch_add(n, Ordering::AcqRel); + } else { + let v = counter.load(Ordering::Relaxed); + counter.store(v.wrapping_add(n), Ordering::Release); + } +} + /// Deferred instance tracking. /// /// CPython tracks every instance of a Python-defined class at @@ -3709,7 +3739,7 @@ pub fn maybe_auto_collect() -> bool { /// Convenience: find a tracked handle by object id (O(1) via the /// id index, which covers all generations plus the frozen set). -pub fn find_handle(id: ObjectId) -> Option> { +pub fn find_handle(id: ObjectId) -> Option> { with_state(|s| s.handle_for(id)) } @@ -3745,7 +3775,7 @@ pub fn complete_finalizer(id: ObjectId) { /// Convenience: snapshot all tracked objects with an unrun `__del__` /// in the shared GC (see [`GcState::finalization_candidates`]). -pub fn finalization_candidates() -> Vec> { +pub fn finalization_candidates() -> Vec> { with_state(|s| s.finalization_candidates()) } @@ -3871,7 +3901,7 @@ const SUSPECT_DORMANT_PROBES: u8 = 16; /// One enrolled suspect: its handle, remaining active probe budget, and /// (once dormant) the number of stride probes it has survived. struct Suspect { - handle: Arc, + handle: HandleRc, budget: u8, dormant_probes: u8, } @@ -4089,7 +4119,7 @@ pub fn active_suspects_present() -> bool { /// Enroll a cascade-skipped tracked object for later deadness re-probes. /// Deduplicated; silently dropped when the list is full (the next full /// collection reclaims it instead). -pub fn note_suspect(h: Arc) { +pub fn note_suspect(h: HandleRc) { static NO_SUSPECTS: std::sync::OnceLock = std::sync::OnceLock::new(); if *NO_SUSPECTS.get_or_init(|| std::env::var_os("WEAVEPY_NO_SUSPECTS").is_some()) { return; @@ -4618,9 +4648,9 @@ mod tests { _ => u8::MAX, }; let object = Object::List(Rc::new(RefCell::new(Vec::new()))); - let handle = Arc::new(TrackedHandle::new(object, 0)); + let handle = HandleRc::new(TrackedHandle::new(object, 0)); let id = handle.id; - handles.insert(id, Arc::downgrade(&handle)); + handles.insert(id, HandleRc::downgrade(&handle)); suspects.insert( id, Suspect { diff --git a/crates/weavepy-vm/src/hot_filter.rs b/crates/weavepy-vm/src/hot_filter.rs index 97d4d5c1..8f11da51 100644 --- a/crates/weavepy-vm/src/hot_filter.rs +++ b/crates/weavepy-vm/src/hot_filter.rs @@ -72,8 +72,13 @@ impl AtomicBloom { #[inline] pub fn insert(&self, id: u64) { let ((w1, b1), (w2, b2)) = Self::probes(id); - self.bits[w1].fetch_or(b1, Ordering::Relaxed); - self.bits[w2].fetch_or(b2, Ordering::Relaxed); + // A bit already set needs no locked read-modify-write: bits are + // only ever cleared by a rebuild, which inserts never race. + for (w, b) in [(w1, b1), (w2, b2)] { + if self.bits[w].load(Ordering::Relaxed) & b == 0 { + self.bits[w].fetch_or(b, Ordering::Relaxed); + } + } } /// `false` means *definitely absent* (for every id that went @@ -126,7 +131,9 @@ impl RebuildableBloom { pub fn insert(&self, id: u64) { self.filters[0].insert(id); self.filters[1].insert(id); - self.inserts.fetch_add(1, Ordering::Relaxed); + // A staleness estimate: a lost increment only delays a rebuild. + let n = self.inserts.load(Ordering::Relaxed); + self.inserts.store(n.wrapping_add(1), Ordering::Relaxed); } #[inline] diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index d020bc24..f0aa3514 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -1101,7 +1101,7 @@ impl Interpreter { }; let n = usize::from(at + 2 + argc - first); if !crate::code_is_pure_leaf(ccode) - || !Self::lean_code_ok(ccode) + || !Self::leaf_code_ok(ccode) || n != ccode.arg_count as usize || n > 8 || crate::recursion::current_depth() + usize::from(nest) + 1 diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index fd4d3365..4f49e960 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -3117,7 +3117,7 @@ impl Interpreter { // unpickler temporary holding the memo) cascades through the // untracked `memo` dict to the tracked argument. The refcount guard // below still filters anything that stays externally reachable. - let mut children: Vec> = Vec::new(); + let mut children: Vec> = Vec::new(); // Untracked descendants with live weakrefs: CPython clears an // object's weakrefs at refcount zero whether or not the GC ever // tracked it, but this cascade's "next link" set is (otherwise) @@ -4694,7 +4694,7 @@ impl Interpreter { Some(Object::Module(m)) => m.dict.borrow().iter().map(|(_k, v)| v.clone()).collect(), _ => Vec::new(), }; - let mut sys_deferred: Vec> = Vec::new(); + let mut sys_deferred: Vec> = Vec::new(); for _ in 0..8 { let candidates = crate::gc_trace::finalization_candidates(); if candidates.is_empty() { @@ -4704,7 +4704,7 @@ impl Interpreter { if sys_values.iter().any(|v| v.is_same(&handle.object)) { if !sys_deferred .iter() - .any(|h| std::sync::Arc::ptr_eq(h, &handle)) + .any(|h| crate::sync::Rc::ptr_eq(h, &handle)) { sys_deferred.push(handle); } @@ -11381,6 +11381,15 @@ impl Interpreter { true } + /// [`Self::lean_code_ok`] for a frameless leaf call from the + /// interpreter: compiled code still qualifies (the leaf evaluator + /// costs less than entering native code from here; native callers + /// take the compiled code through their own call path). + #[inline] + fn leaf_code_ok(code: &CodeObject) -> bool { + !code.wire.as_ref().is_some_and(|w| w.exec_error.is_some()) + } + /// [`Self::lean_code_ok`] for a generator resume: a compiled /// generator whose every loop yields runs at most one iteration per /// native resume, and the native resume protocol costs more than @@ -14912,7 +14921,7 @@ impl Interpreter { return None; } let (missing, slot_self) = code_call_slot(code, call_pc)?.hit(fp, Rc::as_ptr(code_rc))?; - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self || !Self::leaf_code_ok(code_rc) { return None; } let missing = missing as usize; @@ -15073,7 +15082,7 @@ impl Interpreter { #[cfg(test)] note_predicate_stage(code_rc, 2); let has_self = !matches!(ops.get(1)?, Object::Unbound); - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self || !Self::leaf_code_ok(code_rc) { return None; } let missing = missing as usize; @@ -15187,7 +15196,7 @@ impl Interpreter { // `PyFunction::code`); only compared and borrowed below, while the // caller's stack keeps the function alive. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) || !Self::lean_code_ok(code_rc) { + if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) || !Self::leaf_code_ok(code_rc) { return None; } let (_, covered) = Self::kw_names_bind_cached(code, pc, f, func_id, perm, names, eff_argc)?; diff --git a/crates/weavepy-vm/src/stdlib/gc_mod.rs b/crates/weavepy-vm/src/stdlib/gc_mod.rs index 6455907a..c62152f4 100644 --- a/crates/weavepy-vm/src/stdlib/gc_mod.rs +++ b/crates/weavepy-vm/src/stdlib/gc_mod.rs @@ -16,7 +16,7 @@ use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; thread_local! { static GC_ENABLED: RefCell = const { RefCell::new(true) }; static GC_DEBUG: RefCell = const { RefCell::new(0) }; - static GC_THRESHOLD: RefCell<(i64, i64, i64)> = const { RefCell::new((700, 10, 10)) }; + static GC_THRESHOLD: RefCell<(i64, i64, i64)> = const { RefCell::new((2000, 10, 10)) }; } pub fn build(_cache: &ModuleCache) -> Rc { @@ -169,7 +169,7 @@ fn get_threshold(_args: &[Object]) -> Result { } fn set_threshold(args: &[Object]) -> Result { - let mut vals = [700i64, 10, 10]; + let mut vals = [2000i64, 10, 10]; for (slot, v) in vals.iter_mut().zip(args.iter()) { if let Object::Int(n) = v { *slot = *n; diff --git a/crates/weavepy-vm/src/stdlib/gc_real.rs b/crates/weavepy-vm/src/stdlib/gc_real.rs index e856b323..aa368d30 100644 --- a/crates/weavepy-vm/src/stdlib/gc_real.rs +++ b/crates/weavepy-vm/src/stdlib/gc_real.rs @@ -312,7 +312,7 @@ fn get_threshold(_args: &[Object]) -> Result { } fn set_threshold(args: &[Object]) -> Result { - let mut vals = [700usize, 10, 10]; + let mut vals = [2000usize, 10, 10]; for (slot, v) in vals.iter_mut().zip(args.iter()) { if let Object::Int(n) = v { *slot = (*n).max(0) as usize; diff --git a/tests/regrtest/test_ws4_gc_cascade.py b/tests/regrtest/test_ws4_gc_cascade.py index f57ad567..2822b5eb 100644 --- a/tests/regrtest/test_ws4_gc_cascade.py +++ b/tests/regrtest/test_ws4_gc_cascade.py @@ -91,7 +91,7 @@ def cb2(phase, info): # --------------------------------------------------------------------------- -# Generational thresholds round-trip (CPython default is (700, 10, 10)). +# Generational thresholds round-trip (CPython 3.14 defaults to (2000, 10, 10)). # --------------------------------------------------------------------------- saved = gc.get_threshold() From bc493fda7047a2fa9aabff356015f7e2bc41d6b6 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 22:44:30 -0700 Subject: [PATCH 26/65] perf: call math functions and read module attributes in the core loop The math module's functions register as leaf builtins over plain int, float and bool arguments, so math.sqrt(x) no longer drops to the generic call path; module attribute reads hit the site's cached index in the core loop's LOAD_ATTR arm; and a polymorphic method site now remembers each receiver class the class cache resolves for it, so its fused frameless leaf call keeps serving every class. --- crates/weavepy-vm/src/lib.rs | 89 +++++++++++++++++++++++++--- crates/weavepy-vm/src/stdlib/math.rs | 8 +++ 2 files changed, 88 insertions(+), 9 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 4f49e960..cfd77820 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -13674,7 +13674,15 @@ impl Interpreter { ), _ => None, }) { - Some(f) => Object::Function(f), + Some(f) => { + // The site remembers this + // class too (see + // `MethodSlot::poly_remember`), + // so its fused leaf-call + // path serves it next time. + ms.set(ver, &f); + Object::Function(f) + } None => break Some(CoreExit::Helper), }, }, @@ -13768,6 +13776,20 @@ impl Interpreter { _ => break Some(CoreExit::Helper), } } + // `module.name` off the site's cached index (a + // shared module leaves by a plain decrement). + Object::Module(m) => { + match Self::core_module_attr(code, m, pc, ins.arg) { + Some(v) if !gc_trace::note_dropped_marks(unsafe { &*top }) => { + // SAFETY: the receiver is replaced in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + _ => break Some(CoreExit::Helper), + } + } _ => break Some(CoreExit::Helper), }; // SAFETY: a read between two instructions (see @@ -16096,6 +16118,28 @@ impl Interpreter { } } + /// A module attribute through the `LOAD_ATTR` site's cached index: + /// the value, cloned, or `None` for the helper's full path. + #[inline] + fn core_module_attr( + code: &CodeObject, + m: &crate::object::PyModule, + pc: usize, + name_idx: u32, + ) -> Option { + use weavepy_compiler::InlineCache as IC; + let IC::LoadAttrModule { module_id, key_idx } = code.caches.get(pc as u32) else { + return None; + }; + if specialize::rc_id(&m.dict) != module_id { + return None; + } + // SAFETY: a read between two instructions (see `GilCell::peek`). + let dict = unsafe { m.dict.peek() }?; + let (k, v) = dict.get_index(key_idx as usize)?; + slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + } + /// The core loop's `x.attr` on a local receiver: the cached /// instance-dict hit read in place (no borrow guard; nothing here runs /// code), anything else through [`Self::leaf_fused_local_attr`]. @@ -18776,8 +18820,9 @@ impl Interpreter { return None; } Some(match fast { - Some(f) => LeafKind::Fast(*f), - None => LeafKind::Opaque, + leaf_builtins::Entry::Fast(f) => LeafKind::Fast(*f), + leaf_builtins::Entry::Opaque => LeafKind::Opaque, + leaf_builtins::Entry::Scalar => LeafKind::Scalar, }) } @@ -19051,6 +19096,9 @@ impl Interpreter { return Some(Ok(inst)); } K::Opaque => true, + K::Scalar => args + .iter() + .all(|a| matches!(a, O::Int(_) | O::Float(_) | O::Bool(_))), // The fast half decides (and declines untouched). K::Fast(f) => return f(args), }; @@ -56034,6 +56082,9 @@ enum LeafKind { StrFormat, /// A registered leaf builtin (see [`leaf_builtins`]): any arguments. Opaque, + /// A registered leaf builtin over plain `int`/`float`/`bool` + /// arguments (see [`leaf_builtins::Entry::Scalar`]). + Scalar, } impl LeafKind { @@ -56046,6 +56097,7 @@ impl LeafKind { matches!( self, Self::Opaque + | Self::Scalar | Self::Fast(_) | Self::Isinstance | Self::Len @@ -56785,25 +56837,44 @@ pub(crate) mod leaf_builtins { ) -> Option>; - static REGISTRY: parking_lot::Mutex, Option)>> = + /// What a registration vouches for. + #[derive(Clone, Copy)] + pub(crate) enum Entry { + /// The whole body is a leaf, for any arguments. + Opaque, + /// This pure fast half (see [`Fast`]). + Fast(Fast), + /// The whole body is a leaf when every argument is a plain `int`, + /// `float` or `bool` (numeric functions that reach Python code only + /// through another object's conversion hooks). + Scalar, + } + + static REGISTRY: parking_lot::Mutex, Entry)>> = parking_lot::Mutex::new(Vec::new()); static GENERATION: AtomicU64 = AtomicU64::new(1); /// Vouch for `b` (by identity): its whole body is a leaf. pub(crate) fn register(b: &Rc) { - push(b, None); + push(b, Entry::Opaque); } /// Vouch for `fast` as `b`'s leaf half: the dispatch loop calls it /// inline and takes the full path when it declines. pub(crate) fn register_fast(b: &Rc, fast: Fast) { - push(b, Some(fast)); + push(b, Entry::Fast(fast)); + } + + /// Vouch for `b`'s whole body as a leaf over scalar arguments (see + /// [`Entry::Scalar`]). + pub(crate) fn register_scalar(b: &Rc) { + push(b, Entry::Scalar); } - fn push(b: &Rc, fast: Option) { + fn push(b: &Rc, entry: Entry) { let mut reg = REGISTRY.lock(); reg.retain(|(w, _)| w.strong_count() > 0); - reg.push((Rc::downgrade(b), fast)); + reg.push((Rc::downgrade(b), entry)); GENERATION.fetch_add(1, Ordering::Release); } @@ -56823,7 +56894,7 @@ pub(crate) mod leaf_builtins { pub(crate) type LeafMap = std::collections::HashMap< usize, - (Weak, Option), + (Weak, Entry), crate::fasthash::FxBuildHasher, >; } diff --git a/crates/weavepy-vm/src/stdlib/math.rs b/crates/weavepy-vm/src/stdlib/math.rs index 0da3c1b8..44c17eae 100644 --- a/crates/weavepy-vm/src/stdlib/math.rs +++ b/crates/weavepy-vm/src/stdlib/math.rs @@ -268,6 +268,14 @@ pub fn build(_cache: &ModuleCache) -> Rc { builtin("acosh", math_acosh), ); } + // Every function's body is a leaf over plain numbers: only another + // object's `__float__`/`__index__`/`__ceil__`-style hooks run Python + // code, and the dispatch loop admits scalar arguments only. + for v in dict.borrow().values() { + if let Object::Builtin(b) = v { + crate::leaf_builtins::register_scalar(b); + } + } Rc::new(PyModule { name: "math".to_owned(), filename: None, From eb3985bfd84421a7d736c352dedb6a6989ec821f Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:18:55 -0700 Subject: [PATCH 27/65] perf: serve native len, truth and next on instances without generic dispatch len() of an instance whose class serves __len__ with a registered native leaf builtin calls it through a version-keyed class cache; the core loop's TO_BOOL and the JIT's truth helper answer instances through the native truth cache; and generic iteration (sum, list, ...) calls a registered native __next__ directly, with class misses cached too. The deque's slot probes use interned names. --- crates/weavepy-vm/src/lib.rs | 94 +++++++++++++++++-- .../src/stdlib/collections_native.rs | 70 +++++++++++--- crates/weavepy-vm/src/tier2.rs | 8 ++ crates/weavepy-vm/src/types.rs | 3 +- 4 files changed, 154 insertions(+), 21 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index cfd77820..b63f8cbb 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -971,7 +971,7 @@ pub struct Interpreter { /// The core loop's last native iterator `__next__`, keyed by the /// (process-unique) attribute version of the class that resolved it /// (see [`Interpreter::core_leaf_next`]). - core_next: ThreadCell)>>, + core_next: ThreadCell>)>>, /// How the last class seen in a boolean context answers it (see /// [`Interpreter::leaf_instance_truth`]), by attribute version. core_truth: ThreadCell>, @@ -12314,6 +12314,12 @@ impl Interpreter { Some(d) => !d.is_empty(), None => break None, }, + // A registered native `__bool__`/`__len__`, or + // neither (see `leaf_instance_truth`). + Object::Instance(_) => match self.leaf_instance_truth(v) { + Some(b) => b, + None => break None, + }, _ => break None, }; // SAFETY: the operand (droppable) is replaced in place. @@ -18326,6 +18332,58 @@ impl Interpreter { None } + /// `len(v)` for an instance whose class serves `__len__` with a + /// registered native leaf builtin (found through the class's leaf + /// attribute cache): the length, or `None` for the full path (which + /// raises whatever a missing or misbehaving `__len__` must). + fn leaf_instance_len(&self, v: &Object) -> Option { + use crate::types::LeafAttrKind as K; + /// The cache key for `__len__` (an address no interned name has). + static LEN_KEY: u8 = 0; + let Object::Instance(inst) = v else { + return None; + }; + let cls = inst.cls_raw(); + if crate::object::exotic_str_keys_possible() { + return None; + } + let key = std::ptr::addr_of!(LEN_KEY) as usize; + let ver = cls.attr_version.get(); + let b = match cls.leaf_attrs.get(key, ver) { + Some(K::BuiltinMethod(b)) => Rc::as_ptr(b), + Some(_) => return None, + None => { + let kind = match cls.lookup("__len__") { + Some(Object::Builtin(b)) + if b.binds_instance + && self.leaf_call_kind(&b) == Some(LeafKind::Opaque) => + { + K::BuiltinMethod(b) + } + _ => K::Other, + }; + let found = matches!(kind, K::BuiltinMethod(_)); + cls.leaf_attrs.set(key, ver, kind); + return if found { + self.leaf_instance_len(v) + } else { + None + }; + } + }; + // SAFETY: the cache (and the class) keep the builtin alive through + // the call, whose body runs no Python. + let b = unsafe { &*b }; + let r = match b.call_kw.as_ref() { + Some(ckw) => ckw(std::slice::from_ref(v), &[]), + None => (b.call)(std::slice::from_ref(v)), + }; + match r.ok()? { + Object::Int(n) if n >= 0 => Some(Object::Int(n)), + _ => None, + } + } + fn leaf_instance_truth(&self, v: &Object) -> Option { let Object::Instance(inst) = v else { return None; @@ -18773,7 +18831,7 @@ impl Interpreter { let ver = inst.cls_raw().attr_version.get(); if let Some((v, b)) = &*self.core_next.borrow() { if *v == ver { - return Some(Rc::as_ptr(b)); + return b.as_ref().map(Rc::as_ptr); } } self.core_leaf_next_resolve(inst, ver) @@ -18787,16 +18845,19 @@ impl Interpreter { inst: &PyInstance, ver: u64, ) -> Option> { - match inst.cls().lookup("__next__")? { - Object::Builtin(b) + // A miss is remembered too: a Python-level `__next__` class asks + // again on every step of a generic iteration. + let found = match inst.cls().lookup("__next__") { + Some(Object::Builtin(b)) if b.binds_instance && matches!(self.leaf_call_kind(&b), Some(LeafKind::Opaque)) => { - *self.core_next.borrow_mut() = Some((ver, b.clone())); Some(b) } _ => None, - } + }; + *self.core_next.borrow_mut() = Some((ver, found.clone())); + found } /// Which leaf builtin `b` is, if any (pointer identity). @@ -18895,6 +18956,9 @@ impl Interpreter { let leaf_str = |o: &Object| matches!(o, O::Str(_)); let leaf_int = |o: &Object| matches!(o, O::Int(_) | O::Bool(_)); let admitted = match kind { + K::Len if matches!(args, [O::Instance(_)]) => { + return self.leaf_instance_len(&args[0]).map(Ok); + } K::Len => { args.len() == 1 && matches!( @@ -33444,6 +33508,24 @@ impl Interpreter { Err(e) => Err(e), }, Object::Instance(inst) => { + // A registered native leaf `__next__` (the class cache the + // core loop's `FOR_ITER` uses): called directly. + if !crate::gil::free_threading_enabled() && !crate::trace::any_observers_active() { + if let Some(b) = self.core_leaf_next_ptr(inst) { + // SAFETY: the cache (and the class) keep the builtin + // alive through the call, whose body runs no Python. + let b = unsafe { &*b }; + return match (b.call)(std::slice::from_ref(iter)) { + Ok(v) => Ok(Some(v)), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + Ok(None) + } + Err(e) => Err(e), + }; + } + } if let Some(r) = instance_native_dunder(iter, "__next__", None) { return match r { Ok(v) => Ok(Some(v)), diff --git a/crates/weavepy-vm/src/stdlib/collections_native.rs b/crates/weavepy-vm/src/stdlib/collections_native.rs index 40f773db..4ba49718 100644 --- a/crates/weavepy-vm/src/stdlib/collections_native.rs +++ b/crates/weavepy-vm/src/stdlib/collections_native.rs @@ -47,6 +47,43 @@ struct DequeState<'a> { data: Rc>>, } +/// The slot names, interned: a slot key stored through the attribute +/// machinery shares the interned storage, so the hinted lookups settle on +/// one pointer compare instead of comparing the bytes. +struct SlotNames { + data: crate::shared_value::SharedStr, + head: crate::shared_value::SharedStr, + maxlen: crate::shared_value::SharedStr, + state: crate::shared_value::SharedStr, + deq: crate::shared_value::SharedStr, + index: crate::shared_value::SharedStr, + deq_state: crate::shared_value::SharedStr, +} + +/// The first interpreter thread's interned names (a thread interning +/// its own copies only misses the pointer compare, never the lookup). +static NAMES: std::sync::OnceLock = std::sync::OnceLock::new(); + +/// The interned slot names. +#[inline] +fn names() -> &'static SlotNames { + NAMES.get_or_init(|| { + let intern = |n: &str| match crate::stdlib::sys::intern_name(n) { + Object::Str(s) => s, + _ => unreachable!("names intern as strings"), + }; + SlotNames { + data: intern("_data"), + head: intern("_head"), + maxlen: intern("_maxlen"), + state: intern("_state"), + deq: intern("_deq"), + index: intern("_index"), + deq_state: intern("_deq_state"), + } + }) +} + // The slot positions `deque.__init__` assigns in order (hints only; the // name is always verified). const SLOT_DATA: usize = 0; @@ -55,14 +92,14 @@ const SLOT_MAXLEN: usize = 2; const SLOT_STATE: usize = 3; fn head_of(slots: &crate::types::SlotStorage) -> usize { - match slots.get_hinted(SLOT_HEAD, "_head") { + match slots.get_hinted(SLOT_HEAD, &names().head) { Some(Object::Int(h)) if *h >= 0 => *h as usize, _ => 0, } } fn set_head_of(slots: &mut crate::types::SlotStorage, h: usize) { - match slots.get_hinted_mut(SLOT_HEAD, "_head") { + match slots.get_hinted_mut(SLOT_HEAD, &names().head) { Some(slot) => *slot = Object::Int(h as i64), None => slots .insert("_head", Object::Int(h as i64)) @@ -71,14 +108,14 @@ fn set_head_of(slots: &mut crate::types::SlotStorage, h: usize) { } fn maxlen_of(slots: &crate::types::SlotStorage) -> Option { - match slots.get_hinted(SLOT_MAXLEN, "_maxlen") { + match slots.get_hinted(SLOT_MAXLEN, &names().maxlen) { Some(Object::Int(m)) if *m >= 0 => Some(*m as usize), _ => None, } } fn bump_state_of(slots: &mut crate::types::SlotStorage) { - match slots.get_hinted_mut(SLOT_STATE, "_state") { + match slots.get_hinted_mut(SLOT_STATE, &names().state) { Some(Object::Int(s)) => *s = s.wrapping_add(1), Some(slot) => *slot = Object::Int(1), None => slots.insert("_state", Object::Int(1)).map_or((), drop), @@ -118,7 +155,7 @@ fn fast_parts(args: &[Object]) -> Option<(&mut crate::types::SlotStorage, &mut V // SAFETY: see above — no guard is live on either cell (`peek_mut` // checks), and neither reference outlives the native call. let slots = unsafe { inst.slots.peek_mut() }?; - let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, "_data") else { + let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, &names().data) else { return None; }; // The list lives in its own allocation, held by the `_data` slot, @@ -158,7 +195,7 @@ fn receiver<'a>(args: &'a [Object], method: &str) -> Result, Runt .ok_or_else(|| type_error(format!("unbound method deque.{method}() needs an argument")))?; if let Object::Instance(inst) = recv { let slots = inst.slots.borrow_mut(); - if let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, "_data") { + if let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, &names().data) { let data = data.clone(); return Ok(DequeState { slots, data }); } @@ -505,23 +542,27 @@ fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option // Python runs before the last use, and the iterator and its deque are // distinct objects. let its = unsafe { iterator.slots.peek_mut() }?; - let Some(Object::Instance(deque)) = its.get_hinted(IT_DEQ, "_deq") else { + let Some(Object::Instance(deque)) = its.get_hinted(IT_DEQ, &names().deq) else { return None; }; // The deque lives while the iterator's slot holds it (unchanged here). let deque: *const crate::types::PyInstance = Rc::as_ptr(deque); let index = its - .get_hinted(IT_INDEX, "_index") + .get_hinted(IT_INDEX, &names().index) .and_then(Object::as_i64)?; let it_state = its - .get_hinted(IT_STATE, "_deq_state") + .get_hinted(IT_STATE, &names().deq_state) .and_then(Object::as_i64); // SAFETY: as above. let ds = unsafe { (*deque).slots.peek() }?; - let Some(Object::List(data)) = ds.get_hinted(SLOT_DATA, "_data") else { + let Some(Object::List(data)) = ds.get_hinted(SLOT_DATA, &names().data) else { return None; }; - if ds.get_hinted(SLOT_STATE, "_state").and_then(Object::as_i64) != it_state { + if ds + .get_hinted(SLOT_STATE, &names().state) + .and_then(Object::as_i64) + != it_state + { return None; } let h = head_of(ds); @@ -537,7 +578,7 @@ fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option h + index as usize }; let v = d[slot].clone(); - *its.get_hinted_mut(IT_INDEX, "_index")? = Object::Int(index + 1); + *its.get_hinted_mut(IT_INDEX, &names().index)? = Object::Int(index + 1); Some(v) } @@ -557,8 +598,9 @@ fn deque_next(args: &[Object], reverse: bool) -> Result { }; ( deque, - s.get_hinted(IT_INDEX, "_index").and_then(Object::as_i64), - s.get_hinted(IT_STATE, "_deq_state") + s.get_hinted(IT_INDEX, &names().index) + .and_then(Object::as_i64), + s.get_hinted(IT_STATE, &names().deq_state) .and_then(Object::as_i64), ) }; diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index fb2f474c..8aad5c27 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -9288,6 +9288,14 @@ unsafe extern "C" fn wpjit_truth(frame: *mut JitFrame, pin: i64, _reserved: i64) Some(p) => p.to_object(), None => return 3, }; + // A registered native `__bool__`/`__len__` (or neither): the + // interpreter's cached answer, which runs no Python. + if matches!(v, Object::Instance(_)) && !crate::gil::free_threading_enabled() { + if let Some(b) = interp.leaf_instance_truth(&v) { + jf.ret_bits = u64::from(b); + return 0; + } + } let pure = match &v { Object::Foreign(_) | Object::MappingProxyObj(_) => false, Object::Instance(_) => { diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index 859c2743..5b5eb8ff 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -2243,7 +2243,8 @@ impl SlotStorage { pub fn get_hinted_mut(&mut self, idx: usize, name: &str) -> Option<&mut Object> { let at_hint = matches!( self.get_index(idx), - Some((DictKey(Object::Str(stored)), _)) if slot_name_eq(stored.as_ref(), name) + Some((DictKey(Object::Str(stored)), _)) + if std::ptr::eq(stored.as_ptr(), name.as_ptr()) || slot_name_eq(stored.as_ref(), name) ); if at_hint { return self.get_index_mut(idx).map(|(_, v)| v); From 0c735b09197c70777586552cb5d90538c3b84e26 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:50:11 -0700 Subject: [PATCH 28/65] perf: format and join strings in the core loop, and stamp dicts without a lock The core loop's BINARY_OP serves str + str, small str * int, and str % over scalar and string arguments through an out-of-line helper instead of handing the instruction to the full leaf arms. Dict mutation stamps come from a plain load and store while the GIL serializes mutation (the locked increment stays for free-threaded mode and the debug test binary). --- crates/weavepy-vm/src/gc_trace.rs | 4 ++- crates/weavepy-vm/src/lib.rs | 54 +++++++++++++++++++++++++++++++ crates/weavepy-vm/src/object.rs | 13 +++++++- 3 files changed, 69 insertions(+), 2 deletions(-) diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 0a24efbb..5d97a2ff 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -3436,7 +3436,9 @@ pub fn track(obj: Object) { /// only in free-threaded mode. #[inline(always)] fn serial_add(counter: &AtomicUsize, n: usize) { - if crate::gil::free_threading_enabled() { + // (The debug unit-test binary runs interpreters on concurrent threads + // with no GIL between them: it keeps the locked form too.) + if cfg!(debug_assertions) || crate::gil::free_threading_enabled() { counter.fetch_add(n, Ordering::AcqRel); } else { let v = counter.load(Ordering::Relaxed); diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index b63f8cbb..5fbe7ff8 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12457,6 +12457,25 @@ impl Interpreter { None => break None, } } + // String concatenation, repetition and `%` + // formatting over scalars, out of line. + (Object::Str(_), _) | (Object::Int(_), Object::Str(_)) => { + let Some(r) = Self::core_str_binop(a, b, kind) else { + break None; + }; + // SAFETY: the operands are strings and scalars + // (or a tuple of them): releasing them runs no + // code and frees nothing the collector tracks. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 1; + unsafe { base.add(len - 1).write(r) }; + last = pc; + pc += 1; + continue; + } _ => break None, }; // SAFETY: both operands are scalars (no drop owed). @@ -14254,6 +14273,41 @@ impl Interpreter { } } + /// The core loop's `BINARY_OP` over strings: `str + str`, a small + /// `str * int`, and `str % args` over scalar and string arguments + /// (the leaf arms' string cases). `None` for the full handler, which + /// also owns every error. + #[inline(never)] + fn core_str_binop(a: &Object, b: &Object, kind: BinOpKind) -> Option { + Some(match (a, b) { + (Object::Str(x), Object::Str(y)) if kind == BinOpKind::Add => { + Object::Str(SharedStr::concat(&[x, y])) + } + (Object::Str(s), Object::Int(k)) | (Object::Int(k), Object::Str(s)) + if kind == BinOpKind::Mult => + { + let times = usize::try_from(*k).unwrap_or(0); + if s.len().saturating_mul(times) > 1 << 16 { + return None; + } + if times == 1 { + // `s * 1` is `s` itself (CPython `unicode_repeat`). + Object::Str(s.clone()) + } else { + Object::Str(SharedStr::repeat(s, times)) + } + } + (Object::Str(t), args) + if kind == BinOpKind::Mod + && percent_leaf_args(args) + && !percent_args_need_bridge(args) => + { + Object::from_str(percent_format(t, args).ok()?) + } + _ => return None, + }) + } + /// The core loop's container instructions: `seq[i]` and `d[key]` over /// exact containers (an in-range index or a present `str`/`int` key), /// `seq[i] = v` in range and `d[key] = v`, and a comprehension's diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index db103ba6..4ef6ef34 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -4039,7 +4039,18 @@ static DICT_STAMP: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64:: #[inline] fn next_dict_stamp() -> u64 { - DICT_STAMP.fetch_add(1, std::sync::atomic::Ordering::Relaxed) + use std::sync::atomic::Ordering::Relaxed; + // Dict mutation is serialized by the GIL, so a plain load and store + // hands out unique stamps without a locked read-modify-write (drawn on + // every mutable dict access). Free-threaded mode needs the real one, + // and so does the debug unit-test binary, whose tests run separate + // interpreters on concurrent threads with no GIL between them. + if cfg!(debug_assertions) || crate::gil::free_threading_enabled() { + return DICT_STAMP.fetch_add(1, Relaxed); + } + let v = DICT_STAMP.load(Relaxed); + DICT_STAMP.store(v + 1, Relaxed); + v } /// A [`DictMap`] with a mutation stamp: every mutable access (any From d4f900c279c1da5dfd170bbddc8884dc96034124 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 00:11:50 -0700 Subject: [PATCH 29/65] perf: run natively served fields, arithmetic and comparisons in the core loop A datetime-style instance's public field read, its binary operators and its comparisons now run from the core loop's LOAD_ATTR, BINARY_OP and COMPARE_OP arms (through out-of-line calls into the native bodies) instead of leaving the loop twice for the helper and the full leaf arms. --- crates/weavepy-vm/src/lib.rs | 108 +++++++++++++++++++++++++++++++++++ 1 file changed, 108 insertions(+) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 5fbe7ff8..650f91d0 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12457,6 +12457,39 @@ impl Interpreter { None => break None, } } + // A natively served left operand (see + // `stdlib::datetime_native`), out of line. + (Object::Instance(i), _) if i.cls_raw().native_kind.get() != 0 => { + let r = match Self::core_native_binop(kind, a, b) { + Some(r) + if Self::core_droppable(a) && Self::core_droppable(b) => + { + r + } + _ => break None, + }; + // SAFETY: both operands leave by plain + // decrements (checked). + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 2; + match r { + Ok(v) => { + // SAFETY: the operands' slots are free. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } // String concatenation, repetition and `%` // formatting over scalars, out of line. (Object::Str(_), _) | (Object::Int(_), Object::Str(_)) => { @@ -12557,6 +12590,32 @@ impl Interpreter { Some(o) => o, None => break None, }, + // Two natively served instances (see + // `stdlib::datetime_native`), out of line. + (Object::Instance(i), Object::Instance(_)) + if i.cls_raw().native_kind.get() != 0 => + { + let r = match Self::core_native_compare(kind, a, b) { + Some(Ok(r)) + if Self::core_droppable(a) && Self::core_droppable(b) => + { + r + } + _ => break None, + }; + // SAFETY: both operands leave by plain + // decrements (checked); the result takes the + // lower one's slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 1; + unsafe { base.add(len - 1).write(r) }; + last = pc; + pc += 1; + continue; + } _ => break None, }; let r = match kind { @@ -13817,6 +13876,21 @@ impl Interpreter { } _ => break Some(CoreExit::Helper), }; + // A natively served instance's public field (see + // `stdlib::datetime_native`). + if inst.cls_raw().native_kind.get() != 0 { + match Self::core_native_field(ext, inst, ins.arg) { + Some(v) if Self::core_droppable(unsafe { &*top }) => { + // SAFETY: the receiver (droppable) is + // replaced in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + _ => break Some(CoreExit::Helper), + } + } // SAFETY: a read between two instructions (see // `GilCell::peek`). if let Some(v) = ext.and_then(|e| unsafe { field_slot_hit(e, pc, inst) }) { @@ -16178,6 +16252,40 @@ impl Interpreter { } } + /// [`crate::stdlib::datetime_native::leaf_binop`], out of line. + #[inline(never)] + fn core_native_binop( + kind: BinOpKind, + a: &Object, + b: &Object, + ) -> Option> { + crate::stdlib::datetime_native::leaf_binop(kind, a, b) + } + + /// [`crate::stdlib::datetime_native::leaf_compare`], out of line. + #[inline(never)] + fn core_native_compare( + kind: CompareKind, + a: &Object, + b: &Object, + ) -> Option> { + crate::stdlib::datetime_native::leaf_compare(kind, a, b) + } + + /// A natively served instance's public field (`dt.hour`; see + /// [`crate::stdlib::datetime_native::leaf_field`]), or `None`. + #[inline(never)] + fn core_native_field( + ext: Option<&CodeConstObjects>, + inst: &PyInstance, + name_idx: u32, + ) -> Option { + let Object::Str(name) = ext?.name_objs.get(name_idx as usize)? else { + return None; + }; + crate::stdlib::datetime_native::leaf_field(inst, name) + } + /// A module attribute through the `LOAD_ATTR` site's cached index: /// the value, cloned, or `None` for the helper's full path. #[inline] From ba7403ce512236e5b2790e848b37d0be11bb2d3f Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 00:27:13 -0700 Subject: [PATCH 30/65] perf: serve scalar JIT attribute reads and writes on a lean path The tier-2 attribute get and set helpers try the common shape first (a scalar lane over an indexed instance field whose class is unchanged) before the general lane classification, and the class guard is always inlined. --- crates/weavepy-vm/src/tier2.rs | 54 +++++++++++++++++++++++++++++++++- 1 file changed, 53 insertions(+), 1 deletion(-) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 8aad5c27..42e81fb0 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -7480,7 +7480,7 @@ fn key_is(key: &DictKey, name: &SharedStr) -> bool { /// A site guard's class check: the receiver's class still carries the /// compiled `attr_version` (read without a borrow guard when only one /// thread runs Python; nothing here runs code). -#[inline] +#[inline(always)] fn attr_class_ok(inst: &crate::types::PyInstance, ver: u64) -> bool { if crate::gil::free_threading_enabled() { return inst.class.borrow().attr_version.get() == ver; @@ -7664,6 +7664,29 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - let jf = unsafe { &mut *frame }; #[allow(clippy::cast_ptr_alignment)] let ctx = unsafe { &mut *jf.ctx.cast::() }; + // The common shape first: a scalar field of an indexed site, read + // straight off the instance (the full path below re-derives it). + if let (Some(Pin::Obj(Object::Instance(inst))), Some(g)) = ( + ctx.pins.get(pin as usize), + ctx.attr_guards.get(site as usize), + ) { + if let (AttrStorage::Indexed(key_idx), JitType::Int | JitType::Float | JitType::Bool) = + (g.storage, g.lane) + { + if attr_class_ok(inst, g.ver) { + // SAFETY: a read between two native ops; nothing here runs + // code (see `GilCell::peek`). + if let Some((k, v)) = unsafe { inst.attr_peek_index(key_idx as usize) } { + if key_is(k, &g.name) { + if let Some(bits) = pack(v, g.lane) { + jf.ret_bits = bits; + return 0; + } + } + } + } + } + } // Scoped so the receiver borrow of `ctx.pins` ends before an // object-lane result appends a fresh pin (RFC 0070 WS1). let outcome: Result = { @@ -8260,6 +8283,35 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - let Some(g) = ctx.attr_guards.get(site as usize) else { return 1; }; + // The common shape first: a scalar value over a scalar field of an + // indexed site (no write barrier, nothing to reap). + if let (AttrStorage::Indexed(key_idx), Some(Pin::Obj(Object::Instance(inst)))) = + (g.storage, ctx.pins.get(pin as usize)) + { + let v = match g.lane { + JitType::Int => Some(Object::Int(jf.ret_bits as i64)), + JitType::Float => Some(Object::Float(f64::from_bits(jf.ret_bits))), + JitType::Bool => Some(Object::Bool(jf.ret_bits != 0)), + _ => None, + }; + if let Some(v) = v { + if attr_class_ok(inst, g.ver) { + // SAFETY: nothing below runs code while the view is live. + if let Some((k, dst)) = unsafe { inst.attr_peek_index_mut(key_idx as usize, true) } + { + if key_is(k, &g.name) + && matches!( + dst, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) + { + *dst = v; + return 0; + } + } + } + } + } let v = match g.lane { JitType::Int => Object::Int(jf.ret_bits as i64), JitType::Float => Object::Float(f64::from_bits(jf.ret_bits)), From f5a749bfc6b902d886e169bacc7a485579ef1a63 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 01:12:12 -0700 Subject: [PATCH 31/65] perf: slice sequences and search strings without leaving the core loop BINARY_SUBSCR over a slice of ints and BINARY_SLICE with int bounds on a list, tuple or string now run in the core loop's container helper (through the full handler's own slicing) instead of the full step. The substring scans behind str.find and str.count step through an ASCII first byte with memchr and compare the rest inline. --- crates/weavepy-vm/src/builtins.rs | 44 +++++++++++++++++++ crates/weavepy-vm/src/lib.rs | 70 ++++++++++++++++++++++++++++++- 2 files changed, 113 insertions(+), 1 deletion(-) diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index a2900c13..8a7b2366 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -11313,6 +11313,25 @@ pub(crate) fn substr_find(hay: &str, needle: &str) -> Option { return hay.find(needle); } let last_start = h.len() - n.len(); + if n[0].is_ascii() { + // An ASCII byte is never inside a multibyte character, so every + // hit is a character boundary: scan the bytes directly (no char + // searcher) and compare the short rest inline (no `memcmp` call). + let mut i = 0; + let mut budget = h.len() / n.len() + 32; + while i <= last_start { + let at = i + memchr::memchr(n[0], &h[i..=last_start])?; + if short_bytes_eq(&h[at + 1..at + n.len()], &n[1..]) { + return Some(at); + } + budget -= 1; + if budget == 0 { + return hay[at + 1..].find(needle).map(|k| k + at + 1); + } + i = at + 1; + } + return None; + } // `str::find(char)` is memchr-backed, so the candidate scan runs at // vector width; the manual byte loop it replaced did not. let first = needle.chars().next()?; @@ -11341,6 +11360,13 @@ pub(crate) fn substr_find(hay: &str, needle: &str) -> Option { None } +/// `a == b` for the short needle tails the substring scans compare, inline +/// (a `memcmp` call costs more than comparing a few bytes). +#[inline(always)] +fn short_bytes_eq(a: &[u8], b: &[u8]) -> bool { + a.len() == b.len() && a.iter().zip(b).all(|(x, y)| x == y) +} + /// [`substr_find`] from the right. pub(crate) fn substr_rfind(hay: &str, needle: &str) -> Option { let (h, n) = (hay.as_bytes(), needle.as_bytes()); @@ -11378,6 +11404,24 @@ pub(crate) fn substr_count(hay: &str, needle: &str) -> usize { return hay.matches(needle).count(); } let last_start = h.len() - n.len(); + if n[0].is_ascii() { + // As `substr_find`'s ASCII scan (every hit is a boundary). + let mut count = 0; + let mut i = 0; + while i <= last_start { + let Some(off) = memchr::memchr(n[0], &h[i..=last_start]) else { + break; + }; + let at = i + off; + if short_bytes_eq(&h[at + 1..at + n.len()], &n[1..]) { + count += 1; + i = at + n.len(); + } else { + i = at + 1; + } + } + return count; + } let Some(first) = needle.chars().next() else { return 0; }; diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 650f91d0..8e75a57f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12521,6 +12521,7 @@ impl Interpreter { // out of line: arms added here cost the rest of the loop // its register allocation. OpCode::BinarySubscr + | OpCode::BinarySlice | OpCode::StoreSubscr | OpCode::ListAppend | OpCode::UnpackSequence => { @@ -14347,6 +14348,25 @@ impl Interpreter { } } + /// `seq[slice]` for the core loop's `BINARY_SUBSCR` (a list, tuple or + /// string over a slice of ints): the full handler's slicing, or + /// `None` for it to run (and raise) itself. + #[inline(never)] + fn core_slice(c: &Object, sl: &crate::object::PySlice) -> Option { + match c { + Object::List(items) => { + let items = items.try_borrow().ok()?; + Some(Object::new_list(slice_seq(&items, sl).ok()?)) + } + Object::Tuple(items) => { + let v: Vec = items.iter().cloned().collect(); + Some(Object::new_tuple(slice_seq(&v, sl).ok()?)) + } + Object::Str(st) => str_subscript_slice(st, sl).ok(), + _ => None, + } + } + /// The core loop's `BINARY_OP` over strings: `str + str`, a small /// `str * int`, and `str % args` over scalar and string arguments /// (the leaf arms' string cases). `None` for the full handler, which @@ -14438,10 +14458,21 @@ impl Interpreter { let d = d.try_borrow().ok()?; clone_hot(d.get(&probe)?) } + // `seq[a:b:c]` with plain int (or omitted) bounds: the + // full handler's own slicing, out of line (a bad step + // declines, and the full handler raises). + (Object::List(_) | Object::Tuple(_) | Object::Str(_), Object::Slice(sl)) + if [&sl.start, &sl.stop, &sl.step] + .iter() + .all(|v| matches!(v, Object::None | Object::Int(_))) => + { + Self::core_slice(c, sl)? + } _ => return None, }; // SAFETY: both operand slots are initialized. The key is a - // scalar or string and the container a shared value + // scalar, a string or a slice of ints, and the container a + // shared value // (`core_droppable`), so neither release runs code; the // result takes the container's slot. unsafe { @@ -14451,6 +14482,43 @@ impl Interpreter { } Some(len - 1) } + // `seq[a:b]` with the bounds on the stack (CPython's + // `BINARY_SLICE`): as `BINARY_SUBSCR` over the slice. + OpCode::BinarySlice => { + if len < 3 { + return None; + } + // SAFETY: `len >= 3`. + let (c, a, b) = unsafe { + ( + &*base.add(len - 3), + &*base.add(len - 2), + &*base.add(len - 1), + ) + }; + if !matches!(c, Object::List(_) | Object::Tuple(_) | Object::Str(_)) + || !Self::core_droppable(c) + || !matches!(a, Object::None | Object::Int(_)) + || !matches!(b, Object::None | Object::Int(_)) + { + return None; + } + let sl = crate::object::PySlice { + start: a.clone(), + stop: b.clone(), + step: Object::None, + }; + let r = Self::core_slice(c, &sl)?; + // SAFETY: the bounds are scalars and the container a + // shared value (checked); the result takes its slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + drop_hot(base.add(len - 3).read()); + base.add(len - 3).write(r); + } + Some(len - 2) + } OpCode::StoreSubscr => { if len < 3 { return None; From 493558a693c0fab5ccc9be451a66307530e7d501 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 01:42:11 -0700 Subject: [PATCH 32/65] perf: reuse a dynamic call's resolved native callee A compiled loop calling a stored bound method or a function held in a local resolved the callee's native entry on every call. Each call activation now remembers the last resolution, keyed by the function, its code and the call shape, so a steady site skips the lookup (about 16% of a bound-method call). --- crates/weavepy-vm/src/tier2.rs | 56 +++++++++++++++++++++++++--------- 1 file changed, 42 insertions(+), 14 deletions(-) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 42e81fb0..e57d2b56 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -3193,6 +3193,12 @@ struct CallCtx { /// the exact chain). `None` for a framed entry — its `Frame`'s /// shell is already on the spine. frameless_code: Option>, + /// The last dynamic call's resolved native callee, keyed by the + /// function and code identities and the method form: a site calling + /// the same function again (a stored bound method, a function held in + /// a local) re-enters without re-resolving or rebuilding the handle. + /// Taken out for the call's duration and put back after. + dyn_callee: Option<(usize, usize, bool, NativeCallee)>, } impl CallCtx { @@ -4465,6 +4471,7 @@ unsafe fn try_native_call( // The native call lanes push no interpreter frame for the // callee — keep it observable to callee-side stack walkers. frameless_code: Some(nc.code.clone()), + dyn_callee: None, }) }; child.interp = ctx.interp; @@ -8545,22 +8552,40 @@ unsafe fn try_dyn_native( argc: u32, int_result: bool, ) -> Option { - let (nc, recv) = match callee { - Object::Function(pf) => { - let fcode = pf.code.borrow().clone(); - let nc = JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, false))?; - (nc, None) - } + // A plain function or a bound method's function: the activation's + // last resolution when it names the same function and code (see + // `CallCtx::dyn_callee`). + let plain = match callee { + Object::Function(pf) => Some((pf, None)), // A deferred special-method dispatch (`redispatch_descriptor`) // re-resolves `__get__` at call time — interpreter territory. - Object::BoundMethod(bm) if !bm.redispatch_descriptor => { - let Object::Function(pf) = &bm.function else { - return None; - }; - let fcode = pf.code.borrow().clone(); - let nc = JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, true))?; - (nc, Some(bm.receiver.clone())) - } + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(pf) => Some((pf, Some(bm.receiver.clone()))), + _ => return None, + }, + _ => None, + }; + if let Some((pf, recv)) = plain { + let method = recv.is_some(); + // SAFETY: GIL-serialized raw read of the function's code cell; + // only the pointer is compared. + let code_ptr = unsafe { Rc::as_ptr(&*pf.code.as_ptr()) } as usize; + let key = (Rc::as_ptr(pf) as usize, code_ptr, method); + let nc = match ctx.dyn_callee.take() { + Some((f, c, m, nc)) if (f, c, m) == key => nc, + _ => { + let fcode = pf.code.borrow().clone(); + JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, method))? + } + }; + // SAFETY: the resolved owners outlive the call, and the same + // initialized argument-buffer contract applies to the shared + // entry path. + let r = unsafe { enter_dyn_native(jf, ctx, interp, &nc, argc, recv.as_ref(), int_result) }; + ctx.dyn_callee = Some((key.0, key.1, key.2, nc)); + return r; + } + let (nc, recv) = match callee { Object::Type(t) => { // Mirror `resolve_native_callee`'s constructor arm: the // memoised instance plan must be current and carry a @@ -10152,6 +10177,7 @@ pub(crate) fn try_call_native_direct( // The frameless direct entry pushes no interpreter frame — // keep the activation observable to callee-side stack walkers. frameless_code: Some(code.clone()), + dyn_callee: None, }; let mut jf = JitFrame { locals: locals_buf.as_mut_ptr(), @@ -10954,6 +10980,7 @@ fn enter_compiled( // Framed entry: this activation's `Frame` shell is on the // spine already. frameless_code: None, + dyn_callee: None, }; let mut jf = JitFrame { locals: locals_buf.as_mut_ptr(), @@ -11842,6 +11869,7 @@ fn resume_parked(interp: &mut super::Interpreter, frame: &mut super::Frame) -> J // Framed entry (generator resume): the resumed `Frame`'s shell // is on the spine already. frameless_code: None, + dyn_callee: None, }; let mut jf = JitFrame { locals: act.locals_buf.as_mut_ptr(), From 2d23ffebdac38484ccb08904ae070ba6ea0ecf1e Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 01:58:54 -0700 Subject: [PATCH 33/65] perf: call read-only builtins from frameless leaves A leaf body that called a builtin (len, isinstance, math functions, dict.get, str methods) always declined and took a framed call, so a function as small as `return len(l) + x` cost three times CPython's call. The leaf plan now runs builtins whose kind only reads its arguments on borrowed copies of them, with the call site's cached kind, and declines untouched on anything else (a raise included, which the ordinary call repeats). Such a call drops from about 2100 to 1200 instructions. Also match CPython's str.format message for a missing positional replacement field. --- crates/weavepy-vm/src/leaf_plan.rs | 79 +++++++-- crates/weavepy-vm/src/lib.rs | 200 ++++++++++++++++++++-- tests/regrtest/test_leaf_builtin_calls.py | 123 +++++++++++++ 3 files changed, 378 insertions(+), 24 deletions(-) create mode 100644 tests/regrtest/test_leaf_builtin_calls.py diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index f0aa3514..94dac300 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -24,6 +24,9 @@ use crate::{CodeConstObjects, Interpreter}; pub(crate) enum V { /// A resolved callee: a function the class or namespace holds. Fn(*const crate::object::PyFunction), + /// A builtin type's method body, from the leaf method table (which + /// holds it for the interpreter's lifetime). + Bi(*const crate::object::BuiltinFn), /// A call's empty self slot. Null, /// A heap object, borrowed. @@ -62,7 +65,7 @@ pub(crate) fn truth(v: V) -> Option { Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), _ => return None, }, - V::Fn(_) | V::Null => return None, + V::Fn(_) | V::Bi(_) | V::Null => return None, }) } @@ -189,6 +192,7 @@ enum Op { Call { at: u8, argc: u8, + pc: u16, }, Return { src: u8, @@ -558,6 +562,7 @@ impl Builder<'_> { self.ops.push(Op::Call { at: r, argc: u8::try_from(argc).ok()?, + pc: u16::try_from(pc).ok()?, }); } OpCode::ReturnValue => { @@ -771,6 +776,16 @@ impl Drop for Pending { } } +/// A leaf call's callee: a Python function to evaluate in place, or a +/// builtin (with its owner, when a namespace holds one). +enum Callee<'a> { + Py(*const crate::object::PyFunction), + Native( + *const crate::object::BuiltinFn, + Option<&'a Rc>, + ), +} + /// An owned object for a leaf value (the return value, a buffered /// store's value); `None` for the markers that are never values. #[inline(always)] @@ -783,7 +798,7 @@ fn to_object(v: V) -> Option { V::F(x) => Object::Float(x), V::B(b) => Object::Bool(b), V::N => Object::None, - V::Fn(_) | V::Null => return None, + V::Fn(_) | V::Bi(_) | V::Null => return None, }) } @@ -976,7 +991,11 @@ impl Interpreter { set!(dst, V::Fn(ms.peek_unbound(cls.attr_version.get())?)); set!(dst + 1, V::Null); } - _ => return None, + recv => { + let b = self.leaf_builtin_method_ptr(ms, recv, code, name)?; + set!(dst, V::Bi(b)); + set!(dst + 1, V::R(p)); + } } } Op::Compare { dst, a, b, kind } => { @@ -1074,36 +1093,68 @@ impl Interpreter { } Op::Jump { target } => ip = usize::from(target), Op::Decline => return None, - Op::Call { at, argc } => { + Op::Call { at, argc, pc } => { // A pure-leaf callee, evaluated in place: only while // no store is buffered (it would not see one). if (EFFECT && pend.n > 0) || nest >= NEST { return None; } - let fp = match get!(at) { - V::Fn(fp) => fp, + let callee = match get!(at) { + V::Fn(fp) => Callee::Py(fp), + V::Bi(b) => Callee::Native(b, None), // SAFETY: as `norm`. V::R(p) => match unsafe { &*p } { - Object::Function(func) => Rc::as_ptr(func), + Object::Function(func) => Callee::Py(Rc::as_ptr(func)), + Object::Builtin(b) => Callee::Native(Rc::as_ptr(b), Some(b)), _ => return None, }, _ => return None, }; - // SAFETY: the class or the namespace holds the callee, - // and nothing here runs code that could release it. - let callee = unsafe { &*fp }; - // SAFETY: GIL-serialized raw read of the code cell. - let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; let first = if matches!(get!(at + 1), V::Null) { at + 2 } else { at + 1 }; let n = usize::from(at + 2 + argc - first); + if n > 8 { + return None; + } + let fp = match callee { + Callee::Py(fp) => fp, + Callee::Native(b, rc) => { + // A read-only builtin, on borrowed copies of the + // arguments: never dropped, so no reference moves. + let mut staged = + [const { std::mem::MaybeUninit::::uninit() }; 8]; + for k in 0..n { + let o = match get!(first + k as u8) { + // SAFETY: as `norm`; the copy is forgotten. + V::R(p) => unsafe { std::ptr::read(p) }, + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Bi(_) | V::Null => return None, + }; + staged[k].write(o); + } + // SAFETY: the first `n` entries were written. + let args = unsafe { + std::slice::from_raw_parts(staged.as_ptr().cast::(), n) + }; + let r = self.leaf_pure_builtin(code, usize::from(pc), b, rc, args)?; + set!(at, owned.own(r)?); + continue; + } + }; + // SAFETY: the class or the namespace holds the callee, + // and nothing here runs code that could release it. + let callee = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the code cell. + let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; if !crate::code_is_pure_leaf(ccode) || !Self::leaf_code_ok(ccode) || n != ccode.arg_count as usize - || n > 8 || crate::recursion::current_depth() + usize::from(nest) + 1 >= crate::recursion::recursion_limit() { @@ -1122,7 +1173,7 @@ impl Interpreter { V::F(x) => Object::Float(x), V::B(b) => Object::Bool(b), V::N => Object::None, - V::Fn(_) | V::Null => return None, + V::Fn(_) | V::Bi(_) | V::Null => return None, }; ptrs[k] = staged[k].write(o); } diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 8e75a57f..62a9a0cf 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -19170,12 +19170,96 @@ impl Interpreter { Some(result) } + /// A frameless leaf's method load off a builtin receiver at the site + /// `slot` (`name_idx` in `code`'s names): the leaf method table's + /// body, which that table keeps alive. + pub(crate) fn leaf_builtin_method_ptr( + &self, + slot: &MethodSlot, + recv: &Object, + code: &CodeObject, + name_idx: u16, + ) -> Option<*const crate::object::BuiltinFn> { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + Object::Str(_) => 4, + _ => return None, + }; + if let Some(b) = slot.get_builtin_ptr(tag) { + return Some(b); + } + let b = self.leaf_builtin_method(recv, code.names.get(usize::from(name_idx))?)?; + slot.set_builtin(tag, &b); + Some(Rc::as_ptr(&b)) + } + + /// A frameless leaf's call at `pc` of the builtin `b` (`rc` its owner, + /// when the plan borrowed one) on `args`: the result of an admitted + /// read-only kind (see [`LeafKind::is_pure_read`]). `None` touched + /// nothing; that includes a raise, which the ordinary call repeats. + pub(crate) fn leaf_pure_builtin( + &self, + code: &CodeObject, + pc: usize, + b: *const crate::object::BuiltinFn, + rc: Option<&Rc>, + args: &[Object], + ) -> Option { + let slot = code_method_slot(code, pc as u32); + let kind = match slot.and_then(|s| s.get_leaf_ptr(b)) { + Some(k) => k, + None => self.leaf_pure_builtin_kind(slot, b, rc)?, + }; + if !kind.is_pure_read() { + return None; + } + // SAFETY: the plan's holder (a namespace, or the leaf method + // table) keeps the builtin alive, and nothing here releases it. + let r = self.leaf_builtin_call(kind, unsafe { &*b }, args)?.ok(); + #[cfg(test)] + if r.is_some() { + LEAF_BUILTIN_CALLS.with(|calls| calls.set(calls.get() + 1)); + } + r + } + + #[cold] + #[inline(never)] + fn leaf_pure_builtin_kind( + &self, + slot: Option<&MethodSlot>, + b: *const crate::object::BuiltinFn, + rc: Option<&Rc>, + ) -> Option { + let found; + let rc = match rc { + Some(rc) => rc, + None => { + found = self + .leaf_fns() + .methods + .iter() + .find(|(_, _, f)| Rc::as_ptr(f) == b)? + .2 + .clone(); + &found + } + }; + let kind = self.leaf_call_kind(rc)?; + if let Some(s) = slot { + s.set_leaf(rc, kind); + } + Some(kind) + } + /// Call a leaf builtin if `args` (receiver first for methods) has an /// admitted shape; `None` sends the call down the full path. fn leaf_builtin_call( - &mut self, + &self, kind: LeafKind, - b: &Rc, + b: &crate::object::BuiltinFn, args: &[Object], ) -> Option> { use LeafKind as K; @@ -56464,6 +56548,38 @@ impl LeafKind { | Self::StrSplit ) } + + /// Operations that read their arguments and build a fresh result, + /// with nothing else observable: a frameless leaf may run them and + /// still decline afterwards (see `Interpreter::leaf_pure_builtin`). + fn is_pure_read(self) -> bool { + matches!( + self, + Self::Len + | Self::Isinstance + | Self::ListCopy + | Self::DictGet + | Self::DictKeys + | Self::DictValues + | Self::DictItems + | Self::StrStartswith + | Self::StrEndswith + | Self::StrLower + | Self::StrUpper + | Self::StrStrip + | Self::StrLstrip + | Self::StrRstrip + | Self::StrFind + | Self::StrIsdigit + | Self::StrIsalpha + | Self::StrIsspace + | Self::StrSplit + | Self::StrJoin + | Self::StrReplace + | Self::StrFormat + | Self::Scalar + ) + } } /// A receiver's builtin variant, for the method table. @@ -57094,6 +57210,7 @@ thread_local! { static PURE_LITERAL_ARGUMENT_CALLS: std::cell::Cell<[u64; 4]> = const { std::cell::Cell::new([0; 4]) }; static PURE_SLOT_FIELD_READS: std::cell::Cell<[u64; 3]> = const { std::cell::Cell::new([0; 3]) }; static PURE_CACHED_FIELD_PREDICATES: std::cell::Cell = const { std::cell::Cell::new(0) }; + static LEAF_BUILTIN_CALLS: std::cell::Cell = const { std::cell::Cell::new(0) }; static PURE_PREDICATE_STAGES: std::cell::Cell<[u64; 9]> = const { std::cell::Cell::new([0; 9]) }; static PURE_PREDICATE_DROP_MISSES: std::cell::Cell<[u64; 4]> = const { std::cell::Cell::new([0; 4]) }; static NATIVE_SUBSCRIPT_CACHE_HITS: std::cell::Cell = const { std::cell::Cell::new(0) }; @@ -59090,15 +59207,17 @@ fn resolve_field_name( let mut value = if base.is_empty() { let idx = *auto_idx; *auto_idx += 1; - positional - .get(idx) - .cloned() - .ok_or_else(|| index_error(format!("Replacement index {idx} out of range")))? + positional.get(idx).cloned().ok_or_else(|| { + index_error(format!( + "Replacement index {idx} out of range for positional args tuple" + )) + })? } else if let Ok(idx) = base.parse::() { - positional - .get(idx) - .cloned() - .ok_or_else(|| index_error(format!("Replacement index {idx} out of range")))? + positional.get(idx).cloned().ok_or_else(|| { + index_error(format!( + "Replacement index {idx} out of range for positional args tuple" + )) + })? } else if let Some(map) = mapping { let key = DictKey(Object::from_str(base)); map.borrow() @@ -62577,6 +62696,17 @@ impl MethodSlot { } } + /// [`Self::get_builtin`], uncounted: the leaf method table holds the + /// body for the interpreter's lifetime. + #[inline] + fn get_builtin_ptr(&self, tag: u64) -> Option<*const crate::object::BuiltinFn> { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (v, MethodSlotFn::Builtin(f)) if *v == Self::BUILTIN_TAG | tag => Some(Rc::as_ptr(f)), + _ => None, + } + } + #[inline] fn set_builtin(&self, tag: u64, f: &Rc) { // SAFETY: as `set`. @@ -66815,6 +66945,56 @@ assert loop(2000) == 1999000 .unwrap(); } + #[test] + fn leaf_builtin_calls_run_frameless() { + const CHILD: &str = "WEAVEPY_LEAF_BUILTIN_TEST_CHILD"; + if std::env::var_os(CHILD).is_none() { + // Compiled callers reach the same evaluator; the counter is + // required with the JIT off, and semantics in both modes. + for jit in ["0", "1"] { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "tests::leaf_builtin_calls_run_frameless", + "--nocapture", + ]) + .env(CHILD, "1") + .env("WEAVEPY_JIT", jit) + .status() + .expect("spawn leaf builtin test"); + assert!(status.success(), "leaf builtin child: {status}"); + } + return; + } + std::thread::Builder::new() + .stack_size(8 * 1024 * 1024) + .spawn(|| { + let source = include_str!("../../../tests/regrtest/test_leaf_builtin_calls.py"); + let module = parse_module(source).unwrap(); + let code = weavepy_compiler::compile_module_with_source( + &module, + source, + "leaf_builtin_calls.py", + ) + .unwrap(); + let before = LEAF_BUILTIN_CALLS.with(std::cell::Cell::get); + Interpreter::new() + .run_module(&code) + .expect("leaf builtin assertions"); + let calls = LEAF_BUILTIN_CALLS.with(std::cell::Cell::get) - before; + #[cfg(feature = "jit")] + let require = crate::tier2::jit_off_for_process(); + #[cfg(not(feature = "jit"))] + let require = true; + if require { + assert!(calls > 10_000, "frameless builtin calls: {calls}"); + } + }) + .unwrap() + .join() + .unwrap(); + } + #[cfg(feature = "jit")] #[test] fn pure_cached_field_predicates_preserve_fallbacks() { diff --git a/tests/regrtest/test_leaf_builtin_calls.py b/tests/regrtest/test_leaf_builtin_calls.py new file mode 100644 index 00000000..f0d9da2e --- /dev/null +++ b/tests/regrtest/test_leaf_builtin_calls.py @@ -0,0 +1,123 @@ +"""Leaf functions that call read-only builtins keep their results, errors, +and callbacks.""" +import math + + +class Box: + def __init__(self, v): + self.v = v + + +class Key: + calls = 0 + + def __init__(self, k): + self.k = k + + def __hash__(self): + return hash(self.k) + + def __eq__(self, other): + Key.calls += 1 + return isinstance(other, Key) and other.k == self.k + + +class Sized: + def __len__(self): + return 7 + + +def f_len(x, y): + return len(x) + y + + +def f_isinstance(x): + return isinstance(x, int) + + +def f_get(d, k): + return d.get(k, -1) + + +def f_sqrt(x): + return math.sqrt(x) + 1.0 + + +def f_upper(s): + return s.upper() + + +def f_join(s, parts): + return s.join(parts) + + +def f_format(s, a, b): + return s.format(a, b) + + +def f_append(items, x): + return items.append(x) + + +def f_mixed(box, seq): + return len(seq) + box.v + + +def f_nested(x, seq): + return f_len(seq, x) * 2 + + +def f_floor(x): + return math.floor(x) + + +for i in range(3000): + assert f_len([1, 2, 3], i) == 3 + i + assert f_len("abcd", i) == 4 + i + assert f_len({1: 2}, i) == 1 + i + assert f_len(Sized(), i) == 7 + i + assert f_isinstance(i) is True + assert f_isinstance("x") is False + assert f_isinstance(True) is True + assert f_get({"a": i}, "a") == i + assert f_get({"a": i}, "b") == -1 + assert f_get({1: i}, 1.0) == i + assert f_sqrt(float(i)) == math.sqrt(i) + 1.0 + assert f_upper("ab%d" % i) == "AB%d" % i + assert f_join("-", ["a", str(i)]) == "a-%d" % i + assert f_format("{}:{}", i, 2.5) == "%d:2.5" % i + items = [] + assert f_append(items, i) is None and items == [i] + assert f_mixed(Box(i), (1, 2)) == 2 + i + assert f_nested(i, [0] * (i % 5)) == 2 * (i % 5 + i) + assert f_floor(i / 3) == i // 3 + if i % 500 == 0: + try: + f_sqrt(-1.0) + except ValueError as e: + assert str(e) == "expected a nonnegative input, got -1.0", e + else: + raise AssertionError("sqrt(-1.0) returned") + try: + f_len(5, 1) + except TypeError as e: + assert "has no len()" in str(e), e + else: + raise AssertionError("len(5) returned") + try: + f_floor(float("inf")) + except OverflowError: + pass + else: + raise AssertionError("floor(inf) returned") + try: + f_upper(5) + except AttributeError: + pass + else: + raise AssertionError("(5).upper() returned") + # A key whose comparison runs Python code takes the ordinary call. + before = Key.calls + assert f_get({Key(1): 2}, Key(1)) == 2 + assert Key.calls == before + 1 +print("ok") From 21a63f60263ae316a456fd88e56f8f68232480b9 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 02:27:11 -0700 Subject: [PATCH 34/65] perf: collect a drained generator's yields in place, and fix a leaked fold list(), tuple(), join and the other consumers that collect an iterable now drain a generator the way sum() already did: a lean resume moves each yield straight into the result buffer and resumes the generator in place, instead of leaving the core loop for every item. list(genexp) drops from about 1500 to 540 instructions per item (CPython: 740). The consumer's sink is now handed to the one resume it's for. It used to sit in an interpreter field until some lean resume took it, so when the consumer's own resume took the general path (a generator with a materialized frame), a generator resumed from inside it could claim the sink: sum() then counted that inner generator's yields too. --- crates/weavepy-vm/src/lib.rs | 111 ++++++++++++++------ tests/regrtest/test_generator_drain_fold.py | 85 +++++++++++++++ 2 files changed, 162 insertions(+), 34 deletions(-) create mode 100644 tests/regrtest/test_generator_drain_fold.py diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 62a9a0cf..d22848a0 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -926,16 +926,12 @@ pub struct Interpreter { /// chains keep the ~[`crate::gil::GIL_CHECK_INTERVAL`]-opcode /// checkpoint cadence. gil_countdown: u32, - /// `sum(generator)`'s accumulator (its address), offered to the next - /// lean generator resume (see `generator_send_lean`), which pairs it - /// with the generator's frame in [`Self::sum_fold`]. - sum_fold_acc: Option, - /// While a lean resume runs on behalf of `sum()`: the generator's - /// frame and the accumulator (addresses). The core loop folds that - /// frame's scalar yields into the accumulator and resumes in place - /// (see the `YIELD_VALUE` arm) instead of leaving the quiet loop per - /// item. - sum_fold: Option<(usize, usize)>, + /// While a lean resume runs on behalf of a consumer that drains the + /// generator (`sum()`, `list()`, see [`FoldSink`]): the generator's + /// frame (its address) and the consumer's sink. The core loop folds + /// that frame's yields into the sink and resumes in place (see the + /// `YIELD_VALUE` arm) instead of leaving the quiet loop per item. + sum_fold: Option<(usize, FoldSink)>, /// RFC 0061 (WS2b) — set while any observer (trace/profile/PEP 669 /// tool) is active on this thread. Fused-dispatch arms check it and /// fall back to single-step semantics, so instrumentation sees the @@ -1193,7 +1189,6 @@ impl Default for Interpreter { frame_stack_pool: ThreadCell::new(Vec::new()), scratch_pool: ThreadCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, - sum_fold_acc: None, sum_fold: None, fuse_off: false, recheck_frame_observed: false, @@ -1321,7 +1316,6 @@ impl Interpreter { frame_stack_pool: ThreadCell::new(Vec::new()), scratch_pool: ThreadCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, - sum_fold_acc: None, sum_fold: None, fuse_off: false, recheck_frame_observed: false, @@ -11095,7 +11089,7 @@ impl Interpreter { let saved_pc = frame.pc; frame.pc = pc as u32 + 1; let pending_self = self.lean_pending_enter(frame, shell, pc); - let result = self.generator_send_lean(&g, &Object::None); + let result = self.generator_send_lean(&g, &Object::None, None); self.lean_pending_exit(pending_self); let Some(result) = result else { frame.pc = saved_pc; @@ -13061,20 +13055,28 @@ impl Interpreter { OpCode::YieldValue => { // SAFETY: see `CoreSwitch`. if unsafe { (*sw.inl).is_empty() } { - // `sum()` driving this frame: a scalar yield folds - // into the accumulator and the frame resumes as if - // sent `None` (what `sum` does next), in place. - if let Some((ff, acc)) = self.sum_fold { + // A draining consumer driving this frame: the yield + // folds into its sink and the frame resumes as if + // sent `None` (what the consumer does next), in + // place. + if let Some((ff, sink)) = self.sum_fold { if ff == sw.cur as usize && len > 0 { - let acc = acc as *mut SumState; - // SAFETY: the running total is `do_sum_call`'s - // local, alive and untouched while the - // resume runs; `len > 0`. + // SAFETY: `len > 0`. let top = unsafe { base.add(len - 1) }; - if unsafe { (*acc).add_scalar(&*top) } { - // SAFETY: the yielded value is a scalar - // (no drop glue); the sent `None` takes - // its slot. + // SAFETY: the sink is the consumer's local, + // alive and untouched while the resume runs. + let folded = match sink { + // A scalar has no drop glue. + FoldSink::Sum(acc) => unsafe { (*acc).add_scalar(&*top) }, + FoldSink::Collect(out) => { + // The yielded value moves out. + unsafe { (*out).push(top.read()) }; + true + } + }; + if folded { + // SAFETY: the sent `None` takes the + // (moved or trivially dropped) slot. unsafe { top.write(Object::None) }; last = pc; pc += 1; @@ -30883,6 +30885,22 @@ impl Interpreter { } let it = self.make_iter(v, globals)?; let mut out = Vec::new(); + if let Object::Generator(g) = &it { + // A lean resume moves each yield straight into `out` + // (see `Interpreter::sum_fold`). + loop { + let sink = FoldSink::Collect(std::ptr::from_mut(&mut out)); + match self.generator_send_fold(g, Object::None, Some(sink)) { + Ok(x) => out.push(x), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + return Ok(out); + } + Err(e) => return Err(e), + } + } + } while let Some(x) = self.iter_next(&it, globals)? { out.push(x); } @@ -31083,10 +31101,8 @@ impl Interpreter { // A lean resume folds scalar yields straight into `total` // (see `Interpreter::sum_fold`); it returns at the first // yield it cannot fold, or when the generator ends. - self.sum_fold_acc = Some(std::ptr::from_mut(&mut total) as usize); - let sent = self.generator_send(g, Object::None); - self.sum_fold_acc = None; - let x = match sent { + let sink = FoldSink::Sum(std::ptr::from_mut(&mut total)); + let x = match self.generator_send_fold(g, Object::None, Some(sink)) { Ok(v) => v, Err(RuntimeError::PyException(exc)) if exc.type_name() == "StopIteration" => { self.fire_caught_stop_iteration(&exc)?; @@ -35999,6 +36015,7 @@ impl Interpreter { &mut self, gen: &Rc, sent: &Object, + fold: Option, ) -> Option> { let snap_gen = self.lean_snapshot()?; if self.dbg_sample @@ -36066,12 +36083,10 @@ impl Interpreter { crate::tier2::materialize_parked(frame); frame.gen_first_resume = first_resume; frame.push(sent.clone()); - // `sum()`'s accumulator, folded at this frame's yields. + // The draining consumer's sink, folded at this frame's yields. let prev_fold = std::mem::replace( &mut self.sum_fold, - self.sum_fold_acc - .take() - .map(|acc| (std::ptr::from_mut(frame) as usize, acc)), + fold.map(|sink| (std::ptr::from_mut(frame) as usize, sink)), ); let exc_depth_on_entry = self.exc_info_len(); let fin_live = gc_trace::has_any_finalizable(); @@ -36211,7 +36226,21 @@ impl Interpreter { gen: &Rc, sent: Object, ) -> Result { - if let Some(r) = self.generator_send_lean(gen, &sent) { + self.generator_send_fold(gen, sent, None) + } + + /// [`Self::generator_send`] for a consumer that drains the generator: + /// a lean resume folds the yields it can into `fold` (see + /// [`Self::sum_fold`]) and returns the first it can't. Any other + /// resume folds nothing, so a generator it resumes in turn never + /// sees the sink. + fn generator_send_fold( + &mut self, + gen: &Rc, + sent: Object, + fold: Option, + ) -> Result { + if let Some(r) = self.generator_send_lean(gen, &sent, fold) { return r; } // Take the boxed frame; it is run *in place* (RFC 0069 WS4) so @@ -56952,6 +56981,20 @@ impl CoreSwitch { /// compensated float phase taking floats and ints alike, then — once it /// is a complex — compensated real and imaginary sums, then generic `+` /// for everything after (each phase is left for good). +/// A draining consumer's sink for a lean generator resume's yields (see +/// [`Interpreter::sum_fold`]): `sum()`'s running total, which takes +/// scalars, or `list()`'s item buffer, which takes anything. +#[derive(Clone, Copy)] +enum FoldSink { + Sum(*mut SumState), + Collect(*mut Vec), +} + +// SAFETY: a sink is installed only while a resume runs on the thread that +// owns it, and cleared before that resume returns: an interpreter moved +// between threads never carries a live one. +unsafe impl Send for FoldSink {} + enum SumState { Int(i64), Float(CompensatedSum), diff --git a/tests/regrtest/test_generator_drain_fold.py b/tests/regrtest/test_generator_drain_fold.py new file mode 100644 index 00000000..6c5e1875 --- /dev/null +++ b/tests/regrtest/test_generator_drain_fold.py @@ -0,0 +1,85 @@ +"""Consumers that drain a generator (sum, list, tuple, join) see exactly its +own yields, even when it resumes other generators.""" + + +def inner(): + yield 1 + yield 2 + + +def outer(): + for x in inner(): + pass + yield 10 + + +def outer_sum(): + yield sum(inner()) + yield 5 + + +def outer_list(): + yield list(inner()) + yield [3] + + +# A materialized frame sends the outer resume down the general path; the +# inner generator's resumes must not fold into the outer consumer. +g = outer() +frame = g.gi_frame +frame.f_locals +assert sum(g) == 10 +g = outer() +frame = g.gi_frame +frame.f_locals +assert list(g) == [10] +del frame +assert sum(outer()) == 10 +assert list(outer()) == [10] +assert sum(outer_sum()) == 8 +assert list(outer_list()) == [[1, 2], [3]] + +assert list(x * 2 for x in range(5)) == [0, 2, 4, 6, 8] +assert tuple(str(x) for x in range(3)) == ("0", "1", "2") +assert "".join(chr(65 + x) for x in range(3)) == "ABC" +assert sorted(-x for x in range(4)) == [-3, -2, -1, 0] +assert list(x for x in ()) == [] + + +class Tracked: + alive = 0 + + def __init__(self): + Tracked.alive += 1 + + def __del__(self): + Tracked.alive -= 1 + + +items = list(Tracked() for _ in range(100)) +assert Tracked.alive == 100 +del items +assert Tracked.alive == 0 + + +def raises_midway(): + yield 1 + yield 2 + raise KeyError("stop") + + +try: + list(raises_midway()) +except KeyError as e: + assert e.args == ("stop",) +else: + raise AssertionError("list() swallowed the error") + + +def returns_value(): + yield 1 + return "done" + + +assert list(returns_value()) == [1] +print("ok") From a2e94c825e43d6d277957184a0c640abed1b2cd3 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 03:10:08 -0700 Subject: [PATCH 35/65] perf: remember verified leaf method sites, and run **kwargs leaves frameless A local receiver's x.m(...) call of a leaf method re-derived the same verdicts on every call: the method slot's function, its leaf kind, the warm-up gate, the call slot's shape, and the argument count. A site now records its verified callee under the receiver class's attribute version, and the next call with that version checks only the function's code, shadowing, and the recursion limit before evaluating it. Getter and small predicate calls drop by 55 to 160 instructions, and DeltaBlue by about 5%. Small functions that take **kwargs are now leaves too. A keyword call (from the core loop or compiled code) builds the fresh dictionary the function collects and evaluates the body without a frame; positional calls, which bind no dictionary, still take the ordinary call. A call like with_kwargs(i, delta=2) drops from about 4760 to 3170 instructions. --- crates/weavepy-vm/src/leaf_plan.rs | 8 +- crates/weavepy-vm/src/lib.rs | 293 ++++++++++++++++++++++-- crates/weavepy-vm/src/tier2.rs | 6 + tests/regrtest/test_leaf_varkw_calls.py | 64 ++++++ 4 files changed, 351 insertions(+), 20 deletions(-) create mode 100644 tests/regrtest/test_leaf_varkw_calls.py diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index 94dac300..a91f63a6 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -657,7 +657,7 @@ impl Builder<'_> { Some(LeafPlan { ops: self.ops.into_boxed_slice(), consts: self.consts.into_boxed_slice(), - nargs: u8::try_from(self.code.arg_count).ok()?, + nargs: u8::try_from(crate::leaf_arity(self.code)).ok()?, unique_stores: self.stored.is_some() && self.code.arg_count > 0, }) } @@ -667,7 +667,7 @@ impl Builder<'_> { /// body the plan can't express (the ordinary call runs it instead). pub(crate) fn build(code: &CodeObject, ext: &CodeConstObjects) -> Option { let nl = code.varnames.len(); - let nargs = code.arg_count as usize; + let nargs = crate::leaf_arity(code); if nl > 16 || nargs > nl || nargs > 8 || code.instructions.len() > u16::MAX as usize { return None; } @@ -1152,9 +1152,11 @@ impl Interpreter { let callee = unsafe { &*fp }; // SAFETY: GIL-serialized raw read of the code cell. let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; + // (A positional call binds no `**kwargs` dictionary.) if !crate::code_is_pure_leaf(ccode) || !Self::leaf_code_ok(ccode) - || n != ccode.arg_count as usize + || n != crate::leaf_arity(ccode) + || ccode.has_varkeywords || crate::recursion::current_depth() + usize::from(nest) + 1 >= crate::recursion::recursion_limit() { diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index d22848a0..d91d73df 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -11947,15 +11947,49 @@ impl Interpreter { && len < cap && simple_args_prefix(&code.instructions, pc + 2) { - let pure_site = match ( - other, - mslots!(cold_mslots, ext).get(pc + 1), - ) { - (Object::Instance(i), Some(ms)) => ms - .peek_fn(i.cls_raw().attr_version.get()) - .is_some_and(|fp| fn_is_leaf(&*fp)), - _ => false, - }; + // A site that verified its callee + // for this class version. + let mut missed = false; + if let (Object::Instance(i), Some(ext)) = (other, ext) { + if let Some(site) = leaf_site_hit( + ext, + pc + 1, + i.cls_raw().attr_version.get(), + ) { + match self.core_leaf_site_call( + code, + i, + other, + site, + pc + 1, + next.arg, + lbase, + nlocals, + consts, + sw.depth_cell, + ) { + SiteCall::Done(v, call_pc) => { + base.add(len).write(v); + len += 1; + last = call_pc; + pc = call_pc + 1; + continue; + } + SiteCall::Declined => {} + SiteCall::Missed => missed = true, + } + } + } + let pure_site = !missed + && match ( + other, + mslots!(cold_mslots, ext).get(pc + 1), + ) { + (Object::Instance(i), Some(ms)) => ms + .peek_fn(i.cls_raw().attr_version.get()) + .is_some_and(|fp| fn_is_leaf(&*fp)), + _ => false, + }; if pure_site { if let Some((v, call_pc)) = self.core_pure_method( code, @@ -15230,9 +15264,105 @@ impl Interpreter { self.core_pure_fused_eval(code, fp, call_pc, true, &mut args, nargs + 1, depth_cell)?; #[cfg(test)] note_literal_argument_call(code, attr_pc + 1, call_pc, true); + // Remember a call that needed no defaults, so the next one under + // this class version skips the verdicts above. + // SAFETY: as in `core_pure_fused_eval`. + let f = unsafe { &*fp }; + let callee: &Rc = unsafe { &*f.code.as_ptr() }; + let ver = cls.attr_version.get(); + if nargs + 1 == callee.arg_count as usize { + let held = mslots[attr_pc] + .get_held(ver) + .filter(|h| std::ptr::eq(Rc::as_ptr(h), fp)); + if let (Some(ext), Some(held)) = (code_vm_ext(code), held) { + let func = Rc::downgrade(&held); + leaf_site_set( + ext, + code.instructions.len(), + attr_pc, + Some(LeafSiteData { + ver, + func, + code: Rc::as_ptr(callee), + effect: !code_is_pure_leaf(callee), + }), + ); + } + } Some((r, call_pc)) } + /// [`Self::core_pure_method`] at a site that verified its callee (see + /// [`LeafSite`]) for `inst`'s class version: `fp` and `callee` are the + /// site's function and code. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_leaf_site_call( + &self, + code: &CodeObject, + inst: &PyInstance, + recv: &Object, + (fp, callee, effect): (*const crate::object::PyFunction, *const CodeObject, bool), + attr_pc: usize, + name_idx: u32, + lbase: *const Object, + nlocals: usize, + consts: &[Object], + depth_cell: *const std::cell::Cell, + ) -> SiteCall { + // SAFETY: the class holds the function at the site's version (and + // the weak handle is live); nothing here runs code. + let f = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the function's code cell. + let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; + // The ordinary call's `RecursionError` check. + // SAFETY: this thread's own depth cell. + if !std::ptr::eq(Rc::as_ptr(code_rc), callee) + || !Self::default_getattribute(inst.cls_raw()) + || inst_may_shadow(inst, code, name_idx) + || unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() + { + return SiteCall::Declined; + } + // Scalars only (small ints): nothing to drop on the way out. + let mut scratch = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + args[0] = recv; + // SAFETY: the core loop's own locals and constants. + let Some((nargs, call_pc)) = (unsafe { + Self::core_simple_args( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut scratch, + &mut args, + 1, + ) + }) else { + return SiteCall::Declined; + }; + let total = nargs + 1; + if total != code_rc.arg_count as usize { + return SiteCall::Declined; + } + let r = if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + }; + let Some(v) = r else { + if let Some(ext) = code_vm_ext(code) { + leaf_site_set(ext, code.instructions.len(), attr_pc, None); + } + return SiteCall::Missed; + }; + #[cfg(test)] + note_literal_argument_call(code, attr_pc + 1, call_pc, true); + SiteCall::Done(v, call_pc) + } + /// The core loop's fused `LOAD_GLOBAL f; PUSH_NULL; ; CALL k` (`fp` the global function), or `LOAD_GLOBAL C; /// LOAD_ATTR m (method); ...` (`fp` the class's plain or static @@ -15437,9 +15567,11 @@ impl Interpreter { { return None; } - // A pure leaf has no keyword-only parameters and no `**kwargs`. + // A pure leaf has no keyword-only parameters; its `**kwargs` + // dictionary, if any, follows the positional ones. let total = code_rc.arg_count as usize; - if total > 8 { + let arity = leaf_arity(code_rc); + if arity > 8 { return None; } let mut args: [*const Object; 8] = [std::ptr::null(); 8]; @@ -15448,9 +15580,23 @@ impl Interpreter { args[k] = o; } let kw_vals = &ops[first + eff_argc..ops.len() - 1]; + let mut varkw: Option = None; for (j, o) in kw_vals.iter().enumerate() { let slot = ((perm >> (4 * j)) & 0xF) as usize; - if slot >= total || slot == specialize::KW_TO_VARKW as usize { + if slot == specialize::KW_TO_VARKW as usize { + // Collected into a fresh dictionary, in call order (the + // bind check proved the callee has one). + varkw + .get_or_insert_with(|| { + DictData::with_capacity_and_hasher( + kw_vals.len(), + crate::fasthash::FxBuildHasher, + ) + }) + .insert(DictKey(names.get(j)?.clone()), o.clone()); + continue; + } + if slot >= total { return None; } args[slot] = o; @@ -15462,7 +15608,40 @@ impl Interpreter { .get(f.defaults.len().checked_sub(total - slot)?)?; } } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + let dict; + if code_rc.has_varkeywords { + dict = Object::Dict(Rc::new(RefCell::new(varkw.unwrap_or_default()))); + args[total] = &dict; + } + self.pure_leaf_eval::(code_rc, f, &args[..arity]) + } + + /// A keyword call's parameters, bound as `kw_names_fill_locals` leaves + /// them in `locals` (a `**kwargs` dictionary included), evaluated + /// frameless when `f` is a warm pure leaf: its result, or `None` + /// having done nothing observable. + pub(crate) fn bound_leaf_eval( + &self, + f: &crate::object::PyFunction, + locals: &[Object], + ) -> Option { + // SAFETY: GIL-serialized raw read of the function's code cell. + let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; + let arity = leaf_arity(code_rc); + if arity > 8 + || locals.len() < arity + || !code_is_pure_leaf(code_rc) + || !pure_leaf_warm(code_rc) + || !Self::leaf_code_ok(code_rc) + || crate::recursion::current_depth() >= crate::recursion::recursion_limit() + { + return None; + } + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + for (k, o) in locals[..arity].iter().enumerate() { + args[k] = o; + } + self.pure_leaf_eval::(code_rc, f, &args[..arity]) } /// Borrow a guarded instance-dictionary or slot cache hit. `names` is @@ -15552,7 +15731,9 @@ impl Interpreter { f: &crate::object::PyFunction, args: &[*const Object], ) -> Option { - if !Self::leaf_call_entry(code) { + // A positional call binds no `**kwargs` dictionary: the ordinary + // call builds it (see `leaf_arity`). + if args.len() != leaf_arity(code) || !Self::leaf_call_entry(code) { return None; } let r = self.leaf_eval::(code, f, args, 0); @@ -15569,7 +15750,7 @@ impl Interpreter { f: &crate::object::PyFunction, args: &[*const Object], ) -> Option { - if !Self::leaf_call_entry(code) { + if args.len() != leaf_arity(code) || !Self::leaf_call_entry(code) { return None; } let ext = code_vm_ext(code)?; @@ -62055,6 +62236,9 @@ struct CodeConstObjects { /// Split-layout attribute shortcuts per `LOAD_ATTR` site (see /// [`FieldSlot`]); allocated on the first recorded one. field_slots: std::sync::OnceLock>, + /// Verified frameless method calls per method-load site (see + /// [`LeafSite`]); allocated on the first recorded one. + leaf_sites: std::sync::OnceLock>, /// A leaf body's translation for the frameless evaluator (see /// [`leaf_plan`]), or `None` when the body has none. leaf_plan: std::sync::OnceLock>>, @@ -62323,6 +62507,73 @@ impl FieldSlot { } } +/// A local receiver's `x.m()` site whose last call ran +/// frameless (see `Interpreter::core_leaf_site_call`), keyed at the method +/// load: under the receiver class's (process-unique) attribute version +/// `ver`, the method is `func`, whose code was `code`, a pure (or, with +/// `effect`, an effect) leaf taking exactly the site's arguments. The +/// class holds the function at that version; `func` is weak all the same, +/// as [`MethodSlot`]'s are. +struct LeafSite(std::cell::UnsafeCell>); + +struct LeafSiteData { + ver: u64, + func: crate::sync::Weak, + code: *const CodeObject, + effect: bool, +} + +// SAFETY: as `StampSlot`. +unsafe impl Send for LeafSite {} +unsafe impl Sync for LeafSite {} + +impl LeafSite { + const fn empty() -> Self { + Self(std::cell::UnsafeCell::new(None)) + } +} + +/// The site at `pc`'s verified callee under `ver`: its function, code and +/// effect flag. +#[inline(always)] +fn leaf_site_hit( + ext: &CodeConstObjects, + pc: usize, + ver: u64, +) -> Option<(*const crate::object::PyFunction, *const CodeObject, bool)> { + // SAFETY: GIL-serialized; the borrow ends before any refill. + let site = unsafe { &*ext.leaf_sites.get()?.get(pc)?.0.get() }.as_ref()?; + (site.ver == ver && site.func.strong_count() > 0) + .then(|| (site.func.as_ptr(), site.code, site.effect)) +} + +/// Record (or, with `data` `None`, forget) the site at `pc`'s callee. +#[inline(never)] +fn leaf_site_set(ext: &CodeConstObjects, ninstrs: usize, pc: usize, data: Option) { + if data.is_none() && ext.leaf_sites.get().is_none() { + return; + } + let sites = ext + .leaf_sites + .get_or_init(|| (0..ninstrs).map(|_| LeafSite::empty()).collect()); + if let Some(site) = sites.get(pc) { + // SAFETY: GIL-serialized; no reference into the site is live. + unsafe { *site.0.get() = data }; + } +} + +/// What a verified site's call did (see +/// `Interpreter::core_leaf_site_call`). +enum SiteCall { + /// The result, and the `CALL`'s pc. + Done(Object, usize), + /// A check failed before anything ran: the ordinary paths decide. + Declined, + /// The evaluation declined (the site is forgotten): the call takes + /// the ordinary, framed path this time. + Missed, +} + /// The attribute `inst` holds for the `LOAD_ATTR` at `pc`, through the /// site's [`FieldSlot`] (never recorded for a slot or native class). /// @@ -62913,6 +63164,14 @@ fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> }) } +/// A leaf's parameter count: its positional parameters, then its +/// `**kwargs` dictionary when it has one (bound by keyword calls only; a +/// leaf has no `*args` or keyword-only parameters). +#[inline(always)] +pub(crate) fn leaf_arity(code: &CodeObject) -> usize { + code.arg_count as usize + usize::from(code.has_varkeywords) +} + fn code_is_pure_leaf(code: &CodeObject) -> bool { // The recorded verdict (see `code_pure_leaf_decide`): one load. if let Some(yes) = code.jit_hint.pure_leaf() { @@ -62941,9 +63200,8 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { && code.cellvars.is_empty() && code.freevars.is_empty() && !code.has_varargs - && !code.has_varkeywords && code.kwonly_count == 0 - && code.arg_count <= 8 + && leaf_arity(code) <= 8 && code.varnames.len() <= 16 && code.exception_table.is_empty() && code.instructions.len() <= 64; @@ -63262,6 +63520,7 @@ fn code_vm_ext_init( pure_leaf: std::sync::atomic::AtomicU8::new(0), fast_pairs: std::sync::OnceLock::new(), field_slots: std::sync::OnceLock::new(), + leaf_sites: std::sync::OnceLock::new(), leaf_plan: std::sync::OnceLock::new(), returns_none: std::sync::atomic::AtomicU8::new(0), }) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index e57d2b56..efe1ab47 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -8927,6 +8927,12 @@ unsafe fn call_dyn_impl( .then(|| unsafe { dyn_kw_site_bind(jf, ctx, &callee, argc, kwc, names) }) .flatten() { + // A pure-leaf callee evaluates frameless on the bound locals. + if let Some(v) = interp.bound_leaf_eval(&f, &locals) { + interp.recycle_scratch(locals); + // SAFETY: as above. + return unsafe { dyn_call_result(jf, ctx, Ok(v), false, false, int_result) }; + } // Not charged against the native driver: the interpreter's own // `CALL_KW` binds through this same permutation and activation, // so tier-1 would not run the call any cheaper. diff --git a/tests/regrtest/test_leaf_varkw_calls.py b/tests/regrtest/test_leaf_varkw_calls.py new file mode 100644 index 00000000..64cd9e05 --- /dev/null +++ b/tests/regrtest/test_leaf_varkw_calls.py @@ -0,0 +1,64 @@ +"""Small functions taking **kwargs see a fresh dictionary of exactly the +keywords they collect, however they are called.""" + + +def with_kwargs(a, **kw): + return a + kw.get("delta", 0) + + +def just_kw(a, **kw): + return a + + +def mk(**kw): + return kw + + +def count(**kw): + return len(kw) + + +def first(a, b=10, **kw): + return a + b + kw.get("c", 0) + + +class Box: + def get(self, **kw): + return kw.get("x", -1) + + +seen = [] +box = Box() +for i in range(3000): + assert with_kwargs(i, delta=2) == i + 2 + assert with_kwargs(i) == i + assert with_kwargs(a=i, delta=3) == i + 3 + assert just_kw(i, other=1) == i + d = mk(delta=i, other=2) + assert d == {"delta": i, "other": 2} and list(d) == ["delta", "other"] + seen.append(d) + assert mk() == {} + assert count(x=1, y=2, z=3) == 3 + assert first(i, c=5) == i + 15 + assert first(i, b=1, c=5) == i + 6 + assert box.get(x=i) == i + assert box.get() == -1 + if i % 1000 == 0: + try: + with_kwargs(i, a=1) + except TypeError as e: + assert "multiple values" in str(e), e + else: + raise AssertionError("duplicate argument accepted") + try: + with_kwargs(i, 1) + except TypeError as e: + assert "positional" in str(e), e + else: + raise AssertionError("extra positional argument accepted") + +# Every call built its own dictionary. +assert len({id(d) for d in seen}) == len(seen) +seen[0]["delta"] = "changed" +assert seen[1]["delta"] == 1 +print("ok") From 8ef7e61750292224b11766e3d2f1cc3dca62caca Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 05:00:15 -0700 Subject: [PATCH 36/65] perf: run *args, **kwargs, keyword-only and spread calls inline Calls of functions with *args, **kwargs, or keyword-only parameters, and every f(*args) / f(*args, **kwargs) spread, used to leave the core loop for the generic binder and a new recursive activation. Such calls now bind in place and run as inline activations: - Positional calls fill keyword-only parameters from their defaults, collect surplus positionals into *args (the empty tuple stays the interned one, now in the generic binder too), and give **kwargs a fresh dictionary. - CALL_FUNCTION_EX spreads a tuple into an ordinary call, and a non-empty ** mapping binds by name, as does a CALL_KW the site's cached permutation can't describe. Anything that needs an error message still takes the generic binder. - Calling an instance whose class's __call__ is a Python function becomes a call of that function with the instance first. - DICT_MERGE of plain str-keyed dictionaries runs in the leaf burst, and its error prefix is only rendered for an error (it was formatted on every merge). Forwarding wrappers get much cheaper: def f(*a): return g(*a) drops from 1300 to 680 ns, an f(*a, **k) forwarder from 2500 to 1260 ns, and a callable instance from 2400 to 1090 ns. The released operands are reaped as the full handler reaps them, so an argument's finalizer still runs when the call returns. The pickle accelerator's guards also stop re-probing every guarded function's slot store for replaced defaults unless one was ever assigned; the native entry drops from 3 to 1 microsecond, and the pickle benchmark runs about 20% faster. --- crates/weavepy-vm/src/lib.rs | 596 ++++++++++++++++++- crates/weavepy-vm/src/stdlib/pickle_accel.rs | 40 +- tests/regrtest/test_extended_param_calls.py | 227 +++++++ tests/regrtest/test_pickle_guard_changes.py | 54 ++ 4 files changed, 878 insertions(+), 39 deletions(-) create mode 100644 tests/regrtest/test_extended_param_calls.py create mode 100644 tests/regrtest/test_pickle_guard_changes.py diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index d91d73df..3cd59d36 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -10500,13 +10500,19 @@ impl Interpreter { if code.is_generator || code.is_coroutine || code.is_async_generator - || code.has_varargs - || code.has_varkeywords - || code.kwonly_count != 0 || !Self::lean_code_ok(&code) { return None; } + // Keyword-only parameters bind their compiled defaults (a positional + // call supplies none of them); a replaced `__kwdefaults__` takes the + // generic binder. + if code.kwonly_count != 0 + && ((f.defaults_maybe_overridden() && f.slot("__kwdefaults__").is_some()) + || !Self::kwonly_defaults(f, &code).all(|d| d.is_some())) + { + return None; + } f.lean_cells_ref(&code)?; let missing = if eff_argc < total { if f.defaults.len() < total - eff_argc { @@ -10522,7 +10528,8 @@ impl Interpreter { } total - eff_argc } else { - if total != eff_argc { + // Surplus positionals go to `*args`. + if total != eff_argc && !code.has_varargs { return None; } 0 @@ -10530,6 +10537,30 @@ impl Interpreter { Some((code, missing)) } + /// Whether `code` binds parameters beyond its positional ones: `*args`, + /// `**kwargs`, or keyword-only parameters. + #[inline(always)] + fn has_extended_params(code: &CodeObject) -> bool { + code.has_varargs || code.has_varkeywords || code.kwonly_count != 0 + } + + /// `f`'s compiled default for each of `code`'s keyword-only parameters, + /// in parameter order. + fn kwonly_defaults<'a>( + f: &'a PyFunction, + code: &'a CodeObject, + ) -> impl Iterator> + 'a { + let npos = code.arg_count as usize; + code.varnames[npos..npos + code.kwonly_count as usize] + .iter() + .map(|name| { + f.kw_defaults + .iter() + .find(|(n, _)| n == name.as_str()) + .map(|(_, v)| v) + }) + } + /// Move a shape-checked lean call's operands off `frame`'s stack: the /// arguments (a real self first) into fresh locals for `code`, the /// `missing` trailing parameters from the compiled defaults, the rest @@ -10572,15 +10603,19 @@ impl Interpreter { let nlocals = code.varnames.len(); let first_arg = if has_self { self_slot } else { self_slot + 1 }; v.reserve(nlocals); - v.extend(frame.stack.drain(first_arg..)); - if missing > 0 { - let Object::Function(f) = &frame.stack[callee_slot] else { - unreachable!("callee variant checked by the caller") - }; - // Defaults align right-to-left with the declared - // positionals, exactly like the generic binder's - // missing-tail fill. - v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + if Self::has_extended_params(code) { + Self::lean_call_fill_extended(frame, code, first_arg, callee_slot, missing, v); + } else { + v.extend(frame.stack.drain(first_arg..)); + if missing > 0 { + let Object::Function(f) = &frame.stack[callee_slot] else { + unreachable!("callee variant checked by the caller") + }; + // Defaults align right-to-left with the declared + // positionals, exactly like the generic binder's + // missing-tail fill. + v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + } } fill_unbound(v, nlocals); if !has_self { @@ -10592,6 +10627,49 @@ impl Interpreter { .expect("callee slot checked by the caller") } + /// [`Self::lean_call_fill`] for a callee with `*args`, `**kwargs` or + /// keyword-only parameters (see [`Self::lean_call_shape`]): the + /// positional parameters, their missing defaults, each keyword-only + /// parameter's default, the surplus positionals as the `*args` tuple, + /// and an empty `**kwargs` dictionary, in the locals' parameter order. + #[cold] + #[inline(never)] + fn lean_call_fill_extended( + frame: &mut Frame, + code: &CodeObject, + first_arg: usize, + callee_slot: usize, + missing: usize, + v: &mut Vec, + ) { + let npos = code.arg_count as usize; + let direct = (frame.stack.len() - first_arg).min(npos); + v.extend(frame.stack.drain(first_arg..first_arg + direct)); + let star = code.has_varargs.then(|| { + if frame.stack.len() == first_arg { + // The interned empty tuple. + Object::new_tuple(Vec::new()) + } else { + Object::Tuple(crate::tuple_storage::TupleStorage::from_exact_iter( + frame.stack.drain(first_arg..), + )) + } + }); + let Object::Function(f) = &frame.stack[callee_slot] else { + unreachable!("callee variant checked by the caller") + }; + if missing > 0 { + v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + } + for d in Self::kwonly_defaults(f, code) { + v.push(d.cloned().unwrap_or(Object::Unbound)); + } + v.extend(star); + if code.has_varkeywords { + v.push(Object::Dict(Rc::new(RefCell::new(DictData::default())))); + } + } + /// The inline half of `CALL`: `try_lean_call`'s plain-function shapes /// with the callee's frame boxed for the quiet loop to run in place /// (see [`InlineAct`]). `None` leaves everything untouched. @@ -10995,7 +11073,9 @@ impl Interpreter { shell: &mut QuietShell<'_>, pc: usize, ) -> Option> { - let shape = Self::lean_call_kw_shape(frame, pc)?; + let Some(shape) = Self::lean_call_kw_shape(frame, pc) else { + return self.try_inline_call_kw_named(frame, shell, pc); + }; // Past the recursion limit the nested lean path raises. let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { return None; @@ -11010,6 +11090,85 @@ impl Interpreter { Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) } + /// [`Self::try_inline_call_kw`] for a callee the site's cached + /// permutation can't describe (`*args`, for one): the keywords bind by + /// name (see [`Self::lean_bind_keywords`]) before anything is touched. + fn try_inline_call_kw_named( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + use weavepy_compiler::InlineCache as IC; + // A first execution still specializes the site. + if matches!(frame.code.caches.get(pc as u32), IC::Empty) { + return None; + } + let argc = frame.code.instructions.get(pc)?.arg as usize; + let len = frame.stack.len(); + let Some(Object::Tuple(names)) = frame.stack.last() else { + return None; + }; + let kwc = names.len(); + // Stack: callable, self-or-null, positionals, keyword values, names. + let callee_slot = len.checked_sub(kwc + argc + 3)?; + let Object::Function(f) = &frame.stack[callee_slot] else { + return None; + }; + let code = f.code(); + if code.is_generator + || code.is_coroutine + || code.is_async_generator + || !Self::lean_code_ok(&code) + || (f.defaults_maybe_overridden() + && (f.slot("__defaults__").is_some() || f.slot("__kwdefaults__").is_some())) + { + return None; + } + f.lean_cells_ref(&code)?; + let first = if matches!(frame.stack[callee_slot + 1], Object::Unbound) { + callee_slot + 2 + } else { + callee_slot + 1 + }; + let values = &frame.stack[len - 1 - kwc..len - 1]; + let bound = Self::lean_bind_keywords( + f, + &code, + frame.stack[first..len - 1 - kwc].iter(), + names.iter().zip(values), + )?; + // Past the recursion limit the nested lean path raises. + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + return None; + }; + // Committed: the operands leave the stack. + let callable = Object::Function(f.clone()); + let operands: Vec = frame.stack.drain(callee_slot..).collect(); + self.release_call_operands(operands); + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + locals.extend(bound); + fill_unbound(locals, code.varnames.len()); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) + } + + /// Release a call's operands (the callee first) once its parameters + /// are bound, as the full call handler does after it returns: a + /// temporary the collector tracks (the `**` mapping a call site + /// built, say) is reaped at once, so what it held is released with + /// the callee's own references rather than at the next collection. + fn release_call_operands(&mut self, operands: Vec) { + let mut operands = operands.into_iter(); + if let Some(callee) = operands.next() { + self.reap_call_receiver(callee); + } + let mut rest: Vec = operands.collect(); + self.reap_call_args(&mut rest); + } + /// `LOAD_ATTR` of a `property` on a plain instance, from the quiet /// loop: the getter — a plain one-argument Python function the lean /// path can run — is called with the receiver, and its result @@ -13216,6 +13375,18 @@ impl Interpreter { // A keyword call of a pure leaf through the site's cached // keyword permutation (the full handler's `CallPyKwNames` // hit), evaluated in place like `CALL`'s pure leaves. + // `f(*args)` of a Python callee switches in place (see + // `core_call_ex`); anything else takes the full handler. + OpCode::CallEx => { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_ex(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } OpCode::CallKw => { let argc = ins.arg as usize; // SAFETY: `len > 0` is checked first; the operands of @@ -13230,6 +13401,8 @@ impl Interpreter { break None; } let start = len - kwc - argc - 3; + // SAFETY: the callee and its self slot are live. + unsafe { Self::core_instance_callee(base.add(start)) }; // SAFETY: `start + kwc + argc + 3 == len`. let ops = unsafe { std::slice::from_raw_parts(base.add(start), len - start) }; @@ -13285,6 +13458,9 @@ impl Interpreter { } break Some(CoreExit::Reload); } + // SAFETY: `len >= argc + 2`: the callee and its self + // slot. + unsafe { Self::core_instance_callee(base.add(len - argc - 2)) }; // SAFETY: `len >= argc + 2`. let python = match unsafe { &*base.add(len - argc - 2) } { Object::Function(_) => 1, @@ -14073,7 +14249,9 @@ impl Interpreter { let act = self.try_inline_call(frame, shell, pc); if let (Some(act), Some((func, has_self, eff_argc))) = (&act, shape) { let code = &act.frame.code; - if let Some(slot) = code_call_slot(&frame.code, pc) { + if let Some(slot) = code_call_slot(&frame.code, pc) + .filter(|_| !Self::has_extended_params(code)) + { slot.set(CallShape { func, code: Rc::downgrade(code), @@ -14163,6 +14341,258 @@ impl Interpreter { true } + /// The core loop's `CALL_FUNCTION_EX` at `pc` of `sw`'s (synced) + /// running activation, for `f(*args)` with no `**` mapping: the tuple + /// or list spreads into an ordinary call's operands, and a plain + /// Python callee (or a bound method over one) runs as an inline + /// activation switched to here. `false` touches nothing. + #[inline(never)] + fn core_call_ex(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + self.try_inline_call_ex(&mut *frame, &mut *shell.cast::>(), pc) + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + + /// [`Self::core_call_ex`]'s activation: every check first, so a + /// decline leaves the four `CALL_FUNCTION_EX` operands untouched. + fn try_inline_call_ex( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + const MAX_SPREAD: usize = 16; + let n = frame.stack.len(); + let callee_slot = n.checked_sub(4)?; + let [callee, null, spread, mapping] = &frame.stack[callee_slot..] else { + return None; + }; + if !matches!(null, Object::Unbound) { + return None; + } + // A non-empty `**` mapping binds by name (see `lean_bind_keywords`). + let keywords = match mapping { + Object::Unbound => None, + Object::Dict(d) => Some(d), + _ => return None, + }; + if keywords.is_some_and(|d| !d.borrow().is_empty()) { + return self.try_inline_call_ex_keywords(frame, shell, pc); + } + let items: &[Object] = match spread { + Object::Tuple(t) => t, + _ => return None, + }; + if items.len() > MAX_SPREAD { + return None; + } + let (f, receiver) = match callee { + Object::Function(f) => (f, None), + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(f) => (f, Some(&bm.receiver)), + _ => return None, + }, + _ => return None, + }; + let eff_argc = items.len() + usize::from(receiver.is_some()); + let (code, missing) = Self::lean_call_shape(f, eff_argc)?; + // Past the recursion limit the nested lean path raises. + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + return None; + }; + // Committed: the operands become `callee, self-or-null, *items`. + let (f, receiver) = (f.clone(), receiver.cloned()); + // The empty `**` slot (or empty mapping). + let mapping = frame.stack.pop().expect("checked above"); + let spread = frame.stack.pop().expect("checked above"); + let Object::Tuple(items) = &spread else { + unreachable!("checked above") + }; + frame.stack.pop(); // the NULL self slot + let callee = frame.stack.pop().expect("checked above"); + let has_self = receiver.is_some(); + frame.stack.push(Object::Function(f)); + frame.stack.push(receiver.unwrap_or(Object::Unbound)); + frame.stack.extend(items.iter().cloned()); + // The spread tuple and a bound method were call temporaries (or + // are still held elsewhere): grade their releases like any + // dropped operand. + self.release_call_operands(vec![callee, spread, mapping]); + let self_slot = callee_slot + 1; + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + let callable = Self::lean_call_fill( + frame, + &code, + has_self, + self_slot, + callee_slot, + missing, + locals, + ); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) + } + + /// [`Self::try_inline_call_ex`] for `f(*args, **mapping)` with a + /// non-empty dictionary: the parameters bind by name (see + /// [`Self::lean_bind_keywords`]) before anything is touched. (The + /// operands are only borrowed until they leave the stack, so their + /// release is graded with their true owner counts.) + fn try_inline_call_ex_keywords( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + let callee_slot = frame.stack.len() - 4; + let (f, receiver) = match &frame.stack[callee_slot] { + Object::Function(f) => (f.clone(), None), + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(f) => (f.clone(), Some(bm.receiver.clone())), + _ => return None, + }, + _ => return None, + }; + let (Object::Tuple(items), Object::Dict(mapping)) = + (&frame.stack[callee_slot + 2], &frame.stack[callee_slot + 3]) + else { + return None; + }; + let code = f.code(); + if code.is_generator + || code.is_coroutine + || code.is_async_generator + || !Self::lean_code_ok(&code) + || (f.defaults_maybe_overridden() + && (f.slot("__defaults__").is_some() || f.slot("__kwdefaults__").is_some())) + { + return None; + } + f.lean_cells_ref(&code)?; + let bound = { + let kw = mapping.try_borrow().ok()?; + Self::lean_bind_keywords( + &f, + &code, + receiver.iter().chain(items.iter()), + kw.iter().map(|(k, v)| (&k.0, v)), + )? + }; + // Past the recursion limit the nested lean path raises. + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + return None; + }; + // Committed: the operands leave the stack. + let operands: Vec = frame.stack.drain(callee_slot..).collect(); + self.release_call_operands(operands); + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + locals.extend(bound); + fill_unbound(locals, code.varnames.len()); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, Object::Function(f), guard)) + } + + /// Bind `f(*positional, **kw)` (a plain dictionary) to `code`'s + /// parameters as the generic binder would: the positional + /// parameters, the keyword-only ones, then `*args` and `**kwargs` + /// when present. `None` when anything needs the generic binder: a + /// non-string key, a duplicate or unexpected keyword, a missing + /// argument, or surplus positionals without `*args`. (Replaced + /// defaults are the caller's check.) + fn lean_bind_keywords<'a>( + f: &PyFunction, + code: &CodeObject, + positional: impl Iterator, + kw: impl Iterator, + ) -> Option> { + let npos = code.arg_count as usize; + let total = npos + code.kwonly_count as usize; + let posonly = code.posonly_count as usize; + let mut slots: Vec = vec![Object::Unbound; total]; + let mut surplus: Vec = Vec::new(); + for (i, v) in positional.enumerate() { + if i < npos { + slots[i] = v.clone(); + } else if code.has_varargs { + surplus.push(v.clone()); + } else { + return None; + } + } + let mut varkw: Option = None; + for (key, v) in kw { + let Object::Str(name) = key else { + return None; + }; + match code.varnames[posonly..total] + .iter() + .position(|n| n.as_str() == &**name) + { + Some(p) => { + let slot = &mut slots[posonly + p]; + if !matches!(slot, Object::Unbound) { + return None; + } + *slot = v.clone(); + } + None if code.has_varkeywords => { + varkw + .get_or_insert_with(DictData::default) + .insert(DictKey(key.clone()), v.clone()); + } + None => return None, + } + } + let first_default = npos.checked_sub(f.defaults.len())?; + for (i, slot) in slots[..npos].iter_mut().enumerate() { + if matches!(slot, Object::Unbound) { + if i < first_default { + return None; + } + *slot = f.defaults[i - first_default].clone(); + } + } + for (slot, d) in slots[npos..].iter_mut().zip(Self::kwonly_defaults(f, code)) { + if matches!(slot, Object::Unbound) { + *slot = d?.clone(); + } + } + if code.has_varargs { + slots.push(if surplus.is_empty() { + Object::new_tuple(Vec::new()) + } else { + Object::new_tuple(surplus) + }); + } + if code.has_varkeywords { + slots.push(Object::Dict(Rc::new(RefCell::new( + varkw.unwrap_or_default(), + )))); + } + Some(slots) + } + /// The core loop's `CALL` of a plain class at `pc` of `sw`'s (synced) /// running activation: `try_lean_call`'s construction shape, with the /// `__init__` activation run inline and switched to here (the caller @@ -14933,7 +15363,10 @@ impl Interpreter { // `PyFunction::code`); only compared, then cloned below. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; let (missing, slot_self) = slot.hit(Rc::as_ptr(f), Rc::as_ptr(code_rc))?; - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self + || !Self::lean_code_ok(code_rc) + || Self::has_extended_params(code_rc) + { return None; } let missing = missing as usize; @@ -17875,6 +18308,45 @@ impl Interpreter { last = pc; pc += 1; } + // `{**a}` and `f(**kw)`'s merge of a plain dictionary whose + // keys, like the target's, are all `str`: their hashing and + // equality are native, so no Python runs. A call's merge + // that would repeat a keyword raises in the full handler. + OpCode::DictUpdate => { + let depth = (ins.arg >> 1) as usize + 1; + let n = stack.len(); + if n < depth + 1 { + break; + } + let (Object::Dict(target), Object::Dict(src)) = + (&stack[n - 1 - depth], &stack[n - 1]) + else { + break; + }; + let (Ok(mut t), Ok(s)) = (target.try_borrow_mut(), src.try_borrow()) else { + break; + }; + let str_keys = |d: &DictData| d.keys().all(|k| matches!(k.0, Object::Str(_))); + if !str_keys(&s) + || !str_keys(&t) + || (ins.arg & 1 != 0 && s.keys().any(|k| t.contains_key(k))) + { + break; + } + for (k, v) in s.iter() { + t.insert(k.clone(), v.clone()); + } + drop((t, s)); + let other = stack.pop().expect("length checked above"); + last = pc; + pc += 1; + if gc_trace::note_dropped_marks(&other) { + gc_trace::mark_maybe_dead(); + drop(other); + stop = LeafStop::Marked; + break; + } + } OpCode::Swap => { let depth = ins.arg as usize; let n = stack.len(); @@ -18797,6 +19269,62 @@ impl Interpreter { } } + /// A call of an instance whose class's `__call__` is a plain Python + /// function: the callee slot at `callee` takes that function and the + /// (empty) self slot after it the instance, as `type(obj).__call__(obj, + /// ...)`. Anything else is left alone. + /// + /// # Safety + /// + /// `callee` and `callee + 1` are a call's initialized callee and self + /// slots. + #[inline(always)] + unsafe fn core_instance_callee(callee: *mut Object) { + // SAFETY: the caller's contract. + let (Object::Instance(inst), Object::Unbound) = + (unsafe { &*callee }, unsafe { &*callee.add(1) }) + else { + return; + }; + let Some(f) = Self::instance_call_function(inst) else { + return; + }; + // SAFETY: the instance moves to the (dropless) self slot. + unsafe { + let obj = callee.read(); + callee.write(Object::Function(f)); + callee.add(1).write(obj); + } + } + + /// `type(inst).__call__` when it is a plain Python function, cached + /// on the class under its attribute version. + #[inline(never)] + fn instance_call_function(inst: &PyInstance) -> Option> { + use crate::types::LeafAttrKind as K; + /// The cache key for `__call__` (an address no interned name has). + static CALL_KEY: u8 = 0; + let cls = inst.cls_raw(); + let key = std::ptr::addr_of!(CALL_KEY) as usize; + let ver = cls.attr_version.get(); + match cls.leaf_attrs.get(key, ver) { + Some(K::Method(w)) => w.upgrade(), + Some(_) => None, + None => { + let kind = match cls.lookup("__call__") { + Some(Object::Function(f)) => K::Method(Rc::downgrade(&f)), + _ => K::Other, + }; + let found = match &kind { + K::Method(w) => w.upgrade(), + _ => None, + }; + cls.leaf_attrs.set(key, ver, kind); + found + } + } + } + fn leaf_instance_truth(&self, v: &Object) -> Option { let Object::Instance(inst) = v else { return None; @@ -23079,14 +23607,18 @@ impl Interpreter { // sits below the self-or-null slot, the args tuple and the // kwargs dict (RFC 0068 WS1's self-or-null convention // pushed it one slot deeper). - let kw_error_prefix: String = if is_kw_merge { - frame - .peek_back(depth + 2) + // (Rendered only for an error.) + let callee = if is_kw_merge { + frame.peek_back(depth + 2).cloned() + } else { + None + }; + let kw_error_prefix = || -> String { + callee + .as_ref() .and_then(callable_function_str) .map(|s| format!("{s} ")) .unwrap_or_default() - } else { - String::new() }; let dict = frame.peek_back(depth - 1).cloned().ok_or_else(|| { RuntimeError::Internal("DICT_UPDATE: stack underflow".to_owned()) @@ -23105,7 +23637,8 @@ impl Interpreter { for (k, v) in src.borrow().iter() { if is_kw_merge && t.contains_key(k) { return Err(type_error(format!( - "{kw_error_prefix}got multiple values for keyword argument '{}'", + "{}got multiple values for keyword argument '{}'", + kw_error_prefix(), k.0.to_str() ))); } @@ -23122,7 +23655,8 @@ impl Interpreter { // "test.test_extcall.h() argument after ** // must be a mapping, not list". type_error(format!( - "{kw_error_prefix}argument after ** must be a mapping, not {}", + "{}argument after ** must be a mapping, not {}", + kw_error_prefix(), other.type_name_owned() )) } else { @@ -23162,7 +23696,8 @@ impl Interpreter { let key = crate::object::DictKey(k); if is_kw_merge && t.contains_key(&key) { return Err(type_error(format!( - "{kw_error_prefix}got multiple values for keyword argument '{}'", + "{}got multiple values for keyword argument '{}'", + kw_error_prefix(), key.0.to_str() ))); } @@ -48270,9 +48805,14 @@ impl Interpreter { // Straight from the argument vector's tail into the tuple's // own allocation: collecting the remainder into a `Vec` first // was a second allocation and a second move per call. - positional[star_idx] = Object::Tuple( - crate::tuple_storage::TupleStorage::from_exact_iter(arg_iter.by_ref()), - ); + positional[star_idx] = if arg_iter.len() == 0 { + // The interned empty tuple (`() is ()`). + Object::new_tuple(Vec::new()) + } else { + Object::Tuple(crate::tuple_storage::TupleStorage::from_exact_iter( + arg_iter.by_ref(), + )) + }; filled[star_idx] = true; } else if provided > total_args { // Mirror CPython's `too_many_positional`: when the callable @@ -57838,6 +58378,7 @@ static SLOW_LEAF_OPS: [bool; 256] = { OpCode::CompareOp, OpCode::CopyFreeVars, OpCode::CopyTop, + OpCode::DictUpdate, OpCode::ForIter, OpCode::FormatValue, OpCode::GetIter, @@ -57917,6 +58458,7 @@ static CORE_LEAF_OPS: [bool; 256] = { OpCode::ToBool, OpCode::PopJumpIfNone, OpCode::PopJumpIfNotNone, + OpCode::CallEx, ]; let mut i = 0; while i < ops.len() { diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel.rs b/crates/weavepy-vm/src/stdlib/pickle_accel.rs index b05ef835..0cd949d6 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel.rs @@ -44,11 +44,30 @@ impl FunctionGuard { let Object::Function(function) = value else { return false; }; - if Rc::as_ptr(function) != self.function.as_ptr() - || Rc::as_ptr(&function.code.borrow()) != self.code.as_ptr() - { + Rc::as_ptr(function) == self.function.as_ptr() && self.holds_function(function) + } + + /// Whether the guarded function is still alive and unchanged. + fn holds_live(&self) -> bool { + if self.function.strong_count() == 0 { + return false; + } + // SAFETY: a live strong count keeps the function allocated, and + // nothing here can release it. + self.holds_function(unsafe { &*self.function.as_ptr() }) + } + + /// `function` (the guarded one) still runs the guarded code with its + /// compiled defaults. + fn holds_function(&self, function: &PyFunction) -> bool { + if Rc::as_ptr(&function.code.borrow()) != self.code.as_ptr() { return false; } + // Only an assignment to `__defaults__` or `__kwdefaults__` raises + // the flag, so the common function skips the slot probes. + if !function.defaults_maybe_overridden() { + return true; + } let slots = function.slots.borrow(); !slots.contains_key(&StrKey("__defaults__")) && !slots.contains_key(&StrKey("__kwdefaults__")) @@ -412,15 +431,12 @@ impl ClassFunctionsGuard { } fn holds(&self) -> bool { - self.class.upgrade().is_some_and(|class| { - class.attr_version.get() == self.version - && self.functions.iter().all(|guard| { - guard - .function - .upgrade() - .is_some_and(|f| guard.holds(&Object::Function(f))) - }) - }) + // An unchanged version proves the class holds the same functions; + // each one's code and defaults can still change in place. + self.class.strong_count() > 0 + // SAFETY: a live strong count keeps the class allocated. + && unsafe { &*self.class.as_ptr() }.attr_version.get() == self.version + && self.functions.iter().all(FunctionGuard::holds_live) } } diff --git a/tests/regrtest/test_extended_param_calls.py b/tests/regrtest/test_extended_param_calls.py new file mode 100644 index 00000000..e709faba --- /dev/null +++ b/tests/regrtest/test_extended_param_calls.py @@ -0,0 +1,227 @@ +"""Calls of functions with *args, **kwargs, and keyword-only parameters, +and f(*args) spreads, bind exactly as the full binder does.""" + + +def star(*args): + return args + + +def mixed(a, b=2, *rest, flag=False, **kw): + return a, b, rest, flag, kw + + +def kwonly(a, *, sep="-", end="!"): + return sep.join(a) + end + + +def required(a, *, need): + return a, need + + +def fwd(*args): + return mixed(*args) + + +def fwd_kwonly(*args): + return kwonly(*args) + + +def counter(*args): + count = 0 + for _ in args: + count += 1 + return count + + +def make_adder(n): + def add(*xs): + return n + sum(xs) + return add + + +class Box: + def method(self, *args, scale=1): + return tuple(x * scale for x in args) + + +EMPTY = () +box = Box() +add3 = make_adder(3) +dicts = [] +for i in range(2000): + assert star() == () and star() is EMPTY + assert star(i) == (i,) + assert star(i, i + 1, "x") == (i, i + 1, "x") + a, b, rest, flag, kw = mixed(i) + assert (a, b, rest, flag, kw) == (i, 2, (), False, {}) + dicts.append(kw) + assert mixed(i, 5, 6, 7) == (i, 5, (6, 7), False, {}) + assert kwonly("ab") == "a-b!" + assert fwd(i, 1, 2) == (i, 1, (2,), False, {}) + assert fwd(i) == (i, 2, (), False, {}) + assert fwd_kwonly("xyz") == "x-y-z!" + assert counter(*range(i % 7)) == i % 7 + assert counter(*(1, 2, 3)) == 3 + assert add3(i, 1) == i + 4 + assert box.method(i, 2) == (i, 2) + assert Box.method(box, i) == (i,) + spread = (i, 1) + assert mixed(*spread) == (i, 1, (), False, {}) + assert star(*[i, i]) == (i, i) + if i % 500 == 0: + try: + required("a") + except TypeError as e: + assert "missing 1 required keyword-only argument: 'need'" in str(e), e + else: + raise AssertionError("missing keyword-only argument accepted") + try: + kwonly("a", "b") + except TypeError as e: + assert "takes 1 positional argument but 2 were given" in str(e), e + else: + raise AssertionError("extra positional argument accepted") + try: + fwd() + except TypeError as e: + assert "missing 1 required positional argument: 'a'" in str(e), e + else: + raise AssertionError("missing positional argument accepted") + + +def fwd_all(*args, **kwargs): + return mixed(*args, **kwargs) + + +def posonly(a, /, b, **kw): + return a, b, kw + + +class Wrapper: + __slots__ = ("func",) + + def __init__(self, func): + self.func = func + + def __call__(self, *args, **kwargs): + return self.func(*args, **kwargs) + + +class Plain: + def __call__(self, x, y=1): + return x * y + + +wrapped = Wrapper(mixed) +plain = Plain() +for i in range(2000): + assert fwd_all(i, flag=True) == (i, 2, (), True, {}) + assert fwd_all(i, b=3, extra=4) == (i, 3, (), False, {"extra": 4}) + assert fwd_all(a=i) == (i, 2, (), False, {}) + assert fwd_all(i, 5, 6, z=1, y=2) == (i, 5, (6,), False, {"z": 1, "y": 2}) + assert list(fwd_all(i, q=1, p=2)[4]) == ["q", "p"] + assert posonly(i, b=2, a=3) == (i, 2, {"a": 3}) + assert posonly(*(i, 1), **{"a": 5}) == (i, 1, {"a": 5}) + assert wrapped(i) == (i, 2, (), False, {}) + assert wrapped(i, flag=1) == (i, 2, (), 1, {}) + assert plain(i) == i and plain(i, 3) == 3 * i and plain(i, y=2) == 2 * i + if i % 500 == 0: + try: + fwd_all(i, a=1) + except TypeError as e: + assert "multiple values for argument 'a'" in str(e), e + else: + raise AssertionError("duplicate argument accepted") + try: + kwonly(*("ab",), **{"nope": 1}) + except TypeError as e: + assert "unexpected keyword argument 'nope'" in str(e), e + else: + raise AssertionError("unexpected keyword accepted") + try: + plain() + except TypeError as e: + assert "missing 1 required positional argument: 'x'" in str(e), e + else: + raise AssertionError("missing argument accepted") + +Plain.__call__ = lambda self, x, y=1: x + y +assert plain(2, 3) == 5 +del Plain.__call__ +try: + plain(1) +except TypeError as e: + assert "not callable" in str(e), e +else: + raise AssertionError("uncallable instance called") + +# A finalizable argument whose last reference goes with a spread call's +# operands is finalized when the call returns. +FINALIZED = [] + + +class Finalized: + def __del__(self): + FINALIZED.append(1) + + +def sink(obj=None, **kw): + return None + + +def spread(*args, **kwargs): + return sink(*args, **kwargs) + + +class Keeper: + def take(self, *args, **kwargs): + return None + + +keeper = Keeper() +for i in range(200): + del FINALIZED[:] + spread(Finalized(), flag=i) + assert FINALIZED == [1], FINALIZED + spread(obj=Finalized()) + assert FINALIZED == [1, 1], FINALIZED + spread(i, other=Finalized()) + assert FINALIZED == [1, 1, 1], FINALIZED + sink(*(Finalized(),), **{"x": 1}) + assert FINALIZED == [1, 1, 1, 1], FINALIZED + # Only the `**` mapping holds it (pickle's `save_reduce(obj=obj, *rv)`). + sink(*(), obj=Finalized()) + assert FINALIZED == [1, 1, 1, 1, 1], FINALIZED + keeper.take(*(i,), obj=Finalized()) + assert FINALIZED == [1, 1, 1, 1, 1, 1], FINALIZED + +# Every call's **kwargs is its own dictionary. +assert len({id(d) for d in dicts}) == len(dicts) +dicts[0]["x"] = 1 +assert dicts[1] == {} + +# Replaced defaults take effect at once. +kwonly.__kwdefaults__ = {"sep": "+", "end": "?"} +assert kwonly("ab") == "a+b?" +kwonly.__kwdefaults__ = None +try: + kwonly("ab") +except TypeError as e: + assert "keyword-only" in str(e), e +else: + raise AssertionError("cleared __kwdefaults__ ignored") +mixed.__defaults__ = (9,) +assert mixed(1) == (1, 9, (), False, {}) + + +def deep(n, *args): + return deep(n + 1, *args) + + +try: + deep(0, 1) +except RecursionError: + pass +else: + raise AssertionError("unbounded recursion") +print("ok") diff --git a/tests/regrtest/test_pickle_guard_changes.py b/tests/regrtest/test_pickle_guard_changes.py new file mode 100644 index 00000000..698b5f82 --- /dev/null +++ b/tests/regrtest/test_pickle_guard_changes.py @@ -0,0 +1,54 @@ +"""The pickle accelerator notices in-place changes to the Python engine it +stands in for: a method's code, and a helper's defaults.""" +import pickle +import copyreg + + +class Point: + def __init__(self, x): + self.x = x + + +def check_round_trips(): + for _ in range(50): + assert pickle.loads(pickle.dumps("abc")) == "abc" + assert pickle.loads(pickle.dumps([1, 2.5, None])) == [1, 2.5, None] + assert pickle.loads(pickle.dumps(Point(3))).x == 3 + + +check_round_trips() + +# A patched method's code is what runs wherever the Python engine is the +# reference (the accelerator then stands in for it); a C pickler ignores it. +import types + +save_str = pickle._Pickler.save_str +original = save_str.__code__ +# The patched code runs with the pickle module's globals. +pickle._guard_test_save_str = types.FunctionType( + original, save_str.__globals__, "save_str_copy") + + +def patched(self, obj): + _guard_test_save_str(self, "patched" if obj == "abc" else obj) + + +expected = "patched" if issubclass(pickle.Pickler, pickle._Pickler) else "abc" +save_str.__code__ = patched.__code__ +try: + for _ in range(50): + assert pickle.loads(pickle.dumps("abc")) == expected + assert pickle.loads(pickle.dumps(["abc", "x"])) == [expected, "x"] +finally: + save_str.__code__ = original + del pickle._guard_test_save_str +check_round_trips() + +# A helper whose defaults were replaced takes the checked path; the +# results don't change. +saved = copyreg.__newobj__.__defaults__ +copyreg.__newobj__.__defaults__ = None +check_round_trips() +copyreg.__newobj__.__defaults__ = saved +check_round_trips() +print("ok") From b14deb198f2fd35dc80e83c075d328b8ac429f3f Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:44:15 -0700 Subject: [PATCH 37/65] perf: fuse len() of native-length instances, and memoize pickled slot checks The core loop's fused LOAD_GLOBAL len; LOAD_FAST x; CALL 1 now also serves an instance whose class's __len__ is a registered native body (a deque's), skipping the separate push, call, and pop: len(q) on a deque drops from about 1110 to 620 instructions. The pickle decoder's first pass verified every instance's __slots__ names against the class MRO, once per instance. It now remembers the members it verified for the stream, so a list of slotted instances checks each name once (records loads: 4% fewer instructions). --- crates/weavepy-vm/src/lib.rs | 9 ++++++ crates/weavepy-vm/src/stdlib/pickle_accel.rs | 32 +++++++++++++++++--- 2 files changed, 37 insertions(+), 4 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 3cd59d36..14726d33 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -13729,6 +13729,15 @@ impl Interpreter { Object::Dict(d) => { d.try_borrow().ok().map(|d| d.len()) } + // A native `__len__` (a deque's). + v @ Object::Instance(_) => { + match self.leaf_instance_len(v) { + Some(Object::Int(n)) => { + usize::try_from(n).ok() + } + _ => None, + } + } _ => None, }; if let Some(n) = n.and_then(|n| i64::try_from(n).ok()) { diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel.rs b/crates/weavepy-vm/src/stdlib/pickle_accel.rs index 0cd949d6..84d01a36 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel.rs @@ -944,6 +944,10 @@ struct Probe<'a, 'c, const RECORD_OPCODES: bool> { builds: Vec<(StateNodes, Option)>, context: &'c dyn Fn() -> Option, resolved: Option, + /// The `__slots__` members already verified in this stream (see + /// `classes::is_member_slot`): every instance of a class names the + /// same ones. + member_slots: Vec<(*const TypeObject, &'a str)>, } impl<'a, 'c, const RECORD_OPCODES: bool> Probe<'a, 'c, RECORD_OPCODES> { @@ -955,9 +959,28 @@ impl<'a, 'c, const RECORD_OPCODES: bool> Probe<'a, 'c, RECORD_OPCODES> { builds: Vec::new(), context, resolved: None, + member_slots: Vec::new(), } } + /// [`classes::is_member_slot`], remembered for this stream. (No Python + /// runs during the probe, so the class can't change in between.) + fn is_member_slot(&mut self, class: &Rc, name: &'a str) -> bool { + let key = Rc::as_ptr(class); + if self + .member_slots + .iter() + .any(|&(c, n)| c == key && n == name) + { + return true; + } + let yes = classes::is_member_slot(class, name); + if yes && self.member_slots.len() < 64 { + self.member_slots.push((key, name)); + } + yes + } + /// Count the references to each node that the result keeps: the root, /// container items, and the state pair's items when the pair itself is /// kept. BUILD copies its state, so its edge keeps nothing. @@ -1259,10 +1282,11 @@ impl<'a, const RECORD_OPCODES: bool> Sink<'a> for Probe<'a, '_, RECORD_OPCODES> } if let Some(slots) = nodes.slots { // `setattr` must reach a member descriptor of `__slots__`. - if !self - .state_keys(slots)? - .iter() - .all(|name| classes::is_member_slot(&class.class, name)) + let names = self.state_keys(slots)?.to_vec(); + let class = class.class.clone(); + if !names + .into_iter() + .all(|name| self.is_member_slot(&class, name)) { return None; } From c7a2621f8b7fcb185003ef77928a856cb79445bf Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:44:15 -0700 Subject: [PATCH 38/65] tools: add a profile-guided release build tools/pgo_build.py builds an instrumented CLI, trains it on the benchmark fixtures at the harness's work sizes (JIT on and off), the bundled regression suite, and a stdlib import sweep, then rebuilds the release binary with the merged profile, as CPython's release builds do. On the measured macOS host this cuts wall time on most fixtures by 15% to 30% (attr_access 1.24x to 0.84x of CPython, call_overhead 1.10x to 0.87x, float_math 1.27x to 1.01x). Training at toy work sizes left large-integer multiplication cold and slowed pidigits by 38%, hence the harness sizes. --- tools/pgo_build.py | 139 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 139 insertions(+) create mode 100644 tools/pgo_build.py diff --git a/tools/pgo_build.py b/tools/pgo_build.py new file mode 100644 index 00000000..ea6da597 --- /dev/null +++ b/tools/pgo_build.py @@ -0,0 +1,139 @@ +#!/usr/bin/env python3 +"""Build a profile-guided (PGO) release of the `weavepy` CLI. + +CPython's release builds are profile guided (`--enable-optimizations` +trains on the regression suite), and so is this build: it compiles an +instrumented interpreter, runs a training workload, merges the profile, +and rebuilds the release binary with it. + +The training workload is the benchmark fixtures at the benchmark harness's +work sizes (with the JIT on and off), the bundled regression suite, and a +stdlib import sweep. Work sizes matter: code the training never reaches is +laid out as cold, so a fixture run at a toy size (large-integer +multiplication in `pidigits`, for one) gets slower, not faster. + +Requirements: the `llvm-tools` rustup component (`rustup component add +llvm-tools`), which provides the `llvm-profdata` matching the compiler. + +Usage (from the repository root): + + python3 tools/pgo_build.py + python3 tools/pgo_build.py --skip-regrtest # fixtures and imports only + +The optimized binary is written to `target/pgo/optimized/release/weavepy` +(with the runtime library beside it); `--install` also copies it over +`target/release/weavepy`. Intermediate files stay under `target/pgo/`. +""" + +import argparse +import os +import re +import shutil +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +PGO = ROOT / "target" / "pgo" +FIXTURES = ROOT / "crates" / "weavepy-bench" / "fixtures" +FIXTURES_RS = ROOT / "crates" / "weavepy-bench" / "src" / "fixtures.rs" +PACKAGES = ["-p", "weavepy-cli", "-p", "weavepy-pylib"] +IMPORTS = ( + "import argparse, ast, asyncio, collections, csv, dataclasses, datetime, " + "decimal, email, enum, fractions, functools, heapq, inspect, itertools, " + "json, logging, pathlib, pickle, random, re, statistics, string, " + "textwrap, typing, unittest, urllib.parse" +) + + +def run(cmd, env=None, cwd=ROOT, check=True): + print("+", " ".join(str(c) for c in cmd), flush=True) + return subprocess.run(cmd, env=env, cwd=cwd, check=check) + + +def llvm_profdata(): + sysroot = subprocess.run( + ["rustc", "--print", "sysroot"], capture_output=True, text=True, check=True + ).stdout.strip() + host = re.search( + r"^host: (\S+)$", + subprocess.run(["rustc", "-vV"], capture_output=True, text=True, check=True).stdout, + re.M, + ).group(1) + tool = Path(sysroot) / "lib" / "rustlib" / host / "bin" / "llvm-profdata" + if os.name == "nt": + tool = tool.with_suffix(".exe") + if not tool.exists(): + sys.exit(f"{tool} not found: run `rustup component add llvm-tools` first") + return tool + + +def build(target_dir, rustflags): + env = dict(os.environ) + env["CARGO_TARGET_DIR"] = str(target_dir) + env["RUSTFLAGS"] = (env.get("RUSTFLAGS", "") + " " + rustflags).strip() + run(["cargo", "build", "--release", *PACKAGES], env=env) + exe = target_dir / "release" / ("weavepy.exe" if os.name == "nt" else "weavepy") + if not exe.exists(): + sys.exit(f"build produced no {exe}") + return exe + + +def fixture_work(): + """The benchmark harness's `(fixture, work)` pairs, read from its source.""" + text = FIXTURES_RS.read_text() + return re.findall(r'"(\w+)" => ([\d_]+),', text) + + +def train(exe, skip_regrtest): + for name, work in fixture_work(): + path = FIXTURES / f"{name}.py" + if not path.exists(): + continue + for jit in ("1", "0"): + env = dict(os.environ, WEAVEPY_BENCH_WORK=work.replace("_", ""), WEAVEPY_JIT=jit) + run([exe, path], env=env, check=False) + for jit in ("1", "0"): + run([exe, "-c", IMPORTS], env=dict(os.environ, WEAVEPY_JIT=jit), check=False) + if not skip_regrtest: + report = PGO / "regrtest" + run( + [exe, "regrtest", "--mode", "subprocess", "--workers", "0", + "--timeout", "600", "-q", "--no-check", "--report-dir", report], + check=False, + ) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("--skip-regrtest", action="store_true", + help="train on the fixtures and imports only") + parser.add_argument("--install", action="store_true", + help="also copy the optimized binary to target/release/weavepy") + args = parser.parse_args() + + profdata_tool = llvm_profdata() + raw = PGO / "raw" + shutil.rmtree(raw, ignore_errors=True) + raw.mkdir(parents=True) + + instrumented = build(PGO / "instrumented", f"-Cprofile-generate={raw}") + train(instrumented, args.skip_regrtest) + + merged = PGO / "weavepy.profdata" + profiles = sorted(raw.glob("*.profraw")) + if not profiles: + sys.exit("training produced no profiles") + run([profdata_tool, "merge", "-o", merged, *profiles]) + + optimized = build(PGO / "optimized", f"-Cprofile-use={merged}") + print(f"optimized binary: {optimized}") + if args.install: + dest = ROOT / "target" / "release" / optimized.name + dest.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(optimized, dest) + print(f"installed: {dest}") + + +if __name__ == "__main__": + main() From 81d6c6de81b8b0b19f9e9f14f995fdc8a1311b98 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:03:42 -0700 Subject: [PATCH 39/65] perf: match leaf plan ops in place, and hash collector tables by address The frameless evaluator copied each plan op and decoded all of its operands before dispatching on it, keeping the copy's fields and the instruction index in stack slots: about 21 instructions per op. Matching the op in place lets each arm load only its own operands, and dispatch drops to 5 instructions. The evaluator's scratch buffers also stop calling their out-of-line drops when they are empty (the usual case). The collector's temporary id sets, its suspect map, and the descriptor registry hashed object addresses with SipHash. They now use the address mixer the collector's main index already uses (these tables never hold untrusted keys), and so does the prompt reaper's cascade marker set. Measured with callgrind: DeltaBlue retires 5.6% fewer instructions, and the pickle benchmark 1.9% fewer. --- crates/weavepy-vm/src/descr_registry.rs | 25 +++++++------ crates/weavepy-vm/src/fasthash.rs | 5 +++ crates/weavepy-vm/src/gc_trace.rs | 48 +++++++++++++------------ crates/weavepy-vm/src/leaf_plan.rs | 27 ++++++++++++-- crates/weavepy-vm/src/vm_singletons.rs | 4 +-- 5 files changed, 73 insertions(+), 36 deletions(-) diff --git a/crates/weavepy-vm/src/descr_registry.rs b/crates/weavepy-vm/src/descr_registry.rs index 5dcc5f62..a68cef0e 100644 --- a/crates/weavepy-vm/src/descr_registry.rs +++ b/crates/weavepy-vm/src/descr_registry.rs @@ -25,7 +25,12 @@ //! failed — test_interpreters TestInterpreterCall after a prior test's //! interpreter was torn down). -use std::collections::{HashMap, HashSet}; +/// Address-keyed tables: object ids are aligned addresses, hashed with the +/// cheap address mixer rather than SipHash (they carry no untrusted input). +type HashMap = + std::collections::HashMap>; +type HashSet = + std::collections::HashSet>; use std::sync::LazyLock; use crate::object::Object; @@ -63,13 +68,13 @@ pub struct DescrMeta { /// `inspect.getattr_static` (and so `import traceback`, via `_colorize`'s /// dataclasses) in any sub-interpreter created off the main thread. static DESCR_META: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Every pointer key any table below has ever been given (and not yet /// forgotten). `Drop` consults this one set first, so the common case — /// a builtin no table knows — costs a single read-locked hash probe. static TAGGED: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); fn note_key(k: usize) { TAGGED.write().insert(k); @@ -129,7 +134,7 @@ struct StoredMeta { /// `multiprocessing.Queue` feeder thread. The `Rc` pointer key is stable for /// the process lifetime and the value is `&'static str`, so sharing is sound. static BUILTIN_MODULE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attribute `obj` (a native builtin function) to module `module`, so its /// `__module__` reports that instead of the default `"builtins"`. @@ -174,7 +179,7 @@ pub fn module_of_builtin(b: &Rc) -> Option<&'static st /// thread through the module cache, and its descriptors may be read from /// any of them. static NATIVE_DESCR_ACCESSOR: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// Type-dict entries that exist for *introspection only* (RFC 0056 WS4): /// CPython materializes every slot wrapper in `tp_dict` (`'__lt__' in @@ -190,7 +195,7 @@ static NATIVE_DESCR_ACCESSOR: LazyLock>> = /// PROCESS-GLOBAL for the same reason as [`BUILTIN_MODULE`]: the type /// singletons and their dict entries are shared across threads. static SURFACE_ONLY: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// The default-allocator `__new__` builtins (`make_default_new` / /// `make_owned_new`), by identity. Several *real* constructing builtins @@ -198,7 +203,7 @@ static SURFACE_ONLY: LazyLock>> = /// sequences…), so the instantiation path cannot key on the name alone /// now that the allocators are stored as raw builtins. static DEFAULT_NEW: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// Tag `obj` as a default-allocator `__new__` (see [`is_default_new`]). pub fn mark_default_new(obj: &Object) { @@ -272,7 +277,7 @@ pub fn is_native_descr_accessor(b: &Rc) -> bool { /// so `pickle.dumps` failed with "it's not found as /// `_multiarray_umath._reconstruct`" (RFC 0076 WS2). static BUILTIN_WRITABLE_MODULE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Record a runtime `__module__` assignment on a builtin function. /// Returns `false` if `obj` is not a taggable representation. @@ -355,7 +360,7 @@ pub fn lookup(obj: &Object) -> Option { /// name-keyed table in `builtin_text_signature` can't reach. /// PROCESS-GLOBAL for the same reason as [`DESCR_META`]. static TEXT_SIGNATURE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attach an Argument-Clinic `__text_signature__` string to `obj`. pub fn register_text_signature(obj: &Object, sig: &'static str) { @@ -397,7 +402,7 @@ pub type LiveDocReader = unsafe fn(usize) -> Option; /// through the shared module cache. The addresses point into the /// extension's method tables, which live for the process lifetime. static LIVE_C_DOC: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attach a live C-doc reader to `obj` (a bridged method descriptor). pub fn register_live_c_doc(obj: &Object, addr: usize, read: LiveDocReader) { diff --git a/crates/weavepy-vm/src/fasthash.rs b/crates/weavepy-vm/src/fasthash.rs index 7153727b..ccb23a57 100644 --- a/crates/weavepy-vm/src/fasthash.rs +++ b/crates/weavepy-vm/src/fasthash.rs @@ -124,6 +124,11 @@ impl Hasher for ObjectIdHasher { fn write_u64(&mut self, word: u64) { self.0.write_u64(word); } + + #[inline] + fn write_usize(&mut self, word: usize) { + self.0.write_usize(word); + } } /// `BuildHasher` for [`FxHasher`] — usable as the `S` parameter of diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 5d97a2ff..53c4b1c8 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -83,6 +83,9 @@ use std::sync::atomic::{ use crate::object::Object; use crate::weakref_registry::{id_of, ObjectId}; +/// A set of object ids (addresses), hashed with the address mixer. +type IdSet = std::collections::HashSet>; + type GcIndex = std::collections::HashMap< ObjectId, HandleRc, @@ -1088,7 +1091,7 @@ impl GcState { fn collect_child_candidates(&self, obj: &Object, work: &mut Vec) { let mut pending: Vec = Vec::new(); traverse_object(obj, &mut |child| pending.push(child.clone())); - let mut seen: std::collections::HashSet = std::collections::HashSet::new(); + let mut seen: IdSet = IdSet::default(); while let Some(child) = pending.pop() { let id = id_of(&child); if !seen.insert(id) { @@ -2210,19 +2213,17 @@ impl GcState { // (test_callbacks_on_callback: `c.wr`/`d.wr` stay silent while the // external `safe_callback` fires). Snapshot the trash ids so the // queue loops below can drop callbacks belonging to trash wrappers. - let mut trash_ids: std::collections::HashSet = - unreachable.iter().map(|h| h.id).collect(); - let wrapper_is_trash = - |slot: &crate::sync::Rc, - trash: &std::collections::HashSet| { - slot.py_ref - .borrow() - .as_ref() - .and_then(crate::sync::Weak::upgrade) - .is_none_or(|inst| { - trash.contains(&(crate::sync::Rc::as_ptr(&inst) as usize as u64)) - }) - }; + let mut trash_ids: IdSet = unreachable.iter().map(|h| h.id).collect(); + let wrapper_is_trash = |slot: &crate::sync::Rc, + trash: &IdSet| { + slot.py_ref + .borrow() + .as_ref() + .and_then(crate::sync::Weak::upgrade) + .is_none_or(|inst| { + trash.contains(&(crate::sync::Rc::as_ptr(&inst) as usize as u64)) + }) + }; if weakref_only { let mut weakref_callbacks = Vec::new(); @@ -2475,9 +2476,9 @@ impl GcState { // callbacks and recursing into its children. Finalizable orphans are // left for a finalizing collection so `__del__` ordering is preserved. if !saveall { - let dead_ids: std::collections::HashSet = dead.iter().map(|h| h.id).collect(); + let dead_ids: IdSet = dead.iter().map(|h| h.id).collect(); let mut worklist = cascade_seed; - let mut seen: std::collections::HashSet = std::collections::HashSet::new(); + let mut seen: IdSet = IdSet::default(); while let Some(cid) = worklist.pop() { if dead_ids.contains(&cid) || by_id.contains_key(&cid) || !seen.insert(cid) { // Dead (already reaped), a candidate this collection owns, @@ -3350,13 +3351,16 @@ pub fn zombie_memoryview_refs_to(target: ObjectId) -> usize { if handles.is_empty() { return 0; } - let mut zombies: std::collections::HashSet = std::collections::HashSet::new(); + let mut zombies: IdSet = IdSet::default(); loop { // Inbound references each candidate receives from the current // zombie set (a dropped chain of sub-views keeps inner views' // counts up via exporter edges). - let mut inbound: std::collections::HashMap = - std::collections::HashMap::new(); + let mut inbound: std::collections::HashMap< + ObjectId, + usize, + BuildHasherDefault, + > = Default::default(); for h in &handles { if zombies.contains(&h.id) { traverse_object(&h.object, &mut |c| { @@ -3907,9 +3911,9 @@ struct Suspect { budget: u8, dormant_probes: u8, } -type SuspectMap = indexmap::IndexMap; +type SuspectMap = indexmap::IndexMap>; static SUSPECTS: std::sync::LazyLock> = - std::sync::LazyLock::new(|| parking_lot::Mutex::new(SuspectMap::new())); + std::sync::LazyLock::new(|| parking_lot::Mutex::new(SuspectMap::default())); static SUSPECT_COUNT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); /// Entries with probe budget remaining. When only dormant entries are /// left, [`has_suspects`] admits a sweep every [`DORMANT_STRIDE`]-th @@ -4637,7 +4641,7 @@ mod tests { #[test] fn suspect_eviction_preserves_minimum_order_counts_and_release() { for mode in 0..6 { - let mut suspects = SuspectMap::new(); + let mut suspects = SuspectMap::default(); let mut reference = Vec::new(); let mut handles = std::collections::HashMap::new(); for index in 0..SUSPECT_CAP { diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index a91f63a6..befc6f54 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -696,7 +696,18 @@ struct Owned { } impl Drop for Owned { + // Usually nothing is held: no call for the empty case. + #[inline(always)] fn drop(&mut self) { + if self.n != 0 { + self.release(); + } + } +} + +impl Owned { + #[inline(never)] + fn release(&mut self) { for k in 0..self.n { // SAFETY: the first `n` entries are initialized. crate::drop_hot(unsafe { self.buf[k].assume_init_read() }); @@ -762,7 +773,18 @@ impl Pending { } impl Drop for Pending { + // Usually nothing is buffered: no call for the empty case. + #[inline(always)] fn drop(&mut self) { + if self.n != 0 { + self.release(); + } + } +} + +impl Pending { + #[inline(never)] + fn release(&mut self) { for k in 0..self.n { // SAFETY: the first `n` entries are initialized; a moved value // is left in place, not dropped again. @@ -863,9 +885,10 @@ impl Interpreter { loop { // SAFETY: every path ends in a return or a jump, and every // jump target is an op (checked when the plan was built). - let op = unsafe { *ops.get_unchecked(ip) }; + // Matched in place: each arm loads only its own operands. + let op: &Op = unsafe { ops.get_unchecked(ip) }; ip += 1; - match op { + match *op { Op::Move { dst, src } => set!(dst, get!(src)), // SAFETY: `k` names a plan constant (the translation). Op::Const { dst, k } => set!(dst, unsafe { *consts.get_unchecked(usize::from(k)) }), diff --git a/crates/weavepy-vm/src/vm_singletons.rs b/crates/weavepy-vm/src/vm_singletons.rs index e74ab8c8..4ef06da4 100644 --- a/crates/weavepy-vm/src/vm_singletons.rs +++ b/crates/weavepy-vm/src/vm_singletons.rs @@ -818,8 +818,8 @@ thread_local! { /// A queue request for an id in this set is the cascade observing /// itself and is dropped; requests for *other* objects (a child body /// whose last C pin fell during the teardown) still queue normally. - static CASCADING_IDS: RefCell> = - RefCell::new(std::collections::HashSet::new()); + static CASCADING_IDS: RefCell> = + RefCell::new(crate::fasthash::FxHashSet::default()); } /// RAII marker for one object's trip through the prompt reaper's cascade; From dbd4ec8c2f12ad251fa919a210a7609c64587890 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:08:43 -0700 Subject: [PATCH 40/65] perf: forward f(*args, **kwargs) without building the merged dict A forwarding wrapper's f(*args, **kwargs) compiles to BUILD_MAP 0, LOAD_FAST kwargs, DICT_MERGE and CALL_FUNCTION_EX: a fresh dict, tracked by the collector, filled by copying kwargs, only to be read by the callee's binder. When the callee is a Python function the core loop now binds straight from the local mapping and skips the copy; anything else runs the instructions as before, untouched. The keyword binder also writes the parameters into the activation's locals directly (it staged them in a temporary vector and copied them over), sizes a **kwargs dictionary up front, and releases the call's operands in place instead of collecting them first. A callable instance forwarding w(5, protocol=5) through __call__(self, *args, **kwargs) runs 30% fewer instructions. Also make the shared-string thread handoff test revoke the reference count bias before it spawns threads, as the VM does: under the biased counts its raw threads raced with the owner's plain increments, and the test failed depending on test order. --- crates/weavepy-vm/src/lib.rs | 197 +++++++++++++++++++------- crates/weavepy-vm/src/shared_value.rs | 2 + 2 files changed, 149 insertions(+), 50 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 14726d33..a8404568 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -11132,24 +11132,28 @@ impl Interpreter { callee_slot + 1 }; let values = &frame.stack[len - 1 - kwc..len - 1]; + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; let bound = Self::lean_bind_keywords( f, &code, frame.stack[first..len - 1 - kwc].iter(), names.iter().zip(values), - )?; + locals, + ); // Past the recursion limit the nested lean path raises. - let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { - return None; + let guard = match (bound, crate::recursion::enter()) { + (Some(()), crate::recursion::Enter::Ok(guard)) => guard, + _ => { + locals.clear(); + self.inline_unslot(act); + return None; + } }; // Committed: the operands leave the stack. let callable = Object::Function(f.clone()); - let operands: Vec = frame.stack.drain(callee_slot..).collect(); - self.release_call_operands(operands); - let act = self.inline_slot(); - // SAFETY: a parked slot's locals storage is its own, and empty. - let locals = unsafe { &mut *act.frame.locals.as_ptr() }; - locals.extend(bound); + self.release_call_operands(&mut frame.stack, callee_slot); fill_unbound(locals, code.varnames.len()); frame.pc = pc as u32 + 1; Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) @@ -11160,13 +11164,19 @@ impl Interpreter { /// temporary the collector tracks (the `**` mapping a call site /// built, say) is reaped at once, so what it held is released with /// the callee's own references rather than at the next collection. - fn release_call_operands(&mut self, operands: Vec) { - let mut operands = operands.into_iter(); - if let Some(callee) = operands.next() { - self.reap_call_receiver(callee); + fn release_call_operands(&mut self, stack: &mut Vec, callee_slot: usize) { + let callee = std::mem::replace(&mut stack[callee_slot], Object::Unbound); + self.reap_call_receiver(callee); + self.reap_call_args(&mut stack[callee_slot + 1..]); + stack.truncate(callee_slot); + } + + /// Return a parked activation slot [`Self::inline_slot`] handed out but + /// the call didn't use (its locals must be empty again). + fn inline_unslot(&mut self, act: Box) { + if self.inline_pool.len() < INLINE_POOL_CAP { + self.inline_pool.push(act); } - let mut rest: Vec = operands.collect(); - self.reap_call_args(&mut rest); } /// `LOAD_ATTR` of a `property` on a plain instance, from the quiet @@ -13375,6 +13385,19 @@ impl Interpreter { // A keyword call of a pure leaf through the site's cached // keyword permutation (the full handler's `CallPyKwNames` // hit), evaluated in place like `CALL`'s pure leaves. + // `f(*args, **kwargs)` forwarding binds straight from the + // local mapping (see `core_call_forward`); a decline + // builds the dict as usual. + OpCode::BuildMap if ins.arg == 0 && forward_call_shape(code, pc) => { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_forward(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } // `f(*args)` of a Python callee switches in place (see // `core_call_ex`); anything else takes the full handler. OpCode::CallEx => { @@ -14379,6 +14402,50 @@ impl Interpreter { true } + /// The core loop's `BUILD_MAP 0` at `pc` opening a forwarding call, + /// `f(*args, **kwargs)` (see [`forward_call_shape`]): the fresh + /// merged dictionary would only be read by the call's binder, so the + /// local mapping itself stands in as the call's `**` operand and the + /// `CALL_FUNCTION_EX` three instructions on runs inline. `false` + /// touches nothing (the instructions run one by one). + #[inline(never)] + fn core_call_forward(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + let frame = &mut *frame; + let local = frame.code.instructions.get(pc + 1).map(|i| i.arg as usize); + // SAFETY: GIL-serialized read of the running activation's locals. + let locals: &Vec = &*frame.locals.as_ptr(); + let mapping = match local.and_then(|i| locals.get(i)) { + Some(Object::Dict(d)) => Object::Dict(d.clone()), + _ => return false, + }; + frame.stack.push(mapping); + let act = self.try_inline_call_ex(frame, &mut *shell.cast::>(), pc + 3); + if act.is_none() { + // Declined untouched: the operand leaves again. + frame.stack.pop(); + } + act + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + /// [`Self::core_call_ex`]'s activation: every check first, so a /// decline leaves the four `CALL_FUNCTION_EX` operands untouched. fn try_inline_call_ex( @@ -14443,7 +14510,8 @@ impl Interpreter { // The spread tuple and a bound method were call temporaries (or // are still held elsewhere): grade their releases like any // dropped operand. - self.release_call_operands(vec![callee, spread, mapping]); + self.reap_call_receiver(callee); + self.reap_call_args(&mut [spread, mapping]); let self_slot = callee_slot + 1; let act = self.inline_slot(); // SAFETY: a parked slot's locals storage is its own, and empty. @@ -14497,26 +14565,29 @@ impl Interpreter { return None; } f.lean_cells_ref(&code)?; - let bound = { - let kw = mapping.try_borrow().ok()?; - Self::lean_bind_keywords( - &f, - &code, - receiver.iter().chain(items.iter()), - kw.iter().map(|(k, v)| (&k.0, v)), - )? - }; - // Past the recursion limit the nested lean path raises. - let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { - return None; - }; - // Committed: the operands leave the stack. - let operands: Vec = frame.stack.drain(callee_slot..).collect(); - self.release_call_operands(operands); + let kw = mapping.try_borrow().ok()?; let act = self.inline_slot(); // SAFETY: a parked slot's locals storage is its own, and empty. let locals = unsafe { &mut *act.frame.locals.as_ptr() }; - locals.extend(bound); + let bound = Self::lean_bind_keywords( + &f, + &code, + receiver.iter().chain(items.iter()), + kw.iter().map(|(k, v)| (&k.0, v)), + locals, + ); + drop(kw); + // Past the recursion limit the nested lean path raises. + let guard = match (bound, crate::recursion::enter()) { + (Some(()), crate::recursion::Enter::Ok(guard)) => guard, + _ => { + locals.clear(); + self.inline_unslot(act); + return None; + } + }; + // Committed: the operands leave the stack. + self.release_call_operands(&mut frame.stack, callee_slot); fill_unbound(locals, code.varnames.len()); frame.pc = pc as u32 + 1; Some(self.inline_bind(frame, shell, pc, act, code, Object::Function(f), guard)) @@ -14534,15 +14605,21 @@ impl Interpreter { code: &CodeObject, positional: impl Iterator, kw: impl Iterator, - ) -> Option> { + out: &mut Vec, + ) -> Option<()> { let npos = code.arg_count as usize; let total = npos + code.kwonly_count as usize; let posonly = code.posonly_count as usize; - let mut slots: Vec = vec![Object::Unbound; total]; + debug_assert!(out.is_empty()); + fill_unbound(out, total); + // (Every slot written below still holds `Unbound`, which owns + // nothing: it is overwritten without drop glue.) + let bind = + |slot: &mut Object, v: &Object| std::mem::forget(std::mem::replace(slot, v.clone())); let mut surplus: Vec = Vec::new(); for (i, v) in positional.enumerate() { if i < npos { - slots[i] = v.clone(); + bind(&mut out[i], v); } else if code.has_varargs { surplus.push(v.clone()); } else { @@ -14550,6 +14627,7 @@ impl Interpreter { } } let mut varkw: Option = None; + let kw_hint = kw.size_hint().0; for (key, v) in kw { let Object::Str(name) = key else { return None; @@ -14559,47 +14637,51 @@ impl Interpreter { .position(|n| n.as_str() == &**name) { Some(p) => { - let slot = &mut slots[posonly + p]; + let slot = &mut out[posonly + p]; if !matches!(slot, Object::Unbound) { return None; } - *slot = v.clone(); + bind(slot, v); } None if code.has_varkeywords => { varkw - .get_or_insert_with(DictData::default) + .get_or_insert_with(|| { + DictData::with_capacity_and_hasher( + kw_hint.max(1), + crate::fasthash::FxBuildHasher, + ) + }) .insert(DictKey(key.clone()), v.clone()); } None => return None, } } let first_default = npos.checked_sub(f.defaults.len())?; - for (i, slot) in slots[..npos].iter_mut().enumerate() { + for (i, slot) in out[..npos].iter_mut().enumerate() { if matches!(slot, Object::Unbound) { if i < first_default { return None; } - *slot = f.defaults[i - first_default].clone(); + bind(slot, &f.defaults[i - first_default]); } } - for (slot, d) in slots[npos..].iter_mut().zip(Self::kwonly_defaults(f, code)) { + for (slot, d) in out[npos..total] + .iter_mut() + .zip(Self::kwonly_defaults(f, code)) + { if matches!(slot, Object::Unbound) { - *slot = d?.clone(); + bind(slot, d?); } } if code.has_varargs { - slots.push(if surplus.is_empty() { - Object::new_tuple(Vec::new()) - } else { - Object::new_tuple(surplus) - }); + out.push(Object::new_tuple(surplus)); } if code.has_varkeywords { - slots.push(Object::Dict(Rc::new(RefCell::new( + out.push(Object::Dict(Rc::new(RefCell::new( varkw.unwrap_or_default(), )))); } - Some(slots) + Some(()) } /// The core loop's `CALL` of a plain class at `pc` of `sw`'s (synced) @@ -63940,6 +64022,21 @@ fn code_is_effect_leaf(code: &CodeObject) -> bool { code.jit_hint.effect_leaf() } +/// Whether the `BUILD_MAP 0` at `pc` opens `f(*args, **kwargs)`'s mapping: +/// `LOAD_FAST kwargs; DICT_MERGE 1; CALL_FUNCTION_EX` follow. +#[inline(always)] +fn forward_call_shape(code: &CodeObject, pc: usize) -> bool { + let ops = &code.instructions; + matches!( + (ops.get(pc + 1), ops.get(pc + 2), ops.get(pc + 3)), + (Some(load), Some(merge), Some(call)) + if matches!(load.op, OpCode::LoadFast | OpCode::LoadFastBorrow) + && merge.op == OpCode::DictUpdate + && merge.arg == 1 + && call.op == OpCode::CallEx + ) +} + /// Whether the instructions from `pc` could open a simple call's /// argument run (see `Interpreter::core_simple_args`): each of the first /// two is a plain operand load or the `CALL` itself. A cheap filter for diff --git a/crates/weavepy-vm/src/shared_value.rs b/crates/weavepy-vm/src/shared_value.rs index 7ffec9a4..78b01224 100644 --- a/crates/weavepy-vm/src/shared_value.rs +++ b/crates/weavepy-vm/src/shared_value.rs @@ -673,6 +673,8 @@ mod tests { fn weak_upgrades_and_shared_text_survive_thread_handoffs() { let value = SharedStr::from("shared 🧶 text"); let weak = SharedStr::downgrade(&value); + // As the VM does before starting a thread: counts go atomic. + crate::sync::revoke_bias_for_spawn(); std::thread::scope(|scope| { for _ in 0..2 { let weak = weak.clone(); From 838dcb187b05e66e6a9eef7e35d99e466d5b2765 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:27:36 -0700 Subject: [PATCH 41/65] perf: read a deque's slots in one pass, and grade instance subscripts once The deque natives resolved each of _data, _head, _maxlen and _state by a separate hinted name lookup, several per operation. SlotStorage now offers leading/leading_mut, which match the first N slots against interned names by thin pointer in one pass, and the deque's fast paths and iterator step work on that view. The core loop's container subscript no longer grades an instance operand it then declines (the native subscript grades it), shared instances skip the full drop grade, and native method calls stage operands straight into their slice. deque_ops: about 13% fewer instructions. --- crates/weavepy-vm/src/lib.rs | 88 ++++++-- .../src/stdlib/collections_native.rs | 189 +++++++++++------- crates/weavepy-vm/src/types.rs | 89 ++++++++- 3 files changed, 270 insertions(+), 96 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index a8404568..de945e8c 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -14989,7 +14989,11 @@ impl Interpreter { } // SAFETY: `len >= 2`. let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; - if !Self::core_droppable(c) { + // (An instance goes to the caller's native subscript, which + // grades its own release.) + if !matches!(c, Object::List(_) | Object::Tuple(_) | Object::Dict(_) | Object::Str(_)) + || !Self::core_droppable(c) + { return None; } let r = match (c, k) { @@ -15288,7 +15292,11 @@ impl Interpreter { // hit must not reject this nonfinal release. The last // owner still takes ordinary teardown, and every store or // weakref operation that requires tracking revokes the flag. - Rc::strong_count(i) > 1 && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) + // Past two owners only a weakref-watched object can be at + // its dead line (see `gc_trace::note_dropped_marks`). + let sc = Rc::strong_count(i); + sc > 2 && !crate::weakref_registry::may_have_weakrefs(Rc::as_ptr(i) as usize as u64) + || sc > 1 && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) } Object::List(l) if Rc::strong_count(l) > 1 => !gc_trace::note_dropped_marks(v), Object::Dict(d) if Rc::strong_count(d) > 1 => !gc_trace::note_dropped_marks(v), @@ -15571,6 +15579,57 @@ impl Interpreter { } } + /// [`Self::core_simple_args`] for a native callee, which takes its + /// operands as a slice: bitwise copies of the arguments (after the + /// receiver in `ops[0]`), never dropped, so no reference moves. + /// Returns the argument count and the `CALL`'s pc. + /// + /// # Safety + /// + /// As [`Self::core_simple_args`]. + #[inline(always)] + unsafe fn core_simple_ops( + instrs: &[weavepy_compiler::Instruction], + mut pc: usize, + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ops: &mut [std::mem::MaybeUninit; 8], + ) -> Option<(usize, usize)> { + let mut n = 1; + loop { + let ins = *instrs.get(pc)?; + // (Checked before any copy is taken: a borrowed copy must never + // be dropped.) + if n == 8 && ins.op != OpCode::Call { + return None; + } + let v = match ins.op { + OpCode::LoadFast => { + let i = ins.arg as usize; + if i >= nlocals { + return None; + } + // SAFETY: `i < nlocals` (see the function docs). + let p = unsafe { lbase.add(i) }; + if matches!(unsafe { &*p }, Object::Unbound) { + return None; + } + // SAFETY: a borrowed copy of a live local. + unsafe { std::ptr::read(p) } + } + // SAFETY: a borrowed copy of a live constant. + OpCode::LoadConst => unsafe { std::ptr::read(consts.get(ins.arg as usize)?) }, + OpCode::LoadSmallInt => Object::Int(i64::from(ins.arg)), + OpCode::Call if ins.arg as usize == n - 1 => return Some((n - 1, pc)), + _ => return None, + }; + ops[n].write(v); + n += 1; + pc += 1; + } + } + /// The native fast `__getitem__` the `BINARY_SUBSCR` at `pc` cached for /// `ops[0]`'s class (see [`Self::leaf_instance_subscript`]), when both /// operands leave by plain decrements. @@ -15632,20 +15691,21 @@ impl Interpreter { if !Self::default_getattribute(cls) { return None; } - let mut scratch = [const { std::mem::MaybeUninit::::uninit() }; 8]; - let mut args: [*const Object; 8] = [std::ptr::null(); 8]; - args[0] = recv; + // Bitwise views of the operands, never dropped: the callee reads + // them (and clones what it keeps), and nothing it runs can reach + // the locals or constants they alias. + let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; + // SAFETY: the receiver is the core loop's local. + ops[0].write(unsafe { std::ptr::read(recv) }); // SAFETY: the core loop's own locals and constants. let (nargs, call_pc) = unsafe { - Self::core_simple_args( + Self::core_simple_ops( &code.instructions, attr_pc + 1, lbase, nlocals, consts, - &mut scratch, - &mut args, - 1, + &mut ops, ) }?; let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; @@ -15657,16 +15717,8 @@ impl Interpreter { { return None; } - // Bitwise views of the operands, never dropped: the callee reads - // them (and clones what it keeps), and nothing it runs can reach - // the locals or constants they alias. let n = nargs + 1; - let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; - for (slot, &p) in ops.iter_mut().zip(&args[..n]) { - // SAFETY: each pointer names a live operand (see above). - slot.write(unsafe { std::ptr::read(p) }); - } - // SAFETY: the first `n` entries were just written. + // SAFETY: the first `n` entries were written. let ops = unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), n) }; let r = match kind { LeafKind::Fast(f) => f(ops)?, diff --git a/crates/weavepy-vm/src/stdlib/collections_native.rs b/crates/weavepy-vm/src/stdlib/collections_native.rs index 4ba49718..e811d625 100644 --- a/crates/weavepy-vm/src/stdlib/collections_native.rs +++ b/crates/weavepy-vm/src/stdlib/collections_native.rs @@ -148,14 +148,16 @@ impl DequeState<'_> { // unguarded (`peek_mut` returns `None` if any guard is live) and neither // reference outlives the native call, which runs no Python. #[allow(clippy::mut_from_ref)] -fn fast_parts(args: &[Object]) -> Option<(&mut crate::types::SlotStorage, &mut Vec)> { +fn fast_parts(args: &[Object]) -> Option> { let Object::Instance(inst) = args.first()? else { return None; }; // SAFETY: see above — no guard is live on either cell (`peek_mut` // checks), and neither reference outlives the native call. let slots = unsafe { inst.slots.peek_mut() }?; - let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, &names().data) else { + let n = names(); + let [data, head, maxlen, state] = slots.leading_mut([&n.data, &n.head, &n.maxlen, &n.state])?; + let Object::List(data) = data else { return None; }; // The list lives in its own allocation, held by the `_data` slot, @@ -163,30 +165,77 @@ fn fast_parts(args: &[Object]) -> Option<(&mut crate::types::SlotStorage, &mut V let data: *const RefCell> = Rc::as_ptr(data); // SAFETY: as above. let d = unsafe { (*data).peek_mut() }?; - Some((slots, d)) + Some(Fast { + d, + head, + maxlen, + state, + }) } -/// [`popleft_locked`] over the unguarded parts (see [`fast_parts`]). -fn popleft_fast( - slots: &mut crate::types::SlotStorage, - d: &mut Vec, -) -> Result { - let mut h = head_of(slots); - if h >= d.len() { - return Err(index_error("pop from an empty deque")); +/// A deque's backing list and its `_head`, `_maxlen` and `_state` slots, +/// borrowed unguarded (see [`fast_parts`]) and found in one pass over +/// the slot store. +struct Fast<'a> { + d: &'a mut Vec, + head: &'a mut Object, + maxlen: &'a Object, + state: &'a mut Object, +} + +impl Fast<'_> { + #[inline] + fn head(&self) -> usize { + match *self.head { + Object::Int(h) if h >= 0 => h as usize, + _ => 0, + } } - bump_state_of(slots); - let x = std::mem::replace(&mut d[h], Object::None); - h += 1; - if h >= d.len() { - d.clear(); - h = 0; - } else if h >= 32 && h * 2 >= d.len() { - d.drain(..h); - h = 0; + + #[inline] + fn set_head(&mut self, h: usize) { + match &mut *self.head { + Object::Int(slot) => *slot = h as i64, + slot => *slot = Object::Int(h as i64), + } + } + + #[inline] + fn maxlen(&self) -> Option { + match *self.maxlen { + Object::Int(m) if m >= 0 => Some(m as usize), + _ => None, + } + } + + #[inline] + fn bump_state(&mut self) { + match &mut *self.state { + Object::Int(s) => *s = s.wrapping_add(1), + slot => *slot = Object::Int(1), + } + } + + /// [`popleft_locked`] over the unguarded parts. + fn popleft(&mut self) -> Result { + let mut h = self.head(); + let d = &mut *self.d; + if h >= d.len() { + return Err(index_error("pop from an empty deque")); + } + let x = std::mem::replace(&mut d[h], Object::None); + h += 1; + if h >= d.len() { + d.clear(); + h = 0; + } else if h >= 32 && h * 2 >= d.len() { + d.drain(..h); + h = 0; + } + self.bump_state(); + self.set_head(h); + Ok(x) } - set_head_of(slots, h); - Ok(x) } fn receiver<'a>(args: &'a [Object], method: &str) -> Result, RuntimeError> { @@ -229,11 +278,11 @@ fn popleft_locked(st: &mut DequeState<'_>, d: &mut Vec) -> Result Result { if let [_, x] = args { - if let Some((slots, d)) = fast_parts(args) { - bump_state_of(slots); - d.push(x.clone()); - let trimmed = match maxlen_of(slots) { - Some(m) if d.len() - head_of(slots) > m => Some(popleft_fast(slots, d)?), + if let Some(mut f) = fast_parts(args) { + f.bump_state(); + f.d.push(x.clone()); + let trimmed = match f.maxlen() { + Some(m) if f.d.len() - f.head() > m => Some(f.popleft()?), _ => None, }; if trimmed.is_some() { @@ -280,18 +329,18 @@ fn deque_append(args: &[Object]) -> Result { fn deque_appendleft(args: &[Object]) -> Result { if let [_, x] = args { - if let Some((slots, d)) = fast_parts(args) { - bump_state_of(slots); - let mut h = head_of(slots).min(d.len()); + if let Some(mut f) = fast_parts(args) { + f.bump_state(); + let mut h = f.head().min(f.d.len()); if h == 0 { - h = std::cmp::max(8, d.len() / 2); - d.splice(0..0, std::iter::repeat_n(Object::None, h)); + h = std::cmp::max(8, f.d.len() / 2); + f.d.splice(0..0, std::iter::repeat_n(Object::None, h)); } h -= 1; - d[h] = x.clone(); - set_head_of(slots, h); - let trimmed = match maxlen_of(slots) { - Some(m) if d.len() - h > m => d.pop(), + f.d[h] = x.clone(); + f.set_head(h); + let trimmed = match f.maxlen() { + Some(m) if f.d.len() - h > m => f.d.pop(), _ => None, }; if trimmed.is_some() { @@ -339,16 +388,16 @@ fn deque_appendleft(args: &[Object]) -> Result { fn deque_pop(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - let h = head_of(slots); - if d.len() > h { - bump_state_of(slots); - let x = d.pop().expect("len checked"); + if let Some(mut f) = fast_parts(args) { + let h = f.head(); + if f.d.len() > h { + f.bump_state(); + let x = f.d.pop().expect("len checked"); // An emptied deque keeps a short free prefix for the next // `appendleft` instead of splicing a new one. - if d.len() == h && h > 32 { - d.clear(); - set_head_of(slots, 0); + if f.d.len() == h && h > 32 { + f.d.clear(); + f.set_head(0); } return Ok(x); } @@ -378,9 +427,9 @@ fn deque_pop(args: &[Object]) -> Result { fn deque_popleft(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - if head_of(slots) < d.len() { - return popleft_fast(slots, d); + if let Some(mut f) = fast_parts(args) { + if f.head() < f.d.len() { + return f.popleft(); } } } @@ -398,8 +447,8 @@ fn deque_popleft(args: &[Object]) -> Result { fn deque_len(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - return Ok(Object::Int(d.len().saturating_sub(head_of(slots)) as i64)); + if let Some(f) = fast_parts(args) { + return Ok(Object::Int(f.d.len().saturating_sub(f.head()) as i64)); } } let st = receiver(args, "__len__")?; @@ -412,8 +461,8 @@ fn deque_len(args: &[Object]) -> Result { fn deque_bool(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - return Ok(Object::Bool(d.len() > head_of(slots))); + if let Some(f) = fast_parts(args) { + return Ok(Object::Bool(f.d.len() > f.head())); } } let st = receiver(args, "__bool__")?; @@ -426,12 +475,12 @@ fn deque_bool(args: &[Object]) -> Result { fn deque_getitem(args: &[Object]) -> Result { if let [_, Object::Int(i)] = args { - if let Some((slots, d)) = fast_parts(args) { - let h = head_of(slots); - let n = d.len().saturating_sub(h) as i64; + if let Some(f) = fast_parts(args) { + let h = f.head(); + let n = f.d.len().saturating_sub(h) as i64; let index = if *i < 0 { *i + n } else { *i }; if (0..n).contains(&index) { - return Ok(d[h + index as usize].clone()); + return Ok(f.d[h + index as usize].clone()); } } } @@ -542,33 +591,33 @@ fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option // Python runs before the last use, and the iterator and its deque are // distinct objects. let its = unsafe { iterator.slots.peek_mut() }?; - let Some(Object::Instance(deque)) = its.get_hinted(IT_DEQ, &names().deq) else { + let n = names(); + let [deque, index, it_state] = its.leading_mut([&n.deq, &n.index, &n.deq_state])?; + let Object::Instance(deque) = deque else { return None; }; // The deque lives while the iterator's slot holds it (unchanged here). let deque: *const crate::types::PyInstance = Rc::as_ptr(deque); - let index = its - .get_hinted(IT_INDEX, &names().index) - .and_then(Object::as_i64)?; - let it_state = its - .get_hinted(IT_STATE, &names().deq_state) - .and_then(Object::as_i64); + let Object::Int(i) = index else { + return None; + }; // SAFETY: as above. let ds = unsafe { (*deque).slots.peek() }?; - let Some(Object::List(data)) = ds.get_hinted(SLOT_DATA, &names().data) else { + let [data, head, _, state] = ds.leading([&n.data, &n.head, &n.maxlen, &n.state])?; + let Object::List(data) = data else { return None; }; - if ds - .get_hinted(SLOT_STATE, &names().state) - .and_then(Object::as_i64) - != it_state - { + if state.as_i64() != it_state.as_i64() { return None; } - let h = head_of(ds); + let h = match *head { + Object::Int(h) if h >= 0 => h as usize, + _ => 0, + }; // SAFETY: as above. let d = unsafe { data.peek() }?; let h = h.min(d.len()); + let index = *i; if index < 0 || index as usize >= d.len() - h { return None; } @@ -578,7 +627,7 @@ fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option h + index as usize }; let v = d[slot].clone(); - *its.get_hinted_mut(IT_INDEX, &names().index)? = Object::Int(index + 1); + *i = index + 1; Some(v) } diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index 5b5eb8ff..e961e7a9 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -1991,6 +1991,15 @@ enum SlotData { }, } +/// Is `key` the slot name `name`? (Interned names usually share the +/// probe's storage, settled without reading either length.) +#[inline(always)] +fn key_named(key: &DictKey, name: &crate::shared_value::SharedStr) -> bool { + matches!(&key.0, Object::Str(stored) + if crate::shared_value::SharedStr::ptr_eq(stored, name) + || slot_name_eq(stored.as_ref(), name.as_ref())) +} + #[cfg(target_pointer_width = "64")] /// `stored == name` for a slot name, without the call into `memcmp`. /// Slot names are short (`_data`, `_head`, `x`) and almost always differ @@ -2241,15 +2250,79 @@ impl SlotStorage { /// `get_mut(name)` with a position hint (see [`Self::get_hinted`]). #[inline] pub fn get_hinted_mut(&mut self, idx: usize, name: &str) -> Option<&mut Object> { - let at_hint = matches!( - self.get_index(idx), - Some((DictKey(Object::Str(stored)), _)) - if std::ptr::eq(stored.as_ptr(), name.as_ptr()) || slot_name_eq(stored.as_ref(), name) - ); - if at_hint { - return self.get_index_mut(idx).map(|(_, v)| v); + // One lookup at the hint. (The pointer ends the borrow, so the + // name scan below can take its own.) + let hit = match self.get_index_mut(idx) { + Some((DictKey(Object::Str(stored)), value)) + if std::ptr::eq(stored.as_ptr(), name.as_ptr()) + || slot_name_eq(stored.as_ref(), name) => + { + Some(std::ptr::from_mut(value)) + } + _ => None, + }; + match hit { + // SAFETY: the value lives in `self`, borrowed mutably for the + // returned lifetime; no other reference to it exists. + Some(value) => Some(unsafe { &mut *value }), + None => self.get_mut(name), } - self.get_mut(name) + } + + /// The values of the first `N` slots when they are named `names`, in + /// order (the layout of an instance whose `__init__` assigns them in + /// that order): one pass for a native method that reads several + /// slots. `None` when the store is laid out otherwise. + #[inline] + pub fn leading( + &self, + names: [&crate::shared_value::SharedStr; N], + ) -> Option<[&Object; N]> { + match &self.data { + SlotData::Small(entries) => { + let entries = entries.get(..N)?; + if !entries.iter().zip(names).all(|((key, _), name)| key_named(key, name)) { + return None; + } + Some(std::array::from_fn(|i| &entries[i].1)) + } + SlotData::Fixed { layout, values } => { + let (keys, values) = (layout.get(..N)?, values.get(..N)?); + if !keys.iter().zip(names).all(|(key, name)| key_named(key, name)) { + return None; + } + Some(std::array::from_fn(|i| &values[i])) + } + _ => None, + } + } + + /// [`Self::leading`], mutably. + #[inline] + pub fn leading_mut( + &mut self, + names: [&crate::shared_value::SharedStr; N], + ) -> Option<[&mut Object; N]> { + let values: &mut [Object] = match &mut self.data { + SlotData::Small(entries) => { + let entries = entries.get_mut(..N)?; + if !entries.iter().zip(names).all(|((key, _), name)| key_named(key, name)) { + return None; + } + let mut values = entries.iter_mut().map(|(_, value)| value); + return Some(std::array::from_fn(|_| values.next().expect("N entries"))); + } + SlotData::Fixed { layout, values } => { + let keys = layout.get(..N)?; + if !keys.iter().zip(names).all(|(key, name)| key_named(key, name)) { + return None; + } + values.get_mut(..N)? + } + _ => return None, + }; + let mut values = values.iter_mut(); + Some(std::array::from_fn(|_| values.next().expect("N values"))) } /// Mutable access to a populated slot's value, by name. From ed8486a10bb4dfd8db995bd20b7249e7e4826715 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 11:35:13 -0700 Subject: [PATCH 42/65] perf: retire frameless direct JIT callees that keep calling the interpreter A compiled function entered directly from the interpreter (try_call_native_direct) never had its interpreter round-trips counted: only framed entries and native-to-native callees did. deltablue's incremental_add ran every constraint.satisfy call through an activation shell and a framed interpreter call, several times the cost of the interpreter's inline call. Direct entries now feed the same callee round-trip backoff. deltablue (JIT on): 3% fewer instructions. --- crates/weavepy-vm/src/tier2.rs | 32 +++++++++++++++++--------------- 1 file changed, 17 insertions(+), 15 deletions(-) diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index efe1ab47..1866d6de 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -4571,7 +4571,7 @@ unsafe fn try_native_call( } } }; - note_callee_exit(nc, nctx); + note_callee_exit(&nc.art, &nc.code, nctx); if !inline_bufs { put_u64(u64_buf); put_u32(u32_buf); @@ -5456,30 +5456,27 @@ unsafe extern "C" fn wpjit_self_slow( } } -/// The generic-call backoff for native-to-native entries (the framed -/// entries' twin lives in [`note_native_exit`]): a compiled callee whose +/// The generic-call backoff for native-to-native and frameless direct +/// entries (the framed entries' twin lives in [`note_native_exit`]): a +/// compiled callee whose /// activations average [`CALLEE_ROUNDTRIP_RETIRE_RATIO`] or more /// interpreter calls is a thin native driver around them. Each such call pays pin /// traffic, an activation shell and a generic call that the interpreter's /// inline call path avoids, so the callee retires to tier-1. #[inline] -fn note_callee_exit(nc: &NativeCallee, child: &CallCtx) { - let entries = nc.art.callee_entries.get().saturating_add(1); - nc.art.callee_entries.set(entries); +fn note_callee_exit(art: &Artifacts, code: &Rc, child: &CallCtx) { + let entries = art.callee_entries.get().saturating_add(1); + art.callee_entries.set(entries); if child.dyn_py_calls == 0 { return; } - let trips = nc - .art - .callee_roundtrips - .get() - .saturating_add(child.dyn_py_calls); - nc.art.callee_roundtrips.set(trips); + let trips = art.callee_roundtrips.get().saturating_add(child.dyn_py_calls); + art.callee_roundtrips.set(trips); if entries >= GENERIC_RETIRE_MIN_ENTRIES && trips / entries >= CALLEE_ROUNDTRIP_RETIRE_RATIO - && !nc.code.jit_hint.is_not_jitable() + && !code.jit_hint.is_not_jitable() { - let key = Rc::as_ptr(&nc.code).cast::(); + let key = Rc::as_ptr(code).cast::(); JIT.with(|cell| { let mut st = cell.borrow_mut(); if let Some(ce) = st.cache.get_mut(&key) { @@ -5487,7 +5484,7 @@ fn note_callee_exit(nc: &NativeCallee, child: &CallCtx) { } st.stats.generic_retires += 1; }); - nc.code.jit_hint.mark_not_jitable(); + code.jit_hint.mark_not_jitable(); } } @@ -10214,6 +10211,11 @@ pub(crate) fn try_call_native_direct( }; native_stat(|s| s.direct_calls.set(s.direct_calls.get() + 1)); + // A direct callee that keeps calling back into the interpreter is + // retired like a native-to-native one (deltablue's `incremental_add` + // ran each `satisfy` through an activation shell and a framed call, + // 6% of the benchmark's instructions). + note_callee_exit(&entry.art, code, &ctx); let out = match status { JitStatus::Returned => Ok(unpack_pins(jf.ret_bits, jf.ret_tag, &ctx.pins)), From 0e11c8c9ef6602c25f838a2a46adb31e5c43ae54 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 12:20:49 -0700 Subject: [PATCH 43/65] perf: cheapen the core loop's frame switch and generator resume - The reload prologue no longer caches the globals and builtins handles (only LOAD_GLOBAL and LOAD_NAME read them, off the frame), so every call, return, resume and yield spills four fewer values. - VmExt caches the extension table's thin address, so code_vm_ext is one load instead of a fat dyn pointer and its alignment arithmetic. - An inline generator resume writes its slot's fields without drop glue, the yield parks the frame over a Running state without it, and the yield's consumer protocol runs without a QuietEntry round trip. Frame::push inlines. A for loop over a generator: 7% fewer instructions per item. --- crates/weavepy-compiler/src/lib.rs | 10 ++- crates/weavepy-vm/src/lib.rs | 128 +++++++++++++++++------------ 2 files changed, 84 insertions(+), 54 deletions(-) diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index 71bba165..0d9148a9 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -156,8 +156,16 @@ pub use weavepy_parser::ast::expr_name; /// clones (a `replace()`d code object may change `constants`, so a /// cloned code object starts with an empty slot), never participates in /// equality, and is not serialized. +/// +/// The second field caches the payload's address once it's set (the VM's +/// hot accessor then reads one thin pointer instead of the fat `dyn` +/// handle and its alignment arithmetic). The `Arc` in the first field +/// keeps that address alive for the code object's lifetime. #[derive(Default)] -pub struct VmExt(pub std::sync::OnceLock>); +pub struct VmExt( + pub std::sync::OnceLock>, + pub std::sync::atomic::AtomicPtr<()>, +); impl Clone for VmExt { fn clone(&self) -> Self { diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index de945e8c..246cd0a1 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -263,6 +263,7 @@ impl std::fmt::Debug for Frame { } impl Frame { + #[inline] fn push(&mut self, v: Object) { self.stack.push(v); } @@ -9922,16 +9923,26 @@ impl Interpreter { drop(frame.stack.drain(n - 3..)); } let mut act = self.inline_slot(); - act.gen = Some(g); - act.gen_box = Some(boxed); + // A pooled slot holds none of these (every path back to the pool + // takes them), so they're written without drop glue. + debug_assert!( + act.gen.is_none() + && act.gen_box.is_none() + && act.guard.is_none() + && act.act.shell.is_none() + ); + // SAFETY: each field is `None` (above), so nothing is leaked. + unsafe { + std::ptr::write(&raw mut act.gen, Some(g)); + std::ptr::write(&raw mut act.gen_box, Some(boxed)); + std::ptr::write(&raw mut act.guard, Some(guard)); + } act.gen_frame = gen_frame; act.exhaust_arg = arg; act.act.frame = gen_frame; - act.act.shell = None; act.call_pc = pc; act.caller_pending = self.lean_pending_enter(frame, shell, pc); act.exc_depth = self.exc_info_len(); - act.guard = Some(guard); Some(act) } @@ -11969,11 +11980,8 @@ impl Interpreter { // during an activation, and nothing here runs Python code). let locals: &mut Vec = unsafe { &mut *frame.locals.as_ptr() }; let (lbase, nlocals) = (locals.as_mut_ptr(), locals.len()); - let (gdict, bdict) = (frame.globals.as_ptr(), frame.builtins.as_ptr()); - let (gid, bid) = ( - specialize::rc_id(&frame.globals), - specialize::rc_id(&frame.builtins), - ); + // (The namespaces are read off the frame at their uses: holding + // them here costs every switch a spill.) // The arms' colder per-activation state (the global and class // attribute stamps, the method slots) is derived at its first // use: every call, return and helper handoff runs this prologue @@ -13327,11 +13335,11 @@ impl Interpreter { }; // SAFETY (raw dict reads): as in the `LOAD_GLOBAL` arm. let v = unsafe { - match (*gdict).get_index_of(&probe) { - Some(i) => (*gdict).get_index(i), - None => (*bdict) + match (*frame.globals.as_ptr()).get_index_of(&probe) { + Some(i) => (*frame.globals.as_ptr()).get_index(i), + None => (*frame.builtins.as_ptr()) .get_index_of(&probe) - .and_then(|i| (*bdict).get_index(i)), + .and_then(|i| (*frame.builtins.as_ptr()).get_index(i)), } }; // A miss (the full handler's `NameError`) or a key the @@ -13703,6 +13711,8 @@ impl Interpreter { // SAFETY (raw dict reads): as in `leaf_global` — no dict // borrow is held while bytecode runs, and nothing here // runs code. + let (gdict, bdict) = (frame.globals.as_ptr(), frame.builtins.as_ptr()); + let gid = specialize::rc_id(&frame.globals); let g_stamp = unsafe { (*gdict).mutation_stamp() }; let hit = match code.caches.get(pc as u32) { IC::LoadGlobalModule { @@ -13714,7 +13724,7 @@ impl Interpreter { IC::LoadGlobalBuiltin { builtins_id, key_idx, - } if builtins_id == bid + } if builtins_id == specialize::rc_id(&frame.builtins) && !self.globals_missing_any.get() && slot.get() == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => @@ -16598,7 +16608,8 @@ impl Interpreter { drop(gen); let call_pc = done.call_pc; self.lean_pending_exit(done.caller_pending); - done.act.shell = None; + // SAFETY: the shell was `None` (checked above): no drop glue owed. + unsafe { std::ptr::write(&raw mut done.act.shell, None) }; done.act.frame = std::ptr::from_mut::(&mut done.frame); done.caller_pending = None; if self.inline_pool.len() < INLINE_POOL_CAP { @@ -16609,7 +16620,6 @@ impl Interpreter { let (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; // SAFETY: as above. unsafe { (*cframe).stack.push(v) }; - let entry = QuietEntry::Returned { cur_pc: call_pc }; sw.cur = cframe; sw.last = if clast == &raw mut sw.scratch { sw.scratch = usize::MAX; @@ -16617,36 +16627,26 @@ impl Interpreter { } else { clast }; - match entry { - QuietEntry::Returned { cur_pc } => { - // SAFETY: as above. - unsafe { *sw.last = cur_pc }; - // The consumer's `Returned` entry protocol (`quiet_frame`). - if self.gil_countdown <= 2 { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - return true; - } - self.gil_countdown -= 1; - if crate::hot_gates::loop_gen() != snap_gen { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - return true; - } - // SAFETY: see `quiet_run`. - if sw.fin && unsafe { (*sw.maybe_dead).get() } { - // SAFETY: as above. - unsafe { - self.flush_lean(&mut *cframe, &mut *cshell.cast::>()); - } - if self.drain_if_maybe_dead() && crate::hot_gates::loop_gen() != snap_gen { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - } - } - } - QuietEntry::Raised { err, .. } => { - sw.pending = Some(CoreExit::Stop(LeafStop::Raised(err))); + // The consumer's `QuietEntry::Returned` protocol (`quiet_frame`). + // SAFETY: as above. + unsafe { *sw.last = call_pc }; + if self.gil_countdown <= 2 { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); + return true; + } + self.gil_countdown -= 1; + if crate::hot_gates::loop_gen() != snap_gen { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); + return true; + } + // SAFETY: see `quiet_run`. + if sw.fin && unsafe { (*sw.maybe_dead).get() } { + // SAFETY: as above. + unsafe { + self.flush_lean(&mut *cframe, &mut *cshell.cast::>()); } - QuietEntry::Fresh | QuietEntry::Stop(_) => { - unreachable!("inline_gen_deliver returns or raises") + if self.drain_if_maybe_dead() && crate::hot_gates::loop_gen() != snap_gen { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); } } true @@ -37075,6 +37075,11 @@ impl Interpreter { // `Running`, which owns nothing) and no guard is live on the cell // (`peek_mut` checks). match unsafe { gen.state.peek_mut() } { + // `Running` owns nothing: no drop glue for the displaced state. + // SAFETY: as above. + Some(state) if matches!(state, GeneratorState::Running) => unsafe { + std::ptr::write(state, GeneratorState::Suspended(boxed)); + }, Some(state) => *state = GeneratorState::Suspended(boxed), None => *gen.state.borrow_mut() = GeneratorState::Suspended(boxed), } @@ -64156,14 +64161,18 @@ struct BinderLayout { #[inline] fn code_vm_ext(code: &CodeObject) -> Option<&CodeConstObjects> { - // The warm read — one acquire load and a cast — is what the dispatch - // loop pays per activation, so the table's construction lives out of - // line: inlining it here cost 4-7% on call-heavy fixtures. - let arc = match code.vm_ext.0.get() { - Some(arc) => arc, - None => code_vm_ext_init(code), - }; - Some(code_vm_ext_ref(arc)) + // The warm read — one acquire load of the cached thin pointer — is + // what the dispatch loop pays per activation, so the table's + // construction lives out of line: inlining it here cost 4-7% on + // call-heavy fixtures. + let p = code.vm_ext.1.load(std::sync::atomic::Ordering::Acquire); + if !p.is_null() { + // SAFETY: only `code_vm_ext_init` stores this pointer: the address + // of the `CodeConstObjects` the slot's `Arc` owns, alive as long as + // `code`. + return Some(unsafe { &*p.cast::() }); + } + Some(code_vm_ext_ref(code_vm_ext_init(code))) } /// Borrow an already initialized extension without filling any cache. @@ -64195,6 +64204,19 @@ fn code_vm_ext_ref(arc: &std::sync::Arc) -> &Co #[inline(never)] fn code_vm_ext_init( code: &CodeObject, +) -> &std::sync::Arc { + let arc = code_vm_ext_build(code); + let p: *const CodeConstObjects = code_vm_ext_ref(arc); + code.vm_ext + .1 + .store(p.cast_mut().cast(), std::sync::atomic::Ordering::Release); + arc +} + +#[cold] +#[inline(never)] +fn code_vm_ext_build( + code: &CodeObject, ) -> &std::sync::Arc { code.vm_ext.0.get_or_init(|| { std::sync::Arc::new(CodeConstObjects { From 4071d67d96515b9b21a549a032b236ed58dab933 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 12:51:57 -0700 Subject: [PATCH 44/65] perf: release still-shared JIT pins and core operands without the full drop grade A pinned receiver or argument that other references still hold, with no weakref watching it, can't be at its dead line (past two owners only a weakref clone could put it there). gc_trace::drop_survives_plainly says so in two loads; the JIT's pin drain and the core loop's operand check take it before the prompt-reap and drop grades, which cost about 120 instructions per pin. The fused-pair bisection flag moves into the per-code table, off the frame switch. float_math (JIT on): 6% fewer instructions. --- crates/weavepy-vm/src/gc_trace.rs | 18 ++++++++++++++++++ crates/weavepy-vm/src/lib.rs | 16 ++++++++-------- crates/weavepy-vm/src/tier2.rs | 8 ++++++++ 3 files changed, 34 insertions(+), 8 deletions(-) diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 53c4b1c8..2fb3564f 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -4463,6 +4463,24 @@ pub fn note_dropped_marks(obj: &crate::object::Object) -> bool { } } +/// Does dropping this reference to `obj` leave an instance that others +/// still hold, with no weakref watching it? Past two owners only a +/// weakref-watched object can be at its dead line (see +/// [`note_dropped_counted`]), so such a drop needs neither a prompt reap +/// nor a mark: the cheap test ahead of the full grades. +#[inline(always)] +pub fn drop_survives_plainly(obj: &crate::object::Object) -> bool { + match obj { + crate::object::Object::Instance(i) => { + crate::sync::Rc::strong_count(i) > 2 + && !crate::weakref_registry::may_have_weakrefs( + crate::sync::Rc::as_ptr(i) as usize as u64, + ) + } + _ => false, + } +} + /// How many elements a dying container may hold and still be graded by /// inspection (see [`inert_death`]). Capped so the grade stays O(1) on a /// path that runs for every discarded heap value. diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 246cd0a1..8efae2d1 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -11991,11 +11991,7 @@ impl Interpreter { // // The fused local pairs below, one byte per instruction (empty // for code that has none). - let fast_pairs: &[u8] = if crate::hot_gates::env_flags::no_pairs() { - &[] - } else { - code_fast_pairs(code, ext) - }; + let fast_pairs: &[u8] = code_fast_pairs(code, ext); let stack = &mut frame.stack; let base = stack.as_mut_ptr(); let cap = stack.capacity(); @@ -15304,9 +15300,9 @@ impl Interpreter { // weakref operation that requires tracking revokes the flag. // Past two owners only a weakref-watched object can be at // its dead line (see `gc_trace::note_dropped_marks`). - let sc = Rc::strong_count(i); - sc > 2 && !crate::weakref_registry::may_have_weakrefs(Rc::as_ptr(i) as usize as u64) - || sc > 1 && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) + gc_trace::drop_survives_plainly(v) + || Rc::strong_count(i) > 1 + && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) } Object::List(l) if Rc::strong_count(l) > 1 => !gc_trace::note_dropped_marks(v), Object::Dict(d) if Rc::strong_count(d) > 1 => !gc_trace::note_dropped_marks(v), @@ -63826,6 +63822,10 @@ pub(crate) fn code_is_pure_leaf_pub(code: &CodeObject) -> bool { fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> &'a [u8] { let Some(ext) = ext else { return &[] }; ext.fast_pairs.get_or_init(|| { + // (`WEAVEPY_NO_PAIRS`, a bisection aid, leaves every table empty.) + if crate::hot_gates::env_flags::no_pairs() { + return Box::new([]); + } let ops = &code.instructions; let mut any = false; let mut kinds = vec![0u8; ops.len()]; diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 1866d6de..fa76c5b9 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -2155,6 +2155,11 @@ fn drain_activation_pins(interp: &mut super::Interpreter, pins: &mut PinTable) - Pin::List(list, _) => Object::List(list), Pin::Obj(o) => o, }; + // The common pin, a receiver or argument others still hold, + // releases plainly (the grades below would conclude the same). + if crate::gc_trace::drop_survives_plainly(&o) { + continue; + } if super::Interpreter::local_needs_prompt_reap(&o) && super::Interpreter::looks_reapable_temporary(&o) { @@ -2179,6 +2184,9 @@ fn defer_activation_pins(pins: &mut PinTable) { Pin::List(list, _) => Object::List(list), Pin::Obj(obj) => obj, }; + if crate::gc_trace::drop_survives_plainly(&obj) { + continue; + } if super::Interpreter::local_needs_prompt_reap(&obj) && super::Interpreter::looks_reapable_temporary(&obj) { From 18ec3eccb82dfb0b1a44ee7940972c8110fb8b80 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 13:54:26 -0700 Subject: [PATCH 45/65] perf: return inline frames and nested leaf calls with less bookkeeping - The core loop's clean-return test counts each object's own slots instead of assuming every heap local aliases every other, so frames holding a few distinct objects skip the exit reap. - The exit reap's fast path untracks a dying collector-tracked list or dict whose contents can't finalize (deltablue's drained todo lists) instead of running the full prompt-reap cascade, and skips its alias count for objects held past every slot. - A leaf plan's nested pure-leaf call returns its result borrowed when the callee's scratch doesn't hold it (see LeafRet), so the caller neither clones nor later releases a getter's result. deltablue: about 6% fewer instructions. --- crates/weavepy-vm/src/leaf_plan.rs | 66 ++++++++++++-- crates/weavepy-vm/src/lib.rs | 137 ++++++++++++++++++++++++----- 2 files changed, 177 insertions(+), 26 deletions(-) diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index befc6f54..d86e7961 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -716,6 +716,15 @@ impl Owned { } impl Owned { + /// Whether `p` names one of the held values. + #[inline(always)] + fn holds(&self, p: *const Object) -> bool { + let base = self.buf.as_ptr().cast::(); + // SAFETY: one past the buffer's end. + let end = unsafe { base.add(self.buf.len()) }; + p >= base && p < end + } + /// Hold `v` (a scalar needs no holding) and name it. #[inline] fn own(&mut self, v: Object) -> Option { @@ -808,6 +817,25 @@ enum Callee<'a> { ), } +/// A leaf evaluation's result: a value the caller may borrow for the rest +/// of its own evaluation (an argument, a field of one, a namespace entry +/// or a constant: nothing the evaluation owned), or an owned object. +pub(crate) enum LeafRet { + Borrowed(V), + Owned(Object), +} + +impl LeafRet { + /// The result as an owned object (a borrowed value is cloned). + #[inline(always)] + pub(crate) fn into_object(self) -> Option { + match self { + LeafRet::Borrowed(v) => to_object(v), + LeafRet::Owned(o) => Some(o), + } + } +} + /// An owned object for a leaf value (the return value, a buffered /// store's value); `None` for the markers that are never values. #[inline(always)] @@ -829,11 +857,11 @@ impl Interpreter { /// on borrowed `args`, at call-nesting depth `nest`: the leaf's /// result, or `None` having done nothing observable (see /// `Interpreter::leaf_eval`). - #[inline(never)] /// /// `FRESH`: the first argument is an instance nothing else has seen /// (a constructor's `self`), so a store into it lands at once — a /// later decline leaves it half-built, and the caller discards it. + #[inline(always)] pub(crate) fn leaf_run( &self, code: &CodeObject, @@ -843,6 +871,22 @@ impl Interpreter { args: &[*const Object], nest: u8, ) -> Option { + self.leaf_run_ret::(code, ext, plan, f, args, nest)? + .into_object() + } + + /// [`Self::leaf_run`] with its result borrowed when the caller may + /// (see [`LeafRet`]). + #[inline(never)] + pub(crate) fn leaf_run_ret( + &self, + code: &CodeObject, + ext: &CodeConstObjects, + plan: &LeafPlan, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { use weavepy_compiler::InlineCache as IC; /// Nested leaf calls evaluated in place at most this deep. const NEST: u8 = 3; @@ -1202,11 +1246,23 @@ impl Interpreter { }; ptrs[k] = staged[k].write(o); } - let r = self.leaf_eval::(ccode, callee, &ptrs[..n], nest + 1)?; - set!(at, owned.own(r)?); + match self.leaf_eval_nested(ccode, callee, &ptrs[..n], nest + 1)? { + // A constructor's stores land at once, so what it + // borrows from its fresh instance could move. + LeafRet::Borrowed(v) if !FRESH => set!(at, v), + LeafRet::Borrowed(v) => set!(at, owned.own(to_object(v)?)?), + LeafRet::Owned(r) => set!(at, owned.own(r)?), + } } Op::Return { src } => { - let r = to_object(get!(src))?; + let v = get!(src); + // A value this evaluation's scratch doesn't hold outlives + // it (a pure body stores nothing), so a nested caller + // borrows it rather than taking a reference to release. + if !EFFECT && !matches!(v, V::R(p) if owned.holds(p)) { + return Some(LeafRet::Borrowed(v)); + } + let r = to_object(v)?; if EFFECT && pend.n > 0 { // The latest store to each attribute is the one // that lands; every one of them must go through @@ -1266,7 +1322,7 @@ impl Interpreter { } } } - return Some(r); + return Some(LeafRet::Owned(r)); } Op::StoreAttr { recv, diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 8efae2d1..70d6efa9 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -3325,6 +3325,10 @@ impl Interpreter { return false; } let id = crate::weakref_registry::id_of(v); + // Held past every slot and a collector handle: no count. + if sc > nlocals + 1 { + return !crate::weakref_registry::may_have_weakrefs(id); + } // How many of this frame's slots hold `v`: exact for // small frames (the usual `self` is held once here and // once by the caller), the slot count otherwise. @@ -3350,9 +3354,16 @@ impl Interpreter { || escaped(v) || matches!(v, Object::Instance(i) if i.dies_by_plain_drop()) || Self::atomic_container_dies_plainly(v) + || Self::tracked_container_dies_inertly(v) }) { for v in locals { - gc_trace::note_dropped(v); + // A dying tracked container sheds its collector handle + // now (the cascade would find nothing else to do). + if Self::tracked_container_dies_inertly(v) { + gc_trace::untrack_id(crate::weakref_registry::id_of(v)); + } else { + gc_trace::note_dropped(v); + } } return; } @@ -3866,6 +3877,41 @@ impl Interpreter { } } + /// A collector-tracked list or dict whose only holders are the caller's + /// reference and its collector handle, watched by no weakref, holding + /// only values its death can't finalize (atomic ones, or instances + /// others still hold): untracking it and dropping the reference is its + /// whole teardown, as in CPython's `list_dealloc`. + fn tracked_container_dies_inertly(o: &Object) -> bool { + const MAX: usize = 16; + fn inert(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Str(_) + ) || gc_trace::drop_survives_plainly(v) + } + let (sc, id) = match o { + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + _ => return false, + }; + if sc != 2 + || crate::weakref_registry::may_have_weakrefs(id) + || !gc_trace::is_tracked(id) + { + return false; + } + match o { + Object::List(l) => l + .try_borrow() + .is_ok_and(|v| v.len() <= MAX && v.iter().all(inert)), + Object::Dict(d) => d + .try_borrow() + .is_ok_and(|d| d.len() <= MAX && d.iter().all(|(k, v)| inert(&k.0) && inert(v))), + _ => false, + } + } + fn local_needs_prompt_reap(o: &Object) -> bool { matches!( o, @@ -15373,37 +15419,62 @@ impl Interpreter { /// for each release. #[inline] fn core_escaped_locals(locals: &[Object]) -> bool { - let mut heap = 0usize; + // Each heap local's identity and strong count: up to eight are + // compared exactly (how many slots here name each object); past + // that, every heap slot is assumed to name every object. + const EXACT: usize = 8; + let mut heap = [const { std::mem::MaybeUninit::<(u64, usize)>::uninit() }; EXACT]; + let mut n = 0usize; for o in locals { - match o { + let (id, sc) = match o { Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Unbound - | Object::Str(_) => {} - Object::Instance(_) | Object::List(_) | Object::Dict(_) | Object::Tuple(_) => { - heap += 1; - } + | Object::Str(_) => continue, + Object::Instance(i) => (Rc::as_ptr(i) as usize as u64, Rc::strong_count(i)), + Object::List(l) => (Rc::as_ptr(l) as usize as u64, Rc::strong_count(l)), + Object::Dict(d) => (Rc::as_ptr(d) as usize as u64, Rc::strong_count(d)), + Object::Tuple(t) => ( + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ThinArc::strong_count(t), + ), _ => return false, + }; + if n < EXACT { + heap[n].write((id, sc)); } + n += 1; } - if heap == 0 { + if n == 0 { return true; } - locals.iter().all(|o| { - let (sc, id) = match o { - Object::Instance(i) => (Rc::strong_count(i), Rc::as_ptr(i) as usize as u64), - Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), - Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), - Object::Tuple(t) => ( - ThinArc::strong_count(t), - ThinArc::as_ptr(t).cast::<()>() as usize as u64, - ), - _ => return true, - }; - sc > heap + usize::from(gc_trace::maybe_tracked(id)) - && !crate::weakref_registry::may_have_weakrefs(id) + if n > EXACT { + return locals.iter().all(|o| { + let (sc, id) = match o { + Object::Instance(i) => (Rc::strong_count(i), Rc::as_ptr(i) as usize as u64), + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + Object::Tuple(t) => ( + ThinArc::strong_count(t), + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ), + _ => return true, + }; + sc > n + usize::from(gc_trace::maybe_tracked(id)) + && !crate::weakref_registry::may_have_weakrefs(id) + }); + } + // SAFETY: the first `n` entries were written. + let heap: &[(u64, usize)] = unsafe { std::slice::from_raw_parts(heap.as_ptr().cast(), n) }; + heap.iter().all(|&(id, sc)| { + // Held past every slot here and a collector handle needs no + // count of the slots that name it. + (sc > n + 1 || { + let held = heap.iter().filter(|&&(other, _)| other == id).count(); + sc > held + usize::from(gc_trace::maybe_tracked(id)) + }) && !crate::weakref_registry::may_have_weakrefs(id) }) } @@ -16391,6 +16462,30 @@ impl Interpreter { } } + /// [`Self::leaf_eval`] for a leaf plan's own pure-leaf call: the + /// callee's result borrowed when it outlives the callee's evaluation + /// (see [`leaf_plan::LeafRet`]), which the caller then neither clones + /// nor releases. + pub(crate) fn leaf_eval_nested( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + let ext = code_vm_ext(code)?; + if ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) >= 3 { + return self + .leaf_eval::(code, f, args, nest) + .map(leaf_plan::LeafRet::Owned); + } + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run_ret::(code, ext, plan, f, args, nest) + } + /// [`Self::pure_leaf_eval`] at call-nesting depth `nest` (a pure-leaf /// callee of a leaf is evaluated one level down, to a small bound). #[inline(never)] From 15877905b415c5478dd08535024d4b2a38687067 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 14:37:16 -0700 Subject: [PATCH 46/65] perf: reuse the pickle encoder's memo table and size its output Each dumps call grew a fresh memo table from empty (rehashing as it doubled) and a fresh output buffer (copying as it doubled). The encoder now takes the thread's spare table, emptied with its capacity kept, and reserves the previous call's output size. pickle_bench: about 3% faster. --- .../src/stdlib/pickle_accel/encode.rs | 43 ++++++++++++++++--- 1 file changed, 36 insertions(+), 7 deletions(-) diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs index f5ecba5a..f0c41756 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs @@ -550,9 +550,17 @@ fn encode_with_context( if !matches!(protocol, 4 | 5) { return None; } + // The previous call's memo table (emptied, its capacity kept) and + // output size: a `dumps` loop then neither regrows the table nor + // copies the output as it doubles. + let memo = SPARE_MEMO.with(std::cell::Cell::take).unwrap_or_default(); + let hint = OUTPUT_HINT.with(std::cell::Cell::get); let mut encoder = Encoder { - writer: Framer::default(), - memo: FxHashMap::default(), + writer: Framer { + output: Vec::new(), + frame_start: None, + }, + memo, anonymous: 0, active: Vec::new(), python_headroom, @@ -561,14 +569,35 @@ fn encode_with_context( classes: FxHashMap::default(), slot_name_caches: Vec::new(), }; - extend(&mut encoder.writer.output, &[0x80, protocol])?; - encoder.save(value, 0)?; - encoder.writer.write(b".")?; - encoder.writer.commit(true); - encoder.publish_slot_name_caches(); + let done = (|| { + encoder.writer.output.try_reserve(hint.clamp(64, MAX_OUTPUT_HINT)).ok()?; + extend(&mut encoder.writer.output, &[0x80, protocol])?; + encoder.save(value, 0)?; + encoder.writer.write(b".")?; + encoder.writer.commit(true); + encoder.publish_slot_name_caches(); + Some(()) + })(); + let mut memo = std::mem::take(&mut encoder.memo); + if memo.capacity() <= MAX_SPARE_MEMO { + memo.clear(); + SPARE_MEMO.with(|spare| spare.set(Some(memo))); + } + done?; + OUTPUT_HINT.with(|h| h.set(encoder.writer.output.len())); Some(encoder.writer.output) } +/// The largest memo table and output reservation kept between calls. +const MAX_SPARE_MEMO: usize = 1 << 14; +const MAX_OUTPUT_HINT: usize = 1 << 20; + +thread_local! { + static SPARE_MEMO: std::cell::Cell>> = + const { std::cell::Cell::new(None) }; + static OUTPUT_HINT: std::cell::Cell = const { std::cell::Cell::new(0) }; +} + #[cfg(test)] mod tests { use super::*; From 2a1c6cfc95de18565b43fbc78e961a9bb298cc10 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 16:17:50 -0700 Subject: [PATCH 47/65] perf: fuse container method calls, and cheapen dying unpickled graphs - A local list, dict, set or str receiver of a site-cached leaf method (x.append(v), d.get(k), s.startswith(p)) is called on the borrowed local, as instance receivers already were, so no receiver clone is pushed and graded on release. - The unpickler builds plain instances with deferred collector tracking, as ordinary construction does; slot state tracks them only when a value isn't atomic (dict state already did). A decoded Point of scalars now dies by plain drop instead of the prompt-reap cascade. - The prompt-reap cascade allocates its per-node scratch once per cascade. pickle_bench: about 5% fewer instructions; list.append/pop loop: 10%. --- crates/weavepy-vm/src/lib.rs | 117 +++++++++++++++++-- crates/weavepy-vm/src/stdlib/pickle_accel.rs | 22 +++- 2 files changed, 125 insertions(+), 14 deletions(-) diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 70d6efa9..ae9f85d2 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -2978,6 +2978,17 @@ impl Interpreter { // test below keeps `strong` above the dead threshold and leaves // the cycle to the tracing collector. let mut work = vec![dropped]; + // Each node's scratch, emptied per node and allocated once per + // cascade (a dead graph of many small containers paid four + // allocations and a hash set per node). + let mut children: Vec> = Vec::new(); + let mut weakref_candidates: Vec = Vec::new(); + let mut pool_candidates: Vec = Vec::new(); + let mut scan_through: Vec = Vec::new(); + let mut scanned: std::collections::HashSet< + crate::weakref_registry::ObjectId, + std::hash::BuildHasherDefault, + > = std::collections::HashSet::default(); loop { let Some(obj) = work.pop() else { // Freeing the objects above may have parked C-side dealloc @@ -3112,7 +3123,7 @@ impl Interpreter { // unpickler temporary holding the memo) cascades through the // untracked `memo` dict to the tracked argument. The refcount guard // below still filters anything that stays externally reachable. - let mut children: Vec> = Vec::new(); + debug_assert!(children.is_empty()); // Untracked descendants with live weakrefs: CPython clears an // object's weakrefs at refcount zero whether or not the GC ever // tracked it, but this cascade's "next link" set is (otherwise) @@ -3127,18 +3138,15 @@ impl Interpreter { // `has_reference()` flips to False). Collect them during the // scan and run the dead ones through the cascade so their // weakrefs clear at death, like any tracked link. - let mut weakref_candidates: Vec = Vec::new(); + debug_assert!(weakref_candidates.is_empty()); // Plain-tuple children about to die with `obj` are recycled // through the tuple pool (see `maybe_donate_tuple`): the plain // `Rc` drop below frees them invisibly, and this cascade is // where the common `f(*args)`-shaped temporaries actually die // (the argument tuple is anchored by a dead list/frame local). - let mut pool_candidates: Vec = Vec::new(); - let mut scan_through: Vec = vec![obj.clone()]; - let mut scanned: std::collections::HashSet< - crate::weakref_registry::ObjectId, - std::hash::BuildHasherDefault, - > = std::collections::HashSet::default(); + debug_assert!(pool_candidates.is_empty()); + scan_through.push(obj.clone()); + scanned.clear(); while let Some(parent) = scan_through.pop() { gc_trace::traverse_object(&parent, &mut |c| { // Scalars are never tracked, weakly referenced, or @@ -3231,14 +3239,14 @@ impl Interpreter { // A tuple child whose last holder was `obj` is now ours alone // (strong count 1 = the scan's clone); park its allocation. // Anything still referenced elsewhere just sheds the clone. - for cand in pool_candidates { + for cand in pool_candidates.drain(..) { if matches!(&cand, Object::Tuple(t) if ThinArc::strong_count(t) == 1) { self.maybe_donate_tuple(cand); } } // Any tracked child that just lost its last program reference // is the next link in the chain. - for h in children { + for h in children.drain(..) { if !h.untracked.load(std::sync::atomic::Ordering::Acquire) { let weak = crate::weakref_registry::strong_clone_count(h.id); // `h.object` is the GC's own strong reference; the @@ -3265,7 +3273,7 @@ impl Interpreter { // its weakrefs clear now (the refcount guard at the top of the // loop re-checks liveness — our `weakref_candidates` clone is // accounted for by the cascade's `extra` slack of 1). - for cand in weakref_candidates { + for cand in weakref_candidates.drain(..) { if Self::is_refcount_dead(&cand, 1) { work.push(cand); } @@ -12305,6 +12313,41 @@ impl Interpreter { continue; } } + // `x.m(...)` of a container's site-cached + // leaf method on the borrowed local. + if matches!( + other, + Object::List(_) + | Object::Dict(_) + | Object::Set(_) + | Object::Str(_) + ) && pc + 1 < ninstrs + && len < cap + && (*instrs.add(pc + 1)).op == OpCode::LoadMethodAttr + { + if let Some((r, call_pc)) = self.core_builtin_method( + code, + other, + pc + 1, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } + } clone_hot(other) } }; @@ -15807,6 +15850,58 @@ impl Interpreter { Some((r, call_pc)) } + /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); ; CALL k` on a local `list`, `dict`, `set` or `str` + /// whose method the `LOAD_ATTR` site cached under the receiver's tag and + /// the `CALL` site admitted as a leaf kind: called on bitwise views of + /// the borrowed operands, as [`Self::core_native_method`] does, so no + /// receiver clone is pushed and released (a release the collector + /// grades). `None` (nothing touched) runs the instructions one by one. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_builtin_method( + &self, + code: &CodeObject, + recv: &Object, + attr_pc: usize, + mslots: &[MethodSlot], + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ) -> Option<(Result, usize)> { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + Object::Str(_) => 4, + _ => return None, + }; + let b = mslots.get(attr_pc)?.get_builtin_ptr(tag)?; + // Bitwise views, never dropped (see `core_native_method`). + let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; + // SAFETY: the receiver is the core loop's local. + ops[0].write(unsafe { std::ptr::read(recv) }); + // SAFETY: the core loop's own locals and constants. + let (nargs, call_pc) = unsafe { + Self::core_simple_ops( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut ops, + ) + }?; + let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; + // SAFETY: the first `nargs + 1` entries were written. + let ops = + unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), nargs + 1) }; + // SAFETY: the slot (and the builtin type) holds the method for the + // call, which runs no Python code. + let r = self.leaf_builtin_call(kind, unsafe { &*b }, ops)?; + Some((r, call_pc)) + } + /// A fused simple call of pure leaf `fp` (see [`code_is_pure_leaf`]): /// `args[..nargs]` are the borrowed operands (the receiver first for a /// bound call, `has_self`), and `call_pc` the `CALL` whose site slot diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel.rs b/crates/weavepy-vm/src/stdlib/pickle_accel.rs index 84d01a36..cc8e79aa 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel.rs @@ -1437,7 +1437,19 @@ impl<'a> Sink<'a> for Objects { let Object::Type(class) = class else { return None; }; - // `object.__new__(cls)`, including its cycle-collector registration. + // `object.__new__(cls)`, including its cycle-collector registration: + // deferred, as for a plain class's ordinary construction, until the + // instance could hold a non-atomic value (the state paths below + // track it then). + if class.native_kind.get() == 0 + && !class.flags.is_builtin + && !class.instances_need_finalize() + { + self.next_node += 1; + return Some(Object::Instance(crate::types::PyInstance::new_deferred( + class.clone(), + ))); + } let instance = Object::Instance(Rc::new(crate::types::PyInstance::new(class.clone()))); self.created(&instance); Some(instance) @@ -1468,8 +1480,12 @@ impl<'a> Sink<'a> for Objects { } } if let Some(slots) = slots { - // `setattr(inst, key, value)` on a verified member descriptor. - instance.ensure_gc_tracked(); + // `setattr(inst, key, value)` on a verified member descriptor, + // whose write barrier tracks a deferred instance at its first + // non-atomic value. + if slots.borrow().iter().any(|(_, v)| !v.is_gc_atomic()) { + instance.ensure_gc_tracked(); + } let mut storage = instance.slots.borrow_mut(); if movable.slots.is_some() { let data = std::mem::take(&mut *slots.borrow_mut()); From 16fed239c82e50d90c9ff7325fa62645554b8f86 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 17:09:17 -0700 Subject: [PATCH 48/65] ci: fix formatting, clippy lints, and the GC threshold fixture The run fixture for weakrefs and gc still expected the old default collection threshold of 700; the branch adopted CPython 3.14's 2000 (the fixture's output now matches CPython's). The remaining changes are rustfmt and clippy fixes, with alignment and mutable-borrow allowances where the pointer casts are sound by construction. --- crates/weavepy-vm/src/gc_trace.rs | 4 ++-- crates/weavepy-vm/src/inst_dict.rs | 8 +++++-- crates/weavepy-vm/src/leaf_plan.rs | 2 ++ crates/weavepy-vm/src/lib.rs | 22 ++++++++--------- crates/weavepy-vm/src/object.rs | 3 +-- .../weavepy-vm/src/stdlib/datetime_native.rs | 2 ++ .../src/stdlib/pickle_accel/encode.rs | 6 ++++- crates/weavepy-vm/src/stdlib/socket_mod.rs | 8 +++++++ crates/weavepy-vm/src/tier2.rs | 7 ++++-- crates/weavepy-vm/src/types.rs | 24 +++++++++++++++---- .../tests/fixtures/run/58_weakref_and_gc.out | 2 +- 11 files changed, 63 insertions(+), 25 deletions(-) diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 2fb3564f..88ad7e91 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -3360,7 +3360,7 @@ pub fn zombie_memoryview_refs_to(target: ObjectId) -> usize { ObjectId, usize, BuildHasherDefault, - > = Default::default(); + > = std::collections::HashMap::default(); for h in &handles { if zombies.contains(&h.id) { traverse_object(&h.object, &mut |c| { @@ -4474,7 +4474,7 @@ pub fn drop_survives_plainly(obj: &crate::object::Object) -> bool { crate::object::Object::Instance(i) => { crate::sync::Rc::strong_count(i) > 2 && !crate::weakref_registry::may_have_weakrefs( - crate::sync::Rc::as_ptr(i) as usize as u64, + crate::sync::Rc::as_ptr(i) as usize as u64 ) } _ => false, diff --git a/crates/weavepy-vm/src/inst_dict.rs b/crates/weavepy-vm/src/inst_dict.rs index 442a6013..a7f7e8a3 100644 --- a/crates/weavepy-vm/src/inst_dict.rs +++ b/crates/weavepy-vm/src/inst_dict.rs @@ -307,6 +307,8 @@ impl SplitValues { let new_layout = Self::layout(new_cap); // SAFETY: a nonzero-size layout; a grown block keeps its // header and initialized prefix (realloc copies them). + // (The block is allocated with the header's alignment.) + #[allow(clippy::cast_ptr_alignment)] let h = unsafe { match self.block { Some(b) => { @@ -686,7 +688,8 @@ impl InstDict { let off = std::mem::offset_of!(crate::types::PyInstance, dict); // SAFETY: an `InstDict` exists only as `PyInstance::dict`, so the // containing instance starts `off` bytes before it and outlives - // this borrow. + // this borrow (and is aligned as a `PyInstance`). + #[allow(clippy::cast_ptr_alignment)] unsafe { &*std::ptr::from_ref(self) .cast::() @@ -1004,6 +1007,7 @@ impl crate::types::PyInstance { /// /// As [`crate::sync::GilCell::peek_mut`]. #[inline(always)] + #[allow(clippy::mut_from_ref)] pub unsafe fn attr_peek_index_mut( &self, i: usize, @@ -1018,7 +1022,7 @@ impl crate::types::PyInstance { } else { d.map_mut_value_store() }; - map.get_index_mut(i).map(|(k, v)| (&*k, v)) + map.get_index_mut(i) } None => { if !atomic && self.deferred.get() { diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index d86e7961..b07f5960 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -596,6 +596,7 @@ impl Builder<'_> { let instrs = &self.code.instructions; // Whether the previous instruction falls through to this one. let mut live = true; + #[allow(clippy::needless_range_loop)] for pc in 0..instrs.len() { if let Some(&(depth, assigned)) = self.targets.get(&pc) { if live { @@ -1193,6 +1194,7 @@ impl Interpreter { // arguments: never dropped, so no reference moves. let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; + #[allow(clippy::needless_range_loop)] for k in 0..n { let o = match get!(first + k as u8) { // SAFETY: as `norm`; the copy is forgotten. diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index ae9f85d2..b6f7dab7 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -3903,10 +3903,7 @@ impl Interpreter { Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), _ => return false, }; - if sc != 2 - || crate::weakref_registry::may_have_weakrefs(id) - || !gc_trace::is_tracked(id) - { + if sc != 2 || crate::weakref_registry::may_have_weakrefs(id) || !gc_trace::is_tracked(id) { return false; } match o { @@ -12343,7 +12340,9 @@ impl Interpreter { continue; } Err(e) => { - break Some(CoreExit::Stop(LeafStop::Raised(e))); + break Some(CoreExit::Stop(LeafStop::Raised( + e, + ))); } } } @@ -14972,7 +14971,7 @@ impl Interpreter { } let (inst, _) = self.alloc_plain_instance_obj(ty); let mut args: [*const Object; 8] = [std::ptr::null(); 8]; - args[0] = &inst; + args[0] = &raw const inst; for (k, a) in frame.stack[self_slot + 1..].iter().enumerate() { args[k + 1] = a; } @@ -15086,8 +15085,10 @@ impl Interpreter { let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; // (An instance goes to the caller's native subscript, which // grades its own release.) - if !matches!(c, Object::List(_) | Object::Tuple(_) | Object::Dict(_) | Object::Str(_)) - || !Self::core_droppable(c) + if !matches!( + c, + Object::List(_) | Object::Tuple(_) | Object::Dict(_) | Object::Str(_) + ) || !Self::core_droppable(c) { return None; } @@ -15894,8 +15895,7 @@ impl Interpreter { }?; let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; // SAFETY: the first `nargs + 1` entries were written. - let ops = - unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), nargs + 1) }; + let ops = unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), nargs + 1) }; // SAFETY: the slot (and the builtin type) holds the method for the // call, which runs no Python code. let r = self.leaf_builtin_call(kind, unsafe { &*b }, ops)?; @@ -16359,7 +16359,7 @@ impl Interpreter { let dict; if code_rc.has_varkeywords { dict = Object::Dict(Rc::new(RefCell::new(varkw.unwrap_or_default()))); - args[total] = &dict; + args[total] = &raw const dict; } self.pure_leaf_eval::(code_rc, f, &args[..arity]) } diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index 4ef6ef34..5c547af9 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -1204,10 +1204,9 @@ pub fn materialize_stack_at(stack: &FrameStack, idx: usize) -> Option { - let py = shell.materialize(back); // Materialised while live on the stack: count the // activation so `frame.clear()` refuses it. - py + shell.materialize(back) } }; back = Some(py); diff --git a/crates/weavepy-vm/src/stdlib/datetime_native.rs b/crates/weavepy-vm/src/stdlib/datetime_native.rs index 317014ab..df69bda3 100644 --- a/crates/weavepy-vm/src/stdlib/datetime_native.rs +++ b/crates/weavepy-vm/src/stdlib/datetime_native.rs @@ -280,6 +280,7 @@ fn as_int(o: &Object) -> Option { // The field readers take a natively built instance's values straight // from the shared layout; any other storage is read by name. +#[allow(clippy::index_refutable_slice)] fn td_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { let n = &st.names; let s = i.slots.try_borrow().ok()?; @@ -301,6 +302,7 @@ fn td_us(f: (i64, i64, i64)) -> i128 { (i128::from(f.0) * 86_400 + i128::from(f.1)) * 1_000_000 + i128::from(f.2) } +#[allow(clippy::index_refutable_slice)] fn date_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { let n = &st.names; let s = i.slots.try_borrow().ok()?; diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs index f0c41756..52b61377 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs @@ -570,7 +570,11 @@ fn encode_with_context( slot_name_caches: Vec::new(), }; let done = (|| { - encoder.writer.output.try_reserve(hint.clamp(64, MAX_OUTPUT_HINT)).ok()?; + encoder + .writer + .output + .try_reserve(hint.clamp(64, MAX_OUTPUT_HINT)) + .ok()?; extend(&mut encoder.writer.output, &[0x80, protocol])?; encoder.save(value, 0)?; encoder.writer.write(b".")?; diff --git a/crates/weavepy-vm/src/stdlib/socket_mod.rs b/crates/weavepy-vm/src/stdlib/socket_mod.rs index 736fdd10..88075a95 100644 --- a/crates/weavepy-vm/src/stdlib/socket_mod.rs +++ b/crates/weavepy-vm/src/stdlib/socket_mod.rs @@ -3908,6 +3908,8 @@ fn resolve_ipv4(name: &str, who: &str) -> Result, RuntimeError> { // SAFETY: rc == 0 guarantees a valid chain until `freeaddrinfo`. let ai = unsafe { &*cur }; if ai.ai_family == libc::AF_INET && !ai.ai_addr.is_null() { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv4Addr::from(u32::from_be(sin.sin_addr.s_addr)).to_string(); if !ips.contains(&ip) { @@ -4316,6 +4318,8 @@ fn mod_getaddrinfo(args: &[Object]) -> Result { cur = ai.ai_next; let addr_tuple = match ai.ai_family { f if f == libc::AF_INET => { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv4Addr::from(u32::from_be(sin.sin_addr.s_addr)); Object::new_tuple_array([ @@ -4324,6 +4328,8 @@ fn mod_getaddrinfo(args: &[Object]) -> Result { ]) } f if f == libc::AF_INET6 => { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin6 = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv6Addr::from(sin6.sin6_addr.s6_addr); Object::new_tuple_array([ @@ -4621,6 +4627,8 @@ fn mod_getnameinfo(args: &[Object]) -> Result { // SAFETY: rc == 0 guarantees a valid chain until `freeaddrinfo`. let ai = unsafe { &*res }; if ai.ai_family == libc::AF_INET6 { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin6 = unsafe { &mut *ai.ai_addr.cast::() }; sin6.sin6_flowinfo = (flowinfo as u32).to_be(); sin6.sin6_scope_id = scope_id; diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index fa76c5b9..191d3361 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -4023,7 +4023,7 @@ unsafe fn call_py_with_gaps( if kwargs.is_empty() && args.len() == j { args.push(v); } else if let Some(name) = code.varnames.get(j) { - kwargs.push((name.to_string(), v)); + kwargs.push((name.clone(), v)); } } let res = call_with_activation_shell(interp, ctx, jf, |i| { @@ -5478,7 +5478,10 @@ fn note_callee_exit(art: &Artifacts, code: &Rc, child: &CallCtx) { if child.dyn_py_calls == 0 { return; } - let trips = art.callee_roundtrips.get().saturating_add(child.dyn_py_calls); + let trips = art + .callee_roundtrips + .get() + .saturating_add(child.dyn_py_calls); art.callee_roundtrips.set(trips); if entries >= GENERIC_RETIRE_MIN_ENTRIES && trips / entries >= CALLEE_ROUNDTRIP_RETIRE_RATIO diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index e961e7a9..c0543ad7 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -2281,14 +2281,22 @@ impl SlotStorage { match &self.data { SlotData::Small(entries) => { let entries = entries.get(..N)?; - if !entries.iter().zip(names).all(|((key, _), name)| key_named(key, name)) { + if !entries + .iter() + .zip(names) + .all(|((key, _), name)| key_named(key, name)) + { return None; } Some(std::array::from_fn(|i| &entries[i].1)) } SlotData::Fixed { layout, values } => { let (keys, values) = (layout.get(..N)?, values.get(..N)?); - if !keys.iter().zip(names).all(|(key, name)| key_named(key, name)) { + if !keys + .iter() + .zip(names) + .all(|(key, name)| key_named(key, name)) + { return None; } Some(std::array::from_fn(|i| &values[i])) @@ -2306,7 +2314,11 @@ impl SlotStorage { let values: &mut [Object] = match &mut self.data { SlotData::Small(entries) => { let entries = entries.get_mut(..N)?; - if !entries.iter().zip(names).all(|((key, _), name)| key_named(key, name)) { + if !entries + .iter() + .zip(names) + .all(|((key, _), name)| key_named(key, name)) + { return None; } let mut values = entries.iter_mut().map(|(_, value)| value); @@ -2314,7 +2326,11 @@ impl SlotStorage { } SlotData::Fixed { layout, values } => { let keys = layout.get(..N)?; - if !keys.iter().zip(names).all(|(key, name)| key_named(key, name)) { + if !keys + .iter() + .zip(names) + .all(|(key, name)| key_named(key, name)) + { return None; } values.get_mut(..N)? diff --git a/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out b/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out index 5a998283..2e8c0231 100644 --- a/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out +++ b/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out @@ -10,7 +10,7 @@ gc enabled: True disabled: False re-enabled: True collected is int: True -threshold: (700, 10, 10) +threshold: (2000, 10, 10) threshold: (800, 12, 14) count len: 3 objects is list: True From bd076cf292337cc068db1ed2345e5aab7712d483 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 17:58:50 -0700 Subject: [PATCH 49/65] perf: warm the JIT's code generator off the main thread The branch stopped compiling standard library functions during startup, so a program's first compile now pays the code generator's cold start (about 0.7 ms: its code paged in, its tables built) inside its hot loop; the Windows bench gate flagged sumvm at +18% against the merge base. The CLI now compiles and discards a small loop on a background thread when it starts a program file or module, and any other run does so when its first code object is halfway to the compile threshold. -c pass is unaffected (no extra memory). sumvm, jitloop, nested_loops: back within 1-3% of the merge base. --- crates/weavepy-cli/src/lib.rs | 4 +++ crates/weavepy-vm/src/lib.rs | 8 +++++ crates/weavepy-vm/src/tier2.rs | 58 ++++++++++++++++++++++++++++++++++ 3 files changed, 70 insertions(+) diff --git a/crates/weavepy-cli/src/lib.rs b/crates/weavepy-cli/src/lib.rs index 6fa2bdec..6714a6a5 100644 --- a/crates/weavepy-cli/src/lib.rs +++ b/crates/weavepy-cli/src/lib.rs @@ -1080,6 +1080,7 @@ fn real_main() -> Result { if let Some(module) = cli.module.clone() { let extra = cli.args.clone(); + weavepy::vm::spawn_jit_codegen_prewarm(); run_module(&module, extra, &flags, &extra_path)?; return Ok(0); } @@ -1092,6 +1093,9 @@ fn real_main() -> Result { Ok(0) } Some(path) => { + // A program file: warm the JIT's code generator off the main + // thread while it starts. + weavepy::vm::spawn_jit_codegen_prewarm(); run_path(path, trailing.clone(), &flags, &extra_path)?; Ok(0) } diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index b6f7dab7..eedb7760 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -543,6 +543,14 @@ fn frame_reapables(g: &Rc) -> Vec { out } +/// Warm the JIT's code generator on a background thread when the JIT +/// is on (see `tier2::prewarm_codegen`): a program's first compile then +/// doesn't pay the generator's cold start. +pub fn spawn_jit_codegen_prewarm() { + #[cfg(feature = "jit")] + tier2::spawn_codegen_prewarm_if_enabled(); +} + /// RFC 0032 — render the tier-2 JIT's counters as a markdown block for /// the `WEAVEPY_VM_STATS` report, or `None` when the `jit` feature is /// disabled or the JIT was never exercised on this thread. diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 191d3361..75aba9fd 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -578,6 +578,61 @@ struct JitState { static JIT_PROCESS_GATE: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(0); /// The `WEAVEPY_JIT` / free-threading verdict shared by every thread. +/// Warm the code generator on a background thread (see +/// [`prewarm_codegen`]), once per process: when the CLI starts a program +/// file or module, and otherwise when the first code object is halfway +/// to its compile threshold (so `-c pass` doesn't pay for it). +pub(crate) fn spawn_codegen_prewarm_if_enabled() { + if jit_enabled_by_config() { + spawn_codegen_prewarm(); + } +} + +fn spawn_codegen_prewarm() { + static SPAWNED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + if SPAWNED.swap(true, std::sync::atomic::Ordering::Relaxed) { + return; + } + let _ = std::thread::Builder::new() + .name("weavepy-jit-warm".to_owned()) + .spawn(prewarm_codegen); +} + +/// Compile, on a throwaway engine, a small counted loop of the shape hot +/// code takes, and discard it. A process's first compile otherwise pays +/// the code generator's cold start (its code paged in and its tables +/// built: about 0.7 ms on the development host, most of the first +/// compile) on the thread that needs the compiled code. Touches no +/// interpreter state. +fn prewarm_codegen() { + const SOURCE: &str = + "def f(n):\n t = 0\n for i in range(n):\n t = t + i * 2\n return t\n"; + let _ = std::panic::catch_unwind(|| { + let Ok(module) = weavepy_parser::parse_module(SOURCE) else { + return; + }; + let Ok(code) = weavepy_compiler::compile_module(&module) else { + return; + }; + let Some(f) = code.constants.iter().find_map(|c| match c { + weavepy_compiler::Constant::Code(f) => Some(f.clone()), + _ => None, + }) else { + return; + }; + let Some(mut engine) = JitEngine::new() else { + return; + }; + let _ = engine.compile(&f, &mut |name| { + if name == "range" { + weavepy_jit::ResolvedGlobal::RangeBuiltin + } else { + weavepy_jit::ResolvedGlobal::Opaque + } + }); + }); +} + fn jit_enabled_by_config() -> bool { // RFC 0067 WS3 — the tier-2 JIT is on by default; `WEAVEPY_JIT=0` // (or `off`, or an empty value) restores the pure interpreter. @@ -905,6 +960,9 @@ impl JitState { Tier::NotJitable => return None, Tier::Cold => { entry.counter += 1; + if entry.counter == self.threshold / 2 && self.engine.is_none() { + spawn_codegen_prewarm(); + } if entry.counter < self.threshold || !compile_allowed(entry.counter, self.threshold) { From 93320e450592af64cb6b68bef55d0b128240e1bc Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:37:30 -0700 Subject: [PATCH 50/65] perf: run simple generator bodies in fast steps, without switching Resuming a generator switched activations (the frame made the running one, the pending-caller and recursion bookkeeping, the core loop's reload) and back at each yield: most of the cost of each item for a body that only moves scalars through locals. A new stepper (gen_fast) runs such a body directly on its frame to its next yield: locals, constants, int and float arithmetic and comparisons, jumps, and for loops over ranges, lists, tuples and other such generators. It never raises or runs Python code; anything else ends the step at an instruction boundary and the ordinary resume continues from there, so a fast step is always a prefix of the ordinary run. The core loop's FOR_ITER over a generator and the lean resume path (next(), and sum()/list() folding) take fast steps first. A nested generator whose step stops partway is parked having consumed its sent value (Frame::sent_consumed), which every resume path honors. generators: 28% faster; a for loop over a generator: 34% fewer instructions per item. --- crates/weavepy-vm/src/gen_fast.rs | 698 ++++++++++++++++++++ crates/weavepy-vm/src/lib.rs | 178 +++-- crates/weavepy-vm/src/tier2.rs | 1 + tests/regrtest/test_generator_fast_steps.py | 201 ++++++ 4 files changed, 1013 insertions(+), 65 deletions(-) create mode 100644 crates/weavepy-vm/src/gen_fast.rs create mode 100644 tests/regrtest/test_generator_fast_steps.py diff --git a/crates/weavepy-vm/src/gen_fast.rs b/crates/weavepy-vm/src/gen_fast.rs new file mode 100644 index 00000000..01810f07 --- /dev/null +++ b/crates/weavepy-vm/src/gen_fast.rs @@ -0,0 +1,698 @@ +//! Fast steps for simple generator bodies. +//! +//! Resuming a generator the ordinary way switches activations: the quiet +//! loop (or the core loop's inline resume) makes the generator's frame the +//! running one, runs to the next `yield`, and switches back, with the +//! bookkeeping every activation needs (pending-caller lists, recursion +//! guards, the dispatch loop's reload). For a body that only moves scalars +//! through locals and loops over ranges, lists, or other such generators, +//! that bookkeeping is most of the cost of each item. +//! +//! [`Interpreter::gen_fast_step`] runs such a body directly on its frame: +//! from the frame's pc (the sent value already on its stack) to the next +//! `yield`. It never raises and never runs Python code: an instruction it +//! can't finish that way (an overflow, a non-scalar operand, a `return`, an +//! instruction outside its set) ends the step *before* that instruction, +//! with the frame at an ordinary instruction boundary, and the caller +//! continues the resume in the general loop from there. A fast step is +//! therefore always a prefix of the ordinary run. +//! +//! A generator the step iterates in turn is stepped the same way +//! ([`Interpreter::gen_fast_next`]). If that inner step stops partway, the +//! inner generator is parked having consumed its sent value +//! ([`crate::Frame::sent_consumed`]), and the outer step stops before its +//! `FOR_ITER`, which the general loop then runs (resuming the inner one +//! without pushing another value). + +use weavepy_compiler::{BinOpKind, CodeObject, CompareKind, OpCode, COMPARE_OP_TO_BOOL_FLAG}; + +use crate::object::{GeneratorState, Object, PyGenerator, PyIterator}; +use crate::sync::Rc; +use crate::{FoldSink, Frame, Interpreter}; + +/// How a fast step ended. +pub(crate) enum GenStep { + /// The body yielded this value; the frame is suspended past the yield. + Yielded(Object), + /// The next instruction needs the general loop. + Bail, +} + +/// How a fast `next()` of a generator ended. +pub(crate) enum GenNext { + Yielded(Object), + /// Nothing happened: the generator is as it was. + Declined, + /// The generator consumed its sent value and advanced, then stopped + /// short of a yield; it's parked for the general loop to continue. + Partial, +} + +/// `FOR_ITER`'s outcome in a fast step. +enum ForNext { + Value(Object), + Exhausted, + Bail, +} + +/// Nested fast steps at most this deep. +const MAX_DEPTH: u8 = 8; + +/// The instructions a fast step runs (see the module docs). The scan +/// also admits the generator prologue's `RETURN_GENERATOR` and the +/// implicit PEP 479 handler, which a resume or only an exception reaches. +fn op_supported(op: OpCode) -> bool { + matches!( + op, + OpCode::Nop + | OpCode::NotTaken + | OpCode::Resume + | OpCode::PopTop + | OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::LoadFastLoadFast + | OpCode::LoadFastBorrowLoadFastBorrow + | OpCode::StoreFast + | OpCode::LoadSmallInt + | OpCode::LoadConst + | OpCode::BinaryOp + | OpCode::CompareOp + | OpCode::ToBool + | OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::JumpForward + | OpCode::JumpBackward + | OpCode::GetIter + | OpCode::ForIter + | OpCode::EndFor + | OpCode::PopIter + | OpCode::YieldValue + | OpCode::ReturnValue + | OpCode::ReturnGenerator + | OpCode::StopIterationError + | OpCode::CallIntrinsic1 + | OpCode::Reraise + ) +} + +/// Whether `code` is a plain generator whose every instruction is one a +/// fast step runs (cached in the code's extension table). +fn code_ok(code: &CodeObject) -> bool { + use std::sync::atomic::Ordering; + let Some(ext) = crate::code_vm_ext(code) else { + return false; + }; + match ext.gen_fast.load(Ordering::Relaxed) { + 1 => return false, + 2 => return true, + _ => {} + } + let ok = code.is_generator + && !code.is_coroutine + && !code.is_async_generator + && code.cellvars.is_empty() + && code.freevars.is_empty() + && code.instructions.iter().all(|i| op_supported(i.op)); + ext.gen_fast + .store(if ok { 2 } else { 1 }, Ordering::Relaxed); + ok +} + +/// A machine scalar: its release owes nothing. +#[inline(always)] +fn scalar(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) +} + +/// A value's copy for the operand stack: scalars inline, anything else a +/// new reference. +#[inline(always)] +fn copy(v: &Object) -> Object { + match v { + Object::Int(x) => Object::Int(*x), + Object::Float(x) => Object::Float(*x), + Object::Bool(x) => Object::Bool(*x), + Object::None => Object::None, + other => crate::clone_hot(other), + } +} + +/// The result of `a b` for machine scalars, when it can't raise or +/// leave the machine range. +fn scalar_binop(kind: BinOpKind, a: &Object, b: &Object) -> Option { + Some(match (a, b) { + (Object::Int(a), Object::Int(b)) => { + let (a, b) = (*a, *b); + Object::Int(match kind { + BinOpKind::Add => a.checked_add(b)?, + BinOpKind::Sub => a.checked_sub(b)?, + BinOpKind::Mult => a.checked_mul(b)?, + BinOpKind::BitAnd => a & b, + BinOpKind::BitOr => a | b, + BinOpKind::BitXor => a ^ b, + BinOpKind::FloorDiv if b != 0 && !(a == i64::MIN && b == -1) => { + a.div_euclid(b) - i64::from(b < 0 && a.rem_euclid(b) != 0) + } + BinOpKind::Mod if b != 0 => { + let r = a.checked_rem(b)?; + if r != 0 && (r < 0) != (b < 0) { + r + b + } else { + r + } + } + BinOpKind::RShift if (0..64).contains(&b) => a >> b, + BinOpKind::LShift if (0..63).contains(&b) => { + let r = a.checked_shl(b as u32)?; + if r >> b != a { + return None; + } + r + } + _ => return None, + }) + } + (Object::Float(_) | Object::Int(_), Object::Float(_) | Object::Int(_)) => { + let f = |o: &Object| match o { + Object::Float(x) => Some(*x), + // Exactly representable ints only (Python converts exactly + // or raises for huge ones; both stay on the general path). + Object::Int(i) if i.unsigned_abs() < (1 << 53) => Some(*i as f64), + _ => None, + }; + let (a, b) = (f(a)?, f(b)?); + Object::Float(match kind { + BinOpKind::Add => a + b, + BinOpKind::Sub => a - b, + BinOpKind::Mult => a * b, + BinOpKind::Div if b != 0.0 => a / b, + _ => return None, + }) + } + _ => return None, + }) +} + +/// `a b` for machine scalars (NaN and mixed shapes decline). +fn scalar_compare(kind: CompareKind, a: &Object, b: &Object) -> Option { + let ord = match (a, b) { + (Object::Int(a), Object::Int(b)) => a.cmp(b), + (Object::Float(a), Object::Float(b)) => a.partial_cmp(b)?, + _ => return None, + }; + Some(match kind { + CompareKind::Lt => ord.is_lt(), + CompareKind::LtE => ord.is_le(), + CompareKind::Eq => ord.is_eq(), + CompareKind::NotEq => ord.is_ne(), + CompareKind::Gt => ord.is_gt(), + CompareKind::GtE => ord.is_ge(), + }) +} + +/// The truth of a scalar (`None` for anything else). +#[inline(always)] +fn scalar_truth(v: &Object) -> Option { + Some(match v { + Object::Bool(b) => *b, + Object::Int(i) => *i != 0, + Object::None => false, + _ => return None, + }) +} + +impl Interpreter { + /// Whether `frame`, a suspended generator's, may take fast steps: a + /// fast-step body the general machinery holds no extra state for (no + /// Python-visible frame, no saved exception state, no parked native + /// activation). + #[inline] + pub(crate) fn gen_fast_frame_ok(frame: &Frame) -> bool { + frame.py_frame.is_none() + && frame.saved_exc_info.is_empty() + && frame.pc != 0 + && !frame.shell_cache.as_ref().is_some_and(|c| { + c.has_materialized + .load(std::sync::atomic::Ordering::Relaxed) + }) + && { + #[cfg(feature = "jit")] + { + frame.parked_native.is_none() + } + #[cfg(not(feature = "jit"))] + { + true + } + } + && code_ok(&frame.code) + } + + /// Run `frame` from its pc to its next `yield` (see the module docs). + /// `snap_gen` is the caller's quiet-loop generation; `depth` counts + /// the fast steps this one is nested in. + /// + /// The loop works on raw views of the frame's operand stack and + /// locals (as the core loop does), writing the stack length and pc + /// back when it ends. + /// + /// With `fold` (a draining consumer's sink, top level only), a yield + /// the sink takes resumes the body at once with `None` sent. + pub(crate) fn gen_fast_step( + &mut self, + frame: &mut Frame, + snap_gen: u64, + depth: u8, + fold: Option, + ) -> GenStep { + // SAFETY: the frame's code is immutable and outlives the step (the + // frame holds it, and nothing here replaces it). + let code: &CodeObject = unsafe { &*Rc::as_ptr(&frame.code) }; + let (instrs, ninstrs) = (code.instructions.as_ptr(), code.instructions.len()); + let Some(ext) = crate::code_vm_ext(code) else { + return GenStep::Bail; + }; + let (cbase, nconsts) = (ext.objects.as_ptr(), ext.objects.len()); + // SAFETY: no guard is live on the locals (`peek_mut` checks), and + // nothing below runs code that could reach them. + let Some(locals) = (unsafe { frame.locals.peek_mut() }) else { + return GenStep::Bail; + }; + let (lbase, nlocals) = (locals.as_mut_ptr(), locals.len()); + let stack = &mut frame.stack; + if stack.capacity() - stack.len() < 8 { + stack.reserve(8); + } + let (base, cap) = (stack.as_mut_ptr(), stack.capacity()); + let mut len = stack.len(); + let mut pc = frame.pc as usize; + // SAFETY (throughout): `lbase`/`base`/`cbase` index only below + // `nlocals`/`len` (initialized) or `cap`/`nconsts` as checked, and + // `instrs` only below `ninstrs`. + let yielded = loop { + if pc >= ninstrs { + break None; + } + let ins = unsafe { *instrs.add(pc) }; + match ins.op { + OpCode::Nop | OpCode::NotTaken | OpCode::Resume => pc += 1, + OpCode::PopTop => { + if len == 0 { + break None; + } + let top = unsafe { &*base.add(len - 1) }; + if !scalar(top) { + if !Self::core_droppable(top) { + break None; + } + crate::drop_hot(unsafe { base.add(len - 1).read() }); + } + len -= 1; + pc += 1; + } + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { + let i = ins.arg as usize; + if i >= nlocals || len == cap { + break None; + } + let v = unsafe { &*lbase.add(i) }; + if matches!(v, Object::Unbound | Object::Cell(_)) { + break None; + } + unsafe { base.add(len).write(copy(v)) }; + len += 1; + pc += 1; + } + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + let (i, j) = ((ins.arg >> 4) as usize, (ins.arg & 15) as usize); + if i >= nlocals || j >= nlocals || len + 2 > cap { + break None; + } + let (a, b) = unsafe { (&*lbase.add(i), &*lbase.add(j)) }; + if matches!(a, Object::Unbound | Object::Cell(_)) + || matches!(b, Object::Unbound | Object::Cell(_)) + { + break None; + } + unsafe { + base.add(len).write(copy(a)); + base.add(len + 1).write(copy(b)); + } + len += 2; + pc += 1; + } + OpCode::StoreFast => { + let i = ins.arg as usize; + if i >= nlocals || len == 0 { + break None; + } + let slot = unsafe { lbase.add(i) }; + let old = unsafe { &*slot }; + if matches!(unsafe { &*base.add(len - 1) }, Object::Cell(_)) { + break None; + } + if scalar(old) || matches!(old, Object::Unbound) { + // Nothing to release. + len -= 1; + unsafe { slot.write(base.add(len).read()) }; + } else { + if !Self::core_droppable(old) { + break None; + } + len -= 1; + unsafe { crate::drop_hot(std::ptr::replace(slot, base.add(len).read())) }; + } + pc += 1; + } + OpCode::LoadSmallInt => { + if len == cap { + break None; + } + unsafe { base.add(len).write(Object::Int(i64::from(ins.arg))) }; + len += 1; + pc += 1; + } + OpCode::LoadConst => { + let i = ins.arg as usize; + if i >= nconsts || len == cap { + break None; + } + unsafe { base.add(len).write(copy(&*cbase.add(i))) }; + len += 1; + pc += 1; + } + OpCode::BinaryOp => { + if len < 2 { + break None; + } + // SAFETY: `BinOpKind` is `repr(u8)` and the compiler only + // emits valid kinds (as the core loop's arm). + let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; + let (a, b) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + let Some(r) = scalar_binop(kind, a, b) else { + break None; + }; + // Both operands are scalars (no drop owed). + len -= 1; + unsafe { base.add(len - 1).write(r) }; + pc += 1; + } + OpCode::CompareOp => { + if len < 2 { + break None; + } + let kind = match ins.arg & !COMPARE_OP_TO_BOOL_FLAG { + x if x == CompareKind::Lt as u32 => CompareKind::Lt, + x if x == CompareKind::LtE as u32 => CompareKind::LtE, + x if x == CompareKind::Eq as u32 => CompareKind::Eq, + x if x == CompareKind::NotEq as u32 => CompareKind::NotEq, + x if x == CompareKind::Gt as u32 => CompareKind::Gt, + x if x == CompareKind::GtE as u32 => CompareKind::GtE, + _ => break None, + }; + let (a, b) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + let Some(r) = scalar_compare(kind, a, b) else { + break None; + }; + len -= 1; + unsafe { base.add(len - 1).write(Object::Bool(r)) }; + pc += 1; + } + OpCode::ToBool => { + let Some(t) = (len > 0) + .then(|| unsafe { &*base.add(len - 1) }) + .and_then(scalar_truth) + else { + break None; + }; + unsafe { base.add(len - 1).write(Object::Bool(t)) }; + pc += 1; + } + OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue => { + let Some(t) = (len > 0) + .then(|| unsafe { &*base.add(len - 1) }) + .and_then(scalar_truth) + else { + break None; + }; + len -= 1; + pc += 1; + if t == (ins.op == OpCode::PopJumpIfTrue) { + pc += ins.arg as usize; + } + } + OpCode::JumpForward => pc += 1 + ins.arg as usize, + OpCode::JumpBackward => { + // The back edge is the eval-breaker (as the core loop's). + if self.gil_countdown <= 1 || crate::hot_gates::loop_gen() != snap_gen { + break None; + } + self.gil_countdown -= 1; + pc = (pc + 1).saturating_sub(ins.arg as usize); + } + // `iter()` of an iterator or a generator is itself. + OpCode::GetIter => { + if len == 0 + || !matches!( + unsafe { &*base.add(len - 1) }, + Object::Iter(_) | Object::Generator(_) + ) + { + break None; + } + pc += 1; + } + OpCode::ForIter => { + if len == 0 || len == cap { + break None; + } + let top = unsafe { &*base.add(len - 1) }; + // A live range's next value in line (the commonest loop). + if let Object::Iter(it) = top { + // SAFETY: nothing runs code while the view is held. + if let Some(PyIterator::Range { + current, + stop, + step, + }) = unsafe { it.peek_mut() } + { + if *step > 0 && *current < *stop { + let v = *current; + *current = current.wrapping_add(*step); + unsafe { base.add(len).write(Object::Int(v)) }; + len += 1; + pc += 1; + continue; + } + } + } + match self.gen_fast_for_iter(top, snap_gen, depth) { + ForNext::Value(v) => { + unsafe { base.add(len).write(v) }; + len += 1; + pc += 1; + } + ForNext::Exhausted => { + // The iterator (its sole owner is the stack) + // leaves, and the loop exits past its + // `END_FOR`/`POP_ITER` pair. + len -= 1; + crate::drop_hot(unsafe { base.add(len).read() }); + pc += 1 + ins.arg as usize; + let op_at = + |pc: usize| (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }); + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!(op_at(pc), Some(OpCode::PopIter | OpCode::PopTop)) { + pc += 1; + } + } + } + ForNext::Bail => break None, + } + } + OpCode::YieldValue => { + if len == 0 { + break None; + } + pc += 1; + // SAFETY: the slot is initialized; a folded value moves + // into the sink (or is a scalar), and the sent `None` + // takes its place. + let top = unsafe { base.add(len - 1) }; + let folded = match fold { + // SAFETY: the sink is the consumer's own, live and + // untouched while its resume runs (as the core + // loop's fold). + Some(FoldSink::Sum(acc)) => unsafe { (*acc).add_scalar(&*top) }, + Some(FoldSink::Collect(out)) => { + unsafe { (*out).push(top.read()) }; + true + } + None => false, + }; + if folded { + unsafe { top.write(Object::None) }; + continue; + } + len -= 1; + frame.agen_yielded_value = ins.arg == 0; + break Some(unsafe { top.read() }); + } + _ => break None, + } + }; + // SAFETY: the first `len` slots are initialized (pushes wrote them, + // pops moved values out). + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + match yielded { + Some(v) => GenStep::Yielded(v), + None => GenStep::Bail, + } + } + + /// `FOR_ITER`'s step on `top` for a fast step (out of line: the + /// loop keeps its registers): a range, list or tuple iterator's next + /// value, a range's end (when the loop alone holds it), or a fast + /// generator's next yield. + #[inline(never)] + fn gen_fast_for_iter(&mut self, top: &Object, snap_gen: u64, depth: u8) -> ForNext { + match top { + Object::Iter(it) => { + let unique = Rc::strong_count(it) == 1; + // SAFETY: nothing below runs code until `it`'s last use + // (the guard-free `peek`). + let Some(it) = (unsafe { it.peek_mut() }) else { + return ForNext::Bail; + }; + match it { + PyIterator::Range { + current, + stop, + step, + } => { + let live = if *step > 0 { + *current < *stop + } else { + *step < 0 && *current > *stop + }; + if live { + let v = *current; + *current = current.wrapping_add(*step); + ForNext::Value(Object::Int(v)) + } else if unique { + ForNext::Exhausted + } else { + ForNext::Bail + } + } + PyIterator::List { items, index, .. } => { + // SAFETY: as above. + let Some(v) = (unsafe { items.peek() }).and_then(|xs| xs.get(*index)) + else { + return ForNext::Bail; + }; + let v = copy(v); + *index += 1; + ForNext::Value(v) + } + PyIterator::Tuple { items, index } => { + let Some(v) = items.get(*index) else { + return ForNext::Bail; + }; + let v = copy(v); + *index += 1; + ForNext::Value(v) + } + _ => ForNext::Bail, + } + } + Object::Generator(g) => { + let g = g.clone(); + match self.gen_fast_next(&g, snap_gen, depth + 1) { + GenNext::Yielded(v) => ForNext::Value(v), + GenNext::Declined | GenNext::Partial => ForNext::Bail, + } + } + _ => ForNext::Bail, + } + } + + /// A resume's fast steps (its sent value already pushed): the value + /// the body yields next, with a draining consumer's `fold` taking the + /// yields it can along the way, or `None` for the general loop to + /// continue from wherever the steps stopped. + pub(crate) fn gen_fast_run( + &mut self, + frame: &mut Frame, + snap_gen: u64, + fold: Option, + ) -> Option { + if !Self::gen_fast_frame_ok(frame) { + return None; + } + match self.gen_fast_step(frame, snap_gen, 0, fold) { + GenStep::Yielded(v) => Some(v), + GenStep::Bail => None, + } + } + + /// `next(g)` by fast step (see the module docs): resumes `g` with + /// `None` and runs it to its next yield when both it and its body + /// allow. `depth` counts the fast steps this one is nested in. + pub(crate) fn gen_fast_next( + &mut self, + g: &Rc, + snap_gen: u64, + depth: u8, + ) -> GenNext { + if depth > MAX_DEPTH + || crate::recursion::current_depth() + usize::from(depth) + 2 + >= crate::recursion::recursion_limit() + { + return GenNext::Declined; + } + // Validate and take the frame under one exclusive view; no Python + // runs while it's held. + // SAFETY: nothing below reaches the cell again until `state`'s + // last use (`peek_mut` rejects a live guard or shared cells). + let Some(state) = (unsafe { g.state.peek_mut() }) else { + return GenNext::Declined; + }; + let (GeneratorState::Suspended(boxed) | GeneratorState::Created(boxed)) = &*state else { + return GenNext::Declined; + }; + let first_resume = matches!(*state, GeneratorState::Created(_)); + if boxed.sent_consumed || !Self::gen_fast_frame_ok(boxed) { + return GenNext::Declined; + } + let prev = std::mem::replace(state, GeneratorState::Running); + let (GeneratorState::Suspended(mut boxed) | GeneratorState::Created(mut boxed)) = prev + else { + unreachable!("checked above"); + }; + let frame: &mut Frame = &mut boxed; + frame.gen_first_resume = first_resume; + let start = frame.pc; + frame.stack.push(Object::None); + let out = match self.gen_fast_step(frame, snap_gen, depth, None) { + GenStep::Yielded(v) => GenNext::Yielded(v), + GenStep::Bail if frame.pc == start => { + // Nothing ran: the resume is undone. + frame.stack.pop(); + GenNext::Declined + } + GenStep::Bail => { + frame.sent_consumed = true; + GenNext::Partial + } + }; + Self::park_suspended_boxed(g, boxed); + out + } +} diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index eedb7760..40763d9f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -54,6 +54,7 @@ pub mod foreign; pub mod frozen_code_cache; pub mod frozen_table; pub mod gc_trace; +mod gen_fast; pub mod gil; pub mod hot_filter; pub mod hot_gates; @@ -234,6 +235,10 @@ pub struct Frame { /// for later resumptions. Set by `generator_send` when unparking a /// `Created` frame; consumed (reset) by `run_frame`'s entry event. gen_first_resume: bool, + /// A suspended generator frame whose last resume already consumed its + /// sent value (a fast step stopped partway; see `gen_fast`): the next + /// resume pushes none. + sent_consumed: bool, /// RFC 0061 (WS3c): the generator-family activation's `FrameShell`, /// kept across suspensions. A suspended frame survives inside the /// generator object, so its shell — whose immutable fields (code, @@ -5714,6 +5719,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -9881,6 +9887,7 @@ impl Interpreter { fr.agen_yielded_value = true; fr.suppress_call_event = false; fr.gen_first_resume = false; + fr.sent_consumed = false; #[cfg(feature = "jit")] { fr.parked_native = None; @@ -9973,7 +9980,10 @@ impl Interpreter { #[cfg(feature = "jit")] crate::tier2::materialize_parked(gf); gf.gen_first_resume = first_resume; - gf.push(Object::None); + // (A fast step that stopped partway already consumed it.) + if !std::mem::take(&mut gf.sent_consumed) { + gf.push(Object::None); + } let gen_frame: *mut Frame = gf; frame.pc = pc as u32 + 1; if next_call { @@ -11432,6 +11442,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -13036,7 +13047,21 @@ impl Interpreter { } // A generator resumes inline, switched to in place // (the quiet loop's lean path when it declines). - Object::Generator(_) => { + Object::Generator(g) => { + // A simple body runs to its next yield in + // place, without switching (see `gen_fast`). + let g = g.clone(); + if let gen_fast::GenNext::Yielded(v) = + self.gen_fast_next(&g, snap_gen, 0) + { + // SAFETY: `len < cap` (checked above). + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + drop(g); // SAFETY: `len <= cap`, every slot initialized. unsafe { frame.stack.set_len(len) }; frame.pc = pc as u32; @@ -36374,6 +36399,8 @@ impl Interpreter { // exception machinery below both read them. #[cfg(feature = "jit")] crate::tier2::materialize_parked(&mut frame); + // A frame a fast step left partway takes the exception at its pc. + frame.sent_consumed = false; // PEP 3134: an exception thrown into a generator suspended inside // an `except`/`with` block chains to the exception that block was // handling. Done before delegation/handling so the `__context__` @@ -37135,22 +37162,39 @@ impl Interpreter { #[cfg(feature = "jit")] crate::tier2::materialize_parked(frame); frame.gen_first_resume = first_resume; - frame.push(sent.clone()); - // The draining consumer's sink, folded at this frame's yields. - let prev_fold = std::mem::replace( - &mut self.sum_fold, - fold.map(|sink| (std::ptr::from_mut(frame) as usize, sink)), - ); - let exc_depth_on_entry = self.exc_info_len(); - let fin_live = gc_trace::has_any_finalizable(); - let fin = fin_live || gc_trace::active_suspects_present(); - let mut act = LeanAct { - frame: std::ptr::from_mut(frame), - shell: None, - }; - let mut prev_pc: Option = None; - let run = if guard.depth() % 4 == 0 { - stacker::maybe_grow(512 * 1024, 8 * 1024 * 1024, || { + if !std::mem::take(&mut frame.sent_consumed) { + frame.push(sent.clone()); + } + // A simple body runs to its next yield in place (see + // `gen_fast`), folding the draining consumer's yields as it goes. + if let Some(v) = self.gen_fast_run(frame, snap_gen, fold) { + Ok(FrameOutcome::Yielded(v)) + } else { + // The draining consumer's sink, folded at this frame's yields. + let prev_fold = std::mem::replace( + &mut self.sum_fold, + fold.map(|sink| (std::ptr::from_mut(frame) as usize, sink)), + ); + let exc_depth_on_entry = self.exc_info_len(); + let fin_live = gc_trace::has_any_finalizable(); + let fin = fin_live || gc_trace::active_suspects_present(); + let mut act = LeanAct { + frame: std::ptr::from_mut(frame), + shell: None, + }; + let mut prev_pc: Option = None; + let run = if guard.depth() % 4 == 0 { + stacker::maybe_grow(512 * 1024, 8 * 1024 * 1024, || { + self.quiet_run( + frame, + QuietShell::Lazy(&mut act), + snap_gen, + fin, + fin_live, + &mut prev_pc, + ) + }) + } else { self.quiet_run( frame, QuietShell::Lazy(&mut act), @@ -37159,53 +37203,44 @@ impl Interpreter { fin_live, &mut prev_pc, ) - }) - } else { - self.quiet_run( - frame, - QuietShell::Lazy(&mut act), - snap_gen, - fin, - fin_live, - &mut prev_pc, - ) - }; - self.sum_fold = prev_fold; - match (run, act.shell) { - ( - QuietExit::Outcome { - stepped: Ok(StepOutcome::Yield(v)), - .. - }, - None, - ) => Ok(FrameOutcome::Yielded(v)), - ( - QuietExit::Outcome { - stepped: Ok(StepOutcome::Return(v)), - .. - }, - None, - ) => Ok(FrameOutcome::Returned(v)), - // Everything else finishes in the general loop (which - // saves a suspended generator's handler state, as the - // general prologue's twin epilogue does). - (exit, shell) => { - let shell = match shell { - Some(shell) => shell, - None => { - self.flush_pending_callers(); - let shell = self.push_frame_shell(frame); - if frame.shell_cache.is_none() { - frame.shell_cache = Some(shell.clone()); + }; + self.sum_fold = prev_fold; + match (run, act.shell) { + ( + QuietExit::Outcome { + stepped: Ok(StepOutcome::Yield(v)), + .. + }, + None, + ) => Ok(FrameOutcome::Yielded(v)), + ( + QuietExit::Outcome { + stepped: Ok(StepOutcome::Return(v)), + .. + }, + None, + ) => Ok(FrameOutcome::Returned(v)), + // Everything else finishes in the general loop (which + // saves a suspended generator's handler state, as the + // general prologue's twin epilogue does). + (exit, shell) => { + let shell = match shell { + Some(shell) => shell, + None => { + self.flush_pending_callers(); + let shell = self.push_frame_shell(frame); + if frame.shell_cache.is_none() { + frame.shell_cache = Some(shell.clone()); + } + shell } - shell - } - }; - let pending = match exit { - QuietExit::Yield => None, - QuietExit::Outcome { stepped, cur_pc } => Some((stepped, cur_pc)), - }; - self.run_activation(frame, shell, None, exc_depth_on_entry, false, pending) + }; + let pending = match exit { + QuietExit::Yield => None, + QuietExit::Outcome { stepped, cur_pc } => Some((stepped, cur_pc)), + }; + self.run_activation(frame, shell, None, exc_depth_on_entry, false, pending) + } } } }; @@ -37370,7 +37405,14 @@ impl Interpreter { // CPython 3.13 prologue: RETURN_GENERATOR / POP_TOP / RESUME — // *every* resume pushes the sent value; the first activation's // None lands in the prologue's POP_TOP. - let sent_for_frame = Some(if first_resume { Object::None } else { sent }); + // (A fast step that stopped partway already consumed it.) + let sent_for_frame = (!std::mem::take(&mut frame.sent_consumed)).then(|| { + if first_resume { + Object::None + } else { + sent + } + }); self.run_until_yield_or_return(frame, sent_for_frame) }; match outcome { @@ -50411,6 +50453,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -57801,6 +57844,7 @@ impl InlineAct { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -63117,6 +63161,9 @@ struct CodeConstObjects { /// splits them back apart; the core loop reads one byte to run the /// pair as a single dispatch. Empty when the code has no pair. fast_pairs: std::sync::OnceLock>, + /// Whether the code is a generator body fast steps run (see + /// `gen_fast`): 0 not yet scanned, 1 no, 2 yes. + gen_fast: std::sync::atomic::AtomicU8, /// Split-layout attribute shortcuts per `LOAD_ATTR` site (see /// [`FieldSlot`]); allocated on the first recorded one. field_slots: std::sync::OnceLock>, @@ -64439,6 +64486,7 @@ fn code_vm_ext_build( call_slots: std::sync::OnceLock::new(), pure_leaf: std::sync::atomic::AtomicU8::new(0), fast_pairs: std::sync::OnceLock::new(), + gen_fast: std::sync::atomic::AtomicU8::new(0), field_slots: std::sync::OnceLock::new(), leaf_sites: std::sync::OnceLock::new(), leaf_plan: std::sync::OnceLock::new(), diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 75aba9fd..0d170141 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -4920,6 +4920,7 @@ fn finish_deopted( pending_lasti: None, suppress_call_event: true, gen_first_resume: false, + sent_consumed: false, shell_cache: None, parked_native: None, }; diff --git a/tests/regrtest/test_generator_fast_steps.py b/tests/regrtest/test_generator_fast_steps.py new file mode 100644 index 00000000..f96ede81 --- /dev/null +++ b/tests/regrtest/test_generator_fast_steps.py @@ -0,0 +1,201 @@ +"""Generators whose bodies take fast steps (see `gen_fast` in the VM). + +Each case checks a behavior at the boundary between the fast steps and +the general loop: overflow into big integers, raises, returns, sends, +throws and closes after partial iteration, nesting, and folding +consumers. +""" + +import sys +import unittest + + +def counter(n): + i = 0 + while i < n: + yield i + i += 1 + + +def squares(it): + for x in it: + yield x * x + + +def evens(it): + for x in it: + if x & 1 == 0: + yield x + + +def over_range(n): + for i in range(n): + yield i * 3 + + +def over_list(xs): + for x in xs: + yield x + 1 + + +def doubling(start, n): + x = start + for _ in range(n): + yield x + x = x * 2 + + +def floats(n): + x = 0.5 + i = 0 + while i < n: + yield x + x = x * 1.5 + i += 1 + + +def divider(xs): + for x in xs: + yield 10 // x + + +def returning(n): + i = 0 + while i < n: + yield i + i += 1 + return "done" + + +class FastStepTests(unittest.TestCase): + def test_counter_and_pipeline(self): + self.assertEqual(list(counter(5)), [0, 1, 2, 3, 4]) + self.assertEqual(list(evens(squares(counter(10)))), [0, 4, 16, 36, 64]) + self.assertEqual(sum(evens(squares(counter(1000)))), sum( + x * x for x in range(1000) if (x * x) % 2 == 0)) + total = 0 + for x in squares(counter(100)): + total += x + self.assertEqual(total, sum(x * x for x in range(100))) + + def test_iterables_inside(self): + self.assertEqual(list(over_range(5)), [0, 3, 6, 9, 12]) + self.assertEqual(list(over_list([1, 2, 3])), [2, 3, 4]) + self.assertEqual(list(over_list((1.5, 2.5))), [2.5, 3.5]) + self.assertEqual(list(over_list([])), []) + + def test_overflow_to_big_integers(self): + values = list(doubling(1 << 60, 8)) + self.assertEqual(values, [(1 << 60) << k for k in range(8)]) + self.assertEqual(list(squares(doubling(3 << 30, 4))), + [(3 << 30 << k) ** 2 for k in range(4)]) + + def test_floats(self): + self.assertEqual(list(floats(4)), [0.5, 0.75, 1.125, 1.6875]) + + def test_raise_midway(self): + g = divider([5, 2, 0, 1]) + self.assertEqual(next(g), 2) + self.assertEqual(next(g), 5) + with self.assertRaises(ZeroDivisionError): + next(g) + with self.assertRaises(StopIteration): + next(g) + + def test_return_value(self): + g = returning(2) + self.assertEqual(next(g), 0) + self.assertEqual(next(g), 1) + with self.assertRaises(StopIteration) as cm: + next(g) + self.assertEqual(cm.exception.value, "done") + + def test_send_and_throw_after_fast_steps(self): + g = counter(10) + self.assertEqual(next(g), 0) + self.assertEqual(next(g), 1) + self.assertEqual(g.send(None), 2) + with self.assertRaises(KeyError): + g.throw(KeyError("k")) + self.assertIsNone(g.gi_frame) + + def test_close_after_fast_steps(self): + g = squares(counter(10)) + self.assertEqual([next(g), next(g), next(g)], [0, 1, 4]) + g.close() + with self.assertRaises(StopIteration): + next(g) + + def test_frame_inspection(self): + g = counter(10) + next(g) + next(g) + frame = g.gi_frame + self.assertEqual(frame.f_locals["i"], 1) + self.assertEqual(next(g), 2) + self.assertEqual(frame.f_locals["i"], 2) + self.assertFalse(g.gi_running) + + def test_nested_inner_partial(self): + # The inner generator overflows partway through a fast step of + # the outer one. + def outer(it): + for x in it: + yield x + 1 + + self.assertEqual(list(outer(doubling(1 << 61, 4))), + [(1 << 61 << k) + 1 for k in range(4)]) + g = outer(returning(3)) + self.assertEqual(list(g), [1, 2, 3]) + + def test_yield_from_fast_inner(self): + def delegator(n): + r = yield from returning(n) + yield r + + self.assertEqual(list(delegator(3)), [0, 1, 2, "done"]) + + def test_already_executing(self): + def selfish(): + for x in g: + yield x + + g = selfish() + with self.assertRaises(ValueError): + next(g) + + def test_folding_consumers(self): + self.assertEqual(sum(counter(100)), 4950) + self.assertEqual(list(squares(counter(5))), [0, 1, 4, 9, 16]) + self.assertEqual(tuple(evens(counter(7))), (0, 2, 4, 6)) + self.assertEqual(sum(floats(3)), 0.5 + 0.75 + 1.125) + self.assertEqual(sum(doubling(1 << 62, 3)), (1 << 62) * 7) + + def test_tracing_sees_lines(self): + seen = [] + + def tracer(frame, event, arg): + if frame.f_code is counter.__code__ and event == "line": + seen.append(frame.f_lineno) + return tracer + + sys.settrace(tracer) + try: + list(counter(2)) + finally: + sys.settrace(None) + self.assertTrue(seen) + + def test_recursion_depth(self): + def chain(depth): + if depth == 0: + yield from counter(3) + return + for x in chain(depth - 1): + yield x + + self.assertEqual(list(chain(30)), [0, 1, 2]) + + +if __name__ == "__main__": + unittest.main() From 75d2023e31fd03d72ba69c68b13bffc3eec78713 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 20:41:01 -0700 Subject: [PATCH 51/65] perf: skip the fallible reservation on the pickle codec's hot pushes The decoder's push and the encoder's extend reserved capacity (a fallible call) before every append. They now reserve only when the buffer is full, and the encoder's frame writer inlines. pickle_bench: about 8% fewer instructions. --- crates/weavepy-vm/src/stdlib/pickle_accel.rs | 6 +++++- crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs | 7 ++++++- 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel.rs b/crates/weavepy-vm/src/stdlib/pickle_accel.rs index cc8e79aa..3f9d8c96 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel.rs @@ -609,8 +609,12 @@ trait Sink<'a> { fn build(&mut self, target: &Self::Value, state: Self::Value) -> Option<()>; } +#[inline(always)] fn push(items: &mut Vec, value: T) -> Option<()> { - items.try_reserve(1).ok()?; + // (The fallible reservation only when the vector is full.) + if items.len() == items.capacity() { + items.try_reserve(1).ok()?; + } items.push(value); Some(()) } diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs index 52b61377..92847caa 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs @@ -23,8 +23,12 @@ use crate::types::{PyInstance, TypeObject}; const FRAME_TARGET: usize = 65_536; const MAX_DEPTH: usize = 128; +#[inline(always)] fn extend(output: &mut Vec, data: &[u8]) -> Option<()> { - output.try_reserve(data.len()).ok()?; + // (The fallible reservation only when the buffer is short.) + if output.capacity() - output.len() < data.len() { + output.try_reserve(data.len()).ok()?; + } output.extend_from_slice(data); Some(()) } @@ -36,6 +40,7 @@ struct Framer { } impl Framer { + #[inline] fn write(&mut self, data: &[u8]) -> Option<()> { if self.frame_start.is_none() { let start = self.output.len(); From f4d9b37ed2947d0b7b7b3fbe2b618ff4a75fb959 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 21:56:14 -0700 Subject: [PATCH 52/65] perf: check generator fast-step bounds once per code object The eligibility scan now also proves every jump target, local index and constant index in range and that the last instruction can't fall through, so the step reads instructions, locals and constants without per-use checks. The scalar arithmetic and comparison helpers inline, which keeps the step's operand stack in registers. A for loop over a generator: 7% fewer instructions per item; sum() and list() over generator expressions: 5-9%. --- crates/weavepy-vm/src/gen_fast.rs | 75 ++++++++++++++++++++++++------- 1 file changed, 60 insertions(+), 15 deletions(-) diff --git a/crates/weavepy-vm/src/gen_fast.rs b/crates/weavepy-vm/src/gen_fast.rs index 01810f07..4540e6e3 100644 --- a/crates/weavepy-vm/src/gen_fast.rs +++ b/crates/weavepy-vm/src/gen_fast.rs @@ -113,12 +113,48 @@ fn code_ok(code: &CodeObject) -> bool { && !code.is_async_generator && code.cellvars.is_empty() && code.freevars.is_empty() - && code.instructions.iter().all(|i| op_supported(i.op)); + && code.instructions.iter().all(|i| op_supported(i.op)) + && in_bounds(code, ext.objects.len()); ext.gen_fast .store(if ok { 2 } else { 1 }, Ordering::Relaxed); ok } +/// Whether every jump, local and constant `code`'s instructions name is +/// in range, and the last instruction can't fall through: the step then +/// reads instructions, locals and constants without per-use checks. +fn in_bounds(code: &CodeObject, nconsts: usize) -> bool { + let instrs = &code.instructions; + let (n, nvars) = (instrs.len(), code.varnames.len()); + let terminal = |op| { + matches!( + op, + OpCode::ReturnValue | OpCode::Reraise | OpCode::JumpBackward | OpCode::JumpForward + ) + }; + if instrs.last().is_none_or(|i| !terminal(i.op)) { + return false; + } + instrs.iter().enumerate().all(|(p, i)| { + let arg = i.arg as usize; + match i.op { + OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::JumpForward + | OpCode::ForIter => p + 1 + arg < n, + OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::StoreFast => arg < nvars, + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + arg >> 4 < nvars && arg & 15 < nvars + } + OpCode::LoadConst => arg < nconsts, + _ => true, + } + }) +} + /// A machine scalar: its release owes nothing. #[inline(always)] fn scalar(v: &Object) -> bool { @@ -143,6 +179,7 @@ fn copy(v: &Object) -> Object { /// The result of `a b` for machine scalars, when it can't raise or /// leave the machine range. +#[inline(always)] fn scalar_binop(kind: BinOpKind, a: &Object, b: &Object) -> Option { Some(match (a, b) { (Object::Int(a), Object::Int(b)) => { @@ -198,6 +235,7 @@ fn scalar_binop(kind: BinOpKind, a: &Object, b: &Object) -> Option { } /// `a b` for machine scalars (NaN and mixed shapes decline). +#[inline(always)] fn scalar_compare(kind: CompareKind, a: &Object, b: &Object) -> Option { let ord = match (a, b) { (Object::Int(a), Object::Int(b)) => a.cmp(b), @@ -276,13 +314,18 @@ impl Interpreter { let Some(ext) = crate::code_vm_ext(code) else { return GenStep::Bail; }; - let (cbase, nconsts) = (ext.objects.as_ptr(), ext.objects.len()); + let cbase = ext.objects.as_ptr(); // SAFETY: no guard is live on the locals (`peek_mut` checks), and // nothing below runs code that could reach them. let Some(locals) = (unsafe { frame.locals.peek_mut() }) else { return GenStep::Bail; }; - let (lbase, nlocals) = (locals.as_mut_ptr(), locals.len()); + // (The scan checked every local index against the code's + // variables, which the frame's locals cover.) + if locals.len() < code.varnames.len() { + return GenStep::Bail; + } + let lbase = locals.as_mut_ptr(); let stack = &mut frame.stack; if stack.capacity() - stack.len() < 8 { stack.reserve(8); @@ -290,13 +333,15 @@ impl Interpreter { let (base, cap) = (stack.as_mut_ptr(), stack.capacity()); let mut len = stack.len(); let mut pc = frame.pc as usize; - // SAFETY (throughout): `lbase`/`base`/`cbase` index only below - // `nlocals`/`len` (initialized) or `cap`/`nconsts` as checked, and - // `instrs` only below `ninstrs`. + if pc >= ninstrs { + return GenStep::Bail; + } + // SAFETY (throughout): `base` indexes only below `len` (initialized) + // or `cap` as checked; `lbase`, `cbase` and `instrs` only at the + // indices and pcs the eligibility scan proved in range (`in_bounds`: + // every jump lands on an instruction and the last can't fall + // through, so `pc < ninstrs` whenever an instruction is read). let yielded = loop { - if pc >= ninstrs { - break None; - } let ins = unsafe { *instrs.add(pc) }; match ins.op { OpCode::Nop | OpCode::NotTaken | OpCode::Resume => pc += 1, @@ -316,7 +361,7 @@ impl Interpreter { } OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { let i = ins.arg as usize; - if i >= nlocals || len == cap { + if len == cap { break None; } let v = unsafe { &*lbase.add(i) }; @@ -329,7 +374,7 @@ impl Interpreter { } OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { let (i, j) = ((ins.arg >> 4) as usize, (ins.arg & 15) as usize); - if i >= nlocals || j >= nlocals || len + 2 > cap { + if len + 2 > cap { break None; } let (a, b) = unsafe { (&*lbase.add(i), &*lbase.add(j)) }; @@ -347,7 +392,7 @@ impl Interpreter { } OpCode::StoreFast => { let i = ins.arg as usize; - if i >= nlocals || len == 0 { + if len == 0 { break None; } let slot = unsafe { lbase.add(i) }; @@ -377,11 +422,10 @@ impl Interpreter { pc += 1; } OpCode::LoadConst => { - let i = ins.arg as usize; - if i >= nconsts || len == cap { + if len == cap { break None; } - unsafe { base.add(len).write(copy(&*cbase.add(i))) }; + unsafe { base.add(len).write(copy(&*cbase.add(ins.arg as usize))) }; len += 1; pc += 1; } @@ -680,6 +724,7 @@ impl Interpreter { frame.gen_first_resume = first_resume; let start = frame.pc; frame.stack.push(Object::None); + debug_assert!(!frame.stack.is_empty()); let out = match self.gen_fast_step(frame, snap_gen, depth, None) { GenStep::Yielded(v) => GenNext::Yielded(v), GenStep::Bail if frame.pc == start => { From a766bfb5c3b420c7e55a06441dc811b65304dd42 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Tue, 29 Sep 2026 23:22:50 -0700 Subject: [PATCH 53/65] perf: fold int-expression arguments into fused native method calls The core loop's fused `x.m()` step now folds an argument like `i + 1` (checked add, subtract, multiply, and bitwise ops on two ints) in place, and a local instance whose method is a native slot calls it directly before the leaf-site lookup. The deque fixture drops 8.4% in instructions, and `q.append(i + 1)` drops 31%. --- crates/weavepy-vm/src/lib.rs | 68 +++++++++++++++++++++++ tests/regrtest/test_fused_method_args.py | 69 ++++++++++++++++++++++++ 2 files changed, 137 insertions(+) create mode 100644 tests/regrtest/test_fused_method_args.py diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 40763d9f..547201dd 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12190,6 +12190,48 @@ impl Interpreter { && len < cap && simple_args_prefix(&code.instructions, pc + 2) { + // A native method the site cached + // for this class version: straight + // to its fused call (the Python + // leaf probes below would miss). + if let (Object::Instance(i), Some(ms)) = + (other, mslots!(cold_mslots, ext).get(pc + 1)) + { + if ms + .peek_inst_builtin( + i.cls_raw().attr_version.get(), + ) + .is_some() + { + if let Some((r, call_pc)) = self + .core_native_method( + code, + other, + pc + 1, + next.arg, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) + { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop( + LeafStop::Raised(e), + )); + } + } + } + } + } // A site that verified its callee // for this class version. let mut missed = false; @@ -15775,6 +15817,32 @@ impl Interpreter { // SAFETY: a borrowed copy of a live constant. OpCode::LoadConst => unsafe { std::ptr::read(consts.get(ins.arg as usize)?) }, OpCode::LoadSmallInt => Object::Int(i64::from(ins.arg)), + // `x.m(i + 1)`: an int operation on the two latest + // arguments folds into one (scalars: the copies own nothing). + OpCode::BinaryOp if n >= 3 => { + // SAFETY: entries `1..n` were written. + let (a, b) = + unsafe { (ops[n - 2].assume_init_ref(), ops[n - 1].assume_init_ref()) }; + let (&Object::Int(a), &Object::Int(b)) = (a, b) else { + return None; + }; + // SAFETY: `BinOpKind` is `repr(u8)` and the compiler only + // emits valid kinds. + let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; + let r = match kind { + BinOpKind::Add => a.checked_add(b)?, + BinOpKind::Sub => a.checked_sub(b)?, + BinOpKind::Mult => a.checked_mul(b)?, + BinOpKind::BitAnd => a & b, + BinOpKind::BitOr => a | b, + BinOpKind::BitXor => a ^ b, + _ => return None, + }; + ops[n - 2].write(Object::Int(r)); + n -= 1; + pc += 1; + continue; + } OpCode::Call if ins.arg as usize == n - 1 => return Some((n - 1, pc)), _ => return None, }; diff --git a/tests/regrtest/test_fused_method_args.py b/tests/regrtest/test_fused_method_args.py new file mode 100644 index 00000000..ee728d9f --- /dev/null +++ b/tests/regrtest/test_fused_method_args.py @@ -0,0 +1,69 @@ +"""Fused `x.m(...)` calls whose argument is a small int expression. + +The core loop runs `LOAD_FAST x; LOAD_ATTR m; ; CALL` as one step +for a local list, dict, set, str or native-method instance, folding an +argument like `i + 1` in place. These cases check the fold's edges: +overflow into big integers, non-int operands, and raising calls. +""" + +import sys +import unittest +from collections import deque + + +class FusedArgumentTests(unittest.TestCase): + def test_folded_int_arguments(self): + q = deque() + xs = [] + for i in range(5): + q.append(i + 1) + xs.append(i * 3) + xs.append(i - 10) + xs.append(i & 1) + xs.append(i | 8) + xs.append(i ^ 5) + self.assertEqual(list(q), [1, 2, 3, 4, 5]) + self.assertEqual(xs[:6], [0, -10, 0, 8, 5, 3]) + + def test_overflow_into_big_integers(self): + big = sys.maxsize + q = deque() + xs = [] + for i in range(big - 1, big + 1): + q.append(i + 1) + xs.append(i * 2) + self.assertEqual(list(q), [big, big + 1]) + self.assertEqual(xs, [(big - 1) * 2, big * 2]) + + def test_non_int_operands(self): + xs = [] + for x in (1.5, 2.5): + xs.append(x + 1) + s = "ab" + for t in ("c", "d"): + xs.append(s + t) + for b in (True, False): + xs.append(b + 1) + self.assertEqual(xs, [2.5, 3.5, "abc", "abd", 2, 1]) + + def test_raising_calls(self): + q = deque() + with self.assertRaises(IndexError): + for i in range(3): + q.pop() + d = {} + for i in range(3): + self.assertIsNone(d.get(i + 1)) + with self.assertRaises(TypeError): + for i in range(3): + q.append(i + 1, 2) + + def test_maxlen_trim_with_folded_argument(self): + ring = deque(maxlen=3) + for i in range(10): + ring.append(i + 100) + self.assertEqual(list(ring), [107, 108, 109]) + + +if __name__ == "__main__": + unittest.main() From 38f3c90ab376bf0f37d998a74525256cab6a9ffd Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:02:48 -0700 Subject: [PATCH 54/65] perf: run more constructors and nested leaf calls without frames - A class whose __init__ takes trailing default parameters now takes the frameless constructor paths (and the lean framed one): the call fills the missing arguments from the function's defaults. - Leaf plans allocate empty list and dict literals, so an __init__ like `self.items = []` stays a leaf. - A leaf plan runs a pure-leaf callee as a frame of its own evaluation, saving only the caller's live registers, instead of re-entering the plan runner. - The GC's finalizer check on tracking uses the class's cached __del__ verdict rather than an MRO lookup per instance. Constructing a DeltaBlue Variable drops from 12.5K to 7.4K instructions, and building its constraint chain drops 7%. --- crates/weavepy-vm/src/gc_trace.rs | 4 +- crates/weavepy-vm/src/leaf_plan.rs | 163 +++++++++++++++++++++-- crates/weavepy-vm/src/lib.rs | 64 +++++++-- tests/regrtest/test_leaf_constructors.py | 145 ++++++++++++++++++++ 4 files changed, 352 insertions(+), 24 deletions(-) create mode 100644 tests/regrtest/test_leaf_constructors.py diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 88ad7e91..99dcd3b0 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -3259,7 +3259,9 @@ fn run_finalizer(obj: &Object) { /// blocks — CPython's `gen_dealloc` behavior). fn has_finalizer(obj: &Object) -> bool { match obj { - Object::Instance(inst) => inst.cls().lookup("__del__").is_some(), + // The class's cached `__del__` verdict (reset whenever `__del__` + // or the MRO changes), not an MRO walk per tracked instance. + Object::Instance(inst) => inst.cls().instances_need_finalize(), // RFC 0065 (WS4, item 3) tried gating this on "close can run // user code" (empty exception table ⇒ skip enrollment) so // `yield`-loop workloads could reach the fully-quiet dispatch diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index b07f5960..f946c16f 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -197,6 +197,12 @@ enum Op { Return { src: u8, }, + /// `regs[dst]` a new empty list (`BUILD_LIST 0`) or, with `dict`, + /// an empty dict (`BUILD_MAP 0`), held by the owned scratch. + New { + dst: u8, + dict: bool, + }, /// Abandon the evaluation: the path reached an instruction the plan /// can't run. Decline, @@ -217,6 +223,8 @@ pub(crate) struct LeafPlan { consts: Box<[V]>, /// Arguments, copied into the first registers on entry. nargs: u8, + /// Registers the plan uses (the locals, then its deepest stack). + nregs: u8, /// Every attribute store goes to the first argument (never /// reassigned), each to a different name: the buffered stores share /// one receiver and each is the latest to its attribute. @@ -257,11 +265,14 @@ struct Builder<'a> { fixups: Vec<(usize, usize)>, /// The op index each bytecode instruction starts at. starts: Vec, + /// One past the highest register handed out. + top: std::cell::Cell, } impl Builder<'_> { fn slot(&self, pos: usize) -> Option { let r = self.nl + pos; + self.top.set(self.top.get().max(r + 1)); (r < REGS).then_some(r as u8) } @@ -388,6 +399,13 @@ impl Builder<'_> { } OpCode::LoadSmallInt => self.constant(V::I(i64::from(arg)))?, OpCode::PushNull => self.constant(V::Null)?, + OpCode::BuildList | OpCode::BuildMap if arg == 0 => { + let dst = self.push_new()?; + self.ops.push(Op::New { + dst, + dict: ins.op == OpCode::BuildMap, + }); + } OpCode::LoadGlobal => { let dst = self.push_new()?; self.ops.push(Op::Global { @@ -659,6 +677,7 @@ impl Builder<'_> { ops: self.ops.into_boxed_slice(), consts: self.consts.into_boxed_slice(), nargs: u8::try_from(crate::leaf_arity(self.code)).ok()?, + nregs: u8::try_from(self.top.get().clamp(self.nl, REGS)).ok()?, unique_stores: self.stored.is_some() && self.code.arg_count > 0, }) } @@ -684,15 +703,20 @@ pub(crate) fn build(code: &CodeObject, ext: &CodeConstObjects) -> Option; 4], + buf: [std::mem::MaybeUninit; OWNED], n: usize, } @@ -729,21 +753,24 @@ impl Owned { /// Hold `v` (a scalar needs no holding) and name it. #[inline] fn own(&mut self, v: Object) -> Option { - Some(match v { + let scalar = match v { Object::Int(i) => V::I(i), Object::Float(x) => V::F(x), Object::Bool(b) => V::B(b), Object::None => V::N, - v => { + _ => { if self.n == self.buf.len() { return None; } let slot = &mut self.buf[self.n]; slot.write(v); self.n += 1; - V::R(slot.as_ptr()) + return Some(V::R(slot.as_ptr())); } - }) + }; + // A scalar owns nothing: no drop glue to run. + std::mem::forget(v); + Some(scalar) } } @@ -891,14 +918,31 @@ impl Interpreter { use weavepy_compiler::InlineCache as IC; /// Nested leaf calls evaluated in place at most this deep. const NEST: u8 = 3; + /// A pure-leaf callee runs as a frame of this evaluation: its + /// registers are the next `REGS` window, and its caller's state + /// waits here until its return. + struct Caller<'p> { + ip: usize, + code: &'p CodeObject, + ext: &'p CodeConstObjects, + plan: &'p LeafPlan, + f: &'p crate::object::PyFunction, + /// The caller's register for the result. + at: u8, + /// The caller's registers (its plan's `nregs`). + saved: [std::mem::MaybeUninit; REGS], + } let nargs = usize::from(plan.nargs); if args.len() != nargs { return None; } + let (mut code, mut ext, mut plan, mut f) = (code, ext, plan, f); let mut regs = [const { std::mem::MaybeUninit::::uninit() }; REGS]; for (k, &a) in args.iter().enumerate() { regs[k].write(norm(a)); } + let mut callers = [const { std::mem::MaybeUninit::>::uninit() }; NEST as usize]; + let mut depth = 0usize; // The translation proves every register is written before it's // read, and that every index is below `REGS`. macro_rules! get { @@ -914,10 +958,10 @@ impl Interpreter { unsafe { regs.get_unchecked_mut(usize::from($r)).write(v) }; }}; } - let consts: &[V] = &plan.consts; - let stamps: &[crate::StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + let mut consts: &[V] = &plan.consts; + let mut stamps: &[crate::StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); let mut owned = Owned { - buf: [const { std::mem::MaybeUninit::uninit() }; 4], + buf: [const { std::mem::MaybeUninit::uninit() }; OWNED], n: 0, }; let mut pend = Pending { @@ -925,7 +969,7 @@ impl Interpreter { n: 0, moved: 0, }; - let ops: &[Op] = &plan.ops; + let mut ops: &[Op] = &plan.ops; let mut ip = 0usize; loop { // SAFETY: every path ends in a return or a jump, and every @@ -1160,11 +1204,33 @@ impl Interpreter { } } Op::Jump { target } => ip = usize::from(target), + Op::New { dst, dict } => { + // Tracked like the core loop's; an allocation that + // would trigger a collection is left to it. + if crate::stdlib::tracemalloc_real::is_tracking() + || crate::stdlib::testinternalcapi_mod::reftrace_print_active() + || crate::gc_trace::auto_collect_due() + { + return None; + } + let obj = if dict { + Object::Dict(Rc::new(crate::sync::RefCell::new( + crate::object::DictData::with_capacity_and_hasher( + 0, + crate::fasthash::FxBuildHasher, + ), + ))) + } else { + Object::new_list(Vec::new()) + }; + crate::gc_trace::track(obj.clone()); + set!(dst, owned.own(obj)?); + } Op::Decline => return None, Op::Call { at, argc, pc } => { // A pure-leaf callee, evaluated in place: only while // no store is buffered (it would not see one). - if (EFFECT && pend.n > 0) || nest >= NEST { + if (EFFECT && pend.n > 0) || usize::from(nest) + depth >= usize::from(NEST) { return None; } let callee = match get!(at) { @@ -1226,11 +1292,58 @@ impl Interpreter { || !Self::leaf_code_ok(ccode) || n != crate::leaf_arity(ccode) || ccode.has_varkeywords - || crate::recursion::current_depth() + usize::from(nest) + 1 + || crate::recursion::current_depth() + usize::from(nest) + depth + 1 >= crate::recursion::recursion_limit() { return None; } + if !GETTER { + // The callee runs as a frame of this evaluation: + // the caller's registers wait in its record, and + // the arguments (normalized already) become the + // callee's first registers. + let cext = crate::code_vm_ext(ccode)?; + let cplan = cext + .leaf_plan + .get_or_init(|| build(ccode, cext).map(Box::new)) + .as_deref()?; + let mut cargs = [V::N; 8]; + for (k, slot) in cargs.iter_mut().enumerate().take(n) { + let v = get!(first + k as u8); + if matches!(v, V::Fn(_) | V::Bi(_) | V::Null) { + return None; + } + *slot = v; + } + // SAFETY: `depth < NEST` (checked above). + let c = unsafe { callers.get_unchecked_mut(depth) }.as_mut_ptr(); + // SAFETY: `c` is this frame's record; the saved + // registers are the caller plan's own, all below + // `REGS`. + unsafe { + std::ptr::addr_of_mut!((*c).ip).write(ip); + std::ptr::addr_of_mut!((*c).code).write(code); + std::ptr::addr_of_mut!((*c).ext).write(ext); + std::ptr::addr_of_mut!((*c).plan).write(plan); + std::ptr::addr_of_mut!((*c).f).write(f); + std::ptr::addr_of_mut!((*c).at).write(at); + std::ptr::copy_nonoverlapping( + regs.as_ptr(), + std::ptr::addr_of_mut!((*c).saved).cast(), + usize::from(plan.nregs), + ); + } + for (k, &v) in cargs.iter().enumerate().take(n) { + regs[k].write(v); + } + depth += 1; + (code, ext, plan, f) = (&**ccode, cext, cplan, callee); + consts = &plan.consts; + stamps = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + ops = &plan.ops; + ip = 0; + continue; + } // Scalar arguments are staged as objects (no drop glue). let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; let mut ptrs = [std::ptr::null::(); 8]; @@ -1258,6 +1371,34 @@ impl Interpreter { } Op::Return { src } => { let v = get!(src); + if depth > 0 { + // Back to the caller's frame, the result in its + // register. A constructor's stores land at once, + // so what it borrows from its fresh instance could + // move: it holds its own reference. + let v = match v { + V::R(p) if FRESH && !owned.holds(p) => owned.own(to_object(v)?)?, + v => v, + }; + depth -= 1; + // SAFETY: frame `depth`'s record was written at its + // call (its registers as far as its `nregs`). + let c = unsafe { &*callers.get_unchecked(depth).as_ptr() }; + (ip, code, ext, plan, f) = (c.ip, c.code, c.ext, c.plan, c.f); + consts = &plan.consts; + stamps = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + ops = &plan.ops; + // SAFETY: as above. + unsafe { + std::ptr::copy_nonoverlapping( + c.saved.as_ptr(), + regs.as_mut_ptr(), + usize::from(plan.nregs), + ); + } + set!(c.at, v); + continue; + } // A value this evaluation's scratch doesn't hold outlives // it (a pure body stores nothing), so a nested caller // borrows it rather than taking a reference to release. diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 547201dd..5ce2c36f 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -10624,6 +10624,22 @@ impl Interpreter { code.has_varargs || code.has_varkeywords || code.kwonly_count != 0 } + /// How many trailing positional defaults a call of `f` (running + /// `code`) with `given` positional arguments, self included, takes + /// from `f.defaults`. `None` when too many arguments come, too few + /// for the defaults to fill, or `__defaults__` was rebound. + #[inline] + fn missing_defaults(f: &PyFunction, code: &CodeObject, given: usize) -> Option { + let missing = (code.arg_count as usize).checked_sub(given)?; + if missing > 0 + && (f.defaults.len() < missing + || (f.defaults_maybe_overridden() && f.slot("__defaults__").is_some())) + { + return None; + } + Some(missing) + } + /// `f`'s compiled default for each of `code`'s keyword-only parameters, /// in parameter order. fn kwonly_defaults<'a>( @@ -11544,12 +11560,12 @@ impl Interpreter { if !std::ptr::eq( unsafe { Rc::as_ptr(&*init.code.as_ptr()) }, Rc::as_ptr(code), - ) || code.arg_count as usize != args.len() + 1 - || !Self::lean_code_ok(code) + ) || !Self::lean_code_ok(code) || crate::stdlib::testinternalcapi_mod::nomem_alloc_fails() { return None; } + let missing = Self::missing_defaults(init, code, args.len() + 1)?; let cells = init.lean_cells_ref(code)?; let snap_gen = self.lean_snapshot()?; let code = code.clone(); @@ -11570,6 +11586,11 @@ impl Interpreter { let v = unsafe { &mut *rc.as_ptr() }; v.push(inst.clone()); v.extend(args.iter().cloned()); + v.extend( + init.defaults[init.defaults.len() - missing..] + .iter() + .cloned(), + ); v.resize(nlocals, Object::Unbound); } rc @@ -14929,11 +14950,13 @@ impl Interpreter { if !std::ptr::eq( unsafe { Rc::as_ptr(&*init.code.as_ptr()) }, Rc::as_ptr(code), - ) || code.arg_count as usize != argc + 1 - || !Self::lean_code_ok(code) + ) || !Self::lean_code_ok(code) { return false; } + let Some(missing) = Self::missing_defaults(init, code, argc + 1) else { + return false; + }; // A leaf `__init__` (plain stores of its arguments into `self`) // runs frameless, as a leaf method call does. if self.core_leaf_init( @@ -14942,6 +14965,7 @@ impl Interpreter { init, code, argc, + missing, self_slot, callee_slot, sw.depth_cell, @@ -14969,9 +14993,14 @@ impl Interpreter { // SAFETY: a parked slot's locals storage is its own, and empty. let locals = unsafe { &mut *act.frame.locals.as_ptr() }; let nlocals = code.varnames.len(); - locals.reserve(nlocals.max(argc + 1)); + locals.reserve(nlocals.max(argc + 1 + missing)); locals.push(inst.clone()); locals.extend(frame.stack.drain(self_slot + 1..)); + locals.extend( + init.defaults[init.defaults.len() - missing..] + .iter() + .cloned(), + ); fill_unbound(locals, nlocals); // The NULL self slot and the class. frame.stack.truncate(callee_slot); @@ -15018,13 +15047,14 @@ impl Interpreter { init: &Rc, code: &Rc, argc: usize, + missing: usize, self_slot: usize, callee_slot: usize, depth_cell: *const std::cell::Cell, ) -> bool { let pure = code_is_pure_leaf(code); if !(pure || code_is_effect_leaf(code)) - || argc >= 8 + || argc + missing >= 8 || !pure_leaf_warm(code) || !code_returns_only_none(code) || ty.flags.is_builtin @@ -15050,10 +15080,15 @@ impl Interpreter { for (k, a) in frame.stack[self_slot + 1..].iter().enumerate() { args[k + 1] = a; } + let defaults = &init.defaults[init.defaults.len() - missing..]; + for (k, d) in defaults.iter().enumerate() { + args[argc + 1 + k] = d; + } + let args = &args[..=argc + missing]; let r = if pure { - self.pure_leaf_eval::(code, init, &args[..=argc]) + self.pure_leaf_eval::(code, init, args) } else { - self.leaf_init_eval(code, init, &args[..=argc]) + self.leaf_init_eval(code, init, args) }; match r { Some(done) => { @@ -64208,9 +64243,14 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { && code.varnames.len() <= 16 && code.exception_table.is_empty() && code.instructions.len() <= 64; - let pure_op = |op: OpCode| { + let pure_op = |ins: &weavepy_compiler::Instruction| { + // A fresh empty list or dict is unobservable until stored, so + // a declined evaluation that made one still did nothing. + if matches!(ins.op, OpCode::BuildList | OpCode::BuildMap) { + return ins.arg == 0; + } matches!( - op, + ins.op, OpCode::Resume | OpCode::Nop | OpCode::NotTaken @@ -64246,7 +64286,7 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { | OpCode::ReturnValue ) }; - let ok = structural && code.instructions.iter().all(|i| pure_op(i.op)); + let ok = structural && code.instructions.iter().all(pure_op); // An effect leaf: a pure leaf but for attribute stores, which its // evaluation buffers and commits at the return (see // `Interpreter::pure_leaf_eval`). @@ -64255,7 +64295,7 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { && code .instructions .iter() - .all(|i| pure_op(i.op) || i.op == OpCode::StoreAttr) + .all(|i| pure_op(i) || i.op == OpCode::StoreAttr) && code.instructions.iter().any(|i| i.op == OpCode::StoreAttr); let shape = if ok { let body = code.instructions.as_slice(); diff --git a/tests/regrtest/test_leaf_constructors.py b/tests/regrtest/test_leaf_constructors.py new file mode 100644 index 00000000..ff824563 --- /dev/null +++ b/tests/regrtest/test_leaf_constructors.py @@ -0,0 +1,145 @@ +"""Constructors whose `__init__` runs without a frame. + +A class's `__init__` that only stores into `self` runs as a leaf: its +body is evaluated in place on the fresh instance. These cases check the +shapes around that path: trailing default parameters (including ones +rebound through `__defaults__`), empty list and dict literals, and +bodies that stop partway and fall back to the ordinary call. +""" + +import gc +import unittest + + +class WithDefaults: + def __init__(self, name, value=0, mark=None): + self.name = name + self.value = value + self.mark = mark + + +class WithContainers: + def __init__(self, name): + self.name = name + self.items = [] + self.index = {} + self.more = [] + + +class ManyLists: + def __init__(self): + self.a = [] + self.b = [] + self.c = [] + self.d = [] + self.e = [] + self.f = [] + + +class Declines: + def __init__(self, x, extra=1): + self.items = [] + self.x = x + self.y = x + extra + + +def build(cls, n, *args): + return [cls(*args) for _ in range(n)] + + +class LeafConstructorTests(unittest.TestCase): + def test_trailing_defaults(self): + objs = build(WithDefaults, 50, "v") + self.assertTrue(all(o.value == 0 and o.mark is None for o in objs)) + objs = build(WithDefaults, 50, "v", 5) + self.assertTrue(all(o.value == 5 and o.mark is None for o in objs)) + objs = build(WithDefaults, 50, "v", 5, "m") + self.assertTrue(all((o.value, o.mark) == (5, "m") for o in objs)) + self.assertEqual(vars(objs[0]), {"name": "v", "value": 5, "mark": "m"}) + + def test_arity_errors(self): + for _ in range(50): + with self.assertRaises(TypeError): + WithDefaults() + with self.assertRaises(TypeError): + WithDefaults(1, 2, 3, 4) + + def test_rebound_defaults(self): + class C: + def __init__(self, a, b=1): + self.a = a + self.b = b + + self.assertEqual([C(0).b for _ in range(50)], [1] * 50) + C.__init__.__defaults__ = (2,) + self.assertEqual([C(0).b for _ in range(50)], [2] * 50) + C.__init__.__defaults__ = None + for _ in range(3): + with self.assertRaises(TypeError): + C(0) + + def test_fresh_containers(self): + objs = build(WithContainers, 50, "n") + for o in objs: + self.assertEqual((o.items, o.index, o.more), ([], {}, [])) + # Each instance gets its own containers. + objs[0].items.append(1) + objs[0].index["k"] = 2 + self.assertEqual(objs[1].items, []) + self.assertEqual(objs[1].index, {}) + self.assertIsNot(objs[0].items, objs[0].more) + self.assertEqual(len({id(o.items) for o in objs}), len(objs)) + self.assertTrue(gc.is_tracked(objs[1].items)) + self.assertTrue(gc.is_tracked(objs[1].index)) + + def test_more_containers_than_scratch(self): + for o in build(ManyLists, 50): + self.assertEqual([o.a, o.b, o.c, o.d, o.e, o.f], [[]] * 6) + self.assertEqual(len({id(v) for v in vars(o).values()}), 6) + + def test_fallback_after_partial_body(self): + objs = build(Declines, 50, 1) + self.assertTrue(all((o.x, o.y, o.items) == (1, 2, []) for o in objs)) + big = 1 << 62 + objs = build(Declines, 50, big, big) + self.assertTrue(all(o.y == big * 2 for o in objs)) + objs = build(Declines, 50, 1.5) + self.assertTrue(all(o.y == 2.5 for o in objs)) + with self.assertRaises(TypeError): + Declines("a") + + def test_cycles_are_collected(self): + class Node: + def __init__(self, prev=None): + self.prev = prev + self.next = [] + + gc.collect() + for _ in range(20): + prev = None + for _ in range(50): + node = Node(prev) + if prev is not None: + prev.next.append(node) + prev = node + del prev, node + gc.collect() + self.assertFalse(any(type(o) is Node for o in gc.get_objects())) + + def test_leaf_functions_return_fresh_containers(self): + def empty_list(): + return [] + + def empty_dict(): + return {} + + lists = [empty_list() for _ in range(50)] + dicts = [empty_dict() for _ in range(50)] + self.assertEqual(len({id(x) for x in lists}), 50) + self.assertEqual(len({id(x) for x in dicts}), 50) + lists[0].append(1) + self.assertEqual(lists[1], []) + + +if __name__ == "__main__": + unittest.main() From 998c5c45f75d616fe6d3e59de531a6e90574dcfc Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:55:29 -0700 Subject: [PATCH 55/65] fix: keep tiny pure-leaf shapes on their dedicated evaluator A leaf plan's in-place callee frames now apply only to general plans. A certified tiny shape (a constant or field return, a field comparison) keeps its dedicated evaluator, as before the frames: it's at least as fast, and the unit test that counts cached field predicate hits on DeltaBlue passes again. --- crates/weavepy-vm/src/leaf_plan.rs | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs index f946c16f..84115000 100644 --- a/crates/weavepy-vm/src/leaf_plan.rs +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -1297,12 +1297,14 @@ impl Interpreter { { return None; } - if !GETTER { + let cext = crate::code_vm_ext(ccode)?; + // A tiny certified shape (a constant or field return, + // a field comparison) keeps its dedicated evaluator. + if !GETTER && cext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) < 3 { // The callee runs as a frame of this evaluation: // the caller's registers wait in its record, and // the arguments (normalized already) become the // callee's first registers. - let cext = crate::code_vm_ext(ccode)?; let cplan = cext .leaf_plan .get_or_init(|| build(ccode, cext).map(Box::new)) From 517deb289f8cbcd693164c2290ae67174e9d94b9 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:55:30 -0700 Subject: [PATCH 56/65] perf: slice tuples, bytes, and code points without copying them whole Slicing a tuple, bytes, bytearray, or a string with surrogates used to copy the entire sequence into a vector of objects and then select from it, so every slice cost the length of its source. `slice_seq` is now generic over the element type and copies a unit-step slice in one piece, and each type slices its own storage. `re.compile` with IGNORECASE slices a 64K charset map 256 times, which made `import logging` cost 1.19B instructions; it now costs 525M. --- crates/weavepy-vm/src/builtins.rs | 4 +- crates/weavepy-vm/src/lib.rs | 66 ++++++++-------------- tests/regrtest/test_sequence_slices.py | 78 ++++++++++++++++++++++++++ 3 files changed, 102 insertions(+), 46 deletions(-) create mode 100644 tests/regrtest/test_sequence_slices.py diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index 8a7b2366..12e6672e 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -12444,8 +12444,8 @@ fn list_getitem(args: &[Object]) -> Result { .get(1) .ok_or_else(|| type_error("__getitem__ expected 1 argument"))?; if let Object::Slice(s) = key { - let seq = l.borrow().clone(); - return Ok(Object::new_list(crate::slice_seq(&seq, s)?)); + let sliced = crate::slice_seq(&l.borrow(), s)?; + return Ok(Object::new_list(sliced)); } let l = l.borrow(); let n = list_index_arg(l.len(), key, "__getitem__")?; diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 5ce2c36f..3b19496a 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -15119,10 +15119,7 @@ impl Interpreter { let items = items.try_borrow().ok()?; Some(Object::new_list(slice_seq(&items, sl).ok()?)) } - Object::Tuple(items) => { - let v: Vec = items.iter().cloned().collect(); - Some(Object::new_tuple(slice_seq(&v, sl).ok()?)) - } + Object::Tuple(items) => Some(Object::new_tuple(slice_seq(&items[..], sl).ok()?)), Object::Str(st) => str_subscript_slice(st, sl).ok(), _ => None, } @@ -42944,24 +42941,13 @@ impl Interpreter { Ok(Object::new_list(sliced)) } (Object::Tuple(items), Object::Slice(s)) => { - let v: Vec = items.iter().cloned().collect(); - let sliced = slice_seq(&v, s)?; - Ok(Object::new_tuple(sliced)) + Ok(Object::new_tuple(slice_seq(&items[..], s)?)) } (Object::Str(s), Object::Slice(slc)) => str_subscript_slice(s, slc), (Object::WStr(cps), Object::Slice(slc)) => { // Slice over code points, then canonicalise: a slice that drops // every surrogate becomes a plain `Str` again. - let items: Vec = cps.iter().map(|&c| Object::Int(i64::from(c))).collect(); - let sliced = slice_seq(&items, slc)?; - let out: Vec = sliced - .iter() - .map(|o| match o { - Object::Int(n) => *n as u32, - _ => unreachable!("slice of int vec yields ints"), - }) - .collect(); - Ok(Object::str_from_codepoints(out)) + Ok(Object::str_from_codepoints(slice_seq(&cps[..], slc)?)) } (Object::Range(r), Object::Int(_) | Object::Long(_)) => { // Full-width arithmetic: `Object::len()` raises @@ -43003,28 +42989,11 @@ impl Interpreter { Ok(Object::Int(i64::from(buf[idx]))) } (Object::Bytes(buf), Object::Slice(slc)) => { - let as_objs: Vec = buf.iter().map(|b| Object::Int(i64::from(*b))).collect(); - let sliced = slice_seq(&as_objs, slc)?; - let mut out = Vec::with_capacity(sliced.len()); - for o in sliced { - match o { - Object::Int(i) => out.push(i as u8), - _ => return Err(type_error("bytes slice produced non-int")), - } - } + let out = slice_seq(&buf[..], slc)?; Ok(Object::Bytes(SharedSlice::from(out.as_slice()))) } (Object::ByteArray(buf), Object::Slice(slc)) => { - let buf = buf.borrow(); - let as_objs: Vec = buf.iter().map(|b| Object::Int(i64::from(*b))).collect(); - let sliced = slice_seq(&as_objs, slc)?; - let mut out = Vec::with_capacity(sliced.len()); - for o in sliced { - match o { - Object::Int(i) => out.push(i as u8), - _ => return Err(type_error("bytearray slice produced non-int")), - } - } + let out = slice_seq(&buf.borrow()[..], slc)?; Ok(Object::ByteArray(Rc::new(RefCell::new(out)))) } (Object::MemoryView(mv), Object::Int(i)) => { @@ -57136,16 +57105,17 @@ pub(crate) fn str_subscript_slice(s: &SharedStr, slc: &PySlice) -> Result = s.chars().map(|c| Object::from_str(c.to_string())).collect(); - let sliced = slice_seq(&obj_chars, slc)?; - let out: String = sliced.iter().map(|o| o.to_str()).collect(); - Ok(Object::from_str(out)) + // General path: negative bounds or non-unit step, over the code points. + let chars: Vec = s.chars().collect(); + Ok(Object::from_str( + slice_seq(&chars, slc)?.into_iter().collect::(), + )) } -pub(crate) fn slice_seq(seq: &[Object], s: &PySlice) -> Result, RuntimeError> { +/// The elements of `seq` that slice `s` selects, in order: CPython's +/// `PySlice_Unpack` + `PySlice_AdjustIndices` over any element type (a +/// list's objects, a `bytes` buffer, a string's code points). +pub(crate) fn slice_seq(seq: &[T], s: &PySlice) -> Result, RuntimeError> { let len = seq.len() as i64; let step = match &s.step { Object::None => 1i64, @@ -57218,6 +57188,14 @@ pub(crate) fn slice_seq(seq: &[Object], s: &PySlice) -> Result, Runt )) } }; + if step == 1 { + // Both bounds are within `0..=len` here: one contiguous copy. + return Ok(if start < stop { + seq[start as usize..stop as usize].to_vec() + } else { + Vec::new() + }); + } let mut i = start; let mut out = Vec::new(); if step > 0 { diff --git a/tests/regrtest/test_sequence_slices.py b/tests/regrtest/test_sequence_slices.py new file mode 100644 index 00000000..e573b67d --- /dev/null +++ b/tests/regrtest/test_sequence_slices.py @@ -0,0 +1,78 @@ +"""Slices of every built-in sequence, over the step and bound shapes. + +Each type slices its own storage (a tuple's items, a `bytes` buffer, a +string's code points), so these cases compare every shape against the +same selection made by explicit indexing. +""" + +import unittest + +STEPS = (None, 1, 2, 3, -1, -2, -3) +BOUNDS = (None, -100, -5, -1, 0, 1, 3, 5, 100) + + +def expected(seq, start, stop, step): + return [seq[i] for i in range(*slice(start, stop, step).indices(len(seq)))] + + +class SequenceSliceTests(unittest.TestCase): + def check(self, seq, rebuild): + for start in BOUNDS: + for stop in BOUNDS: + for step in STEPS: + with self.subTest(start=start, stop=stop, step=step): + got = seq[start:stop:step] + self.assertIs(type(got), type(seq)) + self.assertEqual(got, rebuild(expected(seq, start, stop, step))) + + def test_tuple(self): + self.check(tuple(range(9)), tuple) + self.check((), tuple) + + def test_list(self): + self.check(list(range(9)), list) + + def test_bytes(self): + self.check(bytes(range(9)), bytes) + self.check(b"", bytes) + + def test_bytearray(self): + self.check(bytearray(range(9)), bytearray) + data = bytearray(b"abc") + part = data[:] + part[0] = 0x7A + self.assertEqual(data, b"abc") + + def test_str(self): + self.check("abcdéfghi", "".join) + self.check("日本語テキストです", "".join) + self.check("", "".join) + + def test_str_with_surrogates(self): + text = "a\ud800b\udc00c\ud83d" + self.check(text, "".join) + self.assertEqual(text[1:2], "\ud800") + self.assertEqual(text[::2], "abc") + + def test_large_sequences(self): + # A small slice of a large sequence (the case `re` hits when it + # chunks a 64K-entry charset map). + big = bytes(65536) + self.assertEqual(big[256:512], bytes(256)) + chunks = {big[i:i + 256] for i in range(0, 65536, 256)} + self.assertEqual(chunks, {bytes(256)}) + items = tuple(range(100000)) + self.assertEqual(items[50000:50003], (50000, 50001, 50002)) + self.assertEqual(items[-3:], (99997, 99998, 99999)) + self.assertEqual(items[99990::4], (99990, 99994, 99998)) + + def test_bad_steps_and_indices(self): + for seq in ((1, 2), b"ab", bytearray(b"ab"), "ab", [1, 2]): + with self.assertRaises(ValueError): + seq[::0] + with self.assertRaises(TypeError): + seq["a":] + + +if __name__ == "__main__": + unittest.main() From 40bcd2c0ed67d9567289fd44e7eb889a78810f32 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:55:30 -0700 Subject: [PATCH 57/65] perf: cache compiled stdlib modules in a WeavePy-native format The private frozen-stdlib cache stored CPython marshal data, so every import unmarshalled it and transcoded the CPython bytecode back into WeavePy instructions. The new `native_code` module encodes a code object's fields directly (varints, delta-coded line tables, nested code inline) and decodes them with no transcoding; filenames are stamped at load. The source check hashes eight bytes at a time. `import json` drops from 269M to 205M instructions, `-c pass` from 106M to 91M, and `import dataclasses` from 454M to 349M. --- crates/weavepy-compiler/src/bytecode.rs | 10 + crates/weavepy-compiler/src/lib.rs | 1 + crates/weavepy-compiler/src/native_code.rs | 547 +++++++++++++++++++++ crates/weavepy-vm/src/frozen_code_cache.rs | 75 ++- 4 files changed, 595 insertions(+), 38 deletions(-) create mode 100644 crates/weavepy-compiler/src/native_code.rs diff --git a/crates/weavepy-compiler/src/bytecode.rs b/crates/weavepy-compiler/src/bytecode.rs index 6614bdaa..2eb3fd99 100644 --- a/crates/weavepy-compiler/src/bytecode.rs +++ b/crates/weavepy-compiler/src/bytecode.rs @@ -954,6 +954,16 @@ pub struct Instruction { pub arg: u32, } +impl OpCode { + /// The opcode numbered `b` (variants count up from zero in + /// declaration order), if there is one. + pub fn from_u8(b: u8) -> Option { + // SAFETY: `OpCode` is `repr(u8)` with implicit, contiguous + // discriminants, the last of which is `StoreFastMaybeNull`. + (b <= Self::StoreFastMaybeNull as u8).then(|| unsafe { std::mem::transmute::(b) }) + } +} + impl Instruction { #[inline] pub const fn new(op: OpCode, arg: u32) -> Self { diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index 0d9148a9..6f840348 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -40,6 +40,7 @@ pub mod cpython_code; mod flowgraph; mod intern; mod mangle; +pub mod native_code; mod validate; pub use bytecode::{ diff --git a/crates/weavepy-compiler/src/native_code.rs b/crates/weavepy-compiler/src/native_code.rs new file mode 100644 index 00000000..b430758a --- /dev/null +++ b/crates/weavepy-compiler/src/native_code.rs @@ -0,0 +1,547 @@ +//! A compact, WeavePy-internal serialization of [`CodeObject`]s. +//! +//! The private frozen-stdlib cache stores compiled modules in this form +//! rather than as CPython `marshal` data: reading it back is a straight +//! copy of each field, with no CPython bytecode to parse and transcode. +//! It isn't an interchange format. Its layout is free to change with any +//! release, so a reader must check the version byte and treat any +//! mismatch or malformed input as a cache miss. +//! +//! Filenames aren't stored: every code object in a cached module shares +//! the module's filename, which the reader supplies. + +use std::sync::Arc; + +use crate::bytecode::{CacheTable, Instruction, OpCode}; +use crate::{CodeObject, ColSpan, Constant, ExcHandler}; + +/// The layout revision; bump it whenever the encoding changes. +pub const VERSION: u8 = 2; + +/// Encode `code` (and its nested code objects). `None` for a code object +/// the format doesn't carry: one with raw CPython wire overrides, or with +/// a constant that has no fixed value. +pub fn encode(code: &CodeObject) -> Option> { + let mut w = Writer(Vec::with_capacity(4096)); + w.byte(VERSION); + w.code(code)?; + Some(w.0) +} + +/// Decode what [`encode`] wrote, stamping `filename` on every code +/// object. `None` for input it didn't write (or a different version's). +pub fn decode(bytes: &[u8], filename: &str) -> Option { + let mut r = Reader { bytes, pos: 0 }; + if r.byte()? != VERSION { + return None; + } + let code = r.code(filename, 0)?; + (r.pos == bytes.len()).then_some(code) +} + +struct Writer(Vec); + +impl Writer { + fn byte(&mut self, b: u8) { + self.0.push(b); + } + + fn uint(&mut self, mut v: u64) { + while v >= 0x80 { + self.0.push((v as u8) | 0x80); + v >>= 7; + } + self.0.push(v as u8); + } + + fn int(&mut self, v: i64) { + self.uint(((v << 1) ^ (v >> 63)) as u64); + } + + fn bytes(&mut self, b: &[u8]) { + self.uint(b.len() as u64); + self.0.extend_from_slice(b); + } + + fn strs(&mut self, v: &[String]) { + self.uint(v.len() as u64); + for s in v { + self.bytes(s.as_bytes()); + } + } + + fn u32s(&mut self, v: &[u32]) { + self.uint(v.len() as u64); + for &x in v { + self.uint(u64::from(x)); + } + } + + fn deltas(&mut self, v: impl ExactSizeIterator) { + self.uint(v.len() as u64); + let mut prev = 0i64; + for x in v { + self.int(i64::from(x) - prev); + prev = i64::from(x); + } + } + + fn code(&mut self, c: &CodeObject) -> Option<()> { + if c.wire.is_some() { + return None; + } + self.bytes(c.name.as_bytes()); + self.bytes(c.qualname.as_bytes()); + self.uint(c.instructions.len() as u64); + for ins in &c.instructions { + self.byte(ins.op as u8); + self.uint(u64::from(ins.arg)); + } + self.uint(c.constants.len() as u64); + for k in &c.constants { + self.constant(k)?; + } + self.strs(&c.names); + self.strs(&c.varnames); + self.strs(&c.freevars); + self.strs(&c.cellvars); + self.uint(c.exception_table.len() as u64); + for h in &c.exception_table { + self.uint(u64::from(h.start)); + self.uint(u64::from(h.end)); + self.uint(u64::from(h.handler)); + self.uint(u64::from(h.depth)); + self.byte(u8::from(h.push_lasti)); + } + // Line numbers as deltas from the previous entry: mostly zero, so + // one byte each whatever the line. + self.deltas(c.linetable.iter().copied()); + self.deltas(c.coltable.iter().map(|s| s.end_lineno)); + for s in &c.coltable { + self.int(i64::from(s.col)); + self.int(i64::from(s.end_col)); + } + self.uint(u64::from(c.arg_count)); + self.uint(u64::from(c.posonly_count)); + self.uint(u64::from(c.kwonly_count)); + let flags = [ + c.has_varargs, + c.has_varkeywords, + c.is_class_body, + c.is_generator, + c.is_coroutine, + c.is_async_generator, + c.is_iterable_coroutine, + c.has_docstring, + c.is_method, + c.is_nested, + c.annotate_scope, + ]; + let bits = flags + .iter() + .enumerate() + .fold(0u64, |acc, (i, &f)| acc | (u64::from(f) << i)); + self.uint(bits); + self.uint(u64::from(c.future_flags)); + self.uint(c.stacksize.map_or(0, |s| u64::from(s) + 1)); + self.u32s(&c.no_interrupt_jumps); + self.bytes(&c.wire_marks); + self.strs(&c.hidden_locals); + self.strs(&c.const_identifiers); + Some(()) + } + + fn constant(&mut self, k: &Constant) -> Option<()> { + match k { + Constant::None => self.byte(0), + Constant::Bool(b) => self.byte(1 + u8::from(*b)), + Constant::Int(i) => { + self.byte(3); + self.int(*i); + } + Constant::BigInt(b) => { + self.byte(4); + self.bytes(&b.to_signed_bytes_le()); + } + Constant::Float(x) => { + self.byte(5); + self.0.extend_from_slice(&x.to_bits().to_le_bytes()); + } + Constant::Complex(re, im) => { + self.byte(6); + self.0.extend_from_slice(&re.to_bits().to_le_bytes()); + self.0.extend_from_slice(&im.to_bits().to_le_bytes()); + } + Constant::Str(s) => { + self.byte(7); + self.bytes(s.as_bytes()); + } + Constant::WStr(cps) => { + self.byte(8); + self.u32s(cps); + } + Constant::Bytes(b) => { + self.byte(9); + self.bytes(b); + } + Constant::Tuple(items) | Constant::FrozenSet(items) => { + self.byte(if matches!(k, Constant::Tuple(_)) { + 10 + } else { + 11 + }); + self.uint(items.len() as u64); + for it in items { + self.constant(it)?; + } + } + Constant::Code(c) => { + self.byte(12); + self.code(c)?; + } + Constant::Ellipsis => self.byte(13), + Constant::Slice(parts) => { + self.byte(14); + self.constant(&parts.0)?; + self.constant(&parts.1)?; + self.constant(&parts.2)?; + } + Constant::Unmarshallable => return None, + } + Some(()) + } +} + +struct Reader<'a> { + bytes: &'a [u8], + pos: usize, +} + +/// Nested code objects deeper than this are taken as corrupt input. +const MAX_NESTING: u32 = 200; + +impl Reader<'_> { + #[inline(always)] + fn byte(&mut self) -> Option { + let b = *self.bytes.get(self.pos)?; + self.pos += 1; + Some(b) + } + + #[inline(always)] + fn uint(&mut self) -> Option { + // Most values (opcodes' arguments, line numbers, lengths) fit + // in one byte. + let b = self.byte()?; + if b < 0x80 { + return Some(u64::from(b)); + } + self.uint_long(b) + } + + #[inline(never)] + fn uint_long(&mut self, first: u8) -> Option { + let mut v = u64::from(first & 0x7f); + let mut shift = 7; + loop { + let b = self.byte()?; + if shift >= 64 { + return None; + } + v |= u64::from(b & 0x7f) << shift; + if b < 0x80 { + return Some(v); + } + shift += 7; + } + } + + #[inline(always)] + fn u32(&mut self) -> Option { + u32::try_from(self.uint()?).ok() + } + + #[inline(always)] + fn int(&mut self) -> Option { + let v = self.uint()?; + Some(((v >> 1) as i64) ^ -((v & 1) as i64)) + } + + #[inline(always)] + fn i32(&mut self) -> Option { + i32::try_from(self.int()?).ok() + } + + /// A length prefix, bounded by what's left of the input (every + /// element takes at least one byte). + #[inline(always)] + fn len(&mut self) -> Option { + let n = usize::try_from(self.uint()?).ok()?; + (n <= self.bytes.len() - self.pos).then_some(n) + } + + #[inline(always)] + fn raw(&mut self) -> Option<&[u8]> { + let n = self.len()?; + let s = &self.bytes[self.pos..self.pos + n]; + self.pos += n; + Some(s) + } + + fn string(&mut self) -> Option { + Some(std::str::from_utf8(self.raw()?).ok()?.to_owned()) + } + + fn strs(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(self.string()?); + } + Some(v) + } + + fn u32s(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(self.u32()?); + } + Some(v) + } + + fn deltas(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + let mut prev = 0i64; + for _ in 0..n { + prev = prev.checked_add(self.int()?)?; + v.push(u32::try_from(prev).ok()?); + } + Some(v) + } + + fn f64(&mut self) -> Option { + let b = self.bytes.get(self.pos..self.pos + 8)?; + self.pos += 8; + Some(f64::from_bits(u64::from_le_bytes(b.try_into().ok()?))) + } + + fn code(&mut self, filename: &str, depth: u32) -> Option { + if depth > MAX_NESTING { + return None; + } + let name = self.string()?; + let qualname = self.string()?; + let n = self.len()?; + let mut instructions = Vec::with_capacity(n); + for _ in 0..n { + let op = OpCode::from_u8(self.byte()?)?; + instructions.push(Instruction::new(op, self.u32()?)); + } + let n = self.len()?; + let mut constants = Vec::with_capacity(n); + for _ in 0..n { + constants.push(self.constant(filename, depth)?); + } + let names = self.strs()?; + let varnames = self.strs()?; + let freevars = self.strs()?; + let cellvars = self.strs()?; + let n = self.len()?; + let mut exception_table = Vec::with_capacity(n); + for _ in 0..n { + exception_table.push(ExcHandler { + start: self.u32()?, + end: self.u32()?, + handler: self.u32()?, + depth: self.u32()?, + push_lasti: self.byte()? != 0, + }); + } + let linetable = self.deltas()?; + let end_lines = self.deltas()?; + let mut coltable = Vec::with_capacity(end_lines.len()); + for end_lineno in end_lines { + coltable.push(ColSpan { + end_lineno, + col: self.i32()?, + end_col: self.i32()?, + }); + } + let arg_count = self.u32()?; + let posonly_count = self.u32()?; + let kwonly_count = self.u32()?; + let bits = self.uint()?; + let flag = |i: u32| bits & (1 << i) != 0; + let future_flags = self.u32()?; + let stacksize = self.u32()?.checked_sub(1); + let no_interrupt_jumps = self.u32s()?; + let wire_marks = self.raw()?.to_vec(); + let hidden_locals = self.strs()?; + let const_identifiers = self.strs()?; + Some(CodeObject { + name, + qualname, + filename: filename.to_owned(), + caches: CacheTable::with_len(instructions.len()), + instructions, + constants, + names, + varnames, + freevars, + cellvars, + exception_table, + linetable, + coltable, + arg_count, + posonly_count, + kwonly_count, + has_varargs: flag(0), + has_varkeywords: flag(1), + is_class_body: flag(2), + is_generator: flag(3), + is_coroutine: flag(4), + is_async_generator: flag(5), + is_iterable_coroutine: flag(6), + has_docstring: flag(7), + is_method: flag(8), + is_nested: flag(9), + annotate_scope: flag(10), + future_flags, + stacksize, + no_interrupt_jumps, + wire_marks, + hidden_locals, + const_identifiers, + ..CodeObject::default() + }) + } + + fn constant(&mut self, filename: &str, depth: u32) -> Option { + Some(match self.byte()? { + 0 => Constant::None, + 1 => Constant::Bool(false), + 2 => Constant::Bool(true), + 3 => Constant::Int(self.int()?), + 4 => Constant::BigInt(num_bigint::BigInt::from_signed_bytes_le(self.raw()?)), + 5 => Constant::Float(self.f64()?), + 6 => Constant::Complex(self.f64()?, self.f64()?), + 7 => Constant::Str(self.string()?), + 8 => Constant::WStr(self.u32s()?), + 9 => Constant::Bytes(self.raw()?.to_vec()), + tag @ (10 | 11) => { + let n = self.len()?; + let mut items = Vec::with_capacity(n); + for _ in 0..n { + items.push(self.constant(filename, depth)?); + } + if tag == 10 { + Constant::Tuple(items) + } else { + Constant::FrozenSet(items) + } + } + 12 => Constant::Code(Arc::new(self.code(filename, depth + 1)?)), + 13 => Constant::Ellipsis, + 14 => Constant::Slice(Box::new(( + self.constant(filename, depth)?, + self.constant(filename, depth)?, + self.constant(filename, depth)?, + ))), + _ => return None, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn round_trips_constants_and_metadata() { + let inner = CodeObject { + name: "f".to_owned(), + qualname: "C.f".to_owned(), + filename: "m.py".to_owned(), + instructions: vec![ + Instruction::new(OpCode::Resume, 0), + Instruction::new(OpCode::LoadConst, 300), + Instruction::new(OpCode::ReturnValue, 0), + ], + constants: vec![Constant::Int(-5)], + varnames: vec!["x".to_owned()], + linetable: vec![1, 2, 2], + coltable: vec![ColSpan::default(); 3], + arg_count: 1, + is_generator: true, + is_nested: true, + stacksize: Some(3), + ..CodeObject::default() + }; + let code = CodeObject { + name: "".to_owned(), + qualname: "".to_owned(), + filename: "m.py".to_owned(), + instructions: vec![Instruction::new(OpCode::Nop, 0)], + constants: vec![ + Constant::None, + Constant::Bool(true), + Constant::Int(i64::MIN), + Constant::BigInt(num_bigint::BigInt::from(-1) << 100), + Constant::Float(-0.0), + Constant::Complex(1.5, f64::INFINITY), + Constant::Str("héllo".to_owned()), + Constant::WStr(vec![0x61, 0xd800]), + Constant::Bytes(vec![0, 255]), + Constant::Tuple(vec![Constant::Ellipsis, Constant::Str(String::new())]), + Constant::FrozenSet(vec![Constant::Int(1)]), + Constant::Slice(Box::new(( + Constant::Int(1), + Constant::None, + Constant::Int(-1), + ))), + Constant::Code(Arc::new(inner)), + ], + names: vec!["print".to_owned()], + exception_table: vec![ExcHandler { + start: 0, + end: 1, + handler: 1, + depth: 2, + push_lasti: true, + }], + no_interrupt_jumps: vec![7], + wire_marks: vec![0, 3], + hidden_locals: vec!["h".to_owned()], + const_identifiers: vec!["k".to_owned()], + future_flags: 0x100, + ..CodeObject::default() + }; + let bytes = encode(&code).expect("encodable"); + let back = decode(&bytes, "m.py").expect("decodable"); + assert_eq!(back, code); + } + + #[test] + fn rejects_truncated_or_foreign_input() { + let code = CodeObject { + name: "".to_owned(), + instructions: vec![Instruction::new(OpCode::Nop, 0)], + constants: vec![Constant::Str("x".repeat(40))], + ..CodeObject::default() + }; + let bytes = encode(&code).expect("encodable"); + for cut in 0..bytes.len() { + assert!(decode(&bytes[..cut], "").is_none()); + } + let mut other = bytes.clone(); + other[0] = VERSION + 1; + assert!(decode(&other, "").is_none()); + assert!(encode(&CodeObject { + constants: vec![Constant::Unmarshallable], + ..CodeObject::default() + }) + .is_none()); + } +} diff --git a/crates/weavepy-vm/src/frozen_code_cache.rs b/crates/weavepy-vm/src/frozen_code_cache.rs index 749f86e7..dbb2a7b9 100644 --- a/crates/weavepy-vm/src/frozen_code_cache.rs +++ b/crates/weavepy-vm/src/frozen_code_cache.rs @@ -43,10 +43,6 @@ use std::sync::OnceLock; use weavepy_compiler::CodeObject; -use crate::object::Object; -use crate::stdlib::marshal_mod; -use crate::sync::Rc; - thread_local! { static CACHE: RefCell> = RefCell::new(HashMap::new()); } @@ -95,27 +91,38 @@ pub fn insert(name: &str, code: &CodeObject) { // persists the marshalled `CodeObject`s in a per-user cache directory // so warm process starts skip parse + compile entirely. // -// Artifact layout: `/weavepy/frozen-/` with -// a 20-byte header — magic `WPYF`, reserved flags word, source length, -// and an FNV-1a 64 source hash — followed by `marshal.dumps(code)`. -// The `CACHE_TAG` in the directory name invalidates on bytecode-format -// revisions (same lever as `.pyc`); the length + hash pair invalidates +// Artifact layout: `/weavepy/frozen--n/` +// with a 20-byte header — magic `WPYN`, reserved flags word, source +// length, and a 64-bit source hash — followed by the code in +// `weavepy_compiler::native_code` form, which loads without the CPython +// bytecode transcoding a `marshal` payload needs. The `CACHE_TAG` and +// native-format `VERSION` in the directory name invalidate on bytecode or +// layout revisions (and keep binaries of different formats from +// rewriting each other's artifacts); the length + hash pair invalidates // when the embedded source itself changes (a rebuilt binary with edited // stdlib). Corrupt or mismatched artifacts are treated as misses. /// Header magic for frozen-cache artifacts (distinct from `.pyc`'s -/// CPython magic — these files are WeavePy-internal). -const FROZEN_MAGIC: &[u8; 4] = b"WPYF"; +/// CPython magic: these files are WeavePy-internal). +const FROZEN_MAGIC: &[u8; 4] = b"WPYN"; const FROZEN_HEADER_LEN: usize = 20; -/// FNV-1a 64-bit — tiny, dependency-free, and plenty for cache -/// validation (collisions only matter combined with an equal length). +/// FNV-1a 64-bit over 8-byte words (the tail zero-padded): tiny, +/// dependency-free, and plenty for cache validation (collisions only +/// matter combined with an equal length). fn fnv1a(s: &str) -> u64 { let mut h: u64 = 0xcbf2_9ce4_8422_2325; - for b in s.bytes() { - h ^= u64::from(b); + let mut words = s.as_bytes().chunks_exact(8); + let mut mix = |w: u64| { + h ^= w; h = h.wrapping_mul(0x0000_0100_0000_01b3); + }; + for w in &mut words { + mix(u64::from_le_bytes(w.try_into().expect("an 8-byte chunk"))); } + let mut tail = [0u8; 8]; + tail[..words.remainder().len()].copy_from_slice(words.remainder()); + mix(u64::from_le_bytes(tail)); h } @@ -144,10 +151,11 @@ fn disk_dir() -> Option<&'static PathBuf> { base } }; - Some( - base.join("weavepy") - .join(format!("frozen-{}", crate::pycache::CACHE_TAG)), - ) + Some(base.join("weavepy").join(format!( + "frozen-{}-n{}", + crate::pycache::CACHE_TAG, + weavepy_compiler::native_code::VERSION + ))) }) .as_ref() } @@ -172,20 +180,11 @@ pub fn get_disk(name: &str, source: &str, filename: &str) -> Option if len as usize != source.len() || hash != fnv1a(source) { return None; } - match marshal_mod::load_from_bytes(&bytes[FROZEN_HEADER_LEN..]).ok()? { - Object::Code(c) => { - let mut code = crate::pycache::own_decoded_code(c); - if code.filename != filename { - crate::pycache::rewrite_filenames(&mut code, filename); - } - // Not mirrored into the in-memory cache: a later interpreter - // reads the same artifact again, and a resident clone of every - // loaded module's code would double its footprint in the - // (usual) single-interpreter process. - Some(code) - } - _ => None, - } + // Not mirrored into the in-memory cache: a later interpreter reads the + // same artifact again, and a resident clone of every loaded module's + // code would double its footprint in the (usual) single-interpreter + // process. + weavepy_compiler::native_code::decode(&bytes[FROZEN_HEADER_LEN..], filename) } /// Persist a freshly-compiled frozen module to the disk cache. @@ -198,15 +197,15 @@ pub fn write_disk(name: &str, source: &str, code: &CodeObject) { if let Some(parent) = path.parent() { let _ = std::fs::create_dir_all(parent); } - let mut bytes = Vec::with_capacity(FROZEN_HEADER_LEN + 4096); + // A code object the native form doesn't carry just isn't cached. + let Some(payload) = weavepy_compiler::native_code::encode(code) else { + return; + }; + let mut bytes = Vec::with_capacity(FROZEN_HEADER_LEN + payload.len()); bytes.extend_from_slice(FROZEN_MAGIC); bytes.extend_from_slice(&0u32.to_le_bytes()); bytes.extend_from_slice(&(source.len() as u32).to_le_bytes()); bytes.extend_from_slice(&fnv1a(source).to_le_bytes()); - let Ok(Object::Bytes(payload)) = marshal_mod::b_dumps(&[Object::Code(Rc::new(code.clone()))]) - else { - return; - }; bytes.extend_from_slice(&payload); // Atomic-ish: temp + rename so concurrent starts never observe a // half-written artifact. From 902ecdbff822d9082d1b6c71306a63403b75e3f2 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 01:55:30 -0700 Subject: [PATCH 58/65] perf: keep dormant GC suspects out of the per-drop sweep Suspects live in two maps, active and dormant, so the sweep that runs at drop safe points walks only the entries with probe budget left; the dormant ones are visited on the stride, as before. During imports the old single map made every sweep walk up to 256 entries. `import logging` drops 3.5% and `import json` 7%. --- crates/weavepy-vm/src/gc_trace.rs | 235 +++++++++++++++--------------- 1 file changed, 121 insertions(+), 114 deletions(-) diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index 99dcd3b0..7dda0326 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -3914,8 +3914,29 @@ struct Suspect { dormant_probes: u8, } type SuspectMap = indexmap::IndexMap>; -static SUSPECTS: std::sync::LazyLock> = - std::sync::LazyLock::new(|| parking_lot::Mutex::new(SuspectMap::default())); +/// The enrolled suspects, split by phase: entries with probe budget left +/// (re-probed at every sweep) and aged-out dormant ones (re-probed only on +/// the stride), so an ordinary sweep never walks the dormant population. +#[derive(Default)] +struct Suspects { + /// Every entry has budget remaining. + active: SuspectMap, + /// Every entry's budget is spent. + dormant: SuspectMap, +} +impl Suspects { + fn len(&self) -> usize { + self.active.len() + self.dormant.len() + } + fn contains_key(&self, id: ObjectId) -> bool { + self.active.contains_key(&id) || self.dormant.contains_key(&id) + } + fn values(&self) -> impl Iterator { + self.active.values().chain(self.dormant.values()) + } +} +static SUSPECTS: std::sync::LazyLock> = + std::sync::LazyLock::new(|| parking_lot::Mutex::new(Suspects::default())); static SUSPECT_COUNT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); /// Entries with probe budget remaining. When only dormant entries are /// left, [`has_suspects`] admits a sweep every [`DORMANT_STRIDE`]-th @@ -4069,46 +4090,38 @@ fn residual_suspects() -> Vec<(String, usize)> { .collect() } -/// Publish the count gates after the locked suspect map changed. -/// `active` is the caller's count of budget-remaining entries (kept -/// incrementally; RFC 0077 WS2 retired the O(n) recount this used to -/// do on every enrollment and removal). -fn publish_suspect_counts(s: &SuspectMap, active: usize) { +/// Publish the count gates after the locked suspect maps changed. +fn publish_suspect_counts(s: &Suspects) { + let active = s.active.len(); let was_active = SUSPECT_ACTIVE.swap(active, Ordering::AcqRel) > 0; let was_present = SUSPECT_COUNT.swap(s.len(), Ordering::AcqRel) > 0; // RFC 0065 (WS1): the dispatch loops' quiet-path snapshots consult // `active_suspects_present` / the population gates, so a // transition of either invalidates them. Same-state churn (one // active suspect replacing another) doesn't. - if was_active != (active > 0) || was_present != (!s.is_empty()) { + if was_active != (active > 0) || was_present != (s.len() > 0) { crate::hot_gates::bump_loop_gen(); } } -/// Evict the first minimum-budget entry, preserving `min_by_key`'s tie -/// order. Zero is the minimum possible budget, so scanning can stop there. -/// The caller holds the map lock and its exact active count throughout. -fn evict_lowest_budget_suspect(s: &mut SuspectMap, active: &mut usize) -> bool { - let Some((_, first)) = s.get_index(0) else { +/// Evict the entry that has had the most chances to die: the oldest +/// dormant one, or with none, the first lowest-budget active one. The +/// caller holds the maps' lock and republishes the counts. +fn evict_lowest_budget_suspect(s: &mut Suspects) -> bool { + if !s.dormant.is_empty() { + s.dormant.shift_remove_index(0); + return true; + } + let Some(victim) = s + .active + .values() + .enumerate() + .min_by_key(|(_, entry)| entry.budget) + .map(|(index, _)| index) + else { return false; }; - let mut victim = 0; - let mut budget = first.budget; - if budget > 0 { - for (index, entry) in s.values().enumerate().skip(1) { - if entry.budget < budget { - victim = index; - budget = entry.budget; - if budget == 0 { - break; - } - } - } - } - let (_, removed) = s - .swap_remove_index(victim) - .expect("selected suspect exists"); - *active = active.saturating_sub(usize::from(removed.budget > 0)); + s.active.swap_remove_index(victim); true } @@ -4133,7 +4146,7 @@ pub fn note_suspect(h: HandleRc) { return; } let mut s = SUSPECTS.lock(); - if s.contains_key(&h.id) { + if s.contains_key(h.id) { return; } // RFC 0065 (WS4): publish to the miss-filter before the enrollment @@ -4150,7 +4163,6 @@ pub fn note_suspect(h: HandleRc) { crate::weakref_registry::strong_clone_count(h.id), Ordering::Release, ); - let mut active = SUSPECT_ACTIVE.load(Ordering::Relaxed); if s.len() >= SUSPECT_CAP { // Full: evict the most-probed entry (lowest remaining budget, // dormant first) — it has had the most chances to die and is @@ -4160,12 +4172,12 @@ pub fn note_suspect(h: HandleRc) { // arrived after ~200 module-teardown stragglers and was never // re-probed, pinning the Timeout→Task→frame web the test_ssl // leak tests watch). - if !evict_lowest_budget_suspect(&mut s, &mut active) { + if !evict_lowest_budget_suspect(&mut s) { return; } } floor_stats::bump(&floor_stats::SUSPECT_ENROLLED, 1); - s.insert( + s.active.insert( h.id, Suspect { handle: h, @@ -4173,7 +4185,7 @@ pub fn note_suspect(h: HandleRc) { dormant_probes: 0, }, ); - publish_suspect_counts(&s, active + 1); + publish_suspect_counts(&s); } /// Cheap gate for the eval loop's safe point: always sweep while an @@ -4215,14 +4227,9 @@ pub fn remove_suspect(id: ObjectId) { return; } let mut s = SUSPECTS.lock(); - if let Some(removed) = s.swap_remove(&id) { - let active = SUSPECT_ACTIVE.load(Ordering::Relaxed); - let active = if removed.budget > 0 { - active.saturating_sub(1) - } else { - active - }; - publish_suspect_counts(&s, active); + let removed = s.active.swap_remove(&id).is_some() || s.dormant.swap_remove(&id).is_some(); + if removed { + publish_suspect_counts(&s); } } @@ -4235,24 +4242,21 @@ pub fn take_dead_suspects() -> Vec { let mut s = SUSPECTS.lock(); // With only dormant entries left, `has_suspects` already stride-gated // this sweep; with actives present the stride ticks here instead. - let probe_dormant = SUSPECT_ACTIVE.load(Ordering::Relaxed) == 0 + let probe_dormant = s.active.is_empty() || SUSPECT_TICK .fetch_add(1, Ordering::Relaxed) .is_multiple_of(DORMANT_STRIDE); floor_stats::bump(&floor_stats::SUSPECT_SWEEPS, 1); - let mut active = 0usize; let mut probed = 0u64; - s.retain(|_, e| { - // Dormant (aged-out) entries only pay on the stride tick. - if e.budget == 0 && !probe_dormant { - return true; - } + // A suspect whose last reference is gone (beyond the handle and any + // weakref strong clones) goes to `out`. Returns whether it stays. + let mut dead = |e: &Suspect, out: &mut Vec| -> Option { probed += 1; let h = &e.handle; // RFC 0061 (WS1b): the handle self-identifies as reclaimed (set // in lock-step with every index removal) — no registry lookup. if h.untracked.load(Ordering::Acquire) { - return false; // already reclaimed elsewhere + return None; // already reclaimed elsewhere } // Fast reject via the cached weakref-clone upper bound (refreshed // at enrollment): more strong refs than the handle plus every @@ -4265,31 +4269,52 @@ pub fn take_dead_suspects() -> Vec { let weak = crate::weakref_registry::strong_clone_count(h.id); if sc <= 1 + weak { out.push(h.object.clone()); - return false; + return None; } } + Some(sc.saturating_sub(1 + cached)) + }; + // Active entries spend a probe; one whose budget runs out turns + // dormant after this sweep (it isn't stride-probed in the same one). + let mut aged = Vec::new(); + s.active.retain(|&id, e| { + if dead(e, &mut out).is_none() { + return false; + } + e.budget -= 1; + if e.budget == 0 { + aged.push(( + id, + Suspect { + handle: e.handle.clone(), + budget: 0, + dormant_probes: e.dormant_probes, + }, + )); + return false; + } + true + }); + if probe_dormant { // A dormant entry still well above its dead line, or one that has // sat through its dormant allowance, is plainly alive: drop it // from the map (see `SUSPECT_LIVE_MARGIN` / `SUSPECT_DORMANT_PROBES`). - if e.budget == 0 { + s.dormant.retain(|_, e| { + let Some(excess) = dead(e, &mut out) else { + return false; + }; e.dormant_probes = e.dormant_probes.saturating_add(1); - if sc > 1 + cached + SUSPECT_LIVE_MARGIN || e.dormant_probes > SUSPECT_DORMANT_PROBES { + if excess > SUSPECT_LIVE_MARGIN || e.dormant_probes > SUSPECT_DORMANT_PROBES { floor_stats::bump(&floor_stats::SUSPECT_FORGOTTEN, 1); return false; } - return true; - } - e.budget -= 1; - if e.budget > 0 { - active += 1; - } - true - }); + true + }); + } + s.dormant.extend(aged); floor_stats::bump(&floor_stats::SUSPECT_PROBED, probed); floor_stats::bump(&floor_stats::SUSPECT_DEAD, out.len() as u64); - // Entries skipped as dormant stayed dormant; entries probed were - // recounted above, so `active` is exact. - publish_suspect_counts(&s, active); + publish_suspect_counts(&s); out } @@ -4659,57 +4684,39 @@ mod tests { use crate::object::DictData; #[test] - fn suspect_eviction_preserves_minimum_order_counts_and_release() { - for mode in 0..6 { - let mut suspects = SuspectMap::default(); - let mut reference = Vec::new(); - let mut handles = std::collections::HashMap::new(); - for index in 0..SUSPECT_CAP { - let budget = match mode { - 0 => 0, - 1 => SUSPECT_BUDGET, - 2 => u8::from(index + 1 != SUSPECT_CAP), - 3 => ((SUSPECT_CAP - index) % 16 + 1) as u8, - 4 => ((index * 37 + 113) % 17) as u8, - _ => u8::MAX, - }; - let object = Object::List(Rc::new(RefCell::new(Vec::new()))); - let handle = HandleRc::new(TrackedHandle::new(object, 0)); - let id = handle.id; - handles.insert(id, HandleRc::downgrade(&handle)); - suspects.insert( - id, - Suspect { - handle, - budget, - dormant_probes: (index % 16) as u8, - }, - ); - reference.push((id, budget)); - } - let mut active = reference.iter().filter(|(_, b)| *b > 0).count(); - while !reference.is_empty() { - // The old policy is the oracle, including the first tied - // minimum and the order produced by each swap removal. - let victim = reference - .iter() - .enumerate() - .min_by_key(|(_, (_, budget))| *budget) - .map(|(index, _)| index) - .unwrap(); - let (removed_id, _) = reference.swap_remove(victim); - assert!(evict_lowest_budget_suspect(&mut suspects, &mut active)); - assert_eq!(active, reference.iter().filter(|(_, b)| *b > 0).count()); - let actual: Vec<_> = suspects - .iter() - .map(|(id, entry)| (*id, entry.budget)) - .collect(); - assert_eq!(actual, reference); - assert!(handles[&removed_id].upgrade().is_none()); + fn suspect_eviction_takes_dormant_then_lowest_budget_and_releases() { + let mut suspects = Suspects::default(); + let mut handles = Vec::new(); + for index in 0..8usize { + let object = Object::List(Rc::new(RefCell::new(Vec::new()))); + let handle = HandleRc::new(TrackedHandle::new(object, 0)); + handles.push((handle.id, HandleRc::downgrade(&handle))); + let budget = [0, 5, 0, 3, 9, 3, 0, 1][index]; + let entry = Suspect { + handle, + budget, + dormant_probes: 0, + }; + if budget == 0 { + suspects.dormant.insert(handles[index].0, entry); + } else { + suspects.active.insert(handles[index].0, entry); } - assert!(!evict_lowest_budget_suspect(&mut suspects, &mut active)); - assert_eq!(active, 0); } + // Dormant entries go first, oldest first. + for expected in [0, 2, 6] { + assert!(evict_lowest_budget_suspect(&mut suspects)); + assert!(!suspects.contains_key(handles[expected].0)); + assert!(handles[expected].1.upgrade().is_none()); + } + // Then active ones by budget: 1, the first 3, the other 3, 5, 9. + for expected in [7, 3, 5, 1, 4] { + assert!(evict_lowest_budget_suspect(&mut suspects)); + assert!(!suspects.contains_key(handles[expected].0)); + assert!(handles[expected].1.upgrade().is_none()); + } + assert_eq!(suspects.len(), 0); + assert!(!evict_lowest_budget_suspect(&mut suspects)); } #[test] From 80e63788a0af84d0cffec21f30f1ab64ab00c7fa Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 03:02:45 -0700 Subject: [PATCH 59/65] perf: read class scalars through instances and fuse local branches - A class's scalar attribute read through an instance (`self.LIMIT`) is served from the site's stamp slot while the class version holds and the instance doesn't shadow the name, instead of resolving the name on every read. - The fused `x.attr` arm reads the site's field shortcut in line. - `if x is None`, `if x is not None`, and `if x:` on a local decide the branch from the local in place (no load, clone, or release), and a conditional jump steps over the `NOT_TAKEN` that follows it. - `POP_TOP` of a scalar no longer calls into the drop glue. A class scalar read through an instance drops from 730 to 280 instructions, and an `is None` branch by a fifth. --- crates/weavepy-vm/src/lib.rs | 149 +++++++++++++++++++++- tests/regrtest/test_core_branch_fusion.py | 100 +++++++++++++++ 2 files changed, 246 insertions(+), 3 deletions(-) create mode 100644 tests/regrtest/test_core_branch_fusion.py diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 3b19496a..db93d99e 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -12184,6 +12184,52 @@ impl Interpreter { } } } + // A branch on a local, read in place (nothing is + // pushed, so nothing is cloned or released). + 3 | 4 => { + // SAFETY: `i < nlocals` (checked above), and a + // pair kind is only recorded where the branch + // instruction exists. + let v = unsafe { &*lbase.add(i) }; + let jump_pc = pc + 1 + usize::from(fast_pairs[pc] == 4); + let jump = unsafe { *instrs.add(jump_pc) }; + let taken = match (jump.op, v) { + (_, Object::Unbound) => None, + (OpCode::PopJumpIfNone, v) => Some(matches!(v, Object::None)), + (OpCode::PopJumpIfNotNone, v) => { + Some(!matches!(v, Object::None)) + } + (op, v) => { + let truthy = match v { + Object::Bool(b) => Some(*b), + Object::Int(n) => Some(*n != 0), + Object::None => Some(false), + Object::Float(f) => Some(*f != 0.0), + Object::Str(s) => Some(!s.is_empty()), + Object::Tuple(t) => Some(!t.is_empty()), + // SAFETY: a read between two instructions. + Object::List(l) => { + unsafe { l.peek() }.map(|l| !l.is_empty()) + } + _ => None, + }; + truthy.map(|t| t == (op == OpCode::PopJumpIfTrue)) + } + }; + if let Some(taken) = taken { + last = jump_pc; + pc = jump_pc + 1; + if taken { + pc += jump.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; + } + continue; + } + } _ => {} } // SAFETY: `i < nlocals`, `len < cap`. @@ -12357,6 +12403,18 @@ impl Interpreter { pc = end + 1; continue; } + } else if let (Some(ext), Object::Instance(inst)) = + (ext, other) + { + // The site's field shortcut, in line (the + // common monomorphic read). + if let Some(v) = field_slot_hit(ext, pc + 1, inst) { + base.add(len).write(clone_hot(v)); + len += 1; + last = pc + 1; + pc += 2; + continue; + } } if let Some(v) = Self::core_local_attr( ext, @@ -12604,8 +12662,16 @@ impl Interpreter { break None; } len -= 1; + // A scalar just leaves (tested apart from the drop, + // or the scalar case folds into the drop glue call). // SAFETY: the slot is initialized and leaves the stack. - unsafe { drop_hot(base.add(len).read()) }; + if !matches!( + unsafe { &*base.add(len) }, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) { + // SAFETY: as above. + unsafe { drop_hot(base.add(len).read()) }; + } last = pc; pc += 1; } @@ -12715,6 +12781,11 @@ impl Interpreter { pc += 1; if is_none == (ins.op == OpCode::PopJumpIfNone) { pc += ins.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; } } OpCode::BinaryOp => { @@ -13021,6 +13092,11 @@ impl Interpreter { pc += 1; if truthy == (ins.op == OpCode::PopJumpIfTrue) { pc += ins.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; } } OpCode::JumpForward => { @@ -17471,6 +17547,18 @@ impl Interpreter { if let Some(v) = unsafe { field_slot_hit(ext, attr_pc, inst) } { return Some(Self::clone_operand(v)); } + // A scalar class attribute the instance doesn't shadow (see + // `leaf_attr_resolve_site`). + if let Some(v) = ext + .stamp_slots + .get() + .and_then(|s| s.get(attr_pc)) + .and_then(|s| class_attr_hit_via(s, inst.cls_raw(), CLASS_ATTR_VIA_INSTANCE)) + { + if !inst_may_shadow(inst, code, name_idx) { + return Some(v); + } + } } if let (IC::LoadAttrInstance { key_idx, ver }, Object::Instance(inst)) = (code.caches.get(attr_pc as u32), local) @@ -20824,6 +20912,15 @@ impl Interpreter { name_idx: u32, ) -> Option { use weavepy_compiler::InlineCache as IC; + // A scalar class attribute read through an instance that doesn't + // shadow it (the stamp was filled below, at this class version). + if let Some(v) = code_stamp_slot(code, cache_pc) + .and_then(|s| class_attr_hit_via(s, inst.cls_raw(), CLASS_ATTR_VIA_INSTANCE)) + { + if !inst_may_shadow(inst, code, name_idx) { + return Some(v); + } + } let poly = code_attr_poly(code, cache_pc); let ver = inst.cls_raw().attr_version.get(); // A never-specialized site resolves once and takes the indexed @@ -20843,6 +20940,15 @@ impl Interpreter { } else if let Some(p) = poly { p.record(ver, ix); } + } else { + // Found on the class: remember a scalar for this site. + class_attr_fill_via( + code, + inst.cls_raw(), + cache_pc as usize, + &v, + CLASS_ATTR_VIA_INSTANCE, + ); } Some(v) } @@ -63688,11 +63794,26 @@ const CLASS_ATTR_NONE: u64 = 0x5ca1_a770_0000_0004; /// `cls`'s current version (see [`class_attr_fill`]). #[inline(always)] fn class_attr_hit(slot: &StampSlot, cls: &crate::types::TypeObject) -> Option { + class_attr_hit_via(slot, cls, 0) +} + +/// Marks a stamp remembered for a read through an *instance* of the +/// class, which also needs the instance not to shadow the name. +const CLASS_ATTR_VIA_INSTANCE: u64 = 0x100; + +/// [`class_attr_hit`] for a stamp filled with `via` (`0`, or +/// [`CLASS_ATTR_VIA_INSTANCE`]). +#[inline(always)] +fn class_attr_hit_via( + slot: &StampSlot, + cls: &crate::types::TypeObject, + via: u64, +) -> Option { let [ver, tag, bits] = slot.get(); if ver != cls.attr_version.get() { return None; } - match tag { + match tag ^ via { CLASS_ATTR_INT => Some(Object::Int(bits as i64)), CLASS_ATTR_FLOAT => Some(Object::Float(f64::from_bits(bits))), CLASS_ATTR_BOOL => Some(Object::Bool(bits != 0)), @@ -63707,6 +63828,17 @@ fn class_attr_hit(slot: &StampSlot, cls: &crate::types::TypeObject) -> Option (CLASS_ATTR_INT, *x as u64), Object::Float(x) => (CLASS_ATTR_FLOAT, x.to_bits()), @@ -63715,7 +63847,7 @@ fn class_attr_fill(code: &CodeObject, cls: &crate::types::TypeObject, pc: usize, _ => return, }; if let Some(slot) = code_stamp_slot(code, pc as u32) { - slot.set([cls.attr_version.get(), tag, bits]); + slot.set([cls.attr_version.get(), tag ^ via, bits]); } } @@ -64167,6 +64299,17 @@ fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> Some(OpCode::LoadAttr | OpCode::StoreAttr | OpCode::LoadMethodAttr) )), Some(OpCode::StoreFast) => 2, + // A local's `is None` / `is not None` test. + Some(OpCode::PopJumpIfNone | OpCode::PopJumpIfNotNone) => 3, + // A local's truth test. + Some(OpCode::ToBool) + if matches!( + ops.get(pc + 2).map(|i| i.op), + Some(OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue) + ) => + { + 4 + } _ => 0, }; kinds[pc] = kind; diff --git a/tests/regrtest/test_core_branch_fusion.py b/tests/regrtest/test_core_branch_fusion.py new file mode 100644 index 00000000..406f1de0 --- /dev/null +++ b/tests/regrtest/test_core_branch_fusion.py @@ -0,0 +1,100 @@ +"""Branches on locals and class attributes read through instances. + +The core loop decides `if x is None`, `if x is not None` and `if x:` on +a local without loading it, and serves a class's scalar attribute read +through an instance from a site cache. These cases check each shape's +edges: every truth kind, unbound locals, and an instance attribute or +class change that shadows the cached value. +""" + +import unittest + + +class Holder: + LIMIT = 3 + RATE = 0.5 + FLAG = True + NOTHING = None + + +def branches(values): + out = [] + for v in values: + if v is None: + out.append("none") + if v is not None: + out.append("some") + if v: + out.append("truthy") + else: + out.append("falsy") + return out + + +class BranchFusionTests(unittest.TestCase): + def test_truth_kinds(self): + values = [None, 0, 1, -1, 0.0, 0.5, "", "x", (), (1,), [], [1], + {}, {1: 2}, True, False, object()] + for _ in range(3): + self.assertEqual(branches(values), [ + x for v in values for x in ( + (["none"] if v is None else ["some"]) + + (["truthy"] if v else ["falsy"])) + ]) + + def test_custom_truth(self): + class Falsy: + def __bool__(self): + return False + + class Empty: + def __len__(self): + return 0 + + for _ in range(3): + self.assertEqual(branches([Falsy(), Empty()]), + ["some", "falsy", "some", "falsy"]) + + def test_unbound_local(self): + def f(flag): + if flag: + x = None + if x is None: + return "none" + return "other" + + for _ in range(3): + self.assertEqual(f(True), "none") + with self.assertRaises(UnboundLocalError): + f(False) + + def test_class_scalars_through_instances(self): + h = Holder() + for _ in range(50): + self.assertEqual((h.LIMIT, h.RATE, h.FLAG, h.NOTHING), + (3, 0.5, True, None)) + h.LIMIT = 7 + self.assertEqual(h.LIMIT, 7) + del h.LIMIT + self.assertEqual(h.LIMIT, 3) + Holder.LIMIT = 9 + try: + self.assertEqual([h.LIMIT for _ in range(5)], [9] * 5) + finally: + Holder.LIMIT = 3 + + def test_shadowing_through_dict_and_subclass(self): + class Sub(Holder): + pass + + s = Sub() + for _ in range(20): + self.assertEqual(s.LIMIT, 3) + Sub.LIMIT = 4 + self.assertEqual(s.LIMIT, 4) + s.__dict__["LIMIT"] = 5 + self.assertEqual(s.LIMIT, 5) + + +if __name__ == "__main__": + unittest.main() From 32fae0b638baa09f28e29cd2c34d0fdca4dfb82a Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 03:21:33 -0700 Subject: [PATCH 60/65] perf: rebind existing globals without invalidating cached global loads `STORE_GLOBAL` and module-level `STORE_NAME` of an existing name used to stamp the module dict like any mutation, so every cached global load in the module missed after each store. Rebinding now replaces the value in place: the key layout, and so the stamp the global-load caches check, stays put. The JIT's burned-in globals, which remember values rather than positions, also watch a process-wide epoch that each in-place rebinding advances. The core loop runs `STORE_GLOBAL` itself, as it already did module-level `STORE_NAME`. A global store drops from 935 to 320 instructions, on par with CPython, and a call to a function that updates a global counter from 3.7K to 1.8K. --- crates/weavepy-vm/src/lib.rs | 40 ++++++++- crates/weavepy-vm/src/object.rs | 17 ++++ crates/weavepy-vm/src/tier2.rs | 11 ++- tests/regrtest/test_global_rebinding.py | 112 ++++++++++++++++++++++++ 4 files changed, 172 insertions(+), 8 deletions(-) create mode 100644 tests/regrtest/test_global_rebinding.py diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index db93d99e..4d9ff652 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -13613,7 +13613,9 @@ impl Interpreter { // Rebinding an existing module-scope name whose old value // leaves by a plain drop; a new name, a finalizer's // candidate or a watched dict takes the full handler. - OpCode::StoreName if self.core_name_scope(frame) => { + OpCode::StoreName | OpCode::StoreGlobal + if ins.op == OpCode::StoreGlobal || self.core_name_scope(frame) => + { if len == 0 || crate::capi_watchers::dicts_active() { break None; } @@ -13635,9 +13637,12 @@ impl Interpreter { { break None; } - let Some((_, slot)) = (**g).get_index_mut(i) else { + // The key layout stays: no stamp (every cached global + // load stands), but the value epoch moves. + let Some((_, slot)) = g.map_mut_value_store().get_index_mut(i) else { break None; }; + crate::object::bump_global_value_epoch(); // SAFETY: `len > 0`; the value moves into the binding and // the displaced one was checked droppable. let old = std::mem::replace(slot, unsafe { base.add(len - 1).read() }); @@ -22393,7 +22398,19 @@ impl Interpreter { let old = if let Some(ns) = &frame.class_namespace { ns.borrow_mut().insert(DictKey(key), v) } else { - frame.globals.borrow_mut().insert(DictKey(key), v) + // A module-level rebinding keeps the key layout (see + // `STORE_GLOBAL`). + let key = DictKey(key); + let mut g = frame.globals.borrow_mut(); + match g.get_index_of(&key) { + Some(i) => { + crate::object::bump_global_value_epoch(); + g.map_mut_value_store() + .get_index_mut(i) + .map(|(_, slot)| std::mem::replace(slot, v)) + } + None => g.insert(key, v), + } }; if let Some(old) = old { if Self::local_needs_prompt_reap(&old) { @@ -22408,7 +22425,22 @@ impl Interpreter { Some(n @ Object::Str(_)) => n.clone(), _ => Object::from_str(self.name_at(&frame.code, ins.arg)?), }; - let old = frame.globals.borrow_mut().insert(DictKey(key), v); + // Rebinding an existing global keeps the dict's key layout, so + // it leaves the stamp (and every cached global load) standing; + // only a new name stamps it. + let key = DictKey(key); + let old = { + let mut g = frame.globals.borrow_mut(); + match g.get_index_of(&key) { + Some(i) => { + crate::object::bump_global_value_epoch(); + g.map_mut_value_store() + .get_index_mut(i) + .map(|(_, slot)| std::mem::replace(slot, v)) + } + None => g.insert(key, v), + } + }; // Rebinding a global that uniquely held a finalizable runs its // `__del__` now, matching the `DeleteGlobal` path and CPython's // decref-on-store (module-scope `handle = None`). diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index 5c547af9..22b4bbbe 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -4036,6 +4036,23 @@ pub type DictMap = indexmap::IndexMap u64 { + GLOBAL_VALUE_EPOCH.load(std::sync::atomic::Ordering::Relaxed) +} + +/// Advance [`GLOBAL_VALUE_EPOCH`]. +#[inline] +pub fn bump_global_value_epoch() { + GLOBAL_VALUE_EPOCH.fetch_add(1, std::sync::atomic::Ordering::Relaxed); +} + #[inline] fn next_dict_stamp() -> u64 { use std::sync::atomic::Ordering::Relaxed; diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index 0d170141..ae5f98b0 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -203,16 +203,17 @@ impl MathGuard { /// probes are skipped (a resume or entry then costs two stamp reads). struct GuardSnapshot { entries: Vec<(String, Object)>, - /// `(globals id, globals stamp, builtins id, builtins stamp)` of the - /// last full validation that held; all-zero until one has. - last_ok: std::cell::Cell<(usize, u64, usize, u64)>, + /// `(globals id, globals stamp, builtins id, builtins stamp, global + /// value epoch)` of the last full validation that held; all-zero + /// until one has. + last_ok: std::cell::Cell<(usize, u64, usize, u64, u64)>, } impl GuardSnapshot { fn new(entries: Vec<(String, Object)>) -> Self { Self { entries, - last_ok: std::cell::Cell::new((0, 0, 0, 0)), + last_ok: std::cell::Cell::new((0, 0, 0, 0, 0)), } } } @@ -3311,6 +3312,8 @@ fn guards_hold( unsafe { (*globals.as_ptr()).mutation_stamp() }, Rc::as_ptr(builtins) as usize, unsafe { (*builtins.as_ptr()).mutation_stamp() }, + // A rebinding in place leaves the stamps alone (see `STORE_GLOBAL`). + crate::object::global_value_epoch(), ); if guard_snapshot.last_ok.get() != key || interp.globals_missing_any.get() { for (name, expected) in guard_snapshot.entries.iter() { diff --git a/tests/regrtest/test_global_rebinding.py b/tests/regrtest/test_global_rebinding.py new file mode 100644 index 00000000..a456d7ab --- /dev/null +++ b/tests/regrtest/test_global_rebinding.py @@ -0,0 +1,112 @@ +"""Rebinding globals while code that reads them stays hot. + +Rebinding an existing global replaces its value in place, which leaves +the module dict's key layout (and every cached global load) standing; +caches that remember a global's value must still see the new one. +These cases rebind functions, classes and scalars from inside +functions, at module level, and through `globals()`, while loops that +read them run long enough to be specialized and compiled. +""" + +import unittest + +COUNTER = 0 +SCALE = 2 + + +def double(x): + return x * 2 + + +def triple(x): + return x * 3 + + +def apply_all(n): + total = 0 + for i in range(n): + total += double(i) + return total + + +def scaled(n): + total = 0 + for i in range(n): + total += i * SCALE + return total + + +def bump(n): + global COUNTER + for _ in range(n): + COUNTER += 1 + return COUNTER + + +def rebind_double(fn): + global double + double = fn + + +class GlobalRebindingTests(unittest.TestCase): + def tearDown(self): + global double, SCALE + double = GlobalRebindingTests._double + SCALE = 2 + + def test_rebound_function_is_seen(self): + for _ in range(5): + self.assertEqual(apply_all(1000), 999000) + rebind_double(triple) + for _ in range(5): + self.assertEqual(apply_all(1000), 1498500) + rebind_double(lambda x: 0) + self.assertEqual(apply_all(1000), 0) + + def test_rebound_scalar_is_seen(self): + global SCALE + for _ in range(5): + self.assertEqual(scaled(100), 9900) + SCALE = 5 + self.assertEqual(scaled(100), 24750) + globals()["SCALE"] = 7 + self.assertEqual(scaled(100), 34650) + + def test_counter_in_loop(self): + global COUNTER + COUNTER = 0 + self.assertEqual(bump(10000), 10000) + self.assertEqual(bump(5), 10005) + self.assertEqual(COUNTER, 10005) + + def test_new_global_shadows_builtin(self): + def uses_len(xs): + return len(xs) + + for _ in range(200): + self.assertEqual(uses_len([1, 2, 3]), 3) + globals()["len"] = lambda xs: -1 + try: + self.assertEqual(uses_len([1, 2, 3]), -1) + finally: + del globals()["len"] + self.assertEqual(uses_len([1, 2, 3]), 3) + + def test_module_level_rebinding(self): + ns = {} + exec( + "def f():\n" + " return X\n" + "X = 0\n" + "seen = []\n" + "for X in range(500):\n" + " seen.append(f())\n", + ns, + ) + self.assertEqual(ns["seen"], list(range(500))) + + +GlobalRebindingTests._double = double + +if __name__ == "__main__": + unittest.main() From eaa08a04e1c96de8d8981e8d816cf9dbbcfd18eb Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 03:54:13 -0700 Subject: [PATCH 61/65] docs: report the branch's standing against CPython 3.14 Record the paired wall-time checkpoint for the benchmark fixtures, the startup and import costs, the per-operation gaps that remain, and the method behind each number. --- docs/PERFORMANCE-CPYTHON-PARITY.md | 141 +++++++++++++++++++++++++++++ docs/PERFORMANCE.md | 2 + 2 files changed, 143 insertions(+) create mode 100644 docs/PERFORMANCE-CPYTHON-PARITY.md diff --git a/docs/PERFORMANCE-CPYTHON-PARITY.md b/docs/PERFORMANCE-CPYTHON-PARITY.md new file mode 100644 index 00000000..3c922af1 --- /dev/null +++ b/docs/PERFORMANCE-CPYTHON-PARITY.md @@ -0,0 +1,141 @@ +# CPython parity performance + +This branch works toward making WeavePy faster than CPython 3.14 on every +benchmark fixture and on the everyday costs outside them: startup, imports, +and memory. This report records where it stands, how the numbers were +measured, and what still trails CPython. + +For the earlier passes, see [Performance measurements](PERFORMANCE.md) and +the reports it links to. + +## Method + +Timings compare a profile-guided release build of the branch +(`tools/pgo_build.py --skip-regrtest`, JIT on) with CPython 3.14.7 on the +macOS x86-64 development host (Intel Core i9-9980HK). Both executables are +native x86-64. + +Wall-time ratios come from paired, interleaved runs. Each cycle runs every +fixture once under each interpreter at the harness work sizes, alternating +which goes first; one warmup cycle is discarded and five are measured. A +fixture's ratio is the median of its per-cycle WeavePy/CPython ratios, using +each fixture's own timer (`WEAVEPY_BENCH_NS`), which excludes startup and +imports. Values below 1.00 mean WeavePy is faster. + +Instruction counts come from `/usr/bin/time -l` ("instructions retired") +on the host, or from callgrind in a Linux container when a per-function or +per-line breakdown is needed. A per-operation cost is the difference between +two work sizes divided by the difference in work, which cancels startup. +Instruction counts are deterministic enough to compare builds on a busy +machine; they don't capture cache or branch behavior, so a wall-time +checkpoint confirms them. + +## Benchmark fixtures + +Geometric mean over the 23 timed fixtures: **0.592**. + +| Fixture | Work | WeavePy | CPython | Ratio | +| --- | ---: | ---: | ---: | ---: | +| `deltablue` | 50 | 213.8 ms | 67.8 ms | 3.15 | +| `deque_ops` | 200,000 | 97.9 ms | 64.1 ms | 1.53 | +| `pickle_bench` | 40 | 11.6 ms | 8.5 ms | 1.34 | +| `generators` | 300,000 | 59.3 ms | 49.6 ms | 1.19 | +| `datetime_ops` | 60,000 | 50.7 ms | 52.7 ms | 0.98 | +| `str_methods` | 15,000 | 57.0 ms | 59.9 ms | 0.95 | +| `float_math` | 100,000 | 78.0 ms | 81.6 ms | 0.94 | +| `dict_ops` | 100,000 | 55.9 ms | 61.7 ms | 0.90 | +| `call_overhead` | 150,000 | 69.8 ms | 82.2 ms | 0.84 | +| `fannkuch` | 100,000 | 17.1 ms | 20.9 ms | 0.83 | +| `richards` | 50,000 | 16.1 ms | 20.3 ms | 0.80 | +| `nbody` | 20,000 | 33.0 ms | 42.9 ms | 0.77 | +| `attr_access` | 200,000 | 35.4 ms | 47.1 ms | 0.77 | +| `list_ops` | 10,000 | 33.8 ms | 45.1 ms | 0.75 | +| `json_bench` | 150 | 57.4 ms | 81.1 ms | 0.71 | +| `pidigits` | 500,000 | 1,421.0 ms | 2,167.8 ms | 0.67 | +| `pyaes` | 400 | 15.8 ms | 28.9 ms | 0.54 | +| `jitkernels` | 2,000 | 24.3 ms | 49.0 ms | 0.48 | +| `fib` | 27 | 8.0 ms | 21.6 ms | 0.37 | +| `spectral_norm` | 100 | 12.1 ms | 57.7 ms | 0.21 | +| `nested_loops` | 120 | 6.2 ms | 69.3 ms | 0.09 | +| `jitloop` | 1,000 | 7.5 ms | 92.0 ms | 0.08 | +| `sumvm` | 2,000,000 | 3.8 ms | 72.5 ms | 0.05 | + +Times are the medians of each interpreter's samples; ratios are the medians +of the paired ratios, so they needn't equal the quotient of the two times. + +## Startup and imports + +Whole-process instructions retired, warm stdlib cache: + +| Command | WeavePy | CPython | +| --- | ---: | ---: | +| `-c pass` | 91M | 127M | +| `import json` | 205M | 153M | +| `import dataclasses` | 349M | 196M | +| `import logging` | 513M | 258M | +| `import asyncio` | 689M | 365M | +| `import unittest` | 496M | 270M | + +Startup is cheaper than CPython's. Importing large parts of the standard +library still costs about twice as much. Three changes on this branch cut +those costs by 20% to 60%: + +- Compiled stdlib modules are cached in a WeavePy-native code format + (`weavepy_compiler::native_code`), which loads without unmarshalling CPython + bytecode and transcoding it back into WeavePy instructions. +- Slicing a tuple, `bytes`, `bytearray`, or a string with surrogates no longer + copies the whole source sequence. `re.compile` with `IGNORECASE` slices a + 64K-entry charset map 256 times, which had made `import logging` cost 1.19B + instructions. +- The collector's per-drop sweep of suspected-dead objects skips entries that + have used up their probe budget, so it no longer walks up to 256 entries at + every drop during imports. + +Peak memory after these imports is about twice CPython's (for example, +23 MiB against 12 MiB for `import json`). An allocation-site profile +attributes most of the difference to compiled code objects: their per-code +structure, per-instruction line and column tables, and the materialized +constants and names built on first use. + +## What still trails CPython + +The four slower fixtures share a cause: operations that CPython's specialized +bytecodes do in 10 to 30 instructions take WeavePy two to four times as many. +Per operation, with the JIT off: + +| Operation | WeavePy | CPython | +| --- | ---: | ---: | +| Instance attribute read (`o.x`) | 160 | 64 | +| Class attribute read through an instance (`o.K`) | 240 | 60 | +| Global read | 150 | 50 | +| `isinstance(o, C)` | 980 | 240 | +| Call of a small non-leaf function | 1,700 | 440 | + +The costs come from the interpreter's structure rather than from any single +slow path: + +- **Loop state.** The core loop keeps about 17 values live across its arms, + which contain calls; x86-64 has six callee-saved registers, so most of the + state lives on the stack and every dispatch reloads it. +- **Data layout.** An instance field read passes the site's slot table, the + class version, the class's shared keys, and the instance's split values, + each behind its own pointer; CPython checks a type version and reads an + inline slot. +- **Calls.** An inline activation binds cells, pools its locals vector, + grades its locals for collector bookkeeping on exit, and switches the + running frame, several hundred instructions per call and return. + +`deltablue` also runs at a lower IPC than CPython (2.13 against 2.47): its +interpreter loops overflow the instruction cache, and their single dispatch +branches predict poorly. + +## Validation + +Every change on the branch passes the VM unit tests (420), the bundled +regression suite (248), the semantics comparison scripts, and the CPython +regression suites most exposed to it. The changes described above were +checked with, among others, `test_bytes`, `test_tuple`, `test_slice`, +`test_str`, `test_re`, `test_marshal`, `test_import`, `test_code`, +`test_compile`, `test_logging`, `test_gc`, `test_weakref`, `test_io`, +`test_ssl`, `test_asyncio`, `test_scope`, `test_global`, `test_module`, +`test_builtin`, `test_sys_settrace`, and `test_monitoring`. diff --git a/docs/PERFORMANCE.md b/docs/PERFORMANCE.md index 2d6c108c..f7cbf22c 100644 --- a/docs/PERFORMANCE.md +++ b/docs/PERFORMANCE.md @@ -8,6 +8,8 @@ For the latest object metadata, numeric text, and enumeration changes, see [Runtime metadata and numeric text performance](PERFORMANCE-RUNTIME-METADATA.md). For the subsequent dispatch, allocation, and garbage-collection changes, see [Dispatch, allocation, and memory performance](PERFORMANCE-DISPATCH-MEMORY.md). +For the branch comparing WeavePy with CPython 3.14 across the benchmark +fixtures, startup, and imports, see [CPython parity performance](PERFORMANCE-CPYTHON-PARITY.md). The September 8, 2026, optimization pass reduces warm-cache process startup from 50.7 ms to 32.3 ms on the measured macOS ARM64 host. Across all 24 existing From 31503af16c0c865f55b1746c8bc672d60d90719d Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 04:37:42 -0700 Subject: [PATCH 62/65] perf: keep nested generators on fast steps past the eval breaker A fast step stops at a back edge when the GIL countdown runs out. For an inner generator that left its frame partway, and `gen_fast_next` then declined it for good, so an outer generator fell back to the general loop for the rest of its run. Fast steps now continue a partial frame where it stopped, and a draining consumer's send services the eval breaker itself (the countdown reset, the dormant suspect probe, the GIL checkpoint) and keeps stepping unless the loop generation moved. Summing a generator over a generator drops 6% to 8%. --- crates/weavepy-vm/src/gen_fast.rs | 17 +++++++++-- crates/weavepy-vm/src/lib.rs | 24 ++++++++++++++- tests/regrtest/test_generator_fast_steps.py | 33 +++++++++++++++++++++ 3 files changed, 70 insertions(+), 4 deletions(-) diff --git a/crates/weavepy-vm/src/gen_fast.rs b/crates/weavepy-vm/src/gen_fast.rs index 4540e6e3..d2902740 100644 --- a/crates/weavepy-vm/src/gen_fast.rs +++ b/crates/weavepy-vm/src/gen_fast.rs @@ -712,9 +712,12 @@ impl Interpreter { return GenNext::Declined; }; let first_resume = matches!(*state, GeneratorState::Created(_)); - if boxed.sent_consumed || !Self::gen_fast_frame_ok(boxed) { + if !Self::gen_fast_frame_ok(boxed) { return GenNext::Declined; } + // A body an earlier fast step left partway (at the eval breaker, + // say) continues where it stopped, with no value sent. + let partial = boxed.sent_consumed; let prev = std::mem::replace(state, GeneratorState::Running); let (GeneratorState::Suspended(mut boxed) | GeneratorState::Created(mut boxed)) = prev else { @@ -723,13 +726,21 @@ impl Interpreter { let frame: &mut Frame = &mut boxed; frame.gen_first_resume = first_resume; let start = frame.pc; - frame.stack.push(Object::None); + if partial { + frame.sent_consumed = false; + } else { + frame.stack.push(Object::None); + } debug_assert!(!frame.stack.is_empty()); let out = match self.gen_fast_step(frame, snap_gen, depth, None) { GenStep::Yielded(v) => GenNext::Yielded(v), GenStep::Bail if frame.pc == start => { // Nothing ran: the resume is undone. - frame.stack.pop(); + if partial { + frame.sent_consumed = true; + } else { + frame.stack.pop(); + } GenNext::Declined } GenStep::Bail => { diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 4d9ff652..08636202 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -37405,7 +37405,29 @@ impl Interpreter { } // A simple body runs to its next yield in place (see // `gen_fast`), folding the draining consumer's yields as it goes. - if let Some(v) = self.gen_fast_run(frame, snap_gen, fold) { + let mut fast = self.gen_fast_run(frame, snap_gen, fold); + // Fast steps stop at the eval breaker; a draining consumer's + // single send would then finish the whole generator on the + // slow path. Service the breaker here (as the quiet loop's + // checkpoint does) and keep stepping. + while fast.is_none() + && self.gil_countdown <= 1 + && crate::hot_gates::loop_gen() == snap_gen + && !frame.sent_consumed + { + self.gil_countdown = crate::gil::GIL_CHECK_INTERVAL; + if !gc_trace::active_suspects_present() && gc_trace::has_suspects() { + for obj in gc_trace::take_dead_suspects() { + self.reap_dead_subgraph(obj); + } + } + crate::gil::yield_checkpoint(); + if crate::hot_gates::loop_gen() != snap_gen { + break; + } + fast = self.gen_fast_run(frame, snap_gen, fold); + } + if let Some(v) = fast { Ok(FrameOutcome::Yielded(v)) } else { // The draining consumer's sink, folded at this frame's yields. diff --git a/tests/regrtest/test_generator_fast_steps.py b/tests/regrtest/test_generator_fast_steps.py index f96ede81..fd8b3bb9 100644 --- a/tests/regrtest/test_generator_fast_steps.py +++ b/tests/regrtest/test_generator_fast_steps.py @@ -7,6 +7,7 @@ """ import sys +import threading import unittest @@ -186,6 +187,38 @@ def tracer(frame, event, arg): sys.settrace(None) self.assertTrue(seen) + def test_long_nested_pipelines(self): + # Long enough to cross many eval-breaker checkpoints, which stop + # a fast step partway through the inner generators. + n = 20000 + self.assertEqual(sum(evens(squares(counter(n)))), + sum(x * x for x in range(n) if x * x % 2 == 0)) + self.assertEqual(sum(x + 1 for x in counter(n)), n * (n + 1) // 2) + self.assertEqual(list(squares(over_range(n)))[-1], ((n - 1) * 3) ** 2) + total = 0 + for x in squares(squares(counter(3000))): + total += x + self.assertEqual(total, sum(x ** 4 for x in range(3000))) + + def test_other_threads_run_during_a_folded_pipeline(self): + progress = [] + stop = threading.Event() + + def worker(): + while not stop.is_set(): + progress.append(1) + + t = threading.Thread(target=worker) + t.start() + try: + before = len(progress) + sum(evens(squares(counter(300000)))) + during = len(progress) - before + finally: + stop.set() + t.join() + self.assertGreater(during, 0) + def test_recursion_depth(self): def chain(depth): if depth == 0: From 2aea55a3e8d779b072343196fa5fcd7e95c43f32 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 05:35:24 -0700 Subject: [PATCH 63/65] perf: run list.append and list.pop in line from the fused method call The core loop's fused `xs.append(x)` and `xs.pop()` on a local list now push and pop directly instead of admitting the call and then entering the generic builtin, about 8% off each call. `list.pop` also accepts a bool index, as CPython's does (`[1, 2, 3].pop(True)` returns 2; it raised TypeError). --- crates/weavepy-vm/src/builtins.rs | 6 ++- crates/weavepy-vm/src/lib.rs | 40 ++++++++++++++-- tests/regrtest/test_list_append_pop.py | 66 ++++++++++++++++++++++++++ 3 files changed, 108 insertions(+), 4 deletions(-) create mode 100644 tests/regrtest/test_list_append_pop.py diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index 12e6672e..a71e7de9 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -12518,7 +12518,11 @@ fn list_pop(args: &[Object]) -> Result { let l = list_self(args)?; let mut l = l.borrow_mut(); let idx = if args.len() > 1 { - match &args[1] { + let index = match &args[1] { + Object::Bool(b) => Object::Int(i64::from(*b)), + other => other.clone(), + }; + match &index { Object::Int(i) => { if l.is_empty() { return Err(index_error("pop from empty list")); diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 08636202..dcbfecb3 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -20554,10 +20554,44 @@ impl Interpreter { ) } K::Isinstance => return Self::core_isinstance(args), - K::ListAppend => args.len() == 2 && matches!(args[0], O::List(_)), + // `list.append` and `list.pop` in line (as `list_append` and + // `list_pop` do them). + K::ListAppend => { + let [O::List(l), item] = args else { + return None; + }; + l.try_borrow_mut().ok()?.push(item.clone()); + return Some(Ok(O::None)); + } K::ListPop => { - (args.len() == 1 || (args.len() == 2 && leaf_int(&args[1]))) - && matches!(args[0], O::List(_)) + let (O::List(l), index) = (&args[0], args.get(1)) else { + return None; + }; + let mut l = l.try_borrow_mut().ok()?; + let i = match index { + None => l.len().checked_sub(1), + Some(O::Int(_) | O::Bool(_)) if args.len() == 2 => { + let i = &match args[1] { + O::Bool(b) => i64::from(b), + O::Int(i) => i, + _ => unreachable!("matched above"), + }; + let len = l.len() as i64; + let n = if *i < 0 { i + len } else { *i }; + if l.is_empty() { + None + } else if n < 0 || n >= len { + return Some(Err(index_error("pop index out of range"))); + } else { + Some(n as usize) + } + } + _ => return None, + }; + return Some(match i { + Some(i) => Ok(l.remove(i)), + None => Err(index_error("pop from empty list")), + }); } K::ListInsert => args.len() == 3 && matches!(args[0], O::List(_)) && leaf_int(&args[1]), K::ListReverse | K::ListCopy => args.len() == 1 && matches!(args[0], O::List(_)), diff --git a/tests/regrtest/test_list_append_pop.py b/tests/regrtest/test_list_append_pop.py new file mode 100644 index 00000000..f5069d2c --- /dev/null +++ b/tests/regrtest/test_list_append_pop.py @@ -0,0 +1,66 @@ +"""`list.append` and `list.pop` on a local list, run in line. + +The core loop's fused method call appends and pops directly. These +cases check the results and every error against the full methods: +negative and boolean indexes, empty lists, out-of-range indexes, and +subclasses, which keep the full path. +""" + +import unittest + + +class ListAppendPopTests(unittest.TestCase): + def test_append_and_pop_round_trip(self): + xs = [] + for i in range(100): + xs.append(i) + xs.append((i, "x")) + popped = [xs.pop() for _ in range(4)] + self.assertEqual(popped, [(99, "x"), 99, (98, "x"), 98]) + self.assertEqual(len(xs), 196) + + def test_pop_indexes(self): + for _ in range(20): + xs = [1, 2, 3, 4] + self.assertEqual(xs.pop(0), 1) + self.assertEqual(xs.pop(-1), 4) + self.assertEqual(xs.pop(True), 3) + self.assertEqual(xs.pop(False), 2) + self.assertEqual(xs, []) + + def test_pop_errors(self): + for _ in range(20): + with self.assertRaisesRegex(IndexError, "pop from empty list"): + [].pop() + with self.assertRaisesRegex(IndexError, "pop from empty list"): + [].pop(0) + with self.assertRaisesRegex(IndexError, "pop index out of range"): + [1].pop(5) + with self.assertRaisesRegex(IndexError, "pop index out of range"): + [1].pop(-2) + with self.assertRaises(TypeError): + [1].pop("0") + + def test_subclass_overrides(self): + class Logged(list): + def append(self, item): + super().append(("logged", item)) + + def pop(self, *args): + return ("popped", super().pop(*args)) + + xs = Logged() + for i in range(3): + xs.append(i) + self.assertEqual(xs, [("logged", 0), ("logged", 1), ("logged", 2)]) + self.assertEqual(xs.pop(), ("popped", ("logged", 2))) + + def test_append_keeps_objects_alive(self): + xs = [] + for _ in range(10): + xs.append(object()) + self.assertEqual(len({id(x) for x in xs}), 10) + + +if __name__ == "__main__": + unittest.main() From 9e7ed7feb4f1851a14bcd45d0867566746bd5006 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 09:47:01 -0700 Subject: [PATCH 64/65] ci: satisfy rustfmt and clippy from Rust 1.98 CI's stable toolchain moved to 1.98.1, whose rustfmt orders the `initconfig` imports differently and whose clippy asks for `as_chunks` in the frozen-cache hash and `usize::midpoint` in the binary insertion sort. Both are stable well before the 1.93 MSRV. --- crates/weavepy-capi/src/pep741.rs | 4 ++-- crates/weavepy-vm/src/frozen_code_cache.rs | 8 ++++---- crates/weavepy-vm/src/timsort.rs | 2 +- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/crates/weavepy-capi/src/pep741.rs b/crates/weavepy-capi/src/pep741.rs index 06d90417..2d5b096e 100644 --- a/crates/weavepy-capi/src/pep741.rs +++ b/crates/weavepy-capi/src/pep741.rs @@ -29,8 +29,8 @@ use weavepy_vm::object::Object; use crate::embed::PyInitFn; use crate::initconfig::{ - self, EmbedConfig, PyConfig, PyConfig_InitIsolatedConfig, PyStatus, PyWideStringList, - _PyStatus_TYPE_ERROR, _PyStatus_TYPE_EXIT, + self, _PyStatus_TYPE_ERROR, _PyStatus_TYPE_EXIT, EmbedConfig, PyConfig, + PyConfig_InitIsolatedConfig, PyStatus, PyWideStringList, }; use crate::object::PyObject; diff --git a/crates/weavepy-vm/src/frozen_code_cache.rs b/crates/weavepy-vm/src/frozen_code_cache.rs index dbb2a7b9..6d083993 100644 --- a/crates/weavepy-vm/src/frozen_code_cache.rs +++ b/crates/weavepy-vm/src/frozen_code_cache.rs @@ -112,16 +112,16 @@ const FROZEN_HEADER_LEN: usize = 20; /// matter combined with an equal length). fn fnv1a(s: &str) -> u64 { let mut h: u64 = 0xcbf2_9ce4_8422_2325; - let mut words = s.as_bytes().chunks_exact(8); + let (words, rest) = s.as_bytes().as_chunks::<8>(); let mut mix = |w: u64| { h ^= w; h = h.wrapping_mul(0x0000_0100_0000_01b3); }; - for w in &mut words { - mix(u64::from_le_bytes(w.try_into().expect("an 8-byte chunk"))); + for w in words { + mix(u64::from_le_bytes(*w)); } let mut tail = [0u8; 8]; - tail[..words.remainder().len()].copy_from_slice(words.remainder()); + tail[..rest.len()].copy_from_slice(rest); mix(u64::from_le_bytes(tail)); h } diff --git a/crates/weavepy-vm/src/timsort.rs b/crates/weavepy-vm/src/timsort.rs index 3189029a..3759ffca 100644 --- a/crates/weavepy-vm/src/timsort.rs +++ b/crates/weavepy-vm/src/timsort.rs @@ -170,7 +170,7 @@ where let (mut l, mut r) = (0, ok); let pivot = unsafe { a.add(ok) }; while l < r { - let m = (l + r) >> 1; + let m = usize::midpoint(l, r); if self.lt(pivot, unsafe { a.add(m) })? { r = m; } else { From 3e99277cb994228e9cecf6d646eeda281e3e6eb0 Mon Sep 17 00:00:00 2001 From: Owen Carey <37121709+owenthcarey@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:29:09 -0700 Subject: [PATCH 65/65] fix: release a SimpleQueue's wake gate once when puts race without the GIL Under `-X gil=0`, two threads putting into the same `queue.SimpleQueue` could both see a parked getter's `_locked` flag, both clear it, and both release the gate; the second release raised `RuntimeError('release unlocked lock')`. `multiprocessing.dummy`'s thread pool returns results that way, so the free-threaded lane's `test_multiprocessing_dummy` failed intermittently (about 2% of runs on main, 5% here). The flag's test, clear, and release now happen under one lock, as CPython's critical section makes them. The append stays outside it, since it may collect and run Python code. The test passes 400 of 400 stressed runs. --- crates/weavepy-vm/src/stdlib/queue_native.rs | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/crates/weavepy-vm/src/stdlib/queue_native.rs b/crates/weavepy-vm/src/stdlib/queue_native.rs index 7bd57e23..e5b966d6 100644 --- a/crates/weavepy-vm/src/stdlib/queue_native.rs +++ b/crates/weavepy-vm/src/stdlib/queue_native.rs @@ -37,10 +37,18 @@ fn simplequeue_put(args: &[Object]) -> Result { let dq = interp.load_attr_public(&recv, "_queue")?; let append = interp.load_attr_public(&dq, "append")?; interp.call(&append, &[item], &[], &g)?; + // The flag's test, clear and release are one step, as CPython's + // critical section makes them without the GIL: two putters that both + // saw it set would release the gate twice (`release unlocked lock`). + // (Nothing inside runs Python code; the append above, which may + // collect, stays outside.) + static GATE: std::sync::Mutex<()> = std::sync::Mutex::new(()); + let _held = GATE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); if interp.load_attr_public(&recv, "_locked")?.is_truthy() { - // Clear the flag before releasing: a second putter that runs in - // between must not release the gate twice (`release unlocked - // lock`), and the woken getter sets it again itself. + // Clear the flag before releasing: the woken getter sets it again + // itself. interp.store_attr_public(&recv, "_locked", Object::Bool(false))?; let lock = interp.load_attr_public(&recv, "_lock")?; let release = interp.load_attr_public(&lock, "release")?;