diff --git a/Cargo.lock b/Cargo.lock index f5f54d62..1036cdee 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1262,6 +1262,15 @@ version = "0.2.16" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981" +[[package]] +name = "libmimalloc-sys" +version = "0.1.49" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6a45a52f43e1c16f667ccfe4dd8c85b7f7c204fd5e3bf46c5b0db9a5c3c0b8e9" +dependencies = [ + "cc", +] + [[package]] name = "libredox" version = "0.1.16" @@ -2819,6 +2828,7 @@ dependencies = [ "clap", "dirs", "libc", + "libmimalloc-sys", "rustyline", "serde_json", "tracing", diff --git a/Cargo.toml b/Cargo.toml index 6cb9ac9a..bce0707c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -141,6 +141,7 @@ unicode_names2 = "2.0" memmap2 = "0.9" memchr = "2.8" libc = "0.2" +libmimalloc-sys = { version = "0.1.49", default-features = false } rustls = { version = "0.23", default-features = false, features = ["ring", "std", "tls12"] } rustls-pki-types = "1.7" # PKCS#8 "ENCRYPTED PRIVATE KEY" (PBES2) decryption for password-protected @@ -317,7 +318,9 @@ opt-level = 1 [profile.release] opt-level = 3 -lto = "thin" +# Fat LTO inlines across the VM, compiler, and JIT crates: 1-10% faster +# on the benchmark fixtures for about a third more build time. +lto = "fat" codegen-units = 1 debug = "line-tables-only" strip = "debuginfo" diff --git a/crates/weavepy-capi/src/memoryview.rs b/crates/weavepy-capi/src/memoryview.rs index 019fe566..547e6f48 100644 --- a/crates/weavepy-capi/src/memoryview.rs +++ b/crates/weavepy-capi/src/memoryview.rs @@ -329,7 +329,9 @@ pub unsafe extern "C" fn PyMemoryView_FromObjectAndFlags( Some(obj.clone()) }; PyMemoryView { - buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + buffer: MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), start: Cell::new(start), len: Cell::new(view_len), readonly: Cell::new(readonly), @@ -403,7 +405,9 @@ pub unsafe extern "C" fn PyMemoryView_FromMemory( readonly, }; let mv = PyMemoryView::contiguous_1d( - MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), len, readonly, "B".to_owned(), @@ -486,7 +490,9 @@ pub unsafe extern "C" fn PyMemoryView_FromBuffer(view: *const Py_buffer) -> *mut readonly, }; let mv = PyMemoryView { - buffer: MemoryViewBuffer::Shared(weavepy_vm::sync::Rc::new(region)), + buffer: MemoryViewBuffer::Shared( + weavepy_vm::rc_unsize!(weavepy_vm::sync::Rc::new(region) => dyn SharedMemBuffer), + ), start: Cell::new(0), len: Cell::new(len), readonly: Cell::new(readonly), diff --git a/crates/weavepy-cli/Cargo.toml b/crates/weavepy-cli/Cargo.toml index fa3631a3..25c2a067 100644 --- a/crates/weavepy-cli/Cargo.toml +++ b/crates/weavepy-cli/Cargo.toml @@ -39,6 +39,7 @@ weavepy-compiler = { workspace = true } weavepy-conformance = { workspace = true } weavepy-version = { workspace = true } libc = { workspace = true } +libmimalloc-sys = { workspace = true } anyhow = { workspace = true } clap = { workspace = true } serde_json = { workspace = true } @@ -59,6 +60,9 @@ windows-sys = { workspace = true } default = ["jit"] # RFC 0032 — build the `weavepy` binary with the tier-2 JIT compiled in. jit = ["weavepy/jit", "weavepy-vm/jit"] +# Sample live allocations by call stack (`WEAVEPY_ALLOC_PROFILE=`, +# macOS only); see `src/alloc_profile.rs`. +alloc-profile = [] [lints] workspace = true diff --git a/crates/weavepy-cli/src/alloc.rs b/crates/weavepy-cli/src/alloc.rs new file mode 100644 index 00000000..7c78cd0b --- /dev/null +++ b/crates/weavepy-cli/src/alloc.rs @@ -0,0 +1,81 @@ +//! The process allocator: mimalloc, entered through its plain allocation +//! functions whenever they already satisfy the requested alignment. +//! +//! Every mimalloc block is at least word aligned, and every block of 16 +//! bytes or more is 16-byte aligned (`MI_MAX_ALIGN_SIZE`), so a layout +//! aligned to that much needs no aligned entry point. The aligned entry points +//! take their fast path only when the size class's free list happens to +//! offer a suitably aligned block, and otherwise fall to a generic path +//! that may over-allocate. Rust asks for alignment on every allocation, +//! so going through them would put every allocation on that path. + +use std::alloc::{GlobalAlloc, Layout}; +use std::ffi::c_void; + +use libmimalloc_sys::{ + mi_free, mi_malloc, mi_malloc_aligned, mi_realloc, mi_realloc_aligned, mi_zalloc, + mi_zalloc_aligned, +}; + +/// The largest alignment every mimalloc block already has. +const WORD_ALIGN: usize = std::mem::size_of::(); + +/// mimalloc's `MI_MAX_ALIGN_SIZE`: the alignment of every block at least +/// this large. +const MAX_ALIGN: usize = 16; + +/// Whether the plain entry points already satisfy `align` for `size`. +#[inline(always)] +fn plain(size: usize, align: usize) -> bool { + align <= WORD_ALIGN || (align <= MAX_ALIGN && size >= align) +} + +/// mimalloc as the global allocator (see the module docs). +pub(crate) struct Mimalloc; + +// SAFETY: every block comes from mimalloc with at least the layout's +// alignment (word-aligned blocks, or the aligned entry points for larger +// alignments), and every block is released or resized through mimalloc. +unsafe impl GlobalAlloc for Mimalloc { + #[inline] + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + // SAFETY: plain FFI allocation calls. + unsafe { + if plain(layout.size(), layout.align()) { + mi_malloc(layout.size()).cast() + } else { + mi_malloc_aligned(layout.size(), layout.align()).cast() + } + } + } + + #[inline] + unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { + // SAFETY: as in `alloc`. + unsafe { + if plain(layout.size(), layout.align()) { + mi_zalloc(layout.size()).cast() + } else { + mi_zalloc_aligned(layout.size(), layout.align()).cast() + } + } + } + + #[inline] + unsafe fn dealloc(&self, ptr: *mut u8, _layout: Layout) { + // SAFETY: `ptr` came from this allocator (the caller's contract). + unsafe { mi_free(ptr.cast::()) } + } + + #[inline] + unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { + // SAFETY: `ptr` came from this allocator with `layout`. + unsafe { + if plain(new_size, layout.align()) { + mi_realloc(ptr.cast::(), new_size).cast() + } else { + mi_realloc_aligned(ptr.cast::(), new_size, layout.align()).cast() + } + } + } +} diff --git a/crates/weavepy-cli/src/alloc_profile.rs b/crates/weavepy-cli/src/alloc_profile.rs new file mode 100644 index 00000000..13bfb8a0 --- /dev/null +++ b/crates/weavepy-cli/src/alloc_profile.rs @@ -0,0 +1,211 @@ +//! Opt-in allocation-site profiler (`--features alloc-profile`, macOS). +//! +//! The global allocator samples one allocation per [`SAMPLE`] bytes and +//! records its call stack. A sampled block is charged [`SAMPLE`] bytes to +//! its stack while it lives, so at exit the table estimates the *live* heap +//! by allocation site. With `WEAVEPY_ALLOC_PROFILE=` set, the live +//! samples are written to ``: one line per sample, the charged bytes +//! then the slide-adjusted return addresses (resolve them with `atos`). + +use std::alloc::{GlobalAlloc, Layout}; +use std::cell::Cell; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +/// Bytes between samples. +const SAMPLE: usize = 16 * 1024; +/// Return addresses kept per sample. +const DEPTH: usize = 24; +/// Sampled blocks tracked at once (open addressing, power of two). +const SLOTS: usize = 1 << 18; + +struct Sample { + ptr: usize, + stack: [usize; DEPTH], +} + +static ENABLED: AtomicBool = AtomicBool::new(false); +static LOCK: AtomicBool = AtomicBool::new(false); +static TABLE: AtomicUsize = AtomicUsize::new(0); + +thread_local! { + static IN_HOOK: Cell = const { Cell::new(false) }; + static UNTIL_SAMPLE: Cell = const { Cell::new(SAMPLE) }; +} + +extern "C" { + fn _dyld_get_image_vmaddr_slide(image_index: u32) -> isize; +} + +fn lock() { + while LOCK + .compare_exchange_weak(false, true, Ordering::Acquire, Ordering::Relaxed) + .is_err() + { + std::hint::spin_loop(); + } +} + +fn unlock() { + LOCK.store(false, Ordering::Release); +} + +fn table() -> *mut Sample { + TABLE.load(Ordering::Acquire) as *mut Sample +} + +fn slot_of(ptr: usize) -> usize { + (ptr.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> 20) & (SLOTS - 1) +} + +/// Start sampling (allocates the table from the system allocator). +pub(crate) fn start() { + if std::env::var_os("WEAVEPY_ALLOC_PROFILE").is_none() { + return; + } + let bytes = SLOTS * std::mem::size_of::(); + // SAFETY: a fresh zeroed mapping owned for the rest of the process. + let p = unsafe { + libc::mmap( + std::ptr::null_mut(), + bytes, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANON, + -1, + 0, + ) + }; + if p == libc::MAP_FAILED { + return; + } + TABLE.store(p as usize, Ordering::Release); + ENABLED.store(true, Ordering::Release); +} + +fn record(ptr: usize) { + let mut stack = [0usize; DEPTH]; + // SAFETY: `backtrace` fills at most `DEPTH` entries of the buffer. + let n = unsafe { libc::backtrace(stack.as_mut_ptr().cast(), DEPTH as libc::c_int) }; + let _ = n; + let t = table(); + lock(); + let mut i = slot_of(ptr); + for _ in 0..SLOTS { + // SAFETY: `i < SLOTS`, inside the mapping. + let s = unsafe { &mut *t.add(i) }; + if s.ptr == 0 || s.ptr == ptr { + s.ptr = ptr; + s.stack = stack; + break; + } + i = (i + 1) & (SLOTS - 1); + } + unlock(); +} + +fn forget(ptr: usize) { + let t = table(); + lock(); + let mut i = slot_of(ptr); + for _ in 0..SLOTS { + // SAFETY: as in `record`. + let s = unsafe { &mut *t.add(i) }; + if s.ptr == 0 { + break; + } + if s.ptr == ptr { + // A tombstone keeps later probes of this chain reachable. + s.ptr = usize::MAX; + break; + } + i = (i + 1) & (SLOTS - 1); + } + unlock(); +} + +/// The profiling allocator: mimalloc plus the sampler. +pub(crate) struct Profiled; + +// SAFETY: every allocation is mimalloc's; the sampler only records +// addresses and never touches the blocks. +unsafe impl GlobalAlloc for Profiled { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + // SAFETY: forwarded unchanged. + let p = unsafe { crate::alloc::Mimalloc.alloc(layout) }; + if ENABLED.load(Ordering::Relaxed) && !p.is_null() { + sample(p as usize, layout.size()); + } + p + } + + unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { + if ENABLED.load(Ordering::Relaxed) { + forget(ptr as usize); + } + // SAFETY: forwarded unchanged. + unsafe { crate::alloc::Mimalloc.dealloc(ptr, layout) } + } + + unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { + if ENABLED.load(Ordering::Relaxed) { + forget(ptr as usize); + } + // SAFETY: forwarded unchanged. + let p = unsafe { crate::alloc::Mimalloc.realloc(ptr, layout, new_size) }; + if ENABLED.load(Ordering::Relaxed) && !p.is_null() { + sample(p as usize, new_size); + } + p + } +} + +fn sample(ptr: usize, size: usize) { + let _ = IN_HOOK.try_with(|hook| { + if hook.get() { + return; + } + let due = UNTIL_SAMPLE.with(|left| { + let l = left.get(); + if size >= l { + left.set(SAMPLE); + true + } else { + left.set(l - size); + false + } + }); + if due { + hook.set(true); + record(ptr); + hook.set(false); + } + }); +} + +/// Write the live samples (see the module docs). +pub(crate) fn finish() { + let Some(path) = std::env::var_os("WEAVEPY_ALLOC_PROFILE") else { + return; + }; + if !ENABLED.swap(false, Ordering::AcqRel) { + return; + } + let t = table(); + // SAFETY: the main executable is image 0. + let slide = unsafe { _dyld_get_image_vmaddr_slide(0) } as usize; + let mut out = String::new(); + lock(); + for i in 0..SLOTS { + // SAFETY: `i < SLOTS`. + let s = unsafe { &*t.add(i) }; + if s.ptr == 0 || s.ptr == usize::MAX { + continue; + } + out.push_str(&SAMPLE.to_string()); + for &a in s.stack.iter().take_while(|a| **a != 0) { + out.push_str(&format!(" {:x}", a.wrapping_sub(slide))); + } + out.push('\n'); + } + unlock(); + let _ = std::fs::write(path, out); +} diff --git a/crates/weavepy-cli/src/lib.rs b/crates/weavepy-cli/src/lib.rs index ddb2edd3..6714a6a5 100644 --- a/crates/weavepy-cli/src/lib.rs +++ b/crates/weavepy-cli/src/lib.rs @@ -36,11 +36,22 @@ use tracing_subscriber::EnvFilter; use weavepy::{InterpreterFlags, RunOptions}; -/// The process allocator: the system allocator behind per-thread -/// small-block caches (see `weavepy_vm::tcache`), enabled on the VM's -/// own threads. +mod alloc; + +/// The process allocator. The interpreter allocates and frees small blocks +/// constantly; mimalloc serves them from per-thread free lists, several +/// times faster than the system allocator on macOS. +#[cfg(not(all(feature = "alloc-profile", target_os = "macos")))] +#[global_allocator] +static GLOBAL_ALLOC: alloc::Mimalloc = alloc::Mimalloc; + +#[cfg(all(feature = "alloc-profile", target_os = "macos"))] +mod alloc_profile; + +/// The allocation-site profiler's allocator (see [`alloc_profile`]). +#[cfg(all(feature = "alloc-profile", target_os = "macos"))] #[global_allocator] -static GLOBAL_ALLOC: weavepy::vm::tcache::ThreadCacheAlloc = weavepy::vm::tcache::ThreadCacheAlloc; +static GLOBAL_ALLOC: alloc_profile::Profiled = alloc_profile::Profiled; const VERSION: &str = env!("CARGO_PKG_VERSION"); @@ -721,7 +732,6 @@ fn run_on_large_stack(entry: fn() -> i32) -> i32 { weavepy::vm::stdlib::signal_mod::block_async_signals_current_thread(); let vm_entry = move || -> i32 { - weavepy::vm::tcache::enable_for_current_thread(); // Opt-in (`WEAVEPY_CRASH_BT`): register the native crash handler + // per-thread sigaltstack on the VM thread itself so a stack-overflow // SIGSEGV can be caught and reported (no-op stub on Windows). @@ -745,8 +755,12 @@ fn run_on_large_stack(entry: fn() -> i32) -> i32 { // fork-warning check measures additional threads against. weavepy::vm::stdlib::os_process::capture_thread_baseline(); pcprof::start(); + #[cfg(all(feature = "alloc-profile", target_os = "macos"))] + alloc_profile::start(); let code = entry(); pcprof::finish(); + #[cfg(all(feature = "alloc-profile", target_os = "macos"))] + alloc_profile::finish(); code }; @@ -1066,6 +1080,7 @@ fn real_main() -> Result { if let Some(module) = cli.module.clone() { let extra = cli.args.clone(); + weavepy::vm::spawn_jit_codegen_prewarm(); run_module(&module, extra, &flags, &extra_path)?; return Ok(0); } @@ -1078,6 +1093,9 @@ fn real_main() -> Result { Ok(0) } Some(path) => { + // A program file: warm the JIT's code generator off the main + // thread while it starts. + weavepy::vm::spawn_jit_codegen_prewarm(); run_path(path, trailing.clone(), &flags, &extra_path)?; Ok(0) } diff --git a/crates/weavepy-compiler/src/bytecode.rs b/crates/weavepy-compiler/src/bytecode.rs index 6614bdaa..2eb3fd99 100644 --- a/crates/weavepy-compiler/src/bytecode.rs +++ b/crates/weavepy-compiler/src/bytecode.rs @@ -954,6 +954,16 @@ pub struct Instruction { pub arg: u32, } +impl OpCode { + /// The opcode numbered `b` (variants count up from zero in + /// declaration order), if there is one. + pub fn from_u8(b: u8) -> Option { + // SAFETY: `OpCode` is `repr(u8)` with implicit, contiguous + // discriminants, the last of which is `StoreFastMaybeNull`. + (b <= Self::StoreFastMaybeNull as u8).then(|| unsafe { std::mem::transmute::(b) }) + } +} + impl Instruction { #[inline] pub const fn new(op: OpCode, arg: u32) -> Self { diff --git a/crates/weavepy-compiler/src/cache_snapshot.rs b/crates/weavepy-compiler/src/cache_snapshot.rs index 5b4c74fc..45966360 100644 --- a/crates/weavepy-compiler/src/cache_snapshot.rs +++ b/crates/weavepy-compiler/src/cache_snapshot.rs @@ -97,8 +97,15 @@ impl SelectStorage for Select { // Avoid portable-atomic's global-lock fallback. An inherited global lock can // belong to a vanished writer after fork; advisory caches can simply miss. -pub(crate) type CacheSnapshot = - as SelectStorage>::Storage; +// +// On x86_64 without AVX enabled at compile time, portable-atomic picks its +// 128-bit load at run time, so every cache read is an out-of-line indirect +// call. The epoch snapshot's loads are plain inline moves there, and cache +// reads vastly outnumber cache writes. +const NATIVE_SNAPSHOT: bool = portable_atomic::AtomicU128::is_always_lock_free() + && !(cfg!(target_arch = "x86_64") && !cfg!(target_feature = "avx")); + +pub(crate) type CacheSnapshot = as SelectStorage>::Storage; #[cfg(test)] mod tests { diff --git a/crates/weavepy-compiler/src/lib.rs b/crates/weavepy-compiler/src/lib.rs index b5f3cb84..6f840348 100644 --- a/crates/weavepy-compiler/src/lib.rs +++ b/crates/weavepy-compiler/src/lib.rs @@ -40,6 +40,7 @@ pub mod cpython_code; mod flowgraph; mod intern; mod mangle; +pub mod native_code; mod validate; pub use bytecode::{ @@ -156,8 +157,16 @@ pub use weavepy_parser::ast::expr_name; /// clones (a `replace()`d code object may change `constants`, so a /// cloned code object starts with an empty slot), never participates in /// equality, and is not serialized. +/// +/// The second field caches the payload's address once it's set (the VM's +/// hot accessor then reads one thin pointer instead of the fat `dyn` +/// handle and its alignment arithmetic). The `Arc` in the first field +/// keeps that address alive for the code object's lifetime. #[derive(Default)] -pub struct VmExt(pub std::sync::OnceLock>); +pub struct VmExt( + pub std::sync::OnceLock>, + pub std::sync::atomic::AtomicPtr<()>, +); impl Clone for VmExt { fn clone(&self) -> Self { @@ -216,6 +225,13 @@ pub struct JitHint { /// The tier-2 state has no further interest in this code's back /// edges (compiled, OSR budget spent). backedge_quiet: std::sync::atomic::AtomicBool, + /// The VM's pure-leaf verdict (0 not yet decided, 1 no, 2 yes, 3 an + /// effect leaf), mirrored here so native call sites read it without + /// the VM extension lookup. + pure_leaf: std::sync::atomic::AtomicU8, + /// Consecutive leaf evaluations the VM abandoned (see + /// [`Self::note_leaf_miss`]). + leaf_misses: std::sync::atomic::AtomicU8, } impl JitHint { @@ -269,6 +285,59 @@ impl JitHint { .store(true, std::sync::atomic::Ordering::Relaxed); } + /// The recorded pure-leaf verdict, if the VM has decided one. + #[must_use] + pub fn pure_leaf(&self) -> Option { + match self.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) { + 1 | 3 => Some(false), + 2 => Some(true), + _ => None, + } + } + + /// Whether the VM decided the code is an *effect leaf*: a pure leaf + /// but for attribute stores (never a pure leaf itself). + #[must_use] + pub fn effect_leaf(&self) -> bool { + self.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) == 3 + } + + /// Record an effect-leaf verdict (see [`Self::effect_leaf`]). + pub fn set_effect_leaf(&self) { + self.pure_leaf + .store(3, std::sync::atomic::Ordering::Relaxed); + } + + /// Count one abandoned leaf evaluation. A long enough run of them + /// (a callee the evaluator never settles, a store that always + /// declines) withdraws the leaf verdict for good, so calls stop + /// paying for an attempt before their ordinary path. + pub fn note_leaf_miss(&self) { + use std::sync::atomic::Ordering::Relaxed; + let n = self.leaf_misses.load(Relaxed).saturating_add(1); + self.leaf_misses.store(n, Relaxed); + if n >= 64 { + self.pure_leaf.store(1, Relaxed); + } + } + + /// A completed leaf evaluation ends a run of misses. + #[inline] + pub fn note_leaf_hit(&self) { + use std::sync::atomic::Ordering::Relaxed; + if self.leaf_misses.load(Relaxed) != 0 { + self.leaf_misses.store(0, Relaxed); + } + } + + /// Record the VM's pure-leaf verdict (see [`Self::pure_leaf`]). + pub fn set_pure_leaf(&self, yes: bool) { + self.pure_leaf.store( + if yes { 2 } else { 1 }, + std::sync::atomic::Ordering::Relaxed, + ); + } + #[must_use] pub fn is_compiled(&self) -> bool { self.compiled.load(std::sync::atomic::Ordering::Relaxed) diff --git a/crates/weavepy-compiler/src/native_code.rs b/crates/weavepy-compiler/src/native_code.rs new file mode 100644 index 00000000..b430758a --- /dev/null +++ b/crates/weavepy-compiler/src/native_code.rs @@ -0,0 +1,547 @@ +//! A compact, WeavePy-internal serialization of [`CodeObject`]s. +//! +//! The private frozen-stdlib cache stores compiled modules in this form +//! rather than as CPython `marshal` data: reading it back is a straight +//! copy of each field, with no CPython bytecode to parse and transcode. +//! It isn't an interchange format. Its layout is free to change with any +//! release, so a reader must check the version byte and treat any +//! mismatch or malformed input as a cache miss. +//! +//! Filenames aren't stored: every code object in a cached module shares +//! the module's filename, which the reader supplies. + +use std::sync::Arc; + +use crate::bytecode::{CacheTable, Instruction, OpCode}; +use crate::{CodeObject, ColSpan, Constant, ExcHandler}; + +/// The layout revision; bump it whenever the encoding changes. +pub const VERSION: u8 = 2; + +/// Encode `code` (and its nested code objects). `None` for a code object +/// the format doesn't carry: one with raw CPython wire overrides, or with +/// a constant that has no fixed value. +pub fn encode(code: &CodeObject) -> Option> { + let mut w = Writer(Vec::with_capacity(4096)); + w.byte(VERSION); + w.code(code)?; + Some(w.0) +} + +/// Decode what [`encode`] wrote, stamping `filename` on every code +/// object. `None` for input it didn't write (or a different version's). +pub fn decode(bytes: &[u8], filename: &str) -> Option { + let mut r = Reader { bytes, pos: 0 }; + if r.byte()? != VERSION { + return None; + } + let code = r.code(filename, 0)?; + (r.pos == bytes.len()).then_some(code) +} + +struct Writer(Vec); + +impl Writer { + fn byte(&mut self, b: u8) { + self.0.push(b); + } + + fn uint(&mut self, mut v: u64) { + while v >= 0x80 { + self.0.push((v as u8) | 0x80); + v >>= 7; + } + self.0.push(v as u8); + } + + fn int(&mut self, v: i64) { + self.uint(((v << 1) ^ (v >> 63)) as u64); + } + + fn bytes(&mut self, b: &[u8]) { + self.uint(b.len() as u64); + self.0.extend_from_slice(b); + } + + fn strs(&mut self, v: &[String]) { + self.uint(v.len() as u64); + for s in v { + self.bytes(s.as_bytes()); + } + } + + fn u32s(&mut self, v: &[u32]) { + self.uint(v.len() as u64); + for &x in v { + self.uint(u64::from(x)); + } + } + + fn deltas(&mut self, v: impl ExactSizeIterator) { + self.uint(v.len() as u64); + let mut prev = 0i64; + for x in v { + self.int(i64::from(x) - prev); + prev = i64::from(x); + } + } + + fn code(&mut self, c: &CodeObject) -> Option<()> { + if c.wire.is_some() { + return None; + } + self.bytes(c.name.as_bytes()); + self.bytes(c.qualname.as_bytes()); + self.uint(c.instructions.len() as u64); + for ins in &c.instructions { + self.byte(ins.op as u8); + self.uint(u64::from(ins.arg)); + } + self.uint(c.constants.len() as u64); + for k in &c.constants { + self.constant(k)?; + } + self.strs(&c.names); + self.strs(&c.varnames); + self.strs(&c.freevars); + self.strs(&c.cellvars); + self.uint(c.exception_table.len() as u64); + for h in &c.exception_table { + self.uint(u64::from(h.start)); + self.uint(u64::from(h.end)); + self.uint(u64::from(h.handler)); + self.uint(u64::from(h.depth)); + self.byte(u8::from(h.push_lasti)); + } + // Line numbers as deltas from the previous entry: mostly zero, so + // one byte each whatever the line. + self.deltas(c.linetable.iter().copied()); + self.deltas(c.coltable.iter().map(|s| s.end_lineno)); + for s in &c.coltable { + self.int(i64::from(s.col)); + self.int(i64::from(s.end_col)); + } + self.uint(u64::from(c.arg_count)); + self.uint(u64::from(c.posonly_count)); + self.uint(u64::from(c.kwonly_count)); + let flags = [ + c.has_varargs, + c.has_varkeywords, + c.is_class_body, + c.is_generator, + c.is_coroutine, + c.is_async_generator, + c.is_iterable_coroutine, + c.has_docstring, + c.is_method, + c.is_nested, + c.annotate_scope, + ]; + let bits = flags + .iter() + .enumerate() + .fold(0u64, |acc, (i, &f)| acc | (u64::from(f) << i)); + self.uint(bits); + self.uint(u64::from(c.future_flags)); + self.uint(c.stacksize.map_or(0, |s| u64::from(s) + 1)); + self.u32s(&c.no_interrupt_jumps); + self.bytes(&c.wire_marks); + self.strs(&c.hidden_locals); + self.strs(&c.const_identifiers); + Some(()) + } + + fn constant(&mut self, k: &Constant) -> Option<()> { + match k { + Constant::None => self.byte(0), + Constant::Bool(b) => self.byte(1 + u8::from(*b)), + Constant::Int(i) => { + self.byte(3); + self.int(*i); + } + Constant::BigInt(b) => { + self.byte(4); + self.bytes(&b.to_signed_bytes_le()); + } + Constant::Float(x) => { + self.byte(5); + self.0.extend_from_slice(&x.to_bits().to_le_bytes()); + } + Constant::Complex(re, im) => { + self.byte(6); + self.0.extend_from_slice(&re.to_bits().to_le_bytes()); + self.0.extend_from_slice(&im.to_bits().to_le_bytes()); + } + Constant::Str(s) => { + self.byte(7); + self.bytes(s.as_bytes()); + } + Constant::WStr(cps) => { + self.byte(8); + self.u32s(cps); + } + Constant::Bytes(b) => { + self.byte(9); + self.bytes(b); + } + Constant::Tuple(items) | Constant::FrozenSet(items) => { + self.byte(if matches!(k, Constant::Tuple(_)) { + 10 + } else { + 11 + }); + self.uint(items.len() as u64); + for it in items { + self.constant(it)?; + } + } + Constant::Code(c) => { + self.byte(12); + self.code(c)?; + } + Constant::Ellipsis => self.byte(13), + Constant::Slice(parts) => { + self.byte(14); + self.constant(&parts.0)?; + self.constant(&parts.1)?; + self.constant(&parts.2)?; + } + Constant::Unmarshallable => return None, + } + Some(()) + } +} + +struct Reader<'a> { + bytes: &'a [u8], + pos: usize, +} + +/// Nested code objects deeper than this are taken as corrupt input. +const MAX_NESTING: u32 = 200; + +impl Reader<'_> { + #[inline(always)] + fn byte(&mut self) -> Option { + let b = *self.bytes.get(self.pos)?; + self.pos += 1; + Some(b) + } + + #[inline(always)] + fn uint(&mut self) -> Option { + // Most values (opcodes' arguments, line numbers, lengths) fit + // in one byte. + let b = self.byte()?; + if b < 0x80 { + return Some(u64::from(b)); + } + self.uint_long(b) + } + + #[inline(never)] + fn uint_long(&mut self, first: u8) -> Option { + let mut v = u64::from(first & 0x7f); + let mut shift = 7; + loop { + let b = self.byte()?; + if shift >= 64 { + return None; + } + v |= u64::from(b & 0x7f) << shift; + if b < 0x80 { + return Some(v); + } + shift += 7; + } + } + + #[inline(always)] + fn u32(&mut self) -> Option { + u32::try_from(self.uint()?).ok() + } + + #[inline(always)] + fn int(&mut self) -> Option { + let v = self.uint()?; + Some(((v >> 1) as i64) ^ -((v & 1) as i64)) + } + + #[inline(always)] + fn i32(&mut self) -> Option { + i32::try_from(self.int()?).ok() + } + + /// A length prefix, bounded by what's left of the input (every + /// element takes at least one byte). + #[inline(always)] + fn len(&mut self) -> Option { + let n = usize::try_from(self.uint()?).ok()?; + (n <= self.bytes.len() - self.pos).then_some(n) + } + + #[inline(always)] + fn raw(&mut self) -> Option<&[u8]> { + let n = self.len()?; + let s = &self.bytes[self.pos..self.pos + n]; + self.pos += n; + Some(s) + } + + fn string(&mut self) -> Option { + Some(std::str::from_utf8(self.raw()?).ok()?.to_owned()) + } + + fn strs(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(self.string()?); + } + Some(v) + } + + fn u32s(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + for _ in 0..n { + v.push(self.u32()?); + } + Some(v) + } + + fn deltas(&mut self) -> Option> { + let n = self.len()?; + let mut v = Vec::with_capacity(n); + let mut prev = 0i64; + for _ in 0..n { + prev = prev.checked_add(self.int()?)?; + v.push(u32::try_from(prev).ok()?); + } + Some(v) + } + + fn f64(&mut self) -> Option { + let b = self.bytes.get(self.pos..self.pos + 8)?; + self.pos += 8; + Some(f64::from_bits(u64::from_le_bytes(b.try_into().ok()?))) + } + + fn code(&mut self, filename: &str, depth: u32) -> Option { + if depth > MAX_NESTING { + return None; + } + let name = self.string()?; + let qualname = self.string()?; + let n = self.len()?; + let mut instructions = Vec::with_capacity(n); + for _ in 0..n { + let op = OpCode::from_u8(self.byte()?)?; + instructions.push(Instruction::new(op, self.u32()?)); + } + let n = self.len()?; + let mut constants = Vec::with_capacity(n); + for _ in 0..n { + constants.push(self.constant(filename, depth)?); + } + let names = self.strs()?; + let varnames = self.strs()?; + let freevars = self.strs()?; + let cellvars = self.strs()?; + let n = self.len()?; + let mut exception_table = Vec::with_capacity(n); + for _ in 0..n { + exception_table.push(ExcHandler { + start: self.u32()?, + end: self.u32()?, + handler: self.u32()?, + depth: self.u32()?, + push_lasti: self.byte()? != 0, + }); + } + let linetable = self.deltas()?; + let end_lines = self.deltas()?; + let mut coltable = Vec::with_capacity(end_lines.len()); + for end_lineno in end_lines { + coltable.push(ColSpan { + end_lineno, + col: self.i32()?, + end_col: self.i32()?, + }); + } + let arg_count = self.u32()?; + let posonly_count = self.u32()?; + let kwonly_count = self.u32()?; + let bits = self.uint()?; + let flag = |i: u32| bits & (1 << i) != 0; + let future_flags = self.u32()?; + let stacksize = self.u32()?.checked_sub(1); + let no_interrupt_jumps = self.u32s()?; + let wire_marks = self.raw()?.to_vec(); + let hidden_locals = self.strs()?; + let const_identifiers = self.strs()?; + Some(CodeObject { + name, + qualname, + filename: filename.to_owned(), + caches: CacheTable::with_len(instructions.len()), + instructions, + constants, + names, + varnames, + freevars, + cellvars, + exception_table, + linetable, + coltable, + arg_count, + posonly_count, + kwonly_count, + has_varargs: flag(0), + has_varkeywords: flag(1), + is_class_body: flag(2), + is_generator: flag(3), + is_coroutine: flag(4), + is_async_generator: flag(5), + is_iterable_coroutine: flag(6), + has_docstring: flag(7), + is_method: flag(8), + is_nested: flag(9), + annotate_scope: flag(10), + future_flags, + stacksize, + no_interrupt_jumps, + wire_marks, + hidden_locals, + const_identifiers, + ..CodeObject::default() + }) + } + + fn constant(&mut self, filename: &str, depth: u32) -> Option { + Some(match self.byte()? { + 0 => Constant::None, + 1 => Constant::Bool(false), + 2 => Constant::Bool(true), + 3 => Constant::Int(self.int()?), + 4 => Constant::BigInt(num_bigint::BigInt::from_signed_bytes_le(self.raw()?)), + 5 => Constant::Float(self.f64()?), + 6 => Constant::Complex(self.f64()?, self.f64()?), + 7 => Constant::Str(self.string()?), + 8 => Constant::WStr(self.u32s()?), + 9 => Constant::Bytes(self.raw()?.to_vec()), + tag @ (10 | 11) => { + let n = self.len()?; + let mut items = Vec::with_capacity(n); + for _ in 0..n { + items.push(self.constant(filename, depth)?); + } + if tag == 10 { + Constant::Tuple(items) + } else { + Constant::FrozenSet(items) + } + } + 12 => Constant::Code(Arc::new(self.code(filename, depth + 1)?)), + 13 => Constant::Ellipsis, + 14 => Constant::Slice(Box::new(( + self.constant(filename, depth)?, + self.constant(filename, depth)?, + self.constant(filename, depth)?, + ))), + _ => return None, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn round_trips_constants_and_metadata() { + let inner = CodeObject { + name: "f".to_owned(), + qualname: "C.f".to_owned(), + filename: "m.py".to_owned(), + instructions: vec![ + Instruction::new(OpCode::Resume, 0), + Instruction::new(OpCode::LoadConst, 300), + Instruction::new(OpCode::ReturnValue, 0), + ], + constants: vec![Constant::Int(-5)], + varnames: vec!["x".to_owned()], + linetable: vec![1, 2, 2], + coltable: vec![ColSpan::default(); 3], + arg_count: 1, + is_generator: true, + is_nested: true, + stacksize: Some(3), + ..CodeObject::default() + }; + let code = CodeObject { + name: "".to_owned(), + qualname: "".to_owned(), + filename: "m.py".to_owned(), + instructions: vec![Instruction::new(OpCode::Nop, 0)], + constants: vec![ + Constant::None, + Constant::Bool(true), + Constant::Int(i64::MIN), + Constant::BigInt(num_bigint::BigInt::from(-1) << 100), + Constant::Float(-0.0), + Constant::Complex(1.5, f64::INFINITY), + Constant::Str("héllo".to_owned()), + Constant::WStr(vec![0x61, 0xd800]), + Constant::Bytes(vec![0, 255]), + Constant::Tuple(vec![Constant::Ellipsis, Constant::Str(String::new())]), + Constant::FrozenSet(vec![Constant::Int(1)]), + Constant::Slice(Box::new(( + Constant::Int(1), + Constant::None, + Constant::Int(-1), + ))), + Constant::Code(Arc::new(inner)), + ], + names: vec!["print".to_owned()], + exception_table: vec![ExcHandler { + start: 0, + end: 1, + handler: 1, + depth: 2, + push_lasti: true, + }], + no_interrupt_jumps: vec![7], + wire_marks: vec![0, 3], + hidden_locals: vec!["h".to_owned()], + const_identifiers: vec!["k".to_owned()], + future_flags: 0x100, + ..CodeObject::default() + }; + let bytes = encode(&code).expect("encodable"); + let back = decode(&bytes, "m.py").expect("decodable"); + assert_eq!(back, code); + } + + #[test] + fn rejects_truncated_or_foreign_input() { + let code = CodeObject { + name: "".to_owned(), + instructions: vec![Instruction::new(OpCode::Nop, 0)], + constants: vec![Constant::Str("x".repeat(40))], + ..CodeObject::default() + }; + let bytes = encode(&code).expect("encodable"); + for cut in 0..bytes.len() { + assert!(decode(&bytes[..cut], "").is_none()); + } + let mut other = bytes.clone(); + other[0] = VERSION + 1; + assert!(decode(&other, "").is_none()); + assert!(encode(&CodeObject { + constants: vec![Constant::Unmarshallable], + ..CodeObject::default() + }) + .is_none()); + } +} diff --git a/crates/weavepy-jit/src/analyze.rs b/crates/weavepy-jit/src/analyze.rs index fa661d61..45ac5419 100644 --- a/crates/weavepy-jit/src/analyze.rs +++ b/crates/weavepy-jit/src/analyze.rs @@ -2941,18 +2941,20 @@ fn kw_call_names(code: &CodeObject, cidx: u32) -> Vec<&str> { /// RFC 0073 WS5 — resolve a `CALL_KW` site's keyword permutation /// against the burned callee: keyword value `j` binds parameter slot -/// `(perm >> 4j) & 0xF` (tier-1's `CallPyKwNames` packing). Admitted -/// only when the filled set — the positional prefix plus the keyword -/// slots — is exactly `0..argc+kwc`: the marshaled call is then a -/// plain positional prefix through the unchanged `wpjit_call_py` -/// helper, and the trailing-defaults window binds the remaining tail. -/// Returns `(perm, filled count)`. +/// `(perm >> 4j) & 0xF` (tier-1's `CallPyKwNames` packing). The filled +/// set — the positional prefix plus the keyword slots — must reach +/// every parameter without a default; a defaulted parameter it skips +/// (`f(x, c=1)` over `def f(a, b=0, c=0)`) is a *gap* the call helper +/// fills from the callee's current defaults, so the marshaled call is +/// a positional prefix of `k` slots through `wpjit_call_py`, whose +/// trailing-defaults window binds any remaining tail. Returns +/// `(perm, k, gap mask)`. fn resolve_kw_perm( mark: &CalleeMark, names: &[&str], argc: usize, kw_slot: &mut dyn FnMut(u32, &str) -> Option, -) -> Result<(u32, usize), JitVerdict> { +) -> Result<(u32, usize, u32), JitVerdict> { if mark.kind != MarkKind::Py || mark.ctor { return Err(JitVerdict::UnsupportedOpcode("CALL_KW (callee kind)")); } @@ -2974,7 +2976,11 @@ fn resolve_kw_perm( perm |= slot << (4 * j); covered |= 1 << slot; } - if covered != (1u32 << k) - 1 { + // The highest filled slot bounds the marshaled prefix; the slots + // it skips must all have defaults. + let k = (32 - covered.leading_zeros()) as usize; + let gaps = ((1u32 << k) - 1) & !covered; + if gaps & ((1u32 << mark.min_args.min(16)) - 1) != 0 { return Err(JitVerdict::UnsupportedOpcode("CALL_KW (keyword gap)")); } // The uncovered tail binds trailing defaults, exactly like the @@ -2982,7 +2988,7 @@ fn resolve_kw_perm( if k < mark.min_args as usize || k > mark.arg_count as usize { return Err(JitVerdict::UnsupportedOpcode("CALL (arity)")); } - Ok((perm, k)) + Ok((perm, k, gaps)) } /// Map a representable [`Constant`] to its lane, or `None`. @@ -7467,6 +7473,7 @@ fn emit_instr( token: mark.token, argc: argc as u8, ret, + is_self: mark.is_self, }, Some(ret), stack, @@ -7545,7 +7552,7 @@ fn emit_instr( *max_stack = (*max_stack).max(stack.len() as u32); return Ok(()); }; - let (perm, _) = resolve_kw_perm(&mark, &names, argc, probes.kw_slot)?; + let (perm, k, gaps) = resolve_kw_perm(&mark, &names, argc, probes.kw_slot)?; for &ty in &arg_tys { if !ty.is_representable() { return Err(JitVerdict::TypeUnknown); @@ -7572,13 +7579,15 @@ fn emit_instr( live_to: pc + 1, interp_depth: mark.interp_depth, }); - *max_call_args = (*max_call_args).max(n as u32); + // The marshaled prefix, skipped defaulted slots included. + *max_call_args = (*max_call_args).max(k as u32); push( TOp::CallPyKw { token: mark.token, argc: argc as u8, kwc: kwc as u8, perm, + gaps, ret, }, Some(ret), diff --git a/crates/weavepy-jit/src/engine.rs b/crates/weavepy-jit/src/engine.rs index 272092b2..e5408456 100644 --- a/crates/weavepy-jit/src/engine.rs +++ b/crates/weavepy-jit/src/engine.rs @@ -9,12 +9,12 @@ use std::mem::{self, ManuallyDrop}; -use cranelift_codegen::ir::{types, AbiParam, Type}; +use cranelift_codegen::ir::{types, AbiParam, FuncRef, Type}; use cranelift_codegen::settings::{self, Configurable}; use cranelift_codegen::Context; use cranelift_frontend::FunctionBuilderContext; use cranelift_jit::{JITBuilder, JITModule}; -use cranelift_module::{Linkage, Module}; +use cranelift_module::{FuncId, Linkage, Module}; use crate::analyze::{JitVerdict, Probes}; use crate::ir::{ @@ -118,6 +118,21 @@ pub struct CompiledFrame { pub ret_lane: Option, scalar_leaf: bool, op_mix: OpMix, + /// The engine's handle for this function, for direct calls from + /// frames compiled later (see [`Self::direct_leaf`]). + func_id: FuncId, +} + +/// A compiled scalar leaf another frame may call directly (see +/// [`CompiledFrame::direct_leaf`]): its function and frame layout, and +/// the lanes of its parameters and result. +#[derive(Clone, Debug)] +pub struct DirectLeaf { + func_id: FuncId, + n_locals: u32, + max_stack: u32, + params: Vec, + ret: JitType, } impl CompiledFrame { @@ -131,6 +146,33 @@ impl CompiledFrame { self.scalar_leaf } + /// This frame as a direct-call target taking `arg_count` arguments: + /// a scalar leaf with no global or math guards (so nothing about + /// the namespace can invalidate it) and scalar parameter lanes. A + /// caller compiled on the same engine calls it in native code; its + /// numeric deopt restarts the call through the ordinary path. + #[must_use] + pub fn direct_leaf(&self, arg_count: u32) -> Option { + let scalar = |t: JitType| matches!(t, JitType::Int | JitType::Float | JitType::Bool); + if !self.scalar_leaf || !self.global_guards.is_empty() || !self.math_guards.is_empty() { + return None; + } + let ret = self.ret_lane.filter(|&t| scalar(t))?; + let params = self + .local_types + .get(..arg_count as usize)? + .iter() + .map(|t| t.filter(|&t| scalar(t))) + .collect::>>()?; + Some(DirectLeaf { + func_id: self.func_id, + n_locals: self.n_locals, + max_stack: self.max_stack, + params, + ret, + }) + } + /// `(generic, total)` statement counts: how many of the compiled /// statements go through the interpreter's generic object protocol /// (dynamic calls and attribute accesses). @@ -279,13 +321,40 @@ impl JitEngine { code: &CodeObject, resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, probes: &mut Probes<'_>, + ) -> Result { + self.compile_frame_direct(code, resolve, probes, &mut |_| None) + } + + /// [`Self::compile_frame`] with the direct-call targets of the callee + /// tokens: `direct(token)` names a scalar leaf compiled on this + /// engine (see [`CompiledFrame::direct_leaf`]) that a matching call + /// site enters in native code. + pub fn compile_frame_direct( + &mut self, + code: &CodeObject, + resolve: &mut dyn FnMut(&str) -> ResolvedGlobal, + probes: &mut Probes<'_>, + direct: &mut dyn FnMut(u32) -> Option, ) -> Result { let tfunc = crate::analyze::analyze_frame(code, resolve, probes)?; - self.compile_tfunc(&tfunc) + if calls_dynamically(&tfunc) { + return Err(JitVerdict::UnsupportedOpcode("dynamic call (loop-free)")); + } + self.compile_tfunc_direct(&tfunc, direct) } /// Compile an already-analyzed [`TFunc`] (also the unit-test entry). pub fn compile_tfunc(&mut self, tfunc: &TFunc) -> Result { + self.compile_tfunc_direct(tfunc, &mut |_| None) + } + + /// [`Self::compile_tfunc`] with direct-call targets (see + /// [`Self::compile_frame_direct`]). + pub fn compile_tfunc_direct( + &mut self, + tfunc: &TFunc, + direct: &mut dyn FnMut(u32) -> Option, + ) -> Result { // These operations each lower to a dedicated embedder helper. // Reject missing registrations before embedding an absolute address. for stmt in tfunc.blocks.iter().flat_map(|block| &block.stmts) { @@ -440,14 +509,64 @@ impl JitEngine { .returns .push(AbiParam::new(types::I64)); - build_function(&mut self.ctx.func, &mut self.fbctx, tfunc, self.ptr_ty); - let name = format!("wpjit_{}", self.next_id); self.next_id += 1; let id = self .module .declare_function(&name, Linkage::Local, &self.ctx.func.signature) .map_err(|_| JitVerdict::NotConverged)?; + // A scalar frame calls itself directly (declared above so the body + // can name its own function). + let self_func = (runtime::self_call_helper_addrs().is_some() + && self_direct_eligible(tfunc)) + .then(|| self.module.declare_func_in_func(id, &mut self.ctx.func)); + // Direct leaf targets, per call token whose sites agree with the + // leaf's arity and result lane (the lowering checks the argument + // lanes at each site). + let mut leaves: Vec<(u32, FuncRef, DirectLeaf)> = Vec::new(); + if runtime::self_call_helper_addrs().is_some() { + for stmt in tfunc.blocks.iter().flat_map(|b| &b.stmts) { + let TOp::CallPy { + token, + argc, + ret, + is_self: false, + } = stmt.op + else { + continue; + }; + if leaves.iter().any(|(t, ..)| *t == token) { + continue; + } + if let Some(leaf) = direct(token) { + if leaf.params.len() == argc as usize && leaf.ret == ret { + let fref = self + .module + .declare_func_in_func(leaf.func_id, &mut self.ctx.func); + leaves.push((token, fref, leaf)); + } + } + } + } + + build_function( + &mut self.ctx.func, + &mut self.fbctx, + tfunc, + self.ptr_ty, + self_func, + leaves + .into_iter() + .map(|(token, fref, leaf)| crate::lower::LeafTarget { + token, + func: fref, + n_locals: leaf.n_locals, + max_stack: leaf.max_stack, + params: leaf.params, + }) + .collect(), + ); + self.module .define_function(id, &mut self.ctx) .map_err(|_| JitVerdict::NotConverged)?; @@ -519,10 +638,41 @@ impl JitEngine { ret_none: tfunc.ret_none, scalar_leaf: is_scalar_leaf(tfunc), op_mix: op_mix(tfunc), + func_id: id, }) } } +/// Whether native code would cost a loop-free body more than it saves: +/// it makes a dynamic Python call, which from native code takes the +/// interpreter's generic call path (a nested run instead of the inline +/// activation an interpreted caller uses), while the compile itself is +/// the largest cost a short-lived method ever pays. +fn calls_dynamically(tfunc: &TFunc) -> bool { + let has_loop = !tfunc.range_loops.is_empty() + || !tfunc.list_loops.is_empty() + || !tfunc.iter_loops.is_empty() + || tfunc.blocks.iter().enumerate().any(|(i, b)| { + use crate::ir::TTerm; + match b.term { + TTerm::Jump(t) => t as usize <= i, + TTerm::BranchFalse { + target, + fallthrough, + } + | TTerm::BranchTrue { + target, + fallthrough, + } => target as usize <= i || fallthrough as usize <= i, + _ => false, + } + }); + if has_loop { + return false; + } + op_mix(tfunc).dyn_calls > 0 +} + /// `(generic, total)`: statements that hand an operation to the /// interpreter's generic object protocol (dynamic calls and attribute /// accesses) against all statements (see [`CompiledFrame::op_mix`]). @@ -649,5 +799,61 @@ fn is_scalar_leaf(tfunc: &TFunc) -> bool { }) } +/// Whether `tfunc` may call itself directly (see +/// [`crate::runtime::SelfEnterHelper`]): it makes a self call, and every +/// local, operand and result is a scalar and every operation pure scalar +/// work or a self call, so no activation of it ever pins an object. Its +/// activations can then share one embedder context (the pin table, the +/// parked result) and live on the native stack. +fn self_direct_eligible(tfunc: &TFunc) -> bool { + let scalar = |t: JitType| matches!(t, JitType::Int | JitType::Float | JitType::Bool); + let mut any_self = false; + let ops_ok = tfunc.blocks.iter().all(|b| { + let term_ok = matches!( + b.term, + TTerm::Return + | TTerm::Jump(_) + | TTerm::BranchFalse { .. } + | TTerm::BranchTrue { .. } + | TTerm::Deopt { .. } + ); + term_ok + && b.stmts.iter().all(|st| match st.op { + TOp::CallPy { + is_self: true, ret, .. + } => { + any_self = true; + scalar(ret) + } + TOp::PushConstInt(_) + | TOp::PushConstFloat(_) + | TOp::PushConstBool(_) + | TOp::LoadLocal(_) + | TOp::StoreLocal(_) + | TOp::IntArith(_) + | TOp::FloatArith(_) + | TOp::IntTrueDiv + | TOp::IntCmp(_) + | TOp::FloatCmp(_) + | TOp::IntNeg + | TOp::FloatNeg + | TOp::IntInvert + | TOp::IntNot + | TOp::FloatNot + | TOp::Pop + | TOp::Dup { .. } + | TOp::Swap2 + | TOp::SwapN { .. } + | TOp::IntToFloatTos { .. } + | TOp::IntToFloatSecond { .. } => true, + _ => false, + }) + }); + any_self + && ops_ok + && tfunc.ret_lane.is_some_and(scalar) + && tfunc.local_types.iter().flatten().all(|t| scalar(*t)) +} + #[cfg(test)] mod tests; diff --git a/crates/weavepy-jit/src/ir.rs b/crates/weavepy-jit/src/ir.rs index 289a4cac..66f7be00 100644 --- a/crates/weavepy-jit/src/ir.rs +++ b/crates/weavepy-jit/src/ir.rs @@ -173,17 +173,25 @@ pub enum TOp { /// callee takes the `Raised` exit at this pc; a result outside the /// `ret` lane (or a caller guard invalidated by the callee's side /// effects) deopts *after* the call with the result spilled. - CallPy { token: u32, argc: u8, ret: JitType }, + /// `is_self`: the callee is this very code object, which a scalar + /// frame calls directly (see `engine::self_direct_eligible`). + CallPy { + token: u32, + argc: u8, + ret: JitType, + is_self: bool, + }, /// RFC 0073 WS5 — a Python-to-Python *keyword* call (`CALL_KW`) /// through the same `wpjit_call_py` helper. Pops `argc + kwc` /// values (positionals below, keyword values above, interpreter /// stack order); the analyzer resolved each keyword to its /// parameter slot at compile time, packed 4 bits per keyword in /// `perm` (keyword value `j` → slot `(perm >> 4j) & 0xF`, tier-1's - /// `CallPyKwNames` encoding). The filled slots are validated to be - /// exactly `0..argc+kwc`, so lowering marshals a plain positional - /// prefix and the call helper needs no keyword awareness (the - /// trailing-defaults window binds any remaining tail). The names + /// `CallPyKwNames` encoding). Lowering marshals a positional prefix + /// up to the highest filled slot; the defaulted slots it skips + /// (`gaps`, one bit per slot) are tagged for the call helper to + /// bind from the callee's current defaults, and the + /// trailing-defaults window binds any remaining tail. The names /// tuple's `LOAD_CONST` is erased from the trace; it never exists /// on the native stack. Exits mirror [`TOp::CallPy`]. CallPyKw { @@ -191,6 +199,7 @@ pub enum TOp { argc: u8, kwc: u8, perm: u32, + gaps: u32, ret: JitType, }, /// RFC 0061 WS5 — `BINARY_SUBSCR` on a pinned list: pops the `int` diff --git a/crates/weavepy-jit/src/lib.rs b/crates/weavepy-jit/src/lib.rs index 4735c481..5e04cb84 100644 --- a/crates/weavepy-jit/src/lib.rs +++ b/crates/weavepy-jit/src/lib.rs @@ -34,7 +34,7 @@ pub use analyze::{ returns_none_syntactically, returns_self_syntactically, JitVerdict, MethodResolution, PathArena, Probes, ELEM_SENTINEL, }; -pub use engine::{CompiledFrame, JitEngine, OpMix}; +pub use engine::{CompiledFrame, DirectLeaf, JitEngine, OpMix}; pub use ir::{ ArithKind, AttrSiteMeta, BlockId, CalleeSpanMeta, CmpKind, CompSavedMeta, CtorFieldSrc, GlobalGuard, IterLoopMeta, ListLoopMeta, MathFunc, MathGuardMeta, MethodRet, MethodSiteMeta, @@ -51,15 +51,16 @@ pub use runtime::{ register_iter_helpers, register_iter_new_helper, register_iter_next_pair_helper, register_list_extra_helpers, register_list_from_range_helper, register_list_helpers, register_list_next_helper, register_math_helpers, register_poll_helper, - register_str_format_helpers, register_str_helpers, register_str_method_helper, - register_str_write_helpers, register_truth_helper, register_tuple_read_helpers, - register_unbox_int_helper, AttrGetChainHelper, AttrGetHelper, AttrSetHelper, BuildListHelper, - BuildTupleHelper, BytesGetHelper, CachedAttrChainHelper, CallDynHelper, CallMethodHelper, - CallPyHelper, CallStatus, CellGetHelper, CellSetHelper, DictAccessHelper, DynAttrHelper, - GetIterHelper, IterNextHelper, IterNextPairHelper, JitFrame, JitStatus, ListAppendHelper, - ListFromRangeHelper, ListGetHelper, ListLenHelper, ListNextHelper, ListRepeatHelper, - ListSetHelper, ListSliceHelper, MathBinaryHelper, MathUnaryHelper, PollHelper, SlotTag, - StrEqHelper, StrLenHelper, StrModHelper, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, + register_self_call_helpers, register_str_format_helpers, register_str_helpers, + register_str_method_helper, register_str_write_helpers, register_truth_helper, + register_tuple_read_helpers, register_unbox_int_helper, AttrGetChainHelper, AttrGetHelper, + AttrSetHelper, BuildListHelper, BuildTupleHelper, BytesGetHelper, CachedAttrChainHelper, + CallDynHelper, CallMethodHelper, CallPyHelper, CallStatus, CellGetHelper, CellSetHelper, + DictAccessHelper, DynAttrHelper, GetIterHelper, IterNextHelper, IterNextPairHelper, JitFrame, + JitStatus, ListAppendHelper, ListFromRangeHelper, ListGetHelper, ListLenHelper, ListNextHelper, + ListRepeatHelper, ListSetHelper, ListSliceHelper, MathBinaryHelper, MathUnaryHelper, + PollHelper, SelfEnterHelper, SelfExitHelper, SelfSlowHelper, SlotTag, StrEqHelper, + StrLenHelper, StrModHelper, CALL_GAPS, DICT_KEY_INT, DICT_KEY_STR, DICT_VAL_FLOAT, DICT_VAL_INT, DICT_VAL_OBJ, ITER_ELEM_STR, JIT_POLL_STRIDE, MAX_ATTR_CHAIN_LEN, MAX_CACHED_ATTR_CHAIN_LEN, }; diff --git a/crates/weavepy-jit/src/lower.rs b/crates/weavepy-jit/src/lower.rs index 1ad84fe7..9f327aac 100644 --- a/crates/weavepy-jit/src/lower.rs +++ b/crates/weavepy-jit/src/lower.rs @@ -13,8 +13,8 @@ use cranelift_codegen::ir::condcodes::{FloatCC, IntCC}; use cranelift_codegen::ir::{ - types, AbiParam, Block, BlockArg, Function, InstBuilder, MemFlags, SigRef, Signature, Type, - Value, + types, AbiParam, Block, BlockArg, FuncRef, Function, InstBuilder, MemFlags, SigRef, Signature, + StackSlot, StackSlotData, StackSlotKind, Type, Value, }; use cranelift_frontend::{FunctionBuilder, FunctionBuilderContext, Variable}; @@ -32,6 +32,20 @@ const OFF_STACK_TAGS: i32 = core::mem::offset_of!(JitFrame, stack_tags) as i32; const OFF_STACK_LEN: i32 = core::mem::offset_of!(JitFrame, stack_len) as i32; const OFF_CALL_ARGS: i32 = core::mem::offset_of!(JitFrame, call_args) as i32; const OFF_CALL_TAGS: i32 = core::mem::offset_of!(JitFrame, call_tags) as i32; +const OFF_N_LOCALS: i32 = core::mem::offset_of!(JitFrame, n_locals) as i32; +const OFF_STACK_CAP: i32 = core::mem::offset_of!(JitFrame, stack_cap) as i32; +const OFF_CTX: i32 = core::mem::offset_of!(JitFrame, ctx) as i32; + +/// A call token whose sites may enter a compiled scalar leaf directly +/// (see `engine::CompiledFrame::direct_leaf`): the leaf's function in +/// this module, its frame layout, and its parameter lanes. +pub(crate) struct LeafTarget { + pub token: u32, + pub func: FuncRef, + pub n_locals: u32, + pub max_stack: u32, + pub params: Vec, +} /// Build the Cranelift function body for `tfunc` into `func`. pub(crate) fn build_function( @@ -39,9 +53,13 @@ pub(crate) fn build_function( fbctx: &mut FunctionBuilderContext, tfunc: &TFunc, ptr_ty: Type, + self_func: Option, + leaves: Vec, ) { let mut builder = FunctionBuilder::new(func, fbctx); let mut lc = Lowerer::new(&mut builder, tfunc, ptr_ty); + lc.self_func = self_func; + lc.leaves = leaves; lc.build(); builder.seal_all_blocks(); builder.finalize(); @@ -100,6 +118,19 @@ struct Lowerer<'a, 'b> { poll_countdown: Option, /// The abstract operand stack: SSA value + lane. vstack: Vec<(Value, JitType)>, + /// This function itself, for direct self calls (see + /// `engine::self_direct_eligible`); `None` when it makes none. + self_func: Option, + /// The callee frame of a direct self call (lazy): its `JitFrame` + /// then its buffers. + self_slot: Option, + /// Imported signatures of the self-call helpers (lazy): `(frame) -> + /// i64` and the slow path's `(frame, callee, status, token, tag) -> + /// i64`. + self_sig: Option, + self_slow_sig: Option, + /// Direct scalar-leaf call targets by token. + leaves: Vec, } impl<'a, 'b> Lowerer<'a, 'b> { @@ -130,6 +161,11 @@ impl<'a, 'b> Lowerer<'a, 'b> { poll_sig: None, poll_countdown: None, vstack: Vec::new(), + self_func: None, + self_slot: None, + self_sig: None, + self_slow_sig: None, + leaves: Vec::new(), } } @@ -948,14 +984,28 @@ impl<'a, 'b> Lowerer<'a, 'b> { let depth = self.vstack.len() - 2; self.emit_int_to_float(depth, guarded, stmt.pc); } - TOp::CallPy { token, argc, ret } => self.emit_call_py(token, argc, 0, 0, ret, stmt.pc), + TOp::CallPy { + token, + argc, + ret, + is_self, + } => { + if is_self && self.self_func.is_some() { + self.emit_call_self(token, argc, ret, stmt.pc); + } else if let Some(ix) = self.leaf_for(token, argc) { + self.emit_call_leaf(ix, token, argc, ret, stmt.pc); + } else { + self.emit_call_py(token, argc, 0, 0, 0, ret, stmt.pc); + } + } TOp::CallPyKw { token, argc, kwc, perm, + gaps, ret, - } => self.emit_call_py(token, argc, kwc, perm, ret, stmt.pc), + } => self.emit_call_py(token, argc, kwc, perm, gaps, ret, stmt.pc), TOp::ListGet { elem } => self.emit_list_get(elem, stmt.pc), TOp::ListSet => self.emit_list_set(stmt.pc), TOp::CellGet { idx, lane } => self.emit_cell_get(idx, lane, stmt.pc), @@ -3089,9 +3139,343 @@ impl<'a, 'b> Lowerer<'a, 'b> { /// (4 bits each, tier-1's `CallPyKwNames` packing). The analyzer /// validated the filled set to be exactly `0..argc+kwc`, so the /// helper still sees a plain positional prefix. - fn emit_call_py(&mut self, token: u32, argc: u8, kwc: u8, perm: u32, ret: JitType, pc: u32) { + /// Lower a direct self call (see `engine::self_direct_eligible`): + /// charge the activation through the enter helper, fill a callee + /// `JitFrame` on this function's native stack frame (the arguments + /// in its first locals, the caller's embedder context shared), call + /// this function itself, and release the charge. A callee that + /// deopts or raises finishes through the slow helper, whose + /// [`crate::runtime::CallStatus`] the caller handles as it does the + /// call helper's. When the enter helper declines (recursion limit, + /// pending interpreter work, observers), the ordinary call runs. + fn emit_call_self(&mut self, token: u32, argc: u8, ret: JitType, pc: u32) { + let trusted = MemFlags::trusted(); + let (enter_addr, exit_addr, slow_addr) = + runtime::self_call_helper_addrs().expect("checked by the engine"); + let n = argc as usize; + let base = self.vstack.len() - n; + let args: Vec<(Value, JitType)> = self.vstack[base..].to_vec(); + self.vstack.truncate(base); + let snapshot = self.vstack.clone(); + self.writeback_locals(); + self.store_call_site_pc(pc); + + let sig = self.self_sig(); + let enter = self.b.ins().iconst(self.ptr_ty, enter_addr as i64); + let call = self.b.ins().call_indirect(sig, enter, &[self.frame_ptr]); + let declined = self.b.inst_results(call)[0]; + + let direct_b = self.b.create_block(); + let generic_b = self.b.create_block(); + let join_b = self.b.create_block(); + self.b.append_block_param(join_b, Self::cl_ty(ret)); + let go = self.b.ins().icmp_imm(IntCC::Equal, declined, 0); + self.b.ins().brif(go, direct_b, &[], generic_b, &[]); + + // Declined: the ordinary call helper. + self.b.switch_to_block(generic_b); + self.vstack.extend(args.iter().copied()); + self.emit_call_py(token, argc, 0, 0, 0, ret, pc); + let (v, _) = self.vstack.pop().expect("the call's result"); + self.b.ins().jump(join_b, &[v.into()]); + + // Direct: the callee frame and buffers, laid out in one slot. + self.b.switch_to_block(direct_b); + let n_locals = self.tfunc.n_locals.max(1) as i32; + let cap = self.tfunc.max_stack as i32 + 1; + let call_cap = self.tfunc.max_call_args.max(1) as i32; + let frame_size = core::mem::size_of::() as i32; + let off_locals = frame_size; + let off_spill = off_locals + n_locals * 8; + let off_tags = off_spill + cap * 8; + let off_call_args = (off_tags + cap * 4 + 7) & !7; + let off_call_tags = off_call_args + call_cap * 8; + let size = off_call_tags + call_cap * 4; + let slot = match self.self_slot { + Some(slot) => slot, + None => { + let slot = self.b.create_sized_stack_slot(StackSlotData::new( + StackSlotKind::ExplicitSlot, + size as u32, + 3, + )); + self.self_slot = Some(slot); + slot + } + }; + let fp = self.b.ins().stack_addr(self.ptr_ty, slot, 0); + let at = |b: &mut FunctionBuilder<'_>, ptr_ty: Type, off: i32| { + b.ins().stack_addr(ptr_ty, slot, off) + }; + let locals = at(self.b, self.ptr_ty, off_locals); + let spill = at(self.b, self.ptr_ty, off_spill); + let tags = at(self.b, self.ptr_ty, off_tags); + let cargs = at(self.b, self.ptr_ty, off_call_args); + let ctags = at(self.b, self.ptr_ty, off_call_tags); + let ctx = self + .b + .ins() + .load(self.ptr_ty, trusted, self.frame_ptr, OFF_CTX); + let zero64 = self.b.ins().iconst(types::I64, 0); + let zero32 = self.b.ins().iconst(types::I32, 0); + self.b.ins().store(trusted, locals, fp, OFF_LOCALS); + let nl = self.b.ins().iconst(types::I32, i64::from(n_locals)); + self.b.ins().store(trusted, nl, fp, OFF_N_LOCALS); + self.b.ins().store(trusted, zero32, fp, OFF_ENTRY_PC); + self.b.ins().store(trusted, zero64, fp, OFF_RET_BITS); + self.b.ins().store(trusted, zero32, fp, OFF_RET_TAG); + self.b.ins().store(trusted, zero32, fp, OFF_DEOPT_PC); + self.b.ins().store(trusted, spill, fp, OFF_STACK_SPILL); + self.b.ins().store(trusted, tags, fp, OFF_STACK_TAGS); + self.b.ins().store(trusted, zero32, fp, OFF_STACK_LEN); + let capv = self.b.ins().iconst(types::I32, i64::from(cap)); + self.b.ins().store(trusted, capv, fp, OFF_STACK_CAP); + self.b.ins().store(trusted, ctx, fp, OFF_CTX); + self.b.ins().store(trusted, cargs, fp, OFF_CALL_ARGS); + self.b.ins().store(trusted, ctags, fp, OFF_CALL_TAGS); + // The arguments bind the first locals (their lanes are the + // parameters' own); every other local starts zeroed, as the + // framed entries leave it. + for slot_ix in 0..n_locals { + let off = slot_ix * 8; + match args.get(slot_ix as usize) { + Some(&(v, _)) => { + self.b.ins().store(trusted, v, locals, off); + } + None => { + self.b.ins().store(trusted, zero64, locals, off); + } + } + } + let self_func = self.self_func.expect("checked by the caller"); + let call = self.b.ins().call(self_func, &[fp]); + let status = self.b.inst_results(call)[0]; + let exit = self.b.ins().iconst(self.ptr_ty, exit_addr as i64); + self.b.ins().call_indirect(sig, exit, &[self.frame_ptr]); + + let returned_b = self.b.create_block(); + let slow_b = self.b.create_block(); + let is_ret = self + .b + .ins() + .icmp_imm(IntCC::Equal, status, JitStatus::Returned as i64); + self.b.ins().brif(is_ret, returned_b, &[], slow_b, &[]); + + self.b.switch_to_block(returned_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, fp, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + // Deopted or raised: finished by the slow helper. + self.b.switch_to_block(slow_b); + let slow_sig = self.self_slow_sig(); + let slow = self.b.ins().iconst(self.ptr_ty, slow_addr as i64); + let tokenv = self.b.ins().iconst(types::I64, i64::from(token)); + let tagv = self.b.ins().iconst(types::I64, Self::tag(ret)); + let call = + self.b + .ins() + .call_indirect(slow_sig, slow, &[self.frame_ptr, fp, status, tokenv, tagv]); + let cstatus = self.b.inst_results(call)[0]; + let ok_b = self.b.create_block(); + let bad_b = self.b.create_block(); + let is_ok = self.b.ins().icmp_imm(IntCC::Equal, cstatus, 0); + self.b.ins().brif(is_ok, ok_b, &[], bad_b, &[]); + self.b.switch_to_block(bad_b); + let raised_b = self.b.create_block(); + let boxed_b = self.b.create_block(); + let is_raised = self.b.ins().icmp_imm(IntCC::Equal, cstatus, 1); + self.b.ins().brif(is_raised, raised_b, &[], boxed_b, &[]); + self.b.switch_to_block(raised_b); + self.emit_exit(pc, &snapshot, JitStatus::Raised); + self.b.switch_to_block(boxed_b); + self.emit_exit(pc + 1, &snapshot, JitStatus::Deopt); + self.b.switch_to_block(ok_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, self.frame_ptr, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + self.b.switch_to_block(join_b); + let v = self.b.block_params(join_b)[0]; + self.vstack.push((v, ret)); + } + + /// The direct leaf target of `token` when this site's argument lanes + /// are exactly the leaf's parameter lanes. + fn leaf_for(&self, token: u32, argc: u8) -> Option { + let ix = self.leaves.iter().position(|l| l.token == token)?; + let base = self.vstack.len().checked_sub(argc as usize)?; + let lanes = self.vstack[base..].iter().map(|&(_, ty)| ty); + lanes + .eq(self.leaves[ix].params.iter().copied()) + .then_some(ix) + } + + /// Lower a direct call of a compiled scalar leaf (see [`LeafTarget`]): + /// charge the activation through the self-call enter helper, fill the + /// leaf's `JitFrame` on this function's native stack frame, call it, + /// and release the charge. The leaf only computes, so when the enter + /// helper declines or the leaf deopts (an overflow, a zero divisor), + /// the ordinary call helper runs the call from the start. + fn emit_call_leaf(&mut self, ix: usize, token: u32, argc: u8, ret: JitType, pc: u32) { + let trusted = MemFlags::trusted(); + let (enter_addr, exit_addr, _) = + runtime::self_call_helper_addrs().expect("checked by the engine"); + let n = argc as usize; + let base = self.vstack.len() - n; + let args: Vec<(Value, JitType)> = self.vstack[base..].to_vec(); + self.vstack.truncate(base); + self.writeback_locals(); + self.store_call_site_pc(pc); + + let sig = self.self_sig(); + let enter = self.b.ins().iconst(self.ptr_ty, enter_addr as i64); + let call = self.b.ins().call_indirect(sig, enter, &[self.frame_ptr]); + let declined = self.b.inst_results(call)[0]; + + let direct_b = self.b.create_block(); + let generic_b = self.b.create_block(); + let join_b = self.b.create_block(); + self.b.append_block_param(join_b, Self::cl_ty(ret)); + let go = self.b.ins().icmp_imm(IntCC::Equal, declined, 0); + self.b.ins().brif(go, direct_b, &[], generic_b, &[]); + + // The leaf's frame and buffers, in one slot per site. + self.b.switch_to_block(direct_b); + let leaf = &self.leaves[ix]; + let (func, n_locals, cap) = ( + leaf.func, + leaf.n_locals.max(1) as i32, + leaf.max_stack as i32 + 1, + ); + let frame_size = core::mem::size_of::() as i32; + let off_locals = frame_size; + let off_spill = off_locals + n_locals * 8; + let off_tags = off_spill + cap * 8; + let size = off_tags + cap * 4; + let slot = self.b.create_sized_stack_slot(StackSlotData::new( + StackSlotKind::ExplicitSlot, + size as u32, + 3, + )); + let fp = self.b.ins().stack_addr(self.ptr_ty, slot, 0); + let locals = self.b.ins().stack_addr(self.ptr_ty, slot, off_locals); + let spill = self.b.ins().stack_addr(self.ptr_ty, slot, off_spill); + let tags = self.b.ins().stack_addr(self.ptr_ty, slot, off_tags); + let null = self.b.ins().iconst(self.ptr_ty, 0); + let zero64 = self.b.ins().iconst(types::I64, 0); + let zero32 = self.b.ins().iconst(types::I32, 0); + self.b.ins().store(trusted, locals, fp, OFF_LOCALS); + let nl = self.b.ins().iconst(types::I32, i64::from(n_locals)); + self.b.ins().store(trusted, nl, fp, OFF_N_LOCALS); + self.b.ins().store(trusted, zero32, fp, OFF_ENTRY_PC); + self.b.ins().store(trusted, zero64, fp, OFF_RET_BITS); + self.b.ins().store(trusted, zero32, fp, OFF_RET_TAG); + self.b.ins().store(trusted, zero32, fp, OFF_DEOPT_PC); + self.b.ins().store(trusted, spill, fp, OFF_STACK_SPILL); + self.b.ins().store(trusted, tags, fp, OFF_STACK_TAGS); + self.b.ins().store(trusted, zero32, fp, OFF_STACK_LEN); + let capv = self.b.ins().iconst(types::I32, i64::from(cap)); + self.b.ins().store(trusted, capv, fp, OFF_STACK_CAP); + // A scalar leaf reads no embedder context and marshals no calls. + self.b.ins().store(trusted, null, fp, OFF_CTX); + self.b.ins().store(trusted, null, fp, OFF_CALL_ARGS); + self.b.ins().store(trusted, null, fp, OFF_CALL_TAGS); + for slot_ix in 0..n_locals { + let v = args.get(slot_ix as usize).map_or(zero64, |&(v, _)| v); + self.b.ins().store(trusted, v, locals, slot_ix * 8); + } + let call = self.b.ins().call(func, &[fp]); + let status = self.b.inst_results(call)[0]; + let exit = self.b.ins().iconst(self.ptr_ty, exit_addr as i64); + self.b.ins().call_indirect(sig, exit, &[self.frame_ptr]); + let returned_b = self.b.create_block(); + let is_ret = self + .b + .ins() + .icmp_imm(IntCC::Equal, status, JitStatus::Returned as i64); + self.b.ins().brif(is_ret, returned_b, &[], generic_b, &[]); + self.b.switch_to_block(returned_b); + let v = self + .b + .ins() + .load(Self::cl_ty(ret), trusted, fp, OFF_RET_BITS); + self.b.ins().jump(join_b, &[v.into()]); + + // Declined or deopted: the ordinary call, from the start. + self.b.switch_to_block(generic_b); + self.vstack.extend(args.iter().copied()); + self.emit_call_py(token, argc, 0, 0, 0, ret, pc); + let (v, _) = self.vstack.pop().expect("the call's result"); + self.b.ins().jump(join_b, &[v.into()]); + + self.b.switch_to_block(join_b); + let v = self.b.block_params(join_b)[0]; + self.vstack.push((v, ret)); + } + + /// The `(frame) -> i64` signature of the self-call enter/exit + /// helpers (lazy). + fn self_sig(&mut self) -> SigRef { + if let Some(sig) = self.self_sig { + return sig; + } + let mut sig = Signature::new(self.b.func.signature.call_conv); + sig.params.push(AbiParam::new(self.ptr_ty)); + sig.returns.push(AbiParam::new(types::I64)); + let r = self.b.import_signature(sig); + self.self_sig = Some(r); + r + } + + /// The slow self-call helper's signature (lazy). + fn self_slow_sig(&mut self) -> SigRef { + if let Some(sig) = self.self_slow_sig { + return sig; + } + let mut sig = Signature::new(self.b.func.signature.call_conv); + sig.params.push(AbiParam::new(self.ptr_ty)); // frame + sig.params.push(AbiParam::new(self.ptr_ty)); // callee frame + sig.params.push(AbiParam::new(types::I64)); // callee status + sig.params.push(AbiParam::new(types::I64)); // token + sig.params.push(AbiParam::new(types::I64)); // expected tag + sig.returns.push(AbiParam::new(types::I64)); // call status + let r = self.b.import_signature(sig); + self.self_slow_sig = Some(r); + r + } + + #[allow(clippy::too_many_arguments)] + fn emit_call_py( + &mut self, + token: u32, + argc: u8, + kwc: u8, + perm: u32, + gaps: u32, + ret: JitType, + pc: u32, + ) { let trusted = MemFlags::trusted(); let n = argc as usize + kwc as usize; + // Skipped defaulted slots: the helper binds them (see + // `SlotTag::Default`). + let mut g = gaps; + while g != 0 { + let slot = g.trailing_zeros() as i32; + g &= g - 1; + let tagv = self + .b + .ins() + .iconst(types::I32, runtime::SlotTag::Default as i64); + self.b + .ins() + .store(trusted, tagv, self.call_tags_base, slot * 4); + } let base = self.vstack.len() - n; for (j, &(v, ty)) in self.vstack[base..].iter().enumerate() { let dst = if j < argc as usize { @@ -3121,9 +3505,15 @@ impl<'a, 'b> Lowerer<'a, 'b> { .ins() .iconst(self.ptr_ty, runtime::call_py_helper_addr() as i64); let tokenv = self.b.ins().iconst(types::I32, i64::from(token)); - // The helper receives the *filled* count — keyword values were - // shuffled into a contiguous positional prefix above. - let argcv = self.b.ins().iconst(types::I32, n as i64); + // The helper receives the prefix length — keyword values were + // shuffled into parameter slots above, with any skipped + // defaulted slots flagged. + let filled = if gaps == 0 { + n as i64 + } else { + i64::from((n as u32 + gaps.count_ones()) | runtime::CALL_GAPS) + }; + let argcv = self.b.ins().iconst(types::I32, filled); let expect = self.b.ins().iconst(types::I32, Self::tag(ret)); let call = self.b diff --git a/crates/weavepy-jit/src/runtime.rs b/crates/weavepy-jit/src/runtime.rs index 67d74be6..bd9e6b57 100644 --- a/crates/weavepy-jit/src/runtime.rs +++ b/crates/weavepy-jit/src/runtime.rs @@ -122,8 +122,16 @@ pub enum SlotTag { /// exit, or a provably-`None` method-call result). The bits are /// ignored; the embedder rebuilds `Object::None`. None = 6, + /// A call argument slot a keyword call skipped: the call helper + /// binds the callee's default there before anything reads it. Only + /// ever appears in a call marshal buffer under [`CALL_GAPS`]. + Default = 7, } +/// Set in a `wpjit_call_py` argument count when some marshaled slots +/// are tagged [`SlotTag::Default`]. +pub const CALL_GAPS: u32 = 1 << 31; + impl SlotTag { /// Decode a raw tag written by native code. #[inline] @@ -136,6 +144,7 @@ impl SlotTag { 4 => SlotTag::ListPin, 5 => SlotTag::ObjPin, 6 => SlotTag::None, + 7 => SlotTag::Default, _ => SlotTag::Int, } } @@ -241,6 +250,57 @@ pub(crate) fn call_py_helper_addr() -> usize { CALL_PY_HELPER.load(std::sync::atomic::Ordering::Acquire) } +/// The embedder's direct self-call helpers (see +/// `engine::self_direct_eligible`): a scalar frame calls itself natively, +/// its callee's [`JitFrame`] on the native stack and the caller's +/// embedder context shared. +/// +/// - [`SelfEnterHelper`] charges one activation (recursion depth, GIL +/// countdown) before the call: `0` to call directly, non-zero to take +/// the ordinary `wpjit_call_py` path instead (nothing charged). +/// - [`SelfExitHelper`] releases the charge after the callee returns. +/// - [`SelfSlowHelper`] finishes a callee that did not return: its +/// [`JitStatus`] (`Deopt` or `Raised`) with its frame still live. +/// Returns a [`CallStatus`] for the caller, as the call helper does. +pub type SelfEnterHelper = unsafe extern "C" fn(frame: *mut JitFrame) -> i64; +/// See [`SelfEnterHelper`]. +pub type SelfExitHelper = unsafe extern "C" fn(frame: *mut JitFrame) -> i64; +/// See [`SelfEnterHelper`]. +pub type SelfSlowHelper = unsafe extern "C" fn( + frame: *mut JitFrame, + callee: *mut JitFrame, + status: i64, + token: i64, + expect_tag: i64, +) -> i64; + +static SELF_ENTER_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static SELF_EXIT_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); +static SELF_SLOW_HELPER: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); + +/// Register the direct self-call helpers (see [`SelfEnterHelper`]). +pub fn register_self_call_helpers( + enter: SelfEnterHelper, + exit: SelfExitHelper, + slow: SelfSlowHelper, +) { + use std::sync::atomic::Ordering::Release; + SELF_ENTER_HELPER.store(enter as usize, Release); + SELF_EXIT_HELPER.store(exit as usize, Release); + SELF_SLOW_HELPER.store(slow as usize, Release); +} + +/// The registered self-call helpers' addresses (enter, exit, slow), or +/// `None` when any is absent. +#[must_use] +pub(crate) fn self_call_helper_addrs() -> Option<(usize, usize, usize)> { + use std::sync::atomic::Ordering::Acquire; + let a = SELF_ENTER_HELPER.load(Acquire); + let b = SELF_EXIT_HELPER.load(Acquire); + let c = SELF_SLOW_HELPER.load(Acquire); + (a != 0 && b != 0 && c != 0).then_some((a, b, c)) +} + /// RFC 0061 WS5 — the embedder's pinned-list *read* helper. `pin` /// indexes the per-entry pinned-object table on the embedder context; /// `idx` is the (possibly negative) Python index. Returns `0` (Ok) with diff --git a/crates/weavepy-vm/src/builtin_types.rs b/crates/weavepy-vm/src/builtin_types.rs index 8fcba4d6..46ae4e32 100644 --- a/crates/weavepy-vm/src/builtin_types.rs +++ b/crates/weavepy-vm/src/builtin_types.rs @@ -1340,6 +1340,9 @@ thread_local! { const { std::cell::RefCell::new(None) }; static PROPERTY_CLASS: std::cell::RefCell>> = const { std::cell::RefCell::new(None) }; + /// The adopted registry behind [`builtin_types`]'s fast path. + static BUILTIN_TYPES_PTR: std::cell::Cell<*const BuiltinTypes> = + const { std::cell::Cell::new(std::ptr::null()) }; } /// Drop this thread's lazily-built type registry (and the derived @@ -1352,6 +1355,7 @@ thread_local! { pub fn clear_thread_type_registry() { let _ = PROPERTY_CLASS.try_with(|slot| slot.borrow_mut().take()); let _ = BUILTIN_TYPES.try_with(|slot| slot.borrow_mut().take()); + let _ = BUILTIN_TYPES_PTR.try_with(|p| p.set(std::ptr::null())); } /// Per-thread accessor. The registry is constructed lazily on first @@ -1370,7 +1374,30 @@ pub fn property_class() -> Rc { }) } -pub fn builtin_types() -> Rc { +/// This thread's type registry. A registry is never freed once a thread +/// has adopted it (see [`adopt_registry`]), so the reference is valid for +/// the rest of the process: the hot accessor is one thread-local load, with +/// no borrow flag and no reference-count traffic. +#[inline] +pub fn builtin_types() -> &'static BuiltinTypes { + let p = BUILTIN_TYPES_PTR.with(std::cell::Cell::get); + if p.is_null() { + return builtin_types_init(); + } + // SAFETY: `adopt_registry` leaked a strong count before publishing the + // pointer, so the registry outlives every thread that can read it. + unsafe { &*p } +} + +/// [`builtin_types`] as an owned handle (for publishing to other threads). +pub fn builtin_types_rc() -> Rc { + builtin_types(); + BUILTIN_TYPES.with(|cell| cell.borrow().clone().expect("registry installed above")) +} + +#[cold] +#[inline(never)] +fn builtin_types_init() -> &'static BuiltinTypes { let (bt, fresh) = BUILTIN_TYPES.with(|cell| { if let Some(bt) = cell.borrow().as_ref() { return (bt.clone(), false); @@ -1379,15 +1406,29 @@ pub fn builtin_types() -> Rc { *cell.borrow_mut() = Some(bt.clone()); (bt, true) }); + let bt = adopt_registry(&bt); if fresh { // Deferred surface pass (RFC 0056 WS4): synthesizing descriptor- // type members re-enters `builtin_types()`, which must resolve to // the just-published cell rather than recursively rebuild. - crate::type_surface::install_docs_table_surface(&bt); + crate::type_surface::install_docs_table_surface(bt); } bt } +/// Publish `bt` as this thread's registry for [`builtin_types`], leaking +/// one strong count so the reference it hands out stays valid. The type +/// objects inside already live for the process (each one's MRO holds +/// itself), so the leak is the registry struct alone, once per thread +/// that adopts one. +fn adopt_registry(bt: &Rc) -> &'static BuiltinTypes { + let p = Rc::as_ptr(bt); + std::mem::forget(bt.clone()); + BUILTIN_TYPES_PTR.with(|c| c.set(p)); + // SAFETY: the leaked count above keeps the registry alive. + unsafe { &*p } +} + /// Resolve `__objclass__` for a built-in method/slot-wrapper object by /// locating the built-in type whose dict holds this exact descriptor /// (CPython stores the owner in the descriptor itself; we recover it @@ -4818,9 +4859,15 @@ fn install_exception_str_repr(base_exception: &Rc) { pub fn make_exception_with_class(class: Rc, message: impl Into) -> Object { use crate::types::PyInstance; - let is_syntax = is_subclass_by_name(&class, "SyntaxError"); - let is_stop_iteration = is_subclass_by_name(&class, "StopIteration"); - let is_import = is_subclass_by_name(&class, "ImportError"); + let (mut is_syntax, mut is_stop_iteration, mut is_import) = (false, false, false); + for t in class.mro.borrow().iter() { + match t.name.as_str() { + "SyntaxError" => is_syntax = true, + "StopIteration" => is_stop_iteration = true, + "ImportError" => is_import = true, + _ => {} + } + } let inst = PyInstance::new(class); let msg = Object::from_str(message); // A messageless raise (`StopIteration()`, `GeneratorExit()`, …) diff --git a/crates/weavepy-vm/src/builtins.rs b/crates/weavepy-vm/src/builtins.rs index d46b4ca1..a71e7de9 100644 --- a/crates/weavepy-vm/src/builtins.rs +++ b/crates/weavepy-vm/src/builtins.rs @@ -2655,7 +2655,7 @@ fn slot_sizeof(args: &[Object]) -> Result { return Ok(Object::Int(28 + 4 * ndigits)); } } - 16 + 8 * inst.dict.get().map_or(0, |dict| dict.borrow().len()) as i64 + 16 + 8 * inst.attr_count() as i64 } // CPython's compact-unicode layout (test_str.test_raiseMemError): // ASCII is a 40-byte struct + len+1 one-byte units; anything wider @@ -4117,12 +4117,7 @@ fn attr_get(obj: &Object, name: &str) -> Option { if let Some(v) = f.slot(name) { return Some(v); } - } else if let Some(v) = f - .attrs() - .borrow() - .get(&crate::object::DictKey(Object::from_str(name))) - .cloned() - { + } else if let Some(v) = f.attr_get(name) { return Some(v); } // Synthetic dunders. Mirror `Vm::load_attr`'s function @@ -8244,17 +8239,9 @@ fn b_sorted(args: &[Object]) -> Result { while let Some(v) = it.next_value() { buf.push(v); } - let mut err: Option = None; - buf.sort_by(|a: &Object, b: &Object| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - err = Some(e); - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } + crate::timsort::sort(&mut buf, |a, b| { + crate::compare_op(a, b, weavepy_compiler::CompareKind::Lt) + })?; let obj = Object::new_list(buf); crate::gc_trace::track(obj.clone()); Ok(obj) @@ -8562,7 +8549,7 @@ pub(crate) fn make_unbound_super(class: Rc) -> Object d })) .into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -8641,7 +8628,7 @@ pub(crate) fn build_super_proxy( d })) .into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -9276,7 +9263,7 @@ pub fn ensure_hashable(obj: &Object) -> Result<(), RuntimeError> { // container lookup runs (test_import's unhashable-`__name__` // str subclass in set membership). Object::Instance(inst) => { - if matches!(inst.cls().lookup("__hash__"), Some(Object::None)) { + if inst.class_dunder(crate::types::Dunder::Hash).is_none() { return Err(type_error(format!( "unhashable type: '{}'", inst.cls().name @@ -9555,9 +9542,11 @@ pub fn b_dir(args: &[Object]) -> Result { // transplants pytest marks onto its wrapper that way, so a dir // that hid `f.pytestmark` silently dropped every // `@pytest.mark.parametrize` stacked under `@given`. - for k in f.attrs().borrow().keys() { - if let Object::Str(s) = &k.0 { - names.insert(s.to_string()); + if let Some(attrs) = f.attrs.borrow().as_ref() { + for k in attrs.borrow().keys() { + if let Object::Str(s) = &k.0 { + names.insert(s.to_string()); + } } } for n in [ @@ -10434,7 +10423,7 @@ fn b_mark_iterable_coroutine(args: &[Object]) -> Result { closure: f.closure.clone(), // Shared, not copied: `func.__dict__` mutations stay visible on // both, matching CPython where the function object is the same. - attrs: RefCell::new(f.attrs()), + attrs: RefCell::new(Some(f.attrs())), slots: RefCell::new(f.slots.borrow().clone()), closure_cells: std::sync::OnceLock::new(), // The copied slot store carries any override along. @@ -11324,6 +11313,25 @@ pub(crate) fn substr_find(hay: &str, needle: &str) -> Option { return hay.find(needle); } let last_start = h.len() - n.len(); + if n[0].is_ascii() { + // An ASCII byte is never inside a multibyte character, so every + // hit is a character boundary: scan the bytes directly (no char + // searcher) and compare the short rest inline (no `memcmp` call). + let mut i = 0; + let mut budget = h.len() / n.len() + 32; + while i <= last_start { + let at = i + memchr::memchr(n[0], &h[i..=last_start])?; + if short_bytes_eq(&h[at + 1..at + n.len()], &n[1..]) { + return Some(at); + } + budget -= 1; + if budget == 0 { + return hay[at + 1..].find(needle).map(|k| k + at + 1); + } + i = at + 1; + } + return None; + } // `str::find(char)` is memchr-backed, so the candidate scan runs at // vector width; the manual byte loop it replaced did not. let first = needle.chars().next()?; @@ -11352,6 +11360,13 @@ pub(crate) fn substr_find(hay: &str, needle: &str) -> Option { None } +/// `a == b` for the short needle tails the substring scans compare, inline +/// (a `memcmp` call costs more than comparing a few bytes). +#[inline(always)] +fn short_bytes_eq(a: &[u8], b: &[u8]) -> bool { + a.len() == b.len() && a.iter().zip(b).all(|(x, y)| x == y) +} + /// [`substr_find`] from the right. pub(crate) fn substr_rfind(hay: &str, needle: &str) -> Option { let (h, n) = (hay.as_bytes(), needle.as_bytes()); @@ -11389,6 +11404,24 @@ pub(crate) fn substr_count(hay: &str, needle: &str) -> usize { return hay.matches(needle).count(); } let last_start = h.len() - n.len(); + if n[0].is_ascii() { + // As `substr_find`'s ASCII scan (every hit is a boundary). + let mut count = 0; + let mut i = 0; + while i <= last_start { + let Some(off) = memchr::memchr(n[0], &h[i..=last_start]) else { + break; + }; + let at = i + off; + if short_bytes_eq(&h[at + 1..at + n.len()], &n[1..]) { + count += 1; + i = at + n.len(); + } else { + i = at + 1; + } + } + return count; + } let Some(first) = needle.chars().next() else { return 0; }; @@ -12411,8 +12444,8 @@ fn list_getitem(args: &[Object]) -> Result { .get(1) .ok_or_else(|| type_error("__getitem__ expected 1 argument"))?; if let Object::Slice(s) = key { - let seq = l.borrow().clone(); - return Ok(Object::new_list(crate::slice_seq(&seq, s)?)); + let sliced = crate::slice_seq(&l.borrow(), s)?; + return Ok(Object::new_list(sliced)); } let l = l.borrow(); let n = list_index_arg(l.len(), key, "__getitem__")?; @@ -12485,7 +12518,11 @@ fn list_pop(args: &[Object]) -> Result { let l = list_self(args)?; let mut l = l.borrow_mut(); let idx = if args.len() > 1 { - match &args[1] { + let index = match &args[1] { + Object::Bool(b) => Object::Int(i64::from(*b)), + other => other.clone(), + }; + match &index { Object::Int(i) => { if l.is_empty() { return Err(index_error("pop from empty list")); @@ -12873,18 +12910,9 @@ fn range_count(args: &[Object]) -> Result { fn list_sort(args: &[Object]) -> Result { let l = list_self(args)?; - let mut err: Option = None; - l.borrow_mut() - .sort_by(|a: &Object, b: &Object| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - err = Some(e); - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } + crate::timsort::sort(&mut l.borrow_mut(), |a, b| { + crate::compare_op(a, b, weavepy_compiler::CompareKind::Lt) + })?; Ok(Object::None) } @@ -12969,6 +12997,17 @@ pub(crate) fn dict_lookup( d: &Rc>, key: &Object, ) -> Result, RuntimeError> { + // A `str` or `int` key settles by native equality unless the table + // compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(key) { + if let Ok(m) = d.try_borrow() { + match m.get(&probe) { + Some(v) => return Ok(Some(v.clone())), + None if probe.miss_is_exact() => return Ok(None), + None => {} + } + } + } if crate::object::dict_key_is_reentrant(key) { return crate::object::dict_reentrant_get(d, key); } diff --git a/crates/weavepy-vm/src/descr_registry.rs b/crates/weavepy-vm/src/descr_registry.rs index 5dcc5f62..a68cef0e 100644 --- a/crates/weavepy-vm/src/descr_registry.rs +++ b/crates/weavepy-vm/src/descr_registry.rs @@ -25,7 +25,12 @@ //! failed — test_interpreters TestInterpreterCall after a prior test's //! interpreter was torn down). -use std::collections::{HashMap, HashSet}; +/// Address-keyed tables: object ids are aligned addresses, hashed with the +/// cheap address mixer rather than SipHash (they carry no untrusted input). +type HashMap = + std::collections::HashMap>; +type HashSet = + std::collections::HashSet>; use std::sync::LazyLock; use crate::object::Object; @@ -63,13 +68,13 @@ pub struct DescrMeta { /// `inspect.getattr_static` (and so `import traceback`, via `_colorize`'s /// dataclasses) in any sub-interpreter created off the main thread. static DESCR_META: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Every pointer key any table below has ever been given (and not yet /// forgotten). `Drop` consults this one set first, so the common case — /// a builtin no table knows — costs a single read-locked hash probe. static TAGGED: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); fn note_key(k: usize) { TAGGED.write().insert(k); @@ -129,7 +134,7 @@ struct StoredMeta { /// `multiprocessing.Queue` feeder thread. The `Rc` pointer key is stable for /// the process lifetime and the value is `&'static str`, so sharing is sound. static BUILTIN_MODULE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attribute `obj` (a native builtin function) to module `module`, so its /// `__module__` reports that instead of the default `"builtins"`. @@ -174,7 +179,7 @@ pub fn module_of_builtin(b: &Rc) -> Option<&'static st /// thread through the module cache, and its descriptors may be read from /// any of them. static NATIVE_DESCR_ACCESSOR: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// Type-dict entries that exist for *introspection only* (RFC 0056 WS4): /// CPython materializes every slot wrapper in `tp_dict` (`'__lt__' in @@ -190,7 +195,7 @@ static NATIVE_DESCR_ACCESSOR: LazyLock>> = /// PROCESS-GLOBAL for the same reason as [`BUILTIN_MODULE`]: the type /// singletons and their dict entries are shared across threads. static SURFACE_ONLY: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// The default-allocator `__new__` builtins (`make_default_new` / /// `make_owned_new`), by identity. Several *real* constructing builtins @@ -198,7 +203,7 @@ static SURFACE_ONLY: LazyLock>> = /// sequences…), so the instantiation path cannot key on the name alone /// now that the allocators are stored as raw builtins. static DEFAULT_NEW: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashSet::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashSet::default())); /// Tag `obj` as a default-allocator `__new__` (see [`is_default_new`]). pub fn mark_default_new(obj: &Object) { @@ -272,7 +277,7 @@ pub fn is_native_descr_accessor(b: &Rc) -> bool { /// so `pickle.dumps` failed with "it's not found as /// `_multiarray_umath._reconstruct`" (RFC 0076 WS2). static BUILTIN_WRITABLE_MODULE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Record a runtime `__module__` assignment on a builtin function. /// Returns `false` if `obj` is not a taggable representation. @@ -355,7 +360,7 @@ pub fn lookup(obj: &Object) -> Option { /// name-keyed table in `builtin_text_signature` can't reach. /// PROCESS-GLOBAL for the same reason as [`DESCR_META`]. static TEXT_SIGNATURE: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attach an Argument-Clinic `__text_signature__` string to `obj`. pub fn register_text_signature(obj: &Object, sig: &'static str) { @@ -397,7 +402,7 @@ pub type LiveDocReader = unsafe fn(usize) -> Option; /// through the shared module cache. The addresses point into the /// extension's method tables, which live for the process lifetime. static LIVE_C_DOC: LazyLock>> = - LazyLock::new(|| parking_lot::RwLock::new(HashMap::new())); + LazyLock::new(|| parking_lot::RwLock::new(HashMap::default())); /// Attach a live C-doc reader to `obj` (a bridged method descriptor). pub fn register_live_c_doc(obj: &Object, addr: usize, read: LiveDocReader) { diff --git a/crates/weavepy-vm/src/error.rs b/crates/weavepy-vm/src/error.rs index f0413556..2a1a5c75 100644 --- a/crates/weavepy-vm/src/error.rs +++ b/crates/weavepy-vm/src/error.rs @@ -11,14 +11,32 @@ use thiserror::Error; use crate::object::Object; -/// A traceback frame captured as the exception unwinds. +/// A traceback frame captured as the exception unwinds: the frame's code +/// object (shared, so a raise copies no strings) and its line. #[derive(Debug, Clone)] pub struct TracebackEntry { - pub filename: String, - pub funcname: String, + pub code: crate::sync::Rc, pub lineno: u32, } +impl TracebackEntry { + pub fn filename(&self) -> &str { + &self.code.filename + } + + pub fn funcname(&self) -> &str { + &self.code.name + } +} + +impl PartialEq for TracebackEntry { + fn eq(&self, other: &Self) -> bool { + self.lineno == other.lineno + && self.filename() == other.filename() + && self.funcname() == other.funcname() + } +} + /// A Python-visible exception. The wrapped [`Object`] is always an /// `Object::Instance` whose class's MRO contains `BaseException`. #[derive(Debug, Clone)] diff --git a/crates/weavepy-vm/src/fasthash.rs b/crates/weavepy-vm/src/fasthash.rs index 7153727b..ccb23a57 100644 --- a/crates/weavepy-vm/src/fasthash.rs +++ b/crates/weavepy-vm/src/fasthash.rs @@ -124,6 +124,11 @@ impl Hasher for ObjectIdHasher { fn write_u64(&mut self, word: u64) { self.0.write_u64(word); } + + #[inline] + fn write_usize(&mut self, word: usize) { + self.0.write_usize(word); + } } /// `BuildHasher` for [`FxHasher`] — usable as the `S` parameter of diff --git a/crates/weavepy-vm/src/frozen_code_cache.rs b/crates/weavepy-vm/src/frozen_code_cache.rs index af0c2f1e..6d083993 100644 --- a/crates/weavepy-vm/src/frozen_code_cache.rs +++ b/crates/weavepy-vm/src/frozen_code_cache.rs @@ -43,10 +43,6 @@ use std::sync::OnceLock; use weavepy_compiler::CodeObject; -use crate::object::Object; -use crate::stdlib::marshal_mod; -use crate::sync::Rc; - thread_local! { static CACHE: RefCell> = RefCell::new(HashMap::new()); } @@ -95,27 +91,38 @@ pub fn insert(name: &str, code: &CodeObject) { // persists the marshalled `CodeObject`s in a per-user cache directory // so warm process starts skip parse + compile entirely. // -// Artifact layout: `/weavepy/frozen-/` with -// a 20-byte header — magic `WPYF`, reserved flags word, source length, -// and an FNV-1a 64 source hash — followed by `marshal.dumps(code)`. -// The `CACHE_TAG` in the directory name invalidates on bytecode-format -// revisions (same lever as `.pyc`); the length + hash pair invalidates +// Artifact layout: `/weavepy/frozen--n/` +// with a 20-byte header — magic `WPYN`, reserved flags word, source +// length, and a 64-bit source hash — followed by the code in +// `weavepy_compiler::native_code` form, which loads without the CPython +// bytecode transcoding a `marshal` payload needs. The `CACHE_TAG` and +// native-format `VERSION` in the directory name invalidate on bytecode or +// layout revisions (and keep binaries of different formats from +// rewriting each other's artifacts); the length + hash pair invalidates // when the embedded source itself changes (a rebuilt binary with edited // stdlib). Corrupt or mismatched artifacts are treated as misses. /// Header magic for frozen-cache artifacts (distinct from `.pyc`'s -/// CPython magic — these files are WeavePy-internal). -const FROZEN_MAGIC: &[u8; 4] = b"WPYF"; +/// CPython magic: these files are WeavePy-internal). +const FROZEN_MAGIC: &[u8; 4] = b"WPYN"; const FROZEN_HEADER_LEN: usize = 20; -/// FNV-1a 64-bit — tiny, dependency-free, and plenty for cache -/// validation (collisions only matter combined with an equal length). +/// FNV-1a 64-bit over 8-byte words (the tail zero-padded): tiny, +/// dependency-free, and plenty for cache validation (collisions only +/// matter combined with an equal length). fn fnv1a(s: &str) -> u64 { let mut h: u64 = 0xcbf2_9ce4_8422_2325; - for b in s.bytes() { - h ^= u64::from(b); + let (words, rest) = s.as_bytes().as_chunks::<8>(); + let mut mix = |w: u64| { + h ^= w; h = h.wrapping_mul(0x0000_0100_0000_01b3); + }; + for w in words { + mix(u64::from_le_bytes(*w)); } + let mut tail = [0u8; 8]; + tail[..rest.len()].copy_from_slice(rest); + mix(u64::from_le_bytes(tail)); h } @@ -144,10 +151,11 @@ fn disk_dir() -> Option<&'static PathBuf> { base } }; - Some( - base.join("weavepy") - .join(format!("frozen-{}", crate::pycache::CACHE_TAG)), - ) + Some(base.join("weavepy").join(format!( + "frozen-{}-n{}", + crate::pycache::CACHE_TAG, + weavepy_compiler::native_code::VERSION + ))) }) .as_ref() } @@ -172,17 +180,11 @@ pub fn get_disk(name: &str, source: &str, filename: &str) -> Option if len as usize != source.len() || hash != fnv1a(source) { return None; } - match marshal_mod::load_from_bytes(&bytes[FROZEN_HEADER_LEN..]).ok()? { - Object::Code(c) => { - let mut code = crate::pycache::own_decoded_code(c); - if code.filename != filename { - crate::pycache::rewrite_filenames(&mut code, filename); - } - insert(name, &code); - Some(code) - } - _ => None, - } + // Not mirrored into the in-memory cache: a later interpreter reads the + // same artifact again, and a resident clone of every loaded module's + // code would double its footprint in the (usual) single-interpreter + // process. + weavepy_compiler::native_code::decode(&bytes[FROZEN_HEADER_LEN..], filename) } /// Persist a freshly-compiled frozen module to the disk cache. @@ -195,15 +197,15 @@ pub fn write_disk(name: &str, source: &str, code: &CodeObject) { if let Some(parent) = path.parent() { let _ = std::fs::create_dir_all(parent); } - let mut bytes = Vec::with_capacity(FROZEN_HEADER_LEN + 4096); + // A code object the native form doesn't carry just isn't cached. + let Some(payload) = weavepy_compiler::native_code::encode(code) else { + return; + }; + let mut bytes = Vec::with_capacity(FROZEN_HEADER_LEN + payload.len()); bytes.extend_from_slice(FROZEN_MAGIC); bytes.extend_from_slice(&0u32.to_le_bytes()); bytes.extend_from_slice(&(source.len() as u32).to_le_bytes()); bytes.extend_from_slice(&fnv1a(source).to_le_bytes()); - let Ok(Object::Bytes(payload)) = marshal_mod::b_dumps(&[Object::Code(Rc::new(code.clone()))]) - else { - return; - }; bytes.extend_from_slice(&payload); // Atomic-ish: temp + rename so concurrent starts never observe a // half-written artifact. diff --git a/crates/weavepy-vm/src/gc_trace.rs b/crates/weavepy-vm/src/gc_trace.rs index c9d0ae14..7dda0326 100644 --- a/crates/weavepy-vm/src/gc_trace.rs +++ b/crates/weavepy-vm/src/gc_trace.rs @@ -32,7 +32,7 @@ //! A type's flags decide whether tracking is needed at //! construction time. //! - The **eval breaker** triggers a collection when the -//! generation-0 counter exceeds the threshold (default 700). +//! generation-0 counter exceeds the threshold (default 2000, as in CPython 3.14). //! Collections also happen on explicit `gc.collect()`. //! //! Today's implementation is *non-incremental*: a full @@ -73,28 +73,34 @@ use crate::fasthash::ObjectIdHasher; use crate::shared_value::ThinArc; +use crate::sync::Rc as HandleRc; use crate::sync::RefCell; use std::hash::BuildHasherDefault; use std::sync::atomic::{ AtomicBool, AtomicI64, AtomicU32, AtomicU64, AtomicU8, AtomicUsize, Ordering, }; -use std::sync::Arc; use crate::object::Object; use crate::weakref_registry::{id_of, ObjectId}; -type GcIndex = - std::collections::HashMap, BuildHasherDefault>; +/// A set of object ids (addresses), hashed with the address mixer. +type IdSet = std::collections::HashSet>; + +type GcIndex = std::collections::HashMap< + ObjectId, + HandleRc, + BuildHasherDefault, +>; /// The standard CPython generation count (3) and default -/// thresholds: gen 0 collects when 700 untracked allocations +/// thresholds (CPython 3.14's): gen 0 collects when 2000 net tracked allocations /// have happened; gen 1 every 10 gen 0 collections; gen 2 /// every 10 gen 1 collections. pub const N_GENERATIONS: usize = 3; // Color and generation are bounded states; reference counts remain full-width. const _: () = assert!(N_GENERATIONS > 0 && N_GENERATIONS <= u8::MAX as usize + 1); const MAX_GENERATION: u8 = (N_GENERATIONS - 1) as u8; -pub const DEFAULT_THRESHOLDS: [usize; N_GENERATIONS] = [700, 10, 10]; +pub const DEFAULT_THRESHOLDS: [usize; N_GENERATIONS] = [2000, 10, 10]; /// Upper bound on the number of mark-sweep passes a single /// [`GcState::collect`] runs to reach a fixpoint. Convergence is normally 2–3 @@ -316,7 +322,7 @@ impl TrackedHandle { /// in from the end, and its `slot` field is corrected here so the /// per-handle position invariant holds after the call. #[inline] -fn swap_remove_handle(vec: &mut Vec>, slot: usize) { +fn swap_remove_handle(vec: &mut Vec>, slot: usize) { if slot >= vec.len() { return; } @@ -333,8 +339,11 @@ fn swap_remove_handle(vec: &mut Vec>, slot: usize) { /// handle was found and removed. O(n) in the generation length, but only ever /// taken on the rare stale-cache path — the common case stays O(1). #[inline] -fn remove_handle_by_ptr(vec: &mut Vec>, handle: &Arc) -> bool { - if let Some(pos) = vec.iter().position(|h| Arc::ptr_eq(h, handle)) { +fn remove_handle_by_ptr( + vec: &mut Vec>, + handle: &HandleRc, +) -> bool { + if let Some(pos) = vec.iter().position(|h| HandleRc::ptr_eq(h, handle)) { swap_remove_handle(vec, pos); true } else { @@ -347,7 +356,7 @@ struct Generation { /// All tracked handles in this generation. Append-only /// during normal allocation; rewritten in place when /// objects are promoted or moved to the unreachable list. - handles: Vec>, + handles: Vec>, } #[derive(Debug, Default, Clone, Copy)] @@ -430,7 +439,7 @@ pub struct GcState { /// Frozen handles. `gc.freeze()` moves all tracked objects /// here; they are skipped by future collections until /// `gc.unfreeze()` runs. - frozen: RefCell>>, + frozen: RefCell>>, /// `gc.garbage` — uncollectable objects (cycles whose /// finalisers refused to release). pub garbage: RefCell>, @@ -461,7 +470,7 @@ pub struct GcState { /// path by scanning *this* small set (not the whole tracked /// population) at the interpreter's reference-drop safe points. Keyed /// by id like `index`; an object is in both while finalizable. - finalizable: RefCell>>, + finalizable: RefCell>>, /// Live population of [`Self::finalizable`]. A relaxed load of this /// atomic is the gate the interpreter checks before every prompt- /// finalization sweep: when it is zero (the overwhelmingly common @@ -484,7 +493,7 @@ pub struct GcState { /// [`Self::finalizable`], so the per-drop scan walks only entries /// that could plausibly die this safe point. Membership is mirrored /// by each handle's `fin_cold` flag and `fin_hot_slot` index. - finalizable_hot: RefCell>>, + finalizable_hot: RefCell>>, /// `finalizable_hot.len()`, published for the lock-free gate: when /// zero, a scan that isn't on the cold stride returns without /// borrowing anything — the steady state of a program whose @@ -511,7 +520,7 @@ impl Drop for GcState { // first so the chains are already severed when the handle // vectors drop. Safe at this point: the thread is exiting, no // Python code will observe the cleared objects. - let mut handles: Vec> = Vec::new(); + let mut handles: Vec> = Vec::new(); if let Ok(gens) = self.generations.try_borrow() { for g in gens.iter() { handles.extend(g.handles.iter().cloned()); @@ -582,7 +591,7 @@ impl GcState { /// Insert `h` into the finalizable index (RFC 0077 WS2: one place /// for the population and hot-set bookkeeping). New entries start /// hot; the next scan grades them. Returns whether it was new. - fn fin_insert(&self, id: ObjectId, h: Arc) -> bool { + fn fin_insert(&self, id: ObjectId, h: HandleRc) -> bool { let mut fin = self.finalizable.borrow_mut(); if fin.contains_key(&id) { return false; @@ -601,6 +610,12 @@ impl GcState { /// Remove `id` from the finalizable index, if present, keeping the /// population and hot set exact. fn fin_remove(&self, id: ObjectId) -> bool { + // The usual case: nothing finalizable is tracked at all. (Inserts + // count after they land, and the same thread removes an id it + // inserted.) + if self.finalizable_count.load(Ordering::Acquire) == 0 { + return false; + } let removed = self.finalizable.borrow_mut().remove(&id); let Some(h) = removed else { return false; @@ -611,7 +626,7 @@ impl GcState { /// The hot-set and population half of [`Self::fin_remove`], for the /// one caller that already holds the finalizable lock. - fn fin_remove_bookkeeping(&self, h: &Arc) { + fn fin_remove_bookkeeping(&self, h: &HandleRc) { self.fin_make_cold(h); if self.finalizable_count.fetch_sub(1, Ordering::AcqRel) == 1 { // RFC 0065 (WS1): population reached zero — dispatch loops @@ -621,7 +636,7 @@ impl GcState { } /// Move `h` into the hot set (no-op if already hot). - fn fin_make_hot(&self, h: &Arc) { + fn fin_make_hot(&self, h: &HandleRc) { let mut hot = self.finalizable_hot.borrow_mut(); if !h.fin_cold.swap(false, Ordering::AcqRel) && h.fin_hot_slot.load(Ordering::Relaxed) != usize::MAX @@ -636,19 +651,19 @@ impl GcState { /// Move `h` out of the hot set (no-op if already cold). O(1) via /// the handle's cached slot, with the swap-remove fixup. - fn fin_make_cold(&self, h: &Arc) { + fn fin_make_cold(&self, h: &HandleRc) { let mut hot = self.finalizable_hot.borrow_mut(); h.fin_cold.store(true, Ordering::Release); let slot = h.fin_hot_slot.swap(usize::MAX, Ordering::AcqRel); if slot == usize::MAX { return; } - if slot < hot.len() && Arc::ptr_eq(&hot[slot], h) { + if slot < hot.len() && HandleRc::ptr_eq(&hot[slot], h) { hot.swap_remove(slot); if let Some(moved) = hot.get(slot) { moved.fin_hot_slot.store(slot, Ordering::Relaxed); } - } else if let Some(pos) = hot.iter().position(|x| Arc::ptr_eq(x, h)) { + } else if let Some(pos) = hot.iter().position(|x| HandleRc::ptr_eq(x, h)) { hot.swap_remove(pos); if let Some(moved) = hot.get(pos) { moved.fin_hot_slot.store(pos, Ordering::Relaxed); @@ -760,6 +775,16 @@ impl GcState { // queue onto a list we cannot touch. return false; }; + // The churn shape — a loop that builds a container and drops + // the previous one — leaves its dead predecessors just below + // the tail: reclaim them here, so steady churn neither grows + // the list nor parks dead allocations until a sweep. + let n = deferred.len(); + for i in (n.saturating_sub(2)..n).rev() { + if deferred[i].is_dead() { + deferred.swap_remove(i); + } + } deferred.push(weak); deferred.len() >= self.deferred_limit.load(Ordering::Relaxed) }; @@ -835,7 +860,7 @@ impl GcState { // insert becomes observable (we hold the index borrow, so // no prober can race past a fresh registration). self.tracked_filter.insert(new_id); - let handle = Arc::new(TrackedHandle::new(obj, 0)); + let handle = HandleRc::new(TrackedHandle::new(obj, 0)); entry.insert(handle.clone()); // Enroll finalizable objects in the dedicated prompt-finalization // index so the per-safe-point sweep scans only them, not the whole @@ -867,8 +892,8 @@ impl GcState { if !self.finalized_ids.borrow().is_empty() { self.finalized_ids.borrow_mut().remove(&new_id); } - self.tracked_count.fetch_add(1, Ordering::AcqRel); - self.tracked_version.fetch_add(1, Ordering::AcqRel); + serial_add(&self.tracked_count, 1); + serial_add(&self.tracked_version, 1); self.note_gen0_alloc(); } @@ -927,7 +952,10 @@ impl GcState { if handle.color.load(Ordering::Acquire) == color::Frozen { let mut frozen = self.frozen.borrow_mut(); let slot = handle.slot.load(Ordering::Acquire); - if frozen.get(slot).is_some_and(|h| Arc::ptr_eq(h, &handle)) { + if frozen + .get(slot) + .is_some_and(|h| HandleRc::ptr_eq(h, &handle)) + { swap_remove_handle(&mut frozen, slot); } else { remove_handle_by_ptr(&mut frozen, &handle); @@ -944,7 +972,7 @@ impl GcState { if gens[g] .handles .get(slot) - .is_some_and(|h| Arc::ptr_eq(h, &handle)) + .is_some_and(|h| HandleRc::ptr_eq(h, &handle)) { swap_remove_handle(&mut gens[g].handles, slot); } else if !remove_handle_by_ptr(&mut gens[g].handles, &handle) { @@ -958,8 +986,8 @@ impl GcState { } } } - self.tracked_count.fetch_sub(1, Ordering::AcqRel); - self.tracked_version.fetch_add(1, Ordering::AcqRel); + serial_add(&self.tracked_count, usize::MAX); + serial_add(&self.tracked_version, 1); } /// Reclaim every tracked object on this thread whose only remaining @@ -1063,7 +1091,7 @@ impl GcState { fn collect_child_candidates(&self, obj: &Object, work: &mut Vec) { let mut pending: Vec = Vec::new(); traverse_object(obj, &mut |child| pending.push(child.clone())); - let mut seen: std::collections::HashSet = std::collections::HashSet::new(); + let mut seen: IdSet = IdSet::default(); while let Some(child) = pending.pop() { let id = id_of(&child); if !seen.insert(id) { @@ -1221,8 +1249,10 @@ impl GcState { Hot, Cold, } - let mut out: Vec> = Vec::new(); - let mut probe = |h: &Arc, out: &mut Vec>| -> Grade { + let mut out: Vec> = Vec::new(); + let mut probe = |h: &HandleRc, + out: &mut Vec>| + -> Grade { probed += 1; let sc = strong_count_for(&h.object); let cached = h.weak_clones.load(Ordering::Acquire); @@ -1290,11 +1320,11 @@ impl GcState { // Walk the whole index (or a rotating window of it) and // re-grade every entry; collect the transitions and apply // them after the index borrow ends. - let mut to_cold: Vec> = Vec::new(); - let mut to_hot: Vec> = Vec::new(); + let mut to_cold: Vec> = Vec::new(); + let mut to_hot: Vec> = Vec::new(); { let fin = self.finalizable.borrow(); - let mut grade = |h: &Arc| { + let mut grade = |h: &HandleRc| { let was_cold = h.fin_cold.load(Ordering::Relaxed); match probe(h, &mut out) { Grade::Cold if !was_cold => to_cold.push(h.clone()), @@ -1428,7 +1458,7 @@ impl GcState { } /// O(1) handle lookup by object id (any generation or frozen). - pub fn handle_for(&self, id: ObjectId) -> Option> { + pub fn handle_for(&self, id: ObjectId) -> Option> { // RFC 0065 (WS4): see `is_tracked`. if !self.tracked_filter.may_contain(id) { return None; @@ -1442,9 +1472,9 @@ impl GcState { /// finalizers for everything during interpreter teardown, not just /// for cyclic garbage. The per-handle `finalized` flag (shared with /// the cycle collector) guarantees each `__del__` runs at most once. - pub fn finalization_candidates(&self) -> Vec> { + pub fn finalization_candidates(&self) -> Vec> { let mut out = Vec::new(); - let pending = |h: &Arc| { + let pending = |h: &HandleRc| { !h.finalized.load(Ordering::Acquire) // A finalizer already queued by a collection (but not yet // drained) must not be listed again — the pending queue owns @@ -1901,7 +1931,7 @@ impl GcState { // are not counted as collected — they're dropped when this pass ends, // and the underlying iterator is freed by refcount once the real // objects in its (dead) cycle are cleared. - let mut temp_handles: Vec> = Vec::new(); + let mut temp_handles: Vec> = Vec::new(); { // Scan the existing candidate list, then the growing temporary // list, preserving discovery order without copying either list. @@ -2029,7 +2059,7 @@ impl GcState { if by_id.contains_key(&cid) { return; } - let handle = Arc::new(TrackedHandle::new(child.clone(), 0)); + let handle = HandleRc::new(TrackedHandle::new(child.clone(), 0)); by_id.insert(cid, handle.clone()); temp_handles.push(handle); }); @@ -2117,7 +2147,7 @@ impl GcState { } // Phase 5: white objects are unreachable cyclic garbage. - let unreachable: Vec> = candidate_set + let unreachable: Vec> = candidate_set .iter() .filter(|h| h.color.load(Ordering::Acquire) == color::White) .cloned() @@ -2183,19 +2213,17 @@ impl GcState { // (test_callbacks_on_callback: `c.wr`/`d.wr` stay silent while the // external `safe_callback` fires). Snapshot the trash ids so the // queue loops below can drop callbacks belonging to trash wrappers. - let mut trash_ids: std::collections::HashSet = - unreachable.iter().map(|h| h.id).collect(); - let wrapper_is_trash = - |slot: &Arc, - trash: &std::collections::HashSet| { - slot.py_ref - .borrow() - .as_ref() - .and_then(std::sync::Weak::upgrade) - .is_none_or(|inst| { - trash.contains(&(crate::sync::Rc::as_ptr(&inst) as usize as u64)) - }) - }; + let mut trash_ids: IdSet = unreachable.iter().map(|h| h.id).collect(); + let wrapper_is_trash = |slot: &crate::sync::Rc, + trash: &IdSet| { + slot.py_ref + .borrow() + .as_ref() + .and_then(crate::sync::Weak::upgrade) + .is_none_or(|inst| { + trash.contains(&(crate::sync::Rc::as_ptr(&inst) as usize as u64)) + }) + }; if weakref_only { let mut weakref_callbacks = Vec::new(); @@ -2217,7 +2245,7 @@ impl GcState { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(crate::sync::Weak::upgrade) .map(crate::object::Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); @@ -2261,8 +2289,8 @@ impl GcState { // wasn't resurrected falls into `dead` and is reclaimed (its weakrefs // cleared in that second pass, so single-`collect()` weakref tests // still observe `ref() is None`). - let mut deferred: Vec> = Vec::new(); - let mut maybe_dead: Vec> = Vec::new(); + let mut deferred: Vec> = Vec::new(); + let mut maybe_dead: Vec> = Vec::new(); for h in &unreachable { let pending_finalizer = has_finalizer(&h.object) && !h.finalized.load(Ordering::Acquire); @@ -2362,7 +2390,7 @@ impl GcState { // Whatever stayed White after the resurrection re-mark and the finalizer // subgraph protection is genuinely dead this pass. - let dead: Vec> = maybe_dead + let dead: Vec> = maybe_dead .into_iter() .filter(|h| h.color.load(Ordering::Acquire) == color::White) .collect(); @@ -2421,7 +2449,7 @@ impl GcState { // every other dead object has released its references, a // dict held only by dead holders is down to one owner and // the retry clears it (a live holder keeps it intact). - let mut shared_dict_holders: Vec<&Arc> = Vec::new(); + let mut shared_dict_holders: Vec<&HandleRc> = Vec::new(); for h in &dead { if !clear_object_fields(&h.object) { shared_dict_holders.push(h); @@ -2448,9 +2476,9 @@ impl GcState { // callbacks and recursing into its children. Finalizable orphans are // left for a finalizing collection so `__del__` ordering is preserved. if !saveall { - let dead_ids: std::collections::HashSet = dead.iter().map(|h| h.id).collect(); + let dead_ids: IdSet = dead.iter().map(|h| h.id).collect(); let mut worklist = cascade_seed; - let mut seen: std::collections::HashSet = std::collections::HashSet::new(); + let mut seen: IdSet = IdSet::default(); while let Some(cid) = worklist.pop() { if dead_ids.contains(&cid) || by_id.contains_key(&cid) || !seen.insert(cid) { // Dead (already reaped), a candidate this collection owns, @@ -2520,7 +2548,7 @@ impl GcState { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(crate::sync::Weak::upgrade) .map(crate::object::Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); @@ -2555,7 +2583,7 @@ impl GcState { reported } - fn snapshot_for_collection(&self, upto: usize) -> Vec> { + fn snapshot_for_collection(&self, upto: usize) -> Vec> { let gens = self.generations.borrow(); let selected = &gens[..=upto.min(N_GENERATIONS - 1)]; let mut out = Vec::with_capacity(selected.iter().map(|g| g.handles.len()).sum()); @@ -2565,7 +2593,7 @@ impl GcState { out } - fn rebuild_generations(&self, upto: usize, candidates: &[Arc]) { + fn rebuild_generations(&self, upto: usize, candidates: &[HandleRc]) { // Lock order MUST match `track` (index before generations): the // collector and a mutator thread can both reach the GC under the // shared, process-global state, and acquiring these two cells in @@ -2749,7 +2777,15 @@ pub fn traverse_object(obj: &Object, visit: &mut dyn FnMut(&Object)) { // `dict -> instance -> class -> method -> __globals__` cycle // in a dead ModuleType namespace would be immortal // (test_module.test_clear_dict_in_ref_cycle). - if let Some(dict) = i.dict.get_shared() { + if i.dict.published().is_none() { + // Split values: the instance's own children. + if let Ok(split) = i.dict.split_cell().try_borrow() { + for (k, v) in split.iter() { + visit(&k.0); + visit(v); + } + } + } else if let Some(dict) = i.dict.get_shared() { let dict_obj = Object::Dict(dict); if is_tracked(id_of(&dict_obj)) { visit(&dict_obj); @@ -2954,7 +2990,7 @@ pub fn traverse_object(obj: &Object, visit: &mut dyn FnMut(&Object)) { } } } - if let Ok(attrs_rc) = f.attrs.try_borrow() { + if let Some(attrs_rc) = f.attrs.try_borrow().ok().and_then(|a| a.clone()) { if let Ok(attrs) = attrs_rc.try_borrow() { for (k, v) in attrs.iter() { visit(&k.0); @@ -3116,12 +3152,21 @@ pub fn clear_object_fields(obj: &Object) -> bool { if let Ok(mut slots) = i.slots.try_borrow_mut() { *slots = crate::types::SlotStorage::default(); } + if i.dict.published().is_none() { + let values = i + .dict + .split_cell() + .try_borrow_mut() + .map(|mut s| s.take()) + .unwrap_or_default(); + drop(values); + } if i.dict.strong_count() > 1 { // Shared `__dict__`: leave its contents to the other // holder (see the doc comment). return false; } - if let Some(dict) = i.dict.get() { + if let Some(dict) = i.dict.published() { if let Ok(mut m) = dict.try_borrow_mut() { m.clear(); } @@ -3159,7 +3204,7 @@ pub fn clear_object_fields(obj: &Object) -> bool { // dict (a module's `__dict__` or the `exec` target), reclaimed as // its own candidate if it too is unreachable — clearing it here // could wipe a live module. - if let Ok(attrs_rc) = f.attrs.try_borrow() { + if let Some(attrs_rc) = f.attrs.try_borrow().ok().and_then(|a| a.clone()) { if let Ok(mut attrs) = attrs_rc.try_borrow_mut() { attrs.clear(); } @@ -3214,7 +3259,9 @@ fn run_finalizer(obj: &Object) { /// blocks — CPython's `gen_dealloc` behavior). fn has_finalizer(obj: &Object) -> bool { match obj { - Object::Instance(inst) => inst.cls().lookup("__del__").is_some(), + // The class's cached `__del__` verdict (reset whenever `__del__` + // or the MRO changes), not an MRO walk per tracked instance. + Object::Instance(inst) => inst.cls().instances_need_finalize(), // RFC 0065 (WS4, item 3) tried gating this on "close can run // user code" (empty exception table ⇒ skip enrollment) so // `yield`-loop workloads could reach the fully-quiet dispatch @@ -3283,7 +3330,7 @@ pub fn with_state(f: impl FnOnce(&GcState) -> R) -> R { /// well away from `getrefcount`-hot paths like pandas'. pub fn zombie_memoryview_refs_to(target: ObjectId) -> usize { with_state(|s| { - let mut handles: Vec> = Vec::new(); + let mut handles: Vec> = Vec::new(); { let Ok(gens) = s.generations.try_borrow() else { return 0; @@ -3306,13 +3353,16 @@ pub fn zombie_memoryview_refs_to(target: ObjectId) -> usize { if handles.is_empty() { return 0; } - let mut zombies: std::collections::HashSet = std::collections::HashSet::new(); + let mut zombies: IdSet = IdSet::default(); loop { // Inbound references each candidate receives from the current // zombie set (a dropped chain of sub-views keeps inner views' // counts up via exporter edges). - let mut inbound: std::collections::HashMap = - std::collections::HashMap::new(); + let mut inbound: std::collections::HashMap< + ObjectId, + usize, + BuildHasherDefault, + > = std::collections::HashMap::default(); for h in &handles { if zombies.contains(&h.id) { traverse_object(&h.object, &mut |c| { @@ -3387,6 +3437,21 @@ pub fn track(obj: Object) { with_state(|s| s.track(obj)); } +/// `counter += n` (wrapping) for a collector counter: a plain load and +/// store while the GIL serializes every writer, a locked read-modify-write +/// only in free-threaded mode. +#[inline(always)] +fn serial_add(counter: &AtomicUsize, n: usize) { + // (The debug unit-test binary runs interpreters on concurrent threads + // with no GIL between them: it keeps the locked form too.) + if cfg!(debug_assertions) || crate::gil::free_threading_enabled() { + counter.fetch_add(n, Ordering::AcqRel); + } else { + let v = counter.load(Ordering::Relaxed); + counter.store(v.wrapping_add(n), Ordering::Release); + } +} + /// Deferred instance tracking. /// /// CPython tracks every instance of a Python-defined class at @@ -3548,9 +3613,9 @@ const DEFERRED_CAP: usize = 4096; /// weak `Object`, because `Object`'s payload `Arc` is what has to stay /// weak — holding the `Object` itself would pin the container alive. enum DeferredContainer { - List(std::sync::Weak>>), - Dict(std::sync::Weak>), - Set(std::sync::Weak>), + List(crate::sync::Weak>>), + Dict(crate::sync::Weak>), + Set(crate::sync::Weak>), } impl DeferredContainer { @@ -3564,6 +3629,16 @@ impl DeferredContainer { } } + /// Whether the container has died. + #[inline] + fn is_dead(&self) -> bool { + match self { + Self::List(w) => w.strong_count() == 0, + Self::Dict(w) => w.strong_count() == 0, + Self::Set(w) => w.strong_count() == 0, + } + } + /// The container, if it is still alive. fn upgrade(&self) -> Option { match self { @@ -3672,7 +3747,7 @@ pub fn maybe_auto_collect() -> bool { /// Convenience: find a tracked handle by object id (O(1) via the /// id index, which covers all generations plus the frozen set). -pub fn find_handle(id: ObjectId) -> Option> { +pub fn find_handle(id: ObjectId) -> Option> { with_state(|s| s.handle_for(id)) } @@ -3708,7 +3783,7 @@ pub fn complete_finalizer(id: ObjectId) { /// Convenience: snapshot all tracked objects with an unrun `__del__` /// in the shared GC (see [`GcState::finalization_candidates`]). -pub fn finalization_candidates() -> Vec> { +pub fn finalization_candidates() -> Vec> { with_state(|s| s.finalization_candidates()) } @@ -3834,13 +3909,34 @@ const SUSPECT_DORMANT_PROBES: u8 = 16; /// One enrolled suspect: its handle, remaining active probe budget, and /// (once dormant) the number of stride probes it has survived. struct Suspect { - handle: Arc, + handle: HandleRc, budget: u8, dormant_probes: u8, } -type SuspectMap = indexmap::IndexMap; -static SUSPECTS: std::sync::LazyLock> = - std::sync::LazyLock::new(|| parking_lot::Mutex::new(SuspectMap::new())); +type SuspectMap = indexmap::IndexMap>; +/// The enrolled suspects, split by phase: entries with probe budget left +/// (re-probed at every sweep) and aged-out dormant ones (re-probed only on +/// the stride), so an ordinary sweep never walks the dormant population. +#[derive(Default)] +struct Suspects { + /// Every entry has budget remaining. + active: SuspectMap, + /// Every entry's budget is spent. + dormant: SuspectMap, +} +impl Suspects { + fn len(&self) -> usize { + self.active.len() + self.dormant.len() + } + fn contains_key(&self, id: ObjectId) -> bool { + self.active.contains_key(&id) || self.dormant.contains_key(&id) + } + fn values(&self) -> impl Iterator { + self.active.values().chain(self.dormant.values()) + } +} +static SUSPECTS: std::sync::LazyLock> = + std::sync::LazyLock::new(|| parking_lot::Mutex::new(Suspects::default())); static SUSPECT_COUNT: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0); /// Entries with probe budget remaining. When only dormant entries are /// left, [`has_suspects`] admits a sweep every [`DORMANT_STRIDE`]-th @@ -3994,46 +4090,38 @@ fn residual_suspects() -> Vec<(String, usize)> { .collect() } -/// Publish the count gates after the locked suspect map changed. -/// `active` is the caller's count of budget-remaining entries (kept -/// incrementally; RFC 0077 WS2 retired the O(n) recount this used to -/// do on every enrollment and removal). -fn publish_suspect_counts(s: &SuspectMap, active: usize) { +/// Publish the count gates after the locked suspect maps changed. +fn publish_suspect_counts(s: &Suspects) { + let active = s.active.len(); let was_active = SUSPECT_ACTIVE.swap(active, Ordering::AcqRel) > 0; let was_present = SUSPECT_COUNT.swap(s.len(), Ordering::AcqRel) > 0; // RFC 0065 (WS1): the dispatch loops' quiet-path snapshots consult // `active_suspects_present` / the population gates, so a // transition of either invalidates them. Same-state churn (one // active suspect replacing another) doesn't. - if was_active != (active > 0) || was_present != (!s.is_empty()) { + if was_active != (active > 0) || was_present != (s.len() > 0) { crate::hot_gates::bump_loop_gen(); } } -/// Evict the first minimum-budget entry, preserving `min_by_key`'s tie -/// order. Zero is the minimum possible budget, so scanning can stop there. -/// The caller holds the map lock and its exact active count throughout. -fn evict_lowest_budget_suspect(s: &mut SuspectMap, active: &mut usize) -> bool { - let Some((_, first)) = s.get_index(0) else { +/// Evict the entry that has had the most chances to die: the oldest +/// dormant one, or with none, the first lowest-budget active one. The +/// caller holds the maps' lock and republishes the counts. +fn evict_lowest_budget_suspect(s: &mut Suspects) -> bool { + if !s.dormant.is_empty() { + s.dormant.shift_remove_index(0); + return true; + } + let Some(victim) = s + .active + .values() + .enumerate() + .min_by_key(|(_, entry)| entry.budget) + .map(|(index, _)| index) + else { return false; }; - let mut victim = 0; - let mut budget = first.budget; - if budget > 0 { - for (index, entry) in s.values().enumerate().skip(1) { - if entry.budget < budget { - victim = index; - budget = entry.budget; - if budget == 0 { - break; - } - } - } - } - let (_, removed) = s - .swap_remove_index(victim) - .expect("selected suspect exists"); - *active = active.saturating_sub(usize::from(removed.budget > 0)); + s.active.swap_remove_index(victim); true } @@ -4052,13 +4140,13 @@ pub fn active_suspects_present() -> bool { /// Enroll a cascade-skipped tracked object for later deadness re-probes. /// Deduplicated; silently dropped when the list is full (the next full /// collection reclaims it instead). -pub fn note_suspect(h: Arc) { +pub fn note_suspect(h: HandleRc) { static NO_SUSPECTS: std::sync::OnceLock = std::sync::OnceLock::new(); if *NO_SUSPECTS.get_or_init(|| std::env::var_os("WEAVEPY_NO_SUSPECTS").is_some()) { return; } let mut s = SUSPECTS.lock(); - if s.contains_key(&h.id) { + if s.contains_key(h.id) { return; } // RFC 0065 (WS4): publish to the miss-filter before the enrollment @@ -4075,7 +4163,6 @@ pub fn note_suspect(h: Arc) { crate::weakref_registry::strong_clone_count(h.id), Ordering::Release, ); - let mut active = SUSPECT_ACTIVE.load(Ordering::Relaxed); if s.len() >= SUSPECT_CAP { // Full: evict the most-probed entry (lowest remaining budget, // dormant first) — it has had the most chances to die and is @@ -4085,12 +4172,12 @@ pub fn note_suspect(h: Arc) { // arrived after ~200 module-teardown stragglers and was never // re-probed, pinning the Timeout→Task→frame web the test_ssl // leak tests watch). - if !evict_lowest_budget_suspect(&mut s, &mut active) { + if !evict_lowest_budget_suspect(&mut s) { return; } } floor_stats::bump(&floor_stats::SUSPECT_ENROLLED, 1); - s.insert( + s.active.insert( h.id, Suspect { handle: h, @@ -4098,7 +4185,7 @@ pub fn note_suspect(h: Arc) { dormant_probes: 0, }, ); - publish_suspect_counts(&s, active + 1); + publish_suspect_counts(&s); } /// Cheap gate for the eval loop's safe point: always sweep while an @@ -4140,14 +4227,9 @@ pub fn remove_suspect(id: ObjectId) { return; } let mut s = SUSPECTS.lock(); - if let Some(removed) = s.swap_remove(&id) { - let active = SUSPECT_ACTIVE.load(Ordering::Relaxed); - let active = if removed.budget > 0 { - active.saturating_sub(1) - } else { - active - }; - publish_suspect_counts(&s, active); + let removed = s.active.swap_remove(&id).is_some() || s.dormant.swap_remove(&id).is_some(); + if removed { + publish_suspect_counts(&s); } } @@ -4160,24 +4242,21 @@ pub fn take_dead_suspects() -> Vec { let mut s = SUSPECTS.lock(); // With only dormant entries left, `has_suspects` already stride-gated // this sweep; with actives present the stride ticks here instead. - let probe_dormant = SUSPECT_ACTIVE.load(Ordering::Relaxed) == 0 + let probe_dormant = s.active.is_empty() || SUSPECT_TICK .fetch_add(1, Ordering::Relaxed) .is_multiple_of(DORMANT_STRIDE); floor_stats::bump(&floor_stats::SUSPECT_SWEEPS, 1); - let mut active = 0usize; let mut probed = 0u64; - s.retain(|_, e| { - // Dormant (aged-out) entries only pay on the stride tick. - if e.budget == 0 && !probe_dormant { - return true; - } + // A suspect whose last reference is gone (beyond the handle and any + // weakref strong clones) goes to `out`. Returns whether it stays. + let mut dead = |e: &Suspect, out: &mut Vec| -> Option { probed += 1; let h = &e.handle; // RFC 0061 (WS1b): the handle self-identifies as reclaimed (set // in lock-step with every index removal) — no registry lookup. if h.untracked.load(Ordering::Acquire) { - return false; // already reclaimed elsewhere + return None; // already reclaimed elsewhere } // Fast reject via the cached weakref-clone upper bound (refreshed // at enrollment): more strong refs than the handle plus every @@ -4190,31 +4269,52 @@ pub fn take_dead_suspects() -> Vec { let weak = crate::weakref_registry::strong_clone_count(h.id); if sc <= 1 + weak { out.push(h.object.clone()); - return false; + return None; } } + Some(sc.saturating_sub(1 + cached)) + }; + // Active entries spend a probe; one whose budget runs out turns + // dormant after this sweep (it isn't stride-probed in the same one). + let mut aged = Vec::new(); + s.active.retain(|&id, e| { + if dead(e, &mut out).is_none() { + return false; + } + e.budget -= 1; + if e.budget == 0 { + aged.push(( + id, + Suspect { + handle: e.handle.clone(), + budget: 0, + dormant_probes: e.dormant_probes, + }, + )); + return false; + } + true + }); + if probe_dormant { // A dormant entry still well above its dead line, or one that has // sat through its dormant allowance, is plainly alive: drop it // from the map (see `SUSPECT_LIVE_MARGIN` / `SUSPECT_DORMANT_PROBES`). - if e.budget == 0 { + s.dormant.retain(|_, e| { + let Some(excess) = dead(e, &mut out) else { + return false; + }; e.dormant_probes = e.dormant_probes.saturating_add(1); - if sc > 1 + cached + SUSPECT_LIVE_MARGIN || e.dormant_probes > SUSPECT_DORMANT_PROBES { + if excess > SUSPECT_LIVE_MARGIN || e.dormant_probes > SUSPECT_DORMANT_PROBES { floor_stats::bump(&floor_stats::SUSPECT_FORGOTTEN, 1); return false; } - return true; - } - e.budget -= 1; - if e.budget > 0 { - active += 1; - } - true - }); + true + }); + } + s.dormant.extend(aged); floor_stats::bump(&floor_stats::SUSPECT_PROBED, probed); floor_stats::bump(&floor_stats::SUSPECT_DEAD, out.len() as u64); - // Entries skipped as dormant stayed dormant; entries probed were - // recounted above, so `active` is exact. - publish_suspect_counts(&s, active); + publish_suspect_counts(&s); out } @@ -4378,6 +4478,10 @@ pub fn note_dropped_marks(obj: &crate::object::Object) -> bool { crate::sync::Rc::strong_count(f), crate::sync::Rc::as_ptr(f) as usize as u64, ), + O::Type(t) => note_dropped_counted( + crate::sync::Rc::strong_count(t), + crate::sync::Rc::as_ptr(t) as usize as u64, + ), O::Tuple(t) => note_dropped_counted( ThinArc::strong_count(t), ThinArc::as_ptr(t).cast::<()>() as usize as u64, @@ -4386,6 +4490,24 @@ pub fn note_dropped_marks(obj: &crate::object::Object) -> bool { } } +/// Does dropping this reference to `obj` leave an instance that others +/// still hold, with no weakref watching it? Past two owners only a +/// weakref-watched object can be at its dead line (see +/// [`note_dropped_counted`]), so such a drop needs neither a prompt reap +/// nor a mark: the cheap test ahead of the full grades. +#[inline(always)] +pub fn drop_survives_plainly(obj: &crate::object::Object) -> bool { + match obj { + crate::object::Object::Instance(i) => { + crate::sync::Rc::strong_count(i) > 2 + && !crate::weakref_registry::may_have_weakrefs( + crate::sync::Rc::as_ptr(i) as usize as u64 + ) + } + _ => false, + } +} + /// How many elements a dying container may hold and still be graded by /// inspection (see [`inert_death`]). Capped so the grade stays O(1) on a /// path that runs for every discarded heap value. @@ -4562,57 +4684,39 @@ mod tests { use crate::object::DictData; #[test] - fn suspect_eviction_preserves_minimum_order_counts_and_release() { - for mode in 0..6 { - let mut suspects = SuspectMap::new(); - let mut reference = Vec::new(); - let mut handles = std::collections::HashMap::new(); - for index in 0..SUSPECT_CAP { - let budget = match mode { - 0 => 0, - 1 => SUSPECT_BUDGET, - 2 => u8::from(index + 1 != SUSPECT_CAP), - 3 => ((SUSPECT_CAP - index) % 16 + 1) as u8, - 4 => ((index * 37 + 113) % 17) as u8, - _ => u8::MAX, - }; - let object = Object::List(Rc::new(RefCell::new(Vec::new()))); - let handle = Arc::new(TrackedHandle::new(object, 0)); - let id = handle.id; - handles.insert(id, Arc::downgrade(&handle)); - suspects.insert( - id, - Suspect { - handle, - budget, - dormant_probes: (index % 16) as u8, - }, - ); - reference.push((id, budget)); - } - let mut active = reference.iter().filter(|(_, b)| *b > 0).count(); - while !reference.is_empty() { - // The old policy is the oracle, including the first tied - // minimum and the order produced by each swap removal. - let victim = reference - .iter() - .enumerate() - .min_by_key(|(_, (_, budget))| *budget) - .map(|(index, _)| index) - .unwrap(); - let (removed_id, _) = reference.swap_remove(victim); - assert!(evict_lowest_budget_suspect(&mut suspects, &mut active)); - assert_eq!(active, reference.iter().filter(|(_, b)| *b > 0).count()); - let actual: Vec<_> = suspects - .iter() - .map(|(id, entry)| (*id, entry.budget)) - .collect(); - assert_eq!(actual, reference); - assert!(handles[&removed_id].upgrade().is_none()); + fn suspect_eviction_takes_dormant_then_lowest_budget_and_releases() { + let mut suspects = Suspects::default(); + let mut handles = Vec::new(); + for index in 0..8usize { + let object = Object::List(Rc::new(RefCell::new(Vec::new()))); + let handle = HandleRc::new(TrackedHandle::new(object, 0)); + handles.push((handle.id, HandleRc::downgrade(&handle))); + let budget = [0, 5, 0, 3, 9, 3, 0, 1][index]; + let entry = Suspect { + handle, + budget, + dormant_probes: 0, + }; + if budget == 0 { + suspects.dormant.insert(handles[index].0, entry); + } else { + suspects.active.insert(handles[index].0, entry); } - assert!(!evict_lowest_budget_suspect(&mut suspects, &mut active)); - assert_eq!(active, 0); } + // Dormant entries go first, oldest first. + for expected in [0, 2, 6] { + assert!(evict_lowest_budget_suspect(&mut suspects)); + assert!(!suspects.contains_key(handles[expected].0)); + assert!(handles[expected].1.upgrade().is_none()); + } + // Then active ones by budget: 1, the first 3, the other 3, 5, 9. + for expected in [7, 3, 5, 1, 4] { + assert!(evict_lowest_budget_suspect(&mut suspects)); + assert!(!suspects.contains_key(handles[expected].0)); + assert!(handles[expected].1.upgrade().is_none()); + } + assert_eq!(suspects.len(), 0); + assert!(!evict_lowest_budget_suspect(&mut suspects)); } #[test] @@ -4627,7 +4731,7 @@ mod tests { let slots: Vec<_> = roots .iter() .map(|target| { - let slot = Arc::new(WeakRefSlot::new( + let slot = Rc::new(WeakRefSlot::new( id_of(target), target.clone(), false, diff --git a/crates/weavepy-vm/src/gen_fast.rs b/crates/weavepy-vm/src/gen_fast.rs new file mode 100644 index 00000000..d2902740 --- /dev/null +++ b/crates/weavepy-vm/src/gen_fast.rs @@ -0,0 +1,754 @@ +//! Fast steps for simple generator bodies. +//! +//! Resuming a generator the ordinary way switches activations: the quiet +//! loop (or the core loop's inline resume) makes the generator's frame the +//! running one, runs to the next `yield`, and switches back, with the +//! bookkeeping every activation needs (pending-caller lists, recursion +//! guards, the dispatch loop's reload). For a body that only moves scalars +//! through locals and loops over ranges, lists, or other such generators, +//! that bookkeeping is most of the cost of each item. +//! +//! [`Interpreter::gen_fast_step`] runs such a body directly on its frame: +//! from the frame's pc (the sent value already on its stack) to the next +//! `yield`. It never raises and never runs Python code: an instruction it +//! can't finish that way (an overflow, a non-scalar operand, a `return`, an +//! instruction outside its set) ends the step *before* that instruction, +//! with the frame at an ordinary instruction boundary, and the caller +//! continues the resume in the general loop from there. A fast step is +//! therefore always a prefix of the ordinary run. +//! +//! A generator the step iterates in turn is stepped the same way +//! ([`Interpreter::gen_fast_next`]). If that inner step stops partway, the +//! inner generator is parked having consumed its sent value +//! ([`crate::Frame::sent_consumed`]), and the outer step stops before its +//! `FOR_ITER`, which the general loop then runs (resuming the inner one +//! without pushing another value). + +use weavepy_compiler::{BinOpKind, CodeObject, CompareKind, OpCode, COMPARE_OP_TO_BOOL_FLAG}; + +use crate::object::{GeneratorState, Object, PyGenerator, PyIterator}; +use crate::sync::Rc; +use crate::{FoldSink, Frame, Interpreter}; + +/// How a fast step ended. +pub(crate) enum GenStep { + /// The body yielded this value; the frame is suspended past the yield. + Yielded(Object), + /// The next instruction needs the general loop. + Bail, +} + +/// How a fast `next()` of a generator ended. +pub(crate) enum GenNext { + Yielded(Object), + /// Nothing happened: the generator is as it was. + Declined, + /// The generator consumed its sent value and advanced, then stopped + /// short of a yield; it's parked for the general loop to continue. + Partial, +} + +/// `FOR_ITER`'s outcome in a fast step. +enum ForNext { + Value(Object), + Exhausted, + Bail, +} + +/// Nested fast steps at most this deep. +const MAX_DEPTH: u8 = 8; + +/// The instructions a fast step runs (see the module docs). The scan +/// also admits the generator prologue's `RETURN_GENERATOR` and the +/// implicit PEP 479 handler, which a resume or only an exception reaches. +fn op_supported(op: OpCode) -> bool { + matches!( + op, + OpCode::Nop + | OpCode::NotTaken + | OpCode::Resume + | OpCode::PopTop + | OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::LoadFastLoadFast + | OpCode::LoadFastBorrowLoadFastBorrow + | OpCode::StoreFast + | OpCode::LoadSmallInt + | OpCode::LoadConst + | OpCode::BinaryOp + | OpCode::CompareOp + | OpCode::ToBool + | OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::JumpForward + | OpCode::JumpBackward + | OpCode::GetIter + | OpCode::ForIter + | OpCode::EndFor + | OpCode::PopIter + | OpCode::YieldValue + | OpCode::ReturnValue + | OpCode::ReturnGenerator + | OpCode::StopIterationError + | OpCode::CallIntrinsic1 + | OpCode::Reraise + ) +} + +/// Whether `code` is a plain generator whose every instruction is one a +/// fast step runs (cached in the code's extension table). +fn code_ok(code: &CodeObject) -> bool { + use std::sync::atomic::Ordering; + let Some(ext) = crate::code_vm_ext(code) else { + return false; + }; + match ext.gen_fast.load(Ordering::Relaxed) { + 1 => return false, + 2 => return true, + _ => {} + } + let ok = code.is_generator + && !code.is_coroutine + && !code.is_async_generator + && code.cellvars.is_empty() + && code.freevars.is_empty() + && code.instructions.iter().all(|i| op_supported(i.op)) + && in_bounds(code, ext.objects.len()); + ext.gen_fast + .store(if ok { 2 } else { 1 }, Ordering::Relaxed); + ok +} + +/// Whether every jump, local and constant `code`'s instructions name is +/// in range, and the last instruction can't fall through: the step then +/// reads instructions, locals and constants without per-use checks. +fn in_bounds(code: &CodeObject, nconsts: usize) -> bool { + let instrs = &code.instructions; + let (n, nvars) = (instrs.len(), code.varnames.len()); + let terminal = |op| { + matches!( + op, + OpCode::ReturnValue | OpCode::Reraise | OpCode::JumpBackward | OpCode::JumpForward + ) + }; + if instrs.last().is_none_or(|i| !terminal(i.op)) { + return false; + } + instrs.iter().enumerate().all(|(p, i)| { + let arg = i.arg as usize; + match i.op { + OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::JumpForward + | OpCode::ForIter => p + 1 + arg < n, + OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::StoreFast => arg < nvars, + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + arg >> 4 < nvars && arg & 15 < nvars + } + OpCode::LoadConst => arg < nconsts, + _ => true, + } + }) +} + +/// A machine scalar: its release owes nothing. +#[inline(always)] +fn scalar(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) +} + +/// A value's copy for the operand stack: scalars inline, anything else a +/// new reference. +#[inline(always)] +fn copy(v: &Object) -> Object { + match v { + Object::Int(x) => Object::Int(*x), + Object::Float(x) => Object::Float(*x), + Object::Bool(x) => Object::Bool(*x), + Object::None => Object::None, + other => crate::clone_hot(other), + } +} + +/// The result of `a b` for machine scalars, when it can't raise or +/// leave the machine range. +#[inline(always)] +fn scalar_binop(kind: BinOpKind, a: &Object, b: &Object) -> Option { + Some(match (a, b) { + (Object::Int(a), Object::Int(b)) => { + let (a, b) = (*a, *b); + Object::Int(match kind { + BinOpKind::Add => a.checked_add(b)?, + BinOpKind::Sub => a.checked_sub(b)?, + BinOpKind::Mult => a.checked_mul(b)?, + BinOpKind::BitAnd => a & b, + BinOpKind::BitOr => a | b, + BinOpKind::BitXor => a ^ b, + BinOpKind::FloorDiv if b != 0 && !(a == i64::MIN && b == -1) => { + a.div_euclid(b) - i64::from(b < 0 && a.rem_euclid(b) != 0) + } + BinOpKind::Mod if b != 0 => { + let r = a.checked_rem(b)?; + if r != 0 && (r < 0) != (b < 0) { + r + b + } else { + r + } + } + BinOpKind::RShift if (0..64).contains(&b) => a >> b, + BinOpKind::LShift if (0..63).contains(&b) => { + let r = a.checked_shl(b as u32)?; + if r >> b != a { + return None; + } + r + } + _ => return None, + }) + } + (Object::Float(_) | Object::Int(_), Object::Float(_) | Object::Int(_)) => { + let f = |o: &Object| match o { + Object::Float(x) => Some(*x), + // Exactly representable ints only (Python converts exactly + // or raises for huge ones; both stay on the general path). + Object::Int(i) if i.unsigned_abs() < (1 << 53) => Some(*i as f64), + _ => None, + }; + let (a, b) = (f(a)?, f(b)?); + Object::Float(match kind { + BinOpKind::Add => a + b, + BinOpKind::Sub => a - b, + BinOpKind::Mult => a * b, + BinOpKind::Div if b != 0.0 => a / b, + _ => return None, + }) + } + _ => return None, + }) +} + +/// `a b` for machine scalars (NaN and mixed shapes decline). +#[inline(always)] +fn scalar_compare(kind: CompareKind, a: &Object, b: &Object) -> Option { + let ord = match (a, b) { + (Object::Int(a), Object::Int(b)) => a.cmp(b), + (Object::Float(a), Object::Float(b)) => a.partial_cmp(b)?, + _ => return None, + }; + Some(match kind { + CompareKind::Lt => ord.is_lt(), + CompareKind::LtE => ord.is_le(), + CompareKind::Eq => ord.is_eq(), + CompareKind::NotEq => ord.is_ne(), + CompareKind::Gt => ord.is_gt(), + CompareKind::GtE => ord.is_ge(), + }) +} + +/// The truth of a scalar (`None` for anything else). +#[inline(always)] +fn scalar_truth(v: &Object) -> Option { + Some(match v { + Object::Bool(b) => *b, + Object::Int(i) => *i != 0, + Object::None => false, + _ => return None, + }) +} + +impl Interpreter { + /// Whether `frame`, a suspended generator's, may take fast steps: a + /// fast-step body the general machinery holds no extra state for (no + /// Python-visible frame, no saved exception state, no parked native + /// activation). + #[inline] + pub(crate) fn gen_fast_frame_ok(frame: &Frame) -> bool { + frame.py_frame.is_none() + && frame.saved_exc_info.is_empty() + && frame.pc != 0 + && !frame.shell_cache.as_ref().is_some_and(|c| { + c.has_materialized + .load(std::sync::atomic::Ordering::Relaxed) + }) + && { + #[cfg(feature = "jit")] + { + frame.parked_native.is_none() + } + #[cfg(not(feature = "jit"))] + { + true + } + } + && code_ok(&frame.code) + } + + /// Run `frame` from its pc to its next `yield` (see the module docs). + /// `snap_gen` is the caller's quiet-loop generation; `depth` counts + /// the fast steps this one is nested in. + /// + /// The loop works on raw views of the frame's operand stack and + /// locals (as the core loop does), writing the stack length and pc + /// back when it ends. + /// + /// With `fold` (a draining consumer's sink, top level only), a yield + /// the sink takes resumes the body at once with `None` sent. + pub(crate) fn gen_fast_step( + &mut self, + frame: &mut Frame, + snap_gen: u64, + depth: u8, + fold: Option, + ) -> GenStep { + // SAFETY: the frame's code is immutable and outlives the step (the + // frame holds it, and nothing here replaces it). + let code: &CodeObject = unsafe { &*Rc::as_ptr(&frame.code) }; + let (instrs, ninstrs) = (code.instructions.as_ptr(), code.instructions.len()); + let Some(ext) = crate::code_vm_ext(code) else { + return GenStep::Bail; + }; + let cbase = ext.objects.as_ptr(); + // SAFETY: no guard is live on the locals (`peek_mut` checks), and + // nothing below runs code that could reach them. + let Some(locals) = (unsafe { frame.locals.peek_mut() }) else { + return GenStep::Bail; + }; + // (The scan checked every local index against the code's + // variables, which the frame's locals cover.) + if locals.len() < code.varnames.len() { + return GenStep::Bail; + } + let lbase = locals.as_mut_ptr(); + let stack = &mut frame.stack; + if stack.capacity() - stack.len() < 8 { + stack.reserve(8); + } + let (base, cap) = (stack.as_mut_ptr(), stack.capacity()); + let mut len = stack.len(); + let mut pc = frame.pc as usize; + if pc >= ninstrs { + return GenStep::Bail; + } + // SAFETY (throughout): `base` indexes only below `len` (initialized) + // or `cap` as checked; `lbase`, `cbase` and `instrs` only at the + // indices and pcs the eligibility scan proved in range (`in_bounds`: + // every jump lands on an instruction and the last can't fall + // through, so `pc < ninstrs` whenever an instruction is read). + let yielded = loop { + let ins = unsafe { *instrs.add(pc) }; + match ins.op { + OpCode::Nop | OpCode::NotTaken | OpCode::Resume => pc += 1, + OpCode::PopTop => { + if len == 0 { + break None; + } + let top = unsafe { &*base.add(len - 1) }; + if !scalar(top) { + if !Self::core_droppable(top) { + break None; + } + crate::drop_hot(unsafe { base.add(len - 1).read() }); + } + len -= 1; + pc += 1; + } + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { + let i = ins.arg as usize; + if len == cap { + break None; + } + let v = unsafe { &*lbase.add(i) }; + if matches!(v, Object::Unbound | Object::Cell(_)) { + break None; + } + unsafe { base.add(len).write(copy(v)) }; + len += 1; + pc += 1; + } + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + let (i, j) = ((ins.arg >> 4) as usize, (ins.arg & 15) as usize); + if len + 2 > cap { + break None; + } + let (a, b) = unsafe { (&*lbase.add(i), &*lbase.add(j)) }; + if matches!(a, Object::Unbound | Object::Cell(_)) + || matches!(b, Object::Unbound | Object::Cell(_)) + { + break None; + } + unsafe { + base.add(len).write(copy(a)); + base.add(len + 1).write(copy(b)); + } + len += 2; + pc += 1; + } + OpCode::StoreFast => { + let i = ins.arg as usize; + if len == 0 { + break None; + } + let slot = unsafe { lbase.add(i) }; + let old = unsafe { &*slot }; + if matches!(unsafe { &*base.add(len - 1) }, Object::Cell(_)) { + break None; + } + if scalar(old) || matches!(old, Object::Unbound) { + // Nothing to release. + len -= 1; + unsafe { slot.write(base.add(len).read()) }; + } else { + if !Self::core_droppable(old) { + break None; + } + len -= 1; + unsafe { crate::drop_hot(std::ptr::replace(slot, base.add(len).read())) }; + } + pc += 1; + } + OpCode::LoadSmallInt => { + if len == cap { + break None; + } + unsafe { base.add(len).write(Object::Int(i64::from(ins.arg))) }; + len += 1; + pc += 1; + } + OpCode::LoadConst => { + if len == cap { + break None; + } + unsafe { base.add(len).write(copy(&*cbase.add(ins.arg as usize))) }; + len += 1; + pc += 1; + } + OpCode::BinaryOp => { + if len < 2 { + break None; + } + // SAFETY: `BinOpKind` is `repr(u8)` and the compiler only + // emits valid kinds (as the core loop's arm). + let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; + let (a, b) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + let Some(r) = scalar_binop(kind, a, b) else { + break None; + }; + // Both operands are scalars (no drop owed). + len -= 1; + unsafe { base.add(len - 1).write(r) }; + pc += 1; + } + OpCode::CompareOp => { + if len < 2 { + break None; + } + let kind = match ins.arg & !COMPARE_OP_TO_BOOL_FLAG { + x if x == CompareKind::Lt as u32 => CompareKind::Lt, + x if x == CompareKind::LtE as u32 => CompareKind::LtE, + x if x == CompareKind::Eq as u32 => CompareKind::Eq, + x if x == CompareKind::NotEq as u32 => CompareKind::NotEq, + x if x == CompareKind::Gt as u32 => CompareKind::Gt, + x if x == CompareKind::GtE as u32 => CompareKind::GtE, + _ => break None, + }; + let (a, b) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + let Some(r) = scalar_compare(kind, a, b) else { + break None; + }; + len -= 1; + unsafe { base.add(len - 1).write(Object::Bool(r)) }; + pc += 1; + } + OpCode::ToBool => { + let Some(t) = (len > 0) + .then(|| unsafe { &*base.add(len - 1) }) + .and_then(scalar_truth) + else { + break None; + }; + unsafe { base.add(len - 1).write(Object::Bool(t)) }; + pc += 1; + } + OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue => { + let Some(t) = (len > 0) + .then(|| unsafe { &*base.add(len - 1) }) + .and_then(scalar_truth) + else { + break None; + }; + len -= 1; + pc += 1; + if t == (ins.op == OpCode::PopJumpIfTrue) { + pc += ins.arg as usize; + } + } + OpCode::JumpForward => pc += 1 + ins.arg as usize, + OpCode::JumpBackward => { + // The back edge is the eval-breaker (as the core loop's). + if self.gil_countdown <= 1 || crate::hot_gates::loop_gen() != snap_gen { + break None; + } + self.gil_countdown -= 1; + pc = (pc + 1).saturating_sub(ins.arg as usize); + } + // `iter()` of an iterator or a generator is itself. + OpCode::GetIter => { + if len == 0 + || !matches!( + unsafe { &*base.add(len - 1) }, + Object::Iter(_) | Object::Generator(_) + ) + { + break None; + } + pc += 1; + } + OpCode::ForIter => { + if len == 0 || len == cap { + break None; + } + let top = unsafe { &*base.add(len - 1) }; + // A live range's next value in line (the commonest loop). + if let Object::Iter(it) = top { + // SAFETY: nothing runs code while the view is held. + if let Some(PyIterator::Range { + current, + stop, + step, + }) = unsafe { it.peek_mut() } + { + if *step > 0 && *current < *stop { + let v = *current; + *current = current.wrapping_add(*step); + unsafe { base.add(len).write(Object::Int(v)) }; + len += 1; + pc += 1; + continue; + } + } + } + match self.gen_fast_for_iter(top, snap_gen, depth) { + ForNext::Value(v) => { + unsafe { base.add(len).write(v) }; + len += 1; + pc += 1; + } + ForNext::Exhausted => { + // The iterator (its sole owner is the stack) + // leaves, and the loop exits past its + // `END_FOR`/`POP_ITER` pair. + len -= 1; + crate::drop_hot(unsafe { base.add(len).read() }); + pc += 1 + ins.arg as usize; + let op_at = + |pc: usize| (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }); + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!(op_at(pc), Some(OpCode::PopIter | OpCode::PopTop)) { + pc += 1; + } + } + } + ForNext::Bail => break None, + } + } + OpCode::YieldValue => { + if len == 0 { + break None; + } + pc += 1; + // SAFETY: the slot is initialized; a folded value moves + // into the sink (or is a scalar), and the sent `None` + // takes its place. + let top = unsafe { base.add(len - 1) }; + let folded = match fold { + // SAFETY: the sink is the consumer's own, live and + // untouched while its resume runs (as the core + // loop's fold). + Some(FoldSink::Sum(acc)) => unsafe { (*acc).add_scalar(&*top) }, + Some(FoldSink::Collect(out)) => { + unsafe { (*out).push(top.read()) }; + true + } + None => false, + }; + if folded { + unsafe { top.write(Object::None) }; + continue; + } + len -= 1; + frame.agen_yielded_value = ins.arg == 0; + break Some(unsafe { top.read() }); + } + _ => break None, + } + }; + // SAFETY: the first `len` slots are initialized (pushes wrote them, + // pops moved values out). + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + match yielded { + Some(v) => GenStep::Yielded(v), + None => GenStep::Bail, + } + } + + /// `FOR_ITER`'s step on `top` for a fast step (out of line: the + /// loop keeps its registers): a range, list or tuple iterator's next + /// value, a range's end (when the loop alone holds it), or a fast + /// generator's next yield. + #[inline(never)] + fn gen_fast_for_iter(&mut self, top: &Object, snap_gen: u64, depth: u8) -> ForNext { + match top { + Object::Iter(it) => { + let unique = Rc::strong_count(it) == 1; + // SAFETY: nothing below runs code until `it`'s last use + // (the guard-free `peek`). + let Some(it) = (unsafe { it.peek_mut() }) else { + return ForNext::Bail; + }; + match it { + PyIterator::Range { + current, + stop, + step, + } => { + let live = if *step > 0 { + *current < *stop + } else { + *step < 0 && *current > *stop + }; + if live { + let v = *current; + *current = current.wrapping_add(*step); + ForNext::Value(Object::Int(v)) + } else if unique { + ForNext::Exhausted + } else { + ForNext::Bail + } + } + PyIterator::List { items, index, .. } => { + // SAFETY: as above. + let Some(v) = (unsafe { items.peek() }).and_then(|xs| xs.get(*index)) + else { + return ForNext::Bail; + }; + let v = copy(v); + *index += 1; + ForNext::Value(v) + } + PyIterator::Tuple { items, index } => { + let Some(v) = items.get(*index) else { + return ForNext::Bail; + }; + let v = copy(v); + *index += 1; + ForNext::Value(v) + } + _ => ForNext::Bail, + } + } + Object::Generator(g) => { + let g = g.clone(); + match self.gen_fast_next(&g, snap_gen, depth + 1) { + GenNext::Yielded(v) => ForNext::Value(v), + GenNext::Declined | GenNext::Partial => ForNext::Bail, + } + } + _ => ForNext::Bail, + } + } + + /// A resume's fast steps (its sent value already pushed): the value + /// the body yields next, with a draining consumer's `fold` taking the + /// yields it can along the way, or `None` for the general loop to + /// continue from wherever the steps stopped. + pub(crate) fn gen_fast_run( + &mut self, + frame: &mut Frame, + snap_gen: u64, + fold: Option, + ) -> Option { + if !Self::gen_fast_frame_ok(frame) { + return None; + } + match self.gen_fast_step(frame, snap_gen, 0, fold) { + GenStep::Yielded(v) => Some(v), + GenStep::Bail => None, + } + } + + /// `next(g)` by fast step (see the module docs): resumes `g` with + /// `None` and runs it to its next yield when both it and its body + /// allow. `depth` counts the fast steps this one is nested in. + pub(crate) fn gen_fast_next( + &mut self, + g: &Rc, + snap_gen: u64, + depth: u8, + ) -> GenNext { + if depth > MAX_DEPTH + || crate::recursion::current_depth() + usize::from(depth) + 2 + >= crate::recursion::recursion_limit() + { + return GenNext::Declined; + } + // Validate and take the frame under one exclusive view; no Python + // runs while it's held. + // SAFETY: nothing below reaches the cell again until `state`'s + // last use (`peek_mut` rejects a live guard or shared cells). + let Some(state) = (unsafe { g.state.peek_mut() }) else { + return GenNext::Declined; + }; + let (GeneratorState::Suspended(boxed) | GeneratorState::Created(boxed)) = &*state else { + return GenNext::Declined; + }; + let first_resume = matches!(*state, GeneratorState::Created(_)); + if !Self::gen_fast_frame_ok(boxed) { + return GenNext::Declined; + } + // A body an earlier fast step left partway (at the eval breaker, + // say) continues where it stopped, with no value sent. + let partial = boxed.sent_consumed; + let prev = std::mem::replace(state, GeneratorState::Running); + let (GeneratorState::Suspended(mut boxed) | GeneratorState::Created(mut boxed)) = prev + else { + unreachable!("checked above"); + }; + let frame: &mut Frame = &mut boxed; + frame.gen_first_resume = first_resume; + let start = frame.pc; + if partial { + frame.sent_consumed = false; + } else { + frame.stack.push(Object::None); + } + debug_assert!(!frame.stack.is_empty()); + let out = match self.gen_fast_step(frame, snap_gen, depth, None) { + GenStep::Yielded(v) => GenNext::Yielded(v), + GenStep::Bail if frame.pc == start => { + // Nothing ran: the resume is undone. + if partial { + frame.sent_consumed = true; + } else { + frame.stack.pop(); + } + GenNext::Declined + } + GenStep::Bail => { + frame.sent_consumed = true; + GenNext::Partial + } + }; + Self::park_suspended_boxed(g, boxed); + out + } +} diff --git a/crates/weavepy-vm/src/hot_filter.rs b/crates/weavepy-vm/src/hot_filter.rs index 97d4d5c1..8f11da51 100644 --- a/crates/weavepy-vm/src/hot_filter.rs +++ b/crates/weavepy-vm/src/hot_filter.rs @@ -72,8 +72,13 @@ impl AtomicBloom { #[inline] pub fn insert(&self, id: u64) { let ((w1, b1), (w2, b2)) = Self::probes(id); - self.bits[w1].fetch_or(b1, Ordering::Relaxed); - self.bits[w2].fetch_or(b2, Ordering::Relaxed); + // A bit already set needs no locked read-modify-write: bits are + // only ever cleared by a rebuild, which inserts never race. + for (w, b) in [(w1, b1), (w2, b2)] { + if self.bits[w].load(Ordering::Relaxed) & b == 0 { + self.bits[w].fetch_or(b, Ordering::Relaxed); + } + } } /// `false` means *definitely absent* (for every id that went @@ -126,7 +131,9 @@ impl RebuildableBloom { pub fn insert(&self, id: u64) { self.filters[0].insert(id); self.filters[1].insert(id); - self.inserts.fetch_add(1, Ordering::Relaxed); + // A staleness estimate: a lost increment only delays a rebuild. + let n = self.inserts.load(Ordering::Relaxed); + self.inserts.store(n.wrapping_add(1), Ordering::Relaxed); } #[inline] diff --git a/crates/weavepy-vm/src/inst_dict.rs b/crates/weavepy-vm/src/inst_dict.rs new file mode 100644 index 00000000..a7f7e8a3 --- /dev/null +++ b/crates/weavepy-vm/src/inst_dict.rs @@ -0,0 +1,1200 @@ +//! Split instance dictionaries: CPython's shared keys and inline values. +//! +//! An ordinary instance stores its attributes as a plain vector of values +//! whose names live once per class in a [`SharedKeys`] table, as long as +//! it assigns them in the order the class's first instances did (the +//! usual `__init__` shape). Constructing such an instance allocates no +//! hash table, and an attribute's position is the same in every instance, +//! so the indexed caches that address `__dict__` entries by insertion +//! order serve both layouts unchanged. +//! +//! Anything that needs the real `__dict__` (`vars(obj)`, a `del`, an +//! out-of-order or thirty-first attribute, the C API, pickling) goes +//! through [`InstDict::get`] or one of its siblings, which *materializes* +//! it: the values move into an ordinary [`DictData`] published in the +//! instance, and the instance keeps that dictionary from then on, as +//! CPython does. Only the hot paths that know about the split layout +//! avoid that step, so every other path sees exactly the dictionary it +//! always did. + +use crate::object::{DictData, DictKey, Object}; +use crate::shared_value::SharedStr; +use crate::sync::{LazyArc, Rc, RefCell}; +use std::cell::UnsafeCell; +use std::mem::MaybeUninit; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; + +/// The most attribute names a class shares (CPython's `SHARED_KEYS_MAX_SIZE`). +pub const SHARED_KEYS_CAP: usize = 30; + +/// A class's attribute names for its split instance dictionaries, in the +/// order its instances first assigned them. +/// +/// Append-only: a published name never moves or changes, and the table +/// never reallocates, so a reader needs only the published length. +/// Appends happen under the GIL (never in free-threaded mode). +pub struct SharedKeys { + len: AtomicUsize, + /// One bit per published name's Python hash (`hash & 63`): a clear + /// bit proves a name absent without comparing any. + filter: AtomicU64, + keys: [UnsafeCell>; SHARED_KEYS_CAP], + /// Each published name's Python hash. + hashes: [UnsafeCell; SHARED_KEYS_CAP], +} + +// SAFETY: names are published with release/acquire ordering and are +// immutable afterwards; the single writer is serialized by the GIL. +unsafe impl Send for SharedKeys {} +// SAFETY: as above. +unsafe impl Sync for SharedKeys {} + +impl std::fmt::Debug for SharedKeys { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_list() + .entries((0..self.len()).filter_map(|i| self.get(i))) + .finish() + } +} + +impl Default for SharedKeys { + fn default() -> Self { + Self { + len: AtomicUsize::new(0), + filter: AtomicU64::new(0), + keys: [const { UnsafeCell::new(MaybeUninit::uninit()) }; SHARED_KEYS_CAP], + hashes: [const { UnsafeCell::new(0) }; SHARED_KEYS_CAP], + } + } +} + +impl SharedKeys { + /// How many names are published. + #[inline] + pub fn len(&self) -> usize { + self.len.load(Ordering::Acquire) + } + + /// Whether no name is published yet. + #[inline] + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// The `i`th name. + #[inline] + pub fn get(&self, i: usize) -> Option<&DictKey> { + if i < self.len() { + // SAFETY: slots below the published length are initialized + // and never written again. + Some(unsafe { (*self.keys[i].get()).assume_init_ref() }) + } else { + None + } + } + + /// The first of the first `n` names equal to `name`, whose Python + /// hash is `hash`. + #[inline(always)] + fn position_hashed(&self, n: usize, name: &str, hash: i64) -> Option { + if self.filter.load(Ordering::Relaxed) & (1 << (hash & 63)) == 0 { + return None; + } + (0..n.min(self.len())).find(|&i| { + // SAFETY: slot `i` is published (below the length). + let h = unsafe { *self.hashes[i].get() }; + h == hash && self.get(i).is_some_and(|k| key_names_str(k, name)) + }) + } + + /// Publish the `str` name `name` as the next name; `None` when the + /// table is full. The caller holds the GIL outside free-threaded mode. + fn push(&self, name: &SharedStr) -> Option { + let n = self.len.load(Ordering::Relaxed); + if n >= SHARED_KEYS_CAP { + return None; + } + let hash = crate::object::py_str_hash(name); + // SAFETY: slot `n` is unpublished, so no reader can see it, and + // the GIL serializes writers. + unsafe { + (*self.keys[n].get()).write(DictKey(Object::Str(name.clone()))); + *self.hashes[n].get() = hash; + } + self.filter.fetch_or(1 << (hash & 63), Ordering::Relaxed); + self.len.store(n + 1, Ordering::Release); + Some(n) + } +} + +impl Drop for SharedKeys { + fn drop(&mut self) { + let n = *self.len.get_mut(); + for slot in &mut self.keys[..n] { + // SAFETY: the first `n` slots are initialized, and each is + // dropped once here. + unsafe { slot.get_mut().assume_init_drop() }; + } + } +} + +/// `a == b` for attribute names without a call into `memcmp`: they are +/// short and almost always differ in length or the first byte. +#[inline(always)] +fn name_eq(a: &str, b: &str) -> bool { + let (a, b) = (a.as_bytes(), b.as_bytes()); + if a.len() != b.len() { + return false; + } + if a.len() > 16 { + return a == b; + } + let mut i = 0; + while i < a.len() { + if a[i] != b[i] { + return false; + } + i += 1; + } + true +} + +/// Whether the `str` key `key` names `name`: identity first (names are +/// interned on both sides), then contents. +#[inline] +fn key_names(key: &DictKey, name: &SharedStr) -> bool { + matches!(&key.0, Object::Str(s) if SharedStr::ptr_eq(s, name) || name_eq(s, name)) +} + +/// [`key_names`] for a borrowed name. +#[inline] +fn key_names_str(key: &DictKey, name: &str) -> bool { + matches!(&key.0, Object::Str(s) if name_eq(s, name)) +} + +/// The header of an instance's split values allocation; the values +/// follow it. +#[repr(C)] +struct SplitHeader { + /// An owned strong reference to the class's shared names. + keys: *const SharedKeys, + len: u32, + cap: u32, +} + +/// An instance's split attribute values: `values[i]` belongs to +/// `keys[i]`, and the instance holds exactly the first `len` shared +/// names. One pointer wide; the names, length and capacity live in the +/// values allocation's header. +#[derive(Default)] +pub struct SplitValues { + block: Option>, +} + +// SAFETY: the block is owned exclusively; its contents are `Send`/`Sync` +// objects and a shared-keys reference. +unsafe impl Send for SplitValues {} +// SAFETY: as above; shared access is read-only. +unsafe impl Sync for SplitValues {} + +impl std::fmt::Debug for SplitValues { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_list().entries(self.iter()).finish() + } +} + +impl Drop for SplitValues { + fn drop(&mut self) { + let Some(block) = self.block.take() else { + return; + }; + // SAFETY: the block is ours; its first `len` values are + // initialized and its `keys` holds one strong count (or is null). + unsafe { + let h = block.as_ptr(); + std::ptr::drop_in_place(std::ptr::slice_from_raw_parts_mut( + Self::values_ptr(h), + (*h).len as usize, + )); + if !(*h).keys.is_null() { + drop(Rc::from_raw((*h).keys)); + } + std::alloc::dealloc(h.cast(), Self::layout((*h).cap as usize)); + } + } +} + +impl SplitValues { + #[inline] + fn layout(cap: usize) -> std::alloc::Layout { + std::alloc::Layout::new::() + .extend(std::alloc::Layout::array::(cap).expect("split capacity")) + .expect("split layout") + .0 + .pad_to_align() + } + + /// The first value slot of the block at `h`. + #[inline(always)] + fn values_ptr(h: *mut SplitHeader) -> *mut Object { + // SAFETY: the values start right after the header (both are + // word-aligned), inside the block's allocation. + unsafe { h.add(1).cast::() } + } + + /// The shared names, while any value is (or was) set. + #[inline(always)] + fn keys(&self) -> Option<&SharedKeys> { + let h = self.block?.as_ptr(); + // SAFETY: a non-null `keys` is a live strong reference owned by + // the block, which outlives `&self`. + unsafe { (*h).keys.as_ref() } + } + + /// How many attributes are set. + #[inline(always)] + pub fn len(&self) -> usize { + match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { (*b.as_ptr()).len as usize }, + None => 0, + } + } + + /// Whether no attribute is set. + #[inline(always)] + pub fn is_empty(&self) -> bool { + self.len() == 0 + } + + /// Every value, in assignment order. + #[inline(always)] + pub fn values(&self) -> &[Object] { + match self.block { + // SAFETY: the first `len` values are initialized. + Some(b) => unsafe { + std::slice::from_raw_parts(Self::values_ptr(b.as_ptr()), (*b.as_ptr()).len as usize) + }, + None => &[], + } + } + + /// [`Self::values`], writable. + #[inline(always)] + fn values_mut(&mut self) -> &mut [Object] { + match self.block { + // SAFETY: as above, and `&mut self` is exclusive. + Some(b) => unsafe { + std::slice::from_raw_parts_mut( + Self::values_ptr(b.as_ptr()), + (*b.as_ptr()).len as usize, + ) + }, + None => &mut [], + } + } + + /// Append `value`, growing the block to at least `want` slots; the + /// first append adopts `keys`. + fn push(&mut self, keys: impl FnOnce() -> Rc, want: usize, value: Object) { + let (len, cap) = match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { ((*b.as_ptr()).len as usize, (*b.as_ptr()).cap as usize) }, + None => (0, 0), + }; + if len == cap { + let new_cap = want.max(len + 1).max(cap * 2).max(2); + let new_layout = Self::layout(new_cap); + // SAFETY: a nonzero-size layout; a grown block keeps its + // header and initialized prefix (realloc copies them). + // (The block is allocated with the header's alignment.) + #[allow(clippy::cast_ptr_alignment)] + let h = unsafe { + match self.block { + Some(b) => { + std::alloc::realloc(b.as_ptr().cast(), Self::layout(cap), new_layout.size()) + } + None => std::alloc::alloc(new_layout), + } + } + .cast::(); + let Some(h) = std::ptr::NonNull::new(h) else { + std::alloc::handle_alloc_error(new_layout); + }; + // SAFETY: `h` is a live block of `new_cap` slots. + unsafe { + if self.block.is_none() { + h.as_ptr().write(SplitHeader { + keys: std::ptr::null(), + len: 0, + cap: 0, + }); + } + (*h.as_ptr()).cap = new_cap as u32; + } + self.block = Some(h); + } + let h = self.block.expect("allocated above").as_ptr(); + // SAFETY: slot `len` is inside the capacity and uninitialized. + unsafe { + if (*h).keys.is_null() { + (*h).keys = Rc::into_raw(keys()); + } + Self::values_ptr(h).add(len).write(value); + (*h).len = len as u32 + 1; + } + } + + /// The `i`th attribute in assignment order (the index a `__dict__` + /// entry of the same instance would have). + #[inline(always)] + pub fn get_index(&self, i: usize) -> Option<(&DictKey, &Object)> { + let v = self.values().get(i)?; + let k = self.keys()?.get(i)?; + Some((k, v)) + } + + /// The `i`th value, when these values are laid out over `keys` (so the + /// name at `i` is whatever `keys` published there). + #[inline(always)] + pub fn get_over(&self, keys: *const SharedKeys, i: usize) -> Option<&Object> { + let h = self.block?.as_ptr(); + // SAFETY: the block is live while owned, and its first `len` values + // are initialized. + unsafe { + if !std::ptr::eq((*h).keys, keys) || i >= (*h).len as usize { + return None; + } + Some(&*Self::values_ptr(h).add(i)) + } + } + + /// [`Self::get_over`], writable in place. + #[inline(always)] + pub fn get_over_mut(&mut self, keys: *const SharedKeys, i: usize) -> Option<&mut Object> { + let h = self.block?.as_ptr(); + // SAFETY: as `get_over`, and `&mut self` is exclusive. + unsafe { + if !std::ptr::eq((*h).keys, keys) || i >= (*h).len as usize { + return None; + } + Some(&mut *Self::values_ptr(h).add(i)) + } + } + + /// Append `value` as the value of position `i` of `keys`, when that + /// is the next unset position of values laid out over `keys` (or the + /// first value of an instance with none, which adopts `keys` through + /// `share`); `Err` hands the value back, touching nothing. + #[inline(always)] + pub fn append_over( + &mut self, + keys: &SharedKeys, + share: impl FnOnce() -> Rc, + i: usize, + value: Object, + ) -> Result<(), Object> { + if let Some(b) = self.block { + let h = b.as_ptr(); + // SAFETY: the block is live while owned; slot `len` is inside + // the capacity (checked) and uninitialized. + unsafe { + let len = (*h).len as usize; + if std::ptr::eq((*h).keys, keys) && i == len && len < (*h).cap as usize { + Self::values_ptr(h).add(len).write(value); + (*h).len = len as u32 + 1; + return Ok(()); + } + if !(*h).keys.is_null() || len != 0 { + return Err(value); + } + } + } + // No value yet (a fresh or recycled instance): adopt the names. + if i != 0 || keys.is_empty() { + return Err(value); + } + self.first_push(share, keys.len(), value); + Ok(()) + } + + /// [`Self::push`] for an instance's first value, out of line. + #[inline(never)] + fn first_push(&mut self, keys: impl FnOnce() -> Rc, want: usize, value: Object) { + // A recycled block too small for every shared name goes: the + // appends after this one then always find room (see + // `can_append_at`). + if let Some(b) = self.block { + // SAFETY: the block is live while owned; with no value and no + // names (the caller's state) it owns nothing else. + unsafe { + let h = b.as_ptr(); + if ((*h).cap as usize) < want { + debug_assert!((*h).len == 0 && (*h).keys.is_null()); + std::alloc::dealloc(h.cast(), Self::layout((*h).cap as usize)); + self.block = None; + } + } + } + self.push(keys, want, value); + } + + /// Whether [`Self::append_over`] of position `i` of `keys` succeeds + /// once the positions from the current length up to `i` have been + /// appended first (the next of a run of in-order appends). + #[inline] + pub fn can_append_at(&self, keys: &SharedKeys, i: usize) -> bool { + if i >= keys.len() { + return false; + } + match self.block { + // SAFETY: the block is live while owned. + Some(b) => unsafe { + let h = b.as_ptr(); + if std::ptr::eq((*h).keys, keys) { + i >= (*h).len as usize && i < (*h).cap as usize + } else { + // No values yet: the first append sizes the block for + // every name. + (*h).keys.is_null() && (*h).len == 0 + } + }, + None => true, + } + } + + /// [`Self::get_index`] with the value writable in place. + #[inline(always)] + pub fn get_index_mut(&mut self, i: usize) -> Option<(&DictKey, &mut Object)> { + let keys: *const SharedKeys = self.keys()?; + let v = self.values_mut().get_mut(i)?; + // SAFETY: the keys outlive `&mut self` (the block owns them), and + // they are disjoint from the values. + let k = unsafe { &*keys }.get(i)?; + Some((k, v)) + } + + /// The position of attribute `name`. + #[inline] + pub fn position(&self, name: &SharedStr) -> Option { + let keys = self.keys()?; + let n = self.len(); + // The constructor shape: the class's next name is unset here, and + // names are unique, so none of the set ones can match. + if keys.get(n).is_some_and(|k| key_names(k, name)) { + return None; + } + (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names(k, name))) + } + + /// The position of attribute `name` whose Python hash is `hash`: a + /// name no instance of the class ever set answers without comparing. + #[inline(always)] + pub fn position_hashed(&self, name: &str, hash: i64) -> Option { + self.keys()?.position_hashed(self.len(), name, hash) + } + + /// [`Self::position`] for a borrowed name (identity of the bytes + /// first: a name read off an interned string settles without a + /// comparison). + #[inline] + pub fn position_str(&self, name: &str) -> Option { + let keys = self.keys()?; + let n = self.len(); + for i in 0..n { + if let Some(DictKey(Object::Str(s))) = keys.get(i) { + if std::ptr::eq(s.as_ptr(), name.as_ptr()) && s.len() == name.len() { + return Some(i); + } + } + } + (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names_str(k, name))) + } + + /// The value of attribute `name`. + #[inline] + pub fn get(&self, name: &SharedStr) -> Option<&Object> { + self.position(name).map(|i| &self.values()[i]) + } + + /// [`Self::get`] for a borrowed name. + pub fn get_str(&self, name: &str) -> Option<&Object> { + self.position_str(name).map(|i| &self.values()[i]) + } + + /// Every attribute, in assignment order. + pub fn iter(&self) -> impl Iterator { + let keys = self.keys(); + self.values() + .iter() + .enumerate() + .filter_map(move |(i, v)| Some((keys?.get(i)?, v))) + } + + /// A copy of the names and values (for a materialization that can't + /// move them). + fn snapshot(&self) -> (Option>, Vec) { + let keys = self.block.and_then(|b| { + // SAFETY: a non-null `keys` is a live strong reference. + let k = unsafe { (*b.as_ptr()).keys }; + (!k.is_null()).then(|| unsafe { + Rc::increment_strong_count(k); + Rc::from_raw(k) + }) + }); + (keys, self.values().to_vec()) + } + + /// Store `value` under the interned name `name` if the split layout + /// can hold it: an existing attribute is overwritten in place + /// (returning the old value), and a new one is appended when it is + /// the next shared name (or becomes it). `Err` hands the value back + /// when the instance needs a real dictionary instead. + pub fn store( + &mut self, + class_keys: impl FnOnce() -> Rc, + name: &SharedStr, + value: Object, + ) -> Result, Object> { + let n = self.len(); + let adopted; + let keys: &SharedKeys = match self.keys() { + Some(k) => k, + None => { + adopted = class_keys(); + &adopted + } + }; + // The constructor shape first: the class's next name (names are + // unique, so it can't also be among the set ones). + let next = keys.get(n); + if !next.is_some_and(|k| key_names(k, name)) { + // An existing attribute. + if let Some(i) = (0..n).find(|&i| keys.get(i).is_some_and(|k| key_names(k, name))) { + return Ok(Some(std::mem::replace(&mut self.values_mut()[i], value))); + } + // A new name at the end of the table. + if next.is_some() || keys.len() != n || keys.push(name).is_none() { + return Err(value); + } + } + let want = keys.len(); + let keys_ptr: *const SharedKeys = keys; + self.push( + // SAFETY: `keys_ptr` is either the block's own reference or + // `adopted`, both live here; the new strong count is the + // block's. + || unsafe { + Rc::increment_strong_count(keys_ptr); + Rc::from_raw(keys_ptr) + }, + want, + value, + ); + Ok(None) + } + + /// Move every value out (the caller drops them after releasing the + /// cell) and forget the names. + pub fn take(&mut self) -> Vec { + let Some(b) = self.block else { + return Vec::new(); + }; + let h = b.as_ptr(); + // SAFETY: the first `len` values move out exactly once (the length + // is zeroed before anything can observe it), and the names' + // strong count is released once. + unsafe { + let n = (*h).len as usize; + let mut out = Vec::with_capacity(n); + std::ptr::copy_nonoverlapping(Self::values_ptr(h), out.as_mut_ptr(), n); + out.set_len(n); + (*h).len = 0; + let keys = std::mem::replace(&mut (*h).keys, std::ptr::null()); + if !keys.is_null() { + drop(Rc::from_raw(keys)); + } + out + } + } + + /// Drop every value, keeping the allocation, and forget the names. + pub fn reset(&mut self) { + drop(self.take()); + } +} + +/// An instance's `__dict__`: split values until something needs the +/// dictionary itself, then that dictionary (see the module docs). +/// +/// The accessors mirror [`LazyArc`]'s and materialize a split layout +/// first, so a caller that never heard of the split layout sees the same +/// dictionary it always did. +pub struct InstDict { + lazy: LazyArc>, + split: RefCell, +} + +impl Default for InstDict { + fn default() -> Self { + Self::new() + } +} + +impl From>> for InstDict { + fn from(d: Rc>) -> Self { + Self { + lazy: LazyArc::from(d), + split: RefCell::new(SplitValues::default()), + } + } +} + +impl From>> for InstDict { + fn from(lazy: LazyArc>) -> Self { + Self { + lazy, + split: RefCell::new(SplitValues::default()), + } + } +} + +impl Clone for InstDict { + #[track_caller] + fn clone(&self) -> Self { + // A shallow instance copy shares the dictionary itself. + self.get(); + Self::from(self.lazy.clone()) + } +} + +impl std::fmt::Debug for InstDict { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_tuple("InstDict").field(&self.lazy).finish() + } +} + +impl InstDict { + pub fn new() -> Self { + Self { + lazy: LazyArc::new(), + split: RefCell::new(SplitValues::default()), + } + } + + /// The instance that owns this field. + #[inline] + fn owner(&self) -> &crate::types::PyInstance { + let off = std::mem::offset_of!(crate::types::PyInstance, dict); + // SAFETY: an `InstDict` exists only as `PyInstance::dict`, so the + // containing instance starts `off` bytes before it and outlives + // this borrow (and is aligned as a `PyInstance`). + #[allow(clippy::cast_ptr_alignment)] + unsafe { + &*std::ptr::from_ref(self) + .cast::() + .sub(off) + .cast::() + } + } + + /// The published dictionary, if any, without materializing. + #[inline] + pub fn published(&self) -> Option<&RefCell> { + self.lazy.get() + } + + /// The split values, exclusively. + #[inline] + pub fn split_mut(&mut self) -> &mut SplitValues { + self.split.get_mut() + } + + /// The split values cell (meaningful while nothing is published). + #[inline] + pub fn split_cell(&self) -> &RefCell { + &self.split + } + + /// The split values, read without borrow bookkeeping: `None` when a + /// dictionary is published, the cell is mutably borrowed, or cells are + /// shared across threads. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]: the view must not outlive + /// anything that could store an attribute or materialize the dict. + #[inline] + pub unsafe fn split_peek(&self) -> Option<&SplitValues> { + if self.lazy.get().is_some() { + return None; + } + // SAFETY: forwarded contract. + unsafe { self.split.peek() } + } + + /// [`Self::split_peek`]'s exclusive form. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline] + #[allow(clippy::mut_from_ref)] + pub unsafe fn split_peek_mut(&self) -> Option<&mut SplitValues> { + if self.lazy.get().is_some() { + return None; + } + // SAFETY: forwarded contract. + unsafe { self.split.peek_mut() } + } + + /// Whether the split layout holds any attribute. + #[inline] + fn has_split(&self) -> bool { + // SAFETY: a length read, finished before anything else runs. + match unsafe { self.split.peek() } { + Some(s) => !s.is_empty(), + None => self.split.try_borrow().map_or(true, |s| !s.is_empty()), + } + } + + /// The dictionary, materializing a split layout; `None` when the + /// instance has no attributes and never had a dictionary. + #[inline] + #[track_caller] + pub fn get(&self) -> Option<&RefCell> { + if let Some(d) = self.lazy.get() { + return Some(d); + } + if !self.has_split() { + return None; + } + Some(self.materialize()) + } + + /// The dictionary, created by `make` when there is none (after + /// materializing a split layout). + #[inline] + #[track_caller] + pub fn get_or_init(&self, make: impl FnOnce() -> Rc>) -> &RefCell { + if let Some(d) = self.get() { + return d; + } + self.lazy.get_or_init(make) + } + + /// Another owner of the dictionary, if one exists (materializing). + #[track_caller] + pub fn get_shared(&self) -> Option>> { + self.get(); + self.lazy.get_shared() + } + + /// Another owner of the dictionary, created with the owner's + /// deferred-tracking record when there is none. + #[track_caller] + pub fn share(&self) -> Rc> { + self.owner().dict_cell(); + self.lazy.share() + } + + /// The published dictionary's strong count (`0` when none is). + pub fn strong_count(&self) -> usize { + self.lazy.strong_count() + } + + /// Move the split values into a published dictionary. + #[cold] + #[inline(never)] + #[track_caller] + fn materialize(&self) -> &RefCell { + note_materialize(std::panic::Location::caller()); + let owner = self.owner(); + let (keys, values) = match self.split.try_borrow_mut() { + Ok(mut s) => { + let keys = s.snapshot().0; + (keys, s.take()) + } + // A live view of the values (nothing in the VM holds one across + // a materializing call): copy them, and leave the stale split + // to the next exclusive reset. Readers check the dictionary + // first. + Err(_) => { + // SAFETY: only shared borrows can be live here; the copy + // reads through the cell without creating a `&mut`. + let s = unsafe { &*self.split.as_ptr() }; + s.snapshot() + } + }; + let n = values.len(); + let mut d = if owner.deferred.get() { + DictData::deferred_for_capacity(std::ptr::from_ref(owner) as usize, n) + } else { + DictData::with_capacity_and_hasher(n, crate::fasthash::FxBuildHasher) + }; + if let Some(keys) = keys { + // A deferred owner holds only atomic values, and a tracked + // one needs no barrier: insert without either. + let map = d.map_mut_atomic_store(); + for (i, v) in values.into_iter().enumerate() { + if let Some(k) = keys.get(i) { + map.insert(k.clone(), v); + } + } + } + self.lazy.get_or_init(|| Rc::new(RefCell::new(d))) + } +} + +/// `WEAVEPY_SPLIT_TRACE=1`: count materializations by call site and +/// report them at exit (to find hot paths that should read the split +/// layout directly). +fn note_materialize(at: &'static std::panic::Location<'static>) { + use std::collections::HashMap; + use std::sync::Mutex; + static ON: std::sync::OnceLock = std::sync::OnceLock::new(); + static COUNTS: Mutex>> = Mutex::new(None); + if !*ON.get_or_init(|| { + let on = std::env::var_os("WEAVEPY_SPLIT_TRACE").is_some(); + if on { + extern "C" fn report() { + let Ok(g) = COUNTS.lock() else { return }; + let Some(m) = g.as_ref() else { return }; + let mut v: Vec<_> = m.iter().collect(); + v.sort_by(|a, b| b.1.cmp(a.1)); + eprintln!("## split dict materializations"); + for (site, n) in v.into_iter().take(40) { + eprintln!("{n:10} {site}"); + } + } + // SAFETY: registering a plain `extern "C"` exit handler. + unsafe { libc::atexit(report) }; + } + on + }) { + return; + } + if let Ok(mut g) = COUNTS.lock() { + *g.get_or_insert_with(HashMap::new) + .entry(format!("{}:{}", at.file(), at.line())) + .or_default() += 1; + } +} + +impl crate::types::PyInstance { + /// The attribute at position `i` of the class's shared names, while + /// this instance's values are still split over its own class's names: + /// the name there is fixed for as long as the class lives, so a + /// caller that proved it once needs no name check again. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline(always)] + pub unsafe fn split_field(&self, i: usize) -> Option<&Object> { + let keys: *const SharedKeys = self.cls_raw().shared_keys.get()?; + // SAFETY: forwarded contract. + unsafe { self.dict.split_peek() }?.get_over(keys, i) + } + + /// Set the attribute at position `i` of the class's shared names by + /// appending `value`, when the instance's split values stop just + /// before `i` and their block has room (the constructor shape, after + /// its first store); `Err` hands the value back, touching nothing. + /// The caller proved the name at `i` and that a plain `__dict__` + /// store is what the assignment means (see `split_store`). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + pub unsafe fn split_append(&self, i: usize, value: Object) -> Result<(), Object> { + if self.c_body.get() != 0 || crate::gil::free_threading_enabled() { + return Err(value); + } + let cls = self.cls_raw(); + let Some(keys) = cls.shared_keys.get() else { + return Err(value); + }; + if !value.is_gc_atomic() && self.deferred.get() { + // The write barrier, before the values are borrowed (tracking + // early is always sound). + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + let Some(split) = (unsafe { self.dict.split_peek_mut() }) else { + return Err(value); + }; + split.append_over(keys, || cls.shared_keys.share(), i, value) + } + + /// The split length and whether position `i` of the class's shared + /// names is ready for a store: an overwrite of a set value (which + /// `droppable` must accept), or the next of a run of in-order appends + /// starting at `*cursor` (the split length once the earlier appends + /// land; `None` starts it at the current length). `None` when the + /// split layout can't say. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline] + pub unsafe fn split_store_ready( + &self, + i: usize, + cursor: &mut Option, + droppable: impl FnOnce(&Object) -> bool, + ) -> Option { + if self.c_body.get() != 0 || crate::gil::free_threading_enabled() { + return None; + } + let keys = self.cls_raw().shared_keys.get()?; + // SAFETY: forwarded contract. + let split = unsafe { self.dict.split_peek() }?; + let at = cursor.get_or_insert(split.len()); + if let Some(v) = split.get_over(keys, i) { + return Some(droppable(v)); + } + if i == *at && split.can_append_at(keys, i) { + *at += 1; + return Some(true); + } + None + } + + /// [`Self::split_field`] for an in-place overwrite by a value of + /// atomicity `atomic`, whose write barrier runs first (a non-atomic + /// value starts tracking a deferred instance). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + #[allow(clippy::mut_from_ref)] + pub unsafe fn split_field_mut(&self, i: usize, atomic: bool) -> Option<&mut Object> { + let keys: *const SharedKeys = self.cls_raw().shared_keys.get()?; + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + unsafe { self.dict.split_peek_mut() }?.get_over_mut(keys, i) + } + + /// The `i`th attribute in assignment order, from whichever layout the + /// instance uses, read without borrow bookkeeping. `None` when it is + /// absent or the storage is borrowed (take the general path). + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]: the view ends before anything + /// that could store an attribute runs. + #[inline(always)] + pub unsafe fn attr_peek_index(&self, i: usize) -> Option<(&DictKey, &Object)> { + match self.dict.published() { + // SAFETY: forwarded contract. + Some(d) => unsafe { d.peek() }?.get_index(i), + // SAFETY: forwarded contract. + None => unsafe { self.dict.split_cell().peek() }?.get_index(i), + } + } + + /// [`Self::attr_peek_index`] for an in-place value store: the write + /// barrier for a value of that atomicity runs first (a non-atomic + /// value starts tracking a deferred instance), and the key layout is + /// left alone. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek_mut`]. + #[inline(always)] + #[allow(clippy::mut_from_ref)] + pub unsafe fn attr_peek_index_mut( + &self, + i: usize, + atomic: bool, + ) -> Option<(&DictKey, &mut Object)> { + match self.dict.published() { + Some(d) => { + // SAFETY: forwarded contract. + let d = unsafe { d.peek_mut() }?; + let map = if atomic { + d.map_mut_unstamped() + } else { + d.map_mut_value_store() + }; + map.get_index_mut(i) + } + None => { + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + // SAFETY: forwarded contract. + unsafe { self.dict.split_cell().peek_mut() }?.get_index_mut(i) + } + } + } + + /// `f` of the `i`th attribute (either layout, borrowed safely); `None` + /// when it is absent or the storage is mutably borrowed. + #[inline] + pub fn attr_index_map(&self, i: usize, f: impl FnOnce(&DictKey, &Object) -> R) -> Option { + match self.dict.published() { + Some(d) => { + let d = d.try_borrow().ok()?; + let (k, v) = d.get_index(i)?; + Some(f(k, v)) + } + None => { + let s = self.dict.split_cell().try_borrow().ok()?; + let (k, v) = s.get_index(i)?; + Some(f(k, v)) + } + } + } + + /// Replace the value of the `i`th attribute (either layout, borrowed + /// safely) with `value`, returning the displaced value; `Err` hands + /// `value` back when there is no such attribute or the storage is + /// borrowed. The write barrier for `value` runs first. + pub fn attr_replace_index(&self, i: usize, value: Object) -> Result { + let atomic = value.is_gc_atomic(); + match self.dict.published() { + Some(d) => { + let Ok(mut d) = d.try_borrow_mut() else { + return Err(value); + }; + let map = if atomic { + d.map_mut_unstamped() + } else { + d.map_mut_value_store() + }; + match map.get_index_mut(i) { + Some((_, slot)) => Ok(std::mem::replace(slot, value)), + None => Err(value), + } + } + None => { + if !atomic && self.deferred.get() { + self.ensure_gc_tracked(); + } + let Ok(mut s) = self.dict.split_cell().try_borrow_mut() else { + return Err(value); + }; + match s.get_index_mut(i) { + Some((_, slot)) => Ok(std::mem::replace(slot, value)), + None => Err(value), + } + } + } + } + + /// Whether attribute `name` (Python hash `hash`) is set, read without + /// borrow bookkeeping; `None` when that can't be told natively. + /// + /// # Safety + /// + /// As [`crate::sync::GilCell::peek`]. + #[inline] + pub unsafe fn attr_peek_has(&self, name: &str, hash: i64) -> Option { + match self.dict.published() { + Some(d) => { + // SAFETY: forwarded contract. + let d = unsafe { d.peek() }?; + if d.is_empty() || !d.may_hold_str_hash(hash) { + return Some(false); + } + let probe = crate::object::LeafNameProbe::new(name, hash); + let hit = d.contains_key(&probe); + (!probe.saw_exotic()).then_some(hit) + } + None => { + // SAFETY: forwarded contract. + let s = unsafe { self.dict.split_cell().peek() }?; + Some(s.position_hashed(name, hash).is_some()) + } + } + } + + /// Attribute `name`'s value (either layout), without materializing + /// a split layout and without running code. + pub fn attr_get_str(&self, name: &str) -> Option { + match self.dict.published() { + Some(d) => d.borrow().get(&crate::object::StrKey(name)).cloned(), + None => self.dict.split_cell().borrow().get_str(name).cloned(), + } + } + + /// The position of attribute `name` in assignment order (either + /// layout, no materializing). + pub fn attr_position_str(&self, name: &str) -> Option { + let i = match self.dict.published() { + Some(d) => d + .try_borrow() + .ok()? + .get_index_of(&crate::object::StrKey(name))?, + None => self + .dict + .split_cell() + .try_borrow() + .ok()? + .position_str(name)?, + }; + u32::try_from(i).ok() + } + + /// How many attributes are set (either layout, no materializing). + pub fn attr_count(&self) -> usize { + match self.dict.published() { + Some(d) => d.try_borrow().map_or(0, |d| d.len()), + None => self.dict.split_cell().try_borrow().map_or(0, |s| s.len()), + } + } + + /// Visit every attribute (either layout, no materializing); a + /// borrowed storage is skipped. + pub fn for_each_attr(&self, mut f: impl FnMut(&DictKey, &Object)) { + match self.dict.published() { + Some(d) => { + if let Ok(d) = d.try_borrow() { + for (k, v) in d.iter() { + f(k, v); + } + } + } + None => { + if let Ok(s) = self.dict.split_cell().try_borrow() { + for (k, v) in s.iter() { + f(k, v); + } + } + } + } + } + + /// Store `value` under the interned name `name` through the split + /// layout: the displaced value on an overwrite, or `Err` handing the + /// value back when the instance needs (or has) a real dictionary. + /// Runs no code. The caller has established that a plain `__dict__` + /// store is what this assignment means. + #[inline] + pub fn split_store(&self, name: &SharedStr, value: Object) -> Result, Object> { + if self.dict.published().is_some() + || self.c_body.get() != 0 + || crate::gil::free_threading_enabled() + { + return Err(value); + } + let cls = self.cls_raw(); + if cls.native_kind.get() != 0 { + return Err(value); + } + if self.deferred.get() && !value.is_gc_atomic() { + self.ensure_gc_tracked(); + } + // SAFETY: nothing below runs code or reaches this cell again. + let Some(split) = (unsafe { self.dict.split_cell().peek_mut() }) else { + return Err(value); + }; + split.store(|| cls.shared_keys.share(), name, value) + } +} diff --git a/crates/weavepy-vm/src/lazy_arc.rs b/crates/weavepy-vm/src/lazy_arc.rs index 9b219011..728983c2 100644 --- a/crates/weavepy-vm/src/lazy_arc.rs +++ b/crates/weavepy-vm/src/lazy_arc.rs @@ -3,12 +3,12 @@ //! A published pointer is never reset through shared access. Every borrowed //! value remains live until its owner can be exclusively destroyed or replaced. +use crate::sync::Rc as Arc; use std::fmt; use std::marker::PhantomData; use std::mem::ManuallyDrop; use std::ptr; use std::sync::atomic::{AtomicPtr, Ordering}; -use std::sync::Arc; pub struct LazyArc { pointer: AtomicPtr, @@ -160,6 +160,96 @@ impl Drop for LazyArc { } } +/// A write-once value behind one pointer: `OnceLock` for a field that +/// is rarely set, where the lock's state word and an inline `T` would +/// enlarge every owner. +pub struct OnceBox { + pointer: AtomicPtr, + owner: PhantomData>, +} + +// SAFETY: publication is a single release compare-exchange of an owned +// box, and readers only take shared references; the published value is +// never replaced through shared access. +unsafe impl Sync for OnceBox {} +// SAFETY: as above. +unsafe impl Send for OnceBox {} + +impl OnceBox { + pub const fn new() -> Self { + Self { + pointer: AtomicPtr::new(ptr::null_mut()), + owner: PhantomData, + } + } + + /// The value, if one was set. + #[inline] + pub fn get(&self) -> Option<&T> { + // SAFETY: a published pointer owns a box that lives until `&mut` + // access destroys it. + unsafe { self.pointer.load(Ordering::Acquire).as_ref() } + } + + /// Set the value once; a second set hands the value back. + pub fn set(&self, value: T) -> Result<(), T> { + let fresh = Box::into_raw(Box::new(value)); + match self.pointer.compare_exchange( + ptr::null_mut(), + fresh, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => Ok(()), + // SAFETY: the unpublished box is still ours alone. + Err(_) => Err(*unsafe { Box::from_raw(fresh) }), + } + } + + /// Remove the value. + pub fn take(&mut self) -> Option { + let pointer = std::mem::replace(self.pointer.get_mut(), ptr::null_mut()); + // SAFETY: exclusive access; the published box is released once. + (!pointer.is_null()).then(|| *unsafe { Box::from_raw(pointer) }) + } +} + +impl Default for OnceBox { + fn default() -> Self { + Self::new() + } +} + +impl From for OnceBox { + fn from(value: T) -> Self { + Self { + pointer: AtomicPtr::new(Box::into_raw(Box::new(value))), + owner: PhantomData, + } + } +} + +impl Clone for OnceBox { + fn clone(&self) -> Self { + match self.get() { + Some(v) => Self::from(v.clone()), + None => Self::new(), + } + } +} + +impl fmt::Debug for OnceBox { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_tuple("OnceBox").field(&self.get()).finish() + } +} + +impl Drop for OnceBox { + fn drop(&mut self) { + drop(self.take()); + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/weavepy-vm/src/leaf_plan.rs b/crates/weavepy-vm/src/leaf_plan.rs new file mode 100644 index 00000000..84115000 --- /dev/null +++ b/crates/weavepy-vm/src/leaf_plan.rs @@ -0,0 +1,1524 @@ +//! Pre-decoded leaf bodies for the frameless leaf evaluator. +//! +//! A pure or effect leaf (see `code_pure_leaf_decide`) runs without an +//! activation: its operands are scalars or borrowed pointers, and any +//! miss abandons the evaluation having done nothing observable. This +//! module translates such a body once into a [`LeafPlan`]: every local +//! and every operand-stack position gets a fixed register, jumps name +//! their target op directly, and local loads, `POP_TOP`, `COPY` and +//! `SWAP` mostly disappear into operand addressing. Running a plan is +//! then one dispatch per real operation, with no stack pointer, no +//! bounds checks, and no per-read "was this local assigned" test. + +use weavepy_compiler::{BinOpKind, CodeObject, CompareKind, OpCode, UnaryKind}; + +use crate::object::Object; +use crate::sync::Rc; +use crate::{CodeConstObjects, Interpreter}; + +/// A leaf operand: a scalar by value, or a borrowed heap object. +#[derive(Clone, Copy)] +// Every payload sits at offset 8 (a primitive representation), so a copy +// is two words rather than a byte-wise shuffle. +#[repr(u64)] +pub(crate) enum V { + /// A resolved callee: a function the class or namespace holds. + Fn(*const crate::object::PyFunction), + /// A builtin type's method body, from the leaf method table (which + /// holds it for the interpreter's lifetime). + Bi(*const crate::object::BuiltinFn), + /// A call's empty self slot. + Null, + /// A heap object, borrowed. + R(*const Object), + I(i64), + F(f64), + B(bool), + N, +} + +#[inline(always)] +pub(crate) fn norm(p: *const Object) -> V { + // SAFETY: `p` names a live object (see `Interpreter::leaf_eval`). + match unsafe { &*p } { + Object::Int(i) => V::I(*i), + Object::Float(x) => V::F(*x), + Object::Bool(b) => V::B(*b), + Object::None => V::N, + _ => V::R(p), + } +} + +#[inline(always)] +pub(crate) fn truth(v: V) -> Option { + Some(match v { + V::B(b) => b, + V::I(i) => i != 0, + V::F(x) => x != 0.0, + V::N => false, + // SAFETY: as `norm`. + V::R(p) => match unsafe { &*p } { + Object::Str(s) => !s.is_empty(), + Object::Tuple(t) => !t.is_empty(), + // SAFETY: a read with nothing running (see `peek`). + Object::List(l) => !unsafe { l.peek() }?.is_empty(), + Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), + _ => return None, + }, + V::Fn(_) | V::Bi(_) | V::Null => return None, + }) +} + +/// The callback-free comparison rules both the decoded field shapes and +/// the plan runner use. NaNs and unsupported operands fall back to the +/// interpreter. +#[inline(always)] +pub(crate) fn compare(a: V, b: V, kind: CompareKind) -> Option { + const EXACT: u64 = 1 << 53; + let ord = match (a, b) { + (V::I(x), V::I(y)) => x.cmp(&y), + (V::F(x), V::F(y)) => x.partial_cmp(&y)?, + (V::I(x), V::F(y)) if x.unsigned_abs() < EXACT => (x as f64).partial_cmp(&y)?, + (V::F(x), V::I(y)) if y.unsigned_abs() < EXACT => x.partial_cmp(&(y as f64))?, + (V::B(x), V::B(y)) => x.cmp(&y), + (V::B(x), V::I(y)) => i64::from(x).cmp(&y), + (V::I(x), V::B(y)) => x.cmp(&i64::from(y)), + (V::N, V::N) if matches!(kind, CompareKind::Eq | CompareKind::NotEq) => { + std::cmp::Ordering::Equal + } + // SAFETY: these pointers name values owned by live arguments or + // the evaluator's scratch; no Python runs. + (V::R(p), V::R(q)) => match (unsafe { &*p }, unsafe { &*q }) { + (Object::Str(s), Object::Str(t)) => (**s).cmp(&**t), + _ => return None, + }, + _ => return None, + }; + Some(match kind { + CompareKind::Lt => ord.is_lt(), + CompareKind::LtE => ord.is_le(), + CompareKind::Eq => ord.is_eq(), + CompareKind::NotEq => ord.is_ne(), + CompareKind::Gt => ord.is_gt(), + CompareKind::GtE => ord.is_ge(), + }) +} + +/// Registers a plan may address: the locals, then the operand stack. +const REGS: usize = 32; + +/// One plan operation. Registers are `u8` indices below [`REGS`]; `pc` +/// is the source instruction (its inline caches and site slots), and +/// `name` its `co_names` index. +#[derive(Clone, Copy, Debug)] +enum Op { + /// `regs[dst] = regs[src]`. + Move { + dst: u8, + src: u8, + }, + /// `regs[dst] = consts[k]`. + Const { + dst: u8, + k: u16, + }, + Global { + dst: u8, + pc: u16, + }, + Attr { + dst: u8, + src: u8, + pc: u16, + name: u16, + }, + /// The method-form load: `regs[dst]` the function, `regs[dst + 1]` + /// the receiver (or the empty self slot for a class receiver). + Method { + dst: u8, + src: u8, + pc: u16, + name: u16, + }, + Compare { + dst: u8, + a: u8, + b: u8, + kind: u8, + }, + Is { + dst: u8, + a: u8, + b: u8, + invert: bool, + }, + Truth { + dst: u8, + src: u8, + }, + Unary { + dst: u8, + src: u8, + kind: u8, + }, + Binary { + dst: u8, + a: u8, + b: u8, + kind: u8, + }, + /// Exchange two registers. + Swap { + a: u8, + b: u8, + }, + /// Jump to op `target` when `regs[src]`'s truth is `when`. + BranchIf { + src: u8, + when: bool, + target: u16, + }, + /// Jump to op `target` when `regs[src]` `is None` is `when`. + BranchNone { + src: u8, + when: bool, + target: u16, + }, + Jump { + target: u16, + }, + /// `regs[at]` the callee, `regs[at + 1]` its self slot, then `argc` + /// arguments; the result lands in `regs[at]`. + Call { + at: u8, + argc: u8, + pc: u16, + }, + Return { + src: u8, + }, + /// `regs[dst]` a new empty list (`BUILD_LIST 0`) or, with `dict`, + /// an empty dict (`BUILD_MAP 0`), held by the owned scratch. + New { + dst: u8, + dict: bool, + }, + /// Abandon the evaluation: the path reached an instruction the plan + /// can't run. + Decline, + /// An effect leaf's buffered `recv.name = val`. + StoreAttr { + recv: u8, + val: u8, + pc: u16, + name: u16, + }, +} + +/// A leaf body, translated (see the module docs). +pub(crate) struct LeafPlan { + ops: Box<[Op]>, + /// The loaded constants, normalized. A heap constant points into the + /// code extension's materialized constants, which outlive the plan. + consts: Box<[V]>, + /// Arguments, copied into the first registers on entry. + nargs: u8, + /// Registers the plan uses (the locals, then its deepest stack). + nregs: u8, + /// Every attribute store goes to the first argument (never + /// reassigned), each to a different name: the buffered stores share + /// one receiver and each is the latest to its attribute. + unique_stores: bool, +} + +// SAFETY: the constants' pointers name the code extension's immutable +// materialized constants (`Object`s shared the way the extension itself +// shares them); a plan is only read. +unsafe impl Send for LeafPlan {} +// SAFETY: as above. +unsafe impl Sync for LeafPlan {} + +/// A value on the translator's abstract operand stack: the register +/// that holds it. A stack position's own register is `nl + position`; +/// a local's value is read in place from the local's register until +/// something would overwrite it. +type Opnd = u8; + +struct Builder<'a> { + code: &'a CodeObject, + ext: &'a CodeConstObjects, + ops: Vec, + consts: Vec, + /// Number of locals (registers below it). + nl: usize, + stack: Vec, + /// The names stored into so far, while every store goes to the + /// first argument (see [`LeafPlan::unique_stores`]); `None` once one + /// doesn't. + stored: Option>, + /// Locals definitely assigned on the current path (bit per local). + assigned: u32, + /// Per bytecode target: the stack depth and assigned set on the + /// edges seen so far. + targets: std::collections::HashMap, + /// `(op index, bytecode target)` of every emitted jump. + fixups: Vec<(usize, usize)>, + /// The op index each bytecode instruction starts at. + starts: Vec, + /// One past the highest register handed out. + top: std::cell::Cell, +} + +impl Builder<'_> { + fn slot(&self, pos: usize) -> Option { + let r = self.nl + pos; + self.top.set(self.top.get().max(r + 1)); + (r < REGS).then_some(r as u8) + } + + fn push_new(&mut self) -> Option { + let r = self.slot(self.stack.len())?; + self.stack.push(r); + Some(r) + } + + fn pop(&mut self) -> Option { + self.stack.pop() + } + + /// Put every stack entry in its own position's register. + fn canonicalize(&mut self) -> Option<()> { + for pos in 0..self.stack.len() { + let own = self.slot(pos)?; + if self.stack[pos] != own { + // Entries only ever name locals or their own position + // (see `swap` and `copy`), so this write clobbers none. + self.ops.push(Op::Move { + dst: own, + src: self.stack[pos], + }); + self.stack[pos] = own; + } + } + Some(()) + } + + /// Before local `i` is overwritten: every stack entry still reading + /// it moves into its own position's register. + fn release_local(&mut self, i: u8) -> Option<()> { + for pos in 0..self.stack.len() { + if self.stack[pos] == i { + let own = self.slot(pos)?; + self.ops.push(Op::Move { dst: own, src: i }); + self.stack[pos] = own; + } + } + Some(()) + } + + fn load_local(&mut self, i: u32) -> Option<()> { + let i = i as usize; + if i >= self.nl || self.assigned & (1 << i) == 0 { + return None; + } + self.stack.push(i as u8); + Some(()) + } + + fn store_local(&mut self, i: u32) -> Option<()> { + let i = i as usize; + if i >= self.nl { + return None; + } + let src = self.pop()?; + if src != i as u8 { + self.release_local(i as u8)?; + self.ops.push(Op::Move { dst: i as u8, src }); + } + if i == 0 { + // The first argument no longer names one receiver. + self.stored = None; + } + self.assigned |= 1 << i; + Some(()) + } + + fn constant(&mut self, v: V) -> Option<()> { + let k = u16::try_from(self.consts.len()).ok()?; + self.consts.push(v); + let dst = self.push_new()?; + self.ops.push(Op::Const { dst, k }); + Some(()) + } + + /// Record a jump edge to bytecode `target` with the current state + /// (already canonical), and emit its fixup. + fn edge(&mut self, target: usize) -> Option<()> { + let state = (self.stack.len(), self.assigned); + match self.targets.get_mut(&target) { + Some((depth, assigned)) => { + if *depth != state.0 { + return None; + } + *assigned &= state.1; + } + None => { + self.targets.insert(target, state); + } + } + self.fixups.push((self.ops.len() - 1, target)); + Some(()) + } + + /// Translate the instruction at `pc`: whether control falls through + /// to the next one. `None` for one the plan can't express (the + /// caller rolls back what it emitted and declines on that path). + fn step(&mut self, pc: usize, ins: weavepy_compiler::Instruction) -> Option { + let arg = ins.arg; + match ins.op { + OpCode::Resume | OpCode::Nop | OpCode::NotTaken => {} + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { + self.load_local(arg)?; + } + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { + self.load_local(arg >> 4)?; + self.load_local(arg & 15)?; + } + OpCode::StoreFast => self.store_local(arg)?, + OpCode::StoreFastLoadFast => { + self.store_local(arg >> 4)?; + self.load_local(arg & 15)?; + } + OpCode::StoreFastStoreFast => { + self.store_local(arg >> 4)?; + self.store_local(arg & 15)?; + } + OpCode::LoadConst => { + let v = norm(self.ext.objects.get(arg as usize)?); + self.constant(v)?; + } + OpCode::LoadSmallInt => self.constant(V::I(i64::from(arg)))?, + OpCode::PushNull => self.constant(V::Null)?, + OpCode::BuildList | OpCode::BuildMap if arg == 0 => { + let dst = self.push_new()?; + self.ops.push(Op::New { + dst, + dict: ins.op == OpCode::BuildMap, + }); + } + OpCode::LoadGlobal => { + let dst = self.push_new()?; + self.ops.push(Op::Global { + dst, + pc: u16::try_from(pc).ok()?, + }); + } + OpCode::LoadGlobalPushNull => { + let dst = self.push_new()?; + self.ops.push(Op::Global { + dst, + pc: u16::try_from(pc).ok()?, + }); + self.constant(V::Null)?; + } + OpCode::LoadAttr => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Attr { + dst, + src, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + OpCode::LoadMethodAttr => { + let src = self.pop()?; + let dst = self.push_new()?; + self.push_new()?; + self.ops.push(Op::Method { + dst, + src, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + OpCode::CompareOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + let kind = (arg & !weavepy_compiler::COMPARE_OP_TO_BOOL_FLAG) as u8; + if kind > CompareKind::GtE as u8 { + return None; + } + self.ops.push(Op::Compare { dst, a, b, kind }); + } + OpCode::IsOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Is { + dst, + a, + b, + invert: arg == 1, + }); + } + OpCode::ToBool => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Truth { dst, src }); + } + OpCode::UnaryOp => { + let src = self.pop()?; + let dst = self.push_new()?; + self.ops.push(Op::Unary { + dst, + src, + kind: arg as u8, + }); + } + OpCode::BinaryOp => { + let b = self.pop()?; + let a = self.pop()?; + let dst = self.push_new()?; + // (The in-place flag sits above the kind's byte, and + // means nothing for the scalars this runs.) + self.ops.push(Op::Binary { + dst, + a, + b, + kind: arg as u8, + }); + } + OpCode::CopyTop => { + let n = (arg as usize).max(1); + let depth = self.stack.len(); + if n > depth { + return None; + } + let src = self.stack[depth - n]; + let dst = self.push_new()?; + self.ops.push(Op::Move { dst, src }); + } + OpCode::Swap => { + let n = arg as usize; + let depth = self.stack.len(); + if n < 2 || n > depth { + return None; + } + // Both entries move into their own registers first, so + // the exchange keeps every entry in its own position. + let (lo, hi) = (depth - n, depth - 1); + for pos in [lo, hi] { + let own = self.slot(pos)?; + if self.stack[pos] != own { + self.ops.push(Op::Move { + dst: own, + src: self.stack[pos], + }); + self.stack[pos] = own; + } + } + self.ops.push(Op::Swap { + a: self.stack[lo], + b: self.stack[hi], + }); + } + OpCode::PopTop => { + self.pop()?; + } + OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::PopJumpIfNone + | OpCode::PopJumpIfNotNone => { + // The condition is a local or its own position's + // register, which lies above the remaining entries: + // canonicalizing them writes neither. + let src = self.pop()?; + self.canonicalize()?; + let target = pc + 1 + arg as usize; + self.ops.push(match ins.op { + OpCode::PopJumpIfFalse => Op::BranchIf { + src, + when: false, + target: 0, + }, + OpCode::PopJumpIfTrue => Op::BranchIf { + src, + when: true, + target: 0, + }, + OpCode::PopJumpIfNone => Op::BranchNone { + src, + when: true, + target: 0, + }, + _ => Op::BranchNone { + src, + when: false, + target: 0, + }, + }); + self.edge(target)?; + } + OpCode::JumpForward => { + self.canonicalize()?; + self.ops.push(Op::Jump { target: 0 }); + self.edge(pc + 1 + arg as usize)?; + return Some(false); + } + OpCode::Call => { + let argc = arg as usize; + let depth = self.stack.len(); + if depth < argc + 2 { + return None; + } + self.canonicalize()?; + let at = depth - argc - 2; + self.stack.truncate(at); + let r = self.push_new()?; + self.ops.push(Op::Call { + at: r, + argc: u8::try_from(argc).ok()?, + pc: u16::try_from(pc).ok()?, + }); + } + OpCode::ReturnValue => { + let src = self.pop()?; + self.ops.push(Op::Return { src }); + return Some(false); + } + OpCode::StoreAttr => { + let recv = self.pop()?; + let val = self.pop()?; + if let Some(names) = &mut self.stored { + if recv != 0 || names.contains(&arg) { + self.stored = None; + } else { + names.push(arg); + } + } + self.ops.push(Op::StoreAttr { + recv, + val, + pc: u16::try_from(pc).ok()?, + name: u16::try_from(arg).ok()?, + }); + } + _ => return None, + } + Some(true) + } + + fn build(mut self) -> Option { + let instrs = &self.code.instructions; + // Whether the previous instruction falls through to this one. + let mut live = true; + #[allow(clippy::needless_range_loop)] + for pc in 0..instrs.len() { + if let Some(&(depth, assigned)) = self.targets.get(&pc) { + if live { + // A fallthrough into a merge point arrives canonical + // (its moves run before the point the jumps land on). + self.canonicalize()?; + if self.stack.len() != depth { + return None; + } + self.assigned &= assigned; + } else { + self.assigned = assigned; + } + self.stack.clear(); + for pos in 0..depth { + let r = self.slot(pos)?; + self.stack.push(r); + } + live = true; + } + self.starts.push(u16::try_from(self.ops.len()).ok()?); + if !live { + // Unreachable code: nothing jumps here. + continue; + } + let mark = (self.ops.len(), self.fixups.len()); + match self.step(pc, instrs[pc]) { + Some(next) => live = next, + None => { + // The path that reaches this instruction declines + // (the old evaluator's behaviour: only a path that + // runs an unsupported instruction abandons). + self.ops.truncate(mark.0); + self.fixups.truncate(mark.1); + self.ops.push(Op::Decline); + live = false; + } + } + } + // Every path ends in a return or a jump: the runner never steps + // past the last op. + if live || self.ops.is_empty() { + return None; + } + for &(at, target) in &self.fixups { + let to = *self.starts.get(target)?; + if usize::from(to) >= self.ops.len() { + return None; + } + match &mut self.ops[at] { + Op::BranchIf { target, .. } + | Op::BranchNone { target, .. } + | Op::Jump { target } => { + *target = to; + } + _ => return None, + } + } + Some(LeafPlan { + ops: self.ops.into_boxed_slice(), + consts: self.consts.into_boxed_slice(), + nargs: u8::try_from(crate::leaf_arity(self.code)).ok()?, + nregs: u8::try_from(self.top.get().clamp(self.nl, REGS)).ok()?, + unique_stores: self.stored.is_some() && self.code.arg_count > 0, + }) + } +} + +/// Translate `code` (a pure or effect leaf) into a plan; `None` for a +/// body the plan can't express (the ordinary call runs it instead). +pub(crate) fn build(code: &CodeObject, ext: &CodeConstObjects) -> Option { + let nl = code.varnames.len(); + let nargs = crate::leaf_arity(code); + if nl > 16 || nargs > nl || nargs > 8 || code.instructions.len() > u16::MAX as usize { + return None; + } + Builder { + code, + ext, + ops: Vec::new(), + consts: Vec::new(), + nl, + stack: Vec::new(), + stored: Some(Vec::new()), + assigned: (1u32 << nargs) - 1, + targets: std::collections::HashMap::new(), + fixups: Vec::new(), + starts: Vec::with_capacity(code.instructions.len()), + top: std::cell::Cell::new(nl), + } + .build() +} + +/// How many owned values one evaluation (its callee frames included) +/// may hold. +const OWNED: usize = 8; + +/// Values a leaf path hands back owned (a polymorphic or class read, a +/// nested call's result) stay here until the evaluation ends; registers +/// point in. +struct Owned { + buf: [std::mem::MaybeUninit; OWNED], + n: usize, +} + +impl Drop for Owned { + // Usually nothing is held: no call for the empty case. + #[inline(always)] + fn drop(&mut self) { + if self.n != 0 { + self.release(); + } + } +} + +impl Owned { + #[inline(never)] + fn release(&mut self) { + for k in 0..self.n { + // SAFETY: the first `n` entries are initialized. + crate::drop_hot(unsafe { self.buf[k].assume_init_read() }); + } + } +} + +impl Owned { + /// Whether `p` names one of the held values. + #[inline(always)] + fn holds(&self, p: *const Object) -> bool { + let base = self.buf.as_ptr().cast::(); + // SAFETY: one past the buffer's end. + let end = unsafe { base.add(self.buf.len()) }; + p >= base && p < end + } + + /// Hold `v` (a scalar needs no holding) and name it. + #[inline] + fn own(&mut self, v: Object) -> Option { + let scalar = match v { + Object::Int(i) => V::I(i), + Object::Float(x) => V::F(x), + Object::Bool(b) => V::B(b), + Object::None => V::N, + _ => { + if self.n == self.buf.len() { + return None; + } + let slot = &mut self.buf[self.n]; + slot.write(v); + self.n += 1; + return Some(V::R(slot.as_ptr())); + } + }; + // A scalar owns nothing: no drop glue to run. + std::mem::forget(v); + Some(scalar) + } +} + +/// An effect leaf's attribute stores, in order, until the return commits +/// them: receiver, store pc, name index, value (all owned). Entries are +/// only appended, so a value read back from one stays put while the +/// evaluation runs. +const PENDING: usize = 6; + +/// The receiver is a stable location for the whole evaluation: an +/// argument (rooted by the caller) or an owned-scratch clone. +type Store = (*const Object, u32, u32, Object); + +struct Pending { + buf: [std::mem::MaybeUninit; PENDING], + n: usize, + /// Entries whose value the return moved into a dict. + moved: u8, +} + +impl Pending { + fn get(&self, k: usize) -> &Store { + debug_assert!(k < self.n); + // SAFETY: the first `n` entries are initialized. + unsafe { self.buf[k].assume_init_ref() } + } + + /// Whether entry `k` is the last store to its attribute. + fn latest(&self, k: usize) -> bool { + let (r0, _, n0, _) = self.get(k); + !(k + 1..self.n).any(|j| { + let (r1, _, n1, _) = self.get(j); + // SAFETY: stable receivers (see `Store`). + n1 == n0 && unsafe { (**r1).is_same(&**r0) } + }) + } +} + +impl Drop for Pending { + // Usually nothing is buffered: no call for the empty case. + #[inline(always)] + fn drop(&mut self) { + if self.n != 0 { + self.release(); + } + } +} + +impl Pending { + #[inline(never)] + fn release(&mut self) { + for k in 0..self.n { + // SAFETY: the first `n` entries are initialized; a moved value + // is left in place, not dropped again. + let (_, _, _, value) = unsafe { self.buf[k].assume_init_read() }; + if self.moved & (1 << k) == 0 { + crate::drop_hot(value); + } else { + std::mem::forget(value); + } + } + } +} + +/// A leaf call's callee: a Python function to evaluate in place, or a +/// builtin (with its owner, when a namespace holds one). +enum Callee<'a> { + Py(*const crate::object::PyFunction), + Native( + *const crate::object::BuiltinFn, + Option<&'a Rc>, + ), +} + +/// A leaf evaluation's result: a value the caller may borrow for the rest +/// of its own evaluation (an argument, a field of one, a namespace entry +/// or a constant: nothing the evaluation owned), or an owned object. +pub(crate) enum LeafRet { + Borrowed(V), + Owned(Object), +} + +impl LeafRet { + /// The result as an owned object (a borrowed value is cloned). + #[inline(always)] + pub(crate) fn into_object(self) -> Option { + match self { + LeafRet::Borrowed(v) => to_object(v), + LeafRet::Owned(o) => Some(o), + } + } +} + +/// An owned object for a leaf value (the return value, a buffered +/// store's value); `None` for the markers that are never values. +#[inline(always)] +fn to_object(v: V) -> Option { + Some(match v { + // SAFETY: as `norm` (an owned value is cloned before its holder + // drops). + V::R(p) => crate::clone_hot(unsafe { &*p }), + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Bi(_) | V::Null => return None, + }) +} + +impl Interpreter { + /// Run `plan` (the translation of `code`, whose namespaces are `f`'s) + /// on borrowed `args`, at call-nesting depth `nest`: the leaf's + /// result, or `None` having done nothing observable (see + /// `Interpreter::leaf_eval`). + /// + /// `FRESH`: the first argument is an instance nothing else has seen + /// (a constructor's `self`), so a store into it lands at once — a + /// later decline leaves it half-built, and the caller discards it. + #[inline(always)] + pub(crate) fn leaf_run( + &self, + code: &CodeObject, + ext: &CodeConstObjects, + plan: &LeafPlan, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + self.leaf_run_ret::(code, ext, plan, f, args, nest)? + .into_object() + } + + /// [`Self::leaf_run`] with its result borrowed when the caller may + /// (see [`LeafRet`]). + #[inline(never)] + pub(crate) fn leaf_run_ret( + &self, + code: &CodeObject, + ext: &CodeConstObjects, + plan: &LeafPlan, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + use weavepy_compiler::InlineCache as IC; + /// Nested leaf calls evaluated in place at most this deep. + const NEST: u8 = 3; + /// A pure-leaf callee runs as a frame of this evaluation: its + /// registers are the next `REGS` window, and its caller's state + /// waits here until its return. + struct Caller<'p> { + ip: usize, + code: &'p CodeObject, + ext: &'p CodeConstObjects, + plan: &'p LeafPlan, + f: &'p crate::object::PyFunction, + /// The caller's register for the result. + at: u8, + /// The caller's registers (its plan's `nregs`). + saved: [std::mem::MaybeUninit; REGS], + } + let nargs = usize::from(plan.nargs); + if args.len() != nargs { + return None; + } + let (mut code, mut ext, mut plan, mut f) = (code, ext, plan, f); + let mut regs = [const { std::mem::MaybeUninit::::uninit() }; REGS]; + for (k, &a) in args.iter().enumerate() { + regs[k].write(norm(a)); + } + let mut callers = [const { std::mem::MaybeUninit::>::uninit() }; NEST as usize]; + let mut depth = 0usize; + // The translation proves every register is written before it's + // read, and that every index is below `REGS`. + macro_rules! get { + ($r:expr) => { + // SAFETY: see above. + unsafe { regs.get_unchecked(usize::from($r)).assume_init() } + }; + } + macro_rules! set { + ($r:expr, $v:expr) => {{ + let v = $v; + // SAFETY: see above. + unsafe { regs.get_unchecked_mut(usize::from($r)).write(v) }; + }}; + } + let mut consts: &[V] = &plan.consts; + let mut stamps: &[crate::StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + let mut owned = Owned { + buf: [const { std::mem::MaybeUninit::uninit() }; OWNED], + n: 0, + }; + let mut pend = Pending { + buf: [const { std::mem::MaybeUninit::uninit() }; PENDING], + n: 0, + moved: 0, + }; + let mut ops: &[Op] = &plan.ops; + let mut ip = 0usize; + loop { + // SAFETY: every path ends in a return or a jump, and every + // jump target is an op (checked when the plan was built). + // Matched in place: each arm loads only its own operands. + let op: &Op = unsafe { ops.get_unchecked(ip) }; + ip += 1; + match *op { + Op::Move { dst, src } => set!(dst, get!(src)), + // SAFETY: `k` names a plan constant (the translation). + Op::Const { dst, k } => set!(dst, unsafe { *consts.get_unchecked(usize::from(k)) }), + Op::Global { dst, pc } => { + // The callee's own namespaces, stamp-validated as the + // core loop's `LOAD_GLOBAL` arm. + let pc = usize::from(pc); + let slot = stamps.get(pc)?; + let (gdict, bdict) = (f.globals.as_ptr(), f.builtins.as_ptr()); + let gid = crate::specialize::rc_id(&f.globals); + // SAFETY (raw dict reads): nothing runs code here. + let g_stamp = unsafe { (*gdict).mutation_stamp() }; + let hit = match code.caches.get(pc as u32) { + IC::LoadGlobalModule { + globals_id, + key_idx, + } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { + (*gdict).get_index(key_idx as usize) + }, + IC::LoadGlobalBuiltin { + builtins_id, + key_idx, + } if builtins_id == crate::specialize::rc_id(&f.builtins) + && !self.globals_missing_any.get() + && slot.get() + == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => + unsafe { (*bdict).get_index(key_idx as usize) }, + _ => return None, + }; + set!(dst, norm(hit?.1)); + } + Op::Attr { dst, src, pc, name } => { + let V::R(p) = get!(src) else { + return None; + }; + let (pc, name) = (u32::from(pc), u32::from(name)); + // SAFETY: as `norm`. + let recv = unsafe { &*p }; + if EFFECT && pend.n > 0 { + // A buffered store to this attribute, the latest. + let hit = (0..pend.n) + .rev() + .map(|k| pend.get(k)) + // SAFETY: stable receivers (see `Store`). + .find(|(r, _, n, _)| *n == name && unsafe { (**r).is_same(recv) }); + if let Some((_, _, _, v)) = hit { + set!(dst, norm(v)); + continue; + } + } + let v = match recv { + Object::Instance(inst) => { + if GETTER && inst.cls_raw().native_kind.get() != 0 { + return None; + } + // SAFETY: the receiver remains rooted by an + // argument or owned scratch; no Python runs. + let hit = unsafe { + Self::leaf_cached_instance_field(ext, code, inst, pc, name) + } + .map(std::ptr::from_ref); + match hit { + Some(v) => norm(v), + // A stale or absent site cache resolves + // through the site's own entries. + None => owned.own(Self::leaf_attr_resolve_site( + code, inst, recv, pc, name, + )?)?, + } + } + Object::Type(cls) => { + if GETTER && !Self::plain_metaclass(cls) { + return None; + } + match stamps + .get(pc as usize) + .and_then(|s| crate::class_attr_hit(s, cls)) + { + Some(v) => owned.own(v)?, + None => { + owned.own(Self::leaf_load_type_attr(code, cls, pc, name)?)? + } + } + } + Object::Module(module) => { + if GETTER + && (crate::object::module_class(module).is_some() + || code.names.get(name as usize)?.starts_with("__")) + { + return None; + } + owned.own(Self::leaf_load_attr_recv(code, recv, pc, name)?)? + } + _ => return None, + }; + set!(dst, v); + } + Op::Method { dst, src, pc, name } => { + // A method off the site's slot: the function under + // the receiver, or a class's function with an empty + // self slot (as the core loop's arm). + let V::R(p) = get!(src) else { + return None; + }; + let ms = crate::code_method_slot(code, u32::from(pc))?; + // SAFETY: as `norm`. + match unsafe { &*p } { + Object::Instance(inst) => { + let cls = inst.cls_raw(); + if !Self::default_getattribute(cls) { + return None; + } + let fp = ms.peek_fn(cls.attr_version.get())?; + // The instance's attributes must not shadow + // the method. + if crate::inst_may_shadow(inst, code, u32::from(name)) { + return None; + } + set!(dst, V::Fn(fp)); + set!(dst + 1, V::R(p)); + } + Object::Type(cls) => { + set!(dst, V::Fn(ms.peek_unbound(cls.attr_version.get())?)); + set!(dst + 1, V::Null); + } + recv => { + let b = self.leaf_builtin_method_ptr(ms, recv, code, name)?; + set!(dst, V::Bi(b)); + set!(dst + 1, V::R(p)); + } + } + } + Op::Compare { dst, a, b, kind } => { + // SAFETY: the translation checked `kind` names a + // comparison. + let kind: CompareKind = unsafe { std::mem::transmute(kind) }; + set!(dst, V::B(compare(get!(a), get!(b), kind)?)); + } + Op::Is { dst, a, b, invert } => { + let same = match (get!(a), get!(b)) { + (V::N, V::N) => true, + (V::B(x), V::B(y)) => x == y, + (V::I(x), V::I(y)) => Object::Int(x).is_same(&Object::Int(y)), + (V::F(x), V::F(y)) => Object::Float(x).is_same(&Object::Float(y)), + // SAFETY: as `norm`. + (V::R(p), V::R(q)) => unsafe { (*p).is_same(&*q) }, + _ => false, + }; + set!(dst, V::B(same != invert)); + } + Op::Truth { dst, src } => set!(dst, V::B(truth(get!(src))?)), + Op::Unary { dst, src, kind } => { + let v = get!(src); + let r = match kind { + k if k == UnaryKind::Not as u8 => V::B(!truth(v)?), + k if k == UnaryKind::Neg as u8 => match v { + V::I(i) => V::I(i.checked_neg()?), + V::F(x) => V::F(-x), + _ => return None, + }, + k if k == UnaryKind::Pos as u8 => match v { + V::I(_) | V::F(_) => v, + _ => return None, + }, + k if k == UnaryKind::Invert as u8 => match v { + V::I(i) => V::I(!i), + _ => return None, + }, + _ => return None, + }; + set!(dst, r); + } + Op::Binary { dst, a, b, kind } => { + let r = match (get!(a), get!(b)) { + (V::I(a), V::I(b)) => V::I(match kind { + k if k == BinOpKind::Add as u8 => a.checked_add(b)?, + k if k == BinOpKind::Sub as u8 => a.checked_sub(b)?, + k if k == BinOpKind::Mult as u8 => a.checked_mul(b)?, + k if k == BinOpKind::BitAnd as u8 => a & b, + k if k == BinOpKind::BitOr as u8 => a | b, + k if k == BinOpKind::BitXor as u8 => a ^ b, + k if k == BinOpKind::RShift as u8 && (0..64).contains(&b) => a >> b, + k if k == BinOpKind::LShift as u8 + && (0..63).contains(&b) + && ((a << b) >> b) == a => + { + a << b + } + k if k == BinOpKind::FloorDiv as u8 && b > 0 && a >= 0 => a / b, + k if k == BinOpKind::Mod as u8 && b > 0 && a >= 0 => a % b, + _ => return None, + }), + (x, y) => { + let (x, y) = match (x, y) { + (V::F(x), V::F(y)) => (x, y), + (V::I(x), V::F(y)) => (x as f64, y), + (V::F(x), V::I(y)) => (x, y as f64), + _ => return None, + }; + // SAFETY: as the core loop's `BINARY_OP` arm + // (the compiler emits only valid kinds). + let kind: BinOpKind = unsafe { std::mem::transmute(kind) }; + match Self::leaf_float_op(x, y, kind)? { + Object::Float(x) => V::F(x), + _ => return None, + } + } + }; + set!(dst, r); + } + Op::Swap { a, b } => { + let (x, y) = (get!(a), get!(b)); + set!(a, y); + set!(b, x); + } + Op::BranchIf { src, when, target } => { + if truth(get!(src))? == when { + ip = usize::from(target); + } + } + Op::BranchNone { src, when, target } => { + if matches!(get!(src), V::N) == when { + ip = usize::from(target); + } + } + Op::Jump { target } => ip = usize::from(target), + Op::New { dst, dict } => { + // Tracked like the core loop's; an allocation that + // would trigger a collection is left to it. + if crate::stdlib::tracemalloc_real::is_tracking() + || crate::stdlib::testinternalcapi_mod::reftrace_print_active() + || crate::gc_trace::auto_collect_due() + { + return None; + } + let obj = if dict { + Object::Dict(Rc::new(crate::sync::RefCell::new( + crate::object::DictData::with_capacity_and_hasher( + 0, + crate::fasthash::FxBuildHasher, + ), + ))) + } else { + Object::new_list(Vec::new()) + }; + crate::gc_trace::track(obj.clone()); + set!(dst, owned.own(obj)?); + } + Op::Decline => return None, + Op::Call { at, argc, pc } => { + // A pure-leaf callee, evaluated in place: only while + // no store is buffered (it would not see one). + if (EFFECT && pend.n > 0) || usize::from(nest) + depth >= usize::from(NEST) { + return None; + } + let callee = match get!(at) { + V::Fn(fp) => Callee::Py(fp), + V::Bi(b) => Callee::Native(b, None), + // SAFETY: as `norm`. + V::R(p) => match unsafe { &*p } { + Object::Function(func) => Callee::Py(Rc::as_ptr(func)), + Object::Builtin(b) => Callee::Native(Rc::as_ptr(b), Some(b)), + _ => return None, + }, + _ => return None, + }; + let first = if matches!(get!(at + 1), V::Null) { + at + 2 + } else { + at + 1 + }; + let n = usize::from(at + 2 + argc - first); + if n > 8 { + return None; + } + let fp = match callee { + Callee::Py(fp) => fp, + Callee::Native(b, rc) => { + // A read-only builtin, on borrowed copies of the + // arguments: never dropped, so no reference moves. + let mut staged = + [const { std::mem::MaybeUninit::::uninit() }; 8]; + #[allow(clippy::needless_range_loop)] + for k in 0..n { + let o = match get!(first + k as u8) { + // SAFETY: as `norm`; the copy is forgotten. + V::R(p) => unsafe { std::ptr::read(p) }, + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Bi(_) | V::Null => return None, + }; + staged[k].write(o); + } + // SAFETY: the first `n` entries were written. + let args = unsafe { + std::slice::from_raw_parts(staged.as_ptr().cast::(), n) + }; + let r = self.leaf_pure_builtin(code, usize::from(pc), b, rc, args)?; + set!(at, owned.own(r)?); + continue; + } + }; + // SAFETY: the class or the namespace holds the callee, + // and nothing here runs code that could release it. + let callee = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the code cell. + let ccode: &Rc = unsafe { &*callee.code.as_ptr() }; + // (A positional call binds no `**kwargs` dictionary.) + if !crate::code_is_pure_leaf(ccode) + || !Self::leaf_code_ok(ccode) + || n != crate::leaf_arity(ccode) + || ccode.has_varkeywords + || crate::recursion::current_depth() + usize::from(nest) + depth + 1 + >= crate::recursion::recursion_limit() + { + return None; + } + let cext = crate::code_vm_ext(ccode)?; + // A tiny certified shape (a constant or field return, + // a field comparison) keeps its dedicated evaluator. + if !GETTER && cext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) < 3 { + // The callee runs as a frame of this evaluation: + // the caller's registers wait in its record, and + // the arguments (normalized already) become the + // callee's first registers. + let cplan = cext + .leaf_plan + .get_or_init(|| build(ccode, cext).map(Box::new)) + .as_deref()?; + let mut cargs = [V::N; 8]; + for (k, slot) in cargs.iter_mut().enumerate().take(n) { + let v = get!(first + k as u8); + if matches!(v, V::Fn(_) | V::Bi(_) | V::Null) { + return None; + } + *slot = v; + } + // SAFETY: `depth < NEST` (checked above). + let c = unsafe { callers.get_unchecked_mut(depth) }.as_mut_ptr(); + // SAFETY: `c` is this frame's record; the saved + // registers are the caller plan's own, all below + // `REGS`. + unsafe { + std::ptr::addr_of_mut!((*c).ip).write(ip); + std::ptr::addr_of_mut!((*c).code).write(code); + std::ptr::addr_of_mut!((*c).ext).write(ext); + std::ptr::addr_of_mut!((*c).plan).write(plan); + std::ptr::addr_of_mut!((*c).f).write(f); + std::ptr::addr_of_mut!((*c).at).write(at); + std::ptr::copy_nonoverlapping( + regs.as_ptr(), + std::ptr::addr_of_mut!((*c).saved).cast(), + usize::from(plan.nregs), + ); + } + for (k, &v) in cargs.iter().enumerate().take(n) { + regs[k].write(v); + } + depth += 1; + (code, ext, plan, f) = (&**ccode, cext, cplan, callee); + consts = &plan.consts; + stamps = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + ops = &plan.ops; + ip = 0; + continue; + } + // Scalar arguments are staged as objects (no drop glue). + let mut staged = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut ptrs = [std::ptr::null::(); 8]; + for k in 0..n { + let o = match get!(first + k as u8) { + V::R(p) => { + ptrs[k] = p; + continue; + } + V::I(i) => Object::Int(i), + V::F(x) => Object::Float(x), + V::B(b) => Object::Bool(b), + V::N => Object::None, + V::Fn(_) | V::Bi(_) | V::Null => return None, + }; + ptrs[k] = staged[k].write(o); + } + match self.leaf_eval_nested(ccode, callee, &ptrs[..n], nest + 1)? { + // A constructor's stores land at once, so what it + // borrows from its fresh instance could move. + LeafRet::Borrowed(v) if !FRESH => set!(at, v), + LeafRet::Borrowed(v) => set!(at, owned.own(to_object(v)?)?), + LeafRet::Owned(r) => set!(at, owned.own(r)?), + } + } + Op::Return { src } => { + let v = get!(src); + if depth > 0 { + // Back to the caller's frame, the result in its + // register. A constructor's stores land at once, + // so what it borrows from its fresh instance could + // move: it holds its own reference. + let v = match v { + V::R(p) if FRESH && !owned.holds(p) => owned.own(to_object(v)?)?, + v => v, + }; + depth -= 1; + // SAFETY: frame `depth`'s record was written at its + // call (its registers as far as its `nregs`). + let c = unsafe { &*callers.get_unchecked(depth).as_ptr() }; + (ip, code, ext, plan, f) = (c.ip, c.code, c.ext, c.plan, c.f); + consts = &plan.consts; + stamps = ext.stamp_slots.get().map_or(&[], |s| &s[..]); + ops = &plan.ops; + // SAFETY: as above. + unsafe { + std::ptr::copy_nonoverlapping( + c.saved.as_ptr(), + regs.as_mut_ptr(), + usize::from(plan.nregs), + ); + } + set!(c.at, v); + continue; + } + // A value this evaluation's scratch doesn't hold outlives + // it (a pure body stores nothing), so a nested caller + // borrows it rather than taking a reference to release. + if !EFFECT && !matches!(v, V::R(p) if owned.holds(p)) { + return Some(LeafRet::Borrowed(v)); + } + let r = to_object(v)?; + if EFFECT && pend.n > 0 { + // The latest store to each attribute is the one + // that lands; every one of them must go through + // before any does (a decline touched nothing, and + // the ordinary call runs the body instead). A + // lone store is its own check: it declines whole. + let unique = plan.unique_stores; + if pend.n > 1 { + // With one receiver, appends of new attributes + // land in order: `cursor` is its split length + // once the earlier ones have. + let mut cursor = None; + // A store the shortcut can't vouch for may + // move the layout: the rest check in full. + let mut split_ok = unique; + for k in 0..pend.n { + if !unique && !pend.latest(k) { + continue; + } + let (rp, spc, name, _) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + let (spc, name) = (*spc as usize, *name); + let ready = if split_ok { + Self::core_store_attr_ready_split(ext, inst, spc, &mut cursor) + } else { + None + }; + split_ok &= ready.is_some(); + if !ready.unwrap_or_else(|| { + Self::core_store_attr_ready(code, inst, spc, name) + }) { + return None; + } + } + } + let n = pend.n; + for k in 0..n { + if !unique && !pend.latest(k) { + continue; + } + let (rp, spc, name, value) = pend.get(k); + // SAFETY: stable receivers (see `Store`). + let Object::Instance(inst) = (unsafe { &**rp }) else { + return None; + }; + // On `true` the value moved into the dict. + if Self::core_store_attr(code, inst, *spc as usize, *name, value) { + pend.moved |= 1 << k; + } else if n == 1 { + // The lone store declined, untouched. + return None; + } else { + debug_assert!(false, "a ready store declined"); + } + } + } + return Some(LeafRet::Owned(r)); + } + Op::StoreAttr { + recv, + val, + pc, + name, + } => { + if !EFFECT { + return None; + } + let V::R(rp) = get!(recv) else { + return None; + }; + // SAFETY: as `norm`. + let recv = unsafe { &*rp }; + if FRESH && args.first().is_some_and(|&a| std::ptr::eq(a, rp)) { + let Object::Instance(inst) = recv else { + return None; + }; + let value = std::mem::ManuallyDrop::new(to_object(get!(val))?); + // On `true` the value moved into the instance. + if !Self::core_store_attr( + code, + inst, + usize::from(pc), + u32::from(name), + &value, + ) { + drop(std::mem::ManuallyDrop::into_inner(value)); + return None; + } + continue; + } + if !matches!(recv, Object::Instance(_)) || pend.n == PENDING { + return None; + } + let value = to_object(get!(val))?; + // An argument receiver is used in place; any other + // (a field's value) is held by the owned scratch. + let rp = if args.iter().any(|&a| std::ptr::eq(a, rp)) { + rp + } else { + let V::R(held) = owned.own(crate::clone_hot(recv))? else { + return None; + }; + held + }; + pend.buf[pend.n].write((rp, u32::from(pc), u32::from(name), value)); + pend.n += 1; + } + } + } + } +} diff --git a/crates/weavepy-vm/src/lib.rs b/crates/weavepy-vm/src/lib.rs index 44777f5a..dcbfecb3 100644 --- a/crates/weavepy-vm/src/lib.rs +++ b/crates/weavepy-vm/src/lib.rs @@ -54,12 +54,15 @@ pub mod foreign; pub mod frozen_code_cache; pub mod frozen_table; pub mod gc_trace; +mod gen_fast; pub mod gil; pub mod hot_filter; pub mod hot_gates; pub mod import; pub mod import_time; +pub mod inst_dict; mod lazy_arc; +mod leaf_plan; pub mod linejump; pub mod malloc_stats; pub mod object; @@ -67,18 +70,19 @@ pub mod proc_init; pub mod py_errno; pub mod pycache; pub mod rare_events; +pub mod rc; pub mod recursion; pub mod shared_value; pub mod specialize; pub mod stdlib; pub mod stdlib_tree; pub mod sync; -pub mod tcache; pub mod thread_registry; /// RFC 0032 — tier-2 Cranelift JIT integration. Present only under the /// `jit` feature; the dispatch loop calls into it behind `#[cfg]` gates. #[cfg(feature = "jit")] mod tier2; +mod timsort; pub mod trace; mod tuple_storage; pub mod type_surface; @@ -106,7 +110,10 @@ use crate::types::{PyInstance, TypeObject}; // ---------- frame ---------- -struct Frame { +/// An activation's execution state. Public only so a suspended generator +/// can own one ([`GeneratorState`]); its fields are private to the VM. +#[doc(hidden)] +pub struct Frame { code: Rc, /// Local variables, indexed by `LOAD_FAST` / `STORE_FAST`. /// @@ -228,6 +235,10 @@ struct Frame { /// for later resumptions. Set by `generator_send` when unparking a /// `Created` frame; consumed (reset) by `run_frame`'s entry event. gen_first_resume: bool, + /// A suspended generator frame whose last resume already consumed its + /// sent value (a fast step stopped partway; see `gen_fast`): the next + /// resume pushes none. + sent_consumed: bool, /// RFC 0061 (WS3c): the generator-family activation's `FrameShell`, /// kept across suspensions. A suspended frame survives inside the /// generator object, so its shell — whose immutable fields (code, @@ -247,7 +258,17 @@ struct Frame { parked_native: Option>, } +impl std::fmt::Debug for Frame { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("Frame") + .field("code", &self.code.qualname) + .field("pc", &self.pc) + .finish_non_exhaustive() + } +} + impl Frame { + #[inline] fn push(&mut self, v: Object) { self.stack.push(v); } @@ -359,7 +380,8 @@ fn generator_frame_traverse(obj: &Object, visit: &mut dyn FnMut(&Object)) { GeneratorState::Created(b) | GeneratorState::Suspended(b) => b, _ => return, }; - if let Some(frame) = boxed.downcast_ref::() { + let frame: &Frame = boxed; + { if let Ok(locals) = frame.locals.try_borrow() { for v in locals.iter() { visit(v); @@ -482,9 +504,7 @@ fn frame_reapables(g: &Rc) -> Vec { GeneratorState::Created(b) | GeneratorState::Suspended(b) => b, _ => return Vec::new(), }; - let Some(frame) = boxed.downcast_ref::() else { - return Vec::new(); - }; + let frame: &Frame = boxed; // CPython clears a generator's frame the instant it finishes/closes // (`gen_send`/`gen_close` → `_PyFrame_ClearExceptCode`), so every // local and value-stack entry is decref'd promptly. WeavePy mirrors @@ -528,6 +548,14 @@ fn frame_reapables(g: &Rc) -> Vec { out } +/// Warm the JIT's code generator on a background thread when the JIT +/// is on (see `tier2::prewarm_codegen`): a program's first compile then +/// doesn't pay the generator's cold start. +pub fn spawn_jit_codegen_prewarm() { + #[cfg(feature = "jit")] + tier2::spawn_codegen_prewarm_if_enabled(); +} + /// RFC 0032 — render the tier-2 JIT's counters as a markdown block for /// the `WEAVEPY_VM_STATS` report, or `None` when the `jit` feature is /// disabled or the JIT was never exercised on this thread. @@ -912,22 +940,23 @@ pub struct Interpreter { /// chains keep the ~[`crate::gil::GIL_CHECK_INTERVAL`]-opcode /// checkpoint cadence. gil_countdown: u32, - /// `sum(generator)`'s accumulator (its address), offered to the next - /// lean generator resume (see `generator_send_lean`), which pairs it - /// with the generator's frame in [`Self::sum_fold`]. - sum_fold_acc: Option, - /// While a lean resume runs on behalf of `sum()`: the generator's - /// frame and the accumulator (addresses). The core loop folds that - /// frame's scalar yields into the accumulator and resumes in place - /// (see the `YIELD_VALUE` arm) instead of leaving the quiet loop per - /// item. - sum_fold: Option<(usize, usize)>, + /// While a lean resume runs on behalf of a consumer that drains the + /// generator (`sum()`, `list()`, see [`FoldSink`]): the generator's + /// frame (its address) and the consumer's sink. The core loop folds + /// that frame's yields into the sink and resumes in place (see the + /// `YIELD_VALUE` arm) instead of leaving the quiet loop per item. + sum_fold: Option<(usize, FoldSink)>, /// RFC 0061 (WS2b) — set while any observer (trace/profile/PEP 669 /// tool) is active on this thread. Fused-dispatch arms check it and /// fall back to single-step semantics, so instrumentation sees the /// exact per-instruction event stream. Refreshed from the dispatch /// loop's [`crate::trace::ObserverSnapshot`] every iteration. fuse_off: bool, + /// A handled exception was just released (`POP_EXCEPT`): the running + /// frame's object may have lost its last outside holder (the + /// traceback), so the dispatch loop re-probes it on the next + /// instruction instead of waiting for its stride. + recheck_frame_observed: bool, /// `WEAVEPY_NO_QUIET`: pin every dispatch-loop iteration to the full /// prologue (RFC 0065 bisection aid). Read once at construction so a /// frame entry does not pay a `OnceLock` probe for it (RFC 0077 WS3). @@ -949,6 +978,13 @@ pub struct Interpreter { /// The registered leaf builtins (see [`leaf_builtins`]) as last /// snapshotted, with the registry generation they reflect. leaf_opaque: ThreadCell<(u64, leaf_builtins::LeafMap)>, + /// The core loop's last native iterator `__next__`, keyed by the + /// (process-unique) attribute version of the class that resolved it + /// (see [`Interpreter::core_leaf_next`]). + core_next: ThreadCell>)>>, + /// How the last class seen in a boolean context answers it (see + /// [`Interpreter::leaf_instance_truth`]), by attribute version. + core_truth: ThreadCell>, /// `WP_DBG_SAMPLE`: periodic frame-entry sampling to stderr. dbg_sample: bool, } @@ -976,7 +1012,7 @@ impl Default for Interpreter { // `_Py_SetLocaleFromEnv`) so `_locale.getencoding()` and // `locale.setlocale(..., None)` observe the user's locale. crate::stdlib::locale_mod::init_from_env(); - let stdout: Stdout = Rc::new(RefCell::new(std::io::stdout())); + let stdout: Stdout = Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stdout()))); let mut builtins_dict = builtins::default_builtins(); // The `builtins` module exposes the core types/exceptions as the // real `type` objects (CPython's `builtins.int is int`), not the @@ -1167,9 +1203,9 @@ impl Default for Interpreter { frame_stack_pool: ThreadCell::new(Vec::new()), scratch_pool: ThreadCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, - sum_fold_acc: None, sum_fold: None, fuse_off: false, + recheck_frame_observed: false, quiet_off: crate::hot_gates::env_flags::no_quiet(), burst_on: !crate::hot_gates::env_flags::no_burst() && !crate::hot_gates::env_flags::no_quiet() @@ -1177,6 +1213,8 @@ impl Default for Interpreter { lean_pending: Vec::new(), leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), + core_next: ThreadCell::new(None), + core_truth: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), }; // RFC 0025: publish the shared parts of this interpreter @@ -1292,9 +1330,9 @@ impl Interpreter { frame_stack_pool: ThreadCell::new(Vec::new()), scratch_pool: ThreadCell::new(Vec::new()), gil_countdown: crate::gil::GIL_CHECK_INTERVAL, - sum_fold_acc: None, sum_fold: None, fuse_off: false, + recheck_frame_observed: false, quiet_off: crate::hot_gates::env_flags::no_quiet(), burst_on: !crate::hot_gates::env_flags::no_burst() && !crate::hot_gates::env_flags::no_quiet() @@ -1302,6 +1340,8 @@ impl Interpreter { lean_pending: Vec::new(), leaf_fns: std::cell::OnceCell::new(), leaf_opaque: ThreadCell::new((0, leaf_builtins::LeafMap::default())), + core_next: ThreadCell::new(None), + core_truth: ThreadCell::new(None), dbg_sample: crate::hot_gates::env_flags::dbg_sample(), } } @@ -2726,8 +2766,10 @@ impl Interpreter { let Ok(attrs) = f.attrs.try_borrow() else { return false; }; - if Rc::strong_count(&attrs) == 1 && !attrs.try_borrow().is_ok_and(|d| d.is_empty()) { - return false; + if let Some(attrs) = attrs.as_ref() { + if Rc::strong_count(attrs) == 1 && !attrs.try_borrow().is_ok_and(|d| d.is_empty()) { + return false; + } } let Ok(slots) = f.slots.try_borrow() else { return false; @@ -2769,11 +2811,20 @@ impl Interpreter { let mut n = 0; // A dict shared through `vars(obj)` outlives the instance, and // with it every value. - let Some(dict) = inst.dict.get().filter(|_| inst.dict.strong_count() == 1) else { - return Some((leaves, n)); + let split; + let d; + let attrs: &mut dyn Iterator = match inst.dict.published() { + Some(dict) if inst.dict.strong_count() == 1 => { + d = dict.try_borrow().ok()?; + &mut d.iter() + } + Some(_) => return Some((leaves, n)), + None => { + split = inst.dict.split_cell().try_borrow().ok()?; + &mut split.iter() + } }; - let d = dict.try_borrow().ok()?; - for (k, v) in d.iter() { + for (k, v) in attrs { if !gc_trace::is_atomic(&k.0) { return None; } @@ -2940,6 +2991,17 @@ impl Interpreter { // test below keeps `strong` above the dead threshold and leaves // the cycle to the tracing collector. let mut work = vec![dropped]; + // Each node's scratch, emptied per node and allocated once per + // cascade (a dead graph of many small containers paid four + // allocations and a hash set per node). + let mut children: Vec> = Vec::new(); + let mut weakref_candidates: Vec = Vec::new(); + let mut pool_candidates: Vec = Vec::new(); + let mut scan_through: Vec = Vec::new(); + let mut scanned: std::collections::HashSet< + crate::weakref_registry::ObjectId, + std::hash::BuildHasherDefault, + > = std::collections::HashSet::default(); loop { let Some(obj) = work.pop() else { // Freeing the objects above may have parked C-side dealloc @@ -3074,7 +3136,7 @@ impl Interpreter { // unpickler temporary holding the memo) cascades through the // untracked `memo` dict to the tracked argument. The refcount guard // below still filters anything that stays externally reachable. - let mut child_ids: Vec = Vec::new(); + debug_assert!(children.is_empty()); // Untracked descendants with live weakrefs: CPython clears an // object's weakrefs at refcount zero whether or not the GC ever // tracked it, but this cascade's "next link" set is (otherwise) @@ -3089,18 +3151,35 @@ impl Interpreter { // `has_reference()` flips to False). Collect them during the // scan and run the dead ones through the cascade so their // weakrefs clear at death, like any tracked link. - let mut weakref_candidates: Vec = Vec::new(); + debug_assert!(weakref_candidates.is_empty()); // Plain-tuple children about to die with `obj` are recycled // through the tuple pool (see `maybe_donate_tuple`): the plain // `Rc` drop below frees them invisibly, and this cascade is // where the common `f(*args)`-shaped temporaries actually die // (the argument tuple is anchored by a dead list/frame local). - let mut pool_candidates: Vec = Vec::new(); - let mut scan_through: Vec = vec![obj.clone()]; - let mut scanned: std::collections::HashSet = - std::collections::HashSet::new(); + debug_assert!(pool_candidates.is_empty()); + scan_through.push(obj.clone()); + scanned.clear(); while let Some(parent) = scan_through.pop() { gc_trace::traverse_object(&parent, &mut |c| { + // Scalars are never tracked, weakly referenced, or + // anchors of anything; a deferred-tracking instance + // holds only scalars and has no collector handle. + // Neither can lead the cascade anywhere, which matters + // for the commonest large death: a list of plain + // objects falling out of scope. + if gc_trace::is_atomic(c) { + return; + } + if let Object::Instance(i) = c { + if i.is_gc_deferred() && i.c_body.get() == 0 { + let cid = crate::weakref_registry::id_of(c); + if crate::weakref_registry::count_for(cid) > 0 && scanned.insert(cid) { + weakref_candidates.push(c.clone()); + } + return; + } + } let cid = crate::weakref_registry::id_of(c); if let Object::Tuple(t) = c { if !t.is_empty() @@ -3111,8 +3190,8 @@ impl Interpreter { pool_candidates.push(c.clone()); } } - if gc_trace::is_tracked(cid) { - child_ids.push(cid); + if let Some(h) = gc_trace::find_handle(cid) { + children.push(h); } else if matches!( c, Object::Tuple(_) @@ -3173,16 +3252,16 @@ impl Interpreter { // A tuple child whose last holder was `obj` is now ours alone // (strong count 1 = the scan's clone); park its allocation. // Anything still referenced elsewhere just sheds the clone. - for cand in pool_candidates { + for cand in pool_candidates.drain(..) { if matches!(&cand, Object::Tuple(t) if ThinArc::strong_count(t) == 1) { self.maybe_donate_tuple(cand); } } // Any tracked child that just lost its last program reference // is the next link in the chain. - for cid in child_ids { - if let Some(h) = gc_trace::find_handle(cid) { - let weak = crate::weakref_registry::strong_clone_count(cid); + for h in children.drain(..) { + if !h.untracked.load(std::sync::atomic::Ordering::Acquire) { + let weak = crate::weakref_registry::strong_clone_count(h.id); // `h.object` is the GC's own strong reference; the // child is dead iff nothing beyond that handle and its // weakref clones still points at it. @@ -3207,7 +3286,7 @@ impl Interpreter { // its weakrefs clear now (the refcount guard at the top of the // loop re-checks liveness — our `weakref_candidates` clone is // accounted for by the cascade's `extra` slack of 1). - for cand in weakref_candidates { + for cand in weakref_candidates.drain(..) { if Self::is_refcount_dead(&cand, 1) { work.push(cand); } @@ -3267,6 +3346,10 @@ impl Interpreter { return false; } let id = crate::weakref_registry::id_of(v); + // Held past every slot and a collector handle: no count. + if sc > nlocals + 1 { + return !crate::weakref_registry::may_have_weakrefs(id); + } // How many of this frame's slots hold `v`: exact for // small frames (the usual `self` is held once here and // once by the caller), the slot count otherwise. @@ -3291,9 +3374,17 @@ impl Interpreter { !(Self::local_needs_prompt_reap(v) || matches!(v, Object::Function(_))) || escaped(v) || matches!(v, Object::Instance(i) if i.dies_by_plain_drop()) + || Self::atomic_container_dies_plainly(v) + || Self::tracked_container_dies_inertly(v) }) { for v in locals { - gc_trace::note_dropped(v); + // A dying tracked container sheds its collector handle + // now (the cascade would find nothing else to do). + if Self::tracked_container_dies_inertly(v) { + gc_trace::untrack_id(crate::weakref_registry::id_of(v)); + } else { + gc_trace::note_dropped(v); + } } return; } @@ -3771,6 +3862,74 @@ impl Interpreter { /// generators/coroutines/async-generators (their `close()` delivers /// `GeneratorExit`). Kept deliberately narrow so the hot return path of /// scalar-only frames pays a single cheap `matches!` per local. + /// A small untracked list, tuple or dict of atomic values held only + /// here: its release frees nothing a reap would visit (no finalizer, + /// weakref or collector handle can hang off it or its values). + fn atomic_container_dies_plainly(o: &Object) -> bool { + const MAX: usize = 16; + fn atomic(o: &Object) -> bool { + matches!( + o, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Str(_) + ) + } + let (sc, id) = match o { + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + Object::Tuple(t) => ( + ThinArc::strong_count(t), + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ), + _ => return false, + }; + if sc != 1 || gc_trace::maybe_tracked(id) || crate::weakref_registry::may_have_weakrefs(id) + { + return false; + } + match o { + Object::List(l) => l + .try_borrow() + .is_ok_and(|v| v.len() <= MAX && v.iter().all(atomic)), + Object::Dict(d) => d + .try_borrow() + .is_ok_and(|d| d.len() <= MAX && d.iter().all(|(k, v)| atomic(&k.0) && atomic(v))), + Object::Tuple(t) => t.len() <= MAX && t.iter().all(atomic), + _ => false, + } + } + + /// A collector-tracked list or dict whose only holders are the caller's + /// reference and its collector handle, watched by no weakref, holding + /// only values its death can't finalize (atomic ones, or instances + /// others still hold): untracking it and dropping the reference is its + /// whole teardown, as in CPython's `list_dealloc`. + fn tracked_container_dies_inertly(o: &Object) -> bool { + const MAX: usize = 16; + fn inert(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Str(_) + ) || gc_trace::drop_survives_plainly(v) + } + let (sc, id) = match o { + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + _ => return false, + }; + if sc != 2 || crate::weakref_registry::may_have_weakrefs(id) || !gc_trace::is_tracked(id) { + return false; + } + match o { + Object::List(l) => l + .try_borrow() + .is_ok_and(|v| v.len() <= MAX && v.iter().all(inert)), + Object::Dict(d) => d + .try_borrow() + .is_ok_and(|d| d.len() <= MAX && d.iter().all(|(k, v)| inert(&k.0) && inert(v))), + _ => false, + } + } + fn local_needs_prompt_reap(o: &Object) -> bool { matches!( o, @@ -3865,6 +4024,45 @@ impl Interpreter { } } + /// The common handled exception, recognized without the reap walks: a + /// class with no finalizer, no instance dict, and slots holding only + /// atomic values, a tuple of them (`args`), or a traceback whose frames + /// are all still executing. Neither `exc_has_finalizable` nor + /// `anchors_tracked_child` can find anything in it. + fn exc_plainly_inert(obj: &Object) -> bool { + let Object::Instance(inst) = obj else { + return false; + }; + if inst.cls().instances_need_finalize() + || inst + .dict + .get() + .is_some_and(|d| d.try_borrow().map_or(true, |d| !d.is_empty())) + { + return false; + } + let Ok(slots) = inst.slots.try_borrow() else { + return false; + }; + // (Bound, not returned directly: the iterator borrows `slots`.) + #[allow(clippy::let_and_return)] + let inert = slots.iter().all(|(_, v)| match v { + Object::Tuple(t) => t.iter().all(Object::is_gc_atomic), + Object::Traceback(tb) => { + let mut cur = Some(tb.clone()); + while let Some(node) = cur { + if node.frame.on_stack.get() == 0 { + return false; + } + cur = node.next.borrow().clone(); + } + true + } + other => other.is_gc_atomic(), + }); + inert + } + /// Cheap, allocation-free pre-check for the `POP_EXCEPT` reap: does the /// acyclic subgraph rooted at `obj` contain a finalizable object /// (`__del__` / an unfinished generator) reachable through value @@ -3935,6 +4133,7 @@ impl Interpreter { || *budget == 0 || c.is_gc_atomic() || matches!(c, Object::Frame(f) if f.on_stack.get() > 0) + || matches!(c, Object::Instance(i) if i.is_gc_deferred() && i.c_body.get() == 0) { return; } @@ -4554,7 +4753,7 @@ impl Interpreter { Some(Object::Module(m)) => m.dict.borrow().iter().map(|(_k, v)| v.clone()).collect(), _ => Vec::new(), }; - let mut sys_deferred: Vec> = Vec::new(); + let mut sys_deferred: Vec> = Vec::new(); for _ in 0..8 { let candidates = crate::gc_trace::finalization_candidates(); if candidates.is_empty() { @@ -4564,7 +4763,7 @@ impl Interpreter { if sys_values.iter().any(|v| v.is_same(&handle.object)) { if !sys_deferred .iter() - .any(|h| std::sync::Arc::ptr_eq(h, &handle)) + .any(|h| crate::sync::Rc::ptr_eq(h, &handle)) { sys_deferred.push(handle); } @@ -4893,8 +5092,7 @@ impl Interpreter { let mut cur = Some(tb); while let Some(node) = cur { entries.push(crate::error::TracebackEntry { - filename: node.frame.code.filename.clone(), - funcname: node.frame.code.name.clone(), + code: node.frame.code.clone(), lineno: node.lineno, }); cur = node.next.borrow().clone(); @@ -4948,7 +5146,9 @@ impl Interpreter { for e in traceback { s.push_str(&format!( " File \"{}\", line {}, in {}\n", - e.filename, e.lineno, e.funcname + e.filename(), + e.lineno, + e.funcname() )); } } @@ -5519,6 +5719,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -6024,7 +6225,8 @@ impl Interpreter { // ran. A few extra instructions on the full path after the // holder goes away cost far less. let rederive = frame_blocks_quiet - && self.gil_countdown.trailing_zeros() >= 4 + && (std::mem::take(&mut self.recheck_frame_observed) + || self.gil_countdown.trailing_zeros() >= 4) && !Self::frame_object_observed(&shell, py_frame_slot.as_ref(), frame); if lgen != loop_snap_gen || rederive { loop_snap_gen = lgen; @@ -7770,9 +7972,9 @@ impl Interpreter { return g.code.clone(); } match &*g.state.borrow() { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => boxed - .downcast_ref::() - .map_or(Object::None, |f| Object::Code(f.code.clone())), + GeneratorState::Created(frame) | GeneratorState::Suspended(frame) => { + Object::Code(frame.code.clone()) + } GeneratorState::Running | GeneratorState::Finished => Object::None, } } @@ -7786,33 +7988,28 @@ impl Interpreter { fn gen_py_frame(&self, g: &Rc) -> Object { let mut state = g.state.borrow_mut(); match &mut *state { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => { - match boxed.downcast_mut::() { - Some(frame) => { - // RFC 0073 WS4 — a Python-visible frame shares - // the locals storage; write a parked native - // activation back before exposing one (park - // refuses whenever a `PyFrame` already exists, - // so this is the only creation path to guard). - #[cfg(feature = "jit")] - crate::tier2::materialize_parked(frame); - if let Some(py) = frame.py_frame.clone() { - if py.gen_owner.borrow().is_none() { - *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); - } - return Object::Frame(py); - } - // Not yet entered: build the frame snapshot now so - // `gi_frame` is observable before the first - // `next()`. `back` is None — a created/suspended - // generator frame has no live caller. - let py = self.build_py_frame(frame, None); + GeneratorState::Created(frame) | GeneratorState::Suspended(frame) => { + // RFC 0073 WS4 — a Python-visible frame shares + // the locals storage; write a parked native + // activation back before exposing one (park + // refuses whenever a `PyFrame` already exists, + // so this is the only creation path to guard). + #[cfg(feature = "jit")] + crate::tier2::materialize_parked(frame); + if let Some(py) = frame.py_frame.clone() { + if py.gen_owner.borrow().is_none() { *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); - frame.py_frame = Some(py.clone()); - Object::Frame(py) } - None => Object::None, + return Object::Frame(py); } + // Not yet entered: build the frame snapshot now so + // `gi_frame` is observable before the first + // `next()`. `back` is None — a created/suspended + // generator frame has no live caller. + let py = self.build_py_frame(frame, None); + *py.gen_owner.borrow_mut() = Some(Rc::downgrade(g)); + frame.py_frame = Some(py.clone()); + Object::Frame(py) } // Running: the frame is live on the interpreter call stack, not // in the box — find it by its generator backlink. CPython's @@ -7873,8 +8070,8 @@ impl Interpreter { /// `SEND`/`YIELD_VALUE` pair with the delegate at top-of-stack. fn gen_yieldfrom(&self, g: &Rc) -> Object { let state = g.state.borrow(); - if let GeneratorState::Suspended(boxed) = &*state { - if let Some(frame) = boxed.downcast_ref::() { + if let GeneratorState::Suspended(frame) = &*state { + { let pc = frame.pc as usize; if pc >= 2 { if let Some(send_ins) = frame.code.instructions.get(pc - 2) { @@ -9319,9 +9516,12 @@ impl Interpreter { // complete before they run. self.flush_lean(frame, shell); self.drain_if_maybe_dead(); - if crate::hot_gates::loop_gen() != snap_gen { - break 'run QuietExit::Yield; - } + } + // The drop may also have queued work of its own (an + // unclosed file's ResourceWarning), which CPython + // delivers at the instruction that dropped it. + if crate::hot_gates::loop_gen() != snap_gen { + break 'run QuietExit::Yield; } continue; } @@ -9420,7 +9620,13 @@ impl Interpreter { && matches!(frame.stack.last(), Some(Object::Generator(_))) && self.inline_calls_ok() { - if let Some(act) = self.try_inline_gen(frame, shell, cur_pc) { + if let Some(act) = self.try_inline_gen( + frame, + shell, + cur_pc, + crate::recursion::depth_cell(), + false, + ) { return FrameEv::Call(act); } } @@ -9633,7 +9839,6 @@ impl Interpreter { /// Park a finished inline slot (its callable already taken): release /// what the activation owned and return the slot to the pool. fn inline_park(&mut self, mut act: Box) { - const POOL_CAP: usize = 64; debug_assert!(!act.parked); let fr: &mut Frame = &mut act.frame; // Leftover operands drop before anything is reused (their drop @@ -9682,6 +9887,7 @@ impl Interpreter { fr.agen_yielded_value = true; fr.suppress_call_event = false; fr.gen_first_resume = false; + fr.sent_consumed = false; #[cfg(feature = "jit")] { fr.parked_native = None; @@ -9697,7 +9903,7 @@ impl Interpreter { act.guard = None; act.caller_pending = None; act.init_inst = None; - if self.inline_pool.len() < POOL_CAP { + if self.inline_pool.len() < INLINE_POOL_CAP { self.inline_pool.push(act); } } @@ -9705,20 +9911,31 @@ impl Interpreter { /// `FOR_ITER` over a suspended generator at `pc` of `frame`, as an /// inline activation: the generator's boxed frame runs in place (the /// `generator_send_lean` shape, without a nested native activation). + /// With `next_call`, the instruction is instead the `CALL` of builtin + /// `next` on the generator: the call's operands leave the stack, the + /// yielded value is its result, and exhaustion raises `StopIteration`. /// `None` leaves everything untouched. fn try_inline_gen( &mut self, frame: &mut Frame, shell: &mut QuietShell<'_>, pc: usize, + depth_cell: *const std::cell::Cell, + next_call: bool, ) -> Option> { - let arg = frame.code.instructions.get(pc)?.arg; + let arg = if next_call { + GEN_NEXT_CALL + } else { + frame.code.instructions.get(pc)?.arg + }; let Some(Object::Generator(g)) = frame.stack.last() else { return None; }; - // Validate and take the frame under one borrow. No Python runs - // while it is held; release it before resuming the activation. - let mut state = g.state.try_borrow_mut().ok()?; + // Validate and take the frame under one exclusive view. No Python + // runs while it is held; it ends before the activation resumes. + // SAFETY: nothing below reaches the cell again until `state`'s last + // use (`peek_mut` rejects a live guard or shared cells). + let state = unsafe { g.state.peek_mut() }?; let first_resume; { let boxed = match &*state { @@ -9732,7 +9949,7 @@ impl Interpreter { } _ => return None, }; - let gf = boxed.downcast_ref::()?; + let gf: &Frame = boxed; if gf.py_frame.is_some() || !gf.saved_exc_info.is_empty() || gf.pc == 0 @@ -9746,40 +9963,55 @@ impl Interpreter { } } // Past the recursion limit the lean path raises. - let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter_with(depth_cell) else { return None; }; // Committed. - let prev_state = std::mem::replace(&mut *state, GeneratorState::Running); - drop(state); + let prev_state = std::mem::replace(state, GeneratorState::Running); let g = g.clone(); let (GeneratorState::Suspended(mut boxed) | GeneratorState::Created(mut boxed)) = prev_state else { unreachable!("checked above"); }; - let gf = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let gf: &mut Frame = &mut boxed; // A native activation parked at the yield is rebuilt into the // interpreted suspension resumed here. #[cfg(feature = "jit")] crate::tier2::materialize_parked(gf); gf.gen_first_resume = first_resume; - gf.push(Object::None); + // (A fast step that stopped partway already consumed it.) + if !std::mem::take(&mut gf.sent_consumed) { + gf.push(Object::None); + } let gen_frame: *mut Frame = gf; frame.pc = pc as u32 + 1; + if next_call { + // `next`, its empty self slot and the generator (held above). + let n = frame.stack.len(); + drop(frame.stack.drain(n - 3..)); + } let mut act = self.inline_slot(); - act.gen = Some(g); - act.gen_box = Some(boxed); + // A pooled slot holds none of these (every path back to the pool + // takes them), so they're written without drop glue. + debug_assert!( + act.gen.is_none() + && act.gen_box.is_none() + && act.guard.is_none() + && act.act.shell.is_none() + ); + // SAFETY: each field is `None` (above), so nothing is leaked. + unsafe { + std::ptr::write(&raw mut act.gen, Some(g)); + std::ptr::write(&raw mut act.gen_box, Some(boxed)); + std::ptr::write(&raw mut act.guard, Some(guard)); + } act.gen_frame = gen_frame; act.exhaust_arg = arg; act.act.frame = gen_frame; - act.act.shell = None; act.call_pc = pc; act.caller_pending = self.lean_pending_enter(frame, shell, pc); act.exc_depth = self.exc_info_len(); - act.guard = Some(guard); Some(act) } @@ -9843,15 +10075,13 @@ impl Interpreter { Self::park_suspended_boxed(&gen, boxed); Ok(GenStep::Yielded(v)) } - Ok(FrameOutcome::Returned(_)) => { + Ok(FrameOutcome::Returned(v)) => { *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("a generator activation's frame"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(&gen); - Ok(GenStep::Exhausted) + Ok(GenStep::Exhausted(v)) } Ok(FrameOutcome::StartGenerator) => { *gen.state.borrow_mut() = GeneratorState::Finished; @@ -9862,9 +10092,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(&gen, err); - let frame = boxed - .downcast_mut::() - .expect("a generator activation's frame"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(&gen); @@ -9883,7 +10111,6 @@ impl Interpreter { mut done: Box, result: Result, ) -> QuietEntry { - const POOL_CAP: usize = 64; let call_pc = done.call_pc; let arg = done.exhaust_arg; self.lean_pending_exit(done.caller_pending); @@ -9892,7 +10119,7 @@ impl Interpreter { done.act.shell = None; done.act.frame = std::ptr::from_mut::(&mut done.frame); done.caller_pending = None; - if self.inline_pool.len() < POOL_CAP { + if self.inline_pool.len() < INLINE_POOL_CAP { self.inline_pool.push(done); } match result { @@ -9900,7 +10127,12 @@ impl Interpreter { frame.stack.push(v); QuietEntry::Returned { cur_pc: call_pc } } - Ok(GenStep::Exhausted) => { + // `next(gen)`: the return value rides the StopIteration. + Ok(GenStep::Exhausted(v)) if arg == GEN_NEXT_CALL => QuietEntry::Raised { + err: crate::error::stop_iteration_with(v), + cur_pc: call_pc, + }, + Ok(GenStep::Exhausted(_)) => { let it = frame.stack.pop(); frame.pc += arg; frame.skip_end_for(); @@ -10071,7 +10303,7 @@ impl Interpreter { /// code); a bound one is just graded. The usual callee — a function /// its namespace still holds (our clone, that binding, its collector /// handle: three or more) with no weakref — needs neither. - #[inline] + #[inline(always)] fn drop_lean_callable( &mut self, frame: &mut Frame, @@ -10087,7 +10319,20 @@ impl Interpreter { }; if alive_elsewhere { drop_hot(callable); - } else if Self::looks_reapable_temporary(&callable) { + } else { + self.drop_lean_callable_slow(frame, shell, callable); + } + } + + /// [`Self::drop_lean_callable`] for a callable that may die here. + #[inline(never)] + fn drop_lean_callable_slow( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + callable: Object, + ) { + if Self::looks_reapable_temporary(&callable) { self.flush_lean(frame, shell); self.prompt_reap_dropped(callable); } else { @@ -10335,13 +10580,19 @@ impl Interpreter { if code.is_generator || code.is_coroutine || code.is_async_generator - || code.has_varargs - || code.has_varkeywords - || code.kwonly_count != 0 || !Self::lean_code_ok(&code) { return None; } + // Keyword-only parameters bind their compiled defaults (a positional + // call supplies none of them); a replaced `__kwdefaults__` takes the + // generic binder. + if code.kwonly_count != 0 + && ((f.defaults_maybe_overridden() && f.slot("__kwdefaults__").is_some()) + || !Self::kwonly_defaults(f, &code).all(|d| d.is_some())) + { + return None; + } f.lean_cells_ref(&code)?; let missing = if eff_argc < total { if f.defaults.len() < total - eff_argc { @@ -10357,7 +10608,8 @@ impl Interpreter { } total - eff_argc } else { - if total != eff_argc { + // Surplus positionals go to `*args`. + if total != eff_argc && !code.has_varargs { return None; } 0 @@ -10365,6 +10617,46 @@ impl Interpreter { Some((code, missing)) } + /// Whether `code` binds parameters beyond its positional ones: `*args`, + /// `**kwargs`, or keyword-only parameters. + #[inline(always)] + fn has_extended_params(code: &CodeObject) -> bool { + code.has_varargs || code.has_varkeywords || code.kwonly_count != 0 + } + + /// How many trailing positional defaults a call of `f` (running + /// `code`) with `given` positional arguments, self included, takes + /// from `f.defaults`. `None` when too many arguments come, too few + /// for the defaults to fill, or `__defaults__` was rebound. + #[inline] + fn missing_defaults(f: &PyFunction, code: &CodeObject, given: usize) -> Option { + let missing = (code.arg_count as usize).checked_sub(given)?; + if missing > 0 + && (f.defaults.len() < missing + || (f.defaults_maybe_overridden() && f.slot("__defaults__").is_some())) + { + return None; + } + Some(missing) + } + + /// `f`'s compiled default for each of `code`'s keyword-only parameters, + /// in parameter order. + fn kwonly_defaults<'a>( + f: &'a PyFunction, + code: &'a CodeObject, + ) -> impl Iterator> + 'a { + let npos = code.arg_count as usize; + code.varnames[npos..npos + code.kwonly_count as usize] + .iter() + .map(|name| { + f.kw_defaults + .iter() + .find(|(n, _)| n == name.as_str()) + .map(|(_, v)| v) + }) + } + /// Move a shape-checked lean call's operands off `frame`'s stack: the /// arguments (a real self first) into fresh locals for `code`, the /// `missing` trailing parameters from the compiled defaults, the rest @@ -10407,17 +10699,21 @@ impl Interpreter { let nlocals = code.varnames.len(); let first_arg = if has_self { self_slot } else { self_slot + 1 }; v.reserve(nlocals); - v.extend(frame.stack.drain(first_arg..)); - if missing > 0 { - let Object::Function(f) = &frame.stack[callee_slot] else { - unreachable!("callee variant checked by the caller") - }; - // Defaults align right-to-left with the declared - // positionals, exactly like the generic binder's - // missing-tail fill. - v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + if Self::has_extended_params(code) { + Self::lean_call_fill_extended(frame, code, first_arg, callee_slot, missing, v); + } else { + v.extend(frame.stack.drain(first_arg..)); + if missing > 0 { + let Object::Function(f) = &frame.stack[callee_slot] else { + unreachable!("callee variant checked by the caller") + }; + // Defaults align right-to-left with the declared + // positionals, exactly like the generic binder's + // missing-tail fill. + v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + } } - v.resize(nlocals, Object::Unbound); + fill_unbound(v, nlocals); if !has_self { frame.stack.pop(); // the NULL self slot } @@ -10427,6 +10723,49 @@ impl Interpreter { .expect("callee slot checked by the caller") } + /// [`Self::lean_call_fill`] for a callee with `*args`, `**kwargs` or + /// keyword-only parameters (see [`Self::lean_call_shape`]): the + /// positional parameters, their missing defaults, each keyword-only + /// parameter's default, the surplus positionals as the `*args` tuple, + /// and an empty `**kwargs` dictionary, in the locals' parameter order. + #[cold] + #[inline(never)] + fn lean_call_fill_extended( + frame: &mut Frame, + code: &CodeObject, + first_arg: usize, + callee_slot: usize, + missing: usize, + v: &mut Vec, + ) { + let npos = code.arg_count as usize; + let direct = (frame.stack.len() - first_arg).min(npos); + v.extend(frame.stack.drain(first_arg..first_arg + direct)); + let star = code.has_varargs.then(|| { + if frame.stack.len() == first_arg { + // The interned empty tuple. + Object::new_tuple(Vec::new()) + } else { + Object::Tuple(crate::tuple_storage::TupleStorage::from_exact_iter( + frame.stack.drain(first_arg..), + )) + } + }); + let Object::Function(f) = &frame.stack[callee_slot] else { + unreachable!("callee variant checked by the caller") + }; + if missing > 0 { + v.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); + } + for d in Self::kwonly_defaults(f, code) { + v.push(d.cloned().unwrap_or(Object::Unbound)); + } + v.extend(star); + if code.has_varkeywords { + v.push(Object::Dict(Rc::new(RefCell::new(DictData::default())))); + } + } + /// The inline half of `CALL`: `try_lean_call`'s plain-function shapes /// with the callee's frame boxed for the quiet loop to run in place /// (see [`InlineAct`]). `None` leaves everything untouched. @@ -10572,6 +10911,42 @@ impl Interpreter { Some((code, covered)) } + /// [`Self::kw_names_bind_check`] for the `CALL_KW` at `pc` of + /// `caller`, remembered in the site's call slot: the names tuple is the + /// site's constant, so a function and code already verified there + /// (under the site's `func_id`) bind the same way again, unless their + /// defaults may have been replaced since. + #[allow(clippy::too_many_arguments)] + fn kw_names_bind_cached( + caller: &CodeObject, + pc: usize, + f: &Rc, + func_id: u64, + perm: u32, + name_items: &[Object], + eff_argc: usize, + ) -> Option<(Rc, u32)> { + // SAFETY: GIL-serialized raw read of the function's code cell; + // only compared, then cloned. + let live: &Rc = unsafe { &*f.code.as_ptr() }; + let slot = code_call_slot(caller, pc); + if specialize::rc_id(f) == func_id && !f.defaults_maybe_overridden() { + if let Some((covered, _)) = slot.and_then(|s| s.hit(Rc::as_ptr(f), Rc::as_ptr(live))) { + return Some((live.clone(), covered)); + } + } + let (code, covered) = Self::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; + if let Some(slot) = slot { + slot.set(CallShape { + func: Rc::downgrade(f), + code: Rc::downgrade(&code), + missing: covered, + has_self: false, + }); + } + Some((code, covered)) + } + /// Move a verified `CallPyKwNames` call's operands off `stack` into /// `locals` (cleared and sized here): keyword values to their /// permuted slots, the positional run (with a real self as slot 0) @@ -10710,7 +11085,8 @@ impl Interpreter { let Object::Function(f) = &frame.stack[self_at - 1] else { return None; }; - let (code, covered) = Self::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; + let (code, covered) = + Self::kw_names_bind_cached(&frame.code, pc, f, func_id, perm, name_items, eff_argc)?; if !Self::lean_code_ok(&code) { return None; } @@ -10793,7 +11169,9 @@ impl Interpreter { shell: &mut QuietShell<'_>, pc: usize, ) -> Option> { - let shape = Self::lean_call_kw_shape(frame, pc)?; + let Some(shape) = Self::lean_call_kw_shape(frame, pc) else { + return self.try_inline_call_kw_named(frame, shell, pc); + }; // Past the recursion limit the nested lean path raises. let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { return None; @@ -10808,6 +11186,95 @@ impl Interpreter { Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) } + /// [`Self::try_inline_call_kw`] for a callee the site's cached + /// permutation can't describe (`*args`, for one): the keywords bind by + /// name (see [`Self::lean_bind_keywords`]) before anything is touched. + fn try_inline_call_kw_named( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + use weavepy_compiler::InlineCache as IC; + // A first execution still specializes the site. + if matches!(frame.code.caches.get(pc as u32), IC::Empty) { + return None; + } + let argc = frame.code.instructions.get(pc)?.arg as usize; + let len = frame.stack.len(); + let Some(Object::Tuple(names)) = frame.stack.last() else { + return None; + }; + let kwc = names.len(); + // Stack: callable, self-or-null, positionals, keyword values, names. + let callee_slot = len.checked_sub(kwc + argc + 3)?; + let Object::Function(f) = &frame.stack[callee_slot] else { + return None; + }; + let code = f.code(); + if code.is_generator + || code.is_coroutine + || code.is_async_generator + || !Self::lean_code_ok(&code) + || (f.defaults_maybe_overridden() + && (f.slot("__defaults__").is_some() || f.slot("__kwdefaults__").is_some())) + { + return None; + } + f.lean_cells_ref(&code)?; + let first = if matches!(frame.stack[callee_slot + 1], Object::Unbound) { + callee_slot + 2 + } else { + callee_slot + 1 + }; + let values = &frame.stack[len - 1 - kwc..len - 1]; + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + let bound = Self::lean_bind_keywords( + f, + &code, + frame.stack[first..len - 1 - kwc].iter(), + names.iter().zip(values), + locals, + ); + // Past the recursion limit the nested lean path raises. + let guard = match (bound, crate::recursion::enter()) { + (Some(()), crate::recursion::Enter::Ok(guard)) => guard, + _ => { + locals.clear(); + self.inline_unslot(act); + return None; + } + }; + // Committed: the operands leave the stack. + let callable = Object::Function(f.clone()); + self.release_call_operands(&mut frame.stack, callee_slot); + fill_unbound(locals, code.varnames.len()); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) + } + + /// Release a call's operands (the callee first) once its parameters + /// are bound, as the full call handler does after it returns: a + /// temporary the collector tracks (the `**` mapping a call site + /// built, say) is reaped at once, so what it held is released with + /// the callee's own references rather than at the next collection. + fn release_call_operands(&mut self, stack: &mut Vec, callee_slot: usize) { + let callee = std::mem::replace(&mut stack[callee_slot], Object::Unbound); + self.reap_call_receiver(callee); + self.reap_call_args(&mut stack[callee_slot + 1..]); + stack.truncate(callee_slot); + } + + /// Return a parked activation slot [`Self::inline_slot`] handed out but + /// the call didn't use (its locals must be empty again). + fn inline_unslot(&mut self, act: Box) { + if self.inline_pool.len() < INLINE_POOL_CAP { + self.inline_pool.push(act); + } + } + /// `LOAD_ATTR` of a `property` on a plain instance, from the quiet /// loop: the getter — a plain one-argument Python function the lean /// path can run — is called with the receiver, and its result @@ -10887,7 +11354,7 @@ impl Interpreter { let saved_pc = frame.pc; frame.pc = pc as u32 + 1; let pending_self = self.lean_pending_enter(frame, shell, pc); - let result = self.generator_send_lean(&g, &Object::None); + let result = self.generator_send_lean(&g, &Object::None, None); self.lean_pending_exit(pending_self); let Some(result) = result else { frame.pc = saved_pc; @@ -10991,6 +11458,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -11092,12 +11560,12 @@ impl Interpreter { if !std::ptr::eq( unsafe { Rc::as_ptr(&*init.code.as_ptr()) }, Rc::as_ptr(code), - ) || code.arg_count as usize != args.len() + 1 - || !Self::lean_code_ok(code) + ) || !Self::lean_code_ok(code) || crate::stdlib::testinternalcapi_mod::nomem_alloc_fails() { return None; } + let missing = Self::missing_defaults(init, code, args.len() + 1)?; let cells = init.lean_cells_ref(code)?; let snap_gen = self.lean_snapshot()?; let code = code.clone(); @@ -11118,6 +11586,11 @@ impl Interpreter { let v = unsafe { &mut *rc.as_ptr() }; v.push(inst.clone()); v.extend(args.iter().cloned()); + v.extend( + init.defaults[init.defaults.len() - missing..] + .iter() + .cloned(), + ); v.resize(nlocals, Object::Unbound); } rc @@ -11173,6 +11646,15 @@ impl Interpreter { true } + /// [`Self::lean_code_ok`] for a frameless leaf call from the + /// interpreter: compiled code still qualifies (the leaf evaluator + /// costs less than entering native code from here; native callers + /// take the compiled code through their own call path). + #[inline] + fn leaf_code_ok(code: &CodeObject) -> bool { + !code.wire.as_ref().is_some_and(|w| w.exec_error.is_some()) + } + /// [`Self::lean_code_ok`] for a generator resume: a compiled /// generator whose every loop yields runs at most one iteration per /// native resume, and the native resume protocol costs more than @@ -11546,6 +12028,28 @@ impl Interpreter { Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None ) } + // The colder per-activation state (see the prologue note), read at + // its uses from the running activation's frame and code extension. + macro_rules! stamps { + ($cache:ident, $frame:expr, $ext:expr) => { + *$cache.get_or_insert_with(|| { + if $frame.builtins_obj.is_none() { + $ext.and_then(|e| e.stamp_slots.get()) + .map_or(&[][..], |s| &s[..]) + } else { + &[][..] + } + }) + }; + } + macro_rules! mslots { + ($cache:ident, $ext:expr) => { + *$cache.get_or_insert_with(|| { + $ext.and_then(|e| e.method_slots.get()) + .map_or(&[][..], |s| &s[..]) + }) + }; + } 'reload: loop { // (A discriminant test first: `take` would copy the whole exit.) if sw.pending.is_some() { @@ -11567,38 +12071,18 @@ impl Interpreter { // during an activation, and nothing here runs Python code). let locals: &mut Vec = unsafe { &mut *frame.locals.as_ptr() }; let (lbase, nlocals) = (locals.as_mut_ptr(), locals.len()); - // The `LOAD_GLOBAL` arm's cache-hit state (see `leaf_global`): the - // site stamps exist once any global hit filled one, and a custom - // `__builtins__` mapping keeps every load on the full path. - let stamps: &[StampSlot] = if frame.builtins_obj.is_none() { - ext.and_then(|e| e.stamp_slots.get()) - .map_or(&[], |s| &s[..]) - } else { - &[] - }; - // The method-form `LOAD_ATTR` arm's per-site functions. - let mslots: &[MethodSlot] = ext - .and_then(|e| e.method_slots.get()) - .map_or(&[], |s| &s[..]); - let (gdict, bdict) = (frame.globals.as_ptr(), frame.builtins.as_ptr()); - let (gid, bid) = ( - specialize::rc_id(&frame.globals), - specialize::rc_id(&frame.builtins), - ); - // Module scope (names resolve in the globals, then the builtins, - // both exact dicts): the `LOAD_NAME` / `STORE_NAME` arms below. + // (The namespaces are read off the frame at their uses: holding + // them here costs every switch a spill.) + // The arms' colder per-activation state (the global and class + // attribute stamps, the method slots) is derived at its first + // use: every call, return and helper handoff runs this prologue + // again, and most activations never read it. + let mut cold_stamps: Option<&[StampSlot]> = None; + let mut cold_mslots: Option<&[MethodSlot]> = None; + // // The fused local pairs below, one byte per instruction (empty // for code that has none). - let fast_pairs: &[u8] = if crate::hot_gates::env_flags::no_pairs() { - &[] - } else { - code_fast_pairs(code, ext) - }; - let name_scope = frame.class_namespace.is_none() - && frame.class_namespace_obj.is_none() - && frame.builtins_obj.is_none() - && !frame.code.is_class_body - && !self.globals_missing_any.get(); + let fast_pairs: &[u8] = code_fast_pairs(code, ext); let stack = &mut frame.stack; let base = stack.as_mut_ptr(); let cap = stack.capacity(); @@ -11700,6 +12184,52 @@ impl Interpreter { } } } + // A branch on a local, read in place (nothing is + // pushed, so nothing is cloned or released). + 3 | 4 => { + // SAFETY: `i < nlocals` (checked above), and a + // pair kind is only recorded where the branch + // instruction exists. + let v = unsafe { &*lbase.add(i) }; + let jump_pc = pc + 1 + usize::from(fast_pairs[pc] == 4); + let jump = unsafe { *instrs.add(jump_pc) }; + let taken = match (jump.op, v) { + (_, Object::Unbound) => None, + (OpCode::PopJumpIfNone, v) => Some(matches!(v, Object::None)), + (OpCode::PopJumpIfNotNone, v) => { + Some(!matches!(v, Object::None)) + } + (op, v) => { + let truthy = match v { + Object::Bool(b) => Some(*b), + Object::Int(n) => Some(*n != 0), + Object::None => Some(false), + Object::Float(f) => Some(*f != 0.0), + Object::Str(s) => Some(!s.is_empty()), + Object::Tuple(t) => Some(!t.is_empty()), + // SAFETY: a read between two instructions. + Object::List(l) => { + unsafe { l.peek() }.map(|l| !l.is_empty()) + } + _ => None, + }; + truthy.map(|t| t == (op == OpCode::PopJumpIfTrue)) + } + }; + if let Some(taken) = taken { + last = jump_pc; + pc = jump_pc + 1; + if taken { + pc += jump.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; + } + continue; + } + } _ => {} } // SAFETY: `i < nlocals`, `len < cap`. @@ -11727,19 +12257,98 @@ impl Interpreter { && len < cap && simple_args_prefix(&code.instructions, pc + 2) { - let pure_site = match (other, mslots.get(pc + 1)) { - (Object::Instance(i), Some(ms)) => ms - .peek_fn(i.cls_raw().attr_version.get()) - .is_some_and(|fp| fn_is_pure_leaf(&*fp)), - _ => false, - }; + // A native method the site cached + // for this class version: straight + // to its fused call (the Python + // leaf probes below would miss). + if let (Object::Instance(i), Some(ms)) = + (other, mslots!(cold_mslots, ext).get(pc + 1)) + { + if ms + .peek_inst_builtin( + i.cls_raw().attr_version.get(), + ) + .is_some() + { + if let Some((r, call_pc)) = self + .core_native_method( + code, + other, + pc + 1, + next.arg, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) + { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop( + LeafStop::Raised(e), + )); + } + } + } + } + } + // A site that verified its callee + // for this class version. + let mut missed = false; + if let (Object::Instance(i), Some(ext)) = (other, ext) { + if let Some(site) = leaf_site_hit( + ext, + pc + 1, + i.cls_raw().attr_version.get(), + ) { + match self.core_leaf_site_call( + code, + i, + other, + site, + pc + 1, + next.arg, + lbase, + nlocals, + consts, + sw.depth_cell, + ) { + SiteCall::Done(v, call_pc) => { + base.add(len).write(v); + len += 1; + last = call_pc; + pc = call_pc + 1; + continue; + } + SiteCall::Declined => {} + SiteCall::Missed => missed = true, + } + } + } + let pure_site = !missed + && match ( + other, + mslots!(cold_mslots, ext).get(pc + 1), + ) { + (Object::Instance(i), Some(ms)) => ms + .peek_fn(i.cls_raw().attr_version.get()) + .is_some_and(|fp| fn_is_leaf(&*fp)), + _ => false, + }; if pure_site { if let Some((v, call_pc)) = self.core_pure_method( code, other, pc + 1, next.arg, - mslots, + mslots!(cold_mslots, ext), lbase, nlocals, consts, @@ -11751,6 +12360,32 @@ impl Interpreter { pc = call_pc + 1; continue; } + } else if let Some((r, call_pc)) = self + .core_native_method( + code, + other, + pc + 1, + next.arg, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) + { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop( + LeafStop::Raised(e), + )); + } + } } } if next.op == OpCode::LoadAttr { @@ -11768,10 +12403,26 @@ impl Interpreter { pc = end + 1; continue; } - } - if let Some(v) = - Self::core_local_attr(code, other, pc + 1, next.arg) + } else if let (Some(ext), Object::Instance(inst)) = + (ext, other) { + // The site's field shortcut, in line (the + // common monomorphic read). + if let Some(v) = field_slot_hit(ext, pc + 1, inst) { + base.add(len).write(clone_hot(v)); + len += 1; + last = pc + 1; + pc += 2; + continue; + } + } + if let Some(v) = Self::core_local_attr( + ext, + code, + other, + pc + 1, + next.arg, + ) { base.add(len).write(v); len += 1; last = pc + 1; @@ -11799,6 +12450,43 @@ impl Interpreter { continue; } } + // `x.m(...)` of a container's site-cached + // leaf method on the borrowed local. + if matches!( + other, + Object::List(_) + | Object::Dict(_) + | Object::Set(_) + | Object::Str(_) + ) && pc + 1 < ninstrs + && len < cap + && (*instrs.add(pc + 1)).op == OpCode::LoadMethodAttr + { + if let Some((r, call_pc)) = self.core_builtin_method( + code, + other, + pc + 1, + mslots!(cold_mslots, ext), + lbase, + nlocals, + consts, + ) { + last = call_pc; + pc = call_pc + 1; + match r { + Ok(v) => { + base.add(len).write(v); + len += 1; + continue; + } + Err(e) => { + break Some(CoreExit::Stop(LeafStop::Raised( + e, + ))); + } + } + } + } clone_hot(other) } }; @@ -11908,6 +12596,20 @@ impl Interpreter { pc += 1; continue; } + // An untracked container of scalars: + // a plain drop. + Object::List(_) | Object::Dict(_) + if Self::core_plain_last_container(&*slot) => + { + len -= 1; + drop(std::mem::replace( + &mut *slot, + base.add(len).read(), + )); + last = pc; + pc += 1; + continue; + } _ => break None, } } @@ -11926,6 +12628,32 @@ impl Interpreter { last = pc; pc += 1; } + // A list literal, tracked like the full handler's (an + // allocation due to trigger a collection is left to it). + OpCode::BuildList => { + let n = ins.arg as usize; + if n > len + || (n == 0 && len == cap) + || crate::stdlib::tracemalloc_real::is_tracking() + || crate::stdlib::testinternalcapi_mod::reftrace_print_active() + || gc_trace::auto_collect_due() + { + break None; + } + // SAFETY: the top `n` slots are initialized; they move + // into the list and leave the stack. + let items: Vec = (len - n..len) + .map(|j| unsafe { base.add(j).read() }) + .collect(); + len -= n; + let obj = Object::new_list(items); + gc_trace::track(obj.clone()); + // SAFETY: `len < cap` (the operands' slots were freed). + unsafe { base.add(len).write(obj) }; + len += 1; + last = pc; + pc += 1; + } OpCode::PopTop => { // A scalar needs no drop; a shared heap value is a // plain decrement (see `core_droppable`). @@ -11934,8 +12662,16 @@ impl Interpreter { break None; } len -= 1; + // A scalar just leaves (tested apart from the drop, + // or the scalar case folds into the drop glue call). // SAFETY: the slot is initialized and leaves the stack. - unsafe { drop_hot(base.add(len).read()) }; + if !matches!( + unsafe { &*base.add(len) }, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) { + // SAFETY: as above. + unsafe { drop_hot(base.add(len).read()) }; + } last = pc; pc += 1; } @@ -12015,6 +12751,12 @@ impl Interpreter { Some(d) => !d.is_empty(), None => break None, }, + // A registered native `__bool__`/`__len__`, or + // neither (see `leaf_instance_truth`). + Object::Instance(_) => match self.leaf_instance_truth(v) { + Some(b) => b, + None => break None, + }, _ => break None, }; // SAFETY: the operand (droppable) is replaced in place. @@ -12039,6 +12781,11 @@ impl Interpreter { pc += 1; if is_none == (ins.op == OpCode::PopJumpIfNone) { pc += ins.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; } } OpCode::BinaryOp => { @@ -12152,6 +12899,58 @@ impl Interpreter { None => break None, } } + // A natively served left operand (see + // `stdlib::datetime_native`), out of line. + (Object::Instance(i), _) if i.cls_raw().native_kind.get() != 0 => { + let r = match Self::core_native_binop(kind, a, b) { + Some(r) + if Self::core_droppable(a) && Self::core_droppable(b) => + { + r + } + _ => break None, + }; + // SAFETY: both operands leave by plain + // decrements (checked). + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 2; + match r { + Ok(v) => { + // SAFETY: the operands' slots are free. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } + // String concatenation, repetition and `%` + // formatting over scalars, out of line. + (Object::Str(_), _) | (Object::Int(_), Object::Str(_)) => { + let Some(r) = Self::core_str_binop(a, b, kind) else { + break None; + }; + // SAFETY: the operands are strings and scalars + // (or a tuple of them): releasing them runs no + // code and frees nothing the collector tracks. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 1; + unsafe { base.add(len - 1).write(r) }; + last = pc; + pc += 1; + continue; + } _ => break None, }; // SAFETY: both operands are scalars (no drop owed). @@ -12160,6 +12959,64 @@ impl Interpreter { last = pc; pc += 1; } + // Container reads, stores and comprehension appends run + // out of line: arms added here cost the rest of the loop + // its register allocation. + OpCode::BinarySubscr + | OpCode::BinarySlice + | OpCode::StoreSubscr + | OpCode::ListAppend + | OpCode::UnpackSequence => { + // SAFETY: the `len` slots at `base` are initialized, + // and the helper touches nothing else. + match unsafe { Self::core_container_op(ins, base, len, cap) } { + Some(n) => { + len = n; + last = pc; + pc += 1; + } + // An instance whose class's `__getitem__` is a + // native fast subscript the site cached. + None if ins.op == OpCode::BinarySubscr && len >= 2 => { + let Some(fast) = self.core_native_subscript( + code, + pc, + // SAFETY: `len >= 2`. + unsafe { std::slice::from_raw_parts(base.add(len - 2), 2) }, + ) else { + break None; + }; + // SAFETY: both operands leave by plain + // decrements (checked); the result takes the + // container's slot. + let r = fast(unsafe { + std::slice::from_raw_parts(base.add(len - 2), 2) + }); + let Some(r) = r else { break None }; + #[cfg(test)] + NATIVE_FAST_SUBSCRIPTS.with(|calls| calls.set(calls.get() + 1)); + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 2; + match r { + Ok(v) => { + // SAFETY: the container's slot is free. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + } + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } + None => break None, + } + } OpCode::CompareOp => { if len < 2 { break None; @@ -12176,6 +13033,32 @@ impl Interpreter { Some(o) => o, None => break None, }, + // Two natively served instances (see + // `stdlib::datetime_native`), out of line. + (Object::Instance(i), Object::Instance(_)) + if i.cls_raw().native_kind.get() != 0 => + { + let r = match Self::core_native_compare(kind, a, b) { + Some(Ok(r)) + if Self::core_droppable(a) && Self::core_droppable(b) => + { + r + } + _ => break None, + }; + // SAFETY: both operands leave by plain + // decrements (checked); the result takes the + // lower one's slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + len -= 1; + unsafe { base.add(len - 1).write(r) }; + last = pc; + pc += 1; + continue; + } _ => break None, }; let r = match kind { @@ -12209,6 +13092,11 @@ impl Interpreter { pc += 1; if truthy == (ins.op == OpCode::PopJumpIfTrue) { pc += ins.arg as usize; + } else if pc < ninstrs + // SAFETY: `pc < ninstrs`. + && unsafe { (*instrs.add(pc)).op } == OpCode::NotTaken + { + pc += 1; } } OpCode::JumpForward => { @@ -12265,25 +13153,75 @@ impl Interpreter { // SAFETY: `len > 0`. let it = match unsafe { &*base.add(len - 1) } { Object::Iter(it) => it, + // A native iterator class's `__next__` (a registered + // leaf builtin), called in place. Exhaustion goes to + // the full arm, which asks again (a leaf iterator + // stays exhausted) and ends the loop. + obj @ Object::Instance(inst) => { + let Some(b) = self.core_leaf_next_ptr(inst) else { + break None; + }; + // SAFETY: the cache and the class keep the + // builtin alive; its body runs no Python. + match unsafe { ((*b).call)(std::slice::from_ref(obj)) } { + Ok(v) => { + // SAFETY: `len < cap`. + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + break None; + } + // As a raising call: the pc moves past it. + Err(e) => { + pc += 1; + break Some(CoreExit::Stop(LeafStop::Raised(e))); + } + } + } // A generator resumes inline, switched to in place // (the quiet loop's lean path when it declines). - Object::Generator(_) => { + Object::Generator(g) => { + // A simple body runs to its next yield in + // place, without switching (see `gen_fast`). + let g = g.clone(); + if let gen_fast::GenNext::Yielded(v) = + self.gen_fast_next(&g, snap_gen, 0) + { + // SAFETY: `len < cap` (checked above). + unsafe { base.add(len).write(v) }; + len += 1; + last = pc; + pc += 1; + continue; + } + drop(g); // SAFETY: `len <= cap`, every slot initialized. unsafe { frame.stack.set_len(len) }; frame.pc = pc as u32; *last_pc = last; - if !self.core_gen_resume(sw, pc) { + if !self.core_gen_resume(sw, pc, false) { sw.pending = Some(CoreExit::Stop(LeafStop::Step)); } break Some(CoreExit::Reload); } _ => break None, }; + let unique = Rc::strong_count(it) == 1; // SAFETY: nothing below runs code until `it`'s last use // (the guard-free reads of `GilCell::peek`). let Some(it) = (unsafe { it.peek_mut() }) else { break None; }; + // An exhausted range or list iterator the loop holds + // alone, whose death frees nothing else (see the leaf + // arm): retired here instead of by the full handler. + let mut retire = false; let v = match it { crate::object::PyIterator::Range { current, @@ -12295,22 +13233,42 @@ impl Interpreter { } else { *step < 0 && *current > *stop }; - if !live { + if live { + let v = *current; + *current = current.wrapping_add(*step); + Object::Int(v) + } else if unique { + retire = true; + Object::None + } else { break None; } - let v = *current; - *current = current.wrapping_add(*step); - Object::Int(v) } - crate::object::PyIterator::List { items, index, .. } => { + crate::object::PyIterator::List { + items, + index, + owner, + } => { // SAFETY: as above. - let v = match unsafe { items.peek() } { - Some(xs) => xs.get(*index).map(Self::clone_operand), - None => None, + let Some(xs) = (unsafe { items.peek() }) else { + break None; }; - let Some(v) = v else { break None }; - *index += 1; - v + match xs.get(*index) { + Some(v) => { + let v = Self::clone_operand(v); + *index += 1; + v + } + None if unique + && owner.is_none() + && Rc::strong_count(items) >= 2 + && !Self::iter_backing_list_dead(items) => + { + retire = true; + Object::None + } + None => break None, + } } crate::object::PyIterator::Tuple { items, index } => { let Some(v) = items.get(*index).cloned() else { @@ -12344,6 +13302,54 @@ impl Interpreter { *index += ch.len_utf8(); Object::from_char(ch) } + // A dict or dict view's next key, value, or item; an + // item unpacked by the `UNPACK_SEQUENCE 2` that + // follows goes straight to the stack, as below. + it @ crate::object::PyIterator::DictKeys { .. } => { + let fuse = len + 2 <= cap + && pc + 1 < ninstrs + // SAFETY: `pc + 1 < ninstrs`. + && unsafe { (*instrs.add(pc + 1)).op } + == OpCode::UnpackSequence + // SAFETY: as above. + && unsafe { (*instrs.add(pc + 1)).arg } == 2; + match Self::core_dict_next(it, fuse, unique) { + DictStep::Item(v, Some(k)) => { + // SAFETY: `len + 2 <= cap`. + unsafe { base.add(len).write(v) }; + len += 1; + pc += 1; + k + } + DictStep::Item(v, None) => v, + DictStep::Exhausted => { + // The iterator (its sole owner is the + // stack) leaves it, and the loop exits + // past its `END_FOR`/`POP_ITER` pair. + len -= 1; + // SAFETY: the slot is initialized; its + // release frees only the iterator. + unsafe { drop_hot(base.add(len).read()) }; + last = pc; + pc += 1 + ins.arg as usize; + let op_at = |pc: usize| { + // SAFETY: `pc < ninstrs` is checked first. + (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }) + }; + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!( + op_at(pc), + Some(OpCode::PopIter | OpCode::PopTop) + ) { + pc += 1; + } + } + continue; + } + DictStep::Decline => break None, + } + } // `for i, x in enumerate(xs)`: the pair goes straight // to the `UNPACK_SEQUENCE 2` that follows, which is // skipped, so no tuple is ever built. @@ -12443,6 +13449,34 @@ impl Interpreter { } _ => break None, }; + if retire { + // Popped, and the loop exit skips the `END_FOR` / + // `POP_ITER` pair (the full handler's shape). + len -= 1; + // SAFETY: the iterator slot leaves the stack. + let it = unsafe { base.add(len).read() }; + let marked = + gc_trace::maybe_tracked(crate::weakref_registry::id_of(&it)) + && gc_trace::note_dropped_marks(&it); + drop(it); + last = pc; + pc += 1 + ins.arg as usize; + let op_at = |pc: usize| { + // SAFETY: `pc < ninstrs` is checked first. + (pc < ninstrs).then(|| unsafe { (*instrs.add(pc)).op }) + }; + if op_at(pc) == Some(OpCode::EndFor) { + pc += 1; + if matches!(op_at(pc), Some(OpCode::PopIter | OpCode::PopTop)) { + pc += 1; + } + } + if marked { + gc_trace::mark_maybe_dead(); + break Some(CoreExit::Stop(LeafStop::Marked)); + } + continue; + } // SAFETY: `len < cap`. unsafe { base.add(len).write(v) }; len += 1; @@ -12488,20 +13522,28 @@ impl Interpreter { OpCode::YieldValue => { // SAFETY: see `CoreSwitch`. if unsafe { (*sw.inl).is_empty() } { - // `sum()` driving this frame: a scalar yield folds - // into the accumulator and the frame resumes as if - // sent `None` (what `sum` does next), in place. - if let Some((ff, acc)) = self.sum_fold { + // A draining consumer driving this frame: the yield + // folds into its sink and the frame resumes as if + // sent `None` (what the consumer does next), in + // place. + if let Some((ff, sink)) = self.sum_fold { if ff == sw.cur as usize && len > 0 { - let acc = acc as *mut SumState; - // SAFETY: the running total is `do_sum_call`'s - // local, alive and untouched while the - // resume runs; `len > 0`. + // SAFETY: `len > 0`. let top = unsafe { base.add(len - 1) }; - if unsafe { (*acc).add_scalar(&*top) } { - // SAFETY: the yielded value is a scalar - // (no drop glue); the sent `None` takes - // its slot. + // SAFETY: the sink is the consumer's local, + // alive and untouched while the resume runs. + let folded = match sink { + // A scalar has no drop glue. + FoldSink::Sum(acc) => unsafe { (*acc).add_scalar(&*top) }, + FoldSink::Collect(out) => { + // The yielded value moves out. + unsafe { (*out).push(top.read()) }; + true + } + }; + if folded { + // SAFETY: the sent `None` takes the + // (moved or trivially dropped) slot. unsafe { top.write(Object::None) }; last = pc; pc += 1; @@ -12540,7 +13582,7 @@ impl Interpreter { // builtin or type callee has leaf arms. // Module-scope names: the globals, then the builtins, probed // with the interned name's hash (no per-read name object). - OpCode::LoadName if name_scope => { + OpCode::LoadName if self.core_name_scope(frame) => { if len == cap { break None; } @@ -12549,11 +13591,11 @@ impl Interpreter { }; // SAFETY (raw dict reads): as in the `LOAD_GLOBAL` arm. let v = unsafe { - match (*gdict).get_index_of(&probe) { - Some(i) => (*gdict).get_index(i), - None => (*bdict) + match (*frame.globals.as_ptr()).get_index_of(&probe) { + Some(i) => (*frame.globals.as_ptr()).get_index(i), + None => (*frame.builtins.as_ptr()) .get_index_of(&probe) - .and_then(|i| (*bdict).get_index(i)), + .and_then(|i| (*frame.builtins.as_ptr()).get_index(i)), } }; // A miss (the full handler's `NameError`) or a key the @@ -12571,7 +13613,9 @@ impl Interpreter { // Rebinding an existing module-scope name whose old value // leaves by a plain drop; a new name, a finalizer's // candidate or a watched dict takes the full handler. - OpCode::StoreName if name_scope => { + OpCode::StoreName | OpCode::StoreGlobal + if ins.op == OpCode::StoreGlobal || self.core_name_scope(frame) => + { if len == 0 || crate::capi_watchers::dicts_active() { break None; } @@ -12593,9 +13637,12 @@ impl Interpreter { { break None; } - let Some((_, slot)) = (**g).get_index_mut(i) else { + // The key layout stays: no stamp (every cached global + // load stands), but the value epoch moves. + let Some((_, slot)) = g.map_mut_value_store().get_index_mut(i) else { break None; }; + crate::object::bump_global_value_epoch(); // SAFETY: `len > 0`; the value moves into the binding and // the displaced one was checked droppable. let old = std::mem::replace(slot, unsafe { base.add(len - 1).read() }); @@ -12607,6 +13654,31 @@ impl Interpreter { // A keyword call of a pure leaf through the site's cached // keyword permutation (the full handler's `CallPyKwNames` // hit), evaluated in place like `CALL`'s pure leaves. + // `f(*args, **kwargs)` forwarding binds straight from the + // local mapping (see `core_call_forward`); a decline + // builds the dict as usual. + OpCode::BuildMap if ins.arg == 0 && forward_call_shape(code, pc) => { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_forward(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } + // `f(*args)` of a Python callee switches in place (see + // `core_call_ex`); anything else takes the full handler. + OpCode::CallEx => { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_ex(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } OpCode::CallKw => { let argc = ins.arg as usize; // SAFETY: `len > 0` is checked first; the operands of @@ -12621,12 +13693,26 @@ impl Interpreter { break None; } let start = len - kwc - argc - 3; + // SAFETY: the callee and its self slot are live. + unsafe { Self::core_instance_callee(base.add(start)) }; // SAFETY: `start + kwc + argc + 3 == len`. let ops = unsafe { std::slice::from_raw_parts(base.add(start), len - start) }; let Some(r) = self.core_pure_kw_call(code, pc, ops, argc, sw.depth_cell) else { - break None; + // A plain Python callee switches in place, as + // `CALL`'s does (see `core_call_kw`). + if !matches!(&ops[0], Object::Function(_)) { + break None; + } + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_call_kw(sw, pc) && sw.pending.is_none() { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); }; // SAFETY: every operand was checked to leave by a plain // decrement; the result takes the callee's slot. @@ -12645,6 +13731,28 @@ impl Interpreter { if len < argc + 2 { break None; } + // `next(gen)`: the generator resumes inline, as for + // `FOR_ITER`, with the yield as the call's result. + // SAFETY: `len >= argc + 2 == 3`. + if argc == 1 + && matches!( + unsafe { (&*base.add(len - 3), &*base.add(len - 2), &*base.add(len - 1)) }, + (Object::Builtin(b), Object::Unbound, Object::Generator(_)) + if Rc::as_ptr(b) as usize == self.leaf_fns().next_ptr + ) + { + // SAFETY: `len <= cap`, every slot initialized. + unsafe { frame.stack.set_len(len) }; + frame.pc = pc as u32; + *last_pc = last; + if !self.core_gen_resume(sw, pc, true) { + sw.pending = Some(CoreExit::Stop(LeafStop::Step)); + } + break Some(CoreExit::Reload); + } + // SAFETY: `len >= argc + 2`: the callee and its self + // slot. + unsafe { Self::core_instance_callee(base.add(len - argc - 2)) }; // SAFETY: `len >= argc + 2`. let python = match unsafe { &*base.add(len - argc - 2) } { Object::Function(_) => 1, @@ -12661,7 +13769,7 @@ impl Interpreter { let ops = unsafe { std::slice::from_raw_parts(base.add(len - argc - 2), argc + 2) }; - if matches!(&ops[0], Object::Function(f) if fn_is_pure_leaf(f)) { + if matches!(&ops[0], Object::Function(f) if fn_is_leaf(f)) { if let Some(r) = self.core_pure_call(code, pc, ops, sw.depth_cell) { let start = len - argc - 2; // SAFETY: every operand was checked to leave @@ -12695,6 +13803,8 @@ impl Interpreter { } break Some(CoreExit::Reload); } + // The site's method slot (its leaf kind), read once. + let site_slot = mslots!(cold_mslots, ext).get(pc); // SAFETY: `len >= argc + 2`. match unsafe { &*base.add(len - argc - 2) } { // `lst.append(x)` / `lst.pop()` on an exact list (the @@ -12707,7 +13817,7 @@ impl Interpreter { // SAFETY: `len >= argc + 2`: the self slot. && matches!(unsafe { &*base.add(len - argc - 1) }, Object::List(_)) => { - let kind = mslots.get(pc).and_then(|s| s.get_leaf(b)); + let kind = site_slot.and_then(|s| s.get_leaf(b)); // SAFETY: `len >= argc + 2`: the self slot. let recv = unsafe { &*base.add(len - argc - 1) }; let Object::List(l) = recv else { @@ -12756,12 +13866,11 @@ impl Interpreter { // `leaf_builtin_call` for these kinds). Object::Builtin(b) if Rc::strong_count(b) > 1 - && matches!( - mslots.get(pc).and_then(|s| s.get_leaf(b)), - Some(LeafKind::Opaque | LeafKind::Fast(_)) - ) => + && site_slot + .and_then(|s| s.get_leaf(b)) + .is_some_and(LeafKind::runs_in_core) => { - let kind = mslots.get(pc).and_then(|s| s.get_leaf(b)); + let kind = site_slot.and_then(|s| s.get_leaf(b)); let callee_at = len - argc - 2; // SAFETY: `len >= argc + 2`: the self slot. let first = if matches!( @@ -12781,10 +13890,14 @@ impl Interpreter { } let r = match kind { Some(LeafKind::Fast(f)) => f(ops), - _ => Some(match b.call_kw.as_ref() { - Some(ckw) => ckw(ops, &[]), - None => (b.call)(ops), - }), + Some(LeafKind::Isinstance) => Self::core_isinstance(ops), + Some(LeafKind::Opaque) | None => { + Some(match b.call_kw.as_ref() { + Some(ckw) => ckw(ops, &[]), + None => (b.call)(ops), + }) + } + Some(k) => self.leaf_builtin_call(k, b, ops), }; match r { // A fast half declined, untouched. @@ -12822,7 +13935,7 @@ impl Interpreter { Object::BoundMethod(bm) if argc == 0 && matches!(&bm.function, Object::Builtin(_)) => { - if let Some(slot) = mslots.get(pc) { + if let Some(slot) = site_slot { if slot.is_non_leaf(bm) { // A prior body/receiver rejection still // applies to this immutable bound method. @@ -12850,7 +13963,7 @@ impl Interpreter { // with `len(local)` fused as `leaf_fused_len_at` does. OpCode::LoadGlobal => { use weavepy_compiler::InlineCache as IC; - let Some(slot) = stamps.get(pc) else { + let Some(slot) = stamps!(cold_stamps, frame, ext).get(pc) else { break Some(CoreExit::Helper); }; if len == cap { @@ -12859,6 +13972,8 @@ impl Interpreter { // SAFETY (raw dict reads): as in `leaf_global` — no dict // borrow is held while bytecode runs, and nothing here // runs code. + let (gdict, bdict) = (frame.globals.as_ptr(), frame.builtins.as_ptr()); + let gid = specialize::rc_id(&frame.globals); let g_stamp = unsafe { (*gdict).mutation_stamp() }; let hit = match code.caches.get(pc as u32) { IC::LoadGlobalModule { @@ -12870,7 +13985,7 @@ impl Interpreter { IC::LoadGlobalBuiltin { builtins_id, key_idx, - } if builtins_id == bid + } if builtins_id == specialize::rc_id(&frame.builtins) && !self.globals_missing_any.get() && slot.get() == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => @@ -12908,6 +14023,15 @@ impl Interpreter { Object::Dict(d) => { d.try_borrow().ok().map(|d| d.len()) } + // A native `__len__` (a deque's). + v @ Object::Instance(_) => { + match self.leaf_instance_len(v) { + Some(Object::Int(n)) => { + usize::try_from(n).ok() + } + _ => None, + } + } _ => None, }; if let Some(n) = n.and_then(|n| i64::try_from(n).ok()) { @@ -12969,12 +14093,12 @@ impl Interpreter { // SAFETY: `pc + 1 < ninstrs`. && unsafe { (*instrs.add(pc + 1)).op } == OpCode::LoadMethodAttr && simple_args_prefix(&code.instructions, pc + 2) - && mslots + && mslots!(cold_mslots, ext) .get(pc + 1) .and_then(|ms| ms.peek_unbound(cls.attr_version.get())) .is_some_and(|fp| fn_is_pure_leaf(unsafe { &*fp })) => { - let fp = mslots + let fp = mslots!(cold_mslots, ext) .get(pc + 1) .and_then(|ms| ms.peek_unbound(cls.attr_version.get())) .expect("checked by the guard"); @@ -13006,7 +14130,10 @@ impl Interpreter { // SAFETY: `pc + 1 < ninstrs`. && unsafe { (*instrs.add(pc + 1)).op } == OpCode::LoadAttr => { - match stamps.get(pc + 1).and_then(|s| class_attr_hit(s, cls)) { + match stamps!(cold_stamps, frame, ext) + .get(pc + 1) + .and_then(|s| class_attr_hit(s, cls)) + { Some(c) => { // SAFETY: `len < cap`. unsafe { base.add(len).write(c) }; @@ -13107,7 +14234,7 @@ impl Interpreter { // receiver moves up into the self slot under the function; // a class receiver leaves an empty self slot. OpCode::LoadMethodAttr => { - let Some(ms) = mslots.get(pc) else { + let Some(ms) = mslots!(cold_mslots, ext).get(pc) else { break Some(CoreExit::Helper); }; if len == 0 || len == cap { @@ -13126,22 +14253,10 @@ impl Interpreter { if !Self::default_getattribute(cls) { break Some(CoreExit::Helper); } - // The instance dict must not shadow the method. - if let Some(dict) = inst.dict.get() { - // SAFETY: a read between two instructions - // (see `GilCell::peek`). - let Some(d) = (unsafe { dict.peek() }) else { - break Some(CoreExit::Helper); - }; - if !d.is_empty() { - let Some(probe) = code_name_leaf_probe(code, ins.arg) - else { - break Some(CoreExit::Helper); - }; - if d.contains_key(&probe) || probe.saw_exotic() { - break Some(CoreExit::Helper); - } - } + // The instance's attributes must not shadow + // the method. + if inst_may_shadow(inst, code, ins.arg) { + break Some(CoreExit::Helper); } let f = match ms.get_held(ver) { Some(f) => Object::Function(f), @@ -13159,7 +14274,15 @@ impl Interpreter { ), _ => None, }) { - Some(f) => Object::Function(f), + Some(f) => { + // The site remembers this + // class too (see + // `MethodSlot::poly_remember`), + // so its fused leaf-call + // path serves it next time. + ms.set(ver, &f); + Object::Function(f) + } None => break Some(CoreExit::Helper), }, }, @@ -13183,10 +14306,20 @@ impl Interpreter { None => break Some(CoreExit::Helper), } } - // A list's site-cached native method (the helper's - // builtin-receiver case, which fills the slot). - Object::List(_) => { - let Some(b) = ms.get_builtin(1) else { + // A native container's site-cached method (the + // helper's builtin-receiver case, which fills the + // slot under the receiver's tag). + recv @ (Object::List(_) + | Object::Dict(_) + | Object::Set(_) + | Object::Str(_)) => { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + _ => 4, + }; + let Some(b) = ms.get_builtin(tag) else { break Some(CoreExit::Helper); }; // SAFETY: `len < cap`; the receiver moves up. @@ -13229,7 +14362,10 @@ impl Interpreter { // stamp); the class (count above one) leaves by a // plain decrement. Object::Type(cls) => { - match stamps.get(pc).and_then(|s| class_attr_hit(s, cls)) { + match stamps!(cold_stamps, frame, ext) + .get(pc) + .and_then(|s| class_attr_hit(s, cls)) + { Some(v) if Rc::strong_count(cls) > 1 => { // SAFETY: the receiver is replaced in place. unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; @@ -13240,8 +14376,50 @@ impl Interpreter { _ => break Some(CoreExit::Helper), } } + // `module.name` off the site's cached index (a + // shared module leaves by a plain decrement). + Object::Module(m) => { + match Self::core_module_attr(code, m, pc, ins.arg) { + Some(v) if !gc_trace::note_dropped_marks(unsafe { &*top }) => { + // SAFETY: the receiver is replaced in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + _ => break Some(CoreExit::Helper), + } + } _ => break Some(CoreExit::Helper), }; + // A natively served instance's public field (see + // `stdlib::datetime_native`). + if inst.cls_raw().native_kind.get() != 0 { + match Self::core_native_field(ext, inst, ins.arg) { + Some(v) if Self::core_droppable(unsafe { &*top }) => { + // SAFETY: the receiver (droppable) is + // replaced in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + _ => break Some(CoreExit::Helper), + } + } + // SAFETY: a read between two instructions (see + // `GilCell::peek`). + if let Some(v) = ext.and_then(|e| unsafe { field_slot_hit(e, pc, inst) }) { + if Self::core_droppable(unsafe { &*top }) { + let v = Self::clone_operand(v); + // SAFETY: the receiver (droppable) is replaced + // in place. + unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; + last = pc; + pc += 1; + continue; + } + } let IC::LoadAttrInstance { key_idx, ver } = code.caches.get(pc as u32) else { break Some(CoreExit::Helper); @@ -13255,15 +14433,16 @@ impl Interpreter { } // SAFETY: a read between two instructions (see // `GilCell::peek`). - let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek() }) else { - break Some(CoreExit::Helper); - }; - let Some((k, v)) = d.get_index(key_idx as usize) else { + let Some((k, v)) = (unsafe { inst.attr_peek_index(key_idx as usize) }) + else { break Some(CoreExit::Helper); }; if !slot_name_matches(code, ins.arg, k) { break Some(CoreExit::Helper); } + if let Some(e) = ext { + field_slot_note(e, ninstrs, pc, inst, key_idx); + } let v = Self::clone_operand(v); // SAFETY: the receiver (droppable) is replaced in place. unsafe { drop_hot(std::mem::replace(&mut *top, v)) }; @@ -13303,8 +14482,14 @@ impl Interpreter { // SAFETY: `len > 0`. let top = unsafe { base.add(len - 1) }; let v = unsafe { &*top }; - if !matches!(v, Object::List(_) | Object::Tuple(_) | Object::Range(_)) - || !Self::core_droppable(v) + if !matches!( + v, + Object::List(_) + | Object::Tuple(_) + | Object::Range(_) + | Object::Dict(_) + | Object::DictView(_) + ) || !Self::core_droppable(v) { break Some(CoreExit::Helper); } @@ -13348,6 +14533,9 @@ impl Interpreter { if !self.inline_calls_ok() { return false; } + // SAFETY: see `CoreSwitch`: the running activation is synced and + // unborrowed here. + Self::unpack_bound_callee(unsafe { &mut *sw.cur }, pc); let mut tmp = None; // SAFETY: see `CoreSwitch`: the running activation is the // innermost; its handles are live and unborrowed here. @@ -13364,7 +14552,9 @@ impl Interpreter { let act = self.try_inline_call(frame, shell, pc); if let (Some(act), Some((func, has_self, eff_argc))) = (&act, shape) { let code = &act.frame.code; - if let Some(slot) = code_call_slot(&frame.code, pc) { + if let Some(slot) = code_call_slot(&frame.code, pc) + .filter(|_| !Self::has_extended_params(code)) + { slot.set(CallShape { func, code: Rc::downgrade(code), @@ -13389,6 +14579,382 @@ impl Interpreter { true } + /// A bound method over a plain function with an empty self slot, + /// called at `pc` of `frame`, becomes that function with the receiver + /// as self (CPython's `CALL_BOUND_METHOD_EXACT_ARGS`), so the cached + /// and pure-leaf call paths apply to it. + #[inline] + fn unpack_bound_callee(frame: &mut Frame, pc: usize) { + let Some(argc) = frame.code.instructions.get(pc).map(|i| i.arg as usize) else { + return; + }; + let n = frame.stack.len(); + let Some(self_slot) = n.checked_sub(argc + 1) else { + return; + }; + let Some(callee_slot) = self_slot.checked_sub(1) else { + return; + }; + let (f, receiver) = match &frame.stack[callee_slot] { + Object::BoundMethod(bm) + if !bm.redispatch_descriptor + && matches!(frame.stack[self_slot], Object::Unbound) => + { + match &bm.function { + Object::Function(f) => (f.clone(), bm.receiver.clone()), + _ => return, + } + } + _ => return, + }; + let bm = std::mem::replace(&mut frame.stack[callee_slot], Object::Function(f)); + frame.stack[self_slot] = receiver; + // The bound method was a call temporary (or is still held + // elsewhere): grade its release like any dropped operand. + if gc_trace::note_dropped_marks(&bm) { + gc_trace::mark_maybe_dead(); + } + drop(bm); + } + + /// [`Self::core_call`] for the `CALL_KW` at `pc`: the quiet loop's + /// inline keyword call ([`Self::try_inline_call_kw`]), pushed and made + /// the running activation. `false` touches nothing. + #[inline(never)] + fn core_call_kw(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + self.try_inline_call_kw(&mut *frame, &mut *shell.cast::>(), pc) + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + + /// The core loop's `CALL_FUNCTION_EX` at `pc` of `sw`'s (synced) + /// running activation, for `f(*args)` with no `**` mapping: the tuple + /// or list spreads into an ordinary call's operands, and a plain + /// Python callee (or a bound method over one) runs as an inline + /// activation switched to here. `false` touches nothing. + #[inline(never)] + fn core_call_ex(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + self.try_inline_call_ex(&mut *frame, &mut *shell.cast::>(), pc) + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + + /// The core loop's `BUILD_MAP 0` at `pc` opening a forwarding call, + /// `f(*args, **kwargs)` (see [`forward_call_shape`]): the fresh + /// merged dictionary would only be read by the call's binder, so the + /// local mapping itself stands in as the call's `**` operand and the + /// `CALL_FUNCTION_EX` three instructions on runs inline. `false` + /// touches nothing (the instructions run one by one). + #[inline(never)] + fn core_call_forward(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + if !self.inline_calls_ok() { + return false; + } + let mut tmp = None; + // SAFETY: see `CoreSwitch` (as in `core_call`). + let act = unsafe { + let depth = (*sw.inl).len(); + let (frame, _, shell) = sw.activation(depth, &mut tmp); + let frame = &mut *frame; + let local = frame.code.instructions.get(pc + 1).map(|i| i.arg as usize); + // SAFETY: GIL-serialized read of the running activation's locals. + let locals: &Vec = &*frame.locals.as_ptr(); + let mapping = match local.and_then(|i| locals.get(i)) { + Some(Object::Dict(d)) => Object::Dict(d.clone()), + _ => return false, + }; + frame.stack.push(mapping); + let act = self.try_inline_call_ex(frame, &mut *shell.cast::>(), pc + 3); + if act.is_none() { + // Declined untouched: the operand leaves again. + frame.stack.pop(); + } + act + }; + let Some(mut act) = act else { + return false; + }; + let callee: *mut Frame = &raw mut *act.frame; + // SAFETY: as above. + unsafe { (*sw.inl).push(act) }; + sw.cur = callee; + sw.scratch = usize::MAX; + sw.last = &raw mut sw.scratch; + true + } + + /// [`Self::core_call_ex`]'s activation: every check first, so a + /// decline leaves the four `CALL_FUNCTION_EX` operands untouched. + fn try_inline_call_ex( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + const MAX_SPREAD: usize = 16; + let n = frame.stack.len(); + let callee_slot = n.checked_sub(4)?; + let [callee, null, spread, mapping] = &frame.stack[callee_slot..] else { + return None; + }; + if !matches!(null, Object::Unbound) { + return None; + } + // A non-empty `**` mapping binds by name (see `lean_bind_keywords`). + let keywords = match mapping { + Object::Unbound => None, + Object::Dict(d) => Some(d), + _ => return None, + }; + if keywords.is_some_and(|d| !d.borrow().is_empty()) { + return self.try_inline_call_ex_keywords(frame, shell, pc); + } + let items: &[Object] = match spread { + Object::Tuple(t) => t, + _ => return None, + }; + if items.len() > MAX_SPREAD { + return None; + } + let (f, receiver) = match callee { + Object::Function(f) => (f, None), + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(f) => (f, Some(&bm.receiver)), + _ => return None, + }, + _ => return None, + }; + let eff_argc = items.len() + usize::from(receiver.is_some()); + let (code, missing) = Self::lean_call_shape(f, eff_argc)?; + // Past the recursion limit the nested lean path raises. + let crate::recursion::Enter::Ok(guard) = crate::recursion::enter() else { + return None; + }; + // Committed: the operands become `callee, self-or-null, *items`. + let (f, receiver) = (f.clone(), receiver.cloned()); + // The empty `**` slot (or empty mapping). + let mapping = frame.stack.pop().expect("checked above"); + let spread = frame.stack.pop().expect("checked above"); + let Object::Tuple(items) = &spread else { + unreachable!("checked above") + }; + frame.stack.pop(); // the NULL self slot + let callee = frame.stack.pop().expect("checked above"); + let has_self = receiver.is_some(); + frame.stack.push(Object::Function(f)); + frame.stack.push(receiver.unwrap_or(Object::Unbound)); + frame.stack.extend(items.iter().cloned()); + // The spread tuple and a bound method were call temporaries (or + // are still held elsewhere): grade their releases like any + // dropped operand. + self.reap_call_receiver(callee); + self.reap_call_args(&mut [spread, mapping]); + let self_slot = callee_slot + 1; + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + let callable = Self::lean_call_fill( + frame, + &code, + has_self, + self_slot, + callee_slot, + missing, + locals, + ); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, callable, guard)) + } + + /// [`Self::try_inline_call_ex`] for `f(*args, **mapping)` with a + /// non-empty dictionary: the parameters bind by name (see + /// [`Self::lean_bind_keywords`]) before anything is touched. (The + /// operands are only borrowed until they leave the stack, so their + /// release is graded with their true owner counts.) + fn try_inline_call_ex_keywords( + &mut self, + frame: &mut Frame, + shell: &mut QuietShell<'_>, + pc: usize, + ) -> Option> { + let callee_slot = frame.stack.len() - 4; + let (f, receiver) = match &frame.stack[callee_slot] { + Object::Function(f) => (f.clone(), None), + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(f) => (f.clone(), Some(bm.receiver.clone())), + _ => return None, + }, + _ => return None, + }; + let (Object::Tuple(items), Object::Dict(mapping)) = + (&frame.stack[callee_slot + 2], &frame.stack[callee_slot + 3]) + else { + return None; + }; + let code = f.code(); + if code.is_generator + || code.is_coroutine + || code.is_async_generator + || !Self::lean_code_ok(&code) + || (f.defaults_maybe_overridden() + && (f.slot("__defaults__").is_some() || f.slot("__kwdefaults__").is_some())) + { + return None; + } + f.lean_cells_ref(&code)?; + let kw = mapping.try_borrow().ok()?; + let act = self.inline_slot(); + // SAFETY: a parked slot's locals storage is its own, and empty. + let locals = unsafe { &mut *act.frame.locals.as_ptr() }; + let bound = Self::lean_bind_keywords( + &f, + &code, + receiver.iter().chain(items.iter()), + kw.iter().map(|(k, v)| (&k.0, v)), + locals, + ); + drop(kw); + // Past the recursion limit the nested lean path raises. + let guard = match (bound, crate::recursion::enter()) { + (Some(()), crate::recursion::Enter::Ok(guard)) => guard, + _ => { + locals.clear(); + self.inline_unslot(act); + return None; + } + }; + // Committed: the operands leave the stack. + self.release_call_operands(&mut frame.stack, callee_slot); + fill_unbound(locals, code.varnames.len()); + frame.pc = pc as u32 + 1; + Some(self.inline_bind(frame, shell, pc, act, code, Object::Function(f), guard)) + } + + /// Bind `f(*positional, **kw)` (a plain dictionary) to `code`'s + /// parameters as the generic binder would: the positional + /// parameters, the keyword-only ones, then `*args` and `**kwargs` + /// when present. `None` when anything needs the generic binder: a + /// non-string key, a duplicate or unexpected keyword, a missing + /// argument, or surplus positionals without `*args`. (Replaced + /// defaults are the caller's check.) + fn lean_bind_keywords<'a>( + f: &PyFunction, + code: &CodeObject, + positional: impl Iterator, + kw: impl Iterator, + out: &mut Vec, + ) -> Option<()> { + let npos = code.arg_count as usize; + let total = npos + code.kwonly_count as usize; + let posonly = code.posonly_count as usize; + debug_assert!(out.is_empty()); + fill_unbound(out, total); + // (Every slot written below still holds `Unbound`, which owns + // nothing: it is overwritten without drop glue.) + let bind = + |slot: &mut Object, v: &Object| std::mem::forget(std::mem::replace(slot, v.clone())); + let mut surplus: Vec = Vec::new(); + for (i, v) in positional.enumerate() { + if i < npos { + bind(&mut out[i], v); + } else if code.has_varargs { + surplus.push(v.clone()); + } else { + return None; + } + } + let mut varkw: Option = None; + let kw_hint = kw.size_hint().0; + for (key, v) in kw { + let Object::Str(name) = key else { + return None; + }; + match code.varnames[posonly..total] + .iter() + .position(|n| n.as_str() == &**name) + { + Some(p) => { + let slot = &mut out[posonly + p]; + if !matches!(slot, Object::Unbound) { + return None; + } + bind(slot, v); + } + None if code.has_varkeywords => { + varkw + .get_or_insert_with(|| { + DictData::with_capacity_and_hasher( + kw_hint.max(1), + crate::fasthash::FxBuildHasher, + ) + }) + .insert(DictKey(key.clone()), v.clone()); + } + None => return None, + } + } + let first_default = npos.checked_sub(f.defaults.len())?; + for (i, slot) in out[..npos].iter_mut().enumerate() { + if matches!(slot, Object::Unbound) { + if i < first_default { + return None; + } + bind(slot, &f.defaults[i - first_default]); + } + } + for (slot, d) in out[npos..total] + .iter_mut() + .zip(Self::kwonly_defaults(f, code)) + { + if matches!(slot, Object::Unbound) { + bind(slot, d?); + } + } + if code.has_varargs { + out.push(Object::new_tuple(surplus)); + } + if code.has_varkeywords { + out.push(Object::Dict(Rc::new(RefCell::new( + varkw.unwrap_or_default(), + )))); + } + Some(()) + } + /// The core loop's `CALL` of a plain class at `pc` of `sw`'s (synced) /// running activation: `try_lean_call`'s construction shape, with the /// `__init__` activation run inline and switched to here (the caller @@ -13465,11 +15031,29 @@ impl Interpreter { if !std::ptr::eq( unsafe { Rc::as_ptr(&*init.code.as_ptr()) }, Rc::as_ptr(code), - ) || code.arg_count as usize != argc + 1 - || !Self::lean_code_ok(code) + ) || !Self::lean_code_ok(code) { return false; } + let Some(missing) = Self::missing_defaults(init, code, argc + 1) else { + return false; + }; + // A leaf `__init__` (plain stores of its arguments into `self`) + // runs frameless, as a leaf method call does. + if self.core_leaf_init( + frame, + &ty, + init, + code, + argc, + missing, + self_slot, + callee_slot, + sw.depth_cell, + ) { + frame.pc = pc as u32 + 1; + return true; + } let Some(cells) = init.lean_cells_ref(code) else { return false; }; @@ -13490,12 +15074,15 @@ impl Interpreter { // SAFETY: a parked slot's locals storage is its own, and empty. let locals = unsafe { &mut *act.frame.locals.as_ptr() }; let nlocals = code.varnames.len(); - locals.reserve(nlocals.max(argc + 1)); + locals.reserve(nlocals.max(argc + 1 + missing)); locals.push(inst.clone()); locals.extend(frame.stack.drain(self_slot + 1..)); - if locals.len() < nlocals { - locals.resize(nlocals, Object::Unbound); - } + locals.extend( + init.defaults[init.defaults.len() - missing..] + .iter() + .cloned(), + ); + fill_unbound(locals, nlocals); // The NULL self slot and the class. frame.stack.truncate(callee_slot); drop(ty); @@ -13526,6 +15113,453 @@ impl Interpreter { true } + /// [`Self::core_new`] for an `__init__` that is a pure or effect leaf + /// whose every return is `return None`: the new instance, evaluated + /// into by [`Self::pure_leaf_eval`] with no activation, replaces the + /// call's operands (`frame.stack[callee_slot..]`). `false` touches + /// nothing observable (a declined evaluation stores nothing, and the + /// unused instance was never seen). + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_leaf_init( + &self, + frame: &mut Frame, + ty: &Rc, + init: &Rc, + code: &Rc, + argc: usize, + missing: usize, + self_slot: usize, + callee_slot: usize, + depth_cell: *const std::cell::Cell, + ) -> bool { + let pure = code_is_pure_leaf(code); + if !(pure || code_is_effect_leaf(code)) + || argc + missing >= 8 + || !pure_leaf_warm(code) + || !code_returns_only_none(code) + || ty.flags.is_builtin + || ty.native_kind.get() != 0 + || ty.instances_need_finalize() + // The ordinary call's `RecursionError` check. + // SAFETY: this thread's own depth cell. + || unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() + { + return false; + } + // The arguments leave by plain decrements (whatever `__init__` + // stored holds its own reference). + if !frame.stack[self_slot + 1..] + .iter() + .all(Self::core_droppable) + { + return false; + } + let (inst, _) = self.alloc_plain_instance_obj(ty); + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + args[0] = &raw const inst; + for (k, a) in frame.stack[self_slot + 1..].iter().enumerate() { + args[k + 1] = a; + } + let defaults = &init.defaults[init.defaults.len() - missing..]; + for (k, d) in defaults.iter().enumerate() { + args[argc + 1 + k] = d; + } + let args = &args[..=argc + missing]; + let r = if pure { + self.pure_leaf_eval::(code, init, args) + } else { + self.leaf_init_eval(code, init, args) + }; + match r { + Some(done) => { + // Every return is `return None` (checked above). + drop(done); + frame.stack.truncate(callee_slot); + frame.stack.push(inst); + true + } + None => { + // A declined `__init__` may have stored into the instance + // before it stopped; nothing else ever saw it, so it goes + // (the framed call starts over on a fresh one). + gc_trace::note_dropped(&inst); + drop(inst); + false + } + } + } + + /// `seq[slice]` for the core loop's `BINARY_SUBSCR` (a list, tuple or + /// string over a slice of ints): the full handler's slicing, or + /// `None` for it to run (and raise) itself. + #[inline(never)] + fn core_slice(c: &Object, sl: &crate::object::PySlice) -> Option { + match c { + Object::List(items) => { + let items = items.try_borrow().ok()?; + Some(Object::new_list(slice_seq(&items, sl).ok()?)) + } + Object::Tuple(items) => Some(Object::new_tuple(slice_seq(&items[..], sl).ok()?)), + Object::Str(st) => str_subscript_slice(st, sl).ok(), + _ => None, + } + } + + /// The core loop's `BINARY_OP` over strings: `str + str`, a small + /// `str * int`, and `str % args` over scalar and string arguments + /// (the leaf arms' string cases). `None` for the full handler, which + /// also owns every error. + #[inline(never)] + fn core_str_binop(a: &Object, b: &Object, kind: BinOpKind) -> Option { + Some(match (a, b) { + (Object::Str(x), Object::Str(y)) if kind == BinOpKind::Add => { + Object::Str(SharedStr::concat(&[x, y])) + } + (Object::Str(s), Object::Int(k)) | (Object::Int(k), Object::Str(s)) + if kind == BinOpKind::Mult => + { + let times = usize::try_from(*k).unwrap_or(0); + if s.len().saturating_mul(times) > 1 << 16 { + return None; + } + if times == 1 { + // `s * 1` is `s` itself (CPython `unicode_repeat`). + Object::Str(s.clone()) + } else { + Object::Str(SharedStr::repeat(s, times)) + } + } + (Object::Str(t), args) + if kind == BinOpKind::Mod + && percent_leaf_args(args) + && !percent_args_need_bridge(args) => + { + Object::from_str(percent_format(t, args).ok()?) + } + _ => return None, + }) + } + + /// The core loop's container instructions: `seq[i]` and `d[key]` over + /// exact containers (an in-range index or a present `str`/`int` key), + /// `seq[i] = v` in range and `d[key] = v`, and a comprehension's + /// `LIST_APPEND` — each only when every release it makes is a scalar or + /// a plain decrement. Returns the new stack length, or `None`, having + /// touched nothing, for the full leaf arms. + /// + /// # Safety + /// + /// The `len` slots at `base` are initialized operand stack entries. + #[inline(never)] + unsafe fn core_container_op( + ins: weavepy_compiler::Instruction, + base: *mut Object, + len: usize, + cap: usize, + ) -> Option { + fn scalar(v: &Object) -> bool { + matches!( + v, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) + } + match ins.op { + OpCode::BinarySubscr => { + if len < 2 { + return None; + } + // SAFETY: `len >= 2`. + let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + // (An instance goes to the caller's native subscript, which + // grades its own release.) + if !matches!( + c, + Object::List(_) | Object::Tuple(_) | Object::Dict(_) | Object::Str(_) + ) || !Self::core_droppable(c) + { + return None; + } + let r = match (c, k) { + (Object::List(xs), Object::Int(i)) => { + let xs = xs.try_borrow().ok()?; + let n = xs.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n { + return None; + } + clone_hot(&xs[i as usize]) + } + (Object::Tuple(t), Object::Int(i)) => { + let n = t.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n { + return None; + } + clone_hot(&t[i as usize]) + } + (Object::Dict(d), Object::Str(_) | Object::Int(_)) => { + let probe = crate::object::LeafProbe::new(k)?; + let d = d.try_borrow().ok()?; + clone_hot(d.get(&probe)?) + } + // `seq[a:b:c]` with plain int (or omitted) bounds: the + // full handler's own slicing, out of line (a bad step + // declines, and the full handler raises). + (Object::List(_) | Object::Tuple(_) | Object::Str(_), Object::Slice(sl)) + if [&sl.start, &sl.stop, &sl.step] + .iter() + .all(|v| matches!(v, Object::None | Object::Int(_))) => + { + Self::core_slice(c, sl)? + } + _ => return None, + }; + // SAFETY: both operand slots are initialized. The key is a + // scalar, a string or a slice of ints, and the container a + // shared value + // (`core_droppable`), so neither release runs code; the + // result takes the container's slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + base.add(len - 2).write(r); + } + Some(len - 1) + } + // `seq[a:b]` with the bounds on the stack (CPython's + // `BINARY_SLICE`): as `BINARY_SUBSCR` over the slice. + OpCode::BinarySlice => { + if len < 3 { + return None; + } + // SAFETY: `len >= 3`. + let (c, a, b) = unsafe { + ( + &*base.add(len - 3), + &*base.add(len - 2), + &*base.add(len - 1), + ) + }; + if !matches!(c, Object::List(_) | Object::Tuple(_) | Object::Str(_)) + || !Self::core_droppable(c) + || !matches!(a, Object::None | Object::Int(_)) + || !matches!(b, Object::None | Object::Int(_)) + { + return None; + } + let sl = crate::object::PySlice { + start: a.clone(), + stop: b.clone(), + step: Object::None, + }; + let r = Self::core_slice(c, &sl)?; + // SAFETY: the bounds are scalars and the container a + // shared value (checked); the result takes its slot. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + drop_hot(base.add(len - 3).read()); + base.add(len - 3).write(r); + } + Some(len - 2) + } + OpCode::StoreSubscr => { + if len < 3 { + return None; + } + // SAFETY: `len >= 3`. + let (c, k) = unsafe { (&*base.add(len - 2), &*base.add(len - 1)) }; + if !Self::core_droppable(c) { + return None; + } + let displaced_ok = |old: &Object| scalar(old) || Self::core_droppable(old); + let old = match (c, k) { + (Object::List(xs), Object::Int(i)) => { + let mut xs = xs.try_borrow_mut().ok()?; + let n = xs.len() as i64; + let i = if *i < 0 { *i + n } else { *i }; + if i < 0 || i >= n || !displaced_ok(&xs[i as usize]) { + return None; + } + // SAFETY: the value slot is initialized and leaves the + // stack into the list. + std::mem::replace(&mut xs[i as usize], unsafe { base.add(len - 3).read() }) + } + (Object::Dict(cell), Object::Str(_) | Object::Int(_)) => { + if crate::capi_watchers::dicts_active() { + return None; + } + let probe = crate::object::LeafProbe::new(k)?; + let mut d = cell.try_borrow_mut().ok()?; + let (old, changed) = match d.get_mut(&probe) { + Some(slot) => { + if !displaced_ok(slot) { + return None; + } + // SAFETY: as above. + let old = + std::mem::replace(slot, unsafe { base.add(len - 3).read() }); + let changed = !old.is_same(slot); + (old, changed) + } + None => { + if !probe.miss_is_exact() { + return None; + } + // SAFETY: as above. + d.insert(DictKey(clone_hot(k)), unsafe { + base.add(len - 3).read() + }); + (Object::None, true) + } + }; + drop(d); + if changed { + crate::object::dict_mutation_event(cell); + } + old + } + _ => return None, + }; + // SAFETY: the key and container slots are initialized (the + // value slot was moved out above); each release is a scalar + // or a plain decrement. + unsafe { + drop_hot(base.add(len - 1).read()); + drop_hot(base.add(len - 2).read()); + } + drop_hot(old); + Some(len - 3) + } + OpCode::UnpackSequence => { + let n = ins.arg as usize; + if len == 0 || n == 0 || len - 1 + n > cap { + return None; + } + // SAFETY: `len > 0`. + let seq = unsafe { &*base.add(len - 1) }; + // The sequence's release frees nothing that could finalize: + // its items outlive it on the stack, and a sole owner here + // means neither the collector nor a weakref holds it. + let unique = match seq { + Object::Tuple(t) if t.len() == n => ThinArc::strong_count(t) == 1, + Object::List(l) if Rc::strong_count(l) == 1 => true, + _ => false, + }; + if !unique && !Self::core_droppable(seq) { + return None; + } + // SAFETY: the sequence leaves its slot, which the last item + // (pushed first) takes; `len - 1 + n <= cap`. + unsafe { + let seq = base.add(len - 1).read(); + match &seq { + Object::Tuple(t) if t.len() == n => { + for (k, item) in t.iter().rev().enumerate() { + base.add(len - 1 + k).write(clone_hot(item)); + } + } + Object::List(l) => { + let Ok(items) = l.try_borrow() else { + base.add(len - 1).write(seq); + return None; + }; + if items.len() != n { + drop(items); + base.add(len - 1).write(seq); + return None; + } + for (k, item) in items.iter().rev().enumerate() { + base.add(len - 1 + k).write(clone_hot(item)); + } + } + _ => { + base.add(len - 1).write(seq); + return None; + } + } + drop_hot(seq); + } + Some(len - 1 + n) + } + OpCode::ListAppend => { + let depth = ins.arg as usize; + if len < 2 || depth == 0 || depth >= len { + return None; + } + // SAFETY: `depth < len`, so both slots are initialized. + let Object::List(lst) = (unsafe { &*base.add(len - 1 - depth) }) else { + return None; + }; + let mut l = lst.try_borrow_mut().ok()?; + // SAFETY: the value leaves the stack into the list. + l.push(unsafe { base.add(len - 1).read() }); + Some(len - 1) + } + _ => None, + } + } + + /// The next step of a dict or dict-view iterator, for the core loop: + /// the key, the value, or the item (as `(value, Some(key))` when + /// `fuse`, for an unpacking `FOR_ITER`), or exhaustion, detaching the + /// iterator from its dict. `Decline`, having changed nothing, for a + /// size or key change the checked step reports, an exhaustion that + /// could release anything (a shared iterator, the dict's last + /// reference, a subclass keepalive), or a reverse iterator. + #[inline(never)] + fn core_dict_next(it: &mut crate::object::PyIterator, fuse: bool, unique: bool) -> DictStep { + use crate::object::DictViewKind; + let crate::object::PyIterator::DictKeys { + kind, + index, + dict: dict @ Some(_), + len, + watch, + reverse: false, + owner, + } = it + else { + return DictStep::Decline; + }; + let d = dict.as_ref().expect("matched above"); + // SAFETY: a read with nothing running (see `peek`). + let Some(data) = (unsafe { d.peek() }) else { + return DictStep::Decline; + }; + if data.len() != *len + || watch + .as_ref() + .is_some_and(crate::object::DictWatch::changed) + { + return DictStep::Decline; + } + let Some((k, v)) = data.get_index(*index) else { + if !unique || owner.is_some() || Rc::strong_count(d) == 1 { + return DictStep::Decline; + } + // CPython clears `di_dict` on the first StopIteration. + *dict = None; + *watch = None; + return DictStep::Exhausted; + }; + let item = match kind { + DictViewKind::Keys => (clone_hot(&k.0), None), + DictViewKind::Values => (clone_hot(v), None), + DictViewKind::Items if fuse => (clone_hot(v), Some(clone_hot(&k.0))), + DictViewKind::Items => ( + Object::new_tuple_array([clone_hot(&k.0), clone_hot(v)]), + None, + ), + }; + if watch.is_none() { + *watch = Some(crate::object::DictWatch::new(d)); + } + *index += 1; + DictStep::Item(item.0, item.1) + } + /// Whether the core loop may release `v` with a plain drop: a scalar, /// a string (freeing one runs no code), or a shared heap value whose /// release is a bare decrement the collector need not hear about (the @@ -13542,10 +15576,19 @@ impl Interpreter { // hit must not reject this nonfinal release. The last // owner still takes ordinary teardown, and every store or // weakref operation that requires tracking revokes the flag. - Rc::strong_count(i) > 1 && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) - } - Object::List(l) => Rc::strong_count(l) > 1 && !gc_trace::note_dropped_marks(v), - Object::Dict(d) => Rc::strong_count(d) > 1 && !gc_trace::note_dropped_marks(v), + // Past two owners only a weakref-watched object can be at + // its dead line (see `gc_trace::note_dropped_marks`). + gc_trace::drop_survives_plainly(v) + || Rc::strong_count(i) > 1 + && (i.is_gc_deferred() || !gc_trace::note_dropped_marks(v)) + } + Object::List(l) if Rc::strong_count(l) > 1 => !gc_trace::note_dropped_marks(v), + Object::Dict(d) if Rc::strong_count(d) > 1 => !gc_trace::note_dropped_marks(v), + // The last owner of a container the collector never took, and + // no weakref watches, holding only scalars: freeing it runs no + // code (`prompt_reap_dropped` reaches the same plain drop the + // long way round). + Object::List(_) | Object::Dict(_) => Self::core_plain_last_container(v), Object::Tuple(t) => ThinArc::strong_count(t) > 1 && !gc_trace::note_dropped_marks(v), Object::Function(f) => Rc::strong_count(f) > 1 && !gc_trace::note_dropped_marks(v), Object::Type(t) => Rc::strong_count(t) > 1 && !gc_trace::note_dropped_marks(v), @@ -13553,6 +15596,53 @@ impl Interpreter { } } + /// [`Self::core_droppable`] for the last owner of an exact `list` or + /// `dict`: untracked (the miss filter proves it), unwatched, and + /// holding only scalars. + #[inline(never)] + fn core_plain_last_container(v: &Object) -> bool { + let id = crate::weakref_registry::id_of(v); + !gc_trace::maybe_tracked(id) + && !crate::weakref_registry::may_have_weakrefs(id) + && !crate::capi_watchers::dicts_active() + && !crate::stdlib::testinternalcapi_mod::reftrace_print_active() + && Self::is_scalar_leaf_container(v) + } + + /// Module scope for the core loop's `LOAD_NAME` / `STORE_NAME` arms: + /// names resolve in the globals, then the builtins, both exact dicts. + #[inline(always)] + fn core_name_scope(&self, frame: &Frame) -> bool { + frame.class_namespace.is_none() + && frame.class_namespace_obj.is_none() + && frame.builtins_obj.is_none() + && !frame.code.is_class_body + && !self.globals_missing_any.get() + } + + /// The core loop's `isinstance(obj, cls)` (the leaf kind's settled + /// shapes, see `leaf_builtin_call`); `None` declines untouched. + #[inline(never)] + fn core_isinstance(ops: &[Object]) -> Option> { + let [obj, Object::Type(cls)] = ops else { + return None; + }; + if !cls.metaclass_is_type() { + return None; + } + let r = match obj { + Object::Instance(inst) => { + if !inst.cls_raw().is_subclass_of(cls) { + return None; + } + true + } + Object::File(_) => return None, + obj => builtins::class_of(obj).is_subclass_of(cls), + }; + Some(Ok(Object::Bool(r))) + } + /// Whether a returning frame's `locals` owe the exit reap /// (`reap_frame_locals_on_exit`) nothing: scalars and strings, and /// instances or containers held beyond every slot of the frame that @@ -13561,37 +15651,62 @@ impl Interpreter { /// for each release. #[inline] fn core_escaped_locals(locals: &[Object]) -> bool { - let mut heap = 0usize; + // Each heap local's identity and strong count: up to eight are + // compared exactly (how many slots here name each object); past + // that, every heap slot is assumed to name every object. + const EXACT: usize = 8; + let mut heap = [const { std::mem::MaybeUninit::<(u64, usize)>::uninit() }; EXACT]; + let mut n = 0usize; for o in locals { - match o { + let (id, sc) = match o { Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None | Object::Unbound - | Object::Str(_) => {} - Object::Instance(_) | Object::List(_) | Object::Dict(_) | Object::Tuple(_) => { - heap += 1; - } + | Object::Str(_) => continue, + Object::Instance(i) => (Rc::as_ptr(i) as usize as u64, Rc::strong_count(i)), + Object::List(l) => (Rc::as_ptr(l) as usize as u64, Rc::strong_count(l)), + Object::Dict(d) => (Rc::as_ptr(d) as usize as u64, Rc::strong_count(d)), + Object::Tuple(t) => ( + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ThinArc::strong_count(t), + ), _ => return false, + }; + if n < EXACT { + heap[n].write((id, sc)); } + n += 1; } - if heap == 0 { + if n == 0 { return true; } - locals.iter().all(|o| { - let (sc, id) = match o { - Object::Instance(i) => (Rc::strong_count(i), Rc::as_ptr(i) as usize as u64), - Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), - Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), - Object::Tuple(t) => ( - ThinArc::strong_count(t), - ThinArc::as_ptr(t).cast::<()>() as usize as u64, - ), - _ => return true, - }; - sc > heap + usize::from(gc_trace::maybe_tracked(id)) - && !crate::weakref_registry::may_have_weakrefs(id) + if n > EXACT { + return locals.iter().all(|o| { + let (sc, id) = match o { + Object::Instance(i) => (Rc::strong_count(i), Rc::as_ptr(i) as usize as u64), + Object::List(l) => (Rc::strong_count(l), Rc::as_ptr(l) as usize as u64), + Object::Dict(d) => (Rc::strong_count(d), Rc::as_ptr(d) as usize as u64), + Object::Tuple(t) => ( + ThinArc::strong_count(t), + ThinArc::as_ptr(t).cast::<()>() as usize as u64, + ), + _ => return true, + }; + sc > n + usize::from(gc_trace::maybe_tracked(id)) + && !crate::weakref_registry::may_have_weakrefs(id) + }); + } + // SAFETY: the first `n` entries were written. + let heap: &[(u64, usize)] = unsafe { std::slice::from_raw_parts(heap.as_ptr().cast(), n) }; + heap.iter().all(|&(id, sc)| { + // Held past every slot here and a collector handle needs no + // count of the slots that name it. + (sc > n + 1 || { + let held = heap.iter().filter(|&&(other, _)| other == id).count(); + sc > held + usize::from(gc_trace::maybe_tracked(id)) + }) && !crate::weakref_registry::may_have_weakrefs(id) }) } @@ -13656,7 +15771,10 @@ impl Interpreter { // `PyFunction::code`); only compared, then cloned below. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; let (missing, slot_self) = slot.hit(Rc::as_ptr(f), Rc::as_ptr(code_rc))?; - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self + || !Self::lean_code_ok(code_rc) + || Self::has_extended_params(code_rc) + { return None; } let missing = missing as usize; @@ -13702,9 +15820,7 @@ impl Interpreter { }; locals.extend(f.defaults[f.defaults.len() - missing..].iter().cloned()); } - if locals.len() < nlocals { - locals.resize(nlocals, Object::Unbound); - } + fill_unbound(locals, nlocals); frame.pc = pc as u32 + 1; // The cells handle is `f`'s (now `callable`'s) or the shared empty // vector: moving the callable leaves it where it was. @@ -13772,6 +15888,234 @@ impl Interpreter { } } + /// [`Self::core_simple_args`] for a native callee, which takes its + /// operands as a slice: bitwise copies of the arguments (after the + /// receiver in `ops[0]`), never dropped, so no reference moves. + /// Returns the argument count and the `CALL`'s pc. + /// + /// # Safety + /// + /// As [`Self::core_simple_args`]. + #[inline(always)] + unsafe fn core_simple_ops( + instrs: &[weavepy_compiler::Instruction], + mut pc: usize, + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ops: &mut [std::mem::MaybeUninit; 8], + ) -> Option<(usize, usize)> { + let mut n = 1; + loop { + let ins = *instrs.get(pc)?; + // (Checked before any copy is taken: a borrowed copy must never + // be dropped.) + if n == 8 && ins.op != OpCode::Call { + return None; + } + let v = match ins.op { + OpCode::LoadFast => { + let i = ins.arg as usize; + if i >= nlocals { + return None; + } + // SAFETY: `i < nlocals` (see the function docs). + let p = unsafe { lbase.add(i) }; + if matches!(unsafe { &*p }, Object::Unbound) { + return None; + } + // SAFETY: a borrowed copy of a live local. + unsafe { std::ptr::read(p) } + } + // SAFETY: a borrowed copy of a live constant. + OpCode::LoadConst => unsafe { std::ptr::read(consts.get(ins.arg as usize)?) }, + OpCode::LoadSmallInt => Object::Int(i64::from(ins.arg)), + // `x.m(i + 1)`: an int operation on the two latest + // arguments folds into one (scalars: the copies own nothing). + OpCode::BinaryOp if n >= 3 => { + // SAFETY: entries `1..n` were written. + let (a, b) = + unsafe { (ops[n - 2].assume_init_ref(), ops[n - 1].assume_init_ref()) }; + let (&Object::Int(a), &Object::Int(b)) = (a, b) else { + return None; + }; + // SAFETY: `BinOpKind` is `repr(u8)` and the compiler only + // emits valid kinds. + let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; + let r = match kind { + BinOpKind::Add => a.checked_add(b)?, + BinOpKind::Sub => a.checked_sub(b)?, + BinOpKind::Mult => a.checked_mul(b)?, + BinOpKind::BitAnd => a & b, + BinOpKind::BitOr => a | b, + BinOpKind::BitXor => a ^ b, + _ => return None, + }; + ops[n - 2].write(Object::Int(r)); + n -= 1; + pc += 1; + continue; + } + OpCode::Call if ins.arg as usize == n - 1 => return Some((n - 1, pc)), + _ => return None, + }; + ops[n].write(v); + n += 1; + pc += 1; + } + } + + /// The native fast `__getitem__` the `BINARY_SUBSCR` at `pc` cached for + /// `ops[0]`'s class (see [`Self::leaf_instance_subscript`]), when both + /// operands leave by plain decrements. + #[inline(never)] + fn core_native_subscript( + &self, + code: &CodeObject, + pc: usize, + ops: &[Object], + ) -> Option { + let [Object::Instance(inst), key] = ops else { + return None; + }; + let cls = inst.cls_raw(); + if !Self::core_droppable(&ops[0]) + || !Self::core_droppable(key) + || !Self::default_getattribute(cls) + || crate::object::exotic_str_keys_possible() + { + return None; + } + let fast = code_vm_ext(code)? + .method_slots + .get()? + .get(pc)? + .get_native_subscript(cls.attr_version.get(), leaf_builtins::generation())?; + #[cfg(test)] + NATIVE_SUBSCRIPT_CACHE_HITS.with(|hits| hits.set(hits.get() + 1)); + Some(fast) + } + + /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); ; CALL k` on a local instance `recv` whose class's `m` + /// is a registered native builtin that runs no Python code (the call + /// site's leaf kind): called on bitwise views of the borrowed operands, + /// so no reference to the receiver, the method or an argument is taken + /// or released. Returns the call's outcome and its pc; `None` (nothing + /// touched) runs the instructions one by one. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_native_method( + &self, + code: &CodeObject, + recv: &Object, + attr_pc: usize, + name_idx: u32, + mslots: &[MethodSlot], + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ) -> Option<(Result, usize)> { + let Object::Instance(inst) = recv else { + return None; + }; + let cls = inst.cls_raw(); + let b = mslots + .get(attr_pc)? + .peek_inst_builtin(cls.attr_version.get())?; + if !Self::default_getattribute(cls) { + return None; + } + // Bitwise views of the operands, never dropped: the callee reads + // them (and clones what it keeps), and nothing it runs can reach + // the locals or constants they alias. + let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; + // SAFETY: the receiver is the core loop's local. + ops[0].write(unsafe { std::ptr::read(recv) }); + // SAFETY: the core loop's own locals and constants. + let (nargs, call_pc) = unsafe { + Self::core_simple_ops( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut ops, + ) + }?; + let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; + // SAFETY: the builtin lives in the slot (and its class) for the + // whole call, which runs no Python code. + let b = unsafe { &*b }; + if !matches!(kind, LeafKind::Opaque | LeafKind::Fast(_)) + || inst_may_shadow(inst, code, name_idx) + { + return None; + } + let n = nargs + 1; + // SAFETY: the first `n` entries were written. + let ops = unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), n) }; + let r = match kind { + LeafKind::Fast(f) => f(ops)?, + _ => match b.call_kw.as_ref() { + Some(ckw) => ckw(ops, &[]), + None => (b.call)(ops), + }, + }; + Some((r, call_pc)) + } + + /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); ; CALL k` on a local `list`, `dict`, `set` or `str` + /// whose method the `LOAD_ATTR` site cached under the receiver's tag and + /// the `CALL` site admitted as a leaf kind: called on bitwise views of + /// the borrowed operands, as [`Self::core_native_method`] does, so no + /// receiver clone is pushed and released (a release the collector + /// grades). `None` (nothing touched) runs the instructions one by one. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_builtin_method( + &self, + code: &CodeObject, + recv: &Object, + attr_pc: usize, + mslots: &[MethodSlot], + lbase: *const Object, + nlocals: usize, + consts: &[Object], + ) -> Option<(Result, usize)> { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + Object::Str(_) => 4, + _ => return None, + }; + let b = mslots.get(attr_pc)?.get_builtin_ptr(tag)?; + // Bitwise views, never dropped (see `core_native_method`). + let mut ops = [const { std::mem::MaybeUninit::::uninit() }; 8]; + // SAFETY: the receiver is the core loop's local. + ops[0].write(unsafe { std::ptr::read(recv) }); + // SAFETY: the core loop's own locals and constants. + let (nargs, call_pc) = unsafe { + Self::core_simple_ops( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut ops, + ) + }?; + let kind = mslots.get(call_pc)?.get_leaf_ptr(b)?; + // SAFETY: the first `nargs + 1` entries were written. + let ops = unsafe { std::slice::from_raw_parts(ops.as_ptr().cast::(), nargs + 1) }; + // SAFETY: the slot (and the builtin type) holds the method for the + // call, which runs no Python code. + let r = self.leaf_builtin_call(kind, unsafe { &*b }, ops)?; + Some((r, call_pc)) + } + /// A fused simple call of pure leaf `fp` (see [`code_is_pure_leaf`]): /// `args[..nargs]` are the borrowed operands (the receiver first for a /// bound call, `has_self`), and `call_pc` the `CALL` whose site slot @@ -13793,11 +16137,13 @@ impl Interpreter { let f = unsafe { &*fp }; // SAFETY: GIL-serialized raw read of the function's code cell. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) { + let pure = code_is_pure_leaf(code_rc); + let effect = !pure && code_is_effect_leaf(code_rc); + if !(pure || effect) || !pure_leaf_warm(code_rc) { return None; } let (missing, slot_self) = code_call_slot(code, call_pc)?.hit(fp, Rc::as_ptr(code_rc))?; - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self || !Self::leaf_code_ok(code_rc) { return None; } let missing = missing as usize; @@ -13819,7 +16165,11 @@ impl Interpreter { for (k, o) in f.defaults[f.defaults.len() - missing..].iter().enumerate() { args[nargs + k] = o; } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } } /// The core loop's fused `LOAD_FAST x; LOAD_ATTR m (method); = unsafe { &*f.code.as_ptr() }; + let ver = cls.attr_version.get(); + if nargs + 1 == callee.arg_count as usize { + let held = mslots[attr_pc] + .get_held(ver) + .filter(|h| std::ptr::eq(Rc::as_ptr(h), fp)); + if let (Some(ext), Some(held)) = (code_vm_ext(code), held) { + let func = Rc::downgrade(&held); + leaf_site_set( + ext, + code.instructions.len(), + attr_pc, + Some(LeafSiteData { + ver, + func, + code: Rc::as_ptr(callee), + effect: !code_is_pure_leaf(callee), + }), + ); + } + } Some((r, call_pc)) } + /// [`Self::core_pure_method`] at a site that verified its callee (see + /// [`LeafSite`]) for `inst`'s class version: `fp` and `callee` are the + /// site's function and code. + #[inline(never)] + #[allow(clippy::too_many_arguments)] + fn core_leaf_site_call( + &self, + code: &CodeObject, + inst: &PyInstance, + recv: &Object, + (fp, callee, effect): (*const crate::object::PyFunction, *const CodeObject, bool), + attr_pc: usize, + name_idx: u32, + lbase: *const Object, + nlocals: usize, + consts: &[Object], + depth_cell: *const std::cell::Cell, + ) -> SiteCall { + // SAFETY: the class holds the function at the site's version (and + // the weak handle is live); nothing here runs code. + let f = unsafe { &*fp }; + // SAFETY: GIL-serialized raw read of the function's code cell. + let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; + // The ordinary call's `RecursionError` check. + // SAFETY: this thread's own depth cell. + if !std::ptr::eq(Rc::as_ptr(code_rc), callee) + || !Self::default_getattribute(inst.cls_raw()) + || inst_may_shadow(inst, code, name_idx) + || unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() + { + return SiteCall::Declined; + } + // Scalars only (small ints): nothing to drop on the way out. + let mut scratch = [const { std::mem::MaybeUninit::::uninit() }; 8]; + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + args[0] = recv; + // SAFETY: the core loop's own locals and constants. + let Some((nargs, call_pc)) = (unsafe { + Self::core_simple_args( + &code.instructions, + attr_pc + 1, + lbase, + nlocals, + consts, + &mut scratch, + &mut args, + 1, + ) + }) else { + return SiteCall::Declined; + }; + let total = nargs + 1; + if total != code_rc.arg_count as usize { + return SiteCall::Declined; + } + let r = if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + }; + let Some(v) = r else { + if let Some(ext) = code_vm_ext(code) { + leaf_site_set(ext, code.instructions.len(), attr_pc, None); + } + return SiteCall::Missed; + }; + #[cfg(test)] + note_literal_argument_call(code, attr_pc + 1, call_pc, true); + SiteCall::Done(v, call_pc) + } + /// The core loop's fused `LOAD_GLOBAL f; PUSH_NULL; ; CALL k` (`fp` the global function), or `LOAD_GLOBAL C; /// LOAD_ATTR m (method); ...` (`fp` the class's plain or static @@ -13949,7 +16388,9 @@ impl Interpreter { let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; #[cfg(test)] note_predicate_stage(code_rc, 0); - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) { + let pure = code_is_pure_leaf(code_rc); + let effect = !pure && code_is_effect_leaf(code_rc); + if !(pure || effect) || !pure_leaf_warm(code_rc) { return None; } #[cfg(test)] @@ -13959,7 +16400,7 @@ impl Interpreter { #[cfg(test)] note_predicate_stage(code_rc, 2); let has_self = !matches!(ops.get(1)?, Object::Unbound); - if slot_self != has_self || !Self::lean_code_ok(code_rc) { + if slot_self != has_self || !Self::leaf_code_ok(code_rc) { return None; } let missing = missing as usize; @@ -14030,7 +16471,11 @@ impl Interpreter { for (k, o) in f.defaults[f.defaults.len() - missing..].iter().enumerate() { args[nargs + k] = o; } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + if effect { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } else { + self.pure_leaf_eval::(code_rc, f, &args[..total]) + } } /// [`Self::core_pure_call`] for a `CALL_KW` at `pc`: `ops` is the @@ -14069,10 +16514,10 @@ impl Interpreter { // `PyFunction::code`); only compared and borrowed below, while the // caller's stack keeps the function alive. let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; - if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) || !Self::lean_code_ok(code_rc) { + if !code_is_pure_leaf(code_rc) || !pure_leaf_warm(code_rc) || !Self::leaf_code_ok(code_rc) { return None; } - let (_, covered) = Self::kw_names_bind_check(f, func_id, perm, names, eff_argc)?; + let (_, covered) = Self::kw_names_bind_cached(code, pc, f, func_id, perm, names, eff_argc)?; // The ordinary call's `RecursionError` check. // SAFETY: this thread's own depth cell. if unsafe { (*depth_cell).get() } >= crate::recursion::recursion_limit() { @@ -14084,9 +16529,11 @@ impl Interpreter { { return None; } - // A pure leaf has no keyword-only parameters and no `**kwargs`. + // A pure leaf has no keyword-only parameters; its `**kwargs` + // dictionary, if any, follows the positional ones. let total = code_rc.arg_count as usize; - if total > 8 { + let arity = leaf_arity(code_rc); + if arity > 8 { return None; } let mut args: [*const Object; 8] = [std::ptr::null(); 8]; @@ -14095,9 +16542,23 @@ impl Interpreter { args[k] = o; } let kw_vals = &ops[first + eff_argc..ops.len() - 1]; + let mut varkw: Option = None; for (j, o) in kw_vals.iter().enumerate() { let slot = ((perm >> (4 * j)) & 0xF) as usize; - if slot >= total || slot == specialize::KW_TO_VARKW as usize { + if slot == specialize::KW_TO_VARKW as usize { + // Collected into a fresh dictionary, in call order (the + // bind check proved the callee has one). + varkw + .get_or_insert_with(|| { + DictData::with_capacity_and_hasher( + kw_vals.len(), + crate::fasthash::FxBuildHasher, + ) + }) + .insert(DictKey(names.get(j)?.clone()), o.clone()); + continue; + } + if slot >= total { return None; } args[slot] = o; @@ -14109,10 +16570,44 @@ impl Interpreter { .get(f.defaults.len().checked_sub(total - slot)?)?; } } - self.pure_leaf_eval::(code_rc, f, &args[..total]) + let dict; + if code_rc.has_varkeywords { + dict = Object::Dict(Rc::new(RefCell::new(varkw.unwrap_or_default()))); + args[total] = &raw const dict; + } + self.pure_leaf_eval::(code_rc, f, &args[..arity]) + } + + /// A keyword call's parameters, bound as `kw_names_fill_locals` leaves + /// them in `locals` (a `**kwargs` dictionary included), evaluated + /// frameless when `f` is a warm pure leaf: its result, or `None` + /// having done nothing observable. + pub(crate) fn bound_leaf_eval( + &self, + f: &crate::object::PyFunction, + locals: &[Object], + ) -> Option { + // SAFETY: GIL-serialized raw read of the function's code cell. + let code_rc: &Rc = unsafe { &*f.code.as_ptr() }; + let arity = leaf_arity(code_rc); + if arity > 8 + || locals.len() < arity + || !code_is_pure_leaf(code_rc) + || !pure_leaf_warm(code_rc) + || !Self::leaf_code_ok(code_rc) + || crate::recursion::current_depth() >= crate::recursion::recursion_limit() + { + return None; + } + let mut args: [*const Object; 8] = [std::ptr::null(); 8]; + for (k, o) in locals[..arity].iter().enumerate() { + args[k] = o; + } + self.pure_leaf_eval::(code_rc, f, &args[..arity]) } - /// Borrow a guarded instance-dictionary or slot cache hit. + /// Borrow a guarded instance-dictionary or slot cache hit. `names` is + /// the code's interned name objects (see [`slot_name_matches_in`]). /// /// # Safety /// @@ -14121,12 +16616,18 @@ impl Interpreter { /// shared storage and conflicting mutable borrows. #[inline(always)] unsafe fn leaf_cached_instance_field<'a>( + ext: &CodeConstObjects, code: &CodeObject, inst: &'a PyInstance, cache_pc: u32, name_idx: u32, ) -> Option<&'a Object> { use weavepy_compiler::InlineCache as IC; + // SAFETY: forwarded contract. + if let Some(v) = unsafe { field_slot_hit(ext, cache_pc as usize, inst) } { + return Some(v); + } + let names: &[Object] = &ext.name_objs; let (key_idx, cached, is_slot) = match code.caches.get(cache_pc) { IC::LoadAttrInstance { key_idx, ver } => (key_idx, ver, false), IC::LoadAttrSlot { key_idx, ver } => (key_idx, ver, true), @@ -14138,15 +16639,24 @@ impl Interpreter { } if !is_slot { // SAFETY: the caller keeps this rooted read callback-free. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(key_idx as usize)?; - slot_name_matches(code, name_idx, key).then_some(value) + let (key, value) = unsafe { inst.attr_peek_index(key_idx as usize) }?; + if !slot_name_matches_in(names, code, name_idx, key) { + return None; + } + field_slot_note( + ext, + code.instructions.len(), + cache_pc as usize, + inst, + key_idx, + ); + Some(value) } else { // SAFETY: the same rooted read as the dictionary path. let slots = unsafe { inst.slots.peek() }?; let indexed = slots .get_index(key_idx as usize) - .filter(|(key, _)| slot_name_matches(code, name_idx, key)) + .filter(|(key, _)| slot_name_matches_in(names, code, name_idx, key)) .map(|(_, value)| value); // Slot order can vary by instance or after deletion. let value = indexed.or_else(|| slots.get(code.names.get(name_idx as usize)?))?; @@ -14169,96 +16679,136 @@ impl Interpreter { /// Evaluate a pure leaf's body (see [`code_is_pure_leaf`]) on borrowed /// arguments: operands are scalars or pointers to objects that stay - /// put (nothing here stores, calls, or runs Python code), so no - /// reference is taken until the returned value's. Loads take the core - /// loop's cache-hit paths; anything else — a miss, an operand shape - /// the scalar arms don't settle, an operation that could raise — - /// abandons the evaluation with `None`, having done nothing - /// observable. - #[inline(never)] - fn pure_leaf_eval( + /// put (nothing here runs Python code, and an effect leaf's stores + /// wait for its return), so no reference is taken until the returned + /// value's. Loads take the core loop's cache-hit paths, and a call + /// evaluates a pure-leaf callee the same way; anything else — a miss, + /// an operand shape the scalar arms don't settle, an operation that + /// could raise — abandons the evaluation with `None`, having done + /// nothing observable. + #[inline(always)] + fn pure_leaf_eval( &self, code: &CodeObject, f: &crate::object::PyFunction, args: &[*const Object], ) -> Option { - use weavepy_compiler::InlineCache as IC; - #[derive(Clone, Copy)] - enum V { - /// A heap object, borrowed. - R(*const Object), - I(i64), - F(f64), - B(bool), - N, + // A positional call binds no `**kwargs` dictionary: the ordinary + // call builds it (see `leaf_arity`). + if args.len() != leaf_arity(code) || !Self::leaf_call_entry(code) { + return None; } - #[inline(always)] - fn norm(p: *const Object) -> V { - // SAFETY: `p` names a live object (see the method docs). - match unsafe { &*p } { - Object::Int(i) => V::I(*i), - Object::Float(x) => V::F(*x), - Object::Bool(b) => V::B(*b), - Object::None => V::N, - _ => V::R(p), + let r = self.leaf_eval::(code, f, args, 0); + Self::leaf_call_exit(code, r.is_some()); + r + } + + /// [`Self::pure_leaf_eval`] for a constructor's leaf `__init__` on the + /// fresh instance `args[0]` (see `leaf_run`'s `FRESH`): `None` may + /// leave that instance half-initialized, for the caller to discard. + fn leaf_init_eval( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + ) -> Option { + if args.len() != leaf_arity(code) || !Self::leaf_call_entry(code) { + return None; + } + let ext = code_vm_ext(code)?; + let r = if ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) == 16 { + // The setter shape stores at once anyway. + self.leaf_eval::(code, f, args, 0) + } else { + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run::(code, ext, plan, f, args, 0) + }; + Self::leaf_call_exit(code, r.is_some()); + r + } + + /// A frameless call's JIT bookkeeping, before it runs: `false` sends + /// it to the framed path instead. + #[inline(always)] + fn leaf_call_entry(code: &CodeObject) -> bool { + // A frameless call is an activation the JIT's warm-up never sees: + // count it as a lean one, and credit each interval to the tier-2 + // counter. The call at which a compile falls due goes to the + // framed path, which compiles the hot leaf so native callers can + // take its direct lanes. + #[cfg(feature = "jit")] + { + let hint = &code.jit_hint; + let n = hint.lean_entries(); + if n <= crate::tier2::LEAN_WARM_COMPILE_THRESHOLD_CAP { + if n + 1 == crate::tier2::lean_warm_at() + && !hint.is_not_jitable() + && !crate::tier2::jit_off_for_process() + { + if crate::tier2::note_frameless_calls(code, n + 1) { + return false; + } + hint.defer_lean_compile(); + } else { + hint.bump_lean_entries(); + } } } - #[inline(always)] - fn truth(v: V) -> Option { - Some(match v { - V::B(b) => b, - V::I(i) => i != 0, - V::F(x) => x != 0.0, - V::N => false, - // SAFETY: as `norm`. - V::R(p) => match unsafe { &*p } { - Object::Str(s) => !s.is_empty(), - Object::Tuple(t) => !t.is_empty(), - // SAFETY: a read with nothing running (see `peek`). - Object::List(l) => !unsafe { l.peek() }?.is_empty(), - Object::Dict(d) => !unsafe { d.peek() }?.is_empty(), - _ => return None, - }, - }) + let _ = code; + true + } + + /// A frameless call's bookkeeping after it ran (`hit`: it finished). + #[inline(always)] + fn leaf_call_exit(code: &CodeObject, hit: bool) { + if hit { + code.jit_hint.note_leaf_hit(); + } else { + code.jit_hint.note_leaf_miss(); } - // Both the decoded field shape and the general evaluator use - // the same callback-free comparison rules. NaNs and unsupported - // operands still fall back to the interpreter. - #[inline(always)] - fn compare(a: V, b: V, kind: CompareKind) -> Option { - const EXACT: u64 = 1 << 53; - let ord = match (a, b) { - (V::I(x), V::I(y)) => x.cmp(&y), - (V::F(x), V::F(y)) => x.partial_cmp(&y)?, - (V::I(x), V::F(y)) if x.unsigned_abs() < EXACT => (x as f64).partial_cmp(&y)?, - (V::F(x), V::I(y)) if y.unsigned_abs() < EXACT => x.partial_cmp(&(y as f64))?, - (V::B(x), V::B(y)) => x.cmp(&y), - (V::B(x), V::I(y)) => i64::from(x).cmp(&y), - (V::I(x), V::B(y)) => x.cmp(&i64::from(y)), - (V::N, V::N) if matches!(kind, CompareKind::Eq | CompareKind::NotEq) => { - std::cmp::Ordering::Equal - } - // SAFETY: these pointers name values owned by live - // arguments or the evaluator's scratch; no Python runs. - (V::R(p), V::R(q)) => match (unsafe { &*p }, unsafe { &*q }) { - (Object::Str(s), Object::Str(t)) => (**s).cmp(&**t), - _ => return None, - }, - _ => return None, - }; - Some(match kind { - CompareKind::Lt => ord.is_lt(), - CompareKind::LtE => ord.is_le(), - CompareKind::Eq => ord.is_eq(), - CompareKind::NotEq => ord.is_ne(), - CompareKind::Gt => ord.is_gt(), - CompareKind::GtE => ord.is_ge(), - }) + } + + /// [`Self::leaf_eval`] for a leaf plan's own pure-leaf call: the + /// callee's result borrowed when it outlives the callee's evaluation + /// (see [`leaf_plan::LeafRet`]), which the caller then neither clones + /// nor releases. + pub(crate) fn leaf_eval_nested( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + let ext = code_vm_ext(code)?; + if ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) >= 3 { + return self + .leaf_eval::(code, f, args, nest) + .map(leaf_plan::LeafRet::Owned); } + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run_ret::(code, ext, plan, f, args, nest) + } + + /// [`Self::pure_leaf_eval`] at call-nesting depth `nest` (a pure-leaf + /// callee of a leaf is evaluated one level down, to a small bound). + #[inline(never)] + fn leaf_eval( + &self, + code: &CodeObject, + f: &crate::object::PyFunction, + args: &[*const Object], + nest: u8, + ) -> Option { + use leaf_plan::{compare, norm, V}; let ext = code_vm_ext(code)?; let consts: &[Object] = &ext.objects; - let stamps: &[StampSlot] = ext.stamp_slots.get().map_or(&[], |s| &s[..]); - let instrs = &code.instructions; + let instrs: &[weavepy_compiler::Instruction] = &code.instructions; // Tiny return bodies need no operand stack or owned-value scratch. // The shape is certified once, alongside the pure-leaf decision; // call, observer, and recursion guards still belong to the caller. @@ -14267,6 +16817,39 @@ impl Interpreter { let start = usize::from(instrs.first()?.op == OpCode::Resume); let load = instrs.get(start)?; match shape { + 16 if EFFECT => { + // The setter: store, and return `None`. A declined + // store touched nothing (the ordinary call runs it). + // SAFETY (argument reads): the arguments stay live for + // this evaluation. + let arg = |k: u32| args.get(k as usize).map(|&p| unsafe { &*p }); + let (value, recv, store_pc) = match load.op { + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => ( + clone_hot(arg(load.arg >> 4)?), + arg(load.arg & 15)?, + start + 1, + ), + op => { + let value = match op { + OpCode::LoadConst => clone_hot(consts.get(load.arg as usize)?), + OpCode::LoadSmallInt => Object::Int(i64::from(load.arg)), + _ => clone_hot(arg(load.arg)?), + }; + (value, arg(instrs.get(start + 1)?.arg)?, start + 2) + } + }; + let Object::Instance(inst) = recv else { + return None; + }; + let store = instrs.get(store_pc)?; + let value = std::mem::ManuallyDrop::new(value); + // On `true` the value moved into the dict. + if Self::core_store_attr(code, inst, store_pc, store.arg, &value) { + return Some(Object::None); + } + drop(std::mem::ManuallyDrop::into_inner(value)); + return None; + } 3 => { // SAFETY: the argument remains live for this evaluation. return Some(clone_hot(unsafe { &**args.get(load.arg as usize)? })); @@ -14282,7 +16865,7 @@ impl Interpreter { // SAFETY: the argument roots the receiver until the // result is retained; nothing here invokes Python. if let Some(value) = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, attr.arg) + Self::leaf_cached_instance_field(ext, code, inst, pc as u32, attr.arg) } { return Some(clone_hot(value)); } @@ -14306,7 +16889,7 @@ impl Interpreter { let attr = instrs.get(pc)?; // SAFETY: the same rooted, callback-free read. let value = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, attr.arg) + Self::leaf_cached_instance_field(ext, code, inst, pc as u32, attr.arg) }?; #[cfg(test)] note_predicate_stage(code, if load_pc == start { 6 } else { 7 }); @@ -14339,281 +16922,13 @@ impl Interpreter { _ => {} } } - let mut st = [V::N; 8]; - let mut sp = 0usize; - // Values a leaf path hands back owned (a polymorphic or class - // read) stay here until the evaluation ends; the stack points in. - struct Owned { - buf: [std::mem::MaybeUninit; 4], - n: usize, - } - impl Drop for Owned { - fn drop(&mut self) { - for k in 0..self.n { - // SAFETY: the first `n` entries are initialized. - drop_hot(unsafe { self.buf[k].assume_init_read() }); - } - } - } - let mut owned = Owned { - buf: [const { std::mem::MaybeUninit::uninit() }; 4], - n: 0, - }; - let op: *mut Owned = &raw mut owned; - let own = |v: Object| -> Option { - match v { - Object::Int(i) => Some(V::I(i)), - Object::Float(x) => Some(V::F(x)), - Object::Bool(b) => Some(V::B(b)), - Object::None => Some(V::N), - v => { - // SAFETY: `owned` outlives every use of `op`, and is - // not otherwise touched while the evaluation runs. - let o = unsafe { &mut *op }; - if o.n == 4 { - return None; - } - let slot = &mut o.buf[o.n]; - slot.write(v); - o.n += 1; - Some(V::R(slot.as_ptr())) - } - } - }; - macro_rules! push { - ($v:expr) => {{ - if sp == 8 { - return None; - } - st[sp] = $v; - sp += 1; - }}; - } - macro_rules! pop { - () => {{ - if sp == 0 { - return None; - } - sp -= 1; - st[sp] - }}; - } - let mut pc = 0usize; - loop { - let ins = *instrs.get(pc)?; - match ins.op { - OpCode::Resume | OpCode::Nop | OpCode::NotTaken => {} - OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck => { - push!(norm(*args.get(ins.arg as usize)?)); - } - OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow => { - push!(norm(*args.get((ins.arg >> 4) as usize)?)); - push!(norm(*args.get((ins.arg & 15) as usize)?)); - } - OpCode::LoadConst => push!(norm(consts.get(ins.arg as usize)?)), - OpCode::LoadSmallInt => push!(V::I(i64::from(ins.arg))), - OpCode::LoadGlobal => { - // The callee's own namespaces, stamp-validated as the - // core loop's `LOAD_GLOBAL` arm. - let slot = stamps.get(pc)?; - let (gdict, bdict) = (f.globals.as_ptr(), f.builtins.as_ptr()); - let gid = specialize::rc_id(&f.globals); - // SAFETY (raw dict reads): nothing runs code here. - let g_stamp = unsafe { (*gdict).mutation_stamp() }; - let hit = match code.caches.get(pc as u32) { - IC::LoadGlobalModule { - globals_id, - key_idx, - } if globals_id == gid && slot.get() == [gid, g_stamp, 0] => unsafe { - (*gdict).get_index(key_idx as usize) - }, - IC::LoadGlobalBuiltin { - builtins_id, - key_idx, - } if builtins_id == specialize::rc_id(&f.builtins) - && !self.globals_missing_any.get() - && slot.get() - == [gid, g_stamp, unsafe { (*bdict).mutation_stamp() }] => - unsafe { (*bdict).get_index(key_idx as usize) }, - _ => return None, - }; - push!(norm(hit?.1)); - } - OpCode::LoadAttr => { - let V::R(p) = pop!() else { - return None; - }; - // SAFETY: as `norm`. - let recv = unsafe { &*p }; - let v = match recv { - Object::Instance(inst) => { - let cls = inst.cls_raw(); - if GETTER && cls.native_kind.get() != 0 { - return None; - } - // SAFETY: the receiver remains rooted by an - // argument or owned scratch; no Python runs. - let hit = unsafe { - Self::leaf_cached_instance_field(code, inst, pc as u32, ins.arg) - } - .map(std::ptr::from_ref); - match hit { - Some(v) => norm(v), - // A stale or absent site cache resolves - // through the site's own entries — the - // full handler never sees this body, so - // there is nothing to deopt to. - None => own(Self::leaf_attr_resolve_site( - code, inst, recv, pc as u32, ins.arg, - )?)?, - } - } - Object::Type(cls) => { - if GETTER && !Self::plain_metaclass(cls) { - return None; - } - match stamps.get(pc).and_then(|s| class_attr_hit(s, cls)) { - Some(v) => own(v)?, - None => { - own(Self::leaf_load_type_attr(code, cls, pc as u32, ins.arg)?)? - } - } - } - Object::Module(module) => { - if GETTER - && (crate::object::module_class(module).is_some() - || code.names.get(ins.arg as usize)?.starts_with("__")) - { - return None; - } - own(Self::leaf_load_attr_recv(code, recv, pc as u32, ins.arg)?)? - } - _ => return None, - }; - push!(v); - } - OpCode::CompareOp => { - let b = pop!(); - let a = pop!(); - // SAFETY: as in `compare_op_step`. - let kind: CompareKind = - unsafe { std::mem::transmute((ins.arg & !COMPARE_OP_TO_BOOL_FLAG) as u8) }; - push!(V::B(compare(a, b, kind)?)); - } - OpCode::IsOp => { - let b = pop!(); - let a = pop!(); - let same = match (a, b) { - (V::N, V::N) => true, - (V::B(x), V::B(y)) => x == y, - (V::I(x), V::I(y)) => Object::Int(x).is_same(&Object::Int(y)), - (V::F(x), V::F(y)) => Object::Float(x).is_same(&Object::Float(y)), - // SAFETY: as `norm`. - (V::R(p), V::R(q)) => unsafe { (*p).is_same(&*q) }, - _ => false, - }; - push!(V::B(same != (ins.arg == 1))); - } - OpCode::ToBool => { - let v = pop!(); - push!(V::B(truth(v)?)); - } - OpCode::UnaryOp => { - let v = pop!(); - // SAFETY: as in the full handler. - let kind: UnaryKind = unsafe { std::mem::transmute(ins.arg as u8) }; - push!(match (v, kind) { - (v, UnaryKind::Not) => V::B(!truth(v)?), - (V::I(i), UnaryKind::Neg) => V::I(i.checked_neg()?), - (V::F(x), UnaryKind::Neg) => V::F(-x), - (V::I(i), UnaryKind::Pos) => V::I(i), - (V::F(x), UnaryKind::Pos) => V::F(x), - (V::I(i), UnaryKind::Invert) => V::I(!i), - _ => return None, - }); - } - OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue => { - let v = pop!(); - if truth(v)? == (ins.op == OpCode::PopJumpIfTrue) { - pc += ins.arg as usize; - } - } - OpCode::PopJumpIfNone | OpCode::PopJumpIfNotNone => { - let v = pop!(); - if matches!(v, V::N) == (ins.op == OpCode::PopJumpIfNone) { - pc += ins.arg as usize; - } - } - OpCode::JumpForward => pc += ins.arg as usize, - OpCode::BinaryOp => { - let b = pop!(); - let a = pop!(); - // SAFETY: as the core loop's `BINARY_OP` arm. - let kind: BinOpKind = unsafe { std::mem::transmute(ins.arg as u8) }; - let r = match (a, b) { - (V::I(a), V::I(b)) => V::I(match kind { - BinOpKind::Add => a.checked_add(b)?, - BinOpKind::Sub => a.checked_sub(b)?, - BinOpKind::Mult => a.checked_mul(b)?, - BinOpKind::BitAnd => a & b, - BinOpKind::BitOr => a | b, - BinOpKind::BitXor => a ^ b, - BinOpKind::RShift if (0..64).contains(&b) => a >> b, - BinOpKind::LShift if (0..63).contains(&b) && ((a << b) >> b) == a => { - a << b - } - BinOpKind::FloorDiv if b > 0 && a >= 0 => a / b, - BinOpKind::Mod if b > 0 && a >= 0 => a % b, - _ => return None, - }), - (V::F(a), V::F(b)) => match Self::leaf_float_op(a, b, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - (V::I(a), V::F(b)) => match Self::leaf_float_op(a as f64, b, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - (V::F(a), V::I(b)) => match Self::leaf_float_op(a, b as f64, kind)? { - Object::Float(x) => V::F(x), - _ => return None, - }, - _ => return None, - }; - push!(r); - } - OpCode::CopyTop => { - let n = (ins.arg as usize).max(1); - if n > sp { - return None; - } - push!(st[sp - n]); - } - OpCode::Swap => { - let n = ins.arg as usize; - if n < 2 || n > sp { - return None; - } - st.swap(sp - 1, sp - n); - } - OpCode::PopTop => { - let _ = pop!(); - } - OpCode::ReturnValue => { - return Some(match pop!() { - // SAFETY: as `norm` (an owned value is cloned before - // its holder drops). - V::R(p) => clone_hot(unsafe { &*p }), - V::I(i) => Object::Int(i), - V::F(x) => Object::Float(x), - V::B(b) => Object::Bool(b), - V::N => Object::None, - }); - } - _ => return None, - } - pc += 1; - } + // The general body: its translated plan (see `leaf_plan`), built + // on the first evaluation that reaches it. + let plan = ext + .leaf_plan + .get_or_init(|| leaf_plan::build(code, ext).map(Box::new)) + .as_deref()?; + self.leaf_run::(code, ext, plan, f, args, nest) } /// The core loop's `FOR_ITER` over a generator at `pc` of `sw`'s @@ -14621,7 +16936,7 @@ impl Interpreter { /// ([`Self::try_inline_gen`]), with the generator's activation pushed /// and made the running one here. `false` touches nothing. #[inline(never)] - fn core_gen_resume(&mut self, sw: &mut CoreSwitch, pc: usize) -> bool { + fn core_gen_resume(&mut self, sw: &mut CoreSwitch, pc: usize, next_call: bool) -> bool { if !self.inline_calls_ok() { return false; } @@ -14630,7 +16945,13 @@ impl Interpreter { let act = unsafe { let depth = (*sw.inl).len(); let (frame, _, shell) = sw.activation(depth, &mut tmp); - self.try_inline_gen(&mut *frame, &mut *shell.cast::>(), pc) + self.try_inline_gen( + &mut *frame, + &mut *shell.cast::>(), + pc, + sw.depth_cell, + next_call, + ) }; let Some(act) = act else { return false; @@ -14677,25 +16998,28 @@ impl Interpreter { if depth == sw.entry_depth { sw.entry_dead = true; } - let result = self.inline_gen_finish( - &mut done, - QuietExit::Outcome { - stepped: Ok(StepOutcome::Yield(v)), - cur_pc, - }, - ); + // `inline_gen_finish`'s shell-less yield (the generator parks + // again) and `inline_gen_deliver`'s yielded value, directly. + drop(done.guard.take()); + let gen = done.gen.take().expect("a generator activation"); + let boxed = done.gen_box.take().expect("a generator activation"); + done.gen_frame = std::ptr::null_mut(); + Self::park_suspended_boxed(&gen, boxed); + drop(gen); + let call_pc = done.call_pc; + self.lean_pending_exit(done.caller_pending); + // SAFETY: the shell was `None` (checked above): no drop glue owed. + unsafe { std::ptr::write(&raw mut done.act.shell, None) }; + done.act.frame = std::ptr::from_mut::(&mut done.frame); + done.caller_pending = None; + if self.inline_pool.len() < INLINE_POOL_CAP { + self.inline_pool.push(done); + } let mut tmp = None; // SAFETY: the consumer is the innermost remaining activation. let (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; // SAFETY: as above. - let entry = unsafe { - self.inline_gen_deliver( - &mut *cframe, - &mut *cshell.cast::>(), - done, - result, - ) - }; + unsafe { (*cframe).stack.push(v) }; sw.cur = cframe; sw.last = if clast == &raw mut sw.scratch { sw.scratch = usize::MAX; @@ -14703,36 +17027,26 @@ impl Interpreter { } else { clast }; - match entry { - QuietEntry::Returned { cur_pc } => { - // SAFETY: as above. - unsafe { *sw.last = cur_pc }; - // The consumer's `Returned` entry protocol (`quiet_frame`). - if self.gil_countdown <= 2 { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - return true; - } - self.gil_countdown -= 1; - if crate::hot_gates::loop_gen() != snap_gen { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - return true; - } - // SAFETY: see `quiet_run`. - if sw.fin && unsafe { (*sw.maybe_dead).get() } { - // SAFETY: as above. - unsafe { - self.flush_lean(&mut *cframe, &mut *cshell.cast::>()); - } - if self.drain_if_maybe_dead() && crate::hot_gates::loop_gen() != snap_gen { - sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); - } - } - } - QuietEntry::Raised { err, .. } => { - sw.pending = Some(CoreExit::Stop(LeafStop::Raised(err))); + // The consumer's `QuietEntry::Returned` protocol (`quiet_frame`). + // SAFETY: as above. + unsafe { *sw.last = call_pc }; + if self.gil_countdown <= 2 { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); + return true; + } + self.gil_countdown -= 1; + if crate::hot_gates::loop_gen() != snap_gen { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); + return true; + } + // SAFETY: see `quiet_run`. + if sw.fin && unsafe { (*sw.maybe_dead).get() } { + // SAFETY: as above. + unsafe { + self.flush_lean(&mut *cframe, &mut *cshell.cast::>()); } - QuietEntry::Fresh | QuietEntry::Stop(_) => { - unreachable!("inline_gen_deliver returns or raises") + if self.drain_if_maybe_dead() && crate::hot_gates::loop_gen() != snap_gen { + sw.pending = Some(CoreExit::Stop(LeafStop::Breaker)); } } true @@ -14778,12 +17092,14 @@ impl Interpreter { // leftover operands or cells, and locals that die here with no drop // glue (scalars) or whose release the exit reap would pass over // (see `core_escaped_locals`). - let result = if frame.stack.is_empty() + let clean = frame.stack.is_empty() && frame.code.cellvars.is_empty() && Rc::strong_count(&frame.locals) == 1 // SAFETY: sole owner; nothing else reaches the vector. - && Self::core_escaped_locals(unsafe { &*frame.locals.as_ptr() }) - { + && Self::core_escaped_locals(unsafe { &*frame.locals.as_ptr() }); + let mut tmp = None; + let (cframe, clast, cshell, entry); + if clean && done.init_inst.is_none() { done.clean = true; // SAFETY: as above; the values drop by plain decrements (or // own nothing). @@ -14791,28 +17107,55 @@ impl Interpreter { drop_hot(v); } drop(done.guard.take()); - Ok(v) + // SAFETY: the caller is the innermost remaining activation. + (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; + // `inline_deliver` for a plain function's return, directly. + let call_pc = done.call_pc; + self.lean_pending_exit(done.caller_pending); + // SAFETY: the callable is moved out exactly once; the parked + // slot treats the field as stale (see `inline_deliver`). + let callable = unsafe { std::ptr::read(&raw const done.callable) }; + self.inline_park(done); + // SAFETY: as above. + unsafe { + (*cframe).stack.push(v); + self.drop_lean_callable( + &mut *cframe, + &mut *cshell.cast::>(), + callable, + ); + } + entry = QuietEntry::Returned { cur_pc: call_pc }; } else { - self.inline_finish( - &mut done, - QuietExit::Outcome { - stepped: Ok(StepOutcome::Return(v)), - cur_pc, - }, - ) - }; - let mut tmp = None; - // SAFETY: the caller is the innermost remaining activation. - let (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; - // SAFETY: as above. - let entry = unsafe { - self.inline_deliver( - &mut *cframe, - &mut *cshell.cast::>(), - done, - result, - ) - }; + let result = if clean { + done.clean = true; + // SAFETY: as above. + for v in unsafe { (*frame.locals.as_ptr()).drain(..) } { + drop_hot(v); + } + drop(done.guard.take()); + Ok(v) + } else { + self.inline_finish( + &mut done, + QuietExit::Outcome { + stepped: Ok(StepOutcome::Return(v)), + cur_pc, + }, + ) + }; + // SAFETY: the caller is the innermost remaining activation. + (cframe, clast, cshell) = unsafe { sw.activation(depth - 1, &mut tmp) }; + // SAFETY: as above. + entry = unsafe { + self.inline_deliver( + &mut *cframe, + &mut *cshell.cast::>(), + done, + result, + ) + }; + } sw.cur = cframe; sw.last = if clast == &raw mut sw.scratch { sw.scratch = usize::MAX; @@ -15136,17 +17479,92 @@ impl Interpreter { } } + /// [`crate::stdlib::datetime_native::leaf_binop`], out of line. + #[inline(never)] + fn core_native_binop( + kind: BinOpKind, + a: &Object, + b: &Object, + ) -> Option> { + crate::stdlib::datetime_native::leaf_binop(kind, a, b) + } + + /// [`crate::stdlib::datetime_native::leaf_compare`], out of line. + #[inline(never)] + fn core_native_compare( + kind: CompareKind, + a: &Object, + b: &Object, + ) -> Option> { + crate::stdlib::datetime_native::leaf_compare(kind, a, b) + } + + /// A natively served instance's public field (`dt.hour`; see + /// [`crate::stdlib::datetime_native::leaf_field`]), or `None`. + #[inline(never)] + fn core_native_field( + ext: Option<&CodeConstObjects>, + inst: &PyInstance, + name_idx: u32, + ) -> Option { + let Object::Str(name) = ext?.name_objs.get(name_idx as usize)? else { + return None; + }; + crate::stdlib::datetime_native::leaf_field(inst, name) + } + + /// A module attribute through the `LOAD_ATTR` site's cached index: + /// the value, cloned, or `None` for the helper's full path. + #[inline] + fn core_module_attr( + code: &CodeObject, + m: &crate::object::PyModule, + pc: usize, + name_idx: u32, + ) -> Option { + use weavepy_compiler::InlineCache as IC; + let IC::LoadAttrModule { module_id, key_idx } = code.caches.get(pc as u32) else { + return None; + }; + if specialize::rc_id(&m.dict) != module_id { + return None; + } + // SAFETY: a read between two instructions (see `GilCell::peek`). + let dict = unsafe { m.dict.peek() }?; + let (k, v) = dict.get_index(key_idx as usize)?; + slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + } + /// The core loop's `x.attr` on a local receiver: the cached /// instance-dict hit read in place (no borrow guard; nothing here runs /// code), anything else through [`Self::leaf_fused_local_attr`]. #[inline(never)] fn core_local_attr( + ext: Option<&CodeConstObjects>, code: &CodeObject, local: &Object, attr_pc: usize, name_idx: u32, ) -> Option { use weavepy_compiler::InlineCache as IC; + if let (Some(ext), Object::Instance(inst)) = (ext, local) { + // SAFETY: a read between two instructions (see `peek`). + if let Some(v) = unsafe { field_slot_hit(ext, attr_pc, inst) } { + return Some(Self::clone_operand(v)); + } + // A scalar class attribute the instance doesn't shadow (see + // `leaf_attr_resolve_site`). + if let Some(v) = ext + .stamp_slots + .get() + .and_then(|s| s.get(attr_pc)) + .and_then(|s| class_attr_hit_via(s, inst.cls_raw(), CLASS_ATTR_VIA_INSTANCE)) + { + if !inst_may_shadow(inst, code, name_idx) { + return Some(v); + } + } + } if let (IC::LoadAttrInstance { key_idx, ver }, Object::Instance(inst)) = (code.caches.get(attr_pc as u32), local) { @@ -15154,10 +17572,16 @@ impl Interpreter { if cls.native_kind.get() != 0 || cls.attr_version.get() != ver { return None; } + let names: &[Object] = ext.map_or(&[], |t| &t.name_objs); // SAFETY: a read between two instructions (see `peek`). - let d = unsafe { inst.dict.get()?.peek() }?; - let (k, v) = d.get_index(key_idx as usize)?; - return slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)); + let (k, v) = unsafe { inst.attr_peek_index(key_idx as usize) }?; + if !slot_name_matches_in(names, code, name_idx, k) { + return None; + } + if let Some(ext) = ext { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, key_idx); + } + return Some(Self::clone_operand(v)); } Self::leaf_fused_local_attr(code, local, attr_pc, name_idx) } @@ -15174,6 +17598,7 @@ impl Interpreter { attr_pc: usize, ) -> Option<(Object, usize)> { use weavepy_compiler::InlineCache as IC; + let ext = code_vm_ext(code); let mut value = root; let mut count = 0; for pc in attr_pc..attr_pc.saturating_add(8) { @@ -15186,6 +17611,12 @@ impl Interpreter { let Object::Instance(inst) = value else { break; }; + // SAFETY: the rooted, callback-free walk described below. + if let Some(next) = ext.and_then(|ext| unsafe { field_slot_hit(ext, pc, inst) }) { + value = next; + count += 1; + continue; + } let cls = inst.cls_raw(); if cls.native_kind.get() != 0 { break; @@ -15214,11 +17645,9 @@ impl Interpreter { } _ => { // SAFETY: the same rooted, callback-free walk as above. - let Some(dict) = inst.dict.get().and_then(|d| unsafe { d.peek() }) else { - break; - }; let indexed = |index| { - dict.get_index(index as usize) + // SAFETY: as above. + unsafe { inst.attr_peek_index(index as usize) } .filter(|(key, _)| slot_name_matches(code, ins.arg, key)) .map(|(_, value)| value) }; @@ -15226,7 +17655,13 @@ impl Interpreter { IC::LoadAttrInstance { key_idx, ver: cached, - } if ver == cached => indexed(key_idx), + } if ver == cached => { + let hit = indexed(key_idx); + if let (Some(_), Some(ext)) = (hit, ext) { + field_slot_note(ext, code.instructions.len(), pc, inst, key_idx); + } + hit + } _ => None, }; primary.or_else(|| { @@ -15284,6 +17719,34 @@ impl Interpreter { ) -> bool { use weavepy_compiler::InlineCache as IC; let cls = inst.cls_raw(); + let ext = code_vm_ext(code); + if let Some((ver, idx)) = ext + .and_then(|e| e.field_slots.get()) + .and_then(|slots| slots.get(attr_pc)) + .map(FieldSlot::get) + { + if cls.attr_version.get() == ver && !crate::capi_watchers::dicts_active() { + // SAFETY: a store between two instructions (see `peek_mut`). + if let Some(slot) = + unsafe { inst.split_field_mut(idx as usize, value.is_gc_atomic()) } + { + if Self::core_droppable(slot) { + // SAFETY: as the indexed store below. + drop(std::mem::replace(slot, unsafe { std::ptr::read(value) })); + return true; + } + return false; + } + // The constructor shape: the attribute is the instance's + // next one. + // SAFETY: as above; the value moves out of the caller's + // stack slot, and a declined append hands the bits back. + match unsafe { inst.split_append(idx as usize, std::ptr::read(value)) } { + Ok(()) => return true, + Err(v) => std::mem::forget(v), + } + } + } match code.caches.get(attr_pc as u32) { IC::StoreAttrInstance { key_idx, ver } => { if cls.native_kind.get() != 0 @@ -15292,29 +17755,24 @@ impl Interpreter { { return false; } + // Replacing a value leaves the keys (and so the stamp) alone. // SAFETY: a read between two instructions (see `peek`). - let Some(d) = inst.dict.get().and_then(|d| unsafe { d.peek_mut() }) else { + let Some((k, slot)) = + (unsafe { inst.attr_peek_index_mut(key_idx as usize, value.is_gc_atomic()) }) + else { return false; }; - match d.get_index(key_idx as usize) { - Some((k, old)) - if slot_name_matches(code, name_idx, k) && Self::core_droppable(old) => {} - _ => return false, - } - // Replacing a value leaves the keys (and so the stamp) alone. - let map = if value.is_gc_atomic() { - d.map_mut_unstamped() - } else { - d.map_mut_value_store() - }; - let Some((_, slot)) = map.get_index_mut(key_idx as usize) else { + if !slot_name_matches(code, name_idx, k) || !Self::core_droppable(slot) { return false; - }; + } // SAFETY: the value moves out of the caller's stack slot // (the caller drops the slot without dropping the value) // and the displaced value was checked droppable above. let old = std::mem::replace(slot, unsafe { std::ptr::read(value) }); drop(old); + if let Some(ext) = ext { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, key_idx); + } true } IC::StoreAttrNewKey { ver } => { @@ -15324,6 +17782,87 @@ impl Interpreter { } } + /// [`Self::core_store_attr_ready`] through the site's [`FieldSlot`], + /// for the next of an effect leaf's stores to one receiver in order + /// (`cursor` as [`PyInstance::split_store_ready`]): `None` when the + /// shortcut can't answer. A `true` here is a store the shortcut in + /// [`Self::core_store_attr`] then performs. + fn core_store_attr_ready_split( + ext: &CodeConstObjects, + inst: &PyInstance, + attr_pc: usize, + cursor: &mut Option, + ) -> Option { + let (ver, idx) = ext.field_slots.get()?.get(attr_pc)?.get(); + if inst.cls_raw().attr_version.get() != ver || crate::capi_watchers::dicts_active() { + return None; + } + // SAFETY: a read with nothing running (the commit's check pass). + unsafe { inst.split_store_ready(idx as usize, cursor, Self::core_droppable) } + } + + /// Whether [`Self::core_store_attr`] would perform this store right + /// now (it declines exactly when this is `false`), touching nothing: + /// an effect leaf validates every buffered store before committing + /// any of them. + fn core_store_attr_ready( + code: &CodeObject, + inst: &PyInstance, + attr_pc: usize, + name_idx: u32, + ) -> bool { + use weavepy_compiler::InlineCache as IC; + let cls = inst.cls_raw(); + let ver = match code.caches.get(attr_pc as u32) { + IC::StoreAttrInstance { ver, .. } | IC::StoreAttrNewKey { ver } => ver, + _ => return false, + }; + if cls.native_kind.get() != 0 + || cls.attr_version.get() != ver + || crate::capi_watchers::dicts_active() + { + return false; + } + match code.caches.get(attr_pc as u32) { + IC::StoreAttrInstance { key_idx, .. } => { + // SAFETY: an unused view with nothing running. + matches!(unsafe { inst.attr_peek_index(key_idx as usize) }, Some((k, old)) + if slot_name_matches(code, name_idx, k) && Self::core_droppable(old)) + } + _ => { + let Some(dict) = inst.dict.published() else { + // Split storage: the store overwrites a droppable + // value or appends (or declines, touching nothing). + // SAFETY: as above. + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return false; + }; + let Some(Object::Str(name)) = code_name_obj(code, name_idx) else { + return false; + }; + return split + .position(name) + .is_none_or(|i| Self::core_droppable(&split.values()[i])); + }; + let Some(probe) = code_name_leaf_probe(code, name_idx) else { + return false; + }; + // SAFETY: as above. + let Some(d) = (unsafe { dict.peek_mut() }) else { + return false; + }; + let found = d.get_index_of(&probe); + if probe.saw_exotic() { + return false; + } + found.is_none_or(|i| { + d.get_index(i) + .is_some_and(|(_, old)| Self::core_droppable(old)) + }) + } + } + } + /// The constructor shape of [`Self::core_store_attr`]: a /// `StoreAttrNewKey` site's insert-or-overwrite, in one probe. Out of /// line so the warm indexed store stays small. @@ -15344,6 +17883,42 @@ impl Interpreter { { return false; } + // The split layout: no hash table at all while the instance + // assigns its attributes in its class's usual order. + if inst.dict.published().is_none() { + let Some(Object::Str(name)) = code_name_obj(code, name_idx) else { + return false; + }; + // SAFETY: a read between two instructions (see `peek`). + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return false; + }; + if split + .position(name) + .is_some_and(|i| !Self::core_droppable(&split.values()[i])) + { + return false; + } + // SAFETY: the value moves out of the caller's stack slot (the + // caller drops the slot without dropping the value); a + // declined store hands the bits back, still owned there. + match inst.split_store(name, unsafe { std::ptr::read(value) }) { + Ok(old) => { + drop(old); + // Later stores at this site (to the same position of + // later instances) take the split shortcut. + if let Some(ext) = code_vm_ext(code) { + // SAFETY: a read between two instructions. + let pos = unsafe { inst.dict.split_peek() }.and_then(|s| s.position(name)); + if let Some(pos) = pos.and_then(|p| u32::try_from(p).ok()) { + field_slot_note(ext, code.instructions.len(), attr_pc, inst, pos); + } + } + return true; + } + Err(v) => std::mem::forget(v), + } + } let atomic = value.is_gc_atomic(); // The class facts were validated when the site specialized and // `ver` guards them since. @@ -15817,9 +18392,23 @@ impl Interpreter { Object::Str(SharedStr::repeat(a, times)) } } + // `"k%d" % i`: formatting scalars and strings runs no + // Python code; any error is the full handler's to + // raise. + (Object::Str(a), args) + if kind == BinOpKind::Mod + && percent_leaf_args(args) + && !percent_args_need_bridge(args) => + { + match percent_format(a, args) { + Ok(s) => Object::from_str(s), + Err(_) => break, + } + } _ => break, }; - // Operands are scalars or strings: nothing to grade. + // Operands are scalars or strings (or a tuple of them): + // nothing to grade. drop_operand(stack.pop().expect("length checked above")); drop_operand(std::mem::replace(&mut stack[n - 2], r)); last = pc; @@ -16274,6 +18863,45 @@ impl Interpreter { last = pc; pc += 1; } + // `{**a}` and `f(**kw)`'s merge of a plain dictionary whose + // keys, like the target's, are all `str`: their hashing and + // equality are native, so no Python runs. A call's merge + // that would repeat a keyword raises in the full handler. + OpCode::DictUpdate => { + let depth = (ins.arg >> 1) as usize + 1; + let n = stack.len(); + if n < depth + 1 { + break; + } + let (Object::Dict(target), Object::Dict(src)) = + (&stack[n - 1 - depth], &stack[n - 1]) + else { + break; + }; + let (Ok(mut t), Ok(s)) = (target.try_borrow_mut(), src.try_borrow()) else { + break; + }; + let str_keys = |d: &DictData| d.keys().all(|k| matches!(k.0, Object::Str(_))); + if !str_keys(&s) + || !str_keys(&t) + || (ins.arg & 1 != 0 && s.keys().any(|k| t.contains_key(k))) + { + break; + } + for (k, v) in s.iter() { + t.insert(k.clone(), v.clone()); + } + drop((t, s)); + let other = stack.pop().expect("length checked above"); + last = pc; + pc += 1; + if gc_trace::note_dropped_marks(&other) { + gc_trace::mark_maybe_dead(); + drop(other); + stop = LeafStop::Marked; + break; + } + } OpCode::Swap => { let depth = ins.arg as usize; let n = stack.len(); @@ -16471,16 +19099,17 @@ impl Interpreter { let v = std::mem::replace(value_slot, Object::Unbound); std::mem::replace(slot, v) } - (Object::Dict(d), key @ (Object::Str(_) | Object::Int(_))) => { + (Object::Dict(cell), key @ (Object::Str(_) | Object::Int(_))) => { if crate::capi_watchers::dicts_active() { break; } let Some(probe) = crate::object::LeafProbe::new(key) else { break; }; - let Ok(mut d) = d.try_borrow_mut() else { break }; - let d = &mut *d; - match d.get_mut(&probe) { + let Ok(mut d) = cell.try_borrow_mut() else { + break; + }; + let (old, changed) = match d.get_mut(&probe) { Some(slot) => { if Self::local_needs_prompt_reap(slot) && Self::looks_reapable_temporary(slot) @@ -16488,17 +19117,24 @@ impl Interpreter { break; } let v = std::mem::replace(value_slot, Object::Unbound); - std::mem::replace(slot, v) + let old = std::mem::replace(slot, v); + let changed = !old.is_same(slot); + (old, changed) } None => { - if d.len() > 8 || !d.keys().all(|k| k.0.is_gc_atomic()) { + if !probe.miss_is_exact() { break; } let v = std::mem::replace(value_slot, Object::Unbound); d.insert(DictKey(key.clone()), v); - Object::Unbound + (Object::Unbound, true) } + }; + drop(d); + if changed { + crate::object::dict_mutation_event(cell); } + old } _ => break, }; @@ -17078,6 +19714,18 @@ impl Interpreter { Object::Str(_) | Object::Int(_) | Object::Bool(_) | Object::None ) => { + // A `str` or `int` key settles natively (see `LeafProbe`). + if let (Object::Dict(d), Some(probe)) = + (container, crate::object::LeafProbe::new(item)) + { + let m = d.try_borrow().ok()?; + if m.get(&probe).is_some() { + return Some(true); + } + if probe.miss_is_exact() { + return Some(false); + } + } let key = DictKey(item.clone()); let (found, deferred) = crate::object::with_key_eq_deferred(|| { crate::object::key_cmp_scope(|| match container { @@ -17124,39 +19772,176 @@ impl Interpreter { None } - fn leaf_instance_truth(&self, v: &Object) -> Option { + /// `len(v)` for an instance whose class serves `__len__` with a + /// registered native leaf builtin (found through the class's leaf + /// attribute cache): the length, or `None` for the full path (which + /// raises whatever a missing or misbehaving `__len__` must). + fn leaf_instance_len(&self, v: &Object) -> Option { + use crate::types::LeafAttrKind as K; + /// The cache key for `__len__` (an address no interned name has). + static LEN_KEY: u8 = 0; let Object::Instance(inst) = v else { return None; }; - // `NotImplemented` has neither method yet refuses a boolean - // context (a TypeError since 3.14): the full path raises it. - if v.is_same(&crate::vm_singletons::not_implemented()) { + let cls = inst.cls_raw(); + if crate::object::exotic_str_keys_possible() { return None; } + let key = std::ptr::addr_of!(LEN_KEY) as usize; + let ver = cls.attr_version.get(); + let b = match cls.leaf_attrs.get(key, ver) { + Some(K::BuiltinMethod(b)) => Rc::as_ptr(b), + Some(_) => return None, + None => { + let kind = match cls.lookup("__len__") { + Some(Object::Builtin(b)) + if b.binds_instance + && self.leaf_call_kind(&b) == Some(LeafKind::Opaque) => + { + K::BuiltinMethod(b) + } + _ => K::Other, + }; + let found = matches!(kind, K::BuiltinMethod(_)); + cls.leaf_attrs.set(key, ver, kind); + return if found { + self.leaf_instance_len(v) + } else { + None + }; + } + }; + // SAFETY: the cache (and the class) keep the builtin alive through + // the call, whose body runs no Python. + let b = unsafe { &*b }; + let r = match b.call_kw.as_ref() { + Some(ckw) => ckw(std::slice::from_ref(v), &[]), + None => (b.call)(std::slice::from_ref(v)), + }; + match r.ok()? { + Object::Int(n) if n >= 0 => Some(Object::Int(n)), + _ => None, + } + } + + /// A call of an instance whose class's `__call__` is a plain Python + /// function: the callee slot at `callee` takes that function and the + /// (empty) self slot after it the instance, as `type(obj).__call__(obj, + /// ...)`. Anything else is left alone. + /// + /// # Safety + /// + /// `callee` and `callee + 1` are a call's initialized callee and self + /// slots. + #[inline(always)] + unsafe fn core_instance_callee(callee: *mut Object) { + // SAFETY: the caller's contract. + let (Object::Instance(inst), Object::Unbound) = + (unsafe { &*callee }, unsafe { &*callee.add(1) }) + else { + return; + }; + let Some(f) = Self::instance_call_function(inst) else { + return; + }; + // SAFETY: the instance moves to the (dropless) self slot. + unsafe { + let obj = callee.read(); + callee.write(Object::Function(f)); + callee.add(1).write(obj); + } + } + + /// `type(inst).__call__` when it is a plain Python function, cached + /// on the class under its attribute version. + #[inline(never)] + fn instance_call_function(inst: &PyInstance) -> Option> { + use crate::types::LeafAttrKind as K; + /// The cache key for `__call__` (an address no interned name has). + static CALL_KEY: u8 = 0; let cls = inst.cls_raw(); - if !Self::default_getattribute(cls) || crate::object::exotic_str_keys_possible() { + let key = std::ptr::addr_of!(CALL_KEY) as usize; + let ver = cls.attr_version.get(); + match cls.leaf_attrs.get(key, ver) { + Some(K::Method(w)) => w.upgrade(), + Some(_) => None, + None => { + let kind = match cls.lookup("__call__") { + Some(Object::Function(f)) => K::Method(Rc::downgrade(&f)), + _ => K::Other, + }; + let found = match &kind { + K::Method(w) => w.upgrade(), + _ => None, + }; + cls.leaf_attrs.set(key, ver, kind); + found + } + } + } + + fn leaf_instance_truth(&self, v: &Object) -> Option { + let Object::Instance(inst) = v else { + return None; + }; + let cls = inst.cls_raw(); + if crate::object::exotic_str_keys_possible() { return None; } - for name in ["__bool__", "__len__"] { + let ver = cls.attr_version.get(); + let fresh = !matches!(&*self.core_truth.borrow(), Some((v, _)) if *v == ver); + if fresh { + let t = self.native_truth_of(v, cls); + *self.core_truth.borrow_mut() = Some((ver, t)); + } + // The cache (and the class) keep the builtin alive through the + // call, whose body runs no Python. + let (b, is_len): (*const crate::object::BuiltinFn, bool) = match &*self.core_truth.borrow() + { + Some((_, NativeTruth::AlwaysTrue)) => return Some(true), + Some((_, NativeTruth::Bool(b))) => (Rc::as_ptr(b), false), + Some((_, NativeTruth::Len(b))) => (Rc::as_ptr(b), true), + _ => return None, + }; + // SAFETY: see above. + let b = unsafe { &*b }; + let r = match b.call_kw.as_ref() { + Some(ckw) => ckw(std::slice::from_ref(v), &[]), + None => (b.call)(std::slice::from_ref(v)), + }; + match r.ok()? { + Object::Bool(b) => Some(b), + Object::Int(n) if is_len && n >= 0 => Some(n != 0), + _ => None, + } + } + + /// How instances of `cls` answer a boolean context natively (keyed by + /// the class's attribute version in `core_truth`). + #[cold] + #[inline(never)] + fn native_truth_of(&self, v: &Object, cls: &TypeObject) -> NativeTruth { + // `NotImplemented` has neither method yet refuses a boolean + // context (a TypeError since 3.14): the full path raises it. + if v.is_same(&crate::vm_singletons::not_implemented()) || !Self::default_getattribute(cls) { + return NativeTruth::Decline; + } + for (name, is_len) in [("__bool__", false), ("__len__", true)] { match cls.lookup(name) { None => continue, Some(Object::Builtin(b)) if b.binds_instance && self.leaf_call_kind(&b) == Some(LeafKind::Opaque) => { - let r = match b.call_kw.as_ref() { - Some(ckw) => ckw(std::slice::from_ref(v), &[]), - None => (b.call)(std::slice::from_ref(v)), - }; - return match r.ok()? { - Object::Bool(b) => Some(b), - Object::Int(n) if name == "__len__" && n >= 0 => Some(n != 0), - _ => None, + return if is_len { + NativeTruth::Len(b) + } else { + NativeTruth::Bool(b) }; } - Some(_) => return None, + Some(_) => return NativeTruth::Decline, } } - Some(true) + NativeTruth::AlwaysTrue } /// Offer borrowed subscript operands to a registered native body or its @@ -17295,8 +20080,12 @@ impl Interpreter { self.leaf_fns.get_or_init(|| { let mut calls = std::collections::HashMap::default(); let mut len_ptr = 0usize; + let mut next_ptr = 0usize; { let b = self.builtins.borrow(); + if let Some(Object::Builtin(f)) = b.get(&crate::object::StrKey("next")) { + next_ptr = Rc::as_ptr(f) as usize; + } for (name, kind) in [("len", LeafKind::Len), ("isinstance", LeafKind::Isinstance)] { if let Some(Object::Builtin(f)) = b.get(&crate::object::StrKey(name)) { calls.insert(Rc::as_ptr(f) as usize, kind); @@ -17413,6 +20202,7 @@ impl Interpreter { LeafFns { calls, len_ptr, + next_ptr, methods, range_ty: bt.range_.clone(), list_ty: bt.list_.clone(), @@ -17528,8 +20318,45 @@ impl Interpreter { .map(|(_, _, f)| f.clone()) } - /// Which leaf builtin `b` is, if any (pointer identity). + /// `inst`'s class's `__next__` when it is a registered leaf builtin + /// that binds its instance (the core loop's native iterators), + /// uncounted: the cache (and the class) keep it alive until the next + /// lookup. #[inline] + fn core_leaf_next_ptr(&self, inst: &PyInstance) -> Option<*const crate::object::BuiltinFn> { + let ver = inst.cls_raw().attr_version.get(); + if let Some((v, b)) = &*self.core_next.borrow() { + if *v == ver { + return b.as_ref().map(Rc::as_ptr); + } + } + self.core_leaf_next_resolve(inst, ver) + .map(|b| Rc::as_ptr(&b)) + } + + #[cold] + #[inline(never)] + fn core_leaf_next_resolve( + &self, + inst: &PyInstance, + ver: u64, + ) -> Option> { + // A miss is remembered too: a Python-level `__next__` class asks + // again on every step of a generic iteration. + let found = match inst.cls().lookup("__next__") { + Some(Object::Builtin(b)) + if b.binds_instance + && matches!(self.leaf_call_kind(&b), Some(LeafKind::Opaque)) => + { + Some(b) + } + _ => None, + }; + *self.core_next.borrow_mut() = Some((ver, found.clone())); + found + } + + /// Which leaf builtin `b` is, if any (pointer identity). fn leaf_call_kind(&self, b: &Rc) -> Option { let p = Rc::as_ptr(b) as usize; if let Some(k) = self.leaf_fns().calls.get(&p) { @@ -17550,8 +20377,9 @@ impl Interpreter { return None; } Some(match fast { - Some(f) => LeafKind::Fast(*f), - None => LeafKind::Opaque, + leaf_builtins::Entry::Fast(f) => LeafKind::Fast(*f), + leaf_builtins::Entry::Opaque => LeafKind::Opaque, + leaf_builtins::Entry::Scalar => LeafKind::Scalar, }) } @@ -17608,12 +20436,96 @@ impl Interpreter { Some(result) } + /// A frameless leaf's method load off a builtin receiver at the site + /// `slot` (`name_idx` in `code`'s names): the leaf method table's + /// body, which that table keeps alive. + pub(crate) fn leaf_builtin_method_ptr( + &self, + slot: &MethodSlot, + recv: &Object, + code: &CodeObject, + name_idx: u16, + ) -> Option<*const crate::object::BuiltinFn> { + let tag = match recv { + Object::List(_) => 1, + Object::Dict(_) => 2, + Object::Set(_) => 3, + Object::Str(_) => 4, + _ => return None, + }; + if let Some(b) = slot.get_builtin_ptr(tag) { + return Some(b); + } + let b = self.leaf_builtin_method(recv, code.names.get(usize::from(name_idx))?)?; + slot.set_builtin(tag, &b); + Some(Rc::as_ptr(&b)) + } + + /// A frameless leaf's call at `pc` of the builtin `b` (`rc` its owner, + /// when the plan borrowed one) on `args`: the result of an admitted + /// read-only kind (see [`LeafKind::is_pure_read`]). `None` touched + /// nothing; that includes a raise, which the ordinary call repeats. + pub(crate) fn leaf_pure_builtin( + &self, + code: &CodeObject, + pc: usize, + b: *const crate::object::BuiltinFn, + rc: Option<&Rc>, + args: &[Object], + ) -> Option { + let slot = code_method_slot(code, pc as u32); + let kind = match slot.and_then(|s| s.get_leaf_ptr(b)) { + Some(k) => k, + None => self.leaf_pure_builtin_kind(slot, b, rc)?, + }; + if !kind.is_pure_read() { + return None; + } + // SAFETY: the plan's holder (a namespace, or the leaf method + // table) keeps the builtin alive, and nothing here releases it. + let r = self.leaf_builtin_call(kind, unsafe { &*b }, args)?.ok(); + #[cfg(test)] + if r.is_some() { + LEAF_BUILTIN_CALLS.with(|calls| calls.set(calls.get() + 1)); + } + r + } + + #[cold] + #[inline(never)] + fn leaf_pure_builtin_kind( + &self, + slot: Option<&MethodSlot>, + b: *const crate::object::BuiltinFn, + rc: Option<&Rc>, + ) -> Option { + let found; + let rc = match rc { + Some(rc) => rc, + None => { + found = self + .leaf_fns() + .methods + .iter() + .find(|(_, _, f)| Rc::as_ptr(f) == b)? + .2 + .clone(); + &found + } + }; + let kind = self.leaf_call_kind(rc)?; + if let Some(s) = slot { + s.set_leaf(rc, kind); + } + Some(kind) + } + /// Call a leaf builtin if `args` (receiver first for methods) has an /// admitted shape; `None` sends the call down the full path. fn leaf_builtin_call( - &mut self, + &self, kind: LeafKind, - b: &Rc, + b: &crate::object::BuiltinFn, args: &[Object], ) -> Option> { use LeafKind as K; @@ -17624,6 +20536,9 @@ impl Interpreter { let leaf_str = |o: &Object| matches!(o, O::Str(_)); let leaf_int = |o: &Object| matches!(o, O::Int(_) | O::Bool(_)); let admitted = match kind { + K::Len if matches!(args, [O::Instance(_)]) => { + return self.leaf_instance_len(&args[0]).map(Ok); + } K::Len => { args.len() == 1 && matches!( @@ -17638,37 +20553,45 @@ impl Interpreter { | O::FrozenSet(_) ) } - K::Isinstance => { - if args.len() != 2 { - return None; - } - let O::Type(cls) = &args[1] else { + K::Isinstance => return Self::core_isinstance(args), + // `list.append` and `list.pop` in line (as `list_append` and + // `list_pop` do them). + K::ListAppend => { + let [O::List(l), item] = args else { return None; }; - // A plain metaclass: no `__instancecheck__` hook. - if !Rc::ptr_eq(&cls.metaclass_or_type(), &builtin_types().type_) { + l.try_borrow_mut().ok()?.push(item.clone()); + return Some(Ok(O::None)); + } + K::ListPop => { + let (O::List(l), index) = (&args[0], args.get(1)) else { return None; - } - // Only a positive MRO answer settles an instance without - // consulting `__class__` (see `recursive_isinstance_type`); - // every other object's type is its `__class__`. - let r = match &args[0] { - O::Instance(inst) => { - if inst.cls_raw().is_subclass_of(cls) { - true + }; + let mut l = l.try_borrow_mut().ok()?; + let i = match index { + None => l.len().checked_sub(1), + Some(O::Int(_) | O::Bool(_)) if args.len() == 2 => { + let i = &match args[1] { + O::Bool(b) => i64::from(b), + O::Int(i) => i, + _ => unreachable!("matched above"), + }; + let len = l.len() as i64; + let n = if *i < 0 { i + len } else { *i }; + if l.is_empty() { + None + } else if n < 0 || n >= len { + return Some(Err(index_error("pop index out of range"))); } else { - return None; + Some(n as usize) } } - O::File(_) => return None, - obj => builtins::class_of(obj).is_subclass_of(cls), + _ => return None, }; - return Some(Ok(O::Bool(r))); - } - K::ListAppend => args.len() == 2 && matches!(args[0], O::List(_)), - K::ListPop => { - (args.len() == 1 || (args.len() == 2 && leaf_int(&args[1]))) - && matches!(args[0], O::List(_)) + return Some(match i { + Some(i) => Ok(l.remove(i)), + None => Err(index_error("pop from empty list")), + }); } K::ListInsert => args.len() == 3 && matches!(args[0], O::List(_)) && leaf_int(&args[1]), K::ListReverse | K::ListCopy => args.len() == 1 && matches!(args[0], O::List(_)), @@ -17694,6 +20617,18 @@ impl Interpreter { return None; } let O::Dict(d) = &args[0] else { return None }; + // A `str` or `int` key settles by native equality unless the + // table compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(&args[1]) { + let m = d.try_borrow().ok()?; + match m.get(&probe) { + Some(v) => return Some(Ok(v.clone())), + None if probe.miss_is_exact() => { + return Some(Ok(args.get(2).cloned().unwrap_or(O::None))); + } + None => {} + } + } // The probe may need a Python comparison against a stored // key: deferred, and settled by the full path. let (found, deferred) = crate::object::with_key_eq_deferred(|| { @@ -17839,6 +20774,9 @@ impl Interpreter { return Some(Ok(inst)); } K::Opaque => true, + K::Scalar => args + .iter() + .all(|a| matches!(a, O::Int(_) | O::Float(_) | O::Bool(_))), // The fast half decides (and declines untouched). K::Fast(f) => return f(args), }; @@ -17886,9 +20824,9 @@ impl Interpreter { if inst.cls_raw().attr_version.get() != ver { return None; } - let dict = inst.dict.get()?.try_borrow().ok()?; - let (k, v) = dict.get_index(key_idx as usize)?; - slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + inst.attr_index_map(key_idx as usize, |k, v| { + slot_name_matches(code, name_idx, k).then(|| Self::clone_operand(v)) + })? } (IC::LoadAttrSlot { key_idx, ver }, Object::Instance(inst)) => { if inst.cls_raw().attr_version.get() != ver { @@ -17952,12 +20890,8 @@ impl Interpreter { LeafAttr::InstanceOnly => { // A data descriptor or a real field wins over __getattr__. // This probe must prove a miss without comparing Python keys. - if let Some(dict) = inst.dict.get() { - let probe = code_name_leaf_probe(code, name_idx)?; - let dict = dict.try_borrow().ok()?; - if dict.contains_key(&probe) || probe.saw_exotic() { - return None; - } + if inst_may_shadow(inst, code, name_idx) { + return None; } let name = code_name_obj(code, name_idx)?; (cls.lookup("__getattr__")?, Some(name)) @@ -18000,7 +20934,7 @@ impl Interpreter { // Getter mode additionally validates metaclasses and module overrides; // ordinary pure-call evaluation avoids these extra checks. let _ = code_is_pure_leaf(&fcode); - self.pure_leaf_eval::(&fcode, &f, &args[..nargs]) + self.pure_leaf_eval::(&fcode, &f, &args[..nargs]) } /// The cache-free half of [`Self::leaf_load_attr_recv`]: the site's @@ -18017,6 +20951,15 @@ impl Interpreter { name_idx: u32, ) -> Option { use weavepy_compiler::InlineCache as IC; + // A scalar class attribute read through an instance that doesn't + // shadow it (the stamp was filled below, at this class version). + if let Some(v) = code_stamp_slot(code, cache_pc) + .and_then(|s| class_attr_hit_via(s, inst.cls_raw(), CLASS_ATTR_VIA_INSTANCE)) + { + if !inst_may_shadow(inst, code, name_idx) { + return Some(v); + } + } let poly = code_attr_poly(code, cache_pc); let ver = inst.cls_raw().attr_version.get(); // A never-specialized site resolves once and takes the indexed @@ -18036,6 +20979,15 @@ impl Interpreter { } else if let Some(p) = poly { p.record(ver, ix); } + } else { + // Found on the class: remember a scalar for this site. + class_attr_fill_via( + code, + inst.cls_raw(), + cache_pc as usize, + &v, + CLASS_ATTR_VIA_INSTANCE, + ); } Some(v) } @@ -18093,18 +21045,33 @@ impl Interpreter { prop @ LeafAttr::Property(_) => return Some((prop, None)), other => Some(other), }; - if let Some(dict) = inst.dict.get() { - let probe = code_name_leaf_probe(code, name_idx)?; - let d = dict.try_borrow().ok()?; - if let Some((ix, _, v)) = d.get_full(&probe) { - return Some(( - LeafAttr::Value(Self::clone_operand(v)), - u32::try_from(ix).ok(), - )); + match inst.dict.published() { + Some(dict) => { + let probe = code_name_leaf_probe(code, name_idx)?; + let d = dict.try_borrow().ok()?; + if let Some((ix, _, v)) = d.get_full(&probe) { + return Some(( + LeafAttr::Value(Self::clone_operand(v)), + u32::try_from(ix).ok(), + )); + } + // A key only a user `__eq__` could compare: the full path. + if probe.saw_exotic() { + return None; + } } - // A key only a user `__eq__` could compare: the full path. - if probe.saw_exotic() { - return None; + None => { + let split = inst.dict.split_cell().try_borrow().ok()?; + let ix = match code_name_obj(code, name_idx) { + Some(Object::Str(n)) => split.position(n), + _ => split.position_str(name.as_str()), + }; + if let Some(ix) = ix { + return Some(( + LeafAttr::Value(Self::clone_operand(&split.values()[ix])), + u32::try_from(ix).ok(), + )); + } } } on_class.map(|a| (a, None)) @@ -18383,14 +21350,8 @@ impl Interpreter { _ => None, }; } - if let Some(dict) = inst.dict.get() { - let d = dict.try_borrow().ok()?; - if !d.is_empty() { - let probe = code_name_leaf_probe(code, name_idx)?; - if d.contains_key(&probe) || probe.saw_exotic() { - return None; - } - } + if inst_may_shadow(inst, code, name_idx) { + return None; } let slot = code_method_slot(code, cache_pc); if let Some(f) = slot.and_then(|s| s.get(ver)) { @@ -18488,14 +21449,7 @@ impl Interpreter { // because the site never reaches it). if matches!(cache, IC::Empty) { let ver = cls.attr_version.get(); - let indexed = inst - .dict - .get() - .and_then(|d| d.try_borrow().ok()) - .and_then(|d| { - use crate::specialize::DictDataExt; - d.index_of_key_str(name) - }); + let indexed = inst.attr_position_str(name); code.caches.set( cache_pc, match indexed { @@ -18509,10 +21463,10 @@ impl Interpreter { }; match cache { IC::StoreAttrInstance { key_idx, .. } => { - let dict = inst.dict.get()?; { - let d = dict.try_borrow().ok()?; - let (k, old) = d.get_index(key_idx as usize)?; + // SAFETY: a leaf store runs no code while the + // views below are live. + let (k, old) = unsafe { inst.attr_peek_index(key_idx as usize) }?; if !slot_name_matches(code, name_idx, k) { return None; } @@ -18522,18 +21476,48 @@ impl Interpreter { return None; } } - let mut d = dict.try_borrow_mut().ok()?; - let d = &mut *d; // As the core loop's arm: an in-place value store. - let d = if value_slot.is_gc_atomic() { - d.map_mut_unstamped() - } else { - d.map_mut_value_store() - }; - let (_, slot) = d.get_index_mut(key_idx as usize)?; + // SAFETY: as above. + let (_, slot) = unsafe { + inst.attr_peek_index_mut(key_idx as usize, value_slot.is_gc_atomic()) + }?; let val = std::mem::replace(value_slot, Object::Unbound); Some(std::mem::replace(slot, val)) } + IC::StoreAttrNewKey { .. } if inst.dict.published().is_none() => { + // The split layout (see `core_store_new_attr`). + let Some(Object::Str(shared)) = code_name_obj(code, name_idx) else { + return None; + }; + { + // SAFETY: as above. + let split = unsafe { inst.dict.split_cell().peek() }?; + if let Some(i) = split.position(shared) { + let old = &split.values()[i]; + if Self::local_needs_prompt_reap(old) + && Self::looks_reapable_temporary(old) + { + return None; + } + } + } + let val = std::mem::replace(value_slot, Object::Unbound); + match inst.split_store(shared, val) { + Ok(old) => old, + Err(val) => { + // Needs the real dictionary: store there. + let dict = inst.dict_cell(); + let mut d = dict.try_borrow_mut().ok()?; + let d = &mut *d; + let d = if val.is_gc_atomic() { + d.map_mut_atomic_store() + } else { + &mut **d + }; + d.insert(DictKey(Object::Str(shared.clone())), val) + } + } + } IC::StoreAttrNewKey { .. } => { let probe = code_name_leaf_probe(code, name_idx)?; // `dict_cell`, not a bare `get_or_init`: see @@ -19244,21 +22228,15 @@ impl Interpreter { cls.attr_version.get() == ver }; if guard_ok { - inst.dict.get().and_then(|dict| { - let dict = dict.borrow(); - match dict.get_index(key_idx as usize) { - Some((k, v)) - if self.cached_slot_name_matches( - &frame.code, - attr_ins.arg, - k, - ) => - { - Some(Self::clone_operand(v)) - } - _ => None, - } + inst.attr_index_map(key_idx as usize, |k, v| { + self.cached_slot_name_matches( + &frame.code, + attr_ins.arg, + k, + ) + .then(|| Self::clone_operand(v)) }) + .flatten() } else { None } @@ -19454,7 +22432,19 @@ impl Interpreter { let old = if let Some(ns) = &frame.class_namespace { ns.borrow_mut().insert(DictKey(key), v) } else { - frame.globals.borrow_mut().insert(DictKey(key), v) + // A module-level rebinding keeps the key layout (see + // `STORE_GLOBAL`). + let key = DictKey(key); + let mut g = frame.globals.borrow_mut(); + match g.get_index_of(&key) { + Some(i) => { + crate::object::bump_global_value_epoch(); + g.map_mut_value_store() + .get_index_mut(i) + .map(|(_, slot)| std::mem::replace(slot, v)) + } + None => g.insert(key, v), + } }; if let Some(old) = old { if Self::local_needs_prompt_reap(&old) { @@ -19469,7 +22459,22 @@ impl Interpreter { Some(n @ Object::Str(_)) => n.clone(), _ => Object::from_str(self.name_at(&frame.code, ins.arg)?), }; - let old = frame.globals.borrow_mut().insert(DictKey(key), v); + // Rebinding an existing global keeps the dict's key layout, so + // it leaves the stamp (and every cached global load) standing; + // only a new name stamps it. + let key = DictKey(key); + let old = { + let mut g = frame.globals.borrow_mut(); + match g.get_index_of(&key) { + Some(i) => { + crate::object::bump_global_value_epoch(); + g.map_mut_value_store() + .get_index_mut(i) + .map(|(_, slot)| std::mem::replace(slot, v)) + } + None => g.insert(key, v), + } + }; // Rebinding a global that uniquely held a finalizable runs its // `__del__` now, matching the `DeleteGlobal` path and CPython's // decref-on-store (module-scope `handle = None`). @@ -20414,12 +23419,7 @@ impl Interpreter { // dict intact. if let Object::BoundMethod(bm) = &callable { let raw = match &bm.function { - Object::Function(f) => f - .attrs - .borrow() - .borrow() - .get(&crate::object::StrKey("__weave_raw_call__")) - .cloned(), + Object::Function(f) => f.attr_get("__weave_raw_call__"), _ => None, }; if let Some(raw) = raw { @@ -20708,8 +23708,13 @@ impl Interpreter { }, Object::Instance(_) => { // Call __next__; treat StopIteration as exhaustion. - match instance_method(&it_obj, "__next__") { - Some(m) => match self.call(&m, &[], &[], &frame.globals) { + let next = match instance_native_dunder(&it_obj, "__next__", None) { + Some(r) => Some(r), + None => instance_method(&it_obj, "__next__") + .map(|m| self.call(&m, &[], &[], &frame.globals)), + }; + match next { + Some(r) => match r { Ok(v) => Some(v), Err(RuntimeError::PyException(exc)) if exc.type_name() == "StopIteration" => @@ -21236,14 +24241,18 @@ impl Interpreter { // sits below the self-or-null slot, the args tuple and the // kwargs dict (RFC 0068 WS1's self-or-null convention // pushed it one slot deeper). - let kw_error_prefix: String = if is_kw_merge { - frame - .peek_back(depth + 2) + // (Rendered only for an error.) + let callee = if is_kw_merge { + frame.peek_back(depth + 2).cloned() + } else { + None + }; + let kw_error_prefix = || -> String { + callee + .as_ref() .and_then(callable_function_str) .map(|s| format!("{s} ")) .unwrap_or_default() - } else { - String::new() }; let dict = frame.peek_back(depth - 1).cloned().ok_or_else(|| { RuntimeError::Internal("DICT_UPDATE: stack underflow".to_owned()) @@ -21262,7 +24271,8 @@ impl Interpreter { for (k, v) in src.borrow().iter() { if is_kw_merge && t.contains_key(k) { return Err(type_error(format!( - "{kw_error_prefix}got multiple values for keyword argument '{}'", + "{}got multiple values for keyword argument '{}'", + kw_error_prefix(), k.0.to_str() ))); } @@ -21279,7 +24289,8 @@ impl Interpreter { // "test.test_extcall.h() argument after ** // must be a mapping, not list". type_error(format!( - "{kw_error_prefix}argument after ** must be a mapping, not {}", + "{}argument after ** must be a mapping, not {}", + kw_error_prefix(), other.type_name_owned() )) } else { @@ -21319,7 +24330,8 @@ impl Interpreter { let key = crate::object::DictKey(k); if is_kw_merge && t.contains_key(&key) { return Err(type_error(format!( - "{kw_error_prefix}got multiple values for keyword argument '{}'", + "{}got multiple values for keyword argument '{}'", + kw_error_prefix(), key.0.to_str() ))); } @@ -21425,7 +24437,7 @@ impl Interpreter { defaults, kw_defaults, closure, - attrs: RefCell::new(Rc::new(RefCell::new(DictData::default()))), + attrs: RefCell::new(None), slots, closure_cells: std::sync::OnceLock::new(), defaults_override: crate::object::OverrideFlag::new(false), @@ -21894,6 +24906,7 @@ impl Interpreter { // timing rather than waiting for the next cyclic collection // (RFC 0040 deterministic-finalization arc: `test_io` / // `test_subprocess` destructor-timing cases). + self.recheck_frame_observed = true; if let Some((_, pe)) = popped { // Deconstruct the handler's `PyException` *first*: its // `context`/`cause` boxes and traceback entries hold @@ -21972,9 +24985,10 @@ impl Interpreter { // so an `except E as e: saved = e` pays nothing. if weakly_observed || gc_trace::is_tracked(id) - || Self::exc_has_finalizable(&pe_instance, 6) - || (Self::is_refcount_dead(&pe_instance, 1) - && Self::anchors_tracked_child(&pe_instance, 5)) + || (!Self::exc_plainly_inert(&pe_instance) + && (Self::exc_has_finalizable(&pe_instance, 6) + || (Self::is_refcount_dead(&pe_instance, 1) + && Self::anchors_tracked_child(&pe_instance, 5)))) { self.reap_dead_subgraph(pe_instance); } @@ -22855,8 +25869,7 @@ impl Interpreter { /// can walk `exc.__traceback__`. fn append_traceback(&self, exc: &mut PyException, frame: &mut Frame, lasti: u32, lineno: u32) { exc.push_traceback(TracebackEntry { - filename: frame.code.filename.clone(), - funcname: frame.code.name.clone(), + code: frame.code.clone(), lineno, }); // The Python-visible frame for this entry must be *this* frame's @@ -23216,9 +26229,9 @@ impl Interpreter { // matching happens — `except (ValueError, 42):` is a TypeError // even when the raised exception would match ValueError // (CPython `check_except_type` over the whole tuple). + let base_exception = &builtin_types().base_exception; let check = |t: &Object| -> Result<(), RuntimeError> { - let ok = matches!(t, Object::Type(c) - if c.mro.borrow().iter().any(|m| m.name == "BaseException")); + let ok = matches!(t, Object::Type(c) if c.is_subclass_of(base_exception)); if ok { Ok(()) } else { @@ -24296,8 +27309,8 @@ impl Interpreter { if let Some(v) = f.slot(name) { return Ok(v); } - } else if let Some(v) = f.attrs().borrow().get(&crate::object::StrKey(name)) { - return Ok(v.clone()); + } else if let Some(v) = f.attr_get(name) { + return Ok(v); } match name { // Stash the computed value on the slot so repeated @@ -25756,10 +28769,8 @@ impl Interpreter { } // (2) Instance dict. - if let Some(dict) = inst.dict.get() { - if let Some(v) = dict.borrow().get(&crate::object::StrKey(name)) { - return Ok(v.clone()); - } + if let Some(v) = inst.attr_get_str(name) { + return Ok(v); } // (3) Non-data descriptor / function on class. @@ -27521,6 +30532,9 @@ impl Interpreter { // `len(x)` is `type(x).__len__(x)` — for an *instance* that's the // class's method; for a *class* it's the metaclass's // (`len(SomeEnum)` → `EnumType.__len__`). + if let Some(r) = instance_native_dunder(v, "__len__", None) { + return coerce_len_result(r?).map(Object::Int); + } let method = instance_method(v, "__len__").or_else(|| metaclass_method(v, "__len__")); if let Some(method) = method { let r = self.call(&method, &[], &[], globals)?; @@ -28430,8 +31444,18 @@ impl Interpreter { "NotImplemented should not be used in a boolean context", )); } - if let Some(method) = instance_method(v, "__bool__") { - let r = self.call(&method, &[], &[], globals)?; + let native = instance_native_dunder(v, "__bool__", None); + let method = if native.is_none() { + instance_method(v, "__bool__") + } else { + None + }; + if native.is_some() || method.is_some() { + let r = match (native, method) { + (Some(r), _) => r?, + (None, Some(method)) => self.call(&method, &[], &[], globals)?, + (None, None) => unreachable!("checked above"), + }; // CPython's `slot_nb_bool` is strict: anything but an exact // bool — even an int — raises (`__bool__` returning `1` is a // TypeError, test_bool `test_convert_to_bool`). @@ -28443,6 +31467,9 @@ impl Interpreter { ))), }; } + if let Some(r) = instance_native_dunder(v, "__len__", None) { + return coerce_len_result(r?).map(|n| n != 0); + } if let Some(method) = instance_method(v, "__len__") { let r = self.call(&method, &[], &[], globals)?; // `PyObject_Size` semantics; errors must match `len()`'s @@ -29208,6 +32235,22 @@ impl Interpreter { } let it = self.make_iter(v, globals)?; let mut out = Vec::new(); + if let Object::Generator(g) = &it { + // A lean resume moves each yield straight into `out` + // (see `Interpreter::sum_fold`). + loop { + let sink = FoldSink::Collect(std::ptr::from_mut(&mut out)); + match self.generator_send_fold(g, Object::None, Some(sink)) { + Ok(x) => out.push(x), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + return Ok(out); + } + Err(e) => return Err(e), + } + } + } while let Some(x) = self.iter_next(&it, globals)? { out.push(x); } @@ -29408,10 +32451,8 @@ impl Interpreter { // A lean resume folds scalar yields straight into `total` // (see `Interpreter::sum_fold`); it returns at the first // yield it cannot fold, or when the generator ends. - self.sum_fold_acc = Some(std::ptr::from_mut(&mut total) as usize); - let sent = self.generator_send(g, Object::None); - self.sum_fold_acc = None; - let x = match sent { + let sink = FoldSink::Sum(std::ptr::from_mut(&mut total)); + let x = match self.generator_send_fold(g, Object::None, Some(sink)) { Ok(v) => v, Err(RuntimeError::PyException(exc)) if exc.type_name() == "StopIteration" => { self.fire_caught_stop_iteration(&exc)?; @@ -29799,45 +32840,9 @@ impl Interpreter { Err(e) => return Err(e), } } - // Native ABCs (e.g. the `io` `IOBase`/`RawIOBase`/`BufferedIOBase`/ - // `TextIOBase` family) record virtual subclasses in a `_abc_registry` - // set via `register()`. CPython's `ABCMeta.__subclasscheck__` honours - // that registry; the native isinstance path must too, otherwise - // `isinstance(_pyio.BufferedReader(...), io.IOBase)` — the layered `io` - // module's classes are the registered `_pyio` ones — wrongly returns - // False. Match if `real` is, or descends from, any registered class. - if Self::class_is_abc_registered(cls, &real) { - return Ok(Object::Bool(true)); - } Ok(Object::Bool(false)) } - /// Whether `candidate` is registered against the native ABC `cls` via its - /// `_abc_registry` set (directly, or as a descendant of a registered - /// virtual subclass). Mirrors one level of CPython's - /// `ABCMeta.__subclasscheck__` registry walk — sufficient for the `io` ABC - /// family, where `_pyio` registers each of `IOBase`/`RawIOBase`/ - /// `BufferedIOBase`/`TextIOBase`. - fn class_is_abc_registered(cls: &Rc, candidate: &Rc) -> bool { - let reg = cls - .dict - .borrow() - .get(&crate::object::DictKey(Object::from_static( - "_abc_registry", - ))) - .cloned(); - if let Some(Object::Set(s)) = reg { - for k in s.borrow().iter() { - if let Object::Type(r) = &k.0 { - if Rc::ptr_eq(r, candidate) || candidate.is_subclass_of(r) { - return true; - } - } - } - } - false - } - /// `issubclass(cls, classinfo)` — same protocol as /// [`do_isinstance_call`] but for class membership. fn do_issubclass_call( @@ -30236,7 +33241,7 @@ impl Interpreter { if !matches!(obj, Object::Instance(_)) { return None; } - if !crate::object::instance_has_custom_dunder(obj, "__hash__") { + if !crate::object::instance_has_custom_dunder(obj, crate::types::Dunder::Hash) { return None; } let globals = self.builtins.clone(); @@ -30320,6 +33325,9 @@ impl Interpreter { ))), }; } + if args.len() >= 3 && attr_certainly_missing(&args[0], &name) { + return Ok(args[2].clone()); + } match self.load_attr(&args[0], &name) { Ok(v) => Ok(v), Err(e) if args.len() >= 3 && self.is_attribute_error(&e) => Ok(args[2].clone()), @@ -30350,6 +33358,9 @@ impl Interpreter { crate::builtins::wstr_attr_get(&args[0], &args[1]).is_some(), )); } + if attr_certainly_missing(&args[0], &name) { + return Ok(Object::Bool(false)); + } match self.load_attr(&args[0], &name) { Ok(_) => Ok(Object::Bool(true)), Err(e) if self.is_attribute_error(&e) => Ok(Object::Bool(false)), @@ -30362,12 +33373,10 @@ impl Interpreter { /// attribute-lookup failures. fn is_attribute_error(&self, err: &RuntimeError) -> bool { match err { - RuntimeError::PyException(pe) => self - .exception_matches( - &pe.instance, - &Object::Type(builtin_types().attribute_error.clone()), - ) - .unwrap_or(false), + RuntimeError::PyException(pe) => crate::builtin_types::instance_is_subclass( + &pe.instance, + &builtin_types().attribute_error, + ), RuntimeError::Internal(_) => false, } } @@ -30700,10 +33709,10 @@ impl Interpreter { Ok(Object::None) } - /// Stable sort over `items`. With `key`, every element is mapped - /// through it once and the results are sorted alongside the - /// originals (decorate-sort-undecorate). Errors from the key - /// function propagate. + /// `list.sort`: a stable sort of `items`, by `key(item)` when a key + /// function is given (called once per item), descending when `reverse` + /// (still stable, as CPython does it: reverse, sort, reverse). On a + /// comparison error `items` holds a permutation of its input. fn sort_with_key( &mut self, items: &mut Vec, @@ -30711,112 +33720,103 @@ impl Interpreter { reverse: bool, globals: &Rc>, ) -> Result<(), RuntimeError> { - // `key=None` is CPython's spelling of "no key function" (sort by the - // elements themselves) — `sorted(xs, key=None)` and `xs.sort(key=None)` - // are identity sorts, not "call `None` on every element". Callers thread - // the raw kwarg through, so collapse the `None` sentinel here. - let key_fn = match key_fn { - Some(Object::None) => None, - other => other, - }; - // CPython's `reverse=True` is *tie-stable*: equal elements keep - // their original relative order (list.sort reverses the slice - // before and after sorting). A post-sort `.reverse()` alone would - // flip ties — observable in `heapq.nlargest`/`Counter.most_common`. - if let Some(f) = key_fn { - let mut decorated: Vec<(Object, Object)> = Vec::with_capacity(items.len()); - for item in items.iter() { - let k = self.call(f, std::slice::from_ref(item), &[], globals)?; - decorated.push((k, item.clone())); - } - if reverse { - decorated.reverse(); - } - if decorated.iter().any(|(k, _)| sort_key_needs_dunder_lt(k)) { - decorated = - merge_sort_by_pylt(self, decorated, &|p: &(Object, Object)| &p.0, globals)?; - } else { - // Uncomparable keys must surface the `TypeError` (CPython - // propagates the failed `<`), not silently compare equal. - let mut err: Option = None; - decorated.sort_by(|a, b| match a.0.cmp(&b.0) { - Ok(o) => o, - Err(e) => { - if err.is_none() { - err = Some(e); - } - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); - } - } + // `key=None` is CPython's spelling of "no key function". + let Some(f) = key_fn.filter(|f| !matches!(f, Object::None)) else { if reverse { - decorated.reverse(); - } - // Undecorate, then promptly reap the dying key objects: CPython - // decrefs the keys inside `list.sort` while the list is still - // detached, so a key `__del__` that mutates the list is caught - // by the caller's mutation check - // (test_sort.test_key_with_mutating_del). - let mut dead_keys: Vec = Vec::with_capacity(decorated.len()); - *items = decorated - .into_iter() - .map(|(k, v)| { - dead_keys.push(k); - v - }) - .collect(); - for k in dead_keys { - if matches!( - k, - Object::Instance(_) - | Object::Generator(_) - | Object::Coroutine(_) - | Object::AsyncGenerator(_) - ) && Self::is_refcount_dead(&k, 1) - { - self.reap_dead_subgraph(k); - } + items.reverse(); } - } else { + let result = self.sort_keyed(items, |o| o, globals); if reverse { items.reverse(); } - if items.iter().any(sort_key_needs_dunder_lt) { - // Keep a (cheap, Rc-clone) backup: `merge_sort_by_pylt` - // consumes the vec, and a raising `__lt__` must not empty - // the caller's list (CPython reattaches the partially - // sorted array on error). - let backup = items.clone(); - match merge_sort_by_pylt(self, std::mem::take(items), &|o: &Object| o, globals) { - Ok(sorted) => *items = sorted, - Err(e) => { - *items = backup; - return Err(e); + return result; + }; + let mut decorated: Vec<(Object, Object)> = Vec::with_capacity(items.len()); + for item in items.iter() { + let k = self.call(f, std::slice::from_ref(item), &[], globals)?; + decorated.push((k, item.clone())); + } + if reverse { + decorated.reverse(); + } + let result = self.sort_keyed(&mut decorated, |p| &p.0, globals); + if reverse { + decorated.reverse(); + } + // Undecorate, then promptly reap the dying key objects: CPython + // decrefs the keys inside `list.sort` while the list is still + // detached, so a key `__del__` that mutates the list is caught by + // the caller's mutation check (test_sort.test_key_with_mutating_del). + let mut dead_keys: Vec = Vec::with_capacity(decorated.len()); + *items = decorated + .into_iter() + .map(|(k, v)| { + dead_keys.push(k); + v + }) + .collect(); + for k in dead_keys { + if matches!( + k, + Object::Instance(_) + | Object::Generator(_) + | Object::Coroutine(_) + | Object::AsyncGenerator(_) + ) && Self::is_refcount_dead(&k, 1) + { + self.reap_dead_subgraph(k); + } + } + result + } + + /// Sort `v` stably by the keys `key` projects, with the comparison + /// CPython's `list.sort` pre-sort check picks for the keys' types: a + /// native one for `str`, `int` or `float` keys (and for tuples whose + /// first items are one of those), else Python's `<`. + fn sort_keyed( + &mut self, + v: &mut [T], + key: impl Fn(&T) -> &Object + Copy, + globals: &Rc>, + ) -> Result<(), RuntimeError> { + match SortKind::of(v.iter().map(key)) { + SortKind::Str => crate::timsort::sort(v, |a, b| Ok(SortKind::str_lt(key(a), key(b)))), + SortKind::Int => crate::timsort::sort(v, |a, b| Ok(SortKind::int_lt(key(a), key(b)))), + SortKind::Float => { + crate::timsort::sort(v, |a, b| Ok(SortKind::float_lt(key(a), key(b)))) + } + SortKind::Numeric => { + crate::timsort::sort(v, |a, b| compare_op(key(a), key(b), CompareKind::Lt)) + } + SortKind::Tuples(first) => crate::timsort::sort(v, |a, b| { + let (Object::Tuple(x), Object::Tuple(y)) = (key(a), key(b)) else { + return self.dispatch_compare_op(key(a), key(b), CompareKind::Lt, globals); + }; + // The first differing position decides; only it is + // compared with `<` (CPython `unsafe_tuple_compare`). + let mut i = 0; + while i < x.len() && i < y.len() { + if !x[i].is_same(&y[i]) && !self.vm_eq(&x[i], &y[i], globals)? { + break; } + i += 1; } - } else { - let mut err: Option = None; - items.sort_by(|a, b| match a.cmp(b) { - Ok(o) => o, - Err(e) => { - if err.is_none() { - err = Some(e); - } - std::cmp::Ordering::Equal - } - }); - if let Some(e) = err { - return Err(e); + if i >= x.len() || i >= y.len() { + return Ok(x.len() < y.len()); } - } - if reverse { - items.reverse(); - } + match (i, first) { + (0, SortKindScalar::Str) => Ok(SortKind::str_lt(&x[0], &y[0])), + (0, SortKindScalar::Int) => Ok(SortKind::int_lt(&x[0], &y[0])), + (0, SortKindScalar::Float) => Ok(SortKind::float_lt(&x[0], &y[0])), + (0, SortKindScalar::Numeric) => compare_op(&x[0], &y[0], CompareKind::Lt), + _ => self.dispatch_compare_op(&x[i], &y[i], CompareKind::Lt, globals), + } + }), + SortKind::General => crate::timsort::sort(v, |a, b| { + self.dispatch_compare_op(key(a), key(b), CompareKind::Lt, globals) + }), } - Ok(()) } /// Run `__str__` on instances, falling back to `__repr__` then @@ -32015,10 +35015,8 @@ impl Interpreter { // CPython GET_AWAITABLE: a coroutine that is suspended // inside its own `await` (it has a yield-from // sub-iterator) cannot gain a second awaiter. - let busy = matches!(&*g.state.borrow(), GeneratorState::Suspended(boxed) - if boxed - .downcast_ref::() - .is_some_and(|f| detect_yield_from_subiter(f).is_some())); + let busy = matches!(&*g.state.borrow(), GeneratorState::Suspended(frame) + if detect_yield_from_subiter(frame).is_some()); if busy { return Err(crate::error::runtime_error( "coroutine is being awaited already", @@ -32190,6 +35188,35 @@ impl Interpreter { Err(e) => Err(e), }, Object::Instance(inst) => { + // A registered native leaf `__next__` (the class cache the + // core loop's `FOR_ITER` uses): called directly. + if !crate::gil::free_threading_enabled() && !crate::trace::any_observers_active() { + if let Some(b) = self.core_leaf_next_ptr(inst) { + // SAFETY: the cache (and the class) keep the builtin + // alive through the call, whose body runs no Python. + let b = unsafe { &*b }; + return match (b.call)(std::slice::from_ref(iter)) { + Ok(v) => Ok(Some(v)), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + Ok(None) + } + Err(e) => Err(e), + }; + } + } + if let Some(r) = instance_native_dunder(iter, "__next__", None) { + return match r { + Ok(v) => Ok(Some(v)), + Err(RuntimeError::PyException(exc)) + if exc.type_name() == "StopIteration" => + { + Ok(None) + } + Err(e) => Err(e), + }; + } if let Some(method) = instance_method(iter, "__next__") { match self.call(&method, &[], &[], globals) { Ok(v) => Ok(Some(v)), @@ -33315,8 +36342,7 @@ impl Interpreter { // as "already running" in CPython's `ag_running_async` sense. let mid_await = matches!( &*g.state.borrow(), - GeneratorState::Suspended(boxed) - if boxed.downcast_ref::().is_some_and(|f| !f.agen_yielded_value) + GeneratorState::Suspended(frame) if !frame.agen_yielded_value ) || matches!(&*g.state.borrow(), GeneratorState::Running); if !mid_await { return None; @@ -33442,9 +36468,7 @@ impl Interpreter { /// (e.g. the generator already finished). fn agen_yielded_a_value(g: &Rc) -> bool { match &*g.state.borrow() { - GeneratorState::Suspended(boxed) => boxed - .downcast_ref::() - .is_none_or(|f| f.agen_yielded_value), + GeneratorState::Suspended(frame) => frame.agen_yielded_value, _ => true, } } @@ -33620,9 +36644,7 @@ impl Interpreter { ) -> Result { let prev_state = std::mem::replace(&mut *gen.state.borrow_mut(), GeneratorState::Running); let mut frame = match prev_state { - GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => *boxed - .downcast::() - .map_err(|_| RuntimeError::Internal("generator frame downcast".to_owned()))?, + GeneratorState::Created(boxed) | GeneratorState::Suspended(boxed) => *boxed, GeneratorState::Finished => { *gen.state.borrow_mut() = GeneratorState::Finished; // bpo-25887: throwing into an exhausted coroutine is a @@ -33649,6 +36671,8 @@ impl Interpreter { // exception machinery below both read them. #[cfg(feature = "jit")] crate::tier2::materialize_parked(&mut frame); + // A frame a fast step left partway takes the exception at its pc. + frame.sent_consumed = false; // PEP 3134: an exception thrown into a generator suspended inside // an `except`/`with` block chains to the exception that block was // handling. Done before delegation/handling so the `__context__` @@ -33931,9 +36955,7 @@ impl Interpreter { if !created && crate::trace::any_observers_active() { return false; } - let Some(frame) = boxed.downcast_mut::() else { - return false; - }; + let frame: &mut Frame = boxed; // A Python-visible frame keeps showing the locals (the general // path leaves them to it). if frame.py_frame.is_some() @@ -33964,10 +36986,8 @@ impl Interpreter { else { unreachable!("checked above"); }; - if let Some(frame) = boxed.downcast_mut::() { - self.reap_dead_frame(frame); - self.recycle_frame_allocs(frame); - } + self.reap_dead_frame(&mut boxed); + self.recycle_frame_allocs(&mut boxed); Self::release_finished_gen(g); drop(boxed); true @@ -34347,6 +37367,7 @@ impl Interpreter { &mut self, gen: &Rc, sent: &Object, + fold: Option, ) -> Option> { let snap_gen = self.lean_snapshot()?; if self.dbg_sample @@ -34373,7 +37394,7 @@ impl Interpreter { } _ => return None, }; - let frame = boxed.downcast_ref::()?; + let frame: &Frame = boxed; if frame.py_frame.is_some() || !frame.saved_exc_info.is_empty() || frame.pc == 0 @@ -34406,33 +37427,68 @@ impl Interpreter { unreachable!("checked above"); }; let outcome = { - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; // A native activation parked at the yield (the generator's // first run entered compiled code) is rebuilt into the // interpreted suspension this path resumes. #[cfg(feature = "jit")] crate::tier2::materialize_parked(frame); frame.gen_first_resume = first_resume; - frame.push(sent.clone()); - // `sum()`'s accumulator, folded at this frame's yields. - let prev_fold = std::mem::replace( - &mut self.sum_fold, - self.sum_fold_acc - .take() - .map(|acc| (std::ptr::from_mut(frame) as usize, acc)), - ); - let exc_depth_on_entry = self.exc_info_len(); - let fin_live = gc_trace::has_any_finalizable(); - let fin = fin_live || gc_trace::active_suspects_present(); - let mut act = LeanAct { - frame: std::ptr::from_mut(frame), - shell: None, - }; - let mut prev_pc: Option = None; - let run = if guard.depth() % 4 == 0 { - stacker::maybe_grow(512 * 1024, 8 * 1024 * 1024, || { + if !std::mem::take(&mut frame.sent_consumed) { + frame.push(sent.clone()); + } + // A simple body runs to its next yield in place (see + // `gen_fast`), folding the draining consumer's yields as it goes. + let mut fast = self.gen_fast_run(frame, snap_gen, fold); + // Fast steps stop at the eval breaker; a draining consumer's + // single send would then finish the whole generator on the + // slow path. Service the breaker here (as the quiet loop's + // checkpoint does) and keep stepping. + while fast.is_none() + && self.gil_countdown <= 1 + && crate::hot_gates::loop_gen() == snap_gen + && !frame.sent_consumed + { + self.gil_countdown = crate::gil::GIL_CHECK_INTERVAL; + if !gc_trace::active_suspects_present() && gc_trace::has_suspects() { + for obj in gc_trace::take_dead_suspects() { + self.reap_dead_subgraph(obj); + } + } + crate::gil::yield_checkpoint(); + if crate::hot_gates::loop_gen() != snap_gen { + break; + } + fast = self.gen_fast_run(frame, snap_gen, fold); + } + if let Some(v) = fast { + Ok(FrameOutcome::Yielded(v)) + } else { + // The draining consumer's sink, folded at this frame's yields. + let prev_fold = std::mem::replace( + &mut self.sum_fold, + fold.map(|sink| (std::ptr::from_mut(frame) as usize, sink)), + ); + let exc_depth_on_entry = self.exc_info_len(); + let fin_live = gc_trace::has_any_finalizable(); + let fin = fin_live || gc_trace::active_suspects_present(); + let mut act = LeanAct { + frame: std::ptr::from_mut(frame), + shell: None, + }; + let mut prev_pc: Option = None; + let run = if guard.depth() % 4 == 0 { + stacker::maybe_grow(512 * 1024, 8 * 1024 * 1024, || { + self.quiet_run( + frame, + QuietShell::Lazy(&mut act), + snap_gen, + fin, + fin_live, + &mut prev_pc, + ) + }) + } else { self.quiet_run( frame, QuietShell::Lazy(&mut act), @@ -34441,53 +37497,44 @@ impl Interpreter { fin_live, &mut prev_pc, ) - }) - } else { - self.quiet_run( - frame, - QuietShell::Lazy(&mut act), - snap_gen, - fin, - fin_live, - &mut prev_pc, - ) - }; - self.sum_fold = prev_fold; - match (run, act.shell) { - ( - QuietExit::Outcome { - stepped: Ok(StepOutcome::Yield(v)), - .. - }, - None, - ) => Ok(FrameOutcome::Yielded(v)), - ( - QuietExit::Outcome { - stepped: Ok(StepOutcome::Return(v)), - .. - }, - None, - ) => Ok(FrameOutcome::Returned(v)), - // Everything else finishes in the general loop (which - // saves a suspended generator's handler state, as the - // general prologue's twin epilogue does). - (exit, shell) => { - let shell = match shell { - Some(shell) => shell, - None => { - self.flush_pending_callers(); - let shell = self.push_frame_shell(frame); - if frame.shell_cache.is_none() { - frame.shell_cache = Some(shell.clone()); + }; + self.sum_fold = prev_fold; + match (run, act.shell) { + ( + QuietExit::Outcome { + stepped: Ok(StepOutcome::Yield(v)), + .. + }, + None, + ) => Ok(FrameOutcome::Yielded(v)), + ( + QuietExit::Outcome { + stepped: Ok(StepOutcome::Return(v)), + .. + }, + None, + ) => Ok(FrameOutcome::Returned(v)), + // Everything else finishes in the general loop (which + // saves a suspended generator's handler state, as the + // general prologue's twin epilogue does). + (exit, shell) => { + let shell = match shell { + Some(shell) => shell, + None => { + self.flush_pending_callers(); + let shell = self.push_frame_shell(frame); + if frame.shell_cache.is_none() { + frame.shell_cache = Some(shell.clone()); + } + shell } - shell - } - }; - let pending = match exit { - QuietExit::Yield => None, - QuietExit::Outcome { stepped, cur_pc } => Some((stepped, cur_pc)), - }; - self.run_activation(frame, shell, None, exc_depth_on_entry, false, pending) + }; + let pending = match exit { + QuietExit::Yield => None, + QuietExit::Outcome { stepped, cur_pc } => Some((stepped, cur_pc)), + }; + self.run_activation(frame, shell, None, exc_depth_on_entry, false, pending) + } } } }; @@ -34499,9 +37546,7 @@ impl Interpreter { } Ok(FrameOutcome::Returned(v)) => { *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -34516,9 +37561,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(gen, err); - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -34545,15 +37588,24 @@ impl Interpreter { gc_trace::untrack_id(id); } - fn park_suspended_boxed(gen: &Rc, boxed: Box) { - if let Some(frame) = boxed.downcast_ref::() { - if let Some(py) = &frame.py_frame { - if py.gen_owner.borrow().is_none() { - *py.gen_owner.borrow_mut() = Some(Rc::downgrade(gen)); - } + fn park_suspended_boxed(gen: &Rc, boxed: Box) { + if let Some(py) = &boxed.py_frame { + if py.gen_owner.borrow().is_none() { + *py.gen_owner.borrow_mut() = Some(Rc::downgrade(gen)); } } - *gen.state.borrow_mut() = GeneratorState::Suspended(boxed); + // SAFETY: the store runs no code (the displaced state is + // `Running`, which owns nothing) and no guard is live on the cell + // (`peek_mut` checks). + match unsafe { gen.state.peek_mut() } { + // `Running` owns nothing: no drop glue for the displaced state. + // SAFETY: as above. + Some(state) if matches!(state, GeneratorState::Running) => unsafe { + std::ptr::write(state, GeneratorState::Suspended(boxed)); + }, + Some(state) => *state = GeneratorState::Suspended(boxed), + None => *gen.state.borrow_mut() = GeneratorState::Suspended(boxed), + } } fn generator_send( @@ -34561,7 +37613,21 @@ impl Interpreter { gen: &Rc, sent: Object, ) -> Result { - if let Some(r) = self.generator_send_lean(gen, &sent) { + self.generator_send_fold(gen, sent, None) + } + + /// [`Self::generator_send`] for a consumer that drains the generator: + /// a lean resume folds the yields it can into `fold` (see + /// [`Self::sum_fold`]) and returns the first it can't. Any other + /// resume folds nothing, so a generator it resumes in turn never + /// sees the sink. + fn generator_send_fold( + &mut self, + gen: &Rc, + sent: Object, + fold: Option, + ) -> Result { + if let Some(r) = self.generator_send_lean(gen, &sent, fold) { return r; } // Take the boxed frame; it is run *in place* (RFC 0069 WS4) so @@ -34589,11 +37655,6 @@ impl Interpreter { ))); } }; - if boxed.downcast_ref::().is_none() { - return Err(RuntimeError::Internal( - "generator frame downcast".to_owned(), - )); - } // On the first call, `sent` must be None (or omitted). if first_resume && !matches!(sent, Object::None) { Self::park_suspended_boxed(gen, boxed); @@ -34603,9 +37664,7 @@ impl Interpreter { ))); } let outcome = { - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; // PEP 667: writes made through the suspended frame's `f_locals` // take effect when the generator resumes. Self::apply_py_frame_locals_writes(frame); @@ -34640,7 +37699,14 @@ impl Interpreter { // CPython 3.13 prologue: RETURN_GENERATOR / POP_TOP / RESUME — // *every* resume pushes the sent value; the first activation's // None lands in the prologue's POP_TOP. - let sent_for_frame = Some(if first_resume { Object::None } else { sent }); + // (A fast step that stopped partway already consumed it.) + let sent_for_frame = (!std::mem::take(&mut frame.sent_consumed)).then(|| { + if first_resume { + Object::None + } else { + sent + } + }); self.run_until_yield_or_return(frame, sent_for_frame) }; match outcome { @@ -34657,9 +37723,7 @@ impl Interpreter { // string we get from `from_builtin("StopIteration", // "")`. *gen.state.borrow_mut() = GeneratorState::Finished; - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); // RFC 0059 WS4: the exhausted frame's storage is dead — // donate it back to the frame pools. @@ -34676,9 +37740,7 @@ impl Interpreter { Err(err) => { *gen.state.borrow_mut() = GeneratorState::Finished; let escaped = self.pep479_escape(gen, err); - let frame = boxed - .downcast_mut::() - .expect("checked by the downcast_ref above"); + let frame: &mut Frame = &mut boxed; self.reap_dead_frame(frame); self.recycle_frame_allocs(frame); Self::release_finished_gen(gen); @@ -35672,6 +38734,9 @@ impl Interpreter { op: CompareKind, globals: &Rc>, ) -> Result { + if let Some(r) = native_scalar_compare(a, b, op) { + return Ok(r); + } // CPython's `PyObject_RichCompareBool` truth-tests the comparison // *result* with `PyObject_IsTrue`, so `arr == x` yielding a // multi-element numpy bool array raises "truth value ... ambiguous" @@ -35779,6 +38844,9 @@ impl Interpreter { op: CompareKind, globals: &Rc>, ) -> Result { + if let Some(r) = native_scalar_compare(a, b, op) { + return Ok(Object::Bool(r)); + } // An object-backed `mappingproxy` compares as the wrapped mapping // (CPython `mappingproxy_richcompare` delegates unconditionally). if matches!(a, Object::MappingProxyObj(_)) || matches!(b, Object::MappingProxyObj(_)) { @@ -36932,11 +40000,8 @@ impl Interpreter { if cls.attr_version.get() != ver { return None; } - if let Some(dict) = inst.dict.get() { - let d = dict.borrow(); - if !d.is_empty() && d.contains_key(&code_name_key(&frame.code, name_idx)?) { - return None; - } + if inst_may_shadow(inst, &frame.code, name_idx) { + return None; } let slot = code_method_slot(&frame.code, cache_pc); if let Some(f) = slot.and_then(|s| s.get(ver)) { @@ -37002,17 +40067,12 @@ impl Interpreter { cls.attr_version.get() == ver }; if guard_ok { - let hit = inst.dict.get().and_then(|dict| { - let dict = dict.borrow(); - match dict.get_index(key_idx as usize) { - Some((k, v)) - if self.cached_slot_name_matches(&frame.code, name_idx, k) => - { - Some(Self::clone_operand(v)) - } - _ => None, - } - }); + let hit = inst + .attr_index_map(key_idx as usize, |k, v| { + self.cached_slot_name_matches(&frame.code, name_idx, k) + .then(|| Self::clone_operand(v)) + }) + .flatten(); if let Some(v) = hit { specialize::record_hit(op_idx); return Ok(v); @@ -37094,12 +40154,7 @@ impl Interpreter { // differ, so probe when non-empty (with the // name's memoised hash — no siphash, no // allocation). - let shadowed = inst.dict.get().is_some_and(|dict| { - let d = dict.borrow(); - !d.is_empty() - && code_name_key(&frame.code, name_idx) - .is_none_or(|k| d.contains_key(&k)) - }); + let shadowed = inst_may_shadow(inst, &frame.code, name_idx); if !shadowed { let slot = code_method_slot(&frame.code, cache_pc); match slot.and_then(|s| s.get(ver)) { @@ -37148,15 +40203,7 @@ impl Interpreter { // it (pandas `MultiIndex.__new__` writes `_names` // over a class-level default), so probe when // non-empty — one hash of the interned name. - let shadowed = inst.dict.get().is_some_and(|dict| { - match code_name_key(&frame.code, name_idx) { - Some(key) => { - let d = dict.borrow(); - !d.is_empty() && d.contains_key(&key) - } - None => true, - } - }); + let shadowed = inst_may_shadow(inst, &frame.code, name_idx); if shadowed { None } else { @@ -37301,12 +40348,11 @@ impl Interpreter { // Validate the cached index still holds *this* name: // a `del` on an earlier attribute shift-renumbers // every later slot (same guard as LOAD_ATTR). - let name_ok = { - let dict = inst.dict_cell().borrow(); - dict.get_index(key_idx as usize).is_some_and(|(k, _)| { + let name_ok = inst + .attr_index_map(key_idx as usize, |k, _| { self.cached_slot_name_matches(&frame.code, name_idx, k) }) - }; + .unwrap_or(false); if name_ok { let val = frame.pop()?; // Mirror the slow path: a bound method stored @@ -37321,11 +40367,13 @@ impl Interpreter { // earlier read-only check has been dropped, and // scope it tightly so it is released before the // prompt-reap cascade (which can run `__del__`). - let old = inst - .dict_cell() - .borrow_mut() - .get_index_mut(key_idx as usize) - .map(|(_, slot)| std::mem::replace(slot, val)); + let old = match inst.attr_replace_index(key_idx as usize, val) { + Ok(old) => Some(old), + Err(val) => { + frame.push(val); + None + } + }; if let Some(old) = old { specialize::record_hit(op_idx); // CPython decrefs the overwritten value now; @@ -37375,7 +40423,19 @@ impl Interpreter { // the site takes the indexed shape from here on // (the core loop's `STORE_ATTR` arm serves it). let mut upgrade_idx = false; - let old = { + let mut val = Some(val); + let old = 'store: { + // The split layout (see `core_store_new_attr`). + if watch_value.is_none() && inst.dict.published().is_none() { + if let Some(Object::Str(n)) = code_name_obj(code, name_idx) { + let v = val.take().expect("value not consumed"); + match inst.split_store(n, v) { + Ok(old) => break 'store old, + Err(v) => val = Some(v), + } + } + } + let val = val.take().expect("value not consumed"); let mut dict = inst.dict_cell().borrow_mut(); let dict = &mut *dict; let dict = if val.is_gc_atomic() { @@ -38082,8 +41142,8 @@ impl Interpreter { if old.solid_base_name() != new.solid_base_name() { return false; } - let oldbase = old.layout_struct_base(); - let newbase = new.layout_struct_base(); + let oldbase = TypeObject::layout_struct_base(old); + let newbase = TypeObject::layout_struct_base(new); if Rc::ptr_eq(&oldbase, &newbase) { return true; } @@ -38652,7 +41712,7 @@ impl Interpreter { value.type_name() ))); }; - *f.attrs.borrow_mut() = d; + *f.attrs.borrow_mut() = Some(d); return Ok(()); } "__defaults__" if !matches!(value, Object::Tuple(_) | Object::None) => { @@ -39381,7 +42441,21 @@ impl Interpreter { } else { None }; - let old = { + let key = crate::stdlib::sys::intern_name(name); + let mut value = Some(value); + let old = 'store: { + // The split layout, while the instance keeps one (and no + // watcher needs to see a real dictionary change). + if watch_value.is_none() && inst.dict.published().is_none() { + if let Object::Str(n) = &key { + let v = value.take().expect("value not consumed"); + match inst.split_store(n, v) { + Ok(old) => break 'store old, + Err(v) => value = Some(v), + } + } + } + let value = value.take().expect("value not consumed"); let mut dict = inst.dict_cell().borrow_mut(); let dict = &mut *dict; let dict = if value.is_gc_atomic() { @@ -39389,7 +42463,7 @@ impl Interpreter { } else { &mut **dict }; - dict.insert(DictKey(crate::stdlib::sys::intern_name(name)), value) + dict.insert(DictKey(key), value) }; // Instance `__dict__` is a real dict; a watched one observes // attribute stores as ADDED/MODIFIED (test_watchers @@ -40061,24 +43135,13 @@ impl Interpreter { Ok(Object::new_list(sliced)) } (Object::Tuple(items), Object::Slice(s)) => { - let v: Vec = items.iter().cloned().collect(); - let sliced = slice_seq(&v, s)?; - Ok(Object::new_tuple(sliced)) + Ok(Object::new_tuple(slice_seq(&items[..], s)?)) } (Object::Str(s), Object::Slice(slc)) => str_subscript_slice(s, slc), (Object::WStr(cps), Object::Slice(slc)) => { // Slice over code points, then canonicalise: a slice that drops // every surrogate becomes a plain `Str` again. - let items: Vec = cps.iter().map(|&c| Object::Int(i64::from(c))).collect(); - let sliced = slice_seq(&items, slc)?; - let out: Vec = sliced - .iter() - .map(|o| match o { - Object::Int(n) => *n as u32, - _ => unreachable!("slice of int vec yields ints"), - }) - .collect(); - Ok(Object::str_from_codepoints(out)) + Ok(Object::str_from_codepoints(slice_seq(&cps[..], slc)?)) } (Object::Range(r), Object::Int(_) | Object::Long(_)) => { // Full-width arithmetic: `Object::len()` raises @@ -40120,28 +43183,11 @@ impl Interpreter { Ok(Object::Int(i64::from(buf[idx]))) } (Object::Bytes(buf), Object::Slice(slc)) => { - let as_objs: Vec = buf.iter().map(|b| Object::Int(i64::from(*b))).collect(); - let sliced = slice_seq(&as_objs, slc)?; - let mut out = Vec::with_capacity(sliced.len()); - for o in sliced { - match o { - Object::Int(i) => out.push(i as u8), - _ => return Err(type_error("bytes slice produced non-int")), - } - } + let out = slice_seq(&buf[..], slc)?; Ok(Object::Bytes(SharedSlice::from(out.as_slice()))) } (Object::ByteArray(buf), Object::Slice(slc)) => { - let buf = buf.borrow(); - let as_objs: Vec = buf.iter().map(|b| Object::Int(i64::from(*b))).collect(); - let sliced = slice_seq(&as_objs, slc)?; - let mut out = Vec::with_capacity(sliced.len()); - for o in sliced { - match o { - Object::Int(i) => out.push(i as u8), - _ => return Err(type_error("bytearray slice produced non-int")), - } - } + let out = slice_seq(&buf.borrow()[..], slc)?; Ok(Object::ByteArray(Rc::new(RefCell::new(out)))) } (Object::MemoryView(mv), Object::Int(i)) => { @@ -42862,7 +45908,7 @@ impl Interpreter { defaults, kw_defaults, closure, - attrs: RefCell::new(Rc::new(RefCell::new(DictData::default()))), + attrs: RefCell::new(None), slots: RefCell::new(DictData::default()), closure_cells: std::sync::OnceLock::new(), defaults_override: crate::object::OverrideFlag::new(false), @@ -45310,6 +48356,12 @@ impl Interpreter { args: &[Object], kwargs: &[(String, Object)], ) -> Result { + // A natively served class's common constructor shapes. + if cls.native_kind.get() != 0 { + if let Some(r) = crate::stdlib::datetime_native::construct(&cls, args, kwargs) { + return r; + } + } if kwargs.is_empty() && !cls.flags.is_builtin { if let Some(r) = self.instantiate_plain_lean(&cls, args) { return r; @@ -45337,6 +48389,78 @@ impl Interpreter { // Built-in conversion types route to the underlying builtin // function so `int("3")`, `range(5)`, `list(xs)` keep working. if cls.flags.is_builtin { + // Exception classes first: a raise constructs one, and none of + // the conversion or descriptor constructors below is one. + if cls.flags.is_exception { + // PEP 654: `BaseExceptionGroup(msg, excs)` goes through + // the full `BaseExceptionGroup.__new__` — argument + // validation, class lowering, `exceptions` freezing. + let instance = if cls + .is_subclass_of(&crate::builtin_types::builtin_types().base_exception_group) + { + let _interp_guard = crate::vm_singletons::publish_interpreter_ptr( + std::ptr::from_mut::(self), + ); + crate::builtin_types::exception_group_new(&cls, args)? + } else { + self.build_exception_instance(cls.clone(), args) + }; + // Keyword fields accepted by the builtin constructors: + // `AttributeError(name=, obj=)`, `NameError(name=)`, + // `ImportError(name=, path=, name_from=)` — mirroring + // CPython's `*_init` C slots. Only consumed when no user + // `__init__` overrides the construction protocol. + if !kwargs.is_empty() && lookup_exception_init(&cls).is_none() { + let bt = crate::builtin_types::builtin_types(); + let allowed: &[&str] = if cls.is_subclass_of(&bt.import_error) { + &["name", "path", "name_from"] + } else if cls.is_subclass_of(&bt.attribute_error) { + &["name", "obj"] + } else if cls.is_subclass_of(&bt.name_error) { + &["name"] + } else { + &[] + }; + if let Object::Instance(inst) = &instance { + for (k, v) in kwargs { + if !allowed.contains(&k.as_str()) { + return Err(type_error(format!( + "{}() got an unexpected keyword argument '{}'", + cls.name, k + ))); + } + // These fields are exception pseudo-slots + // (class-level slot descriptors), so the value + // must land in the slot table — an instance-dict + // entry would be shadowed by the descriptor. + inst.slot_set(k, v.clone()); + } + } + } + // If a class anywhere between `cls` and `BaseException` + // (exclusive) defines its own `__init__`, run it so + // subclasses such as `BaseExceptionGroup` get to stitch + // `exceptions` onto the instance. We stop at + // `BaseException` because its default `__init__` only + // populates `args` — which the fast path already did. + if let Some(init) = lookup_exception_init(&cls) { + let bound = + Object::BoundMethod(Rc::new(BoundMethod::new(instance.clone(), init))); + let result = self.call( + &bound, + args, + kwargs, + &Rc::new(RefCell::new(DictData::default())), + )?; + if !matches!(result, Object::None) { + return Err(type_error(format!( + "__init__() should return None, not '{}'", + result.type_name() + ))); + } + } + return Ok(instance); + } // Descriptor wrapper classes (property/staticmethod/ // classmethod) — route to dedicated constructors. match cls.name.as_str() { @@ -45696,76 +48820,6 @@ impl Interpreter { } return (builtin.call)(args); } - if cls.flags.is_exception { - // PEP 654: `BaseExceptionGroup(msg, excs)` goes through - // the full `BaseExceptionGroup.__new__` — argument - // validation, class lowering, `exceptions` freezing. - let instance = if cls - .is_subclass_of(&crate::builtin_types::builtin_types().base_exception_group) - { - let _interp_guard = crate::vm_singletons::publish_interpreter_ptr( - std::ptr::from_mut::(self), - ); - crate::builtin_types::exception_group_new(&cls, args)? - } else { - self.build_exception_instance(cls.clone(), args) - }; - // Keyword fields accepted by the builtin constructors: - // `AttributeError(name=, obj=)`, `NameError(name=)`, - // `ImportError(name=, path=, name_from=)` — mirroring - // CPython's `*_init` C slots. Only consumed when no user - // `__init__` overrides the construction protocol. - if !kwargs.is_empty() && lookup_exception_init(&cls).is_none() { - let bt = crate::builtin_types::builtin_types(); - let allowed: &[&str] = if cls.is_subclass_of(&bt.import_error) { - &["name", "path", "name_from"] - } else if cls.is_subclass_of(&bt.attribute_error) { - &["name", "obj"] - } else if cls.is_subclass_of(&bt.name_error) { - &["name"] - } else { - &[] - }; - if let Object::Instance(inst) = &instance { - for (k, v) in kwargs { - if !allowed.contains(&k.as_str()) { - return Err(type_error(format!( - "{}() got an unexpected keyword argument '{}'", - cls.name, k - ))); - } - // These fields are exception pseudo-slots - // (class-level slot descriptors), so the value - // must land in the slot table — an instance-dict - // entry would be shadowed by the descriptor. - inst.slot_set(k, v.clone()); - } - } - } - // If a class anywhere between `cls` and `BaseException` - // (exclusive) defines its own `__init__`, run it so - // subclasses such as `BaseExceptionGroup` get to stitch - // `exceptions` onto the instance. We stop at - // `BaseException` because its default `__init__` only - // populates `args` — which the fast path already did. - if let Some(init) = lookup_exception_init(&cls) { - let bound = - Object::BoundMethod(Rc::new(BoundMethod::new(instance.clone(), init))); - let result = self.call( - &bound, - args, - kwargs, - &Rc::new(RefCell::new(DictData::default())), - )?; - if !matches!(result, Object::None) { - return Err(type_error(format!( - "__init__() should return None, not '{}'", - result.type_name() - ))); - } - } - return Ok(instance); - } } // Everything below that is a pure function of the class dict / @@ -46401,9 +49455,14 @@ impl Interpreter { // Straight from the argument vector's tail into the tuple's // own allocation: collecting the remainder into a `Vec` first // was a second allocation and a second move per call. - positional[star_idx] = Object::Tuple( - crate::tuple_storage::TupleStorage::from_exact_iter(arg_iter.by_ref()), - ); + positional[star_idx] = if arg_iter.len() == 0 { + // The interned empty tuple (`() is ()`). + Object::new_tuple(Vec::new()) + } else { + Object::Tuple(crate::tuple_storage::TupleStorage::from_exact_iter( + arg_iter.by_ref(), + )) + }; filled[star_idx] = true; } else if provided > total_args { // Mirror CPython's `too_many_positional`: when the callable @@ -46940,9 +49999,7 @@ impl Interpreter { /// without eager `PyFrame` construction. fn set_frame_gen_owner(gen: &Rc) { if let GeneratorState::Created(boxed) = &mut *gen.state.borrow_mut() { - if let Some(fr) = boxed.downcast_mut::() { - fr.gen_owner = Some(Rc::downgrade(gen)); - } + boxed.gen_owner = Some(Rc::downgrade(gen)); } } @@ -47576,6 +50633,52 @@ impl Interpreter { self.run_py_exact_nofree_with(f, f.code(), args) } + /// Run plain function `f` on `locals` already bound to its parameter + /// slots (sized to its locals): as a lean activation when the + /// dispatch loop is quiet and the code allows one, as + /// [`Self::run_py_exact_nofree`] otherwise. + pub(crate) fn run_py_bound( + &mut self, + f: &Rc, + mut locals: Vec, + ) -> Result { + let code = f.code(); + if let Some(snap_gen) = self.lean_snapshot() { + // A pure leaf evaluates frameless, as the core loop's call does. + let nargs = code.arg_count as usize; + if nargs <= 8 + && locals.len() >= nargs + && code_is_pure_leaf(&code) + && pure_leaf_warm(&code) + && crate::recursion::current_depth() < crate::recursion::recursion_limit() + { + let mut ptrs = [std::ptr::null::(); 8]; + for (p, v) in ptrs.iter_mut().zip(&locals[..nargs]) { + *p = v; + } + if let Some(v) = self.pure_leaf_eval::(&code, f, &ptrs[..nargs]) { + self.recycle_scratch(locals); + return Ok(v); + } + } + if Self::lean_code_ok(&code) && locals.len() <= code.varnames.len() { + if let Some(cells) = f.lean_cells_ref(&code) { + let cells: *const Rc>>> = cells; + let n = code.varnames.len(); + let locals_rc = self.pooled_locals_from_args(&mut locals, n); + self.recycle_scratch(locals); + // SAFETY: `f` (borrowed for this whole call) owns the + // cells handle and outlives the frame. + let mut callee = unsafe { self.lean_frame(f, code, locals_rc, &*cells) }; + let result = self.run_frame_lean(&mut callee, snap_gen); + self.retire_lean_frame(callee); + return result; + } + } + } + self.run_py_exact_nofree_with(f, code, locals) + } + /// [`Self::run_py_exact_nofree`] for a caller that already holds the /// function's current code object (its inline-cache guard read it). fn run_py_exact_nofree_with( @@ -47616,6 +50719,7 @@ impl Interpreter { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -54195,16 +57299,17 @@ pub(crate) fn str_subscript_slice(s: &SharedStr, slc: &PySlice) -> Result = s.chars().map(|c| Object::from_str(c.to_string())).collect(); - let sliced = slice_seq(&obj_chars, slc)?; - let out: String = sliced.iter().map(|o| o.to_str()).collect(); - Ok(Object::from_str(out)) + // General path: negative bounds or non-unit step, over the code points. + let chars: Vec = s.chars().collect(); + Ok(Object::from_str( + slice_seq(&chars, slc)?.into_iter().collect::(), + )) } -pub(crate) fn slice_seq(seq: &[Object], s: &PySlice) -> Result, RuntimeError> { +/// The elements of `seq` that slice `s` selects, in order: CPython's +/// `PySlice_Unpack` + `PySlice_AdjustIndices` over any element type (a +/// list's objects, a `bytes` buffer, a string's code points). +pub(crate) fn slice_seq(seq: &[T], s: &PySlice) -> Result, RuntimeError> { let len = seq.len() as i64; let step = match &s.step { Object::None => 1i64, @@ -54277,6 +57382,14 @@ pub(crate) fn slice_seq(seq: &[Object], s: &PySlice) -> Result, Runt )) } }; + if step == 1 { + // Both bounds are within `0..=len` here: one contiguous copy. + return Ok(if start < stop { + seq[start as usize..stop as usize].to_vec() + } else { + Vec::new() + }); + } let mut i = start; let mut out = Vec::new(); if step > 0 { @@ -54775,9 +57888,53 @@ enum LeafKind { StrFormat, /// A registered leaf builtin (see [`leaf_builtins`]): any arguments. Opaque, + /// A registered leaf builtin over plain `int`/`float`/`bool` + /// arguments (see [`leaf_builtins::Entry::Scalar`]). + Scalar, } impl LeafKind { + /// Whether the core loop runs this kind's call itself (operands that + /// all leave by plain decrements): the registered leaf bodies, and the + /// native methods that release no reference beyond their operands (a + /// removal, which drops an element, goes through the helper, whose + /// release grading covers it). + fn runs_in_core(self) -> bool { + matches!( + self, + Self::Opaque + | Self::Scalar + | Self::Fast(_) + | Self::Isinstance + | Self::Len + | Self::ListAppend + | Self::ListPop + | Self::ListInsert + | Self::ListReverse + | Self::ListCopy + | Self::DictGet + | Self::DictKeys + | Self::DictValues + | Self::DictItems + | Self::SetPop + | Self::StrStartswith + | Self::StrEndswith + | Self::StrLower + | Self::StrUpper + | Self::StrStrip + | Self::StrLstrip + | Self::StrRstrip + | Self::StrFind + | Self::StrIsdigit + | Self::StrIsalpha + | Self::StrIsspace + | Self::StrSplit + | Self::StrJoin + | Self::StrReplace + | Self::StrFormat + ) + } + /// These self-only operations neither invoke Python nor discard any /// existing reference held by the receiver. Argument-shape admission /// still belongs to `leaf_builtin_call`; subclasses must fall back. @@ -54801,6 +57958,38 @@ impl LeafKind { | Self::StrSplit ) } + + /// Operations that read their arguments and build a fresh result, + /// with nothing else observable: a frameless leaf may run them and + /// still decline afterwards (see `Interpreter::leaf_pure_builtin`). + fn is_pure_read(self) -> bool { + matches!( + self, + Self::Len + | Self::Isinstance + | Self::ListCopy + | Self::DictGet + | Self::DictKeys + | Self::DictValues + | Self::DictItems + | Self::StrStartswith + | Self::StrEndswith + | Self::StrLower + | Self::StrUpper + | Self::StrStrip + | Self::StrLstrip + | Self::StrRstrip + | Self::StrFind + | Self::StrIsdigit + | Self::StrIsalpha + | Self::StrIsspace + | Self::StrSplit + | Self::StrJoin + | Self::StrReplace + | Self::StrFormat + | Self::Scalar + ) + } } /// A receiver's builtin variant, for the method table. @@ -54822,6 +58011,8 @@ struct LeafFns { calls: std::collections::HashMap, /// The builtin `len`'s address (for the fused `len(x)`), or 0. len_ptr: usize, + /// The builtin `next`'s address (an inline generator resume), or 0. + next_ptr: usize, /// The builtin `range` and `list` types, whose calls with leaf /// argument shapes construct directly. range_ty: Rc, @@ -54852,6 +58043,9 @@ struct LeanAct { shell: Option>, } +/// How many parked [`InlineAct`] slots `Interpreter::inline_pool` keeps. +const INLINE_POOL_CAP: usize = 64; + /// A lean activation the quiet loop runs *inline*: its frame lives in /// this box instead of on a nested native activation, and /// `Interpreter::quiet_run` switches to it at the caller's `CALL` and @@ -54889,9 +58083,10 @@ struct InlineAct { /// its own boxed frame runs instead of the slot's (which stays /// parked); `gen_frame` points into `gen_box`. gen: Option>, - gen_box: Option>, + gen_box: Option>, gen_frame: *mut Frame, - /// The resuming `FOR_ITER`'s jump distance (the exhaustion exit). + /// The resuming `FOR_ITER`'s jump distance (the exhaustion exit), or + /// [`GEN_NEXT_CALL`] for a resume by builtin `next`. exhaust_arg: u32, /// A constructor call's fresh instance (see `Interpreter::core_new`): /// the activation runs its `__init__`, and the caller receives the @@ -54924,6 +58119,7 @@ impl InlineAct { pending_lasti: None, suppress_call_event: false, gen_first_resume: false, + sent_consumed: false, shell_cache: None, #[cfg(feature = "jit")] parked_native: None, @@ -55001,11 +58197,15 @@ enum FrameEv { } /// How an inline generator resume concluded. +/// [`InlineAct::exhaust_arg`] of a generator resumed by builtin `next` +/// (no `FOR_ITER` jump is that long). +const GEN_NEXT_CALL: u32 = u32::MAX; + enum GenStep { /// The generator yielded this value. Yielded(Object), - /// The generator returned: the loop is exhausted. - Exhausted, + /// The generator returned this value: the loop is exhausted. + Exhausted(Object), } /// Where an activation's stretch of the quiet loop starts. @@ -55163,6 +58363,20 @@ impl CoreSwitch { /// compensated float phase taking floats and ints alike, then — once it /// is a complex — compensated real and imaginary sums, then generic `+` /// for everything after (each phase is left for good). +/// A draining consumer's sink for a lean generator resume's yields (see +/// [`Interpreter::sum_fold`]): `sum()`'s running total, which takes +/// scalars, or `list()`'s item buffer, which takes anything. +#[derive(Clone, Copy)] +enum FoldSink { + Sum(*mut SumState), + Collect(*mut Vec), +} + +// SAFETY: a sink is installed only while a resume runs on the thread that +// owns it, and cleared before that resume returns: an interpreter moved +// between threads never carries a live one. +unsafe impl Send for FoldSink {} + enum SumState { Int(i64), Float(CompensatedSum), @@ -55421,6 +58635,7 @@ thread_local! { static PURE_LITERAL_ARGUMENT_CALLS: std::cell::Cell<[u64; 4]> = const { std::cell::Cell::new([0; 4]) }; static PURE_SLOT_FIELD_READS: std::cell::Cell<[u64; 3]> = const { std::cell::Cell::new([0; 3]) }; static PURE_CACHED_FIELD_PREDICATES: std::cell::Cell = const { std::cell::Cell::new(0) }; + static LEAF_BUILTIN_CALLS: std::cell::Cell = const { std::cell::Cell::new(0) }; static PURE_PREDICATE_STAGES: std::cell::Cell<[u64; 9]> = const { std::cell::Cell::new([0; 9]) }; static PURE_PREDICATE_DROP_MISSES: std::cell::Cell<[u64; 4]> = const { std::cell::Cell::new([0; 4]) }; static NATIVE_SUBSCRIPT_CACHE_HITS: std::cell::Cell = const { std::cell::Cell::new(0) }; @@ -55476,25 +58691,44 @@ pub(crate) mod leaf_builtins { ) -> Option>; - static REGISTRY: parking_lot::Mutex, Option)>> = + /// What a registration vouches for. + #[derive(Clone, Copy)] + pub(crate) enum Entry { + /// The whole body is a leaf, for any arguments. + Opaque, + /// This pure fast half (see [`Fast`]). + Fast(Fast), + /// The whole body is a leaf when every argument is a plain `int`, + /// `float` or `bool` (numeric functions that reach Python code only + /// through another object's conversion hooks). + Scalar, + } + + static REGISTRY: parking_lot::Mutex, Entry)>> = parking_lot::Mutex::new(Vec::new()); static GENERATION: AtomicU64 = AtomicU64::new(1); /// Vouch for `b` (by identity): its whole body is a leaf. pub(crate) fn register(b: &Rc) { - push(b, None); + push(b, Entry::Opaque); } /// Vouch for `fast` as `b`'s leaf half: the dispatch loop calls it /// inline and takes the full path when it declines. pub(crate) fn register_fast(b: &Rc, fast: Fast) { - push(b, Some(fast)); + push(b, Entry::Fast(fast)); + } + + /// Vouch for `b`'s whole body as a leaf over scalar arguments (see + /// [`Entry::Scalar`]). + pub(crate) fn register_scalar(b: &Rc) { + push(b, Entry::Scalar); } - fn push(b: &Rc, fast: Option) { + fn push(b: &Rc, entry: Entry) { let mut reg = REGISTRY.lock(); reg.retain(|(w, _)| w.strong_count() > 0); - reg.push((Rc::downgrade(b), fast)); + reg.push((Rc::downgrade(b), entry)); GENERATION.fetch_add(1, Ordering::Release); } @@ -55514,7 +58748,7 @@ pub(crate) mod leaf_builtins { pub(crate) type LeafMap = std::collections::HashMap< usize, - (Weak, Option), + (Weak, Entry), crate::fasthash::FxBuildHasher, >; } @@ -55702,6 +58936,27 @@ fn is_object_new(b: &Rc) -> bool { Rc::as_ptr(b) as usize == want } +/// Extend `v` to `n` slots with `Unbound` (the fresh locals of an +/// activation), in line: a call's handful of slots isn't worth +/// `Vec::resize`'s out-of-line loop. +#[inline(always)] +fn fill_unbound(v: &mut Vec, n: usize) { + let len = v.len(); + if len >= n { + return; + } + v.reserve(n - len); + // SAFETY: capacity reserved above; `Unbound` owns nothing, and every + // slot below `n` is written before the length covers it. + unsafe { + let p = v.as_mut_ptr(); + for k in len..n { + p.add(k).write(Object::Unbound); + } + v.set_len(n); + } +} + /// Release an operand the burst is done with: an unboxed scalar has no /// heap and `Object` has no `Drop` of its own, so it needs no drop glue. #[inline(always)] @@ -55731,6 +58986,14 @@ enum LeafStop { Core, } +/// A step of a dict iterator in the core loop (see +/// [`Interpreter::core_dict_next`]). +enum DictStep { + Item(Object, Option), + Exhausted, + Decline, +} + /// Why [`Interpreter::leaf_core`] handed control back. enum CoreExit { /// End the burst. @@ -55776,6 +59039,7 @@ static SLOW_LEAF_OPS: [bool; 256] = { OpCode::CompareOp, OpCode::CopyFreeVars, OpCode::CopyTop, + OpCode::DictUpdate, OpCode::ForIter, OpCode::FormatValue, OpCode::GetIter, @@ -55829,6 +59093,10 @@ static CORE_LEAF_OPS: [bool; 256] = { OpCode::StoreFast, OpCode::PopTop, OpCode::BinaryOp, + OpCode::BinarySubscr, + OpCode::StoreSubscr, + OpCode::ListAppend, + OpCode::UnpackSequence, OpCode::CompareOp, OpCode::PopJumpIfFalse, OpCode::PopJumpIfTrue, @@ -55851,6 +59119,7 @@ static CORE_LEAF_OPS: [bool; 256] = { OpCode::ToBool, OpCode::PopJumpIfNone, OpCode::PopJumpIfNotNone, + OpCode::CallEx, ]; let mut i = 0; while i < ops.len() { @@ -55990,6 +59259,43 @@ pub(crate) fn callable_function_str(callable: &Object) -> Option { } } +/// `type(obj).(obj, arg?)` for an instance whose class resolves the +/// dunder `name` to a native builtin that binds its instance: the +/// builtin's body, called directly (no bound method, no general call). +/// `None`, with nothing done, for any other resolution (a Python method, a +/// descriptor, an interpreter-aware builtin) or while a profiler or tracer +/// could observe the call; the caller then takes its ordinary path. +fn instance_native_dunder( + obj: &Object, + name: &str, + arg: Option<&Object>, +) -> Option> { + let Object::Instance(inst) = obj else { + return None; + }; + if crate::trace::any_observers_active() { + return None; + } + let Object::Builtin(b) = inst.cls().lookup(name)? else { + return None; + }; + if !b.binds_instance || builtin_needs_interp(b.name) { + return None; + } + let pair; + let args: &[Object] = match arg { + Some(a) => { + pair = [obj.clone(), a.clone()]; + &pair + } + None => std::slice::from_ref(obj), + }; + Some(match b.call_kw.as_ref() { + Some(ckw) => ckw(args, &[]), + None => (b.call)(args), + }) +} + pub(crate) fn instance_method(obj: &Object, name: &str) -> Option { let inst = match obj { Object::Instance(i) => i.clone(), @@ -56037,8 +59343,45 @@ thread_local! { /// identity `enum`'s bootstrap (`found in (data_type_method, object_method)`) /// depends on. Only built-in types are keyed, so the entries live as long /// as the type singletons themselves. - static SLOT_WRAPPER_CACHE: std::cell::RefCell> = - std::cell::RefCell::new(std::collections::HashMap::new()); + /// `builtin_type_dunder` results (misses included) by built-in type + /// address, then name. + #[allow(clippy::type_complexity)] + static SLOT_WRAPPER_CACHE: std::cell::RefCell< + crate::fasthash::FxHashMap, Option>>, + > = std::cell::RefCell::new(crate::fasthash::FxHashMap::default()); +} + +/// Whether `obj.name` certainly raises `AttributeError` without running +/// any code: an ordinary instance under the default attribute protocol, +/// with no class attribute of that name and none in its `__dict__`. A +/// `getattr` default or `hasattr` then needs no exception at all. Dunder +/// names, which the lookup special-cases, never qualify. +fn attr_certainly_missing(obj: &Object, name: &str) -> bool { + use crate::types::Dunder; + let Object::Instance(inst) = obj else { + return false; + }; + if name.starts_with("__") || inst.native.get().is_some() || inst.c_body.get() != 0 { + return false; + } + let cls = inst.cls(); + if cls.native_kind.get() != 0 + || !cls.dunder(Dunder::GetAttribute).object_owner() + || cls.dunder(Dunder::GetAttr).present() + || cls.lookup(name).is_some() + { + return false; + } + match inst.dict.published() { + Some(d) => d + .try_borrow() + .is_ok_and(|d| !d.contains_key(&crate::object::StrKey(name))), + None => inst + .dict + .split_cell() + .try_borrow() + .is_ok_and(|s| s.position_str(name).is_none()), + } } /// Resolve a built-in slot-wrapper dunder reached via *type-level* attribute @@ -56064,22 +59407,31 @@ pub(crate) fn builtin_slot_wrapper(ty: &Rc, name: &str) -> Option Object::None, }); } - let mro: Vec> = ty.mro.borrow().iter().cloned().collect(); - for base in mro { + for base in ty.mro.borrow().iter() { if !base.flags.is_builtin { continue; } - let ptr = Rc::as_ptr(&base) as usize; - if let Some(o) = - SLOT_WRAPPER_CACHE.with(|c| c.borrow().get(&(ptr, name.to_owned())).cloned()) - { - return Some(o); - } - if let Some(o) = crate::builtins::builtin_type_dunder(&base.name, name) { - SLOT_WRAPPER_CACHE.with(|c| { - c.borrow_mut().insert((ptr, name.to_owned()), o.clone()); - }); - return Some(o); + let ptr = Rc::as_ptr(base) as usize; + let cached = SLOT_WRAPPER_CACHE.with(|c| { + c.borrow() + .get(&ptr) + .and_then(|names| names.get(name).cloned()) + }); + let dunder = match cached { + Some(hit) => hit, + None => { + let found = crate::builtins::builtin_type_dunder(&base.name, name); + SLOT_WRAPPER_CACHE.with(|c| { + c.borrow_mut() + .entry(ptr) + .or_default() + .insert(name.into(), found.clone()); + }); + found + } + }; + if dunder.is_some() { + return dunder; } // The base's own type dict — where `object.__reduce_ex__` / // `__reduce__` / `__getattribute__` sentinels live. Already @@ -56250,6 +59602,19 @@ fn dunder_key(which: usize) -> &'static Object { /// class in that prefix carries its own `__init__`, return it. /// Otherwise the caller can stick with the cheap `args`-only setup. fn lookup_exception_init(cls: &Rc) -> Option { + if exc_family_flags(cls) & EXC_CUSTOM_INIT == 0 { + return None; + } + exception_init_uncached(cls) +} + +/// [`exc_family_flags`]'s pseudo-family: a class between `cls` and +/// `BaseException` defines its own `__init__` (see +/// [`lookup_exception_init`]). +const EXC_CUSTOM_INIT: u16 = 1 << 9; + +/// [`lookup_exception_init`]'s MRO walk. +fn exception_init_uncached(cls: &TypeObject) -> Option { let mro = cls.mro.borrow(); for ty in mro.iter() { if ty.name == "BaseException" || ty.name == "object" { @@ -56563,8 +59928,12 @@ fn exception_holds_nonatomic(obj: &Object) -> bool { return true; } } - let Some(dict) = inst.dict.get() else { - return false; + let Some(dict) = inst.dict.published() else { + return inst + .dict + .split_cell() + .try_borrow() + .map_or(true, |s| s.values().iter().any(nonatomic)); }; let Ok(dict) = dict.try_borrow() else { // Borrowed elsewhere (shouldn't happen on a just-built instance); err @@ -56647,109 +60016,95 @@ fn oserror_subclass_name(errno: i64) -> Option<&'static str> { Some(name) } -fn sort_key_needs_dunder_lt(o: &Object) -> bool { - sort_key_needs_dunder_lt_depth(o, 0) +/// The comparison `list.sort` uses for a list of keys, chosen by the +/// CPython pre-sort check: keys of one native type compare natively. +#[derive(Clone, Copy, PartialEq, Eq)] +enum SortKind { + Str, + Int, + Float, + /// Mixed `bool`, `int` and `float` keys. + Numeric, + /// Non-empty tuples whose first items have the given scalar kind. + Tuples(SortKindScalar), + General, +} + +/// The scalar [`SortKind`]s, for the first items of tuple keys. +#[derive(Clone, Copy, PartialEq, Eq)] +enum SortKindScalar { + Str, + Int, + Float, + Numeric, + General, } -/// Whether a sort key must be ordered through Python's `<` (rich-comparison -/// dispatch) rather than the fast Rust `Object::cmp` total order. True for a -/// custom instance carrying a `__lt__`, *and* — crucially — for a `tuple`/ -/// `list` that (recursively) contains one: the Rust `seq_cmp` path bottoms -/// out in `Object::cmp`, which has no case for user instances and would -/// raise a spurious "'<' not supported" (e.g. `pprint`'s `_safe_tuple` keys, -/// which wrap each dict key/value in a `__lt__`-only `_safe_key`). The depth -/// guard avoids unbounded recursion on self-referential containers. -fn sort_key_needs_dunder_lt_depth(o: &Object, depth: u32) -> bool { - if depth > 100 { - return false; - } - match o { - Object::Instance(inst) => { - // A user-defined Python `__lt__`/bound method always needs the - // Python `<` path (the historical case: `pprint`'s `_safe_key`). - if matches!( - inst.cls().lookup("__lt__"), - Some(Object::Function(_) | Object::BoundMethod(_)) - ) { - return true; - } - // CPython's `list.sort`/`sorted` *always* order through `<` - // (`PyObject_RichCompareBool(x, y, Py_LT)`); WeavePy's native - // `Object::cmp` is only a safe fast path when it yields the - // identical answer. That holds for a plain value subclass - // (`class E(int)`) — ordered by its `native_value()` payload — - // but NOT for a Cython/C extension class surfaced as an - // `Object::Instance` (pandas `Period`/`Timestamp`, - // `decimal.Decimal`) whose `__lt__` is a C slot wrapper - // (`builtin_function_or_method`) and which carries no native - // scalar for `Object::cmp` to order. Detect exactly that: an - // ordering dunder is present, yet `native_value()` is `None`, so - // the native path would raise a spurious "'<' not supported". - if o.native_value().is_none() - && (inst.cls().lookup("__lt__").is_some() || inst.cls().lookup("__gt__").is_some()) - { - return true; +impl SortKind { + fn of<'a>(keys: impl Iterator + Clone) -> Self { + let mut first = keys.clone().next(); + if let Some(Object::Tuple(t)) = first { + if t.is_empty() { + first = None; } - // A `list`/`tuple` *subclass* instance orders as its native - // payload — recurse into it so contained `__lt__`-bearing - // elements still get Python `<` dispatch - // (test_sort.test_unsafe_object_compare's WackyList1). - match inst.native.get() { - Some(n @ (Object::List(_) | Object::Tuple(_))) => { - sort_key_needs_dunder_lt_depth(n, depth + 1) + } + if matches!(first, Some(Object::Tuple(_))) { + let mut firsts = Vec::new(); + for k in keys { + match k { + Object::Tuple(t) if !t.is_empty() => firsts.push(&t[0]), + _ => return Self::General, } - _ => false, } + return Self::Tuples(SortKindScalar::of(firsts.into_iter())); } - // A genuinely foreign extension object (a bare numpy scalar not - // surfaced as an `Object::Instance`, …) orders through its - // `tp_richcompare` slot, which the native `Object::cmp` total order - // cannot reach — so it must sort through Python's `<`. - Object::Foreign(_) => true, - Object::Tuple(items) => items - .iter() - .any(|x| sort_key_needs_dunder_lt_depth(x, depth + 1)), - Object::List(items) => items - .borrow() - .iter() - .any(|x| sort_key_needs_dunder_lt_depth(x, depth + 1)), - _ => false, + match SortKindScalar::of(keys) { + SortKindScalar::Str => Self::Str, + SortKindScalar::Int => Self::Int, + SortKindScalar::Float => Self::Float, + SortKindScalar::Numeric => Self::Numeric, + SortKindScalar::General => Self::General, + } + } + + #[inline] + fn str_lt(a: &Object, b: &Object) -> bool { + // UTF-8 byte order is code point order. + matches!((a, b), (Object::Str(x), Object::Str(y)) if x.as_bytes() < y.as_bytes()) + } + + #[inline] + fn int_lt(a: &Object, b: &Object) -> bool { + matches!((a, b), (Object::Int(x), Object::Int(y)) if x < y) + } + + #[inline] + fn float_lt(a: &Object, b: &Object) -> bool { + matches!((a, b), (Object::Float(x), Object::Float(y)) if x < y) } } -/// Stable merge sort that orders by Python `<` (full rich-comparison -/// dispatch, reflected operands included). `key` projects the comparison -/// object out of each element (the decorated sort key, or the element -/// itself). Comparison errors (unorderable types, raising `__lt__`) -/// propagate exactly as CPython's `list.sort` does. -fn merge_sort_by_pylt( - interp: &mut Interpreter, - mut v: Vec, - key: &impl Fn(&T) -> &Object, - globals: &Rc>, -) -> Result, RuntimeError> { - if v.len() <= 1 { - return Ok(v); - } - let right = v.split_off(v.len() / 2); - let left = merge_sort_by_pylt(interp, v, key, globals)?; - let right = merge_sort_by_pylt(interp, right, key, globals)?; - let mut out = Vec::with_capacity(left.len() + right.len()); - let (mut li, mut ri) = (0, 0); - while li < left.len() && ri < right.len() { - // Stability: take from the right run only when strictly smaller - // (timsort's `b < a` merge test). - if interp.dispatch_compare_op(key(&right[ri]), key(&left[li]), CompareKind::Lt, globals)? { - out.push(right[ri].clone()); - ri += 1; - } else { - out.push(left[li].clone()); - li += 1; +impl SortKindScalar { + fn of<'a>(keys: impl Iterator) -> Self { + let mut kind = None; + for k in keys { + let this = match k { + Object::Str(_) => Self::Str, + Object::Int(_) => Self::Int, + Object::Float(_) => Self::Float, + Object::Long(_) | Object::Bool(_) => Self::Numeric, + _ => return Self::General, + }; + kind = Some(match kind { + None => this, + Some(k) if k == this => k, + Some(Self::Str) | Some(Self::General) => return Self::General, + Some(_) if this == Self::Str => return Self::General, + Some(_) => Self::Numeric, + }); } + kind.unwrap_or(Self::General) } - out.extend_from_slice(&left[li..]); - out.extend_from_slice(&right[ri..]); - Ok(out) } /// Convert a freshly built `set` result into a `frozenset` — used when @@ -57279,15 +60634,17 @@ fn resolve_field_name( let mut value = if base.is_empty() { let idx = *auto_idx; *auto_idx += 1; - positional - .get(idx) - .cloned() - .ok_or_else(|| index_error(format!("Replacement index {idx} out of range")))? + positional.get(idx).cloned().ok_or_else(|| { + index_error(format!( + "Replacement index {idx} out of range for positional args tuple" + )) + })? } else if let Ok(idx) = base.parse::() { - positional - .get(idx) - .cloned() - .ok_or_else(|| index_error(format!("Replacement index {idx} out of range")))? + positional.get(idx).cloned().ok_or_else(|| { + index_error(format!( + "Replacement index {idx} out of range for positional args tuple" + )) + })? } else if let Some(map) = mapping { let key = DictKey(Object::from_str(base)); map.borrow() @@ -57442,6 +60799,21 @@ pub(crate) enum PercentMode { Bytes, } +/// A `%`-format argument (or argument tuple) of scalars and exact +/// strings: its conversions run no Python code. +fn percent_leaf_args(args: &Object) -> bool { + let leaf = |o: &Object| { + matches!( + o, + Object::Int(_) | Object::Float(_) | Object::Str(_) | Object::Bool(_) | Object::None + ) + }; + match args { + Object::Tuple(items) => items.iter().all(leaf), + other => leaf(other), + } +} + pub(crate) fn percent_format(template: &str, value: &Object) -> Result { let mut noop = |_: &Object, _: char| Ok(None); percent_format_with(template, value, PercentMode::Str, &mut noop) @@ -60064,6 +63436,21 @@ struct CodeConstObjects { /// splits them back apart; the core loop reads one byte to run the /// pair as a single dispatch. Empty when the code has no pair. fast_pairs: std::sync::OnceLock>, + /// Whether the code is a generator body fast steps run (see + /// `gen_fast`): 0 not yet scanned, 1 no, 2 yes. + gen_fast: std::sync::atomic::AtomicU8, + /// Split-layout attribute shortcuts per `LOAD_ATTR` site (see + /// [`FieldSlot`]); allocated on the first recorded one. + field_slots: std::sync::OnceLock>, + /// Verified frameless method calls per method-load site (see + /// [`LeafSite`]); allocated on the first recorded one. + leaf_sites: std::sync::OnceLock>, + /// A leaf body's translation for the frameless evaluator (see + /// [`leaf_plan`]), or `None` when the body has none. + leaf_plan: std::sync::OnceLock>>, + /// Whether every return is `return None` (see + /// [`code_returns_only_none`]): `0` not yet decided, `1` no, `2` yes. + returns_none: std::sync::atomic::AtomicU8, } /// A `CALL` site's inline-call shape (see `Interpreter::core_call`): the @@ -60074,7 +63461,22 @@ struct CodeConstObjects { /// that can change under a fixed code object — the defaults, the /// closure's cells, the JIT's claim on the code — is re-validated per /// call. -struct CallSlot(std::cell::UnsafeCell>); +/// A `CALL` site's inline-call shape, plus up to [`POLY_CALLS`] earlier +/// ones for a site that calls several functions (a method call over +/// receivers of several classes). +struct CallSlot( + std::cell::UnsafeCell>, + std::cell::UnsafeCell>>, +); + +/// Earlier shapes a polymorphic [`CallSlot`] keeps. +const POLY_CALLS: usize = 3; + +/// A [`CallSlot`]'s earlier shapes, replaced round-robin. +struct PolyCalls { + shapes: [Option; POLY_CALLS], + next: usize, +} struct CallShape { func: crate::sync::Weak, @@ -60091,9 +63493,24 @@ struct CallShape { unsafe impl Send for CallSlot {} unsafe impl Sync for CallSlot {} +impl CallShape { + #[inline] + fn matches( + &self, + func: *const crate::object::PyFunction, + code: *const CodeObject, + ) -> Option<(u32, bool)> { + (std::ptr::eq(self.func.as_ptr(), func) && std::ptr::eq(self.code.as_ptr(), code)) + .then_some((self.missing, self.has_self)) + } +} + impl CallSlot { const fn empty() -> Self { - Self(std::cell::UnsafeCell::new(None)) + Self( + std::cell::UnsafeCell::new(None), + std::cell::UnsafeCell::new(None), + ) } /// The shape recorded for `func` running `code` (by identity). @@ -60105,16 +63522,53 @@ impl CallSlot { ) -> Option<(u32, bool)> { // SAFETY: see the type docs. let shape = unsafe { &*self.0.get() }.as_ref()?; - (std::ptr::eq(shape.func.as_ptr(), func) && std::ptr::eq(shape.code.as_ptr(), code)) - .then_some((shape.missing, shape.has_self)) + shape + .matches(func, code) + .or_else(|| self.poly_hit(func, code)) + } + + /// [`Self::hit`] among the earlier shapes. + #[cold] + #[inline(never)] + fn poly_hit( + &self, + func: *const crate::object::PyFunction, + code: *const CodeObject, + ) -> Option<(u32, bool)> { + // SAFETY: see the type docs. + let poly = unsafe { &*self.1.get() }.as_deref()?; + poly.shapes + .iter() + .flatten() + .find_map(|shape| shape.matches(func, code)) } #[inline] fn set(&self, shape: CallShape) { + let (func, code) = (shape.func.as_ptr(), shape.code.as_ptr()); // SAFETY: see the type docs (the old handles drop after the // store; dropping a weak handle runs no code). let old = unsafe { (*self.0.get()).replace(shape) }; - drop(old); + // A different live callee keeps its shape among the earlier ones. + if let Some(old) = old.filter(|o| { + o.func.strong_count() > 0 && !(o.func.as_ptr() == func && o.code.as_ptr() == code) + }) { + // SAFETY: as above. + let poly = unsafe { &mut *self.1.get() }.get_or_insert_with(|| { + Box::new(PolyCalls { + shapes: std::array::from_fn(|_| None), + next: 0, + }) + }); + let seen = poly.shapes.iter().flatten().any(|s| { + s.func.as_ptr() == old.func.as_ptr() && s.code.as_ptr() == old.code.as_ptr() + }); + if !seen { + let i = poly.next; + poly.shapes[i] = Some(old); + poly.next = (i + 1) % POLY_CALLS; + } + } } } @@ -60131,13 +63585,16 @@ fn code_call_slot(code: &CodeObject, pc: usize) -> Option<&CallSlot> { .get(pc) } -/// A `LOAD_ATTR` site's polymorphic instance-attribute cache: up to two -/// `(class attr_version, instance-dict index)` pairs. Versions are +/// A `LOAD_ATTR` site's polymorphic instance-attribute cache: up to +/// [`ATTR_POLY`] `(class attr_version, instance-dict index)` pairs. Versions are /// process-unique and never reused, so a match names exactly one class /// in one state — the state in which the class cache resolved the name /// to the instance dict (no data descriptor, default /// `__getattribute__`); the index is a hint the name check validates. -struct AttrPoly(std::cell::UnsafeCell<[(u64, u32); 2]>); +struct AttrPoly(std::cell::UnsafeCell<[(u64, u32); ATTR_POLY]>); + +/// Receiver classes an [`AttrPoly`] remembers. +const ATTR_POLY: usize = 4; // SAFETY: read and written only from the dispatch loop with the GIL held // (the `MethodSlot` invariant). @@ -60146,7 +63603,7 @@ unsafe impl Sync for AttrPoly {} impl AttrPoly { const fn empty() -> Self { - Self(std::cell::UnsafeCell::new([(0, 0); 2])) + Self(std::cell::UnsafeCell::new([(0, 0); ATTR_POLY])) } /// The instance-dict value an entry for `ver` points at, when its @@ -60154,10 +63611,8 @@ impl AttrPoly { #[inline] fn hit(&self, code: &CodeObject, inst: &PyInstance, ver: u64, name_idx: u32) -> Option { let ix = self.index(ver)?; - let dict = inst.dict.get()?; // SAFETY: a read between two instructions (see `GilCell::peek`). - let d = unsafe { dict.peek() }?; - let (k, v) = d.get_index(ix as usize)?; + let (k, v) = unsafe { inst.attr_peek_index(ix as usize) }?; slot_name_matches(code, name_idx, k).then(|| Interpreter::clone_operand(v)) } @@ -60169,7 +63624,7 @@ impl AttrPoly { entries.iter().find(|e| e.0 == ver && ver != 0).map(|e| e.1) } - /// Remember `ver` → `ix` (the older of two full entries yields). + /// Remember `ver` → `ix` (the oldest of full entries yields). #[inline] fn record(&self, ver: u64, ix: u32) { // SAFETY: see the type docs. @@ -60177,8 +63632,8 @@ impl AttrPoly { if let Some(e) = entries.iter_mut().find(|e| e.0 == ver || e.0 == 0) { *e = (ver, ix); } else { - entries[0] = entries[1]; - entries[1] = (ver, ix); + entries.rotate_left(1); + entries[ATTR_POLY - 1] = (ver, ix); } } } @@ -60227,6 +63682,143 @@ impl StampSlot { } } +/// A `LOAD_ATTR` site's split-layout shortcut: the receiver class's +/// attribute version and the attribute's position among the class's +/// shared names, recorded after a full site read proved the name there +/// (see [`field_slot_note`]). The names are append-only and the version +/// is process-unique, so while both match, the value at that position of +/// a split instance *is* the attribute: no inline-cache decode and no +/// name comparison. +struct FieldSlot(std::cell::UnsafeCell<(u64, u32)>); + +// SAFETY: as `StampSlot`. +unsafe impl Send for FieldSlot {} +unsafe impl Sync for FieldSlot {} + +impl FieldSlot { + const fn empty() -> Self { + Self(std::cell::UnsafeCell::new((0, 0))) + } + + #[inline(always)] + fn get(&self) -> (u64, u32) { + // SAFETY: GIL-serialized; no `&mut` escapes `set`. + unsafe { *self.0.get() } + } + + #[inline] + fn set(&self, v: (u64, u32)) { + // SAFETY: as `StampSlot::set`. + unsafe { *self.0.get() = v }; + } +} + +/// A local receiver's `x.m()` site whose last call ran +/// frameless (see `Interpreter::core_leaf_site_call`), keyed at the method +/// load: under the receiver class's (process-unique) attribute version +/// `ver`, the method is `func`, whose code was `code`, a pure (or, with +/// `effect`, an effect) leaf taking exactly the site's arguments. The +/// class holds the function at that version; `func` is weak all the same, +/// as [`MethodSlot`]'s are. +struct LeafSite(std::cell::UnsafeCell>); + +struct LeafSiteData { + ver: u64, + func: crate::sync::Weak, + code: *const CodeObject, + effect: bool, +} + +// SAFETY: as `StampSlot`. +unsafe impl Send for LeafSite {} +unsafe impl Sync for LeafSite {} + +impl LeafSite { + const fn empty() -> Self { + Self(std::cell::UnsafeCell::new(None)) + } +} + +/// The site at `pc`'s verified callee under `ver`: its function, code and +/// effect flag. +#[inline(always)] +fn leaf_site_hit( + ext: &CodeConstObjects, + pc: usize, + ver: u64, +) -> Option<(*const crate::object::PyFunction, *const CodeObject, bool)> { + // SAFETY: GIL-serialized; the borrow ends before any refill. + let site = unsafe { &*ext.leaf_sites.get()?.get(pc)?.0.get() }.as_ref()?; + (site.ver == ver && site.func.strong_count() > 0) + .then(|| (site.func.as_ptr(), site.code, site.effect)) +} + +/// Record (or, with `data` `None`, forget) the site at `pc`'s callee. +#[inline(never)] +fn leaf_site_set(ext: &CodeConstObjects, ninstrs: usize, pc: usize, data: Option) { + if data.is_none() && ext.leaf_sites.get().is_none() { + return; + } + let sites = ext + .leaf_sites + .get_or_init(|| (0..ninstrs).map(|_| LeafSite::empty()).collect()); + if let Some(site) = sites.get(pc) { + // SAFETY: GIL-serialized; no reference into the site is live. + unsafe { *site.0.get() = data }; + } +} + +/// What a verified site's call did (see +/// `Interpreter::core_leaf_site_call`). +enum SiteCall { + /// The result, and the `CALL`'s pc. + Done(Object, usize), + /// A check failed before anything ran: the ordinary paths decide. + Declined, + /// The evaluation declined (the site is forgotten): the call takes + /// the ordinary, framed path this time. + Missed, +} + +/// The attribute `inst` holds for the `LOAD_ATTR` at `pc`, through the +/// site's [`FieldSlot`] (never recorded for a slot or native class). +/// +/// # Safety +/// +/// As [`crate::sync::GilCell::peek`]: the view must not outlive anything +/// that could store an attribute or release the receiver. +#[inline(always)] +unsafe fn field_slot_hit<'a>( + ext: &CodeConstObjects, + pc: usize, + inst: &'a PyInstance, +) -> Option<&'a Object> { + let (ver, idx) = ext.field_slots.get()?.get(pc)?.get(); + if inst.cls_raw().attr_version.get() != ver { + return None; + } + // SAFETY: forwarded contract. + unsafe { inst.split_field(idx as usize) } +} + +/// Record the [`FieldSlot`] shortcut for the `LOAD_ATTR` at `pc`, after +/// a guarded read found its name at dictionary position `idx` of `inst` +/// (whose class passed the site's version check): kept only when that +/// position is the split layout's, over the class's own names. +#[inline(never)] +fn field_slot_note(ext: &CodeConstObjects, ninstrs: usize, pc: usize, inst: &PyInstance, idx: u32) { + // SAFETY: a read with nothing running (the caller's own guarded read). + if unsafe { inst.split_field(idx as usize) }.is_none() { + return; + } + let slots = ext + .field_slots + .get_or_init(|| (0..ninstrs).map(|_| FieldSlot::empty()).collect()); + if let Some(slot) = slots.get(pc) { + slot.set((inst.cls_raw().attr_version.get(), idx)); + } +} + /// `v.clone()` with the scalars copied and the common heap variants' /// increments in line (the enum's `Clone` is an out-of-line call and a /// variant switch per clone). @@ -60290,11 +63882,26 @@ const CLASS_ATTR_NONE: u64 = 0x5ca1_a770_0000_0004; /// `cls`'s current version (see [`class_attr_fill`]). #[inline(always)] fn class_attr_hit(slot: &StampSlot, cls: &crate::types::TypeObject) -> Option { + class_attr_hit_via(slot, cls, 0) +} + +/// Marks a stamp remembered for a read through an *instance* of the +/// class, which also needs the instance not to shadow the name. +const CLASS_ATTR_VIA_INSTANCE: u64 = 0x100; + +/// [`class_attr_hit`] for a stamp filled with `via` (`0`, or +/// [`CLASS_ATTR_VIA_INSTANCE`]). +#[inline(always)] +fn class_attr_hit_via( + slot: &StampSlot, + cls: &crate::types::TypeObject, + via: u64, +) -> Option { let [ver, tag, bits] = slot.get(); if ver != cls.attr_version.get() { return None; } - match tag { + match tag ^ via { CLASS_ATTR_INT => Some(Object::Int(bits as i64)), CLASS_ATTR_FLOAT => Some(Object::Float(f64::from_bits(bits))), CLASS_ATTR_BOOL => Some(Object::Bool(bits != 0)), @@ -60309,6 +63916,17 @@ fn class_attr_hit(slot: &StampSlot, cls: &crate::types::TypeObject) -> Option (CLASS_ATTR_INT, *x as u64), Object::Float(x) => (CLASS_ATTR_FLOAT, x.to_bits()), @@ -60317,7 +63935,7 @@ fn class_attr_fill(code: &CodeObject, cls: &crate::types::TypeObject, pc: usize, _ => return, }; if let Some(slot) = code_stamp_slot(code, pc as u32) { - slot.set([cls.attr_version.get(), tag, bits]); + slot.set([cls.attr_version.get(), tag ^ via, bits]); } } @@ -60330,7 +63948,25 @@ fn class_attr_fill(code: &CodeObject, cls: &crate::types::TypeObject, pc: usize, /// the name compare that the `InlineCache` shape needs. The reference /// is weak (CPython's `_PyType_Lookup` cache is a borrowed pointer for /// the same reason): a site must not keep a class or module alive. -struct MethodSlot(std::cell::UnsafeCell<(u64, MethodSlotFn)>); +/// +/// A site that sees several receiver classes (`c.execute()` over a list +/// of constraint subclasses) keeps up to [`POLY_METHODS`] earlier +/// resolutions beside the current one, so it stops re-resolving on every +/// class change. +struct MethodSlot( + std::cell::UnsafeCell<(u64, MethodSlotFn)>, + std::cell::UnsafeCell>>, +); + +/// Earlier resolutions a polymorphic [`MethodSlot`] keeps. +const POLY_METHODS: usize = 3; + +/// A [`MethodSlot`]'s earlier plain or static functions, by class +/// version (`0` marks an empty entry), replaced round-robin. +struct PolyMethods { + entries: [(u64, crate::sync::Weak, bool); POLY_METHODS], + next: usize, +} /// What a [`MethodSlot`] resolved: a Python function off a class (keyed /// by the class's attribute version), or a builtin type's method body @@ -60368,10 +64004,68 @@ impl MethodSlot { const BUILTIN_TAG: u64 = 1 << 63; const fn empty() -> Self { - Self(std::cell::UnsafeCell::new(( - 0, - MethodSlotFn::Py(crate::sync::Weak::new()), - ))) + Self( + std::cell::UnsafeCell::new((0, MethodSlotFn::Py(crate::sync::Weak::new()))), + std::cell::UnsafeCell::new(None), + ) + } + + /// An earlier resolution under `ver` (a static method's only when + /// `unbound`), without a reference (see `peek_fn`). + #[cold] + #[inline(never)] + fn poly_peek(&self, ver: u64, unbound: bool) -> Option<*const crate::object::PyFunction> { + // SAFETY: GIL-serialized; no `&mut` escapes `poly_remember`. + let poly = unsafe { &*self.1.get() }.as_deref()?; + poly.entries + .iter() + .find(|(v, w, is_static)| { + *v == ver && ver != 0 && (unbound || !*is_static) && w.strong_count() > 0 + }) + .map(|(_, w, _)| w.as_ptr()) + } + + /// Keep the resolution the slot is about to replace, if it names a + /// live plain or static function under another version. + fn poly_remember(&self, ver: u64) { + // SAFETY: GIL-serialized; the references end before the slot is + // rewritten. + let (old_ver, old) = unsafe { &*self.0.get() }; + let (w, is_static) = match old { + MethodSlotFn::Py(w) => (w, false), + MethodSlotFn::Static(w) => (w, true), + _ => return, + }; + if *old_ver == 0 || *old_ver == ver || w.strong_count() == 0 { + return; + } + // SAFETY: as above. + let poly = unsafe { &mut *self.1.get() }.get_or_insert_with(|| { + Box::new(PolyMethods { + entries: std::array::from_fn(|_| (0, crate::sync::Weak::new(), false)), + next: 0, + }) + }); + if poly.entries.iter().any(|(v, _, _)| v == old_ver) { + return; + } + let i = poly.next; + poly.entries[i] = (*old_ver, w.clone(), is_static); + poly.next = (i + 1) % POLY_METHODS; + } + + /// A strong handle to the function at `p` (see `get_held`). + /// + /// # Safety + /// + /// `p` names a live function its class holds. + #[inline] + unsafe fn held(p: *const crate::object::PyFunction) -> Rc { + // SAFETY: the caller's contract. + unsafe { + Rc::increment_strong_count(p); + Rc::from_raw(p) + } } /// The cached function if the slot was filled under `ver`. @@ -60380,6 +64074,10 @@ impl MethodSlot { // SAFETY: GIL-serialized; no `&mut` escapes `set`. match unsafe { &*self.0.get() } { (v, MethodSlotFn::Py(w)) if *v == ver => w.upgrade(), + // SAFETY: the class holds a function it resolved at `ver`. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, false).map(|p| unsafe { Self::held(p) }) + } _ => None, } } @@ -60402,12 +64100,17 @@ impl MethodSlot { Some(Rc::from_raw(p)) } } + // SAFETY: as above. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, false).map(|p| unsafe { Self::held(p) }) + } _ => None, } } #[inline] fn set(&self, ver: u64, f: &Rc) { + self.poly_remember(ver); // SAFETY: GIL-serialized; the exclusive reference lives only for // the assignment. unsafe { *self.0.get() = (ver, MethodSlotFn::Py(Rc::downgrade(f))) }; @@ -60431,6 +64134,10 @@ impl MethodSlot { Some(Rc::from_raw(p)) } } + // SAFETY: as above. + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => { + self.poly_peek(ver, true).map(|p| unsafe { Self::held(p) }) + } _ => None, } } @@ -60443,6 +64150,7 @@ impl MethodSlot { // SAFETY: GIL-serialized; no `&mut` escapes `set`. match unsafe { &*self.0.get() } { (v, MethodSlotFn::Py(w)) if *v == ver && w.strong_count() > 0 => Some(w.as_ptr()), + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => self.poly_peek(ver, false), _ => None, } } @@ -60458,6 +64166,7 @@ impl MethodSlot { { Some(w.as_ptr()) } + (_, MethodSlotFn::Py(_) | MethodSlotFn::Static(_)) => self.poly_peek(ver, true), _ => None, } } @@ -60465,6 +64174,7 @@ impl MethodSlot { /// Remember a static method's function read through a class at `ver`. #[inline] fn set_static(&self, ver: u64, f: &Rc) { + self.poly_remember(ver); // SAFETY: as `set`. unsafe { *self.0.get() = (ver, MethodSlotFn::Static(Rc::downgrade(f))) }; } @@ -60483,6 +64193,19 @@ impl MethodSlot { } } + /// [`Self::get_inst_builtin`] without taking a reference: the slot + /// (and the class) keep the builtin alive while it is used. + #[inline] + fn peek_inst_builtin(&self, ver: u64) -> Option<*const crate::object::BuiltinFn> { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (v, MethodSlotFn::Builtin(f)) if *v == ver && ver & Self::BUILTIN_TAG == 0 => { + Some(Rc::as_ptr(f)) + } + _ => None, + } + } + #[inline] fn set_inst_builtin(&self, ver: u64, f: &Rc) { // SAFETY: as `set`. @@ -60499,6 +64222,17 @@ impl MethodSlot { } } + /// [`Self::get_builtin`], uncounted: the leaf method table holds the + /// body for the interpreter's lifetime. + #[inline] + fn get_builtin_ptr(&self, tag: u64) -> Option<*const crate::object::BuiltinFn> { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (v, MethodSlotFn::Builtin(f)) if *v == Self::BUILTIN_TAG | tag => Some(Rc::as_ptr(f)), + _ => None, + } + } + #[inline] fn set_builtin(&self, tag: u64, f: &Rc) { // SAFETY: as `set`. @@ -60515,6 +64249,18 @@ impl MethodSlot { } } + /// [`Self::get_leaf`] by pointer identity. + #[inline] + fn get_leaf_ptr(&self, f: *const crate::object::BuiltinFn) -> Option { + // SAFETY: as `get`. + match unsafe { &*self.0.get() } { + (_, MethodSlotFn::Leaf(cached, kind)) if std::ptr::eq(Rc::as_ptr(cached), f) => { + Some(*kind) + } + _ => None, + } + } + #[inline] fn set_leaf(&self, f: &Rc, kind: LeafKind) { // SAFETY: as `set`. @@ -60622,6 +64368,10 @@ pub(crate) fn code_is_pure_leaf_pub(code: &CodeObject) -> bool { fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> &'a [u8] { let Some(ext) = ext else { return &[] }; ext.fast_pairs.get_or_init(|| { + // (`WEAVEPY_NO_PAIRS`, a bisection aid, leaves every table empty.) + if crate::hot_gates::env_flags::no_pairs() { + return Box::new([]); + } let ops = &code.instructions; let mut any = false; let mut kinds = vec![0u8; ops.len()]; @@ -60637,6 +64387,17 @@ fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> Some(OpCode::LoadAttr | OpCode::StoreAttr | OpCode::LoadMethodAttr) )), Some(OpCode::StoreFast) => 2, + // A local's `is None` / `is not None` test. + Some(OpCode::PopJumpIfNone | OpCode::PopJumpIfNotNone) => 3, + // A local's truth test. + Some(OpCode::ToBool) + if matches!( + ops.get(pc + 2).map(|i| i.op), + Some(OpCode::PopJumpIfFalse | OpCode::PopJumpIfTrue) + ) => + { + 4 + } _ => 0, }; kinds[pc] = kind; @@ -60650,12 +64411,24 @@ fn code_fast_pairs<'a>(code: &CodeObject, ext: Option<&'a CodeConstObjects>) -> }) } +/// A leaf's parameter count: its positional parameters, then its +/// `**kwargs` dictionary when it has one (bound by keyword calls only; a +/// leaf has no `*args` or keyword-only parameters). +#[inline(always)] +pub(crate) fn leaf_arity(code: &CodeObject) -> usize { + code.arg_count as usize + usize::from(code.has_varkeywords) +} + fn code_is_pure_leaf(code: &CodeObject) -> bool { + // The recorded verdict (see `code_pure_leaf_decide`): one load. + if let Some(yes) = code.jit_hint.pure_leaf() { + return yes; + } let Some(ext) = code_vm_ext(code) else { return false; }; match ext.pure_leaf.load(std::sync::atomic::Ordering::Relaxed) { - 1 => false, + 1 | 16 => false, 2..=7 => true, _ => code_pure_leaf_decide(code, ext), } @@ -60666,7 +64439,7 @@ fn code_is_pure_leaf(code: &CodeObject) -> bool { #[inline(never)] fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { use std::sync::atomic::Ordering::Relaxed; - let ok = !code.is_generator + let structural = !code.is_generator && !code.is_coroutine && !code.is_async_generator && !code.is_iterable_coroutine @@ -60674,43 +64447,65 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { && code.cellvars.is_empty() && code.freevars.is_empty() && !code.has_varargs - && !code.has_varkeywords && code.kwonly_count == 0 - && code.arg_count <= 8 - && code.varnames.len() == code.arg_count as usize + && leaf_arity(code) <= 8 + && code.varnames.len() <= 16 && code.exception_table.is_empty() - && code.instructions.len() <= 64 - && code.instructions.iter().all(|i| { - matches!( - i.op, - OpCode::Resume - | OpCode::Nop - | OpCode::NotTaken - | OpCode::LoadFast - | OpCode::LoadFastBorrow - | OpCode::LoadFastCheck - | OpCode::LoadFastLoadFast - | OpCode::LoadFastBorrowLoadFastBorrow - | OpCode::LoadConst - | OpCode::LoadSmallInt - | OpCode::LoadGlobal - | OpCode::LoadAttr - | OpCode::CompareOp - | OpCode::IsOp - | OpCode::ToBool - | OpCode::UnaryOp - | OpCode::PopJumpIfFalse - | OpCode::PopJumpIfTrue - | OpCode::PopJumpIfNone - | OpCode::PopJumpIfNotNone - | OpCode::JumpForward - | OpCode::BinaryOp - | OpCode::CopyTop - | OpCode::Swap - | OpCode::PopTop - | OpCode::ReturnValue - ) - }); + && code.instructions.len() <= 64; + let pure_op = |ins: &weavepy_compiler::Instruction| { + // A fresh empty list or dict is unobservable until stored, so + // a declined evaluation that made one still did nothing. + if matches!(ins.op, OpCode::BuildList | OpCode::BuildMap) { + return ins.arg == 0; + } + matches!( + ins.op, + OpCode::Resume + | OpCode::Nop + | OpCode::NotTaken + | OpCode::LoadFast + | OpCode::LoadFastBorrow + | OpCode::LoadFastCheck + | OpCode::StoreFast + | OpCode::StoreFastLoadFast + | OpCode::StoreFastStoreFast + | OpCode::LoadGlobalPushNull + | OpCode::PushNull + | OpCode::LoadMethodAttr + | OpCode::Call + | OpCode::LoadFastLoadFast + | OpCode::LoadFastBorrowLoadFastBorrow + | OpCode::LoadConst + | OpCode::LoadSmallInt + | OpCode::LoadGlobal + | OpCode::LoadAttr + | OpCode::CompareOp + | OpCode::IsOp + | OpCode::ToBool + | OpCode::UnaryOp + | OpCode::PopJumpIfFalse + | OpCode::PopJumpIfTrue + | OpCode::PopJumpIfNone + | OpCode::PopJumpIfNotNone + | OpCode::JumpForward + | OpCode::BinaryOp + | OpCode::CopyTop + | OpCode::Swap + | OpCode::PopTop + | OpCode::ReturnValue + ) + }; + let ok = structural && code.instructions.iter().all(pure_op); + // An effect leaf: a pure leaf but for attribute stores, which its + // evaluation buffers and commits at the return (see + // `Interpreter::pure_leaf_eval`). + let effect = !ok + && structural + && code + .instructions + .iter() + .all(|i| pure_op(i) || i.op == OpCode::StoreAttr) + && code.instructions.iter().any(|i| i.op == OpCode::StoreAttr); let shape = if ok { let body = code.instructions.as_slice(); let body = if body.first().is_some_and(|i| i.op == OpCode::Resume) { @@ -60753,10 +64548,114 @@ fn code_pure_leaf_decide(code: &CodeObject, ext: &CodeConstObjects) -> bool { } else { 1 }; + // An effect leaf's one fast shape: the setter, `self.x = ; return None`. + let shape = if effect && effect_setter_shape(code) { + 16 + } else { + shape + }; ext.pure_leaf.store(shape, Relaxed); + if effect { + code.jit_hint.set_effect_leaf(); + } else { + code.jit_hint.set_pure_leaf(ok); + } ok } +/// Whether an effect leaf's body is the setter shape (`16`, see +/// `Interpreter::pure_leaf_eval`): one attribute store of an argument, +/// constant or small int into an argument, then `return None`. +fn effect_setter_shape(code: &CodeObject) -> bool { + let body = code.instructions.as_slice(); + let body = match body.first() { + Some(i) if i.op == OpCode::Resume => &body[1..], + _ => body, + }; + let local = |op| { + matches!( + op, + OpCode::LoadFast | OpCode::LoadFastBorrow | OpCode::LoadFastCheck + ) + }; + let tail = match body { + [pair, rest @ ..] + if matches!( + pair.op, + OpCode::LoadFastLoadFast | OpCode::LoadFastBorrowLoadFastBorrow + ) => + { + rest + } + [value, recv, rest @ ..] + if (local(value.op) + || matches!(value.op, OpCode::LoadConst | OpCode::LoadSmallInt)) + && local(recv.op) => + { + rest + } + _ => return false, + }; + matches!(tail, [store, none, ret] + if store.op == OpCode::StoreAttr + && none.op == OpCode::LoadConst + && matches!(code.constants.get(none.arg as usize), Some(Constant::None)) + && ret.op == OpCode::ReturnValue) +} + +/// Whether every `RETURN_VALUE` in `code` returns the constant `None` +/// (what a constructor's `__init__` must return), decided once. +fn code_returns_only_none(code: &CodeObject) -> bool { + use std::sync::atomic::Ordering::Relaxed; + let Some(ext) = code_vm_ext(code) else { + return false; + }; + match ext.returns_none.load(Relaxed) { + 1 => return false, + 2 => return true, + _ => {} + } + let instrs = &code.instructions; + let yes = instrs.iter().enumerate().all(|(pc, ins)| { + ins.op != OpCode::ReturnValue + || pc + .checked_sub(1) + .and_then(|p| instrs.get(p)) + .is_some_and(|prev| { + prev.op == OpCode::LoadConst + && matches!(code.constants.get(prev.arg as usize), Some(Constant::None)) + }) + }); + ext.returns_none.store(if yes { 2 } else { 1 }, Relaxed); + yes +} + +/// Whether `code` is an *effect leaf* (see `code_pure_leaf_decide`), +/// deciding its leaf verdicts on first use. +#[inline] +fn code_is_effect_leaf(code: &CodeObject) -> bool { + if code.jit_hint.pure_leaf().is_none() { + code_is_pure_leaf(code); + } + code.jit_hint.effect_leaf() +} + +/// Whether the `BUILD_MAP 0` at `pc` opens `f(*args, **kwargs)`'s mapping: +/// `LOAD_FAST kwargs; DICT_MERGE 1; CALL_FUNCTION_EX` follow. +#[inline(always)] +fn forward_call_shape(code: &CodeObject, pc: usize) -> bool { + let ops = &code.instructions; + matches!( + (ops.get(pc + 1), ops.get(pc + 2), ops.get(pc + 3)), + (Some(load), Some(merge), Some(call)) + if matches!(load.op, OpCode::LoadFast | OpCode::LoadFastBorrow) + && merge.op == OpCode::DictUpdate + && merge.arg == 1 + && call.op == OpCode::CallEx + ) +} + /// Whether the instructions from `pc` could open a simple call's /// argument run (see `Interpreter::core_simple_args`): each of the first /// two is a plain operand load or the `CALL` itself. A cheap filter for @@ -60792,6 +64691,15 @@ fn pure_leaf_warm(code: &CodeObject) -> bool { true } +/// Whether `f`'s code is a pure or an effect leaf (the core loop's +/// frameless call candidates). +#[inline(always)] +fn fn_is_leaf(f: &crate::object::PyFunction) -> bool { + // SAFETY: as in `fn_is_pure_leaf`. + let code = unsafe { &*f.code.as_ptr() }; + code_is_pure_leaf(code) || code.jit_hint.effect_leaf() +} + /// Whether function `f`'s current code is a pure leaf (see /// [`code_is_pure_leaf`]). #[inline(always)] @@ -60815,14 +64723,18 @@ struct BinderLayout { #[inline] fn code_vm_ext(code: &CodeObject) -> Option<&CodeConstObjects> { - // The warm read — one acquire load and a cast — is what the dispatch - // loop pays per activation, so the table's construction lives out of - // line: inlining it here cost 4-7% on call-heavy fixtures. - let arc = match code.vm_ext.0.get() { - Some(arc) => arc, - None => code_vm_ext_init(code), - }; - Some(code_vm_ext_ref(arc)) + // The warm read — one acquire load of the cached thin pointer — is + // what the dispatch loop pays per activation, so the table's + // construction lives out of line: inlining it here cost 4-7% on + // call-heavy fixtures. + let p = code.vm_ext.1.load(std::sync::atomic::Ordering::Acquire); + if !p.is_null() { + // SAFETY: only `code_vm_ext_init` stores this pointer: the address + // of the `CodeConstObjects` the slot's `Arc` owns, alive as long as + // `code`. + return Some(unsafe { &*p.cast::() }); + } + Some(code_vm_ext_ref(code_vm_ext_init(code))) } /// Borrow an already initialized extension without filling any cache. @@ -60854,6 +64766,19 @@ fn code_vm_ext_ref(arc: &std::sync::Arc) -> &Co #[inline(never)] fn code_vm_ext_init( code: &CodeObject, +) -> &std::sync::Arc { + let arc = code_vm_ext_build(code); + let p: *const CodeConstObjects = code_vm_ext_ref(arc); + code.vm_ext + .1 + .store(p.cast_mut().cast(), std::sync::atomic::Ordering::Release); + arc +} + +#[cold] +#[inline(never)] +fn code_vm_ext_build( + code: &CodeObject, ) -> &std::sync::Arc { code.vm_ext.0.get_or_init(|| { std::sync::Arc::new(CodeConstObjects { @@ -60878,6 +64803,11 @@ fn code_vm_ext_init( call_slots: std::sync::OnceLock::new(), pure_leaf: std::sync::atomic::AtomicU8::new(0), fast_pairs: std::sync::OnceLock::new(), + gen_fast: std::sync::atomic::AtomicU8::new(0), + field_slots: std::sync::OnceLock::new(), + leaf_sites: std::sync::OnceLock::new(), + leaf_plan: std::sync::OnceLock::new(), + returns_none: std::sync::atomic::AtomicU8::new(0), }) }) } @@ -60894,10 +64824,18 @@ fn code_const_objects(code: &CodeObject) -> &[Object] { /// needed the interpreter): does dict key `key` name `co_names[name_idx]`? #[inline] fn slot_name_matches(code: &CodeObject, name_idx: u32, key: &DictKey) -> bool { + let names: &[Object] = code_vm_ext(code).map_or(&[], |t| &t.name_objs); + slot_name_matches_in(names, code, name_idx, key) +} + +/// [`slot_name_matches`] with the code's interned name objects already +/// in hand (`code_vm_ext(code).name_objs`, or empty). +#[inline(always)] +fn slot_name_matches_in(names: &[Object], code: &CodeObject, name_idx: u32, key: &DictKey) -> bool { let Object::Str(s) = &key.0 else { return false; }; - if let Some(Object::Str(n)) = code_name_obj(code, name_idx) { + if let Some(Object::Str(n)) = names.get(name_idx as usize) { if SharedStr::ptr_eq(n, s) { return true; } @@ -60968,14 +64906,17 @@ fn exc_family_bit(name: &str) -> u16 { fn exc_family_flags(cls: &crate::types::TypeObject) -> u16 { let ver = cls.attr_version.get(); let memo = cls.exc_families.get(); - if memo & 1 != 0 && memo >> 10 == ver { - return ((memo >> 1) & 0x1FF) as u16; + if memo & 1 != 0 && memo >> 11 == ver { + return ((memo >> 1) & 0x3FF) as u16; } let mut flags = 0u16; for t in cls.mro.borrow().iter() { flags |= exc_family_bit(&t.name); } - cls.exc_families.set(ver << 10 | u64::from(flags) << 1 | 1); + if exception_init_uncached(cls).is_some() { + flags |= EXC_CUSTOM_INIT; + } + cls.exc_families.set(ver << 11 | u64::from(flags) << 1 | 1); flags } @@ -60987,6 +64928,63 @@ fn code_name_obj(code: &CodeObject, name_idx: u32) -> Option<&Object> { code_vm_ext(code).and_then(|t| t.name_objs.get(name_idx as usize)) } +/// How a class's instances answer a boolean context without running +/// Python (see [`Interpreter::leaf_instance_truth`]). +#[derive(Clone)] +enum NativeTruth { + /// Neither `__bool__` nor `__len__`: always true. + AlwaysTrue, + /// A registered native `__bool__`. + Bool(Rc), + /// A registered native `__len__`. + Len(Rc), + /// Anything else: the full path decides. + Decline, +} + +/// Whether `inst`'s own attributes may shadow `co_names[name_idx]` (a +/// method cache's guard): `false` proves the name absent, and `true` also +/// answers when that can't be told without borrowing. +#[inline(always)] +fn inst_may_shadow(inst: &PyInstance, code: &CodeObject, name_idx: u32) -> bool { + match inst.dict.published() { + Some(dict) => dict_may_shadow(dict, code, name_idx), + None => { + // SAFETY: a read between two instructions (see `GilCell::peek`). + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return true; + }; + if split.is_empty() { + return false; + } + let i = name_idx as usize; + match code_vm_ext(code).map(|t| (t.name_objs.get(i), t.name_hashes.get(i))) { + Some((Some(Object::Str(n)), Some(&hash))) => { + split.position_hashed(n, hash).is_some() + } + _ => code_name_key(code, name_idx) + .is_none_or(|k| split.position_hashed(k.s, k.hash).is_some()), + } + } + } +} + +/// [`inst_may_shadow`] for an instance with a real dictionary. +#[inline(never)] +fn dict_may_shadow(dict: &RefCell, code: &CodeObject, name_idx: u32) -> bool { + // SAFETY: a read between two instructions (see `GilCell::peek`). + let Some(d) = (unsafe { dict.peek() }) else { + return true; + }; + if d.is_empty() { + return false; + } + let Some(probe) = code_name_leaf_probe(code, name_idx) else { + return true; + }; + d.may_hold_str_hash(probe.hash) && (d.contains_key(&probe) || probe.saw_exotic()) +} + /// A pre-hashed, Python-free probe for `co_names[name_idx]` (see /// [`crate::object::LeafNameProbe`]). #[inline] @@ -61058,7 +65056,7 @@ fn constant_to_object(c: Constant) -> Object { } // Hand out the pool's own `Arc` so every access sees the *same* // code object (identity, like CPython's co_consts). - Constant::Code(c) => Object::Code(c), + Constant::Code(c) => Object::Code(Rc::from_arc(c)), Constant::Ellipsis => crate::vm_singletons::ellipsis(), Constant::Slice(parts) => { let (start, stop, step) = *parts; @@ -61097,7 +65095,7 @@ fn object_to_constant(o: &Object) -> Constant { Object::FrozenSet(s) => { Constant::FrozenSet(s.iter().map(|k| object_to_constant(&k.0)).collect()) } - Object::Code(c) => Constant::Code(c.clone()), + Object::Code(c) => Constant::Code(Rc::into_arc(c.clone())), // A `slice` reaches the pool from 3.14's constant-slice folding // (`a[1:2]`); its bounds must themselves be pool-representable. Object::Slice(s) => { @@ -63196,6 +67194,35 @@ pub(crate) fn coerce_len_result(r: Object) -> Result { } } +/// `a b` for a pair of native `int`s, `float`s or `str`s, whose +/// comparison consults no protocol; `None` for anything else. +#[inline] +fn native_scalar_compare(a: &Object, b: &Object, op: CompareKind) -> Option { + use std::cmp::Ordering; + let ord = match (a, b) { + (Object::Int(x), Object::Int(y)) => Some(x.cmp(y)), + (Object::Float(x), Object::Float(y)) => x.partial_cmp(y), + (Object::Str(x), Object::Str(y)) => { + if matches!(op, CompareKind::Eq | CompareKind::NotEq) { + let eq = SharedStr::ptr_eq(x, y) || x.as_bytes() == y.as_bytes(); + return Some(eq == matches!(op, CompareKind::Eq)); + } + // UTF-8 byte order is code point order. + Some(x.as_bytes().cmp(y.as_bytes())) + } + _ => return None, + }; + // An unordered pair (a NaN) is unequal and not ordered either way. + Some(match op { + CompareKind::Eq => ord == Some(Ordering::Equal), + CompareKind::NotEq => ord != Some(Ordering::Equal), + CompareKind::Lt => ord == Some(Ordering::Less), + CompareKind::LtE => matches!(ord, Some(Ordering::Less | Ordering::Equal)), + CompareKind::Gt => ord == Some(Ordering::Greater), + CompareKind::GtE => matches!(ord, Some(Ordering::Greater | Ordering::Equal)), + }) +} + pub(crate) fn compare_op(a: &Object, b: &Object, op: CompareKind) -> Result { // CPython lifts ``<``, ``<=``, ``>``, ``>=`` to subset/superset // tests on the set family. They are *not* total orderings, so we @@ -63546,7 +67573,7 @@ next(gen) let code = compile_module(&module).expect("compile"); let mut interp = Interpreter::new(); let buf: Rc>> = Rc::new(RefCell::new(Vec::new())); - let writer: Stdout = buf.clone() as Rc>; + let writer: Stdout = crate::rc_unsize!(buf.clone() => RefCell); interp.set_stdout(writer); interp .run_module(&code) @@ -64238,7 +68265,7 @@ assert loop(2000) == 1999000 let args = [std::ptr::from_ref(receiver); 2]; eprintln!( " direct evaluator: {:?}, class_version={}, native_kind={}", - interp.pure_leaf_eval::(&code, function, &args), + interp.pure_leaf_eval::(&code, function, &args), inst.cls_raw().attr_version.get(), inst.cls_raw().native_kind.get() ); @@ -64505,6 +68532,56 @@ assert loop(2000) == 1999000 .unwrap(); } + #[test] + fn leaf_builtin_calls_run_frameless() { + const CHILD: &str = "WEAVEPY_LEAF_BUILTIN_TEST_CHILD"; + if std::env::var_os(CHILD).is_none() { + // Compiled callers reach the same evaluator; the counter is + // required with the JIT off, and semantics in both modes. + for jit in ["0", "1"] { + let status = std::process::Command::new(std::env::current_exe().unwrap()) + .args([ + "--exact", + "tests::leaf_builtin_calls_run_frameless", + "--nocapture", + ]) + .env(CHILD, "1") + .env("WEAVEPY_JIT", jit) + .status() + .expect("spawn leaf builtin test"); + assert!(status.success(), "leaf builtin child: {status}"); + } + return; + } + std::thread::Builder::new() + .stack_size(8 * 1024 * 1024) + .spawn(|| { + let source = include_str!("../../../tests/regrtest/test_leaf_builtin_calls.py"); + let module = parse_module(source).unwrap(); + let code = weavepy_compiler::compile_module_with_source( + &module, + source, + "leaf_builtin_calls.py", + ) + .unwrap(); + let before = LEAF_BUILTIN_CALLS.with(std::cell::Cell::get); + Interpreter::new() + .run_module(&code) + .expect("leaf builtin assertions"); + let calls = LEAF_BUILTIN_CALLS.with(std::cell::Cell::get) - before; + #[cfg(feature = "jit")] + let require = crate::tier2::jit_off_for_process(); + #[cfg(not(feature = "jit"))] + let require = true; + if require { + assert!(calls > 10_000, "frameless builtin calls: {calls}"); + } + }) + .unwrap() + .join() + .unwrap(); + } + #[cfg(feature = "jit")] #[test] fn pure_cached_field_predicates_preserve_fallbacks() { @@ -68995,7 +73072,7 @@ print(sliced('é😀z', 2)) _ => None, }) .expect("slice function"); - let function = Rc::make_mut(nested); + let function = std::sync::Arc::make_mut(nested); let pc = function .instructions .iter() @@ -69008,7 +73085,8 @@ print(sliced('é😀z', 2)) function.instructions[pc].arg = 2; let mut interp = Interpreter::new(); let buffer: Rc>> = Rc::new(RefCell::new(Vec::new())); - let writer: Stdout = buffer.clone() as Rc>; + let writer: Stdout = + crate::rc_unsize!(buffer.clone() => RefCell); interp.set_stdout(writer); interp.run_module(&code).expect("two-bound native fallback"); assert_eq!(String::from_utf8(buffer.borrow().clone()).unwrap(), "é😀\n"); @@ -69857,8 +73935,9 @@ print("native pickle coverage: ok") #[test] fn jit_native_lane_mismatch_falls_back() { // `dbl` compiled for int arguments; the native caller passes a - // float — the argument-lane check rejects the fast path and - // the interpreter call stays exact. + // float — the argument-lane check rejects the native fast path, + // and the call (frameless or through the interpreter) stays + // exact. let src = "def dbl(x):\n if x < 0:\n return 0\n\ \x20 return x + x\n\ k = 0\nwhile k < 10:\n dbl(3)\n k = k + 1\n\ @@ -69868,8 +73947,8 @@ print("native pickle coverage: ok") r = 0.0\nk = 0\n\ while k < 10:\n r = spin(20)\n k = k + 1\n\ print(r)\n"; - let (out, _calls, fallbacks, _deopts) = run_jit_native(src); - assert!(fallbacks >= 1, "float-for-int argument must fall back"); + let (out, _calls, _fallbacks, deopts) = run_jit_native(src); + assert_eq!(deopts, 0, "a lane mismatch never enters the native callee"); assert_eq!(out, "20.0\n"); assert_eq!(out, run(src)); } diff --git a/crates/weavepy-vm/src/object.rs b/crates/weavepy-vm/src/object.rs index a36ab67e..22b4bbbe 100644 --- a/crates/weavepy-vm/src/object.rs +++ b/crates/weavepy-vm/src/object.rs @@ -1204,10 +1204,9 @@ pub fn materialize_stack_at(stack: &FrameStack, idx: usize) -> Option { - let py = shell.materialize(back); // Materialised while live on the stack: count the // activation so `frame.clear()` refuses it. - py + shell.materialize(back) } }; back = Some(py); @@ -3052,11 +3051,16 @@ impl indexmap::Equivalent for LeafNameProbe<'_> { /// A dict probe for the leaf burst: a `str` or `int` key that matches a /// stored key only by exact native equality. It never runs Python — an /// exotic stored key sharing the bucket (a `__eq__` that could equate) -/// reads as a miss, and the burst leaves every miss to the full path. -#[derive(Debug, Clone, Copy)] +/// reads as a miss, and the burst leaves every miss to the full path +/// unless [`LeafProbe::miss_is_exact`] proves no such key was seen. +#[derive(Debug)] pub struct LeafProbe<'a> { pub key: &'a Object, pub hash: i64, + /// Set when the table compared this probe with a stored key of another + /// kind, whose equality to the probe might need Python (`1.0`, `True`, + /// a `str` subclass, an instance with `__eq__`). + foreign: std::cell::Cell, } impl<'a> LeafProbe<'a> { @@ -3068,7 +3072,21 @@ impl<'a> LeafProbe<'a> { Object::Int(_) => py_hash_value(key)?, _ => return None, }; - Some(Self { key, hash }) + Some(Self { + key, + hash, + foreign: std::cell::Cell::new(false), + }) + } + + /// After a lookup that found nothing: whether no stored key can equal + /// this one, so inserting it natively is exact. Any key that Python + /// would compare with the probe has an equal hash, so the table + /// compared it with the probe, and a key of another kind set + /// `foreign`. + #[inline] + pub fn miss_is_exact(&self) -> bool { + !self.foreign.get() } } @@ -3085,7 +3103,10 @@ impl indexmap::Equivalent for LeafProbe<'_> { match (self.key, &key.0) { (Object::Str(a), Object::Str(b)) => a.as_bytes() == b.as_bytes(), (Object::Int(a), Object::Int(b)) => a == b, - _ => false, + _ => { + self.foreign.set(true); + false + } } } } @@ -3152,74 +3173,37 @@ fn current_interp_eq(a: &Object, b: &Object) -> Option { /// `name` dunder (a real Python `def`, not the inherited identity default). /// Used to gate the reentrant `__eq__` dispatch so plain instances keep the /// native identity fast path. -pub(crate) fn instance_has_custom_dunder(obj: &Object, name: &str) -> bool { +pub(crate) fn instance_has_custom_dunder(obj: &Object, dunder: crate::types::Dunder) -> bool { let Object::Instance(inst) = obj else { return false; }; - match inst.cls().lookup_with_owner(name) { - Some((Object::Function(_) | Object::BoundMethod(_), _)) => true, - Some((Object::None, _)) | None => false, - // A non-function dunder. A user class supplying it (e.g. - // `unittest.mock` installs `Mock` instances as `__hash__` / - // `__eq__` on per-instance subclasses) needs Python dispatch. - Some((_, owner)) if !owner.flags.is_builtin => true, - // A built-in type that *overrides* the dunder in its own dict - // rather than inheriting `object`'s identity default needs Python - // dispatch too: `weakref` compares/hashes by referent, so its - // `ref` objects can't be keyed by the native identity path - // (`test_weakref`/`test_weakset` rely on `ref(a) == ref(a)` and - // matching hashes finding the same set slot). `object`'s own - // `__eq__`/`__hash__` — where every *plain* instance resolves — - // stay native so ordinary objects keep identity semantics, and - // value-wrapping subclasses (`class C(int)`, struct sequences) - // keep their native structural comparison via `native`. - Some((_, owner)) => { - inst.native.get().is_none() - && !Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) - } - } -} - -/// [`instance_has_custom_dunder`]`(obj, "__eq__")`, memoised per class: -/// membership tests and `list.remove`/`index`/`count` ask it for every -/// element they compare. -pub(crate) fn instance_has_custom_eq(obj: &Object) -> bool { - let Object::Instance(inst) = obj else { + let info = inst.class_dunder(dunder); + if !info.present() || info.is_none() { return false; - }; - if crate::gil::free_threading_enabled() { - return custom_eq_of(&inst.cls(), inst); } - // Under the GIL nothing reassigns `__class__` while this reads it - // (the lookup below runs no Python code). - custom_eq_of(inst.cls_raw(), inst) -} - -fn custom_eq_of(cls: &crate::types::TypeObject, inst: &crate::types::PyInstance) -> bool { - let ver = cls.attr_version.get(); - let memo = cls.eq_kind.get(); - let kind = if memo != 0 && memo >> 2 == ver { - memo & 3 - } else { - let kind = match cls.lookup_with_owner("__eq__") { - Some((Object::Function(_) | Object::BoundMethod(_), _)) => 2, - Some((Object::None, _)) | None => 1, - Some((_, owner)) if !owner.flags.is_builtin => 2, - Some((_, owner)) - if Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) => - { - 1 - } - Some(_) => 3, - }; - cls.eq_kind.set(ver << 2 | kind); - kind - }; - match kind { - 1 => false, - 2 => true, - _ => inst.native.get().is_none(), + // A Python function, or any other object a user class supplies (e.g. + // `unittest.mock` installs `Mock` instances as `__hash__` / `__eq__` + // on per-instance subclasses), needs Python dispatch. + if info.is_function() || !info.builtin_owner() { + return true; } + // A built-in type that *overrides* the dunder in its own dict rather + // than inheriting `object`'s identity default needs Python dispatch + // too: `weakref` compares/hashes by referent, so its `ref` objects + // can't be keyed by the native identity path (`test_weakref`/ + // `test_weakset` rely on `ref(a) == ref(a)` and matching hashes + // finding the same set slot). `object`'s own `__eq__`/`__hash__` — + // where every *plain* instance resolves — stay native so ordinary + // objects keep identity semantics, and value-wrapping subclasses + // (`class C(int)`, struct sequences) keep their native structural + // comparison via `native`. + inst.native.get().is_none() && !info.object_owner() +} + +/// [`instance_has_custom_dunder`] for `__eq__`: membership tests and +/// `list.remove`/`index`/`count` ask it for every element they compare. +pub(crate) fn instance_has_custom_eq(obj: &Object) -> bool { + instance_has_custom_dunder(obj, crate::types::Dunder::Eq) } /// `a == b` can only be `object`'s identity default: each side is a @@ -3275,7 +3259,7 @@ pub(crate) fn key_needs_interp_eq(obj: &Object) -> bool { // `_UnionGenericAlias` — structural namespace equality can't // express either (RFC 0076 WS5). Object::SimpleNamespace(_) => crate::is_pep604_union(obj).is_some(), - _ => instance_has_custom_dunder(obj, "__eq__"), + _ => instance_has_custom_dunder(obj, crate::types::Dunder::Eq), } } @@ -3604,11 +3588,8 @@ fn key_eq_defer_active() -> bool { pub(crate) fn dict_key_is_reentrant(key: &Object) -> bool { match key { Object::Instance(inst) => { - let py_dunder = |name: &str| match inst.cls().lookup_with_owner(name) { - Some((Object::None, _)) | None => false, - Some((_, owner)) => !owner.flags.is_builtin, - }; - py_dunder("__eq__") || py_dunder("__hash__") + inst.class_dunder(crate::types::Dunder::Eq).user_defined() + || inst.class_dunder(crate::types::Dunder::Hash).user_defined() } Object::Tuple(items) => items.iter().any(dict_key_is_reentrant), _ => false, @@ -4055,9 +4036,37 @@ pub type DictMap = indexmap::IndexMap u64 { + GLOBAL_VALUE_EPOCH.load(std::sync::atomic::Ordering::Relaxed) +} + +/// Advance [`GLOBAL_VALUE_EPOCH`]. +#[inline] +pub fn bump_global_value_epoch() { + GLOBAL_VALUE_EPOCH.fetch_add(1, std::sync::atomic::Ordering::Relaxed); +} + #[inline] fn next_dict_stamp() -> u64 { - DICT_STAMP.fetch_add(1, std::sync::atomic::Ordering::Relaxed) + use std::sync::atomic::Ordering::Relaxed; + // Dict mutation is serialized by the GIL, so a plain load and store + // hands out unique stamps without a locked read-modify-write (drawn on + // every mutable dict access). Free-threaded mode needs the real one, + // and so does the debug unit-test binary, whose tests run separate + // interpreters on concurrent threads with no GIL between them. + if cfg!(debug_assertions) || crate::gil::free_threading_enabled() { + return DICT_STAMP.fetch_add(1, Relaxed); + } + let v = DICT_STAMP.load(Relaxed); + DICT_STAMP.store(v + 1, Relaxed); + v } /// A [`DictMap`] with a mutation stamp: every mutable access (any @@ -4077,6 +4086,10 @@ pub struct DictData { /// value (every `DerefMut`) tracks the owner first; the owner clears /// it when it is tracked by other means or dies. deferred_owner: std::sync::atomic::AtomicUsize, + /// A one-bit-per-hash-class summary of the `str` keys (see + /// [`Self::may_hold_str_hash`]); `0` until built, and reset by every + /// access that stamps. + key_filter: std::sync::atomic::AtomicU64, } impl Clone for DictData { @@ -4085,6 +4098,7 @@ impl Clone for DictData { map: self.map.clone(), stamp: self.stamp, deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } @@ -4096,6 +4110,7 @@ impl DictData { map: DictMap::with_capacity_and_hasher(n, h), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4106,6 +4121,7 @@ impl DictData { map: DictMap::default(), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(owner), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4118,6 +4134,7 @@ impl DictData { map: DictMap::with_capacity_and_hasher(n, crate::fasthash::FxBuildHasher), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(owner), + key_filter: std::sync::atomic::AtomicU64::new(0), } } @@ -4151,6 +4168,7 @@ impl DictData { #[inline] pub fn map_mut_atomic_store(&mut self) -> &mut DictMap { self.stamp = next_dict_stamp(); + *self.key_filter.get_mut() = 0; &mut self.map } @@ -4167,6 +4185,39 @@ impl DictData { self.stamp } + /// Whether a `str` key with Python hash `hash` may be present: `false` + /// proves it absent. The summary is built once per key layout (value + /// stores in place leave it standing, so an instance dict's summary + /// survives its attribute updates). Bit 63 marks a built summary, so + /// the hash class it shares always answers "maybe"; a non-`str` key + /// makes every hash possible. + #[inline] + pub fn may_hold_str_hash(&self, hash: i64) -> bool { + let mut bits = self.key_filter.load(std::sync::atomic::Ordering::Relaxed); + if bits == 0 { + bits = self.build_key_filter(); + } + bits & (1u64 << (hash as u64 & 63)) != 0 + } + + #[cold] + #[inline(never)] + fn build_key_filter(&self) -> u64 { + let mut bits = 1u64 << 63; + for k in self.map.keys() { + match &k.0 { + Object::Str(s) => bits |= 1u64 << (SharedStr::hash_cached(s) as u64 & 63), + _ => { + bits = u64::MAX; + break; + } + } + } + self.key_filter + .store(bits, std::sync::atomic::Ordering::Relaxed); + bits + } + /// Mutable access for replacing an existing key's value in place: the /// keys do not move, and only key layout is what the stamp's readers /// (the global/builtin load caches) watch, so it stays put. A @@ -4196,6 +4247,7 @@ impl Default for DictData { map: DictMap::default(), stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } @@ -4216,6 +4268,7 @@ impl std::ops::DerefMut for DictData { self.track_deferred_owner(); } self.stamp = next_dict_stamp(); + *self.key_filter.get_mut() = 0; &mut self.map } } @@ -4232,6 +4285,7 @@ impl From for DictData { map, stamp: next_dict_stamp(), deferred_owner: std::sync::atomic::AtomicUsize::new(0), + key_filter: std::sync::atomic::AtomicU64::new(0), } } } @@ -4302,7 +4356,8 @@ pub struct PyFunction { /// CPython's `func_set_dict` *aliases* the assigned dict /// (`f.__dict__ = d; f.__dict__ is d` — test_funcattrs), so the /// whole payload must be swappable, not just its contents. - pub attrs: RefCell>>, + /// Allocated on first use: most functions never get a `__dict__`. + pub attrs: RefCell>>>, /// CPython function *getset/member slots* (`__name__`, /// `__qualname__`, `__doc__`, `__module__`, `__annotations__`, /// `__type_params__`, …). These live outside `__dict__`: they're @@ -4421,7 +4476,22 @@ impl PyFunction { /// The live `__dict__` payload (honours `f.__dict__ = d` swapping). pub fn attrs(&self) -> Rc> { - self.attrs.borrow().clone() + if let Some(d) = self.attrs.borrow().as_ref() { + return d.clone(); + } + let d = Rc::new(RefCell::new(DictData::default())); + *self.attrs.borrow_mut() = Some(d.clone()); + d + } + + /// A `__dict__` entry, without allocating an empty dictionary. + pub fn attr_get(&self, name: &str) -> Option { + self.attrs + .borrow() + .as_ref()? + .borrow() + .get(&StrKey(name)) + .cloned() } /// Read a slot value if one has been stored (explicitly assigned or @@ -4672,7 +4742,7 @@ impl PyGenerator { qualname: impl Into, kind: CoroutineKind, code: Object, - frame: Box, + frame: Box, ) -> Self { Self { name: RefCell::new(Object::from_str(name.into())), @@ -4694,7 +4764,7 @@ impl PyGenerator { qualname: Object, kind: CoroutineKind, code: Object, - frame: Box, + frame: Box, ) -> Self { Self { name: RefCell::new(name), @@ -4812,9 +4882,9 @@ impl CoroutineKind { pub enum GeneratorState { /// Created but not yet started — body hasn't executed past the /// initial `RETURN_GENERATOR`. - Created(Box), + Created(Box), /// Paused at a `YIELD_VALUE`. - Suspended(Box), + Suspended(Box), /// Body returned (cleanly or via exception). Subsequent /// `next`/`send` raise `StopIteration`. Finished, @@ -11906,7 +11976,7 @@ pub(crate) fn py_hash_value(obj: &Object) -> Option { // A user-defined `__hash__` outranks the wrapped value's hash — // e.g. functools' `_HashedSeq(list)` caches its hash precisely so // the (unhashable) list payload is never consulted. - if instance_has_custom_dunder(obj, "__hash__") { + if instance_has_custom_dunder(obj, crate::types::Dunder::Hash) { // When the instance wraps an *immutable* builtin value (an // `int`/`str`/`tuple`/… subclass), the value can't change, so // a custom `__hash__` is genuinely constant and may be diff --git a/crates/weavepy-vm/src/pycache.rs b/crates/weavepy-vm/src/pycache.rs index 0186b05e..93ba292d 100644 --- a/crates/weavepy-vm/src/pycache.rs +++ b/crates/weavepy-vm/src/pycache.rs @@ -408,6 +408,7 @@ fn mtime_seconds(meta: &fs::Metadata) -> u32 { #[cfg(test)] mod ownership_tests { use super::*; + use std::sync::Arc; use weavepy_compiler::{compile_module, Constant}; #[test] @@ -433,7 +434,7 @@ mod ownership_tests { fn collect(code: &CodeObject, rows: &mut Vec<(*const CodeObject, String)>) { for c in &code.constants { if let Constant::Code(inner) = c { - rows.push((Rc::as_ptr(inner), inner.filename.clone())); + rows.push((Arc::as_ptr(inner), inner.filename.clone())); collect(inner, rows); } } @@ -455,17 +456,17 @@ mod ownership_tests { #[test] fn relocation_preserves_shared_code_and_nested_tuple_constants() { - let child = Rc::new(CodeObject { + let child = Arc::new(CodeObject { name: "child".to_owned(), filename: "original.py".to_owned(), ..CodeObject::default() }); - let root = Rc::new(CodeObject { + let root = Arc::new(CodeObject { filename: "original.py".to_owned(), constants: vec![Constant::Tuple(vec![Constant::Code(child.clone())])], ..CodeObject::default() }); - let mut moved = own_decoded_code(root.clone()); + let mut moved = own_decoded_code(Rc::from_arc(root.clone())); rewrite_filenames(&mut moved, "relocated.py"); assert_eq!(root.filename, "original.py"); assert_eq!(child.filename, "original.py"); @@ -476,6 +477,6 @@ mod ownership_tests { panic!("expected code"); }; assert_eq!(relocated.filename, "relocated.py"); - assert!(!Rc::ptr_eq(&child, relocated)); + assert!(!Arc::ptr_eq(&child, relocated)); } } diff --git a/crates/weavepy-vm/src/rc.rs b/crates/weavepy-vm/src/rc.rs new file mode 100644 index 00000000..34308d73 --- /dev/null +++ b/crates/weavepy-vm/src/rc.rs @@ -0,0 +1,475 @@ +//! Biased reference counting for the VM heap. +//! +//! [`Rc`] and [`Weak`] wrap [`std::sync::Arc`] and [`std::sync::Weak`] +//! with the same API, but a strong-count increment or decrement is a plain +//! load and store instead of a locked read-modify-write while one thread +//! owns every object. That's the common case: a program that never starts a +//! second thread, running under the GIL. A `lock xadd` pair costs about four +//! times a plain pair on x86, and the interpreter performs roughly one pair +//! per bytecode, so atomics were a leading cost of every workload. +//! +//! The bias is the one [`crate::sync`] already maintains for cell borrows: +//! it holds until a second thread registers to run VM code or free-threading +//! starts, and it's never restored. The registering thread revokes the bias +//! while holding the GIL; the previous owner released the GIL before that, +//! and reacquires it before touching another object, so every plain update +//! happens before the revocation and every later update is atomic. Threads +//! that touch objects without the GIL announce themselves with +//! [`crate::sync::mark_cells_shared`] before they begin, and code that +//! spawns a thread which will run VM code revokes the bias first (see +//! [`revoke_refcount_bias`]). +//! +//! Deallocation always goes through `Arc`'s own drop, so the payload, +//! weak references, and the allocation are released exactly as before. +//! Only the counter arithmetic changes, and only for owners other than the +//! last one. + +use std::borrow::Borrow; +use std::fmt; +use std::hash::{Hash, Hasher}; +use std::mem::ManuallyDrop; +use std::ops::Deref; +use std::ptr; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; + +/// Whether strong counts may be updated with plain loads and stores. +/// +/// Debug builds always use atomics: the unit-test binary runs many +/// interpreters on concurrent threads with no GIL between them, and they +/// share static objects. +#[inline(always)] +pub(crate) fn refcounts_biased() -> bool { + #[cfg(debug_assertions)] + { + false + } + #[cfg(not(debug_assertions))] + { + crate::sync::bias_held() + } +} + +/// Revoke the single-thread bias before starting a thread that may touch VM +/// objects before it acquires the GIL. The caller's later updates observe +/// the revocation in program order, and the new thread observes it through +/// the spawn. +pub fn revoke_refcount_bias() { + crate::sync::revoke_bias_for_spawn(); +} + +/// The strong counter of the `Arc` allocation holding `value`. +/// +/// `ArcInner` is `#[repr(C)]` with the strong count first, then the weak +/// count, then the payload at the next offset aligned for the payload. The +/// standard library relies on this layout for `Arc::from_raw`; the +/// `strong_word_matches_arc_layout` test pins it for every payload shape +/// the VM uses. +#[inline(always)] +fn strong_word(value: *const T) -> *const AtomicUsize { + let align = std::mem::align_of_val(unsafe { &*value }); + let header = 2 * std::mem::size_of::(); + let offset = (header + align - 1) & !(align - 1); + // SAFETY: the payload lives `offset` bytes into its ArcInner, whose + // first field is the strong count, aligned for `AtomicUsize`. + #[allow(clippy::cast_ptr_alignment)] + unsafe { + value.cast::().sub(offset).cast::() + } +} + +/// Add one strong owner of the `Arc` payload at `value`. +/// +/// # Safety +/// +/// `value` must point at the payload of a live `Arc` allocation. +#[inline(always)] +pub(crate) unsafe fn increment_strong(value: *const T) { + if refcounts_biased() { + // SAFETY: live allocation, per the caller. + let word = unsafe { &*strong_word(value) }; + word.store(word.load(Ordering::Relaxed) + 1, Ordering::Relaxed); + } else { + // SAFETY: as above. + unsafe { Arc::increment_strong_count(value) }; + } +} + +/// Release one strong owner of the `Arc` payload at `value` without +/// destroying it. Returns `false`, having changed nothing, when this is the +/// last owner (or the bias is off); the caller must then drop an `Arc` +/// normally. +/// +/// # Safety +/// +/// `value` must point at the payload of a live `Arc` allocation, and the +/// caller must own one of its strong references. +#[inline(always)] +pub(crate) unsafe fn try_release_shared(value: *const T) -> bool { + if !refcounts_biased() { + return false; + } + // SAFETY: live allocation, per the caller. + let word = unsafe { &*strong_word(value) }; + let n = word.load(Ordering::Relaxed); + if n == 1 { + return false; + } + word.store(n - 1, Ordering::Relaxed); + true +} + +/// Convert an [`Rc`] to one of an unsized type, such as a trait object: +/// `rc_unsize!(rc => dyn Trait)`. Stable Rust only coerces its own smart +/// pointers, so the conversion passes through `Arc`. +#[macro_export] +macro_rules! rc_unsize { + ($rc:expr => $ty:ty) => { + $crate::sync::Rc::<$ty>::from_arc($crate::sync::Rc::into_arc($rc) as ::std::sync::Arc<$ty>) + }; +} + +/// A reference-counted shared pointer with biased counting. See the module +/// documentation. +#[repr(transparent)] +pub struct Rc(ManuallyDrop>); + +/// A weak reference to an [`Rc`] allocation. +#[repr(transparent)] +pub struct Weak(std::sync::Weak); + +impl Rc { + #[inline] + pub fn new(value: T) -> Self { + Self(ManuallyDrop::new(Arc::new(value))) + } + + pub fn try_unwrap(this: Self) -> Result { + Arc::try_unwrap(Self::into_arc(this)).map_err(Self::from_arc) + } + + pub fn into_inner(this: Self) -> Option { + Arc::into_inner(Self::into_arc(this)) + } + + pub fn unwrap_or_clone(this: Self) -> T + where + T: Clone, + { + Arc::unwrap_or_clone(Self::into_arc(this)) + } + + pub fn new_cyclic(data_fn: impl FnOnce(&Weak) -> T) -> Self { + Self::from_arc(Arc::new_cyclic(|weak| { + // SAFETY: Weak is a transparent wrapper over std::sync::Weak. + data_fn(unsafe { &*ptr::from_ref(weak).cast::>() }) + })) + } +} + +impl Rc { + #[inline] + pub fn from_arc(arc: Arc) -> Self { + Self(ManuallyDrop::new(arc)) + } + + #[inline] + pub fn into_arc(this: Self) -> Arc { + let this = ManuallyDrop::new(this); + // SAFETY: moves the owned Arc out; `this` is never dropped. + unsafe { ptr::read(&raw const *this.0) } + } + + #[inline] + pub fn as_arc(this: &Self) -> &Arc { + &this.0 + } + + #[inline] + pub fn as_ptr(this: &Self) -> *const T { + Arc::as_ptr(&this.0) + } + + #[inline] + pub fn ptr_eq(this: &Self, other: &Self) -> bool { + Arc::ptr_eq(&this.0, &other.0) + } + + #[inline] + pub fn strong_count(this: &Self) -> usize { + Arc::strong_count(&this.0) + } + + #[inline] + pub fn weak_count(this: &Self) -> usize { + Arc::weak_count(&this.0) + } + + #[inline] + pub fn downgrade(this: &Self) -> Weak { + Weak(Arc::downgrade(&this.0)) + } + + #[inline] + pub fn get_mut(this: &mut Self) -> Option<&mut T> { + Arc::get_mut(&mut this.0) + } + + #[inline] + pub fn into_raw(this: Self) -> *const T { + Arc::into_raw(Self::into_arc(this)) + } + + /// # Safety + /// + /// As [`Arc::from_raw`]. + #[inline] + pub unsafe fn from_raw(ptr: *const T) -> Self { + // SAFETY: forwarded to the caller. + Self::from_arc(unsafe { Arc::from_raw(ptr) }) + } + + /// # Safety + /// + /// As [`Arc::increment_strong_count`]. + #[inline] + pub unsafe fn increment_strong_count(ptr: *const T) { + // SAFETY: forwarded to the caller. + unsafe { increment_strong(ptr) } + } + + /// # Safety + /// + /// As [`Arc::decrement_strong_count`]. + #[inline] + pub unsafe fn decrement_strong_count(ptr: *const T) { + // SAFETY: forwarded to the caller. + unsafe { drop(Self::from_raw(ptr)) } + } +} + +impl Rc { + #[inline] + pub fn make_mut(this: &mut Self) -> &mut T { + Arc::make_mut(&mut this.0) + } +} + +impl Clone for Rc { + #[inline(always)] + fn clone(&self) -> Self { + // SAFETY: `self` keeps the allocation alive, and the new owner + // accounts for the added reference. + unsafe { + increment_strong(Arc::as_ptr(&self.0)); + Self(ManuallyDrop::new(ptr::read(&raw const *self.0))) + } + } +} + +impl Drop for Rc { + #[inline(always)] + fn drop(&mut self) { + // SAFETY: this owner holds one strong reference, released exactly + // once: either here, or by Arc's own drop. + unsafe { + if !try_release_shared(Arc::as_ptr(&self.0)) { + ManuallyDrop::drop(&mut self.0); + } + } + } +} + +impl Deref for Rc { + type Target = T; + #[inline(always)] + fn deref(&self) -> &T { + &self.0 + } +} + +impl AsRef for Rc { + fn as_ref(&self) -> &T { + self + } +} + +impl Borrow for Rc { + fn borrow(&self) -> &T { + self + } +} + +impl fmt::Debug for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Debug::fmt(&**self, f) + } +} + +impl fmt::Display for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(&**self, f) + } +} + +impl fmt::Pointer for Rc { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Pointer::fmt(&Self::as_ptr(self), f) + } +} + +impl PartialEq for Rc { + fn eq(&self, other: &Self) -> bool { + **self == **other + } +} + +impl Eq for Rc {} + +impl PartialOrd for Rc { + fn partial_cmp(&self, other: &Self) -> Option { + (**self).partial_cmp(&**other) + } +} + +impl Ord for Rc { + fn cmp(&self, other: &Self) -> std::cmp::Ordering { + (**self).cmp(&**other) + } +} + +impl Hash for Rc { + fn hash(&self, state: &mut H) { + (**self).hash(state) + } +} + +impl Default for Rc { + fn default() -> Self { + Self::new(T::default()) + } +} + +impl From for Rc { + fn from(value: T) -> Self { + Self::new(value) + } +} + +impl From> for Rc { + fn from(value: Box) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From> for Rc<[T]> { + fn from(value: Vec) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From<&str> for Rc { + fn from(value: &str) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From for Rc { + fn from(value: String) -> Self { + Self::from_arc(Arc::from(value)) + } +} + +impl From> for Rc { + fn from(value: Arc) -> Self { + Self::from_arc(value) + } +} + +impl Weak { + pub const fn new() -> Self { + Self(std::sync::Weak::new()) + } +} + +impl Weak { + #[inline] + pub fn upgrade(&self) -> Option> { + self.0.upgrade().map(Rc::from_arc) + } + + pub fn strong_count(&self) -> usize { + self.0.strong_count() + } + + pub fn weak_count(&self) -> usize { + self.0.weak_count() + } + + pub fn ptr_eq(&self, other: &Self) -> bool { + self.0.ptr_eq(&other.0) + } + + pub fn as_ptr(&self) -> *const T { + self.0.as_ptr() + } +} + +impl Clone for Weak { + fn clone(&self) -> Self { + Self(self.0.clone()) + } +} + +impl Default for Weak { + fn default() -> Self { + Self::new() + } +} + +impl fmt::Debug for Weak { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str("(Weak)") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn check(arc: &Arc) { + let before = Arc::strong_count(arc); + // SAFETY: `arc` is live. + let word = unsafe { &*strong_word(Arc::as_ptr(arc)) }; + assert_eq!(word.load(Ordering::Relaxed), before); + let extra = arc.clone(); + assert_eq!(word.load(Ordering::Relaxed), before + 1); + drop(extra); + } + + #[test] + fn strong_word_matches_arc_layout() { + #[repr(align(64))] + struct Wide(#[allow(dead_code)] u8); + check(&Arc::new(1u8)); + check(&Arc::new(7u64)); + check(&Arc::new(Wide(3))); + check(&Arc::new(String::from("x"))); + check::(&Arc::from("text")); + check::<[u64]>(&Arc::from(vec![1u64, 2, 3])); + check::(&(Arc::new(5u32) as Arc<_>)); + } + + #[test] + fn biased_updates_keep_arc_counts() { + let rc = Rc::new(vec![1, 2, 3]); + let copies: Vec<_> = (0..10).map(|_| rc.clone()).collect(); + assert_eq!(Rc::strong_count(&rc), 11); + drop(copies); + assert_eq!(Rc::strong_count(&rc), 1); + let weak = Rc::downgrade(&rc); + assert!(weak.upgrade().is_some()); + drop(rc); + assert!(weak.upgrade().is_none()); + } +} diff --git a/crates/weavepy-vm/src/shared_value.rs b/crates/weavepy-vm/src/shared_value.rs index 7a8d88bd..78b01224 100644 --- a/crates/weavepy-vm/src/shared_value.rs +++ b/crates/weavepy-vm/src/shared_value.rs @@ -113,7 +113,13 @@ impl ThinArc { impl Clone for ThinArc { #[inline] fn clone(&self) -> Self { - Self::from_arc(Arc::clone(&self.arc_view())) + // SAFETY: this owner keeps the payload alive; the copy accounts for + // the added strong reference. + unsafe { crate::rc::increment_strong(self.raw()) }; + Self { + data: self.data, + ownership: PhantomData, + } } } @@ -122,7 +128,11 @@ impl Drop for ThinArc { fn drop(&mut self) { // SAFETY: release this owner's one strong reference exactly once. // Arc drops the payload and handles remaining weak references normally. - unsafe { drop(Arc::from_raw(self.raw())) } + unsafe { + if !crate::rc::try_release_shared(self.raw()) { + drop(Arc::from_raw(self.raw())) + } + } } } @@ -663,6 +673,8 @@ mod tests { fn weak_upgrades_and_shared_text_survive_thread_handoffs() { let value = SharedStr::from("shared 🧶 text"); let weak = SharedStr::downgrade(&value); + // As the VM does before starting a thread: counts go atomic. + crate::sync::revoke_bias_for_spawn(); std::thread::scope(|scope| { for _ in 0..2 { let weak = weak.clone(); diff --git a/crates/weavepy-vm/src/specialize.rs b/crates/weavepy-vm/src/specialize.rs index bb161da9..de5bd4fd 100644 --- a/crates/weavepy-vm/src/specialize.rs +++ b/crates/weavepy-vm/src/specialize.rs @@ -171,10 +171,8 @@ pub fn attempt_specialize_load_attr(obj: &Object, name: &str) -> InlineCache { } // First check the instance dict — that's the // `LoadAttrInstance` shape. - if let Some(dict) = inst.dict.get() { - if let Some(idx) = dict.borrow().index_of_key_str(name) { - return InlineCache::LoadAttrInstance { key_idx: idx, ver }; - } + if let Some(idx) = inst.attr_position_str(name) { + return InlineCache::LoadAttrInstance { key_idx: idx, ver }; } // Not on the instance: resolve through the MRO. A plain // Python function anywhere on it is the *method* shape — @@ -322,10 +320,8 @@ pub fn attempt_specialize_store_attr(obj: &Object, name: &str) -> InlineCache { ) { return InlineCache::Cooldown(COOLDOWN); } - if let Some(dict) = inst.dict.get() { - if let Some(idx) = dict.borrow().index_of_key_str(name) { - return InlineCache::StoreAttrInstance { key_idx: idx, ver }; - } + if let Some(idx) = inst.attr_position_str(name) { + return InlineCache::StoreAttrInstance { key_idx: idx, ver }; } // Key not present: the constructor pattern (`self.x = …` on a // fresh instance). Specialize to a single-probe insert when diff --git a/crates/weavepy-vm/src/stdlib/abc_mod.rs b/crates/weavepy-vm/src/stdlib/abc_mod.rs index 42774788..5c38d77a 100644 --- a/crates/weavepy-vm/src/stdlib/abc_mod.rs +++ b/crates/weavepy-vm/src/stdlib/abc_mod.rs @@ -1,20 +1,106 @@ -//! The `_abc` accelerator module — RFC 0023. +//! The `_abc` accelerator module, a port of CPython's `Modules/_abc.c`. //! -//! Backs `abc.ABCMeta` with the registry of virtual subclasses and -//! the abstractmethod cache. Surface mirrors CPython's `_abc`: -//! `get_cache_token`, `_abc_init`, `_abc_register`, `_abc_instancecheck`, -//! `_abc_subclasscheck`, `_get_dump`, `_reset_registry`, -//! `_reset_caches`. +//! `abc.ABCMeta` delegates registration and the instance and subclass +//! checks here. Each ABC keeps its virtual-subclass registry, positive +//! cache, and negative cache on its [`TypeObject`]. Classes are held +//! weakly, so registering or checking a class never keeps it alive. A +//! foreign extension type has no weak form and is held strongly; such types +//! live for the whole process in practice. -use crate::sync::Rc; -use crate::sync::RefCell; +use crate::sync::{Cell, Rc, RefCell, Weak}; +use std::sync::atomic::{AtomicU64, Ordering}; -use crate::error::RuntimeError; +use crate::error::{assertion_error, runtime_error, type_error, RuntimeError}; use crate::import::ModuleCache; use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; +use crate::types::TypeObject; +use crate::weakref_registry::{id_of, ObjectId}; +use crate::Interpreter; -thread_local! { - static CACHE_TOKEN: RefCell = const { RefCell::new(1) }; +/// CPython's `abc_invalidation_counter`: bumped by every registration, so +/// each ABC's negative cache knows when it may be stale. +static INVALIDATION_COUNTER: AtomicU64 = AtomicU64::new(0); + +/// A class held by an ABC cache or registry. +enum Held { + Type(Weak), + Other(Object), +} + +impl Held { + fn new(class: &Object) -> Self { + match class { + Object::Type(t) => Self::Type(Rc::downgrade(t)), + other => Self::Other(other.clone()), + } + } + + fn get(&self) -> Option { + match self { + Self::Type(t) => t.upgrade().map(Object::Type), + Self::Other(o) => Some(o.clone()), + } + } +} + +/// A set of classes keyed by identity. A weak entry keeps its allocation, +/// so a dead class's identity is never reused while its entry remains. +#[derive(Default)] +struct ClassSet { + entries: indexmap::IndexMap, + /// Length at which dead entries are next pruned. + prune_at: usize, +} + +impl ClassSet { + fn contains(&self, class: &Object) -> bool { + self.entries.contains_key(&id_of(class)) + } + + fn insert(&mut self, class: &Object) { + self.entries + .entry(id_of(class)) + .or_insert_with(|| Held::new(class)); + if self.entries.len() >= self.prune_at.max(64) { + self.entries.retain(|_, held| held.get().is_some()); + self.prune_at = self.entries.len() * 2; + } + } + + fn clear(&mut self) { + self.entries.clear(); + } + + fn live(&self) -> Vec { + self.entries.values().filter_map(Held::get).collect() + } +} + +/// CPython's `_abc_data`: one ABC's registry and caches. +pub struct AbcState { + registry: RefCell, + cache: RefCell, + negative_cache: RefCell, + negative_cache_version: Cell, +} + +impl std::fmt::Debug for AbcState { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("AbcState") + .field("registered", &self.registry.borrow().entries.len()) + .finish_non_exhaustive() + } +} + +impl AbcState { + fn new() -> Self { + Self { + registry: RefCell::default(), + cache: RefCell::default(), + negative_cache: RefCell::default(), + negative_cache_version: Cell::new(INVALIDATION_COUNTER.load(Ordering::Relaxed)), + } + } } pub fn build(_cache: &ModuleCache) -> Rc { @@ -56,181 +142,379 @@ pub fn build(_cache: &ModuleCache) -> Rc { }) } -fn bump_cache() { - CACHE_TOKEN.with(|c| { - let mut g = c.borrow_mut(); - *g = g.wrapping_add(1); - }); +fn interpreter() -> Result<&'static mut Interpreter, RuntimeError> { + let ptr = crate::vm_singletons::current_interpreter_ptr() + .ok_or_else(|| runtime_error("_abc requires a running interpreter"))?; + // SAFETY: published by the enclosing VM frame on this thread, which + // outlives this builtin call. + Ok(unsafe { &mut *ptr }) } -fn abc_get_cache_token(_args: &[Object]) -> Result { - Ok(Object::Int(CACHE_TOKEN.with(|c| *c.borrow() as i64))) +fn args_exact(args: &[Object], name: &str) -> Result<[Object; N], RuntimeError> { + <&[Object; N]>::try_from(args).cloned().map_err(|_| { + type_error(format!( + "{name} expected {N} argument{}, got {}", + if N == 1 { "" } else { "s" }, + args.len() + )) + }) } -fn abc_init(args: &[Object]) -> Result { - // _abc_init(cls) — initialise the registry / cache on the class. - if let Some(Object::Type(cls)) = args.first() { - // CPython's `_abc_init` consumes `__abc_tpflags__`: it validates - // that a class doesn't claim both `Py_TPFLAGS_SEQUENCE` and - // `Py_TPFLAGS_MAPPING`, folds the collection bits into - // `tp_flags` (we keep them under a private dict key that - // `flags_bits` reads), and deletes the public attribute. - const COLLECTION_FLAGS: i64 = (1 << 5) | (1 << 6); - let tpflags = cls - .dict +/// The ABC state of `cls`, inherited through the MRO the way CPython's +/// `_abc_impl` attribute is. +fn state_of(cls: &Object) -> Result, RuntimeError> { + if let Object::Type(t) = cls { + if t.abc_state.get().is_some() { + return Ok(t.clone()); + } + if let Some(owner) = t + .mro .borrow() - .get(&DictKey(Object::from_static("__abc_tpflags__"))) - .cloned(); - if let Some(flags) = tpflags { - if let Some(val) = flags.as_i64() { - if (val & COLLECTION_FLAGS) == COLLECTION_FLAGS { - return Err(crate::error::type_error( - "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING", - )); - } - cls.dict.borrow_mut().insert( - DictKey(Object::from_static("_abc_collection_flags")), - Object::Int(val & COLLECTION_FLAGS), - ); + .iter() + .find(|base| base.abc_state.get().is_some()) + { + return Ok(owner.clone()); + } + } + Err(type_error("_abc_impl is set to a wrong type")) +} + +fn state(owner: &TypeObject) -> &AbcState { + owner.abc_state.get().expect("checked by state_of") +} + +fn is_class(interp: &mut Interpreter, obj: &Object) -> Result { + match obj { + Object::Type(_) => Ok(true), + Object::Foreign(_) => { + let type_type = Object::Type(crate::builtin_types::builtin_types().type_.clone()); + interp.isinstance_public(obj, &type_type) + } + _ => Ok(false), + } +} + +fn call_method( + interp: &mut Interpreter, + receiver: &Object, + name: &str, + args: &[Object], +) -> Result { + let method = interp.load_attr_public(receiver, name)?; + let globals = interp.builtins_dict(); + interp.call(&method, args, &[], &globals) +} + +/// `getattr(obj, name, None)`, propagating everything but `AttributeError`. +fn attr_or_none( + interp: &mut Interpreter, + obj: &Object, + name: &str, +) -> Result, RuntimeError> { + match interp.load_attr_public(obj, name) { + Ok(v) => Ok(Some(v)), + Err(e) if interp.is_attribute_error(&e) => Ok(None), + Err(e) => Err(e), + } +} + +/// CPython `_PyObject_IsAbstract`. +fn is_abstract(interp: &mut Interpreter, obj: &Object) -> Result { + let flag = match obj { + // A plain function carries the flag only in its `__dict__`, and + // built-in data values never do: answer both without an + // attribute lookup that would build an `AttributeError`. + Object::Function(f) => f.attr_get("__isabstractmethod__"), + Object::None + | Object::Bool(_) + | Object::Int(_) + | Object::Long(_) + | Object::Float(_) + | Object::Str(_) + | Object::Bytes(_) + | Object::Tuple(_) => None, + _ => attr_or_none(interp, obj, "__isabstractmethod__")?, + }; + match flag { + Some(flag) => interp.op_truth(&flag), + None => Ok(false), + } +} + +fn abc_get_cache_token(args: &[Object]) -> Result { + args_exact::<0>(args, "get_cache_token")?; + Ok(Object::Int( + INVALIDATION_COUNTER.load(Ordering::Relaxed) as i64 + )) +} + +/// `_abc_init(cls)`: compute `__abstractmethods__`, fold +/// `__abc_tpflags__`, and give the class fresh caches. +fn abc_init(args: &[Object]) -> Result { + let [cls] = args_exact::<1>(args, "_abc_init")?; + let interp = interpreter()?; + compute_abstract_methods(interp, &cls)?; + let Object::Type(t) = &cls else { + return Ok(Object::None); + }; + // CPython's `_abc_init` consumes `__abc_tpflags__`: a class may not + // claim both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING. The collection + // bits are kept under a private key that `flags_bits` and pattern + // matching read. + const COLLECTION_FLAGS: i64 = (1 << 5) | (1 << 6); + let key = DictKey(Object::from_static("__abc_tpflags__")); + let tpflags = t.dict.borrow().get(&key).cloned(); + if let Some(flags) = tpflags { + if let Some(val) = flags.as_i64() { + if (val & COLLECTION_FLAGS) == COLLECTION_FLAGS { + return Err(type_error( + "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING", + )); } - cls.dict - .borrow_mut() - .shift_remove(&DictKey(Object::from_static("__abc_tpflags__"))); - } - let mut td = cls.dict.borrow_mut(); - td.insert( - DictKey(Object::from_static("_abc_registry")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_cache")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_negative_cache")), - Object::new_set(), - ); - td.insert( - DictKey(Object::from_static("_abc_negative_cache_version")), - Object::Int(CACHE_TOKEN.with(|c| *c.borrow() as i64)), - ); + t.dict.borrow_mut().insert( + DictKey(Object::from_static("_abc_collection_flags")), + Object::Int(val & COLLECTION_FLAGS), + ); + } + t.dict.borrow_mut().shift_remove(&key); + } + match t.abc_state.get() { + Some(existing) => { + existing.registry.borrow_mut().clear(); + existing.cache.borrow_mut().clear(); + existing.negative_cache.borrow_mut().clear(); + existing + .negative_cache_version + .set(INVALIDATION_COUNTER.load(Ordering::Relaxed)); + } + None => { + let _ = t.abc_state.set(Box::new(AbcState::new())); + } } Ok(Object::None) } -fn abc_register(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let Object::Type(t) = &cls { - if let Some(Object::Set(reg)) = t +/// CPython `compute_abstract_methods`. +fn compute_abstract_methods(interp: &mut Interpreter, cls: &Object) -> Result<(), RuntimeError> { + let mut abstracts: Vec = Vec::new(); + // Stage 1: the class's own abstract methods. + let own: Vec<(Object, Object)> = match cls { + Object::Type(t) => t .dict .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - { - reg.borrow_mut().insert(DictKey(sub.clone())); + .iter() + .map(|(k, v)| (k.0.clone(), v.clone())) + .collect(), + _ => Vec::new(), + }; + for (name, value) in own { + if is_abstract(interp, &value)? { + abstracts.push(name); + } + } + // Stage 2: inherited abstract methods the class didn't override. + let bases = interp.load_attr_public(cls, "__bases__")?; + let Object::Tuple(bases) = bases else { + return Err(type_error("__bases__ is not tuple")); + }; + let globals = interp.builtins_dict(); + for base in bases.iter() { + let Some(inherited) = attr_or_none(interp, base, "__abstractmethods__")? else { + continue; + }; + for name in interp.collect_iterable(&inherited, &globals)? { + let Object::Str(s) = &name else { + continue; + }; + let Some(value) = attr_or_none(interp, cls, s)? else { + continue; + }; + if is_abstract(interp, &value)? && !abstracts.iter().any(|a| a.is_same(&name)) { + abstracts.push(name); + } } } - // CPython's `_abc_register` copies the ABC's collection flag - // (Py_TPFLAGS_SEQUENCE / Py_TPFLAGS_MAPPING) onto the registered - // class so late registration makes match patterns work (PEP 634). - if let (Some(Object::Type(t)), Object::Type(s)) = (args.first(), &sub) { - let flag = t.collection_flags(); - if flag != 0 { - s.dict.borrow_mut().insert( + interp.store_attr_public( + cls, + "__abstractmethods__", + Object::new_frozenset_from(abstracts), + ) +} + +/// `_abc_register(cls, subclass)`. +fn abc_register(args: &[Object]) -> Result { + let [cls, subclass] = args_exact::<2>(args, "_abc_register")?; + let interp = interpreter()?; + if !is_class(interp, &subclass)? { + return Err(type_error("Can only register classes")); + } + if interp.issubclass_public(&subclass, &cls)? { + return Ok(subclass); // Already a subclass. + } + // Test for cycles *after* testing for "already a subclass", so that + // `X.register(X)` is a no-op. + if interp.issubclass_public(&cls, &subclass)? { + return Err(runtime_error("Refusing to create an inheritance cycle")); + } + let owner = state_of(&cls)?; + state(&owner).registry.borrow_mut().insert(&subclass); + INVALIDATION_COUNTER.fetch_add(1, Ordering::Relaxed); + // Late registration on a Sequence or Mapping ABC must make pattern + // matching treat the class accordingly (CPython sets the type flag + // recursively; the VM reads this marker through the MRO). + if let (Object::Type(abc), Object::Type(sub)) = (&cls, &subclass) { + let flag = abc.collection_flags(); + if flag != 0 && !sub.flags.is_builtin { + sub.dict.borrow_mut().insert( DictKey(Object::from_static("_abc_collection_flags")), Object::Int(flag), ); } } - bump_cache(); - Ok(sub) + Ok(subclass) } +/// `_abc_instancecheck(cls, instance)`. fn abc_instancecheck(args: &[Object]) -> Result { - // Delegates to issubclass(type(obj), cls) — the Python wrapper - // dispatches the full protocol. - let cls = args.first().cloned().unwrap_or(Object::None); - let inst = args.get(1).cloned().unwrap_or(Object::None); - if let (Object::Type(t), Object::Instance(i)) = (&cls, &inst) { - if i.cls().is_subclass_of(t) { - return Ok(Object::Bool(true)); - } - if let Some(Object::Set(reg)) = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() + let [cls, instance] = args_exact::<2>(args, "_abc_instancecheck")?; + let interp = interpreter()?; + let owner = state_of(&cls)?; + let subclass = interp.load_attr_public(&instance, "__class__")?; + if state(&owner).cache.borrow().contains(&subclass) { + return Ok(Object::Bool(true)); + } + let subtype = Object::Type(crate::builtins::class_of(&instance)); + if subtype.is_same(&subclass) { + let data = state(&owner); + if data.negative_cache_version.get() == INVALIDATION_COUNTER.load(Ordering::Relaxed) + && data.negative_cache.borrow().contains(&subclass) { - for entry in reg.borrow().iter() { - if let Object::Type(et) = &entry.0 { - if i.cls().is_subclass_of(et) { - return Ok(Object::Bool(true)); - } - } - } + return Ok(Object::Bool(false)); } + return call_method(interp, &cls, "__subclasscheck__", &[subclass]); } - Ok(Object::Bool(false)) + let result = call_method(interp, &cls, "__subclasscheck__", &[subclass])?; + if interp.op_truth(&result)? { + return Ok(result); + } + call_method(interp, &cls, "__subclasscheck__", &[subtype]) } +/// `_abc_subclasscheck(cls, subclass)`. fn abc_subclasscheck(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let (Object::Type(t), Object::Type(st)) = (&cls, &sub) { - if st.is_subclass_of(t) { - return Ok(Object::Bool(true)); + let [cls, subclass] = args_exact::<2>(args, "_abc_subclasscheck")?; + let interp = interpreter()?; + if !is_class(interp, &subclass)? { + return Err(type_error("issubclass() arg 1 must be a class")); + } + let owner = state_of(&cls)?; + let data = state(&owner); + // 1. The positive cache. + if data.cache.borrow().contains(&subclass) { + return Ok(Object::Bool(true)); + } + // 2. The negative cache, invalidated by any registration since. + let counter = INVALIDATION_COUNTER.load(Ordering::Relaxed); + if data.negative_cache_version.get() < counter { + data.negative_cache.borrow_mut().clear(); + data.negative_cache_version.set(counter); + } else if data.negative_cache.borrow().contains(&subclass) { + return Ok(Object::Bool(false)); + } + let found = |data: &AbcState| { + data.cache.borrow_mut().insert(&subclass); + Ok(Object::Bool(true)) + }; + // 3. The subclass hook. + let ok = call_method( + interp, + &cls, + "__subclasshook__", + std::slice::from_ref(&subclass), + )?; + match ok { + Object::Bool(true) => return found(data), + Object::Bool(false) => { + data.negative_cache.borrow_mut().insert(&subclass); + return Ok(Object::Bool(false)); } - if let Some(Object::Set(reg)) = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - { - for entry in reg.borrow().iter() { - if let Object::Type(et) = &entry.0 { - if st.is_subclass_of(et) { - return Ok(Object::Bool(true)); - } - } - } + ref other if crate::vm_singletons::is_not_implemented(other) => {} + _ => { + return Err(assertion_error( + "__subclasshook__ must return either False, True, or NotImplemented", + )) + } + } + // 4. A direct subclass. + let direct = match (&subclass, &cls) { + (Object::Type(sub), Object::Type(abc)) => sub.is_subclass_of(abc), + _ => match attr_or_none(interp, &subclass, "__mro__")? { + Some(Object::Tuple(mro)) => mro.iter().any(|c| c.is_same(&cls)), + _ => false, + }, + }; + if direct { + return found(data); + } + // 5. A subclass of a registered class (recursive). + let registered = data.registry.borrow().live(); + for rcls in registered { + if interp.issubclass_public(&subclass, &rcls)? { + return found(data); + } + } + // 6. A subclass of a subclass (recursive). + let subclasses = call_method(interp, &cls, "__subclasses__", &[])?; + let globals = interp.builtins_dict(); + for scls in interp.collect_iterable(&subclasses, &globals)? { + if interp.issubclass_public(&subclass, &scls)? { + return found(data); } } + data.negative_cache.borrow_mut().insert(&subclass); Ok(Object::Bool(false)) } +/// `_get_dump(cls)`: weak references to the registry and caches, plus the +/// negative-cache version (used by the refleak hunter and `_dump_registry`). fn abc_get_dump(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - if let Object::Type(t) = cls { - let reg = t - .dict - .borrow() - .get(&DictKey(Object::from_static("_abc_registry"))) - .cloned() - .unwrap_or(Object::new_set()); - return Ok(Object::new_tuple_array([ - reg, - Object::new_set(), - Object::new_set(), - Object::Int(0), - ])); - } + let [cls] = args_exact::<1>(args, "_get_dump")?; + let interp = interpreter()?; + let owner = state_of(&cls)?; + let data = state(&owner); + let globals = interp.builtins_dict(); + let weakref = interp.do_import("_weakref", &Object::None, 0, &globals)?; + let make_ref = interp.load_attr_public(&weakref, "ref")?; + let mut weak_set = |classes: Vec| -> Result { + let mut refs = Vec::with_capacity(classes.len()); + for class in classes { + refs.push(interp.call(&make_ref, &[class], &[], &globals)?); + } + Ok(Object::new_set_from(refs)) + }; + let registry = weak_set(data.registry.borrow().live())?; + let cache = weak_set(data.cache.borrow().live())?; + let negative = weak_set(data.negative_cache.borrow().live())?; Ok(Object::new_tuple_array([ - Object::new_set(), - Object::new_set(), - Object::new_set(), - Object::Int(0), + registry, + cache, + negative, + Object::Int(data.negative_cache_version.get() as i64), ])) } fn abc_reset_registry(args: &[Object]) -> Result { - let _ = args; - bump_cache(); + let [cls] = args_exact::<1>(args, "_reset_registry")?; + let owner = state_of(&cls)?; + state(&owner).registry.borrow_mut().clear(); Ok(Object::None) } fn abc_reset_caches(args: &[Object]) -> Result { - let _ = args; - bump_cache(); + let [cls] = args_exact::<1>(args, "_reset_caches")?; + let owner = state_of(&cls)?; + let data = state(&owner); + data.cache.borrow_mut().clear(); + data.negative_cache.borrow_mut().clear(); Ok(Object::None) } diff --git a/crates/weavepy-vm/src/stdlib/collections_native.rs b/crates/weavepy-vm/src/stdlib/collections_native.rs index 64b7227d..e811d625 100644 --- a/crates/weavepy-vm/src/stdlib/collections_native.rs +++ b/crates/weavepy-vm/src/stdlib/collections_native.rs @@ -35,7 +35,6 @@ use crate::sync::RefCell; use crate::error::{index_error, runtime_error, stop_iteration, type_error, RuntimeError}; use crate::import::ModuleCache; use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; -use crate::types::PyInstance; /// The receiver must be a deque instance (any class whose `_data` slot /// is the backing list). CPython's method descriptor rejects a foreign @@ -48,6 +47,43 @@ struct DequeState<'a> { data: Rc>>, } +/// The slot names, interned: a slot key stored through the attribute +/// machinery shares the interned storage, so the hinted lookups settle on +/// one pointer compare instead of comparing the bytes. +struct SlotNames { + data: crate::shared_value::SharedStr, + head: crate::shared_value::SharedStr, + maxlen: crate::shared_value::SharedStr, + state: crate::shared_value::SharedStr, + deq: crate::shared_value::SharedStr, + index: crate::shared_value::SharedStr, + deq_state: crate::shared_value::SharedStr, +} + +/// The first interpreter thread's interned names (a thread interning +/// its own copies only misses the pointer compare, never the lookup). +static NAMES: std::sync::OnceLock = std::sync::OnceLock::new(); + +/// The interned slot names. +#[inline] +fn names() -> &'static SlotNames { + NAMES.get_or_init(|| { + let intern = |n: &str| match crate::stdlib::sys::intern_name(n) { + Object::Str(s) => s, + _ => unreachable!("names intern as strings"), + }; + SlotNames { + data: intern("_data"), + head: intern("_head"), + maxlen: intern("_maxlen"), + state: intern("_state"), + deq: intern("_deq"), + index: intern("_index"), + deq_state: intern("_deq_state"), + } + }) +} + // The slot positions `deque.__init__` assigns in order (hints only; the // name is always verified). const SLOT_DATA: usize = 0; @@ -56,14 +92,14 @@ const SLOT_MAXLEN: usize = 2; const SLOT_STATE: usize = 3; fn head_of(slots: &crate::types::SlotStorage) -> usize { - match slots.get_hinted(SLOT_HEAD, "_head") { + match slots.get_hinted(SLOT_HEAD, &names().head) { Some(Object::Int(h)) if *h >= 0 => *h as usize, _ => 0, } } fn set_head_of(slots: &mut crate::types::SlotStorage, h: usize) { - match slots.get_hinted_mut(SLOT_HEAD, "_head") { + match slots.get_hinted_mut(SLOT_HEAD, &names().head) { Some(slot) => *slot = Object::Int(h as i64), None => slots .insert("_head", Object::Int(h as i64)) @@ -72,14 +108,14 @@ fn set_head_of(slots: &mut crate::types::SlotStorage, h: usize) { } fn maxlen_of(slots: &crate::types::SlotStorage) -> Option { - match slots.get_hinted(SLOT_MAXLEN, "_maxlen") { + match slots.get_hinted(SLOT_MAXLEN, &names().maxlen) { Some(Object::Int(m)) if *m >= 0 => Some(*m as usize), _ => None, } } fn bump_state_of(slots: &mut crate::types::SlotStorage) { - match slots.get_hinted_mut(SLOT_STATE, "_state") { + match slots.get_hinted_mut(SLOT_STATE, &names().state) { Some(Object::Int(s)) => *s = s.wrapping_add(1), Some(slot) => *slot = Object::Int(1), None => slots.insert("_state", Object::Int(1)).map_or((), drop), @@ -112,14 +148,16 @@ impl DequeState<'_> { // unguarded (`peek_mut` returns `None` if any guard is live) and neither // reference outlives the native call, which runs no Python. #[allow(clippy::mut_from_ref)] -fn fast_parts(args: &[Object]) -> Option<(&mut crate::types::SlotStorage, &mut Vec)> { +fn fast_parts(args: &[Object]) -> Option> { let Object::Instance(inst) = args.first()? else { return None; }; // SAFETY: see above — no guard is live on either cell (`peek_mut` // checks), and neither reference outlives the native call. let slots = unsafe { inst.slots.peek_mut() }?; - let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, "_data") else { + let n = names(); + let [data, head, maxlen, state] = slots.leading_mut([&n.data, &n.head, &n.maxlen, &n.state])?; + let Object::List(data) = data else { return None; }; // The list lives in its own allocation, held by the `_data` slot, @@ -127,30 +165,77 @@ fn fast_parts(args: &[Object]) -> Option<(&mut crate::types::SlotStorage, &mut V let data: *const RefCell> = Rc::as_ptr(data); // SAFETY: as above. let d = unsafe { (*data).peek_mut() }?; - Some((slots, d)) + Some(Fast { + d, + head, + maxlen, + state, + }) } -/// [`popleft_locked`] over the unguarded parts (see [`fast_parts`]). -fn popleft_fast( - slots: &mut crate::types::SlotStorage, - d: &mut Vec, -) -> Result { - let mut h = head_of(slots); - if h >= d.len() { - return Err(index_error("pop from an empty deque")); +/// A deque's backing list and its `_head`, `_maxlen` and `_state` slots, +/// borrowed unguarded (see [`fast_parts`]) and found in one pass over +/// the slot store. +struct Fast<'a> { + d: &'a mut Vec, + head: &'a mut Object, + maxlen: &'a Object, + state: &'a mut Object, +} + +impl Fast<'_> { + #[inline] + fn head(&self) -> usize { + match *self.head { + Object::Int(h) if h >= 0 => h as usize, + _ => 0, + } } - bump_state_of(slots); - let x = std::mem::replace(&mut d[h], Object::None); - h += 1; - if h >= d.len() { - d.clear(); - h = 0; - } else if h >= 32 && h * 2 >= d.len() { - d.drain(..h); - h = 0; + + #[inline] + fn set_head(&mut self, h: usize) { + match &mut *self.head { + Object::Int(slot) => *slot = h as i64, + slot => *slot = Object::Int(h as i64), + } + } + + #[inline] + fn maxlen(&self) -> Option { + match *self.maxlen { + Object::Int(m) if m >= 0 => Some(m as usize), + _ => None, + } + } + + #[inline] + fn bump_state(&mut self) { + match &mut *self.state { + Object::Int(s) => *s = s.wrapping_add(1), + slot => *slot = Object::Int(1), + } + } + + /// [`popleft_locked`] over the unguarded parts. + fn popleft(&mut self) -> Result { + let mut h = self.head(); + let d = &mut *self.d; + if h >= d.len() { + return Err(index_error("pop from an empty deque")); + } + let x = std::mem::replace(&mut d[h], Object::None); + h += 1; + if h >= d.len() { + d.clear(); + h = 0; + } else if h >= 32 && h * 2 >= d.len() { + d.drain(..h); + h = 0; + } + self.bump_state(); + self.set_head(h); + Ok(x) } - set_head_of(slots, h); - Ok(x) } fn receiver<'a>(args: &'a [Object], method: &str) -> Result, RuntimeError> { @@ -159,7 +244,7 @@ fn receiver<'a>(args: &'a [Object], method: &str) -> Result, Runt .ok_or_else(|| type_error(format!("unbound method deque.{method}() needs an argument")))?; if let Object::Instance(inst) = recv { let slots = inst.slots.borrow_mut(); - if let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, "_data") { + if let Some(Object::List(data)) = slots.get_hinted(SLOT_DATA, &names().data) { let data = data.clone(); return Ok(DequeState { slots, data }); } @@ -172,13 +257,6 @@ fn receiver<'a>(args: &'a [Object], method: &str) -> Result, Runt /// Read-only helper for the iterator paths that still address the /// instance directly. -fn head(inst: &PyInstance) -> usize { - match inst.slot_get("_head") { - Some(Object::Int(h)) if h >= 0 => h as usize, - _ => 0, - } -} - fn popleft_locked(st: &mut DequeState<'_>, d: &mut Vec) -> Result { let mut h = st.head(); if h >= d.len() { @@ -200,11 +278,11 @@ fn popleft_locked(st: &mut DequeState<'_>, d: &mut Vec) -> Result Result { if let [_, x] = args { - if let Some((slots, d)) = fast_parts(args) { - bump_state_of(slots); - d.push(x.clone()); - let trimmed = match maxlen_of(slots) { - Some(m) if d.len() - head_of(slots) > m => Some(popleft_fast(slots, d)?), + if let Some(mut f) = fast_parts(args) { + f.bump_state(); + f.d.push(x.clone()); + let trimmed = match f.maxlen() { + Some(m) if f.d.len() - f.head() > m => Some(f.popleft()?), _ => None, }; if trimmed.is_some() { @@ -250,6 +328,28 @@ fn deque_append(args: &[Object]) -> Result { } fn deque_appendleft(args: &[Object]) -> Result { + if let [_, x] = args { + if let Some(mut f) = fast_parts(args) { + f.bump_state(); + let mut h = f.head().min(f.d.len()); + if h == 0 { + h = std::cmp::max(8, f.d.len() / 2); + f.d.splice(0..0, std::iter::repeat_n(Object::None, h)); + } + h -= 1; + f.d[h] = x.clone(); + f.set_head(h); + let trimmed = match f.maxlen() { + Some(m) if f.d.len() - h > m => f.d.pop(), + _ => None, + }; + if trimmed.is_some() { + crate::gc_trace::mark_maybe_dead(); + } + drop(trimmed); + return Ok(Object::None); + } + } let mut st = receiver(args, "appendleft")?; let x = match args { [_, x] => x.clone(), @@ -288,14 +388,16 @@ fn deque_appendleft(args: &[Object]) -> Result { fn deque_pop(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - let h = head_of(slots); - if d.len() > h { - bump_state_of(slots); - let x = d.pop().expect("len checked"); - if d.len() == h && h != 0 { - d.clear(); - set_head_of(slots, 0); + if let Some(mut f) = fast_parts(args) { + let h = f.head(); + if f.d.len() > h { + f.bump_state(); + let x = f.d.pop().expect("len checked"); + // An emptied deque keeps a short free prefix for the next + // `appendleft` instead of splicing a new one. + if f.d.len() == h && h > 32 { + f.d.clear(); + f.set_head(0); } return Ok(x); } @@ -325,9 +427,9 @@ fn deque_pop(args: &[Object]) -> Result { fn deque_popleft(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - if head_of(slots) < d.len() { - return popleft_fast(slots, d); + if let Some(mut f) = fast_parts(args) { + if f.head() < f.d.len() { + return f.popleft(); } } } @@ -345,8 +447,8 @@ fn deque_popleft(args: &[Object]) -> Result { fn deque_len(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - return Ok(Object::Int(d.len().saturating_sub(head_of(slots)) as i64)); + if let Some(f) = fast_parts(args) { + return Ok(Object::Int(f.d.len().saturating_sub(f.head()) as i64)); } } let st = receiver(args, "__len__")?; @@ -359,8 +461,8 @@ fn deque_len(args: &[Object]) -> Result { fn deque_bool(args: &[Object]) -> Result { if args.len() == 1 { - if let Some((slots, d)) = fast_parts(args) { - return Ok(Object::Bool(d.len() > head_of(slots))); + if let Some(f) = fast_parts(args) { + return Ok(Object::Bool(f.d.len() > f.head())); } } let st = receiver(args, "__bool__")?; @@ -372,6 +474,16 @@ fn deque_bool(args: &[Object]) -> Result { } fn deque_getitem(args: &[Object]) -> Result { + if let [_, Object::Int(i)] = args { + if let Some(f) = fast_parts(args) { + let h = f.head(); + let n = f.d.len().saturating_sub(h) as i64; + let index = if *i < 0 { *i + n } else { *i }; + if (0..n).contains(&index) { + return Ok(f.d[h + index as usize].clone()); + } + } + } let [_, index] = args else { receiver(args, "__getitem__")?; return Err(type_error("deque.__getitem__() takes one argument")); @@ -464,39 +576,123 @@ fn deque_reverse_iterator_next(args: &[Object]) -> Result deque_next(args, true) } +// The iterator classes' slot positions as their `__init__` assigns them +// (hints, like the deque's own). +const IT_DEQ: usize = 0; +const IT_INDEX: usize = 1; +const IT_STATE: usize = 2; + +/// [`deque_next`]'s common case over unguarded views (see +/// [`fast_parts`]): a live iterator over an unmutated deque with an item +/// left. `None` (nothing touched) leaves every other case to the full +/// body. +fn deque_next_fast(iterator: &crate::types::PyInstance, reverse: bool) -> Option { + // SAFETY: no guard is live on the cells (`peek`/`peek_mut` check), no + // Python runs before the last use, and the iterator and its deque are + // distinct objects. + let its = unsafe { iterator.slots.peek_mut() }?; + let n = names(); + let [deque, index, it_state] = its.leading_mut([&n.deq, &n.index, &n.deq_state])?; + let Object::Instance(deque) = deque else { + return None; + }; + // The deque lives while the iterator's slot holds it (unchanged here). + let deque: *const crate::types::PyInstance = Rc::as_ptr(deque); + let Object::Int(i) = index else { + return None; + }; + // SAFETY: as above. + let ds = unsafe { (*deque).slots.peek() }?; + let [data, head, _, state] = ds.leading([&n.data, &n.head, &n.maxlen, &n.state])?; + let Object::List(data) = data else { + return None; + }; + if state.as_i64() != it_state.as_i64() { + return None; + } + let h = match *head { + Object::Int(h) if h >= 0 => h as usize, + _ => 0, + }; + // SAFETY: as above. + let d = unsafe { data.peek() }?; + let h = h.min(d.len()); + let index = *i; + if index < 0 || index as usize >= d.len() - h { + return None; + } + let slot = if reverse { + d.len() - 1 - index as usize + } else { + h + index as usize + }; + let v = d[slot].clone(); + *i = index + 1; + Some(v) +} + fn deque_next(args: &[Object], reverse: bool) -> Result { let [Object::Instance(iterator)] = args else { return Err(type_error("deque iterator __next__ requires one iterator")); }; - let deque = match iterator.slot_get("_deq") { - Some(Object::Instance(deque)) => deque, - Some(Object::None) => return Err(stop_iteration()), - _ => return Err(type_error("deque iterator expected")), + if let Some(v) = deque_next_fast(iterator, reverse) { + return Ok(v); + } + let (deque, index, it_state) = { + let s = iterator.slots.borrow(); + let deque = match s.get_hinted(IT_DEQ, "_deq") { + Some(Object::Instance(deque)) => deque.clone(), + Some(Object::None) => return Err(stop_iteration()), + _ => return Err(type_error("deque iterator expected")), + }; + ( + deque, + s.get_hinted(IT_INDEX, &names().index) + .and_then(Object::as_i64), + s.get_hinted(IT_STATE, &names().deq_state) + .and_then(Object::as_i64), + ) }; - let Some(Object::List(data)) = deque.slot_get("_data") else { - return Err(type_error("deque expected")); + let (data, h, state) = { + let s = deque.slots.borrow(); + let Some(Object::List(data)) = s.get_hinted(SLOT_DATA, "_data") else { + return Err(type_error("deque expected")); + }; + ( + data.clone(), + head_of(&s), + s.get_hinted(SLOT_STATE, "_state").and_then(Object::as_i64), + ) }; // Serialize the state check, cursor advance, and item read with deque // end operations, including simultaneous next() calls under gil=0. // The local deque reference outlives the guard, so clearing _deq can't // finalize its items while the backing list is borrowed. let d = data.borrow(); - if deque.slot_get("_state").and_then(|s| s.as_i64()) - != iterator.slot_get("_deq_state").and_then(|s| s.as_i64()) - { + if state != it_state { iterator.slot_set("_deq", Object::None); return Err(runtime_error("deque mutated during iteration")); } - let h = head(&deque).min(d.len()); - let index = iterator - .slot_get("_index") - .and_then(|i| i.as_i64()) - .ok_or_else(|| type_error("invalid deque iterator index"))?; + let h = h.min(d.len()); + let index = index.ok_or_else(|| type_error("invalid deque iterator index"))?; if index < 0 || index as usize >= d.len() - h { iterator.slot_set("_deq", Object::None); return Err(stop_iteration()); } - iterator.slot_set("_index", Object::Int(index + 1)); + let advanced = match iterator + .slots + .borrow_mut() + .get_hinted_mut(IT_INDEX, "_index") + { + Some(slot) => { + *slot = Object::Int(index + 1); + true + } + None => false, + }; + if !advanced { + iterator.slot_set("_index", Object::Int(index + 1)); + } let slot = if reverse { d.len() - 1 - index as usize } else { diff --git a/crates/weavepy-vm/src/stdlib/ctypes_native.rs b/crates/weavepy-vm/src/stdlib/ctypes_native.rs index 288ec10c..7cf7a7b8 100644 --- a/crates/weavepy-vm/src/stdlib/ctypes_native.rs +++ b/crates/weavepy-vm/src/stdlib/ctypes_native.rs @@ -327,11 +327,12 @@ fn b_memoryview_at(args: &[Object]) -> Result { if addr == 0 && size != 0 { return Err(value_error("memoryview_at: NULL pointer access")); } - let region: Rc = Rc::new(RawRegion { - ptr: addr, - len: size as usize, - readonly, - }); + let region: Rc = + Rc::from_arc(std::sync::Arc::new(RawRegion { + ptr: addr, + len: size as usize, + readonly, + })); Ok(Object::MemoryView(Rc::new( crate::object::PyMemoryView::from_shared(region), ))) diff --git a/crates/weavepy-vm/src/stdlib/datetime_native.rs b/crates/weavepy-vm/src/stdlib/datetime_native.rs index 704c9115..df69bda3 100644 --- a/crates/weavepy-vm/src/stdlib/datetime_native.rs +++ b/crates/weavepy-vm/src/stdlib/datetime_native.rs @@ -182,6 +182,8 @@ const SHORTCUT_NAMES: &[(u8, &str)] = &[ (KIND_DATE, "__le__"), (KIND_DATE, "__gt__"), (KIND_DATE, "__ge__"), + (KIND_DATE, "__new__"), + (KIND_DATE, "__init__"), (KIND_DATE, "year"), (KIND_DATE, "month"), (KIND_DATE, "day"), @@ -193,6 +195,8 @@ const SHORTCUT_NAMES: &[(u8, &str)] = &[ (KIND_DATETIME, "__le__"), (KIND_DATETIME, "__gt__"), (KIND_DATETIME, "__ge__"), + (KIND_DATETIME, "__new__"), + (KIND_DATETIME, "__init__"), (KIND_DATETIME, "year"), (KIND_DATETIME, "month"), (KIND_DATETIME, "day"), @@ -273,8 +277,20 @@ fn as_int(o: &Object) -> Option { } } -fn td_fields(i: &PyInstance, n: &Names) -> Option<(i64, i64, i64)> { +// The field readers take a natively built instance's values straight +// from the shared layout; any other storage is read by name. + +#[allow(clippy::index_refutable_slice)] +fn td_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s.values_for_layout(&st.td_layout) { + return Some(( + as_int(&v[TD_DAYS])?, + as_int(&v[TD_SECONDS])?, + as_int(&v[TD_US])?, + )); + } Some(( as_int(s.get_hinted(TD_DAYS, &n.days)?)?, as_int(s.get_hinted(TD_SECONDS, &n.seconds)?)?, @@ -286,8 +302,20 @@ fn td_us(f: (i64, i64, i64)) -> i128 { (i128::from(f.0) * 86_400 + i128::from(f.1)) * 1_000_000 + i128::from(f.2) } -fn date_fields(i: &PyInstance, n: &Names) -> Option<(i64, i64, i64)> { +#[allow(clippy::index_refutable_slice)] +fn date_fields(i: &PyInstance, st: &State) -> Option<(i64, i64, i64)> { + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s + .values_for_layout(&st.date_layout) + .or_else(|| s.values_for_layout(&st.dt_layout)) + { + return Some(( + as_int(&v[D_YEAR])?, + as_int(&v[D_MONTH])?, + as_int(&v[D_DAY])?, + )); + } Some(( as_int(s.get_hinted(D_YEAR, &n.year)?)?, as_int(s.get_hinted(D_MONTH, &n.month)?)?, @@ -307,8 +335,22 @@ struct Dt { fold: i64, } -fn dt_fields(i: &PyInstance, n: &Names) -> Option
{ +fn dt_fields(i: &PyInstance, st: &State) -> Option
{ + let n = &st.names; let s = i.slots.try_borrow().ok()?; + if let Some(v) = s.values_for_layout(&st.dt_layout) { + return Some(Dt { + y: as_int(&v[D_YEAR])?, + m: as_int(&v[D_MONTH])?, + d: as_int(&v[D_DAY])?, + hh: as_int(&v[DT_HOUR])?, + mm: as_int(&v[DT_MINUTE])?, + ss: as_int(&v[DT_SECOND])?, + us: as_int(&v[DT_US])?, + tz: v[DT_TZINFO].clone(), + fold: as_int(&v[DT_FOLD])?, + }); + } Some(Dt { y: as_int(s.get_hinted(D_YEAR, &n.year)?)?, m: as_int(s.get_hinted(D_MONTH, &n.month)?)?, @@ -329,7 +371,8 @@ enum Tz { Fixed(i128), } -fn tz_of(tz: &Object, n: &Names) -> Option { +fn tz_of(tz: &Object, st: &State) -> Option { + let n = &st.names; match tz { Object::None => Some(Tz::Naive), o => { @@ -339,7 +382,7 @@ fn tz_of(tz: &Object, n: &Names) -> Option { s.get_hinted(TZ_OFFSET, &n.offset)?.clone() }; let td = inst(&off, KIND_TIMEDELTA)?; - Some(Tz::Fixed(td_us(td_fields(td, n)?))) + Some(Tz::Fixed(td_us(td_fields(td, st)?))) } } } @@ -423,11 +466,88 @@ fn instance_fixed( Object::Instance(Rc::new(i)) } +/// `date(y, m, d)` and `datetime(y, m, d[, hh[, mm[, ss[, us[, tz]]]]])` +/// (`tzinfo=` by keyword too) of the exact classes, with in-range `int` +/// fields and a naive or fixed-offset zone: the instance the replaced +/// `__new__` builds, built natively. `None` for every other shape, which +/// the Python constructor serves (and diagnoses). +pub(crate) fn construct( + cls: &Rc, + args: &[Object], + kwargs: &[(String, Object)], +) -> Option> { + let st = state_of_cls(cls)?; + let kind = cls.native_kind.get(); + let exact = match kind { + KIND_DATE => st.date.upgrade(), + KIND_DATETIME => st.datetime.upgrade(), + _ => None, + }; + if !exact.is_some_and(|c| Rc::ptr_eq(&c, cls)) || !verified(st, cls, kind) { + return None; + } + let int = |o: &Object| match o { + Object::Int(v) => Some(*v), + _ => None, + }; + if kind == KIND_DATE { + let ([y, m, d], true) = (args, kwargs.is_empty()) else { + return None; + }; + let (y, m, d) = (int(y)?, int(m)?, int(d)?); + return valid_date(y, m, d) + .then(|| new_date(st, y, m, d)) + .flatten() + .map(Ok); + } + if !(3..=8).contains(&args.len()) { + return None; + } + let mut tz = args.get(7).cloned().unwrap_or(Object::None); + for (name, v) in kwargs { + if name != "tzinfo" || args.len() == 8 { + return None; + } + tz = v.clone(); + } + let field = |ix: usize| args.get(ix).map_or(Some(0), int); + let f = Dt { + y: int(&args[0])?, + m: int(&args[1])?, + d: int(&args[2])?, + hh: field(3)?, + mm: field(4)?, + ss: field(5)?, + us: field(6)?, + tz, + fold: 0, + }; + let ok = valid_date(f.y, f.m, f.d) + && (0..24).contains(&f.hh) + && (0..60).contains(&f.mm) + && (0..60).contains(&f.ss) + && (0..1_000_000).contains(&f.us); + if !ok { + return None; + } + tz_of(&f.tz, st)?; + new_dt(st, &f).map(Ok) +} + /// A normalized exact `timedelta` from unnormalized components. fn new_td(st: &State, d: i128, s: i128, us: i128) -> Option> { let total = d * US_PER_DAY + s * 1_000_000 + us; - let days = total.div_euclid(US_PER_DAY); - let rest = total.rem_euclid(US_PER_DAY); + // 64-bit division when it fits (128-bit division is a library call). + let (days, rest) = match i64::try_from(total) { + Ok(t) => { + let per_day = US_PER_DAY as i64; + ( + i128::from(t.div_euclid(per_day)), + i128::from(t.rem_euclid(per_day)), + ) + } + Err(_) => (total.div_euclid(US_PER_DAY), total.rem_euclid(US_PER_DAY)), + }; if days.abs() > MAX_DAYS { return Some(Err(overflow_error(format!( "days={days}; must have magnitude <= 999999999" @@ -492,17 +612,28 @@ fn new_dt(st: &State, f: &Dt) -> Option { /// `datetime` fields shifted by `delta_us` (fold cleared, `tzinfo` /// kept), or `None` past the representable range. fn dt_shift(f: &Dt, delta_us: i128) -> Option
{ - let base = (i128::from(ymd2ord(f.y, f.m, f.d)) * 86_400 - + i128::from(f.hh * 3600 + f.mm * 60 + f.ss)) - * 1_000_000 - + i128::from(f.us); - let total = base + delta_us; - let ord = total.div_euclid(US_PER_DAY); - if !(1..=i128::from(MAX_ORDINAL)).contains(&ord) { + // Every representable datetime is under 2^59 microseconds, so a + // delta that fits no `i64` lands out of range (the Python path + // raises); the rest is 64-bit arithmetic over in-range fields. + let in_range = (1..=9999).contains(&f.y) + && valid_date(f.y, f.m, f.d) + && (0..24).contains(&f.hh) + && (0..60).contains(&f.mm) + && (0..60).contains(&f.ss) + && (0..1_000_000).contains(&f.us); + if !in_range { + return None; + } + let base = + (ymd2ord(f.y, f.m, f.d) * 86_400 + f.hh * 3600 + f.mm * 60 + f.ss) * 1_000_000 + f.us; + let total = base.checked_add(i64::try_from(delta_us).ok()?)?; + let per_day = US_PER_DAY as i64; + let ord = total.div_euclid(per_day); + if !(1..=MAX_ORDINAL).contains(&ord) { return None; } - let rest = total.rem_euclid(US_PER_DAY); - let (y, m, d) = ord2ymd(ord as i64); + let rest = total.rem_euclid(per_day); + let (y, m, d) = ord2ymd(ord); let secs = (rest / 1_000_000) as i64; Some(Dt { y, @@ -533,8 +664,8 @@ type Fast = fn(&[Object]) -> Option>; fn td_add(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) + i128::from(q.0), @@ -546,8 +677,8 @@ fn td_add(a: &[Object]) -> Option> { fn td_sub(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) - i128::from(q.0), @@ -559,7 +690,7 @@ fn td_sub(a: &[Object]) -> Option> { fn td_neg(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; new_td(st, -i128::from(p.0), -i128::from(p.1), -i128::from(p.2)) } @@ -567,7 +698,7 @@ fn td_mul(a: &[Object]) -> Option> { let [x, k] = a else { return None }; let k = i128::from(as_int(k)?); let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; new_td( st, i128::from(p.0) * k, @@ -579,15 +710,15 @@ fn td_mul(a: &[Object]) -> Option> { fn td_bool(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; Some(Ok(Object::Bool(p != (0, 0, 0)))) } fn td_cmp(a: &[Object]) -> Option { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = td_fields(inst(x, KIND_TIMEDELTA)?, &st.names)?; - let q = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let p = td_fields(inst(x, KIND_TIMEDELTA)?, st)?; + let q = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; Some(p.cmp(&q)) } @@ -617,7 +748,7 @@ fn date_toordinal(a: &[Object]) -> Option> { }, _ => return None, }; - let (y, m, d) = date_fields(i, &st.names)?; + let (y, m, d) = date_fields(i, st)?; if !valid_date(y, m, d) { return None; } @@ -645,8 +776,8 @@ fn date_isoweekday(a: &[Object]) -> Option> { fn date_shift(a: &[Object], sign: i64) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let (yy, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let (days, _, _) = td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?; + let (yy, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; + let (days, _, _) = td_fields(inst(y, KIND_TIMEDELTA)?, st)?; if !valid_date(yy, m, d) { return None; } @@ -668,8 +799,8 @@ fn date_sub(a: &[Object]) -> Option> { return date_shift(a, -1); } let st = state_of(x)?; - let (y1, m1, d1) = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let (y2, m2, d2) = date_fields(inst(y, KIND_DATE)?, &st.names)?; + let (y1, m1, d1) = date_fields(inst(x, KIND_DATE)?, st)?; + let (y2, m2, d2) = date_fields(inst(y, KIND_DATE)?, st)?; if !valid_date(y1, m1, d1) || !valid_date(y2, m2, d2) { return None; } @@ -684,8 +815,8 @@ fn date_sub(a: &[Object]) -> Option> { fn date_cmp(a: &[Object]) -> Option { let [x, y] = a else { return None }; let st = state_of(x)?; - let p = date_fields(inst(x, KIND_DATE)?, &st.names)?; - let q = date_fields(inst(y, KIND_DATE)?, &st.names)?; + let p = date_fields(inst(x, KIND_DATE)?, st)?; + let q = date_fields(inst(y, KIND_DATE)?, st)?; Some(p.cmp(&q)) } @@ -698,7 +829,7 @@ cmp_fast!(date_ge, date_cmp, |o| o.is_ge()); fn date_isoformat(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; if !valid_date(y, m, d) { return None; } @@ -709,7 +840,7 @@ fn date_isoformat(a: &[Object]) -> Option> { fn date_replace(a: &[Object]) -> Option> { let (x, rest) = a.split_first()?; let st = state_of(x)?; - let (mut y, mut m, mut d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (mut y, mut m, mut d) = date_fields(inst(x, KIND_DATE)?, st)?; if rest.len() > 3 { return None; } @@ -733,8 +864,8 @@ fn date_replace(a: &[Object]) -> Option> { fn dt_add(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; - let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?); + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; + let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, st)?); if !valid_date(f.y, f.m, f.d) { return None; } @@ -747,26 +878,26 @@ fn dt_add(a: &[Object]) -> Option> { fn dt_sub(a: &[Object]) -> Option> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } match kind_of(y) { KIND_TIMEDELTA => { - let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, &st.names)?); + let delta = td_us(td_fields(inst(y, KIND_TIMEDELTA)?, st)?); match dt_shift(&f, -delta) { Some(g) => Some(Ok(new_dt(st, &g)?)), None => Some(Err(overflow_error("date value out of range"))), } } KIND_DATETIME => { - let g = dt_fields(inst(y, KIND_DATETIME)?, &st.names)?; + let g = dt_fields(inst(y, KIND_DATETIME)?, st)?; if !valid_date(g.y, g.m, g.d) { return None; } let mut diff = dt_us(&f) - dt_us(&g); if !f.tz.is_same(&g.tz) { - match (tz_of(&f.tz, &st.names)?, tz_of(&g.tz, &st.names)?) { + match (tz_of(&f.tz, st)?, tz_of(&g.tz, st)?) { (Tz::Naive, Tz::Naive) => {} (Tz::Fixed(a), Tz::Fixed(b)) => diff += b - a, // The mixed case raises in the Python code. @@ -784,15 +915,15 @@ fn dt_sub(a: &[Object]) -> Option> { fn dt_cmp_raw(a: &[Object]) -> Option<(std::cmp::Ordering, bool)> { let [x, y] = a else { return None }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; - let g = dt_fields(inst(y, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; + let g = dt_fields(inst(y, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) || !valid_date(g.y, g.m, g.d) { return None; } if f.tz.is_same(&g.tz) { return Some((dt_us(&f).cmp(&dt_us(&g)), true)); } - match (tz_of(&f.tz, &st.names)?, tz_of(&g.tz, &st.names)?) { + match (tz_of(&f.tz, st)?, tz_of(&g.tz, st)?) { (Tz::Naive, Tz::Naive) => Some((dt_us(&f).cmp(&dt_us(&g)), true)), (Tz::Fixed(a), Tz::Fixed(b)) => Some(((dt_us(&f) - a).cmp(&(dt_us(&g) - b)), true)), _ => Some((std::cmp::Ordering::Equal, false)), @@ -826,7 +957,7 @@ dt_order!(dt_ge, |o| o.is_ge()); fn dt_date(a: &[Object]) -> Option> { let [x] = a else { return None }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(y, m, d) { return None; } @@ -881,11 +1012,11 @@ fn dt_isoformat(a: &[Object]) -> Option> { return None; } let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } - let tz = tz_of(&f.tz, &st.names)?; + let tz = tz_of(&f.tz, st)?; let mut out = String::with_capacity(32); let _ = write!(out, "{:04}-{:02}-{:02}{sep}", f.y, f.m, f.d); match spec { @@ -1010,11 +1141,11 @@ fn dt_strftime(a: &[Object]) -> Option> { return None; }; let st = state_of(x)?; - let f = dt_fields(inst(x, KIND_DATETIME)?, &st.names)?; + let f = dt_fields(inst(x, KIND_DATETIME)?, st)?; if !valid_date(f.y, f.m, f.d) { return None; } - let tz = tz_of(&f.tz, &st.names)?; + let tz = tz_of(&f.tz, st)?; let s = strftime_numeric( f.y, f.m, @@ -1031,7 +1162,7 @@ fn date_strftime(a: &[Object]) -> Option> { return None; }; let st = state_of(x)?; - let (y, m, d) = date_fields(inst(x, KIND_DATE)?, &st.names)?; + let (y, m, d) = date_fields(inst(x, KIND_DATE)?, st)?; if !valid_date(y, m, d) { return None; } @@ -1677,7 +1808,7 @@ pub(crate) fn install(args: &[Object]) -> Result { for (kind, cls) in &classes { let _ = cls .native_ext - .set(state.clone() as Rc); + .set(crate::rc_unsize!(state.clone() => dyn std::any::Any + Send + Sync)); cls.native_kind.set(*kind); } for spec in SPECS { diff --git a/crates/weavepy-vm/src/stdlib/gc_mod.rs b/crates/weavepy-vm/src/stdlib/gc_mod.rs index 6455907a..c62152f4 100644 --- a/crates/weavepy-vm/src/stdlib/gc_mod.rs +++ b/crates/weavepy-vm/src/stdlib/gc_mod.rs @@ -16,7 +16,7 @@ use crate::object::{BuiltinFn, DictData, DictKey, Object, PyModule}; thread_local! { static GC_ENABLED: RefCell = const { RefCell::new(true) }; static GC_DEBUG: RefCell = const { RefCell::new(0) }; - static GC_THRESHOLD: RefCell<(i64, i64, i64)> = const { RefCell::new((700, 10, 10)) }; + static GC_THRESHOLD: RefCell<(i64, i64, i64)> = const { RefCell::new((2000, 10, 10)) }; } pub fn build(_cache: &ModuleCache) -> Rc { @@ -169,7 +169,7 @@ fn get_threshold(_args: &[Object]) -> Result { } fn set_threshold(args: &[Object]) -> Result { - let mut vals = [700i64, 10, 10]; + let mut vals = [2000i64, 10, 10]; for (slot, v) in vals.iter_mut().zip(args.iter()) { if let Object::Int(n) = v { *slot = *n; diff --git a/crates/weavepy-vm/src/stdlib/gc_real.rs b/crates/weavepy-vm/src/stdlib/gc_real.rs index e856b323..aa368d30 100644 --- a/crates/weavepy-vm/src/stdlib/gc_real.rs +++ b/crates/weavepy-vm/src/stdlib/gc_real.rs @@ -312,7 +312,7 @@ fn get_threshold(_args: &[Object]) -> Result { } fn set_threshold(args: &[Object]) -> Result { - let mut vals = [700usize, 10, 10]; + let mut vals = [2000usize, 10, 10]; for (slot, v) in vals.iter_mut().zip(args.iter()) { if let Object::Int(n) = v { *slot = (*n).max(0) as usize; diff --git a/crates/weavepy-vm/src/stdlib/io.rs b/crates/weavepy-vm/src/stdlib/io.rs index e0077afc..01e39199 100644 --- a/crates/weavepy-vm/src/stdlib/io.rs +++ b/crates/weavepy-vm/src/stdlib/io.rs @@ -182,34 +182,6 @@ pub fn build(_cache: &ModuleCache) -> Rc { }) } -/// `IOBase.register(subclass)` — ABC virtual-subclass registration. The -/// class binds first (classmethod); we record the subclass in the class's -/// `_abc_registry` set and return it so `register` also works as a -/// decorator, mirroring `_abc._abc_register`. -fn io_abc_register(args: &[Object]) -> Result { - let cls = args.first().cloned().unwrap_or(Object::None); - let sub = args.get(1).cloned().unwrap_or(Object::None); - if let Object::Type(t) = &cls { - let key = DictKey(Object::from_static("_abc_registry")); - let reg = { - let existing = t.dict.borrow().get(&key).cloned(); - match existing { - Some(Object::Set(s)) => s, - _ => { - let s = Object::new_set(); - t.dict.borrow_mut().insert(key, s.clone()); - match s { - Object::Set(s) => s, - _ => return Ok(sub), - } - } - } - }; - reg.borrow_mut().insert(DictKey(sub.clone())); - } - Ok(sub) -} - fn builtin(name: &'static str, body: fn(&[Object]) -> Result) -> Object { Object::Builtin(Rc::new(BuiltinFn { name, @@ -1861,21 +1833,10 @@ pub(crate) fn file_io_abc_match( /// Build the `IOBase → {RawIOBase, BufferedIOBase, TextIOBase} → FileIO` /// hierarchy with the CPython mixin methods installed on the root. fn build_iobase_family_inner() -> IoFamily { - use crate::object::MethodWrapper; use crate::types::{TypeFlags, TypeObject}; let bt = crate::builtin_types::builtin_types(); let mut dict = DictData::default(); install_iobase_mixins(&mut dict); - // `io.IOBase.register(...)` — ABC virtual-subclass registration. - dict.insert( - DictKey(Object::from_static("register")), - Object::ClassMethod(MethodWrapper::new(Object::Builtin(Rc::new(BuiltinFn { - name: "register", - binds_instance: true, - call: Box::new(io_abc_register), - call_kw: None, - })))), - ); let flags = || TypeFlags { is_exception: false, is_builtin: true, diff --git a/crates/weavepy-vm/src/stdlib/math.rs b/crates/weavepy-vm/src/stdlib/math.rs index 0da3c1b8..44c17eae 100644 --- a/crates/weavepy-vm/src/stdlib/math.rs +++ b/crates/weavepy-vm/src/stdlib/math.rs @@ -268,6 +268,14 @@ pub fn build(_cache: &ModuleCache) -> Rc { builtin("acosh", math_acosh), ); } + // Every function's body is a leaf over plain numbers: only another + // object's `__float__`/`__index__`/`__ceil__`-style hooks run Python + // code, and the dispatch loop admits scalar arguments only. + for v in dict.borrow().values() { + if let Object::Builtin(b) = v { + crate::leaf_builtins::register_scalar(b); + } + } Rc::new(PyModule { name: "math".to_owned(), filename: None, diff --git a/crates/weavepy-vm/src/stdlib/mmap_mod.rs b/crates/weavepy-vm/src/stdlib/mmap_mod.rs index 14ab65d1..87c233c4 100644 --- a/crates/weavepy-vm/src/stdlib/mmap_mod.rs +++ b/crates/weavepy-vm/src/stdlib/mmap_mod.rs @@ -410,7 +410,7 @@ fn state_cell(inst: &Rc) -> Result>, RuntimeEr /// `None` for a closed mapping. pub fn shared_buffer(inst: &Rc) -> Option> { let cell = state_cell(inst).ok()?; - let region: Rc = cell.borrow().region.clone(); + let region = crate::rc_unsize!(cell.borrow().region.clone() => dyn SharedMemBuffer); Some(region) } diff --git a/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs b/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs index 7db1b355..8330be53 100644 --- a/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs +++ b/crates/weavepy-vm/src/stdlib/multiprocessing_mod.rs @@ -592,7 +592,7 @@ fn make_semlock_instance(inner: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(semlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -1225,7 +1225,7 @@ fn nt_make_semlock_instance(inner: &Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(nt_semlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel.rs b/crates/weavepy-vm/src/stdlib/pickle_accel.rs index b05ef835..3f9d8c96 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel.rs @@ -44,11 +44,30 @@ impl FunctionGuard { let Object::Function(function) = value else { return false; }; - if Rc::as_ptr(function) != self.function.as_ptr() - || Rc::as_ptr(&function.code.borrow()) != self.code.as_ptr() - { + Rc::as_ptr(function) == self.function.as_ptr() && self.holds_function(function) + } + + /// Whether the guarded function is still alive and unchanged. + fn holds_live(&self) -> bool { + if self.function.strong_count() == 0 { return false; } + // SAFETY: a live strong count keeps the function allocated, and + // nothing here can release it. + self.holds_function(unsafe { &*self.function.as_ptr() }) + } + + /// `function` (the guarded one) still runs the guarded code with its + /// compiled defaults. + fn holds_function(&self, function: &PyFunction) -> bool { + if Rc::as_ptr(&function.code.borrow()) != self.code.as_ptr() { + return false; + } + // Only an assignment to `__defaults__` or `__kwdefaults__` raises + // the flag, so the common function skips the slot probes. + if !function.defaults_maybe_overridden() { + return true; + } let slots = function.slots.borrow(); !slots.contains_key(&StrKey("__defaults__")) && !slots.contains_key(&StrKey("__kwdefaults__")) @@ -412,15 +431,12 @@ impl ClassFunctionsGuard { } fn holds(&self) -> bool { - self.class.upgrade().is_some_and(|class| { - class.attr_version.get() == self.version - && self.functions.iter().all(|guard| { - guard - .function - .upgrade() - .is_some_and(|f| guard.holds(&Object::Function(f))) - }) - }) + // An unchanged version proves the class holds the same functions; + // each one's code and defaults can still change in place. + self.class.strong_count() > 0 + // SAFETY: a live strong count keeps the class allocated. + && unsafe { &*self.class.as_ptr() }.attr_version.get() == self.version + && self.functions.iter().all(FunctionGuard::holds_live) } } @@ -593,8 +609,12 @@ trait Sink<'a> { fn build(&mut self, target: &Self::Value, state: Self::Value) -> Option<()>; } +#[inline(always)] fn push(items: &mut Vec, value: T) -> Option<()> { - items.try_reserve(1).ok()?; + // (The fallible reservation only when the vector is full.) + if items.len() == items.capacity() { + items.try_reserve(1).ok()?; + } items.push(value); Some(()) } @@ -928,6 +948,10 @@ struct Probe<'a, 'c, const RECORD_OPCODES: bool> { builds: Vec<(StateNodes, Option)>, context: &'c dyn Fn() -> Option, resolved: Option, + /// The `__slots__` members already verified in this stream (see + /// `classes::is_member_slot`): every instance of a class names the + /// same ones. + member_slots: Vec<(*const TypeObject, &'a str)>, } impl<'a, 'c, const RECORD_OPCODES: bool> Probe<'a, 'c, RECORD_OPCODES> { @@ -939,9 +963,28 @@ impl<'a, 'c, const RECORD_OPCODES: bool> Probe<'a, 'c, RECORD_OPCODES> { builds: Vec::new(), context, resolved: None, + member_slots: Vec::new(), } } + /// [`classes::is_member_slot`], remembered for this stream. (No Python + /// runs during the probe, so the class can't change in between.) + fn is_member_slot(&mut self, class: &Rc, name: &'a str) -> bool { + let key = Rc::as_ptr(class); + if self + .member_slots + .iter() + .any(|&(c, n)| c == key && n == name) + { + return true; + } + let yes = classes::is_member_slot(class, name); + if yes && self.member_slots.len() < 64 { + self.member_slots.push((key, name)); + } + yes + } + /// Count the references to each node that the result keeps: the root, /// container items, and the state pair's items when the pair itself is /// kept. BUILD copies its state, so its edge keeps nothing. @@ -1243,10 +1286,11 @@ impl<'a, const RECORD_OPCODES: bool> Sink<'a> for Probe<'a, '_, RECORD_OPCODES> } if let Some(slots) = nodes.slots { // `setattr` must reach a member descriptor of `__slots__`. - if !self - .state_keys(slots)? - .iter() - .all(|name| classes::is_member_slot(&class.class, name)) + let names = self.state_keys(slots)?.to_vec(); + let class = class.class.clone(); + if !names + .into_iter() + .all(|name| self.is_member_slot(&class, name)) { return None; } @@ -1397,7 +1441,19 @@ impl<'a> Sink<'a> for Objects { let Object::Type(class) = class else { return None; }; - // `object.__new__(cls)`, including its cycle-collector registration. + // `object.__new__(cls)`, including its cycle-collector registration: + // deferred, as for a plain class's ordinary construction, until the + // instance could hold a non-atomic value (the state paths below + // track it then). + if class.native_kind.get() == 0 + && !class.flags.is_builtin + && !class.instances_need_finalize() + { + self.next_node += 1; + return Some(Object::Instance(crate::types::PyInstance::new_deferred( + class.clone(), + ))); + } let instance = Object::Instance(Rc::new(crate::types::PyInstance::new(class.clone()))); self.created(&instance); Some(instance) @@ -1428,8 +1484,12 @@ impl<'a> Sink<'a> for Objects { } } if let Some(slots) = slots { - // `setattr(inst, key, value)` on a verified member descriptor. - instance.ensure_gc_tracked(); + // `setattr(inst, key, value)` on a verified member descriptor, + // whose write barrier tracks a deferred instance at its first + // non-atomic value. + if slots.borrow().iter().any(|(_, v)| !v.is_gc_atomic()) { + instance.ensure_gc_tracked(); + } let mut storage = instance.slots.borrow_mut(); if movable.slots.is_some() { let data = std::mem::take(&mut *slots.borrow_mut()); diff --git a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs index f5ecba5a..92847caa 100644 --- a/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs +++ b/crates/weavepy-vm/src/stdlib/pickle_accel/encode.rs @@ -23,8 +23,12 @@ use crate::types::{PyInstance, TypeObject}; const FRAME_TARGET: usize = 65_536; const MAX_DEPTH: usize = 128; +#[inline(always)] fn extend(output: &mut Vec, data: &[u8]) -> Option<()> { - output.try_reserve(data.len()).ok()?; + // (The fallible reservation only when the buffer is short.) + if output.capacity() - output.len() < data.len() { + output.try_reserve(data.len()).ok()?; + } output.extend_from_slice(data); Some(()) } @@ -36,6 +40,7 @@ struct Framer { } impl Framer { + #[inline] fn write(&mut self, data: &[u8]) -> Option<()> { if self.frame_start.is_none() { let start = self.output.len(); @@ -550,9 +555,17 @@ fn encode_with_context( if !matches!(protocol, 4 | 5) { return None; } + // The previous call's memo table (emptied, its capacity kept) and + // output size: a `dumps` loop then neither regrows the table nor + // copies the output as it doubles. + let memo = SPARE_MEMO.with(std::cell::Cell::take).unwrap_or_default(); + let hint = OUTPUT_HINT.with(std::cell::Cell::get); let mut encoder = Encoder { - writer: Framer::default(), - memo: FxHashMap::default(), + writer: Framer { + output: Vec::new(), + frame_start: None, + }, + memo, anonymous: 0, active: Vec::new(), python_headroom, @@ -561,14 +574,39 @@ fn encode_with_context( classes: FxHashMap::default(), slot_name_caches: Vec::new(), }; - extend(&mut encoder.writer.output, &[0x80, protocol])?; - encoder.save(value, 0)?; - encoder.writer.write(b".")?; - encoder.writer.commit(true); - encoder.publish_slot_name_caches(); + let done = (|| { + encoder + .writer + .output + .try_reserve(hint.clamp(64, MAX_OUTPUT_HINT)) + .ok()?; + extend(&mut encoder.writer.output, &[0x80, protocol])?; + encoder.save(value, 0)?; + encoder.writer.write(b".")?; + encoder.writer.commit(true); + encoder.publish_slot_name_caches(); + Some(()) + })(); + let mut memo = std::mem::take(&mut encoder.memo); + if memo.capacity() <= MAX_SPARE_MEMO { + memo.clear(); + SPARE_MEMO.with(|spare| spare.set(Some(memo))); + } + done?; + OUTPUT_HINT.with(|h| h.set(encoder.writer.output.len())); Some(encoder.writer.output) } +/// The largest memo table and output reservation kept between calls. +const MAX_SPARE_MEMO: usize = 1 << 14; +const MAX_OUTPUT_HINT: usize = 1 << 20; + +thread_local! { + static SPARE_MEMO: std::cell::Cell>> = + const { std::cell::Cell::new(None) }; + static OUTPUT_HINT: std::cell::Cell = const { std::cell::Cell::new(0) }; +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/weavepy-vm/src/stdlib/python/abc.py b/crates/weavepy-vm/src/stdlib/python/abc.py index cf212823..f8a4e11c 100644 --- a/crates/weavepy-vm/src/stdlib/python/abc.py +++ b/crates/weavepy-vm/src/stdlib/python/abc.py @@ -81,201 +81,66 @@ def my_abstract_property(self): __isabstractmethod__ = True -# WeavePy has no `_abc` C accelerator, so `ABCMeta` is implemented in -# pure Python (mirroring `_py_abc`, which stays a separate module that -# `test_abc` imports to exercise the reference implementation). It is -# defined *here* rather than imported so that its frames' module is -# `'abc'`: `typing._allow_reckless_class_checks()` walks the frame stack -# and only skips the runtime-protocol restrictions when the caller -# module is `'abc'` or `'functools'` (e.g. `isinstance(x, Traversable)` -# reaches `_ProtocolMeta.__subclasscheck__` *via* -# `ABCMeta.__instancecheck__`, which CPython sees as module `'abc'` -# because that's where the delegating method lives). -from _weakrefset import WeakSet - - -def get_cache_token(): - """Returns the current ABC cache token. - - The token is an opaque object (supporting equality testing) identifying the - current version of the ABC cache for virtual subclasses. The token changes - with every call to ``register()`` on any ABC. - """ - return ABCMeta._abc_invalidation_counter - - -class ABCMeta(type): - """Metaclass for defining Abstract Base Classes (ABCs). - - Use this metaclass to create an ABC. An ABC can be subclassed - directly, and then acts as a mix-in class. You can also register - unrelated concrete classes (even built-in classes) and unrelated - ABCs as 'virtual subclasses' -- these and their descendants will - be considered subclasses of the registering ABC by the built-in - issubclass() function, but the registering ABC won't show up in - their MRO (Method Resolution Order) nor will method - implementations defined by the registering ABC be callable (not - even via super()). - """ - - # A global counter that is incremented each time a class is - # registered as a virtual subclass of anything. It forces the - # negative cache to be cleared before its next use. - # Note: this counter is private. Use `abc.get_cache_token()` for - # external code. - _abc_invalidation_counter = 0 - - def __new__(mcls, name, bases, namespace, /, **kwargs): - cls = super().__new__(mcls, name, bases, namespace, **kwargs) - # Compute set of abstract method names - abstracts = {name - for name, value in namespace.items() - if getattr(value, "__isabstractmethod__", False)} - for base in bases: - for name in getattr(base, "__abstractmethods__", set()): - value = getattr(cls, name, None) - if getattr(value, "__isabstractmethod__", False): - abstracts.add(name) - cls.__abstractmethods__ = frozenset(abstracts) - # CPython's C `_abc_init` consumes `__abc_tpflags__` here: reject - # a class claiming both Py_TPFLAGS_SEQUENCE and Py_TPFLAGS_MAPPING, - # fold the collection bits into the type's flags (stored under a - # private name the VM's `__flags__` getter folds in), and delete - # the public attribute. - tpflags = namespace.get("__abc_tpflags__") - if isinstance(tpflags, int): - COLLECTION_FLAGS = (1 << 5) | (1 << 6) - if tpflags & COLLECTION_FLAGS == COLLECTION_FLAGS: - raise TypeError( - "__abc_tpflags__ cannot be both Py_TPFLAGS_SEQUENCE" - " and Py_TPFLAGS_MAPPING" - ) - cls._abc_collection_flags = tpflags & COLLECTION_FLAGS - if "__abc_tpflags__" in namespace: - del cls.__abc_tpflags__ - # Set up inheritance registry - cls._abc_registry = WeakSet() - cls._abc_cache = WeakSet() - cls._abc_negative_cache = WeakSet() - cls._abc_negative_cache_version = ABCMeta._abc_invalidation_counter - return cls - - def register(cls, subclass): - """Register a virtual subclass of an ABC. - - Returns the subclass, to allow usage as a class decorator. +try: + from _abc import (get_cache_token, _abc_init, _abc_register, + _abc_instancecheck, _abc_subclasscheck, _get_dump, + _reset_registry, _reset_caches) +except ImportError: + from _py_abc import ABCMeta, get_cache_token + ABCMeta.__module__ = 'abc' +else: + class ABCMeta(type): + """Metaclass for defining Abstract Base Classes (ABCs). + + Use this metaclass to create an ABC. An ABC can be subclassed + directly, and then acts as a mix-in class. You can also register + unrelated concrete classes (even built-in classes) and unrelated + ABCs as 'virtual subclasses' -- these and their descendants will + be considered subclasses of the registering ABC by the built-in + issubclass() function, but the registering ABC won't show up in + their MRO (Method Resolution Order) nor will method + implementations defined by the registering ABC be callable (not + even via super()). """ - if not isinstance(subclass, type): - raise TypeError("Can only register classes") - if issubclass(subclass, cls): - return subclass # Already a subclass - # Subtle: test for cycles *after* testing for "already a subclass"; - # this means we allow X.register(X) and interpret it as a no-op. - if issubclass(cls, subclass): - # This would create a cycle, which is bad for the algorithm below - raise RuntimeError("Refusing to create an inheritance cycle") - cls._abc_registry.add(subclass) - ABCMeta._abc_invalidation_counter += 1 # Invalidate negative cache - # CPython's C `_abc_register` copies the ABC's collection flag - # (Py_TPFLAGS_SEQUENCE / Py_TPFLAGS_MAPPING) onto the registered - # class (and recursively its subclasses), so late registration on - # Sequence/Mapping makes match patterns work (test_patma). The VM - # walks the MRO for this marker, so stamping the registered class - # also covers its subclasses. - collection_flag = getattr(cls, "_abc_collection_flags", 0) - if collection_flag: - try: - subclass._abc_collection_flags = collection_flag - except TypeError: - pass # immutable (builtin) type - return subclass - - def _dump_registry(cls, file=None): - """Debug helper to print the ABC registry.""" - print(f"Class: {cls.__module__}.{cls.__qualname__}", file=file) - print(f"Inv. counter: {get_cache_token()}", file=file) - for name in cls.__dict__: - if name.startswith("_abc_"): - value = getattr(cls, name) - if isinstance(value, WeakSet): - value = set(value) - print(f"{name}: {value!r}", file=file) - - def _abc_registry_clear(cls): - """Clear the registry (for debugging or testing).""" - cls._abc_registry.clear() - - def _abc_caches_clear(cls): - """Clear the caches (for debugging or testing).""" - cls._abc_cache.clear() - cls._abc_negative_cache.clear() - - def __instancecheck__(cls, instance): - """Override for isinstance(instance, cls).""" - # Inline the cache checking - subclass = instance.__class__ - if subclass in cls._abc_cache: - return True - subtype = type(instance) - if subtype is subclass: - if (cls._abc_negative_cache_version == - ABCMeta._abc_invalidation_counter and - subclass in cls._abc_negative_cache): - return False - # Fall back to the subclass check. - return cls.__subclasscheck__(subclass) - return any(cls.__subclasscheck__(c) for c in (subclass, subtype)) - - def __subclasscheck__(cls, subclass): - """Override for issubclass(subclass, cls).""" - if not isinstance(subclass, type): - raise TypeError('issubclass() arg 1 must be a class') - # Check cache - if subclass in cls._abc_cache: - return True - # Check negative cache; may have to invalidate - if cls._abc_negative_cache_version < ABCMeta._abc_invalidation_counter: - # Invalidate the negative cache - cls._abc_negative_cache = WeakSet() - cls._abc_negative_cache_version = ABCMeta._abc_invalidation_counter - elif subclass in cls._abc_negative_cache: - return False - # Check the subclass hook - ok = cls.__subclasshook__(subclass) - if ok is not NotImplemented: - assert isinstance(ok, bool) - if ok: - cls._abc_cache.add(subclass) - else: - cls._abc_negative_cache.add(subclass) - return ok - # Check if it's a direct subclass - if cls in getattr(subclass, '__mro__', ()): - cls._abc_cache.add(subclass) - return True - # Check if it's a subclass of a registered class (recursive) - for rcls in cls._abc_registry: - if issubclass(subclass, rcls): - cls._abc_cache.add(subclass) - return True - # Check if it's a subclass of a subclass (recursive) - for scls in cls.__subclasses__(): - if issubclass(subclass, scls): - cls._abc_cache.add(subclass) - return True - # No dice; update negative cache - cls._abc_negative_cache.add(subclass) - return False - - -# CPython's pure-Python path does `from _py_abc import ABCMeta` and then -# `ABCMeta.__module__ = 'abc'` — and *that assignment* drops the class's -# `__firstlineno__` (type_set_module invalidates stale source info), so -# `inspect.getsource(abc.ABCMeta)` reports "source code not available" -# whenever the C accelerator is absent (test_inspect -# test_getsource_stdlib_abc). Mirror the assignment — a no-op for the -# module name itself, but with the same firstlineno-dropping effect. -ABCMeta.__module__ = 'abc' + def __new__(mcls, name, bases, namespace, /, **kwargs): + cls = super().__new__(mcls, name, bases, namespace, **kwargs) + _abc_init(cls) + return cls + + def register(cls, subclass): + """Register a virtual subclass of an ABC. + + Returns the subclass, to allow usage as a class decorator. + """ + return _abc_register(cls, subclass) + + def __instancecheck__(cls, instance): + """Override for isinstance(instance, cls).""" + return _abc_instancecheck(cls, instance) + + def __subclasscheck__(cls, subclass): + """Override for issubclass(subclass, cls).""" + return _abc_subclasscheck(cls, subclass) + + def _dump_registry(cls, file=None): + """Debug helper to print the ABC registry.""" + print(f"Class: {cls.__module__}.{cls.__qualname__}", file=file) + print(f"Inv. counter: {get_cache_token()}", file=file) + (_abc_registry, _abc_cache, _abc_negative_cache, + _abc_negative_cache_version) = _get_dump(cls) + print(f"_abc_registry: {_abc_registry!r}", file=file) + print(f"_abc_cache: {_abc_cache!r}", file=file) + print(f"_abc_negative_cache: {_abc_negative_cache!r}", file=file) + print(f"_abc_negative_cache_version: {_abc_negative_cache_version!r}", + file=file) + + def _abc_registry_clear(cls): + """Clear the registry (for debugging or testing).""" + _reset_registry(cls) + + def _abc_caches_clear(cls): + """Clear the caches (for debugging or testing).""" + _reset_caches(cls) def update_abstractmethods(cls): diff --git a/crates/weavepy-vm/src/stdlib/queue_native.rs b/crates/weavepy-vm/src/stdlib/queue_native.rs index 7bd57e23..e5b966d6 100644 --- a/crates/weavepy-vm/src/stdlib/queue_native.rs +++ b/crates/weavepy-vm/src/stdlib/queue_native.rs @@ -37,10 +37,18 @@ fn simplequeue_put(args: &[Object]) -> Result { let dq = interp.load_attr_public(&recv, "_queue")?; let append = interp.load_attr_public(&dq, "append")?; interp.call(&append, &[item], &[], &g)?; + // The flag's test, clear and release are one step, as CPython's + // critical section makes them without the GIL: two putters that both + // saw it set would release the gate twice (`release unlocked lock`). + // (Nothing inside runs Python code; the append above, which may + // collect, stays outside.) + static GATE: std::sync::Mutex<()> = std::sync::Mutex::new(()); + let _held = GATE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); if interp.load_attr_public(&recv, "_locked")?.is_truthy() { - // Clear the flag before releasing: a second putter that runs in - // between must not release the gate twice (`release unlocked - // lock`), and the woken getter sets it again itself. + // Clear the flag before releasing: the woken getter sets it again + // itself. interp.store_attr_public(&recv, "_locked", Object::Bool(false))?; let lock = interp.load_attr_public(&recv, "_lock")?; let release = interp.load_attr_public(&lock, "release")?; diff --git a/crates/weavepy-vm/src/stdlib/random_core.rs b/crates/weavepy-vm/src/stdlib/random_core.rs index 919de177..c270fd30 100644 --- a/crates/weavepy-vm/src/stdlib/random_core.rs +++ b/crates/weavepy-vm/src/stdlib/random_core.rs @@ -139,40 +139,91 @@ impl Mt { mt } + /// Regenerate the whole block of state words (`genrand_uint32`'s + /// refill). + fn regenerate(&mut self) { + for kk in 0..(N - M) { + let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); + self.key[kk] = self.key[kk + M] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + } + for kk in (N - M)..(N - 1) { + let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); + self.key[kk] = self.key[kk + M - N] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + } + let y = (self.key[N - 1] & UPPER_MASK) | (self.key[0] & LOWER_MASK); + self.key[N - 1] = self.key[M - 1] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; + self.pos = 0; + } +} + +/// `genrand_uint32`'s output tempering. +#[inline] +fn temper(mut y: u32) -> u32 { + y ^= y >> 11; + y ^= (y << 7) & 0x9d2c_5680; + y ^= (y << 15) & 0xefc6_0000; + y ^ (y >> 18) +} + +/// The persisted state, borrowed in place: the 624 key words, then the +/// cursor, each little-endian (see [`STATE_LEN`]). +struct MtBytes<'a>(&'a mut [u8]); + +impl MtBytes<'_> { + #[inline] + fn word(&self, i: usize) -> u32 { + let mut w = [0u8; 4]; + w.copy_from_slice(&self.0[4 * i..4 * i + 4]); + u32::from_le_bytes(w) + } + + #[inline] + fn set_word(&mut self, i: usize, v: u32) { + self.0[4 * i..4 * i + 4].copy_from_slice(&v.to_le_bytes()); + } + + fn to_mt(&self) -> Mt { + let mut key = [0u32; N]; + for (i, k) in key.iter_mut().enumerate() { + *k = self.word(i); + } + Mt { + key, + pos: (self.word(N) as usize).min(N), + } + } + + fn write(&mut self, mt: &Mt) { + for (i, k) in mt.key.iter().enumerate() { + self.set_word(i, *k); + } + self.set_word(N, mt.pos as u32); + } + /// `genrand_uint32` — the raw 32-bit output stream. fn genrand_u32(&mut self) -> u32 { - if self.pos >= N { - // Regenerate the whole block. - for kk in 0..(N - M) { - let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); - self.key[kk] = self.key[kk + M] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - } - for kk in (N - M)..(N - 1) { - let y = (self.key[kk] & UPPER_MASK) | (self.key[kk + 1] & LOWER_MASK); - self.key[kk] = - self.key[kk + M - N] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - } - let y = (self.key[N - 1] & UPPER_MASK) | (self.key[0] & LOWER_MASK); - self.key[N - 1] = self.key[M - 1] ^ (y >> 1) ^ if y & 1 != 0 { MATRIX_A } else { 0 }; - self.pos = 0; + let mut pos = self.word(N) as usize; + if pos >= N { + let mut mt = self.to_mt(); + mt.regenerate(); + self.write(&mt); + pos = 0; } - let mut y = self.key[self.pos]; - self.pos += 1; - y ^= y >> 11; - y ^= (y << 7) & 0x9d2c_5680; - y ^= (y << 15) & 0xefc6_0000; - y ^ (y >> 18) + let y = self.word(pos); + self.set_word(N, pos as u32 + 1); + temper(y) } } // =================================================================== -// Instance-state plumbing. The 624-word state lives in a bytearray in -// the instance dict (so Python-level subclasses share it), the cursor -// in an int. +// Instance-state plumbing. The state lives in a bytearray in the +// instance dict (so Python-level subclasses share it): the 624 key +// words, then the cursor. The generator runs on it in place. // =================================================================== const STATE_KEY: &str = "_mt_state"; -const POS_KEY: &str = "_mt_pos"; +/// The state bytearray's length: the key words and the cursor. +const STATE_LEN: usize = (N + 1) * 4; fn self_instance(args: &[Object], what: &str) -> Result, RuntimeError> { match args.first() { @@ -181,54 +232,43 @@ fn self_instance(args: &[Object], what: &str) -> Result, RuntimeE } } -fn load_mt(inst: &Rc) -> Result { - let dict = inst.dict_cell().borrow(); - let bytes = match dict.get(&DictKey(Object::from_static(STATE_KEY))) { - Some(Object::ByteArray(b)) => b.clone(), - _ => { - drop(dict); - // Unseeded use (e.g. subclass skipping __init__): seed from - // system entropy, as CPython does at allocation time. - let mt = seed_from_entropy(); - store_mt(inst, &mt); - return Ok(mt); - } - }; - let pos = match dict.get(&DictKey(Object::from_static(POS_KEY))) { - Some(Object::Int(i)) => *i as usize, - _ => N, +/// The instance's state buffer, seeding it from system entropy when it +/// is missing (a subclass skipping `__init__`), as CPython does at +/// allocation time. +fn state_buffer(inst: &Rc) -> Rc>> { + let found = match inst + .dict_cell() + .borrow() + .get(&crate::object::StrKey(STATE_KEY)) + { + Some(Object::ByteArray(b)) if b.borrow().len() == STATE_LEN => Some(b.clone()), + _ => None, }; - let buf = bytes.borrow(); - let mut key = [0u32; N]; - for (i, chunk) in buf.as_chunks::<4>().0.iter().enumerate().take(N) { - key[i] = u32::from_le_bytes(*chunk); - } - Ok(Mt { key, pos }) + found.unwrap_or_else(|| store_mt(inst, &seed_from_entropy())) } -fn store_mt(inst: &Rc, mt: &Mt) { - let mut buf = Vec::with_capacity(N * 4); - for w in &mt.key { - buf.extend_from_slice(&w.to_le_bytes()); - } - let mut dict = inst.dict_cell().borrow_mut(); - dict.insert( +fn load_mt(inst: &Rc) -> Mt { + let buf = state_buffer(inst); + let mut bytes = buf.borrow_mut(); + MtBytes(&mut bytes).to_mt() +} + +fn store_mt(inst: &Rc, mt: &Mt) -> Rc>> { + let mut bytes = vec![0u8; STATE_LEN]; + MtBytes(&mut bytes).write(mt); + let buf = Rc::new(RefCell::new(bytes)); + inst.dict_cell().borrow_mut().insert( DictKey(Object::from_static(STATE_KEY)), - Object::ByteArray(Rc::new(RefCell::new(buf))), - ); - dict.insert( - DictKey(Object::from_static(POS_KEY)), - Object::Int(mt.pos as i64), + Object::ByteArray(buf.clone()), ); + buf } -/// Mutate-in-place fast path: run `f` against the deserialized state, -/// then persist the (changed) words back into the bytearray buffer. -fn with_mt(inst: &Rc, f: impl FnOnce(&mut Mt) -> R) -> Result { - let mut mt = load_mt(inst)?; - let r = f(&mut mt); - store_mt(inst, &mt); - Ok(r) +/// Run `f` against the instance's state, in place. +fn with_mt(inst: &Rc, f: impl FnOnce(&mut MtBytes<'_>) -> R) -> R { + let buf = state_buffer(inst); + let mut bytes = buf.borrow_mut(); + f(&mut MtBytes(&mut bytes)) } fn seed_from_entropy() -> Mt { @@ -317,7 +357,7 @@ fn random_random(args: &[Object]) -> Result { let a = mt.genrand_u32() >> 5; let b = mt.genrand_u32() >> 6; (f64::from(a) * 67_108_864.0 + f64::from(b)) * (1.0 / 9_007_199_254_740_992.0) - })?; + }); Ok(Object::Float(v)) } @@ -363,7 +403,7 @@ fn random_getrandbits(args: &[Object]) -> Result { return Ok(Object::Int(0)); } if k <= 32 { - let v = with_mt(&inst, |mt| mt.genrand_u32())? >> (32 - k as u32); + let v = with_mt(&inst, |mt| mt.genrand_u32()) >> (32 - k as u32); return Ok(Object::Int(i64::from(v))); } if (k - 1) / 32 + 1 > (isize::MAX as u64) / 4 { @@ -386,7 +426,7 @@ fn random_getrandbits(args: &[Object]) -> Result { remaining = remaining.saturating_sub(32); } out - })?; + }); let mut bytes = Vec::with_capacity(words * 4); for d in &digits { bytes.extend_from_slice(&d.to_le_bytes()); @@ -421,14 +461,14 @@ fn random_randbytes(args: &[Object]) -> Result { buf.extend_from_slice(&w[..take]); } buf - })?; + }); Ok(Object::new_bytes(out)) } /// `getstate()` → 625-tuple: the 624 state words plus the cursor. fn random_getstate(args: &[Object]) -> Result { let inst = self_instance(args, "getstate()")?; - let mt = load_mt(&inst)?; + let mt = load_mt(&inst); let mut items: Vec = mt.key.iter().map(|w| Object::Int(i64::from(*w))).collect(); items.push(Object::Int(mt.pos as i64)); Ok(Object::new_tuple(items)) diff --git a/crates/weavepy-vm/src/stdlib/select_mod.rs b/crates/weavepy-vm/src/stdlib/select_mod.rs index cf1f6e7e..707fb23b 100644 --- a/crates/weavepy-vm/src/stdlib/select_mod.rs +++ b/crates/weavepy-vm/src/stdlib/select_mod.rs @@ -1179,7 +1179,7 @@ mod kqueue_impl { /// closed, matching CPython's `kqueue_queue_traverse`/at-fork sweep). /// `Rc` is `Arc` here (the RFC 0025 shared heap), so a process-global /// registry of `Weak` is sound. - static LIVE_KQUEUES: std::sync::Mutex>> = + static LIVE_KQUEUES: std::sync::Mutex>> = std::sync::Mutex::new(Vec::new()); fn store_kqueue(inst: &Rc, fd: libc::c_int) { @@ -1205,7 +1205,7 @@ mod kqueue_impl { /// observes `kq.closed == True` and `kq.fileno()` raising. We mirror /// that here (`test_kqueue.test_fork`). pub(super) fn close_all_in_child() { - let drained: Vec> = match LIVE_KQUEUES.lock() { + let drained: Vec> = match LIVE_KQUEUES.lock() { Ok(mut reg) => reg.drain(..).collect(), // A poisoned lock can't happen across `fork` (single thread in // the child), but recover defensively rather than panic. diff --git a/crates/weavepy-vm/src/stdlib/socket_mod.rs b/crates/weavepy-vm/src/stdlib/socket_mod.rs index 736fdd10..88075a95 100644 --- a/crates/weavepy-vm/src/stdlib/socket_mod.rs +++ b/crates/weavepy-vm/src/stdlib/socket_mod.rs @@ -3908,6 +3908,8 @@ fn resolve_ipv4(name: &str, who: &str) -> Result, RuntimeError> { // SAFETY: rc == 0 guarantees a valid chain until `freeaddrinfo`. let ai = unsafe { &*cur }; if ai.ai_family == libc::AF_INET && !ai.ai_addr.is_null() { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv4Addr::from(u32::from_be(sin.sin_addr.s_addr)).to_string(); if !ips.contains(&ip) { @@ -4316,6 +4318,8 @@ fn mod_getaddrinfo(args: &[Object]) -> Result { cur = ai.ai_next; let addr_tuple = match ai.ai_family { f if f == libc::AF_INET => { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv4Addr::from(u32::from_be(sin.sin_addr.s_addr)); Object::new_tuple_array([ @@ -4324,6 +4328,8 @@ fn mod_getaddrinfo(args: &[Object]) -> Result { ]) } f if f == libc::AF_INET6 => { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin6 = unsafe { &*ai.ai_addr.cast::() }; let ip = std::net::Ipv6Addr::from(sin6.sin6_addr.s6_addr); Object::new_tuple_array([ @@ -4621,6 +4627,8 @@ fn mod_getnameinfo(args: &[Object]) -> Result { // SAFETY: rc == 0 guarantees a valid chain until `freeaddrinfo`. let ai = unsafe { &*res }; if ai.ai_family == libc::AF_INET6 { + // (getaddrinfo returns suitably aligned address storage.) + #[allow(clippy::cast_ptr_alignment)] let sin6 = unsafe { &mut *ai.ai_addr.cast::() }; sin6.sin6_flowinfo = (flowinfo as u32).to_be(); sin6.sin6_scope_id = scope_id; diff --git a/crates/weavepy-vm/src/stdlib/sys.rs b/crates/weavepy-vm/src/stdlib/sys.rs index 881b3345..0181d019 100644 --- a/crates/weavepy-vm/src/stdlib/sys.rs +++ b/crates/weavepy-vm/src/stdlib/sys.rs @@ -873,9 +873,9 @@ pub fn build(cache: &ModuleCache) -> Rc { // sharing the interpreter's host sinks, so `print()` and // direct writes via `sys.stdout.write(...)` agree. let stdout_sink: Rc> = - Rc::new(RefCell::new(std::io::stdout())); + Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stdout()))); let stderr_sink: Rc> = - Rc::new(RefCell::new(std::io::stderr())); + Rc::from_arc(std::sync::Arc::new(RefCell::new(std::io::stderr()))); // CPython's `init_sys_streams`: a standard stream whose fd is // closed at startup (e.g. spawned with `os.close(0)` in a // preexec hook) is `None`, not a broken file object diff --git a/crates/weavepy-vm/src/stdlib/thread_real.rs b/crates/weavepy-vm/src/stdlib/thread_real.rs index d8160362..06eebcda 100644 --- a/crates/weavepy-vm/src/stdlib/thread_real.rs +++ b/crates/weavepy-vm/src/stdlib/thread_real.rs @@ -708,7 +708,7 @@ fn make_lock_object(lock: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(lock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -893,7 +893,7 @@ fn make_rlock_object(rlock: Arc) -> Object { let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(rlock_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -1345,11 +1345,12 @@ fn spawn_python_worker( // deep recursion. A 64 MiB starting reserve keeps early-startup // segment churn low while letting hundreds of workers coexist. const WORKER_STACK_BYTES: usize = 64 * 1024 * 1024; // 64 MiB + // The worker clones and drops objects before it first takes the GIL. + crate::rc::revoke_refcount_bias(); let handle = std::thread::Builder::new() .name(format!("weavepy-worker-{}", synth_id)) .stack_size(WORKER_STACK_BYTES) .spawn(move || { - crate::tcache::enable_for_current_thread(); crate::vm_singletons::install_worker_thread_id(synth_id); // RFC 0040 WS4: record this worker's pthread_t so // `signal.pthread_kill(ident, sig)` can target it. @@ -1830,7 +1831,7 @@ fn make_thread_handle_object(state: Arc, ident: Object) -> Ob let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(thread_handle_type()), dict: dict.into(), - native: std::sync::OnceLock::new(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(true), slots: crate::sync::RefCell::new(crate::types::SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/stdlib/weakref_real.rs b/crates/weavepy-vm/src/stdlib/weakref_real.rs index c011ac2e..3eeda5a7 100644 --- a/crates/weavepy-vm/src/stdlib/weakref_real.rs +++ b/crates/weavepy-vm/src/stdlib/weakref_real.rs @@ -30,8 +30,8 @@ //! `True`. use crate::sync::Rc; +use crate::sync::Rc as Arc; use crate::sync::RefCell; -use std::sync::Arc; use crate::error::{type_error, value_error, RuntimeError}; use crate::import::ModuleCache; @@ -1202,8 +1202,8 @@ fn make_ref_object_with_class( let inst = Rc::new(PyInstance { class: crate::sync::RefCell::new(class), - dict, - native: std::sync::OnceLock::new(), + dict: dict.into(), + native: crate::sync::OnceBox::new(), inline_values: crate::sync::Cell::new(!fixed_wrapper), slots: crate::sync::RefCell::new(slots), hash_cache: crate::sync::CachedHash::new(None), diff --git a/crates/weavepy-vm/src/sync.rs b/crates/weavepy-vm/src/sync.rs index 894448ca..1d8b2ef0 100644 --- a/crates/weavepy-vm/src/sync.rs +++ b/crates/weavepy-vm/src/sync.rs @@ -26,7 +26,7 @@ //! The RFC 0024 surface (real lock / event / barrier primitives //! that back `threading.Lock` etc.) lives below the new aliases. -pub use crate::lazy_arc::LazyArc; +pub use crate::lazy_arc::{LazyArc, OnceBox}; use std::cell::UnsafeCell; use std::fmt; @@ -43,17 +43,9 @@ use parking_lot::{Condvar, Mutex, ReentrantMutex}; // `std::cell::RefCell`, `std::cell::Cell`. // --------------------------------------------------------------------------- -/// Drop-in replacement for [`std::rc::Rc`]. Backed by -/// [`std::sync::Arc`], so it carries the same atomic refcount -/// behaviour. Every method on `Arc` (`ptr_eq`, `clone`, `as_ptr`, -/// `strong_count`, `weak_count`, `downgrade`, `try_unwrap`, -/// `get_mut`, `into_inner`) is identical to the `Rc` API the -/// workspace already calls. -pub type Rc = std::sync::Arc; - -/// Drop-in replacement for [`std::rc::Weak`]. Backed by -/// [`std::sync::Weak`]. -pub type Weak = std::sync::Weak; +/// Drop-in replacement for [`std::rc::Rc`] and [`std::rc::Weak`], backed +/// by [`std::sync::Arc`] with biased counting (see [`crate::rc`]). +pub use crate::rc::{Rc, Weak}; /// An interior-mutability cell that's `Send + Sync` (when the /// payload is `Send`) and supports both the `RefCell` and `Cell` @@ -354,6 +346,26 @@ fn cells_unguarded() -> bool { CELLS_UNGUARDED.load(Ordering::Relaxed) } +/// Whether reference counts must be updated atomically: set whenever the +/// cell bias is revoked, and also before spawning a thread that may touch +/// objects before it registers (see [`crate::rc`]). Never cleared. +static RC_SHARED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + +/// True while one thread owns every reference count (see [`crate::rc`]). +#[cfg_attr(debug_assertions, allow(dead_code))] +#[inline(always)] +pub(crate) fn bias_held() -> bool { + !RC_SHARED.load(Ordering::Relaxed) +} + +/// Make reference counting atomic before spawning a thread that will run +/// VM code. Cell borrows keep their bias until the thread registers: the +/// spawner may hold a lock-free guard now, and it can't hand the GIL to the +/// new thread until that guard is released. +pub(crate) fn revoke_bias_for_spawn() { + RC_SHARED.store(true, Ordering::SeqCst); +} + /// The first thread to run VM code (see [`note_vm_thread`]). static FIRST: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); @@ -526,6 +538,7 @@ fn retract_sole_guard() { /// including the biased thread backing out of its own fast path — never /// waits on a guard it is itself holding. fn revoke_bias() { + RC_SHARED.store(true, Ordering::SeqCst); if CELLS_SHARED.swap(true, Ordering::SeqCst) { return; } diff --git a/crates/weavepy-vm/src/tcache.rs b/crates/weavepy-vm/src/tcache.rs deleted file mode 100644 index e7c33148..00000000 --- a/crates/weavepy-vm/src/tcache.rs +++ /dev/null @@ -1,226 +0,0 @@ -//! A thread-caching front end for the system allocator. -//! -//! The interpreter allocates and frees small blocks constantly — iterators, -//! tuples, list and dict storage, strings, boxed payloads — and a system -//! `malloc`/`free` pair costs ~20ns on macOS. [`ThreadCacheAlloc`] keeps -//! per-thread free lists of recently freed small blocks (16-byte size -//! classes up to [`MAX_SMALL`] bytes) and serves allocations of the same -//! class from them, which turns the common alloc/free pair into a handful -//! of loads and stores. -//! -//! Every block, cached or not, is a genuine system `malloc` block whose -//! usable size is at least its class size: a small request is rounded up -//! to its class before it reaches the system, and a block only enters a -//! class list when its layout maps to that class. So a block may be -//! handed back to the system (`free`, `realloc`) at any time, from any -//! thread, whatever list it last sat on. -//! -//! Caching is opt-in per thread ([`enable_for_current_thread`]): enabling -//! registers a thread-exit guard that returns the thread's cached blocks -//! to the system, so short-lived threads cannot strand memory. Threads -//! that never opt in (and a thread past its exit flush) go straight to the -//! system allocator. - -// Every cached block is at least `QUANTUM`-aligned (16 bytes, the class -// granularity), so threading the free-list link through a `*mut *mut u8` -// is aligned by construction. -#![allow(clippy::cast_ptr_alignment)] - -use std::alloc::{GlobalAlloc, Layout, System}; -use std::cell::UnsafeCell; -use std::ptr; - -/// Largest request served from the class lists. -const MAX_SMALL: usize = 512; -/// Size-class granularity (also the guaranteed alignment of a class block). -const QUANTUM: usize = 16; -const NCLASSES: usize = MAX_SMALL / QUANTUM; -/// Bytes one class list may retain. -const CLASS_BUDGET: usize = 8 * 1024; - -struct Lists { - /// Whether this thread caches (see [`enable_for_current_thread`]). - enabled: bool, - heads: [*mut u8; NCLASSES], - counts: [u32; NCLASSES], -} - -thread_local! { - /// This thread's class lists (intrusive: a free block's first word - /// links to the next). - static LISTS: UnsafeCell = const { - UnsafeCell::new(Lists { - enabled: false, - heads: [ptr::null_mut(); NCLASSES], - counts: [0; NCLASSES], - }) - }; - /// Flushes the lists when the thread exits. - static EXIT_GUARD: ExitGuard = const { ExitGuard }; -} - -struct ExitGuard; - -impl Drop for ExitGuard { - fn drop(&mut self) { - let _ = LISTS.try_with(|l| { - // SAFETY: this thread's own lists; caching goes off first, so - // the frees below go straight to the system. - let lists = unsafe { &mut *l.get() }; - lists.enabled = false; - for class in 0..NCLASSES { - let mut p = lists.heads[class]; - while !p.is_null() { - // SAFETY: every listed block is a live system block - // whose first word holds the next link. - let next = unsafe { *p.cast::<*mut u8>() }; - unsafe { libc_free(p) }; - p = next; - } - lists.heads[class] = ptr::null_mut(); - lists.counts[class] = 0; - } - }); - } -} - -extern "C" { - #[link_name = "free"] - fn libc_free(p: *mut u8); -} - -/// Turn on block caching for the calling thread (idempotent). Registers -/// the thread-exit flush first, while caching is still off, so the -/// registration's own allocations go straight to the system. -pub fn enable_for_current_thread() { - // SAFETY (both reads/writes): this thread's own lists, outside any - // allocator call. - let on = LISTS - .try_with(|l| unsafe { (*l.get()).enabled }) - .unwrap_or(true); - if on { - return; - } - let _ = EXIT_GUARD.try_with(|_| ()); - let _ = LISTS.try_with(|l| unsafe { (*l.get()).enabled = true }); -} - -/// The class of a small layout, or `None` for one the lists never hold. -#[inline] -fn class_of(layout: &Layout) -> Option { - let size = layout.size(); - if size == 0 || size > MAX_SMALL || layout.align() > QUANTUM { - return None; - } - Some((size - 1) / QUANTUM) -} - -#[inline] -fn class_layout(class: usize) -> Layout { - // SAFETY: a multiple of 16 no larger than MAX_SMALL, aligned to 16. - unsafe { Layout::from_size_align_unchecked((class + 1) * QUANTUM, QUANTUM) } -} - -/// The global allocator: [`System`] behind per-thread class lists. -#[derive(Debug, Default, Clone, Copy)] -pub struct ThreadCacheAlloc; - -unsafe impl GlobalAlloc for ThreadCacheAlloc { - #[inline] - unsafe fn alloc(&self, layout: Layout) -> *mut u8 { - if let Some(class) = class_of(&layout) { - let got = LISTS - .try_with(|l| { - // SAFETY: this thread's own lists; the allocator is - // never re-entered while they are borrowed. - let lists = unsafe { &mut *l.get() }; - if !lists.enabled { - return ptr::null_mut(); - } - let head = lists.heads[class]; - if !head.is_null() { - // SAFETY: a listed block's first word links on. - lists.heads[class] = unsafe { *head.cast::<*mut u8>() }; - lists.counts[class] -= 1; - } - head - }) - .unwrap_or(ptr::null_mut()); - if !got.is_null() { - return got; - } - // SAFETY: a valid non-zero layout. - return unsafe { System.alloc(class_layout(class)) }; - } - // SAFETY: forwarded unchanged. - unsafe { System.alloc(layout) } - } - - #[inline] - unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { - if let Some(class) = class_of(&layout) { - let kept = LISTS - .try_with(|l| { - // SAFETY: as in `alloc`. - let lists = unsafe { &mut *l.get() }; - if !lists.enabled { - return false; - } - let cap = (CLASS_BUDGET / ((class + 1) * QUANTUM)).max(8) as u32; - if lists.counts[class] >= cap { - return false; - } - // SAFETY: the block is ours now and at least 16 bytes. - unsafe { *ptr.cast::<*mut u8>() = lists.heads[class] }; - lists.heads[class] = ptr; - lists.counts[class] += 1; - true - }) - .unwrap_or(false); - if !kept { - // SAFETY: a system block allocated at its class layout. - unsafe { System.dealloc(ptr, class_layout(class)) }; - } - return; - } - // SAFETY: forwarded unchanged. - unsafe { System.dealloc(ptr, layout) } - } - - #[inline] - unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { - if layout.align() > QUANTUM { - // The generic path: fresh block, copy, release. - // SAFETY: the caller's layout invariants hold for `new_layout`. - let new_layout = unsafe { Layout::from_size_align_unchecked(new_size, layout.align()) }; - let new = unsafe { self.alloc(new_layout) }; - if !new.is_null() { - unsafe { - ptr::copy_nonoverlapping(ptr, new, layout.size().min(new_size)); - self.dealloc(ptr, layout); - } - } - return new; - } - let old_class = class_of(&layout); - // SAFETY: `new_size` is non-zero and fits the caller's layout rules. - let new_layout = unsafe { Layout::from_size_align_unchecked(new_size, layout.align()) }; - let new_class = class_of(&new_layout); - if old_class.is_some() && old_class == new_class { - // Same class: the block already has the room. - return ptr; - } - // Resize the underlying system block to what the new layout's - // eventual `dealloc` will assume (its class size, when small). - let target = match new_class { - Some(class) => (class + 1) * QUANTUM, - None => new_size, - }; - let old_system = match old_class { - Some(class) => class_layout(class), - None => layout, - }; - // SAFETY: `ptr` is a live system block of `old_system`. - unsafe { System.realloc(ptr, old_system, target) } - } -} diff --git a/crates/weavepy-vm/src/tier2.rs b/crates/weavepy-vm/src/tier2.rs index ef1086df..ae5f98b0 100644 --- a/crates/weavepy-vm/src/tier2.rs +++ b/crates/weavepy-vm/src/tier2.rs @@ -203,16 +203,17 @@ impl MathGuard { /// probes are skipped (a resume or entry then costs two stamp reads). struct GuardSnapshot { entries: Vec<(String, Object)>, - /// `(globals id, globals stamp, builtins id, builtins stamp)` of the - /// last full validation that held; all-zero until one has. - last_ok: std::cell::Cell<(usize, u64, usize, u64)>, + /// `(globals id, globals stamp, builtins id, builtins stamp, global + /// value epoch)` of the last full validation that held; all-zero + /// until one has. + last_ok: std::cell::Cell<(usize, u64, usize, u64, u64)>, } impl GuardSnapshot { fn new(entries: Vec<(String, Object)>) -> Self { Self { entries, - last_ok: std::cell::Cell::new((0, 0, 0, 0)), + last_ok: std::cell::Cell::new((0, 0, 0, 0, 0)), } } } @@ -327,6 +328,10 @@ struct Artifacts { /// generator — possibly to another thread — so buffer-layout /// identity needs a process-wide id. compile_id: u64, + /// Native-to-native entries of this compilation, and the interpreter + /// round-trips they made (see [`note_callee_exit`]). + callee_entries: Cell, + callee_roundtrips: Cell, } /// RFC 0073 WS4 — source of [`Artifacts::compile_id`]. @@ -422,6 +427,13 @@ pub(crate) const DEOPT_BUDGET: u32 = 64; /// Retire it to tier-1. pub(crate) const GENERIC_CALL_RETIRE_RATIO: u32 = 4; +/// [`GENERIC_CALL_RETIRE_RATIO`] for native-to-native entries. Such a +/// callee is usually loop-free, so native code saves it a few dozen +/// nanoseconds per activation while one interpreter call from it costs +/// several hundred more than the interpreter's inline call (measured on +/// deltablue's `execute` / `input` / `output` methods: 4x slower compiled). +pub(crate) const CALLEE_ROUNDTRIP_RETIRE_RATIO: u32 = 1; + /// Retire a compiled code object — and deopt the running activation — /// once one activation has made this many interpreter round-trips /// through the call helpers. Each such call pays activation-shell @@ -567,6 +579,61 @@ struct JitState { static JIT_PROCESS_GATE: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(0); /// The `WEAVEPY_JIT` / free-threading verdict shared by every thread. +/// Warm the code generator on a background thread (see +/// [`prewarm_codegen`]), once per process: when the CLI starts a program +/// file or module, and otherwise when the first code object is halfway +/// to its compile threshold (so `-c pass` doesn't pay for it). +pub(crate) fn spawn_codegen_prewarm_if_enabled() { + if jit_enabled_by_config() { + spawn_codegen_prewarm(); + } +} + +fn spawn_codegen_prewarm() { + static SPAWNED: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); + if SPAWNED.swap(true, std::sync::atomic::Ordering::Relaxed) { + return; + } + let _ = std::thread::Builder::new() + .name("weavepy-jit-warm".to_owned()) + .spawn(prewarm_codegen); +} + +/// Compile, on a throwaway engine, a small counted loop of the shape hot +/// code takes, and discard it. A process's first compile otherwise pays +/// the code generator's cold start (its code paged in and its tables +/// built: about 0.7 ms on the development host, most of the first +/// compile) on the thread that needs the compiled code. Touches no +/// interpreter state. +fn prewarm_codegen() { + const SOURCE: &str = + "def f(n):\n t = 0\n for i in range(n):\n t = t + i * 2\n return t\n"; + let _ = std::panic::catch_unwind(|| { + let Ok(module) = weavepy_parser::parse_module(SOURCE) else { + return; + }; + let Ok(code) = weavepy_compiler::compile_module(&module) else { + return; + }; + let Some(f) = code.constants.iter().find_map(|c| match c { + weavepy_compiler::Constant::Code(f) => Some(f.clone()), + _ => None, + }) else { + return; + }; + let Some(mut engine) = JitEngine::new() else { + return; + }; + let _ = engine.compile(&f, &mut |name| { + if name == "range" { + weavepy_jit::ResolvedGlobal::RangeBuiltin + } else { + weavepy_jit::ResolvedGlobal::Opaque + } + }); + }); +} + fn jit_enabled_by_config() -> bool { // RFC 0067 WS3 — the tier-2 JIT is on by default; `WEAVEPY_JIT=0` // (or `off`, or an empty value) restores the pure interpreter. @@ -621,6 +688,31 @@ thread_local! { pub(crate) const LEAN_WARM_COMPILE_THRESHOLD_CAP: u32 = 24; #[inline] +/// Credit `n` activations a frameless path ran for `code` to its tier-2 +/// warm-up counter (the framed entries that count otherwise never happen +/// for them). Returns whether the next framed or lean entry should +/// compile it: a compile is due, or no entry exists yet to count in. +pub(crate) fn note_frameless_calls(code: &CodeObject, n: u32) -> bool { + JIT.with(|cell| { + let Ok(mut st) = cell.try_borrow_mut() else { + return false; + }; + if !st.enabled { + return false; + } + let threshold = st.threshold; + match st.cache.get_mut(&std::ptr::from_ref(code)) { + None => true, + Some(entry) if matches!(entry.tier, Tier::Cold) => { + entry.counter = entry.counter.saturating_add(n); + let next = entry.counter.saturating_add(1); + next >= threshold && compile_allowed(next, threshold) + } + Some(_) => false, + } + }) +} + pub(crate) fn lean_warm_at() -> u32 { LEAN_WARM_AT .try_with(std::cell::Cell::get) @@ -773,6 +865,7 @@ impl JitState { ); // RFC 0067 WS2 — the eval-breaker poll for native loop headers. weavepy_jit::register_poll_helper(wpjit_poll); + weavepy_jit::register_self_call_helpers(wpjit_self_enter, wpjit_self_exit, wpjit_self_slow); // RFC 0069 WS1 — the guarded method-call lane. weavepy_jit::register_call_method_helper(wpjit_call_method); // RFC 0073 WS3 — the native `str`-method lane. @@ -868,6 +961,9 @@ impl JitState { Tier::NotJitable => return None, Tier::Cold => { entry.counter += 1; + if entry.counter == self.threshold / 2 && self.engine.is_none() { + spawn_codegen_prewarm(); + } if entry.counter < self.threshold || !compile_allowed(entry.counter, self.threshold) { @@ -1122,7 +1218,21 @@ impl JitState { let t0 = std::env::var_os("WEAVEPY_JIT_TRACE") .is_some() .then(std::time::Instant::now); - let r = engine.compile_frame(code, &mut classify, &mut jit_probes); + // A callee already compiled as a guard-free scalar leaf is + // entered directly by native code (its identity is guarded + // with the callee table like any burned-in callee). + let mut direct = |token: u32| { + let callees = callees.borrow(); + let (Object::Function(_), fcode) = callees.get(token as usize)? else { + return None; + }; + let k = Rc::as_ptr(fcode).cast::(); + match &cache_ref.get(&k)?.tier { + Tier::Compiled(a) => a.cf.direct_leaf(fcode.arg_count), + _ => None, + } + }; + let r = engine.compile_frame_direct(code, &mut classify, &mut jit_probes, &mut direct); if let Some(t0) = t0 { eprintln!("jit compile-time {:?} {:?}", code.name, t0.elapsed()); } @@ -1228,6 +1338,8 @@ impl JitState { math: StdRc::new(math_tbl), compile_id: NEXT_COMPILE_ID .fetch_add(1, std::sync::atomic::Ordering::Relaxed), + callee_entries: Cell::new(0), + callee_roundtrips: Cell::new(0), }); // RFC 0067 WS1 — a fresh compile can flip a // `None` native-callee slot in *other* frames' @@ -1743,10 +1855,7 @@ fn probe_class_ctor_shape( let bt = crate::builtin_types::builtin_types(); // `type` subclasses (metaclasses) construct *classes* through the // three-argument form, never plain instances. - if cls.flags.is_builtin - || cls.is_subclass_of(&bt.type_) - || !Rc::ptr_eq(&cls.metaclass_or_type(), &bt.type_) - { + if cls.flags.is_builtin || cls.is_subclass_of(&bt.type_) || !cls.metaclass_is_type() { return None; } let plan = interp.instance_plan(cls); @@ -2105,6 +2214,11 @@ fn drain_activation_pins(interp: &mut super::Interpreter, pins: &mut PinTable) - Pin::List(list, _) => Object::List(list), Pin::Obj(o) => o, }; + // The common pin, a receiver or argument others still hold, + // releases plainly (the grades below would conclude the same). + if crate::gc_trace::drop_survives_plainly(&o) { + continue; + } if super::Interpreter::local_needs_prompt_reap(&o) && super::Interpreter::looks_reapable_temporary(&o) { @@ -2129,6 +2243,9 @@ fn defer_activation_pins(pins: &mut PinTable) { Pin::List(list, _) => Object::List(list), Pin::Obj(obj) => obj, }; + if crate::gc_trace::drop_survives_plainly(&obj) { + continue; + } if super::Interpreter::local_needs_prompt_reap(&obj) && super::Interpreter::looks_reapable_temporary(&obj) { @@ -2150,7 +2267,7 @@ fn unpack(bits: u64, tag: u32) -> Object { SlotTag::Bool => Object::Bool(bits != 0), // RFC 0069 WS1 — the `None` singleton (a `ReturnNone` exit). SlotTag::None => Object::None, - SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => Object::None, + SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin | SlotTag::Default => Object::None, } } @@ -2745,20 +2862,12 @@ fn attr_fingerprint_obj( // RFC 0071 WS2 — a new-key store has no current value by // definition: the `Unknown` lane tells the analyzer to type the // site from the stored value instead. - let slot_val; - let dict; - let v: &Object = match storage { + let current = match storage { AttrStorage::NewKey => return Some((JitType::Unknown, ver, storage)), - AttrStorage::Slot(_) => { - slot_val = inst.slot_get(name)?; - &slot_val - } - AttrStorage::Indexed(key_idx) => { - dict = inst.dict.get()?.borrow(); - let (_, v) = dict.get_index(key_idx as usize)?; - v - } + AttrStorage::Slot(_) => inst.slot_get(name)?, + AttrStorage::Indexed(key_idx) => inst.attr_index_map(key_idx as usize, |_, v| v.clone())?, }; + let v = ¤t; // RFC 0070 WS1 — instance- or `None`-valued attributes take the // nullable object lane (loads pin the value at runtime; stores // resolve the staged pin); RFC 0071 WS6 — exact `str`/`bytes` @@ -2865,7 +2974,8 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F } let key = Rc::as_ptr(&frame.code).cast::(); let threshold = st.threshold; - let importing = phase == CompilationPhase::Import; + // Embedders that never report start-up finished still compile, + // after sustained work. let warm = if phase == CompilationPhase::Normal && STARTUP_DONE.load(std::sync::atomic::Ordering::Relaxed) { @@ -2888,11 +2998,16 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F code: frame.code.clone(), }); if matches!(entry.tier, Tier::Cold) { - if importing { - // The lean path has no ordinary frame-entry counter. Account - // for the interval just completed, then permit another one. - // Resetting keeps the equality checkpoint reachable after the - // import exits, including when pure-leaf calls can skip frames. + if phase != CompilationPhase::Normal { + // A loop-free body gains nothing from native code until + // compiled callers exist to take its direct lanes, while + // compiling one during start-up or an import costs time and + // memory the program may never recover (`ABCMeta.register` + // while `_collections_abc` loads). The lean path has no + // ordinary frame-entry counter: account for the interval + // just completed and count another, so only sustained work + // compiles (the checkpoint stays reachable afterwards, even + // when pure-leaf calls skip frames). let interval = lean_warm_at(); entry.counter = entry.counter.saturating_add(interval); if entry.counter < interval.saturating_mul(16) { @@ -2901,7 +3016,7 @@ pub(crate) fn warm_compile(interp: &mut super::Interpreter, frame: &mut super::F } } // Preserve the earlier lean warm point relative to frame/loop - // hotness, including the escape hatch for unreported startup. + // hotness. entry.counter = entry.counter.max(warm); } let interp_ref: &super::Interpreter = interp; @@ -3041,6 +3156,11 @@ struct CallCtx { /// re-entrancy pattern as `vm_singletons::publish_interpreter_ptr`). interp: *mut super::Interpreter, callees: StdRc, + /// The running compilation's frame layout, owned by whoever entered + /// this activation. A direct self call shares its caller's context, + /// so its deopt rebuild reads the layout here rather than through + /// the tier cache, which may retire the code mid-recursion. + cf: *const CompiledFrame, guard_snapshot: StdRc, /// The caller frame's namespaces, for post-call guard revalidation /// (the caller `Frame` itself is mutably borrowed across the native @@ -3140,6 +3260,12 @@ struct CallCtx { /// the exact chain). `None` for a framed entry — its `Frame`'s /// shell is already on the spine. frameless_code: Option>, + /// The last dynamic call's resolved native callee, keyed by the + /// function and code identities and the method form: a site calling + /// the same function again (a stored bound method, a function held in + /// a local) re-enters without re-resolving or rebuilding the handle. + /// Taken out for the call's duration and put back after. + dyn_callee: Option<(usize, usize, bool, NativeCallee)>, } impl CallCtx { @@ -3186,6 +3312,8 @@ fn guards_hold( unsafe { (*globals.as_ptr()).mutation_stamp() }, Rc::as_ptr(builtins) as usize, unsafe { (*builtins.as_ptr()).mutation_stamp() }, + // A rebinding in place leaves the stamps alone (see `STORE_GLOBAL`). + crate::object::global_value_epoch(), ); if guard_snapshot.last_ok.get() != key || interp.globals_missing_any.get() { for (name, expected) in guard_snapshot.entries.iter() { @@ -3676,10 +3804,11 @@ fn scalar_field_update_plan( { return None; } - let AttrStorage::Indexed(index) = guard.storage else { - return None; + let current = match guard.storage { + AttrStorage::Indexed(index) => (guard.ver, index, false), + AttrStorage::Slot(index) => (guard.ver, index, true), + AttrStorage::NewKey => return None, }; - let current = (guard.ver, index); if fingerprint.is_some_and(|old| old != current) { return None; } @@ -3790,8 +3919,10 @@ unsafe fn native_scalar_field_update( } let plan = nc.scalar_update.as_deref()?; let guard = nc.attr_guards.get(plan.store_token)?; - let AttrStorage::Indexed(index) = guard.storage else { - return None; + let (index, slot_storage) = match guard.storage { + AttrStorage::Indexed(index) => (index, false), + AttrStorage::Slot(index) => (index, true), + AttrStorage::NewKey => return None, }; let Object::Instance(inst) = receiver else { return None; @@ -3824,18 +3955,32 @@ unsafe fn native_scalar_field_update( return None; } }; + if slot_storage { + // SAFETY: as for the dictionary below. + let slots = unsafe { inst.slots.peek_mut() }?; + let (key, old) = slots.get_index(index as usize)?; + if !key_is(key, &guard.name) { + return None; + } + let Object::Int(old) = old else { + return None; + }; + let value = old.checked_add(increment)?; + let (_, slot) = slots.get_index_mut(index as usize)?; + // An exact integer owns no destructor or GC edge. + *slot = Object::Int(value); + return Some(value); + } // SAFETY: no callback, allocation, or Python execution can overlap this // exclusive view. A shared cell is rejected by peek_mut. - let dict = unsafe { inst.dict.get()?.peek_mut() }?; - let (key, old) = dict.get_index(index as usize)?; + let (key, slot) = unsafe { inst.attr_peek_index_mut(index as usize, true) }?; if !key_is(key, &guard.name) { return None; } - let Object::Int(old) = old else { + let Object::Int(old) = slot else { return None; }; let value = old.checked_add(increment)?; - let (_, slot) = dict.map_mut_unstamped().get_index_mut(index as usize)?; // Exact integers own no destructor or GC edge. Existing-key replacement // preserves key stamps. No failing operation follows the completed store. *slot = Object::Int(value); @@ -3844,6 +3989,110 @@ unsafe fn native_scalar_field_update( Some(value) } +/// Bind the slots a keyword call skipped (tagged [`SlotTag::Default`] +/// under [`weavepy_jit::CALL_GAPS`]) to the callee's current +/// `__defaults__` when those are scalars, so the call is a positional +/// prefix again (whose trailing window the lanes below bind). Returns +/// the argument count to call with and whether every skipped slot is +/// bound (`false` leaves the call to [`call_py_with_gaps`]). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]; the buffers are the compiled +/// frame's `max_call_args` wide. +unsafe fn bind_call_defaults( + jf: &mut JitFrame, + ctx: &CallCtx, + token: u32, + raw: u32, +) -> (u32, bool) { + let gapped = raw & weavepy_jit::CALL_GAPS != 0; + let argc = raw & !weavepy_jit::CALL_GAPS; + let Some((Object::Function(f), code)) = ctx.callees.get(token as usize) else { + return (argc, !gapped); + }; + if !gapped { + return (argc, true); + } + // A reassigned `__defaults__` lives in the function's slots: the + // generic binder reads it. + if f.defaults_maybe_overridden() { + return (argc, false); + } + let npos = code.arg_count as usize; + // SAFETY: the activation's compiled frame outlives its calls. + let cap = unsafe { ctx.cf.as_ref() }.map_or(0, |cf| cf.max_call_args as usize); + let first = npos.saturating_sub(f.defaults.len()); + let scalar = |j: usize| -> Option<(u64, SlotTag)> { + let v = f.defaults.get(j.checked_sub(first)?)?; + match v { + Object::Int(i) => Some((*i as u64, SlotTag::Int)), + Object::Float(x) => Some((x.to_bits(), SlotTag::Float)), + Object::Bool(b) => Some((u64::from(*b), SlotTag::Bool)), + Object::None => Some((u64::MAX, SlotTag::ObjPin)), + _ => None, + } + }; + let write = |jf: &mut JitFrame, j: usize, (bits, tag): (u64, SlotTag)| { + // SAFETY: `j` is below the buffers' width (checked by callers). + unsafe { + *jf.call_args.add(j) = bits; + *jf.call_tags.add(j) = tag as u32; + } + }; + let mut bound = true; + for j in 0..(argc as usize).min(cap).min(npos) { + // SAFETY: native code wrote `argc` tags. + if unsafe { *jf.call_tags.add(j) } == SlotTag::Default as u32 { + match scalar(j) { + Some(v) => write(jf, j, v), + None => bound = false, + } + } + } + (argc, bound) +} + +/// [`wpjit_call_py`] for a keyword call whose skipped defaulted slot +/// could not be bound natively (a non-scalar default, or none any +/// more): the generic call, with the prefix before the first skipped +/// slot positional and the rest by name, binds (or rejects) it exactly +/// as the interpreter would. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe fn call_py_with_gaps( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &mut super::Interpreter, + token: u32, + argc: u32, + expect_tag: u32, +) -> i64 { + ctx.dirty = true; + let (callee, code) = ctx.callees[token as usize].clone(); + let mut args: Vec = Vec::new(); + let mut kwargs: Vec<(String, Object)> = Vec::new(); + for j in 0..argc as usize { + // SAFETY: native code wrote `argc` entries. + let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; + if tag == SlotTag::Default as u32 { + continue; + } + let v = unpack_pins(bits, tag, &ctx.pins); + if kwargs.is_empty() && args.len() == j { + args.push(v); + } else if let Some(name) = code.varnames.get(j) { + kwargs.push((name.clone(), v)); + } + } + let res = call_with_activation_shell(interp, ctx, jf, |i| { + i.call(&callee, &args, &kwargs, &ctx.globals) + }); + finish_interp_call(jf, ctx, interp, res, expect_tag) +} + /// RFC 0067 WS1 — attempt a native-to-native call for one marshaled /// `CallPy` site. Returns `Some(CallStatus as i64)` when the call /// completed through the native path (including via a materialized @@ -4224,6 +4473,7 @@ unsafe fn try_native_call( if !c.obj_global_pins.is_empty() { c.obj_global_pins.clear(); } + c.cf = StdRc::as_ptr(&nc.cf); c.dirty = false; c.interp_calls = 0; c.dyn_py_calls = 0; @@ -4259,6 +4509,7 @@ unsafe fn try_native_call( Box::new(CallCtx { interp: ctx.interp, callees: nc.callees.clone(), + cf: StdRc::as_ptr(&nc.cf), guard_snapshot: nc.snap.clone(), globals: nc.func.globals.clone(), builtins: nc.func.builtins.clone(), @@ -4289,6 +4540,7 @@ unsafe fn try_native_call( // The native call lanes push no interpreter frame for the // callee — keep it observable to callee-side stack walkers. frameless_code: Some(nc.code.clone()), + dyn_callee: None, }) }; child.interp = ctx.interp; @@ -4388,6 +4640,7 @@ unsafe fn try_native_call( } } }; + note_callee_exit(&nc.art, &nc.code, nctx); if !inline_bufs { put_u64(u64_buf); put_u32(u32_buf); @@ -4485,9 +4738,11 @@ unsafe fn try_native_call( SlotTag::Int => JitType::Int, SlotTag::Float => JitType::Float, SlotTag::Bool => JitType::Bool, - SlotTag::None | SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => { - JitType::Unknown - } + SlotTag::None + | SlotTag::Boxed + | SlotTag::ListPin + | SlotTag::ObjPin + | SlotTag::Default => JitType::Unknown, }; match pack(&v, expect) { Some(bits) if guards_ok => { @@ -4602,19 +4857,6 @@ fn finish_deopted_callee( njf: &JitFrame, raised: Option, ) -> Result { - let code = &nc.code; - let n_real = code.varnames.len(); - let mut locals_v: Vec = Vec::with_capacity(n_real); - for slot in 0..n_real { - match nc.cf.local_types.get(slot).copied().flatten() { - Some(ty) => locals_v.push(unpack_ty( - locals_buf.get(slot).copied().unwrap_or(0), - ty, - &nctx.pins, - )), - None => locals_v.push(Object::Unbound), - } - } let entry = CompiledEntry { cf: nc.cf.clone(), guard_snapshot: nc.snap.clone(), @@ -4625,17 +4867,49 @@ fn finish_deopted_callee( math: nc.math.clone(), native: None, method_native: None, - // Synthetic entry, only for the stack rebuild below; `0` is - // never a real compile id, so nothing can park against it. + // Synthetic entry, only for the stack rebuild; `0` is never a + // real compile id, so nothing can park against it. compile_id: 0, }; + finish_deopted( + interp, &nc.code, &nc.func, &entry, nctx, locals_buf, spill, tags, njf, raised, + ) +} + +/// [`finish_deopted_callee`] for an activation described by `entry` +/// (the callee's own tables), `code` and `func`. +#[allow(clippy::too_many_arguments)] +fn finish_deopted( + interp: &mut super::Interpreter, + code: &Rc, + func: &PyFunction, + entry: &CompiledEntry, + nctx: &mut CallCtx, + locals_buf: &[u64], + spill: &[u64], + tags: &[u32], + njf: &JitFrame, + raised: Option, +) -> Result { + let n_real = code.varnames.len(); + let mut locals_v: Vec = Vec::with_capacity(n_real); + for slot in 0..n_real { + match entry.cf.local_types.get(slot).copied().flatten() { + Some(ty) => locals_v.push(unpack_ty( + locals_buf.get(slot).copied().unwrap_or(0), + ty, + &nctx.pins, + )), + None => locals_v.push(Object::Unbound), + } + } let mut frame = super::Frame { code: code.clone(), locals: Rc::new(GilRefCell::new(locals_v)), cells: crate::object::empty_cells(), stack: Vec::new(), - globals: nc.func.globals.clone(), - builtins: nc.func.builtins.clone(), + globals: func.globals.clone(), + builtins: func.builtins.clone(), builtins_obj: None, class_namespace: None, class_namespace_obj: None, @@ -4649,6 +4923,7 @@ fn finish_deopted_callee( pending_lasti: None, suppress_call_event: true, gen_first_resume: false, + sent_consumed: false, shell_cache: None, parked_native: None, }; @@ -4660,7 +4935,7 @@ fn finish_deopted_callee( nctx.parked.take() }; rebuild_stack( - interp, &mut frame, &entry, locals_buf, spill, tags, njf, &nctx.pins, parked, + interp, &mut frame, entry, locals_buf, spill, tags, njf, &nctx.pins, parked, ); if raised.is_some() { // As though the raising CALL just executed: pc points past it @@ -4749,6 +5024,48 @@ unsafe extern "C" fn wpjit_call_py( // while the helper runs; this is the only live path to it. let interp = unsafe { &mut *ctx.interp }; + // Defaulted parameters the site didn't pass: bound here, so every + // lane below sees a full-arity call. + // SAFETY: per the function contract. + let (argc, bound) = unsafe { bind_call_defaults(jf, ctx, token, argc) }; + if !bound { + // SAFETY: as above. + return unsafe { call_py_with_gaps(jf, ctx, interp, token, argc, expect_tag) }; + } + + // A pure-leaf callee evaluates frameless (see `try_pure_leaf_call`), + // unless its compiled scalar leaf runs it natively (cheaper still). + let native_scalar = ctx + .native + .as_deref() + .and_then(|t| t.get(token as usize)) + .and_then(Option::as_ref) + .is_some_and(|nc| nc.ctor.is_none() && nc.cf.is_scalar_leaf()); + let maybe_pure = !native_scalar + && ctx + .callees + .get(token as usize) + .is_some_and(|(_, code)| code.jit_hint.pure_leaf() != Some(false)); + let callees = if maybe_pure { + Some(StdRc::clone(&ctx.callees)) + } else { + None + }; + if let Some((Object::Function(f), code)) = callees.as_ref().and_then(|c| c.get(token as usize)) + { + // SAFETY: GIL-serialized raw read of the function's code cell; + // only compared. + if std::ptr::eq(unsafe { Rc::as_ptr(&*f.code.as_ptr()) }, Rc::as_ptr(code)) { + // SAFETY: `argc` marshaled entries are live (the function + // contract). + if let Some(status) = + unsafe { try_pure_leaf_call(jf, ctx, interp, f, code, None, argc, expect_tag) } + { + return status; + } + } + } + // RFC 0067 WS1 — the native-to-native fast path: a compiled, // shape-eligible callee is entered directly with the marshaled // scalars, skipping the interpreter frame entirely. (The table @@ -4851,9 +5168,11 @@ unsafe extern "C" fn wpjit_call_py( // Other pin-lane call results are rejected at // emission; `Unknown` never packs, forcing the // boxed path. - SlotTag::None | SlotTag::Boxed | SlotTag::ListPin | SlotTag::ObjPin => { - JitType::Unknown - } + SlotTag::None + | SlotTag::Boxed + | SlotTag::ListPin + | SlotTag::ObjPin + | SlotTag::Default => JitType::Unknown, }; if let Some(bits) = pack(&v, expect) { jf.ret_bits = bits; @@ -4903,37 +5222,7 @@ fn finish_interp_call( &ctx.math, ); if still_valid { - match SlotTag::from_raw(expect_tag) { - // The procedure lane: nothing to write back, the - // compiled code pushes no result. - SlotTag::None => { - if matches!(v, Object::None) { - return CallStatus::Ok as i64; - } - } - SlotTag::Int | SlotTag::Float | SlotTag::Bool => { - let expect = match SlotTag::from_raw(expect_tag) { - SlotTag::Int => JitType::Int, - SlotTag::Float => JitType::Float, - _ => JitType::Bool, - }; - if let Some(bits) = pack(&v, expect) { - jf.ret_bits = bits; - jf.ret_tag = expect_tag; - return CallStatus::Ok as i64; - } - } - // RFC 0071 WS1 — an object-lane result pins into - // this activation's table. - SlotTag::ObjPin => { - if let Some(bits) = obj_ret_bits(&v, &mut ctx.pins) { - jf.ret_bits = bits; - jf.ret_tag = expect_tag; - return CallStatus::Ok as i64; - } - } - SlotTag::Boxed | SlotTag::ListPin => {} - } + return deliver_call_result(jf, ctx, v, expect_tag); } ctx.parked = Some(v); CallStatus::Boxed as i64 @@ -4941,6 +5230,337 @@ fn finish_interp_call( } } +/// Hand a completed call's result `v` to the compiled caller in its +/// `expect_tag` lane, or park it (`Boxed`: the caller deopts after the +/// call) when the lane cannot carry it. +fn deliver_call_result(jf: &mut JitFrame, ctx: &mut CallCtx, v: Object, expect_tag: u32) -> i64 { + match SlotTag::from_raw(expect_tag) { + // The procedure lane: nothing to write back, the compiled code + // pushes no result. + SlotTag::None => { + if matches!(v, Object::None) { + return CallStatus::Ok as i64; + } + } + SlotTag::Int | SlotTag::Float | SlotTag::Bool => { + let expect = match SlotTag::from_raw(expect_tag) { + SlotTag::Int => JitType::Int, + SlotTag::Float => JitType::Float, + _ => JitType::Bool, + }; + if let Some(bits) = pack(&v, expect) { + jf.ret_bits = bits; + jf.ret_tag = expect_tag; + return CallStatus::Ok as i64; + } + } + // RFC 0071 WS1 — an object-lane result pins into this + // activation's table. + SlotTag::ObjPin => { + if let Some(bits) = obj_ret_bits(&v, &mut ctx.pins) { + jf.ret_bits = bits; + jf.ret_tag = expect_tag; + return CallStatus::Ok as i64; + } + } + SlotTag::Boxed | SlotTag::ListPin | SlotTag::Default => {} + } + ctx.parked = Some(v); + CallStatus::Boxed as i64 +} + +/// Evaluate a pure-leaf callee (see `code_is_pure_leaf`) frameless, as +/// the interpreter's core loop does: `recv` (a method's receiver) then +/// the `argc` marshaled arguments bind its parameters exactly. A native +/// activation would cost several times the body; `None` (nothing ran) +/// leaves the call to the ordinary paths. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]: `argc` marshal entries are live. +#[inline(always)] +#[allow(clippy::too_many_arguments)] +unsafe fn try_pure_leaf_call( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &super::Interpreter, + func: &crate::object::PyFunction, + code: &CodeObject, + recv: Option<*const Object>, + argc: u32, + expect_tag: u32, +) -> Option { + // The common native callee is no pure leaf: one relaxed load decides. + if code.jit_hint.pure_leaf() == Some(false) + || code.arg_count != argc + u32::from(recv.is_some()) + { + return None; + } + // SAFETY: the caller's contract. + unsafe { pure_leaf_call(jf, ctx, interp, func, code, recv, argc, expect_tag) } +} + +/// [`try_pure_leaf_call`]'s evaluation, out of line (its argument +/// buffers would otherwise widen every native call helper's frame). +/// +/// # Safety +/// +/// As [`try_pure_leaf_call`]. +#[inline(never)] +#[allow(clippy::too_many_arguments)] +unsafe fn pure_leaf_call( + jf: &mut JitFrame, + ctx: &mut CallCtx, + interp: &super::Interpreter, + func: &crate::object::PyFunction, + code: &CodeObject, + recv: Option<*const Object>, + argc: u32, + expect_tag: u32, +) -> Option { + const MAX: usize = 8; + let offset = usize::from(recv.is_some()); + let n = argc as usize + offset; + if n > MAX + || code.arg_count as usize != n + || !code + .jit_hint + .pure_leaf() + .unwrap_or_else(|| crate::code_is_pure_leaf_pub(code)) + || crate::hot_gates::load() != 0 + || crate::trace::any_observers_active() + { + return None; + } + // Only the marshaled arguments need owned values (a scalar lane + // becomes its `Object`, a pin names the pinned one); the receiver is + // borrowed where it lives. + let mut owned = [const { std::mem::MaybeUninit::::uninit() }; MAX]; + /// Drops the first `.1` values at `.0` (the written arguments). + struct Owned(*mut Object, usize); + impl Drop for Owned { + fn drop(&mut self) { + for j in 0..self.1 { + // SAFETY: the first `self.1` values were written below. + unsafe { std::ptr::drop_in_place(self.0.add(j)) }; + } + } + } + let mut written = Owned(owned.as_mut_ptr().cast::(), 0); + let mut ptrs: [*const Object; MAX] = [std::ptr::null(); MAX]; + if let Some(r) = recv { + ptrs[0] = r; + } + for j in 0..argc as usize { + // SAFETY: the caller's contract — `argc` marshaled entries. + let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; + // SAFETY: `j < argc <= MAX`; the slot is uninitialized until now. + unsafe { written.0.add(j).write(unpack_pins(bits, tag, &ctx.pins)) }; + written.1 = j + 1; + // SAFETY: as above. + ptrs[offset + j] = unsafe { written.0.add(j) }; + } + let v = interp.pure_leaf_eval::(code, func, &ptrs[..n])?; + // A call served without an interpreter frame, like a native one. + native_stat(|s| s.calls.set(s.calls.get() + 1)); + Some(deliver_call_result(jf, ctx, v, expect_tag)) +} + +/// The direct self-call enter helper (see `weavepy_jit::SelfEnterHelper`): +/// the per-call work `try_native_call` does for a callee, reduced to what +/// a pin-free activation of the caller's own code needs — a GIL +/// checkpoint, the observer gate, and the recursion tick (checked +/// against the limit, and against the native stack's headroom every few +/// levels: the ordinary path grows the stack, this one cannot). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_self_enter(frame: *mut JitFrame) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + // SAFETY: the `&mut Interpreter` that entered native code is dormant + // while the helper runs. + let interp = unsafe { &mut *ctx.interp }; + interp.gil_countdown = interp.gil_countdown.wrapping_sub(1); + if interp.gil_countdown == 0 { + interp.gil_countdown = crate::gil::GIL_CHECK_INTERVAL; + crate::gil::yield_checkpoint(); + } + if crate::hot_gates::load() != 0 || crate::trace::any_observers_active() { + return 1; + } + // SAFETY: this thread's own depth cell (see `CallCtx::depth_cell`). + let depth = unsafe { &*ctx.depth_cell }; + let n = depth.get() + 1; + if n > crate::recursion::recursion_limit() + || (n % 8 == 0 && stacker::remaining_stack().is_some_and(|r| r < 256 * 1024)) + { + return 1; + } + depth.set(n); + native_stat(|s| s.calls.set(s.calls.get() + 1)); + 0 +} + +/// Release [`wpjit_self_enter`]'s recursion tick. +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]. +unsafe extern "C" fn wpjit_self_exit(frame: *mut JitFrame) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &*frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &*jf.ctx.cast::() }; + // SAFETY: as in `wpjit_self_enter`. + let depth = unsafe { &*ctx.depth_cell }; + depth.set(depth.get().saturating_sub(1)); + 0 +} + +/// Finish a direct self call whose callee did not return (see +/// `weavepy_jit::SelfSlowHelper`): exactly `try_native_call`'s deopt and +/// raise handling, the callee's frame and buffers being the ones on the +/// caller's native stack and its context the caller's own (a pin-free +/// activation leaves the shared table empty). +/// +/// # Safety +/// +/// Same contract as [`wpjit_call_py`]; `callee` is the live callee frame. +unsafe extern "C" fn wpjit_self_slow( + frame: *mut JitFrame, + callee: *mut JitFrame, + status: i64, + token: i64, + expect_tag: i64, +) -> i64 { + // SAFETY: see wpjit_call_py — same live-buffer contract. + let jf = unsafe { &mut *frame }; + #[allow(clippy::cast_ptr_alignment)] + let ctx = unsafe { &mut *jf.ctx.cast::() }; + // SAFETY: as in `wpjit_self_enter`. + let interp = unsafe { &mut *ctx.interp }; + // SAFETY: the caller's contract. + let cjf = unsafe { &*callee }; + let raised = (status == JitStatus::Raised as i64).then(|| { + ctx.raised.take().unwrap_or_else(|| { + RuntimeError::Internal("JIT Raised exit without a parked exception".to_owned()) + }) + }); + let Some((Object::Function(pf), code)) = ctx.callees.get(token as usize).cloned() else { + // Unreachable: the lowering only emits direct calls for tokens + // naming this very function. + ctx.raised = Some(raised.unwrap_or_else(|| { + RuntimeError::Internal("direct self call without its callee".to_owned()) + })); + return CallStatus::Raised as i64; + }; + if status == JitStatus::Deopt as i64 { + native_stat(|s| s.deopts.set(s.deopts.get() + 1)); + let key = Rc::as_ptr(&code).cast::(); + JIT.with(|cell| { + if let Some(ce) = cell.borrow_mut().cache.get_mut(&key) { + ce.deopts += 1; + if ce.deopts >= DEOPT_BUDGET { + ce.tier = Tier::NotJitable; + code.jit_hint.mark_not_jitable(); + } + } + }); + } + // The callee runs the caller's own compilation, whose layout the + // shared context holds (the tier cache may have just retired it). + // SAFETY: `ctx.cf` came from a live `StdRc` its entry still owns. + let cf = unsafe { + StdRc::increment_strong_count(ctx.cf); + StdRc::from_raw(ctx.cf) + }; + let entry = CompiledEntry { + cf, + guard_snapshot: ctx.guard_snapshot.clone(), + callees: ctx.callees.clone(), + obj_globals: ctx.obj_globals.clone(), + attr_guards: ctx.attr_guards.clone(), + methods: ctx.methods.clone(), + math: ctx.math.clone(), + native: None, + method_native: None, + compile_id: 0, + }; + ctx.dirty = true; + // SAFETY: the callee frame's buffers are live on the caller's stack, + // sized by its own compiled frame. + let (locals, spill, tags) = unsafe { + ( + std::slice::from_raw_parts(cjf.locals, cjf.n_locals as usize), + std::slice::from_raw_parts(cjf.stack_spill, cjf.stack_cap as usize), + std::slice::from_raw_parts(cjf.stack_tags, cjf.stack_cap as usize), + ) + }; + match finish_deopted( + interp, &code, &pf, &entry, ctx, locals, spill, tags, cjf, raised, + ) { + Err(e) => { + ctx.raised = Some(e); + CallStatus::Raised as i64 + } + Ok(v) => { + // Python ran for the continuation: the caller's burned-in + // resolutions must still hold for it to continue natively. + if !guards_hold( + interp, + &ctx.globals, + &ctx.builtins, + &ctx.guard_snapshot, + &ctx.callees, + &ctx.math, + ) { + ctx.parked = Some(v); + return CallStatus::Boxed as i64; + } + deliver_call_result(jf, ctx, v, expect_tag as u32) + } + } +} + +/// The generic-call backoff for native-to-native and frameless direct +/// entries (the framed entries' twin lives in [`note_native_exit`]): a +/// compiled callee whose +/// activations average [`CALLEE_ROUNDTRIP_RETIRE_RATIO`] or more +/// interpreter calls is a thin native driver around them. Each such call pays pin +/// traffic, an activation shell and a generic call that the interpreter's +/// inline call path avoids, so the callee retires to tier-1. +#[inline] +fn note_callee_exit(art: &Artifacts, code: &Rc, child: &CallCtx) { + let entries = art.callee_entries.get().saturating_add(1); + art.callee_entries.set(entries); + if child.dyn_py_calls == 0 { + return; + } + let trips = art + .callee_roundtrips + .get() + .saturating_add(child.dyn_py_calls); + art.callee_roundtrips.set(trips); + if entries >= GENERIC_RETIRE_MIN_ENTRIES + && trips / entries >= CALLEE_ROUNDTRIP_RETIRE_RATIO + && !code.jit_hint.is_not_jitable() + { + let key = Rc::as_ptr(code).cast::(); + JIT.with(|cell| { + let mut st = cell.borrow_mut(); + if let Some(ce) = st.cache.get_mut(&key) { + ce.tier = Tier::NotJitable; + } + st.stats.generic_retires += 1; + }); + code.jit_hint.mark_not_jitable(); + } +} + /// Charge one expensive round-trip (an interpreter call, a generic /// attribute access, or a heavy native-to-native call) to the running /// activation. Returns true once the activation has spent @@ -5095,16 +5715,14 @@ unsafe extern "C" fn wpjit_call_method( s: &entry.name, hash: entry.name_hash, }; + // SAFETY: a read between two native ops; nothing here runs + // code (see `GilCell::peek`). attr_class_ok(inst, entry.ver) - && inst.dict.get().is_none_or(|dict| { - // SAFETY: a read between two native ops; nothing - // here runs code (see `GilCell::peek`). - match unsafe { dict.peek() } { - Some(d) => d.get(&probe).is_none(), - None => dict.borrow().get(&probe).is_none(), - } - }) - && Rc::ptr_eq(&entry.func.code.borrow(), &entry.code) + && unsafe { inst.attr_peek_has(probe.s, probe.hash) } == Some(false) + && match unsafe { entry.func.code.peek() } { + Some(code) => Rc::ptr_eq(code, &entry.code), + None => Rc::ptr_eq(&entry.func.code.borrow(), &entry.code), + } } _ => false, }; @@ -5136,6 +5754,57 @@ unsafe extern "C" fn wpjit_call_method( return finish_interp_call(jf, ctx, interp, res, expect_tag); } + // A pure-leaf method (a getter, a predicate) evaluates frameless. + // SAFETY: `argc` marshaled entries are live (the function contract), + // and `recv` outlives the evaluation. + if let Some(status) = unsafe { + try_pure_leaf_call( + jf, + ctx, + interp, + &entry.func, + &entry.code, + Some(&raw const recv), + argc, + expect_tag, + ) + } { + return status; + } + + // A callback-free field update (`self.n += k; return self.n`) bound + // exactly: the guards above and the update's own checks are all the + // native activation would validate for it. + if let Some(nc) = ctx + .method_native + .as_deref() + .and_then(|t| t.get(token as usize)) + .and_then(Option::as_ref) + { + if nc.scalar_update.is_some() + && entry.code.arg_count == argc + 1 + && Rc::ptr_eq(&nc.func, &entry.func) + && Rc::ptr_eq(&nc.code, &entry.code) + { + // SAFETY: the method guard above pinned the binding; the update + // checks its receiver, argument lane, and observers itself. + if let Some(value) = + unsafe { native_scalar_field_update(jf, ctx, nc, &recv, argc as usize) } + { + if expect_tag == SlotTag::Int as u32 { + jf.ret_bits = value as u64; + jf.ret_tag = SlotTag::Int as u32; + return CallStatus::Ok as i64; + } + // The store is complete: never repeat it. + #[cfg(test)] + crate::SCALAR_FIELD_UPDATE_BOXED_RETURNS.with(|hits| hits.set(hits.get() + 1)); + ctx.parked = Some(Object::Int(value)); + return CallStatus::Boxed as i64; + } + } + } + // RFC 0069 WS1 — the native fast path: the guarded method's own // body is compiled and shape-eligible, so enter it directly with // the receiver seeded as its pin 0. The table is parallel to @@ -5371,7 +6040,7 @@ unsafe extern "C" fn wpjit_str_method( } } } - SlotTag::None | SlotTag::Float | SlotTag::Boxed => {} + SlotTag::None | SlotTag::Float | SlotTag::Boxed | SlotTag::Default => {} } // Lane surprise (`WStr` result, huge `int`, pin-cap // pressure): park the exact result and deopt after the @@ -5838,6 +6507,17 @@ fn dict_pin_and_key( /// natively found) reports `Err(())` so the caller deopts and the /// interpreter runs the comparison with full semantics. fn dict_probe_native(d: &Rc>, key: &Object) -> Result, ()> { + // A `str` or `int` key settles by native equality unless the table + // compared it with a key of another kind. + if let Some(probe) = crate::object::LeafProbe::new(key) { + if let Ok(m) = d.try_borrow() { + match m.get(&probe) { + Some(v) => return Ok(Some(v.clone())), + None if probe.miss_is_exact() => return Ok(None), + None => {} + } + } + } let (found, deferred) = crate::object::with_key_eq_deferred(|| { crate::object::key_cmp_scope(|| d.borrow().get(&DictKey(key.clone())).cloned()) }); @@ -5871,15 +6551,25 @@ unsafe extern "C" fn wpjit_dict_get( let jf = unsafe { &mut *frame }; #[allow(clippy::cast_ptr_alignment)] let ctx = unsafe { &mut *jf.ctx.cast::() }; - let Some((d, key)) = dict_pin_and_key(ctx, pin, key_bits, key_tag) else { + let Some(Pin::Obj(Object::Dict(d))) = ctx.pins.get(pin as usize) else { return 1; }; - let found = match dict_probe_native(&d, &key) { + let int_key; + let key: &Object = if key_tag == weavepy_jit::DICT_KEY_STR { + match ctx.pins.get(key_bits as usize) { + Some(Pin::Obj(o @ Object::Str(_))) => o, + _ => return 1, + } + } else { + int_key = Object::Int(key_bits); + &int_key + }; + let found = match dict_probe_native(d, key) { Ok(f) => f, Err(()) => return 1, }; let Some(v) = found else { - ctx.raised = Some(crate::error::key_error_object(key)); + ctx.raised = Some(crate::error::key_error_object(key.clone())); return 2; }; match (val_tag, &v) { @@ -6362,6 +7052,11 @@ unsafe extern "C" fn wpjit_iter_next(frame: *mut JitFrame, pin: i64, elem_tag: i let runs_python = !matches!(it, Object::Iter(_)); if runs_python { ctx.dirty = true; + // A generator resume from native code rebuilds a whole interpreter + // activation, several times what the interpreter's own inline + // resume costs: charged like an interpreter call, so a loop that + // drives a generator retires at the next poll (see `wpjit_poll`). + ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } match interp.iter_next(&it, &ctx.globals) { Err(err) => { @@ -6862,7 +7557,7 @@ fn key_is(key: &DictKey, name: &SharedStr) -> bool { /// A site guard's class check: the receiver's class still carries the /// compiled `attr_version` (read without a borrow guard when only one /// thread runs Python; nothing here runs code). -#[inline] +#[inline(always)] fn attr_class_ok(inst: &crate::types::PyInstance, ver: u64) -> bool { if crate::gil::free_threading_enabled() { return inst.class.borrow().attr_version.get() == ver; @@ -7046,6 +7741,29 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - let jf = unsafe { &mut *frame }; #[allow(clippy::cast_ptr_alignment)] let ctx = unsafe { &mut *jf.ctx.cast::() }; + // The common shape first: a scalar field of an indexed site, read + // straight off the instance (the full path below re-derives it). + if let (Some(Pin::Obj(Object::Instance(inst))), Some(g)) = ( + ctx.pins.get(pin as usize), + ctx.attr_guards.get(site as usize), + ) { + if let (AttrStorage::Indexed(key_idx), JitType::Int | JitType::Float | JitType::Bool) = + (g.storage, g.lane) + { + if attr_class_ok(inst, g.ver) { + // SAFETY: a read between two native ops; nothing here runs + // code (see `GilCell::peek`). + if let Some((k, v)) = unsafe { inst.attr_peek_index(key_idx as usize) } { + if key_is(k, &g.name) { + if let Some(bits) = pack(v, g.lane) { + jf.ret_bits = bits; + return 0; + } + } + } + } + } + } // Scoped so the receiver borrow of `ctx.pins` ends before an // object-lane result appends a fresh pin (RFC 0070 WS1). let outcome: Result = { @@ -7081,7 +7799,11 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - }; match g.storage { AttrStorage::Slot(key_idx) => { - let slots = inst.slots.borrow(); + // SAFETY: a read between two native ops; nothing here + // runs code (see `GilCell::peek`). + let Some(slots) = (unsafe { inst.slots.peek() }) else { + return 1; + }; let indexed = slots .get_index(key_idx as usize) .filter(|(key, _)| key_is(key, &g.name)) @@ -7097,15 +7819,9 @@ unsafe extern "C" fn wpjit_attr_get(frame: *mut JitFrame, pin: i64, site: i64) - } } AttrStorage::Indexed(key_idx) => { - let Some(dict) = inst.dict.get() else { - return 1; - }; // SAFETY: a read between two native ops; nothing here // runs code (see `GilCell::peek`). - let Some(dict) = (unsafe { dict.peek() }) else { - return 1; - }; - match dict.get_index(key_idx as usize) { + match unsafe { inst.attr_peek_index(key_idx as usize) } { Some((k, v)) if key_is(k, &g.name) => match classify(v) { Some(o) => o, None => return 1, @@ -7171,9 +7887,8 @@ unsafe fn chain_attr_peek<'a>( } AttrStorage::Indexed(index) => { // SAFETY: the same callback-free interval as the slot read. - let dict = unsafe { inst.dict.get().ok_or(AttrChainMiss::Guard)?.peek() } - .ok_or(AttrChainMiss::Guard)?; - let (name, value) = dict.get_index(index as usize).ok_or(AttrChainMiss::Guard)?; + let (name, value) = + unsafe { inst.attr_peek_index(index as usize) }.ok_or(AttrChainMiss::Guard)?; if !key_is(name, &guard.name) { return Err(AttrChainMiss::Guard); } @@ -7248,8 +7963,7 @@ fn chain_attr_read( } AttrStorage::Indexed(index) => { // SAFETY: every chain step is a read without Python callbacks. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (name, value) = dict.get_index(index as usize)?; + let (name, value) = unsafe { inst.attr_peek_index(index as usize) }?; key_is(name, &guard.name).then(|| read(value)) } AttrStorage::NewKey => None, @@ -7397,8 +8111,7 @@ unsafe fn cached_chain_peek<'a>( match cache { IC::LoadAttrInstance { key_idx, ver } if ver == version => { // SAFETY: the rooted walk is read-only and callback-free. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(key_idx as usize)?; + let (key, value) = unsafe { inst.attr_peek_index(key_idx as usize) }?; key_is(key, name).then_some(value) } IC::LoadAttrSlot { key_idx, ver } if ver == version => { @@ -7418,8 +8131,7 @@ unsafe fn cached_chain_peek<'a>( .get(pc as usize)? .index(version)?; // SAFETY: the same callback-free, rooted interval. - let dict = unsafe { inst.dict.get()?.peek() }?; - let (key, value) = dict.get_index(index as usize)?; + let (key, value) = unsafe { inst.attr_peek_index(index as usize) }?; key_is(key, name).then_some(value) } } @@ -7648,6 +8360,35 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - let Some(g) = ctx.attr_guards.get(site as usize) else { return 1; }; + // The common shape first: a scalar value over a scalar field of an + // indexed site (no write barrier, nothing to reap). + if let (AttrStorage::Indexed(key_idx), Some(Pin::Obj(Object::Instance(inst)))) = + (g.storage, ctx.pins.get(pin as usize)) + { + let v = match g.lane { + JitType::Int => Some(Object::Int(jf.ret_bits as i64)), + JitType::Float => Some(Object::Float(f64::from_bits(jf.ret_bits))), + JitType::Bool => Some(Object::Bool(jf.ret_bits != 0)), + _ => None, + }; + if let Some(v) = v { + if attr_class_ok(inst, g.ver) { + // SAFETY: nothing below runs code while the view is live. + if let Some((k, dst)) = unsafe { inst.attr_peek_index_mut(key_idx as usize, true) } + { + if key_is(k, &g.name) + && matches!( + dst, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) + { + *dst = v; + return 0; + } + } + } + } + } let v = match g.lane { JitType::Int => Object::Int(jf.ret_bits as i64), JitType::Float => Object::Float(f64::from_bits(jf.ret_bits)), @@ -7691,15 +8432,12 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - 0 } AttrStorage::Indexed(key_idx) => { - let mut dict = inst.dict_cell().borrow_mut(); - let atomic = v.is_gc_atomic(); - let dict = &mut *dict; - let dict = if atomic { - dict.map_mut_atomic_store() - } else { - &mut **dict - }; - let Some((k, dst)) = dict.get_index_mut(key_idx as usize) else { + // Replacing an existing key's value leaves the key layout (and + // so the stamp) alone, as the interpreter's indexed store does. + // SAFETY: nothing below runs code while the view is live. + let Some((k, dst)) = + (unsafe { inst.attr_peek_index_mut(key_idx as usize, v.is_gc_atomic()) }) + else { return 1; }; if !key_is(k, &g.name) { @@ -7735,6 +8473,32 @@ unsafe extern "C" fn wpjit_attr_set(frame: *mut JitFrame, pin: i64, site: i64) - if crate::capi_watchers::dicts_active() { return 1; } + // The split layout (see `Interpreter::core_store_new_attr`). + let v = if inst.dict.published().is_none() { + // SAFETY: a read between two native ops. + let Some(split) = (unsafe { inst.dict.split_cell().peek() }) else { + return 1; + }; + if let Some(dst) = split.get(&g.name) { + if !matches!( + dst, + Object::Int(_) | Object::Float(_) | Object::Bool(_) | Object::None + ) && super::Interpreter::local_needs_prompt_reap(dst) + && super::Interpreter::looks_reapable_temporary(dst) + { + return 1; + } + } + match inst.split_store(&g.name, v) { + Ok(old) => { + drop(old); + return 0; + } + Err(v) => v, + } + } else { + v + }; let mut dict = inst.dict_cell().borrow_mut(); let atomic = v.is_gc_atomic(); let dict = &mut *dict; @@ -7858,22 +8622,40 @@ unsafe fn try_dyn_native( argc: u32, int_result: bool, ) -> Option { - let (nc, recv) = match callee { - Object::Function(pf) => { - let fcode = pf.code.borrow().clone(); - let nc = JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, false))?; - (nc, None) - } + // A plain function or a bound method's function: the activation's + // last resolution when it names the same function and code (see + // `CallCtx::dyn_callee`). + let plain = match callee { + Object::Function(pf) => Some((pf, None)), // A deferred special-method dispatch (`redispatch_descriptor`) // re-resolves `__get__` at call time — interpreter territory. - Object::BoundMethod(bm) if !bm.redispatch_descriptor => { - let Object::Function(pf) = &bm.function else { - return None; - }; - let fcode = pf.code.borrow().clone(); - let nc = JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, true))?; - (nc, Some(bm.receiver.clone())) - } + Object::BoundMethod(bm) if !bm.redispatch_descriptor => match &bm.function { + Object::Function(pf) => Some((pf, Some(bm.receiver.clone()))), + _ => return None, + }, + _ => None, + }; + if let Some((pf, recv)) = plain { + let method = recv.is_some(); + // SAFETY: GIL-serialized raw read of the function's code cell; + // only the pointer is compared. + let code_ptr = unsafe { Rc::as_ptr(&*pf.code.as_ptr()) } as usize; + let key = (Rc::as_ptr(pf) as usize, code_ptr, method); + let nc = match ctx.dyn_callee.take() { + Some((f, c, m, nc)) if (f, c, m) == key => nc, + _ => { + let fcode = pf.code.borrow().clone(); + JIT.with(|c| c.borrow().resolve_native_func(pf, &fcode, method))? + } + }; + // SAFETY: the resolved owners outlive the call, and the same + // initialized argument-buffer contract applies to the shared + // entry path. + let r = unsafe { enter_dyn_native(jf, ctx, interp, &nc, argc, recv.as_ref(), int_result) }; + ctx.dyn_callee = Some((key.0, key.1, key.2, nc)); + return r; + } + let (nc, recv) = match callee { Object::Type(t) => { // Mirror `resolve_native_callee`'s constructor arm: the // memoised instance plan must be current and carry a @@ -8215,12 +8997,19 @@ unsafe fn call_dyn_impl( .then(|| unsafe { dyn_kw_site_bind(jf, ctx, &callee, argc, kwc, names) }) .flatten() { - note_generic_dyn_call(ctx); + // A pure-leaf callee evaluates frameless on the bound locals. + if let Some(v) = interp.bound_leaf_eval(&f, &locals) { + interp.recycle_scratch(locals); + // SAFETY: as above. + return unsafe { dyn_call_result(jf, ctx, Ok(v), false, false, int_result) }; + } + // Not charged against the native driver: the interpreter's own + // `CALL_KW` binds through this same permutation and activation, + // so tier-1 would not run the call any cheaper. ctx.dirty = true; - let called = - call_with_activation_shell(interp, ctx, jf, |i| i.run_py_exact_nofree(&f, locals)); + let called = call_with_activation_shell(interp, ctx, jf, |i| i.run_py_bound(&f, locals)); // SAFETY: as above. - return unsafe { dyn_call_result(jf, ctx, called, false, int_result) }; + return unsafe { dyn_call_result(jf, ctx, called, false, false, int_result) }; } let n = (argc + kwc) as usize; let mut args: Vec = Vec::with_capacity(n); @@ -8252,8 +9041,13 @@ unsafe fn call_dyn_impl( } } // Arbitrary Python runs on behalf of this activation (RFC 0067 - // WS1's dirtiness discipline). - note_generic_dyn_call(ctx); + // WS1's dirtiness discipline). A keyword call pays the generic + // binder in tier-1 too, so only positional calls count against the + // native driver. + let charged = kwc == 0; + if charged { + note_generic_dyn_call(ctx); + } ctx.dirty = true; let native_callee = matches!(&callee, Object::Builtin(_)) || matches!(&callee, Object::BoundMethod(bm) if matches!(bm.function, Object::Builtin(_))); @@ -8261,12 +9055,14 @@ unsafe fn call_dyn_impl( i.call_object_with_globals(&callee, &args, &kwargs, &ctx.globals) }); // SAFETY: as above. - unsafe { dyn_call_result(jf, ctx, called, native_callee, int_result) } + unsafe { dyn_call_result(jf, ctx, called, native_callee, charged, int_result) } } /// `call_dyn_impl`'s result protocol: park a raise, or deliver the /// result unboxed (`int_result`) or pinned, parking it (`Boxed`) when a -/// round-trip charge or an invalidated guard ends the activation. +/// round-trip charge or an invalidated guard ends the activation. An +/// uncharged call (one tier-1 would not run any cheaper) leaves the +/// native driver's call density alone. /// /// # Safety /// @@ -8276,6 +9072,7 @@ unsafe fn dyn_call_result( ctx: &mut CallCtx, called: Result, native_callee: bool, + charged: bool, int_result: bool, ) -> i64 { // SAFETY: the `&mut Interpreter` that entered native code is @@ -8287,10 +9084,10 @@ unsafe fn dyn_call_result( CallStatus::Raised as i64 } Ok(v) => { - if !native_callee { + if charged && !native_callee { ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } - if charge_roundtrip(ctx) || (native_callee && charge_native_roundtrip(ctx)) { + if charge_roundtrip(ctx) || (charged && native_callee && charge_native_roundtrip(ctx)) { ctx.parked = Some(v); return CallStatus::Boxed as i64; } @@ -8371,7 +9168,10 @@ unsafe fn dyn_kw_site_bind( } let (fcode, covered) = crate::Interpreter::kw_names_bind_check(f, func_id, perm, name_items, eff_argc)?; - let mut staged: Vec = Vec::with_capacity(eff_argc + kwc); + // SAFETY: the `&mut Interpreter` that entered native code is dormant + // while the helper runs; only its vector pools are used here. + let interp = unsafe { &*ctx.interp }; + let mut staged = interp.pooled_scratch(); staged.extend(recv.cloned()); for j in 0..argc + kwc { // SAFETY: native code wrote `argc + kwc` entries, and the @@ -8379,7 +9179,7 @@ unsafe fn dyn_kw_site_bind( let (bits, tag) = unsafe { (*jf.call_args.add(j), *jf.call_tags.add(j)) }; staged.push(unpack_pins(bits, tag, &ctx.pins)); } - let mut locals = Vec::new(); + let mut locals = interp.pooled_scratch(); crate::Interpreter::kw_names_fill_locals( &mut locals, &mut staged, @@ -8391,6 +9191,7 @@ unsafe fn dyn_kw_site_bind( kwc, eff_argc, ); + interp.recycle_scratch(staged); Some((f.clone(), locals)) } @@ -8640,6 +9441,14 @@ unsafe extern "C" fn wpjit_truth(frame: *mut JitFrame, pin: i64, _reserved: i64) Some(p) => p.to_object(), None => return 3, }; + // A registered native `__bool__`/`__len__` (or neither): the + // interpreter's cached answer, which runs no Python. + if matches!(v, Object::Instance(_)) && !crate::gil::free_threading_enabled() { + if let Some(b) = interp.leaf_instance_truth(&v) { + jf.ret_bits = u64::from(b); + return 0; + } + } let pure = match &v { Object::Foreign(_) | Object::MappingProxyObj(_) => false, Object::Instance(_) => { @@ -9088,6 +9897,11 @@ unsafe extern "C" fn wpjit_iter_next_pair( let runs_python = !matches!(it, Object::Iter(_)); if runs_python { ctx.dirty = true; + // A generator resume from native code rebuilds a whole interpreter + // activation, several times what the interpreter's own inline + // resume costs: charged like an interpreter call, so a loop that + // drives a generator retires at the next poll (see `wpjit_poll`). + ctx.dyn_py_calls = ctx.dyn_py_calls.saturating_add(1); } match interp.iter_next(&it, &ctx.globals) { Err(err) => { @@ -9408,6 +10222,7 @@ pub(crate) fn try_call_native_direct( let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.art.callees.clone(), + cf: StdRc::as_ptr(&entry.art.cf), guard_snapshot: entry.art.snap.clone(), globals: f.globals.clone(), builtins: f.builtins.clone(), @@ -9438,6 +10253,7 @@ pub(crate) fn try_call_native_direct( // The frameless direct entry pushes no interpreter frame — // keep the activation observable to callee-side stack walkers. frameless_code: Some(code.clone()), + dyn_callee: None, }; let mut jf = JitFrame { locals: locals_buf.as_mut_ptr(), @@ -9468,6 +10284,11 @@ pub(crate) fn try_call_native_direct( }; native_stat(|s| s.direct_calls.set(s.direct_calls.get() + 1)); + // A direct callee that keeps calling back into the interpreter is + // retired like a native-to-native one (deltablue's `incremental_add` + // ran each `satisfy` through an activation shell and a framed call, + // 6% of the benchmark's instructions). + note_callee_exit(&entry.art, code, &ctx); let out = match status { JitStatus::Returned => Ok(unpack_pins(jf.ret_bits, jf.ret_tag, &ctx.pins)), @@ -10210,6 +11031,7 @@ fn enter_compiled( let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.callees.clone(), + cf: StdRc::as_ptr(&entry.cf), guard_snapshot: entry.guard_snapshot.clone(), globals: frame.globals.clone(), builtins: frame.builtins.clone(), @@ -10239,6 +11061,7 @@ fn enter_compiled( // Framed entry: this activation's `Frame` shell is on the // spine already. frameless_code: None, + dyn_callee: None, }; let mut jf = JitFrame { locals: locals_buf.as_mut_ptr(), @@ -10943,7 +11766,16 @@ fn park_plan(frame: &super::Frame, entry: &CompiledEntry, jf: &JitFrame) -> Opti /// a resume's sent value). Afterwards the frame is indistinguishable /// from an interpreted suspension. No-op without a parked box; never /// needs an interpreter (park refused any shape whose rebuild would). +#[inline] pub(crate) fn materialize_parked(frame: &mut super::Frame) { + if frame.parked_native.is_some() { + materialize_parked_native(frame); + } +} + +#[cold] +#[inline(never)] +fn materialize_parked_native(frame: &mut super::Frame) { let Some(mut act) = frame.parked_native.take() else { return; }; @@ -11088,6 +11920,7 @@ fn resume_parked(interp: &mut super::Interpreter, frame: &mut super::Frame) -> J let mut ctx = CallCtx { interp: std::ptr::from_mut(interp), callees: entry.callees.clone(), + cf: StdRc::as_ptr(&entry.cf), guard_snapshot: entry.guard_snapshot.clone(), globals: frame.globals.clone(), builtins: frame.builtins.clone(), @@ -11117,6 +11950,7 @@ fn resume_parked(interp: &mut super::Interpreter, frame: &mut super::Frame) -> J // Framed entry (generator resume): the resumed `Frame`'s shell // is on the spine already. frameless_code: None, + dyn_callee: None, }; let mut jf = JitFrame { locals: act.locals_buf.as_mut_ptr(), diff --git a/crates/weavepy-vm/src/timsort.rs b/crates/weavepy-vm/src/timsort.rs new file mode 100644 index 00000000..3759ffca --- /dev/null +++ b/crates/weavepy-vm/src/timsort.rs @@ -0,0 +1,736 @@ +//! CPython's list sort: an adaptive, stable, natural merge sort (timsort with +//! the powersort merge policy), ported from `Objects/listobject.c` in 3.14. +//! +//! The comparison sequence follows CPython's, so a comparator that is +//! inconsistent (NaNs, a random `__lt__`) or that raises produces the same +//! observable behavior: no panic, and on error the slice is left as a +//! permutation of its input. Every comparison is a strict "less than". + +use std::ptr; + +/// Once a merge is galloping, it stays there until both runs win fewer +/// than this many consecutive times. +const MIN_GALLOP: usize = 7; + +/// The largest minimum run length; a power of 2. +const MAX_MINRUN: usize = 64; + +/// A run pending a merge: `len` elements starting at index `base`. +#[derive(Clone, Copy)] +struct Run { + base: usize, + len: usize, + /// Depth in the conceptual binary merge tree (powersort). + power: u32, +} + +struct MergeState { + base: *mut T, + len: usize, + lt: F, + min_gallop: usize, + /// Scratch storage for merges. Its length is always 0: elements are + /// moved in and out bitwise, and never dropped from here. + tmp: Vec, + pending: Vec, +} + +/// Sort `v` in place, stably, by `lt`, the strict ordering. On error, `v` +/// holds a permutation of its input. +pub(crate) fn sort(v: &mut [T], lt: F) -> Result<(), E> +where + F: FnMut(&T, &T) -> Result, +{ + let n = v.len(); + if n < 2 { + return Ok(()); + } + let mut ms = MergeState { + base: v.as_mut_ptr(), + len: n, + lt, + min_gallop: MIN_GALLOP, + tmp: Vec::new(), + pending: Vec::new(), + }; + let minrun = compute_minrun(n); + let mut lo = 0; + let mut nremaining = n; + // SAFETY: every index handed to the helpers below lies in `v`, which is + // borrowed mutably for the whole sort. + unsafe { + while nremaining > 0 { + let mut run = ms.count_run(lo, nremaining)?; + if run < minrun { + let force = nremaining.min(minrun); + ms.binarysort(lo, force, run)?; + run = force; + } + ms.found_new_run(run)?; + ms.pending.push(Run { + base: lo, + len: run, + power: 0, + }); + lo += run; + nremaining -= run; + } + ms.merge_force_collapse() + } +} + +/// A good minimum run length: `n` itself below `MAX_MINRUN`, else a value in +/// `MAX_MINRUN / 2 ..= MAX_MINRUN` such that `n / minrun` is close to, but +/// strictly less than, a power of 2. +fn compute_minrun(mut n: usize) -> usize { + let mut r = 0; + while n >= MAX_MINRUN { + r |= n & 1; + n >>= 1; + } + n + r +} + +/// The powersort "power" of the run at `s1` (length `n1`) followed by one of +/// length `n2`, in a list of length `n`. +fn powerloop(s1: usize, n1: usize, n2: usize, n: usize) -> u32 { + let mut result = 0; + // Twice the two runs' midpoints, so that both are integers. + let mut a = 2 * s1 + n1; + let mut b = a + n1 + n2; + loop { + result += 1; + if a >= n { + a -= n; + b -= n; + } else if b >= n { + break; + } + a <<= 1; + b <<= 1; + } + result +} + +/// Unmerged elements of a merge, parked in scratch storage: `len` of them +/// from `src` belong at `dest`. Dropping the hole moves them there, which +/// also restores a complete permutation if a comparison fails or panics. +struct Hole { + src: *const T, + dest: *mut T, + len: usize, +} + +impl Drop for Hole { + fn drop(&mut self) { + // SAFETY: the merge maintains that `dest` is exactly the vacated + // range the parked elements fill, and scratch never overlaps the + // list. + unsafe { ptr::copy_nonoverlapping(self.src, self.dest, self.len) }; + } +} + +impl MergeState +where + F: FnMut(&T, &T) -> Result, +{ + #[inline] + unsafe fn at(&self, i: usize) -> *mut T { + // SAFETY: callers pass indices within the list. + unsafe { self.base.add(i) } + } + + #[inline] + fn lt(&mut self, a: *const T, b: *const T) -> Result { + // SAFETY: both point at initialized elements, in the list or + // parked in scratch, and no element moves during a comparison. + unsafe { (self.lt)(&*a, &*b) } + } + + /// Scratch storage for `need` elements. + fn scratch(&mut self, need: usize) -> *mut T { + if self.tmp.capacity() < need { + self.tmp = Vec::with_capacity(need); + } + self.tmp.as_mut_ptr() + } + + unsafe fn reverse(&mut self, lo: usize, n: usize) { + // SAFETY: `lo..lo + n` lies in the list. + unsafe { std::slice::from_raw_parts_mut(self.at(lo), n).reverse() }; + } + + /// Stable binary insertion sort of `lo..lo + n`, whose first `ok` + /// elements are already sorted. + unsafe fn binarysort(&mut self, lo: usize, n: usize, ok: usize) -> Result<(), E> { + let a = unsafe { self.at(lo) }; + let mut ok = ok.max(1); + while ok < n { + // Find where a[ok] belongs: a[..l] <= pivot < a[r..ok]. + let (mut l, mut r) = (0, ok); + let pivot = unsafe { a.add(ok) }; + while l < r { + let m = usize::midpoint(l, r); + if self.lt(pivot, unsafe { a.add(m) })? { + r = m; + } else { + l = m + 1; + } + } + // SAFETY: rotate a[l..=ok] right by one; nothing compares + // while the pivot is out. + unsafe { + let p = ptr::read(pivot); + ptr::copy(a.add(l), a.add(l + 1), ok - l); + ptr::write(a.add(l), p); + } + ok += 1; + } + Ok(()) + } + + /// The length of the run starting at `lo`, no longer than `nremaining`, + /// made ascending in place. + unsafe fn count_run(&mut self, lo: usize, nremaining: usize) -> Result { + let a = unsafe { self.at(lo) }; + let next_smaller = + |ms: &mut Self, n: usize| ms.lt(unsafe { a.add(n) }, unsafe { a.add(n - 1) }); + let next_larger = + |ms: &mut Self, n: usize| ms.lt(unsafe { a.add(n - 1) }, unsafe { a.add(n) }); + // Try an ascending run first. + let mut n = 1; + while n < nremaining { + if next_smaller(self, n)? { + break; + } + n += 1; + } + if n == nremaining { + return Ok(n); + } + // a[n] is strictly less. With a longer ascending prefix, it either + // rose somewhere (done), or is all equal and can start a + // descending run, reversed in place. + if n > 1 { + if self.lt(a, unsafe { a.add(n - 1) })? { + return Ok(n); + } + unsafe { self.reverse(lo, n) }; + } + n += 1; + // Finish the descending run, reversing all-equal subruns on the fly + // so the final whole-run reversal restores their order. + let mut neq = 0; + while n < nremaining { + if next_smaller(self, n)? { + if neq > 0 { + neq += 1; + unsafe { self.reverse(lo + n - neq, neq) }; + neq = 0; + } + } else if next_larger(self, n)? { + break; + } else { + neq += 1; + } + n += 1; + } + if neq > 0 { + neq += 1; + unsafe { self.reverse(lo + n - neq, neq) }; + } + unsafe { self.reverse(lo, n) }; + // The reversed run may extend with a naturally increasing suffix. + while n < nremaining { + if next_smaller(self, n)? { + break; + } + n += 1; + } + Ok(n) + } + + /// The index `k` in `0..=n` where `key` belongs in the sorted `a[..n]`, + /// left of any equal elements: `a[k - 1] < key <= a[k]`. The search + /// starts at `hint`. + unsafe fn gallop_left( + &mut self, + key: *const T, + a: *const T, + n: usize, + hint: usize, + ) -> Result { + let at = |i: usize| unsafe { a.add(i) }; + let (mut lastofs, mut ofs); + if self.lt(at(hint), key)? { + // Gallop right until a[hint + lastofs] < key <= a[hint + ofs]. + let maxofs = n - hint; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(at(hint + ofs), key)? { + lastofs = ofs; + ofs = (ofs << 1) + 1; + } else { + break; + } + } + ofs = ofs.min(maxofs); + lastofs += hint + 1; + ofs += hint; + } else { + // Gallop left until a[hint - ofs] < key <= a[hint - lastofs]. + let maxofs = hint + 1; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(at(hint - ofs), key)? { + break; + } + lastofs = ofs; + ofs = (ofs << 1) + 1; + } + ofs = ofs.min(maxofs); + let k = lastofs; + // `hint - ofs` may be -1; the binary search starts one past it. + lastofs = hint + 1 - ofs; + ofs = hint - k; + } + // Binary search with a[lastofs - 1] < key <= a[ofs]. + while lastofs < ofs { + let m = lastofs + ((ofs - lastofs) >> 1); + if self.lt(at(m), key)? { + lastofs = m + 1; + } else { + ofs = m; + } + } + Ok(ofs) + } + + /// Like [`Self::gallop_left`], but right of any equal elements: + /// `a[k - 1] <= key < a[k]`. + unsafe fn gallop_right( + &mut self, + key: *const T, + a: *const T, + n: usize, + hint: usize, + ) -> Result { + let at = |i: usize| unsafe { a.add(i) }; + let (mut lastofs, mut ofs); + if self.lt(key, at(hint))? { + // Gallop left until a[hint - ofs] <= key < a[hint - lastofs]. + let maxofs = hint + 1; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(key, at(hint - ofs))? { + lastofs = ofs; + ofs = (ofs << 1) + 1; + } else { + break; + } + } + ofs = ofs.min(maxofs); + let k = lastofs; + lastofs = hint + 1 - ofs; + ofs = hint - k; + } else { + // Gallop right until a[hint + lastofs] <= key < a[hint + ofs]. + let maxofs = n - hint; + lastofs = 0; + ofs = 1; + while ofs < maxofs { + if self.lt(key, at(hint + ofs))? { + break; + } + lastofs = ofs; + ofs = (ofs << 1) + 1; + } + ofs = ofs.min(maxofs); + lastofs += hint + 1; + ofs += hint; + } + // Binary search with a[lastofs - 1] <= key < a[ofs]. + while lastofs < ofs { + let m = lastofs + ((ofs - lastofs) >> 1); + if self.lt(key, at(m))? { + ofs = m; + } else { + lastofs = m + 1; + } + } + Ok(ofs) + } + + /// Merge the adjacent runs `sa..sa + na` and `sa + na..sa + na + nb`, + /// with `na <= nb`, through scratch space for the first. The last + /// element of the first run belongs at the end of the merge. + // A move's bookkeeping is dead when the merge returns right after it. + #[allow(unused_assignments)] + unsafe fn merge_lo(&mut self, sa: usize, na: usize, nb: usize) -> Result<(), E> { + let tmp = self.scratch(na); + let mut dest = unsafe { self.at(sa) }; + let mut b = unsafe { self.at(sa + na) }; + unsafe { ptr::copy_nonoverlapping(dest, tmp, na) }; + // The unmerged part of the first run, parked in scratch. + let mut hole = Hole { + src: tmp, + dest, + len: na, + }; + let mut nb = nb; + // SAFETY (throughout): the merged prefix, the parked first-run + // elements, and the rest of the second run always partition the + // two runs' span, so `hole.dest` is where the parked ones belong. + macro_rules! take_b { + ($k:expr) => {{ + let k = $k; + unsafe { ptr::copy(b, dest, k) }; + dest = unsafe { dest.add(k) }; + b = unsafe { b.add(k) }; + nb -= k; + hole.dest = dest; + }}; + } + macro_rules! take_a { + ($k:expr) => {{ + let k = $k; + unsafe { ptr::copy_nonoverlapping(hole.src, dest, k) }; + dest = unsafe { dest.add(k) }; + hole.src = unsafe { hole.src.add(k) }; + hole.len -= k; + hole.dest = dest; + }}; + } + take_b!(1); + if nb == 0 { + return Ok(()); + } + if hole.len == 1 { + // The last element of the first run goes after the second. + take_b!(nb); + return Ok(()); + } + let mut min_gallop = self.min_gallop; + loop { + let mut acount = 0; + let mut bcount = 0; + // One element at a time, until one run keeps winning. + loop { + if self.lt(b, hole.src)? { + take_b!(1); + bcount += 1; + acount = 0; + if nb == 0 { + return Ok(()); + } + if bcount >= min_gallop { + break; + } + } else { + take_a!(1); + acount += 1; + bcount = 0; + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + if acount >= min_gallop { + break; + } + } + } + // Gallop until neither run is winning consistently. + min_gallop += 1; + loop { + min_gallop -= usize::from(min_gallop > 1); + self.min_gallop = min_gallop; + let k = unsafe { self.gallop_right(b, hole.src, hole.len, 0)? }; + acount = k; + if k > 0 { + take_a!(k); + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + // Impossible for a consistent comparison, but possible. + if hole.len == 0 { + return Ok(()); + } + } + take_b!(1); + if nb == 0 { + return Ok(()); + } + let k = unsafe { self.gallop_left(hole.src, b, nb, 0)? }; + bcount = k; + if k > 0 { + take_b!(k); + if nb == 0 { + return Ok(()); + } + } + take_a!(1); + if hole.len == 1 { + take_b!(nb); + return Ok(()); + } + if acount < MIN_GALLOP && bcount < MIN_GALLOP { + break; + } + } + // Penalize leaving galloping mode. + min_gallop += 1; + self.min_gallop = min_gallop; + } + } + + /// Merge the adjacent runs `sa..sa + na` and `sa + na..sa + na + nb`, + /// with `na >= nb`, from the top, through scratch space for the second. + /// The first element of the second run belongs at the front. + #[allow(unused_assignments)] + unsafe fn merge_hi(&mut self, sa: usize, na: usize, nb: usize) -> Result<(), E> { + let tmp = self.scratch(nb); + let a_base = unsafe { self.at(sa) }; + unsafe { ptr::copy_nonoverlapping(self.at(sa + na), tmp, nb) }; + // The unmerged part of the second run, parked in scratch; it always + // belongs just above the unmerged part of the first. + let mut hole = Hole { + src: tmp, + dest: unsafe { a_base.add(na) }, + len: nb, + }; + let mut na = na; + // `dest` is the highest unfilled slot: a_base[na + hole.len - 1]. + // SAFETY (throughout): as in `merge_lo`, mirrored. + macro_rules! take_a { + ($k:expr) => {{ + let k = $k; + // Move a_base[na - k..na] up to end at the top slot. + unsafe { ptr::copy(a_base.add(na - k), a_base.add(na - k + hole.len), k) }; + na -= k; + hole.dest = unsafe { a_base.add(na) }; + }}; + } + macro_rules! take_b { + ($k:expr) => {{ + let k = $k; + let len = hole.len; + unsafe { ptr::copy_nonoverlapping(tmp.add(len - k), a_base.add(na + len - k), k) }; + hole.len -= k; + }}; + } + take_a!(1); + if na == 0 { + return Ok(()); + } + if hole.len == 1 { + // The first element of the second run goes before the first. + take_a!(na); + return Ok(()); + } + let mut min_gallop = self.min_gallop; + loop { + let mut acount = 0; + let mut bcount = 0; + loop { + let (a_top, b_top) = unsafe { (a_base.add(na - 1), tmp.add(hole.len - 1)) }; + if self.lt(b_top, a_top)? { + take_a!(1); + acount += 1; + bcount = 0; + if na == 0 { + return Ok(()); + } + if acount >= min_gallop { + break; + } + } else { + take_b!(1); + bcount += 1; + acount = 0; + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + if bcount >= min_gallop { + break; + } + } + } + min_gallop += 1; + loop { + min_gallop -= usize::from(min_gallop > 1); + self.min_gallop = min_gallop; + let b_top = unsafe { tmp.add(hole.len - 1) }; + let k = na - unsafe { self.gallop_right(b_top, a_base, na, na - 1)? }; + acount = k; + if k > 0 { + take_a!(k); + if na == 0 { + return Ok(()); + } + } + take_b!(1); + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + let a_top = unsafe { a_base.add(na - 1) }; + let k = hole.len - unsafe { self.gallop_left(a_top, tmp, hole.len, hole.len - 1)? }; + bcount = k; + if k > 0 { + take_b!(k); + if hole.len == 1 { + take_a!(na); + return Ok(()); + } + // Impossible for a consistent comparison, but possible. + if hole.len == 0 { + return Ok(()); + } + } + take_a!(1); + if na == 0 { + return Ok(()); + } + if acount < MIN_GALLOP && bcount < MIN_GALLOP { + break; + } + } + min_gallop += 1; + self.min_gallop = min_gallop; + } + } + + /// Merge the pending runs at stack indices `i` and `i + 1`. + unsafe fn merge_at(&mut self, i: usize) -> Result<(), E> { + let Run { + base: sa, len: na, .. + } = self.pending[i]; + let Run { + base: sb, len: nb, .. + } = self.pending[i + 1]; + self.pending[i].len = na + nb; + self.pending.remove(i + 1); + // Elements of the first run before where the second starts are + // already in place. + let k = unsafe { self.gallop_right(self.at(sb), self.at(sa), na, 0)? }; + let (sa, na) = (sa + k, na - k); + if na == 0 { + return Ok(()); + } + // Elements of the second run after where the first ends are too. + let nb = unsafe { self.gallop_left(self.at(sa + na - 1), self.at(sb), nb, nb - 1)? }; + if nb == 0 { + return Ok(()); + } + if na <= nb { + unsafe { self.merge_lo(sa, na, nb) } + } else { + unsafe { self.merge_hi(sa, na, nb) } + } + } + + /// A run of length `n2` follows the pending ones: merge runs deeper in + /// the powersort tree than the top one. + unsafe fn found_new_run(&mut self, n2: usize) -> Result<(), E> { + let Some(top) = self.pending.last() else { + return Ok(()); + }; + let power = powerloop(top.base, top.len, n2, self.len); + while self.pending.len() > 1 && self.pending[self.pending.len() - 2].power > power { + unsafe { self.merge_at(self.pending.len() - 2)? }; + } + let last = self.pending.len() - 1; + self.pending[last].power = power; + Ok(()) + } + + /// Merge every pending run into one. + unsafe fn merge_force_collapse(&mut self) -> Result<(), E> { + while self.pending.len() > 1 { + let mut n = self.pending.len() - 2; + if n > 0 && self.pending[n - 1].len < self.pending[n + 1].len { + n -= 1; + } + unsafe { self.merge_at(n)? }; + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn check(mut v: Vec<(i32, usize)>) { + let mut expected = v.clone(); + expected.sort_by_key(|p| p.0); + sort(&mut v, |a, b| Ok::(a.0 < b.0)).unwrap(); + assert_eq!(v, expected); + } + + #[test] + fn sorts_stably() { + let mut seed = 12345u64; + let mut rand = move || { + seed ^= seed << 13; + seed ^= seed >> 7; + seed ^= seed << 17; + seed + }; + for n in [0, 1, 2, 3, 10, 63, 64, 65, 100, 257, 1000, 5000] { + for modulus in [2, 10, 1000, u64::MAX] { + let v: Vec<(i32, usize)> = (0..n).map(|i| ((rand() % modulus) as i32, i)).collect(); + check(v.clone()); + let mut sorted = v.clone(); + sorted.sort_unstable(); + check(sorted.clone()); + sorted.reverse(); + check(sorted); + } + } + } + + #[test] + fn inconsistent_order_keeps_every_element() { + let mut seed = 99u64; + let v: Vec = (0..3000).map(|i| i.to_string()).collect(); + let mut w = v.clone(); + sort(&mut w, |_, _| { + seed = seed.wrapping_mul(6_364_136_223_846_793_005).wrapping_add(1); + Ok::(seed >> 63 == 1) + }) + .unwrap(); + w.sort(); + let mut v = v; + v.sort(); + assert_eq!(v, w); + } + + #[test] + fn error_keeps_every_element() { + let v: Vec = (0..2000).rev().map(|i| i.to_string()).collect(); + for limit in [0, 1, 10, 500, 5000] { + let mut w = v.clone(); + let mut count = 0; + let r = sort(&mut w, |a, b| { + count += 1; + if count > limit { + Err(()) + } else { + Ok(a.len() < b.len() || (a.len() == b.len() && a < b)) + } + }); + assert!(r.is_err() || limit >= 5000); + let mut sorted = w.clone(); + sorted.sort(); + let mut expected = v.clone(); + expected.sort(); + assert_eq!(sorted, expected); + } + } +} diff --git a/crates/weavepy-vm/src/tuple_storage.rs b/crates/weavepy-vm/src/tuple_storage.rs index e8f6f72f..0bc37915 100644 --- a/crates/weavepy-vm/src/tuple_storage.rs +++ b/crates/weavepy-vm/src/tuple_storage.rs @@ -5,7 +5,8 @@ use std::alloc::Layout; use std::ops::{Deref, DerefMut}; use crate::object::Object; -use crate::sync::{CachedHash, Rc}; +use crate::sync::CachedHash; +use std::sync::Arc as Rc; pub type SharedTuple = ThinArc; diff --git a/crates/weavepy-vm/src/types.rs b/crates/weavepy-vm/src/types.rs index dda93592..c0543ad7 100644 --- a/crates/weavepy-vm/src/types.rs +++ b/crates/weavepy-vm/src/types.rs @@ -37,6 +37,90 @@ static NEXT_TYPE_VERSION: std::sync::atomic::AtomicU64 = std::sync::atomic::Atom /// cached entry (RFC 0077 WS4, [`type_cache`]). Reads are relaxed atomic /// loads — the previous `Cell` paid a `GilCell` lock round-trip on /// every inline-cache guard. +/// Dunders whose resolution [`TypeObject::dunder`] memoises. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Dunder { + Eq, + Hash, + GetAttr, + GetAttribute, +} + +impl Dunder { + pub const COUNT: usize = 4; + + pub const fn name(self) -> &'static str { + match self { + Self::Eq => "__eq__", + Self::Hash => "__hash__", + Self::GetAttr => "__getattr__", + Self::GetAttribute => "__getattribute__", + } + } +} + +/// How a type resolves one dunder, as a set of flags. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DunderInfo(u8); + +impl DunderInfo { + const VALID: u8 = 1; + const PRESENT: u8 = 1 << 1; + const NONE: u8 = 1 << 2; + const FUNCTION: u8 = 1 << 3; + const BUILTIN_OWNER: u8 = 1 << 4; + const OBJECT_OWNER: u8 = 1 << 5; + + fn resolve(found: Option<(Object, Rc)>) -> Self { + let Some((value, owner)) = found else { + return Self(Self::VALID); + }; + let mut bits = Self::VALID | Self::PRESENT; + match value { + Object::None => bits |= Self::NONE, + Object::Function(_) | Object::BoundMethod(_) => bits |= Self::FUNCTION, + _ => {} + } + if owner.flags.is_builtin { + bits |= Self::BUILTIN_OWNER; + } + if Rc::ptr_eq(&owner, &crate::builtin_types::builtin_types().object_) { + bits |= Self::OBJECT_OWNER; + } + Self(bits) + } + + /// The MRO defines the dunder (possibly as `None`). + pub fn present(self) -> bool { + self.0 & Self::PRESENT != 0 + } + + /// The dunder is set to `None` (the unhashable marker for `__hash__`). + pub fn is_none(self) -> bool { + self.0 & Self::NONE != 0 + } + + /// The dunder is a Python function or bound method. + pub fn is_function(self) -> bool { + self.0 & Self::FUNCTION != 0 + } + + /// The defining class is a built-in type. + pub fn builtin_owner(self) -> bool { + self.0 & Self::BUILTIN_OWNER != 0 + } + + /// The defining class is `object`. + pub fn object_owner(self) -> bool { + self.0 & Self::OBJECT_OWNER != 0 + } + + /// A non-`None` definition supplied by a class written in Python. + pub fn user_defined(self) -> bool { + self.present() && !self.is_none() && !self.builtin_owner() + } +} + pub struct AttrVersion(std::sync::atomic::AtomicU64); impl AttrVersion { @@ -574,6 +658,9 @@ pub struct TypeObject { /// (capped): new instances' dicts are presized to it, so `__init__`'s /// stores do not regrow them. pub inst_dict_hint: std::sync::atomic::AtomicU32, + /// The attribute names this class's split instance dictionaries share + /// (see [`crate::inst_dict`]), created at the first split store. + pub shared_keys: crate::sync::LazyArc, /// Non-zero for an exact class whose hot methods have native /// implementations (see `stdlib::datetime_native`): its instances /// are never cycle-collector tracked (like CPython's C types without @@ -582,6 +669,9 @@ pub struct TypeObject { /// Native-implementation state for such a class (see /// `stdlib::datetime_native`), set once. pub native_ext: std::sync::OnceLock>, + /// An `abc.ABCMeta` class's registry and caches (see + /// [`crate::stdlib::abc_mod`]). + pub abc_state: std::sync::OnceLock>, /// Cached "do instances of this type carry a `__del__` finalizer /// anywhere in their MRO?" answer, so [`crate::object::PyInstance`]'s /// `Drop` safety net can skip an MRO walk on the hot per-instance drop @@ -590,13 +680,9 @@ pub struct TypeObject { /// `__del__` is assigned to / deleted from a type's dict or the MRO is /// recomputed (`__bases__` assignment). pub has_del: Cell, - /// Memoised `__eq__` resolution for instances of this type, packed - /// as `attr_version << 2 | kind` (`0` = not yet computed): kind `1` = - /// `object`'s identity default (or no `__eq__`), `2` = a Python-level - /// override, `3` = a built-in type's own override (Python dispatch - /// only for instances without a native payload). A stale version - /// recomputes. See `object::instance_has_custom_eq`. - pub eq_kind: Cell, + /// Memoised resolution of the dunders in [`Dunder`], one slot each, + /// packed as `attr_version << 8 | DunderInfo` (see [`Self::dunder`]). + pub dunder_memo: [Cell; Dunder::COUNT], /// Memoised instantiation plan (`type(…)` call protocol resolution: /// `__new__`/`__init__`/native-payload classification), stamped with /// the [`Self::attr_version`] observed when it was built. Rebuilt @@ -979,8 +1065,10 @@ impl TypeObject { metaclass: RefCell::new(None), leaf_attrs: LeafAttrCache::new(), inst_dict_hint: std::sync::atomic::AtomicU32::new(0), + shared_keys: crate::sync::LazyArc::new(), native_kind: Cell::new(0), native_ext: std::sync::OnceLock::new(), + abc_state: std::sync::OnceLock::new(), slot_names: RefCell::new(Vec::new()), declares_slots: Cell::new(false), forbids_dict: false, @@ -990,7 +1078,7 @@ impl TypeObject { mro_kind: std::sync::atomic::AtomicU8::new(0), attr_version: AttrVersion::fresh(), has_del: Cell::new(0), - eq_kind: Cell::new(0), + dunder_memo: Default::default(), instance_plan: RefCell::new(None), c_tp_name: crate::sync::RefCell::new(None), c_sq_item: Cell::new(false), @@ -1092,7 +1180,7 @@ impl TypeObject { /// CPython `best_base`: the base contributing the instance layout — /// the one whose solid base is the most derived. Ties resolve to the /// first base (matching `type_new`'s left-to-right scan). - pub fn best_base(self: &Rc) -> Option> { + pub fn best_base(&self) -> Option> { let bases = self.bases.borrow(); let mut best: Option> = None; for b in bases.iter() { @@ -1118,8 +1206,8 @@ impl TypeObject { /// CPython `compatible_for_assignment`'s `newbase`/`oldbase` walk: /// climb the `best_base` chain past every level that doesn't change /// the struct, returning the most-derived type that *does*. - pub fn layout_struct_base(self: &Rc) -> Rc { - let mut cur = self.clone(); + pub fn layout_struct_base(this: &Rc) -> Rc { + let mut cur = this.clone(); while !cur.changes_layout() { match cur.best_base() { Some(b) => cur = b, @@ -1537,6 +1625,22 @@ impl TypeObject { crate::builtin_types::builtin_types().type_.clone() } + /// Whether [`Self::metaclass_or_type`] is `type` itself, without + /// cloning the metaclass handle. + pub fn metaclass_is_type(&self) -> bool { + let ext = self.c_ext_ptr.get(); + if ext != 0 { + if let Some(h) = METACLASS_DRIFT_HOOK.get() { + h(ext, self); + } + } + let ty = &crate::builtin_types::builtin_types().type_; + match self.metaclass.borrow().as_ref() { + Some(m) => Rc::ptr_eq(m, ty), + None => true, + } + } + /// `True` when `self` is a subclass of `other` (including itself). pub fn is_subclass_of(&self, other: &TypeObject) -> bool { let other_ptr = std::ptr::from_ref::(other); @@ -1578,6 +1682,22 @@ impl TypeObject { /// the attribute. Lets callers distinguish a dunder *supplied by a /// user class* from one inherited off a built-in (e.g. `object`'s /// identity `__hash__`). + /// How this type resolves `dunder`, memoised until the type or a base + /// changes. Hot paths that only need to know whether a dunder is + /// overridden, and by whom, use this instead of an MRO lookup by name. + #[inline] + pub fn dunder(&self, dunder: Dunder) -> DunderInfo { + let version = self.attr_version.get(); + let slot = &self.dunder_memo[dunder as usize]; + let memo = slot.get(); + if memo & u64::from(DunderInfo::VALID) != 0 && memo >> 8 == version { + return DunderInfo(memo as u8); + } + let info = DunderInfo::resolve(self.lookup_with_owner(dunder.name())); + slot.set(version << 8 | u64::from(info.0)); + info + } + pub fn lookup_with_owner(&self, name: &str) -> Option<(Object, Rc)> { // Fast pass — see `lookup` for the gate rationale. if !crate::object::exotic_str_keys_possible() { @@ -1871,6 +1991,15 @@ enum SlotData { }, } +/// Is `key` the slot name `name`? (Interned names usually share the +/// probe's storage, settled without reading either length.) +#[inline(always)] +fn key_named(key: &DictKey, name: &crate::shared_value::SharedStr) -> bool { + matches!(&key.0, Object::Str(stored) + if crate::shared_value::SharedStr::ptr_eq(stored, name) + || slot_name_eq(stored.as_ref(), name.as_ref())) +} + #[cfg(target_pointer_width = "64")] /// `stored == name` for a slot name, without the call into `memcmp`. /// Slot names are short (`_data`, `_head`, `x`) and almost always differ @@ -1898,6 +2027,29 @@ fn slot_name_eq(stored: &str, name: &str) -> bool { true } +/// The key for a newly populated slot `name`. The slots every raise +/// populates (`args`, `__traceback__`, the chaining links) share one +/// interned key each instead of allocating a string per exception. +fn slot_key(name: &str) -> DictKey { + const COMMON: [&str; 6] = [ + "args", + "__traceback__", + "__context__", + "__cause__", + "__suppress_context__", + "message", + ]; + static KEYS: std::sync::OnceLock<[Object; 6]> = std::sync::OnceLock::new(); + if let Some(i) = COMMON.iter().position(|c| *c == name) { + let keys = KEYS.get_or_init(|| COMMON.map(crate::stdlib::sys::intern_name)); + return DictKey(keys[i].clone()); + } + // Interned, as instance-dict keys are: a guard holding the interned + // name settles a slot's key by identity, and no instance allocates + // its own copy of the name. + DictKey(crate::stdlib::sys::intern_name(name)) +} + impl SlotStorage { /// Storage holding exactly `entries` (distinct `str` keys, in slot /// order), built in one step. @@ -2098,14 +2250,95 @@ impl SlotStorage { /// `get_mut(name)` with a position hint (see [`Self::get_hinted`]). #[inline] pub fn get_hinted_mut(&mut self, idx: usize, name: &str) -> Option<&mut Object> { - let at_hint = matches!( - self.get_index(idx), - Some((DictKey(Object::Str(stored)), _)) if slot_name_eq(stored.as_ref(), name) - ); - if at_hint { - return self.get_index_mut(idx).map(|(_, v)| v); + // One lookup at the hint. (The pointer ends the borrow, so the + // name scan below can take its own.) + let hit = match self.get_index_mut(idx) { + Some((DictKey(Object::Str(stored)), value)) + if std::ptr::eq(stored.as_ptr(), name.as_ptr()) + || slot_name_eq(stored.as_ref(), name) => + { + Some(std::ptr::from_mut(value)) + } + _ => None, + }; + match hit { + // SAFETY: the value lives in `self`, borrowed mutably for the + // returned lifetime; no other reference to it exists. + Some(value) => Some(unsafe { &mut *value }), + None => self.get_mut(name), } - self.get_mut(name) + } + + /// The values of the first `N` slots when they are named `names`, in + /// order (the layout of an instance whose `__init__` assigns them in + /// that order): one pass for a native method that reads several + /// slots. `None` when the store is laid out otherwise. + #[inline] + pub fn leading( + &self, + names: [&crate::shared_value::SharedStr; N], + ) -> Option<[&Object; N]> { + match &self.data { + SlotData::Small(entries) => { + let entries = entries.get(..N)?; + if !entries + .iter() + .zip(names) + .all(|((key, _), name)| key_named(key, name)) + { + return None; + } + Some(std::array::from_fn(|i| &entries[i].1)) + } + SlotData::Fixed { layout, values } => { + let (keys, values) = (layout.get(..N)?, values.get(..N)?); + if !keys + .iter() + .zip(names) + .all(|(key, name)| key_named(key, name)) + { + return None; + } + Some(std::array::from_fn(|i| &values[i])) + } + _ => None, + } + } + + /// [`Self::leading`], mutably. + #[inline] + pub fn leading_mut( + &mut self, + names: [&crate::shared_value::SharedStr; N], + ) -> Option<[&mut Object; N]> { + let values: &mut [Object] = match &mut self.data { + SlotData::Small(entries) => { + let entries = entries.get_mut(..N)?; + if !entries + .iter() + .zip(names) + .all(|((key, _), name)| key_named(key, name)) + { + return None; + } + let mut values = entries.iter_mut().map(|(_, value)| value); + return Some(std::array::from_fn(|_| values.next().expect("N entries"))); + } + SlotData::Fixed { layout, values } => { + let keys = layout.get(..N)?; + if !keys + .iter() + .zip(names) + .all(|(key, name)| key_named(key, name)) + { + return None; + } + values.get_mut(..N)? + } + _ => return None, + }; + let mut values = values.iter_mut(); + Some(std::array::from_fn(|_| values.next().expect("N values"))) } /// Mutable access to a populated slot's value, by name. @@ -2130,9 +2363,7 @@ impl SlotStorage { } pub fn insert(&mut self, name: &str, value: Object) -> Option { - self.insert_with_key(name, value, || { - DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) - }) + self.insert_with_key(name, value, || slot_key(name)) } /// Reuse an existing name allocation when populating a new slot. @@ -2284,9 +2515,7 @@ impl SlotStorage { } pub fn insert(&mut self, name: &str, value: Object) -> Option { - self.insert_with_key(name, value, || { - DictKey(Object::Str(crate::shared_value::SharedStr::from(name))) - }) + self.insert_with_key(name, value, || slot_key(name)) } /// Share the name allocation only for a newly populated slot. @@ -2332,7 +2561,7 @@ pub struct PyInstance { pub class: RefCell>, /// Instance attributes, allocated on the first write or exported handle. /// Reads through `get()` preserve the absence of an unused dictionary. - pub dict: crate::sync::LazyArc>, + pub dict: crate::inst_dict::InstDict, /// For instances of a subclass of an immutable built-in /// (`int`, `str`, `float`, `bytes`, `tuple`, …) this holds the /// underlying primitive value the instance *is* — the moral @@ -2344,7 +2573,7 @@ pub struct PyInstance { /// inline body by the extension's `tp_new` chain after allocation). /// Unwrapped by the numeric / comparison / hashing / conversion /// fast paths so e.g. `class C(int)` instances behave like real ints. - pub native: std::sync::OnceLock, + pub native: crate::sync::OnceBox, /// Mirrors CPython 3.13's "inline values" state observable through /// `_testinternalcapi.has_inline_values`: starts `true` for ordinary /// instances (native fixed weakrefs start `false`) and is @@ -2417,8 +2646,8 @@ impl PyInstance { pub fn new(class: Rc) -> Self { Self { class: RefCell::new(class), - dict: crate::sync::LazyArc::new(), - native: std::sync::OnceLock::new(), + dict: crate::inst_dict::InstDict::new(), + native: crate::sync::OnceBox::new(), inline_values: Cell::new(true), slots: RefCell::new(SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -2433,8 +2662,8 @@ impl PyInstance { pub fn with_native(class: Rc, native: Object) -> Self { Self { class: RefCell::new(class), - dict: crate::sync::LazyArc::new(), - native: std::sync::OnceLock::from(native), + dict: crate::inst_dict::InstDict::new(), + native: crate::sync::OnceBox::from(native), inline_values: Cell::new(true), slots: RefCell::new(SlotStorage::default()), hash_cache: crate::sync::CachedHash::new(None), @@ -2464,7 +2693,7 @@ impl PyInstance { *m.class.get_mut() = class; m.deferred.set(true); if hint > 0 { - if let Some(dict) = m.dict.get() { + if let Some(dict) = m.dict.published() { if let Ok(mut d) = dict.try_borrow_mut() { if d.capacity() < hint { d.map_mut_atomic_store().reserve(hint); @@ -2533,10 +2762,13 @@ impl PyInstance { if m.native.get().is_some() || m.finalize_ran.get() || m.c_body.get() != 0 { return; } + // Split values are atomic (the instance is deferred): clearing + // them runs no code. The allocation stays for the next tenant. + m.dict.split_mut().reset(); // An instance that never grew a `__dict__` is the common case // now and needs no reset; one that did keeps it for the next // tenant, cleared and carrying a fresh owner record. - if let Some(dict) = m.dict.get() { + if let Some(dict) = m.dict.published() { // The dict must be private too: `vars(obj)` / `obj.__dict__` // hand out the same `Arc`, and a holder must keep seeing the // dead instance's attributes, not the next tenant's. @@ -2602,7 +2834,7 @@ impl PyInstance { #[inline] pub(crate) fn clear_deferred_tracking(&self) { self.deferred.set(false); - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { match d.try_borrow() { Ok(d) => { d.take_deferred_owner(); @@ -2622,7 +2854,7 @@ impl PyInstance { // Retire the dict's copy of the record so the write barrier does // not track a second time, then track from the instance itself — // which works whether or not a `__dict__` was ever created. - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { if let Ok(d) = d.try_borrow() { d.take_deferred_owner(); } @@ -2658,6 +2890,17 @@ impl PyInstance { unsafe { &*self.class.as_ptr() } } + /// How this instance's class resolves `dunder` (see + /// [`TypeObject::dunder`]). + #[inline] + pub fn class_dunder(&self, dunder: Dunder) -> DunderInfo { + if crate::gil::free_threading_enabled() { + self.cls().dunder(dunder) + } else { + self.cls_raw().dunder(dunder) + } + } + /// Re-point the instance at a new class (`obj.__class__ = C`). pub fn set_cls(&self, class: Rc) { *self.class.borrow_mut() = class; @@ -2712,7 +2955,7 @@ impl Drop for PyInstance { fn drop(&mut self) { // A deferred-tracking record names this instance; the dict may // outlive it (`d = obj.__dict__`), so retire the record first. - if let Some(d) = self.dict.get() { + if let Some(d) = self.dict.published() { match d.try_borrow() { Ok(d) => { d.take_deferred_owner(); @@ -2768,8 +3011,8 @@ impl Drop for PyInstance { class: RefCell::new(self.cls()), dict: self.dict.clone(), native: match self.native.get() { - Some(v) => std::sync::OnceLock::from(v.clone()), - None => std::sync::OnceLock::new(), + Some(v) => crate::sync::OnceBox::from(v.clone()), + None => crate::sync::OnceBox::new(), }, inline_values: Cell::new(self.inline_values.get()), slots: RefCell::new(self.slots.borrow().clone()), @@ -2794,7 +3037,9 @@ mod slot_storage_tests { #[test] fn shared_layout_keeps_slot_storage_compact() { assert_eq!(std::mem::size_of::(), 32); - assert_eq!(std::mem::size_of::(), 128); + // The split `__dict__` values pointer and its cell: with the + // allocator's header the instance stays in the 160-byte class. + assert_eq!(std::mem::size_of::(), 136); } #[test] diff --git a/crates/weavepy-vm/src/vm_singletons.rs b/crates/weavepy-vm/src/vm_singletons.rs index ef5f8247..4ef06da4 100644 --- a/crates/weavepy-vm/src/vm_singletons.rs +++ b/crates/weavepy-vm/src/vm_singletons.rs @@ -174,6 +174,12 @@ fn make_singleton(cls: Rc) -> Object { /// `NotImplementedType` (an `object` subclass), so `type(NotImplemented)` /// and the MRO match CPython. pub fn not_implemented() -> Object { + not_implemented_ref().clone() +} + +/// [`not_implemented`] by reference. The singleton is process-wide, built +/// on the first thread to ask, whose type registry supplies its class. +fn not_implemented_ref() -> &'static Object { static SLOT: OnceLock = OnceLock::new(); SLOT.get_or_init(|| { let cls = crate::builtin_types::builtin_types() @@ -181,18 +187,35 @@ pub fn not_implemented() -> Object { .clone(); make_singleton(cls) }) - .clone() } /// Same idea for `Ellipsis` (the value of `...`); its class is the /// registry's `ellipsis` type. pub fn ellipsis() -> Object { + ellipsis_ref().clone() +} + +/// [`ellipsis`] by reference (see [`not_implemented_ref`]). +fn ellipsis_ref() -> &'static Object { static SLOT: OnceLock = OnceLock::new(); SLOT.get_or_init(|| { let cls = crate::builtin_types::builtin_types().ellipsis_.clone(); make_singleton(cls) }) - .clone() +} + +/// Whether `obj` is the process-wide singleton `single`, or another +/// instance of its type or of this thread's registry type `cls` (a thread +/// with its own registry, or a C-API proxy, sees the same singleton). +fn is_singleton_of(obj: &Object, single: &Object, cls: &Rc) -> bool { + let (Object::Instance(inst), Object::Instance(one)) = (obj, single) else { + return false; + }; + if Rc::ptr_eq(inst, one) { + return true; + } + let ty = inst.cls(); + Rc::ptr_eq(&ty, cls) || Rc::ptr_eq(&ty, &one.cls()) } /// `True` if `obj` is the canonical `Ellipsis` singleton — an instance of @@ -203,26 +226,24 @@ pub fn ellipsis() -> Object { /// (numpy's `prepare_index`) takes the right branch rather than rejecting a /// freshly-boxed proxy with "only integers, slices … are valid indices". pub fn is_ellipsis(obj: &Object) -> bool { - if let Object::Instance(inst) = obj { - return Rc::ptr_eq( - &inst.cls(), + matches!(obj, Object::Instance(_)) + && is_singleton_of( + obj, + ellipsis_ref(), &crate::builtin_types::builtin_types().ellipsis_, - ); - } - false + ) } /// `True` if `obj` is the canonical `NotImplemented` singleton. The C-API /// bridge maps it to the static `_Py_NotImplementedStruct` so extensions /// that compare against `Py_NotImplemented` by pointer behave correctly. pub fn is_not_implemented(obj: &Object) -> bool { - if let Object::Instance(inst) = obj { - return Rc::ptr_eq( - &inst.cls(), + matches!(obj, Object::Instance(_)) + && is_singleton_of( + obj, + not_implemented_ref(), &crate::builtin_types::builtin_types().not_implemented_type_, - ); - } - false + ) } /// CPython's `help`/`copyright`/`license`/`credits` builtins are @@ -295,7 +316,7 @@ pub fn publish_interpreter_seed(interp: &crate::Interpreter) { let mut slot = seed_slot().lock(); *slot = Some(interp.fork_for_thread()); drop(slot); - *seed_types_slot().lock() = Some(crate::builtin_types::builtin_types()); + *seed_types_slot().lock() = Some(crate::builtin_types::builtin_types_rc()); } /// Run `f` (typically a `Interpreter::new()` for a *sub*-interpreter) @@ -797,8 +818,8 @@ thread_local! { /// A queue request for an id in this set is the cascade observing /// itself and is dropped; requests for *other* objects (a child body /// whose last C pin fell during the teardown) still queue normally. - static CASCADING_IDS: RefCell> = - RefCell::new(std::collections::HashSet::new()); + static CASCADING_IDS: RefCell> = + RefCell::new(crate::fasthash::FxHashSet::default()); } /// RAII marker for one object's trip through the prompt reaper's cascade; diff --git a/crates/weavepy-vm/src/weakref_registry.rs b/crates/weavepy-vm/src/weakref_registry.rs index a98189bf..71ed558c 100644 --- a/crates/weavepy-vm/src/weakref_registry.rs +++ b/crates/weavepy-vm/src/weakref_registry.rs @@ -37,9 +37,9 @@ use crate::fasthash::ObjectIdHasher; use crate::shared_value::{SharedSlice, SharedStr, ThinArc}; use crate::sync::RefCell; +use crate::sync::{Rc as Arc, Weak}; use std::hash::BuildHasherDefault; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; -use std::sync::{Arc, Weak}; use crate::object::{DictKey, Object, StrKey}; @@ -548,7 +548,7 @@ pub fn queue_callbacks(cleared: Vec<(Arc, Option)>) { .py_ref .borrow() .as_ref() - .and_then(std::sync::Weak::upgrade) + .and_then(Weak::upgrade) .map(Object::Instance); if let Some(wr) = wr { crate::vm_singletons::push_pending_weakref_callback(cb, wr); diff --git a/crates/weavepy/src/lib.rs b/crates/weavepy/src/lib.rs index 91be56d7..d2835c35 100644 --- a/crates/weavepy/src/lib.rs +++ b/crates/weavepy/src/lib.rs @@ -87,7 +87,9 @@ impl Error { let _ = writeln!( s, " File \"{}\", line {}, in {}", - entry.filename, entry.lineno, entry.funcname + entry.filename(), + entry.lineno, + entry.funcname() ); } } diff --git a/crates/weavepy/tests/fixtures.rs b/crates/weavepy/tests/fixtures.rs index 9651824d..e3debb84 100644 --- a/crates/weavepy/tests/fixtures.rs +++ b/crates/weavepy/tests/fixtures.rs @@ -105,7 +105,7 @@ fn run_fixture(py_path: &Path) { let mut interp = vm::Interpreter::new(); let buf: Rc>> = Rc::new(RefCell::new(Vec::new())); - let sink: vm::Stdout = buf.clone() as Rc>; + let sink: vm::Stdout = vm::rc_unsize!(buf.clone() => RefCell); interp.set_stdout(sink); // Sibling files in the fixtures directory must be importable so // multi-file tests (e.g. `import _helper`) work. diff --git a/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out b/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out index 5a998283..2e8c0231 100644 --- a/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out +++ b/crates/weavepy/tests/fixtures/run/58_weakref_and_gc.out @@ -10,7 +10,7 @@ gc enabled: True disabled: False re-enabled: True collected is int: True -threshold: (700, 10, 10) +threshold: (2000, 10, 10) threshold: (800, 12, 14) count len: 3 objects is list: True diff --git a/docs/PERFORMANCE-CPYTHON-PARITY.md b/docs/PERFORMANCE-CPYTHON-PARITY.md new file mode 100644 index 00000000..3c922af1 --- /dev/null +++ b/docs/PERFORMANCE-CPYTHON-PARITY.md @@ -0,0 +1,141 @@ +# CPython parity performance + +This branch works toward making WeavePy faster than CPython 3.14 on every +benchmark fixture and on the everyday costs outside them: startup, imports, +and memory. This report records where it stands, how the numbers were +measured, and what still trails CPython. + +For the earlier passes, see [Performance measurements](PERFORMANCE.md) and +the reports it links to. + +## Method + +Timings compare a profile-guided release build of the branch +(`tools/pgo_build.py --skip-regrtest`, JIT on) with CPython 3.14.7 on the +macOS x86-64 development host (Intel Core i9-9980HK). Both executables are +native x86-64. + +Wall-time ratios come from paired, interleaved runs. Each cycle runs every +fixture once under each interpreter at the harness work sizes, alternating +which goes first; one warmup cycle is discarded and five are measured. A +fixture's ratio is the median of its per-cycle WeavePy/CPython ratios, using +each fixture's own timer (`WEAVEPY_BENCH_NS`), which excludes startup and +imports. Values below 1.00 mean WeavePy is faster. + +Instruction counts come from `/usr/bin/time -l` ("instructions retired") +on the host, or from callgrind in a Linux container when a per-function or +per-line breakdown is needed. A per-operation cost is the difference between +two work sizes divided by the difference in work, which cancels startup. +Instruction counts are deterministic enough to compare builds on a busy +machine; they don't capture cache or branch behavior, so a wall-time +checkpoint confirms them. + +## Benchmark fixtures + +Geometric mean over the 23 timed fixtures: **0.592**. + +| Fixture | Work | WeavePy | CPython | Ratio | +| --- | ---: | ---: | ---: | ---: | +| `deltablue` | 50 | 213.8 ms | 67.8 ms | 3.15 | +| `deque_ops` | 200,000 | 97.9 ms | 64.1 ms | 1.53 | +| `pickle_bench` | 40 | 11.6 ms | 8.5 ms | 1.34 | +| `generators` | 300,000 | 59.3 ms | 49.6 ms | 1.19 | +| `datetime_ops` | 60,000 | 50.7 ms | 52.7 ms | 0.98 | +| `str_methods` | 15,000 | 57.0 ms | 59.9 ms | 0.95 | +| `float_math` | 100,000 | 78.0 ms | 81.6 ms | 0.94 | +| `dict_ops` | 100,000 | 55.9 ms | 61.7 ms | 0.90 | +| `call_overhead` | 150,000 | 69.8 ms | 82.2 ms | 0.84 | +| `fannkuch` | 100,000 | 17.1 ms | 20.9 ms | 0.83 | +| `richards` | 50,000 | 16.1 ms | 20.3 ms | 0.80 | +| `nbody` | 20,000 | 33.0 ms | 42.9 ms | 0.77 | +| `attr_access` | 200,000 | 35.4 ms | 47.1 ms | 0.77 | +| `list_ops` | 10,000 | 33.8 ms | 45.1 ms | 0.75 | +| `json_bench` | 150 | 57.4 ms | 81.1 ms | 0.71 | +| `pidigits` | 500,000 | 1,421.0 ms | 2,167.8 ms | 0.67 | +| `pyaes` | 400 | 15.8 ms | 28.9 ms | 0.54 | +| `jitkernels` | 2,000 | 24.3 ms | 49.0 ms | 0.48 | +| `fib` | 27 | 8.0 ms | 21.6 ms | 0.37 | +| `spectral_norm` | 100 | 12.1 ms | 57.7 ms | 0.21 | +| `nested_loops` | 120 | 6.2 ms | 69.3 ms | 0.09 | +| `jitloop` | 1,000 | 7.5 ms | 92.0 ms | 0.08 | +| `sumvm` | 2,000,000 | 3.8 ms | 72.5 ms | 0.05 | + +Times are the medians of each interpreter's samples; ratios are the medians +of the paired ratios, so they needn't equal the quotient of the two times. + +## Startup and imports + +Whole-process instructions retired, warm stdlib cache: + +| Command | WeavePy | CPython | +| --- | ---: | ---: | +| `-c pass` | 91M | 127M | +| `import json` | 205M | 153M | +| `import dataclasses` | 349M | 196M | +| `import logging` | 513M | 258M | +| `import asyncio` | 689M | 365M | +| `import unittest` | 496M | 270M | + +Startup is cheaper than CPython's. Importing large parts of the standard +library still costs about twice as much. Three changes on this branch cut +those costs by 20% to 60%: + +- Compiled stdlib modules are cached in a WeavePy-native code format + (`weavepy_compiler::native_code`), which loads without unmarshalling CPython + bytecode and transcoding it back into WeavePy instructions. +- Slicing a tuple, `bytes`, `bytearray`, or a string with surrogates no longer + copies the whole source sequence. `re.compile` with `IGNORECASE` slices a + 64K-entry charset map 256 times, which had made `import logging` cost 1.19B + instructions. +- The collector's per-drop sweep of suspected-dead objects skips entries that + have used up their probe budget, so it no longer walks up to 256 entries at + every drop during imports. + +Peak memory after these imports is about twice CPython's (for example, +23 MiB against 12 MiB for `import json`). An allocation-site profile +attributes most of the difference to compiled code objects: their per-code +structure, per-instruction line and column tables, and the materialized +constants and names built on first use. + +## What still trails CPython + +The four slower fixtures share a cause: operations that CPython's specialized +bytecodes do in 10 to 30 instructions take WeavePy two to four times as many. +Per operation, with the JIT off: + +| Operation | WeavePy | CPython | +| --- | ---: | ---: | +| Instance attribute read (`o.x`) | 160 | 64 | +| Class attribute read through an instance (`o.K`) | 240 | 60 | +| Global read | 150 | 50 | +| `isinstance(o, C)` | 980 | 240 | +| Call of a small non-leaf function | 1,700 | 440 | + +The costs come from the interpreter's structure rather than from any single +slow path: + +- **Loop state.** The core loop keeps about 17 values live across its arms, + which contain calls; x86-64 has six callee-saved registers, so most of the + state lives on the stack and every dispatch reloads it. +- **Data layout.** An instance field read passes the site's slot table, the + class version, the class's shared keys, and the instance's split values, + each behind its own pointer; CPython checks a type version and reads an + inline slot. +- **Calls.** An inline activation binds cells, pools its locals vector, + grades its locals for collector bookkeeping on exit, and switches the + running frame, several hundred instructions per call and return. + +`deltablue` also runs at a lower IPC than CPython (2.13 against 2.47): its +interpreter loops overflow the instruction cache, and their single dispatch +branches predict poorly. + +## Validation + +Every change on the branch passes the VM unit tests (420), the bundled +regression suite (248), the semantics comparison scripts, and the CPython +regression suites most exposed to it. The changes described above were +checked with, among others, `test_bytes`, `test_tuple`, `test_slice`, +`test_str`, `test_re`, `test_marshal`, `test_import`, `test_code`, +`test_compile`, `test_logging`, `test_gc`, `test_weakref`, `test_io`, +`test_ssl`, `test_asyncio`, `test_scope`, `test_global`, `test_module`, +`test_builtin`, `test_sys_settrace`, and `test_monitoring`. diff --git a/docs/PERFORMANCE.md b/docs/PERFORMANCE.md index 2d6c108c..f7cbf22c 100644 --- a/docs/PERFORMANCE.md +++ b/docs/PERFORMANCE.md @@ -8,6 +8,8 @@ For the latest object metadata, numeric text, and enumeration changes, see [Runtime metadata and numeric text performance](PERFORMANCE-RUNTIME-METADATA.md). For the subsequent dispatch, allocation, and garbage-collection changes, see [Dispatch, allocation, and memory performance](PERFORMANCE-DISPATCH-MEMORY.md). +For the branch comparing WeavePy with CPython 3.14 across the benchmark +fixtures, startup, and imports, see [CPython parity performance](PERFORMANCE-CPYTHON-PARITY.md). The September 8, 2026, optimization pass reduces warm-cache process startup from 50.7 ms to 32.3 ms on the measured macOS ARM64 host. Across all 24 existing diff --git a/tests/regrtest/test_core_branch_fusion.py b/tests/regrtest/test_core_branch_fusion.py new file mode 100644 index 00000000..406f1de0 --- /dev/null +++ b/tests/regrtest/test_core_branch_fusion.py @@ -0,0 +1,100 @@ +"""Branches on locals and class attributes read through instances. + +The core loop decides `if x is None`, `if x is not None` and `if x:` on +a local without loading it, and serves a class's scalar attribute read +through an instance from a site cache. These cases check each shape's +edges: every truth kind, unbound locals, and an instance attribute or +class change that shadows the cached value. +""" + +import unittest + + +class Holder: + LIMIT = 3 + RATE = 0.5 + FLAG = True + NOTHING = None + + +def branches(values): + out = [] + for v in values: + if v is None: + out.append("none") + if v is not None: + out.append("some") + if v: + out.append("truthy") + else: + out.append("falsy") + return out + + +class BranchFusionTests(unittest.TestCase): + def test_truth_kinds(self): + values = [None, 0, 1, -1, 0.0, 0.5, "", "x", (), (1,), [], [1], + {}, {1: 2}, True, False, object()] + for _ in range(3): + self.assertEqual(branches(values), [ + x for v in values for x in ( + (["none"] if v is None else ["some"]) + + (["truthy"] if v else ["falsy"])) + ]) + + def test_custom_truth(self): + class Falsy: + def __bool__(self): + return False + + class Empty: + def __len__(self): + return 0 + + for _ in range(3): + self.assertEqual(branches([Falsy(), Empty()]), + ["some", "falsy", "some", "falsy"]) + + def test_unbound_local(self): + def f(flag): + if flag: + x = None + if x is None: + return "none" + return "other" + + for _ in range(3): + self.assertEqual(f(True), "none") + with self.assertRaises(UnboundLocalError): + f(False) + + def test_class_scalars_through_instances(self): + h = Holder() + for _ in range(50): + self.assertEqual((h.LIMIT, h.RATE, h.FLAG, h.NOTHING), + (3, 0.5, True, None)) + h.LIMIT = 7 + self.assertEqual(h.LIMIT, 7) + del h.LIMIT + self.assertEqual(h.LIMIT, 3) + Holder.LIMIT = 9 + try: + self.assertEqual([h.LIMIT for _ in range(5)], [9] * 5) + finally: + Holder.LIMIT = 3 + + def test_shadowing_through_dict_and_subclass(self): + class Sub(Holder): + pass + + s = Sub() + for _ in range(20): + self.assertEqual(s.LIMIT, 3) + Sub.LIMIT = 4 + self.assertEqual(s.LIMIT, 4) + s.__dict__["LIMIT"] = 5 + self.assertEqual(s.LIMIT, 5) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_datetime_fields.py b/tests/regrtest/test_datetime_fields.py index e69fa838..9215bd36 100644 --- a/tests/regrtest/test_datetime_fields.py +++ b/tests/regrtest/test_datetime_fields.py @@ -89,7 +89,10 @@ def record_time(*args): return result try: reference._date_fields, reference._time_fields = record_date, record_time - datetime.datetime(2024, 2, 29, 12, 34, 56, 123456) + # The exact class with int fields is built natively, before + # `__new__`; a subclass runs `__new__` and its helpers. + value = datetime.datetime(2024, 2, 29, 12, 34, 56, 123456) + assert calls == [] and value == Moment(2024, 2, 29, 12, 34, 56, 123456) assert calls == [('date', True), ('time', True)] calls.clear() datetime.date(Index('year', 2024), 2, 29) diff --git a/tests/regrtest/test_extended_param_calls.py b/tests/regrtest/test_extended_param_calls.py new file mode 100644 index 00000000..e709faba --- /dev/null +++ b/tests/regrtest/test_extended_param_calls.py @@ -0,0 +1,227 @@ +"""Calls of functions with *args, **kwargs, and keyword-only parameters, +and f(*args) spreads, bind exactly as the full binder does.""" + + +def star(*args): + return args + + +def mixed(a, b=2, *rest, flag=False, **kw): + return a, b, rest, flag, kw + + +def kwonly(a, *, sep="-", end="!"): + return sep.join(a) + end + + +def required(a, *, need): + return a, need + + +def fwd(*args): + return mixed(*args) + + +def fwd_kwonly(*args): + return kwonly(*args) + + +def counter(*args): + count = 0 + for _ in args: + count += 1 + return count + + +def make_adder(n): + def add(*xs): + return n + sum(xs) + return add + + +class Box: + def method(self, *args, scale=1): + return tuple(x * scale for x in args) + + +EMPTY = () +box = Box() +add3 = make_adder(3) +dicts = [] +for i in range(2000): + assert star() == () and star() is EMPTY + assert star(i) == (i,) + assert star(i, i + 1, "x") == (i, i + 1, "x") + a, b, rest, flag, kw = mixed(i) + assert (a, b, rest, flag, kw) == (i, 2, (), False, {}) + dicts.append(kw) + assert mixed(i, 5, 6, 7) == (i, 5, (6, 7), False, {}) + assert kwonly("ab") == "a-b!" + assert fwd(i, 1, 2) == (i, 1, (2,), False, {}) + assert fwd(i) == (i, 2, (), False, {}) + assert fwd_kwonly("xyz") == "x-y-z!" + assert counter(*range(i % 7)) == i % 7 + assert counter(*(1, 2, 3)) == 3 + assert add3(i, 1) == i + 4 + assert box.method(i, 2) == (i, 2) + assert Box.method(box, i) == (i,) + spread = (i, 1) + assert mixed(*spread) == (i, 1, (), False, {}) + assert star(*[i, i]) == (i, i) + if i % 500 == 0: + try: + required("a") + except TypeError as e: + assert "missing 1 required keyword-only argument: 'need'" in str(e), e + else: + raise AssertionError("missing keyword-only argument accepted") + try: + kwonly("a", "b") + except TypeError as e: + assert "takes 1 positional argument but 2 were given" in str(e), e + else: + raise AssertionError("extra positional argument accepted") + try: + fwd() + except TypeError as e: + assert "missing 1 required positional argument: 'a'" in str(e), e + else: + raise AssertionError("missing positional argument accepted") + + +def fwd_all(*args, **kwargs): + return mixed(*args, **kwargs) + + +def posonly(a, /, b, **kw): + return a, b, kw + + +class Wrapper: + __slots__ = ("func",) + + def __init__(self, func): + self.func = func + + def __call__(self, *args, **kwargs): + return self.func(*args, **kwargs) + + +class Plain: + def __call__(self, x, y=1): + return x * y + + +wrapped = Wrapper(mixed) +plain = Plain() +for i in range(2000): + assert fwd_all(i, flag=True) == (i, 2, (), True, {}) + assert fwd_all(i, b=3, extra=4) == (i, 3, (), False, {"extra": 4}) + assert fwd_all(a=i) == (i, 2, (), False, {}) + assert fwd_all(i, 5, 6, z=1, y=2) == (i, 5, (6,), False, {"z": 1, "y": 2}) + assert list(fwd_all(i, q=1, p=2)[4]) == ["q", "p"] + assert posonly(i, b=2, a=3) == (i, 2, {"a": 3}) + assert posonly(*(i, 1), **{"a": 5}) == (i, 1, {"a": 5}) + assert wrapped(i) == (i, 2, (), False, {}) + assert wrapped(i, flag=1) == (i, 2, (), 1, {}) + assert plain(i) == i and plain(i, 3) == 3 * i and plain(i, y=2) == 2 * i + if i % 500 == 0: + try: + fwd_all(i, a=1) + except TypeError as e: + assert "multiple values for argument 'a'" in str(e), e + else: + raise AssertionError("duplicate argument accepted") + try: + kwonly(*("ab",), **{"nope": 1}) + except TypeError as e: + assert "unexpected keyword argument 'nope'" in str(e), e + else: + raise AssertionError("unexpected keyword accepted") + try: + plain() + except TypeError as e: + assert "missing 1 required positional argument: 'x'" in str(e), e + else: + raise AssertionError("missing argument accepted") + +Plain.__call__ = lambda self, x, y=1: x + y +assert plain(2, 3) == 5 +del Plain.__call__ +try: + plain(1) +except TypeError as e: + assert "not callable" in str(e), e +else: + raise AssertionError("uncallable instance called") + +# A finalizable argument whose last reference goes with a spread call's +# operands is finalized when the call returns. +FINALIZED = [] + + +class Finalized: + def __del__(self): + FINALIZED.append(1) + + +def sink(obj=None, **kw): + return None + + +def spread(*args, **kwargs): + return sink(*args, **kwargs) + + +class Keeper: + def take(self, *args, **kwargs): + return None + + +keeper = Keeper() +for i in range(200): + del FINALIZED[:] + spread(Finalized(), flag=i) + assert FINALIZED == [1], FINALIZED + spread(obj=Finalized()) + assert FINALIZED == [1, 1], FINALIZED + spread(i, other=Finalized()) + assert FINALIZED == [1, 1, 1], FINALIZED + sink(*(Finalized(),), **{"x": 1}) + assert FINALIZED == [1, 1, 1, 1], FINALIZED + # Only the `**` mapping holds it (pickle's `save_reduce(obj=obj, *rv)`). + sink(*(), obj=Finalized()) + assert FINALIZED == [1, 1, 1, 1, 1], FINALIZED + keeper.take(*(i,), obj=Finalized()) + assert FINALIZED == [1, 1, 1, 1, 1, 1], FINALIZED + +# Every call's **kwargs is its own dictionary. +assert len({id(d) for d in dicts}) == len(dicts) +dicts[0]["x"] = 1 +assert dicts[1] == {} + +# Replaced defaults take effect at once. +kwonly.__kwdefaults__ = {"sep": "+", "end": "?"} +assert kwonly("ab") == "a+b?" +kwonly.__kwdefaults__ = None +try: + kwonly("ab") +except TypeError as e: + assert "keyword-only" in str(e), e +else: + raise AssertionError("cleared __kwdefaults__ ignored") +mixed.__defaults__ = (9,) +assert mixed(1) == (1, 9, (), False, {}) + + +def deep(n, *args): + return deep(n + 1, *args) + + +try: + deep(0, 1) +except RecursionError: + pass +else: + raise AssertionError("unbounded recursion") +print("ok") diff --git a/tests/regrtest/test_fused_method_args.py b/tests/regrtest/test_fused_method_args.py new file mode 100644 index 00000000..ee728d9f --- /dev/null +++ b/tests/regrtest/test_fused_method_args.py @@ -0,0 +1,69 @@ +"""Fused `x.m(...)` calls whose argument is a small int expression. + +The core loop runs `LOAD_FAST x; LOAD_ATTR m; ; CALL` as one step +for a local list, dict, set, str or native-method instance, folding an +argument like `i + 1` in place. These cases check the fold's edges: +overflow into big integers, non-int operands, and raising calls. +""" + +import sys +import unittest +from collections import deque + + +class FusedArgumentTests(unittest.TestCase): + def test_folded_int_arguments(self): + q = deque() + xs = [] + for i in range(5): + q.append(i + 1) + xs.append(i * 3) + xs.append(i - 10) + xs.append(i & 1) + xs.append(i | 8) + xs.append(i ^ 5) + self.assertEqual(list(q), [1, 2, 3, 4, 5]) + self.assertEqual(xs[:6], [0, -10, 0, 8, 5, 3]) + + def test_overflow_into_big_integers(self): + big = sys.maxsize + q = deque() + xs = [] + for i in range(big - 1, big + 1): + q.append(i + 1) + xs.append(i * 2) + self.assertEqual(list(q), [big, big + 1]) + self.assertEqual(xs, [(big - 1) * 2, big * 2]) + + def test_non_int_operands(self): + xs = [] + for x in (1.5, 2.5): + xs.append(x + 1) + s = "ab" + for t in ("c", "d"): + xs.append(s + t) + for b in (True, False): + xs.append(b + 1) + self.assertEqual(xs, [2.5, 3.5, "abc", "abd", 2, 1]) + + def test_raising_calls(self): + q = deque() + with self.assertRaises(IndexError): + for i in range(3): + q.pop() + d = {} + for i in range(3): + self.assertIsNone(d.get(i + 1)) + with self.assertRaises(TypeError): + for i in range(3): + q.append(i + 1, 2) + + def test_maxlen_trim_with_folded_argument(self): + ring = deque(maxlen=3) + for i in range(10): + ring.append(i + 100) + self.assertEqual(list(ring), [107, 108, 109]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_generator_drain_fold.py b/tests/regrtest/test_generator_drain_fold.py new file mode 100644 index 00000000..6c5e1875 --- /dev/null +++ b/tests/regrtest/test_generator_drain_fold.py @@ -0,0 +1,85 @@ +"""Consumers that drain a generator (sum, list, tuple, join) see exactly its +own yields, even when it resumes other generators.""" + + +def inner(): + yield 1 + yield 2 + + +def outer(): + for x in inner(): + pass + yield 10 + + +def outer_sum(): + yield sum(inner()) + yield 5 + + +def outer_list(): + yield list(inner()) + yield [3] + + +# A materialized frame sends the outer resume down the general path; the +# inner generator's resumes must not fold into the outer consumer. +g = outer() +frame = g.gi_frame +frame.f_locals +assert sum(g) == 10 +g = outer() +frame = g.gi_frame +frame.f_locals +assert list(g) == [10] +del frame +assert sum(outer()) == 10 +assert list(outer()) == [10] +assert sum(outer_sum()) == 8 +assert list(outer_list()) == [[1, 2], [3]] + +assert list(x * 2 for x in range(5)) == [0, 2, 4, 6, 8] +assert tuple(str(x) for x in range(3)) == ("0", "1", "2") +assert "".join(chr(65 + x) for x in range(3)) == "ABC" +assert sorted(-x for x in range(4)) == [-3, -2, -1, 0] +assert list(x for x in ()) == [] + + +class Tracked: + alive = 0 + + def __init__(self): + Tracked.alive += 1 + + def __del__(self): + Tracked.alive -= 1 + + +items = list(Tracked() for _ in range(100)) +assert Tracked.alive == 100 +del items +assert Tracked.alive == 0 + + +def raises_midway(): + yield 1 + yield 2 + raise KeyError("stop") + + +try: + list(raises_midway()) +except KeyError as e: + assert e.args == ("stop",) +else: + raise AssertionError("list() swallowed the error") + + +def returns_value(): + yield 1 + return "done" + + +assert list(returns_value()) == [1] +print("ok") diff --git a/tests/regrtest/test_generator_fast_steps.py b/tests/regrtest/test_generator_fast_steps.py new file mode 100644 index 00000000..fd8b3bb9 --- /dev/null +++ b/tests/regrtest/test_generator_fast_steps.py @@ -0,0 +1,234 @@ +"""Generators whose bodies take fast steps (see `gen_fast` in the VM). + +Each case checks a behavior at the boundary between the fast steps and +the general loop: overflow into big integers, raises, returns, sends, +throws and closes after partial iteration, nesting, and folding +consumers. +""" + +import sys +import threading +import unittest + + +def counter(n): + i = 0 + while i < n: + yield i + i += 1 + + +def squares(it): + for x in it: + yield x * x + + +def evens(it): + for x in it: + if x & 1 == 0: + yield x + + +def over_range(n): + for i in range(n): + yield i * 3 + + +def over_list(xs): + for x in xs: + yield x + 1 + + +def doubling(start, n): + x = start + for _ in range(n): + yield x + x = x * 2 + + +def floats(n): + x = 0.5 + i = 0 + while i < n: + yield x + x = x * 1.5 + i += 1 + + +def divider(xs): + for x in xs: + yield 10 // x + + +def returning(n): + i = 0 + while i < n: + yield i + i += 1 + return "done" + + +class FastStepTests(unittest.TestCase): + def test_counter_and_pipeline(self): + self.assertEqual(list(counter(5)), [0, 1, 2, 3, 4]) + self.assertEqual(list(evens(squares(counter(10)))), [0, 4, 16, 36, 64]) + self.assertEqual(sum(evens(squares(counter(1000)))), sum( + x * x for x in range(1000) if (x * x) % 2 == 0)) + total = 0 + for x in squares(counter(100)): + total += x + self.assertEqual(total, sum(x * x for x in range(100))) + + def test_iterables_inside(self): + self.assertEqual(list(over_range(5)), [0, 3, 6, 9, 12]) + self.assertEqual(list(over_list([1, 2, 3])), [2, 3, 4]) + self.assertEqual(list(over_list((1.5, 2.5))), [2.5, 3.5]) + self.assertEqual(list(over_list([])), []) + + def test_overflow_to_big_integers(self): + values = list(doubling(1 << 60, 8)) + self.assertEqual(values, [(1 << 60) << k for k in range(8)]) + self.assertEqual(list(squares(doubling(3 << 30, 4))), + [(3 << 30 << k) ** 2 for k in range(4)]) + + def test_floats(self): + self.assertEqual(list(floats(4)), [0.5, 0.75, 1.125, 1.6875]) + + def test_raise_midway(self): + g = divider([5, 2, 0, 1]) + self.assertEqual(next(g), 2) + self.assertEqual(next(g), 5) + with self.assertRaises(ZeroDivisionError): + next(g) + with self.assertRaises(StopIteration): + next(g) + + def test_return_value(self): + g = returning(2) + self.assertEqual(next(g), 0) + self.assertEqual(next(g), 1) + with self.assertRaises(StopIteration) as cm: + next(g) + self.assertEqual(cm.exception.value, "done") + + def test_send_and_throw_after_fast_steps(self): + g = counter(10) + self.assertEqual(next(g), 0) + self.assertEqual(next(g), 1) + self.assertEqual(g.send(None), 2) + with self.assertRaises(KeyError): + g.throw(KeyError("k")) + self.assertIsNone(g.gi_frame) + + def test_close_after_fast_steps(self): + g = squares(counter(10)) + self.assertEqual([next(g), next(g), next(g)], [0, 1, 4]) + g.close() + with self.assertRaises(StopIteration): + next(g) + + def test_frame_inspection(self): + g = counter(10) + next(g) + next(g) + frame = g.gi_frame + self.assertEqual(frame.f_locals["i"], 1) + self.assertEqual(next(g), 2) + self.assertEqual(frame.f_locals["i"], 2) + self.assertFalse(g.gi_running) + + def test_nested_inner_partial(self): + # The inner generator overflows partway through a fast step of + # the outer one. + def outer(it): + for x in it: + yield x + 1 + + self.assertEqual(list(outer(doubling(1 << 61, 4))), + [(1 << 61 << k) + 1 for k in range(4)]) + g = outer(returning(3)) + self.assertEqual(list(g), [1, 2, 3]) + + def test_yield_from_fast_inner(self): + def delegator(n): + r = yield from returning(n) + yield r + + self.assertEqual(list(delegator(3)), [0, 1, 2, "done"]) + + def test_already_executing(self): + def selfish(): + for x in g: + yield x + + g = selfish() + with self.assertRaises(ValueError): + next(g) + + def test_folding_consumers(self): + self.assertEqual(sum(counter(100)), 4950) + self.assertEqual(list(squares(counter(5))), [0, 1, 4, 9, 16]) + self.assertEqual(tuple(evens(counter(7))), (0, 2, 4, 6)) + self.assertEqual(sum(floats(3)), 0.5 + 0.75 + 1.125) + self.assertEqual(sum(doubling(1 << 62, 3)), (1 << 62) * 7) + + def test_tracing_sees_lines(self): + seen = [] + + def tracer(frame, event, arg): + if frame.f_code is counter.__code__ and event == "line": + seen.append(frame.f_lineno) + return tracer + + sys.settrace(tracer) + try: + list(counter(2)) + finally: + sys.settrace(None) + self.assertTrue(seen) + + def test_long_nested_pipelines(self): + # Long enough to cross many eval-breaker checkpoints, which stop + # a fast step partway through the inner generators. + n = 20000 + self.assertEqual(sum(evens(squares(counter(n)))), + sum(x * x for x in range(n) if x * x % 2 == 0)) + self.assertEqual(sum(x + 1 for x in counter(n)), n * (n + 1) // 2) + self.assertEqual(list(squares(over_range(n)))[-1], ((n - 1) * 3) ** 2) + total = 0 + for x in squares(squares(counter(3000))): + total += x + self.assertEqual(total, sum(x ** 4 for x in range(3000))) + + def test_other_threads_run_during_a_folded_pipeline(self): + progress = [] + stop = threading.Event() + + def worker(): + while not stop.is_set(): + progress.append(1) + + t = threading.Thread(target=worker) + t.start() + try: + before = len(progress) + sum(evens(squares(counter(300000)))) + during = len(progress) - before + finally: + stop.set() + t.join() + self.assertGreater(during, 0) + + def test_recursion_depth(self): + def chain(depth): + if depth == 0: + yield from counter(3) + return + for x in chain(depth - 1): + yield x + + self.assertEqual(list(chain(30)), [0, 1, 2]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_global_rebinding.py b/tests/regrtest/test_global_rebinding.py new file mode 100644 index 00000000..a456d7ab --- /dev/null +++ b/tests/regrtest/test_global_rebinding.py @@ -0,0 +1,112 @@ +"""Rebinding globals while code that reads them stays hot. + +Rebinding an existing global replaces its value in place, which leaves +the module dict's key layout (and every cached global load) standing; +caches that remember a global's value must still see the new one. +These cases rebind functions, classes and scalars from inside +functions, at module level, and through `globals()`, while loops that +read them run long enough to be specialized and compiled. +""" + +import unittest + +COUNTER = 0 +SCALE = 2 + + +def double(x): + return x * 2 + + +def triple(x): + return x * 3 + + +def apply_all(n): + total = 0 + for i in range(n): + total += double(i) + return total + + +def scaled(n): + total = 0 + for i in range(n): + total += i * SCALE + return total + + +def bump(n): + global COUNTER + for _ in range(n): + COUNTER += 1 + return COUNTER + + +def rebind_double(fn): + global double + double = fn + + +class GlobalRebindingTests(unittest.TestCase): + def tearDown(self): + global double, SCALE + double = GlobalRebindingTests._double + SCALE = 2 + + def test_rebound_function_is_seen(self): + for _ in range(5): + self.assertEqual(apply_all(1000), 999000) + rebind_double(triple) + for _ in range(5): + self.assertEqual(apply_all(1000), 1498500) + rebind_double(lambda x: 0) + self.assertEqual(apply_all(1000), 0) + + def test_rebound_scalar_is_seen(self): + global SCALE + for _ in range(5): + self.assertEqual(scaled(100), 9900) + SCALE = 5 + self.assertEqual(scaled(100), 24750) + globals()["SCALE"] = 7 + self.assertEqual(scaled(100), 34650) + + def test_counter_in_loop(self): + global COUNTER + COUNTER = 0 + self.assertEqual(bump(10000), 10000) + self.assertEqual(bump(5), 10005) + self.assertEqual(COUNTER, 10005) + + def test_new_global_shadows_builtin(self): + def uses_len(xs): + return len(xs) + + for _ in range(200): + self.assertEqual(uses_len([1, 2, 3]), 3) + globals()["len"] = lambda xs: -1 + try: + self.assertEqual(uses_len([1, 2, 3]), -1) + finally: + del globals()["len"] + self.assertEqual(uses_len([1, 2, 3]), 3) + + def test_module_level_rebinding(self): + ns = {} + exec( + "def f():\n" + " return X\n" + "X = 0\n" + "seen = []\n" + "for X in range(500):\n" + " seen.append(f())\n", + ns, + ) + self.assertEqual(ns["seen"], list(range(500))) + + +GlobalRebindingTests._double = double + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_leaf_builtin_calls.py b/tests/regrtest/test_leaf_builtin_calls.py new file mode 100644 index 00000000..f0d9da2e --- /dev/null +++ b/tests/regrtest/test_leaf_builtin_calls.py @@ -0,0 +1,123 @@ +"""Leaf functions that call read-only builtins keep their results, errors, +and callbacks.""" +import math + + +class Box: + def __init__(self, v): + self.v = v + + +class Key: + calls = 0 + + def __init__(self, k): + self.k = k + + def __hash__(self): + return hash(self.k) + + def __eq__(self, other): + Key.calls += 1 + return isinstance(other, Key) and other.k == self.k + + +class Sized: + def __len__(self): + return 7 + + +def f_len(x, y): + return len(x) + y + + +def f_isinstance(x): + return isinstance(x, int) + + +def f_get(d, k): + return d.get(k, -1) + + +def f_sqrt(x): + return math.sqrt(x) + 1.0 + + +def f_upper(s): + return s.upper() + + +def f_join(s, parts): + return s.join(parts) + + +def f_format(s, a, b): + return s.format(a, b) + + +def f_append(items, x): + return items.append(x) + + +def f_mixed(box, seq): + return len(seq) + box.v + + +def f_nested(x, seq): + return f_len(seq, x) * 2 + + +def f_floor(x): + return math.floor(x) + + +for i in range(3000): + assert f_len([1, 2, 3], i) == 3 + i + assert f_len("abcd", i) == 4 + i + assert f_len({1: 2}, i) == 1 + i + assert f_len(Sized(), i) == 7 + i + assert f_isinstance(i) is True + assert f_isinstance("x") is False + assert f_isinstance(True) is True + assert f_get({"a": i}, "a") == i + assert f_get({"a": i}, "b") == -1 + assert f_get({1: i}, 1.0) == i + assert f_sqrt(float(i)) == math.sqrt(i) + 1.0 + assert f_upper("ab%d" % i) == "AB%d" % i + assert f_join("-", ["a", str(i)]) == "a-%d" % i + assert f_format("{}:{}", i, 2.5) == "%d:2.5" % i + items = [] + assert f_append(items, i) is None and items == [i] + assert f_mixed(Box(i), (1, 2)) == 2 + i + assert f_nested(i, [0] * (i % 5)) == 2 * (i % 5 + i) + assert f_floor(i / 3) == i // 3 + if i % 500 == 0: + try: + f_sqrt(-1.0) + except ValueError as e: + assert str(e) == "expected a nonnegative input, got -1.0", e + else: + raise AssertionError("sqrt(-1.0) returned") + try: + f_len(5, 1) + except TypeError as e: + assert "has no len()" in str(e), e + else: + raise AssertionError("len(5) returned") + try: + f_floor(float("inf")) + except OverflowError: + pass + else: + raise AssertionError("floor(inf) returned") + try: + f_upper(5) + except AttributeError: + pass + else: + raise AssertionError("(5).upper() returned") + # A key whose comparison runs Python code takes the ordinary call. + before = Key.calls + assert f_get({Key(1): 2}, Key(1)) == 2 + assert Key.calls == before + 1 +print("ok") diff --git a/tests/regrtest/test_leaf_constructors.py b/tests/regrtest/test_leaf_constructors.py new file mode 100644 index 00000000..ff824563 --- /dev/null +++ b/tests/regrtest/test_leaf_constructors.py @@ -0,0 +1,145 @@ +"""Constructors whose `__init__` runs without a frame. + +A class's `__init__` that only stores into `self` runs as a leaf: its +body is evaluated in place on the fresh instance. These cases check the +shapes around that path: trailing default parameters (including ones +rebound through `__defaults__`), empty list and dict literals, and +bodies that stop partway and fall back to the ordinary call. +""" + +import gc +import unittest + + +class WithDefaults: + def __init__(self, name, value=0, mark=None): + self.name = name + self.value = value + self.mark = mark + + +class WithContainers: + def __init__(self, name): + self.name = name + self.items = [] + self.index = {} + self.more = [] + + +class ManyLists: + def __init__(self): + self.a = [] + self.b = [] + self.c = [] + self.d = [] + self.e = [] + self.f = [] + + +class Declines: + def __init__(self, x, extra=1): + self.items = [] + self.x = x + self.y = x + extra + + +def build(cls, n, *args): + return [cls(*args) for _ in range(n)] + + +class LeafConstructorTests(unittest.TestCase): + def test_trailing_defaults(self): + objs = build(WithDefaults, 50, "v") + self.assertTrue(all(o.value == 0 and o.mark is None for o in objs)) + objs = build(WithDefaults, 50, "v", 5) + self.assertTrue(all(o.value == 5 and o.mark is None for o in objs)) + objs = build(WithDefaults, 50, "v", 5, "m") + self.assertTrue(all((o.value, o.mark) == (5, "m") for o in objs)) + self.assertEqual(vars(objs[0]), {"name": "v", "value": 5, "mark": "m"}) + + def test_arity_errors(self): + for _ in range(50): + with self.assertRaises(TypeError): + WithDefaults() + with self.assertRaises(TypeError): + WithDefaults(1, 2, 3, 4) + + def test_rebound_defaults(self): + class C: + def __init__(self, a, b=1): + self.a = a + self.b = b + + self.assertEqual([C(0).b for _ in range(50)], [1] * 50) + C.__init__.__defaults__ = (2,) + self.assertEqual([C(0).b for _ in range(50)], [2] * 50) + C.__init__.__defaults__ = None + for _ in range(3): + with self.assertRaises(TypeError): + C(0) + + def test_fresh_containers(self): + objs = build(WithContainers, 50, "n") + for o in objs: + self.assertEqual((o.items, o.index, o.more), ([], {}, [])) + # Each instance gets its own containers. + objs[0].items.append(1) + objs[0].index["k"] = 2 + self.assertEqual(objs[1].items, []) + self.assertEqual(objs[1].index, {}) + self.assertIsNot(objs[0].items, objs[0].more) + self.assertEqual(len({id(o.items) for o in objs}), len(objs)) + self.assertTrue(gc.is_tracked(objs[1].items)) + self.assertTrue(gc.is_tracked(objs[1].index)) + + def test_more_containers_than_scratch(self): + for o in build(ManyLists, 50): + self.assertEqual([o.a, o.b, o.c, o.d, o.e, o.f], [[]] * 6) + self.assertEqual(len({id(v) for v in vars(o).values()}), 6) + + def test_fallback_after_partial_body(self): + objs = build(Declines, 50, 1) + self.assertTrue(all((o.x, o.y, o.items) == (1, 2, []) for o in objs)) + big = 1 << 62 + objs = build(Declines, 50, big, big) + self.assertTrue(all(o.y == big * 2 for o in objs)) + objs = build(Declines, 50, 1.5) + self.assertTrue(all(o.y == 2.5 for o in objs)) + with self.assertRaises(TypeError): + Declines("a") + + def test_cycles_are_collected(self): + class Node: + def __init__(self, prev=None): + self.prev = prev + self.next = [] + + gc.collect() + for _ in range(20): + prev = None + for _ in range(50): + node = Node(prev) + if prev is not None: + prev.next.append(node) + prev = node + del prev, node + gc.collect() + self.assertFalse(any(type(o) is Node for o in gc.get_objects())) + + def test_leaf_functions_return_fresh_containers(self): + def empty_list(): + return [] + + def empty_dict(): + return {} + + lists = [empty_list() for _ in range(50)] + dicts = [empty_dict() for _ in range(50)] + self.assertEqual(len({id(x) for x in lists}), 50) + self.assertEqual(len({id(x) for x in dicts}), 50) + lists[0].append(1) + self.assertEqual(lists[1], []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_leaf_varkw_calls.py b/tests/regrtest/test_leaf_varkw_calls.py new file mode 100644 index 00000000..64cd9e05 --- /dev/null +++ b/tests/regrtest/test_leaf_varkw_calls.py @@ -0,0 +1,64 @@ +"""Small functions taking **kwargs see a fresh dictionary of exactly the +keywords they collect, however they are called.""" + + +def with_kwargs(a, **kw): + return a + kw.get("delta", 0) + + +def just_kw(a, **kw): + return a + + +def mk(**kw): + return kw + + +def count(**kw): + return len(kw) + + +def first(a, b=10, **kw): + return a + b + kw.get("c", 0) + + +class Box: + def get(self, **kw): + return kw.get("x", -1) + + +seen = [] +box = Box() +for i in range(3000): + assert with_kwargs(i, delta=2) == i + 2 + assert with_kwargs(i) == i + assert with_kwargs(a=i, delta=3) == i + 3 + assert just_kw(i, other=1) == i + d = mk(delta=i, other=2) + assert d == {"delta": i, "other": 2} and list(d) == ["delta", "other"] + seen.append(d) + assert mk() == {} + assert count(x=1, y=2, z=3) == 3 + assert first(i, c=5) == i + 15 + assert first(i, b=1, c=5) == i + 6 + assert box.get(x=i) == i + assert box.get() == -1 + if i % 1000 == 0: + try: + with_kwargs(i, a=1) + except TypeError as e: + assert "multiple values" in str(e), e + else: + raise AssertionError("duplicate argument accepted") + try: + with_kwargs(i, 1) + except TypeError as e: + assert "positional" in str(e), e + else: + raise AssertionError("extra positional argument accepted") + +# Every call built its own dictionary. +assert len({id(d) for d in seen}) == len(seen) +seen[0]["delta"] = "changed" +assert seen[1]["delta"] == 1 +print("ok") diff --git a/tests/regrtest/test_list_append_pop.py b/tests/regrtest/test_list_append_pop.py new file mode 100644 index 00000000..f5069d2c --- /dev/null +++ b/tests/regrtest/test_list_append_pop.py @@ -0,0 +1,66 @@ +"""`list.append` and `list.pop` on a local list, run in line. + +The core loop's fused method call appends and pops directly. These +cases check the results and every error against the full methods: +negative and boolean indexes, empty lists, out-of-range indexes, and +subclasses, which keep the full path. +""" + +import unittest + + +class ListAppendPopTests(unittest.TestCase): + def test_append_and_pop_round_trip(self): + xs = [] + for i in range(100): + xs.append(i) + xs.append((i, "x")) + popped = [xs.pop() for _ in range(4)] + self.assertEqual(popped, [(99, "x"), 99, (98, "x"), 98]) + self.assertEqual(len(xs), 196) + + def test_pop_indexes(self): + for _ in range(20): + xs = [1, 2, 3, 4] + self.assertEqual(xs.pop(0), 1) + self.assertEqual(xs.pop(-1), 4) + self.assertEqual(xs.pop(True), 3) + self.assertEqual(xs.pop(False), 2) + self.assertEqual(xs, []) + + def test_pop_errors(self): + for _ in range(20): + with self.assertRaisesRegex(IndexError, "pop from empty list"): + [].pop() + with self.assertRaisesRegex(IndexError, "pop from empty list"): + [].pop(0) + with self.assertRaisesRegex(IndexError, "pop index out of range"): + [1].pop(5) + with self.assertRaisesRegex(IndexError, "pop index out of range"): + [1].pop(-2) + with self.assertRaises(TypeError): + [1].pop("0") + + def test_subclass_overrides(self): + class Logged(list): + def append(self, item): + super().append(("logged", item)) + + def pop(self, *args): + return ("popped", super().pop(*args)) + + xs = Logged() + for i in range(3): + xs.append(i) + self.assertEqual(xs, [("logged", 0), ("logged", 1), ("logged", 2)]) + self.assertEqual(xs.pop(), ("popped", ("logged", 2))) + + def test_append_keeps_objects_alive(self): + xs = [] + for _ in range(10): + xs.append(object()) + self.assertEqual(len({id(x) for x in xs}), 10) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_pickle_guard_changes.py b/tests/regrtest/test_pickle_guard_changes.py new file mode 100644 index 00000000..698b5f82 --- /dev/null +++ b/tests/regrtest/test_pickle_guard_changes.py @@ -0,0 +1,54 @@ +"""The pickle accelerator notices in-place changes to the Python engine it +stands in for: a method's code, and a helper's defaults.""" +import pickle +import copyreg + + +class Point: + def __init__(self, x): + self.x = x + + +def check_round_trips(): + for _ in range(50): + assert pickle.loads(pickle.dumps("abc")) == "abc" + assert pickle.loads(pickle.dumps([1, 2.5, None])) == [1, 2.5, None] + assert pickle.loads(pickle.dumps(Point(3))).x == 3 + + +check_round_trips() + +# A patched method's code is what runs wherever the Python engine is the +# reference (the accelerator then stands in for it); a C pickler ignores it. +import types + +save_str = pickle._Pickler.save_str +original = save_str.__code__ +# The patched code runs with the pickle module's globals. +pickle._guard_test_save_str = types.FunctionType( + original, save_str.__globals__, "save_str_copy") + + +def patched(self, obj): + _guard_test_save_str(self, "patched" if obj == "abc" else obj) + + +expected = "patched" if issubclass(pickle.Pickler, pickle._Pickler) else "abc" +save_str.__code__ = patched.__code__ +try: + for _ in range(50): + assert pickle.loads(pickle.dumps("abc")) == expected + assert pickle.loads(pickle.dumps(["abc", "x"])) == [expected, "x"] +finally: + save_str.__code__ = original + del pickle._guard_test_save_str +check_round_trips() + +# A helper whose defaults were replaced takes the checked path; the +# results don't change. +saved = copyreg.__newobj__.__defaults__ +copyreg.__newobj__.__defaults__ = None +check_round_trips() +copyreg.__newobj__.__defaults__ = saved +check_round_trips() +print("ok") diff --git a/tests/regrtest/test_sequence_slices.py b/tests/regrtest/test_sequence_slices.py new file mode 100644 index 00000000..e573b67d --- /dev/null +++ b/tests/regrtest/test_sequence_slices.py @@ -0,0 +1,78 @@ +"""Slices of every built-in sequence, over the step and bound shapes. + +Each type slices its own storage (a tuple's items, a `bytes` buffer, a +string's code points), so these cases compare every shape against the +same selection made by explicit indexing. +""" + +import unittest + +STEPS = (None, 1, 2, 3, -1, -2, -3) +BOUNDS = (None, -100, -5, -1, 0, 1, 3, 5, 100) + + +def expected(seq, start, stop, step): + return [seq[i] for i in range(*slice(start, stop, step).indices(len(seq)))] + + +class SequenceSliceTests(unittest.TestCase): + def check(self, seq, rebuild): + for start in BOUNDS: + for stop in BOUNDS: + for step in STEPS: + with self.subTest(start=start, stop=stop, step=step): + got = seq[start:stop:step] + self.assertIs(type(got), type(seq)) + self.assertEqual(got, rebuild(expected(seq, start, stop, step))) + + def test_tuple(self): + self.check(tuple(range(9)), tuple) + self.check((), tuple) + + def test_list(self): + self.check(list(range(9)), list) + + def test_bytes(self): + self.check(bytes(range(9)), bytes) + self.check(b"", bytes) + + def test_bytearray(self): + self.check(bytearray(range(9)), bytearray) + data = bytearray(b"abc") + part = data[:] + part[0] = 0x7A + self.assertEqual(data, b"abc") + + def test_str(self): + self.check("abcdéfghi", "".join) + self.check("日本語テキストです", "".join) + self.check("", "".join) + + def test_str_with_surrogates(self): + text = "a\ud800b\udc00c\ud83d" + self.check(text, "".join) + self.assertEqual(text[1:2], "\ud800") + self.assertEqual(text[::2], "abc") + + def test_large_sequences(self): + # A small slice of a large sequence (the case `re` hits when it + # chunks a 64K-entry charset map). + big = bytes(65536) + self.assertEqual(big[256:512], bytes(256)) + chunks = {big[i:i + 256] for i in range(0, 65536, 256)} + self.assertEqual(chunks, {bytes(256)}) + items = tuple(range(100000)) + self.assertEqual(items[50000:50003], (50000, 50001, 50002)) + self.assertEqual(items[-3:], (99997, 99998, 99999)) + self.assertEqual(items[99990::4], (99990, 99994, 99998)) + + def test_bad_steps_and_indices(self): + for seq in ((1, 2), b"ab", bytearray(b"ab"), "ab", [1, 2]): + with self.assertRaises(ValueError): + seq[::0] + with self.assertRaises(TypeError): + seq["a":] + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/regrtest/test_ws4_gc_cascade.py b/tests/regrtest/test_ws4_gc_cascade.py index f57ad567..2822b5eb 100644 --- a/tests/regrtest/test_ws4_gc_cascade.py +++ b/tests/regrtest/test_ws4_gc_cascade.py @@ -91,7 +91,7 @@ def cb2(phase, info): # --------------------------------------------------------------------------- -# Generational thresholds round-trip (CPython default is (700, 10, 10)). +# Generational thresholds round-trip (CPython 3.14 defaults to (2000, 10, 10)). # --------------------------------------------------------------------------- saved = gc.get_threshold() diff --git a/tools/pgo_build.py b/tools/pgo_build.py new file mode 100644 index 00000000..ea6da597 --- /dev/null +++ b/tools/pgo_build.py @@ -0,0 +1,139 @@ +#!/usr/bin/env python3 +"""Build a profile-guided (PGO) release of the `weavepy` CLI. + +CPython's release builds are profile guided (`--enable-optimizations` +trains on the regression suite), and so is this build: it compiles an +instrumented interpreter, runs a training workload, merges the profile, +and rebuilds the release binary with it. + +The training workload is the benchmark fixtures at the benchmark harness's +work sizes (with the JIT on and off), the bundled regression suite, and a +stdlib import sweep. Work sizes matter: code the training never reaches is +laid out as cold, so a fixture run at a toy size (large-integer +multiplication in `pidigits`, for one) gets slower, not faster. + +Requirements: the `llvm-tools` rustup component (`rustup component add +llvm-tools`), which provides the `llvm-profdata` matching the compiler. + +Usage (from the repository root): + + python3 tools/pgo_build.py + python3 tools/pgo_build.py --skip-regrtest # fixtures and imports only + +The optimized binary is written to `target/pgo/optimized/release/weavepy` +(with the runtime library beside it); `--install` also copies it over +`target/release/weavepy`. Intermediate files stay under `target/pgo/`. +""" + +import argparse +import os +import re +import shutil +import subprocess +import sys +from pathlib import Path + +ROOT = Path(__file__).resolve().parent.parent +PGO = ROOT / "target" / "pgo" +FIXTURES = ROOT / "crates" / "weavepy-bench" / "fixtures" +FIXTURES_RS = ROOT / "crates" / "weavepy-bench" / "src" / "fixtures.rs" +PACKAGES = ["-p", "weavepy-cli", "-p", "weavepy-pylib"] +IMPORTS = ( + "import argparse, ast, asyncio, collections, csv, dataclasses, datetime, " + "decimal, email, enum, fractions, functools, heapq, inspect, itertools, " + "json, logging, pathlib, pickle, random, re, statistics, string, " + "textwrap, typing, unittest, urllib.parse" +) + + +def run(cmd, env=None, cwd=ROOT, check=True): + print("+", " ".join(str(c) for c in cmd), flush=True) + return subprocess.run(cmd, env=env, cwd=cwd, check=check) + + +def llvm_profdata(): + sysroot = subprocess.run( + ["rustc", "--print", "sysroot"], capture_output=True, text=True, check=True + ).stdout.strip() + host = re.search( + r"^host: (\S+)$", + subprocess.run(["rustc", "-vV"], capture_output=True, text=True, check=True).stdout, + re.M, + ).group(1) + tool = Path(sysroot) / "lib" / "rustlib" / host / "bin" / "llvm-profdata" + if os.name == "nt": + tool = tool.with_suffix(".exe") + if not tool.exists(): + sys.exit(f"{tool} not found: run `rustup component add llvm-tools` first") + return tool + + +def build(target_dir, rustflags): + env = dict(os.environ) + env["CARGO_TARGET_DIR"] = str(target_dir) + env["RUSTFLAGS"] = (env.get("RUSTFLAGS", "") + " " + rustflags).strip() + run(["cargo", "build", "--release", *PACKAGES], env=env) + exe = target_dir / "release" / ("weavepy.exe" if os.name == "nt" else "weavepy") + if not exe.exists(): + sys.exit(f"build produced no {exe}") + return exe + + +def fixture_work(): + """The benchmark harness's `(fixture, work)` pairs, read from its source.""" + text = FIXTURES_RS.read_text() + return re.findall(r'"(\w+)" => ([\d_]+),', text) + + +def train(exe, skip_regrtest): + for name, work in fixture_work(): + path = FIXTURES / f"{name}.py" + if not path.exists(): + continue + for jit in ("1", "0"): + env = dict(os.environ, WEAVEPY_BENCH_WORK=work.replace("_", ""), WEAVEPY_JIT=jit) + run([exe, path], env=env, check=False) + for jit in ("1", "0"): + run([exe, "-c", IMPORTS], env=dict(os.environ, WEAVEPY_JIT=jit), check=False) + if not skip_regrtest: + report = PGO / "regrtest" + run( + [exe, "regrtest", "--mode", "subprocess", "--workers", "0", + "--timeout", "600", "-q", "--no-check", "--report-dir", report], + check=False, + ) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("--skip-regrtest", action="store_true", + help="train on the fixtures and imports only") + parser.add_argument("--install", action="store_true", + help="also copy the optimized binary to target/release/weavepy") + args = parser.parse_args() + + profdata_tool = llvm_profdata() + raw = PGO / "raw" + shutil.rmtree(raw, ignore_errors=True) + raw.mkdir(parents=True) + + instrumented = build(PGO / "instrumented", f"-Cprofile-generate={raw}") + train(instrumented, args.skip_regrtest) + + merged = PGO / "weavepy.profdata" + profiles = sorted(raw.glob("*.profraw")) + if not profiles: + sys.exit("training produced no profiles") + run([profdata_tool, "merge", "-o", merged, *profiles]) + + optimized = build(PGO / "optimized", f"-Cprofile-use={merged}") + print(f"optimized binary: {optimized}") + if args.install: + dest = ROOT / "target" / "release" / optimized.name + dest.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(optimized, dest) + print(f"installed: {dest}") + + +if __name__ == "__main__": + main()