diff --git a/changelog.d/11589-byte-view-element-access.md b/changelog.d/11589-byte-view-element-access.md new file mode 100644 index 0000000000..f39db11c0e --- /dev/null +++ b/changelog.d/11589-byte-view-element-access.md @@ -0,0 +1,3 @@ +perf(buffer): `Uint8Array` and `Buffer` element reads and writes are answered from the byte-view admission cache (now two-way, and admitting Node Buffers) inline at typed sites and at the top of the runtime byte accessors, instead of the buffer-registry probes (#10515, #10694). On qb2: the #10515 `Uint8Array`-parameter loop goes from 876 to 76 instructions per element; nanoid/generate −34.4%, uuid/v7 −12.6%. + +Key material (CryptoKey, secret and asymmetric keys), `ArrayBuffer`, `SharedArrayBuffer` and `DataView` buffers are never admitted to that cache. The perry-stdlib test `thread_exit_releases_the_threads_external_buffer_registrations` (#11555) primed the cache through a CryptoKey buffer and probed it with the old one-way slot formula, so it failed on this change. It now primes the cache with an ordinary `Buffer` on the same thread, asserts the key is refused, and reads the cache through the new `perry_runtime::buffer::u8_inline_cache_holds_for_test` probe. The cache's thread-exit release was already correct: it clears every slot, both ways. With that release removed, the test fails with "PERRY_U8_INLINE_CACHE outlived the thread". diff --git a/crates/perry-codegen/src/expr/arrays_finds.rs b/crates/perry-codegen/src/expr/arrays_finds.rs index 999bdec6a9..6993c12ed7 100644 --- a/crates/perry-codegen/src/expr/arrays_finds.rs +++ b/crates/perry-codegen/src/expr/arrays_finds.rs @@ -270,9 +270,10 @@ pub(crate) fn lower_uint8array_get_i32( let idx_i32 = lower_index_i32(ctx, index)?; let a = lower_expr(ctx, array)?; - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &a); - let byte_i32 = blk.call(I32, "js_uint8array_get", &[(I64, &handle), (I32, &idx_i32)]); + // #10515: admitted owning byte views load inline; misses (and priming) + // stay on the runtime accessor. + let byte_i32 = + super::u8_buffer_read::emit_u8_cached_get_i32(ctx, &a, &idx_i32, "js_uint8array_get"); let slow = LoweredValue { semantic: SemanticKind::JsNumber, rep: NativeRep::I32, @@ -381,9 +382,8 @@ pub(crate) fn lower_buffer_index_get_i32( let idx_i32 = lower_index_i32(ctx, index)?; let a = lower_expr(ctx, buffer)?; - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &a); - let byte_i32 = blk.call(I32, "js_buffer_get", &[(I64, &handle), (I32, &idx_i32)]); + let byte_i32 = + super::u8_buffer_read::emit_u8_cached_get_i32(ctx, &a, &idx_i32, "js_buffer_get"); let slow = LoweredValue { semantic: SemanticKind::JsNumber, rep: NativeRep::I32, @@ -997,12 +997,11 @@ pub(crate) fn lower( &[index], |ctx| lower_index_i32(ctx, index), |ctx, vals, idx_i32| { - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &vals[0]); - Ok(blk.call( - DOUBLE, + Ok(super::u8_buffer_read::emit_u8_cached_get_value( + ctx, + &vals[0], + &idx_i32, "js_uint8array_index_get_value", - &[(I64, &handle), (I32, &idx_i32)], )) }, ) @@ -1026,12 +1025,11 @@ pub(crate) fn lower( &[index], |ctx| lower_index_i32(ctx, index), |ctx, vals, idx_i32| { - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &vals[0]); - Ok(blk.call( - DOUBLE, + Ok(super::u8_buffer_read::emit_u8_cached_get_value( + ctx, + &vals[0], + &idx_i32, "js_buffer_index_get_value", - &[(I64, &handle), (I32, &idx_i32)], )) }, ) @@ -1134,11 +1132,13 @@ pub(crate) fn lower( // Slow path accepts either BufferHeader-backed Uint8Arrays or // NativeArena typed views. let a = lower_expr(ctx, array)?; - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &a); - blk.call_void( + // #10515: admitted owning byte views store inline. + super::u8_buffer_read::emit_u8_cached_set_i32( + ctx, + &a, + &idx_i32, + &val_i32, "js_uint8array_set", - &[(I64, &handle), (I32, &idx_i32), (I32, &val_i32)], ); let reason = buffer_access_materialization_reason(ctx, array); let slow = LoweredValue { @@ -1227,11 +1227,12 @@ pub(crate) fn lower( ctx.toint32_wrap(&v) }; let a = lower_expr(ctx, buffer)?; - let blk = ctx.block(); - let handle = unbox_to_i64(blk, &a); - blk.call_void( + super::u8_buffer_read::emit_u8_cached_set_i32( + ctx, + &a, + &idx_i32, + &val_i32, "js_buffer_set", - &[(I64, &handle), (I32, &idx_i32), (I32, &val_i32)], ); let reason = buffer_access_materialization_reason(ctx, buffer); let slow = LoweredValue { diff --git a/crates/perry-codegen/src/expr/index_get/inline_dyn_typed_array.rs b/crates/perry-codegen/src/expr/index_get/inline_dyn_typed_array.rs index 00d9cedb4c..a63ee9d5be 100644 --- a/crates/perry-codegen/src/expr/index_get/inline_dyn_typed_array.rs +++ b/crates/perry-codegen/src/expr/index_get/inline_dyn_typed_array.rs @@ -100,6 +100,12 @@ pub(super) fn lower_inline_dyn_typed_array_get( ctx.ic_globals.push(cache_name.clone()); let slot_ref = format!("@{cache_name}"); + let u8_brand_idx = ctx.new_block("arrlike.u8.brand"); + let u8_bounds_idx = ctx.new_block("arrlike.u8.bounds"); + let u8_load_idx = ctx.new_block("arrlike.u8.load"); + let u8_brand_label = ctx.block_label(u8_brand_idx); + let u8_bounds_label = ctx.block_label(u8_bounds_idx); + let u8_load_label = ctx.block_label(u8_load_idx); let object_header_idx = ctx.new_block("arrlike.ic.header"); let object_brand_idx = ctx.new_block("arrlike.ic.brand"); let object_array_guard_idx = ctx.new_block("arrlike.ic.array_guard"); @@ -476,7 +482,47 @@ pub(super) fn lower_inline_dyn_typed_array_get( ctx.current_block = elem_kind_idx; let elem_is_object = ctx.block().icmp_eq(I8, &gc_type, "2"); ctx.block() - .cond_br(&elem_is_object, &elem_meta_label, &object_miss_label); + .cond_br(&elem_is_object, &elem_meta_label, &u8_brand_label); + + // ---- #10515: an admitted owning byte view (`Uint8Array` / `Buffer`) ---- + // + // A `GC_TYPE_BUFFER` receiver whose full address is in + // `PERRY_U8_INLINE_CACHE` is, by that cache's contract, a live registered + // byte view with `length` at offset 0 and its bytes inline at `+8` — the + // same proof `u8_buffer_read.rs` loads on for a `Uint8Array`-typed + // receiver. Untyped `b[i]` over a Buffer used to leave through the exit + // and the typed-array + buffer registry probes on every element. The + // cache is primed by the runtime byte accessors on a miss, and anything + // it does not hold (views, ArrayBuffers, DataViews, foreign spans, an + // out-of-range index) still leaves through the exit. + ctx.current_block = u8_brand_idx; + { + let blk = ctx.block(); + let is_buffer = blk.icmp_eq(I8, &gc_type, "10"); // GC_TYPE_BUFFER + let admitted = crate::expr::u8_buffer_read::emit_u8_cache_holds(blk, &object_raw); + let hit = blk.and(I1, &is_buffer, &admitted); + blk.cond_br(&hit, &u8_bounds_label, &object_miss_label); + } + ctx.current_block = u8_bounds_idx; + { + let blk = ctx.block(); + let len_ptr = blk.inttoptr(I64, &object_raw); + let len = blk.load(I32, &len_ptr); + let len_i64 = blk.zext(I32, &len, I64); + let in_bounds = blk.icmp_ult(I64, &object_idx_i64, &len_i64); + blk.cond_br(&in_bounds, &u8_load_label, &object_miss_label); + } + ctx.current_block = u8_load_idx; + let u8_value = { + let blk = ctx.block(); + let data = blk.add(I64, &object_raw, "8"); + let addr = blk.add(I64, &data, &object_idx_i64); + let ptr = blk.inttoptr(I64, &addr); + let byte = blk.load(I8, &ptr); + blk.uitofp(I8, &byte, DOUBLE) + }; + let u8_end_label = ctx.block().label.clone(); + ctx.block().br(&merge_label); ctx.current_block = elem_meta_idx; let meta_ptr_size: u64 = if crate::target_layout::target_is_ilp32(ctx.target_triple) { @@ -596,6 +642,7 @@ pub(super) fn lower_inline_dyn_typed_array_get( (ta_w1_value.as_str(), ta_w1_end.as_str()), (array_value.as_str(), array_end_label.as_str()), (elem_value.as_str(), elem_end_label.as_str()), + (u8_value.as_str(), u8_end_label.as_str()), (slow_val.as_str(), slow_end_label.as_str()), ], ) diff --git a/crates/perry-codegen/src/expr/index_get_claim_tests.rs b/crates/perry-codegen/src/expr/index_get_claim_tests.rs index f98f78ac6b..4743ef7380 100644 --- a/crates/perry-codegen/src/expr/index_get_claim_tests.rs +++ b/crates/perry-codegen/src/expr/index_get_claim_tests.rs @@ -196,6 +196,10 @@ fn unknown_numeric_read_is_one_inline_hit_and_one_out_of_line_exit() { assert_eq!( dynamic_index_site_blocks(&ir), vec![ + // #10515: the admitted byte-view (`Uint8Array` / `Buffer`) arm. + "arrlike.u8.brand", + "arrlike.u8.bounds", + "arrlike.u8.load", "arrlike.ic.header", "arrlike.ic.brand", "arrlike.ic.array_guard", @@ -370,7 +374,7 @@ fn the_number_context_coercion_is_coupled_across_every_arm() { ); assert_eq!( dynamic_index_site_blocks(&ir).len(), - 21, + 24, "{name}: a number context must not change the emitted block shape:\n{ir}" ); let miss = super::class_field_barrier_tests::block_body(&ir, "arrlike.ic.miss.") @@ -715,9 +719,20 @@ fn any_typed_dynamic_key_takes_the_numeric_tiers_when_it_is_an_array_index() { let kind = super::class_field_barrier_tests::block_body(&ir, "arrlike.elem.kind.") .expect("the object-kind guard exists"); assert!( - kind.contains("icmp eq i8") && kind.contains(", 2") && kind.contains("arrlike.ic.miss"), + kind.contains("icmp eq i8") && kind.contains(", 2") && kind.contains("arrlike.u8.brand"), "only GC_TYPE_OBJECT may reach the ObjectMeta.elements load; everything \ - else must leave through the single exit:\n{kind}" + else goes to the byte-view arm:\n{kind}" + ); + // #10515: the byte-view arm admits only a `GC_TYPE_BUFFER` whose address + // the admission cache holds; everything else leaves through the exit. + let u8_brand = super::class_field_barrier_tests::block_body(&ir, "arrlike.u8.brand.") + .expect("the byte-view brand guard exists"); + assert!( + u8_brand.contains(", 10") + && u8_brand.contains("@PERRY_U8_INLINE_CACHE") + && u8_brand.contains("arrlike.ic.miss"), + "the byte-view arm must test GC_TYPE_BUFFER and the admission cache, \ + and exit on a miss:\n{u8_brand}" ); // The elements-backed subclass probe, the lazy-JSON-array probe and the // dense-tail family token now live behind that exit rather than at every diff --git a/crates/perry-codegen/src/expr/u8_buffer_read.rs b/crates/perry-codegen/src/expr/u8_buffer_read.rs index 8bf941158a..989588efed 100644 --- a/crates/perry-codegen/src/expr/u8_buffer_read.rs +++ b/crates/perry-codegen/src/expr/u8_buffer_read.rs @@ -158,18 +158,7 @@ fn lower_u8_buffer_checked_load( let raw = blk.and(I64, &obj_bits, crate::nanbox::POINTER_MASK_I64); let tagged = blk.and(I64, &obj_bits, &tag_mask); let is_ptr = blk.icmp_eq(I64, &tagged, crate::nanbox::POINTER_TAG_I64); - // Slot formula duplicates `buffer/header.rs::u8_inline_cache_slot`. - let slot = blk.lshr(I64, &raw, "3"); - let slot = blk.and(I64, &slot, "63"); - let entry_ptr = blk.gep( - "[64 x i64]", - "@PERRY_U8_INLINE_CACHE", - &[(I64, "0"), (I64, &slot)], - ); - let entry_val = blk.load(I64, &entry_ptr); - // Full-address compare — an empty slot (0) can never match a real - // pointer, so no separate emptiness test. - let hit = blk.icmp_eq(I64, &entry_val, &raw); + let hit = emit_u8_cache_holds(blk, &raw); let g = blk.and(I1, &is_ptr, &hit); blk.cond_br(&g, &chk_label, &slow_label); raw @@ -236,3 +225,225 @@ fn lower_u8_buffer_checked_load( ], )) } + +// --------------------------------------------------------------------------- +// #10515: the same admission cache, for the i32-ABI reads and for WRITES. +// +// The cache contract (`perry-runtime/src/buffer/header.rs`) is that an entry +// names a live registered byte view — `Uint8Array` or `Buffer` — that OWNS its +// bytes inline at `+8` (no foreign span, not a registered view). A write to +// such a buffer is exactly `js_buffer_set`'s store: views over it resolve +// their bytes through this backing rather than holding a copy (see +// `buffer/view.rs`), so there is nothing to propagate. Every guard miss — +// a view, a foreign span, an out-of-range index, a non-pointer, an +// unadmitted buffer — takes the unchanged runtime accessor, which also primes +// the cache for the next access. +// --------------------------------------------------------------------------- + +/// Pointer tag + full-address admission hit for `obj_box`. Returns +/// `(hit, raw_address)`, both in the current block. +fn emit_u8_cache_admission(ctx: &mut FnCtx<'_>, obj_box: &str) -> (String, String) { + let tag_mask = i64_literal(crate::nanbox::TAG_MASK); + let blk = ctx.block(); + let obj_bits = blk.bitcast_double_to_i64(obj_box); + let raw = blk.and(I64, &obj_bits, crate::nanbox::POINTER_MASK_I64); + let tagged = blk.and(I64, &obj_bits, &tag_mask); + let is_ptr = blk.icmp_eq(I64, &tagged, crate::nanbox::POINTER_TAG_I64); + let admitted = emit_u8_cache_holds(blk, &raw); + (blk.and(I1, &is_ptr, &admitted), raw) +} + +/// `i1`: `PERRY_U8_INLINE_CACHE` holds exactly `raw`. The cache is two-way +/// set-associative (#10515): `raw` may sit in either slot of the pair +/// `(raw >> 3) & 62`, which duplicates +/// `perry-runtime/src/buffer/header.rs::u8_inline_cache_pair` — keep in sync. +/// Full-address compares, so an empty slot (0) never matches a real pointer. +pub(crate) fn emit_u8_cache_holds(blk: &mut crate::block::LlBlock, raw: &str) -> String { + let shifted = blk.lshr(I64, raw, "3"); + let pair = blk.and(I64, &shifted, "62"); + let second = blk.or(I64, &pair, "1"); + let first_ptr = blk.gep( + "[64 x i64]", + "@PERRY_U8_INLINE_CACHE", + &[(I64, "0"), (I64, &pair)], + ); + let second_ptr = blk.gep( + "[64 x i64]", + "@PERRY_U8_INLINE_CACHE", + &[(I64, "0"), (I64, &second)], + ); + let first = blk.load(I64, &first_ptr); + let second = blk.load(I64, &second_ptr); + let in_first = blk.icmp_eq(I64, &first, raw); + let in_second = blk.icmp_eq(I64, &second, raw); + blk.or(I1, &in_first, &in_second) +} + +/// `idx ult length` against an admitted buffer's `u32` length at offset 0. +/// `ult` also rejects a negative (or `-1`-sentinel) index. +fn emit_u8_in_bounds(ctx: &mut FnCtx<'_>, raw: &str, idx_i32: &str) -> String { + let blk = ctx.block(); + let hdr_ptr = blk.inttoptr(I64, raw); + let len = blk.load(I32, &hdr_ptr); + blk.icmp_ult(I32, idx_i32, &len) +} + +fn emit_u8_byte_ptr(ctx: &mut FnCtx<'_>, raw: &str, idx_i32: &str) -> String { + let blk = ctx.block(); + let data_base = blk.add(I64, raw, "8"); + let idx_i64 = blk.zext(I32, idx_i32, I64); + let addr = blk.add(I64, &data_base, &idx_i64); + blk.inttoptr(I64, &addr) +} + +/// Guarded inline byte READ in the runtime helper's native i32 ABI: +/// `slow_fn(handle: i64, idx: i32) -> i32` (`js_uint8array_get` / +/// `js_buffer_get`, which answer the `0` byte sentinel out of range). +pub(crate) fn emit_u8_cached_get_i32( + ctx: &mut FnCtx<'_>, + obj_box: &str, + idx_i32: &str, + slow_fn: &str, +) -> String { + let chk_idx = ctx.new_block("u8c.get.chk"); + let load_idx = ctx.new_block("u8c.get.load"); + let slow_idx = ctx.new_block("u8c.get.slow"); + let merge_idx = ctx.new_block("u8c.get.merge"); + let chk_label = ctx.block_label(chk_idx); + let load_label = ctx.block_label(load_idx); + let slow_label = ctx.block_label(slow_idx); + let merge_label = ctx.block_label(merge_idx); + let (hit, raw) = emit_u8_cache_admission(ctx, obj_box); + ctx.block().cond_br(&hit, &chk_label, &slow_label); + + ctx.current_block = chk_idx; + let in_bounds = emit_u8_in_bounds(ctx, &raw, idx_i32); + ctx.block().cond_br(&in_bounds, &load_label, &slow_label); + + ctx.current_block = load_idx; + let ptr = emit_u8_byte_ptr(ctx, &raw, idx_i32); + let (fast_val, fast_end) = { + let blk = ctx.block(); + let byte = blk.load(I8, &ptr); + let val = blk.zext(I8, &byte, I32); + let end = blk.label.clone(); + blk.br(&merge_label); + (val, end) + }; + + ctx.current_block = slow_idx; + let (slow_val, slow_end) = { + let blk = ctx.block(); + let val = blk.call(I32, slow_fn, &[(I64, &raw), (I32, idx_i32)]); + let end = blk.label.clone(); + blk.br(&merge_label); + (val, end) + }; + + ctx.current_block = merge_idx; + ctx.block().phi( + I32, + &[ + (fast_val.as_str(), fast_end.as_str()), + (slow_val.as_str(), slow_end.as_str()), + ], + ) +} + +/// Guarded inline byte read yielding a JS value: the byte as a Number in +/// bounds, else `slow_fn(handle, idx) -> double` (`js_uint8array_index_get_value` +/// / `js_buffer_index_get_value`, which answer `undefined` out of range). +pub(crate) fn emit_u8_cached_get_value( + ctx: &mut FnCtx<'_>, + obj_box: &str, + idx_i32: &str, + slow_fn: &str, +) -> String { + let chk_idx = ctx.new_block("u8c.getv.chk"); + let load_idx = ctx.new_block("u8c.getv.load"); + let slow_idx = ctx.new_block("u8c.getv.slow"); + let merge_idx = ctx.new_block("u8c.getv.merge"); + let chk_label = ctx.block_label(chk_idx); + let load_label = ctx.block_label(load_idx); + let slow_label = ctx.block_label(slow_idx); + let merge_label = ctx.block_label(merge_idx); + let (hit, raw) = emit_u8_cache_admission(ctx, obj_box); + ctx.block().cond_br(&hit, &chk_label, &slow_label); + + ctx.current_block = chk_idx; + let in_bounds = emit_u8_in_bounds(ctx, &raw, idx_i32); + ctx.block().cond_br(&in_bounds, &load_label, &slow_label); + + ctx.current_block = load_idx; + let ptr = emit_u8_byte_ptr(ctx, &raw, idx_i32); + let (fast_val, fast_end) = { + let blk = ctx.block(); + let byte = blk.load(I8, &ptr); + let val = blk.uitofp(I8, &byte, DOUBLE); + let end = blk.label.clone(); + blk.br(&merge_label); + (val, end) + }; + + ctx.current_block = slow_idx; + let (slow_val, slow_end) = { + let blk = ctx.block(); + let val = blk.call(DOUBLE, slow_fn, &[(I64, &raw), (I32, idx_i32)]); + let end = blk.label.clone(); + blk.br(&merge_label); + (val, end) + }; + + ctx.current_block = merge_idx; + ctx.block().phi( + DOUBLE, + &[ + (fast_val.as_str(), fast_end.as_str()), + (slow_val.as_str(), slow_end.as_str()), + ], + ) +} + +/// Guarded inline byte WRITE in the runtime helper's i32 ABI: `val_i32` is +/// already ToInt32'd by the caller, and the store keeps its low byte exactly +/// as `js_buffer_set` does (`value & 0xFF`). Misses call +/// `slow_fn(handle, idx, value)` (`js_uint8array_set` / `js_buffer_set`). +pub(crate) fn emit_u8_cached_set_i32( + ctx: &mut FnCtx<'_>, + obj_box: &str, + idx_i32: &str, + val_i32: &str, + slow_fn: &str, +) { + let chk_idx = ctx.new_block("u8c.set.chk"); + let store_idx = ctx.new_block("u8c.set.store"); + let slow_idx = ctx.new_block("u8c.set.slow"); + let merge_idx = ctx.new_block("u8c.set.merge"); + let chk_label = ctx.block_label(chk_idx); + let store_label = ctx.block_label(store_idx); + let slow_label = ctx.block_label(slow_idx); + let merge_label = ctx.block_label(merge_idx); + let (hit, raw) = emit_u8_cache_admission(ctx, obj_box); + ctx.block().cond_br(&hit, &chk_label, &slow_label); + + ctx.current_block = chk_idx; + let in_bounds = emit_u8_in_bounds(ctx, &raw, idx_i32); + ctx.block().cond_br(&in_bounds, &store_label, &slow_label); + + ctx.current_block = store_idx; + let ptr = emit_u8_byte_ptr(ctx, &raw, idx_i32); + { + let blk = ctx.block(); + let byte = blk.trunc(I32, val_i32, I8); + blk.store(I8, &byte, &ptr); + blk.br(&merge_label); + } + + ctx.current_block = slow_idx; + { + let blk = ctx.block(); + blk.call_void(slow_fn, &[(I64, &raw), (I32, idx_i32), (I32, val_i32)]); + blk.br(&merge_label); + } + ctx.current_block = merge_idx; +} diff --git a/crates/perry-runtime/src/array/subclass_packed_index.rs b/crates/perry-runtime/src/array/subclass_packed_index.rs index 00fe3be52d..1f6f8cae80 100644 --- a/crates/perry-runtime/src/array/subclass_packed_index.rs +++ b/crates/perry-runtime/src/array/subclass_packed_index.rs @@ -83,6 +83,13 @@ pub extern "C" fn js_packed_arraylike_index_get( return probed; } } + // #10515: an admitted owning byte view answers from the + // inline-access cache before the dispatcher's registry probes. + if header.obj_type == crate::gc::GC_TYPE_BUFFER { + if let Some(byte) = cached_u8_packed_get(raw as usize, index_u32) { + return byte; + } + } if matches!( header.obj_type, crate::gc::GC_TYPE_ARRAY | crate::gc::GC_TYPE_LAZY_ARRAY @@ -151,6 +158,14 @@ pub extern "C" fn js_packed_arraylike_index_get( } crate::value::js_dyn_index_get(receiver, index) } +/// #10515: an admitted owning byte view's element, out of line so the Array +/// receivers this helper mostly serves pay only the brand compare. +#[inline(never)] +fn cached_u8_packed_get(addr: usize, index: u32) -> Option { + let idx = i32::try_from(index).ok()?; + crate::buffer::cached_u8_read(addr, idx).map(f64::from) +} + #[cfg(feature = "keepalive-anchors")] #[used(compiler)] static KEEP_JS_PACKED_ARRAYLIKE_INDEX_GET: extern "C" fn( diff --git a/crates/perry-runtime/src/buffer/access.rs b/crates/perry-runtime/src/buffer/access.rs index 03a2c456ab..87f09ecd5c 100644 --- a/crates/perry-runtime/src/buffer/access.rs +++ b/crates/perry-runtime/src/buffer/access.rs @@ -219,10 +219,65 @@ unsafe fn read_buffer_byte(buf_ptr: *const BufferHeader, index: i32) -> Option= (*buf_ptr).length { return None; } - let data = buffer_data(buf_ptr); + let data = byte_access_data(buf_ptr); Some(*data.add(index as usize)) } +/// The byte data of a buffer an element access is about to touch, resolving a +/// registered view to its backing exactly as [`buffer_data`] does — and, for an +/// owning buffer, admitting it to the inline-access cache (#10515) so the next +/// access through ANY site (the emitted guards, and the cache test at the top +/// of the runtime accessors) skips the registry probes entirely. A view is +/// answered from its one registry lookup and never pays for the admission +/// attempt; only the (rare) non-admissible owning buffers — foreign-backed +/// spans, a stale-hint ArrayBuffer — retry it on each access. +#[inline] +pub(crate) unsafe fn byte_access_data(buf_ptr: *const BufferHeader) -> *mut u8 { + let addr = buf_ptr as usize; + if let Some(info) = super::view::lookup(addr) { + return (buffer_data(info.backing as *const BufferHeader) as *mut u8) + .add(info.offset as usize); + } + super::header::u8_inline_cache_try_prime(addr); + buffer_data(buf_ptr) as *mut u8 +} + +/// #10515: the inline-access cache hit shared by every runtime byte accessor. +/// `Some(byte)` when `addr` is an admitted owning byte view and `index` is in +/// bounds; `None` sends the caller down its unchanged dispatch (which answers +/// out-of-range reads itself). The cache contract makes this read exactly what +/// `read_buffer_byte` would return, without the typed-array and buffer +/// registry probes that precede it. +#[inline(always)] +pub(crate) fn cached_u8_read(addr: usize, index: i32) -> Option { + if !super::header::u8_inline_cache_hit(addr) { + return None; + } + unsafe { + let len = *(addr as *const u32); + if index < 0 || index as u32 >= len { + return None; + } + Some(*((addr + std::mem::size_of::()) as *const u8).add(index as usize)) + } +} + +/// Store twin of [`cached_u8_read`]: `true` when the byte was written. +#[inline(always)] +pub(crate) fn cached_u8_write(addr: usize, index: i32, byte: u8) -> bool { + if !super::header::u8_inline_cache_hit(addr) { + return false; + } + unsafe { + let len = *(addr as *const u32); + if index < 0 || index as u32 >= len { + return false; + } + *((addr + std::mem::size_of::()) as *mut u8).add(index as usize) = byte; + } + true +} + /// Get a byte at the specified index. Native i32 accessor: an out-of-range /// index yields the `0` sentinel because every caller here has proven the /// index in bounds or consumes the byte in a native integer context. A @@ -269,7 +324,7 @@ pub extern "C" fn js_buffer_set(buf_ptr: *mut BufferHeader, index: i32, value: i return; } let byte = (value & 0xFF) as u8; - let data = buffer_data_mut(buf_ptr); + let data = byte_access_data(buf_ptr); *data.add(index as usize) = byte; } } diff --git a/crates/perry-runtime/src/buffer/header.rs b/crates/perry-runtime/src/buffer/header.rs index a07dc8a551..5d72f0f5fb 100644 --- a/crates/perry-runtime/src/buffer/header.rs +++ b/crates/perry-runtime/src/buffer/header.rs @@ -411,6 +411,8 @@ static ASYMMETRIC_KEY_EVER_MARKED: RegistryLatch = RegistryLatch::new(); static BUFFER_AB_ALIAS_EVER_SET: RegistryLatch = RegistryLatch::new(); pub fn mark_as_array_buffer(addr: usize) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); ARRAY_BUFFER_EVER_MARKED.arm(); ARRAY_BUFFER_REGISTRY.with(|r| { r.borrow_mut().insert(addr); @@ -481,6 +483,8 @@ pub(crate) fn test_resizable_registry_len() -> usize { } pub fn mark_as_shared_array_buffer(addr: usize) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); SHARED_ARRAY_BUFFER_EVER_MARKED.arm(); SHARED_ARRAY_BUFFER_REGISTRY.with(|r| { r.borrow_mut().insert(addr); @@ -509,6 +513,8 @@ pub fn is_any_array_buffer(addr: usize) -> bool { } pub fn mark_as_data_view(addr: usize) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); DATA_VIEW_EVER_MARKED.arm(); DATA_VIEW_REGISTRY.with(|r| { r.borrow_mut().insert(addr); @@ -795,6 +801,8 @@ fn register_external_uint8array(addr: usize) { } pub fn mark_as_secret_key(addr: usize) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); SECRET_KEY_EVER_MARKED.arm(); SECRET_KEY_REGISTRY.with(|r| { r.borrow_mut().insert(addr); @@ -830,6 +838,8 @@ pub fn mark_as_crypto_key_with_flags( usages: u32, bit_length: u32, ) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); CRYPTO_KEY_EVER_MARKED.arm(); CRYPTO_KEY_META_REGISTRY.with(|r| { r.borrow_mut() @@ -915,6 +925,8 @@ fn default_crypto_key_usages(algo: u8, kind: u8) -> u32 { /// `kind`: 1 public, 2 private. `asym_type`: 1 rsa, 2 ec (P-256), 3 ed25519, /// 4 x25519, 5 ec (P-384), 6 ec (P-521). pub fn mark_as_asymmetric_key(addr: usize, kind: u8, asym_type: u8) { + // A non-byte-view brand revokes inline element admission (#10515). + u8_inline_cache_invalidate(addr); ASYMMETRIC_KEY_EVER_MARKED.arm(); ASYMMETRIC_KEY_REGISTRY.with(|r| { r.borrow_mut().insert(addr, (kind, asym_type)); @@ -929,15 +941,28 @@ pub fn asymmetric_key_meta(addr: usize) -> Option<(u8, u8)> { ASYMMETRIC_KEY_REGISTRY.with(|r| r.borrow().get(&addr).copied()) } -/// #9342: direct-mapped inline-read admission cache for `Uint8Array`-backing +/// #9342: direct-mapped inline element-access admission cache for byte-view /// `BufferHeader`s, exported under a stable link name for the codegen's -/// guarded inline byte load (`perry-codegen/src/expr/u8_buffer_read.rs`). +/// guarded inline byte loads and stores (`perry-codegen/src/expr/ +/// u8_buffer_read.rs`) and consulted first by the runtime byte accessors. /// -/// An entry holds the full address of a **live, `mark_as_uint8array`-marked -/// owning `BufferHeader` whose authoritative bytes are inline at -/// `header + 8`** (no foreign backing and no registered view). Under that -/// contract the emitted reader may do -/// `len = *(u32*)addr; addr + 8 + idx` directly: +/// An entry holds the full address of a **live, registered byte view — a +/// `Uint8Array` or a Node `Buffer` (`buffer_brand` says so) — whose +/// authoritative bytes are inline at `header + 8`** (no foreign backing and no +/// registered view). Under that contract the emitted code may do +/// `len = *(u32*)addr; addr + 8 + idx` directly, for a read AND for a write: +/// a write to an owning buffer is exactly `js_buffer_set`'s store, because +/// every view over it resolves its bytes through the backing (`buffer/view.rs`) +/// rather than holding a copy. +/// +/// * `Buffer` was admitted too in #10515: its element semantics are the +/// `Uint8Array`'s, and requiring the `mark_as_uint8array` marker sent every +/// `Buffer.alloc` byte through the registry probes on every access. An +/// `ArrayBuffer`, `SharedArrayBuffer`, `DataView` or key object shares the +/// `BufferHeader` storage but is NOT integer-indexed (a DataView even keeps +/// its data pointer in that payload), so it is never admitted, and every +/// `mark_as_*` for those brands invalidates the address in case a mark ever +/// follows a prime; /// /// * shared views (`js_buffer_slice` / `new Uint8Array(arrayBuffer)`) are /// excluded — their allocation is only a header. Runtime reads resolve @@ -950,44 +975,88 @@ pub fn asymmetric_key_meta(addr: usize) -> Option<(u8, u8)> { /// and `register_buffer` clears it again when the address is re-issued /// (belt and suspenders, mirroring its own-props clear). /// -/// Slot formula `(addr >> 3) & 63` is duplicated by codegen — keep in sync. +/// #10515: TWO-WAY set-associative. An address maps to the slot PAIR +/// `(addr >> 3) & 62` and may live in either of its two slots, +/// so two hot buffers that hash together (nanoid's pool + its alphabet table) +/// no longer evict each other on every alternate access — each miss re-ran the +/// admission probes, which cost more than the access itself. The pair formula +/// is duplicated by codegen (`u8_buffer_read.rs::emit_u8_cache_admission`) — +/// keep in sync. pub const U8_INLINE_CACHE_SLOTS: usize = 64; #[no_mangle] pub static PERRY_U8_INLINE_CACHE: [std::sync::atomic::AtomicU64; U8_INLINE_CACHE_SLOTS] = [const { std::sync::atomic::AtomicU64::new(0) }; U8_INLINE_CACHE_SLOTS]; -#[inline] -fn u8_inline_cache_slot(addr: usize) -> usize { - (addr >> 3) & (U8_INLINE_CACHE_SLOTS - 1) +/// The first slot of `addr`'s pair; the pair is `[p, p + 1]`. +#[inline(always)] +fn u8_inline_cache_pair(addr: usize) -> usize { + (addr >> 3) & (U8_INLINE_CACHE_SLOTS - 2) } /// Test-only: does the admission cache currently hold exactly `addr`? -/// Reads the slot the way the emitted guard does — full-address compare. +/// Reads the pair the way the emitted guard does — full-address compares. #[cfg(test)] pub(crate) fn test_u8_inline_cache_holds(addr: usize) -> bool { - PERRY_U8_INLINE_CACHE[u8_inline_cache_slot(addr)].load(std::sync::atomic::Ordering::Relaxed) - == addr as u64 + u8_inline_cache_hit(addr) +} + +/// Test probe (#11589): does the admission cache hold exactly `addr`, in +/// either way of its pair? The public twin of `test_u8_inline_cache_holds` +/// for out-of-crate tests, so they never re-derive the slot formula. Reads no +/// thread-local, so a thread-exit range hook may call it. +#[doc(hidden)] +pub fn u8_inline_cache_holds_for_test(addr: usize) -> bool { + u8_inline_cache_hit(addr) } #[inline] pub(crate) fn u8_inline_cache_invalidate(addr: usize) { - let slot = u8_inline_cache_slot(addr); - if PERRY_U8_INLINE_CACHE[slot].load(std::sync::atomic::Ordering::Relaxed) == addr as u64 { - PERRY_U8_INLINE_CACHE[slot].store(0, std::sync::atomic::Ordering::Relaxed); + let pair = u8_inline_cache_pair(addr); + for slot in [pair, pair + 1] { + if PERRY_U8_INLINE_CACHE[slot].load(std::sync::atomic::Ordering::Relaxed) == addr as u64 { + PERRY_U8_INLINE_CACHE[slot].store(0, std::sync::atomic::Ordering::Relaxed); + } } } -/// Admit `addr` to the inline-read cache iff it satisfies the cache contract -/// above. Called from the codegen slow arm (`js_u8_buffer_read_f64`) so a -/// guard miss primes the next access; never called on a hot path. +/// `addr` holds an admission in [`PERRY_U8_INLINE_CACHE`]: it is a live +/// owning byte view whose `length` is the `u32` at offset 0 and whose bytes +/// are inline at `addr + 8`. Two loads and two compares. +#[inline(always)] +pub(crate) fn u8_inline_cache_hit(addr: usize) -> bool { + use std::sync::atomic::Ordering::Relaxed; + let pair = u8_inline_cache_pair(addr); + addr != 0 + && (PERRY_U8_INLINE_CACHE[pair].load(Relaxed) == addr as u64 + || PERRY_U8_INLINE_CACHE[pair + 1].load(Relaxed) == addr as u64) +} + +/// Admit `addr` to the inline-access cache iff it satisfies the cache +/// contract above. Called from the codegen slow arms (`js_u8_buffer_read_f64` +/// and the #10515 i32 get/set twins) and from the runtime byte accessors' +/// registry arm, so a miss primes the next access. A new admission takes an +/// empty slot of its pair, else the first slot, demoting that slot's entry to +/// the second (which drops the older of the two). pub(crate) fn u8_inline_cache_try_prime(addr: usize) { - if is_uint8array_buffer(addr) + use std::sync::atomic::Ordering::Relaxed; + if u8_inline_cache_hit(addr) { + return; + } + if super::exotic_view::is_uint8_view_buffer(addr) && foreign_backing(addr).is_none() && super::view::lookup(addr).is_none() { register_thread_exit_hook(); - PERRY_U8_INLINE_CACHE[u8_inline_cache_slot(addr)] - .store(addr as u64, std::sync::atomic::Ordering::Relaxed); + let pair = u8_inline_cache_pair(addr); + let first = PERRY_U8_INLINE_CACHE[pair].load(Relaxed); + if first == 0 { + PERRY_U8_INLINE_CACHE[pair].store(addr as u64, Relaxed); + } else { + // The first way's entry moves to the second (dropping whatever was + // older there); the new admission takes the first. + PERRY_U8_INLINE_CACHE[pair + 1].store(first, Relaxed); + PERRY_U8_INLINE_CACHE[pair].store(addr as u64, Relaxed); + } } } diff --git a/crates/perry-runtime/src/buffer/mod.rs b/crates/perry-runtime/src/buffer/mod.rs index beb460a384..184172401e 100644 --- a/crates/perry-runtime/src/buffer/mod.rs +++ b/crates/perry-runtime/src/buffer/mod.rs @@ -52,9 +52,10 @@ pub use header::{BufferHeader, BUFFER_TYPE_ID, NODE_BUFFER_CLASS_ID, SMALL_BUF_T // ---- Re-exports: allocation / registry helpers ---- pub(crate) use header::{is_small_buf_slab_addr, visit_ab_alias_slot}; // #9342: primed by `typedarray::js_u8_buffer_read_f64` (codegen slow arm). +pub(crate) use access::{cached_u8_read, cached_u8_write}; #[cfg(test)] pub(crate) use header::test_u8_inline_cache_holds; -pub(crate) use header::u8_inline_cache_try_prime; +pub(crate) use header::{u8_inline_cache_hit, u8_inline_cache_try_prime}; // `shared_sab` publishes process-global backings that `is_registered_buffer` // reports as buffers without them entering `BUFFER_REGISTRY`, so it arms the // same monotone latch — before the backing becomes reachable. @@ -67,7 +68,7 @@ pub use header::{ is_uint8array_buffer, js_set_crypto_key_death_hook, mark_as_array_buffer, mark_as_asymmetric_key, mark_as_crypto_key, mark_as_data_view, mark_as_secret_key, mark_as_shared_array_buffer, mark_as_uint8array, register_buffer, resolve_buffer_ab_alias, - set_buffer_ab_alias, CryptoKeyDeathHookFn, + set_buffer_ab_alias, u8_inline_cache_holds_for_test, CryptoKeyDeathHookFn, }; pub(crate) use header::{ buffer_alloc_foreign, collect_dead_registered_buffers_post_trace, diff --git a/crates/perry-runtime/src/gc/tests/u8_inline_cache.rs b/crates/perry-runtime/src/gc/tests/u8_inline_cache.rs index dd7bbf4a63..b00957a14f 100644 --- a/crates/perry-runtime/src/gc/tests/u8_inline_cache.rs +++ b/crates/perry-runtime/src/gc/tests/u8_inline_cache.rs @@ -33,18 +33,19 @@ fn test_prime_admits_inline_u8_and_contract_holds() { *crate::buffer::buffer_data_mut(buf).add(3) = 0xAB; } - // Unmarked: not a Uint8Array, must not be admitted. + // Unmarked: a Node `Buffer`, whose element semantics are a Uint8Array's + // (#10515 admits it). crate::buffer::u8_inline_cache_try_prime(addr); assert!( - !crate::buffer::test_u8_inline_cache_holds(addr), - "an unmarked buffer must not be admitted" + crate::buffer::test_u8_inline_cache_holds(addr), + "an inline-storage Buffer must be admitted" ); crate::buffer::mark_as_uint8array(addr); crate::buffer::u8_inline_cache_try_prime(addr); assert!( crate::buffer::test_u8_inline_cache_holds(addr), - "a marked inline-storage buffer must be admitted" + "a marked inline-storage Uint8Array must be admitted" ); // The emitted reader's view of an admitted entry: length then byte. @@ -161,3 +162,76 @@ fn test_reissued_address_does_not_inherit_admission() { inline-read admission" ); } + +/// #10515: the non-integer-indexed brands that share `BufferHeader` storage — +/// ArrayBuffer, SharedArrayBuffer, DataView (whose payload holds its data +/// pointer) — are never admitted, and a brand mark that arrives after a prime +/// revokes the admission. Fails if `u8_inline_cache_try_prime` stops asking +/// the brand, or a `mark_as_*` loses its invalidation. +#[test] +fn test_non_byte_view_brands_are_never_admitted() { + let _guard = GcTestIsolationGuard::new(); + + type Mark = fn(usize); + let marks: [(&str, Mark); 3] = [ + ("ArrayBuffer", crate::buffer::mark_as_array_buffer), + ( + "SharedArrayBuffer", + crate::buffer::mark_as_shared_array_buffer, + ), + ("DataView", crate::buffer::mark_as_data_view), + ]; + for (brand, mark) in marks { + let buf = crate::buffer::buffer_alloc(16); + let addr = buf as usize; + unsafe { (*buf).length = 16 }; + mark(addr); + crate::buffer::u8_inline_cache_try_prime(addr); + assert!( + !crate::buffer::test_u8_inline_cache_holds(addr), + "a {brand} is not integer-indexed and must not be admitted" + ); + + // Mark AFTER a prime: the admission must be revoked. + let buf = crate::buffer::buffer_alloc(16); + let addr = buf as usize; + unsafe { (*buf).length = 16 }; + crate::buffer::u8_inline_cache_try_prime(addr); + assert!( + crate::buffer::test_u8_inline_cache_holds(addr), + "test premise: a plain Buffer is admitted" + ); + mark(addr); + assert!( + !crate::buffer::test_u8_inline_cache_holds(addr), + "marking a primed buffer as a {brand} must revoke its admission" + ); + } +} + +/// #10515: the runtime byte accessors answer an admitted buffer from the cache +/// and prime a fresh one on its first access, so the SECOND access of an owning +/// buffer through any runtime route is a cache hit. +#[test] +fn test_runtime_byte_access_primes_and_hits() { + let _guard = GcTestIsolationGuard::new(); + + let buf = crate::buffer::buffer_alloc(8); + let addr = buf as usize; + unsafe { (*buf).length = 8 }; + assert!(!crate::buffer::test_u8_inline_cache_holds(addr)); + crate::buffer::js_buffer_set(buf, 2, 0x1FF); + assert!( + crate::buffer::test_u8_inline_cache_holds(addr), + "the first byte store must prime the admission" + ); + assert_eq!(crate::buffer::cached_u8_read(addr, 2), Some(0xFF)); + assert!(crate::buffer::cached_u8_write(addr, 3, 7)); + assert_eq!(crate::buffer::js_buffer_get(buf, 3), 7); + assert_eq!( + crate::buffer::cached_u8_read(addr, 8), + None, + "out of range leaves the cache" + ); + assert!(!crate::buffer::cached_u8_write(addr, -1, 1)); +} diff --git a/crates/perry-runtime/src/typedarray/access.rs b/crates/perry-runtime/src/typedarray/access.rs index c0d973ade4..72bc93e72a 100644 --- a/crates/perry-runtime/src/typedarray/access.rs +++ b/crates/perry-runtime/src/typedarray/access.rs @@ -652,6 +652,12 @@ pub extern "C" fn js_uint8array_get(target: *const TypedArrayHeader, index: i32) if addr < 0x1000 || index < 0 { return 0; } + // #10515: an admitted owning byte view answers before the typed-array and + // buffer registry probes (`is_registered_buffer_slow` was ~48% of a + // `Uint8Array`-parameter loop). + if let Some(byte) = crate::buffer::cached_u8_read(addr, index) { + return i32::from(byte); + } let value = if lookup_typed_array_kind(addr).is_some() { js_typed_array_get(addr as *const TypedArrayHeader, index) } else if crate::buffer::is_registered_buffer(addr) { @@ -704,6 +710,9 @@ pub extern "C" fn js_uint8array_index_get_value( if addr < 0x1000 || index < 0 { return undefined; } + if let Some(byte) = crate::buffer::cached_u8_read(addr, index) { + return f64::from(byte); + } if lookup_typed_array_kind(addr).is_some() { js_typed_array_get(addr as *const TypedArrayHeader, index) } else if crate::buffer::is_registered_buffer(addr) { @@ -737,6 +746,9 @@ pub extern "C" fn js_uint8array_set(target: *mut TypedArrayHeader, index: i32, v if addr < 0x1000 || index < 0 { return; } + if crate::buffer::cached_u8_write(addr, index, (value & 0xFF) as u8) { + return; + } if lookup_typed_array_kind(addr).is_some() { js_typed_array_set(addr as *mut TypedArrayHeader, index, f64::from(value)); } else if crate::buffer::is_registered_buffer(addr) { diff --git a/crates/perry-runtime/src/value/dyn_index.rs b/crates/perry-runtime/src/value/dyn_index.rs index 1d7a6d69e3..41f1b53939 100644 --- a/crates/perry-runtime/src/value/dyn_index.rs +++ b/crates/perry-runtime/src/value/dyn_index.rs @@ -142,6 +142,30 @@ unsafe fn canonical_buffer_index(key_ptr: *const crate::StringHeader) -> Option< (val <= i32::MAX as u64).then_some(val as u32) } +/// #10515: the cache-hit arm of `js_dyn_index_get` for an admitted owning +/// byte view, kept out of line so the dispatcher's other receivers pay only +/// the inline admission test. +#[inline(never)] +fn cached_u8_index_get(addr: usize, index: f64) -> Option { + let idx = finite_nonnegative_i32_index(index)?; + crate::buffer::cached_u8_read(addr, idx).map(f64::from) +} + +/// #10515: the cache-hit store arm of `js_dyn_index_set_strict`. Only a +/// Number (or int32 box) is stored here: any other value's ToNumber may run +/// user code, which the full path orders against the bounds check. +#[inline(never)] +fn cached_u8_index_set(addr: usize, index: f64, value: f64) -> bool { + let v = JSValue::from_bits(value.to_bits()); + if !(v.is_number() || v.is_int32()) { + return false; + } + let Some(idx) = finite_nonnegative_i32_index(index) else { + return false; + }; + crate::buffer::cached_u8_write(addr, idx, crate::typedarray::jsvalue_to_uint8(value)) +} + /// Tag-aware dynamic index dispatch for `obj[key]` where `obj` has unknown /// static type. Issue #514. Strings → js_string_char_at; objects stringify /// numeric keys (`obj[0]` is `obj["0"]`), while arrays/buffers keep numeric @@ -270,6 +294,15 @@ pub extern "C" fn js_dyn_index_get(value: f64, index: f64) -> f64 { index, ); } + // #10515: an admitted owning byte view (`Uint8Array` / `Buffer`) with a + // canonical in-bounds index answers from the inline-access cache instead + // of the buffer-registry probes below. Placed where buffers are handled so + // no other receiver pays for it. + if crate::buffer::u8_inline_cache_hit(raw_ptr) { + if let Some(byte) = cached_u8_index_get(raw_ptr, index) { + return byte; + } + } // #8149: an `ArrayBuffer` / `SharedArrayBuffer` / `DataView` is a registered // buffer too, but it is NOT an integer-indexed exotic object — node answers // `undefined` for `dv[0]`, never the byte. Ask that ABOVE the byte arm: the @@ -694,6 +727,14 @@ pub extern "C" fn js_dyn_index_set_strict(obj: f64, index: f64, value: f64, stri ); return value; } + // #10515: a Number stored at a canonical in-bounds index of an admitted + // owning byte view (`Uint8Array` / `Buffer`) is one byte write, instead of + // the buffer-registry probes below. Only a Number: any other value's + // ToNumber may run user code, which the full path orders against the + // bounds check. + if crate::buffer::u8_inline_cache_hit(raw_ptr) && cached_u8_index_set(raw_ptr, index, value) { + return value; + } // #8149: an index STORE on an `ArrayBuffer` / `SharedArrayBuffer` / // `DataView` creates an ORDINARY own property — `dv[0] = 7` leaves the byte // at 0, and `Object.keys(dv)` afterwards is `["0"]`. Asked above the diff --git a/crates/perry-stdlib/src/runtime_thread_exit_tests/symbols_tests.rs b/crates/perry-stdlib/src/runtime_thread_exit_tests/symbols_tests.rs index 605d2be6dd..6f17e5aa49 100644 --- a/crates/perry-stdlib/src/runtime_thread_exit_tests/symbols_tests.rs +++ b/crates/perry-stdlib/src/runtime_thread_exit_tests/symbols_tests.rs @@ -26,7 +26,6 @@ extern "C" { ); fn perry_thread_exit_probe_object_prototype_recorded(owner: usize) -> bool; fn js_u8_buffer_read_f64(target: *const u8, index: i32) -> f64; - static PERRY_U8_INLINE_CACHE: [std::sync::atomic::AtomicU64; 64]; } extern "C" fn probe_thunk(_closure: *const perry_runtime::ClosureHeader) -> f64 { @@ -50,11 +49,6 @@ fn addr_of(value: f64) -> usize { (value.to_bits() & ADDR_MASK) as usize } -fn u8_cache_holds(addr: usize) -> bool { - let slot = (addr >> 3) & 63; - unsafe { PERRY_U8_INLINE_CACHE[slot].load(std::sync::atomic::Ordering::Relaxed) == addr as u64 } -} - #[test] fn thread_exit_releases_the_threads_symbol_side_table_entries() { const STATIC_SYMBOL_CLASS: u32 = 0x0B11_4711; @@ -270,12 +264,23 @@ fn thread_exit_releases_the_threads_buffer_own_props() { ); } -/// Membership of `addr` in the three process-global external-buffer -/// registries and `PERRY_U8_INLINE_CACHE`, in that order. Reads no -/// thread-local, so the thread-exit probe below may call it. -fn external_buffer_registrations(addr: usize) -> [bool; 4] { - let [ext, u8a, meta] = perry_runtime::buffer::external_registries_hold_for_test(addr); - [ext, u8a, meta, u8_cache_holds(addr)] +/// Membership of the CryptoKey buffer `key` in the three process-global +/// external-buffer registries, and of the byte view `view` in +/// `PERRY_U8_INLINE_CACHE`, in that order. Reads no thread-local, so the +/// thread-exit probe below may call it. +/// +/// Two addresses because since #11589 key material is never admitted to the +/// inline-access cache (a key is not integer-indexed), so the key buffer can +/// no longer witness the cache's release; an ordinary `Buffer` on the same +/// thread does. +fn external_buffer_registrations(key: usize, view: usize) -> [bool; 4] { + let [ext, u8a, meta] = perry_runtime::buffer::external_registries_hold_for_test(key); + [ + ext, + u8a, + meta, + perry_runtime::buffer::u8_inline_cache_holds_for_test(view), + ] } /// #11547: the external-buffer test's buffer address, and what the tables held @@ -290,18 +295,23 @@ fn external_buffer_registrations(addr: usize) -> [bool; 4] { /// thread-exit range hook that runs after the external-buffer registries' own /// hook, while the block is still owned by the exiting thread and so cannot /// belong to anyone else. -static EXTERNAL_BUFFER_EXIT_PROBE: std::sync::Mutex<(usize, Option<[bool; 4]>)> = - std::sync::Mutex::new((0, None)); +/// +/// Holds `(key, view, seen)`: the CryptoKey buffer, the cache-admitted byte +/// view, and the verdict. +#[allow(clippy::type_complexity)] +static EXTERNAL_BUFFER_EXIT_PROBE: std::sync::Mutex<(usize, usize, Option<[bool; 4]>)> = + std::sync::Mutex::new((0, 0, None)); fn record_external_buffer_release(freed: &perry_runtime::arena::thread_exit::FreedRanges) { let mut probe = EXTERNAL_BUFFER_EXIT_PROBE .lock() .unwrap_or_else(std::sync::PoisonError::into_inner); - let (addr, seen) = *probe; - // First release only: once the address is back with the allocator, a later - // tenant's own thread exit reports it again. - if addr != 0 && seen.is_none() && freed.contains(addr) { - probe.1 = Some(external_buffer_registrations(addr)); + let (key, view, seen) = *probe; + // First release only: once the addresses are back with the allocator, a + // later tenant's own thread exit reports them again. Both live on the same + // thread, so its one release covers both. + if key != 0 && seen.is_none() && freed.contains(key) && freed.contains(view) { + probe.2 = Some(external_buffer_registrations(key, view)); } } @@ -331,17 +341,32 @@ fn thread_exit_releases_the_threads_external_buffer_registrations() { // This also registers their thread-exit hook, so it runs before the // probe registered below. unsafe { js_buffer_mark_as_crypto_key_external(addr, 1, 0, 1, 1, 0, 0) }; - // The codegen inline-read slow arm primes PERRY_U8_INLINE_CACHE. + // The codegen inline-read slow arm primes PERRY_U8_INLINE_CACHE — for + // a byte view. It must refuse the key (#11589: key material is not + // integer-indexed), so an ordinary Buffer carries the cache probe. + let view = scope.root_raw_mut_ptr(perry_runtime::buffer::js_buffer_alloc(16, 0)); + let view_addr = view.get_raw_mut_ptr::() as usize; unsafe { js_u8_buffer_read_f64(addr as *const u8, 0) }; - *EXTERNAL_BUFFER_EXIT_PROBE.lock().unwrap() = (addr, None); + unsafe { js_u8_buffer_read_f64(view_addr as *const u8, 0) }; + let key_admitted = perry_runtime::buffer::u8_inline_cache_holds_for_test(addr); + *EXTERNAL_BUFFER_EXIT_PROBE.lock().unwrap() = (addr, view_addr, None); perry_runtime::arena::thread_exit::register_thread_exit_range_hook( record_external_buffer_release, ); - external_buffer_registrations(addr) + (external_buffer_registrations(addr, view_addr), key_admitted) }) .join() .unwrap(); - let seen = std::mem::replace(&mut *EXTERNAL_BUFFER_EXIT_PROBE.lock().unwrap(), (0, None)).1; + let (alive, key_admitted) = alive; + let seen = std::mem::replace( + &mut *EXTERNAL_BUFFER_EXIT_PROBE.lock().unwrap(), + (0, 0, None), + ) + .2; + assert!( + !key_admitted, + "key material must never enter the inline element-access cache" + ); assert_eq!( alive, [true; 4], "every registration must exist while its thread lives" diff --git a/test-files/test_gap_10515_byte_view_element_access.ts b/test-files/test_gap_10515_byte_view_element_access.ts new file mode 100644 index 0000000000..6ac7a600ff --- /dev/null +++ b/test-files/test_gap_10515_byte_view_element_access.ts @@ -0,0 +1,121 @@ +// #10515 / #10694: Uint8Array and Buffer element reads and writes may be +// answered from the byte-view admission cache (inline, and at the top of the +// runtime accessors). Every shape the cache must NOT answer, or must answer +// exactly like the full path, is exercised here through typed parameters, +// untyped (`any`) parameters and closure captures. + +function getT(b: Uint8Array, i: number): number { return b[i]; } +function setT(b: Uint8Array, i: number, v: number): void { b[i] = v; } +function getB(b: Buffer, i: number): any { return b[i]; } +function setB(b: Buffer, i: number, v: number): void { b[i] = v; } +function getA(b: any, i: any): any { return b[i]; } +function setA(b: any, i: any, v: any): void { b[i] = v; } +function sumT(b: Uint8Array): number { let s = 0; for (let i = 0; i < b.length; i++) s += b[i]; return s; } +function sumA(b: any): number { let s = 0; for (let i = 0; i < b.length; i++) s += b[i]; return s; } +function fillT(b: Uint8Array, k: number): void { for (let i = 0; i < b.length; i++) b[i] = (i * k) & 0xff; } +function hex(b: any): string { return Array.from(b as Uint8Array, (x: number) => x.toString(16).padStart(2, "0")).join(""); } + +// 1. owning Uint8Array / Buffer, typed and untyped, repeated so the cache primes +{ + const u = new Uint8Array(8); const b = Buffer.alloc(8); + for (let r = 0; r < 3; r++) { + setT(u, r, 250 + r); setB(b, r, 300 + r); setA(u, r + 3, -1 - r); setA(b, r + 3, r + 0.75); + } + console.log("owning", hex(u), hex(b), getT(u, 1), getB(b, 1), getA(u, 4), getA(b, 4)); + console.log("sums", sumT(u), sumA(b), sumT(b), sumA(u)); +} + +// 2. value conversion on the store: wrapping, fractions, NaN, and non-Numbers +{ + const u = new Uint8Array(10); + const vals: any[] = [256, -1, 1.5, -1.5, NaN, Infinity, 4294967295, "7", true, null]; + for (let i = 0; i < vals.length; i++) setA(u, i, vals[i]); + console.log("convert-any", Array.from(u).join(",")); + const w = new Uint8Array(4); + setT(w, 0, 511); setT(w, 1, -129); setT(w, 2, 2.9); setT(w, 3, 1e10); + console.log("convert-typed", Array.from(w).join(",")); + let calls = 0; + const obj = { valueOf() { calls++; return 66; } }; + setA(w, 0, obj); setA(w, 9, obj); + console.log("valueOf", w[0], calls); +} + +// 3. out-of-bounds and non-canonical keys +{ + const u = new Uint8Array([1, 2, 3]); const b = Buffer.from([4, 5, 6]); + console.log("oob-read", getT(u, 3), getA(u, 3), getA(u, -1), getA(u, 1.5), getB(b, 7), getA(b, "01")); + setT(u, 3, 9); setA(u, 5, 9); setA(u, -1, 9); setA(u, 1.5, 9); setA(b, 3, 9); + console.log("oob-write", Array.from(u).join(","), u.length, Object.keys(u).join("|"), hex(b), b.length); +} + +// 4. views: subarray and Uint8Array-over-ArrayBuffer alias their backing +{ + const base = new Uint8Array(16); fillT(base, 3); + const sub = base.subarray(4, 8); + setT(sub, 0, 200); setA(sub, 1, 201); setT(base, 6, 202); setA(base, 7, 203); + console.log("subarray", getT(sub, 0), getA(sub, 1), getT(sub, 2), getA(sub, 3), getT(base, 4), getA(base, 5), sumT(sub)); + const ab = new ArrayBuffer(12); + const v1 = new Uint8Array(ab); const v2 = new Uint8Array(ab, 3, 5); + for (let r = 0; r < 3; r++) { setT(v1, 3 + r, 10 + r); setA(v2, 3, 90 + r); } + console.log("ab-views", hex(v1), hex(v2), getT(v2, 0), getA(v1, 6)); + const bsl = Buffer.alloc(10); const bview = bsl.subarray(2, 6); + setB(bview, 1, 77); setA(bsl, 4, 88); + console.log("buffer-subarray", hex(bsl), hex(bview)); + const slice = base.slice(0, 4); setT(slice, 0, 1); + console.log("slice-copy", getT(slice, 0), getT(base, 0)); +} + +// 5. ArrayBuffer, SharedArrayBuffer and DataView are not integer-indexed +{ + const ab: any = new ArrayBuffer(4); const dv: any = new DataView(ab); + setA(ab, 0, 5); setA(dv, 1, 6); + console.log("non-indexed", getA(ab, 0), getA(dv, 1), new Uint8Array(ab).join(","), Object.keys(ab).join("|"), Object.keys(dv).join("|")); + const sab = new SharedArrayBuffer(4); const s1 = new Uint8Array(sab); const s2: any = new Uint8Array(sab); + setT(s1, 2, 33); setA(s2, 3, 44); + console.log("shared", getA(s2, 2), getT(s1, 3)); +} + +// 6. detach / transfer and resizable backings +{ + const ab = new ArrayBuffer(8); const u = new Uint8Array(ab); + setT(u, 0, 1); getT(u, 0); getT(u, 0); + const moved = ab.transfer(); + setT(u, 0, 9); setA(u, 1, 9); + console.log("detached", u.length, getT(u, 0), getA(u, 1), new Uint8Array(moved)[0]); + const rab = new ArrayBuffer(4, { maxByteLength: 16 }); const ru = new Uint8Array(rab); + setT(ru, 3, 7); rab.resize(8); setT(ru, 6, 8); setA(ru, 7, 9); + console.log("resizable", ru.length, Array.from(ru).join(",")); + rab.resize(2); + console.log("shrunk", ru.length, getT(ru, 3), getA(ru, 1)); +} + +// 7. many live buffers hitting the same cache sets, alternating access +{ + const pool: Uint8Array[] = []; + for (let k = 0; k < 40; k++) { const p = k % 2 ? Buffer.alloc(32 + k) : new Uint8Array(24 + k); pool.push(p); } + let acc = 0; + for (let r = 0; r < 20; r++) for (let k = 0; k < pool.length; k++) { const p = pool[k]; setT(p, r % p.length, r + k); acc = (acc + getT(p, (r * 7) % p.length) + getA(pool[(k * 13) % pool.length], r)) % 1000003; } + console.log("pool", acc, pool.map((p) => sumT(p)).join(",")); +} + +// 8. buffers created and dropped in a loop (address reuse after collection) +{ + let acc = 0; + for (let r = 0; r < 300; r++) { + const t = r % 3 === 0 ? Buffer.alloc(64) : new Uint8Array(48); + for (let i = 0; i < t.length; i++) setT(t, i, i + r); + acc = (acc + sumT(t) + getA(t, 5)) % 1000003; + if (r % 100 === 50 && typeof (globalThis as any).gc === "function") (globalThis as any).gc(); + } + console.log("churn", acc); +} + +// 9. closure captures (the nanoid pool shape) and named expandos +{ + const ALPHA = "useandom-26T198340PX75pxJACKVERYMINDBUSHWOLF_GQZbfghjklqvwyzrict"; + const mk = () => { const cc = Uint8Array.from(ALPHA, (s: string) => s.charCodeAt(0)); let mask = 63; return (b: Uint8Array) => { for (let i = 0; i < b.length; i++) b[i] = cc[b[i] & mask]; return b; }; }; + const refill = mk(); const pool = Buffer.alloc(32); for (let i = 0; i < 32; i++) pool[i] = i * 11; + console.log("nanoid", refill(pool).toString("latin1"), refill(pool).toString("latin1")); + const e: any = new Uint8Array(3); e.tag = "x"; setA(e, "name", "n"); setA(e, 1, 5); + console.log("expando", e.tag, e.name, e[1], Object.keys(e).join("|")); +}