diff --git a/changelog.d/11543-regex-split-replace-ascii.md b/changelog.d/11543-regex-split-replace-ascii.md new file mode 100644 index 0000000000..06da7c2d8e --- /dev/null +++ b/changelog.d/11543-regex-split-replace-ascii.md @@ -0,0 +1,9 @@ +### Faster + +- `String.prototype.split` with a regular expression and string-template `String.prototype.replace` build their pieces with far less host work on ASCII strings (#10165, #10518): + - A split piece or a replace capture is now one byte copy. It used to take two unit-by-unit passes and two GC safepoint polls. + - A replacement's output is now one allocation plus a copy per piece. It used to decode and re-encode every unit twice. + - The split loop polls once per 512 units rather than once per piece. + - A split with no capture groups no longer builds a capture array for every piece. + - A split's private output list appends without a catch frame while it has spare capacity. +- Measured on qb2 in instructions: `encodeURI(s).split(/%..|./)` −43%, a short global replace −15%. validator's `sanitize` and `batch` package workloads are −4.2% and −3.3%. Peak RSS is unchanged within 1%. diff --git a/crates/perry-runtime/src/regex/perex_replace_storage.rs b/crates/perry-runtime/src/regex/perex_replace_storage.rs index 15e5ce04b5..6b466c0344 100644 --- a/crates/perry-runtime/src/regex/perex_replace_storage.rs +++ b/crates/perry-runtime/src/regex/perex_replace_storage.rs @@ -73,6 +73,36 @@ impl<'a> List<'a> { } Ok(()) } + + /// `push` for a list no code outside this operation has seen yet, such as + /// a `split` result still being built. Nothing can have frozen, sealed or + /// wrapped it, so an append that fits its capacity stores in place and + /// cannot throw; it skips the catch frame `push` sets up, which was about + /// a tenth of a short `split`. An append that must grow takes `push`. + pub(super) fn push_unseen( + &mut self, + value: f64, + budget: &mut Budget, + ) -> Result<(), EngineError> { + let fits = self + .root + .with_const_ptr::(|array| unsafe { + (*array).length < (*array).capacity + }); + if !fits { + return self.push(value, budget); + } + host::charge(budget, 1)?; + let array = self + .root + .with_mut_ptr(|array| crate::array::js_array_push_f64(array, value)); + self.root.set_raw_mut_ptr(array); + self.count += 1; + if self.count % api::QUANTUM == 0 { + host::poll()?; + } + Ok(()) + } } pub(super) fn call( @@ -229,7 +259,7 @@ pub(super) fn call_native( /// This value is therefore a measured trade rather than a bound inherited from /// elsewhere: small enough to keep the collector's openings, large enough that /// a piece of two or three units no longer buys a poll of its own. -const POLL_UNITS: usize = 512; +pub(super) const POLL_UNITS: usize = 512; /// A reusable original-input reader. A read retains only Perex offsets across /// collection, and adjacent reads do not repeat the initial Unicode seek. @@ -531,12 +561,144 @@ impl<'a> Pieces<'a> { } Ok(()) } + /// `finish` for native pieces over an ASCII subject and template: every + /// piece is then a byte span already in the output's encoding, so the + /// output is one allocation and a copy per piece, instead of two passes + /// that decode and re-encode every unit through a cursor. `None` for any + /// other `Pieces`, which `finish` builds as before. + fn finish_ascii( + &self, + original: &RuntimeHandle<'_>, + template: Option<&RuntimeHandle<'_>>, + budget: &mut Budget, + ) -> Result, EngineError> { + let Some(native) = self.native.as_ref() else { + return Ok(None); + }; + if self.list.len() != 0 { + return Err(EngineError::InvalidSpan); + } + let original_subject = subject(*original)?; + let template_subject = template.map(|t| subject(*t)).transpose()?; + let ascii_length = |bound: &BoundSubject>| { + bound + .with_view(|input| input.ascii_bytes().map(<[u8]>::len)) + .map_err(EngineError::Subject) + }; + let Some(original_length) = ascii_length(&original_subject)? else { + return Ok(None); + }; + let template_length = match template_subject.as_ref() { + Some(bound) => match ascii_length(bound)? { + Some(length) => length, + None => return Ok(None), + }, + None => 0, + }; + let mut total = 0usize; + for &(source, start, end) in &native.records { + let length = match source { + Source::Original => original_length, + Source::Template if template_subject.is_some() => template_length, + Source::Template => return Err(EngineError::InvalidSpan), + }; + if start > end || end as usize > length { + return Err(EngineError::InvalidSpan); + } + total = total + .checked_add((end - start) as usize) + .ok_or(StorageError::Limit)?; + } + if total != self.units { + return Err(EngineError::InvalidSpan); + } + let limit = api::OUTPUT_BYTES.min( + u32::MAX as usize - crate::gc::GC_HEADER_SIZE - std::mem::size_of::() - 7, + ); + if total > limit || total > crate::string::MAX_STRING_LENGTH { + return Err(StorageError::Limit.into()); + } + // The charge the two unit-by-unit passes made. + host::charge(budget, total.saturating_mul(2))?; + let scope = RuntimeHandleScope::new(); + let output = api::caught(|| { + let (p, _) = crate::string::string_storage_alloc(total as u32); + unsafe { + crate::string::init_string_header(p, 0, 0, total as u32, 0, 0); + } + p + })?; + let output = scope.root_string_ptr(output); + // Nothing below allocates or collects: both subject views and the + // output's data pointer are reacquired inside one scope. + output.with_mut_ptr::(|header| { + let data = crate::string::string_data(header).cast_mut(); + let mut written = 0usize; + original_subject + .with_view(|original| { + let original = original.ascii_bytes().ok_or(EngineError::InvalidSpan)?; + let mut copy = |bytes: &[u8]| { + // Bounded by the total measured above. + unsafe { + std::ptr::copy_nonoverlapping( + bytes.as_ptr(), + data.add(written), + bytes.len(), + ); + } + written += bytes.len(); + }; + match template_subject.as_ref() { + Some(template) => template + .with_view(|template| { + let template = + template.ascii_bytes().ok_or(EngineError::InvalidSpan)?; + for &(source, start, end) in &native.records { + let bytes = match source { + Source::Original => original, + Source::Template => template, + }; + copy(&bytes[start as usize..end as usize]); + } + Ok::<(), EngineError>(()) + }) + .map_err(EngineError::Subject)?, + None => { + for &(_, start, end) in &native.records { + copy(&original[start as usize..end as usize]); + } + Ok(()) + } + } + }) + .map_err(EngineError::Subject)??; + if written != total { + return Err(EngineError::InvalidSpan); + } + unsafe { + crate::string::init_string_header( + header, + total as u32, + total as u32, + total as u32, + 0, + 0, + ); + } + Ok::<(), EngineError>(()) + })?; + Ok(Some(output.with_mut_ptr(|output| output))) + } + pub(super) fn finish( &self, original: &RuntimeHandle<'_>, template: Option<&RuntimeHandle<'_>>, budget: &mut Budget, ) -> Result<*mut StringHeader, EngineError> { + if let Some(output) = self.finish_ascii(original, template, budget)? { + return Ok(output); + } let limit = api::OUTPUT_BYTES.min( u32::MAX as usize - crate::gc::GC_HEADER_SIZE - std::mem::size_of::() - 7, ); diff --git a/crates/perry-runtime/src/regex/perex_runtime.rs b/crates/perry-runtime/src/regex/perex_runtime.rs index fccd00fe3d..9cafb56082 100644 --- a/crates/perry-runtime/src/regex/perex_runtime.rs +++ b/crates/perry-runtime/src/regex/perex_runtime.rs @@ -52,6 +52,43 @@ impl From for EngineError { } } +/// The safepoint poll for a loop that emits many small pieces: once per +/// `POLL_UNITS` units of output, counting each piece as one more unit so a +/// run of empty or one-unit pieces still reaches it. A poll per piece runs the +/// budgeted trigger ladder for a handful of units each, which on a short +/// `split` was a third of the call; `perex_replace_storage::POLL_UNITS` +/// records why 512 keeps the collector's openings without a peak-RSS cost. +pub(crate) struct PieceStride { + unpolled: usize, +} + +impl PieceStride { + pub(crate) const fn new() -> Self { + Self { unpolled: 0 } + } + + /// Whether the piece of `units` just emitted brings the poll due, which it + /// then resets. The caller runs the poll. + pub(crate) fn due(&mut self, units: usize) -> bool { + self.unpolled = self.unpolled.saturating_add(units).saturating_add(1); + if self.unpolled >= super::perex_replace_storage::POLL_UNITS { + self.unpolled = 0; + true + } else { + false + } + } + + /// Account for a piece and poll when that brings the poll due. + pub(crate) fn tick(&mut self, units: usize) -> Result<(), EngineError> { + if self.due(units) { + poll() + } else { + Ok(()) + } + } +} + /// Normal runtime poll. Test/embedding callers may supply an alternative poll /// that requests cancellation or forces actual collection. No input/program /// view or scratch slice is live when any poll is invoked. diff --git a/crates/perry-runtime/src/regex/perex_split.rs b/crates/perry-runtime/src/regex/perex_split.rs index 7cd584e4d3..42d43d04c9 100644 --- a/crates/perry-runtime/src/regex/perex_split.rs +++ b/crates/perry-runtime/src/regex/perex_split.rs @@ -133,6 +133,21 @@ fn forward_split( // Each search starts where the previous one stood, so on non-ASCII // storage it does not seek from an end of the subject (#10164). let mut near: Option = None; + // A piece is usually a few units; the loop polls once per POLL_UNITS of + // them rather than once per piece. + let mut stride = host::PieceStride::new(); + // Without capture groups a piece needs only the match itself; asking for + // every capture built, filled and copied a slot array per piece to learn + // that there were none. + let mode = if forward + .with_view(|program| program.capture_count()) + .map_err(EngineError::Program)? + > 1 + { + CaptureMode::All + } else { + CaptureMode::Full + }; while q < size { let local = RuntimeHandleScope::new(); let (found, position) = host::find_near( @@ -140,7 +155,7 @@ fn forward_split( bound, q, near, - CaptureMode::All, + mode, budget, memory, api::QUANTUM, @@ -160,7 +175,7 @@ fn forward_split( // Only an empty match at `p` itself: step past it, as the // sticky loop does. q = advance(&mut units, start, size, unicode, budget)?; - host::poll()?; + stride.tick(0)?; continue; } push_span(&mut output, &mut copies, p, start, budget)?; @@ -196,8 +211,8 @@ fn forward_split( } } } + stride.tick(p - q)?; q = p; - host::poll()?; } push_span(&mut output, &mut copies, p, size, budget)?; charged(budget); @@ -212,7 +227,7 @@ fn push_span( budget: &mut Budget, ) -> Result<(), EngineError> { let result = copies.copy(start, end, budget)?; - output.push(js_nanbox_string(result as i64), budget) + output.push_unseen(js_nanbox_string(result as i64), budget) } /// Split by an untouched RegExp without its protocol Gets (#10518). diff --git a/crates/perry-runtime/src/regex/perex_strings.rs b/crates/perry-runtime/src/regex/perex_strings.rs index 110be27f42..cbfd4a766a 100644 --- a/crates/perry-runtime/src/regex/perex_strings.rs +++ b/crates/perry-runtime/src/regex/perex_strings.rs @@ -244,16 +244,26 @@ fn copy_ascii_span( /// repeatedly from the beginning of a non-ASCII input. Only offsets survive GC. pub(super) struct SpanCopies<'a, 's> { readers: [BoundSpan<'a, HeapSubject<'s>>; 2], + /// The subject, when it is ASCII: a piece is then one byte copy, as + /// `copy_ascii_span` makes for a capture, instead of two unit-by-unit + /// passes and two polls. + ascii: Option<&'a BoundSubject>>, + stride: super::perex_runtime::PieceStride, } impl<'a, 's> SpanCopies<'a, 's> { pub(super) fn new(subject: &'a BoundSubject>) -> Result { let empty = Span::new(0, 0).unwrap(); + let ascii = subject + .with_view(|input| input.ascii_bytes().is_some()) + .map_err(EngineError::Subject)?; Ok(Self { readers: [ BoundSpan::new(subject, empty).map_err(|e| read_error(e, |n| match n {}))?, BoundSpan::new(subject, empty).map_err(|e| read_error(e, |n| match n {}))?, ], + ascii: ascii.then_some(subject), + stride: super::perex_runtime::PieceStride::new(), }) } @@ -264,6 +274,27 @@ impl<'a, 's> SpanCopies<'a, 's> { budget: &mut Budget, ) -> Result<*mut StringHeader, EngineError> { let span = Span::new(start, end).ok_or(EngineError::InvalidSpan)?; + if let Some(subject) = self.ascii { + // One poll per `POLL_UNITS` units of pieces rather than two per + // piece; the copy itself allocates through the ordinary path. + let due = self.stride.due(span.len()); + let mut poll = || { + if due { + super::perex_runtime::poll() + } else { + Ok(()) + } + }; + if let Some(output) = copy_ascii_span( + subject, + span, + budget, + super::perex_api::OUTPUT_BYTES, + &mut poll, + )? { + return Ok(output); + } + } for reader in &mut self.readers { reader .retarget(span) diff --git a/test-files/test_gap_regex_engine_package_shapes.ts b/test-files/test_gap_regex_engine_package_shapes.ts new file mode 100644 index 0000000000..19ee3ed780 --- /dev/null +++ b/test-files/test_gap_regex_engine_package_shapes.ts @@ -0,0 +1,127 @@ +// Regex shapes the package workloads spend their time in (uuid validate, +// jws JWS_REGEX, dotenv LINE, validator isByteLength split, node-cron), plus +// the edge cases the engine's fast paths for them must keep exact: folded +// classes (including the two non-ASCII characters that fold into ASCII under +// `u`), counted repeats, lazy repeats before a literal, captures, lastIndex, +// sticky and global iteration. Output must match Node byte-for-byte. + +function show(label: string, value: unknown): void { + console.log(label + " " + JSON.stringify(value)); +} + +// uuid validate: counted folded classes in an alternation. +const UUID = + /^(?:[0-9a-f]{8}-[0-9a-f]{4}-[1-8][0-9a-f]{3}-[89ab][0-9a-f]{3}-[0-9a-f]{12}|00000000-0000-0000-0000-000000000000|ffffffff-ffff-ffff-ffff-ffffffffffff)$/i; +for (const u of [ + "3b241101-e2bb-4255-8caf-4136c566a962", + "9C5B94B1-35AD-49BB-B118-8E8FC24ABF80", + "00000000-0000-0000-0000-000000000000", + "FFFFFFFF-FFFF-FFFF-FFFF-FFFFFFFFFFFF", + "3b241101-e2bb-4255-8caf-4136c566a96", + "3b241101-e2bb-4255-8caf-4136c566a9621", + "3b241101-e2bb-9255-8caf-4136c566a962", + "3b241101-e2bb-4255-7caf-4136c566a962", + "3b24110g-e2bb-4255-8caf-4136c566a962", + "", +]) { + show("uuid " + u, UUID.test(u)); +} + +// jws: lazy repeats before a literal, then an optional greedy capture. +const JWS = /^[a-zA-Z0-9\-_]+?\.[a-zA-Z0-9\-_]+?\.([a-zA-Z0-9\-_]+)?$/; +for (const t of [ + "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.c2lnbmF0dXJl", + "a.b.c", + "a.b.", + "a..c", + ".b.c", + "a.b.c.d", + "a.b", + "abc.def.gh!i", + "abc.def.ghi\n", +]) { + show("jws " + JSON.stringify(t), JWS.test(t)); + show("jws-exec " + JSON.stringify(t), JWS.exec(t)); +} +show("lazy-cap", /^([a-z]+?)([a-z]{2})$/.exec("abcdef")); +show("lazy-min", /^(a{2,}?)(a*)b$/.exec("aaaaab")); +show("lazy-bound", /^(x{1,3}?)\.$/.exec("xxx.")); +show("lazy-bound-over", /^(x{1,3}?)\.$/.exec("xxxx.")); +show("lazy-fold-lit", /^([a-z]+?)K/i.exec("abckdef")); +show("lazy-fold-lit-u", /^([a-z]+?)k/iu.exec("abKdef")); +show("lazy-class-cont", /^(\w+?)(\d+)$/.exec("abc123")); +show("lazy-any", /<(.+?)>/g[Symbol.match]("")); +show("lazy-nonascii", /^(.+?)é(.*)$/.exec("abcé123é")); +show("lazy-astral", /^(.+?)\u{1F600}/u.exec("ab\u{1F600}c")); + +// Folded classes: legacy and unicode case equivalence. +show("fold-kelvin-legacy", /[a-z]/i.test("K")); +show("fold-kelvin-u", /[a-z]/iu.test("K")); +show("fold-long-s-u", /^[s]$/iu.test("ſ")); +show("fold-long-s-legacy", /^[s]$/i.test("ſ")); +show("fold-kelvin-class-u", /^[K]+$/iu.test("kKK")); +show("fold-negated", /^[^a-f]+$/i.test("GHIJ")); +show("fold-negated-miss", /^[^a-f]+$/i.test("GHIa")); +show("fold-negated-u", /[^k]/iu.test("K")); +show("fold-digits", /^[0-9a-f]{4}$/i.test("aF09")); +show("fold-sigma", /^[σ]+$/i.test("Σσς")); +show("fold-sigma-u", /^[σ]+$/iu.test("Σσς")); +show("fold-greek-range", "ΑΒΓαβγ".match(/[α-γ]+/gi)); +show("fold-latin1", /^[à-þ]+$/i.test("ÀÉÎõü")); +show("fold-word", "a_Z9ſK".match(/\w/giu)); +show("fold-word-legacy", "a_Z9ſK".match(/\w/gi)); +show("fold-dash", /^[a-z-]+$/i.test("Ab-C")); +show("fold-v", /^[\p{Lu}--[A-C]]+$/iv.test("DEF")); + +// Counted repeats: exact, bounded, captured, sticky, global. +show("count-exact", /^\d{3}-\d{4}$/.test("555-1234")); +show("count-short", /^\d{3}-\d{4}$/.test("55-1234")); +show("count-range", "12 1234 123456".match(/\b\d{2,4}\b/g)); +show("count-cap", /^(\d{1,2})-(\d{1,2})$/.exec("7-12")); +show("count-sticky", (() => { const r = /\d{2}/y; r.lastIndex = 1; const m = r.exec("a12b"); return [m, r.lastIndex]; })()); +show("count-sticky-miss", (() => { const r = /\d{2}/y; r.lastIndex = 2; const m = r.exec("a12b"); return [m, r.lastIndex]; })()); +show("count-global", (() => { const r = /[a-c]{2}/g; const out: unknown[] = []; let m; while ((m = r.exec("abcabcab")) !== null) out.push([m[0], m.index, r.lastIndex]); return out; })()); +show("cron-l", [/^l-\d{1,2}$/i.test("L-12"), /^l-\d{1,2}$/i.test("l-123"), /^[0-7]l$/i.test("5L")]); +show("count-nonascii", /^(é{2})(.)$/.exec("ééx")); +show("count-astral-u", /^(.{2})$/u.exec("\u{1F600}a")); + +// dotenv LINE over a small document. +const LINE = + /(?:^|^)\s*(?:export\s+)?([\w.-]+)(?:\s*=\s*?|:\s+?)(\s*'(?:\\'|[^'])*'|\s*"(?:\\"|[^"])*"|\s*`(?:\\`|[^`])*`|[^#\r\n]+)?\s*(?:#.*)?(?:$|$)/mg; +const doc = "# c\nexport A=1\nB = two words # note\nC='q # x'\nD=\"m\\nn\"\nE: colon\n F = spaced \nG=\n"; +const pairs: unknown[] = []; +let m: RegExpExecArray | null; +while ((m = LINE.exec(doc)) != null) pairs.push([m[1], m[2], m.index, LINE.lastIndex]); +show("dotenv", pairs); + +// validator isByteLength: split on a two-way alternation. +for (const s of ["hello", "caf%C3%A9", "", "%20%20", "a%2"]) { + show("bytelen " + s, s.split(/%..|./)); +} +show("split-limit", "a,b;c".split(/[,;]/, 2)); +show("split-cap", "a1b22c".split(/(\d+)/)); + +// Whitespace class (more than eight ranges) as a repeated atom. +show("ws", "a \t  b".split(/\s+/)); +show("ws-trim", " x y ".replace(/^\s+|\s+$/g, "")); +show("ws-nonascii", /^\s{3}$/.test("
 ")); + +// String-template replacements, whose output is built from spans of the +// subject and the template: ASCII and non-ASCII on either side, every +// substitution form, empty pieces and empty results. +show("rep-g", "a-b-c".replace(/-/g, "+")); +show("rep-1", "Hello".replace(/l/, "L")); +show("rep-empty", "---".replace(/-/g, "")); +show("rep-all-empty", "".replace(/x*/g, "y")); +show("rep-dollar", "a-b".replace(/(-)/g, "[$1|$&|$`|$'|$$]")); +show("rep-named", "2026-09-27".replace(/(?\d+)-(?\d+)-(?\d+)/, "$/$/$")); +show("rep-missing-group", "ab".replace(/(a)(x)?/, "[$2]")); +show("rep-nonascii-subject", "café-thé".replace(/-/g, " & ")); +show("rep-nonascii-template", "a-b".replace(/-/g, "→")); +show("rep-astral", "a\u{1F600}b".replace(/\u{1F600}/u, "$&$&")); +show("rep-lone", "a\ud800b".replace(/b/, "$`")); +show("rep-escape", 'say "hi"\\n'.replace(/\\n/g, "\n").replace(/^"|"$/g, "")); +show("rep-dotenv-quote", "'quoted value'".replace(/^(['"`])([\s\S]*)\1$/gm, "$2")); +show("rep-sticky", (() => { const r = /a/y; r.lastIndex = 1; return ["baa".replace(r, "X"), r.lastIndex]; })()); +show("rep-replaceAll", "x.y.z".replaceAll(/\./g, "$&$&")); +show("rep-long", ("ab-".repeat(300)).replace(/-/g, "+").length);