Skip to content

Commit 54e8d02

Browse files
committed
perf: literal-prefix capture-extraction fast path
For anchored patterns of the shape ^<literal-prefix-set>([^X]+)X.*$ with replacement `${1}` (or `$1`) capture 1's bounds are structurally trivial — skip the prefix, find the terminator with memchr — so the engine doesn't need to track captures at all. Two changes work together: 1. A new `LiteralPrefixCapture` strategy in `regex-automata`'s meta engine recognizes the shape via HIR walking (single-pattern only, anchored at both ends, default flags, ASCII terminator, finite literal-alternation prefix set capped at 32 variants). Strategy methods extract the match and capture-1 slots directly with memchr, bypassing PikeVM / BoundedBacktracker. Wires in alongside the existing reverse strategies. 2. `Regex::replacen` gets a borrowed-output fast path for replacements that are exactly `$N` / `${N}`. Detected via a new `Replacer::single_capture_ref` method (default `None`, opted into for `&str`/`String`/`Cow<str>`). For `limit == 1` with a match covering the whole haystack, returns `Cow::Borrowed` of the captured slice — no `Captures::expand`, no output string allocation. Bench (500k synthetic Referer rows, 5-iter mean, on the same machine): Regex::replacen, q28 pattern, 80% match before: 281 ms after: 39 ms (7.3x) Regex::replacen, ^key=([^,]+),.*$, 100% match before: 113 ms after: 27 ms (4.2x) Tests: 257 / 257 pass (regex-automata --lib + --test integration, regex --test integration). No regressions.
1 parent 839d16b commit 54e8d02

2 files changed

Lines changed: 468 additions & 2 deletions

File tree

‎regex-automata/src/meta/strategy.rs‎

Lines changed: 365 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -3,9 +3,9 @@ use core::{
33
panic::{RefUnwindSafe, UnwindSafe},
44
};
55

6-
use alloc::sync::Arc;
6+
use alloc::{boxed::Box, sync::Arc, vec, vec::Vec};
77

8-
use regex_syntax::hir::{literal, Hir};
8+
use regex_syntax::hir::{literal, Class, Hir, HirKind, Literal, Look};
99

1010
use crate::{
1111
meta::{
@@ -160,6 +160,13 @@ pub(super) fn new(
160160
// might give up or quit for reasons. If we had, e.g., a PikeVM that
161161
// supported reverse searching, then we could avoid building a full Core
162162
// engine for this case.
163+
core = match LiteralPrefixCapture::new(core, hirs) {
164+
Err(core) => core,
165+
Ok(lpc) => {
166+
debug!("using literal-prefix-capture strategy");
167+
return Ok(Arc::new(lpc));
168+
}
169+
};
163170
core = match ReverseAnchored::new(core) {
164171
Err(core) => core,
165172
Ok(ra) => {
@@ -1903,3 +1910,359 @@ fn copy_match_to_slots(m: Match, slots: &mut [Option<NonMaxUsize>]) {
19031910
*slot = NonMaxUsize::new(m.end());
19041911
}
19051912
}
1913+
1914+
/// A specialized strategy for anchored, fully-bounded regexes of the form
1915+
///
1916+
/// ```text
1917+
/// ^<literal-prefix-set>([^X]+)X.*$
1918+
/// ```
1919+
///
1920+
/// where the prefix reduces to a finite set of literal byte alternatives,
1921+
/// the capture is a greedy `[^X]+` for a single ASCII byte X, and the trailing
1922+
/// `.*$` is the standard "rest of line, then end of haystack" tail. The
1923+
/// motivating instance is the ClickBench Q28 pattern
1924+
/// `^https?://(?:www\.)?([^/]+)/.*$` -> `${1}`, but the recognizer applies to
1925+
/// any pattern of this shape (single-literal prefixes, alternation, and
1926+
/// `?`-optional segments).
1927+
///
1928+
/// For inputs that match, capture 1's bounds are structurally trivial — skip
1929+
/// the prefix, find the terminator with `memchr` — so we can avoid the full
1930+
/// engine's capture-tracking entirely. For inputs that don't match (e.g., a
1931+
/// newline in the tail breaks `.*$`, or no prefix matches), we report no
1932+
/// match: that result is identical to what the full engine would compute, so
1933+
/// no fallback is required.
1934+
#[derive(Debug)]
1935+
struct LiteralPrefixCapture {
1936+
core: Core,
1937+
/// Distinct literal byte prefixes, longest-first so the runtime probe
1938+
/// is greedy. Bounded to `MAX_PREFIX_VARIANTS` at construction time.
1939+
prefixes: Box<[Box<[u8]>]>,
1940+
/// Single ASCII byte ending the capture (also the literal that must
1941+
/// follow the capture in the original regex).
1942+
terminator: u8,
1943+
}
1944+
1945+
/// Each `(?:...)?` doubles the count and each `(a|b|c)` multiplies it,
1946+
/// so this caps the explosion for adversarial patterns. 32 fits roughly
1947+
/// 8 levels of optional/alternation past Q28's 4 variants on one cache
1948+
/// line of `Box<[u8]>`.
1949+
const MAX_PREFIX_VARIANTS: usize = 32;
1950+
1951+
impl LiteralPrefixCapture {
1952+
fn new(core: Core, hirs: &[&Hir]) -> Result<Self, Core> {
1953+
if hirs.len() != 1 {
1954+
return Err(core);
1955+
}
1956+
if !core.info.is_always_anchored_start()
1957+
|| !core.info.is_always_anchored_end()
1958+
{
1959+
return Err(core);
1960+
}
1961+
// `.*$` excludes the line terminator; the runtime newline check
1962+
// hard-codes `b'\n'`, so reject non-default line terminators.
1963+
if core.info.config().get_line_terminator() != b'\n' {
1964+
return Err(core);
1965+
}
1966+
let Some((prefixes, terminator)) =
1967+
try_recognize_prefix_capture(hirs[0])
1968+
else {
1969+
return Err(core);
1970+
};
1971+
Ok(LiteralPrefixCapture { core, prefixes, terminator })
1972+
}
1973+
1974+
/// Returns capture 1's byte offsets if the input matches, else `None`.
1975+
/// The overall match always spans `0..input.haystack().len()` because
1976+
/// the regex is `^...$`.
1977+
#[cfg_attr(feature = "perf-inline", inline(always))]
1978+
fn try_fast_match(&self, input: &Input<'_>) -> Option<(usize, usize)> {
1979+
if input.start() != 0 || input.end() != input.haystack().len() {
1980+
return None;
1981+
}
1982+
let bytes = input.haystack();
1983+
for prefix in self.prefixes.iter() {
1984+
if !bytes.starts_with(prefix) {
1985+
continue;
1986+
}
1987+
let cap_start = prefix.len();
1988+
// Fused scan: the first byte that matters in the tail is either
1989+
// the terminator (success) or `\n` (failure for `.*$`).
1990+
let off = crate::util::memchr::memchr2(
1991+
self.terminator,
1992+
b'\n',
1993+
&bytes[cap_start..],
1994+
)?;
1995+
if bytes[cap_start + off] != self.terminator {
1996+
return None;
1997+
}
1998+
if off == 0 {
1999+
// `[^X]+` requires >= 1 byte; try a shorter prefix.
2000+
continue;
2001+
}
2002+
let cap_end = cap_start + off;
2003+
// Anything past the terminator must also be `\n`-free for
2004+
// `.*$` to reach end-of-haystack.
2005+
if crate::util::memchr::memchr(b'\n', &bytes[cap_end + 1..])
2006+
.is_some()
2007+
{
2008+
return None;
2009+
}
2010+
return Some((cap_start, cap_end));
2011+
}
2012+
None
2013+
}
2014+
}
2015+
2016+
impl Strategy for LiteralPrefixCapture {
2017+
#[cfg_attr(feature = "perf-inline", inline(always))]
2018+
fn group_info(&self) -> &GroupInfo {
2019+
self.core.group_info()
2020+
}
2021+
2022+
#[cfg_attr(feature = "perf-inline", inline(always))]
2023+
fn create_cache(&self) -> Cache {
2024+
self.core.create_cache()
2025+
}
2026+
2027+
#[cfg_attr(feature = "perf-inline", inline(always))]
2028+
fn reset_cache(&self, cache: &mut Cache) {
2029+
self.core.reset_cache(cache);
2030+
}
2031+
2032+
fn is_accelerated(&self) -> bool {
2033+
true
2034+
}
2035+
2036+
fn memory_usage(&self) -> usize {
2037+
let prefix_bytes: usize = self.prefixes.iter().map(|p| p.len()).sum();
2038+
self.core.memory_usage()
2039+
+ self.prefixes.len() * core::mem::size_of::<Box<[u8]>>()
2040+
+ prefix_bytes
2041+
}
2042+
2043+
#[cfg_attr(feature = "perf-inline", inline(always))]
2044+
fn search(&self, _cache: &mut Cache, input: &Input<'_>) -> Option<Match> {
2045+
self.try_fast_match(input)?;
2046+
Some(Match::new(PatternID::ZERO, 0..input.haystack().len()))
2047+
}
2048+
2049+
#[cfg_attr(feature = "perf-inline", inline(always))]
2050+
fn search_half(
2051+
&self,
2052+
_cache: &mut Cache,
2053+
input: &Input<'_>,
2054+
) -> Option<HalfMatch> {
2055+
self.try_fast_match(input)?;
2056+
Some(HalfMatch::new(PatternID::ZERO, input.haystack().len()))
2057+
}
2058+
2059+
#[cfg_attr(feature = "perf-inline", inline(always))]
2060+
fn is_match(&self, _cache: &mut Cache, input: &Input<'_>) -> bool {
2061+
self.try_fast_match(input).is_some()
2062+
}
2063+
2064+
#[cfg_attr(feature = "perf-inline", inline(always))]
2065+
fn search_slots(
2066+
&self,
2067+
_cache: &mut Cache,
2068+
input: &Input<'_>,
2069+
slots: &mut [Option<NonMaxUsize>],
2070+
) -> Option<PatternID> {
2071+
let (cap_start, cap_end) = self.try_fast_match(input)?;
2072+
let match_end = input.haystack().len();
2073+
if let Some(slot) = slots.get_mut(0) {
2074+
*slot = NonMaxUsize::new(0);
2075+
}
2076+
if let Some(slot) = slots.get_mut(1) {
2077+
*slot = NonMaxUsize::new(match_end);
2078+
}
2079+
if let Some(slot) = slots.get_mut(2) {
2080+
*slot = NonMaxUsize::new(cap_start);
2081+
}
2082+
if let Some(slot) = slots.get_mut(3) {
2083+
*slot = NonMaxUsize::new(cap_end);
2084+
}
2085+
Some(PatternID::ZERO)
2086+
}
2087+
2088+
fn which_overlapping_matches(
2089+
&self,
2090+
cache: &mut Cache,
2091+
input: &Input<'_>,
2092+
patset: &mut PatternSet,
2093+
) {
2094+
self.core.which_overlapping_matches(cache, input, patset)
2095+
}
2096+
}
2097+
2098+
/// Recognizes `^<prefix-set>([^X]+)X.*$` (default flags) and returns the
2099+
/// enumerated prefix set together with the terminator byte X.
2100+
fn try_recognize_prefix_capture(hir: &Hir) -> Option<(Box<[Box<[u8]>]>, u8)> {
2101+
let HirKind::Concat(parts) = hir.kind() else {
2102+
return None;
2103+
};
2104+
let mut iter = parts.iter();
2105+
2106+
// Multiline `(?m)` lowers `^` to `Look::StartLF`, which would break
2107+
// the byte-level fast path; require text-start specifically.
2108+
if !matches!(iter.next()?.kind(), HirKind::Look(Look::Start)) {
2109+
return None;
2110+
}
2111+
2112+
let mut prefixes: Vec<Vec<u8>> = vec![Vec::new()];
2113+
let capture = loop {
2114+
let part = iter.next()?;
2115+
if matches!(part.kind(), HirKind::Capture(_)) {
2116+
break part;
2117+
}
2118+
extend_prefix(&mut prefixes, part)?;
2119+
if prefixes.len() > MAX_PREFIX_VARIANTS {
2120+
return None;
2121+
}
2122+
};
2123+
2124+
let HirKind::Capture(cap) = capture.kind() else { unreachable!() };
2125+
if cap.index != 1 {
2126+
return None;
2127+
}
2128+
let terminator = capture_terminator_byte(&cap.sub)?;
2129+
2130+
let HirKind::Literal(Literal(lit)) = iter.next()?.kind() else {
2131+
return None;
2132+
};
2133+
if lit.as_ref() != [terminator] {
2134+
return None;
2135+
}
2136+
2137+
if !is_dot_star(iter.next()?) {
2138+
return None;
2139+
}
2140+
2141+
if !matches!(iter.next()?.kind(), HirKind::Look(Look::End)) {
2142+
return None;
2143+
}
2144+
if iter.next().is_some() {
2145+
return None;
2146+
}
2147+
2148+
prefixes.sort_unstable();
2149+
prefixes.dedup();
2150+
let mut prefixes: Vec<Box<[u8]>> =
2151+
prefixes.into_iter().map(Vec::into_boxed_slice).collect();
2152+
prefixes.sort_unstable_by_key(|p| core::cmp::Reverse(p.len()));
2153+
2154+
Some((prefixes.into_boxed_slice(), terminator))
2155+
}
2156+
2157+
/// Extend the accumulator with one prefix segment. Returns `None` if the
2158+
/// segment isn't a finite literal shape (literal / concat / alternation /
2159+
/// `?`-optional combination of those).
2160+
fn extend_prefix(variants: &mut Vec<Vec<u8>>, hir: &Hir) -> Option<()> {
2161+
match hir.kind() {
2162+
HirKind::Literal(Literal(bytes)) => {
2163+
for v in variants.iter_mut() {
2164+
v.extend_from_slice(bytes);
2165+
}
2166+
Some(())
2167+
}
2168+
HirKind::Concat(parts) => {
2169+
for part in parts {
2170+
extend_prefix(variants, part)?;
2171+
if variants.len() > MAX_PREFIX_VARIANTS {
2172+
return None;
2173+
}
2174+
}
2175+
Some(())
2176+
}
2177+
HirKind::Repetition(rep) if rep.min == 0 && rep.max == Some(1) => {
2178+
let mut with = variants.clone();
2179+
extend_prefix(&mut with, &rep.sub)?;
2180+
if variants.len() + with.len() > MAX_PREFIX_VARIANTS {
2181+
return None;
2182+
}
2183+
variants.extend(with);
2184+
Some(())
2185+
}
2186+
HirKind::Alternation(branches) => {
2187+
let base = core::mem::take(variants);
2188+
for branch in branches {
2189+
let mut local = base.clone();
2190+
extend_prefix(&mut local, branch)?;
2191+
if variants.len() + local.len() > MAX_PREFIX_VARIANTS {
2192+
return None;
2193+
}
2194+
variants.extend(local);
2195+
}
2196+
Some(())
2197+
}
2198+
_ => None,
2199+
}
2200+
}
2201+
2202+
/// Capture must be a greedy `[^X]+` over a single ASCII byte X.
2203+
fn capture_terminator_byte(hir: &Hir) -> Option<u8> {
2204+
let HirKind::Repetition(rep) = hir.kind() else {
2205+
return None;
2206+
};
2207+
if rep.min < 1 || rep.max.is_some() || !rep.greedy {
2208+
return None;
2209+
}
2210+
let HirKind::Class(class) = rep.sub.kind() else {
2211+
return None;
2212+
};
2213+
single_excluded_ascii_byte(class)
2214+
}
2215+
2216+
/// `.*` for default-flag regexes: any byte except `\n`, zero or more, greedy.
2217+
fn is_dot_star(hir: &Hir) -> bool {
2218+
let HirKind::Repetition(rep) = hir.kind() else {
2219+
return false;
2220+
};
2221+
if rep.min != 0 || rep.max.is_some() || !rep.greedy {
2222+
return false;
2223+
}
2224+
let HirKind::Class(class) = rep.sub.kind() else {
2225+
return false;
2226+
};
2227+
single_excluded_ascii_byte(class) == Some(b'\n')
2228+
}
2229+
2230+
/// Returns `Some(b)` iff `class` matches every codepoint or byte except a
2231+
/// single ASCII byte `b`. ASCII-only because the runtime matcher uses
2232+
/// `memchr` over byte slices.
2233+
fn single_excluded_ascii_byte(class: &Class) -> Option<u8> {
2234+
match class {
2235+
Class::Unicode(uc) => {
2236+
let ranges = uc.ranges();
2237+
if ranges.len() != 2 {
2238+
return None;
2239+
}
2240+
let (r0, r1) = (&ranges[0], &ranges[1]);
2241+
if (r0.start() as u32) != 0 || (r1.end() as u32) != 0x10FFFF {
2242+
return None;
2243+
}
2244+
let gap_start = r0.end() as u32 + 1;
2245+
let gap_end = r1.start() as u32 - 1;
2246+
if gap_start != gap_end || gap_start > 0x7F {
2247+
return None;
2248+
}
2249+
Some(gap_start as u8)
2250+
}
2251+
Class::Bytes(bc) => {
2252+
let ranges = bc.ranges();
2253+
if ranges.len() != 2 {
2254+
return None;
2255+
}
2256+
let (r0, r1) = (&ranges[0], &ranges[1]);
2257+
if r0.start() != 0 || r1.end() != 0xFF {
2258+
return None;
2259+
}
2260+
let gap_start = r0.end() as u16 + 1;
2261+
let gap_end = r1.start() as u16 - 1;
2262+
if gap_start != gap_end || gap_start > 0x7F {
2263+
return None;
2264+
}
2265+
Some(gap_start as u8)
2266+
}
2267+
}
2268+
}

0 commit comments

Comments
 (0)