|
1 | 1 | use fancy_regex::{Expr, LookAround}; |
2 | | -use regex_syntax::hir::{Class, Hir, HirKind}; |
| 2 | +use regex_syntax::hir::{Class, ClassUnicode, ClassUnicodeRange, Hir, HirKind}; |
3 | 3 |
|
4 | 4 | #[derive(Clone, Debug, PartialEq, Eq)] |
5 | 5 | pub(crate) struct ByteSet([bool; 256]); |
@@ -147,36 +147,37 @@ fn regex_syntax_start(expr: &Hir) -> Start { |
147 | 147 | } |
148 | 148 | } |
149 | 149 |
|
| 150 | +#[inline] |
| 151 | +fn first_utf8_byte(c: char) -> u8 { |
| 152 | + let mut buf = [0u8; 4]; |
| 153 | + c.encode_utf8(&mut buf).as_bytes()[0] |
| 154 | +} |
| 155 | + |
150 | 156 | /// TODO: what to do with low selective things like `\S+` or anything negated like `[^a]` that can match pretty much any chars? |
151 | 157 | fn fancy_start(expr: &Expr) -> Start { |
152 | 158 | match expr { |
153 | 159 | Expr::Empty | Expr::Assertion(_) | Expr::KeepOut | Expr::ContinueFromPreviousMatchEnd => { |
154 | 160 | Start::MaybeEmpty(ByteSet::new()) |
155 | 161 | } |
156 | 162 | Expr::Literal { val, casei } => { |
157 | | - let bytes = val.as_bytes(); |
158 | | - if let Some(first) = bytes.first() { |
159 | | - let mut set = ByteSet::new(); |
160 | | - if first.is_ascii() { |
161 | | - if *casei { |
162 | | - set.insert(first.to_ascii_uppercase()); |
163 | | - set.insert(first.to_ascii_lowercase()); |
164 | | - } else { |
165 | | - set.insert(*first); |
166 | | - } |
167 | | - Start::Definite(set) |
168 | | - } else { |
169 | | - // utf-8, we ignore if casei is true since we can't represent it i think? |
170 | | - if *casei { |
171 | | - Start::Bail |
172 | | - } else { |
173 | | - set.insert(*first); |
174 | | - Start::Definite(set) |
| 163 | + let Some(first) = val.chars().next() else { |
| 164 | + return Start::MaybeEmpty(ByteSet::new()); |
| 165 | + }; |
| 166 | + let mut set = ByteSet::new(); |
| 167 | + if *casei { |
| 168 | + let mut class = ClassUnicode::new([ClassUnicodeRange::new(first, first)]); |
| 169 | + class |
| 170 | + .try_case_fold_simple() |
| 171 | + .expect("regex-syntax missing unicode case?"); |
| 172 | + for range in class.ranges() { |
| 173 | + for c in range.start()..=range.end() { |
| 174 | + set.insert(first_utf8_byte(c)); |
175 | 175 | } |
176 | 176 | } |
177 | 177 | } else { |
178 | | - Start::MaybeEmpty(ByteSet::new()) |
| 178 | + set.insert(first_utf8_byte(first)); |
179 | 179 | } |
| 180 | + Start::Definite(set) |
180 | 181 | } |
181 | 182 | // We parse those with regex-syntax |
182 | 183 | Expr::Delegate { inner, casei } => { |
@@ -395,8 +396,16 @@ mod tests { |
395 | 396 | (r".", None), |
396 | 397 | (r".*x", None), |
397 | 398 | (r"\R", None), |
398 | | - (r"(?i)é", None), |
399 | 399 | (r"(", None), |
| 400 | + // Cases generated by Claude for (?i) |
| 401 | + // Unicode simple folding: (?i)s also matches ſ (U+017F, lead 0xC5) |
| 402 | + (r"(?i)sort", Some(vec![b'S', b's', 0xC5])), |
| 403 | + // and (?i)k also matches K (U+212A KELVIN SIGN, lead 0xE2) |
| 404 | + (r"(?i)kind", Some(vec![b'K', b'k', 0xE2])), |
| 405 | + // the closure works from the non-ASCII side too |
| 406 | + (r"(?i)ſ", Some(vec![b'S', b's', 0xC5])), |
| 407 | + // é folds to {é, É}, both with starting with 0xC3 |
| 408 | + (r"(?i)é", Some(vec![0xC3])), |
400 | 409 | ]; |
401 | 410 |
|
402 | 411 | for (input, expected) in inputs { |
|
0 commit comments