diff --git a/regex-syntax/src/hir/interval.rs b/regex-syntax/src/hir/interval.rs index baec76d8f..7916e8c5c 100644 --- a/regex-syntax/src/hir/interval.rs +++ b/regex-syntax/src/hir/interval.rs @@ -344,7 +344,16 @@ impl IntervalSet { for i in 1..drain_end { let lower = self.ranges[i - 1].upper().increment(); let upper = self.ranges[i].lower().decrement(); - self.ranges.push(I::create(lower, upper)); + // Two canonical ranges are never adjacent in terms of their + // integer values, but they may still be adjacent in terms of + // their bounds. For example, the ranges `..=\u{D7FF}` and + // `\u{E000}..` have nothing between them, since the surrogate + // codepoints are not valid bounds. In that case, `lower > upper` + // and there is no range to add. (`create` would swap the bounds + // and produce a range that contains both.) + if lower <= upper { + self.ranges.push(I::create(lower, upper)); + } } if self.ranges[drain_end - 1].upper() < I::Bound::max_value() { let lower = self.ranges[drain_end - 1].upper().increment(); diff --git a/regex-syntax/src/hir/mod.rs b/regex-syntax/src/hir/mod.rs index f6818fb28..b6b67d917 100644 --- a/regex-syntax/src/hir/mod.rs +++ b/regex-syntax/src/hir/mod.rs @@ -3450,6 +3450,23 @@ mod tests { let cls = uclass(&[('\u{E001}', '\u{10FFFF}')]); let expected = uclass(&[('\x00', '\u{E000}')]); assert_eq!(expected, unegate(&cls)); + + // Ranges on either side of the surrogate codepoints have nothing + // between them, so negation must not produce a range for the gap. + // See: https://github.com/rust-lang/regex/issues/1397 + let cls = + uclass(&[('\u{D7FF}', '\u{D7FF}'), ('\u{E000}', '\u{E000}')]); + let expected = + uclass(&[('\x00', '\u{D7FE}'), ('\u{E001}', '\u{10FFFF}')]); + assert_eq!(expected, unegate(&cls)); + + let cls = uclass(&[('\x00', '\u{D7FF}'), ('\u{E000}', '\u{10FFFF}')]); + let expected = uclass(&[]); + assert_eq!(expected, unegate(&cls)); + + let cls = uclass(&[('a', '\u{D7FF}'), ('\u{E000}', '\u{E010}')]); + let expected = uclass(&[('\x00', '`'), ('\u{E011}', '\u{10FFFF}')]); + assert_eq!(expected, unegate(&cls)); } #[test] diff --git a/testdata/regression.toml b/testdata/regression.toml index 9956dd7be..77c31621e 100644 --- a/testdata/regression.toml +++ b/testdata/regression.toml @@ -983,3 +983,28 @@ name = 'improper-reverse-suffix-optimization' regex = '(\\N\{[^}]+})|([{}])' haystack = 'hiya \N{snowman} bye' matches = [[[5, 16], [5, 16], []]] + +# This is a regression test for negating a Unicode class with ranges on either +# side of the surrogate codepoints. Negation computed the gap between +# `..=\u{D7FF}` and `\u{E000}..` as an inverted range and then "corrected" it +# to `\u{D7FF}..=\u{E000}`, so both codepoints ended up matching their own +# negation. +# +# See: https://github.com/rust-lang/regex/issues/1397 +[[test]] +name = "negated-class-across-surrogate-gap" +regex = '^[^\x{D7FF}\x{E000}\n]$' +haystack = "\uD7FF" +matches = [] + +[[test]] +name = "negated-class-across-surrogate-gap-upper" +regex = '^[^\x{D7FF}\x{E000}\n]$' +haystack = "\uE000" +matches = [] + +[[test]] +name = "negated-class-across-surrogate-gap-neighbor" +regex = '^[^\x{D7FF}\x{E000}\n]$' +haystack = "\uD7FE" +matches = [[0, 3]]