From d485689e767751f8a8ac53892ec7ede767873e31 Mon Sep 17 00:00:00 2001 From: Siraj saheb Date: Wed, 16 Sep 2026 13:37:56 +0530 Subject: [PATCH 1/2] replace raw null and surrogate code points during tokenization --- cssselect/parser.py | 11 +++++++++++ tests/test_cssselect.py | 19 +++++++++++++++++++ 2 files changed, 30 insertions(+) diff --git a/cssselect/parser.py b/cssselect/parser.py index 1560823..e7ea02e 100644 --- a/cssselect/parser.py +++ b/cssselect/parser.py @@ -991,6 +991,11 @@ def _compile(pattern: str) -> MatchFunc: _sub_newline_escape = re.compile(r"\\(?:\n|\r\n|\r|\f)").sub _sub_string_control_char = re.compile(r"[\x00-\x1f\x7f]").sub +# CSS Syntax Level 3, §3.3 "Preprocessing the input stream": a raw U+0000 or +# surrogate code point is replaced with U+FFFD. This mirrors what +# _replace_unicode does for the escaped forms (§4.3.7). +_sub_invalid_input_char = re.compile("[\x00\ud800-\udfff]").sub + # Same as r'\1', but faster on CPython _replace_simple = operator.methodcaller("group", 1) @@ -1048,6 +1053,12 @@ def _serialize_ident(value: str) -> str: def tokenize(s: str) -> Iterator[Token]: + # Preprocess the input stream (CSS Syntax Level 3, §3.3): fold a raw + # U+0000 or surrogate code point to U+FFFD before tokenizing. The + # substitution is length-preserving, so token positions are unaffected. + # Without this such a code point survives inside a string token and leaks + # into the generated XPath, which lxml then refuses to compile. + s = _sub_invalid_input_char("\N{REPLACEMENT CHARACTER}", s) pos = 0 len_s = len(s) while pos < len_s: diff --git a/tests/test_cssselect.py b/tests/test_cssselect.py index 661f663..0f5436e 100644 --- a/tests/test_cssselect.py +++ b/tests/test_cssselect.py @@ -1085,6 +1085,25 @@ def test_unicode_escapes(self) -> None: assert unescape_ident(r"\41 B") == "AB" assert unescape_ident(r"\-foo") == "-foo" + def test_input_preprocessing(self) -> None: + # CSS Syntax §3.3: a raw U+0000 or surrogate code point (not an + # escape) is folded to U+FFFD before tokenizing, so it cannot leak + # into the generated XPath. Before this, lxml rejected the result + # with "no NULL bytes or control characters" / "surrogates not + # allowed". + css_to_xpath = GenericTranslator().css_to_xpath + assert css_to_xpath('*[aval="x\x00y"]') == ( + "descendant-or-self::*[@aval = 'x�y']" + ) + assert css_to_xpath('*[aval="x\ud800y"]') == ( + "descendant-or-self::*[@aval = 'x�y']" + ) + assert css_to_xpath(':contains("x\udfffy")') == ( + "descendant-or-self::*[contains(., 'x�y')]" + ) + # A raw NUL in an identifier becomes a valid U+FFFD name character. + assert str(next(tokenize("foo\x00bar"))) == "" + def test_xpath_pseudo_elements(self) -> None: class CustomTranslator(GenericTranslator): def xpath_pseudo_element( From 94b2bf7801e8b3f0ed9a363e24262296ab3b4972 Mon Sep 17 00:00:00 2001 From: Siraj saheb Date: Wed, 16 Sep 2026 16:25:14 +0530 Subject: [PATCH 2/2] =?UTF-8?q?trim=20=C2=A73.3=20preprocessing=20comments?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- cssselect/parser.py | 10 ++-------- tests/test_cssselect.py | 5 +---- 2 files changed, 3 insertions(+), 12 deletions(-) diff --git a/cssselect/parser.py b/cssselect/parser.py index e7ea02e..ebc873c 100644 --- a/cssselect/parser.py +++ b/cssselect/parser.py @@ -991,9 +991,7 @@ def _compile(pattern: str) -> MatchFunc: _sub_newline_escape = re.compile(r"\\(?:\n|\r\n|\r|\f)").sub _sub_string_control_char = re.compile(r"[\x00-\x1f\x7f]").sub -# CSS Syntax Level 3, §3.3 "Preprocessing the input stream": a raw U+0000 or -# surrogate code point is replaced with U+FFFD. This mirrors what -# _replace_unicode does for the escaped forms (§4.3.7). +# CSS Syntax Level 3, §3.3: fold a raw U+0000 or surrogate code point to U+FFFD. _sub_invalid_input_char = re.compile("[\x00\ud800-\udfff]").sub # Same as r'\1', but faster on CPython @@ -1053,11 +1051,7 @@ def _serialize_ident(value: str) -> str: def tokenize(s: str) -> Iterator[Token]: - # Preprocess the input stream (CSS Syntax Level 3, §3.3): fold a raw - # U+0000 or surrogate code point to U+FFFD before tokenizing. The - # substitution is length-preserving, so token positions are unaffected. - # Without this such a code point survives inside a string token and leaks - # into the generated XPath, which lxml then refuses to compile. + # Preprocess the input stream (§3.3); the substitution is length-preserving. s = _sub_invalid_input_char("\N{REPLACEMENT CHARACTER}", s) pos = 0 len_s = len(s) diff --git a/tests/test_cssselect.py b/tests/test_cssselect.py index 0f5436e..f121384 100644 --- a/tests/test_cssselect.py +++ b/tests/test_cssselect.py @@ -1087,10 +1087,7 @@ def test_unicode_escapes(self) -> None: def test_input_preprocessing(self) -> None: # CSS Syntax §3.3: a raw U+0000 or surrogate code point (not an - # escape) is folded to U+FFFD before tokenizing, so it cannot leak - # into the generated XPath. Before this, lxml rejected the result - # with "no NULL bytes or control characters" / "surrogates not - # allowed". + # escape) is folded to U+FFFD before tokenizing. css_to_xpath = GenericTranslator().css_to_xpath assert css_to_xpath('*[aval="x\x00y"]') == ( "descendant-or-self::*[@aval = 'x�y']"