diff --git a/cssselect/parser.py b/cssselect/parser.py index 898b2bc..1560823 100644 --- a/cssselect/parser.py +++ b/cssselect/parser.py @@ -997,7 +997,7 @@ def _compile(pattern: str) -> MatchFunc: def _replace_unicode(match: re.Match[str]) -> str: codepoint = int(match.group(1), 16) - if codepoint > sys.maxunicode or 0xD800 <= codepoint <= 0xDFFF: + if codepoint == 0 or codepoint > sys.maxunicode or 0xD800 <= codepoint <= 0xDFFF: codepoint = 0xFFFD return chr(codepoint) diff --git a/tests/test_cssselect.py b/tests/test_cssselect.py index 5b98711..661f663 100644 --- a/tests/test_cssselect.py +++ b/tests/test_cssselect.py @@ -1076,6 +1076,11 @@ def test_unicode_escapes(self) -> None: ) # A code point beyond the Unicode range is replaced with U+FFFD. assert css_to_xpath(r"\110000") == ("descendant-or-self::*[name() = '\ufffd']") + # A null code point is likewise replaced with U+FFFD, so no NUL + # character can leak into the generated XPath string. + assert css_to_xpath(r"*[aval='\0 ']") == ( + "descendant-or-self::*[@aval = '\ufffd']" + ) # unescape_ident() resolves both unicode and simple escapes. assert unescape_ident(r"\41 B") == "AB" assert unescape_ident(r"\-foo") == "-foo"