From 12f544cbf5889893899cb4e6c097caf8125eded9 Mon Sep 17 00:00:00 2001 From: Adrian Chaves Date: Tue, 22 Sep 2026 09:23:37 +0200 Subject: [PATCH] Distinguish *|E, |E and E namespace selectors --- cssselect/parser.py | 49 +++++++++++++++++++++++++++-------------- cssselect/xpath.py | 29 +++++++++++++++++------- docs/index.rst | 5 +++++ tests/test_cssselect.py | 12 ++++++---- 4 files changed, 67 insertions(+), 28 deletions(-) diff --git a/cssselect/parser.py b/cssselect/parser.py index ebc873c..90829ad 100644 --- a/cssselect/parser.py +++ b/cssselect/parser.py @@ -416,7 +416,10 @@ def __repr__(self) -> str: def canonical(self) -> str: attrib = _serialize_ident(self.attrib) if self.namespace: - attrib = f"{_serialize_ident(self.namespace)}|{attrib}" + namespace = ( + "*" if self.namespace == "*" else _serialize_ident(self.namespace) + ) + attrib = f"{namespace}|{attrib}" if self.operator == "exists": op = attrib @@ -437,7 +440,10 @@ class Element: """ Represents namespace|element - `None` is for the universal selector '*' + `None` is for the universal selector '*'. `namespace` is ``"*"`` + for the explicit namespace wildcard, e.g. ``*|div``, and `None` + both for an unprefixed element and for the explicit ``|div`` + no-namespace syntax. """ @@ -453,7 +459,10 @@ def __repr__(self) -> str: def canonical(self) -> str: element = _serialize_ident(self.element) if self.element else "*" if self.namespace: - element = f"{_serialize_ident(self.namespace)}|{element}" + namespace = ( + "*" if self.namespace == "*" else _serialize_ident(self.namespace) + ) + element = f"{namespace}|{element}" return element def specificity(self) -> tuple[int, int, int]: @@ -613,12 +622,12 @@ def parse_simple_selector( namespace = stream.next().value else: stream.next() - namespace = None + namespace = "*" if stream.peek() == ("DELIM", "|"): stream.next() element = stream.next_ident_or_star() else: - element = namespace + element = None if namespace == "*" else namespace namespace = None else: element = namespace = None @@ -822,23 +831,31 @@ def parse_simple_selector_arguments(stream: TokenStream) -> list[Tree]: def parse_attrib(selector: Tree, stream: TokenStream) -> Attrib: stream.skip_whitespace() - attrib = stream.next_ident_or_star() - if attrib is None and stream.peek() != ("DELIM", "|"): - raise SelectorSyntaxError(f"Expected '|', got {stream.peek()}") + attrib: str | None namespace: str | None op: str | None if stream.peek() == ("DELIM", "|"): + # The explicit "no namespace" syntax, e.g. [|attr]. stream.next() - if stream.peek() == ("DELIM", "="): - namespace = None + namespace = None + attrib = stream.next_ident() + op = None + else: + attrib = stream.next_ident_or_star() + if attrib is None and stream.peek() != ("DELIM", "|"): + raise SelectorSyntaxError(f"Expected '|', got {stream.peek()}") + if stream.peek() == ("DELIM", "|"): stream.next() - op = "|=" + if stream.peek() == ("DELIM", "="): + namespace = None + stream.next() + op = "|=" + else: + namespace = "*" if attrib is None else attrib + attrib = stream.next_ident() + op = None else: - namespace = attrib - attrib = stream.next_ident() - op = None - else: - namespace = op = None + namespace = op = None if op is None: stream.skip_whitespace() next_ = stream.next() diff --git a/cssselect/xpath.py b/cssselect/xpath.py index 9cfffb0..34e37d4 100644 --- a/cssselect/xpath.py +++ b/cssselect/xpath.py @@ -427,14 +427,20 @@ def xpath_attrib(self, selector: Attrib) -> XPathExpr: name = selector.attrib.lower() else: name = selector.attrib - safe = is_safe_name(name) - if selector.namespace: - name = f"{selector.namespace}:{name}" - safe = safe and is_safe_name(selector.namespace) - if safe: - attrib = "@" + name + if selector.namespace == "*": + # Namespace wildcard, e.g. "[*|href]": any namespace, including + # none. XPath 1.0 has no "*:name" name test for this, so match + # by local name instead. + attrib = f"attribute::*[local-name() = {self.xpath_literal(name)}]" else: - attrib = f"attribute::*[name() = {self.xpath_literal(name)}]" + safe = is_safe_name(name) + if selector.namespace: + name = f"{selector.namespace}:{name}" + safe = safe and is_safe_name(selector.namespace) + if safe: + attrib = "@" + name + else: + attrib = f"attribute::*[name() = {self.xpath_literal(name)}]" if selector.value is None: value = None elif self.lower_case_attribute_values: @@ -469,7 +475,14 @@ def xpath_element(self, selector: Element) -> XPathExpr: safe = bool(is_safe_name(element)) if self.lower_case_element_names: element = element.lower() - if selector.namespace: + if selector.namespace == "*": + # Namespace wildcard, e.g. "*|div": any namespace, + # including none. XPath 1.0 has no "*:name" name test + # for this, so match by local name instead. + xpath = self.xpathexpr_cls(element="*") + xpath.add_condition(f"local-name() = {self.xpath_literal(element)}") + return xpath + if selector.namespace and selector.namespace != "*": # Namespace prefixes are case-sensitive. # http://www.w3.org/TR/css3-namespace/#prefixes element = f"{selector.namespace}:{element}" diff --git a/docs/index.rst b/docs/index.rst index 07c7531..df0333d 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -179,4 +179,9 @@ In CSS you can use ``namespace-prefix|element``, similar to one-to-one. How prefixes are mapped to namespace URIs depends on the XPath implementation. +``*|element`` matches an element in any namespace, including none. +``|element`` and a bare ``element`` both match only elements with no +namespace. The same rules apply to attribute names, e.g. +``[*|attrib]`` and ``[|attrib]``. + .. include:: ../CHANGES diff --git a/tests/test_cssselect.py b/tests/test_cssselect.py index f121384..f567ba9 100644 --- a/tests/test_cssselect.py +++ b/tests/test_cssselect.py @@ -89,9 +89,9 @@ def parse_many(first: str, *others: str) -> list[str]: return result assert parse_many("*") == ["Element[*]"] - assert parse_many("*|*") == ["Element[*]"] - assert parse_many("*|foo") == ["Element[foo]"] - assert parse_many("|foo") == ["Element[foo]"] + assert parse_many("*|*") == ["Element[*|*]"] + assert parse_many("*|foo") == ["Element[*|foo]"] + assert parse_many("foo", "|foo") == ["Element[foo]"] assert parse_many("|*") == ["Element[*]"] assert parse_many("foo|*") == ["Element[foo|*]"] assert parse_many("foo|bar") == ["Element[foo|bar]"] @@ -656,9 +656,13 @@ def xpath(css: str) -> str: assert xpath("*") == "*" assert xpath("e") == "e" - assert xpath("*|e") == "e" + assert xpath("|e") == "e" + assert xpath("*|e") == "*[local-name() = 'e']" + assert xpath("*|*") == "*" assert xpath("e|f") == "e:f" assert xpath("e[foo]") == "e[@foo]" + assert xpath("e[|foo]") == "e[@foo]" + assert xpath("e[*|foo]") == "e[attribute::*[local-name() = 'foo']]" assert xpath("e[foo|bar]") == "e[@foo:bar]" assert xpath('e[foo="bar"]') == "e[@foo = 'bar']" assert xpath('e[foo~="bar"]') == (