diff --git a/rdflib/plugins/serializers/jsonld.py b/rdflib/plugins/serializers/jsonld.py index 8db0173ad..12b848d6b 100644 --- a/rdflib/plugins/serializers/jsonld.py +++ b/rdflib/plugins/serializers/jsonld.py @@ -417,6 +417,16 @@ def to_raw_value( return v def to_collection(self, graph: Graph, l_: Identifier): + """Return the members of the ``rdf:List`` headed by ``l_``, or None. + + None means the chain must not be rendered as ``@list``. Besides a + malformed or cyclic chain, that includes a chain any of whose cells is + the object of more than one statement: the ``@list`` form inlines the + whole chain at a single reference and writes its cells nowhere else, so + another statement pointing at one of those cells would be left + referring to a node the output never defines, and the serialization + would not round-trip. + """ if l_ != RDF.nil and not graph.value(l_, RDF.first): return None list_nodes: list[Optional[_ObjectType]] = [] @@ -426,6 +436,8 @@ def to_collection(self, graph: Graph, l_: Identifier): return list_nodes if isinstance(l_, URIRef): return None + if len(list(graph.subject_predicates(l_))) != 1: + return None first, rest = None, None for p, o in graph.predicate_objects(l_): if not first and p == RDF.first: diff --git a/rdflib/plugins/serializers/nt.py b/rdflib/plugins/serializers/nt.py index 583a0cfdf..d87ac1cc4 100644 --- a/rdflib/plugins/serializers/nt.py +++ b/rdflib/plugins/serializers/nt.py @@ -4,6 +4,7 @@ import warnings from typing import IO, TYPE_CHECKING, Any, Optional, Union +from rdflib.compare import to_canonical_graph from rdflib.graph import Graph from rdflib.serializer import Serializer from rdflib.term import Literal @@ -33,6 +34,16 @@ def serialize( encoding: Optional[str] = "utf-8", **kwargs: Any, ) -> None: + """Write the graph as N-Triples. + + When the optional parameter ``canon`` is set to ``True``, the graph is + canonicalized and the statements are written in sorted order, so + isomorphic graphs serialize to identical bytes. This mirrors the + ``canon`` parameter of the longturtle serializer, and carries the same + cost: canonicalization is not free, and the statements are held in + memory to be sorted. It is off by default, so ordinary serialization + still streams statement by statement. + """ if base is not None: warnings.warn("NTSerializer does not support base.") if encoding != "utf-8": @@ -41,6 +52,17 @@ def serialize( f"Given encoding was: {encoding}" ) + if kwargs.get("canon", False): + # Sort the rendered rows rather than the triples: two distinct + # terms can have the same str(), which would leave their relative + # order down to the store's iteration order, whereas identical + # rows mean identical statements. + for row in sorted( + _nt_row(triple) for triple in to_canonical_graph(self.store) + ): + stream.write(row.encode()) + return + for triple in self.store: stream.write(_nt_row(triple).encode()) diff --git a/rdflib/plugins/sparql/operators.py b/rdflib/plugins/sparql/operators.py index e9cb5a3f1..f17f91b0f 100644 --- a/rdflib/plugins/sparql/operators.py +++ b/rdflib/plugins/sparql/operators.py @@ -880,10 +880,36 @@ def RelationalExpression(e: Expr, ctx: Union[QueryContext, FrozenBindings]) -> L for x in other: try: - if x == expr: + # SPARQL 1.1 defines `IN` (17.4.1.9) as exactly + # (lhs = expr1) || (lhs = expr2) || ... + # and `NOT IN` (17.4.1.10) as its negation, so membership must use + # the same *value*-space equality as the `=` operator -- + # `Identifier.eq`, which falls back to `__eq__` for IRIs and blank + # nodes. Using `==` here applied strict *term* equality instead, + # under which the RDF 1.1-identical terms `"x"` and + # `"x"^^xsd:string` compared unequal, and numerics such as `1` and + # `1.0` never matched across xsd types. + match = expr.eq(x) if isinstance(expr, Identifier) else x == expr + # `Literal.eq` reports incomparable operands as `NotImplemented` + # rather than raising, and the `=` operator below turns exactly + # that into a `SPARQLError`. Do the same instead of evaluating it + # in a boolean context, which is truthy (so `IN` would wrongly + # match) and is a `TypeError` on Python 3.14. Recording it as an + # error -- rather than as "no match" -- is also what the spec + # requires: "Errors in comparisons cause the IN expression to + # raise an error if the RDF term being tested is not found + # elsewhere in the list", which the `error` accumulator below + # implements (`2 IN (1/0, 2)` is true; `2 IN (3, 1/0)` errors). + if match is NotImplemented: + raise SPARQLError("Error when comparing") + if match: return Literal(True ^ res) except SPARQLError as e: error = e + except TypeError as te: + # Mirrors the `=` path, which also normalises a comparison + # `TypeError` into a `SPARQLError`. + error = SPARQLError(*te.args) if not error: return Literal(False ^ res) else: diff --git a/test/jsonld/test_shared_list_cells.py b/test/jsonld/test_shared_list_cells.py new file mode 100644 index 000000000..10e5c66b3 --- /dev/null +++ b/test/jsonld/test_shared_list_cells.py @@ -0,0 +1,97 @@ +from rdflib import RDF, BNode, Graph, Literal, Namespace +from rdflib.compare import isomorphic +from rdflib.plugins.serializers.jsonld import Converter +from rdflib.plugins.shared.jsonld.context import Context + + +def _shared_tail_graph(): + """Two lists whose final cell is the same node, not a copy of it.""" + ns = Namespace("http://example.org/ns/") + g = Graph() + tail = BNode() + head = BNode() + g.add((tail, RDF.first, Literal("b"))) + g.add((tail, RDF.rest, RDF.nil)) + g.add((head, RDF.first, Literal("a"))) + g.add((head, RDF.rest, tail)) + g.add((ns.s1, ns.p, head)) + g.add((ns.s2, ns.p, tail)) + return g, ns, head, tail + + +def test_jsonld_shared_list_tail_round_trips(): + """ + A list cell pointed to from more than one place cannot be written as + ``@list``: that form inlines the whole chain at a single reference and + writes its cells nowhere else, so any other statement referring to one of + those cells is left pointing at a node the output never defines. + + Rendering each list independently instead emitted the shared cell once per + referring list, so the round-tripped graph gained triples: 6 in, 8 out, + non-isomorphic. + + Same defect class as ``test_turtle_shared_list_tail_round_trips`` and + ``test_longturtle_shared_list_tail_round_trips``; the JSON-LD converter has + its own chain walk and so needed the same reference check. + """ + g, _ns, _head, _tail = _shared_tail_graph() + + data = g.serialize(format="json-ld") + g2 = Graph() + g2.parse(data=data, format="json-ld") + + assert len(g2) == len(g) + assert isomorphic(g, g2) + # The chain is stated explicitly rather than inlined. ``"@list": []`` may + # still appear: that is ``rdf:nil``, a single IRI, not a list structure. + assert '"@list": [\n' not in data.replace('"@list": []', "") + assert len(list(g2.triples((None, RDF.first, None)))) == 2 + assert len(list(g2.triples((None, RDF.rest, None)))) == 2 + + +def test_to_collection_rejects_a_multiply_referenced_cell(): + """The decision itself, so the reason survives a refactor of the caller.""" + g, _ns, head, tail = _shared_tail_graph() + converter = Converter(Context(), False, None) + + assert converter.to_collection(g, head) is None + assert converter.to_collection(g, tail) is None + + +def test_a_privately_owned_list_is_still_written_as_a_list(): + """The fix must not stop ordinary lists from using ``@list``.""" + ns = Namespace("http://example.org/ns/") + g = Graph() + g.add((ns.s, ns.p, RDF.nil)) + collection = BNode() + second = BNode() + g.add((collection, RDF.first, Literal("a"))) + g.add((collection, RDF.rest, second)) + g.add((second, RDF.first, Literal("b"))) + g.add((second, RDF.rest, RDF.nil)) + g.add((ns.owner, ns.items, collection)) + + data = g.serialize(format="json-ld") + g2 = Graph() + g2.parse(data=data, format="json-ld") + + assert "@list" in data + assert isomorphic(g, g2) + + +def test_a_shared_list_head_round_trips(): + """Sharing the head, not an interior cell, must round-trip too.""" + ns = Namespace("http://example.org/ns/") + g = Graph() + head = BNode() + g.add((head, RDF.first, Literal("a"))) + g.add((head, RDF.rest, RDF.nil)) + g.add((ns.s1, ns.p, head)) + g.add((ns.s2, ns.p, head)) + + data = g.serialize(format="json-ld") + g2 = Graph() + g2.parse(data=data, format="json-ld") + + assert len(g2) == len(g) + assert isomorphic(g, g2) diff --git a/test/test_serializers/test_serializer_ntriples_canon.py b/test/test_serializers/test_serializer_ntriples_canon.py new file mode 100644 index 000000000..39b1cae7c --- /dev/null +++ b/test/test_serializers/test_serializer_ntriples_canon.py @@ -0,0 +1,119 @@ +import os +import subprocess +import sys + +from rdflib import BNode, Graph, Literal, Namespace +from rdflib.compare import isomorphic + +BUILD = """ +from rdflib import RDF, BNode, Graph, Literal, Namespace +ns = Namespace("http://example.org/ns/") +g = Graph() +hub = BNode() +for i in range(4): + g.add((ns[f"s{i}"], ns.ref, hub)) +g.add((hub, ns.value, Literal("hub"))) +cells = [BNode() for _ in range(4)] +for i, cell in enumerate(cells): + g.add((cell, RDF.first, Literal(f"v{i}"))) + g.add((cell, RDF.rest, cells[i + 1] if i + 1 < len(cells) else RDF.nil)) +g.add((ns.a, ns.items, cells[0])) +""" + + +def _graph() -> Graph: + namespace: dict = {} + exec(BUILD, namespace) + return namespace["g"] + + +def test_canon_produces_identical_bytes_across_processes(): + """``canon=True`` must survive a change of process, which is the point of it. + + Hash seeds rather than repeated calls in one process: the store's iteration + order is stable within a process, so repeating a call proves nothing. + """ + script = ( + BUILD + + """ +import sys +sys.stdout.write(g.serialize(format="nt", canon=True)) +""" + ) + outputs = { + subprocess.run( + [sys.executable, "-c", script], + capture_output=True, + text=True, + check=True, + env={**os.environ, "PYTHONHASHSEED": str(seed)}, + ).stdout + for seed in (0, 1, 2, 3, 5, 8) + } + assert len(outputs) == 1, f"canon=True gave {len(outputs)} different documents" + + +def test_canon_output_is_sorted(): + lines = [ + line + for line in _graph().serialize(format="nt", canon=True).splitlines() + if line + ] + assert lines == sorted(lines) + + +def test_canon_preserves_the_graph(): + g = _graph() + reparsed = Graph() + reparsed.parse(data=g.serialize(format="nt", canon=True), format="nt") + assert len(reparsed) == len(g) + assert isomorphic(reparsed, g) + + +def test_canon_is_off_by_default(): + """Ordinary serialization must be untouched. + + Default output keeps the graph's own blank-node labels, so it differs from + the canonicalized form -- which is what shows the parameter is doing + something rather than being ignored. + """ + g = _graph() + default = g.serialize(format="nt") + + reparsed = Graph() + reparsed.parse(data=default, format="nt") + assert isomorphic(reparsed, g) + assert len(default.splitlines()) == len(g) + assert default != g.serialize(format="nt", canon=True) + + +def test_an_unrecognised_keyword_is_still_ignored(): + """The parameter is opt-in via kwargs; nothing else may start failing.""" + g = _graph() + assert g.serialize(format="nt", something_else=True) == g.serialize(format="nt") + + +def test_canon_works_for_nt11(): + """NT11Serializer subclasses NTSerializer, so it inherits the parameter.""" + g = _graph() + reparsed = Graph() + reparsed.parse(data=g.serialize(format="nt11", canon=True), format="nt") + assert isomorphic(reparsed, g) + + +def test_canon_on_an_empty_graph(): + assert Graph().serialize(format="nt", canon=True) == "" + + +def test_canon_relabels_blank_nodes_independently_of_their_input_names(): + """Canonicalization is what makes the sort meaningful across runs.""" + ns = Namespace("http://example.org/ns/") + first, second = Graph(), Graph() + for graph, name in ((first, "aaa"), (second, "zzz")): + node = BNode(name) + graph.add((ns.s, ns.p, node)) + graph.add((node, ns.q, Literal("v"))) + + assert first.serialize(format="nt", canon=True) == second.serialize( + format="nt", canon=True + ) diff --git a/test/test_sparql/test_sparql.py b/test/test_sparql/test_sparql.py index 3f45717fa..f7a7a35b4 100644 --- a/test/test_sparql/test_sparql.py +++ b/test/test_sparql/test_sparql.py @@ -12,7 +12,7 @@ import rdflib.plugins.sparql.parser from rdflib import BNode, Dataset, Graph, Literal, URIRef from rdflib.compare import isomorphic -from rdflib.namespace import RDF, RDFS, Namespace +from rdflib.namespace import RDF, RDFS, XSD, Namespace from rdflib.plugins.sparql import prepareQuery, sparql from rdflib.plugins.sparql.algebra import translateQuery from rdflib.plugins.sparql.evaluate import evalPart @@ -998,3 +998,124 @@ def test_expand_unicode_escapes(arg: str, expected_result: str, expected_valid: else: with pytest.raises(ValueError, match="Invalid unicode code point"): _ = expandUnicodeEscapes(arg) + + +# `IN` / `NOT IN` membership. SPARQL 1.1 17.4.1.9 defines `IN` as exactly +# `(lhs = expr1) || (lhs = expr2) || ...` ("The test is done with `=` operator, which +# tests for the same value"), and 17.4.1.10 defines `NOT IN` as its negation. So +# membership must agree with `=`, and must follow the same error semantics: "Errors in +# comparisons cause the IN expression to raise an error if the RDF term being tested is +# not found elsewhere in the list". All six normative examples from those two sections +# are covered by the parametrisations below. +_EX = Namespace("urn:ex:") +_IN_PREFIXES = "PREFIX xsd: PREFIX ex: " + + +def _in_count(stored: Identifier, filter_expr: str) -> int: + """Rows surviving ``filter_expr``, counted by ITERATING the result. + + Deliberately not ``len(list(...))``: ``Result.__len__`` is consulted by + ``list()`` via ``PyObject_LengthHint``, which *clears* a ``TypeError`` raised + during evaluation, so a crashing filter would silently look like "no rows" + instead of failing the test. + """ + g = Graph() + g.add((URIRef("urn:ex:s"), URIRef("urn:ex:p"), stored)) + query = ( + f"{_IN_PREFIXES}SELECT ?s WHERE {{ ?s ?o . FILTER({filter_expr}) }}" + ) + return sum(1 for _ in g.query(query)) + + +@pytest.mark.parametrize( + "stored, listed, in_matches", + [ + # RDF 1.1: a simple literal and an xsd:string-typed literal are the SAME term. + (Literal("x"), '"x"', True), + (Literal("x", datatype=XSD.string), '"x"', True), + (Literal("x"), '"x"^^xsd:string', True), + (Literal("x", datatype=XSD.string), '"x"^^xsd:string', True), + # Numeric literals compare in value space across xsd types. + (Literal(1), "1.0", True), + (Literal(1.0), "1", True), + (Literal(1), "1", True), + (Literal(2), "1, 2, 3", True), + # Spec example (17.4.1.9): mixed IRI / string / numeric list. + (Literal(2), ', "str", 2.0', True), + # Non-literal terms: `Identifier.eq` falls back to `__eq__` for these. + (URIRef("urn:ex:x"), "", True), + (URIRef("urn:ex:x"), "", False), + (URIRef("urn:ex:x"), '"urn:ex:x"', False), + # Genuine non-matches must stay non-matches. + (Literal("x"), '"y"', False), + (Literal(1), "2", False), + # Incomparable operands: must not match, and must agree with `=`. + (Literal("x"), "1", False), + (Literal(1), '"x"', False), + (Literal("x", lang="en"), '"x"', False), + # Language tags are case-insensitive in RDF 1.1, so these ARE the same term. + (Literal("x", lang="en"), '"x"@EN', True), + (Literal("x"), '"x"@en', False), + ( + Literal("2024-01-01", datatype=XSD.date), + '"2024-01-01T00:00:00"^^xsd:dateTime', + False, + ), + (Literal(True), "1", False), + ], +) +def test_in_operator_uses_value_equality( + stored: Identifier, listed: str, in_matches: bool +) -> None: + """`IN` / `NOT IN` must compare in value space, exactly like `=` / `!=`. + + The membership test previously used Python `==`, i.e. strict *term* equality, under + which `"x"` and `"x"^^xsd:string` -- identical terms in RDF 1.1 -- compared unequal + and `1` never matched `1.0`, while the equivalent `=` comparisons were correct. + """ + assert _in_count(stored, f"?o IN ({listed})") == (1 if in_matches else 0) + assert _in_count(stored, f"?o NOT IN ({listed})") == (0 if in_matches else 1) + # The equivalence SPARQL 1.1 actually mandates, asserted directly. + assert _in_count(stored, f"?o IN ({listed})") == _in_count( + stored, " || ".join(f"?o = {item}" for item in listed.split(", ")) + ) + + +@pytest.mark.parametrize( + "stored, listed, expected_rows", + [ + # A match elsewhere in the list wins over an erroring entry, in either order. + # Spec: `2 IN (1/0, 2)` and `2 IN (2, 1/0)` are both true. + (Literal(2), "1/0, 2", 1), + (Literal(2), "2, 1/0", 1), + # Two literals of the same unknown datatype, different lexical forms: rdflib + # raises TypeError here, which must be recorded as an error and NOT abort the + # scan, so the later genuine match still wins. + (Literal("x", datatype=_EX.custom), '"y"^^ex:custom, "x"^^ex:custom', 1), + # Ill-typed literal compared to a well-typed one of the same datatype. + (Literal("abc", datatype=XSD.integer), '1, "abc"^^xsd:integer', 1), + # An erroring entry with NO match must not match -- in particular it must not + # be evaluated in a boolean context, where `NotImplemented` is truthy (and is + # a TypeError on Python 3.14), which would make `IN` match everything. + (Literal(2), "1/0", 0), + (Literal(2), "3, 1/0", 0), + # Empty list (the `RDF.nil` branch). + (Literal(2), "", 0), + ], +) +def test_in_operator_error_handling( + stored: Identifier, listed: str, expected_rows: int +) -> None: + """An erroring list entry must not suppress a match elsewhere, nor fabricate one. + + SPARQL 1.1 17.4.1.9: an error is only raised "if the RDF term being tested is not + found elsewhere in the list". The membership loop therefore has to record a failed + comparison and keep scanning -- a raised `TypeError` must not escape, and a + `NotImplemented` result must not be treated as a match. + """ + assert _in_count(stored, f"?o IN ({listed})") == expected_rows + # `NOT IN` is the exact complement whenever a match was found; when every entry + # errored and none matched, both forms yield no rows (the error is propagated and + # the FILTER drops the row), so the two are still consistent. + not_in_rows = _in_count(stored, f"?o NOT IN ({listed})") + assert not_in_rows == (0 if expected_rows else (1 if listed == "" else 0))