Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions rdflib/plugins/serializers/jsonld.py
Original file line number Diff line number Diff line change
Expand Up @@ -417,6 +417,16 @@ def to_raw_value(
return v

def to_collection(self, graph: Graph, l_: Identifier):
"""Return the members of the ``rdf:List`` headed by ``l_``, or None.

None means the chain must not be rendered as ``@list``. Besides a
malformed or cyclic chain, that includes a chain any of whose cells is
the object of more than one statement: the ``@list`` form inlines the
whole chain at a single reference and writes its cells nowhere else, so
another statement pointing at one of those cells would be left
referring to a node the output never defines, and the serialization
would not round-trip.
"""
if l_ != RDF.nil and not graph.value(l_, RDF.first):
return None
list_nodes: list[Optional[_ObjectType]] = []
Expand All @@ -426,6 +436,8 @@ def to_collection(self, graph: Graph, l_: Identifier):
return list_nodes
if isinstance(l_, URIRef):
return None
if len(list(graph.subject_predicates(l_))) != 1:
return None
first, rest = None, None
for p, o in graph.predicate_objects(l_):
if not first and p == RDF.first:
Expand Down
22 changes: 22 additions & 0 deletions rdflib/plugins/serializers/nt.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
import warnings
from typing import IO, TYPE_CHECKING, Any, Optional, Union

from rdflib.compare import to_canonical_graph
from rdflib.graph import Graph
from rdflib.serializer import Serializer
from rdflib.term import Literal
Expand Down Expand Up @@ -33,6 +34,16 @@ def serialize(
encoding: Optional[str] = "utf-8",
**kwargs: Any,
) -> None:
"""Write the graph as N-Triples.

When the optional parameter ``canon`` is set to ``True``, the graph is
canonicalized and the statements are written in sorted order, so
isomorphic graphs serialize to identical bytes. This mirrors the
``canon`` parameter of the longturtle serializer, and carries the same
cost: canonicalization is not free, and the statements are held in
memory to be sorted. It is off by default, so ordinary serialization
still streams statement by statement.
"""
if base is not None:
warnings.warn("NTSerializer does not support base.")
if encoding != "utf-8":
Expand All @@ -41,6 +52,17 @@ def serialize(
f"Given encoding was: {encoding}"
)

if kwargs.get("canon", False):
# Sort the rendered rows rather than the triples: two distinct
# terms can have the same str(), which would leave their relative
# order down to the store's iteration order, whereas identical
# rows mean identical statements.
for row in sorted(
_nt_row(triple) for triple in to_canonical_graph(self.store)
):
stream.write(row.encode())
return

for triple in self.store:
stream.write(_nt_row(triple).encode())

Expand Down
28 changes: 27 additions & 1 deletion rdflib/plugins/sparql/operators.py
Original file line number Diff line number Diff line change
Expand Up @@ -880,10 +880,36 @@ def RelationalExpression(e: Expr, ctx: Union[QueryContext, FrozenBindings]) -> L

for x in other:
try:
if x == expr:
# SPARQL 1.1 defines `IN` (17.4.1.9) as exactly
# (lhs = expr1) || (lhs = expr2) || ...
# and `NOT IN` (17.4.1.10) as its negation, so membership must use
# the same *value*-space equality as the `=` operator --
# `Identifier.eq`, which falls back to `__eq__` for IRIs and blank
# nodes. Using `==` here applied strict *term* equality instead,
# under which the RDF 1.1-identical terms `"x"` and
# `"x"^^xsd:string` compared unequal, and numerics such as `1` and
# `1.0` never matched across xsd types.
match = expr.eq(x) if isinstance(expr, Identifier) else x == expr
# `Literal.eq` reports incomparable operands as `NotImplemented`
# rather than raising, and the `=` operator below turns exactly
# that into a `SPARQLError`. Do the same instead of evaluating it
# in a boolean context, which is truthy (so `IN` would wrongly
# match) and is a `TypeError` on Python 3.14. Recording it as an
# error -- rather than as "no match" -- is also what the spec
# requires: "Errors in comparisons cause the IN expression to
# raise an error if the RDF term being tested is not found
# elsewhere in the list", which the `error` accumulator below
# implements (`2 IN (1/0, 2)` is true; `2 IN (3, 1/0)` errors).
if match is NotImplemented:
raise SPARQLError("Error when comparing")
if match:
return Literal(True ^ res)
except SPARQLError as e:
error = e
except TypeError as te:
# Mirrors the `=` path, which also normalises a comparison
# `TypeError` into a `SPARQLError`.
error = SPARQLError(*te.args)
if not error:
return Literal(False ^ res)
else:
Expand Down
97 changes: 97 additions & 0 deletions test/jsonld/test_shared_list_cells.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,97 @@
from rdflib import RDF, BNode, Graph, Literal, Namespace
from rdflib.compare import isomorphic
from rdflib.plugins.serializers.jsonld import Converter
from rdflib.plugins.shared.jsonld.context import Context


def _shared_tail_graph():
"""Two lists whose final cell is the same node, not a copy of it."""
ns = Namespace("http://example.org/ns/")
g = Graph()
tail = BNode()
head = BNode()
g.add((tail, RDF.first, Literal("b")))
g.add((tail, RDF.rest, RDF.nil))
g.add((head, RDF.first, Literal("a")))
g.add((head, RDF.rest, tail))
g.add((ns.s1, ns.p, head))
g.add((ns.s2, ns.p, tail))
return g, ns, head, tail


def test_jsonld_shared_list_tail_round_trips():
"""
A list cell pointed to from more than one place cannot be written as
``@list``: that form inlines the whole chain at a single reference and
writes its cells nowhere else, so any other statement referring to one of
those cells is left pointing at a node the output never defines.

Rendering each list independently instead emitted the shared cell once per
referring list, so the round-tripped graph gained triples: 6 in, 8 out,
non-isomorphic.

Same defect class as ``test_turtle_shared_list_tail_round_trips`` and
``test_longturtle_shared_list_tail_round_trips``; the JSON-LD converter has
its own chain walk and so needed the same reference check.
"""
g, _ns, _head, _tail = _shared_tail_graph()

data = g.serialize(format="json-ld")
g2 = Graph()
g2.parse(data=data, format="json-ld")

assert len(g2) == len(g)
assert isomorphic(g, g2)
# The chain is stated explicitly rather than inlined. ``"@list": []`` may
# still appear: that is ``rdf:nil``, a single IRI, not a list structure.
assert '"@list": [\n' not in data.replace('"@list": []', "")
assert len(list(g2.triples((None, RDF.first, None)))) == 2
assert len(list(g2.triples((None, RDF.rest, None)))) == 2


def test_to_collection_rejects_a_multiply_referenced_cell():
"""The decision itself, so the reason survives a refactor of the caller."""
g, _ns, head, tail = _shared_tail_graph()
converter = Converter(Context(), False, None)

assert converter.to_collection(g, head) is None
assert converter.to_collection(g, tail) is None


def test_a_privately_owned_list_is_still_written_as_a_list():
"""The fix must not stop ordinary lists from using ``@list``."""
ns = Namespace("http://example.org/ns/")
g = Graph()
g.add((ns.s, ns.p, RDF.nil))
collection = BNode()
second = BNode()
g.add((collection, RDF.first, Literal("a")))
g.add((collection, RDF.rest, second))
g.add((second, RDF.first, Literal("b")))
g.add((second, RDF.rest, RDF.nil))
g.add((ns.owner, ns.items, collection))

data = g.serialize(format="json-ld")
g2 = Graph()
g2.parse(data=data, format="json-ld")

assert "@list" in data
assert isomorphic(g, g2)


def test_a_shared_list_head_round_trips():
"""Sharing the head, not an interior cell, must round-trip too."""
ns = Namespace("http://example.org/ns/")
g = Graph()
head = BNode()
g.add((head, RDF.first, Literal("a")))
g.add((head, RDF.rest, RDF.nil))
g.add((ns.s1, ns.p, head))
g.add((ns.s2, ns.p, head))

data = g.serialize(format="json-ld")
g2 = Graph()
g2.parse(data=data, format="json-ld")

assert len(g2) == len(g)
assert isomorphic(g, g2)
119 changes: 119 additions & 0 deletions test/test_serializers/test_serializer_ntriples_canon.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,119 @@
import os
import subprocess
import sys

from rdflib import BNode, Graph, Literal, Namespace
from rdflib.compare import isomorphic

BUILD = """
from rdflib import RDF, BNode, Graph, Literal, Namespace
ns = Namespace("http://example.org/ns/")
g = Graph()
hub = BNode()
for i in range(4):
g.add((ns[f"s{i}"], ns.ref, hub))
g.add((hub, ns.value, Literal("hub")))
cells = [BNode() for _ in range(4)]
for i, cell in enumerate(cells):
g.add((cell, RDF.first, Literal(f"v{i}")))
g.add((cell, RDF.rest, cells[i + 1] if i + 1 < len(cells) else RDF.nil))
g.add((ns.a, ns.items, cells[0]))
"""


def _graph() -> Graph:
namespace: dict = {}
exec(BUILD, namespace)
return namespace["g"]


def test_canon_produces_identical_bytes_across_processes():
"""``canon=True`` must survive a change of process, which is the point of it.

Hash seeds rather than repeated calls in one process: the store's iteration
order is stable within a process, so repeating a call proves nothing.
"""
script = (
BUILD
+ """
import sys
sys.stdout.write(g.serialize(format="nt", canon=True))
"""
)
outputs = {
subprocess.run(
[sys.executable, "-c", script],
capture_output=True,
text=True,
check=True,
env={**os.environ, "PYTHONHASHSEED": str(seed)},
).stdout
for seed in (0, 1, 2, 3, 5, 8)
}
assert len(outputs) == 1, f"canon=True gave {len(outputs)} different documents"


def test_canon_output_is_sorted():
lines = [
line
for line in _graph().serialize(format="nt", canon=True).splitlines()
if line
]
assert lines == sorted(lines)


def test_canon_preserves_the_graph():
g = _graph()
reparsed = Graph()
reparsed.parse(data=g.serialize(format="nt", canon=True), format="nt")
assert len(reparsed) == len(g)
assert isomorphic(reparsed, g)


def test_canon_is_off_by_default():
"""Ordinary serialization must be untouched.

Default output keeps the graph's own blank-node labels, so it differs from
the canonicalized form -- which is what shows the parameter is doing
something rather than being ignored.
"""
g = _graph()
default = g.serialize(format="nt")

reparsed = Graph()
reparsed.parse(data=default, format="nt")
assert isomorphic(reparsed, g)
assert len(default.splitlines()) == len(g)
assert default != g.serialize(format="nt", canon=True)


def test_an_unrecognised_keyword_is_still_ignored():
"""The parameter is opt-in via kwargs; nothing else may start failing."""
g = _graph()
assert g.serialize(format="nt", something_else=True) == g.serialize(format="nt")


def test_canon_works_for_nt11():
"""NT11Serializer subclasses NTSerializer, so it inherits the parameter."""
g = _graph()
reparsed = Graph()
reparsed.parse(data=g.serialize(format="nt11", canon=True), format="nt")
assert isomorphic(reparsed, g)


def test_canon_on_an_empty_graph():
assert Graph().serialize(format="nt", canon=True) == ""


def test_canon_relabels_blank_nodes_independently_of_their_input_names():
"""Canonicalization is what makes the sort meaningful across runs."""
ns = Namespace("http://example.org/ns/")
first, second = Graph(), Graph()
for graph, name in ((first, "aaa"), (second, "zzz")):
node = BNode(name)
graph.add((ns.s, ns.p, node))
graph.add((node, ns.q, Literal("v")))

assert first.serialize(format="nt", canon=True) == second.serialize(
format="nt", canon=True
)
Loading