diff --git a/semantica/export/rdf_exporter.py b/semantica/export/rdf_exporter.py
index 63eac4fe..ed1828bb 100644
--- a/semantica/export/rdf_exporter.py
+++ b/semantica/export/rdf_exporter.py
@@ -30,6 +30,7 @@ License: MIT
"""
from pathlib import Path
+from decimal import Decimal, InvalidOperation
from typing import Any, Dict, List, Optional, Set, Union
from ..utils.exceptions import ProcessingError, ValidationError
@@ -49,6 +50,58 @@ DEFAULT_ENTITY_TYPE = f"{SEMANTICA_NS}Entity"
#: Written when a relationship carries no type of its own. Same reasoning.
DEFAULT_RELATION_TYPE = f"{SEMANTICA_NS}related_to"
+#: The one datatype every serializer writes confidence in.
+#:
+#: The four paths used to disagree: Turtle wrote the value bare, which the
+#: Turtle grammar reads as xsd:decimal, N-Triples typed it xsd:float, RDF/XML
+#: emitted a plain literal with no datatype, and JSON-LD emitted a native
+#: number, which becomes xsd:double. Those are four distinct RDF terms for one
+#: value (issue #1100).
+#:
+#: xsd:decimal is the choice because it is what the Turtle path already
+#: produced, so the most used output is unchanged, and because it is exact:
+#: xsd:float is 32 bit binary, and cannot represent 0.9 or 0.95 at all.
+CONFIDENCE_DATATYPE = "http://www.w3.org/2001/XMLSchema#decimal"
+
+
+def normalize_confidence(value: Any) -> Optional[str]:
+ """
+ Return the canonical xsd:decimal lexical form of a confidence value.
+
+ Returns None when the value cannot be a decimal, so callers omit the triple
+ rather than writing something the vocabulary contradicts. The Turtle path
+ used to interpolate the raw value, so a confidence of "high" produced
+ `semantica:confidence high .` and made the whole document unparseable
+ (issue #1102).
+
+ Booleans are rejected. `bool` is a subclass of `int` in Python, so True
+ would otherwise silently become a confidence of 1.
+ """
+ if value is None or isinstance(value, bool):
+ return None
+ if isinstance(value, str):
+ value = value.strip()
+ if not value:
+ return None
+ if not isinstance(value, (int, float, str, Decimal)):
+ return None
+
+ try:
+ decimal_value = Decimal(str(value))
+ except (InvalidOperation, ValueError, TypeError):
+ return None
+
+ # NaN and the infinities are Decimal values with no xsd:decimal form.
+ if not decimal_value.is_finite():
+ return None
+
+ # `str(Decimal("0.00001"))` gives "0.00001", but a float that has already
+ # been through repr can arrive as "1e-05", which xsd:decimal does not allow.
+ formatted = format(decimal_value, "f")
+ if "." in formatted:
+ formatted = formatted.rstrip("0").rstrip(".") or "0"
+ return formatted
+
def mint_entity_iri(text: str) -> str:
"""Mint a stable IRI for an entity that arrived without an id.
@@ -395,11 +448,20 @@ class RDFSerializer:
entity_type = entity.get("type", DEFAULT_ENTITY_TYPE)
text = entity.get("text") or entity.get("label", "")
- confidence = entity.get("confidence", 1.0)
+ confidence = normalize_confidence(entity.get("confidence", 1.0))
lines.append(f"<{entity_id}> a <{entity_type}> ;")
- lines.append(f' semantica:text "{text}" ;')
- lines.append(f" semantica:confidence {confidence} .")
+ if confidence is None:
+ self.logger.warning(
+ f"Entity {entity_id} has a confidence that is not a number "
+ f"({entity.get('confidence')!r}), so no confidence is written"
+ )
+ lines.append(f' semantica:text "{text}" .')
+ else:
+ lines.append(f' semantica:text "{text}" ;')
+ lines.append(
+ f' semantica:confidence "{confidence}"^^<{CONFIDENCE_DATATYPE}> .'
+ )
lines.append("")
# Convert relationships to RDF triplets
@@ -527,15 +589,22 @@ class RDFSerializer:
entity_type = entity.get("type", DEFAULT_ENTITY_TYPE)
text = entity.get("text") or entity.get("label", "")
- confidence = entity.get("confidence", 1.0)
+ confidence = normalize_confidence(entity.get("confidence", 1.0))
# RDF/XML syntax: rdf:Description with rdf:about
lines.append(f' ')
lines.append(f' ')
lines.append(f" {text}")
- lines.append(
- f" {confidence}"
- )
+ if confidence is None:
+ self.logger.warning(
+ f"Entity {entity_id} has a confidence that is not a number "
+ f"({entity.get('confidence')!r}), so no confidence is written"
+ )
+ else:
+ lines.append(
+ f' '
+ f"{confidence}"
+ )
lines.append(" ")
lines.append("")
@@ -606,14 +675,25 @@ class RDFSerializer:
entity_text = entity.get("text", "")
entity_id = f"semantica:entity/{entity_text}"
- jsonld["@graph"].append(
- {
- "@id": entity_id,
- "@type": entity.get("type", "semantica:Entity"),
- "semantica:text": entity.get("text") or entity.get("label", ""),
- "semantica:confidence": entity.get("confidence", 1.0),
+ node = {
+ "@id": entity_id,
+ "@type": entity.get("type", "semantica:Entity"),
+ "semantica:text": entity.get("text") or entity.get("label", ""),
+ }
+ confidence = normalize_confidence(entity.get("confidence", 1.0))
+ if confidence is None:
+ self.logger.warning(
+ f"Entity {entity_id} has a confidence that is not a number "
+ f"({entity.get('confidence')!r}), so no confidence is written"
+ )
+ else:
+ # A native JSON number becomes xsd:double once expanded, so the
+ # value is written as a typed literal instead.
+ node["semantica:confidence"] = {
+ "@value": confidence,
+ "@type": CONFIDENCE_DATATYPE,
}
- )
+ jsonld["@graph"].append(node)
# Convert relationships to JSON-LD
relationships = rdf_data.get("relationships", [])
@@ -697,11 +777,20 @@ class RDFSerializer:
f'{subject} {expand_uri("semantica:text")} "{safe_text}" .'
)
- # Confidence property
- confidence = entity.get("confidence")
- if confidence is not None:
+ # Confidence property. The default matches the other serializers,
+ # which have always written one; omitting it here was half of why
+ # Turtle and N-Triples of one KG were different graphs (#1100).
+ raw_confidence = entity.get("confidence", 1.0)
+ confidence = normalize_confidence(raw_confidence)
+ if confidence is None:
+ self.logger.warning(
+ f"Entity {entity.get('id')} has a confidence that is not a "
+ f"number ({raw_confidence!r}), so no confidence is written"
+ )
+ else:
lines.append(
- f'{subject} {expand_uri("semantica:confidence")} "{confidence}"^^ .'
+ f'{subject} {expand_uri("semantica:confidence")} '
+ f'"{confidence}"^^<{CONFIDENCE_DATATYPE}> .'
)
# Convert relationships
diff --git a/semantica/ontology/vocabulary/semantica-ns.ttl b/semantica/ontology/vocabulary/semantica-ns.ttl
index c6843a32..9c461937 100644
--- a/semantica/ontology/vocabulary/semantica-ns.ttl
+++ b/semantica/ontology/vocabulary/semantica-ns.ttl
@@ -53,16 +53,17 @@ this IRI.""" ;
sem:confidence a owl:DatatypeProperty ;
rdfs:label "confidence" ;
+ rdfs:range xsd:decimal ;
rdfs:comment """Extractor confidence in the assertion, on the unit interval.
Emitted for both entities and relationships, so the domain is left open rather
than tied to sem:Entity.
-No rdfs:range is declared, deliberately. The Turtle serializer writes the value
-bare, which the Turtle grammar reads as xsd:decimal, while the N-Triples
-serializer types it xsd:float explicitly, and those two datatypes are disjoint.
-Declaring either one would make the vocabulary contradict one of the exporters.
-Issue #1100 tracks the disagreement; a range belongs here once the serializers
-agree on one.""" ;
+The range was left undeclared when this vocabulary first shipped, because the
+four serializers disagreed: Turtle wrote the value bare, which the Turtle
+grammar reads as xsd:decimal, N-Triples typed it xsd:float, RDF/XML wrote a
+plain literal, and JSON-LD wrote a native number, which expands to xsd:double.
+Declaring any one of them would have contradicted three exporters. Issue #1100
+settled that on xsd:decimal, which every serializer now writes.""" ;
rdfs:isDefinedBy .
sem:metadata a owl:AnnotationProperty ;
diff --git a/tests/export/test_confidence_literal_typing.py b/tests/export/test_confidence_literal_typing.py
new file mode 100644
index 00000000..082f947c
--- /dev/null
+++ b/tests/export/test_confidence_literal_typing.py
@@ -0,0 +1,177 @@
+"""
+Regression tests for #1100 and #1102.
+
+#1100: the four RDF serializers rendered the same confidence value four
+different ways. Turtle wrote it bare, which the Turtle grammar reads as
+xsd:decimal. N-Triples typed it xsd:float. RDF/XML wrote a plain literal with
+no datatype at all. JSON-LD emitted a native JSON number, which becomes
+xsd:double. Those are four distinct RDF terms, so a FILTER matches at most one
+of them and merging two exports of one graph yields two confidence values for
+the same entity.
+
+#1102: the Turtle path interpolated the value with no type check, so a
+non-numeric confidence produced `semantica:confidence high .`, which is not
+parseable Turtle. One bad field made the whole export unreadable.
+"""
+
+import json
+
+import pytest
+
+rdflib = pytest.importorskip("rdflib")
+from rdflib import Graph, URIRef # noqa: E402
+from rdflib.compare import isomorphic # noqa: E402
+
+from semantica.export.rdf_exporter import RDFSerializer # noqa: E402
+
+NS = "https://semantica.dev/ns#"
+CONFIDENCE = URIRef(NS + "confidence")
+XSD_DECIMAL = URIRef("http://www.w3.org/2001/XMLSchema#decimal")
+
+
+def _kg(confidence):
+ entity = {"id": NS + "e1", "text": "Acme", "type": NS + "ORG"}
+ if confidence is not _ABSENT:
+ entity["confidence"] = confidence
+ return {"entities": [entity], "relationships": []}
+
+
+_ABSENT = object()
+
+
+def _graphs(kg):
+ """Parse every serialization of one KG into a graph, keyed by format."""
+ serializer = RDFSerializer()
+ out = {}
+
+ out["turtle"] = Graph()
+ out["turtle"].parse(data=serializer.serialize_to_turtle(json.loads(json.dumps(kg))),
+ format="turtle")
+
+ out["ntriples"] = Graph()
+ out["ntriples"].parse(data=serializer.serialize_to_ntriples(json.loads(json.dumps(kg))),
+ format="nt")
+
+ out["rdfxml"] = Graph()
+ out["rdfxml"].parse(data=serializer.serialize_to_rdfxml(json.loads(json.dumps(kg))),
+ format="xml")
+
+ jsonld = serializer.serialize_to_jsonld(json.loads(json.dumps(kg)))
+ out["jsonld"] = Graph()
+ out["jsonld"].parse(
+ data=jsonld if isinstance(jsonld, str) else json.dumps(jsonld), format="json-ld"
+ )
+ return out
+
+
+def _confidence_terms(graph):
+ return [o for _, p, o in graph if p == CONFIDENCE]
+
+
+def test_every_serializer_agrees_on_the_confidence_term():
+ """The heart of #1100. One value, one RDF term, whatever the format."""
+ terms = {}
+ for name, graph in _graphs(_kg(0.9)).items():
+ values = _confidence_terms(graph)
+ assert len(values) == 1, f"{name} emitted {len(values)} confidence triples"
+ terms[name] = values[0]
+
+ distinct = set(terms.values())
+ assert len(distinct) == 1, (
+ "the same confidence serialised as different RDF terms: "
+ + ", ".join(f"{k}={v!r} ({v.datatype})" for k, v in terms.items())
+ )
+
+
+def test_the_agreed_term_is_a_typed_decimal():
+ for name, graph in _graphs(_kg(0.9)).items():
+ term = _confidence_terms(graph)[0]
+ assert term.datatype == XSD_DECIMAL, f"{name} typed it {term.datatype}"
+ assert str(term) == "0.9", f"{name} wrote the lexical form {str(term)!r}"
+
+
+def test_turtle_and_ntriples_are_the_same_graph():
+ """#1100 as filed: two serializations of one KG must not be two graphs."""
+ graphs = _graphs(_kg(0.9))
+ assert isomorphic(graphs["turtle"], graphs["ntriples"]), (
+ "Turtle only:\n"
+ + "\n".join(str(t) for t in set(graphs["turtle"]) - set(graphs["ntriples"]))
+ + "\nN-Triples only:\n"
+ + "\n".join(str(t) for t in set(graphs["ntriples"]) - set(graphs["turtle"]))
+ )
+
+
+def test_a_numeric_string_is_accepted():
+ for name, graph in _graphs(_kg("0.85")).items():
+ terms = _confidence_terms(graph)
+ assert terms, f"{name} dropped a usable numeric string"
+ assert terms[0].datatype == XSD_DECIMAL
+ assert str(terms[0]) == "0.85"
+
+
+def test_an_integer_confidence_is_accepted():
+ for name, graph in _graphs(_kg(1)).items():
+ terms = _confidence_terms(graph)
+ assert terms, f"{name} dropped an integer confidence"
+ assert terms[0].datatype == XSD_DECIMAL
+
+
+def test_a_small_value_is_not_written_in_exponent_notation():
+ """1e-05 is a valid Python repr and an invalid xsd:decimal lexical form."""
+ for name, graph in _graphs(_kg(0.00001)).items():
+ term = _confidence_terms(graph)[0]
+ assert "e" not in str(term).lower(), f"{name} wrote {str(term)!r}"
+ assert term.value is not None, f"{name} produced an ill-typed literal"
+
+
+# ── #1102 ────────────────────────────────────────────────────────────────────
+
+@pytest.mark.parametrize(
+ "bad", ["high", "", "not a number", None, True, False, [0.9], {"v": 0.9},
+ float("nan"), float("inf")],
+)
+def test_an_unusable_confidence_never_breaks_the_export(bad):
+ """`semantica:confidence high .` made the whole Turtle document unparseable."""
+ serializer = RDFSerializer()
+ kg = _kg(bad)
+
+ turtle = serializer.serialize_to_turtle(json.loads(json.dumps(kg, default=str)))
+ graph = Graph()
+ graph.parse(data=turtle, format="turtle") # must not raise
+
+ assert _confidence_terms(graph) == [], (
+ f"{bad!r} was emitted as a confidence value: {_confidence_terms(graph)}"
+ )
+ # The rest of the entity must survive.
+ assert (URIRef(NS + "e1"), URIRef(NS + "text"), rdflib.Literal("Acme")) in graph
+
+
+def test_an_unusable_confidence_is_dropped_consistently_everywhere():
+ for name, graph in _graphs(_kg("high")).items():
+ assert _confidence_terms(graph) == [], f"{name} still emitted it"
+
+
+def test_all_serializers_stay_parseable_with_an_unusable_confidence():
+ graphs = _graphs(_kg("high")) # each parse would raise on malformed output
+ assert isomorphic(graphs["turtle"], graphs["ntriples"])
+
+
+def test_the_emitted_datatype_matches_the_shipped_vocabulary():
+ """
+ Drift guard. The vocabulary shipped with no rdfs:range on sem:confidence
+ precisely because the serializers disagreed. Now that they agree, the range
+ is declared, and the two must not drift apart again.
+ """
+ from rdflib import RDFS
+
+ from semantica.export.rdf_exporter import CONFIDENCE_DATATYPE
+ from semantica.ontology.vocabulary import vocabulary_turtle
+
+ vocabulary = Graph()
+ vocabulary.parse(data=vocabulary_turtle(), format="turtle")
+
+ declared = list(vocabulary.objects(URIRef(NS + "confidence"), RDFS.range))
+ assert declared, "sem:confidence declares no rdfs:range"
+ assert str(declared[0]) == CONFIDENCE_DATATYPE, (
+ f"vocabulary says {declared[0]}, serializers write {CONFIDENCE_DATATYPE}"
+ )