diff --git a/semantica/export/rdf_exporter.py b/semantica/export/rdf_exporter.py index 63eac4fe..ed1828bb 100644 --- a/semantica/export/rdf_exporter.py +++ b/semantica/export/rdf_exporter.py @@ -30,6 +30,7 @@ License: MIT """ from pathlib import Path +from decimal import Decimal, InvalidOperation from typing import Any, Dict, List, Optional, Set, Union from ..utils.exceptions import ProcessingError, ValidationError @@ -49,6 +50,58 @@ DEFAULT_ENTITY_TYPE = f"{SEMANTICA_NS}Entity" #: Written when a relationship carries no type of its own. Same reasoning. DEFAULT_RELATION_TYPE = f"{SEMANTICA_NS}related_to" +#: The one datatype every serializer writes confidence in. +#: +#: The four paths used to disagree: Turtle wrote the value bare, which the +#: Turtle grammar reads as xsd:decimal, N-Triples typed it xsd:float, RDF/XML +#: emitted a plain literal with no datatype, and JSON-LD emitted a native +#: number, which becomes xsd:double. Those are four distinct RDF terms for one +#: value (issue #1100). +#: +#: xsd:decimal is the choice because it is what the Turtle path already +#: produced, so the most used output is unchanged, and because it is exact: +#: xsd:float is 32 bit binary, and cannot represent 0.9 or 0.95 at all. +CONFIDENCE_DATATYPE = "http://www.w3.org/2001/XMLSchema#decimal" + + +def normalize_confidence(value: Any) -> Optional[str]: + """ + Return the canonical xsd:decimal lexical form of a confidence value. + + Returns None when the value cannot be a decimal, so callers omit the triple + rather than writing something the vocabulary contradicts. The Turtle path + used to interpolate the raw value, so a confidence of "high" produced + `semantica:confidence high .` and made the whole document unparseable + (issue #1102). + + Booleans are rejected. `bool` is a subclass of `int` in Python, so True + would otherwise silently become a confidence of 1. + """ + if value is None or isinstance(value, bool): + return None + if isinstance(value, str): + value = value.strip() + if not value: + return None + if not isinstance(value, (int, float, str, Decimal)): + return None + + try: + decimal_value = Decimal(str(value)) + except (InvalidOperation, ValueError, TypeError): + return None + + # NaN and the infinities are Decimal values with no xsd:decimal form. + if not decimal_value.is_finite(): + return None + + # `str(Decimal("0.00001"))` gives "0.00001", but a float that has already + # been through repr can arrive as "1e-05", which xsd:decimal does not allow. + formatted = format(decimal_value, "f") + if "." in formatted: + formatted = formatted.rstrip("0").rstrip(".") or "0" + return formatted + def mint_entity_iri(text: str) -> str: """Mint a stable IRI for an entity that arrived without an id. @@ -395,11 +448,20 @@ class RDFSerializer: entity_type = entity.get("type", DEFAULT_ENTITY_TYPE) text = entity.get("text") or entity.get("label", "") - confidence = entity.get("confidence", 1.0) + confidence = normalize_confidence(entity.get("confidence", 1.0)) lines.append(f"<{entity_id}> a <{entity_type}> ;") - lines.append(f' semantica:text "{text}" ;') - lines.append(f" semantica:confidence {confidence} .") + if confidence is None: + self.logger.warning( + f"Entity {entity_id} has a confidence that is not a number " + f"({entity.get('confidence')!r}), so no confidence is written" + ) + lines.append(f' semantica:text "{text}" .') + else: + lines.append(f' semantica:text "{text}" ;') + lines.append( + f' semantica:confidence "{confidence}"^^<{CONFIDENCE_DATATYPE}> .' + ) lines.append("") # Convert relationships to RDF triplets @@ -527,15 +589,22 @@ class RDFSerializer: entity_type = entity.get("type", DEFAULT_ENTITY_TYPE) text = entity.get("text") or entity.get("label", "") - confidence = entity.get("confidence", 1.0) + confidence = normalize_confidence(entity.get("confidence", 1.0)) # RDF/XML syntax: rdf:Description with rdf:about lines.append(f' ') lines.append(f' ') lines.append(f" {text}") - lines.append( - f" {confidence}" - ) + if confidence is None: + self.logger.warning( + f"Entity {entity_id} has a confidence that is not a number " + f"({entity.get('confidence')!r}), so no confidence is written" + ) + else: + lines.append( + f' ' + f"{confidence}" + ) lines.append(" ") lines.append("") @@ -606,14 +675,25 @@ class RDFSerializer: entity_text = entity.get("text", "") entity_id = f"semantica:entity/{entity_text}" - jsonld["@graph"].append( - { - "@id": entity_id, - "@type": entity.get("type", "semantica:Entity"), - "semantica:text": entity.get("text") or entity.get("label", ""), - "semantica:confidence": entity.get("confidence", 1.0), + node = { + "@id": entity_id, + "@type": entity.get("type", "semantica:Entity"), + "semantica:text": entity.get("text") or entity.get("label", ""), + } + confidence = normalize_confidence(entity.get("confidence", 1.0)) + if confidence is None: + self.logger.warning( + f"Entity {entity_id} has a confidence that is not a number " + f"({entity.get('confidence')!r}), so no confidence is written" + ) + else: + # A native JSON number becomes xsd:double once expanded, so the + # value is written as a typed literal instead. + node["semantica:confidence"] = { + "@value": confidence, + "@type": CONFIDENCE_DATATYPE, } - ) + jsonld["@graph"].append(node) # Convert relationships to JSON-LD relationships = rdf_data.get("relationships", []) @@ -697,11 +777,20 @@ class RDFSerializer: f'{subject} {expand_uri("semantica:text")} "{safe_text}" .' ) - # Confidence property - confidence = entity.get("confidence") - if confidence is not None: + # Confidence property. The default matches the other serializers, + # which have always written one; omitting it here was half of why + # Turtle and N-Triples of one KG were different graphs (#1100). + raw_confidence = entity.get("confidence", 1.0) + confidence = normalize_confidence(raw_confidence) + if confidence is None: + self.logger.warning( + f"Entity {entity.get('id')} has a confidence that is not a " + f"number ({raw_confidence!r}), so no confidence is written" + ) + else: lines.append( - f'{subject} {expand_uri("semantica:confidence")} "{confidence}"^^ .' + f'{subject} {expand_uri("semantica:confidence")} ' + f'"{confidence}"^^<{CONFIDENCE_DATATYPE}> .' ) # Convert relationships diff --git a/semantica/ontology/vocabulary/semantica-ns.ttl b/semantica/ontology/vocabulary/semantica-ns.ttl index c6843a32..9c461937 100644 --- a/semantica/ontology/vocabulary/semantica-ns.ttl +++ b/semantica/ontology/vocabulary/semantica-ns.ttl @@ -53,16 +53,17 @@ this IRI.""" ; sem:confidence a owl:DatatypeProperty ; rdfs:label "confidence" ; + rdfs:range xsd:decimal ; rdfs:comment """Extractor confidence in the assertion, on the unit interval. Emitted for both entities and relationships, so the domain is left open rather than tied to sem:Entity. -No rdfs:range is declared, deliberately. The Turtle serializer writes the value -bare, which the Turtle grammar reads as xsd:decimal, while the N-Triples -serializer types it xsd:float explicitly, and those two datatypes are disjoint. -Declaring either one would make the vocabulary contradict one of the exporters. -Issue #1100 tracks the disagreement; a range belongs here once the serializers -agree on one.""" ; +The range was left undeclared when this vocabulary first shipped, because the +four serializers disagreed: Turtle wrote the value bare, which the Turtle +grammar reads as xsd:decimal, N-Triples typed it xsd:float, RDF/XML wrote a +plain literal, and JSON-LD wrote a native number, which expands to xsd:double. +Declaring any one of them would have contradicted three exporters. Issue #1100 +settled that on xsd:decimal, which every serializer now writes.""" ; rdfs:isDefinedBy . sem:metadata a owl:AnnotationProperty ; diff --git a/tests/export/test_confidence_literal_typing.py b/tests/export/test_confidence_literal_typing.py new file mode 100644 index 00000000..082f947c --- /dev/null +++ b/tests/export/test_confidence_literal_typing.py @@ -0,0 +1,177 @@ +""" +Regression tests for #1100 and #1102. + +#1100: the four RDF serializers rendered the same confidence value four +different ways. Turtle wrote it bare, which the Turtle grammar reads as +xsd:decimal. N-Triples typed it xsd:float. RDF/XML wrote a plain literal with +no datatype at all. JSON-LD emitted a native JSON number, which becomes +xsd:double. Those are four distinct RDF terms, so a FILTER matches at most one +of them and merging two exports of one graph yields two confidence values for +the same entity. + +#1102: the Turtle path interpolated the value with no type check, so a +non-numeric confidence produced `semantica:confidence high .`, which is not +parseable Turtle. One bad field made the whole export unreadable. +""" + +import json + +import pytest + +rdflib = pytest.importorskip("rdflib") +from rdflib import Graph, URIRef # noqa: E402 +from rdflib.compare import isomorphic # noqa: E402 + +from semantica.export.rdf_exporter import RDFSerializer # noqa: E402 + +NS = "https://semantica.dev/ns#" +CONFIDENCE = URIRef(NS + "confidence") +XSD_DECIMAL = URIRef("http://www.w3.org/2001/XMLSchema#decimal") + + +def _kg(confidence): + entity = {"id": NS + "e1", "text": "Acme", "type": NS + "ORG"} + if confidence is not _ABSENT: + entity["confidence"] = confidence + return {"entities": [entity], "relationships": []} + + +_ABSENT = object() + + +def _graphs(kg): + """Parse every serialization of one KG into a graph, keyed by format.""" + serializer = RDFSerializer() + out = {} + + out["turtle"] = Graph() + out["turtle"].parse(data=serializer.serialize_to_turtle(json.loads(json.dumps(kg))), + format="turtle") + + out["ntriples"] = Graph() + out["ntriples"].parse(data=serializer.serialize_to_ntriples(json.loads(json.dumps(kg))), + format="nt") + + out["rdfxml"] = Graph() + out["rdfxml"].parse(data=serializer.serialize_to_rdfxml(json.loads(json.dumps(kg))), + format="xml") + + jsonld = serializer.serialize_to_jsonld(json.loads(json.dumps(kg))) + out["jsonld"] = Graph() + out["jsonld"].parse( + data=jsonld if isinstance(jsonld, str) else json.dumps(jsonld), format="json-ld" + ) + return out + + +def _confidence_terms(graph): + return [o for _, p, o in graph if p == CONFIDENCE] + + +def test_every_serializer_agrees_on_the_confidence_term(): + """The heart of #1100. One value, one RDF term, whatever the format.""" + terms = {} + for name, graph in _graphs(_kg(0.9)).items(): + values = _confidence_terms(graph) + assert len(values) == 1, f"{name} emitted {len(values)} confidence triples" + terms[name] = values[0] + + distinct = set(terms.values()) + assert len(distinct) == 1, ( + "the same confidence serialised as different RDF terms: " + + ", ".join(f"{k}={v!r} ({v.datatype})" for k, v in terms.items()) + ) + + +def test_the_agreed_term_is_a_typed_decimal(): + for name, graph in _graphs(_kg(0.9)).items(): + term = _confidence_terms(graph)[0] + assert term.datatype == XSD_DECIMAL, f"{name} typed it {term.datatype}" + assert str(term) == "0.9", f"{name} wrote the lexical form {str(term)!r}" + + +def test_turtle_and_ntriples_are_the_same_graph(): + """#1100 as filed: two serializations of one KG must not be two graphs.""" + graphs = _graphs(_kg(0.9)) + assert isomorphic(graphs["turtle"], graphs["ntriples"]), ( + "Turtle only:\n" + + "\n".join(str(t) for t in set(graphs["turtle"]) - set(graphs["ntriples"])) + + "\nN-Triples only:\n" + + "\n".join(str(t) for t in set(graphs["ntriples"]) - set(graphs["turtle"])) + ) + + +def test_a_numeric_string_is_accepted(): + for name, graph in _graphs(_kg("0.85")).items(): + terms = _confidence_terms(graph) + assert terms, f"{name} dropped a usable numeric string" + assert terms[0].datatype == XSD_DECIMAL + assert str(terms[0]) == "0.85" + + +def test_an_integer_confidence_is_accepted(): + for name, graph in _graphs(_kg(1)).items(): + terms = _confidence_terms(graph) + assert terms, f"{name} dropped an integer confidence" + assert terms[0].datatype == XSD_DECIMAL + + +def test_a_small_value_is_not_written_in_exponent_notation(): + """1e-05 is a valid Python repr and an invalid xsd:decimal lexical form.""" + for name, graph in _graphs(_kg(0.00001)).items(): + term = _confidence_terms(graph)[0] + assert "e" not in str(term).lower(), f"{name} wrote {str(term)!r}" + assert term.value is not None, f"{name} produced an ill-typed literal" + + +# ── #1102 ──────────────────────────────────────────────────────────────────── + +@pytest.mark.parametrize( + "bad", ["high", "", "not a number", None, True, False, [0.9], {"v": 0.9}, + float("nan"), float("inf")], +) +def test_an_unusable_confidence_never_breaks_the_export(bad): + """`semantica:confidence high .` made the whole Turtle document unparseable.""" + serializer = RDFSerializer() + kg = _kg(bad) + + turtle = serializer.serialize_to_turtle(json.loads(json.dumps(kg, default=str))) + graph = Graph() + graph.parse(data=turtle, format="turtle") # must not raise + + assert _confidence_terms(graph) == [], ( + f"{bad!r} was emitted as a confidence value: {_confidence_terms(graph)}" + ) + # The rest of the entity must survive. + assert (URIRef(NS + "e1"), URIRef(NS + "text"), rdflib.Literal("Acme")) in graph + + +def test_an_unusable_confidence_is_dropped_consistently_everywhere(): + for name, graph in _graphs(_kg("high")).items(): + assert _confidence_terms(graph) == [], f"{name} still emitted it" + + +def test_all_serializers_stay_parseable_with_an_unusable_confidence(): + graphs = _graphs(_kg("high")) # each parse would raise on malformed output + assert isomorphic(graphs["turtle"], graphs["ntriples"]) + + +def test_the_emitted_datatype_matches_the_shipped_vocabulary(): + """ + Drift guard. The vocabulary shipped with no rdfs:range on sem:confidence + precisely because the serializers disagreed. Now that they agree, the range + is declared, and the two must not drift apart again. + """ + from rdflib import RDFS + + from semantica.export.rdf_exporter import CONFIDENCE_DATATYPE + from semantica.ontology.vocabulary import vocabulary_turtle + + vocabulary = Graph() + vocabulary.parse(data=vocabulary_turtle(), format="turtle") + + declared = list(vocabulary.objects(URIRef(NS + "confidence"), RDFS.range)) + assert declared, "sem:confidence declares no rdfs:range" + assert str(declared[0]) == CONFIDENCE_DATATYPE, ( + f"vocabulary says {declared[0]}, serializers write {CONFIDENCE_DATATYPE}" + )