From 1c27a0ae7e7e9e84fd6be953ff9b098908c82d96 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=B1=9F=E4=BF=8A=E6=9D=B0?= Date: Thu, 20 Aug 2026 09:59:49 +0800 Subject: [PATCH] fix(export): escape RDF literals and use URI-aware id fallback Address Qodo review on #1113: - Escape entity text for Turtle, RDF/XML and N-Triples so names containing quotes, XML markup, backslashes or control chars cannot break out of the literal or inject RDF/XML (High/Security). - Replace colon-only id split with URI-aware local-name extraction so an id like https://example.org/acme yields 'acme', not '//example.org/acme' (Medium/Correctness). - Add regression tests: escaping (quotes/XML/backslash/CR/LF), parseability via rdflib, and exact id local-name assertions. --- semantica/export/rdf_exporter.py | 72 +++++++++++++++++++++++++-- tests/export/test_rdf_exporter.py | 83 ++++++++++++++++++++++++++++++- 2 files changed, 149 insertions(+), 6 deletions(-) diff --git a/semantica/export/rdf_exporter.py b/semantica/export/rdf_exporter.py index c1c095bf..762759ef 100644 --- a/semantica/export/rdf_exporter.py +++ b/semantica/export/rdf_exporter.py @@ -277,6 +277,61 @@ class RDFSerializer: self.logger.debug("RDF serializer initialized") + @staticmethod + def _escape_turtle_literal(value: str) -> str: + """Escape a string for use inside a Turtle/N-Triples double-quoted literal. + + Follows the RDF 1.1 Turtle grammar for STRING_LITERAL_QUOTE: backslash + must be escaped first, then the double quote and the recognized control + characters. This prevents literal breakout and injection. + """ + return ( + value.replace("\\", "\\\\") + .replace('"', '\\"') + .replace("\r", "\\r") + .replace("\n", "\\n") + .replace("\t", "\\t") + ) + + @staticmethod + def _escape_xml_text(value: str) -> str: + """Escape a string for safe inclusion in RDF/XML character data.""" + from xml.sax.saxutils import escape + + # escape() handles &, <, > by default; add quotes for attribute safety. + return escape(value, {'"': """, "'": "'"}) + + @staticmethod + def _local_name_from_id(identifier: str) -> str: + """Derive a human-readable local name from an entity identifier. + + Handles HTTP(S)/IRI identifiers (path segments and fragments, tolerating + trailing slashes) as well as compact/CURIE and URN-style identifiers. + """ + raw = str(identifier).strip() + if not raw: + return "" + + # Prefer a fragment if present (e.g. http://ex.org/onto#acme -> acme). + if "#" in raw: + candidate = raw.rsplit("#", 1)[-1] + if candidate: + return candidate + + # For IRIs/paths, take the last non-empty path segment. + if "/" in raw: + segment = raw.rstrip("/").rsplit("/", 1)[-1] + if segment: + return segment + + # Fall back to the tail of a CURIE/URN (e.g. urn:x:acme, semantica:acme). + if ":" in raw: + candidate = raw.rsplit(":", 1)[-1] + if candidate: + return candidate + + return raw + def convert_kg_to_rdf(self, knowledge_graph: Dict[str, Any]) -> Dict[str, Any]: """ Convert knowledge graph to RDF data structure. @@ -316,8 +371,11 @@ class RDFSerializer: if "name" in norm_entity: norm_entity["label"] = norm_entity["name"] elif "id" in norm_entity: - # Use ID part as label if no name/text - norm_entity["label"] = str(norm_entity["id"]).split(":")[-1] + # Derive a readable label from the identifier's local name + # (fragment/last path segment/CURIE tail). See #1097. + local_name = self._local_name_from_id(norm_entity["id"]) + if local_name: + norm_entity["label"] = local_name rdf_data["entities"].append(norm_entity) @@ -398,7 +456,9 @@ class RDFSerializer: confidence = entity.get("confidence", 1.0) lines.append(f"<{entity_id}> a <{entity_type}> ;") - lines.append(f' semantica:text "{text}" ;') + lines.append( + f' semantica:text "{self._escape_turtle_literal(text)}" ;' + ) lines.append(f" semantica:confidence {confidence} .") lines.append("") @@ -532,7 +592,9 @@ class RDFSerializer: # RDF/XML syntax: rdf:Description with rdf:about lines.append(f' ') lines.append(f' ') - lines.append(f" {text}") + lines.append( + f" {self._escape_xml_text(text)}" + ) lines.append( f" {confidence}" ) @@ -692,7 +754,7 @@ class RDFSerializer: # Text property text = entity.get("text") or entity.get("label", "") if text: - safe_text = text.replace('"', '\\"').replace("\n", "\\n") + safe_text = self._escape_turtle_literal(text) lines.append( f'{subject} {expand_uri("semantica:text")} "{safe_text}" .' ) diff --git a/tests/export/test_rdf_exporter.py b/tests/export/test_rdf_exporter.py index ffbd2994..b2d93137 100644 --- a/tests/export/test_rdf_exporter.py +++ b/tests/export/test_rdf_exporter.py @@ -1,6 +1,7 @@ """Tests for RDFExporter format alias resolution (issue #355).""" import pytest + from semantica.export import RDFExporter RDF_DATA = { @@ -150,7 +151,7 @@ def test_existing_text_not_overwritten_by_name(exporter): def test_id_fallback_label_when_no_name(exporter): - """With neither 'name' nor 'text', the id tail is used as the label (#1097).""" + """With neither 'name' nor 'text', the id local-name is used as label (#1097).""" data = { "entities": [ {"id": "https://example.org/acme", "type": "https://example.org/Org"} @@ -159,3 +160,83 @@ def test_id_fallback_label_when_no_name(exporter): } result = exporter.export_to_rdf(data, format="turtle") assert 'semantica:text ""' not in result + # The label must be the URI local name 'acme', not '//example.org/acme'. + assert 'semantica:text "acme"' in result + + +@pytest.mark.parametrize( + "identifier,expected", + [ + ("https://example.org/acme", "acme"), + ("https://example.org/path/acme", "acme"), + ("https://example.org/onto#acme", "acme"), + ("https://example.org/acme/", "acme"), + ("urn:example:acme", "acme"), + ("semantica:acme", "acme"), + ("acme", "acme"), + ], +) +def test_id_fallback_local_name_extraction(exporter, identifier, expected): + """The id fallback must extract a URI-aware local name, not a colon split (#1113).""" + data = { + "entities": [{"id": identifier, "type": "https://example.org/Org"}], + "relationships": [], + } + result = exporter.export_to_rdf(data, format="turtle") + assert f'semantica:text "{expected}"' in result + + +def _export(exporter, name, fmt): + data = { + "entities": [ + {"id": "https://example.org/e1", "name": name, "type": "ORG"} + ], + "relationships": [], + } + return exporter.export_to_rdf(data, format=fmt) + + +def test_turtle_escapes_quotes_and_control_chars(exporter): + """Turtle literals must escape quotes/backslashes/newlines (#1113 security).""" + result = _export(exporter, 'Acme "Best" \\ Corp\nLine2\tTab\rCR', "turtle") + # The raw closing-quote breakout must not appear inside the literal. + assert '"Acme "Best"' not in result + assert '\\"Best\\"' in result + assert "\\\\ Corp" in result + assert "\\n" in result and "\\t" in result and "\\r" in result + # No unescaped newline leaked into the literal value. + assert "Line2" in result and "\nLine2" not in result.split("semantica:text")[1] + + +def test_ntriples_escapes_quotes_and_control_chars(exporter): + """N-Triples literals must escape backslash first, then quotes/controls (#1113).""" + result = _export(exporter, 'Quote " Back \\ New\nTab\t', "ntriples") + assert '\\"' in result + assert "\\\\" in result + assert "\\n" in result and "\\t" in result + # Each triple must be a single physical line: the literal value carrying the + # escaped text must not have leaked a bare newline that splits it in two. + text_lines = [ln for ln in result.splitlines() if "Quote" in ln] + assert len(text_lines) == 1 + assert text_lines[0].rstrip().endswith(" .") + + +def test_rdfxml_escapes_markup(exporter): + """RDF/XML character data must escape &, <, > so names cannot inject markup (#1113).""" + result = _export(exporter, 'Acme