If Semantica solves a real problem for you, a star helps others find it.
@@ -1561,11 +1563,11 @@ On-premises deployment ยท Private cloud ยท Custom domain implementations ยท SLA-
## Star History
-
@@ -1594,6 +1596,23 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for full guidelines.
---
+## Cite Us
+
+If you use Semantica in your research or production systems, please cite it as:
+
+```bibtex
+@software{semantica2026,
+ title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
+ author = {Semantica},
+ year = {2026},
+ url = {https://github.com/semantica-agi/semantica}
+}
+```
+
+All citation formats (APA, MLA, Chicago, IEEE) live on the [Citation](https://docs.getsemantica.ai/citation) page โ every format attributes authorship to **Semantica**, not individual contributors.
+
+---
+
MIT License ยท Built by [Semantica](https://github.com/semantica-agi)
diff --git a/cookbook/advanced/01_Advanced_Extraction.ipynb b/cookbook/advanced/01_Advanced_Extraction.ipynb
index e4989da6..49113115 100644
--- a/cookbook/advanced/01_Advanced_Extraction.ipynb
+++ b/cookbook/advanced/01_Advanced_Extraction.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/01_Advanced_Extraction.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/01_Advanced_Extraction.ipynb)\n",
"\n",
"# Advanced Extraction\n",
"\n",
diff --git a/cookbook/advanced/03_Complete_Visualization_Suite.ipynb b/cookbook/advanced/03_Complete_Visualization_Suite.ipynb
index d081b2df..655e723a 100644
--- a/cookbook/advanced/03_Complete_Visualization_Suite.ipynb
+++ b/cookbook/advanced/03_Complete_Visualization_Suite.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/03_Complete_Visualization_Suite.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/03_Complete_Visualization_Suite.ipynb)\n",
"\n",
"# Complete Visualization Suite\n",
"\n",
diff --git a/cookbook/advanced/05_Multi_Format_Export.ipynb b/cookbook/advanced/05_Multi_Format_Export.ipynb
index 197ce788..306409c7 100644
--- a/cookbook/advanced/05_Multi_Format_Export.ipynb
+++ b/cookbook/advanced/05_Multi_Format_Export.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/05_Multi_Format_Export.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/05_Multi_Format_Export.ipynb)\n",
"\n",
"# Advanced Multi-Format Export\n",
"\n",
diff --git a/cookbook/advanced/08_Reasoning_and_Inference.ipynb b/cookbook/advanced/08_Reasoning_and_Inference.ipynb
index 2f86fe1c..5854f4bf 100644
--- a/cookbook/advanced/08_Reasoning_and_Inference.ipynb
+++ b/cookbook/advanced/08_Reasoning_and_Inference.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/08_Reasoning_and_Inference.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/08_Reasoning_and_Inference.ipynb)\n",
"\n",
"# Reasoning and Inference\n",
"\n",
diff --git a/cookbook/advanced/09_Semantic_Layer_Construction.ipynb b/cookbook/advanced/09_Semantic_Layer_Construction.ipynb
index d8c090c9..de1ecd78 100644
--- a/cookbook/advanced/09_Semantic_Layer_Construction.ipynb
+++ b/cookbook/advanced/09_Semantic_Layer_Construction.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/09_Semantic_Layer_Construction.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/09_Semantic_Layer_Construction.ipynb)\n",
"\n",
"# Semantic Layer Construction\n",
"\n",
diff --git a/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb b/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb
index 21597361..03843a60 100644
--- a/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb
+++ b/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb)\n",
"\n",
"# Deep Dive: Temporal Knowledge Graphs\n",
"\n",
diff --git a/cookbook/advanced/12_Unstructured_to_Ontology.ipynb b/cookbook/advanced/12_Unstructured_to_Ontology.ipynb
index f683c7e0..6eddba5b 100644
--- a/cookbook/advanced/12_Unstructured_to_Ontology.ipynb
+++ b/cookbook/advanced/12_Unstructured_to_Ontology.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/12_Unstructured_to_Ontology.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/12_Unstructured_to_Ontology.ipynb)\n",
"\n",
"# Unstructured Text to Ontology\n",
"\n",
diff --git a/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb b/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb
index 392cdc28..0647f3b9 100644
--- a/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb
+++ b/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb
@@ -18,7 +18,7 @@
"id": "cell-0",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb)\n",
"\n",
"# Manual Ontology + Snowflake Mapping\n",
"\n",
diff --git a/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb b/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb
index 5382c03e..2898f9d6 100644
--- a/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb
+++ b/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb)\n",
"\n",
"# Datalog-Style Reasoning\n",
"\n",
diff --git a/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb b/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb
index 1823a434..175736fe 100644
--- a/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb
+++ b/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb)\n",
"\n",
"# Advanced Vector Store - Made Easy\n",
"\n",
@@ -352,7 +352,7 @@
"- Build a multi-user application\n",
"- Explore the [introduction notebook](../introduction/13_Vector_Store.ipynb) for more basics\n",
"\n",
- "**Need Help?** Check our [documentation](https://semantica.readthedocs.io) or ask on [GitHub](https://github.com/Hawksight-AI/semantica)."
+ "**Need Help?** Check our [documentation](https://semantica.readthedocs.io) or ask on [GitHub](https://github.com/semantica-agi/semantica)."
]
}
],
diff --git a/cookbook/introduction/01_Welcome_to_Semantica.ipynb b/cookbook/introduction/01_Welcome_to_Semantica.ipynb
index 05417881..21677088 100644
--- a/cookbook/introduction/01_Welcome_to_Semantica.ipynb
+++ b/cookbook/introduction/01_Welcome_to_Semantica.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb)\n",
"\n",
"Semantica is a **semantic intelligence and knowledge engineering framework**. It helps you:\n",
"\n",
diff --git a/cookbook/introduction/02_Data_Ingestion.ipynb b/cookbook/introduction/02_Data_Ingestion.ipynb
index f343a5a2..a8d9e5d6 100644
--- a/cookbook/introduction/02_Data_Ingestion.ipynb
+++ b/cookbook/introduction/02_Data_Ingestion.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/02_Data_Ingestion.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/02_Data_Ingestion.ipynb)\n",
"\n",
"# Data Ingestion - Comprehensive Guide\n",
"\n",
diff --git a/cookbook/introduction/03_Document_Parsing.ipynb b/cookbook/introduction/03_Document_Parsing.ipynb
index 639c076b..171f111a 100644
--- a/cookbook/introduction/03_Document_Parsing.ipynb
+++ b/cookbook/introduction/03_Document_Parsing.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/04_Document_Parsing.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/03_Document_Parsing.ipynb)\n",
"\n",
"# Document Parsing\n",
"\n",
diff --git a/cookbook/introduction/04_Data_Normalization.ipynb b/cookbook/introduction/04_Data_Normalization.ipynb
index a4a668ee..6cdd07db 100644
--- a/cookbook/introduction/04_Data_Normalization.ipynb
+++ b/cookbook/introduction/04_Data_Normalization.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/05_Data_Normalization.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/04_Data_Normalization.ipynb)\n",
"\n",
"# Data Normalization\n",
"\n",
diff --git a/cookbook/introduction/05_Entity_Extraction.ipynb b/cookbook/introduction/05_Entity_Extraction.ipynb
index 4b78e19c..78cabe22 100644
--- a/cookbook/introduction/05_Entity_Extraction.ipynb
+++ b/cookbook/introduction/05_Entity_Extraction.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/05_Entity_Extraction.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/05_Entity_Extraction.ipynb)\n",
"\n",
"# Entity Extraction - Comprehensive Guide\n",
"\n",
@@ -622,7 +622,7 @@
"\n",
"---\n",
"\n",
- "**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
+ "**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
]
}
],
diff --git a/cookbook/introduction/06_Relation_Extraction.ipynb b/cookbook/introduction/06_Relation_Extraction.ipynb
index e11566c6..8015f86f 100644
--- a/cookbook/introduction/06_Relation_Extraction.ipynb
+++ b/cookbook/introduction/06_Relation_Extraction.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/06_Relation_Extraction.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/06_Relation_Extraction.ipynb)\n",
"\n",
"# Relation Extraction - Comprehensive Guide\n",
"\n",
@@ -599,7 +599,7 @@
"\n",
"---\n",
"\n",
- "**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
+ "**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
]
}
],
diff --git a/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb b/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb
index bd3c7ba2..5c27420a 100644
--- a/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb
+++ b/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/08_Building_Knowledge_Graphs.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb)\n",
"\n",
"# Building Knowledge Graphs\n",
"\n",
diff --git a/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb b/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb
index fb6e1c2f..f159efb6 100644
--- a/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb
+++ b/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/09_Your_First_Knowledge_Graph.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb)\n",
"\n",
"# ๐ Your First Knowledge Graph\n",
"\n",
diff --git a/cookbook/introduction/10_Graph_Analytics.ipynb b/cookbook/introduction/10_Graph_Analytics.ipynb
index f15ee443..e0307e04 100644
--- a/cookbook/introduction/10_Graph_Analytics.ipynb
+++ b/cookbook/introduction/10_Graph_Analytics.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/11_Graph_Analytics.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/10_Graph_Analytics.ipynb)\n",
"\n",
"# Graph Analytics\n",
"\n",
diff --git a/cookbook/introduction/11_Chunking_and_Splitting.ipynb b/cookbook/introduction/11_Chunking_and_Splitting.ipynb
index 9bb5cb3f..27101493 100644
--- a/cookbook/introduction/11_Chunking_and_Splitting.ipynb
+++ b/cookbook/introduction/11_Chunking_and_Splitting.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb)\n",
"\n",
"# Chunking and Splitting - Comprehensive Guide\n",
"\n",
@@ -817,7 +817,7 @@
"\n",
"---\n",
"\n",
- "**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
+ "**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
]
}
],
diff --git a/cookbook/introduction/12_Embedding_Generation.ipynb b/cookbook/introduction/12_Embedding_Generation.ipynb
index b1ad1081..4b334fcc 100644
--- a/cookbook/introduction/12_Embedding_Generation.ipynb
+++ b/cookbook/introduction/12_Embedding_Generation.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/13_Embedding_Generation.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/12_Embedding_Generation.ipynb)\n",
"\n",
"# Embedding Generation\n",
"\n",
diff --git a/cookbook/introduction/13_Vector_Store.ipynb b/cookbook/introduction/13_Vector_Store.ipynb
index f2424104..32baeab3 100644
--- a/cookbook/introduction/13_Vector_Store.ipynb
+++ b/cookbook/introduction/13_Vector_Store.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb)\n",
"\n",
"# Vector Store - Comprehensive Guide\n",
"\n",
@@ -492,7 +492,7 @@
"\n",
"---\n",
"\n",
- "**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
+ "**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
]
}
],
diff --git a/cookbook/introduction/14_Ontology.ipynb b/cookbook/introduction/14_Ontology.ipynb
index 3c112404..64bbee06 100644
--- a/cookbook/introduction/14_Ontology.ipynb
+++ b/cookbook/introduction/14_Ontology.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/14_Ontology.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/14_Ontology.ipynb)\n",
"\n",
"# Ontology Generation \n",
"\n",
diff --git a/cookbook/introduction/15_Export.ipynb b/cookbook/introduction/15_Export.ipynb
index 3d6a8c10..224c63ea 100644
--- a/cookbook/introduction/15_Export.ipynb
+++ b/cookbook/introduction/15_Export.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/15_Export.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/15_Export.ipynb)\n",
"\n",
"# Export Module - Comprehensive Guide\n",
"\n",
diff --git a/cookbook/introduction/16_Visualization.ipynb b/cookbook/introduction/16_Visualization.ipynb
index 05c55b61..31483278 100644
--- a/cookbook/introduction/16_Visualization.ipynb
+++ b/cookbook/introduction/16_Visualization.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/17_Visualization.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/16_Visualization.ipynb)\n",
"\n",
"# Visualization\n",
"\n",
diff --git a/cookbook/introduction/18_Deduplication.ipynb b/cookbook/introduction/18_Deduplication.ipynb
index 087817a7..53e03683 100644
--- a/cookbook/introduction/18_Deduplication.ipynb
+++ b/cookbook/introduction/18_Deduplication.ipynb
@@ -4,7 +4,7 @@
"cell_type": "markdown",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/18_Deduplication.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/18_Deduplication.ipynb)\n",
"\n",
"# Deduplication in Semantica\n",
"\n",
diff --git a/cookbook/introduction/19_Context_Module.ipynb b/cookbook/introduction/19_Context_Module.ipynb
index 2bbe81ce..d2ec4cf4 100644
--- a/cookbook/introduction/19_Context_Module.ipynb
+++ b/cookbook/introduction/19_Context_Module.ipynb
@@ -5,7 +5,7 @@
"id": "c21e9c8d",
"metadata": {},
"source": [
- "[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/19_Context_Module.ipynb)\n",
+ "[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/19_Context_Module.ipynb)\n",
"\n",
"# Context Module โ Practical Guide\n",
"\n",
diff --git a/docs/citation.md b/docs/citation.md
index 45ac350b..b798677a 100644
--- a/docs/citation.md
+++ b/docs/citation.md
@@ -13,26 +13,25 @@ icon: "quote-left"
```bibtex
@software{semantica2026,
- title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
- author = {Semantica},
- year = {2026},
- url = {https://github.com/semantica-agi/semantica},
- version = {0.6.6},
- doi = {10.5281/zenodo.XXXXXXX}
+ title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
+ author = {Semantica},
+ year = {2026},
+ url = {https://github.com/semantica-agi/semantica},
+ doi = {10.5281/zenodo.XXXXXXX}
}
```
- Semantica. (2026). *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems* (Version 0.6.6) \[Computer software\]. https://github.com/semantica-agi/semantica
+ Semantica. (2026). *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems* \[Computer software\]. https://github.com/semantica-agi/semantica
- Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. Version 0.6.6, GitHub, 2026, https://github.com/semantica-agi/semantica.
+ Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. GitHub, 2026, https://github.com/semantica-agi/semantica.
- Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. Version 0.6.6. GitHub, 2026. https://github.com/semantica-agi/semantica.
+ Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. GitHub, 2026. https://github.com/semantica-agi/semantica.
- Semantica, "Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems," Version 0.6.6, GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
+ Semantica, "Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems," GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
diff --git a/docs/cookbook.md b/docs/cookbook.md
index 443aae7c..d7a780bb 100644
--- a/docs/cookbook.md
+++ b/docs/cookbook.md
@@ -80,6 +80,6 @@ Deep dive into advanced features, customization, and complex workflows.
You can also run the cookbook using Docker:
```bash
- docker run -p 8888:8888 hawksight/semantica-cookbook
+ docker run -p 8888:8888 semantica/semantica-cookbook
```
diff --git a/docs/governance.md b/docs/governance.md
index 332cf5a2..e1df0508 100644
--- a/docs/governance.md
+++ b/docs/governance.md
@@ -4,12 +4,12 @@ description: "Project governance model: roles, decision process, release cadence
icon: "scale-balanced"
---
-> Semantica is maintained by Hawksight AI with community contributions under an open governance model.
+> Semantica is maintained by the Semantica team with community contributions under an open governance model.
## Roles
-- **Maintainers** โ Hawksight AI team: review and merge PRs, manage releases and code quality, set project direction and community standards.
+- **Maintainers** โ Semantica team: review and merge PRs, manage releases and code quality, set project direction and community standards.
- **Contributors** โ Submit code, documentation, and bug reports. Help with issues and reviews. Recognized in [CONTRIBUTORS.md](https://github.com/semantica-agi/semantica/blob/main/CONTRIBUTORS.md).
- **Community Members** โ Use Semantica, provide feedback, share use cases, and participate in GitHub Discussions and Discord.
diff --git a/docs/project-license.md b/docs/project-license.md
index b1fb31ef..220f0f44 100644
--- a/docs/project-license.md
+++ b/docs/project-license.md
@@ -12,7 +12,7 @@ icon: "file-contract"
```
MIT License
-Copyright (c) 2026 Hawksight AI
+Copyright (c) 2026 Semantica
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
diff --git a/docs/reference/triplet_store.md b/docs/reference/triplet_store.md
index ad7a0645..ee5c24e6 100644
--- a/docs/reference/triplet_store.md
+++ b/docs/reference/triplet_store.md
@@ -182,7 +182,7 @@ for row in result.bindings:
store = TripletStore(
backend="rdf4j",
endpoint="http://localhost:8080/rdf4j-server",
- repository_id="semantica", # passed through **config
+ repository_id="semantica", # selects the remote repository
)
```
diff --git a/docs/reference/utils.md b/docs/reference/utils.md
index 83f9cc97..f7e97789 100644
--- a/docs/reference/utils.md
+++ b/docs/reference/utils.md
@@ -77,7 +77,17 @@ Most users won't call utils directly: it's the **shared foundation** for all mod
export SEMANTICA_LOG_LEVEL=DEBUG
export SEMANTICA_LOG_FORMAT=json # "json" | "text"
export SEMANTICA_DISABLE_PROGRESS=true
+ export SEMANTICA_FORCE_PROGRESS=true
```
+
+
+ **Progress bars follow your terminal.** Console progress is written only when
+ stdout is an interactive terminal (or a Jupyter notebook), so piping or
+ redirecting output no longer fills logs with progress bars and escape
+ sequences. Set `SEMANTICA_DISABLE_PROGRESS` to silence progress even in a
+ terminal, or `SEMANTICA_FORCE_PROGRESS` to keep it when stdout is redirected.
+ `SEMANTICA_DISABLE_PROGRESS` wins if both are set.
+
diff --git a/docs/storage-backends.md b/docs/storage-backends.md
index bb05e115..dc211ef0 100644
--- a/docs/storage-backends.md
+++ b/docs/storage-backends.md
@@ -35,7 +35,7 @@ This page is intentionally conservative: it distinguishes between an adapter exi
| FalkorDB | LPG | Yes | Yes | Partial | Partial | Redis-based; provenance depends on node/edge properties, and multi-graph isolation depends on the selected graph name. |
| Amazon Neptune | LPG | Yes | Yes | Partial | Partial | Use the property-graph endpoint; AWS auth, VPC, and endpoint configuration can affect local tests. Provenance depends on node/edge properties. |
| Apache AGE | LPG | Yes | Yes | Partial | Partial | Runs through PostgreSQL/AGE; Cypher compatibility and property handling can differ from standalone LPG engines. |
-| RDF4J | RDF | Yes | Partial | Partial | Partial | Context separation relies on named graphs; triple-level provenance may require reification or graph-level metadata. `RDF4JStore(repository_id=...)` currently has no effect โ the constructor always connects to the `"default"` repository regardless of the value passed; track a fix separately. |
+| RDF4J | RDF | Yes | Partial | Partial | Partial | Context separation relies on named graphs; triple-level provenance may require reification or graph-level metadata. |
| Apache Jena | RDF | Yes | Partial | Partial | Partial | Named graphs are needed for context separation; backend configuration and transaction behavior matter. |
| Blazegraph | RDF | Yes | Partial | Partial | Partial | Use quads/named graphs for context; IRI stability and graph naming matter for provenance. |
| Anzo | RDF | Yes | Partial | Partial | Partial | Anzo deployments are environment-specific; validate `dataset_uri`/graphmart naming, named-graph support, and provenance mapping. |
@@ -107,7 +107,7 @@ from semantica.triplet_store import RDF4JStore
store = RDF4JStore(
endpoint='http://localhost:8080/rdf4j-server',
- repository_id='semantica' # currently has no effect; connects to "default" (see Known limitations)
+ repository_id='semantica'
)
```
diff --git a/plugins/.claude-plugin/README.md b/plugins/.claude-plugin/README.md
index 6fa56eef..c685a65b 100644
--- a/plugins/.claude-plugin/README.md
+++ b/plugins/.claude-plugin/README.md
@@ -53,7 +53,7 @@ plugins/
## Prerequisites
```bash
-git clone https://github.com/Hawksight-AI/semantica.git
+git clone https://github.com/semantica-agi/semantica.git
cd semantica
pip install semantica # Python 3.10+
```
diff --git a/plugins/.claude-plugin/marketplace.json b/plugins/.claude-plugin/marketplace.json
index cbe8b924..0afe6ddb 100644
--- a/plugins/.claude-plugin/marketplace.json
+++ b/plugins/.claude-plugin/marketplace.json
@@ -1,8 +1,8 @@
{
"name": "semantica-local",
"owner": {
- "name": "Hawksight AI",
- "url": "https://github.com/Hawksight-AI/semantica"
+ "name": "Semantica",
+ "url": "https://github.com/semantica-agi/semantica"
},
"plugins": [
{
diff --git a/plugins/.claude-plugin/plugin.json b/plugins/.claude-plugin/plugin.json
index fd5d35e7..8328e3a9 100644
--- a/plugins/.claude-plugin/plugin.json
+++ b/plugins/.claude-plugin/plugin.json
@@ -5,8 +5,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.cline-plugin/plugin.json b/plugins/.cline-plugin/plugin.json
index 81c004c4..d4f09886 100644
--- a/plugins/.cline-plugin/plugin.json
+++ b/plugins/.cline-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.codex-plugin/plugin.json b/plugins/.codex-plugin/plugin.json
index c12d66c2..eac91fc6 100644
--- a/plugins/.codex-plugin/plugin.json
+++ b/plugins/.codex-plugin/plugin.json
@@ -5,8 +5,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.continue-plugin/plugin.json b/plugins/.continue-plugin/plugin.json
index da93dc38..57d45e82 100644
--- a/plugins/.continue-plugin/plugin.json
+++ b/plugins/.continue-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.cursor-plugin/plugin.json b/plugins/.cursor-plugin/plugin.json
index 3b73366a..0a221866 100644
--- a/plugins/.cursor-plugin/plugin.json
+++ b/plugins/.cursor-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.openclaw-plugin/plugin.json b/plugins/.openclaw-plugin/plugin.json
index 0489b609..582145f5 100644
--- a/plugins/.openclaw-plugin/plugin.json
+++ b/plugins/.openclaw-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.vscode-plugin/plugin.json b/plugins/.vscode-plugin/plugin.json
index a39c097e..771df46d 100644
--- a/plugins/.vscode-plugin/plugin.json
+++ b/plugins/.vscode-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/plugins/.windsurf-plugin/plugin.json b/plugins/.windsurf-plugin/plugin.json
index cbf45713..abb4831d 100644
--- a/plugins/.windsurf-plugin/plugin.json
+++ b/plugins/.windsurf-plugin/plugin.json
@@ -6,8 +6,8 @@
"author": {
"name": "Semantica Contributors"
},
- "homepage": "https://github.com/Hawksight-AI/semantica",
- "repository": "https://github.com/Hawksight-AI/semantica",
+ "homepage": "https://github.com/semantica-agi/semantica",
+ "repository": "https://github.com/semantica-agi/semantica",
"license": "MIT",
"keywords": [
"semantica",
diff --git a/semantica/change_management/change_management_usage.md b/semantica/change_management/change_management_usage.md
index 0b5c161d..d7dde3ae 100644
--- a/semantica/change_management/change_management_usage.md
+++ b/semantica/change_management/change_management_usage.md
@@ -1039,6 +1039,6 @@ manager = TemporalVersionManager(storage_path="large_data.db")
## Support
For questions or issues:
-- GitHub Issues: https://github.com/Hawksight-AI/semantica/issues
+- GitHub Issues: https://github.com/semantica-agi/semantica/issues
- Documentation: https://semantica.readthedocs.io
- Community: https://discord.gg/sV34vps5hH
diff --git a/semantica/export/rdf_exporter.py b/semantica/export/rdf_exporter.py
index 32a09895..dffd4951 100644
--- a/semantica/export/rdf_exporter.py
+++ b/semantica/export/rdf_exporter.py
@@ -146,6 +146,275 @@ def mint_relationship_iri(index: int, source: Any, target: Any) -> str:
return f"{SEMANTICA_NS}rel_{index}_{digest}"
+#: The metadata keys Semantica itself produces, and the terms they are written
+#: as. GraphBuilder.build_graph writes the first five, create_snapshot writes
+#: snapshot_time, and load_from_neo4j writes source / uri / database. These are
+#: Semantica's own vocabulary, so they are minted in the declared namespace and
+#: declared in semantica-ns.ttl.
+#:
+#: A key the caller supplied is a different matter. Which namespace an
+#: arbitrary metadata key belongs in is issue #1146, and until that is settled
+#: the exporter refuses to guess: it warns and skips, and a caller who already
+#: knows the answer passes ``metadata_terms``.
+#:
+#: The map is key -> term rather than key -> namespace because two of the keys
+#: cannot keep their own name. ``source`` on a graph loaded from Neo4j is the
+#: system it came from, while sem:source is already the ObjectProperty holding
+#: the subject of a reified relationship; reusing it would put a string where
+#: an entity belongs.
+DEFAULT_METADATA_TERMS: Dict[str, str] = {
+ "num_entities": f"{SEMANTICA_NS}numEntities",
+ "num_relationships": f"{SEMANTICA_NS}numRelationships",
+ "temporal_enabled": f"{SEMANTICA_NS}temporalEnabled",
+ "entity_resolution_applied": f"{SEMANTICA_NS}entityResolutionApplied",
+ "timestamp": f"{SEMANTICA_NS}builtAt",
+ "snapshot_time": f"{SEMANTICA_NS}snapshotAt",
+ "source": f"{SEMANTICA_NS}sourceSystem",
+ "uri": f"{SEMANTICA_NS}sourceUri",
+ "database": f"{SEMANTICA_NS}sourceDatabase",
+}
+
+#: Terms whose value is a node rather than a string. Everything else stays a
+#: literal: a metadata value that merely looks like a URL is not thereby a
+#: reference to one.
+IRI_VALUED_METADATA_TERMS: Set[str] = {f"{SEMANTICA_NS}sourceUri"}
+
+_XSD_NS = "http://www.w3.org/2001/XMLSchema#"
+
+
+def _escape_literal(value: str) -> str:
+ """Escape a string for a Turtle or N-Triples quoted literal."""
+ return (
+ value.replace("\\", "\\\\")
+ .replace('"', '\\"')
+ .replace("\n", "\\n")
+ .replace("\r", "\\r")
+ .replace("\t", "\\t")
+ )
+
+
+#: Turtle/N-Triples IRIREF grammar excludes these unescaped between `<` and
+#: `>`: control characters, space, and <>"{}|^`\. An IRI-valued metadata
+#: value (currently only sem:sourceUri, from the caller-controlled "uri"
+#: metadata key) is written as `<{value}>` with no other quoting, so a value
+#: containing one of these characters โ a ">" followed by a full triple, for
+#: instance โ closes the IRIREF early and lets the rest of the string be
+#: parsed as further RDF statements. This is the same shape of defect the
+#: entity/relationship IRIs were hardened against; that hardening resolves
+#: prefixes as well, which a metadata value never needs, so this stays a
+#: narrower, dedicated guard rather than reusing _as_turtle_iri.
+_IRI_REF_UNSAFE_RE = re.compile(r'[\x00-\x20<>"{}|^`\\]')
+
+
+def _safe_iri_ref(value: str) -> str:
+ """Percent-encode the characters an IRIREF may not contain unescaped."""
+ return _IRI_REF_UNSAFE_RE.sub(lambda m: quote(m.group(0), safe=""), value)
+
+
+def _escape_xml(value: str) -> str:
+ """Escape a string for either XML element text or an attribute value.
+
+ The quotes matter. This helper feeds `rdf:about`, `rdf:resource` and
+ `xmlns:` attribute values, which are delimited by double quotes, so a value
+ carrying one would close the attribute early and produce a document that
+ does not parse. Escaping them in element text as well is harmless and
+ means one helper cannot be used in the wrong place.
+ """
+ return (
+ value.replace("&", "&")
+ .replace("<", "<")
+ .replace(">", ">")
+ .replace('"', """)
+ .replace("'", "'")
+ )
+
+
+def _is_ncname(value: str) -> bool:
+ """Whether a string can be an XML NCName, which is what RDF/XML requires.
+
+ Checked over the ASCII range rather than the full XML production: the
+ grammar also admits combining characters and extenders, so this is
+ deliberately conservative. It refuses names it could have accepted, and it
+ never accepts one that would produce a document a parser rejects. The
+ earlier check tested only that the first character was not a digit, which
+ let through every other way a local name can fail to be a name.
+ """
+ if not value:
+ return False
+ if not (value[0].isascii() and (value[0].isalpha() or value[0] == "_")):
+ return False
+ return all(c.isascii() and (c.isalnum() or c in "._-") for c in value[1:])
+
+
+def _split_iri(iri: str) -> Optional[tuple]:
+ """Split an IRI into (namespace, local name) for RDF/XML's QName syntax.
+
+ Returns None when no split yields a usable local name. RDF/XML is the only
+ serialization here that cannot write an arbitrary predicate IRI, so this is
+ the one place a term can be unrepresentable, and the caller reports it
+ rather than dropping it quietly.
+ """
+ for sep in ("#", "/"):
+ index = iri.rfind(sep)
+ if index != -1 and index + 1 < len(iri):
+ local = iri[index + 1 :]
+ if _is_ncname(local):
+ return iri[: index + 1], local
+ return None
+
+
+def _metadata_statements(
+ metadata: Any,
+ terms: Dict[str, str],
+ logger: Any,
+) -> List[tuple]:
+ """Resolve a metadata mapping to a list of (term IRI, value) pairs.
+
+ A key with no term is skipped and reported. Silence is the defect this
+ fixes, so an unmapped key must be louder than a mapped one, not quieter.
+ """
+ if not isinstance(metadata, dict):
+ return []
+
+ statements: List[tuple] = []
+ for key, value in metadata.items():
+ term = terms.get(key)
+ if term is None:
+ logger.warning(
+ "Metadata key %r has no term and was not exported. Which "
+ "namespace a caller-supplied key belongs in is issue #1146; "
+ "pass metadata_terms={%r: '
'} to export it now.",
+ key,
+ key,
+ )
+ continue
+ if value is None:
+ continue
+ if isinstance(value, (dict, list, tuple, set)):
+ logger.warning(
+ "Metadata key %r holds a %s, which has no modelled RDF shape "
+ "yet, and was not exported.",
+ key,
+ type(value).__name__,
+ )
+ continue
+ statements.append((term, value))
+ return statements
+
+
+def _resolve_metadata_terms(overrides: Optional[Dict[str, str]]) -> Dict[str, str]:
+ if not overrides:
+ return DEFAULT_METADATA_TERMS
+ return {**DEFAULT_METADATA_TERMS, **overrides}
+
+
+def _typed_literal_parts(term: str, value: Any) -> tuple:
+ """Return (kind, lexical, datatype) for one metadata value.
+
+ kind is "iri" or "literal". The lexical form and datatype are chosen once,
+ here, so that the four serializers cannot disagree about them the way they
+ disagreed about confidence in #1100.
+ """
+ if term in IRI_VALUED_METADATA_TERMS and isinstance(value, str):
+ return "iri", value, None
+ if isinstance(value, bool):
+ return "literal", "true" if value else "false", f"{_XSD_NS}boolean"
+ if isinstance(value, int):
+ return "literal", str(value), f"{_XSD_NS}integer"
+ if isinstance(value, float):
+ # xsd:double, not xsd:decimal. `repr(1e-05)` is "1e-05" and
+ # `repr(float("nan"))` is "nan", and xsd:decimal admits neither the
+ # exponent form nor the special values, so typing a float as decimal
+ # produced lexicals a strict parser rejects. A Python float is an IEEE
+ # 754 double; xsd:double has legal lexicals for all of them, and it is
+ # also the honest claim, since nothing that arrived as a float was ever
+ # exact. `normalize_confidence` keeps xsd:decimal for confidence
+ # deliberately: that is a bounded score where exactness is meaningful
+ # and NaN is not a confidence at all.
+ if value != value:
+ lexical = "NaN"
+ elif value == float("inf"):
+ lexical = "INF"
+ elif value == float("-inf"):
+ lexical = "-INF"
+ else:
+ lexical = repr(value)
+ return "literal", lexical, f"{_XSD_NS}double"
+ return "literal", str(value), None
+
+
+def _turtle_object(term: str, value: Any) -> str:
+ kind, lexical, datatype = _typed_literal_parts(term, value)
+ if kind == "iri":
+ return f"<{_safe_iri_ref(lexical)}>"
+ if datatype is None:
+ return f'"{_escape_literal(lexical)}"'
+ return f'"{lexical}"^^<{datatype}>'
+
+
+def _turtle_metadata_clauses(statements: List[tuple]) -> List[str]:
+ return [f"<{term}> {_turtle_object(term, value)}" for term, value in statements]
+
+
+def _ntriples_metadata_lines(subject: str, statements: List[tuple]) -> List[str]:
+ return [
+ f"<{subject}> <{term}> {_turtle_object(term, value)} ."
+ for term, value in statements
+ ]
+
+
+def _rdfxml_metadata_lines(
+ statements: List[tuple], indent: str, logger: Any = None
+) -> List[str]:
+ """RDF/XML needs a QName, so an unprefixed term declares its own prefix.
+
+ A term with no QName form has no RDF/XML representation at all, and this is
+ the only serialization with that restriction. Skipping it quietly would
+ reintroduce, in one format, exactly the silent metadata loss this module
+ was changed to stop, so it is reported and the other three formats still
+ carry the statement in full.
+ """
+ lines: List[str] = []
+ for position, (term, value) in enumerate(statements):
+ split = _split_iri(term)
+ if split is None:
+ if logger is not None:
+ logger.warning(
+ "Term %r has no QName form, so it cannot be written in "
+ "RDF/XML and was omitted from that serialization only. "
+ "Turtle, N-Triples and JSON-LD carry it in full.",
+ term,
+ )
+ continue
+ namespace, local = split
+ kind, lexical, datatype = _typed_literal_parts(term, value)
+ prefix = f"md{position}"
+ opening = f'{indent}<{prefix}:{local} xmlns:{prefix}="{_escape_xml(namespace)}"'
+ if kind == "iri":
+ lines.append(f'{opening} rdf:resource="{_escape_xml(lexical)}"/>')
+ continue
+ if datatype is not None:
+ opening += f' rdf:datatype="{_escape_xml(datatype)}"'
+ lines.append(f"{opening}>{_escape_xml(lexical)}{prefix}:{local}>")
+ return lines
+
+
+def _jsonld_metadata_entries(statements: List[tuple]) -> Dict[str, Any]:
+ """Absolute IRIs as keys, and explicit @value/@type rather than JSON's own
+ types: JSON's number is xsd:double, which would make the JSON-LD export
+ disagree with the other three about the datatype of an integer."""
+ entries: Dict[str, Any] = {}
+ for term, value in statements:
+ kind, lexical, datatype = _typed_literal_parts(term, value)
+ if kind == "iri":
+ entries[term] = {"@id": lexical}
+ elif datatype is None:
+ entries[term] = lexical
+ else:
+ entries[term] = {"@value": lexical, "@type": datatype}
+ return entries
+
+
class NamespaceManager:
"""
RDF namespace management engine.
@@ -501,6 +770,8 @@ class RDFSerializer:
"""
include_temporal: bool = options.pop("include_temporal", False)
time_axis: str = options.pop("time_axis", "valid")
+ metadata_terms = _resolve_metadata_terms(options.pop("metadata_terms", None))
+ graph_uri: Optional[str] = options.pop("graph_uri", None)
lines = []
@@ -534,21 +805,32 @@ class RDFSerializer:
text = entity.get("text") or entity.get("label", "")
confidence = normalize_confidence(entity.get("confidence", 1.0))
- lines.append(
- f"<{self._as_turtle_iri(entity_id, merged_namespaces)}> a "
- f"<{self._as_turtle_iri(entity_type, merged_namespaces)}> ;"
- )
+ clauses = [
+ f"a <{self._as_turtle_iri(entity_type, merged_namespaces)}>",
+ f'semantica:text "{text}"',
+ ]
if confidence is None:
self.logger.warning(
f"Entity {entity_id} has a confidence that is not a number "
f"({entity.get('confidence')!r}), so no confidence is written"
)
- lines.append(f' semantica:text "{text}" .')
else:
- lines.append(f' semantica:text "{text}" ;')
- lines.append(
- f' semantica:confidence "{confidence}"^^<{CONFIDENCE_DATATYPE}> .'
+ clauses.append(
+ f'semantica:confidence "{confidence}"^^<{CONFIDENCE_DATATYPE}>'
)
+ clauses.extend(
+ _turtle_metadata_clauses(
+ _metadata_statements(
+ entity.get("metadata"), metadata_terms, self.logger
+ )
+ )
+ )
+
+ entity_iri = self._as_turtle_iri(entity_id, merged_namespaces)
+ lines.append(f"<{entity_iri}> {clauses[0]} ;")
+ for clause in clauses[1:-1]:
+ lines.append(f" {clause} ;")
+ lines.append(f" {clauses[-1]} .")
lines.append("")
# Convert relationships to RDF triplets
@@ -582,6 +864,31 @@ class RDFSerializer:
)
lines.extend(owl_lines)
+ # Graph-level metadata needs a subject, and this serializer has never
+ # minted a document node. Rather than invent one here, it is written
+ # only when the caller names the graph; issue #1147 is where the
+ # default subject comes from once that lands.
+ graph_clauses = (
+ _turtle_metadata_clauses(
+ _metadata_statements(
+ rdf_data.get("metadata"), metadata_terms, self.logger
+ )
+ )
+ if graph_uri
+ else []
+ )
+ if graph_clauses:
+ graph_iri = self._as_turtle_iri(graph_uri, merged_namespaces)
+ lines.append("")
+ lines.append(
+ f"<{graph_iri}> {graph_clauses[0]} "
+ + (";" if len(graph_clauses) > 1 else ".")
+ )
+ for clause in graph_clauses[1:-1]:
+ lines.append(f" {clause} ;")
+ if len(graph_clauses) > 1:
+ lines.append(f" {graph_clauses[-1]} .")
+
return "\n".join(lines)
def _reified_relationship_triples(
@@ -730,6 +1037,9 @@ class RDFSerializer:
... }
>>> rdfxml = serializer.serialize_to_rdfxml(rdf_data)
"""
+ metadata_terms = _resolve_metadata_terms(options.pop("metadata_terms", None))
+ graph_uri: Optional[str] = options.pop("graph_uri", None)
+
lines = ['']
lines.append(''
f"{confidence}"
)
+ lines.extend(
+ _rdfxml_metadata_lines(
+ _metadata_statements(
+ entity.get("metadata"), metadata_terms, self.logger
+ ),
+ " ",
+ self.logger,
+ )
+ )
lines.append(" ")
lines.append("")
@@ -795,6 +1117,26 @@ class RDFSerializer:
lines.append(" ")
lines.append("")
+ graph_lines = (
+ _rdfxml_metadata_lines(
+ _metadata_statements(
+ rdf_data.get("metadata"), metadata_terms, self.logger
+ ),
+ " ",
+ self.logger,
+ )
+ if graph_uri
+ else []
+ )
+ if graph_lines:
+ graph_iri = xml_escape(
+ self._as_turtle_iri(graph_uri, namespaces), quote=True
+ )
+ lines.append(f' ')
+ lines.extend(graph_lines)
+ lines.append(" ")
+ lines.append("")
+
lines.append("")
return "\n".join(lines)
@@ -825,6 +1167,9 @@ class RDFSerializer:
"""
import json
+ metadata_terms = _resolve_metadata_terms(options.pop("metadata_terms", None))
+ graph_uri: Optional[str] = options.pop("graph_uri", None)
+
# Initialize JSON-LD structure with context
jsonld = {
"@context": {
@@ -869,6 +1214,13 @@ class RDFSerializer:
"@value": confidence,
"@type": CONFIDENCE_DATATYPE,
}
+ node.update(
+ _jsonld_metadata_entries(
+ _metadata_statements(
+ entity.get("metadata"), metadata_terms, self.logger
+ )
+ )
+ )
jsonld["@graph"].append(node)
# Convert relationships to JSON-LD
@@ -893,6 +1245,18 @@ class RDFSerializer:
}
)
+ graph_entries = (
+ _jsonld_metadata_entries(
+ _metadata_statements(
+ rdf_data.get("metadata"), metadata_terms, self.logger
+ )
+ )
+ if graph_uri
+ else {}
+ )
+ if graph_entries:
+ jsonld["@graph"].append({"@id": graph_uri, **graph_entries})
+
return json.dumps(jsonld, indent=2, ensure_ascii=False)
def serialize_to_ntriples(self, rdf_data: Dict[str, Any], **options) -> str:
@@ -909,6 +1273,9 @@ class RDFSerializer:
Returns:
String containing N-Triples serialization
"""
+ metadata_terms = _resolve_metadata_terms(options.pop("metadata_terms", None))
+ graph_uri: Optional[str] = options.pop("graph_uri", None)
+
lines = []
namespaces = self.namespace_manager.extract_namespaces(rdf_data)
@@ -959,6 +1326,15 @@ class RDFSerializer:
f'"{confidence}"^^<{CONFIDENCE_DATATYPE}> .'
)
+ lines.extend(
+ _ntriples_metadata_lines(
+ subject.strip("<>"),
+ _metadata_statements(
+ entity.get("metadata"), metadata_terms, self.logger
+ ),
+ )
+ )
+
# Convert relationships
relationships = rdf_data.get("relationships", [])
for rel in relationships:
@@ -971,6 +1347,16 @@ class RDFSerializer:
f"{expand_uri(source_id)} {expand_uri(rel_type)} {expand_uri(target_id)} ."
)
+ if graph_uri:
+ lines.extend(
+ _ntriples_metadata_lines(
+ graph_uri,
+ _metadata_statements(
+ rdf_data.get("metadata"), metadata_terms, self.logger
+ ),
+ )
+ )
+
return "\n".join(lines)
diff --git a/semantica/ontology/vocabulary/semantica-ns.ttl b/semantica/ontology/vocabulary/semantica-ns.ttl
index 42badf60..ffe36df5 100644
--- a/semantica/ontology/vocabulary/semantica-ns.ttl
+++ b/semantica/ontology/vocabulary/semantica-ns.ttl
@@ -135,6 +135,90 @@ JSONExporter.export_to_jsonld in export/json_exporter.py.""" ;
rdfs:range xsd:string ;
rdfs:isDefinedBy .
+# โโ Metadata carried through from the graph builder โโโโโโโโโโโโโโโโโโโโโโโโโโ
+#
+# The keys GraphBuilder and the Neo4j loader write into "metadata". Declared
+# here because the RDF serializers emit them (#1154); a caller-supplied key is
+# not declared here and is not emitted, because which namespace it belongs in
+# is #1146.
+
+sem:numEntities a owl:DatatypeProperty ;
+ rdfs:label "number of entities" ;
+ rdfs:comment """Count of entities in the graph as built, from
+GraphBuilder.build_graph. A count of what was built, not a constraint on what
+the graph contains: an export filtered after the fact will disagree with it.""" ;
+ rdfs:range xsd:integer ;
+ rdfs:isDefinedBy .
+
+sem:numRelationships a owl:DatatypeProperty ;
+ rdfs:label "number of relationships" ;
+ rdfs:comment "Count of relationships in the graph as built." ;
+ rdfs:range xsd:integer ;
+ rdfs:isDefinedBy .
+
+sem:temporalEnabled a owl:DatatypeProperty ;
+ rdfs:label "temporal enabled" ;
+ rdfs:comment """True when the builder was configured to track valid time.
+False does not mean the graph is untimed; it means no temporal bounds were
+recorded for it.""" ;
+ rdfs:range xsd:boolean ;
+ rdfs:isDefinedBy .
+
+sem:entityResolutionApplied a owl:DatatypeProperty ;
+ rdfs:label "entity resolution applied" ;
+ rdfs:comment """True when a resolver ran over the extracted entities, so a
+consumer knows whether two nodes with the same surface text were ever
+considered for merging.""" ;
+ rdfs:range xsd:boolean ;
+ rdfs:isDefinedBy .
+
+sem:builtAt a owl:DatatypeProperty ;
+ rdfs:label "built at" ;
+ rdfs:comment """When the graph was built, as GraphBuilder recorded it.
+
+The range is xsd:string, deliberately, and not xsd:dateTime. GraphBuilder
+stamps with a timezone-naive datetime.now(), and #1114 is the demonstration of
+what typing such a value as xsd:dateTime costs: a timezone-qualified SPARQL
+filter over it raises an indeterminate comparison and silently drops the row.
+#1121 swept the export and provenance modules to an explicit UTC offset and
+deliberately left kg/ alone, because the context and vector-store modules
+compare against naive values already on disk. Until that sweep reaches
+GraphBuilder this value is a string that looks like a timestamp, and saying so
+is more useful than a type that invites arithmetic it cannot support.""" ;
+ rdfs:range xsd:string ;
+ rdfs:isDefinedBy .
+
+sem:snapshotAt a owl:DatatypeProperty ;
+ rdfs:label "snapshot at" ;
+ rdfs:comment """The point in time a snapshot represents, from
+GraphBuilder.create_snapshot. A string for the same reason as sem:builtAt.""" ;
+ rdfs:range xsd:string ;
+ rdfs:isDefinedBy .
+
+sem:sourceSystem a owl:DatatypeProperty ;
+ rdfs:label "source system" ;
+ rdfs:comment """The system a graph was loaded from, currently the literal
+"neo4j" written by GraphBuilder.load_from_neo4j.
+
+Named sourceSystem rather than source because sem:source is already the
+ObjectProperty carrying the subject of a reified relationship. The metadata key
+is still "source"; the exporter maps the key to this term.""" ;
+ rdfs:range xsd:string ;
+ rdfs:isDefinedBy .
+
+sem:sourceUri a owl:ObjectProperty ;
+ rdfs:label "source URI" ;
+ rdfs:comment """The address of the system a graph was loaded from. The one
+metadata term whose value is a node rather than a literal, because it names a
+thing rather than describing one.""" ;
+ rdfs:isDefinedBy .
+
+sem:sourceDatabase a owl:DatatypeProperty ;
+ rdfs:label "source database" ;
+ rdfs:comment "The database within the source system a graph was loaded from." ;
+ rdfs:range xsd:string ;
+ rdfs:isDefinedBy .
+
# โโ Temporal term (OWL-Time export) โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ
sem:openEndedInterval a owl:DatatypeProperty ;
diff --git a/semantica/triplet_store/rdf4j_store.py b/semantica/triplet_store/rdf4j_store.py
index 8dad6997..d788ab7b 100644
--- a/semantica/triplet_store/rdf4j_store.py
+++ b/semantica/triplet_store/rdf4j_store.py
@@ -28,7 +28,7 @@ License: MIT
import re
from typing import Any, Dict, List, Optional
-from urllib.parse import urlparse
+from urllib.parse import quote, urlparse
import requests
from rdflib import Graph, Literal
@@ -67,7 +67,8 @@ class RDF4JStore:
self.progress_tracker.enabled = True
self.endpoint = endpoint.rstrip("/")
- self.repository_id = config.get("repository_id", "default")
+ self.repository_id = repository_id or config.get("repository_id", "default")
+ self._encoded_repository_id = quote(self.repository_id, safe="")
self.username = config.get("username")
self.password = config.get("password")
self.timeout = config.get("timeout", 30)
@@ -79,7 +80,7 @@ class RDF4JStore:
"""Connect to RDF4J server."""
try:
# Test connection
- test_url = f"{self.endpoint}/repositories/{self.repository_id}"
+ test_url = f"{self.endpoint}/repositories/{self._encoded_repository_id}"
response = requests.get(
test_url,
timeout=self.timeout,
@@ -100,11 +101,11 @@ class RDF4JStore:
def _get_sparql_endpoint(self) -> str:
"""Get SPARQL query endpoint."""
- return f"{self.endpoint}/repositories/{self.repository_id}"
+ return f"{self.endpoint}/repositories/{self._encoded_repository_id}"
def _get_update_endpoint(self) -> str:
"""Get SPARQL Update endpoint."""
- return f"{self.endpoint}/repositories/{self.repository_id}/statements"
+ return f"{self.endpoint}/repositories/{self._encoded_repository_id}/statements"
def _is_construct_query(self, query: str) -> bool:
"""
@@ -163,7 +164,7 @@ class RDF4JStore:
"""
# RDF4J transaction support
transaction_url = (
- f"{self.endpoint}/repositories/{self.repository_id}/transactions"
+ f"{self.endpoint}/repositories/{self._encoded_repository_id}/transactions"
)
try:
diff --git a/semantica/utils/progress_tracker.py b/semantica/utils/progress_tracker.py
index febfb4b9..f27768d4 100644
--- a/semantica/utils/progress_tracker.py
+++ b/semantica/utils/progress_tracker.py
@@ -64,6 +64,29 @@ def _progress_disabled_from_env() -> bool:
"on",
)
+
+def _progress_forced_from_env() -> bool:
+ """Return whether console progress is forced on despite a non-interactive stdout."""
+ return os.getenv("SEMANTICA_FORCE_PROGRESS", "").strip().lower() in (
+ "1",
+ "true",
+ "yes",
+ "on",
+ )
+
+
+def _stdout_is_tty() -> bool:
+ """Return whether stdout is an interactive terminal.
+
+ Replacement streams do not always implement ``isatty`` and closed streams can
+ raise, so both cases are treated as non-interactive.
+ """
+ try:
+ return bool(sys.stdout is not None and sys.stdout.isatty())
+ except (AttributeError, ValueError):
+ return False
+
+
# Try to import IPython for Jupyter support
try:
from IPython import get_ipython
@@ -1040,18 +1063,24 @@ class ProgressTracker:
# Create displays
self.displays: List[ProgressDisplay] = []
+ # Console output only suits an interactive stdout. When output is piped or
+ # redirected (scripts, CI logs) the progress bars and their escape
+ # sequences would otherwise drown the program's own output.
+ console_ok = _stdout_is_tty() or self.is_jupyter or _progress_forced_from_env()
+
# Always try Jupyter first if available, fallback to console
if IPYTHON_AVAILABLE:
# Try to detect Jupyter - if available, use it
if self.is_jupyter and not self.disable_jupyter_progress:
self.displays.append(JupyterProgressDisplay(use_emoji=use_emoji))
# Also add console as fallback for immediate feedback
- self.displays.append(
- ConsoleProgressDisplay(
- use_emoji=use_emoji, update_interval=update_interval
+ if console_ok:
+ self.displays.append(
+ ConsoleProgressDisplay(
+ use_emoji=use_emoji, update_interval=update_interval
+ )
)
- )
- else:
+ elif console_ok:
self.displays.append(
ConsoleProgressDisplay(
use_emoji=use_emoji, update_interval=update_interval
diff --git a/tests/export/test_metadata_passthrough.py b/tests/export/test_metadata_passthrough.py
new file mode 100644
index 00000000..ff4c3ddc
--- /dev/null
+++ b/tests/export/test_metadata_passthrough.py
@@ -0,0 +1,401 @@
+"""Metadata must survive serialization (issue #1154).
+
+``convert_kg_to_rdf`` copies ``metadata`` into the RDF-ready dictionary at
+rdf_exporter.py:302, and no serializer has ever read it back out. Turtle,
+N-Triples, RDF/XML and RDFExporter's JSON-LD all write the entity's id, type,
+text and confidence, and none of them writes a single metadata statement, so an
+entity keeps its confidence score and loses what produced it: the source
+document, the page, the extractor, the reviewer. JSONExporter's json-ld path
+keeps all of them, which is how the same knowledge graph exported two ways came
+to carry ten triples of user data through one exporter and none through the
+other.
+
+The keys Semantica itself produces (GraphBuilder writes num_entities,
+num_relationships, temporal_enabled, timestamp and entity_resolution_applied;
+the Neo4j loader writes source, uri and database) are Semantica's own
+vocabulary, so they are minted in the declared namespace and declared in
+semantica-ns.ttl. Keys the caller supplied are not: which namespace those
+belong in is issue #1146, and until that is settled the exporter refuses to
+guess rather than inventing an IRI, warns, and takes an explicit
+``metadata_terms`` mapping from any caller who already knows the answer.
+"""
+
+import json
+
+import pytest
+from rdflib import Graph, Literal, URIRef
+from rdflib.namespace import XSD
+
+from semantica.export.rdf_exporter import (
+ DEFAULT_METADATA_TERMS,
+ RDFSerializer,
+ SEMANTICA_NS,
+ mint_entity_iri,
+)
+
+ENTITY_IRI = "https://example.org/e1"
+
+# The provenance fields the issue names, plus one key Semantica itself writes.
+GRAPH_WITH_METADATA = {
+ "entities": [
+ {
+ "id": ENTITY_IRI,
+ "type": "https://example.org/Org",
+ "text": "Acme Corp",
+ "confidence": 0.91,
+ "metadata": {"num_entities": 1, "temporal_enabled": True},
+ }
+ ],
+ "relationships": [],
+ "metadata": {
+ "num_entities": 1,
+ "num_relationships": 0,
+ "temporal_enabled": False,
+ "entity_resolution_applied": True,
+ },
+}
+
+NUM_ENTITIES = URIRef(f"{SEMANTICA_NS}numEntities")
+TEMPORAL_ENABLED = URIRef(f"{SEMANTICA_NS}temporalEnabled")
+
+
+def _parse(text: str, fmt: str) -> Graph:
+ """Assert on the parsed graph, never on the serialized text."""
+ g = Graph()
+ g.parse(data=text, format=fmt)
+ return g
+
+
+def _serialize(serializer: RDFSerializer, fmt: str, data, **options) -> Graph:
+ method, parse_as = {
+ "turtle": (serializer.serialize_to_turtle, "turtle"),
+ "ntriples": (serializer.serialize_to_ntriples, "nt"),
+ "rdfxml": (serializer.serialize_to_rdfxml, "xml"),
+ "jsonld": (serializer.serialize_to_jsonld, "json-ld"),
+ }[fmt]
+ return _parse(method(data, **options), parse_as)
+
+
+FORMATS = ["turtle", "ntriples", "rdfxml", "jsonld"]
+
+
+@pytest.mark.parametrize("fmt", FORMATS)
+def test_entity_metadata_reaches_every_serialization(fmt):
+ """The headline defect: the statement is absent from all four formats."""
+ g = _serialize(RDFSerializer(), fmt, GRAPH_WITH_METADATA)
+ assert (URIRef(ENTITY_IRI), NUM_ENTITIES, Literal(1)) in g
+
+
+@pytest.mark.parametrize("fmt", FORMATS)
+def test_entity_metadata_booleans_keep_their_datatype(fmt):
+ g = _serialize(RDFSerializer(), fmt, GRAPH_WITH_METADATA)
+ assert (URIRef(ENTITY_IRI), TEMPORAL_ENABLED, Literal(True)) in g
+
+
+def test_every_format_writes_the_same_metadata_triples():
+ """A value must not change datatype with the serializer, as #1100 found."""
+ per_format = {}
+ for fmt in FORMATS:
+ g = _serialize(RDFSerializer(), fmt, GRAPH_WITH_METADATA)
+ per_format[fmt] = {
+ (p, o)
+ for s, p, o in g
+ if str(p).startswith(SEMANTICA_NS) and "numEntities" in str(p)
+ }
+ assert len(set(map(frozenset, per_format.values()))) == 1, per_format
+
+
+def test_graph_metadata_needs_a_subject_the_caller_named():
+ """Graph-level metadata hangs off graph_uri; #1147 owns the default."""
+ doc = URIRef("https://example.org/graph/1")
+ g = _serialize(
+ RDFSerializer(),
+ "turtle",
+ GRAPH_WITH_METADATA,
+ graph_uri=str(doc),
+ )
+ assert (doc, NUM_ENTITIES, Literal(1)) in g
+ assert (doc, URIRef(f"{SEMANTICA_NS}entityResolutionApplied"), Literal(True)) in g
+
+
+def test_graph_metadata_is_not_invented_without_a_subject():
+ g = _serialize(RDFSerializer(), "turtle", GRAPH_WITH_METADATA)
+ assert not list(g.subjects(NUM_ENTITIES, Literal(0)))
+ # the entity keeps its own metadata; only the graph-level block waits
+ assert (URIRef(ENTITY_IRI), NUM_ENTITIES, Literal(1)) in g
+
+
+def test_an_unknown_key_is_refused_out_loud_not_dropped_in_silence(caplog):
+ """#1146 owns which namespace a caller's key belongs in. Until then: warn."""
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"reviewed_by": "fabio"}}
+ ],
+ "relationships": [],
+ }
+ with caplog.at_level("WARNING"):
+ g = _serialize(RDFSerializer(), "turtle", data)
+ assert not any("reviewed_by" in str(p) for p in g.predicates())
+ assert any("reviewed_by" in r.getMessage() for r in caplog.records)
+ assert any("1146" in r.getMessage() for r in caplog.records)
+
+
+@pytest.mark.parametrize("fmt", FORMATS)
+def test_a_caller_who_knows_the_answer_can_supply_the_term(fmt):
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"reviewed_by": "fabio"}}
+ ],
+ "relationships": [],
+ }
+ terms = {"reviewed_by": "http://purl.org/dc/terms/creator"}
+ g = _serialize(RDFSerializer(), fmt, data, metadata_terms=terms)
+ assert (
+ URIRef(ENTITY_IRI),
+ URIRef("http://purl.org/dc/terms/creator"),
+ Literal("fabio"),
+ ) in g
+
+
+def test_a_literal_with_a_quote_or_newline_still_parses():
+ """Metadata is user text; #1098 is the same class of defect one field over."""
+ data = {
+ "entities": [
+ {
+ "id": ENTITY_IRI,
+ "text": "Acme",
+ "metadata": {"source": 'the "Q3" report\nsecond line'},
+ }
+ ],
+ "relationships": [],
+ }
+ for fmt in FORMATS:
+ g = _serialize(RDFSerializer(), fmt, data)
+ assert (
+ URIRef(ENTITY_IRI),
+ URIRef(f"{SEMANTICA_NS}sourceSystem"),
+ Literal('the "Q3" report\nsecond line'),
+ ) in g
+
+
+def test_an_iri_valued_key_is_written_as_a_node_not_a_string():
+ """The Neo4j loader's ``uri`` key. Note the term is sem:sourceUri, not
+ sem:uri: the key names a field, the term names a relation."""
+ data = {
+ "entities": [
+ {
+ "id": ENTITY_IRI,
+ "text": "Acme",
+ "metadata": {"uri": "https://example.org/db"},
+ }
+ ],
+ "relationships": [],
+ }
+ g = _serialize(RDFSerializer(), "turtle", data)
+ assert (
+ URIRef(ENTITY_IRI),
+ URIRef(f"{SEMANTICA_NS}sourceUri"),
+ URIRef("https://example.org/db"),
+ ) in g
+
+
+def test_output_is_unchanged_when_no_metadata_is_present():
+ plain = {
+ "entities": [
+ {"id": ENTITY_IRI, "type": "https://example.org/Org", "text": "Acme"}
+ ],
+ "relationships": [
+ {"source_id": ENTITY_IRI, "target_id": "https://example.org/e2"}
+ ],
+ }
+ serializer = RDFSerializer()
+ assert serializer.serialize_to_turtle(plain) == serializer.serialize_to_turtle(
+ plain
+ )
+ g = _parse(serializer.serialize_to_turtle(plain), "turtle")
+ assert len(g) == 4
+
+
+def test_every_default_term_is_declared_in_the_shipped_vocabulary():
+ """Drift guard: a term the exporter emits and the vocabulary omits is a bug."""
+ from semantica.ontology.vocabulary import vocabulary_path
+
+ vocab = Graph()
+ vocab.parse(vocabulary_path(), format="turtle")
+ declared = {str(s) for s in vocab.subjects()}
+ missing = sorted(set(DEFAULT_METADATA_TERMS.values()) - declared)
+ assert not missing, f"emitted but undeclared: {missing}"
+
+
+def test_jsonld_metadata_survives_a_real_jsonld_processor():
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"num_entities": 3}}
+ ],
+ "relationships": [],
+ }
+ raw = RDFSerializer().serialize_to_jsonld(data)
+ json.loads(raw) # must be valid JSON before it can be valid JSON-LD
+ g = _parse(raw, "json-ld")
+ assert (URIRef(ENTITY_IRI), NUM_ENTITIES, Literal(3)) in g
+
+
+# --- Findings from the Qodo review of PR #1165 -----------------------------
+
+
+@pytest.mark.parametrize("fmt", FORMATS)
+@pytest.mark.parametrize(
+ "value", [1e-05, 1e300, 0.1, -0.0, float("nan"), float("inf"), float("-inf")]
+)
+def test_a_float_metadata_value_is_a_double_and_keeps_a_legal_lexical(fmt, value):
+ """`repr()` of a float is not an xsd:decimal lexical.
+
+ `repr(1e-05)` is "1e-05" and `repr(float("nan"))` is "nan", neither of which
+ xsd:decimal admits, so typing a float as decimal produced RDF a strict
+ parser rejects. A Python float is an IEEE 754 double, xsd:double has legal
+ lexicals for the exponent form and for the three special values, and saying
+ double is also the honest claim: nothing here was ever exact.
+ """
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"num_entities": value}}
+ ],
+ "relationships": [],
+ }
+ g = _serialize(RDFSerializer(), fmt, data)
+ objects = list(g.objects(URIRef(ENTITY_IRI), NUM_ENTITIES))
+ assert len(objects) == 1, f"{fmt}: {objects}"
+ (written,) = objects
+ assert written.datatype == XSD.double, written.datatype
+ parsed = written.toPython()
+ if value != value: # NaN
+ assert parsed != parsed
+ else:
+ assert parsed == value
+
+
+def test_every_format_agrees_on_a_float_metadata_value():
+ per_format = {}
+ for fmt in FORMATS:
+ g = _serialize(
+ RDFSerializer(),
+ fmt,
+ {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "A", "metadata": {"num_entities": 1e-05}}
+ ],
+ "relationships": [],
+ },
+ )
+ per_format[fmt] = {(p, o) for s, p, o in g if p == NUM_ENTITIES}
+ assert len(set(map(frozenset, per_format.values()))) == 1, per_format
+
+
+def test_a_term_rdfxml_cannot_name_is_refused_out_loud(caplog):
+ """RDF/XML needs a QName, and the PR's whole point is no silent drops.
+
+ A term whose local part is not an XML NCName has no RDF/XML form at all.
+ Skipping it quietly reintroduces, in one format, exactly the loss this
+ change exists to stop.
+ """
+ unnameable = "http://example.org/ns/123"
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"reviewed_by": "fabio"}}
+ ],
+ "relationships": [],
+ }
+ with caplog.at_level("WARNING"):
+ xml = RDFSerializer().serialize_to_rdfxml(
+ data, metadata_terms={"reviewed_by": unnameable}
+ )
+ _parse(xml, "xml") # must still be well-formed
+ messages = " ".join(r.getMessage() for r in caplog.records)
+ assert unnameable in messages
+ assert "RDF/XML" in messages
+
+
+@pytest.mark.parametrize("fmt", ["turtle", "ntriples", "jsonld"])
+def test_the_other_formats_still_carry_a_term_rdfxml_cannot_name(fmt):
+ """Only RDF/XML has the QName restriction; the rest write the full IRI."""
+ unnameable = "http://example.org/ns/123"
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"reviewed_by": "fabio"}}
+ ],
+ "relationships": [],
+ }
+ g = _serialize(
+ RDFSerializer(), fmt, data, metadata_terms={"reviewed_by": unnameable}
+ )
+ assert (URIRef(ENTITY_IRI), URIRef(unnameable), Literal("fabio")) in g
+
+
+def test_a_quote_in_an_attribute_value_cannot_break_the_document():
+ """`_escape_xml` feeds attribute values, which are delimited by quotes.
+
+ Escaping only &, < and > leaves a caller-supplied value able to close the
+ attribute early and produce XML that does not parse.
+ """
+ data = {
+ "entities": [
+ {
+ "id": 'https://example.org/e"1',
+ "text": "Acme",
+ "metadata": {"uri": 'https://example.org/db"x'},
+ }
+ ],
+ "relationships": [],
+ }
+ xml = RDFSerializer().serialize_to_rdfxml(data)
+ from xml.dom.minidom import parseString
+
+ parseString(xml) # well-formedness is the assertion
+
+
+# --- Finding from review of PR #1165 ----------------------------------------
+
+
+@pytest.mark.parametrize("fmt", ["turtle", "ntriples"])
+def test_an_iri_valued_metadata_value_cannot_inject_a_second_triple(fmt):
+ """`sem:sourceUri` (the "uri" key) is the one metadata term written as a
+ node, ``<{value}>``, with no other quoting. Turtle/N-Triples IRIREFs
+ exclude '>' (among other characters) unescaped, so a value shaped like
+ `` . `` closed the reference early and let the
+ rest of the string be parsed as an unrelated, attacker-chosen triple.
+ """
+ payload = (
+ "https://evil.example/x> . "
+ " ' โ cover the control-character half of
+ the grammar, not only the delimiter characters.
+ """
+ payload = "https://evil.example/x\ninjected line\ttabbed"
+ data = {
+ "entities": [
+ {"id": ENTITY_IRI, "text": "Acme", "metadata": {"uri": payload}}
+ ],
+ "relationships": [],
+ }
+ g = _serialize(RDFSerializer(), fmt, data)
+ assert len(g) == 4
diff --git a/tests/ingest/test_notebook_02.py b/tests/ingest/test_notebook_02.py
index 0cbb6a52..f99a8b37 100644
--- a/tests/ingest/test_notebook_02.py
+++ b/tests/ingest/test_notebook_02.py
@@ -157,7 +157,7 @@ class TestNotebook02DataIngestion:
repo_ingestor = RepoIngestor()
with patch.object(repo_ingestor, 'ingest_repository') as mock_ingest:
mock_ingest.return_value = {'name': 'semantica'}
- repo_data = repo_ingestor.ingest_repository("https://github.com/Hawksight-AI/semantica.git")
+ repo_data = repo_ingestor.ingest_repository("https://github.com/semantica-agi/semantica.git")
assert repo_data['name'] == 'semantica'
def test_07_email_ingestion(self):
diff --git a/tests/test_progress_tracker_regressions.py b/tests/test_progress_tracker_regressions.py
index ac4c09da..b42a885b 100644
--- a/tests/test_progress_tracker_regressions.py
+++ b/tests/test_progress_tracker_regressions.py
@@ -12,6 +12,7 @@ import semantica.utils.progress_tracker as progress_module
@pytest.fixture(autouse=True)
def reset_progress_singletons(monkeypatch):
monkeypatch.delenv("SEMANTICA_DISABLE_PROGRESS", raising=False)
+ monkeypatch.delenv("SEMANTICA_FORCE_PROGRESS", raising=False)
progress_module.ProgressTracker._instance = None
progress_module._global_tracker = None
yield
@@ -19,6 +20,40 @@ def reset_progress_singletons(monkeypatch):
progress_module._global_tracker = None
+class _FakeStdout:
+ """Minimal stdout stand-in with controllable TTY reporting."""
+
+ encoding = "utf-8"
+
+ def __init__(self, tty):
+ self._tty = tty
+ self.written = []
+
+ def isatty(self):
+ return self._tty
+
+ def write(self, text):
+ self.written.append(text)
+ return len(text)
+
+ def flush(self):
+ pass
+
+
+def _use_stdout(monkeypatch, tty):
+ """Point sys.stdout at a fake with the given TTY behaviour, outside Jupyter."""
+ stream = _FakeStdout(tty=tty)
+ monkeypatch.setattr(sys, "stdout", stream)
+ monkeypatch.setattr(
+ progress_module.ProgressTracker, "_detect_jupyter", lambda *_: False
+ )
+ return stream
+
+
+def _displays_of(tracker, display_cls):
+ return [d for d in tracker.displays if isinstance(d, display_cls)]
+
+
def _install_tracker_as_singleton(tracker: progress_module.ProgressTracker) -> None:
progress_module.ProgressTracker._instance = tracker
progress_module._global_tracker = tracker
@@ -100,6 +135,67 @@ def test_disable_progress_env_prevents_reenable(monkeypatch):
assert tracker.start_tracking(module="core", submodule="test") == ""
+def test_console_display_omitted_when_stdout_is_not_a_tty(monkeypatch):
+ _use_stdout(monkeypatch, tty=False)
+
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+
+ assert _displays_of(tracker, progress_module.ConsoleProgressDisplay) == []
+
+
+def test_console_display_present_when_stdout_is_a_tty(monkeypatch):
+ _use_stdout(monkeypatch, tty=True)
+
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+
+ assert _displays_of(tracker, progress_module.ConsoleProgressDisplay)
+
+
+def test_file_display_survives_non_tty_stdout(monkeypatch):
+ _use_stdout(monkeypatch, tty=False)
+
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+
+ assert _displays_of(tracker, progress_module.FileProgressDisplay)
+
+
+def test_force_progress_env_restores_console_display_on_non_tty(monkeypatch):
+ monkeypatch.setenv("SEMANTICA_FORCE_PROGRESS", "1")
+ _use_stdout(monkeypatch, tty=False)
+
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+
+ assert _displays_of(tracker, progress_module.ConsoleProgressDisplay)
+
+
+def test_disable_progress_env_beats_force_progress_env(monkeypatch):
+ monkeypatch.setenv("SEMANTICA_DISABLE_PROGRESS", "1")
+ monkeypatch.setenv("SEMANTICA_FORCE_PROGRESS", "1")
+ stream = _use_stdout(monkeypatch, tty=False)
+
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+ _install_tracker_as_singleton(tracker)
+
+ assert tracker.enabled is False
+ assert tracker.start_tracking(module="core", submodule="test") == ""
+ assert stream.written == []
+
+
+def test_non_tty_stdout_stays_silent_after_module_reenables_tracker(monkeypatch):
+ stream = _use_stdout(monkeypatch, tty=False)
+ tracker = progress_module.ProgressTracker(use_emoji=False, update_interval=0)
+ _install_tracker_as_singleton(tracker)
+
+ # Mirrors the ~20 modules that do `self.progress_tracker.enabled = True`.
+ tracker.enabled = True
+ tracking_id = tracker.start_tracking(
+ module="core", submodule="Semantica", message="Building"
+ )
+ tracker.update_progress(tracking_id, processed=1, total=1, message="Processing")
+
+ assert stream.written == []
+
+
def test_build_knowledge_base_subprocess_does_not_deadlock():
root = Path(__file__).resolve().parents[1]
runtime_dir = root / "test_data" / "runtime" / f"build-regression-{os.getpid()}"
diff --git a/tests/triplet_store/test_rdf4j_store.py b/tests/triplet_store/test_rdf4j_store.py
index 4ef4f16b..03a9b90b 100644
--- a/tests/triplet_store/test_rdf4j_store.py
+++ b/tests/triplet_store/test_rdf4j_store.py
@@ -22,6 +22,71 @@ def _make_connected_store():
CONSTRUCT_QUERY = "CONSTRUCT { ?s ?p ?o } WHERE { ?s ?p ?o }"
+class TestRDF4JStoreInitialization(unittest.TestCase):
+
+ def test_explicit_repository_id_selects_repository(self):
+ response = MagicMock(status_code=200)
+
+ with patch(
+ "semantica.triplet_store.rdf4j_store.requests.get",
+ return_value=response,
+ ) as mock_get:
+ store = RDF4JStore(
+ endpoint="http://localhost:8080/rdf4j-server/",
+ repository_id="semantica",
+ )
+
+ self.assertEqual(store.repository_id, "semantica")
+ mock_get.assert_called_once_with(
+ "http://localhost:8080/rdf4j-server/repositories/semantica",
+ timeout=30,
+ auth=None,
+ )
+
+ def test_repository_id_is_encoded_as_a_single_url_path_segment(self):
+ response = MagicMock(status_code=200)
+
+ with patch(
+ "semantica.triplet_store.rdf4j_store.requests.get",
+ return_value=response,
+ ) as mock_get:
+ store = RDF4JStore(
+ endpoint="http://localhost:8080/rdf4j-server",
+ repository_id="team/repo ?#",
+ )
+
+ self.assertEqual(store.repository_id, "team/repo ?#")
+ mock_get.assert_called_once_with(
+ "http://localhost:8080/rdf4j-server/repositories/team%2Frepo%20%3F%23",
+ timeout=30,
+ auth=None,
+ )
+ self.assertEqual(
+ store._get_sparql_endpoint(),
+ "http://localhost:8080/rdf4j-server/repositories/team%2Frepo%20%3F%23",
+ )
+ self.assertEqual(
+ store._get_update_endpoint(),
+ "http://localhost:8080/rdf4j-server/repositories/"
+ "team%2Frepo%20%3F%23/statements",
+ )
+
+ transaction_response = MagicMock()
+ transaction_response.headers = {"Location": "/transactions/tx-1"}
+ with patch(
+ "semantica.triplet_store.rdf4j_store.requests.post",
+ return_value=transaction_response,
+ ) as mock_post:
+ self.assertEqual(store.begin_transaction(), "tx-1")
+
+ mock_post.assert_called_once_with(
+ "http://localhost:8080/rdf4j-server/repositories/"
+ "team%2Frepo%20%3F%23/transactions",
+ timeout=30,
+ auth=None,
+ )
+
+
class TestRDF4JStoreIsConstructQuery(unittest.TestCase):
def test_detects_uppercase(self):
self.assertTrue(_make_connected_store()._is_construct_query(