mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-08-29 04:26:20 +00:00
Compare commits
290
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
da642f12fa | ||
|
|
5376f046ca | ||
|
|
56d9e9a857 | ||
|
|
ecb33a5b7d | ||
|
|
100e95a098 | ||
|
|
cce5ea177c | ||
|
|
e12eec40a1 | ||
|
|
4da27c38bb | ||
|
|
7f928f9f8e | ||
|
|
65e6dcfef5 | ||
|
|
0775b0114e | ||
|
|
f4c3064571 | ||
|
|
b2dc633796 | ||
|
|
13b287b974 | ||
|
|
cec9bee099 | ||
|
|
36ced4e826 | ||
|
|
6032b4e0bc | ||
|
|
8db95f00c6 | ||
|
|
23baf21d5a | ||
|
|
5d54919804 | ||
|
|
c9c777993b | ||
|
|
9cec305a75 | ||
|
|
3d0ce55fd7 | ||
|
|
b13cc1cca2 | ||
|
|
f187d4b5da | ||
|
|
8e79c65542 | ||
|
|
91ea31b460 | ||
|
|
c49e77d059 | ||
|
|
af3308ad06 | ||
|
|
59af023447 | ||
|
|
f4692eea80 | ||
|
|
2de029ac8d | ||
|
|
1ce76055f5 | ||
|
|
8cc5d364db | ||
|
|
d76bff9ab0 | ||
|
|
1e5ad49dc3 | ||
|
|
f0aa581318 | ||
|
|
8a990c8bf5 | ||
|
|
47c7ff5df8 | ||
|
|
92b8aa6993 | ||
|
|
970d3552d4 | ||
|
|
88d73189dd | ||
|
|
599729f2c0 | ||
|
|
84ccc7c0e3 | ||
|
|
fa6d645eea | ||
|
|
97f7154220 | ||
|
|
551b94c524 | ||
|
|
50468f9c90 | ||
|
|
c7d608570c | ||
|
|
5e8caadcb4 | ||
|
|
d05ef9d09f | ||
|
|
06a4b2c9aa | ||
|
|
e2fc76cea0 | ||
|
|
a1a72cdd50 | ||
|
|
2075eca0f3 | ||
|
|
4217f23df2 | ||
|
|
2f63896fb4 | ||
|
|
58aad80d56 | ||
|
|
1452dab5fa | ||
|
|
f454c48929 | ||
|
|
b06a4f0748 | ||
|
|
7da3519ca7 | ||
|
|
b388e936fd | ||
|
|
93281859c8 | ||
|
|
45ce682e6b | ||
|
|
c415d57d16 | ||
|
|
943be0c10f | ||
|
|
3c00ffb019 | ||
|
|
7a6f1d0417 | ||
|
|
b6c8563cb0 | ||
|
|
6dad69cdb4 | ||
|
|
9b30c8af94 | ||
|
|
703b40a116 | ||
|
|
08d6390521 | ||
|
|
7109040984 | ||
|
|
220fb10e5c | ||
|
|
fb02c868f8 | ||
|
|
58ec7639fb | ||
|
|
ac16042f67 | ||
|
|
346f98bdbf | ||
|
|
595f08ee30 | ||
|
|
0468a603ae | ||
|
|
b9cb524514 | ||
|
|
f4c6be158f | ||
|
|
cf4750ebf0 | ||
|
|
95b6d952e6 | ||
|
|
49db007691 | ||
|
|
4c997b5017 | ||
|
|
de31b43663 | ||
|
|
6c2ccfd3af | ||
|
|
cf6c9b7b9c | ||
|
|
52ba7b6890 | ||
|
|
1f868b9779 | ||
|
|
8d5479d22f | ||
|
|
14107e51c3 | ||
|
|
e41993a6bd | ||
|
|
ea9b1f5d4a | ||
|
|
e63bad310e | ||
|
|
d79f2cfb8f | ||
|
|
48c2d2a7ed | ||
|
|
fdafffa980 | ||
|
|
abe1bc8f3e | ||
|
|
a4bcfade7f | ||
|
|
2b077c6d0e | ||
|
|
f6b31925c6 | ||
|
|
4820185924 | ||
|
|
7e1d2550b9 | ||
|
|
f124df4229 | ||
|
|
a47954c19a | ||
|
|
1ee2ae88a7 | ||
|
|
93e4b97517 | ||
|
|
48f219cd5c | ||
|
|
331c857672 | ||
|
|
727b0383cc | ||
|
|
248ae57694 | ||
|
|
ba3737c878 | ||
|
|
e3b24ef872 | ||
|
|
b891902d6d | ||
|
|
9123dcc0bd | ||
|
|
6cbe0ae438 | ||
|
|
fe3baad67c | ||
|
|
7efc66d0e3 | ||
|
|
50f2f82b95 | ||
|
|
d3f37f798e | ||
|
|
5cd79b6436 | ||
|
|
3c99f447e6 | ||
|
|
8dcbee386d | ||
|
|
e2f850e9e4 | ||
|
|
2d976963ab | ||
|
|
283b7ada0c | ||
|
|
483f53aaa6 | ||
|
|
14091d21fb | ||
|
|
58125a0a93 | ||
|
|
394ce5fe61 | ||
|
|
8e9f7c5526 | ||
|
|
d4fdc1f0d3 | ||
|
|
729f4fe932 | ||
|
|
92ad7bc2df | ||
|
|
db81136b0a | ||
|
|
5c6b40f36c | ||
|
|
6390652303 | ||
|
|
1b21fc4cbb | ||
|
|
719efa4794 | ||
|
|
1a220da477 | ||
|
|
d06434ae31 | ||
|
|
b64e4b6600 | ||
|
|
db0a9e8bfd | ||
|
|
46451ae2e1 | ||
|
|
54bae5dffe | ||
|
|
5b01949dd8 | ||
|
|
3b710c79d2 | ||
|
|
4b312ca2fa | ||
|
|
363a9ad641 | ||
|
|
b7af18a70a | ||
|
|
560ffef59f | ||
|
|
a279e74468 | ||
|
|
6653cbe879 | ||
|
|
0dc26350f9 | ||
|
|
eb7427d12c | ||
|
|
3063bf8096 | ||
|
|
cbb0a6dd8e | ||
|
|
a46be971e6 | ||
|
|
e3405ebc23 | ||
|
|
0ee38c2d99 | ||
|
|
a3074ec454 | ||
|
|
5e40d6e4ce | ||
|
|
96f60e6114 | ||
|
|
a2a8d776a3 | ||
|
|
4801ff3492 | ||
|
|
cfccdab8ed | ||
|
|
241ff8e481 | ||
|
|
c5d382ee81 | ||
|
|
54c274e02c | ||
|
|
988ff609cf | ||
|
|
cd2d11a2e7 | ||
|
|
1273c4fb1e | ||
|
|
898a92062a | ||
|
|
556e786fd5 | ||
|
|
d83b21ba77 | ||
|
|
8f6948f85d | ||
|
|
60eb595d62 | ||
|
|
861b2bf757 | ||
|
|
78fc9028a8 | ||
|
|
6b7625ef9b | ||
|
|
48a05b00a6 | ||
|
|
58b77ddcf5 | ||
|
|
5d554ec586 | ||
|
|
1c27a0ae7e | ||
|
|
9e2f349221 | ||
|
|
a7ebec8fe5 | ||
|
|
71cffb15e9 | ||
|
|
d7ee22cf1f | ||
|
|
efdfa39c15 | ||
|
|
66e3333e41 | ||
|
|
9ca83d397f | ||
|
|
d5dc4eabac | ||
|
|
f60ca6a529 | ||
|
|
05c21af117 | ||
|
|
981c9d9208 | ||
|
|
c30ec14858 | ||
|
|
e5c5cf0efa | ||
|
|
e03212cd66 | ||
|
|
83c04a57d6 | ||
|
|
75b026c6dd | ||
|
|
2a303cf4da | ||
|
|
7595bad28f | ||
|
|
2d75952476 | ||
|
|
2ac3eaffd7 | ||
|
|
51c12d5c8f | ||
|
|
e55c03bd39 | ||
|
|
4d3259df31 | ||
|
|
4a886d970e | ||
|
|
e1092ac507 | ||
|
|
b77e3e8c3c | ||
|
|
68f7ae3807 | ||
|
|
56609ab3fc | ||
|
|
7b2b2efe6b | ||
|
|
08a6e7c053 | ||
|
|
5a3bdc393d | ||
|
|
e6b159e5c5 | ||
|
|
7db2e2f46b | ||
|
|
3fbe3cfd2d | ||
|
|
75bc6255d4 | ||
|
|
a96f1590f1 | ||
|
|
063f447202 | ||
|
|
a1194a155d | ||
|
|
17d878cbf3 | ||
|
|
488e381247 | ||
|
|
4acd2f9c33 | ||
|
|
7cac8a6bc3 | ||
|
|
6c77594ea6 | ||
|
|
5109c6fab2 | ||
|
|
12b9694a8e | ||
|
|
49707729ad | ||
|
|
430020c7c4 | ||
|
|
43b207c1c5 | ||
|
|
55bde673c9 | ||
|
|
0f308b2078 | ||
|
|
5c2901ae27 | ||
|
|
dae21166a1 | ||
|
|
67be421533 | ||
|
|
baf8f01f85 | ||
|
|
c58686b4ec | ||
|
|
c3a0078bfd | ||
|
|
04602a0e0e | ||
|
|
eedf1425ca | ||
|
|
d4cb14c1fb | ||
|
|
4a451f410d | ||
|
|
a8194dfc60 | ||
|
|
c7415f2e92 | ||
|
|
3331df28ad | ||
|
|
de5e20dc55 | ||
|
|
0f252ab355 | ||
|
|
0b77e5fe94 | ||
|
|
893b6db3c3 | ||
|
|
b8297b8077 | ||
|
|
4d37920007 | ||
|
|
6416fbb669 | ||
|
|
4b6cc09585 | ||
|
|
aee6e5ad9c | ||
|
|
83649f6821 | ||
|
|
2f04bc01a3 | ||
|
|
eaf51b3383 | ||
|
|
616f5ca9b9 | ||
|
|
d42af280e8 | ||
|
|
21edb700b2 | ||
|
|
13915297d2 | ||
|
|
0e40639930 | ||
|
|
778ff51162 | ||
|
|
b2d54a6683 | ||
|
|
ea7790a5bf | ||
|
|
1cee3e7cb9 | ||
|
|
e7ce092ccf | ||
|
|
92dc3304f8 | ||
|
|
f737f72675 | ||
|
|
828179e115 | ||
|
|
b3e107de8f | ||
|
|
55f7eba389 | ||
|
|
a3d8064f3d | ||
|
|
1b9bb4c345 | ||
|
|
20781e8a9e | ||
|
|
6148975e83 | ||
|
|
0ca7b8d489 | ||
|
|
fb7845240b | ||
|
|
d769bf1c39 | ||
|
|
310ac7bd85 | ||
|
|
aeb1752c83 | ||
|
|
dec05b907d | ||
|
|
c77ce9394a | ||
|
|
c7174e9852 |
@@ -69,5 +69,5 @@ If you have ideas on how this could be implemented, please share.
|
||||
|
||||
---
|
||||
|
||||
**Note**: For feature requests that are ready to be implemented, consider creating a [Feature Request issue](https://github.com/Hawksight-AI/semantica/issues/new?template=feature_request.md) instead.
|
||||
**Note**: For feature requests that are ready to be implemented, consider creating a [Feature Request issue](https://github.com/semantica-agi/semantica/issues/new?template=feature_request.md) instead.
|
||||
|
||||
|
||||
@@ -46,8 +46,8 @@ If applicable, paste any error messages or describe unexpected behavior:
|
||||
|
||||
## Checklist
|
||||
|
||||
- [ ] I have searched existing [discussions](https://github.com/Hawksight-AI/semantica/discussions) and [issues](https://github.com/Hawksight-AI/semantica/issues)
|
||||
- [ ] I have checked the [documentation](https://github.com/Hawksight-AI/semantica/tree/main/docs) and [FAQ](https://github.com/Hawksight-AI/semantica/blob/main/docs/faq.md)
|
||||
- [ ] I have searched existing [discussions](https://github.com/semantica-agi/semantica/discussions) and [issues](https://github.com/semantica-agi/semantica/issues)
|
||||
- [ ] I have checked the [documentation](https://github.com/semantica-agi/semantica/tree/main/docs) and [FAQ](https://github.com/semantica-agi/semantica/blob/main/docs/faq.md)
|
||||
- [ ] I have provided a minimal code example (if applicable)
|
||||
- [ ] I have included error messages (if applicable)
|
||||
- [ ] I have provided environment details
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
# Funding options for Semantica
|
||||
github: Hawksight-AI
|
||||
github: semantica-agi
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
blank_issues_enabled: true
|
||||
contact_links:
|
||||
- name: 📚 Documentation
|
||||
url: https://github.com/Hawksight-AI/semantica/tree/main/docs
|
||||
url: https://github.com/semantica-agi/semantica/tree/main/docs
|
||||
about: Browse the documentation
|
||||
- name: 💬 Discussions
|
||||
url: https://github.com/Hawksight-AI/semantica/discussions
|
||||
url: https://github.com/semantica-agi/semantica/discussions
|
||||
about: Ask questions and discuss with the community
|
||||
|
||||
+9
-9
@@ -3,31 +3,31 @@
|
||||
## Getting Help
|
||||
|
||||
### 📚 Documentation
|
||||
Check the [docs folder](https://github.com/Hawksight-AI/semantica/tree/main/docs) and [README](https://github.com/Hawksight-AI/semantica/blob/main/README.md) for guides and examples.
|
||||
Check the [docs folder](https://github.com/semantica-agi/semantica/tree/main/docs) and [README](https://github.com/semantica-agi/semantica/blob/main/README.md) for guides and examples.
|
||||
|
||||
### 💬 Community Support
|
||||
- **GitHub Discussions**: [Ask questions](https://github.com/Hawksight-AI/semantica/discussions)
|
||||
- **GitHub Discussions**: [Ask questions](https://github.com/semantica-agi/semantica/discussions)
|
||||
- **Discord**: Join our [Discord server](https://discord.gg/sV34vps5hH) for real-time chat
|
||||
|
||||
### 💭 Discussions
|
||||
Join the conversation on [GitHub Discussions](https://github.com/Hawksight-AI/semantica/discussions):
|
||||
Join the conversation on [GitHub Discussions](https://github.com/semantica-agi/semantica/discussions):
|
||||
- **Q&A**: Ask questions and get help from the community
|
||||
- **Ideas**: Share feature requests and suggestions
|
||||
- **Show and Tell**: Showcase your projects and use cases
|
||||
- **General**: General discussions about Semantica
|
||||
|
||||
### 🐛 Bug Reports
|
||||
Found a bug? [Create an issue](https://github.com/Hawksight-AI/semantica/issues/new/choose)
|
||||
Found a bug? [Create an issue](https://github.com/semantica-agi/semantica/issues/new/choose)
|
||||
|
||||
### 📖 Resources
|
||||
- [Quick Start Guide](https://github.com/Hawksight-AI/semantica/blob/main/docs/quickstart.md)
|
||||
- [FAQ](https://github.com/Hawksight-AI/semantica/blob/main/docs/faq.md)
|
||||
- [Cookbook Examples](https://github.com/Hawksight-AI/semantica/tree/main/cookbook)
|
||||
- [Quick Start Guide](https://github.com/semantica-agi/semantica/blob/main/docs/quickstart.md)
|
||||
- [FAQ](https://github.com/semantica-agi/semantica/blob/main/docs/faq.md)
|
||||
- [Cookbook Examples](https://github.com/semantica-agi/semantica/tree/main/cookbook)
|
||||
|
||||
## Commercial Support
|
||||
|
||||
For enterprise support, custom development, or consulting services:
|
||||
- Contact us through [GitHub Issues](https://github.com/Hawksight-AI/semantica/issues)
|
||||
- Contact us through [GitHub Issues](https://github.com/semantica-agi/semantica/issues)
|
||||
- Include "Commercial Support" in the title
|
||||
|
||||
## Sponsorship
|
||||
@@ -35,7 +35,7 @@ For enterprise support, custom development, or consulting services:
|
||||
### Sponsor this project
|
||||
|
||||
Support Semantica development:
|
||||
- [GitHub Sponsors](https://github.com/sponsors/Hawksight-AI)
|
||||
- [GitHub Sponsors](https://github.com/sponsors/semantica-agi)
|
||||
|
||||
Your sponsorship helps us:
|
||||
- Maintain and improve the framework
|
||||
|
||||
+270
-3
@@ -9,9 +9,127 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
## [0.6.7] - 2026-08-28
|
||||
|
||||
### Added
|
||||
|
||||
- **First-class CrewAI integration** (#962)
|
||||
- **First-class LangChain integration** (closes #963; recreates #969)
|
||||
- New `pip install semantica[langchain]` extra (`langchain-core>=0.3.0`), included in the `all` bundle
|
||||
- `integrations/langchain/SemanticaRetriever` — LangChain `BaseRetriever` that seeds from `HybridSearch` then walks graph edges (`hops=2` default) for GraphRAG-style retrieval; falls back to `ContextGraph.query` when hybrid search is unavailable
|
||||
- `integrations/langchain/SemanticaVectorStore` — LangChain `VectorStore` adapter over `HybridSearch` (`add_texts`, `similarity_search`, `similarity_search_with_score`, `from_texts`)
|
||||
- `integrations/langchain/SemanticaKGTool` / `SemanticaDecisionTool` — `BaseTool` subclasses with Pydantic `args_schema` (`semantica_query_graph`, `semantica_query_decisions`); `build()` returns the tool, or `None` when langchain-core is absent
|
||||
- Retriever and VectorStore read HybridSearch nested `metadata` (`content`, `node_id`, `node_type`) rather than top-level fields that HybridSearch does not set
|
||||
- All adapters remain importable without langchain-core (`LANGCHAIN_AVAILABLE` flag)
|
||||
- Docs: `docs/integrations/langchain.md`, README native-integration matrix, and `docs.json` nav entry
|
||||
|
||||
- **SAP OData ingestor** (#1234, closes #1228) by @pkupt
|
||||
- New `SAPODataEntity` / `SAPODataConnector` / `SAPIngestor` (`semantica.ingest`, lazy exports), following the three-layer connector pattern already used for Snowflake/Databricks, to pull master/transactional data (Business Partners, Sales Orders) from SAP OData v2/v4 services into the Context Graph
|
||||
- Dual auth (OAuth2 client-credentials for BTP/S4HANA Cloud, Basic for on-prem NetWeaver); every outbound request, including the token exchange, routes through `request_with_ssrf_guard`
|
||||
- `$metadata` (CSDL XML) is parsed with a hand-rolled `xml.etree` reader rather than pulling in `pyodata`; pagination follows OData v2 `__next`/`__deferred` and v4 `@odata.nextLink`
|
||||
- New `pip install semantica[ingest-sap]` extra (`requests>=2.28.0`)
|
||||
- **Known phase-1 limits** (documented in docstrings): the OAuth2 token is cached but never refreshed, and pagination has no `max_pages` fuse (`top` bounds it when supplied)
|
||||
- New `tests/ingest/test_sap_ingestor.py`: 22 tests (auth, EDMX parsing, v2/v4 pagination, SSRF routing, error paths, service-root normalization)
|
||||
|
||||
- **`ContextGraph` gains deterministic, human-editable Markdown round-trip persistence** (#852) by @SaurabhScripts
|
||||
- `save_to_markdown()`/`load_from_markdown()` write one file per node plus a graph manifest, so a graph can be reviewed and hand-edited outside the application without giving up the existing JSON API or its default behavior
|
||||
- An existing destination is validated as a complete, canonical managed export before atomic replacement, so the loader can't silently clobber an unrelated or manually-extended directory
|
||||
- Import/export paths and their ancestors reject symlinks, Windows junctions, and other reparse points, with pre-open and post-open validation — the same hardening applied to `AgentMemory`'s existing Markdown import in the companion fix below
|
||||
- Dangling edge endpoints import as JSON-compatible entity stubs rather than being rejected outright (matching what the JSON loader already accepts); node/edge indexes, adjacency, and analytics/retraction/tombstone state are rebuilt after a Markdown load, and granular node/edge events are still emitted so temporal audit history stays useful
|
||||
- New `tests/context/test_context_graph_markdown.py`: 29 passed, 1 skipped (the skipped case creates a real Windows junction and runs on Windows CI); full `tests/context/` suite: 614 passed, 1 skipped
|
||||
|
||||
- **Explorer graph inspector gains a read-only Markdown content viewer** (#1078, closes #900) by @sakshi04-ui — Preview (rendered GFM) and Source (exact, whitespace-preserving) tabs for node content, with a copy-to-clipboard action. A URL allowlist restricts links to `http:`/`https:`/`mailto:`/in-document anchors, raw HTML execution is disabled, and external links carry `rel="noopener noreferrer"`. A first, focused step toward human-editable memory (#765); no write path yet. New `explorer/tests/markdownContentViewer.test.ts`: 8 tests
|
||||
- **Follow-up (perf)** (#1195, addresses #1118) by @pravit-amp: `remarkPlugins` and the ~20-entry renderer `components` map were inline literals, so every unrelated re-render (e.g. clicking Copy) re-ran the full remark parse and remounted the whole subtree — up to 1.1s of main-thread block on a 2000-row GFM table. Both are now hoisted to module scope and the rendered element is memoized on content, cutting re-render cost from as much as 1121ms to ~0.1ms across all measured fixtures with no change to rendered output. A separate, upstream `remark-gfm` table-parse cost (~O(n^1.9), not fixed here) is left open on the issue as a product decision
|
||||
- **Follow-up (cleanup)** (#1194, closes #1119) by @pravit-amp: the pure `isSafeUrl` URL-safety helper is extracted out of `MarkdownContentViewer.tsx` into its own `markdownUrlSafety.ts` module (behavior-preserving — moved verbatim), so the component module exports only components and stops tripping `react-refresh/only-export-components`
|
||||
|
||||
- **`reasoning` gains a structured Action layer — rule-driven side effects with optional provenance** (#1096, closes #1095) by @cxzg007 — `AssertAction`/`RetractAction`/`CallAction`/`EmitEventAction` let a matched rule write facts back to a `KnowledgeGraph`, retract facts, call a structured handler (replacing the previously-unused `Rule.handler`), or emit to a sink registered via `Reasoner.on_event`, turning the reasoner from a pure inference engine into a production-rule system. With `provenance=True`, fired actions are recorded to `Reasoner.action_log`. Fully additive — rules without `actions` are unaffected, and the legacy `handler` field still fires (now wrapped internally as a `CallAction`). Also fixes a latent dangling import in `reasoning_provenance.py` (`ReasoningEngine`/`infer` → `Reasoner`/`infer_facts`). New `tests/reasoning/test_rule_actions.py`: 9 tests; full `tests/reasoning/` suite: 54 passed
|
||||
|
||||
- **`run_shacl_validation` is now a public, documented entry point** (#1189, closes #1186) by @mikemikimike — the SHACL guide had documented the private `_run_pyshacl` helper as the canonical API; it's now exposed through `semantica.ontology`, with `_run_pyshacl` kept as a compatibility alias over the same implementation. `tests/ontology/test_ontology_advanced.py`: 33 passed (also fixes a flaky comparison against pySHACL's non-deterministic blank-node shape identifiers by comparing stable report fields instead)
|
||||
|
||||
- **`docs/storage-backends.md`: adapter inventory and RDF/LPG feature matrix** (#899, addresses #888) by @yulinlina — which graph storage backends are built-in vs. bring-your-own, and where provenance/context support is partial
|
||||
- **`docs/guides/shacl-validation.md`: documented that `rdfs:range` + RDFS entailment makes `sh:class` unfalsifiable** (#1182, fixes #1130) by @ALDRIN121 — with entailment on, pyshacl infers the declared range class onto every object, so a `sh:class` constraint can never fail and reports `conforms: True` on non-conforming data; added to Common Pitfalls with the `inference="none"` vs `inference="rdfs"` contrast and guidance to re-run `sh:class` shape sets with entailment off before trusting a pass
|
||||
- **Cookbook: four new module notebooks** — `22_Provenance_Tracking.ipynb` (#989, lineage walks, revision history, invalidation, checksums), `23_Reasoning.ipynb` (#990, `Reasoner`/`DatalogReasoner`/`ExplanationGenerator`), `24_Change_Management.ipynb` (#991, versioned snapshots, named tags, checksum tamper-detection), and `25_Seed_Data.ipynb` (#992, bootstrapping a foundation graph from a trusted CSV source) — all by @LeonSGP43, filling gaps where the corresponding module shipped a usage doc but no runnable tutorial; every cell verified against current module source. `docs/cookbook.md` index entries for all four added in #1225
|
||||
- **README "Cite Us" section and `docs/citation.md` cross-link** (#1210) by @KaifAhmad1 — BibTeX/APA/MLA/Chicago/IEEE citation forms; also corrects the copyright holder in `LICENSE`/`docs/project-license.md` from the stale "Hawksight AI" to "Semantica" and replaces the retired `Hawksight-AI` GitHub org slug with `semantica-agi` across ~40 files (READMEs, issue templates, plugin manifests, cookbook notebooks, docs)
|
||||
|
||||
### Changed
|
||||
|
||||
- **A registered custom method can now refuse, instead of being silently overridden by the default implementation** (#1127, closes #1108) by @fabio-rovai — every module supporting custom methods wrapped the registered callable in a `try`/`except` that logged a warning and ran the built-in default on *any* exception, including one a validator or policy gate raised on purpose to say "do not produce this output." That made every registered gate advisory rather than authoritative. `semantica/utils/custom_methods.py` now centralizes the policy: an exception from a registered method propagates to the caller by default; `fallback_on_custom_error=True` restores the previous warn-and-continue behavior per call. Applied mechanically across all 58 call sites in `export/`, `ingest/`, `normalize/`, `parse/`, `embeddings/`, and `kg/` methods modules. New `tests/utils/test_custom_method_can_refuse.py`: 13 tests, including the reported gate-deletes-and-raises scenario and a guard that no call site still swallows
|
||||
- **Removed 13 confirmed-dead symbols across 9 files** (#1176, closes #1174) by @Vinv-AI — private helpers and Explorer app-layer code with zero callers in code, tests, or docs, none part of the public API or a FastAPI `response_model`; 289 deletions, no behavior change
|
||||
- **Consolidated the two duplicate Turtle/N-Triples literal escapers in `rdf_exporter.py`** (#1221, closes #1218) by @pkupt — `_escape_turtle_literal` (added in #1148) escaped the same five characters in the same order as the older module-level `_escape_literal`; the redundant one is dropped and all four call sites route through the original. Behavior no-op, verified against the full export suite (301 passed, 1 skipped)
|
||||
- **Removed the unreachable `_extract_with_spacy()` method and the unused `self.nlp` attribute from `NERExtractor`** (#1220, fixes #1058) by @yunaremaia — the ML dispatch path has always gone through `methods.py`'s process-level model cache instead; `__init__` still validates the spaCy runtime up front but no longer eagerly loads a model nothing on the instance reads
|
||||
- **Cleaned up an unused `sys` import and import ordering in `semantica/worker.py`** (#1061) by @aoright
|
||||
- **Test-only contributions**: isolated `sys.modules` mock leakage between `tests/visualization/` files so the suite passes in any collection order (#897, closes #859, by @luantaraschi); added coverage for 4 previously-untested `ConflictResolver` strategies and 3 `ConflictDetector` conflict types (#902, fixes #865, by @Devansh070); added a regression test tracking relationship provenance through `ProvenanceManager` (#1071, closes #1055, by @dex0shubham); added `max_tokens`-propagation regression coverage for LLM extraction methods, later folded into the cache-key fix below (#925, by @saiganesh47)
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`SPARQLReasoner.execute_query()` claimed to run a query but always returned an empty result** (#1087, fixes #1083) by @ALDRIN121 — both the store-configured and unconfigured branches returned an empty `SPARQLQueryResult` with no real execution behind it, so a caller trusting "no matches" (e.g. a compliance check) could draw a false-negative conclusion from a method that never actually queried anything. Until a real triplet-store execution path lands, it now raises `NotImplementedError` explaining why, and the dead cache/inference scaffolding after the unreachable execution point is removed. 3 new regression tests
|
||||
- **`DuplicateDetector` merged entities that share no identifier, type, or name** (#1149, fixes #1137) by @pkupt — `_create_duplicate_candidate()` only ever boosted confidence for matching types and never penalized a mismatch, so two sparse, differently-typed entities (e.g. a `Person` and an `Organization`) could land above the merge threshold and collapse into one node, silently dropping the second. Two non-empty, differing types are now never a duplicate candidate. `tests/deduplication/`: 92 passed
|
||||
- **`TemporalGraphQuery.analyze_evolution()`'s `stability` metric was a hardcoded placeholder** (#1143, closes #1142) by @cxzg007 — every bounded relationship contributed a constant `1`, so `stability` was always `1.0` or `0` regardless of how long relationships actually stayed valid. Now computes the mean valid-time duration in seconds across relationships with both `valid_from`/`valid_until` set; unbounded/half-open intervals are skipped and negative intervals clamp to zero. 3 new tests in `tests/kg/test_kg.py`
|
||||
- **CodeQL false-positive on a JSON-LD test's URL check** (#1183) by @KaifAhmad1 — `"https://schema.org/" in flattened` pattern-matched CodeQL's substring-sanitization heuristic even though `flattened` is always a `list` (exact membership, no sanitization or SSRF path involved); rewritten as an explicit `any(entry == ... for entry in flattened)` with identical behavior
|
||||
- **HuggingFace NER extraction crashed on `huggingface_model` being forwarded as an unexpected pipeline loader kwarg** (#1188, fixes #1063) by @shahzaib-ahmadcs — while preserving genuinely supported pipeline kwargs like `aggregation_strategy`. 5 tests pass
|
||||
- **JSON-LD document/graph `@id` was minted from the wall clock, so re-exporting an unchanged graph produced a new subject every time** (#1181, closes #1147) by @reddynitish — merging repeated exports duplicated graph identity instead of recognizing them as the same graph. The `@id` is now content-derived, with optional `graph_uri`/`document_uri` overrides for callers with a stable graph name; `semantica:exportedAt` still records export time separately. Applies to both JSON-LD export paths
|
||||
- **`ContextGraph.get_causal_chain()` only matched the canonical uppercase causal-edge spellings, silently missing edges recorded in `CausalChainAnalyzer`'s present-tense vocabulary** (#1187, fixes #1184) by @ALDRIN121 — `causes`/`influences`/`precedes` differ from `CAUSED`/`INFLUENCED`/`PRECEDENT_FOR` in word form, not just case, so an edge recorded with the analyzer's spelling produced an empty audit chain — silent, and in the dangerous direction for a compliance trace. `add_causal_relationship()` now normalizes through an alias map before storing the canonical form; traversal accepts the union vocabulary. 2 new regression tests, full `tests/context/` suite: 587 passed
|
||||
- **`semantica embed generate` corrupted its own output and could recurse into a stack overflow** (#996/#1004/#1005, closes #994) by @varunsahni18, @yzxcj797 — three compounding defects in one pipeline. (1) `generate_embeddings`/`embed_text`/`calculate_similarity`/`pool_embeddings` all registered themselves as their own custom-method-registry default, so an unqualified call (exactly what the CLI does) re-entered the same wrapper until Python's recursion limit; each of the four dispatch sites now guards on registry identity before recursing (#996, #1005). A second self-recursion in `EmbeddingGeneratorWithProvenance.__getattr__` (re-entering itself when `_generator` is unset, e.g. during a `deepcopy` probe) now raises a normal `AttributeError` for private names instead (#1005). (2) `--output embeddings.parquet` wrote `json.dumps(result, default=str)` regardless of extension, turning a numpy array into its plain-text `repr()` — a file `embed index` then failed to open as Parquet; the writer now detects `.parquet`/`.json`/`.jsonl` and produces real Parquet/JSON, rejecting any other extension with a clear message (#996, #1004). (3) `pyarrow` was only in optional extras despite being required by the documented quick-start flow; promoted to a core dependency (#996)
|
||||
- **`AgentMemory`'s existing Markdown import accepted symbolic links, NTFS junctions, and other Windows reparse points** (#851) by @SaurabhScripts — a direct linked import path is now rejected with an actionable error, and a linked entry found inside an otherwise-valid directory is skipped rather than aborting the whole import; hardened with pre-open/post-open checks, `O_NOFOLLOW` where available, and `fstat`-based regular-file validation. `tests/context/`: 595 passed, 1 skipped (Windows-junction test, runs on Windows CI)
|
||||
- **A caught vector-similarity scoring exception left stale partial state behind, risking a misleading match on the next call** (#885, fixes #875) by @ArmanGrewal007 — the exception is now logged at debug level and `vector_score`/`vector_idx` reset to neutral values before the remaining matching stages continue
|
||||
- **`RDFExporter` could write invalid or unintended relative IRIs for `GraphBuilder`-default entity/relationship identifiers** (#1112, closes #1099) by @mikemikimike — normalization is now applied at the RDF export boundary across Turtle (including temporal Turtle), RDF/XML, and N-Triples: bare/relative identifiers are minted under the Semantica namespace with safe percent-encoding, absolute IRIs pass through unchanged, and configured/input-context prefixes expand through the effective namespace mapping. 38 focused regression tests; 175 export tests plus 46 subtests pass
|
||||
- **`RDF4JStore`'s `repository_id` constructor argument had no effect** (#1192, closes #1191) by @Freakz2z — the explicit id is now honored when selecting the repository; stale documentation caveats claiming otherwise are removed. 65 tests pass across the affected triplet-store suites
|
||||
- **Non-interactive stdout (piped/redirected output, CI logs) was flooded with progress-bar escape sequences** (#1193, fixes #1185) by @ALDRIN121 — a plain `python demo.py > out.txt` captured 173 bytes of progress noise around 10 bytes of real output. `ProgressTracker` now attaches its console display only for an interactive terminal, Jupyter, or the new `SEMANTICA_FORCE_PROGRESS` opt-in (following the `NO_COLOR`/`FORCE_COLOR` convention); file-based progress logging is untouched. Both switches are now documented in the README and `docs/reference/utils.md`. 11 tests pass (6 new)
|
||||
- **Entity `metadata` was dropped by every RDF serializer except the JSON-LD path**, so an entity kept its confidence but lost its source document, page, extractor, and reviewer on Turtle/N-Triples/RDF/XML/`RDFExporter`'s own JSON-LD (#1165, closes #1154) by @fabio-rovai — Semantica's own metadata keys (`num_entities`, `snapshot_time`, Neo4j loader fields, etc.) are now mapped to declared vocabulary terms and carried through on every path; a caller-supplied key with no mapped term is skipped with an explicit warning (rather than silently vanishing) naming the override needed, pending the caller-key namespace decision tracked in #1146. 21 new tests in `tests/export/test_metadata_passthrough.py`; `tests/export`+`tests/ontology`: 274 pass
|
||||
- **`extract_relations_llm` silently dropped caller-supplied generation parameters** (`max_tokens`, `top_p`, `seed`, etc.), and the extraction cache didn't distinguish calls made with different generation settings (#1213, with test coverage from #925) by @Sameer6305 — a small hardcoded allowlist forwarded only `temperature`/`verbose` to `generate_typed`, discarding the rest; fixed by forwarding all caller kwargs. Once forwarded, those parameters also needed to enter the cache key, since two calls differing only in `max_tokens` previously shared one cache entry and the second could silently reuse a result generated under the first's settings — now applied consistently across entity, relation, and triplet LLM extraction. New regression tests for cache bypass/reuse under differing `max_tokens`/`temperature`
|
||||
- **`OxigraphStore` silently ignored the `storage_path` constructor argument and never flushed writes before a reopen**, both causing silent on-disk data loss (#970) by @logan-jl-cc — `__init__`'s parameter is named `path`, so the project-conventional `storage_path` landed in `**config` and was ignored, degrading a supposedly-persistent store to in-memory with no error; `storage_path` is now accepted as an alias. Separately, pyoxigraph's background flush can lag behind a write, so a reopen immediately after `add_triplets` could observe fewer triples than were written; writes to an on-disk store now call `flush()` explicitly. 2 new regression tests, full suite: 9 passed
|
||||
- **MCP server's `export_graph` tool was broken on every output format** (#1151) by @Arasz — the `json` branch called `JSONExporter().export()` without the `file_path` it requires, and every RDF branch passed a `ContextGraph` object where the exporters expect the canonical kg dict, both surfacing as a raw exception string. A third bug compounded both: the RDF export path's progress bar wrote to stdout, which over stdio MCP *is* the JSON-RPC framing, corrupting the protocol and hanging the client (a 300s timeout on an empty graph). Fixed by converting through `ContextGraph.to_kg_dict()`, serializing the JSON branch to match the RDF branches' string contract, and forcing `SEMANTICA_DISABLE_PROGRESS=1` for the server process. 5 new tests, verified failing against 0.6.6 beforehand
|
||||
- **`OntologyIngestor` dropped every class and property from a JSON-LD document using a named graph** (#1156, fixes #1129) by @13g4d0 — a top-level `@id` beside `@graph` names the graph, and `rdflib.Graph.parse()` silently loads only the default graph, discarding the rest; `POST /api/ontology/load` returned `status: "success"` with `class_count: 0`. Now parses into a `Dataset` and flattens all quads into the working graph (the same `Graph`→`Dataset` migration #757 made for `JenaStore`, extended to the ingest path). On the PR's real-world reproduction: 25 triples/1 subject before, 719 triples/45 classes/40 object properties after. 4 new tests including a default-graph canary so the fix can't trade one blind spot for another
|
||||
- **Turtle and N-Triples RDF export interpolated entity `text` into string literals with no escaping**, so a `"`, backslash, newline, CR, or tab in the source text emitted invalid RDF other parsers rejected (#1148, closes #1098) by @pkupt — a shared `_escape_turtle_literal()` (later consolidated in #1221) now escapes per the RDF 1.1 Turtle grammar and is reused for the N-Triples path, which previously escaped only quotes and newlines. `tests/export/`: 161 passed, 1 skipped
|
||||
- **`PipelineSerializer` round trips dropped step dependencies and delta-processing metadata, and could rehydrate a legacy stringified handler as a non-callable string** (#1217, fixes #1216) by @cxzg007 — step dependencies, delta mode, and base/target version IDs are now restored from the serialized schema; runtime handler callables are treated as process-local state and excluded from serialized business configuration rather than (mis)serialized. 52 tests pass
|
||||
- **`PipelineBuilder` never actually dispatched to a handler registered by `step_type`**, and a serialize/deserialize round trip could leak `handler`/`dependencies` into a step's business config (#1215, fixes #1214) by @cxzg007 — a registered handler is now resolved by `step_type` when no explicit `handler=` is supplied (explicit handlers still take precedence), and the two builder-control fields are kept out of `PipelineStep.config` so a strict handler signature can't receive them as unexpected kwargs. `tests/core`+`tests/pipeline`: 50 passed
|
||||
- **`PipelineBuilder.set_parallelism()` was accepted and stored but never read — pipeline steps always ran strictly sequentially**, and the setting didn't survive a serialize/deserialize round trip (#1226, fixes #1223) by @cxzg007 — wired through builder → serializer → execution engine, plus a new opt-in `PipelineStep.parallel_safe` flag. A dependency layer now runs in parallel only when every step in it is marked `parallel_safe`, the layer has more than one step, the input is dict-typed, and no step is in delta mode; otherwise it falls back to sequential execution. Each parallel step's input is deep-copied for isolation, execution is bounded by `ThreadPoolExecutor(max_workers=min(configured parallelism, max_workers))`, a failure cancels pending futures in the layer, and layer results merge back in declaration order (a same-key conflict raises `ProcessingError`). 22 new tests in `tests/pipeline/test_pipeline_parallel.py`
|
||||
- **`Config.get()` silently dropped boolean environment-variable overrides** (#1038, fixes #1035) by @Kyou12138 — the type dispatch checked `isinstance(default, int)` before `isinstance(default, bool)`, and since `bool` subclasses `int` in Python, the bool branch was unreachable: `CONFLICT_ZZTESTFLAG=true` with a `False` default returned `False`, and `=1` returned the int `1` rather than `True`. Bool is now checked first (with whitespace stripped before parsing truthy/falsy spellings), fixed across all ten affected config modules (`conflicts`, `deduplication`, `split`, `embeddings`, `export`, `ingest`, `kg`, `parse`, `ontology`, `normalize`). 12 new tests plus 6 existing conflicts tests and 131 related module tests pass
|
||||
- **Scanned (image-only) PDFs parsed with no error and no warning, returning empty text with a "completed" status** (#1021, closes #1020) by @shanyu910 — `PDFParser._parse_page` swallowed a missing text layer via `page.extract_text() or ""`, so the failure only surfaced far downstream as zero extracted entities. A warning now fires when every parsed page yields no text with `extract_text` enabled, pointing at `parse_pdf(..., method="docling", enable_ocr=True)`. Also fixes a separate `import semantica.parse` failure on a fresh interpreter (`email_parser.py` used `email.message.Message` without importing `email.message`) that was blocking the parse test suite from even collecting. 25 tests pass in `tests/parse/`
|
||||
- **`GET /api/decisions` returned HTTP 422 for any graph containing real decisions**, breaking the Explorer Decisions workspace entirely (#937) by @logan-jl-cc — `record_decision()` stores the timestamp as a POSIX float, but `DecisionResponse.timestamp` is typed `Optional[str]` and Pydantic's strict mode rejected the coercion. Fixed by coercing to `str` (preserving `None`) at the response-adapter boundary
|
||||
- **Decision persistence/query bugs, CJK text handling, and three missing MCP graph tools** (#967) by @toratto — `mcp_server`'s `_get_graph` called a non-existent `graph.load` instead of `load_from_file`, so `SEMANTICA_KG_PATH` was silently ignored and the server always started with an empty graph; `query_decisions` read `category` from the wrong field, always returning nothing for a category filter; `find_precedents`/`query_decisions(query=)`'s similarity threshold was too high for short CJK queries, which also failed outright because `_calculate_decision_content_similarity`'s whitespace-Jaccard fallback is always zero for languages with no whitespace tokenization (now falls back further to a character-bigram overlap coefficient); `load_from_file` didn't rebuild the in-memory decision/entity/temporal indexes after loading, breaking `find_precedents_by_scenario` and decision counts post-reload; `extract_entities`/`extract_relations` returned the spaCy type label as `text` and dropped the actual entity text, and had no way to select a non-English NER model. Also adds three new MCP tools (`query_graph`, `update_node`, `delete_node`, the latter two persisting back to `SEMANTICA_KG_PATH`)
|
||||
- **`sqlalchemy.text` was used but never imported in two `DBIngestor`/`DataExporter` methods**, raising `NameError` on every call before any query reached the database (#1017, closes #1015) by @pravit-amp — `connect()`/`test_connection()` imported `text` function-locally, so the binding never reached `export_table_data()` or `execute_query()`, which called it anyway; both raised immediately, re-wrapped by an `except Exception` into a `ProcessingError` that read like a database fault rather than a missing import. `docs/guides/ontology.md` documents `DBIngestor().execute_query()` as a supported entry point, so documented usage walked straight into it. 5 new tests against a temporary SQLite database, also repairing a previously-failing `tests/ingest/test_notebook_02.py` case
|
||||
- **Ontology generation resolved relationship endpoint types incorrectly, producing wrong object-property domains/ranges** (#1170, closes #1168) by @T1mn — endpoint types are now resolved from the canonical `source_id`/`target_id` fields and supported aliases instead of defaulting to the first entity when a field was missing, preventing e.g. a `Person -> Organization` relationship from generating a `Person -> Person` property. 80 tests pass, 1 skipped
|
||||
- **Ontology property generation dropped data properties when a raw entity type was normalized into a class name** (#1171, closes #1169) by @T1mn — e.g. `software engineer` → `SoftwareEngineer` lost its `email` property; attributes are now grouped by matching raw, normalized, and recorded class names, so the normalized class stays each property's domain. 79 tests pass, 1 skipped
|
||||
- **`flatten_dict()` silently dropped data when a top-level key already containing the separator collided with a key produced by flattening a nested dict** (#1012, fixes #1010) by @yzxcj797 — `{"a.b": 1, "a": {"b": 2}}` flattened to `{"a.b": 2}` with no error, the `1` simply gone; collisions are now detected (unique-key count vs. item count) and raise `ValueError` naming the colliding key before data is lost. 6 new tests
|
||||
- **Creating relationships after `GraphStore.add_edges`/`build_from_entities_and_relationships` silently produced zero edges against ID-minting backends** (#1173, fixes #1136) by @yzxcj797 — an id-space mismatch across three layers: `add_edges` reads application-level string ids and passes them to `create_relationship`, which is a pure passthrough into `Neo4jStore.create_relationship`'s `MATCH ... WHERE id(a) = $start_id` — a Neo4j-internal integer id. Every node was created and every relationship silently failed with one easily-missed warning per edge. `GraphStore` now keeps an application-id→internal-id map, populated by `add_nodes`/`create_node` from the backend's own creation results and consulted by `create_relationship`; unknown ids and identity-mapped backends are unaffected. `tests/graph_store/`: 100 passed
|
||||
- **RDF export left `semantica:text`/`rdfs:label` empty for entities that only carry a `name` field**, across all four RDF formats (#1113, fixes #1097) by @cxzg007 — `RDFSerializer.convert_kg_to_rdf()` already implemented the `name`→`label`/`text` normalization, but `export_to_rdf()` never called it. Now called once at the export boundary (idempotent, non-destructive, falls back to a label derived from the id suffix). 7 new tests, `tests/export/test_rdf_exporter.py`: 17 passed
|
||||
- **Docker Explorer image failed to build on Python 3.14** — `gensim` has no prebuilt wheel for it and the slim base has no `gcc` to build from source (#1172, closes #1025) by @DwitiThaker — runtime pinned to `python:3.13-slim`, where `gensim` installs from a prebuilt wheel
|
||||
- **Unit normalization rejected common aliases before conversion** — `kg`, `g`, and other abbreviated/plural unit spellings failed category validation and the conversion-factor lookup ahead of it (#939) by @Mr-Neutr0n — aliases now normalize first; canonical aliases added for feet, yards, miles, and gallons. 7 tests pass
|
||||
- **An oversized, caller-controlled mapping key could blow up a `ValidationError` message to megabyte scale**, and equally inflate application logs on repeated malformed input (#1088, fixes #1001) by @ALDRIN121 — follow-up to the graph-payload validation added in #958. The displayed key is now truncated at 64 characters with an ellipsis; the underlying input and validation decisions are unchanged. 4 new tests
|
||||
- **`SeedDataManager.load_from_api()` mislabeled genuine connection failures as a missing `requests` dependency** (#972, closes #949) by @pravit-amp — `requests.exceptions.RequestException` (connection errors, timeouts, `raise_for_status()` failures) subclasses `OSError`, so an `except (ImportError, OSError)` block written to guard a lazy import that no longer existed (`requests` is a core dependency) caught real failures too and told users to reinstall an already-installed library while dropping the original exception chain. The block is removed; genuine failures now surface through the existing `Failed to load from API: {e}` path with `from e` intact. 5 new regression tests
|
||||
- **`SHACLGenerator` produced shapes that matched nothing, and pySHACL reported `conforms: True` on data that plainly violated them** (#1124, closes #1104, closes #1105) by @fabio-rovai — `base_uri` was used both as where shape resources live and to expand every `sh:targetClass`/`sh:path`, so with the default shapes namespace, generated shapes targeted classes no data graph in the package actually uses; a shape with zero matching focus nodes is vacuously satisfied, so validation silently passed regardless of real violations. The target namespace now resolves independently (explicit argument → ontology's declared namespace → an existing absolute class/property IRI → ontology `uri` → the vocabulary namespace), never the shapes namespace. Separately, `_attach_property_shapes` attached a domain-less property's constraint to *every* shape ("no domain declared, attach to all"), asserting a constraint the ontology never stated; a domain-less property is now left unattached by default, with `attach_domainless_properties=True` to restore the old behavior. 17 new tests validate real data through pySHACL rather than reading shape text; `tests/ontology`+`tests/export`: 239 passed
|
||||
- **OWL export dropped every generated property and collapsed distinct classes onto one node** (#1123, closes #1103) by @fabio-rovai — `OWLExporter` reads `object_properties`/`data_properties`, but `OntologyGenerator` emits one combined `properties` list, so every property was silently discarded; separately, a class built without a namespace manager gets `"uri": None`, which a `"uri" not in cls"` guard never catches (the key is present), so the exporter wrote a relative `<>` IRI for it — resolved by rdflib against the current working directory, meaning two classes could collapse onto one subject and that subject's identity changed with the export's working directory. Both dict shapes are now merged and classified correctly, and a class/property IRI resolves through `uri`→`iri`→`id`→a name joined onto the ontology base, skipping (with a warning) a term with none of those instead of minting `<>`. 10 new regression tests parse the real output with rdflib and Oxigraph; `tests/export`+`tests/ontology`: 231 passed
|
||||
- **Confidence scores serialized as four different, mutually-disagreeing RDF terms depending on export format, and one non-numeric confidence value could break an entire Turtle export** (#1125, closes #1100, closes #1102) by @fabio-rovai — Turtle wrote a bare `xsd:decimal`, N-Triples an explicit `xsd:float`, RDF/XML an untyped plain literal, and JSON-LD's native number expanded to `xsd:double`; loading a Turtle and an N-Triples export of the same graph into one store gave the same entity two different confidence values. Separately, an unparseable confidence (e.g. the string `"high"`) was interpolated into Turtle with no validation, producing a syntax error that dropped every entity from the export. All four paths now write one canonical `xsd:decimal` lexical form (matching the pre-existing Turtle behavior and the only exact representation of the four); an unusable value is omitted with a warning instead of corrupting the document. The vocabulary's `sem:confidence` now declares `xsd:decimal` (previously left undeclared to avoid contradicting the disagreeing exporters). 20 new tests compare parsed graphs across all four formats; `tests/export`+`tests/ontology`: 240 passed
|
||||
- **An OWL-Time validity interval was reified onto a relationship IRI the graph never actually referenced**, making it unreachable from the edge it described (#1126, closes #1106) by @fabio-rovai — a relationship serializes as a single triple with no node of its own, so `include_temporal=True` minted a well-formed `time:Interval` with zero inbound arcs to its subject. Turtle now also emits the `sem:Relationship`/`sem:source`/`sem:target`/`sem:type` reification the JSON-LD path already produced, but only when there's temporal data to attach — default and `include_temporal=False` output are byte-for-byte unchanged. 7 new tests include a SPARQL walk from the edge to its interval, the path the dangling node made impossible; `tests/export`+`tests/ontology`: 228 passed
|
||||
- **JSON-LD exports were unreadable by Semantica's own default parser** (#1145, fixes #1144) by @fabio-rovai — every export was written as a named graph (a top-level `@id` beside `@graph`), which a plain `rdflib.Graph.parse()` silently discards in favor of the (empty) default graph; a two-entity graph parsed as 2 triples instead of 20. Compounded by `export_knowledge_graph` converting its payload to JSON-LD and then handing the *already-converted* document to `export()`, which converted it again, producing two `@context` blocks and two document nodes. Metadata now attaches beside `@graph` rather than naming it, and a payload that already declares `@context` is merged rather than re-wrapped. 9 new tests parse with both `Graph()` and `Dataset()` and assert identical counts; full-suite failure set unchanged before/after (539/539)
|
||||
- **`GraphBuilder` didn't propagate entity-resolution's merged ids into the `source_id`/`target_id` relationship aliases**, only `source`/`target` (#1115, closes #1110) by @T1mn — a relationship's alias fields could still point at a pre-merge id after resolution. Both alias pairs are now kept in sync. 9 tests pass
|
||||
- **`GraphValidator` indexed entities only by `id`, rejecting graphs that use the `entity_id` alias as invalid even when their relationships were fine** (#1116, closes #1111) by @T1mn — validation and endpoint checks now go through the shared `get_entity_id()` helper, accepting both fields consistently. 5 tests pass
|
||||
- **Broken star history chart in README** (#1057) by @OctoBored — the embedded chart used the GitHub stargazer API, now access-restricted; switched to a token-free alternative data source
|
||||
|
||||
### Security
|
||||
|
||||
- **Agno's `AgnoKnowledgeGraph.load_urls()` made outbound requests with no SSRF protection beyond a scheme check** (#1212) by @Sameer6305 — caller-supplied URLs went straight to `urllib.request.urlopen()`, unguarded against loopback/private addresses, cloud metadata endpoints (`169.254.169.254`), IPv6-internal addresses, hostnames resolving to private space, or redirects into any of the above. Found during a project-wide SSRF audit following #936/#959. Now routed through the shared `request_with_ssrf_guard()`; an unsafe URL is skipped rather than aborting the rest of the ingestion batch. `OpenClawKGTool` (operator-configured, intentionally allowed to target `localhost` for local deployments) gains scheme/malformed-URL validation as defense in depth, without restricting its legitimate private-network use case. 29 new Agno tests, 26 new OpenClaw tests, all passing alongside the 15 pre-existing Agno integration tests
|
||||
|
||||
### Dependencies
|
||||
|
||||
- Routine version bumps with no application-facing behavior change: `anthropic` 0.121.0→0.122.0 (#1045), `botocore` 1.43.69→1.43.73 (#1047), `agno` 2.8.7→2.9.0 (#1050), `google-genai` 2.17.0→2.18.1→2.19.0 (#1163, #1205), `lxml` 6.1.1→6.1.2 (#1197), `charset-normalizer` 3.5.0→3.5.1 (#1201), `pypickle` 2.0.1→2.0.2 (#1203)
|
||||
|
||||
## [0.6.6] - 2026-08-20
|
||||
|
||||
### Added
|
||||
|
||||
- **Semantica RDF vocabulary, and deterministic entity/relationship IRIs** (#1109, closes #1107, closes #1101) by @fabio-rovai, reviewed by @KaifAhmad1
|
||||
- Every RDF/JSON-LD export mints terms in `https://semantica.dev/ns#`, and until now nothing declared what those terms meant — the namespace 404s and no vocabulary shipped with the package, so a consumer receiving an export had no way to tell `sem:text` from a typo of it, and no closed-world checker could validate an export at all
|
||||
- `semantica/ontology/vocabulary/semantica-ns.ttl` declares the terms the exporters actually emit — drawn from the emitting call sites in `export/rdf_exporter.py`, `export/json_exporter.py` and `provenance/manager.py`, not from what a vocabulary "ought" to contain. Ships inside the package (`from semantica.ontology.vocabulary import vocabulary_turtle`) so it loads without a network round trip, and is the same document intended to be served at the namespace IRI once hosting/content-negotiation is sorted
|
||||
- `tests/ontology/test_vocabulary.py` ties the document to the code: every term a serializer can write must be declared, so adding a term to an exporter without declaring it fails the build
|
||||
- The missing-id fallback minted entity/relationship IRIs from Python's builtin `hash()`, randomised per process (`PYTHONHASHSEED`), so the same entity got a different IRI on every run and exports couldn't be diffed, deduplicated, or joined to an earlier provenance record. It also wrote `<semantica:entity_N>`, an IRI in the scheme `semantica` rather than the expansion of the declared prefix, so those nodes never joined with anything written through it. Minting now uses SHA-256 and writes a full IRI in the declared namespace; the same fix applies to the default entity/relationship types in the Turtle path
|
||||
- **Fixed during review** (Qodo): the temporal fallback minted from `source_id` only, while the main serializer accepts `source_id` or `source` — relationships using the second form hashed two empty strings, which the previous randomised `hash()` masked by making the IRI unstable anyway; once deterministic, unrelated relationships at the same list index collided on one IRI across exports. Endpoints are now resolved the same way `serialize_to_turtle` resolves them, before minting. `sem:confidence` also lost its declared `xsd:decimal` range: the N-Triples serializer types the same value `xsd:float`, and the two are disjoint, so declaring either contradicted one of the exporters (tracked in #1100) — a new `test_declared_ranges_do_not_contradict_what_the_exporters_emit` guards the whole class of that mistake
|
||||
- **Fixed in follow-up**: `serialize_to_rdfxml`'s default entity type still wrote the bare string `"semantica:Entity"` into an `rdf:resource` attribute, which (unlike a Turtle angle-bracket or an XML element name) is not namespace-expanded — the exact #1101 failure mode, just on the untested RDF/XML path. `json_exporter.py`'s `semantica:format` and `@type: "semantica:KnowledgeGraph"` were emitted but absent from both the vocabulary and the test's `EMITTED_TERMS` guard set, so the "undeclared terms fail the build" claim didn't actually cover them — both are now declared and guarded. `MANIFEST.in` didn't mirror the `pyproject.toml` package-data addition, so a source-distribution install could omit the vocabulary file. The cross-process minting-stability test replaced the subprocess's entire environment with a POSIX-only `PATH`, breaking it on Windows; now overrides only `PYTHONHASHSEED` on top of the inherited environment
|
||||
- **Also fixed, on the JSON-LD paths**: the first fix covered the Turtle, N-Triples and RDF/XML serializers, and left both JSON-LD writers interpolating the entity's own text into `f"semantica:entity/{text}"` and the endpoints into `f"semantica:rel/{source}_{target}"`. Three consequences, all live in 0.6.5: an entity whose text contained a space produced an invalid IRI, and a JSON-LD parser dropped that node in full rather than reporting it, so the entity disappeared from the export; every relationship carrying `source`/`target` rather than `source_id`/`target_id` minted the identical `semantica:rel/_`, collapsing all of them onto one node whose types and endpoints merged; and the JSON-LD `@id` disagreed with the Turtle IRI for the same entity, so the two serializations of one knowledge graph were two different graphs. Both JSON-LD writers now use `mint_entity_iri`/`mint_relationship_iri`, and `JSONExporter.export_entities`/`export_relationships` declare the `semantica` prefix their `@context` was already writing `semantica:entities` against — without it a processor reads that as an IRI in the scheme `semantica`, which is the original #1101 defect on a third path
|
||||
- `tests/export/test_jsonld_iri_minting.py` parses each export with a real JSON-LD processor and asserts the entity survives, the relationships stay distinct, no term expands into the `semantica` scheme, and the JSON-LD `@id` equals the Turtle IRI
|
||||
- 236 export and ontology tests pass
|
||||
|
||||
- **First-class CrewAI integration** (#988, closes #962) by @Shindevrp
|
||||
- New `pip install semantica[crewai]` extra (`crewai>=0.80.0`) — crewai core provides `BaseTool`/`BaseKnowledgeSource`, so `crewai-tools` is intentionally not included, and the extra is intentionally **not** part of the `all` bundle: crewai hard-requires `chromadb~=1.1.0`, which is affected by the unpatched pre-auth code-injection CVE-2026-45829 (see `integrations/crewai/README.md`)
|
||||
- `integrations/crewai/SemanticaKGTool` — a CrewAI `BaseTool` exposing 5 KG actions (`extract_entities`, `extract_relations`, `add_to_graph`, `query_graph`, `find_related`) backed by `NERExtractor` / `RelationExtractor` / `ContextGraph`; supports both sync `run()` and async `arun()`
|
||||
- `integrations/crewai/SemanticaDecisionTool` — a CrewAI `BaseTool` wrapping `AgentContext` with 5 decision-intelligence actions (`record_decision`, `find_precedents`, `trace_causal_chain`, `analyze_impact`, `check_policy`)
|
||||
@@ -41,6 +159,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- New `tests/export/test_distance_exporter_metric_errors.py`: 6 tests covering success, single/multiple failures, opt-out, the no-path-vs-error distinction, and default-schema stability; existing `tests/export/test_distance_exporter.py` updated for the new tuple return type
|
||||
- Full `tests/export/` suite: 77 passed
|
||||
|
||||
- **`ContextGraph.to_kg_dict()`: an adapter converting a `ContextGraph`'s internal `nodes`/`edges`/`source` shape into the canonical `entities`/`relationships`/`source_id` shape `RDFExporter` and `TemporalGraphQuery` consume** (#1081) by @cxzg007
|
||||
- Previously there was no supported way to feed a `ContextGraph` into those consumers without hand-rolling the field remapping; `to_kg_dict()` does it once, with an `entities_only` option that drops relationships left dangling by the filter
|
||||
- **Fixed during review** (Qodo): null `properties`/`metadata` on a node loaded from JSON raised `TypeError` when copied — both are now guarded with `or {}`; entity ids are coerced to `str(node_id)` to match `ContextEdge`'s already-str-coerced endpoints, so valid relationships were no longer dropped by `entities_only` filtering
|
||||
- `RDFExporter`'s validator and `TemporalGraphQuery` now also accept `source_id`/`target_id` endpoints, the shape `to_kg_dict()` emits
|
||||
|
||||
### Changed
|
||||
|
||||
- **`GraphBuilder`'s 6 public methods now have Google-style docstrings** (#878, closes #876) by @cakeni
|
||||
@@ -50,7 +173,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- **Corrected during review**: `add_temporal_edge`/`create_temporal_snapshot` docstrings overclaimed numeric-timestamp support; `_parse_time()` only special-cases `str` and `datetime`, falling back to a bare `str()` cast for anything else (not true numeric parsing). Narrowed to "datetime or ISO-formatted string"
|
||||
- **Fixed along the way**: `build()`'s `**options` documented a default only for `extract`; `extract_relations`, `extract_triplets`, `ner_method`, `relation_method`, and `triplet_method` all have concrete defaults in `_extract_from_text()` (`True`, `True`, `"llm"`, `"llm"`, `"llm"`) that were left unstated, inconsistent with CONTRIBUTING.md's own docstring example of noting defaults inline
|
||||
- `python -m pytest tests/kg/test_kg.py tests/kg/test_graph_builder_external.py -q`: 45 passed
|
||||
- **`GraphBuilder` raw-text extraction now defaults to local extractors instead of LLM extraction** (closes #930) by @dex0shubham
|
||||
- **`GraphBuilder` raw-text extraction now defaults to local extractors instead of LLM extraction** (#941, closes #930) by @dex0shubham
|
||||
- `GraphBuilder._extract_from_text()` defaulted `ner_method`, `relation_method`, and `triplet_method` to `"llm"`, and ran relation extraction unconditionally (`extract_relations` defaulted to `True`) — all four contradicting the defaults documented in the `build()` docstring at the time (`"ml"` / `"pattern"` / `False`), and diverging from the standalone extractors (`NERExtractor` defaults to `method="ml"`, `RelationExtractor` and `TripletExtractor` to `method="pattern"`). The practical effect was that any raw-text `build()` call silently required a configured provider, an API key, and network access
|
||||
- Defaults are now `ner_method="ml"`, `relation_method="pattern"`, `triplet_method="pattern"`, and `extract_relations=False`, matching the docstring. LLM extraction remains fully available and is now opt-in
|
||||
- **To restore the previous behaviour**, pass the methods explicitly:
|
||||
@@ -71,8 +194,57 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- New regression coverage in `tests/kg/test_graph_builder_extraction_defaults.py` pinning all four defaults, verifying that no default resolves to `"llm"`, confirming explicit LLM opt-in still routes correctly, asserting extractors are constructed once across repeated texts, covering fallback method lists (e.g. `ner_method=["pattern", "ml"]`) for all three extractors, asserting relations are forwarded to triplet extraction (and that `None` is forwarded when relation extraction is disabled or fails), and running the real default path end to end with no provider mocked. Verified to fail against the pre-fix code
|
||||
- Full `kg` suite: 473 passed
|
||||
|
||||
- **Explorer graph canvas now renders edge labels** (#1013, closes #1009) by @yzxcj797
|
||||
- `GraphCanvas.tsx` had no edge-label rendering path at all; Sigma's edge-label renderer draws `data.label`, but the graph state stored the relationship type under `edgeType`, so simply enabling the renderer would have left every edge blank. `graphSceneState`'s edge reducer now maps `edgeType` onto `label` (suppressed for hidden edges)
|
||||
- Rendering is gated behind a new `edgeLabelsEnabled` entry in the Effects panel (default on), wired through the existing `GraphEffectToggle`/`GraphEffectsState` plumbing, so dense graphs can still turn labels off
|
||||
- **Fixed during review** (Qodo): two follow-up passes closed gaps the first cut left — label rendering wasn't wired through `explorationEffectsPluginPhaseC.tsx`'s Phase C variant, and toggling the effect off mid-session didn't clear already-rendered labels
|
||||
- New coverage in `explorer/tests/graphSceneState.display.test.ts`
|
||||
|
||||
- **Removed `GraphWorkspaceShell.tsx`, `GraphRuntimeStage.tsx`, and `useGraphData.ts` — a second, unused implementation of the graph-loading/error-handling logic already fixed in `GraphWorkspace.tsx`** (#984, resolves the cleanup tracked in #981 by #980's review note) by @lakshayxi
|
||||
- 1,564 lines removed; the surviving `GraphWorkspace` path is now the only implementation, so the "two copies that drifted apart" root cause #980 fixed can't recur in the copy nobody was maintaining
|
||||
|
||||
- **Explorer README and `docs/explorer-setup.md` corrected to describe the authentication 0.6.5 actually shipped**, plus a documented `/ws/graph-updates` auth note (#1040, fixes #1028) by @Kyou12138
|
||||
- Both docs still claimed the Explorer API had no built-in authentication after v0.6.5 added mandatory `SEMANTICA_API_KEY` enforcement with a `503` fail-closed default; corrected to describe the actual behavior, including that only protected routes require the key (`/api/health`/`/api/info` stay open), the non-loopback-bind CLI warning only fires in anonymous mode or when the key is unset, and `SEMANTICA_API_KEY`/`SEMANTICA_ALLOW_ANONYMOUS` are documented in the environment-variable table
|
||||
|
||||
- **CI: pinned `github/codeql-action` to current v4** (#986) by @ZohaibHassan16, and **pinned Python dependencies in `requirements-ci.txt` for reproducible CI runs** (#945) by @yunaremaia, closing the gap where an unpinned CI dependency could silently change behavior between runs
|
||||
|
||||
- **README now states up front that Semantica's explainability is system-level, not foundation-model-internal** (#1033, #1034) by @KaifAhmad1
|
||||
- Nothing in the README previously scoped what "explainable" meant, leaving readers to assume Semantica could expose or reconstruct an LLM's internal reasoning. A callout now states explicitly that Semantica explains and audits what the AI *system* did — context fed in, decisions produced, provenance, relationships, policies applied — not the model's private internal reasoning, and moved the note near the top of the README rather than leaving it implicit
|
||||
|
||||
### Fixed
|
||||
|
||||
- **KG provenance tests asserted on generated ID strings instead of stored records, and `kg_provenance.py` was missed by the `utcnow` sweep** (closes #946) by @pravit-amp
|
||||
- The KG workflow and integration suites checked that a tracker call returned an ID matching a prefix (`assert cent_id.startswith("centrality_")`) without ever reading the record back, so an ID generator that returned a well-formed string and wrote nothing would have passed. Worse, some of those calls named tracker methods that do not exist anywhere in `semantica/` (`track_layer_analysis`, `track_centrality_score`), so the assertions were satisfied with no real interaction behind them
|
||||
- Those tests now read provenance back through `get_provenance()` and assert on algorithm metadata, and call the methods that actually persist records. Verified by mutation rather than by a green run alone: neutering the manager's storage write (`self.storage.store(...)` → no-op) fails 10 tests
|
||||
- `GraphBuilderWithProvenance` in `semantica/kg/kg_provenance.py` still stamped `activity_started_at_time`/`activity_ended_at_time` with the deprecated `datetime.utcnow()`; it was outside the `export/`+`provenance/` scope of the #1114 sweep below and now uses the same `utc_now_iso()` helper. `docs/guides/provenance.md` and `docs/reference/provenance.md` were still documenting `utcnow()` and a naive timestamp example, and now show the helper and the offset-bearing form
|
||||
- 16 tests across the affected suites ended in `return <value>` instead of asserting, which pytest reports as `PytestReturnNotNoneWarning`; now zero
|
||||
|
||||
- **The temporal-evolution `stability` metric was a hardcoded placeholder, not a duration**
|
||||
- `TemporalGraphQuery.analyze_evolution()` documents `stability` as a "relationship duration/stability measure", but the implementation appended a constant `1` for every relationship with both `valid_from` and `valid_until` set (`durations.append(1) # Placeholder`). The reported stability was therefore always `1.0` when any bounded relationship existed and `0` otherwise — it never reflected how long relationships actually stayed valid, so it could not distinguish a graph of decade-long relationships from one of one-second relationships
|
||||
- `stability` now computes the mean valid-time duration in seconds (`(valid_until - valid_from).total_seconds()`) across relationships that have both bounds set. Relationships with a missing or open `valid_from`/`valid_until` are skipped (their duration is unbounded), and non-positive intervals are clamped to `0`; an empty set still reports `0`
|
||||
- New tests in `tests/kg/test_kg.py` assert the mean-duration result, the skipping of unbounded/half-open intervals, and the empty-graph zero case
|
||||
|
||||
- **Every timestamp an export or a provenance record wrote was timezone-naive** (closes #1114) by @fabio-rovai
|
||||
- `semantica/export/` stamped with `datetime.now().isoformat()`, which reads the machine's **local** clock; `semantica/provenance/` stamped with `datetime.utcnow().isoformat()`, which reads **UTC**. Both produce a naive value and both serialize identically, so nothing downstream can tell which zone a given timestamp belongs to — the same string means two different instants depending on which module wrote it
|
||||
- In RDF the consequence is silent rather than loud. Under XSD 1.1 a value with no timezone compared against one with a timezone is indeterminate whenever the two fall inside the ±14 hour window; SPARQL turns an indeterminate comparison into an error, and `FILTER` discards errors as non-matches. A timezone-qualified query over an Oxigraph store returns an answer with every Semantica-written record quietly absent from it, which is a poor property for `prov:generatedAtTime`, `prov:startedAtTime`, `prov:endedAtTime` and `prov:atTime` to have
|
||||
- New `utc_now()`/`utc_now_iso()` in `semantica/utils/helpers.py`, exported from `semantica.utils`, and used at all 29 call sites in `export/` (`json_exporter`, `yaml_exporter`, `report_generator`, `export_provenance`) and `provenance/` (`manager`, `schemas`, `bridge_axiom`). Values now read `2026-08-19T14:19:04.229937+00:00`: one unambiguous instant, comparable against any correctly stamped value, and valid `xsd:dateTimeStamp`. `sem:exportedAt`'s range in `semantica/ontology/vocabulary/semantica-ns.ttl` is tightened from `xsd:dateTime` accordingly, and its comment no longer has to explain why the weaker range was necessary
|
||||
- `datetime.utcnow()` is deprecated as of Python 3.12 and scheduled for removal; constructing a `ProvenanceEntry` under `-W error::DeprecationWarning` on 3.13 raised, and no longer does
|
||||
- New `tests/export/test_timestamp_timezones.py` and `tests/provenance/test_timestamp_timezones.py`: offset presence on every export and provenance path, PROV-O literals valid as `xsd:dateTimeStamp`, comparison against a timezone-aware instant without `TypeError`, the Oxigraph filter that dropped the naive value (with a bound inside the indeterminate window, so the test cannot pass by accident), and the document `@id` remaining a valid IRI with `+00:00` in it. 11 of the 13 fail on the parent commit
|
||||
- **Fixed during review** (Qodo): once new entries carry `+00:00` and stored ones do not, `ProvenanceManager.query_recorded_between` and `audit_log` compared ISO timestamps as raw strings, so they ordered by spelling rather than by instant — an inclusive naive bound naming a stored offset-bearing timestamp sorted *below* it and dropped the record, and a bound written in another offset landed wherever its digits fell (`19:45+05:30` is 14:15Z, but sorted after 14:19Z). Both now compare instants through a new `to_utc_datetime()` helper that reads a missing offset as UTC, which is what the values written before this change actually were; a bound that cannot be read as a timestamp keeps the historical string comparison rather than raising on a call that used to work
|
||||
- The remaining 147 naive call sites are in `context/`, `vector_store/`, `seed/` and elsewhere, where timestamps are compared against values parsed from previously stored naive strings. Converting those without a read-side migration would raise `TypeError: can't compare offset-naive and offset-aware datetimes` on existing data, so they are deliberately left for a separate change
|
||||
- **`SHACLGenerator` mangles `#`-terminated namespaces into `#/`, so generated shapes target nothing** (#1082) by @changshenhan
|
||||
- `__init__` normalized `base_uri` with `rstrip("/") + "/"`, which turns `http://example.org/manufacturing#` into `...manufacturing#/` — the most common RDF namespace convention. Every generated URI (`sh:targetClass`, `sh:path`, shape URIs) then landed in a different namespace than the instance data, and SHACL validation silently passed because the shapes targeted nothing
|
||||
- `__init__` now preserves a namespace already ending in `/` or `#`, matching the `#`-aware normalization `generate()` already applies; `shapes_uri` inherits the fix
|
||||
- New `test_hash_namespace_base_uri_is_not_mangled` in `tests/ontology/test_ontology_advanced.py` fails on the pre-fix normalization and passes with it; full ontology suite (76 tests) green
|
||||
|
||||
- **`split`/chunking paths bypassed the centralized spaCy model cache, reloading the model on every call** (#1042, closes #998) by @Accute9, reviewed by @Sameer6305
|
||||
- `semantica/split/methods.py`'s `split_by_sentences()` and `semantica/split/semantic_chunker.py`'s `SemanticChunker.__init__` each called `spacy.load()` directly instead of reusing the process-level cache added in #889/`semantic_extract/methods.py`'s `load_spacy_model()` — every call/construction re-paid the ~120ms model-load cost independently of `NERExtractor`, which already used the cache
|
||||
- Both now route through `load_spacy_model()`, sharing one cached `Language` instance per model name across `split_by_sentences()`, `SemanticChunker`, and `NERExtractor`; a missing model still falls back to regex/paragraph chunking without poisoning the cache for a later successful load
|
||||
- **Fixed during review** (@Sameer6305): `NERExtractor.__init__()` still had a direct `spacy.load()` call site with the same cache-bypass issue, outside the two files named in #998 but sharing the same root cause; routed through the cache alongside stale test patch targets and a strengthened cache-configuration assertion
|
||||
- **Fixed during review** (@KaifAhmad1): `SemanticChunker.__init__` only caught `OSError` around `load_spacy_model()`, while the sibling fix to `NERExtractor` in this same PR added a broader `except Exception` for a model that is installed but fails at runtime (e.g. a config incompatible with the installed spaCy version). A broken-but-present model crashed `SemanticChunker()` outright instead of degrading to fallback chunking like every other path in this PR. Added the matching `except Exception` branch, leaving `self.nlp` as `None`; new `test_semantic_chunker_falls_back_when_spacy_runtime_is_broken` mirrors the existing `NERExtractor` regression test for the same scenario
|
||||
- New `tests/split/test_spacy_model_cache.py`: cache reuse across repeated calls/instances, shared cache between `split_by_sentences()`/`SemanticChunker`/`NERExtractor`, distinct model names loading separately, missing-model fallback without poisoning the cache, and the broken-runtime fallback added above
|
||||
- `pytest tests/split/test_spacy_model_cache.py tests/split/test_splitter.py tests/split/test_chunkers.py`: all passing (3 pre-existing, unrelated `tests/test_ner_configurations.py` failures confirmed present on `main` before this PR)
|
||||
|
||||
- **`export_yaml` raised a raw `AttributeError` on list input, silently wrote empty exports for unrecognized dict keys, and graph payloads were reconciled differently by every exporter** (#958, closes #956, #952, #953) by @pravit-amp, reviewed by @Sameer6305
|
||||
- Graph payloads circulate under two vocabularies, `entities`/`relationships` and `nodes`/`edges`, and each exporter reconciled them locally with a different idiom — `LPGExporter` in particular dropped every entity whenever `nodes` was present but empty, the exact shape `JSONExporter` emits. A new `normalize_graph_payload()` in `utils/helpers.py` centralizes that decision once, adopted by `LPGExporter`, `ArangoAQLExporter`, `Neo4jCSVExporter`, and both YAML exporters; `ContextGraph.to_dict()` now round-trips through YAML correctly as a result
|
||||
- `export_yaml(records, path)` on a bare list previously failed with `AttributeError` from inside the exporter; it and the other YAML methods now reject non-mapping input with an actionable `ProcessingError` naming the expected keys, since these formats distinguish entities/relationships/triplets and guessing which one a list represents would mislabel the records
|
||||
@@ -166,8 +338,86 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- New regression coverage in `tests/export/test_distance_exporter.py`: warnings fire on exception for all four helpers, exported sentinel values/shape stay unchanged, and the legitimate "no KG backend" `None` path still logs nothing
|
||||
- Full `tests/export/` suite: 71 passed
|
||||
|
||||
- **`explain_violations` rendered hardcoded placeholders (`min_count=1`, `max_count=1`) instead of the SHACL shape's real constraint values, and misused the violation message text as the datatype/class value** (#1094) by @cxzg007
|
||||
- `_run_pyshacl` never read `sh:minCount`/`sh:maxCount`/`sh:datatype`/`sh:class` back from the violation's `sh:sourceShape`, so every plain-English explanation was wrong regardless of what the shape actually declared. `SHACLViolation` now carries those four fields (also exposed via `to_dict()`), populated by back-referencing `sh:sourceShape`; `explain_violations` renders the real values, falling back to `"?"` when a value is genuinely absent
|
||||
- **Known limitation**: `sh:qualifiedMinCount`/`sh:qualifiedMaxCount` are not handled yet and still fall back to the `"?"` placeholder
|
||||
- New regression tests cover both the rendering path and the `sh:sourceShape` back-reference (skipped when `pyshacl`/`rdflib` are absent)
|
||||
|
||||
- **Entity merging silently dropped `entity_id` aliases, and exact-match entity resolution had three correctness gaps** (#1086, #1026) by @T1mn
|
||||
- `entity_merger.py`/`merge_strategy.py`/`entity_resolver.py` used inconsistent logic for extracting an entity's id across the merge path, so a merged entity could lose the `entity_id` aliases that let later lookups find it under its old identity. A new `semantica/utils/entity_ids.py` unifies id extraction across all three call sites
|
||||
- `EntityResolver`'s exact-match path is now honored rather than silently falling through to fuzzy matching in some cases; entities with no identifier are preserved instead of being dropped, and blank exact-match names are ignored rather than matching every other blank name
|
||||
- New/expanded coverage in `tests/kg/test_entity_pipeline.py` and `tests/kg/test_entity_resolver_exact.py`
|
||||
|
||||
- **`flatten_dict()` silently collided keys when a flattened path from one branch matched a literal key already present at the target depth** (#1062) by @shahzaib-ahmadcs
|
||||
- Two differently-shaped inputs could flatten to the same output key, with the second write silently overwriting the first — no error, no warning, just a dropped value. Collisions are now detected and handled explicitly instead of overwriting
|
||||
|
||||
- **`ExcelParser.__init__` raised `NameError` on every instantiation — `get_progress_tracker()` was called but never imported** (#1016, closes #1014) by @pravit-amp
|
||||
- Same defect as the one fixed for `SimilarityCalculator` in #530, this time in `semantica/parse/excel_parser.py`; the existing test imported the class but never constructed it, so nothing caught the missing import. Added construction coverage for every parser exported from `semantica.parse`, driven off `__all__` so future additions are covered automatically, living outside `test_parse_comprehensive.py` (whose `setUp` mocks `get_progress_tracker` into each module and would mock away the exact interaction under test)
|
||||
|
||||
- **Graph analytics (`centrality_calculator.py`, `community_detector.py`, `connectivity_analyzer.py`) dropped isolated nodes and diverged on how each computed its working view of the graph** (#1011) by @T1mn
|
||||
- Each analyzer had its own ad hoc logic for building the node/edge set it operated over, and none of them included nodes with no edges — a node with zero connections simply vanished from centrality scores, community assignments, and connectivity reports instead of appearing with a zero/singleton value. A new shared `semantica/kg/_graph_view.py` centralizes graph-view construction (including node fallbacks and community payload shaping) for all three analyzers, which are now ~250 lines lighter combined
|
||||
- New `tests/kg/test_analytics_node_scope.py` covering isolated-node presence across all three analyzers
|
||||
|
||||
- **Explorer fired temporal-bounds and snapshot requests before the graph itself had loaded, tripling failed requests when the backend was down and leaving the timeline scrubber with nothing to scrub** (#1003) by @lakshayxi
|
||||
- Two new predicate functions gate the temporal effects on the graph having actually loaded (an empty graph still counts as loaded); confirmed against a downed backend that this cuts three failing requests per page load down to one
|
||||
|
||||
- **`SeedDataManager.load_from_database()` never actually reached the database, and connection failures were mislabeled as a missing optional dependency** (#995, closes #973) by @yzxcj797
|
||||
- `DBIngestor.execute_query`/`export_table` need the connection string as their first positional argument; `load_from_database()` only passed it into the constructor's config dict, which those methods never read, so every call raised `TypeError` before connecting. Also split the combined `except (ImportError, OSError)` handling apart — a genuine connection failure was reported as `"module not available"`, sending debugging in the wrong direction; `OSError` now propagates as an actual failure, chained via `from e`
|
||||
|
||||
- **SPARQL `CONSTRUCT` detection matched inside a leading `#`-comment, misclassifying `SELECT`/`ASK` queries as `CONSTRUCT` across all four SPARQL backends** (#951) by @pravit-amp
|
||||
- `CONSTRUCT_QUERY_RE` skipped comments with a bare `\#[^\n]*`, whose backtracking `*` let a `# CONSTRUCT ...` comment line "swallow" the real query-form keyword on the next line for a query like `# CONSTRUCT ...\nSELECT ...`. The mistaken `CONSTRUCT` classification sent `Accept: text/turtle` and tried to parse a SELECT/ASK response body as Turtle, failing with a misleading parse error. The regex now requires a comment to reach a line terminator (LF or CR, per the SPARQL grammar) before matching
|
||||
|
||||
- **`k_shortest_paths` mutated caller-visible graph state during traversal and ignored direction when excluding already-used edges** (#1000) by @T1mn
|
||||
- `semantica/kg/path_finder.py`'s search left side effects behind after returning, and edge exclusion during Yen's-algorithm-style path removal didn't respect the traversal direction of directed graphs, letting a later search see edges that should have been available. Both fixed; new coverage in `tests/kg/test_path_finder.py`
|
||||
|
||||
- **`trace_decision_causality()` ignored explicitly recorded causal edges, inferring causes only from shared NER entities plus timestamp ordering** (#983) by @hsd2514
|
||||
- A `CAUSED`/`INFLUENCED`/`PRECEDENT_FOR` edge added via `add_causal_relationship()` had no effect on the trace — when entity extraction found nothing in common between two decisions, `trace_decision_chain()` came back empty even with an explicit edge stored in the graph. Explicit causal edges are now traversed first as ground truth, with entity/timestamp inference kept as an additive fallback for pairs with no explicit link; edges whose source has no decision record (e.g. a graph restored via `from_dict`) are skipped so a stale edge can't abort the trace
|
||||
|
||||
- **`RepoIngestor`'s module-level DNS resolve cache had no lock, raising `RuntimeError: OrderedDict mutated during iteration` under concurrent `ingest_repository()` calls** (#979) by @manjunathbhaskar
|
||||
- `_REPO_HOST_RESOLVE_CACHE` is a shared `OrderedDict` read, written, and pruned by every thread with no synchronization — reliably reproduced with 32 threads hammering resolution under a low TTL and small cache cap. Now guarded by a lock
|
||||
|
||||
- **`GraphBuilder` didn't remap relationship endpoints after entity resolution merged nodes, leaving relationships pointing at ids that no longer existed in the resolved graph** (#978) by @T1mn
|
||||
- New coverage in `tests/kg/test_graph_builder_external.py`; a follow-up commit hardens the remapping against edge cases found during review
|
||||
|
||||
- **Explorer's dev server esbuild target didn't match the browser targets the production build declares**, occasionally producing dev-only syntax errors on older browsers (#966) by @le-czs
|
||||
- `explorer/vite.config.ts` now sets the dev esbuild target explicitly to match
|
||||
|
||||
- **`normalize`'s number normalizer accepted currency symbols without validating them against the surrounding text, and an earlier fix's currency-code matching wasn't token-bounded** (#940) by @Mr-Neutr0n, reviewed by @ZohaibHassan16
|
||||
- Symbol currencies are now validated before being accepted; currency codes are matched on token boundaries so a code embedded inside a longer token no longer false-positives
|
||||
|
||||
- **`ContextGraph.to_dict()` was the one reader on the class that didn't hold `self._lock`, raising `RuntimeError: dictionary changed size during iteration` under a concurrent writer and risking a torn snapshot otherwise** (#929) by @pravit-amp
|
||||
- Every other reader (`stats()`, `density()`, `find_nodes()`, `find_edges()`, `get_neighbors()`, `get_nodes_by_label()`, `state_at()`, `save_to_file()`) already took the lock after it was introduced; `to_dict()` predated that change and was missed. `save_to_file()` was safe only incidentally, since it builds its payload inline under its own lock rather than delegating to `to_dict()`
|
||||
|
||||
- **`PipelineWithProvenance` had a broken import and no working `run()` method** (#862) by @Karunasagar12
|
||||
- `from .pipeline import Pipeline` failed because `Pipeline` lives in `pipeline_builder.py`, not a nonexistent `pipeline.py` — fixed to `from .pipeline_builder import Pipeline`. The class also had no `run()`; it now delegates to `ExecutionEngine.execute_pipeline()`, the intended execution path for a built `Pipeline`. The constructor now accepts a built `Pipeline` instance directly
|
||||
|
||||
### Security
|
||||
|
||||
- **Tarball restore path traversal, latent SQL injection, DNS-rebinding TOCTOU in the shared SSRF guard, stored XSS in report generation, and unvalidated SPARQL object IRIs in AnzoStore** (#1079) by @KaifAhmad1
|
||||
- `semantica backup restore`'s tar extraction (`cli.py`) stripped only the literal `semantica-backup/` prefix and called `tar.extract()` with no path-containment check, no symlink/hardlink validation, and (on Python <3.12) no extraction filter — a crafted archive member (`../../<file>`, or a symlink pointing outside the restore root) could write arbitrary files above the restore directory. Every member is now validated for resolved-path containment before extraction, symlink/hardlink targets are rejected both lexically (absolute path, `..` segments) and by resolution, and `filter="data"` is applied on Python ≥3.12
|
||||
- `DataExporter.export_table_data()` (`db_ingestor.py`) was missing the `text` import from `sqlalchemy` — a `NameError` that made the method non-functional, but latently: the query it built from raw f-string interpolation of `table_name`/`schema`/`where`/`order_by` was already injectable, so fixing the import alone (without also fixing the injection) would have silently armed it. Both are fixed together: the import is restored, `table_name`/`schema` are now validated against a strict identifier allowlist, and `where`/`order_by` are checked against a blocklist (statement separators, comments, UNION, DDL/DML keywords, time-based blind-injection primitives, schema-enumeration terms). This is a blocklist, not a grammar — it closes the concrete UNION-exfiltration path and common injection primitives, but a boolean-blind subquery using none of the blocked keywords could still get through; `where`/`order_by` must be treated as trusted/operator input, not exposed to untrusted end users, and the docstrings now say so explicitly
|
||||
- `request_with_ssrf_guard()` (`ssrf.py`) validated a hostname's resolved IPs, then let the underlying HTTP client re-resolve the same hostname independently at connect time — a low-TTL or DNS-rebinding answer could differ between the two lookups, so a hostname that validated as public could still connect to a private/internal address. Ported the IP-pinning pattern already used by `explorer/routes/ontology.py`'s `_make_pinned_session` into the shared ingest guard: the one resolution that decides accept/reject is now also the one the connection is pinned to, via a custom `HTTPAdapter` that presents the real hostname over TLS SNI / Host header while connecting only to the validated IPs. Also closes the RFC 6598 Carrier-Grade NAT gap noted as a known limitation in #905/#868: `100.64.0.0/10` is now in `BLOCKED_NETWORKS`
|
||||
- `ReportGenerator._generate_html()` (`export/report_generator.py`) f-string-interpolated report title/summary/metrics into HTML with no escaping — an ingested entity or document whose content flowed into a report (e.g. `<img src=x onerror=...>`) executed as stored XSS when the report was opened. All interpolated values are now `html.escape()`d
|
||||
- `AnzoStore._format_object_for_sparql()` (`triplet_store/anzo_store.py`) validated the subject/predicate of a triplet via `sparql_escaping.validate_uri()` before interpolating them into a SPARQL `INSERT DATA` clause, but delegated the **object** position to a separate formatter that wrapped it as `<{obj}>` without the same validation — an object value containing `>`/`}`/`{`/`"` could close the intended `<...>` token early and inject additional SPARQL Update operations. The Blazegraph/RDF4J backends were hardened for the equivalent gap previously; Anzo's object position now goes through the same `validate_uri()` check
|
||||
- Also hardened in the same pass: Apache AGE's `create_index()` `index_type` parameter is now allowlisted (was interpolated raw into a `USING` clause); Neo4j's `limit` is now explicitly validated (raises `ValidationError` for non-integer input instead of falling through to a generic `ProcessingError`); the `ffprobe` metadata-extraction subprocess call is guarded against a filename starting with `-` being parsed as an option; the MCP server no longer echoes raw exception text to JSON-RPC clients, logging full details server-side and returning a generic message plus the exception class name instead
|
||||
- **Fixed during review** (@KaifAhmad1): the SSRF IP-pinning change introduced a connection-pool leak of its own — `requests.Session.mount()` silently drops whatever adapter it replaces without closing it, so a multi-hop redirect chain on a reused session leaked one pooled connection per hop. Pinned adapters are now tagged and explicitly closed before being replaced, both per-hop and on final restore
|
||||
- **Fixed during review** (@KaifAhmad1): mounting a pinned adapter and setting a Host header on a caller-supplied `Session` is not inherently thread-safe — two guarded calls sharing the same session from different threads could interleave their mount/restore cycles. Added a per-session lock (`_get_session_lock`) so concurrent guarded calls on the same session now serialize instead of racing; verified with a two-thread test showing correct serialization and zero cross-contamination of per-request Host headers
|
||||
- **Fixed during automated PR review** (Qodo): `export_table_data()`'s new identifier/fragment validation raised `ValidationError` from inside a `try` whose blanket `except Exception` re-wrapped it as `ProcessingError`, masking the distinction between "bad input" and "the export itself failed" that callers rely on elsewhere in this module. Added the `except ValidationError: raise` guard already used by its sibling methods
|
||||
- **Fixed during automated PR review** (Qodo): on a hop where IP pinning doesn't apply (`allow_private_ips=True`), `_apply_connection_pin()` unconditionally popped the session's `Host` header instead of restoring whatever it was before pinning touched it — a caller-supplied session carrying its own legitimate `Host` override (e.g. fronting a private endpoint under a different name) had that override silently dropped for the in-flight request, only reappearing afterward via the outer `finally` restore. It now restores the session's own pre-call header state (set back if present, popped only if it was truly absent) instead of always popping
|
||||
- **Fixed during automated PR review** (Qodo): the `where`/`order_by` blocklist matched keywords/punctuation inside properly quoted string literals and identifiers too, so legitimate data like `status = 'union'` or `name = 'a--b'` was rejected as if it were SQL syntax. The blocklist now runs against a copy with quoted-literal contents masked out (`_mask_sql_literals`) — a malformed/unterminated quote sequence doesn't match the masking pattern and is left fully exposed to the blocklist, so this closes false positives without opening a masking-based bypass; the fragment actually used in the query is unchanged
|
||||
- Re-ran each finding's proof-of-concept (or an equivalent adversarial test) against the fix and confirmed it is blocked: tar path/symlink traversal (both lexical and resolved-path forms), SQL UNION exfiltration and identifier breakout, DNS-rebinding TOCTOU (including under a configured `HTTP_PROXY`, which the pinning adapter also rejects outright since a proxy would resolve DNS itself), stored XSS, and the AnzoStore SPARQL injection
|
||||
- `pytest tests/ingest/`: 266 passed, 2 skipped (10 pre-existing failures unrelated to this change — identical failure set confirmed on unmodified `main`); full regression sweep across `graph_store`, `export`, `triplet_store`, `parse`, and backup/restore: 313 passed
|
||||
|
||||
- **`Authorization`/`Proxy-Authorization` credentials could leak to a different origin across HTTP redirects, and several ingest paths bypassed the shared SSRF/redirect guard entirely** (#1067, closes #947) by @Sameer6305, reviewed by @KaifAhmad1
|
||||
- `request_with_ssrf_guard()` previously only stripped sensitive headers from per-request `kwargs["headers"]` on a cross-origin redirect; session-level `Authorization`/`Proxy-Authorization` headers, `session.auth`, and `session.trust_env` (`.netrc` lookup) could all still resurrect credentials on the hop to a foreign origin. All five credential sources are now stripped case-insensitively, kept stripped for the remainder of a multi-hop redirect chain (no resurrection even if a later hop returns to the original host), and unconditionally restored via `finally` — including on exceptions and redirect-limit errors
|
||||
- `MCPClient._send_request_http()` and `PublicAPIIngestor.detect_public_api()`/`ingest_public_api()` called `httpx.post()`/`requests.post()`/`session.request()` directly, bypassing `request_with_ssrf_guard()` entirely. Both now route through the shared guard, including when `validate_no_auth=False`
|
||||
- `SeedDataManager.load_from_api()` mutated the caller-supplied `headers` dict in place when adding an API-key `Authorization` header, silently leaking the key back into a dict the caller might reuse elsewhere. Now copies before modifying
|
||||
- **Fixed during review** (@KaifAhmad1): `allow_private_ips=True` (used to let MCP servers run on localhost/internal networks) was applied to every redirect hop, not just the operator-configured host — a compromised or malicious MCP server could 302-redirect to an internal address (e.g. `169.254.169.254` cloud metadata) and the guard would follow it unchecked, defeating the SSRF protection this PR otherwise adds. Added `allow_private_ips_on_redirect` to `request_with_ssrf_guard()`: a redirect target inherits the original host's private-IP trust only when it matches that host; any other host falls back to strict validation. `MCPClient` now pins `allow_private_ips_on_redirect=False`, so only same-host redirects on a trusted MCP server keep working — a cross-host hop into private address space is blocked
|
||||
- **Fixed during review** (@KaifAhmad1): `detect_public_api()` only caught `requests.exceptions.RequestException`, but `request_with_ssrf_guard()` raises `ValidationError` (a disjoint hierarchy) for SSRF-blocked hosts, blocked redirect targets, missing `Location`, or exceeded redirect limits — unlike its sibling `ingest_public_api()`, which already caught it. Callers (including `is_public_api()`) got an undocumented raw `ValidationError` instead of `ProcessingError`, and the error-logging call was skipped. Now catches `(ValidationError, ProcessingError)` and re-raises, matching the sibling method
|
||||
- **Fixed during review** (@KaifAhmad1): `detect_public_api()`/`ingest_public_api()` forwarded `**options` into `request_with_ssrf_guard(..., session=self.session, allow_private_ips=self.allow_private_ips, **request_options)` without stripping `session`/`allow_private_ips` from `request_options` first — a caller passing either through the per-call `**options` (a plausible mistake, since `allow_private_ips` is also a documented constructor-level knob) got a raw `TypeError: got multiple values for keyword argument`. Both are now popped from `request_options` before the call
|
||||
- New regression coverage added during review: `TestAllowPrivateIpsOnRedirect` (cross-host redirect into private space blocked, same-host redirect trust preserved, default behavior unchanged for existing callers that don't pass the new kwarg) and `TestMCPClientAuthRedirect::test_redirect_to_private_ip_is_blocked`/`test_same_host_redirect_on_private_mcp_server_is_not_blocked` in `tests/ingest/test_auth_header_redirect_security.py`; `test_detect_public_api_propagates_ssrf_validation_error` and duplicate-kwarg regression tests for both methods in `tests/ingest/test_public_api_ingestor.py`
|
||||
- `pytest tests/ingest/test_auth_header_redirect_security.py tests/ingest/test_public_api_ingestor.py tests/test_seed_manager.py tests/ingest/test_submodules.py tests/ingest/test_cookbook_integration.py`: 111 passed
|
||||
|
||||
- **`FeedIngestor`/`FeedMonitor` (RSS/Atom feed ingestion) had no SSRF protection, allowing requests to internal/private network targets** (#928, closes #927) by @ZohaibHassan16
|
||||
- `FeedIngestor.ingest_feed()`, `discover_feeds()` (link-tag fetch, common-path HEAD probe, and feed-validation GET), and `FeedMonitor.check_updates()` all called `requests.get()`/`requests.head()` directly with default redirect-following and no scheme allowlist or private/loopback/link-local IP validation — despite `semantica/ingest/ssrf.py`'s `request_with_ssrf_guard()` already existing and being used by `web_ingestor.py`/`api_ingestor.py`. `ingest_feed()`'s own URL check only verified `urlparse(url).scheme`/`.netloc` were non-empty, never that the scheme was http/https or that the resolved target IP was safe. Reachable via the public `ingest_feed()`/`ingest()` entry points with any caller-supplied feed URL
|
||||
- All 5 call sites now route through `request_with_ssrf_guard()`, which validates scheme (http/https only) and resolved IP before the request, and re-validates every redirect `Location` before following it — closing both the direct-IP and redirect-chain SSRF paths. Added an `allow_private_ips` config option to both `FeedIngestor` and `FeedMonitor`, consistent with the other ingestors
|
||||
@@ -201,6 +451,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- **Caught by the new gate on its first run**: `python -m pip install -e ".[all]"` pulled in `setuptools==79.0.1`, vulnerable to CVE-2026-59890/GHSA-h35f-9h28-mq5c/PYSEC-2026-3447 (Unicode-normalization bypass of `MANIFEST.in` exclude/prune patterns on macOS APFS/HFS+, letting excluded files leak into a built sdist), fixed in `83.0.0`. `[build-system] requires` had the exact same too-permissive-floor pattern this whole entry is about (`setuptools>=61.0`), and `actions/setup-python`'s baked-in `setuptools` isn't governed by that pin at all since it's outside any isolated build. Bumped `[build-system] requires` to `setuptools>=83.0.0`, and the `Security` workflow now runs `pip install --upgrade pip setuptools` before auditing so the scanned environment can't have a stale ambient copy regardless of what governs it
|
||||
- Full `explorer` suite: 241 passed
|
||||
|
||||
- **`SeedDataManager.load_from_api()` made unguarded HTTP requests, with no SSRF protection at all** (#942) by @ZohaibHassan16
|
||||
- `load_from_api()` called `requests.get()` directly instead of going through `semantica/ingest/ssrf.py`'s `request_with_ssrf_guard()`, unlike every other ingestor in this module — a caller-supplied `api_url` could target internal/private network addresses with no validation. Now routes through the shared guard, gaining redirect validation and bounded DNS resolution for free
|
||||
- **Follow-up** (#959, closes #943) by @yunaremaia: added an `allow_private_ips` opt-in (parsed via the shared `parse_bool` helper) for trusted internal deployments that legitimately need to load from a private-network API, while keeping the guard's block-by-default behavior for everyone else
|
||||
|
||||
## [0.6.5] - 2026-08-11
|
||||
|
||||
### Added
|
||||
@@ -261,8 +515,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
- Entities and relationships round-trip as memory-local provenance only — Markdown import intentionally does not write into `ContextGraph`, matching the MVP scope agreed on in #765
|
||||
- Documented the file contract and workflow in `docs/reference/context.md`; 43 new tests in `tests/context/test_agent_memory_markdown.py` cover round-trip losslessness, idempotency, validation errors, rollback on failure, and vector-store sync ordering
|
||||
|
||||
- **Markdown directory round trips for `ContextGraph`** (#852) by @SaurabhScripts
|
||||
- `ContextGraph.save_to_file(..., format="markdown")` and `load_from_file(..., format="markdown")` persist a deterministic `graph.md` relationship manifest plus one human-editable Markdown file per node, preserving graph, node, edge, family, temporal, and cross-graph link identities
|
||||
- Imports validate the complete directory before replacing graph state, rebuild indexes and analytics state atomically, create JSON-compatible stub nodes for dangling edge endpoints, and emit the same granular node/edge audit events as JSON loading
|
||||
- Existing exports are replaced atomically only after their complete canonical layout is validated; untracked files, renamed node files, symlinks, Windows directory junctions, and other reparse points cause a fail-closed error instead of authorizing directory deletion
|
||||
- Added 30 focused tests covering deterministic round trips, manual edits, validation rollback, managed-directory identity, publish rollback, audit-manager compatibility, stale-cache clearing, mocked and real Windows junctions, and missing-path behavior
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Markdown import followed filesystem links even though Markdown export already refused to overwrite them** (#851, follow-up to #765, #786) by @SaurabhScripts
|
||||
- `AgentMemory._read_markdown_path()` now rejects symlink files, broken symlinks, symlinked directories, Windows directory junctions, and other Windows reparse points supplied directly; linked entries discovered inside an otherwise valid directory are safely skipped, preserving the current directory-import contract
|
||||
- `_read_markdown_file_content()` re-checks the file and parent directory immediately before and after opening, uses `O_NOFOLLOW` where available, and verifies the resulting descriptor is a regular file via `fstat`/`S_ISREG`, so link swaps are rejected rather than silently followed
|
||||
- Junction detection uses `os.path.isjunction()` where available and falls back to the Windows reparse-point file attribute on older Python versions; export applies the same link check before replacing a Markdown file
|
||||
- Documented the import restriction in `docs/reference/context.md`; added 11 tests to `tests/context/test_agent_memory_markdown.py` covering file/directory/broken-symlink rejection, simulated open races, mocked and real Windows junctions, and the reparse-point fallback
|
||||
- Any additional review follow-up commits land in this same PR/entry rather than as a separate changelog item
|
||||
|
||||
- **`PipelineWithProvenance` raised `ModuleNotFoundError` on import and `AttributeError` on `.run()`** (#858, closes #858) by @Karunasagar12
|
||||
- `from .pipeline import Pipeline` failed because `semantica/pipeline/pipeline.py` does not exist; corrected to `from .pipeline_builder import Pipeline`
|
||||
- `.run()` called `self._pipeline.run()` on the `Pipeline` dataclass, which has no such method; replaced with `self._engine.execute_pipeline(self._pipeline, ...)` delegating to `ExecutionEngine`
|
||||
@@ -1379,4 +1646,4 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
---
|
||||
|
||||
For detailed release notes, see [GitHub Releases](https://github.com/Hawksight-AI/semantica/releases).
|
||||
For detailed release notes, see [GitHub Releases](https://github.com/semantica-agi/semantica/releases).
|
||||
|
||||
+1
-1
@@ -58,7 +58,7 @@ representative at an online or offline event.
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported to the community leaders responsible for enforcement through
|
||||
[GitHub Issues](https://github.com/Hawksight-AI/semantica/issues) with "[CoC]" prefix.
|
||||
[GitHub Issues](https://github.com/semantica-agi/semantica/issues) with "[CoC]" prefix.
|
||||
All complaints will be reviewed and investigated promptly and fairly.
|
||||
|
||||
All community leaders are obligated to respect the privacy and security of the
|
||||
|
||||
+3
-3
@@ -44,7 +44,7 @@ We recognize all types of contributions:
|
||||
All contributors are recognized in:
|
||||
|
||||
- This contributors list
|
||||
- [GitHub contributors page](https://github.com/Hawksight-AI/semantica/graphs/contributors)
|
||||
- [GitHub contributors page](https://github.com/semantica-agi/semantica/graphs/contributors)
|
||||
- Release notes for significant contributions
|
||||
- Community appreciation
|
||||
|
||||
@@ -54,7 +54,7 @@ All contributors are recognized in:
|
||||
|
||||
### Automatic Recognition
|
||||
|
||||
If you've made a commit, you'll automatically appear in [GitHub's contributors graph](https://github.com/Hawksight-AI/semantica/graphs/contributors).
|
||||
If you've made a commit, you'll automatically appear in [GitHub's contributors graph](https://github.com/semantica-agi/semantica/graphs/contributors).
|
||||
|
||||
### Using All-Contributors Bot
|
||||
|
||||
@@ -111,4 +111,4 @@ Every contribution, no matter how small, helps make Semantica better. Thank you
|
||||
|
||||
**Want to contribute?**
|
||||
|
||||
⭐ Give us a Star • 🍴 [Fork us](https://github.com/Hawksight-AI/semantica/fork) • Check out our [Contributing Guide](CONTRIBUTING.md) to get started!
|
||||
⭐ Give us a Star • 🍴 [Fork us](https://github.com/semantica-agi/semantica/fork) • Check out our [Contributing Guide](CONTRIBUTING.md) to get started!
|
||||
|
||||
+1
-1
@@ -9,7 +9,7 @@ RUN npm ci
|
||||
COPY explorer/ ./
|
||||
RUN mkdir -p /app/semantica && npm run build
|
||||
|
||||
FROM python:3.14-slim AS runtime
|
||||
FROM python:3.13-slim AS runtime
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Hawksight AI
|
||||
Copyright (c) 2026 Semantica
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
|
||||
@@ -1 +1,2 @@
|
||||
recursive-include semantica/static *
|
||||
recursive-include semantica/ontology/vocabulary *.ttl
|
||||
|
||||
@@ -2,7 +2,15 @@
|
||||
|
||||
<img src="Semantica Logo.png" alt="Semantica" width="420"/>
|
||||
|
||||
<a href="https://trendshift.io/repositories/18986?utm_source=repository-badge&utm_medium=badge&utm_campaign=badge-repository-18986" target="_blank" rel="noopener noreferrer"><img src="https://trendshift.io/api/badge/repositories/18986" alt="semantica-agi%2Fsemantica | Trendshift" width="250" height="55"/></a>
|
||||
<div style="display:flex; gap:10px; align-items:center; flex-wrap:wrap;">
|
||||
<a href="https://trendshift.io/repositories/18986?utm_source=repository-badge&utm_medium=badge&utm_campaign=badge-repository-18986" target="_blank" rel="noopener noreferrer">
|
||||
<img src="https://trendshift.io/api/badge/repositories/18986" alt="semantica-agi/semantica | Trendshift" width="250" height="55"/>
|
||||
</a>
|
||||
|
||||
<a href="https://trendshift.io/repositories/18986?utm_source=trendshift-badge&utm_medium=badge&utm_campaign=badge-trendshift-18986" target="_blank" rel="noopener noreferrer">
|
||||
<img src="https://trendshift.io/api/badge/trendshift/repositories/18986/weekly?language=Python" alt="semantica-agi/semantica | Trendshift" width="250" height="55"/>
|
||||
</a>
|
||||
</div>
|
||||
|
||||
### Graph-Native Infrastructure for Context and Accountable AI Systems
|
||||
|
||||
@@ -52,6 +60,8 @@ Most AI agents act without a trail. They store embeddings, not meaning: context
|
||||
|
||||
Semantica sits underneath your LLM, vector store, and agent framework as a deterministic infrastructure layer: no LLM required for graph construction, reasoning, or provenance.
|
||||
|
||||
> ⚠️ **System-level explainability, not foundation-model explainability.** Semantica does not expose or reconstruct what happens *inside* the LLM — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. Semantica explains what's *outside* the model: the context and data fed in, the decision produced, its provenance, relevant relationships, applied policies, and the full execution trail.
|
||||
|
||||
**Who it's for:**
|
||||
|
||||
- **AI/ML platform teams** shipping agents that make consequential decisions and need structured, queryable context built from fragmented raw data, not just a vector index
|
||||
@@ -77,7 +87,7 @@ Semantica sits underneath your LLM, vector store, and agent framework as a deter
|
||||
- **Graph Analytics:** Centrality, community detection, link prediction, and shortest-path queries over the graph you just built
|
||||
- **Polyglot Graph Storage:** Native RDF (embedded Oxigraph, Blazegraph, Apache Jena, Eclipse RDF4J via SPARQL) and Labeled Property Graphs (Neo4j, FalkorDB, Apache AGE, AWS Neptune via Cypher), plus vector stores, all swappable without touching your code
|
||||
- **Visualization:** Explore any graph, ontology, or timeline in an interactive browser workbench
|
||||
- **Drop-in Integrations:** Native Agno and CrewAI support, a full-featured MCP server, a comprehensive CLI, a REST API, and plugins across major editors
|
||||
- **Drop-in Integrations:** Native Agno, CrewAI, and LangChain support, a full-featured MCP server, a comprehensive CLI, a REST API, and plugins across major editors
|
||||
|
||||
---
|
||||
|
||||
@@ -132,11 +142,13 @@ compliant = graph.check_decision_rules({"category": "vendor_selection"}) # poli
|
||||
```bash
|
||||
semantica doctor
|
||||
# Python 3.11.9 pass
|
||||
# semantica 0.6.5 pass
|
||||
# semantica 0.6.7 pass
|
||||
# faiss vector store pass
|
||||
# Config file pass ~/.semantica/config.yaml
|
||||
```
|
||||
|
||||
**Running in a script or CI?** Progress bars are written only when stdout is an interactive terminal (or a Jupyter notebook), so piping and redirecting stay clean by default. Override with `SEMANTICA_DISABLE_PROGRESS=1` to silence progress everywhere, or `SEMANTICA_FORCE_PROGRESS=1` to keep it when stdout is redirected. `SEMANTICA_DISABLE_PROGRESS` takes precedence.
|
||||
|
||||
<div align="center">
|
||||
|
||||
If Semantica solves a real problem for you, a star helps others find it.
|
||||
@@ -293,17 +305,10 @@ graph.add_causal_relationship(d1, d2, relationship_type="CAUSED")
|
||||
prov.track_entity("patient_P4821", source="ehr/medication_orders_2024.json",
|
||||
metadata={"extractor": "NamedEntityRecognizer"})
|
||||
|
||||
# Export W3C PROV-O for regulator submission - RDFExporter expects
|
||||
# {"entities": [...], "relationships": [...]}, so map ContextGraph.to_dict()'s
|
||||
# {"nodes": [...], "edges": [...]} shape onto it first
|
||||
graph_dict = graph.to_dict()
|
||||
kg = {
|
||||
"entities": [{"id": n["id"], "type": n["type"], "text": n["content"]} for n in graph_dict["nodes"]],
|
||||
"relationships": [
|
||||
{"source_id": e["source"], "target_id": e["target"], "type": e["type"]}
|
||||
for e in graph_dict["edges"]
|
||||
],
|
||||
}
|
||||
# Export W3C PROV-O for regulator submission - to_kg_dict() is the official
|
||||
# adapter that emits the {"entities": [...], "relationships": [...]} /
|
||||
# source_id shape RDFExporter expects, so no manual field mapping is needed
|
||||
kg = graph.to_kg_dict()
|
||||
RDFExporter().export(kg, "audit_trail.ttl", format="turtle")
|
||||
```
|
||||
|
||||
@@ -877,20 +882,14 @@ fact = BiTemporalFact(
|
||||
recorded_at=datetime(2024, 3, 5),
|
||||
)
|
||||
|
||||
# Query facts valid within a time window - query_time_range() expects
|
||||
# {"relationships": [...]} with source_id/target_id keys, which differs from
|
||||
# ContextGraph.to_dict()'s {"nodes", "edges"} shape, so map it first
|
||||
graph_dict = graph.to_dict()
|
||||
kg_relationships = {
|
||||
"relationships": [
|
||||
{**e, "source_id": e["source"], "target_id": e["target"]}
|
||||
for e in graph_dict["edges"]
|
||||
]
|
||||
}
|
||||
# Query facts valid within a time window - to_kg_dict() is the official
|
||||
# adapter that emits {"entities", "relationships"} with source_id/target_id
|
||||
# keys, the shape query_time_range() expects (no manual mapping required)
|
||||
kg = graph.to_kg_dict()
|
||||
|
||||
tq = TemporalGraphQuery()
|
||||
facts_in_window = tq.query_time_range(
|
||||
kg_relationships, query="valid_facts", start_time="2024-01-01", end_time="2024-12-31"
|
||||
kg, query="valid_facts", start_time="2024-01-01", end_time="2024-12-31"
|
||||
)
|
||||
|
||||
# Normalize natural language temporal expressions - returns a (start, end) range
|
||||
@@ -1189,7 +1188,7 @@ Start with `semantica`, verify with `doctor`, build a graph, and explore the com
|
||||
|
||||
## Integrations
|
||||
|
||||
Native plugin bundles for Claude Code, Cursor, Codex, Windsurf, Cline, Continue, VS Code, and OpenClaw; a full-featured MCP server for any MCP-compatible client; a comprehensive REST API; and first-class Agno and CrewAI support for agentic frameworks. Every major LLM provider is already supported via `semantica.llms` and LiteLLM: OpenAI, Anthropic, Gemini, Mistral, Llama, Groq, Cohere, Azure, Bedrock, Ollama, DeepSeek, HuggingFace, and more.
|
||||
Native plugin bundles for Claude Code, Cursor, Codex, Windsurf, Cline, Continue, VS Code, and OpenClaw; a full-featured MCP server for any MCP-compatible client; a comprehensive REST API; and first-class Agno, CrewAI, and LangChain support for agentic frameworks. Every major LLM provider is already supported via `semantica.llms` and LiteLLM: OpenAI, Anthropic, Gemini, Mistral, Llama, Groq, Cohere, Azure, Bedrock, Ollama, DeepSeek, HuggingFace, and more.
|
||||
|
||||
MCP setup takes 30 seconds — see [MCP Server](#mcp-server) below.
|
||||
|
||||
@@ -1308,17 +1307,17 @@ MCP setup takes 30 seconds — see [MCP Server](#mcp-server) below.
|
||||
<strong>CrewAI</strong><br/>
|
||||
<sub>First-class · <code>pip install semantica[crewai]</code></sub>
|
||||
</td>
|
||||
<td align="center" width="12.5%">
|
||||
<a href="https://github.com/langchain-ai/langchain"><img src="https://github.com/langchain-ai.png?size=120" alt="LangChain" width="48" height="48" /></a><br/>
|
||||
<strong>LangChain</strong><br/>
|
||||
<sub>First-class · <code>pip install semantica[langchain]</code></sub>
|
||||
</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<th colspan="8" align="left">Already Supported via REST API & MCP</th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="12.5%">
|
||||
<a href="https://github.com/langchain-ai/langchain"><img src="https://github.com/langchain-ai.png?size=120" alt="LangChain" width="48" height="48" /></a><br/>
|
||||
<strong>LangChain</strong><br/>
|
||||
<sub>REST API · MCP</sub>
|
||||
</td>
|
||||
<td align="center" width="12.5%">
|
||||
<a href="https://github.com/langchain-ai/langgraph"><img src="https://github.com/langchain-ai.png?size=120" alt="LangGraph" width="48" height="48" /></a><br/>
|
||||
<strong>LangGraph</strong><br/>
|
||||
<sub>REST API · MCP</sub>
|
||||
@@ -1349,11 +1348,6 @@ MCP setup takes 30 seconds — see [MCP Server](#mcp-server) below.
|
||||
</tr>
|
||||
<tr>
|
||||
<td align="center" width="12.5%">
|
||||
<a href="https://github.com/langchain-ai/langchain"><img src="https://github.com/langchain-ai.png?size=120" alt="LangChain" width="48" height="48" /></a><br/>
|
||||
<strong>LangChain</strong><br/>
|
||||
<sub>Dedicated toolkit</sub>
|
||||
</td>
|
||||
<td align="center" width="12.5%">
|
||||
<a href="https://github.com/run-llama/llama_index"><img src="https://github.com/run-llama.png?size=120" alt="LlamaIndex" width="48" height="48" /></a><br/>
|
||||
<strong>LlamaIndex</strong><br/>
|
||||
<sub>Dedicated toolkit</sub>
|
||||
@@ -1469,18 +1463,18 @@ For contributor / dev-server setup: **[explorer/README.md: Local Setup Guide](ex
|
||||
|
||||
---
|
||||
|
||||
## What's New in v0.6.5
|
||||
## What's New in v0.6.7
|
||||
|
||||
**Security release — upgrading is strongly recommended.** Fixes for 5 externally-reported vulnerabilities in the Explorer API and graph/triplet store backends, plus a CodeQL-flagged ReDoS:
|
||||
**Feature release**, plus one SSRF hardening fix and a large batch of correctness fixes across the RDF/ontology export pipeline:
|
||||
|
||||
- **Missing authentication on all Explorer API routes** (GHSA-j4mq-hprp-987v, Critical): every route now requires `SEMANTICA_API_KEY`, fails closed (503) rather than open when unconfigured
|
||||
- **SSRF via redirect bypass in ontology URL fetching** (GHSA-8c7v-62gr-hj6g, High): redirect targets are now re-validated at every hop and the connection is pinned to the validated address, closing a DNS check-then-use race
|
||||
- **Cypher injection via unvalidated node labels and property keys** (GHSA-482h-hw99-h62p, Critical): Neptune, Neo4j, and FalkorDB now sanitize every label/relationship-type/property-key interpolation site
|
||||
- **SPARQL injection via unvalidated triplet IRIs** (GHSA-8vgg-8mr4-r236, Critical): Blazegraph, RDF4J, and Jena now validate subject/predicate/object IRIs before interpolation
|
||||
- **Missing Origin validation on the WebSocket handshake** (GHSA-4643-wpgq-w329, Moderate, anonymous-mode only): `/ws/graph-updates` now checks `Origin` against the same allowlist `CORSMiddleware` enforces for HTTP
|
||||
- **Polynomial ReDoS in SPARQL query validation** (CodeQL `py/polynomial-redos`): fixed a backtracking regex in the Explorer's SPARQL route
|
||||
- **First-class LangChain integration** (`semantica[langchain]`): a `BaseRetriever` and `VectorStore` over `HybridSearch`, plus graph/decision-query tools
|
||||
- **SAP OData ingestor** (`semantica[ingest-sap]`): OAuth2/Basic-auth, SSRF-guarded ingestion for Business Partners and Sales Orders, following the existing Snowflake/Databricks connector pattern
|
||||
- **`ContextGraph` gains deterministic, human-editable Markdown round-trip persistence** alongside the existing JSON API, and the Explorer graph inspector gains a read-only Markdown content viewer
|
||||
- **`reasoning` gains a structured Action layer**: rule-driven `Assert`/`Retract`/`Call`/`EmitEvent` actions with optional provenance, turning the reasoner into a production-rule system
|
||||
- **`run_shacl_validation` is now a public, documented API**, and a dozen ontology/RDF export correctness fixes land: OWL property/class export, SHACL target-namespace resolution, one canonical confidence datatype across all four RDF formats, reachable OWL-Time reification, JSON-LD default-graph and content-derived document identity, and full metadata passthrough on every RDF serializer
|
||||
- **Security**: Agno's `AgnoKnowledgeGraph.load_urls()` and OpenClaw's MCP tool now route outbound requests through the shared SSRF guard
|
||||
|
||||
Also includes: embedded Oxigraph backend for `TripletStore`, PROV-O trust/spec completeness for `ProvenanceManager`, and the Altair Anzo triplet store backend.
|
||||
Also fixes: `PipelineBuilder.set_parallelism()` now actually parallelizes independent pipeline steps, `flatten_dict()` no longer silently drops data on a key collision, `Config.get()` honors boolean environment overrides, and the MCP server's `export_graph` tool works again on every format.
|
||||
|
||||
→ [Full release notes](RELEASE_NOTES.md) · [Changelog](CHANGELOG.md)
|
||||
|
||||
@@ -1498,6 +1492,8 @@ Semantica is designed for environments where AI outputs must be explainable, aud
|
||||
- **Cybersecurity:** Threat attribution, incident response timelines, and IOC provenance tracking
|
||||
- **Autonomous Systems:** Decision logs, safety validation, and explainable AI for certification
|
||||
|
||||
> ⚠️ **This is system-level explainability, not foundation-model explainability.** Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. What Semantica explains is *outside* the model: the context and data fed in, the decision produced, its provenance, the relevant relationships, the policies applied, and the full execution trail. In short, Semantica explains and audits what the AI system did, not the LLM's private internal reasoning.
|
||||
|
||||
---
|
||||
|
||||
## Installation
|
||||
@@ -1510,6 +1506,7 @@ pip install semantica[all] # everything
|
||||
```bash
|
||||
pip install semantica[agno] # Agno multi-agent integration
|
||||
pip install semantica[crewai] # CrewAI integration
|
||||
pip install semantica[langchain] # LangChain / LangGraph integration
|
||||
pip install semantica[llm-litellm] # OpenAI, Anthropic, Gemini, Mistral, Llama, Groq, Cohere, Bedrock, Ollama, DeepSeek, and more
|
||||
pip install semantica[graph-neo4j] # Neo4j graph store (LPG)
|
||||
pip install semantica[graph-falkordb] # FalkorDB graph store (LPG)
|
||||
@@ -1562,11 +1559,11 @@ On-premises deployment · Private cloud · Custom domain implementations · SLA-
|
||||
|
||||
## Star History
|
||||
|
||||
<a href="https://www.star-history.com/?repos=semantica-agi%2Fsemantica&type=date&legend=top-left">
|
||||
<a href="https://star-history.dera.page/#semantica-agi/semantica&type=date&legend=top-left">
|
||||
<picture>
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://api.star-history.com/chart?repos=semantica-agi/semantica&type=date&theme=dark&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://api.star-history.com/chart?repos=semantica-agi/semantica&type=date&legend=top-left" />
|
||||
<img alt="Star History Chart" src="https://api.star-history.com/chart?repos=semantica-agi/semantica&type=date&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: dark)" srcset="https://star-history.dera.page/svg?repos=semantica-agi/semantica&type=date&theme=dark&legend=top-left" />
|
||||
<source media="(prefers-color-scheme: light)" srcset="https://star-history.dera.page/svg?repos=semantica-agi/semantica&type=date&legend=top-left" />
|
||||
<img alt="Star History Chart" src="https://star-history.dera.page/svg?repos=semantica-agi/semantica&type=date&legend=top-left" />
|
||||
</picture>
|
||||
</a>
|
||||
|
||||
@@ -1595,6 +1592,23 @@ See [CONTRIBUTING.md](CONTRIBUTING.md) for full guidelines.
|
||||
|
||||
---
|
||||
|
||||
## Cite Us
|
||||
|
||||
If you use Semantica in your research or production systems, please cite it as:
|
||||
|
||||
```bibtex
|
||||
@software{semantica2026,
|
||||
title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
|
||||
author = {Semantica},
|
||||
year = {2026},
|
||||
url = {https://github.com/semantica-agi/semantica}
|
||||
}
|
||||
```
|
||||
|
||||
All citation formats (APA, MLA, Chicago, IEEE) live on the [Citation](https://docs.getsemantica.ai/citation) page — every format attributes authorship to **Semantica**, not individual contributors.
|
||||
|
||||
---
|
||||
|
||||
<div align="center">
|
||||
|
||||
MIT License · Built by [Semantica](https://github.com/semantica-agi)
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/01_Advanced_Extraction.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/01_Advanced_Extraction.ipynb)\n",
|
||||
"\n",
|
||||
"# Advanced Extraction\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/03_Complete_Visualization_Suite.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/03_Complete_Visualization_Suite.ipynb)\n",
|
||||
"\n",
|
||||
"# Complete Visualization Suite\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/05_Multi_Format_Export.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/05_Multi_Format_Export.ipynb)\n",
|
||||
"\n",
|
||||
"# Advanced Multi-Format Export\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/08_Reasoning_and_Inference.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/08_Reasoning_and_Inference.ipynb)\n",
|
||||
"\n",
|
||||
"# Reasoning and Inference\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/09_Semantic_Layer_Construction.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/09_Semantic_Layer_Construction.ipynb)\n",
|
||||
"\n",
|
||||
"# Semantic Layer Construction\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb)\n",
|
||||
"\n",
|
||||
"# Deep Dive: Temporal Knowledge Graphs\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/12_Unstructured_to_Ontology.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/12_Unstructured_to_Ontology.ipynb)\n",
|
||||
"\n",
|
||||
"# Unstructured Text to Ontology\n",
|
||||
"\n",
|
||||
|
||||
@@ -18,7 +18,7 @@
|
||||
"id": "cell-0",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/13_Manual_Ontology_Snowflake_Mapping.ipynb)\n",
|
||||
"\n",
|
||||
"# Manual Ontology + Snowflake Mapping\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/14_Datalog_Style_Reasoning.ipynb)\n",
|
||||
"\n",
|
||||
"# Datalog-Style Reasoning\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb)\n",
|
||||
"\n",
|
||||
"# Advanced Vector Store - Made Easy\n",
|
||||
"\n",
|
||||
@@ -352,7 +352,7 @@
|
||||
"- Build a multi-user application\n",
|
||||
"- Explore the [introduction notebook](../introduction/13_Vector_Store.ipynb) for more basics\n",
|
||||
"\n",
|
||||
"**Need Help?** Check our [documentation](https://semantica.readthedocs.io) or ask on [GitHub](https://github.com/Hawksight-AI/semantica)."
|
||||
"**Need Help?** Check our [documentation](https://semantica.readthedocs.io) or ask on [GitHub](https://github.com/semantica-agi/semantica)."
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb)\n",
|
||||
"\n",
|
||||
"Semantica is a **semantic intelligence and knowledge engineering framework**. It helps you:\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/02_Data_Ingestion.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/02_Data_Ingestion.ipynb)\n",
|
||||
"\n",
|
||||
"# Data Ingestion - Comprehensive Guide\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/04_Document_Parsing.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/03_Document_Parsing.ipynb)\n",
|
||||
"\n",
|
||||
"# Document Parsing\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/05_Data_Normalization.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/04_Data_Normalization.ipynb)\n",
|
||||
"\n",
|
||||
"# Data Normalization\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/05_Entity_Extraction.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/05_Entity_Extraction.ipynb)\n",
|
||||
"\n",
|
||||
"# Entity Extraction - Comprehensive Guide\n",
|
||||
"\n",
|
||||
@@ -622,7 +622,7 @@
|
||||
"\n",
|
||||
"---\n",
|
||||
"\n",
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/06_Relation_Extraction.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/06_Relation_Extraction.ipynb)\n",
|
||||
"\n",
|
||||
"# Relation Extraction - Comprehensive Guide\n",
|
||||
"\n",
|
||||
@@ -599,7 +599,7 @@
|
||||
"\n",
|
||||
"---\n",
|
||||
"\n",
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/08_Building_Knowledge_Graphs.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb)\n",
|
||||
"\n",
|
||||
"# Building Knowledge Graphs\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/09_Your_First_Knowledge_Graph.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb)\n",
|
||||
"\n",
|
||||
"# 🚀 Your First Knowledge Graph\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/11_Graph_Analytics.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/10_Graph_Analytics.ipynb)\n",
|
||||
"\n",
|
||||
"# Graph Analytics\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb)\n",
|
||||
"\n",
|
||||
"# Chunking and Splitting - Comprehensive Guide\n",
|
||||
"\n",
|
||||
@@ -817,7 +817,7 @@
|
||||
"\n",
|
||||
"---\n",
|
||||
"\n",
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/13_Embedding_Generation.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/12_Embedding_Generation.ipynb)\n",
|
||||
"\n",
|
||||
"# Embedding Generation\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb)\n",
|
||||
"\n",
|
||||
"# Vector Store - Comprehensive Guide\n",
|
||||
"\n",
|
||||
@@ -492,7 +492,7 @@
|
||||
"\n",
|
||||
"---\n",
|
||||
"\n",
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/Hawksight-AI/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
"**Questions or Issues?** Check out our [GitHub repository](https://github.com/semantica-agi/semantica) or [documentation](https://semantica.readthedocs.io)."
|
||||
]
|
||||
}
|
||||
],
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/14_Ontology.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/14_Ontology.ipynb)\n",
|
||||
"\n",
|
||||
"# Ontology Generation \n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/15_Export.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/15_Export.ipynb)\n",
|
||||
"\n",
|
||||
"# Export Module - Comprehensive Guide\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/17_Visualization.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/16_Visualization.ipynb)\n",
|
||||
"\n",
|
||||
"# Visualization\n",
|
||||
"\n",
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/18_Deduplication.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/18_Deduplication.ipynb)\n",
|
||||
"\n",
|
||||
"# Deduplication in Semantica\n",
|
||||
"\n",
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
"id": "c21e9c8d",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/introduction/19_Context_Module.ipynb)\n",
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/19_Context_Module.ipynb)\n",
|
||||
"\n",
|
||||
"# Context Module — Practical Guide\n",
|
||||
"\n",
|
||||
|
||||
@@ -0,0 +1,253 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"# Provenance Tracking (W3C PROV-O)\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"In high-stakes domains — healthcare, legal, finance, research — a Knowledge Graph is only as trustworthy as its ability to answer **\"where did this fact come from?\"**. Semantica's `provenance` module provides audit-grade, W3C PROV-O-aligned tracking for every entity, relationship and chunk that flows through your pipeline.\n",
|
||||
"\n",
|
||||
"In this cookbook you will learn how to:\n",
|
||||
"\n",
|
||||
"- Track entities and relationships with **source details** (DOI, page, verbatim quote, confidence)\n",
|
||||
"- Walk the full **lineage** of a fact (document → chunk → entity → KG)\n",
|
||||
"- Audit **revision history** and **all sources** behind an entity\n",
|
||||
"- **Invalidate** a fact without deleting it (prov:Invalidation) — corrections stay provable\n",
|
||||
"- Verify **tamper-evidence** with chained SHA-256 checksums\n",
|
||||
"\n",
|
||||
"**The Scenario:** a research team ingests findings from two scientific papers (with DOIs) into a Knowledge Graph. A regulator later asks: *\"Which paper, which figure, and which exact sentence supports the claim that fish biomass increased by 463%? And was that fact ever corrected?\"*"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip install -q semantica"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import json\n",
|
||||
"from semantica.provenance import (\n",
|
||||
" ProvenanceManager,\n",
|
||||
" compute_checksum,\n",
|
||||
" verify_checksum,\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# In-memory storage for this demo; pass storage_path=\"provenance.db\"\n",
|
||||
"# (or a config with provenance.storage_path) for a persistent SQLite backend.\n",
|
||||
"prov = ProvenanceManager()\n",
|
||||
"print(\"ProvenanceManager ready (in-memory storage)\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 1: Track Entities with Audit-Grade Source Details\n",
|
||||
"\n",
|
||||
"Every fact we ingest carries its evidence with it: the **source identifier** (a DOI here), the **location** inside the source (a figure), the **verbatim quote**, and the extractor's **confidence**."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# Finding from paper #1\n",
|
||||
"entry_biomass = prov.track_entity(\n",
|
||||
" entity_id=\"claim_biomass_increase\",\n",
|
||||
" source=\"DOI:10.1371/journal.pone.0023601\",\n",
|
||||
" confidence=0.92,\n",
|
||||
" source_location=\"Figure 2\",\n",
|
||||
" source_quote=\"Total fish biomass increased by 463% ...\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"# Supporting entity from paper #2\n",
|
||||
"entry_reserve = prov.track_entity(\n",
|
||||
" entity_id=\"marine_reserve_1\",\n",
|
||||
" source=\"DOI:10.1126/science.1088121\",\n",
|
||||
" confidence=0.88,\n",
|
||||
" source_location=\"Table 1\",\n",
|
||||
" source_quote=\"... no-take marine reserve at Cabo Pulmo ...\",\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"Tracked:\", entry_biomass.entity_id, \"|\", entry_reserve.entity_id)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 2: Track the Relationship Between Facts\n",
|
||||
"\n",
|
||||
"Facts rarely stand alone. The claim about biomass increase is *about* the marine reserve — that relationship is a first-class provenance-tracked object too.\n",
|
||||
"\n",
|
||||
"`track_relationship()` has no dedicated subject/object fields, so by convention we record which two entities it connects inside `metadata`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"rel = prov.track_relationship(\n",
|
||||
" relationship_id=\"rel_biomass_about_reserve\",\n",
|
||||
" source=\"DOI:10.1371/journal.pone.0023601\",\n",
|
||||
" metadata={\n",
|
||||
" \"type\": \"measured_at\",\n",
|
||||
" # No dedicated endpoint fields on track_relationship() yet -- record\n",
|
||||
" # which entities this relationship connects here by convention.\n",
|
||||
" \"subject_entity_id\": \"claim_biomass_increase\",\n",
|
||||
" \"object_entity_id\": \"marine_reserve_1\",\n",
|
||||
" },\n",
|
||||
")\n",
|
||||
"\n",
|
||||
"print(\"Relationship tracked:\", rel.entity_id, \"|\", rel.metadata[\"subject_entity_id\"], \"->\", rel.metadata[\"object_entity_id\"])"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 3: Walk the Lineage\n",
|
||||
"\n",
|
||||
"`get_lineage` reconstructs everything known about a fact; `trace_lineage` returns the ordered chain of `ProvenanceEntry` records — every version, every activity, every agent that touched it."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"lineage = prov.get_lineage(\"claim_biomass_increase\")\n",
|
||||
"print(json.dumps(lineage, indent=2, default=str)[:800])\n",
|
||||
"\n",
|
||||
"print(\"\\n--- ordered chain ---\")\n",
|
||||
"for e in prov.trace_lineage(\"claim_biomass_increase\"):\n",
|
||||
" print(f\"{e.entity_id} | seq#{e.sequence_id} | {e.activity_id}\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 4: Audit Sources and Revision History\n",
|
||||
"\n",
|
||||
"When the regulator asks *\"has this fact ever been corrected?\"*, `revision_history` answers with the full version chain, and `get_all_sources` lists every source document that ever supported the entity."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"revisions = prov.revision_history(\"claim_biomass_increase\")\n",
|
||||
"print(f\"{len(revisions)} revision(s) on record\")\n",
|
||||
"\n",
|
||||
"for s in prov.get_all_sources(\"claim_biomass_increase\"):\n",
|
||||
" print(\"source:\", s)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 5: Invalidate — Correct Without Deleting\n",
|
||||
"\n",
|
||||
"Suppose paper #1 is retracted in part. An audit trail must **not** silently delete the fact: `invalidate` archives the pre-invalidation state and appends a fresh `prov:Invalidation` entry naming **who** retracted it and **why**."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"invalidated = prov.invalidate(\n",
|
||||
" entity_id=\"claim_biomass_increase\",\n",
|
||||
" agent_id=\"reviewer_dr_chen\",\n",
|
||||
" reason=\"Partial retraction: Figure 2 statistics corrected by publisher (see erratum).\",\n",
|
||||
")\n",
|
||||
"print(\"Invalidated:\", invalidated.entity_id, \"| invalidated flag:\", getattr(invalidated, \"invalidated\", True))\n",
|
||||
"\n",
|
||||
"stats = prov.get_statistics()\n",
|
||||
"print(\"\\nStorage statistics:\", json.dumps(stats, indent=2, default=str))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Step 6: Verify Tamper-Evidence\n",
|
||||
"\n",
|
||||
"Each entry carries a deterministic SHA-256 checksum chained to the previous entry. Recompute and compare to detect any after-the-fact corruption of the provenance record."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"# entry_biomass was returned by track_entity in Step 1\n",
|
||||
"ok = verify_checksum(entry_biomass)\n",
|
||||
"print(\"Checksum verified:\", ok)\n",
|
||||
"\n",
|
||||
"print(\"Computed:\", compute_checksum(entry_biomass)[:16], \"...\")\n",
|
||||
"print(\"Stored: \", entry_biomass.checksum[:16] if getattr(entry_biomass, 'checksum', None) else \"(see entry fields)\")\n",
|
||||
"chain = prov.verify_chain()\n",
|
||||
"print(\"Chain verification:\", json.dumps(chain, default=str)[:200])\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Summary\n",
|
||||
"\n",
|
||||
"| Need | Call |\n",
|
||||
"|---|---|\n",
|
||||
"| Record a fact's evidence | `prov.track_entity(entity_id, source, confidence=..., source_location=..., source_quote=...)` |\n",
|
||||
"| Record a relationship | `prov.track_relationship(relationship_id, source, metadata=...)` |\n",
|
||||
"| Full lineage of a fact | `prov.get_lineage(entity_id)` / `prov.trace_lineage(entity_id)` |\n",
|
||||
"| \"Was it ever corrected?\" | `prov.revision_history(entity_id)` |\n",
|
||||
"| \"Which sources support it?\" | `prov.get_all_sources(entity_id)` |\n",
|
||||
"| Retract without deleting | `prov.invalidate(entity_id, agent_id, reason=...)` |\n",
|
||||
"| Tamper check | `verify_checksum(entry)` |\n",
|
||||
"\n",
|
||||
"### Where to go next\n",
|
||||
"\n",
|
||||
"- **Conflict Detection and Resolution** (notebook 17) — what happens when two sources disagree.\n",
|
||||
"- **Your First Knowledge Graph** (notebook 08) — plug `provenance=True` into extractors so tracking happens automatically during ingestion.\n",
|
||||
"- The module docstring (`help(semantica.provenance)`) documents opt-in integration with `kg`, `split` and `conflicts` trackers."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"name": "python",
|
||||
"version": "3.11"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
@@ -0,0 +1,383 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "b76a5997",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/23_Reasoning.ipynb)\n",
|
||||
"\n",
|
||||
"# Reasoning Module — Practical Guide\n",
|
||||
"\n",
|
||||
"Semantica's `reasoning` module derives new knowledge from existing facts and knowledge graphs. It ships several strategies behind one facade:\n",
|
||||
"\n",
|
||||
"- **`Reasoner`** — unified facade with forward chaining, backward chaining, and one-shot `infer_facts`\n",
|
||||
"- **`DatalogReasoner`** — semi-naive Datalog fixpoint evaluation with variable queries\n",
|
||||
"- **`ExplanationGenerator`** — human-readable explanations and reasoning paths for inferred conclusions\n",
|
||||
"- Plus lower-level engines: `ReteEngine`, `SPARQLReasoner`, `GraphReasoner`, temporal reasoning\n",
|
||||
"\n",
|
||||
"This notebook walks through the facade, the Datalog engine, and explanations. All APIs are verified against `semantica/reasoning/`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 1,
|
||||
"id": "52073af7",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:45:55.427457Z",
|
||||
"iopub.status.busy": "2026-08-26T18:45:55.427247Z",
|
||||
"iopub.status.idle": "2026-08-26T18:45:57.266607Z",
|
||||
"shell.execute_reply": "2026-08-26T18:45:57.264783Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip install -q semantica"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "06deb916",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 1) Forward chaining with the `Reasoner` facade\n",
|
||||
"\n",
|
||||
"Facts are simple `Predicate(args)` strings. Rules use `IF <conditions> THEN <conclusion>` with `?x`-style variables. `forward_chain()` derives everything possible and returns a list of `InferenceResult` objects."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 2,
|
||||
"id": "519ca92d",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:45:57.270791Z",
|
||||
"iopub.status.busy": "2026-08-26T18:45:57.270352Z",
|
||||
"iopub.status.idle": "2026-08-26T18:45:59.991941Z",
|
||||
"shell.execute_reply": "2026-08-26T18:45:59.990678Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/html": [
|
||||
"<div style='font-family: monospace;'><h4>🧠 Semantica - 📊 Current Progress</h4><table style='width: 100%; border-collapse: collapse;'><tr><th>Status</th><th>Action</th><th>Module</th><th>Submodule</th><th>Progress</th><th>ETA</th><th>Rate</th><th>Time</th><th>Extracted</th></tr><tr><td>✅</td><td>Semantica is reasoning</td><td>🤔 reasoning</td><td>Reasoner</td><td>100.0%</td><td>-</td><td>-</td><td>0.00s</td><td>-</td></tr><tr><td>✅</td><td>Semantica is reasoning</td><td>🤔 reasoning</td><td>DatalogReasoner</td><td>100.0%</td><td>-</td><td>-</td><td>0.00s</td><td>-</td></tr><tr><td>✅</td><td>Semantica is reasoning</td><td>🤔 reasoning</td><td>ExplanationGenerator</td><td>100.0%</td><td>-</td><td>-</td><td>0.00s</td><td>-</td></tr></table></div>"
|
||||
],
|
||||
"text/plain": [
|
||||
"<IPython.core.display.HTML object>"
|
||||
]
|
||||
},
|
||||
"metadata": {},
|
||||
"output_type": "display_data"
|
||||
},
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"🔄 Semantica is reasoning: Performing forward chaining 🤔 reasoning Reasoner |░░░░░░░░░░░░░░░| 0.0% ETA: - Rate: - Time: 0.00s Extracted: -"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"Inferred 2 new facts\n",
|
||||
" Human(Jane) (rule: Rule 1, confidence: 1.0)\n",
|
||||
" Human(John) (rule: Rule 1, confidence: 1.0)\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import Reasoner\n",
|
||||
"\n",
|
||||
"reasoner = Reasoner()\n",
|
||||
"\n",
|
||||
"reasoner.add_fact(\"Person(John)\")\n",
|
||||
"reasoner.add_fact(\"Person(Jane)\")\n",
|
||||
"reasoner.add_rule(\"IF Person(?x) THEN Human(?x)\")\n",
|
||||
"\n",
|
||||
"results = reasoner.forward_chain()\n",
|
||||
"print(f\"Inferred {len(results)} new facts\")\n",
|
||||
"for res in results:\n",
|
||||
" print(f\" {res.conclusion} (rule: {res.rule_used.name}, confidence: {res.confidence})\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "c1131c45",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 2) One-shot inference with `infer_facts`\n",
|
||||
"\n",
|
||||
"`infer_facts(facts, rules)` **adds** the given facts and rules to this `Reasoner` instance, runs forward chaining to fixpoint, and returns the derived facts as strings. It does not reset the instance's existing state — create a fresh `Reasoner()` first if you need isolation between runs."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 3,
|
||||
"id": "26249990",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:45:59.995447Z",
|
||||
"iopub.status.busy": "2026-08-26T18:45:59.995069Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:00.004107Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:00.002873Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"['Employee(Jane, Acme)', 'Employee(John, Acme)']"
|
||||
]
|
||||
},
|
||||
"execution_count": 3,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import Reasoner\n",
|
||||
"\n",
|
||||
"derived = Reasoner().infer_facts(\n",
|
||||
" facts=[\"WorksFor(John, Acme)\", \"WorksFor(Jane, Acme)\"],\n",
|
||||
" rules=[\"IF WorksFor(?x, ?y) THEN Employee(?x, ?y)\"],\n",
|
||||
")\n",
|
||||
"derived"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "d5504a38",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 3) Backward chaining: proving a goal\n",
|
||||
"\n",
|
||||
"`backward_chain(goal)` works backwards from a conclusion through the rules. It returns the `InferenceResult` that proves the goal, or `None`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 4,
|
||||
"id": "c4ef85dd",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:00.007740Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:00.007346Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:00.015561Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:00.014145Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"Human(John)\n",
|
||||
"premises: ['Person(John)']\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import Reasoner\n",
|
||||
"\n",
|
||||
"reasoner = Reasoner()\n",
|
||||
"reasoner.add_fact(\"Person(John)\")\n",
|
||||
"reasoner.add_rule(\"IF Person(?x) THEN Human(?x)\")\n",
|
||||
"\n",
|
||||
"proof = reasoner.backward_chain(\"Human(John)\")\n",
|
||||
"print(proof.conclusion if proof else \"not provable\")\n",
|
||||
"print(\"premises:\", proof.premises if proof else None)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "b245581d",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 4) Re-run safety\n",
|
||||
"\n",
|
||||
"`add_rule` deduplicates rules with identical conditions and conclusion, so re-executing a setup cell (the common Jupyter re-run) does not duplicate rules — see issue #732."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 5,
|
||||
"id": "fb2aeb39",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:00.019091Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:00.018881Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:00.024042Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:00.022836Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stderr",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"Skipping duplicate rule (same conditions/conclusion as 'rule_1'): IF Person(?x) THEN Human(?x)\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"1"
|
||||
]
|
||||
},
|
||||
"execution_count": 5,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import Reasoner\n",
|
||||
"\n",
|
||||
"reasoner = Reasoner()\n",
|
||||
"reasoner.add_fact(\"Person(John)\")\n",
|
||||
"\n",
|
||||
"# Simulate a Jupyter cell re-run: add the same rule twice\n",
|
||||
"r1 = reasoner.add_rule(\"IF Person(?x) THEN Human(?x)\")\n",
|
||||
"r2 = reasoner.add_rule(\"IF Person(?x) THEN Human(?x)\")\n",
|
||||
"\n",
|
||||
"len(reasoner.rules)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "ba2e5c4a",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 5) Datalog reasoning\n",
|
||||
"\n",
|
||||
"`DatalogReasoner` uses classic Datalog syntax (`head :- body.`) and semi-naive fixpoint evaluation. Queries return variable bindings as a list of dicts — use uppercase variables to ask *which* facts hold."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 6,
|
||||
"id": "9ec5c0c4",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:00.026769Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:00.026588Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:00.034963Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:00.032672Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"[{'X': 'tom', 'Z': 'ann'}]"
|
||||
]
|
||||
},
|
||||
"execution_count": 6,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import DatalogReasoner\n",
|
||||
"\n",
|
||||
"datalog = DatalogReasoner()\n",
|
||||
"datalog.add_fact(\"parent(tom, mary)\")\n",
|
||||
"datalog.add_fact(\"parent(mary, ann)\")\n",
|
||||
"datalog.add_rule(\"grandparent(X, Z) :- parent(X, Y), parent(Y, Z)\")\n",
|
||||
"\n",
|
||||
"datalog.derive_all()\n",
|
||||
"datalog.query(\"grandparent(X, Z)\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "d4f0689b",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 6) Explanations for inferred conclusions\n",
|
||||
"\n",
|
||||
"`ExplanationGenerator` turns `InferenceResult` objects into structured `Explanation` and `ReasoningPath` records, so agents can show *why* they believe a derived fact."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 7,
|
||||
"id": "19dcd3a7",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:00.038649Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:00.038396Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:00.059188Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:00.057805Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"('Explanation', 'ReasoningPath')"
|
||||
]
|
||||
},
|
||||
"execution_count": 7,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.reasoning import Reasoner, ExplanationGenerator\n",
|
||||
"\n",
|
||||
"reasoner = Reasoner()\n",
|
||||
"reasoner.add_fact(\"Person(John)\")\n",
|
||||
"reasoner.add_rule(\"IF Person(?x) THEN Human(?x)\")\n",
|
||||
"results = reasoner.forward_chain()\n",
|
||||
"\n",
|
||||
"gen = ExplanationGenerator()\n",
|
||||
"explanation = gen.generate_explanation(results[0])\n",
|
||||
"path = gen.show_reasoning_path(results[0])\n",
|
||||
"\n",
|
||||
"type(explanation).__name__, type(path).__name__"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "fb882ee4",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Summary\n",
|
||||
"\n",
|
||||
"| Task | API |\n",
|
||||
"|---|---|\n",
|
||||
"| Derive all new facts | `Reasoner.forward_chain()` |\n",
|
||||
"| One-shot inference | `Reasoner.infer_facts(facts, rules)` |\n",
|
||||
"| Prove a goal | `Reasoner.backward_chain(goal)` |\n",
|
||||
"| Datalog fixpoint | `DatalogReasoner.derive_all()` + `query(\"p(X, Y)\")` |\n",
|
||||
"| Explain a conclusion | `ExplanationGenerator.generate_explanation(result)` |\n",
|
||||
"\n",
|
||||
"See also `semantica/reasoning/reasoning_usage.md` and the module docstrings for `ReteEngine`, `SPARQLReasoner`, and temporal reasoning."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"codemirror_mode": {
|
||||
"name": "ipython",
|
||||
"version": 3
|
||||
},
|
||||
"file_extension": ".py",
|
||||
"mimetype": "text/x-python",
|
||||
"name": "python",
|
||||
"nbconvert_exporter": "python",
|
||||
"pygments_lexer": "ipython3",
|
||||
"version": "3.13.12"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
@@ -0,0 +1,299 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "8d7096ea",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/24_Change_Management.ipynb)\n",
|
||||
"\n",
|
||||
"# Change Management — Practical Guide\n",
|
||||
"\n",
|
||||
"Semantica's `change_management` module provides versioning, audit trails, and data-integrity checks for knowledge graphs and ontologies:\n",
|
||||
"\n",
|
||||
"- **`ChangeLogEntry`** — standardized change metadata (validated timestamp/author)\n",
|
||||
"- **`InMemoryVersionStorage` / `SQLiteVersionStorage`** — version snapshot storage with named tags\n",
|
||||
"- **`compute_checksum` / `verify_checksum`** — SHA-256 integrity verification\n",
|
||||
"\n",
|
||||
"This notebook runs a complete save → tag → verify → tamper-detect cycle. All outputs are real executed results verified against the repository's `semantica/change_management/` source at the time of writing (the `pip install` cell may fetch a newer release with slightly different behavior)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 1,
|
||||
"id": "7bdffec1",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:37.171333Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:37.171183Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.060860Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.059594Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip install -q semantica"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "169efee1",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 1) A `ChangeLogEntry` records *who* changed *what*, *when*\n",
|
||||
"\n",
|
||||
"`author` must be a valid email — the dataclass validates on construction (`ValidationError` otherwise), which keeps audit trails clean."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 2,
|
||||
"id": "5b17acdb",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:39.064077Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:39.063818Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.321881Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.321036Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"ChangeLogEntry(timestamp='2026-08-15T09:00:00Z', author='demo@example.com', description='initial version', change_id=None, related_changes=[])"
|
||||
]
|
||||
},
|
||||
"execution_count": 2,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.change_management import ChangeLogEntry\n",
|
||||
"\n",
|
||||
"entry = ChangeLogEntry(\n",
|
||||
" timestamp=\"2026-08-15T09:00:00Z\",\n",
|
||||
" author=\"demo@example.com\",\n",
|
||||
" description=\"initial version\",\n",
|
||||
")\n",
|
||||
"entry"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "53d8df5c",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 2) Save a versioned snapshot\n",
|
||||
"\n",
|
||||
"A snapshot is a dict with a required `label` plus your payload. Here we attach the KG data, the change log, and a SHA-256 `checksum` computed over everything except the checksum field itself."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 3,
|
||||
"id": "fec16f24",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:39.325528Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:39.325140Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.331480Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.330586Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"True"
|
||||
]
|
||||
},
|
||||
"execution_count": 3,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.change_management import InMemoryVersionStorage, compute_checksum\n",
|
||||
"\n",
|
||||
"storage = InMemoryVersionStorage()\n",
|
||||
"\n",
|
||||
"snapshot = {\n",
|
||||
" \"label\": \"v1.0.0\",\n",
|
||||
" \"data\": {\"entities\": {\"acme\": {\"type\": \"Company\"}}},\n",
|
||||
" \"change_log\": {\n",
|
||||
" \"timestamp\": entry.timestamp,\n",
|
||||
" \"author\": entry.author,\n",
|
||||
" \"description\": entry.description,\n",
|
||||
" },\n",
|
||||
"}\n",
|
||||
"snapshot[\"checksum\"] = compute_checksum({k: v for k, v in snapshot.items() if k != \"checksum\"})\n",
|
||||
"\n",
|
||||
"storage.save(snapshot)\n",
|
||||
"storage.exists(\"v1.0.0\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "0f1c603b",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 3) Named tags pin a version for releases\n",
|
||||
"\n",
|
||||
"`save_tag` / `get_tag` map stable names (e.g. `release`) to version labels, decoupling consumers from label churn."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 4,
|
||||
"id": "62f7643e",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:39.335182Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:39.334886Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.339586Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.338568Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"('v1.0.0', ['v1.0.0'])"
|
||||
]
|
||||
},
|
||||
"execution_count": 4,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"storage.save_tag(\"release\", \"v1.0.0\")\n",
|
||||
"\n",
|
||||
"storage.get_tag(\"release\"), [s[\"label\"] for s in storage.list_all()]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "96df12da",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 4) Verify integrity — and catch tampering\n",
|
||||
"\n",
|
||||
"`verify_checksum(snapshot)` recomputes the SHA-256 over the snapshot (minus its `checksum` field) and compares. A single mutated character in the data flips the result to `False`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 5,
|
||||
"id": "26d0de85",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:39.342653Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:39.342466Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.346714Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.345623Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"intact: True\n",
|
||||
"tampered: False\n"
|
||||
]
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"from semantica.change_management import verify_checksum\n",
|
||||
"\n",
|
||||
"stored = storage.get(\"v1.0.0\")\n",
|
||||
"print(\"intact:\", verify_checksum(stored))\n",
|
||||
"\n",
|
||||
"tampered = storage.get(\"v1.0.0\")\n",
|
||||
"tampered[\"data\"][\"entities\"][\"acme\"][\"note\"] = \"mutated after the fact\"\n",
|
||||
"print(\"tampered:\", verify_checksum(tampered))"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "bd14c3e4",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 5) Retiring a version\n",
|
||||
"\n",
|
||||
"`delete(label)` removes a snapshot; tags pointing at it are your responsibility to update."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 6,
|
||||
"id": "de9fe3e5",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:46:39.349814Z",
|
||||
"iopub.status.busy": "2026-08-26T18:46:39.349513Z",
|
||||
"iopub.status.idle": "2026-08-26T18:46:39.354710Z",
|
||||
"shell.execute_reply": "2026-08-26T18:46:39.353669Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"False"
|
||||
]
|
||||
},
|
||||
"execution_count": 6,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"storage.delete(\"v1.0.0\")\n",
|
||||
"storage.exists(\"v1.0.0\")"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "ab667b32",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Summary\n",
|
||||
"\n",
|
||||
"| Task | API |\n",
|
||||
"|---|---|\n",
|
||||
"| Record audit metadata | `ChangeLogEntry(timestamp, author=email, description)` |\n",
|
||||
"| Persist a version | `InMemoryVersionStorage().save({\"label\": ..., ...})` |\n",
|
||||
"| Pin a release name | `save_tag(\"release\", \"v1.0.0\")` / `get_tag(\"release\")` |\n",
|
||||
"| Integrity check | `compute_checksum(snap)` / `verify_checksum(snap)` |\n",
|
||||
"| Persistent backend | `SQLiteVersionStorage(path)` — same interface |\n",
|
||||
"\n",
|
||||
"See also `semantica/change_management/change_management_usage.md` for the manager classes (`TemporalVersionManager`, `OntologyVersionManager`)."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"codemirror_mode": {
|
||||
"name": "ipython",
|
||||
"version": 3
|
||||
},
|
||||
"file_extension": ".py",
|
||||
"mimetype": "text/x-python",
|
||||
"name": "python",
|
||||
"nbconvert_exporter": "python",
|
||||
"pygments_lexer": "ipython3",
|
||||
"version": "3.13.12"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
@@ -0,0 +1,314 @@
|
||||
{
|
||||
"cells": [
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "6eb4dfba",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"[](https://colab.research.google.com/github/semantica-agi/semantica/blob/main/cookbook/introduction/25_Seed_Data.ipynb)\n",
|
||||
"\n",
|
||||
"# Seed Data — Practical Guide\n",
|
||||
"\n",
|
||||
"The `seed` module bootstraps a knowledge graph from **trusted, pre-known data** (CSV/JSON/database/API sources) before any extraction runs. This gives extraction a foundation to link against instead of starting from an empty graph.\n",
|
||||
"\n",
|
||||
"Key pieces:\n",
|
||||
"\n",
|
||||
"- **`SeedDataManager`** — registers data sources and builds foundation graphs\n",
|
||||
"- **`create_foundation_graph()`** — turns registered sources into `entities` + `relationships` + `metadata`\n",
|
||||
"- **`validate_quality()`** — checks a foundation graph before you commit it\n",
|
||||
"\n",
|
||||
"All examples below were executed against `semantica/seed/seed_manager.py`."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 1,
|
||||
"id": "32f80cc6",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:18.716466Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:18.716264Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.533828Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.531402Z"
|
||||
}
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"!pip install -q semantica"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "75136e5f",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 1) Prepare a seed CSV and register the source\n",
|
||||
"\n",
|
||||
"`register_source(name, format, location, entity_type=...)` records where trusted data lives. `verified=True` (the default) marks the source as pre-validated."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 2,
|
||||
"id": "a8089e1f",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:20.538772Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:20.538323Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.675403Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.674060Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"True"
|
||||
]
|
||||
},
|
||||
"execution_count": 2,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"import csv\n",
|
||||
"import tempfile\n",
|
||||
"from pathlib import Path\n",
|
||||
"from semantica.seed import SeedDataManager\n",
|
||||
"\n",
|
||||
"# Write the sample CSV into a session-scoped temp directory so we never\n",
|
||||
"# clobber a companies.csv that might exist in the user's working directory.\n",
|
||||
"seed_csv = Path(tempfile.mkdtemp(prefix=\"semantica-seed-\")) / \"companies.csv\"\n",
|
||||
"with open(seed_csv, \"w\", newline=\"\") as f:\n",
|
||||
" writer = csv.DictWriter(f, fieldnames=[\"id\", \"name\", \"type\", \"industry\"])\n",
|
||||
" writer.writeheader()\n",
|
||||
" writer.writerow({\"id\": \"c1\", \"name\": \"Acme\", \"type\": \"Company\", \"industry\": \"robotics\"})\n",
|
||||
" writer.writerow({\"id\": \"c2\", \"name\": \"Globex\", \"type\": \"Company\", \"industry\": \"energy\"})\n",
|
||||
"\n",
|
||||
"manager = SeedDataManager()\n",
|
||||
"manager.register_source(\"companies\", format=\"csv\", location=str(seed_csv), entity_type=\"Company\")\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "e87221ba",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 2) Load records from a registered source\n",
|
||||
"\n",
|
||||
"`load_source(name)` reads the source and enriches each record with `entity_type` and `source` provenance keys."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 3,
|
||||
"id": "f932e550",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:20.679424Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:20.679156Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.690659Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.688812Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/html": [
|
||||
"<div style='font-family: monospace;'><h4>🧠 Semantica - 📊 Current Progress</h4><table style='width: 100%; border-collapse: collapse;'><tr><th>Status</th><th>Action</th><th>Module</th><th>Submodule</th><th>Progress</th><th>ETA</th><th>Rate</th><th>Time</th><th>Extracted</th></tr><tr><td>✅</td><td>Semantica is seeding</td><td>🌱 seed</td><td>SeedDataManager</td><td>100.0%</td><td>-</td><td>-</td><td>0.00s</td><td>-</td></tr></table></div>"
|
||||
],
|
||||
"text/plain": [
|
||||
"<IPython.core.display.HTML object>"
|
||||
]
|
||||
},
|
||||
"metadata": {},
|
||||
"output_type": "display_data"
|
||||
},
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"🔄 Semantica is seeding: Loading seed data from CSV: /var/folders/7s/bvvstgs10y963tz6_4bbnklr0000gn/T/semantica-seed-eu9__ep1/companies.csv 🌱 seed SeedDataManager |░░░░░░░░░░░░░░░| 0.0% ETA: - Rate: - Time: 0.00s Extracted: -"
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "stdout",
|
||||
"output_type": "stream",
|
||||
"text": [
|
||||
"loaded 2 records\n"
|
||||
]
|
||||
},
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"{'id': 'c1',\n",
|
||||
" 'name': 'Acme',\n",
|
||||
" 'type': 'Company',\n",
|
||||
" 'industry': 'robotics',\n",
|
||||
" 'entity_type': 'Company',\n",
|
||||
" 'source': 'companies'}"
|
||||
]
|
||||
},
|
||||
"execution_count": 3,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"records = manager.load_source(\"companies\")\n",
|
||||
"print(f\"loaded {len(records)} records\")\n",
|
||||
"records[0]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "f2ebce64",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 3) Build the foundation graph\n",
|
||||
"\n",
|
||||
"`create_foundation_graph()` converts every registered source into graph-ready entities and relationships. Entities carry `confidence: 1.0` — seed data is trusted by definition."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 4,
|
||||
"id": "09388c31",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:20.695259Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:20.694928Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.708595Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.707072Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"['entities', 'metadata', 'relationships']"
|
||||
]
|
||||
},
|
||||
"execution_count": 4,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"foundation = manager.create_foundation_graph()\n",
|
||||
"sorted(foundation.keys())"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 5,
|
||||
"id": "4610a59f",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:20.713136Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:20.712795Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.718637Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.716835Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"{'id': 'c1',\n",
|
||||
" 'text': 'Acme',\n",
|
||||
" 'type': 'Company',\n",
|
||||
" 'confidence': 1.0,\n",
|
||||
" 'metadata': {'industry': 'robotics', 'source': 'companies'}}"
|
||||
]
|
||||
},
|
||||
"execution_count": 5,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"foundation[\"entities\"][0]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "f3a52dc7",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## 4) Validate quality before committing\n",
|
||||
"\n",
|
||||
"`validate_quality(foundation_graph)` returns `valid`, `errors`, `warnings`, and `metrics` so you can gate bad seed data before it pollutes the graph."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": 6,
|
||||
"id": "4eb7e664",
|
||||
"metadata": {
|
||||
"execution": {
|
||||
"iopub.execute_input": "2026-08-26T18:51:20.722674Z",
|
||||
"iopub.status.busy": "2026-08-26T18:51:20.722118Z",
|
||||
"iopub.status.idle": "2026-08-26T18:51:20.732003Z",
|
||||
"shell.execute_reply": "2026-08-26T18:51:20.730170Z"
|
||||
}
|
||||
},
|
||||
"outputs": [
|
||||
{
|
||||
"data": {
|
||||
"text/plain": [
|
||||
"True"
|
||||
]
|
||||
},
|
||||
"execution_count": 6,
|
||||
"metadata": {},
|
||||
"output_type": "execute_result"
|
||||
}
|
||||
],
|
||||
"source": [
|
||||
"quality = manager.validate_quality(foundation)\n",
|
||||
"quality[\"valid\"]"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"id": "b534be89",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Summary\n",
|
||||
"\n",
|
||||
"| Task | API |\n",
|
||||
"|---|---|\n",
|
||||
"| Register a trusted source | `register_source(name, format, location, entity_type=...)` |\n",
|
||||
"| Load records | `load_source(name)` — adds `entity_type` / `source` keys |\n",
|
||||
"| Direct file load | `load_from_csv(path)` / `load_from_json(path)` |\n",
|
||||
"| Build the graph | `create_foundation_graph()` → `entities` / `relationships` / `metadata` |\n",
|
||||
"| Gate bad data | `validate_quality(graph)` → `valid` / `errors` / `warnings` / `metrics` |\n",
|
||||
"\n",
|
||||
"See also `semantica/seed/seed_usage.md` for `load_from_database`, `load_from_api`, and `integrate_with_extracted`."
|
||||
]
|
||||
}
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "Python 3",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
"language_info": {
|
||||
"codemirror_mode": {
|
||||
"name": "ipython",
|
||||
"version": 3
|
||||
},
|
||||
"file_extension": ".py",
|
||||
"mimetype": "text/x-python",
|
||||
"name": "python",
|
||||
"nbconvert_exporter": "python",
|
||||
"pygments_lexer": "ipython3",
|
||||
"version": "3.13.12"
|
||||
}
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5
|
||||
}
|
||||
+9
-10
@@ -13,26 +13,25 @@ icon: "quote-left"
|
||||
<Tab title="BibTeX">
|
||||
```bibtex
|
||||
@software{semantica2026,
|
||||
title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
|
||||
author = {Semantica},
|
||||
year = {2026},
|
||||
url = {https://github.com/semantica-agi/semantica},
|
||||
version = {0.6.5},
|
||||
doi = {10.5281/zenodo.XXXXXXX}
|
||||
title = {Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems},
|
||||
author = {Semantica},
|
||||
year = {2026},
|
||||
url = {https://github.com/semantica-agi/semantica},
|
||||
doi = {10.5281/zenodo.XXXXXXX}
|
||||
}
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="APA">
|
||||
Semantica. (2026). *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems* (Version 0.6.5) \[Computer software\]. https://github.com/semantica-agi/semantica
|
||||
Semantica. (2026). *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems* \[Computer software\]. https://github.com/semantica-agi/semantica
|
||||
</Tab>
|
||||
<Tab title="MLA">
|
||||
Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. Version 0.6.5, GitHub, 2026, https://github.com/semantica-agi/semantica.
|
||||
Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. GitHub, 2026, https://github.com/semantica-agi/semantica.
|
||||
</Tab>
|
||||
<Tab title="Chicago">
|
||||
Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. Version 0.6.5. GitHub, 2026. https://github.com/semantica-agi/semantica.
|
||||
Semantica. *Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems*. GitHub, 2026. https://github.com/semantica-agi/semantica.
|
||||
</Tab>
|
||||
<Tab title="IEEE">
|
||||
Semantica, "Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems," Version 0.6.5, GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
|
||||
Semantica, "Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems," GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
|
||||
@@ -16,6 +16,9 @@ At its core, Semantica adds a **context and accountability layer** on top of you
|
||||
- **Accountability Layer** — Provenance tracking, decision intelligence, conflict detection, and W3C PROV-O compliance make every claim in your AI stack auditable and explainable.
|
||||
- **Extension Layer** — `PluginRegistry` and `MethodRegistry` let you replace or augment any component: ingestors, extractors, reasoning engines, backends: without changing framework code.
|
||||
|
||||
<Warning>
|
||||
**This is system-level explainability, not foundation-model explainability.** Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. What Semantica explains is *outside* the model: the context and data fed in, the decision produced, its provenance, the relevant relationships, the policies applied, and the full execution trail. In short, Semantica explains and audits *what the AI system did*, not the foundation model's private internal reasoning.
|
||||
</Warning>
|
||||
|
||||
## Knowledge Graphs
|
||||
|
||||
|
||||
+5
-1
@@ -35,6 +35,7 @@ Essential guides to master the Semantica framework.
|
||||
- **[Vector Store](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb)** — Setting up vector stores for similarity search and retrieval. *Intermediate*
|
||||
- **[Graph Store](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/09_Graph_Store.ipynb)** — Persisting knowledge graphs in Neo4j or FalkorDB. Topics: Neo4j, Cypher, Persistence · *Intermediate*
|
||||
- **[Ontology](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/14_Ontology.ipynb)** — Defining domain schemas and ontologies to structure your data. Topics: OWL, RDF, Schema Design · *Intermediate*
|
||||
- **[Seed Data](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/25_Seed_Data.ipynb)** — Bootstrapping a knowledge graph from trusted CSV, JSON, database, and API sources before extraction runs. Topics: SeedDataManager, Foundation Graphs · *Intermediate*
|
||||
|
||||
|
||||
## Advanced Concepts
|
||||
@@ -50,6 +51,9 @@ Deep dive into advanced features, customization, and complex workflows.
|
||||
- **[Multi-Source Integration](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb)** — Merging data from disparate sources into a unified graph. Topics: Entity Resolution, Merging, Fusion · *Advanced*
|
||||
- **[Reasoning and Inference](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/08_Reasoning_and_Inference.ipynb)** — Using logical reasoning to infer new knowledge from existing facts. Topics: Logic Rules, Inference Engines · *Advanced*
|
||||
- **[Temporal Knowledge Graphs](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb)** — Modeling and querying data that changes over time. Topics: Time Series, Temporal Logic, Allen Algebra · *Advanced*
|
||||
- **[Provenance Tracking](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/22_Provenance_Tracking.ipynb)** — Audit-grade, W3C PROV-O-aligned tracking of where every entity, relationship, and chunk came from. Topics: PROV-O, Lineage, Checksums, Invalidation · *Advanced*
|
||||
- **[Reasoning Module](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/23_Reasoning.ipynb)** — Deriving new knowledge from existing facts with forward chaining, backward chaining, and Datalog strategies. Topics: Reasoner, Datalog, Explanations · *Advanced*
|
||||
- **[Change Management](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/24_Change_Management.ipynb)** — Versioning, audit trails, and data-integrity checks for knowledge graphs and ontologies. Topics: ChangeLogEntry, Version Storage, Data Integrity · *Advanced*
|
||||
|
||||
|
||||
## How to Run
|
||||
@@ -80,6 +84,6 @@ Deep dive into advanced features, customization, and complex workflows.
|
||||
You can also run the cookbook using Docker:
|
||||
|
||||
```bash
|
||||
docker run -p 8888:8888 hawksight/semantica-cookbook
|
||||
docker run -p 8888:8888 semantica/semantica-cookbook
|
||||
```
|
||||
</Tip>
|
||||
|
||||
@@ -103,6 +103,7 @@
|
||||
"pages": [
|
||||
"integrations/agno",
|
||||
"integrations/crewai",
|
||||
"integrations/langchain",
|
||||
"integrations/docling",
|
||||
"integrations/snowflake",
|
||||
"integrations/databricks"
|
||||
|
||||
@@ -162,7 +162,7 @@ semantica-explorer --graph my_graph.json --no-browser
|
||||
```
|
||||
|
||||
<Warning>
|
||||
`--host 0.0.0.0` makes Explorer reachable on every network interface. The server has no built-in authentication. Only use this on a trusted private network.
|
||||
`--host 0.0.0.0` makes Explorer reachable on every network interface. Since v0.6.5 the Explorer API requires `SEMANTICA_API_KEY` (sent as the `X-API-Key` header) and fails closed with `503` when unconfigured; unauthenticated access is only possible when `SEMANTICA_ALLOW_ANONYMOUS=true` is set explicitly. Only use this on a trusted private network.
|
||||
</Warning>
|
||||
|
||||
|
||||
|
||||
+11
-1
@@ -17,7 +17,7 @@ icon: "circle-question"
|
||||
| API key required? | Optional: pattern extraction works with no keys |
|
||||
| Works with LangChain / LlamaIndex? | Yes: Semantica is a layer on top, not a replacement |
|
||||
| Production-ready? | Yes: 1,000+ tests, v0.5.0 ships with 12 security fixes |
|
||||
| Latest version? | **v0.6.5** (August 2026) |
|
||||
| Latest version? | **v0.6.7** (August 2026) |
|
||||
| Local LLMs? | Yes: Ollama via LiteLLM, HuggingFaceLLM for air-gapped |
|
||||
|
||||
|
||||
@@ -52,6 +52,16 @@ Semantica works alongside these frameworks, not against them.
|
||||
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Does Semantica explain an LLM's internal reasoning or chain-of-thought?" icon="triangle-exclamation">
|
||||
|
||||
No. This is **system-level explainability, not foundation-model explainability**. Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system.
|
||||
|
||||
What Semantica explains is *outside* the model: what context and data were used, what decision was produced, the provenance behind it, the relevant relationships, the policies applied, and the resulting decision trail.
|
||||
|
||||
In short: Semantica explains and audits *what the AI system did* — not the foundation model's private internal reasoning.
|
||||
|
||||
</Accordion>
|
||||
|
||||
<Accordion title="Is Semantica free?" icon="tag">
|
||||
|
||||
Yes: MIT licensed, no vendor lock-in, no paywalled features. Some capabilities require third-party API keys (e.g., OpenAI embeddings, Groq inference), but Semantica itself is always free and open source.
|
||||
|
||||
@@ -42,7 +42,7 @@ icon: "rocket"
|
||||
Verify installation:
|
||||
```python
|
||||
import semantica
|
||||
print(semantica.__version__) # 0.6.5
|
||||
print(semantica.__version__) # 0.6.7
|
||||
```
|
||||
</Check>
|
||||
</Step>
|
||||
|
||||
+2
-2
@@ -4,12 +4,12 @@ description: "Project governance model: roles, decision process, release cadence
|
||||
icon: "scale-balanced"
|
||||
---
|
||||
|
||||
> Semantica is maintained by Hawksight AI with community contributions under an open governance model.
|
||||
> Semantica is maintained by the Semantica team with community contributions under an open governance model.
|
||||
|
||||
|
||||
## Roles
|
||||
|
||||
- **Maintainers** — Hawksight AI team: review and merge PRs, manage releases and code quality, set project direction and community standards.
|
||||
- **Maintainers** — Semantica team: review and merge PRs, manage releases and code quality, set project direction and community standards.
|
||||
- **Contributors** — Submit code, documentation, and bug reports. Help with issues and reviews. Recognized in [CONTRIBUTORS.md](https://github.com/semantica-agi/semantica/blob/main/CONTRIBUTORS.md).
|
||||
- **Community Members** — Use Semantica, provide feedback, share use cases, and participate in GitHub Discussions and Discord.
|
||||
|
||||
|
||||
@@ -436,6 +436,35 @@ d = graph.to_dict()
|
||||
# d["statistics"] → {"node_count": int, "edge_count": int}
|
||||
```
|
||||
|
||||
For a human-editable, version-control-friendly representation, save a Markdown
|
||||
directory instead:
|
||||
|
||||
```python
|
||||
graph.save_to_file("context_graph/", format="markdown")
|
||||
|
||||
restored = ContextGraph(advanced_analytics=True)
|
||||
restored.load_from_file("context_graph/", format="markdown")
|
||||
```
|
||||
|
||||
The directory contains a versioned `graph.md` manifest for graph identity,
|
||||
relationships, and cross-graph link descriptors, plus one file per node under
|
||||
`nodes/`. A node's content is its Markdown body; its ID, type, properties,
|
||||
metadata, and temporal validity are YAML frontmatter. Node, edge, family, graph,
|
||||
and cross-graph link IDs are preserved across round trips.
|
||||
|
||||
Markdown loading uses replacement semantics, like `from_dict()`: it parses and
|
||||
validates the complete directory before replacing the current graph. Invalid YAML,
|
||||
duplicate IDs, unsupported versions, and unsafe filesystem links fail without
|
||||
partially mutating the graph. As with JSON loading, an edge endpoint without a node
|
||||
file creates an `entity` stub node. Symlinks, Windows directory junctions, and other
|
||||
Windows reparse points are rejected.
|
||||
|
||||
Re-exporting to an existing managed directory atomically replaces it, removing stale
|
||||
node files. Before replacement, Semantica validates the complete canonical export
|
||||
layout, not just the manifest header. Untracked files, assets, extra directories, or
|
||||
renamed node files therefore cause the export to fail closed instead of being deleted.
|
||||
Keep attachments and hand-written indexes outside the managed export directory.
|
||||
|
||||
If the graph had cross-graph links created with `link_graph()`, call `resolve_links()` after loading to restore live navigation — object references cannot be serialized, so they must be reconnected manually:
|
||||
|
||||
```python
|
||||
|
||||
@@ -382,6 +382,45 @@ For authentication details (PAT vs. OAuth M2M for Databricks; password vs. key-p
|
||||
|
||||
> **Security Note:** Never hardcode credentials (`token`, `password`, `private_key`) in production code; pass them via environment variables (e.g., `DATABRICKS_TOKEN`, `SNOWFLAKE_PASSWORD`) or a secrets manager.
|
||||
|
||||
## Source 7 — SAP OData
|
||||
|
||||
`SAPIngestor` ingests an Entity Set from a SAP OData service (S/4HANA Cloud, SuccessFactors, or an on-prem NetWeaver Gateway over its REST surface). It speaks OData v2 and v4, follows server-driven pagination automatically, and flattens each record into a document dict via `export_as_documents()` — the same structured "transform to text, then store" pattern as the other sources.
|
||||
|
||||
```python
|
||||
from semantica.ingest import SAPIngestor
|
||||
|
||||
ing = SAPIngestor(
|
||||
base_url="https://my-sap.example.com/sap/opu/odata/sap/API_BUSINESS_PARTNER",
|
||||
client_id="...", client_secret="...",
|
||||
token_url="https://my-sap.example.com/oauth/token", # OAuth2 client-credentials (BTP/S/4HANA Cloud)
|
||||
# On-prem NetWeaver often uses Basic auth instead — swap the block above for:
|
||||
# username="erp_user", password="...",
|
||||
)
|
||||
|
||||
# 1. Discover an unfamiliar service: entity sets + field types from $metadata
|
||||
sets = ing.discover_service() # [{"name": "A_BusinessPartnerSet", "fields": [...]}, ...]
|
||||
|
||||
# 2. Pull a page-walked Entity Set (v2/v4 next links handled for you)
|
||||
partners = ing.ingest_entity_set(
|
||||
entity_set="A_BusinessPartnerSet",
|
||||
select="BusinessPartner,BusinessPartnerFullName", # $select
|
||||
top=1000, # cap on total rows
|
||||
)
|
||||
|
||||
# 3. Flatten to document dicts, then build text for the graph
|
||||
docs = ing.export_as_documents(partners)
|
||||
partner_texts = [
|
||||
f"Business Partner {d['BusinessPartner']}: {d['BusinessPartnerFullName']}"
|
||||
for d in docs
|
||||
]
|
||||
```
|
||||
|
||||
- Use `expand="to_Item"` (e.g. on a sales-order header set) to pull nested line items in one request — handy for modeling order → line-item → material relationships.
|
||||
- Every outbound request, including the OAuth2 token exchange, is routed through the SSRF guard, so a user-supplied SAP URL can never reach private/loopback/link-local address space.
|
||||
- Install with `pip install 'semantica[ingest-sap]'`.
|
||||
|
||||
> **Security Note:** Never hardcode credentials (`client_secret`, `password`) in code; pass them via environment variables (e.g., `SAP_CLIENT_SECRET`, `SAP_PASSWORD`) or a secrets manager.
|
||||
|
||||
## Combining All Five Sources
|
||||
|
||||
Once you have text from each source, `AgentContext.store()` accepts a flat list of strings. Semantica embeds and indexes them together — the context graph has no concept of which string came from which source unless you add metadata explicitly.
|
||||
|
||||
+16
-5
@@ -222,10 +222,10 @@ builder.register_step_handler("ner_extract", run_ner)
|
||||
builder.register_step_handler("triplet_extract", run_triplets)
|
||||
builder.register_step_handler("kg_merge", merge_into_graph)
|
||||
|
||||
builder.add_step("ingest", "file_ingest", handler=ingest_stix_bundles, path="./stix_bundles/")
|
||||
builder.add_step("ner", "ner_extract", handler=run_ner, confidence_threshold=0.75)
|
||||
builder.add_step("triplets", "triplet_extract", handler=run_triplets, include_temporal=True)
|
||||
builder.add_step("store", "kg_merge", handler=merge_into_graph, output_path="./cti_output/")
|
||||
builder.add_step("ingest", "file_ingest", path="./stix_bundles/")
|
||||
builder.add_step("ner", "ner_extract", confidence_threshold=0.75)
|
||||
builder.add_step("triplets", "triplet_extract", include_temporal=True)
|
||||
builder.add_step("store", "kg_merge", output_path="./cti_output/")
|
||||
|
||||
# ingest feeds both ner and triplets in parallel
|
||||
builder.connect_steps("ingest", "ner")
|
||||
@@ -241,7 +241,18 @@ engine = ExecutionEngine(max_workers=2, retry_on_failure=True)
|
||||
result = engine.execute_pipeline(pipeline)
|
||||
```
|
||||
|
||||
`set_parallelism(n)` tells the engine how many steps it may run simultaneously. The topological sort guarantees that only steps whose dependencies are all completed are eligible for concurrent execution — you cannot accidentally run a step before its inputs are ready.
|
||||
`set_parallelism(n)` tells the engine how many steps it may run simultaneously; `n` must be a positive integer. The topological sort guarantees that only steps whose dependencies are all completed are eligible for concurrent execution — you cannot accidentally run a step before its inputs are ready. The effective concurrency is capped at `min(n, max_workers)`, so the engine's `max_workers` setting remains a hard resource ceiling.
|
||||
|
||||
Concurrency is opt-in per step. A dependency layer only runs in parallel when every step in that layer is marked `parallel_safe`, the layer has more than one step, and the data flowing into the layer is a dict:
|
||||
|
||||
```python
|
||||
builder.add_step("ner", "ner_extract", parallel_safe=True, confidence_threshold=0.75)
|
||||
builder.add_step("triplets", "triplet_extract", parallel_safe=True, include_temporal=True)
|
||||
```
|
||||
|
||||
If any step in a layer is not marked `parallel_safe`, or if a step runs in delta mode, the entire layer falls back to sequential execution — parallelism never silently bypasses a step that was not declared safe. `parallel_safe` is a control field: like `dependencies`, it is consumed by the builder and never reaches your handler's config.
|
||||
|
||||
Parallel-safe handlers must return a dict. Each step in a parallel layer receives an isolated deep copy of the layer's input, so steps cannot see each other's mutations. The per-step results are merged key by key in step declaration order: a key written by one step is added to the merged output, a key written by several steps with equal values is kept, and two steps writing different values for the same key fail the pipeline with a `ProcessingError` naming the conflicting key and both steps. Handlers that touch shared mutable resources — database connections, in-memory stores, global caches — should not be marked `parallel_safe`.
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
|
||||
@@ -639,7 +639,7 @@ Every `ProvenanceEntry` maps directly to W3C PROV-O terms. If your compliance te
|
||||
| — | `previous_version_id` | This entry corrects/replaces a prior version of the *same* fact |
|
||||
| `prov:wasDerivedFrom` | `derived_from_id` | This entry was derived from a *different* source entity |
|
||||
| `prov:used` | `used_entities` | Entity IDs consumed to produce this one |
|
||||
| `prov:generatedAtTime` | `timestamp` | ISO datetime, auto-set to `datetime.utcnow()` at write time |
|
||||
| `prov:generatedAtTime` | `timestamp` | ISO datetime, auto-set to `utc_now_iso()` at write time |
|
||||
| `prov:qualifiedInvalidation` | `invalidated`, `invalidated_at_time`, `invalidated_by`, `invalidation_reason` | A retraction/correction recorded as a tombstone via `ProvenanceManager.invalidate()`, never a hard delete |
|
||||
| `prov:startedAtTime` / `prov:endedAtTime` | `activity_started_at_time`, `activity_ended_at_time` | Typed Activity timing — pass an `ActivityRecord` via the `activity=` kwarg to set these together with `activity_id` |
|
||||
| `prov:qualifiedGeneration`/`Generation`, `qualifiedUsage`/`Usage`, `qualifiedDerivation`/`Derivation` | (derived from the fields above) | Additive qualified forms of `wasGeneratedBy`/`used`/`wasDerivedFrom`, emitted automatically alongside the plain triples |
|
||||
|
||||
+19
-15
@@ -150,6 +150,12 @@ HighRiskSupplier(DELTA-3) conf=100% rule=Rule 3
|
||||
|
||||
DELTA-3 is flagged even though no document described it that way — the system traced: DELTA-3 supplied GAMMA-7, and GAMMA-7 exploits critical CVEs. For rules that need priority ordering or graded confidence, use the `Rule` dataclass:
|
||||
|
||||
If a rule has side-effecting actions, one concrete activation runs those
|
||||
actions at most once on a Reasoner instance. Re-running `forward_chain()` is
|
||||
therefore safe: already-attempted actions are not repeated. Use
|
||||
`reasoner.reset_action_history()` when you intentionally want to replay them;
|
||||
`reasoner.clear()` and `reasoner.reset()` also clear the history.
|
||||
|
||||
```python
|
||||
# Higher priority rules fire first; confidence propagates into InferenceResult.confidence
|
||||
reasoner.add_rule(Rule(
|
||||
@@ -269,7 +275,7 @@ print("Loaded {} facts from graph".format(count))
|
||||
|
||||
## Step 5 — SPARQL queries over enriched working memory
|
||||
|
||||
After forward chaining has derived new facts, `SPARQLReasoner` lets you query the enriched working memory using SPARQL triple-pattern matching with optional inference expansion:
|
||||
After forward chaining has derived new facts, `SPARQLReasoner` prepares SPARQL queries over the enriched working memory with optional inference expansion:
|
||||
|
||||
```python
|
||||
from semantica.reasoning import SPARQLReasoner
|
||||
@@ -288,22 +294,13 @@ query = """
|
||||
}
|
||||
"""
|
||||
|
||||
# execute_query() runs: expansion → inference → deduplication
|
||||
result = sparql.execute_query(query)
|
||||
|
||||
for binding in result.bindings:
|
||||
print("Actor: {:15s} CVE: {}".format(
|
||||
binding.get("actor", "?"),
|
||||
binding.get("cve", "?"),
|
||||
))
|
||||
|
||||
# metadata shows how many results came from inference vs ground facts
|
||||
print("Original: {} Inferred: {}".format(
|
||||
result.metadata.get("original_count", 0),
|
||||
result.metadata.get("inferred_count", 0),
|
||||
))
|
||||
# expand_query() applies inference rules to the query text:
|
||||
expanded = sparql.expand_query(query)
|
||||
print(expanded)
|
||||
```
|
||||
|
||||
`execute_query()` is not implemented yet: no triplet-store execution path exists, so it raises `NotImplementedError` rather than returning an empty result set that callers would misread as "no matches". Until execution lands, run the expanded query against your RDF store directly (for example with `rdflib`).
|
||||
|
||||
Inspect the expanded query before running it:
|
||||
|
||||
```python
|
||||
@@ -369,6 +366,13 @@ engine.reset()
|
||||
|
||||
The rule network is compiled once by `build_network()`. Each subsequent `add_fact()` call propagates incrementally through only the nodes whose conditions it satisfies — not the full rule set — which keeps evaluation cost proportional to the number of new activations rather than the total rule count.
|
||||
|
||||
With a Reasoner bound, Rete action side effects are attempted once per rule,
|
||||
bindings, and matched fact identity. Passing the same match to
|
||||
`execute_matches()` again still returns the same conclusion, but does not repeat
|
||||
its actions. Call `engine.reset_action_history()` to replay actions without
|
||||
clearing working memory. `engine.reset()` and `engine.build_network()` also
|
||||
clear the action history.
|
||||
|
||||
## Step 7 — Temporal interval reasoning
|
||||
|
||||
`TemporalReasoningEngine` computes Allen interval relations between time windows, letting you identify whether two events overlap, one contains the other, they meet at a boundary, and so on across your graph:
|
||||
|
||||
@@ -8,7 +8,7 @@ icon: "shield-check"
|
||||
|
||||
SHACL (Shapes Constraint Language) is a standard for validating graph-based data. While an ontology defines the conceptual *schema* (the "what" exists in your domain), SHACL defines the structural *rules and constraints* (the "how" it should be structured).
|
||||
|
||||
In Semantica, `SHACLGenerator` produces constraint rules (shapes) based on your ontology, and `_run_pyshacl` evaluates your actual data against these rules. If a node violates a rule (e.g., missing a required property or using the wrong datatype), a detailed violation report is generated.
|
||||
In Semantica, `SHACLGenerator` produces constraint rules (shapes) based on your ontology, and the public `run_shacl_validation` function evaluates your actual data against these rules. If a node violates a rule (e.g., missing a required property or using the wrong datatype), a detailed violation report is generated. The historical `_run_pyshacl` name remains available as a compatibility alias.
|
||||
|
||||
## Why Use SHACL Validation?
|
||||
|
||||
@@ -55,7 +55,7 @@ Let's look at a simple, universally understood example: ensuring every `Employee
|
||||
```python
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
# 1. Prepare your data graph
|
||||
graph = ContextGraph()
|
||||
@@ -95,7 +95,7 @@ data_ttl = """
|
||||
"""
|
||||
|
||||
# 5. Run Validation
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
report = run_shacl_validation(data_ttl, shacl_ttl)
|
||||
|
||||
# 6. Analyze the Report
|
||||
print(f"Graph conforms: {report.conforms}")
|
||||
@@ -265,10 +265,10 @@ cve_id_shape = NodeShape(
|
||||
|
||||
## Step 4 — Run validation and read the report
|
||||
|
||||
Serialize the graph to RDF, then run `_run_pyshacl` against the shapes.
|
||||
Serialize the graph to RDF, then run `run_shacl_validation` against the shapes.
|
||||
|
||||
```python
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
# Prepare your RDF data string (since export_rdf primarily exports structural metadata,
|
||||
# you typically serialize your custom data graph to Turtle using rdflib or similar).
|
||||
@@ -281,7 +281,7 @@ data_ttl = """
|
||||
"""
|
||||
|
||||
# Run SHACL validation
|
||||
report = _run_pyshacl(
|
||||
report = run_shacl_validation(
|
||||
data_ttl,
|
||||
shacl_ttl,
|
||||
data_graph_format="turtle",
|
||||
@@ -366,8 +366,8 @@ print(f"Malware nodes missing 'family': {len(missing_family)}")
|
||||
# e.g. graph.update_node(node_id, {"family": "UNKNOWN — requires triage"})
|
||||
|
||||
# After remediation, re-run validation to confirm the fix
|
||||
# (re-export the patched graph to Turtle first, then call _run_pyshacl again)
|
||||
report2 = _run_pyshacl(patched_data_ttl, shacl_ttl)
|
||||
# (re-export the patched graph to Turtle first, then call run_shacl_validation again)
|
||||
report2 = run_shacl_validation(patched_data_ttl, shacl_ttl)
|
||||
print(f"Violations after remediation: {report2.violation_count}")
|
||||
# Violations after remediation: 0
|
||||
```
|
||||
@@ -377,10 +377,49 @@ print(f"Violations after remediation: {report2.violation_count}")
|
||||
## Common Pitfalls
|
||||
|
||||
- **Assuming the ontology automatically enforces data quality**: `SHACLGenerator` generates shapes based on what it observes in the data. If your data is missing a field, the generator won't know it was mandatory unless you explicitly inject the constraint (as shown in Step 3).
|
||||
- **Passing `ContextGraph` directly to SHACL validators**: The `_run_pyshacl` function expects an RDF string (like Turtle format), not a raw Python dictionary or `ContextGraph` object.
|
||||
- **Passing `ContextGraph` directly to SHACL validators**: The `run_shacl_validation` function expects an RDF string (like Turtle format), not a raw Python dictionary or `ContextGraph` object.
|
||||
- **Forgetting RDF serialization**: You must serialize your graph (often via a temporary file using `export_rdf`) before validating it.
|
||||
- **Treating validation as a one-time step**: Validation should be integrated as an automated step in your CI/CD pipeline or data ingestion flow, acting as a recurring gatekeeper rather than a one-off script.
|
||||
- **Ignoring validation reports**: A graph that does not conform must be remediated. Failing to review the `violation_count` and address the issues negates the purpose of SHACL validation.
|
||||
- **Validating `sh:class`/`sh:node` range checks on a property that declares `rdfs:range` with RDFS entailment on**: RDFS is an entailment rule, not a constraint. When pyshacl runs with `inference="rdfs"`, it infers the range class onto every object of the property, so class-based constraints on that property can never fail — the report says `conforms: True` on data that does not conform:
|
||||
|
||||
```python
|
||||
from pyshacl import validate
|
||||
from rdflib import Graph
|
||||
|
||||
data = Graph()
|
||||
data.parse(
|
||||
data="""
|
||||
@prefix ex: <https://example.org/ns#> .
|
||||
@prefix rdfs: <http://www.w3.org/2000/01/rdf-schema#> .
|
||||
ex:contains rdfs:domain ex:Container ; rdfs:range ex:Item .
|
||||
ex:box a ex:Container ; ex:contains ex:notAnItem .
|
||||
ex:notAnItem a ex:Fish .
|
||||
""",
|
||||
format="turtle",
|
||||
)
|
||||
|
||||
shapes = Graph()
|
||||
shapes.parse(
|
||||
data="""
|
||||
@prefix ex: <https://example.org/ns#> .
|
||||
@prefix sh: <http://www.w3.org/ns/shacl#> .
|
||||
ex:ContainerShape a sh:NodeShape ;
|
||||
sh:targetClass ex:Container ;
|
||||
sh:property [ sh:path ex:contains ; sh:class ex:Item ] .
|
||||
""",
|
||||
format="turtle",
|
||||
)
|
||||
|
||||
for inference in ("none", "rdfs"):
|
||||
conforms, _, _ = validate(data, shacl_graph=shapes, inference=inference)
|
||||
print(inference, conforms)
|
||||
# none False <- correct: notAnItem is a Fish, not an Item
|
||||
# rdfs True <- the entailment manufactured the type
|
||||
```
|
||||
|
||||
Mitigations: prefer not to declare `rdfs:range` on properties you intend to constrain with `sh:class`; when class membership is the thing under test, run validation without RDFS entailment (`inference="none"`); or express the check as a constraint the entailment cannot satisfy (for example a literal property constraint). Note the trade-off: with entailment off, `sh:targetClass` no longer reaches subclasses, so subclass hierarchies need explicit typing or inference-aware target selection. Semantica's own `run_shacl_validation` wrapper already calls pyshacl with `inference="none"`, so this pitfall only bites when calling `pyshacl.validate` directly with entailment enabled.
|
||||
- **Trusting `conforms: True` without checking the inference mode**: an inference-enabled run can hide the exact violations the shapes were written to catch (see above). Record which inference mode validation ran under alongside the result, and re-run shape sets that contain `sh:class`/`sh:node` with entailment off before treating a pass as authoritative.
|
||||
|
||||
---
|
||||
|
||||
@@ -396,7 +435,7 @@ A DoD CTI team enforces STIX-compatible constraints on a threat graph before sha
|
||||
from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
graph = ContextGraph()
|
||||
ctx = AgentContext(
|
||||
@@ -448,7 +487,7 @@ data_ttl = """
|
||||
<http://example.org/hammertoss> a ex:Malware .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
report = run_shacl_validation(data_ttl, shacl_ttl)
|
||||
print(f"CTI graph conforms : {report.conforms}")
|
||||
print(f"Violations : {report.violation_count}")
|
||||
print(f"Warnings : {report.warning_count}")
|
||||
@@ -469,7 +508,7 @@ A SOC team validates zero-trust policy nodes before publishing them to the polic
|
||||
```python
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
graph = ContextGraph()
|
||||
graph.add_node("policy-001", "Policy", "MFA Required for Tier-1 Resources",
|
||||
@@ -516,7 +555,7 @@ data_ttl = """
|
||||
<http://example.org/policy-002> a ex:Policy .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
report = run_shacl_validation(data_ttl, shacl_ttl)
|
||||
print(f"Policy graph conforms: {report.conforms}")
|
||||
# Policy graph conforms: False
|
||||
|
||||
@@ -534,7 +573,7 @@ A clinical informatics team validates trial ontology nodes before loading them i
|
||||
|
||||
```python
|
||||
from semantica.ontology import LLMOntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
from semantica.export import export_rdf
|
||||
import tempfile, os
|
||||
|
||||
@@ -586,7 +625,7 @@ with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
os.unlink(tmp.name)
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
report = run_shacl_validation(data_ttl, shacl_ttl)
|
||||
print(f"Trial data conforms: {report.conforms}")
|
||||
print(f"Warnings : {report.warning_count}")
|
||||
```
|
||||
@@ -600,7 +639,7 @@ A credit risk team validates every `LoanApplication` node against Basel III CRE2
|
||||
```python
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
graph = ContextGraph()
|
||||
graph.add_node("loan-001", "LoanApplication", "Prime mortgage APP-2025-88421",
|
||||
@@ -645,7 +684,7 @@ data_ttl = """
|
||||
ex:ltv "0.65" .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
report = run_shacl_validation(data_ttl, shacl_ttl)
|
||||
print(f"Loan portfolio conforms: {report.conforms}")
|
||||
# Loan portfolio conforms: False
|
||||
|
||||
@@ -675,14 +714,14 @@ Call this function as a pre-publish gate; exit code 1 blocks the pipeline.
|
||||
```python
|
||||
import sys
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.ontology import run_shacl_validation
|
||||
|
||||
def validate_before_publish(data_graph_str: str, ontology: dict) -> None:
|
||||
shacl_gen = SHACLGenerator(base_uri="https://example.org/shapes/")
|
||||
shacl_graph = shacl_gen.generate(ontology)
|
||||
shacl_ttl = shacl_gen.serialize(shacl_graph, format="turtle")
|
||||
|
||||
report = _run_pyshacl(data_graph_str, shacl_ttl)
|
||||
report = run_shacl_validation(data_graph_str, shacl_ttl)
|
||||
|
||||
if not report.conforms:
|
||||
print(f"Graph validation FAILED — {report.violation_count} violation(s)")
|
||||
@@ -700,7 +739,6 @@ def validate_before_publish(data_graph_str: str, ontology: dict) -> None:
|
||||
|
||||
- [Ontology Management](ontology) — generate the OWL ontology that SHACL shapes are derived from
|
||||
- [Reasoning & Rules](reasoning) — complement SHACL structural constraints with logical inference rules
|
||||
- [Export & Serialization](export) — serialize graph data to Turtle/RDF/XML for `_run_pyshacl` input
|
||||
- [Export & Serialization](export) — serialize graph data to Turtle/RDF/XML for `run_shacl_validation` input
|
||||
- [Conflict Resolution](conflict-resolution) — detect and resolve data conflicts before SHACL validation
|
||||
- [Change Management](change-management) — version-gate SHACL shapes alongside ontology versions
|
||||
|
||||
|
||||
+5
-1
@@ -192,7 +192,11 @@ decision_id = context.record_decision(
|
||||
|
||||
## Built for Where Mistakes Have Consequences
|
||||
|
||||
Semantica was designed for domains where every decision must be explainable and every fact must be traceable:
|
||||
Semantica was designed for domains where every decision must be explainable and every fact must be traceable.
|
||||
|
||||
<Warning>
|
||||
**This is system-level explainability, not foundation-model explainability.** Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. What Semantica explains is *outside* the model: the context and data fed in, the decision produced, its provenance, the relevant relationships, the policies applied, and the full execution trail. See [Core Concepts](concepts) for the full scope note.
|
||||
</Warning>
|
||||
|
||||
**Healthcare & Life Sciences**
|
||||
- Clinical decision support with full audit trails
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
---
|
||||
title: "LangChain Integration"
|
||||
description: "Drop Semantica into LangChain / LangGraph pipelines via a GraphRAG retriever, VectorStore adapter, and agent tools."
|
||||
icon: "link"
|
||||
---
|
||||
|
||||
> Three drop-in adapters that bring Semantica's context graph and hybrid search into LangChain chains and LangGraph agents.
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install "semantica[langchain]"
|
||||
```
|
||||
|
||||
Requires `langchain-core >= 0.3`. If langchain-core is not installed, the integration still imports — every class carries the full Semantica API and degrades gracefully (`build()` returns `None`; branch on `LANGCHAIN_AVAILABLE`).
|
||||
|
||||
## Components at a Glance
|
||||
|
||||
- **SemanticaRetriever** — `BaseRetriever`: hybrid-search seeds retrieval, then graph edges are walked `hops` steps (default 2) for GraphRAG-style results.
|
||||
- **SemanticaVectorStore** — `VectorStore`: `add_texts` / `similarity_search` / `similarity_search_with_score` / `from_texts` over `HybridSearch`.
|
||||
- **SemanticaKGTool** / **SemanticaDecisionTool** — `BaseTool` subclasses: `semantica_query_graph` and `semantica_query_decisions` for LangGraph / tool-calling agents.
|
||||
|
||||
## Component Details
|
||||
|
||||
<Tabs>
|
||||
<Tab title="SemanticaRetriever">
|
||||
Hybrid search seeds retrieval; then graph edges are walked `hops` steps so results go beyond flat vector similarity. If hybrid search is omitted or fails, the retriever falls back to a `ContextGraph.query` keyword scan.
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaRetriever
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.vector_store import HybridSearch
|
||||
|
||||
graph = ContextGraph()
|
||||
hybrid = HybridSearch()
|
||||
|
||||
retriever = SemanticaRetriever(graph=graph, hybrid=hybrid, hops=2, top_k=10)
|
||||
|
||||
from langchain.chains import RetrievalQA
|
||||
|
||||
qa = RetrievalQA.from_chain_type(llm=llm, retriever=retriever)
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="SemanticaVectorStore">
|
||||
Drop-in `VectorStore` for RetrievalQA / LCEL chains. `from_texts` requires a pre-configured `hybrid` instance.
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaVectorStore
|
||||
|
||||
store = SemanticaVectorStore(hybrid=hybrid)
|
||||
store.add_texts(
|
||||
["document one", "document two"],
|
||||
metadatas=[{"source": "a"}, {"source": "b"}],
|
||||
)
|
||||
docs = store.similarity_search("document", k=2)
|
||||
docs, scores = store.similarity_search_with_score("document", k=2)
|
||||
```
|
||||
|
||||
`add_texts` delegates to a Semantica vector store with `add_documents` (pass `vector_store=` to `HybridSearch` or to `SemanticaVectorStore`).
|
||||
</Tab>
|
||||
<Tab title="Agent tools">
|
||||
Instances are LangChain `BaseTool`s and can be passed to an agent directly.
|
||||
`.build()` returns the tool, or `None` when langchain-core is absent.
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaKGTool, SemanticaDecisionTool
|
||||
from langgraph.prebuilt import create_react_agent
|
||||
|
||||
tools = [
|
||||
SemanticaKGTool(graph),
|
||||
SemanticaDecisionTool(graph),
|
||||
]
|
||||
agent = create_react_agent(model, tools)
|
||||
```
|
||||
|
||||
| Tool | Description |
|
||||
| :------ | :------------- |
|
||||
| `semantica_query_graph` | Keyword / NL query over the shared context graph |
|
||||
| `semantica_query_decisions` | Search the recorded decision log |
|
||||
</Tab>
|
||||
</Tabs>
|
||||
@@ -12,7 +12,7 @@ icon: "file-contract"
|
||||
```
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2026 Hawksight AI
|
||||
Copyright (c) 2026 Semantica
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
|
||||
@@ -435,8 +435,8 @@ print("Nodes: {}, Edges: {}".format(stats["node_count"], stats["edge_count"]))
|
||||
| `query(query, skip, limit)` | `List[Dict]` | Full-text search over node content |
|
||||
| `stats()` | `Dict` | Node/edge counts, type breakdowns, graph density |
|
||||
| `density()` | `float` | Graph density score |
|
||||
| `save_to_file(path)` | `None` | Persist graph to JSON |
|
||||
| `load_from_file(path)` | `None` | Load graph from JSON |
|
||||
| `save_to_file(path, format="json")` | `None` | Persist graph as JSON or a Markdown directory |
|
||||
| `load_from_file(path, format="json")` | `None` | Replace graph state from JSON or a Markdown directory |
|
||||
| `build_from_conversations(conversations, link_entities)` | `Dict` | Build graph from conversation data |
|
||||
| `link_graph(other_graph, source_node_id, target_node_id, link_type)` | `str` | Create cross-graph navigation link; returns `link_id` |
|
||||
| `navigate_to(link_id)` | `Tuple` | Follow a cross-graph link to `(target_graph, target_node_id)` |
|
||||
@@ -625,8 +625,10 @@ malformed or duplicate fields before changing memory, and re-importing unchanged
|
||||
files is idempotent. Memory-local `entities` and `relationships` are preserved as
|
||||
provenance but are not applied to `ContextGraph` by Markdown import. Use a dedicated
|
||||
export directory: matching files are overwritten, but unrelated or stale Markdown
|
||||
files are not deleted automatically. Export refuses to overwrite symbolic links and
|
||||
uses atomic file replacement. Timestamp offsets are preserved in Markdown and
|
||||
files are not deleted automatically. Export refuses to overwrite filesystem links and
|
||||
uses atomic file replacement; import also refuses symlinks, Windows directory
|
||||
junctions, and other Windows reparse points.
|
||||
Timestamp offsets are preserved in Markdown and
|
||||
normalized to UTC only for comparisons, so aware and local-naive records can be
|
||||
queried together safely. Vector-store writes are deferred until the in-memory import
|
||||
commits; adapter synchronization remains best-effort and logs failures.
|
||||
|
||||
@@ -28,6 +28,7 @@ icon: "database"
|
||||
| `DBIngestor` | SQL databases via SQLAlchemy: tables, views, and custom queries |
|
||||
| `SnowflakeIngestor` | Snowflake data warehouse queries and table exports |
|
||||
| `DatabricksIngestor` | Databricks Unity Catalog metadata, Delta table queries, and lineage |
|
||||
| `SAPIngestor` | SAP OData services (S/4HANA Cloud, SuccessFactors, NetWeaver Gateway): entity-set discovery and ingestion with v2/v4 pagination |
|
||||
| `ParquetIngestor` | Apache Parquet files and partitioned datasets with column selection |
|
||||
| `ArrowIngestor` | Apache Arrow IPC and Feather file processing |
|
||||
| `XMLIngestor` | XXE-safe XML parsing with optional XSD schema validation |
|
||||
|
||||
@@ -250,7 +250,7 @@ entry = ProvenanceEntry(
|
||||
source_document="report.pdf", # str: default ""
|
||||
source_location="Page 4", # Optional[str]: default None
|
||||
source_quote="Relevant text...", # Optional[str]: default None
|
||||
timestamp="2024-01-01T12:00:00", # str: auto-set to utcnow()
|
||||
timestamp="2024-01-01T12:00:00+00:00", # str: auto-set to utc_now_iso()
|
||||
first_seen=None, # Optional[str]: ISO timestamp
|
||||
last_updated=None, # Optional[str]: ISO timestamp
|
||||
confidence=0.9, # float: default 1.0
|
||||
|
||||
@@ -127,9 +127,19 @@ conclusions = reasoner.infer_facts(
|
||||
| `forward_chain()` | `List[InferenceResult]` | Derive all possible conclusions iteratively until fixpoint |
|
||||
| `backward_chain(goal, max_depth)` | `InferenceResult \| None` | Prove a specific goal string, returns `None` if unprovable |
|
||||
| `infer_facts(facts, rules)` | `List[str]` | Load facts and rules then run `forward_chain()`, returns conclusion strings |
|
||||
| `clear()` | `None` | Clear all facts and rules |
|
||||
| `reset_action_history()` | `None` | Allow actions for previously fired activations to run again |
|
||||
| `clear()` | `None` | Clear all facts, rules, and action activation history |
|
||||
| `reset()` | `None` | Alias for `clear()` |
|
||||
|
||||
Rules with actions use at-most-once attempt semantics per concrete activation
|
||||
(rule ID, bindings, and matched facts). Calling `forward_chain()` again on the
|
||||
same instance does not repeat side effects for an activation that was already
|
||||
attempted, even when an action raised an exception. Call
|
||||
`reset_action_history()` to deliberately retry without clearing facts or rules;
|
||||
`clear()` and `reset()` also clear this history. Replacing a rule's actions in
|
||||
place does not invalidate an existing activation; reset the history explicitly
|
||||
when the replacement should be replayed.
|
||||
|
||||
### Rule and Fact dataclass fields
|
||||
|
||||
```python
|
||||
@@ -230,9 +240,16 @@ engine.reset()
|
||||
| `add_fact(fact)` | `None` | Add a `Fact` to working memory and propagate through the network |
|
||||
| `match_patterns(facts)` | `List[Match]` | Match all patterns; optionally add facts before matching |
|
||||
| `execute_matches(matches)` | `List[Any]` | Execute matched rules and return their conclusion values |
|
||||
| `reset()` | `None` | Clear facts and all node activation state |
|
||||
| `reset_action_history()` | `None` | Allow actions for previously executed activations to run again |
|
||||
| `reset()` | `None` | Clear facts, node activation state, and action activation history |
|
||||
| `get_network_stats()` | `dict` | Return counts of alpha, beta, terminal nodes and facts |
|
||||
|
||||
When a Reasoner is bound, `execute_matches()` deduplicates action side effects
|
||||
by rule ID, bindings, and matched fact identity. Re-executing a match still
|
||||
returns its conclusion for compatibility, but its actions are skipped after the
|
||||
first attempt. `reset_action_history()`, `reset()`, and `build_network()` allow
|
||||
those actions to run again.
|
||||
|
||||
|
||||
## SPARQLReasoner
|
||||
|
||||
|
||||
@@ -182,7 +182,7 @@ for row in result.bindings:
|
||||
store = TripletStore(
|
||||
backend="rdf4j",
|
||||
endpoint="http://localhost:8080/rdf4j-server",
|
||||
repository_id="semantica", # passed through **config
|
||||
repository_id="semantica", # selects the remote repository
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
@@ -77,7 +77,17 @@ Most users won't call utils directly: it's the **shared foundation** for all mod
|
||||
export SEMANTICA_LOG_LEVEL=DEBUG
|
||||
export SEMANTICA_LOG_FORMAT=json # "json" | "text"
|
||||
export SEMANTICA_DISABLE_PROGRESS=true
|
||||
export SEMANTICA_FORCE_PROGRESS=true
|
||||
```
|
||||
|
||||
<Tip>
|
||||
**Progress bars follow your terminal.** Console progress is written only when
|
||||
stdout is an interactive terminal (or a Jupyter notebook), so piping or
|
||||
redirecting output no longer fills logs with progress bars and escape
|
||||
sequences. Set `SEMANTICA_DISABLE_PROGRESS` to silence progress even in a
|
||||
terminal, or `SEMANTICA_FORCE_PROGRESS` to keep it when stdout is redirected.
|
||||
`SEMANTICA_DISABLE_PROGRESS` wins if both are set.
|
||||
</Tip>
|
||||
</Step>
|
||||
</Steps>
|
||||
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
# Graph storage backends and feature matrix
|
||||
|
||||
Semantica separates graph modeling from physical storage. LPG backends are accessed through `graph_store` adapters; RDF backends are accessed through `triplet_store` adapters.
|
||||
|
||||
This page is intentionally conservative: it distinguishes between an adapter existing, a feature being generally available with that model, and a backend needing user-supplied wiring.
|
||||
|
||||
## Status labels
|
||||
|
||||
- `built-in`: adapter implementation exists in Semantica core.
|
||||
- `tested`: covered by automated integration fixtures or tests.
|
||||
- `example-only`: usable example exists, but support is not asserted by integration tests.
|
||||
- `interface/BYO`: interface or integration point exists; bring your own backend wiring.
|
||||
|
||||
## Adapter inventory
|
||||
|
||||
| Backend | Model | Adapter | Status | Reference |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| Neo4j | LPG | `semantica.graph_store.Neo4jStore` | built-in | `cookbook/introduction/09_Graph_Store.ipynb` |
|
||||
| FalkorDB | LPG | `semantica.graph_store.FalkorDBStore` | built-in | `docs/reference/graph_store.md` |
|
||||
| Amazon Neptune | LPG | `semantica.graph_store.AmazonNeptuneStore` | built-in | `cookbook/introduction/21_Amazon_Neptune_Store.ipynb` |
|
||||
| Apache AGE | LPG | `semantica.graph_store.ApacheAgeStore` | built-in | `docs/graph_stores/apache_age.md` |
|
||||
| RDF4J | RDF | `semantica.triplet_store.RDF4JStore` | built-in | `cookbook/introduction/20_Triplet_Store.ipynb` |
|
||||
| Apache Jena | RDF | `semantica.triplet_store.JenaStore` | built-in | `cookbook/introduction/20_Triplet_Store.ipynb` |
|
||||
| Blazegraph | RDF | `semantica.triplet_store.BlazegraphStore` | built-in | `cookbook/introduction/20_Triplet_Store.ipynb` |
|
||||
| Anzo | RDF | `semantica.triplet_store.AnzoStore` | built-in | `cookbook/introduction/20_Triplet_Store.ipynb` |
|
||||
| Oxigraph | RDF | `semantica.triplet_store.OxigraphStore` | built-in | `docs/reference/triplet_store.md` |
|
||||
|
||||
## Feature matrix
|
||||
|
||||
`Yes` means the capability is expected to work with the adapter and graph model. `Partial` means the capability works with model-specific constraints. `BYO` means the user must supply or validate wiring for the backend.
|
||||
|
||||
| Backend | Model | Ingestion | Context graph construction | Reasoning/analytics | Provenance | Known limitations |
|
||||
| --- | --- | --- | --- | --- | --- | --- |
|
||||
| Neo4j | LPG | Yes | Yes | Yes | Partial | Provenance and context metadata are stored as node and edge properties; relationship properties and stable node identifiers are required. |
|
||||
| FalkorDB | LPG | Yes | Yes | Partial | Partial | Redis-based; provenance depends on node/edge properties, and multi-graph isolation depends on the selected graph name. |
|
||||
| Amazon Neptune | LPG | Yes | Yes | Partial | Partial | Use the property-graph endpoint; AWS auth, VPC, and endpoint configuration can affect local tests. Provenance depends on node/edge properties. |
|
||||
| Apache AGE | LPG | Yes | Yes | Partial | Partial | Runs through PostgreSQL/AGE; Cypher compatibility and property handling can differ from standalone LPG engines. |
|
||||
| RDF4J | RDF | Yes | Partial | Partial | Partial | Context separation relies on named graphs; triple-level provenance may require reification or graph-level metadata. |
|
||||
| Apache Jena | RDF | Yes | Partial | Partial | Partial | Named graphs are needed for context separation; backend configuration and transaction behavior matter. |
|
||||
| Blazegraph | RDF | Yes | Partial | Partial | Partial | Use quads/named graphs for context; IRI stability and graph naming matter for provenance. |
|
||||
| Anzo | RDF | Yes | Partial | Partial | Partial | Anzo deployments are environment-specific; validate `dataset_uri`/graphmart naming, named-graph support, and provenance mapping. |
|
||||
| Oxigraph | RDF | Yes | Partial | Partial | Partial | Embedded, single-process store (in-memory or on-disk); named graphs are supported, but there is no separate server process to scale independently. |
|
||||
|
||||
## RDF and LPG differences
|
||||
|
||||
- LPG backends store context and provenance as graph elements and properties. If a backend does not support relationship properties, some provenance patterns may be degraded.
|
||||
- RDF backends rely on IRIs, named graphs, and optional reification. Context graphs and provenance are easiest to preserve when the store supports named graphs/quads.
|
||||
- Ingestion works across both models, but the physical representation differs: LPG stores nodes/edges directly, while RDF stores subject-predicate-object statements.
|
||||
- Reasoning and analytics should be validated against the adapter's query capabilities, especially for path traversal, property filters, and named-graph queries.
|
||||
|
||||
## Minimal connection examples
|
||||
|
||||
Prefer the referenced notebook cells for a working setup. The examples below show the intended adapter entrypoints, not a universal connection DSL.
|
||||
|
||||
### Neo4j
|
||||
|
||||
```python
|
||||
import os
|
||||
from semantica.graph_store import Neo4jStore
|
||||
|
||||
store = Neo4jStore(
|
||||
uri='bolt://localhost:7687',
|
||||
user='neo4j',
|
||||
password=os.environ['NEO4J_PASSWORD']
|
||||
)
|
||||
```
|
||||
|
||||
### FalkorDB
|
||||
|
||||
```python
|
||||
from semantica.graph_store import FalkorDBStore
|
||||
|
||||
store = FalkorDBStore(
|
||||
host='localhost',
|
||||
port=6379,
|
||||
graph_name='semantica'
|
||||
)
|
||||
```
|
||||
|
||||
### Amazon Neptune
|
||||
|
||||
```python
|
||||
from semantica.graph_store import AmazonNeptuneStore
|
||||
|
||||
store = AmazonNeptuneStore(
|
||||
endpoint='your-neptune-cluster-endpoint',
|
||||
port=8182,
|
||||
region='us-east-1'
|
||||
)
|
||||
```
|
||||
|
||||
### Apache AGE
|
||||
|
||||
```python
|
||||
from semantica.graph_store import ApacheAgeStore
|
||||
|
||||
store = ApacheAgeStore(
|
||||
connection_string='host=localhost dbname=agedb user=postgres password=postgres',
|
||||
graph_name='semantica'
|
||||
)
|
||||
```
|
||||
|
||||
### RDF4J
|
||||
|
||||
```python
|
||||
from semantica.triplet_store import RDF4JStore
|
||||
|
||||
store = RDF4JStore(
|
||||
endpoint='http://localhost:8080/rdf4j-server',
|
||||
repository_id='semantica'
|
||||
)
|
||||
```
|
||||
|
||||
### Apache Jena
|
||||
|
||||
```python
|
||||
from semantica.triplet_store import JenaStore
|
||||
|
||||
store = JenaStore(
|
||||
endpoint='http://localhost:3030/ds'
|
||||
)
|
||||
```
|
||||
|
||||
### Blazegraph
|
||||
|
||||
```python
|
||||
from semantica.triplet_store import BlazegraphStore
|
||||
|
||||
store = BlazegraphStore(
|
||||
endpoint='http://localhost:9999/blazegraph/sparql'
|
||||
)
|
||||
```
|
||||
|
||||
### Anzo
|
||||
|
||||
```python
|
||||
from semantica.triplet_store import AnzoStore
|
||||
|
||||
store = AnzoStore(
|
||||
endpoint='http://anzo-host:8080',
|
||||
dataset_uri='http://cambridgesemantics.com/Graphmart/your-graphmart-id'
|
||||
)
|
||||
```
|
||||
|
||||
### Oxigraph
|
||||
|
||||
```python
|
||||
from semantica.triplet_store import OxigraphStore
|
||||
|
||||
# Omit `path` for an in-memory store; pass a directory for on-disk persistence.
|
||||
store = OxigraphStore(path='./semantica-oxigraph-data')
|
||||
```
|
||||
|
||||
Replace hostnames, ports, repositories, graphs, and credentials with values from your environment. For regulated or self-hosted deployments, keep credentials in environment variables or secret storage rather than source code.
|
||||
+10
-1
@@ -63,7 +63,9 @@ semantica-explorer --graph my_graph.json --no-browser
|
||||
python -m semantica.explorer --graph my_graph.json
|
||||
```
|
||||
|
||||
> **Security note:** The Explorer API has no built-in authentication. The default `--host 127.0.0.1` binds to localhost only, so it is not reachable from other machines on your network. If you bind to `0.0.0.0`, all graph data is readable and writable by any host that can reach the port. The CLI will print a warning in that case.
|
||||
> **Security note:** Since v0.6.5 the Explorer API requires an API key on protected routes. Set the `SEMANTICA_API_KEY` environment variable and send it as the `X-API-Key` header; without a configured key, protected routes fail closed with `503` rather than serving anonymously. To opt into unauthenticated access for local development only, set `SEMANTICA_ALLOW_ANONYMOUS=true` explicitly. (`/api/health` and `/api/info` are intentionally unauthenticated.)
|
||||
>
|
||||
> The default `--host 127.0.0.1` binds to localhost only, so it is not reachable from other machines on your network. If you bind to `0.0.0.0`, all graph data is readable and writable by any host that can reach the port (subject to API-key auth). The CLI prints a warning when binding to a non-loopback host in anonymous mode or when `SEMANTICA_API_KEY` is unset.
|
||||
|
||||
---
|
||||
|
||||
@@ -148,6 +150,8 @@ This writes the compiled assets to `../semantica/static/`. The Python server the
|
||||
| --- | --- | --- |
|
||||
| `EXPLORER_CORS_ORIGINS` | `http://localhost:5173,http://127.0.0.1:5173` | Comma-separated list of allowed CORS origins |
|
||||
| `EXPLORER_CORS_CREDENTIALS` | `false` | Set to `true` to allow credentialed cross-origin requests (only needed behind an authenticating reverse proxy) |
|
||||
| `SEMANTICA_API_KEY` | *(unset)* | API key required on protected routes since v0.6.5; send it as the `X-API-Key` header. When unset, protected routes fail closed with `503`. |
|
||||
| `SEMANTICA_ALLOW_ANONYMOUS` | `false` | Set to `true` to opt into unauthenticated access (local development only). |
|
||||
|
||||
---
|
||||
|
||||
@@ -251,6 +255,11 @@ Vite automatically tries the next available port and prints the actual URL in th
|
||||
- Confirm the backend exposes the `/ws/graph-updates` WebSocket endpoint.
|
||||
- Check DevTools → Network → WS tab for the connection status and error code.
|
||||
- Ensure the backend version matches the frontend — mixing major versions can cause protocol mismatches.
|
||||
- **Authentication:** `/ws/graph-updates` enforces the same API key as the REST routes. Browsers cannot set custom headers on a WebSocket handshake, so pass the key as a query parameter instead:
|
||||
```
|
||||
ws://127.0.0.1:8000/ws/graph-updates?api_key=<your-key>
|
||||
```
|
||||
Non-browser clients (native apps, scripts) may send it as the `X-API-Key` header. A missing or incorrect key results in close code `4401`; if `SEMANTICA_API_KEY` is unset and `SEMANTICA_ALLOW_ANONYMOUS` is not `true`, the connection is also rejected. Note that API keys in URLs appear in server logs — prefer the header for non-browser clients.
|
||||
|
||||
---
|
||||
|
||||
|
||||
Generated
+1489
-14
File diff suppressed because it is too large
Load Diff
@@ -9,7 +9,7 @@
|
||||
"lint": "eslint .",
|
||||
"preview": "vite preview",
|
||||
"test:graph-store": "node --test tests/graphStore.multi-edge.test.mjs",
|
||||
"test:graph-workspace": "node --import tsx --test tests/graphSceneState.display.test.ts tests/temporalLifecycle.test.ts",
|
||||
"test:graph-workspace": "node --import tsx --test tests/markdownContentViewer.test.ts tests/graphSceneState.display.test.ts tests/temporalLifecycle.test.ts",
|
||||
"test:plugin-registry": "node --import tsx --test tests/pluginRegistry.temporal.test.mjs"
|
||||
},
|
||||
"dependencies": {
|
||||
@@ -29,6 +29,8 @@
|
||||
"react-arborist": "^3.4.3",
|
||||
"react-dom": "^19.2.4",
|
||||
"react-dropzone": "^15.0.0",
|
||||
"react-markdown": "^10.1.0",
|
||||
"remark-gfm": "^4.0.1",
|
||||
"sigma": "^3.0.2",
|
||||
"vis-data": "^8.0.3",
|
||||
"vis-timeline": "^8.5.0"
|
||||
|
||||
@@ -162,7 +162,11 @@ const SIGMA_SETTINGS = {
|
||||
hideLabelsOnMove: true,
|
||||
hideEdgesOnMove: true,
|
||||
enableEdgeEvents: true,
|
||||
renderEdgeLabels: false,
|
||||
// #1009: edge labels (the edge `type` — "works_for", "leads", ...) were
|
||||
// hardcoded off, so edge text never rendered regardless of data. The
|
||||
// labelDensity / labelGridCellSize / labelRenderedSizeThreshold settings
|
||||
// below already throttle label density for both nodes and edges.
|
||||
renderEdgeLabels: true,
|
||||
labelDensity: 0.7,
|
||||
labelGridCellSize: 140,
|
||||
zIndex: true,
|
||||
@@ -741,6 +745,12 @@ function buildEffectAvailability(
|
||||
? { enabled: true, available: true, reason: "Panel enabled" }
|
||||
: { enabled: false, available: false, reason: "Disabled by toggle" };
|
||||
|
||||
// #1009: edge labels are immediately available once the graph is loaded —
|
||||
// they have no async analytics or zoom-tier dependency.
|
||||
const edgeLabels = effectsState.edgeLabelsEnabled
|
||||
? { enabled: true, available: true, reason: "Ready" }
|
||||
: { enabled: false, available: false, reason: "Disabled by toggle" };
|
||||
|
||||
const diagnostics = !GRAPH_THEME.effects.diagnostics.enabledInDev
|
||||
? { enabled: false, available: false, reason: "Disabled in production" }
|
||||
: effectsState.diagnosticsEnabled
|
||||
@@ -758,6 +768,7 @@ function buildEffectAvailability(
|
||||
communities,
|
||||
centrality,
|
||||
legend,
|
||||
edgeLabels,
|
||||
diagnostics,
|
||||
};
|
||||
}
|
||||
@@ -1211,6 +1222,12 @@ function applySceneState(
|
||||
size: resolvedStyle.size,
|
||||
zIndex: resolvedStyle.zIndex,
|
||||
curvature: resolvedStyle.curvature,
|
||||
// #1009: Sigma's edge label renderer draws data.label — the graph
|
||||
// stores the relationship type in edgeType, which the renderer never
|
||||
// saw, so enabling renderEdgeLabels alone left edges blank.
|
||||
// Use || rather than ?? so that an empty-string edgeType (possible
|
||||
// when the API returns type: "") does not produce a blank label.
|
||||
label: resolvedStyle.hidden ? undefined : String(attrs.edgeType || data.label || ""),
|
||||
};
|
||||
});
|
||||
|
||||
@@ -1295,6 +1312,9 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle, GraphCanvasProps>(
|
||||
const onEdgeClickRef = useRef(onEdgeClick);
|
||||
const onSceneRuntimeChangeRef = useRef(onSceneRuntimeChange);
|
||||
const onCameraStateChangeRef = useRef(onCameraStateChange);
|
||||
// #1009: tracked as a ref so the Sigma creation effect always reads the
|
||||
// current value without needing effectsState in its dependency array.
|
||||
const effectsStateRef = useRef(effectsState);
|
||||
const [hoveredNodeId, setHoveredNodeId] = useState<string | null>(null);
|
||||
const [zoomTier, setZoomTier] = useState<GraphZoomTier>("overview");
|
||||
const [analyticsSnapshot, setAnalyticsSnapshot] = useState<GraphAnalyticsSnapshot | null>(null);
|
||||
@@ -1323,6 +1343,7 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle, GraphCanvasProps>(
|
||||
onEdgeClickRef.current = onEdgeClick;
|
||||
onSceneRuntimeChangeRef.current = onSceneRuntimeChange;
|
||||
onCameraStateChangeRef.current = onCameraStateChange;
|
||||
effectsStateRef.current = effectsState;
|
||||
|
||||
const behaviors = useMemo<GraphBehavior[]>(
|
||||
() => [
|
||||
@@ -1835,7 +1856,13 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle, GraphCanvasProps>(
|
||||
return;
|
||||
}
|
||||
|
||||
const sigma = new Sigma(displayGraphRef.current, containerRef.current, SIGMA_SETTINGS);
|
||||
const sigma = new Sigma(displayGraphRef.current, containerRef.current, {
|
||||
...SIGMA_SETTINGS,
|
||||
// #1009: initialize with the current toggle value rather than the
|
||||
// static default so that a user who disabled Edge Labels before
|
||||
// graph/Sigma initialization sees the correct state after mount.
|
||||
renderEdgeLabels: effectsStateRef.current.edgeLabelsEnabled,
|
||||
});
|
||||
sigmaRef.current = sigma;
|
||||
appliedGraphVersionRef.current = graphVersionRef.current;
|
||||
|
||||
@@ -1937,6 +1964,17 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle, GraphCanvasProps>(
|
||||
});
|
||||
}, [behaviors, dispatchToBehaviors, getBehaviorContext, graphReady, syncCameraState]);
|
||||
|
||||
// #1009: renderEdgeLabels follows the Effects-panel toggle instead of
|
||||
// staying hardcoded — dense graphs get their label-free edges back.
|
||||
useEffect(() => {
|
||||
const sigma = sigmaRef.current;
|
||||
if (!sigma) {
|
||||
return;
|
||||
}
|
||||
sigma.setSetting("renderEdgeLabels", effectsState.edgeLabelsEnabled);
|
||||
sigma.scheduleRefresh();
|
||||
}, [effectsState.edgeLabelsEnabled]);
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
const sigma = sigmaRef.current;
|
||||
|
||||
@@ -3,6 +3,7 @@ import { Loader2 } from "lucide-react";
|
||||
import { graph } from "../../store/graphStore";
|
||||
import { GRAPH_THEME, withAlpha } from "./graphTheme";
|
||||
import type { GraphSelectedNodeKind } from "./types";
|
||||
import { MarkdownContentViewer } from "./MarkdownContentViewer";
|
||||
|
||||
export type LinkPrediction = {
|
||||
target: string;
|
||||
@@ -364,6 +365,11 @@ export function GraphInspectorPanel({
|
||||
([key]) =>
|
||||
!["x","y","valid_from","valid_until","content","source","source_url","pmid","pmids","evidence","provenance","confidence"].includes(key),
|
||||
);
|
||||
const nodeContent = (typeof attributes?.content === "string" && attributes.content)
|
||||
? attributes.content
|
||||
: (typeof properties.content === "string" && properties.content)
|
||||
? properties.content
|
||||
: "";
|
||||
|
||||
return (
|
||||
<aside style={{ padding: 24, display: "flex", flexDirection: "column", gap: 18 }}>
|
||||
@@ -408,6 +414,20 @@ export function GraphInspectorPanel({
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{/* Content Section — only rendered when the node carries actual content.
|
||||
This matches the existing inspector convention: sections that have no
|
||||
data for the current node are either hidden (temporal bounds) or closed
|
||||
by default (Source Attribution, Properties). Always showing an open
|
||||
empty panel would add noise for every relationship/predicate node. */}
|
||||
{nodeContent && (
|
||||
<details className="node-panel-collapse" open>
|
||||
<summary className="node-panel-summary">Content</summary>
|
||||
<div className="node-panel-body" style={{ marginTop: 8 }}>
|
||||
<MarkdownContentViewer content={nodeContent} />
|
||||
</div>
|
||||
</details>
|
||||
)}
|
||||
|
||||
{/* Actions */}
|
||||
<section style={sectionStyle}>
|
||||
<div style={sectionTitleStyle}>Actions</div>
|
||||
|
||||
@@ -41,6 +41,7 @@ import {
|
||||
} from "./plugins";
|
||||
import { explorationEffectsShouldLoad, neighborhoodPanelShouldLoad, temporalOverlayShouldLoad } from "./pluginRegistryPredicates";
|
||||
import { shouldFetchTemporalBounds, shouldFetchTemporalSnapshot } from "./temporalLifecyclePredicates";
|
||||
import { createTemporalSnapshotGuards, type TemporalSnapshotResponse } from "./temporalSnapshotGuards";
|
||||
import type { LinkPrediction, PathResponse } from "./GraphInspectorPanel";
|
||||
import type { GraphSceneHandle, GraphSceneRuntime } from "./scene";
|
||||
import type {
|
||||
@@ -148,6 +149,7 @@ const DEFAULT_EFFECTS_STATE: GraphEffectsState = {
|
||||
communitiesEnabled: false,
|
||||
centralityEnabled: false,
|
||||
legendEnabled: false,
|
||||
edgeLabelsEnabled: true,
|
||||
diagnosticsEnabled: false,
|
||||
lensMode: "neighborhood",
|
||||
effectQuality: "bounded",
|
||||
@@ -1478,6 +1480,23 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
summary?.edgeCount,
|
||||
]);
|
||||
|
||||
// Guards the snapshot lifecycle: at most one in-flight request per scrubber
|
||||
// position (identical-`at` polls are deduplicated, breaking the idle/play
|
||||
// polling loop), applied snapshots are cached and re-applied on revisit, and
|
||||
// a response applies only while the scrubber is still on its position
|
||||
// (out-of-order responses cannot clobber the active-node count).
|
||||
const temporalSnapshotGuardsRef = useRef<ReturnType<typeof createTemporalSnapshotGuards> | null>(null);
|
||||
if (temporalSnapshotGuardsRef.current === null) {
|
||||
temporalSnapshotGuardsRef.current = createTemporalSnapshotGuards();
|
||||
}
|
||||
const temporalSnapshotGuards = temporalSnapshotGuardsRef.current;
|
||||
|
||||
// A new graph summary means the graph data was replaced (reload/retry);
|
||||
// snapshots cached against the previous graph are stale, so reset all state.
|
||||
useEffect(() => {
|
||||
temporalSnapshotGuards.reset();
|
||||
}, [summary]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!canFetchTemporalSnapshot) {
|
||||
return;
|
||||
@@ -1487,37 +1506,67 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
return;
|
||||
}
|
||||
|
||||
const atMs = debouncedTime.getTime();
|
||||
const { seq, cached } = temporalSnapshotGuards.begin(atMs);
|
||||
if (seq === null) {
|
||||
// An identical request is already in flight: one request per position.
|
||||
return;
|
||||
}
|
||||
|
||||
let cancelled = false;
|
||||
|
||||
const applyData = (data: TemporalSnapshotResponse) => {
|
||||
const nextActiveIds = new Set(data.active_node_ids);
|
||||
requestAnimationFrame(() => {
|
||||
if (cancelled) return;
|
||||
if (!temporalSnapshotGuards.shouldApply(atMs, seq)) {
|
||||
// The scrubber moved on (or this request was superseded): release the
|
||||
// position so a return to it refetches instead of stalling.
|
||||
temporalSnapshotGuards.finish(atMs, seq);
|
||||
return;
|
||||
}
|
||||
const previous = prevActiveIdsRef.current;
|
||||
previous.forEach((id) => {
|
||||
if (!nextActiveIds.has(id) && graph.hasNode(id)) {
|
||||
graph.setNodeAttribute(id, "hidden", true);
|
||||
}
|
||||
});
|
||||
nextActiveIds.forEach((id) => {
|
||||
if (graph.hasNode(id)) {
|
||||
graph.setNodeAttribute(id, "hidden", false);
|
||||
}
|
||||
});
|
||||
prevActiveIdsRef.current = nextActiveIds;
|
||||
setActiveNodeCount(data.active_node_count);
|
||||
setGraphVersion((current) => current + 1);
|
||||
sceneRef.current?.getRuntime()?.requestRender();
|
||||
temporalSnapshotGuards.apply(atMs, seq, data);
|
||||
});
|
||||
};
|
||||
|
||||
if (cached) {
|
||||
// Returning to a position whose snapshot was already applied: re-apply
|
||||
// the cached result without a network request.
|
||||
applyData(cached);
|
||||
return;
|
||||
}
|
||||
|
||||
const applySnapshot = async () => {
|
||||
try {
|
||||
const at = debouncedTime.toISOString();
|
||||
const response = await fetch(`/api/temporal/snapshot?at=${encodeURIComponent(at)}`);
|
||||
if (!response.ok || cancelled) return;
|
||||
|
||||
const data: { active_node_ids: string[]; active_node_count: number } = await response.json();
|
||||
if (!response.ok) {
|
||||
// A failed request must be retryable if the scrubber returns.
|
||||
if (!cancelled) temporalSnapshotGuards.finish(atMs, seq);
|
||||
return;
|
||||
}
|
||||
if (cancelled) return;
|
||||
|
||||
const nextActiveIds = new Set(data.active_node_ids);
|
||||
requestAnimationFrame(() => {
|
||||
if (cancelled) return;
|
||||
const previous = prevActiveIdsRef.current;
|
||||
previous.forEach((id) => {
|
||||
if (!nextActiveIds.has(id) && graph.hasNode(id)) {
|
||||
graph.setNodeAttribute(id, "hidden", true);
|
||||
}
|
||||
});
|
||||
nextActiveIds.forEach((id) => {
|
||||
if (graph.hasNode(id)) {
|
||||
graph.setNodeAttribute(id, "hidden", false);
|
||||
}
|
||||
});
|
||||
prevActiveIdsRef.current = nextActiveIds;
|
||||
setActiveNodeCount(data.active_node_count);
|
||||
setGraphVersion((current) => current + 1);
|
||||
sceneRef.current?.getRuntime()?.requestRender();
|
||||
});
|
||||
const data: TemporalSnapshotResponse = await response.json();
|
||||
if (cancelled) return;
|
||||
applyData(data);
|
||||
} catch (fetchError) {
|
||||
temporalSnapshotGuards.finish(atMs, seq);
|
||||
if (!cancelled) {
|
||||
console.error("[Temporal] Snapshot fetch failed", fetchError);
|
||||
}
|
||||
@@ -1527,6 +1576,8 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
applySnapshot();
|
||||
return () => {
|
||||
cancelled = true;
|
||||
// A cancelled request must be retryable when its position is revisited.
|
||||
temporalSnapshotGuards.finish(atMs, seq);
|
||||
};
|
||||
}, [
|
||||
canFetchTemporalSnapshot,
|
||||
|
||||
@@ -0,0 +1,404 @@
|
||||
import { useState, useRef, useEffect, useMemo, type CSSProperties } from "react";
|
||||
import ReactMarkdown, { type Components } from "react-markdown";
|
||||
import remarkGfm from "remark-gfm";
|
||||
import { Check, Copy, Code2, Eye, ExternalLink, Image as ImageIcon } from "lucide-react";
|
||||
import { GRAPH_THEME } from "./graphTheme";
|
||||
import { isSafeUrl } from "./markdownUrlSafety";
|
||||
|
||||
export interface MarkdownContentViewerProps {
|
||||
content?: string | null;
|
||||
className?: string;
|
||||
defaultMode?: "preview" | "source";
|
||||
}
|
||||
|
||||
export function MarkdownContentViewer({
|
||||
content,
|
||||
className,
|
||||
defaultMode = "preview",
|
||||
}: MarkdownContentViewerProps) {
|
||||
const [activeMode, setActiveMode] = useState<"preview" | "source">(defaultMode);
|
||||
const [copied, setCopied] = useState(false);
|
||||
// Track the content value for which the copied indicator is valid.
|
||||
// When content changes (i.e. the user selects a different node), reset the
|
||||
// copied indicator inline during render rather than in a useEffect — this
|
||||
// avoids a cascading-render lint error and is the React-recommended pattern
|
||||
// for resetting derived visual state on prop changes.
|
||||
const [copiedForContent, setCopiedForContent] = useState<string | null | undefined>(content);
|
||||
if (copiedForContent !== content) {
|
||||
setCopiedForContent(content);
|
||||
if (copied) {
|
||||
// Clear the stale indicator synchronously so the new node's copy button
|
||||
// never shows "Copied" from the previous selection.
|
||||
setCopied(false);
|
||||
}
|
||||
}
|
||||
|
||||
const copyTimeoutRef = useRef<ReturnType<typeof setTimeout> | null>(null);
|
||||
|
||||
// Clean up any outstanding timeout on unmount.
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
if (copyTimeoutRef.current) {
|
||||
clearTimeout(copyTimeoutRef.current);
|
||||
}
|
||||
};
|
||||
}, []);
|
||||
|
||||
const rawContent = typeof content === "string" ? content : "";
|
||||
const hasContent = rawContent.trim().length > 0;
|
||||
|
||||
// react-markdown runs the whole remark pipeline synchronously inside its own
|
||||
// render, so without this memo every unrelated re-render of this component --
|
||||
// clicking Copy, toggling Preview/Source -- re-parses the entire document.
|
||||
// Measured at ~364ms per re-render for a 1000-row GFM table (issue #1118).
|
||||
// Keyed on rawContent so a genuine node change still re-parses exactly once.
|
||||
const renderedMarkdown = useMemo(
|
||||
() => (
|
||||
<ReactMarkdown remarkPlugins={REMARK_PLUGINS} components={MARKDOWN_COMPONENTS}>
|
||||
{rawContent}
|
||||
</ReactMarkdown>
|
||||
),
|
||||
[rawContent],
|
||||
);
|
||||
|
||||
const handleCopy = async () => {
|
||||
if (!hasContent) return;
|
||||
try {
|
||||
await navigator.clipboard.writeText(rawContent);
|
||||
if (copyTimeoutRef.current) {
|
||||
clearTimeout(copyTimeoutRef.current);
|
||||
}
|
||||
setCopied(true);
|
||||
copyTimeoutRef.current = setTimeout(() => setCopied(false), 1500);
|
||||
} catch {
|
||||
// Clipboard write unavailable
|
||||
}
|
||||
};
|
||||
|
||||
return (
|
||||
<div className={className} style={viewerContainerStyle}>
|
||||
<div style={viewerHeaderStyle}>
|
||||
<div style={{ display: "flex", gap: 4 }} role="tablist">
|
||||
<button
|
||||
type="button"
|
||||
role="tab"
|
||||
aria-selected={activeMode === "preview"}
|
||||
onClick={() => setActiveMode("preview")}
|
||||
style={{ ...tabBtnStyle, ...(activeMode === "preview" ? activeTabBtnStyle : {}) }}
|
||||
>
|
||||
<Eye size={12} style={{ marginRight: 5 }} />
|
||||
Preview
|
||||
</button>
|
||||
<button
|
||||
type="button"
|
||||
role="tab"
|
||||
aria-selected={activeMode === "source"}
|
||||
onClick={() => setActiveMode("source")}
|
||||
style={{ ...tabBtnStyle, ...(activeMode === "source" ? activeTabBtnStyle : {}) }}
|
||||
>
|
||||
<Code2 size={12} style={{ marginRight: 5 }} />
|
||||
Source
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{hasContent && (
|
||||
<button type="button" onClick={() => void handleCopy()} style={copyBtnStyle} title="Copy raw content">
|
||||
{copied ? (
|
||||
<>
|
||||
<Check size={12} color="#3fb950" style={{ marginRight: 4 }} />
|
||||
<span style={{ color: "#3fb950", fontSize: 11 }}>Copied</span>
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<Copy size={12} style={{ marginRight: 4 }} />
|
||||
<span style={{ fontSize: 11 }}>Copy</span>
|
||||
</>
|
||||
)}
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div style={viewerBodyStyle}>
|
||||
{!hasContent ? (
|
||||
<div style={emptyTextStyle}>No content available for this node.</div>
|
||||
) : activeMode === "source" ? (
|
||||
<pre style={sourcePreStyle}>
|
||||
<code style={sourceCodeStyle}>{rawContent}</code>
|
||||
</pre>
|
||||
) : (
|
||||
<div style={previewStyle}>{renderedMarkdown}</div>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
/* ─── Markdown rendering config ───────────────────────────────────── */
|
||||
|
||||
// Both props are hoisted to module scope so they keep a stable identity across
|
||||
// renders. As inline literals they allocated a fresh plugin array and ~20 fresh
|
||||
// arrow components on every render, which made React treat every mapped tag as a
|
||||
// new element type and remount the entire rendered subtree instead of updating
|
||||
// it (issue #1118). The arrow bodies only read the style constants below at call
|
||||
// time, so declaring the map before them is safe.
|
||||
const REMARK_PLUGINS = [remarkGfm];
|
||||
|
||||
const MARKDOWN_COMPONENTS: Components = {
|
||||
// C-1: react-markdown passes a HAST `node` prop (the raw AST
|
||||
// Element) to every custom component override via passNode:true.
|
||||
// In React 19 any unknown prop spreads onto a native element are
|
||||
// serialised as HTML attributes, producing node="[object Object]"
|
||||
// on every rendered link. Fix: destructure `node` by name so it
|
||||
// is explicitly discarded, then spread `...rest` to preserve all
|
||||
// other legitimate HAST/remark-gfm attributes — e.g. the `id`,
|
||||
// `aria-describedby`, `aria-label`, `data-footnote-ref`,
|
||||
// `data-footnote-backref`, and `class` attrs that GFM footnotes
|
||||
// require for correct in-page navigation and accessibility.
|
||||
//
|
||||
// C-2: fragment links (#anchor, GFM footnote backlinks) must
|
||||
// navigate within the current document. External links continue
|
||||
// to use target="_blank" with noopener noreferrer.
|
||||
//
|
||||
// eslint-disable-next-line @typescript-eslint/no-unused-vars
|
||||
a: ({ href, children, title, node: _node, ...rest }) => {
|
||||
if (!isSafeUrl(href)) {
|
||||
return <span style={{ color: GRAPH_THEME.ui.text.muted, textDecoration: "line-through" }}>{children}</span>;
|
||||
}
|
||||
// isSafeUrl returning true guarantees href is a non-empty string.
|
||||
const safeHref = href ?? "";
|
||||
// Fragment links (#section, footnote backlinks like
|
||||
// #user-content-fnref-1) are in-document anchors. Opening them
|
||||
// in a new tab would break GFM footnote back-navigation.
|
||||
const isFragment = safeHref.startsWith("#");
|
||||
if (isFragment) {
|
||||
return (
|
||||
<a href={safeHref} title={title} style={linkStyle} {...rest}>
|
||||
{children}
|
||||
</a>
|
||||
);
|
||||
}
|
||||
return (
|
||||
<a href={safeHref} title={title} target="_blank" rel="noopener noreferrer" style={linkStyle} {...rest}>
|
||||
{children}
|
||||
<ExternalLink size={10} style={{ marginLeft: 3, verticalAlign: "middle", display: "inline" }} />
|
||||
</a>
|
||||
);
|
||||
},
|
||||
img: ({ src, alt }) => (
|
||||
<span style={imageBadgeStyle} title={src || "Image"}>
|
||||
<ImageIcon size={12} style={{ marginRight: 5 }} />
|
||||
<span>Image: {alt || src || "unlabeled"}</span>
|
||||
</span>
|
||||
),
|
||||
h1: ({ children }) => <h1 style={h1Style}>{children}</h1>,
|
||||
h2: ({ children }) => <h2 style={h2Style}>{children}</h2>,
|
||||
h3: ({ children }) => <h3 style={h3Style}>{children}</h3>,
|
||||
h4: ({ children }) => <h4 style={h4Style}>{children}</h4>,
|
||||
p: ({ children }) => <p style={{ margin: "0 0 8px 0" }}>{children}</p>,
|
||||
ul: ({ children }) => <ul style={{ margin: "0 0 8px 0", paddingLeft: 18 }}>{children}</ul>,
|
||||
ol: ({ children }) => <ol style={{ margin: "0 0 8px 0", paddingLeft: 18 }}>{children}</ol>,
|
||||
li: ({ children }) => <li style={{ marginBottom: 3 }}>{children}</li>,
|
||||
blockquote: ({ children }) => <blockquote style={blockquoteStyle}>{children}</blockquote>,
|
||||
hr: () => <hr style={{ border: "none", borderTop: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`, margin: "10px 0" }} />,
|
||||
table: ({ children }) => (
|
||||
<div style={{ width: "100%", overflowX: "auto", margin: "8px 0", borderRadius: 6, border: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}` }}>
|
||||
<table style={{ width: "100%", borderCollapse: "collapse", fontSize: 12 }}>{children}</table>
|
||||
</div>
|
||||
),
|
||||
thead: ({ children }) => <thead style={{ background: "rgba(255, 255, 255, 0.04)" }}>{children}</thead>,
|
||||
tbody: ({ children }) => <tbody>{children}</tbody>,
|
||||
tr: ({ children }) => <tr style={{ borderBottom: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}` }}>{children}</tr>,
|
||||
th: ({ children }) => <th style={{ padding: "6px 8px", textAlign: "left", fontWeight: 700, color: GRAPH_THEME.ui.text.strong, borderRight: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}` }}>{children}</th>,
|
||||
td: ({ children }) => <td style={{ padding: "6px 8px", color: GRAPH_THEME.ui.text.body, borderRight: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}` }}>{children}</td>,
|
||||
pre: ({ children }) => <pre style={preBlockStyle}>{children}</pre>,
|
||||
// C-1: discard `node` here too — code elements are custom components
|
||||
// and would otherwise receive node="[object Object]" in the DOM.
|
||||
code: ({ className: codeClass, children }) => {
|
||||
const isInline = !codeClass && typeof children === "string" && !children.includes("\n");
|
||||
return (
|
||||
<code style={isInline ? inlineCodeStyle : blockCodeStyle}>
|
||||
{children}
|
||||
</code>
|
||||
);
|
||||
},
|
||||
};
|
||||
|
||||
/* ─── Styles ──────────────────────────────────────────────────────── */
|
||||
|
||||
const viewerContainerStyle: CSSProperties = {
|
||||
display: "flex",
|
||||
flexDirection: "column",
|
||||
background: "rgba(255, 255, 255, 0.025)",
|
||||
border: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`,
|
||||
borderRadius: 12,
|
||||
overflow: "hidden",
|
||||
};
|
||||
|
||||
const viewerHeaderStyle: CSSProperties = {
|
||||
display: "flex",
|
||||
alignItems: "center",
|
||||
justifyContent: "space-between",
|
||||
padding: "6px 10px",
|
||||
background: "rgba(0, 0, 0, 0.2)",
|
||||
borderBottom: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`,
|
||||
};
|
||||
|
||||
const tabBtnStyle: CSSProperties = {
|
||||
display: "inline-flex",
|
||||
alignItems: "center",
|
||||
padding: "4px 9px",
|
||||
borderRadius: 6,
|
||||
border: "1px solid transparent",
|
||||
background: "transparent",
|
||||
color: GRAPH_THEME.ui.text.muted,
|
||||
fontSize: 12,
|
||||
fontWeight: 600,
|
||||
cursor: "pointer",
|
||||
transition: "all 150ms ease",
|
||||
};
|
||||
|
||||
const activeTabBtnStyle: CSSProperties = {
|
||||
background: GRAPH_THEME.ui.timeline.playheadSoft,
|
||||
border: `1px solid ${GRAPH_THEME.ui.control.activeBorder}`,
|
||||
color: GRAPH_THEME.ui.timeline.playhead,
|
||||
};
|
||||
|
||||
const copyBtnStyle: CSSProperties = {
|
||||
display: "inline-flex",
|
||||
alignItems: "center",
|
||||
padding: "3px 8px",
|
||||
borderRadius: 6,
|
||||
border: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`,
|
||||
background: "rgba(255, 255, 255, 0.04)",
|
||||
color: GRAPH_THEME.ui.text.subtle,
|
||||
fontSize: 11,
|
||||
cursor: "pointer",
|
||||
};
|
||||
|
||||
const viewerBodyStyle: CSSProperties = {
|
||||
padding: 12,
|
||||
maxHeight: 380,
|
||||
overflowY: "auto",
|
||||
};
|
||||
|
||||
const emptyTextStyle: CSSProperties = {
|
||||
color: GRAPH_THEME.ui.text.muted,
|
||||
fontSize: 12,
|
||||
lineHeight: 1.5,
|
||||
fontStyle: "italic",
|
||||
};
|
||||
|
||||
const sourcePreStyle: CSSProperties = {
|
||||
margin: 0,
|
||||
padding: 10,
|
||||
borderRadius: 8,
|
||||
background: "rgba(0, 0, 0, 0.3)",
|
||||
border: "1px solid rgba(255, 255, 255, 0.05)",
|
||||
overflowX: "auto",
|
||||
};
|
||||
|
||||
const sourceCodeStyle: CSSProperties = {
|
||||
fontFamily: "'JetBrains Mono', 'Fira Code', monospace",
|
||||
fontSize: 12,
|
||||
lineHeight: 1.6,
|
||||
color: GRAPH_THEME.ui.text.strong,
|
||||
whiteSpace: "pre-wrap",
|
||||
wordBreak: "break-word",
|
||||
userSelect: "text",
|
||||
};
|
||||
|
||||
const previewStyle: CSSProperties = {
|
||||
color: GRAPH_THEME.ui.text.body,
|
||||
fontSize: 13,
|
||||
lineHeight: 1.6,
|
||||
wordBreak: "break-word",
|
||||
};
|
||||
|
||||
const h1Style: CSSProperties = {
|
||||
fontSize: 16,
|
||||
fontWeight: 700,
|
||||
color: GRAPH_THEME.ui.text.strong,
|
||||
marginTop: 8,
|
||||
marginBottom: 6,
|
||||
paddingBottom: 3,
|
||||
borderBottom: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`,
|
||||
};
|
||||
|
||||
const h2Style: CSSProperties = {
|
||||
fontSize: 14,
|
||||
fontWeight: 700,
|
||||
color: GRAPH_THEME.ui.text.strong,
|
||||
marginTop: 8,
|
||||
marginBottom: 4,
|
||||
};
|
||||
|
||||
const h3Style: CSSProperties = {
|
||||
fontSize: 13,
|
||||
fontWeight: 600,
|
||||
color: GRAPH_THEME.ui.text.strong,
|
||||
marginTop: 6,
|
||||
marginBottom: 4,
|
||||
};
|
||||
|
||||
const h4Style: CSSProperties = {
|
||||
fontSize: 12,
|
||||
fontWeight: 600,
|
||||
color: GRAPH_THEME.ui.text.strong,
|
||||
marginTop: 4,
|
||||
marginBottom: 2,
|
||||
};
|
||||
|
||||
const blockquoteStyle: CSSProperties = {
|
||||
margin: "8px 0",
|
||||
padding: "6px 12px",
|
||||
borderLeft: `3px solid ${GRAPH_THEME.ui.timeline.playhead}`,
|
||||
background: "rgba(98, 226, 205, 0.05)",
|
||||
borderRadius: "0 6px 6px 0",
|
||||
color: GRAPH_THEME.ui.text.body,
|
||||
fontStyle: "italic",
|
||||
};
|
||||
|
||||
const linkStyle: CSSProperties = {
|
||||
color: "#79c0ff",
|
||||
textDecoration: "underline",
|
||||
textUnderlineOffset: "3px",
|
||||
wordBreak: "break-all",
|
||||
};
|
||||
|
||||
const imageBadgeStyle: CSSProperties = {
|
||||
display: "inline-flex",
|
||||
alignItems: "center",
|
||||
padding: "3px 7px",
|
||||
background: "rgba(255, 255, 255, 0.04)",
|
||||
border: `1px solid ${GRAPH_THEME.ui.surface.panelBorder}`,
|
||||
borderRadius: 6,
|
||||
color: GRAPH_THEME.ui.text.muted,
|
||||
fontSize: 11,
|
||||
margin: "3px 0",
|
||||
};
|
||||
|
||||
const inlineCodeStyle: CSSProperties = {
|
||||
fontFamily: "'JetBrains Mono', monospace",
|
||||
fontSize: 12,
|
||||
padding: "2px 5px",
|
||||
borderRadius: 4,
|
||||
background: "rgba(255, 255, 255, 0.07)",
|
||||
color: "#e6edf3",
|
||||
border: "1px solid rgba(255, 255, 255, 0.08)",
|
||||
};
|
||||
|
||||
const preBlockStyle: CSSProperties = {
|
||||
margin: "8px 0",
|
||||
padding: 10,
|
||||
borderRadius: 8,
|
||||
background: "rgba(0, 0, 0, 0.35)",
|
||||
border: "1px solid rgba(255, 255, 255, 0.08)",
|
||||
overflowX: "auto",
|
||||
};
|
||||
|
||||
const blockCodeStyle: CSSProperties = {
|
||||
fontFamily: "'JetBrains Mono', monospace",
|
||||
fontSize: 12,
|
||||
lineHeight: 1.5,
|
||||
color: "#e6edf3",
|
||||
};
|
||||
@@ -2099,6 +2099,15 @@ function createCollapsedNeighborhoodGraph(
|
||||
return collapsedGraph;
|
||||
}
|
||||
|
||||
// Normalize an edge relationship type: empty string, null, and undefined all
|
||||
// fall back to the project-wide default used consistently across every
|
||||
// aggregation path. Keep this local — it exists only to guarantee that the
|
||||
// three code paths (single-entry, multi-entry, community-grouped) produce the
|
||||
// same semantics and do not diverge again.
|
||||
function normalizeEdgeType(value: string | null | undefined): string {
|
||||
return value || "related_to";
|
||||
}
|
||||
|
||||
function aggregateDisplayGraph(graphRef: GraphRef): Graph<NodeAttributes, EdgeAttributes> {
|
||||
const aggregated = new Graph<NodeAttributes, EdgeAttributes>({
|
||||
type: "directed",
|
||||
@@ -2124,10 +2133,13 @@ function aggregateDisplayGraph(graphRef: GraphRef): Graph<NodeAttributes, EdgeAt
|
||||
const [{ edgeId, attrs }] = entries;
|
||||
aggregated.mergeDirectedEdgeWithKey(edgeId, sourceId, targetId, {
|
||||
...attrs,
|
||||
// #1009: normalize empty/null/undefined edgeType so Sigma's label
|
||||
// renderer never receives a blank string on the single-entry path.
|
||||
edgeType: normalizeEdgeType(attrs.edgeType),
|
||||
dominantEdgeType: normalizeEdgeType(attrs.dominantEdgeType ?? attrs.edgeType),
|
||||
rawEdgeIds: collectRawEdgeIds(attrs, edgeId),
|
||||
isAggregated: isAggregatedEdgeAttributes(attrs),
|
||||
aggregateCount: attrs.aggregateCount ?? collectRawEdgeIds(attrs, edgeId).length,
|
||||
dominantEdgeType: attrs.dominantEdgeType ?? attrs.edgeType,
|
||||
representativeWeight: attrs.representativeWeight ?? Number(attrs.weight ?? 1),
|
||||
});
|
||||
return;
|
||||
@@ -2150,10 +2162,11 @@ function aggregateDisplayGraph(graphRef: GraphRef): Graph<NodeAttributes, EdgeAt
|
||||
const rawEdgeIds = entries.flatMap(({ edgeId, attrs }) => collectRawEdgeIds(attrs, edgeId));
|
||||
const typeCounts = new Map<string, number>();
|
||||
entries.forEach(({ attrs }) => {
|
||||
const edgeType = String(attrs.edgeType ?? "related_to");
|
||||
const edgeType = normalizeEdgeType(attrs.edgeType);
|
||||
typeCounts.set(edgeType, (typeCounts.get(edgeType) ?? 0) + 1);
|
||||
});
|
||||
const dominantEdgeType = [...typeCounts.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? representative.attrs.edgeType ?? "related_to";
|
||||
const dominantEdgeType = [...typeCounts.entries()].sort((left, right) => right[1] - left[1])[0]?.[0]
|
||||
?? normalizeEdgeType(representative.attrs.edgeType);
|
||||
const reverseKey = `${targetId}→${sourceId}`;
|
||||
const isBidirectionalBundle = groupedEdges.has(reverseKey);
|
||||
const syntheticEdgeId = `${AGGREGATED_EDGE_PREFIX}${sourceId}::${targetId}`;
|
||||
@@ -2167,10 +2180,10 @@ function aggregateDisplayGraph(graphRef: GraphRef): Graph<NodeAttributes, EdgeAt
|
||||
rawEdgeIds,
|
||||
isAggregated: true,
|
||||
aggregateCount: rawEdgeIds.length,
|
||||
dominantEdgeType: String(dominantEdgeType),
|
||||
dominantEdgeType: dominantEdgeType,
|
||||
representativeWeight: Number(representative.attrs.weight ?? 1),
|
||||
weight: Number(representative.attrs.weight ?? 1),
|
||||
edgeType: String(representative.attrs.edgeType ?? dominantEdgeType ?? "related_to"),
|
||||
edgeType: representative.attrs.edgeType || dominantEdgeType,
|
||||
parallelCount: rawEdgeIds.length,
|
||||
familySize: rawEdgeIds.length,
|
||||
bundleKind: isBidirectionalBundle ? "bidirectional" : "parallel",
|
||||
@@ -2280,7 +2293,7 @@ function buildCommunityGroupedGraph(): GraphDisplayResult {
|
||||
};
|
||||
bucket.rawEdgeIds.push(String(edgeId));
|
||||
bucket.weight = Math.max(bucket.weight, Number((attrs as EdgeAttributes).weight ?? 1));
|
||||
const edgeType = String((attrs as EdgeAttributes).edgeType ?? "related_to");
|
||||
const edgeType = normalizeEdgeType((attrs as EdgeAttributes).edgeType);
|
||||
bucket.typeCounts.set(edgeType, (bucket.typeCounts.get(edgeType) ?? 0) + 1);
|
||||
groupedEdges.set(key, bucket);
|
||||
});
|
||||
@@ -2396,7 +2409,8 @@ function buildCommunityGroupedGraph(): GraphDisplayResult {
|
||||
if (!visibleGroupedEdgeKeys.has(key)) {
|
||||
return;
|
||||
}
|
||||
const dominantEdgeType = [...bundle.typeCounts.entries()].sort((left, right) => right[1] - left[1])[0]?.[0] ?? "related_to";
|
||||
const dominantEdgeType = [...bundle.typeCounts.entries()].sort((left, right) => right[1] - left[1])[0]?.[0]
|
||||
?? "related_to";
|
||||
const reverseKey = `${bundle.targetId}→${bundle.sourceId}`;
|
||||
const syntheticEdgeId = `${AGGREGATED_EDGE_PREFIX}${key}`;
|
||||
const aggregateCount = bundle.rawEdgeIds.length;
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
/**
|
||||
* URL-safety predicate for the Markdown content viewer.
|
||||
*
|
||||
* Extracted into a pure module so the check can be unit-tested without
|
||||
* importing the MarkdownContentViewer React component, and so the component
|
||||
* module exports only components (react-refresh/only-export-components,
|
||||
* issue #1119). The behaviour is unchanged from the original in-component
|
||||
* implementation: only http, https, mailto, in-document fragments, and
|
||||
* root-relative paths are permitted.
|
||||
*/
|
||||
|
||||
export function isSafeUrl(url?: string): boolean {
|
||||
if (!url) return false;
|
||||
const trimmed = url.trim();
|
||||
// Reject whitespace-only strings — new URL("", base) would resolve to the base
|
||||
// protocol and produce a false positive. This guards direct callers of the exported
|
||||
// function; markdown parsers normalise whitespace-only destinations to "" which
|
||||
// already fails the !url check above.
|
||||
if (!trimmed) return false;
|
||||
if (trimmed.startsWith("//")) return false;
|
||||
if (trimmed.startsWith("#")) return true;
|
||||
if (trimmed.startsWith("/")) return true;
|
||||
try {
|
||||
const parsed = new URL(trimmed, "http://localhost");
|
||||
return ["http:", "https:", "mailto:"].includes(parsed.protocol);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
import type { CSSProperties } from "react";
|
||||
|
||||
import type {
|
||||
GraphDiagnosticsSnapshot,
|
||||
GraphEffectAvailability,
|
||||
GraphEffectToggle,
|
||||
} from "../types";
|
||||
@@ -30,6 +31,11 @@ const EFFECT_ROWS: EffectRowConfig[] = [
|
||||
label: "Neighborhood Lens",
|
||||
description: "Local emphasis around the hovered or selected node.",
|
||||
},
|
||||
{
|
||||
key: "edgeLabelsEnabled",
|
||||
label: "Edge Labels",
|
||||
description: "Draw the relationship type on graph edges. Off restores label-free edges on dense graphs.",
|
||||
},
|
||||
{
|
||||
key: "legendEnabled",
|
||||
label: "Semantic Legend",
|
||||
@@ -37,6 +43,17 @@ const EFFECT_ROWS: EffectRowConfig[] = [
|
||||
},
|
||||
];
|
||||
|
||||
// Maps the effect toggle keys rendered by this plugin to their corresponding
|
||||
// availability keys in GraphDiagnosticsSnapshot["effectAvailability"]. Kept
|
||||
// local because this plugin only renders a subset of all effects.
|
||||
const EFFECT_AVAILABILITY_KEYS: Partial<Record<GraphEffectToggle, keyof GraphDiagnosticsSnapshot["effectAvailability"]>> = {
|
||||
pathPulseEnabled: "pathPulse",
|
||||
pathFlowEnabled: "pathFlow",
|
||||
lensEnabled: "lens",
|
||||
edgeLabelsEnabled: "edgeLabels",
|
||||
legendEnabled: "legend",
|
||||
};
|
||||
|
||||
function renderAvailabilityText(availability: GraphEffectAvailability) {
|
||||
if (availability.available) {
|
||||
if (typeof availability.visibleSegments === "number" && typeof availability.segmentCap === "number") {
|
||||
@@ -139,15 +156,9 @@ export const explorationEffectsPlugin: GraphPlugin = {
|
||||
description={row.description}
|
||||
checked={effectsState[row.key]}
|
||||
availability={
|
||||
availability?.[
|
||||
row.key === "pathPulseEnabled"
|
||||
? "pathPulse"
|
||||
: row.key === "pathFlowEnabled"
|
||||
? "pathFlow"
|
||||
: row.key === "lensEnabled"
|
||||
? "lens"
|
||||
: "legend"
|
||||
] ?? {
|
||||
(EFFECT_AVAILABILITY_KEYS[row.key] !== undefined
|
||||
? availability?.[EFFECT_AVAILABILITY_KEYS[row.key]!]
|
||||
: undefined) ?? {
|
||||
enabled: effectsState[row.key],
|
||||
available: false,
|
||||
reason: "Waiting for graph runtime",
|
||||
|
||||
@@ -47,6 +47,11 @@ const SCENE_EFFECT_ROWS: EffectRowConfig[] = [
|
||||
label: "Contours",
|
||||
description: "Low-contrast density halos around the strongest visible anchors.",
|
||||
},
|
||||
{
|
||||
key: "edgeLabelsEnabled",
|
||||
label: "Edge Labels",
|
||||
description: "Draw the relationship type on graph edges. Off restores label-free edges on dense graphs.",
|
||||
},
|
||||
{
|
||||
key: "legendEnabled",
|
||||
label: "Regions Summary",
|
||||
@@ -83,6 +88,7 @@ const AVAILABILITY_KEYS: Record<GraphEffectToggle, keyof GraphDiagnosticsSnapsho
|
||||
communitiesEnabled: "communities",
|
||||
centralityEnabled: "centrality",
|
||||
legendEnabled: "legend",
|
||||
edgeLabelsEnabled: "edgeLabels",
|
||||
diagnosticsEnabled: "diagnostics",
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
/**
|
||||
* Guards for the temporal snapshot fetch/apply lifecycle.
|
||||
*
|
||||
* The snapshot effect previously fetched /api/temporal/snapshot with no
|
||||
* idempotency or ordering protection. Upstream churn (timeline recreation
|
||||
* while bounds settle, play ticks resetting the playhead, drag events) could
|
||||
* re-request the same `at` repeatedly, and responses could arrive after the
|
||||
* scrubber had moved on.
|
||||
*
|
||||
* The guards enforce:
|
||||
* - at most one in-flight request per scrubber position (identical `at`
|
||||
* values are deduplicated while a request is pending, breaking the
|
||||
* idle/play polling loop);
|
||||
* - successful snapshots are cached per position and re-applied when the
|
||||
* scrubber returns (play wrap-around, back-scrubbing) without a refetch;
|
||||
* - a response is applied only while the scrubber is still on its position,
|
||||
* so out-of-order responses cannot clobber a newer position's count;
|
||||
* - failed, cancelled, or superseded requests release their position so it
|
||||
* can be fetched again on the next visit;
|
||||
* - `reset()` drops all state when the underlying graph data is replaced
|
||||
* (reload/retry), because cached snapshots describe the previous graph.
|
||||
*
|
||||
* `createTemporalSnapshotGuards()` is stateful by design.
|
||||
*/
|
||||
|
||||
export interface TemporalSnapshotResponse {
|
||||
active_node_ids: string[];
|
||||
active_node_count: number;
|
||||
}
|
||||
|
||||
export interface TemporalSnapshotRequest {
|
||||
/** null when the request was deduplicated because one is already in flight. */
|
||||
seq: number | null;
|
||||
/** The snapshot previously applied for this position, when revisiting it. */
|
||||
cached: TemporalSnapshotResponse | null;
|
||||
}
|
||||
|
||||
export interface TemporalSnapshotGuards {
|
||||
/** Begin (or dedupe) a request for `atMs`; marks it as the current position. */
|
||||
begin(atMs: number): TemporalSnapshotRequest;
|
||||
/** True when the response for `atMs`/`seq` may be applied (scrubber still on `atMs`). */
|
||||
shouldApply(atMs: number, seq: number): boolean;
|
||||
/** Record a successful application and cache its snapshot for revisits. */
|
||||
apply(atMs: number, seq: number, data: TemporalSnapshotResponse): void;
|
||||
/** Release a position whose request failed, was cancelled, or was superseded. */
|
||||
finish(atMs: number, seq: number): void;
|
||||
/** Drop all state; call when the underlying graph data is replaced (reload). */
|
||||
reset(): void;
|
||||
}
|
||||
|
||||
interface SnapshotEntry {
|
||||
seq: number;
|
||||
/** null while the request is in flight (or before the first success). */
|
||||
data: TemporalSnapshotResponse | null;
|
||||
}
|
||||
|
||||
/** Upper bound on cached positions so long scrubbing sessions stay bounded. */
|
||||
const MAX_CACHED_POSITIONS = 256;
|
||||
|
||||
export function createTemporalSnapshotGuards(): TemporalSnapshotGuards {
|
||||
const entries = new Map<number, SnapshotEntry>();
|
||||
let latestRequestSeq = 0;
|
||||
let currentAtMs: number | null = null;
|
||||
|
||||
const evictOldest = () => {
|
||||
while (entries.size > MAX_CACHED_POSITIONS) {
|
||||
const oldestAtMs = entries.keys().next().value;
|
||||
if (oldestAtMs === undefined) return;
|
||||
entries.delete(oldestAtMs);
|
||||
}
|
||||
};
|
||||
|
||||
return {
|
||||
begin(atMs) {
|
||||
const existing = entries.get(atMs);
|
||||
if (existing && existing.data === null) {
|
||||
// Identical request already in flight: dedupe, but the scrubber is here now.
|
||||
currentAtMs = atMs;
|
||||
return { seq: null, cached: null };
|
||||
}
|
||||
latestRequestSeq += 1;
|
||||
const seq = latestRequestSeq;
|
||||
entries.set(atMs, { seq, data: existing?.data ?? null });
|
||||
currentAtMs = atMs;
|
||||
evictOldest();
|
||||
return { seq, cached: existing?.data ?? null };
|
||||
},
|
||||
|
||||
shouldApply(atMs, seq) {
|
||||
return atMs === currentAtMs && entries.get(atMs)?.seq === seq;
|
||||
},
|
||||
|
||||
apply(atMs, seq, data) {
|
||||
const entry = entries.get(atMs);
|
||||
if (entry && entry.seq === seq) {
|
||||
entry.data = data;
|
||||
}
|
||||
},
|
||||
|
||||
finish(atMs, seq) {
|
||||
const entry = entries.get(atMs);
|
||||
if (entry && entry.seq === seq && entry.data === null) {
|
||||
entries.delete(atMs);
|
||||
}
|
||||
},
|
||||
|
||||
reset() {
|
||||
entries.clear();
|
||||
latestRequestSeq = 0;
|
||||
currentAtMs = null;
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -103,6 +103,7 @@ export type GraphEffectToggle =
|
||||
| "communitiesEnabled"
|
||||
| "centralityEnabled"
|
||||
| "legendEnabled"
|
||||
| "edgeLabelsEnabled"
|
||||
| "diagnosticsEnabled";
|
||||
|
||||
export interface GraphEffectsState {
|
||||
@@ -113,6 +114,7 @@ export interface GraphEffectsState {
|
||||
semanticRegionsEnabled: boolean;
|
||||
contoursEnabled: boolean;
|
||||
pathfindingEnabled: boolean;
|
||||
edgeLabelsEnabled: boolean;
|
||||
communitiesEnabled: boolean;
|
||||
centralityEnabled: boolean;
|
||||
legendEnabled: boolean;
|
||||
@@ -186,6 +188,7 @@ export interface GraphDiagnosticsSnapshot {
|
||||
communities: GraphEffectAvailability;
|
||||
centrality: GraphEffectAvailability;
|
||||
legend: GraphEffectAvailability;
|
||||
edgeLabels: GraphEffectAvailability;
|
||||
diagnostics: GraphEffectAvailability;
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1061,3 +1061,198 @@ test("checkGroupedViewAvailability returns available when communities exist", ()
|
||||
assert.equal(result.reason, null);
|
||||
});
|
||||
|
||||
|
||||
// ── #1009: edge label data-path regression tests ─────────────────────────────
|
||||
|
||||
test("resolveDisplayGraph parallel-bundle preserves edgeType on aggregated edge", () => {
|
||||
addNode("a");
|
||||
addNode("b");
|
||||
batchMergeEdges([
|
||||
{ id: "e1", source: "a", target: "b", attributes: { edgeType: "causes", weight: 1, properties: {} } },
|
||||
{ id: "e2", source: "a", target: "b", attributes: { edgeType: "causes", weight: 2, properties: {} } },
|
||||
]);
|
||||
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: true });
|
||||
assert.equal(displayGraph.size, 1);
|
||||
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string; isAggregated?: boolean };
|
||||
assert.equal(attrs.isAggregated, true);
|
||||
// The aggregated representative must carry the relationship text through to
|
||||
// the edgeReducer's label assignment.
|
||||
assert.equal(typeof attrs.edgeType, "string");
|
||||
assert.ok((attrs.edgeType ?? "").length > 0, "aggregated edge must have a non-empty edgeType");
|
||||
});
|
||||
|
||||
test("resolveDisplayGraph parallel-bundle picks dominant edgeType across mixed types", () => {
|
||||
addNode("a");
|
||||
addNode("b");
|
||||
batchMergeEdges([
|
||||
{ id: "e1", source: "a", target: "b", attributes: { edgeType: "inhibits", weight: 1, properties: {} } },
|
||||
{ id: "e2", source: "a", target: "b", attributes: { edgeType: "inhibits", weight: 1, properties: {} } },
|
||||
{ id: "e3", source: "a", target: "b", attributes: { edgeType: "activates", weight: 1, properties: {} } },
|
||||
]);
|
||||
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: true });
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string; dominantEdgeType?: string };
|
||||
// "inhibits" appears twice so it must be the dominant type.
|
||||
assert.equal(attrs.edgeType, "inhibits");
|
||||
assert.equal(attrs.dominantEdgeType, "inhibits");
|
||||
});
|
||||
|
||||
test("resolveDisplayGraph grouped view community edges carry non-empty edgeType", () => {
|
||||
const left = ["g1", "g2", "g3", "g4"];
|
||||
const right = ["h1", "h2", "h3", "h4"];
|
||||
[...left, ...right].forEach((nodeId, index) => addNode(nodeId, index < left.length ? "left" : "right"));
|
||||
|
||||
let edgeIndex = 0;
|
||||
for (let i = 0; i < left.length; i += 1) {
|
||||
for (let j = 0; j < left.length; j += 1) {
|
||||
if (i !== j) {
|
||||
batchMergeEdges([{
|
||||
id: `lg-${edgeIndex++}`,
|
||||
source: left[i],
|
||||
target: left[j],
|
||||
attributes: { edgeType: "co_occurs", weight: 3, properties: {} },
|
||||
}]);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (let i = 0; i < right.length; i += 1) {
|
||||
for (let j = 0; j < right.length; j += 1) {
|
||||
if (i !== j) {
|
||||
batchMergeEdges([{
|
||||
id: `rg-${edgeIndex++}`,
|
||||
source: right[i],
|
||||
target: right[j],
|
||||
attributes: { edgeType: "co_occurs", weight: 3, properties: {} },
|
||||
}]);
|
||||
}
|
||||
}
|
||||
}
|
||||
batchMergeEdges([{ id: "bridge-g", source: "g1", target: "h1", attributes: { edgeType: "interacts_with", weight: 0.1, properties: {} } }]);
|
||||
|
||||
const { graph: displayGraph, state } = resolveDisplayGraph("", [], [], "grouped", { aggregationEnabled: true });
|
||||
assert.equal(state.groupedViewAvailable, true);
|
||||
|
||||
const communityEdges = displayGraph.edges().filter((edgeId) => {
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { bundleKind?: string };
|
||||
return attrs.bundleKind === "community";
|
||||
});
|
||||
assert.ok(communityEdges.length > 0, "expected at least one community bundle edge");
|
||||
|
||||
for (const edgeId of communityEdges) {
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string };
|
||||
assert.equal(typeof attrs.edgeType, "string");
|
||||
assert.ok((attrs.edgeType ?? "").length > 0, `community edge ${edgeId} must have a non-empty edgeType`);
|
||||
}
|
||||
});
|
||||
|
||||
test("resolveDisplayGraph raw edge preserves exact edgeType string for label rendering", () => {
|
||||
addNode("src");
|
||||
addNode("tgt");
|
||||
batchMergeEdges([{
|
||||
id: "raw-1",
|
||||
source: "src",
|
||||
target: "tgt",
|
||||
attributes: { edgeType: "works_for", weight: 1, properties: {} },
|
||||
}]);
|
||||
|
||||
// In full view without aggregation the edge passes through unchanged.
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: false });
|
||||
assert.equal(displayGraph.size, 1);
|
||||
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string };
|
||||
assert.equal(attrs.edgeType, "works_for");
|
||||
});
|
||||
|
||||
test("resolveDisplayGraph does not produce empty-string edgeType on aggregated edges when source has empty type", () => {
|
||||
addNode("a");
|
||||
addNode("b");
|
||||
// Simulate an API response where type is empty string — the aggregation
|
||||
// path must not propagate a blank label.
|
||||
batchMergeEdges([
|
||||
{ id: "e-empty-1", source: "a", target: "b", attributes: { edgeType: "", weight: 1, properties: {} } },
|
||||
{ id: "e-empty-2", source: "a", target: "b", attributes: { edgeType: "", weight: 1, properties: {} } },
|
||||
]);
|
||||
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: true });
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as {
|
||||
edgeType?: string;
|
||||
isAggregated?: boolean;
|
||||
};
|
||||
assert.equal(attrs.isAggregated, true);
|
||||
// The aggregation falls back to "related_to" when all source edgeTypes are
|
||||
// empty, so the rendered label should never be an empty string.
|
||||
assert.equal(attrs.edgeType, "related_to");
|
||||
});
|
||||
|
||||
test("resolveEdgeElementStyle hidden class produces hidden:true for suppressed edges", () => {
|
||||
// Verify the data condition the edgeReducer relies on: hidden-classified
|
||||
// edges must have hidden:true so that the label assignment sets undefined.
|
||||
const style = resolveEdgeElementStyle(
|
||||
GRAPH_THEME,
|
||||
"overview",
|
||||
"inactive",
|
||||
{
|
||||
edgeType: "causes",
|
||||
weight: 1,
|
||||
properties: {},
|
||||
edgeVariant: "line",
|
||||
visualPriority: 0.05,
|
||||
baseSize: 0.3,
|
||||
},
|
||||
"source",
|
||||
"target",
|
||||
"full",
|
||||
"inactive-edge",
|
||||
"hidden",
|
||||
);
|
||||
assert.equal(style.hidden, true);
|
||||
});
|
||||
|
||||
// ── #1009 maintainer-blocking regression: single-edge empty edgeType ─────────
|
||||
|
||||
test("resolveDisplayGraph single-edge normalizes empty-string edgeType to related_to", () => {
|
||||
addNode("a");
|
||||
addNode("b");
|
||||
// One edge only — exercises the entries.length === 1 path in aggregateDisplayGraph.
|
||||
batchMergeEdges([{
|
||||
id: "e-single-empty",
|
||||
source: "a",
|
||||
target: "b",
|
||||
attributes: { edgeType: "", weight: 1, properties: {} },
|
||||
}]);
|
||||
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: true });
|
||||
assert.equal(displayGraph.size, 1);
|
||||
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string; dominantEdgeType?: string };
|
||||
assert.equal(attrs.edgeType, "related_to",
|
||||
"single-edge path must normalize empty edgeType to the canonical fallback");
|
||||
assert.equal(attrs.dominantEdgeType, "related_to",
|
||||
"single-edge dominantEdgeType must also be normalized");
|
||||
});
|
||||
|
||||
test("resolveDisplayGraph single-edge preserves a valid non-empty edgeType unchanged", () => {
|
||||
addNode("a");
|
||||
addNode("b");
|
||||
batchMergeEdges([{
|
||||
id: "e-single-valid",
|
||||
source: "a",
|
||||
target: "b",
|
||||
attributes: { edgeType: "works_for", weight: 1, properties: {} },
|
||||
}]);
|
||||
|
||||
const { graph: displayGraph } = resolveDisplayGraph("", [], [], "full", { aggregationEnabled: true });
|
||||
assert.equal(displayGraph.size, 1);
|
||||
|
||||
const edgeId = displayGraph.edges()[0];
|
||||
const attrs = displayGraph.getEdgeAttributes(edgeId) as { edgeType?: string };
|
||||
assert.equal(attrs.edgeType, "works_for",
|
||||
"single-edge path must not alter a valid relationship type");
|
||||
});
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import React from "react";
|
||||
import { renderToString } from "react-dom/server";
|
||||
|
||||
(globalThis as any).React = React;
|
||||
|
||||
import { MarkdownContentViewer } from "../src/workspaces/GraphWorkspace/MarkdownContentViewer.tsx";
|
||||
import { isSafeUrl } from "../src/workspaces/GraphWorkspace/markdownUrlSafety.ts";
|
||||
|
||||
test("isSafeUrl permits safe http, https, and mailto URLs and relative paths", () => {
|
||||
assert.equal(isSafeUrl("https://example.com"), true);
|
||||
assert.equal(isSafeUrl("http://localhost:8000"), true);
|
||||
assert.equal(isSafeUrl("mailto:user@example.com"), true);
|
||||
assert.equal(isSafeUrl("#section-1"), true);
|
||||
assert.equal(isSafeUrl("/relative/path"), true);
|
||||
});
|
||||
|
||||
test("isSafeUrl rejects protocol-relative URLs and dangerous schemes", () => {
|
||||
// Protocol-relative URLs (must be blocked)
|
||||
assert.equal(isSafeUrl("//evil.com"), false);
|
||||
assert.equal(isSafeUrl("//localhost:8000"), false);
|
||||
assert.equal(isSafeUrl("//"), false);
|
||||
|
||||
// Dangerous schemes
|
||||
assert.equal(isSafeUrl("javascript:alert('xss')"), false);
|
||||
assert.equal(isSafeUrl("JAVASCRIPT:alert(1)"), false);
|
||||
assert.equal(isSafeUrl("data:text/html;base64,PHNjcmlwdD4="), false);
|
||||
assert.equal(isSafeUrl("vbscript:MsgBox(1)"), false);
|
||||
assert.equal(isSafeUrl(""), false);
|
||||
assert.equal(isSafeUrl(undefined), false);
|
||||
});
|
||||
|
||||
// ─── C URL contract: whitespace-only strings ────────────────────────────────
|
||||
// The CommonMark parser normalises whitespace-only link destinations to "" so
|
||||
// these values are unreachable through normal markdown rendering. However, the
|
||||
// function is exported and its direct-call contract must be correct.
|
||||
test("isSafeUrl rejects whitespace-only strings (contract correctness)", () => {
|
||||
assert.equal(isSafeUrl(" "), false, "single space must be rejected");
|
||||
assert.equal(isSafeUrl("\t"), false, "tab must be rejected");
|
||||
assert.equal(isSafeUrl("\n"), false, "newline must be rejected");
|
||||
assert.equal(isSafeUrl(" "), false, "multiple spaces must be rejected");
|
||||
assert.equal(isSafeUrl(" \t\n "), false, "mixed whitespace must be rejected");
|
||||
});
|
||||
|
||||
test("renders Preview mode with formatted Markdown elements and tabs", () => {
|
||||
const markdown = `# Main Title\n\n**Bold Statement**\n\n* Item A\n* Item B`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content: markdown, defaultMode: "preview" }));
|
||||
|
||||
// Tab buttons are present
|
||||
assert.equal(html.includes("Preview"), true);
|
||||
assert.equal(html.includes("Source"), true);
|
||||
assert.equal(html.includes("Copy"), true);
|
||||
|
||||
// Formatted preview elements
|
||||
assert.equal(html.includes("Main Title"), true);
|
||||
assert.equal(html.includes("Bold Statement"), true);
|
||||
assert.equal(html.includes("<strong>Bold Statement</strong>"), true);
|
||||
assert.equal(html.includes("Item A"), true);
|
||||
assert.equal(html.includes("Item B"), true);
|
||||
});
|
||||
|
||||
test("renders Source mode with exact unmodified text inside pre/code", () => {
|
||||
const markdown = `# Title 🚀\n\n * Indented item\n\n\`\`\`python\ndef test():\n return "α + β"\n\`\`\``;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content: markdown, defaultMode: "source" }));
|
||||
|
||||
assert.equal(html.includes("<pre"), true);
|
||||
assert.equal(html.includes("<code"), true);
|
||||
assert.equal(html.includes("# Title 🚀"), true);
|
||||
assert.equal(html.includes(" * Indented item"), true);
|
||||
assert.equal(html.includes('return "α + β"'), true);
|
||||
});
|
||||
|
||||
test("renders raw HTML safely as escaped text without executing elements", () => {
|
||||
const dangerousHtml = `<script>alert("XSS")</script><iframe src="https://evil.com"></iframe>`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content: dangerousHtml, defaultMode: "preview" }));
|
||||
|
||||
// Script and iframe tags must NOT be rendered as active DOM tags
|
||||
assert.equal(html.includes("<script>"), false);
|
||||
assert.equal(html.includes("<iframe"), false);
|
||||
// Content is escaped as text
|
||||
assert.equal(html.includes("<script>"), true);
|
||||
});
|
||||
|
||||
// ─── C-1: HAST node prop must not reach the DOM ─────────────────────────────
|
||||
// react-markdown passes a HAST `node` (Element) object to custom component
|
||||
// overrides. Before this fix, ...props spread caused React 19 to serialise it
|
||||
// as node="[object Object]" on every <a> and <code> element.
|
||||
test("rendered links do not expose the HAST node object as a DOM attribute", () => {
|
||||
const content = `[Example](https://example.com)\n\nInline \`code\` here.`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// The rendered HTML must not contain the serialised HAST object
|
||||
assert.equal(html.includes("node="), false, "node= attribute must not appear in rendered HTML");
|
||||
assert.equal(html.includes("[object Object]"), false, "serialised HAST object must not appear in rendered HTML");
|
||||
|
||||
// The link must still render correctly with the right href
|
||||
assert.equal(html.includes('href="https://example.com"'), true, "href must be present");
|
||||
});
|
||||
|
||||
// ─── C-2: Fragment links must not open in a new tab ─────────────────────────
|
||||
// Links to in-document anchors such as #section or GFM footnote backlinks like
|
||||
// #user-content-fn-1 must stay in the current document. Only external links
|
||||
// use target="_blank".
|
||||
test("fragment links render in the current document without target blank", () => {
|
||||
const content = `[Jump to section](#introduction)\n\n[External](https://example.com)`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// Fragment link must have the href
|
||||
assert.equal(html.includes('href="#introduction"'), true, "fragment href must be present");
|
||||
|
||||
// Confirm no target=_blank attribute appears anywhere near the fragment link.
|
||||
// We check that the output contains a fragment href WITHOUT target="_blank"
|
||||
// by verifying the two strings are not both present (the external link has
|
||||
// target blank; the fragment link must not).
|
||||
const fragmentLinkIdx = html.indexOf('href="#introduction"');
|
||||
assert.notEqual(fragmentLinkIdx, -1, "fragment link must be rendered");
|
||||
// Inspect the 80 chars around the fragment href — should not contain target
|
||||
const fragmentContext = html.slice(Math.max(0, fragmentLinkIdx - 10), fragmentLinkIdx + 90);
|
||||
assert.equal(fragmentContext.includes('target="_blank"'), false, "fragment link must not have target=_blank");
|
||||
|
||||
// External link must still have target blank
|
||||
assert.equal(html.includes('href="https://example.com"'), true, "external href must be present");
|
||||
assert.equal(html.includes('target="_blank"'), true, "external link must have target=_blank");
|
||||
assert.equal(html.includes('rel="noopener noreferrer"'), true, "external link must have rel");
|
||||
});
|
||||
|
||||
test("GFM footnote backlinks render without target blank", () => {
|
||||
// GFM footnote syntax: footnote ref in text + definition below
|
||||
const content = `See the note[^1] for more.\n\n[^1]: This is the footnote text.`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// The footnote reference link (#user-content-fn-1) and backlink
|
||||
// (#user-content-fnref-1) are fragment links and must not open in a new tab.
|
||||
// We verify no fragment href is paired with target=_blank.
|
||||
// Extract all href="#..." occurrences and confirm none is adjacent to target=_blank.
|
||||
const anchorMatches = [...html.matchAll(/href="#[^"]*"/g)];
|
||||
assert.ok(anchorMatches.length > 0, "GFM footnotes must produce fragment links");
|
||||
for (const match of anchorMatches) {
|
||||
const start = match.index ?? 0;
|
||||
const context = html.slice(Math.max(0, start - 10), start + 120);
|
||||
assert.equal(
|
||||
context.includes('target="_blank"'),
|
||||
false,
|
||||
`fragment link ${match[0]} must not have target=_blank`,
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
// ─── C-1-R: GFM footnote attributes must be preserved (regression test) ─────
|
||||
// The C-1 fix (removing the HAST `node` prop) must NOT silently drop other
|
||||
// legitimate HAST attributes. remark-gfm generates the following on footnote
|
||||
// links that are required for correct in-page navigation and accessibility:
|
||||
//
|
||||
// Footnote reference anchor:
|
||||
// id="user-content-fnref-1" ← backlink target
|
||||
// data-footnote-ref="true"
|
||||
// aria-describedby="footnote-label"
|
||||
//
|
||||
// Footnote back-link anchor:
|
||||
// data-footnote-backref=""
|
||||
// aria-label="Back to reference 1" ← screen-reader label
|
||||
// class="data-footnote-backref"
|
||||
//
|
||||
// If these are absent, clicking the ↩ back-link cannot scroll back to the
|
||||
// in-text reference, and screen readers cannot announce the backlink purpose.
|
||||
test("GFM footnote links preserve generated id, aria, and class attributes", () => {
|
||||
const content = `See the note[^1] for more.\n\n[^1]: This is the footnote text.`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// The HAST `node` object must not appear serialised as a DOM attribute.
|
||||
assert.equal(html.includes("node="), false, "node= attribute must not appear in HTML");
|
||||
assert.equal(html.includes("[object Object]"), false, "serialised HAST object must not appear in HTML");
|
||||
|
||||
// Footnote reference anchor must retain its id so the backlink can navigate to it.
|
||||
assert.equal(
|
||||
html.includes('id="user-content-fnref-1"'),
|
||||
true,
|
||||
"footnote reference anchor must retain id for back-navigation",
|
||||
);
|
||||
|
||||
// Footnote backlink must retain its aria-label for screen-reader accessibility.
|
||||
assert.equal(
|
||||
html.includes('aria-label="Back to reference 1"'),
|
||||
true,
|
||||
"footnote backlink must retain aria-label for accessibility",
|
||||
);
|
||||
|
||||
// Footnote backlink must retain its class attribute.
|
||||
assert.equal(
|
||||
html.includes('class="data-footnote-backref"'),
|
||||
true,
|
||||
"footnote backlink must retain class attribute",
|
||||
);
|
||||
});
|
||||
|
||||
test("renders safe links as <a> with target blank and unclickable span for unsafe links", () => {
|
||||
const content = `[Safe Link](https://getsemantica.ai)\n\n[Unsafe Scheme](javascript:alert(1))\n\n[Protocol Relative](//evil.com)`;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// Safe link renders as <a> with security attributes
|
||||
assert.equal(html.includes('href="https://getsemantica.ai"'), true);
|
||||
assert.equal(html.includes('target="_blank"'), true);
|
||||
assert.equal(html.includes('rel="noopener noreferrer"'), true);
|
||||
|
||||
// Unsafe links do NOT render as <a> tags
|
||||
assert.equal(html.includes('href="javascript:alert(1)"'), false);
|
||||
assert.equal(html.includes('href="//evil.com"'), false);
|
||||
assert.equal(html.includes("Unsafe Scheme"), true);
|
||||
assert.equal(html.includes("Protocol Relative"), true);
|
||||
});
|
||||
|
||||
test("renders remote images as safe placeholder badges instead of <img> tags", () => {
|
||||
const content = ``;
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content, defaultMode: "preview" }));
|
||||
|
||||
// No <img> tag rendered
|
||||
assert.equal(html.includes("<img"), false);
|
||||
// Image placeholder badge rendered
|
||||
assert.equal(html.includes("Image:"), true);
|
||||
assert.equal(html.includes("System Diagram"), true);
|
||||
});
|
||||
|
||||
test("renders clear empty-state message when content is empty or null", () => {
|
||||
const emptyHtml = renderToString(React.createElement(MarkdownContentViewer, { content: "" }));
|
||||
assert.equal(emptyHtml.includes("No content available for this node."), true);
|
||||
|
||||
const nullHtml = renderToString(React.createElement(MarkdownContentViewer, { content: null }));
|
||||
assert.equal(nullHtml.includes("No content available for this node."), true);
|
||||
});
|
||||
|
||||
test("renders plain text cleanly without requiring Markdown formatting", () => {
|
||||
const plainText = "Plain entity summary text without markdown formatting.";
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content: plainText, defaultMode: "preview" }));
|
||||
|
||||
assert.equal(html.includes(plainText), true);
|
||||
});
|
||||
|
||||
test("handles very large Markdown content without failure", () => {
|
||||
const largeContent = `# Large Knowledge Node\n\n` + "Structured observation paragraph. ".repeat(400);
|
||||
assert.equal(largeContent.length > 10000, true);
|
||||
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, { content: largeContent, defaultMode: "preview" }));
|
||||
assert.equal(html.includes("Large Knowledge Node"), true);
|
||||
});
|
||||
|
||||
// ─── H-2: Stale copied state lifecycle (SSR-compatible portion) ─────────────
|
||||
// Full state-transition testing (Node A → copy → Node B) requires an interactive
|
||||
// framework. The lifecycle correctness is guaranteed by the render-phase
|
||||
// previous-prop synchronisation pattern: a `copiedForContent` state value tracks
|
||||
// the content for which the copied indicator was set; when `content` changes, the
|
||||
// mismatch is detected during render and `copied` is reset to false in the same
|
||||
// React batch, before the new node's UI is painted. What we CAN verify in SSR
|
||||
// is that the initial render for any content value shows the Copy button (not the
|
||||
// Copied indicator), which confirms the initial state is always clean.
|
||||
test("copy button always starts in un-copied state on initial render", () => {
|
||||
const html = renderToString(React.createElement(MarkdownContentViewer, {
|
||||
content: "# Some Node\n\nDescription text.",
|
||||
defaultMode: "preview",
|
||||
}));
|
||||
|
||||
// Initial render must show 'Copy', never 'Copied'
|
||||
assert.equal(html.includes("Copy"), true, "Copy button must be present on initial render");
|
||||
assert.equal(html.includes("Copied"), false, "Copied indicator must NOT be present on initial render");
|
||||
});
|
||||
@@ -0,0 +1,150 @@
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
|
||||
import { createTemporalSnapshotGuards } from "../src/workspaces/GraphWorkspace/temporalSnapshotGuards.ts";
|
||||
|
||||
const POSITION_1 = new Date("2023-07-02T00:00:00Z").getTime();
|
||||
const POSITION_2 = new Date("2024-01-02T00:00:00Z").getTime();
|
||||
const POSITION_3 = new Date("2024-07-02T00:00:00Z").getTime();
|
||||
|
||||
const SNAPSHOT = { active_node_ids: ["n1", "n2"], active_node_count: 2 };
|
||||
|
||||
// ── begin: one request per scrubber position ─────────────────────────────────
|
||||
|
||||
test("begin: a new position returns a fresh request sequence", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
assert.deepEqual(guards.begin(POSITION_1), { seq: 1, cached: null });
|
||||
});
|
||||
|
||||
test("begin: an identical in-flight request is deduplicated (no duplicate fetch)", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
guards.begin(POSITION_1);
|
||||
assert.deepEqual(guards.begin(POSITION_1), { seq: null, cached: null });
|
||||
});
|
||||
|
||||
test("begin: distinct positions request independently", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
assert.equal(guards.begin(POSITION_1).seq, 1);
|
||||
assert.equal(guards.begin(POSITION_2).seq, 2);
|
||||
});
|
||||
|
||||
test("begin: revisiting an applied position returns its cached snapshot", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.apply(POSITION_1, seq, SNAPSHOT);
|
||||
const revisit = guards.begin(POSITION_1);
|
||||
assert.equal(revisit.seq, 2);
|
||||
assert.deepEqual(revisit.cached, SNAPSHOT);
|
||||
});
|
||||
|
||||
test("begin: a failed position (finished) can be requested again", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.finish(POSITION_1, seq);
|
||||
const retry = guards.begin(POSITION_1);
|
||||
assert.equal(retry.seq, 2);
|
||||
assert.equal(retry.cached, null);
|
||||
});
|
||||
|
||||
test("finish: does not clear a position whose snapshot was already applied", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.apply(POSITION_1, seq, SNAPSHOT);
|
||||
guards.finish(POSITION_1, seq);
|
||||
assert.deepEqual(guards.begin(POSITION_1).cached, SNAPSHOT);
|
||||
});
|
||||
|
||||
test("finish: a stale sequence cannot release a newer request's position", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const first = guards.begin(POSITION_1);
|
||||
guards.finish(POSITION_1, first.seq);
|
||||
guards.begin(POSITION_1); // seq 2, in flight again
|
||||
guards.finish(POSITION_1, first.seq); // stale seq: must not release seq 2
|
||||
assert.deepEqual(guards.begin(POSITION_1), { seq: null, cached: null });
|
||||
});
|
||||
|
||||
// ── shouldApply: applied only while the scrubber is on that position ─────────
|
||||
|
||||
test("shouldApply: the current position's response is applied", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
assert.equal(guards.shouldApply(POSITION_1, seq), true);
|
||||
});
|
||||
|
||||
test("shouldApply: a response for a position the scrubber left is discarded", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq: seq1 } = guards.begin(POSITION_1);
|
||||
guards.begin(POSITION_2);
|
||||
assert.equal(guards.shouldApply(POSITION_1, seq1), false);
|
||||
assert.equal(guards.shouldApply(POSITION_2, 2), true);
|
||||
});
|
||||
|
||||
test("shouldApply: a late response for the position the scrubber returned to is applied", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq: seq1 } = guards.begin(POSITION_1);
|
||||
const { seq: seq2 } = guards.begin(POSITION_2);
|
||||
guards.begin(POSITION_1); // back to 1: deduplicated, no new request
|
||||
assert.equal(guards.shouldApply(POSITION_1, seq1), true);
|
||||
assert.equal(guards.shouldApply(POSITION_2, seq2), false);
|
||||
});
|
||||
|
||||
test("shouldApply: an unknown sequence is discarded", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
guards.begin(POSITION_1);
|
||||
assert.equal(guards.shouldApply(POSITION_1, 99), false);
|
||||
});
|
||||
|
||||
test("shouldApply: after a reset no pre-reset response applies", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.reset();
|
||||
assert.equal(guards.shouldApply(POSITION_1, seq), false);
|
||||
});
|
||||
|
||||
// ── apply: caching for revisits ─────────────────────────────────────────────
|
||||
|
||||
test("apply: stores the snapshot so a revisit re-applies it without a request", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.apply(POSITION_1, seq, SNAPSHOT);
|
||||
guards.begin(POSITION_2);
|
||||
assert.deepEqual(guards.begin(POSITION_1).cached, SNAPSHOT);
|
||||
});
|
||||
|
||||
test("apply: play wrap-around re-applies the wrapped-to position's snapshot", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.apply(POSITION_1, seq, SNAPSHOT);
|
||||
guards.begin(POSITION_2);
|
||||
guards.begin(POSITION_3);
|
||||
const wrap = guards.begin(POSITION_1);
|
||||
assert.deepEqual(wrap.cached, SNAPSHOT);
|
||||
assert.equal(guards.shouldApply(POSITION_1, wrap.seq), true);
|
||||
});
|
||||
|
||||
// ── reset: graph reload ─────────────────────────────────────────────────────
|
||||
|
||||
test("reset: clears requested and cached state so positions refetch", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const { seq } = guards.begin(POSITION_1);
|
||||
guards.apply(POSITION_1, seq, SNAPSHOT);
|
||||
guards.reset();
|
||||
const fresh = guards.begin(POSITION_1);
|
||||
assert.equal(fresh.seq, 1);
|
||||
assert.equal(fresh.cached, null);
|
||||
});
|
||||
|
||||
// ── cache bound ─────────────────────────────────────────────────────────────
|
||||
|
||||
test("cache: oldest positions are evicted when the cache is full", () => {
|
||||
const guards = createTemporalSnapshotGuards();
|
||||
const count = 300;
|
||||
for (let i = 0; i < count; i++) {
|
||||
const { seq } = guards.begin(POSITION_1 + i * 1000);
|
||||
guards.apply(POSITION_1 + i * 1000, seq, SNAPSHOT);
|
||||
}
|
||||
const oldest = guards.begin(POSITION_1);
|
||||
assert.equal(oldest.cached, null); // evicted: must refetch on revisit
|
||||
const newest = guards.begin(POSITION_1 + (count - 1) * 1000);
|
||||
assert.deepEqual(newest.cached, SNAPSHOT); // still cached
|
||||
});
|
||||
@@ -1,7 +1,7 @@
|
||||
"""
|
||||
Semantica Framework Integrations
|
||||
|
||||
Optional integration packages for agentic frameworks (Google ADK, Claude Agent SDK, Agno, etc.).
|
||||
Optional integration packages for agentic frameworks (Google ADK, Claude Agent SDK, Agno, CrewAI, LangChain, etc.).
|
||||
Each integration is self-contained, independently installable via extras_require, and maintains
|
||||
zero impact on core Semantica - keeping the semantic layer lean while maximizing ecosystem reach.
|
||||
"""
|
||||
|
||||
@@ -277,25 +277,22 @@ class AgnoKnowledgeGraph(_KnowledgeBase): # type: ignore[misc]
|
||||
def load_urls(self, urls: List[str]) -> None:
|
||||
"""Fetch each URL and ingest the response body.
|
||||
|
||||
Only ``http`` and ``https`` schemes are permitted to prevent SSRF.
|
||||
Uses the shared SSRF guard so that ``http`` and ``https`` are the only
|
||||
permitted schemes, private/loopback/link-local/cloud-metadata addresses
|
||||
are blocked by default, DNS resolution is validated, and every redirect
|
||||
hop is re-checked before being followed.
|
||||
"""
|
||||
import urllib.request
|
||||
from urllib.parse import urlparse
|
||||
from semantica.ingest.ssrf import request_with_ssrf_guard
|
||||
from semantica.utils.exceptions import ValidationError
|
||||
|
||||
for url in urls:
|
||||
parsed = urlparse(url)
|
||||
if parsed.scheme not in ("http", "https"):
|
||||
logger.warning(
|
||||
"Skipping URL with disallowed scheme '%s': %s",
|
||||
parsed.scheme,
|
||||
url,
|
||||
)
|
||||
continue
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=10) as resp: # noqa: S310
|
||||
text = resp.read().decode("utf-8", errors="replace")
|
||||
response = request_with_ssrf_guard("GET", url, timeout=10)
|
||||
text = response.text
|
||||
self._ingest_text(text, source=url)
|
||||
logger.info("Loaded URL: %s", url)
|
||||
except ValidationError as exc:
|
||||
logger.warning("Skipping URL (SSRF check failed) %s: %s", url, exc)
|
||||
except Exception as exc:
|
||||
logger.warning("Failed to fetch %s: %s", url, exc)
|
||||
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
# Semantica × LangChain
|
||||
|
||||
Drop Semantica into existing LangChain / LangGraph pipelines: GraphRAG-style
|
||||
retrieval, a `VectorStore` adapter, and agent tools.
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
pip install semantica[langchain]
|
||||
# or just the core adapter dependency:
|
||||
pip install langchain-core
|
||||
```
|
||||
|
||||
## Retriever (GraphRAG)
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaRetriever
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.vector_store import HybridSearch
|
||||
|
||||
graph = ContextGraph()
|
||||
hybrid = HybridSearch()
|
||||
|
||||
retriever = SemanticaRetriever(graph=graph, hybrid=hybrid, hops=2, top_k=10)
|
||||
|
||||
# Use with any LangChain chain that accepts a retriever:
|
||||
from langchain.chains import RetrievalQA
|
||||
|
||||
qa = RetrievalQA.from_chain_type(llm=llm, retriever=retriever)
|
||||
```
|
||||
|
||||
Hybrid search seeds retrieval; then graph edges are walked `hops` steps so
|
||||
results go beyond flat vector similarity.
|
||||
|
||||
## VectorStore
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaVectorStore
|
||||
|
||||
store = SemanticaVectorStore(hybrid=hybrid)
|
||||
store.add_texts(["document one", "document two"], metadatas=[{"source": "a"}, {"source": "b"}])
|
||||
docs = store.similarity_search("document", k=2)
|
||||
docs, scores = store.similarity_search_with_score("document", k=2)
|
||||
```
|
||||
|
||||
## Agent tools (LangGraph / tool-calling agents)
|
||||
|
||||
```python
|
||||
from integrations.langchain import SemanticaKGTool, SemanticaDecisionTool
|
||||
from langgraph.prebuilt import create_react_agent
|
||||
|
||||
tools = [
|
||||
SemanticaKGTool(graph),
|
||||
SemanticaDecisionTool(graph),
|
||||
]
|
||||
agent = create_react_agent(model, tools)
|
||||
```
|
||||
|
||||
- `semantica_query_graph` — query the shared context graph (keyword / NL)
|
||||
- `semantica_query_decisions` — search the recorded decision log
|
||||
|
||||
## Compatibility
|
||||
|
||||
- Requires `langchain-core >= 0.3`.
|
||||
- All classes degrade gracefully when `langchain-core` is absent: they remain
|
||||
importable (carrying the full Semantica API), and `build()` returns `None`,
|
||||
so agents can branch on `LANGCHAIN_AVAILABLE`.
|
||||
@@ -0,0 +1,48 @@
|
||||
"""
|
||||
Semantica × LangChain Integration
|
||||
=================================
|
||||
|
||||
First-class integration between the Semantica semantic intelligence stack and
|
||||
the `LangChain <https://github.com/langchain-ai/langchain>`_ / LangGraph
|
||||
ecosystem.
|
||||
|
||||
Public surface
|
||||
--------------
|
||||
SemanticaRetriever — ``BaseRetriever`` with multi-hop GraphRAG (walks graph
|
||||
edges from hybrid-search hits)
|
||||
SemanticaVectorStore — ``VectorStore`` adapter over Semantica's hybrid search
|
||||
(drop-in for RetrievalQA / LCEL chains)
|
||||
SemanticaKGTool — ``BaseTool`` for querying the context graph
|
||||
SemanticaDecisionTool — ``BaseTool`` exposing the recorded decision log
|
||||
|
||||
Quick start
|
||||
-----------
|
||||
pip install semantica[langchain]
|
||||
|
||||
>>> from integrations.langchain import (
|
||||
... SemanticaRetriever,
|
||||
... SemanticaVectorStore,
|
||||
... SemanticaKGTool,
|
||||
... SemanticaDecisionTool,
|
||||
... )
|
||||
|
||||
Compatibility
|
||||
-------------
|
||||
Requires ``langchain-core >= 0.3``. All classes degrade gracefully when
|
||||
``langchain-core`` is not installed — they are still importable and carry the
|
||||
full Semantica API, but cannot be bound to LangChain chains/agents.
|
||||
"""
|
||||
|
||||
from .retriever import LANGCHAIN_AVAILABLE, SemanticaRetriever
|
||||
from .tools import SemanticaDecisionTool, SemanticaKGTool
|
||||
from .vectorstore import SemanticaVectorStore
|
||||
|
||||
__all__ = [
|
||||
"SemanticaRetriever",
|
||||
"SemanticaVectorStore",
|
||||
"SemanticaKGTool",
|
||||
"SemanticaDecisionTool",
|
||||
"LANGCHAIN_AVAILABLE",
|
||||
]
|
||||
|
||||
__version__ = "0.1.0"
|
||||
@@ -0,0 +1,216 @@
|
||||
"""
|
||||
SemanticaRetriever — LangChain ``BaseRetriever`` with multi-hop GraphRAG.
|
||||
|
||||
Hybrid search seeds the retrieval, then graph edges are walked for ``hops``
|
||||
steps so results go beyond flat vector similarity.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from semantica.utils.logging import get_logger
|
||||
|
||||
logger = get_logger(__name__)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional: LangChain core
|
||||
# ---------------------------------------------------------------------------
|
||||
LANGCHAIN_AVAILABLE = False
|
||||
LANGCHAIN_IMPORT_ERROR: Optional[str] = None
|
||||
|
||||
_BaseRetriever: Any = object
|
||||
_Document: Any = None
|
||||
|
||||
|
||||
def _get_document(**kwargs: Any) -> Any:
|
||||
"""Instantiate a langchain Document lazily (keeps the import optional)."""
|
||||
if _Document is None: # pragma: no cover - exercised only with langchain
|
||||
raise RuntimeError(LANGCHAIN_IMPORT_ERROR or "langchain-core not installed")
|
||||
return _Document(**kwargs)
|
||||
|
||||
|
||||
try:
|
||||
from langchain_core.documents import Document as _Document # type: ignore
|
||||
from langchain_core.retrievers import (
|
||||
BaseRetriever as _BaseRetriever, # type: ignore
|
||||
)
|
||||
|
||||
LANGCHAIN_AVAILABLE = True
|
||||
except ImportError: # pragma: no cover - exercised only without langchain
|
||||
LANGCHAIN_IMPORT_ERROR = (
|
||||
"langchain-core is not installed. Install with: pip install langchain-core"
|
||||
)
|
||||
logger.debug(LANGCHAIN_IMPORT_ERROR)
|
||||
|
||||
|
||||
def _hit_layers(hit: Dict[str, Any]) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
||||
"""Nested HybridSearch metadata and ContextGraph.query node, if present."""
|
||||
metadata = hit.get("metadata") if isinstance(hit.get("metadata"), dict) else {}
|
||||
node = hit.get("node") if isinstance(hit.get("node"), dict) else {}
|
||||
return metadata, node
|
||||
|
||||
|
||||
def _hit_id(hit: Dict[str, Any]) -> Optional[str]:
|
||||
"""Graph node id, preferring metadata over a HybridSearch vector id."""
|
||||
metadata, node = _hit_layers(hit)
|
||||
return (
|
||||
hit.get("node_id")
|
||||
or metadata.get("node_id")
|
||||
or node.get("id")
|
||||
or node.get("node_id")
|
||||
or hit.get("id")
|
||||
)
|
||||
|
||||
|
||||
def _hit_content(hit: Dict[str, Any], fallback: str = "") -> str:
|
||||
metadata, node = _hit_layers(hit)
|
||||
props = node.get("properties") if isinstance(node.get("properties"), dict) else {}
|
||||
return (
|
||||
hit.get("content")
|
||||
or hit.get("text")
|
||||
or metadata.get("content")
|
||||
or metadata.get("text")
|
||||
or props.get("content")
|
||||
or fallback
|
||||
)
|
||||
|
||||
|
||||
def _hit_type(hit: Dict[str, Any]) -> str:
|
||||
metadata, node = _hit_layers(hit)
|
||||
return (
|
||||
hit.get("node_type")
|
||||
or hit.get("type")
|
||||
or metadata.get("node_type")
|
||||
or metadata.get("type")
|
||||
or node.get("type")
|
||||
or node.get("node_type")
|
||||
or "node"
|
||||
)
|
||||
|
||||
|
||||
def _hit_score(hit: Dict[str, Any], default: float = 1.0) -> float:
|
||||
return float(hit.get("score") if hit.get("score") is not None else hit.get("distance") or default)
|
||||
|
||||
|
||||
class SemanticaRetriever(_BaseRetriever): # type: ignore[misc]
|
||||
"""GraphRAG-style retriever over a Semantica ``ContextGraph``.
|
||||
|
||||
Args:
|
||||
graph: A semantica.context.ContextGraph instance.
|
||||
hybrid: A semantica.vector_store.HybridSearch instance used to seed
|
||||
retrieval. If omitted, a best-effort keyword search on the graph
|
||||
is used.
|
||||
hops: Number of graph-edge expansion hops (default 2).
|
||||
top_k: Number of seed hits (default 10).
|
||||
"""
|
||||
|
||||
graph: Any
|
||||
hybrid: Any = None
|
||||
hops: int = 2
|
||||
top_k: int = 10
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
graph: Any,
|
||||
hybrid: Any = None,
|
||||
hops: int = 2,
|
||||
top_k: int = 10,
|
||||
**kwargs: Any,
|
||||
) -> None:
|
||||
"""Explicit init so the retriever works with and without langchain."""
|
||||
if LANGCHAIN_AVAILABLE:
|
||||
# BaseRetriever is a Pydantic model: pass the declared fields
|
||||
# through so validation succeeds.
|
||||
super().__init__(
|
||||
graph=graph,
|
||||
hybrid=hybrid,
|
||||
hops=hops,
|
||||
top_k=top_k,
|
||||
**kwargs,
|
||||
)
|
||||
else:
|
||||
# Without langchain-core, BaseRetriever is a plain object
|
||||
super().__init__() # type: ignore[call-arg]
|
||||
self.graph = graph
|
||||
self.hybrid = hybrid
|
||||
self.hops = hops
|
||||
self.top_k = top_k
|
||||
|
||||
def _get_relevant_documents(self, query: str, **kwargs: Any) -> List[Any]:
|
||||
"""LangChain BaseRetriever entry point."""
|
||||
seed = self._seed_results(query)
|
||||
if not seed:
|
||||
return []
|
||||
|
||||
# Expand each seed node through the graph
|
||||
expanded: Dict[str, Dict[str, Any]] = {}
|
||||
for hit in seed:
|
||||
node_id = _hit_id(hit)
|
||||
if not node_id:
|
||||
continue
|
||||
metadata, _ = _hit_layers(hit)
|
||||
expanded[node_id] = {
|
||||
"content": _hit_content(hit, fallback=str(node_id)),
|
||||
"node_type": _hit_type(hit),
|
||||
"score": _hit_score(hit),
|
||||
"metadata": metadata,
|
||||
}
|
||||
try:
|
||||
neighbors = self.graph.get_neighbors(node_id, hops=self.hops)
|
||||
for neighbor in neighbors:
|
||||
nid = neighbor.get("node_id") or neighbor.get("id")
|
||||
if nid and nid not in expanded:
|
||||
expanded[nid] = {
|
||||
"content": neighbor.get("content")
|
||||
or neighbor.get("text")
|
||||
or neighbor.get("name")
|
||||
or str(nid),
|
||||
"node_type": neighbor.get("node_type")
|
||||
or neighbor.get("type")
|
||||
or "node",
|
||||
"score": float(neighbor.get("weight") or 0.5),
|
||||
"metadata": {},
|
||||
}
|
||||
except Exception as exc: # graph expansion is best-effort
|
||||
logger.debug("graph expansion failed for %s: %s", node_id, exc)
|
||||
|
||||
# Order: seed hits first (they have real scores), then neighbors.
|
||||
# Keep a deterministic id->payload list (sets are unordered — see Qodo).
|
||||
ordered_pairs: List[tuple] = []
|
||||
seen_ids = set()
|
||||
for hit in seed:
|
||||
nid = _hit_id(hit)
|
||||
if nid and nid in expanded and nid not in seen_ids:
|
||||
ordered_pairs.append((nid, expanded[nid]))
|
||||
seen_ids.add(nid)
|
||||
for nid, item in expanded.items():
|
||||
if nid not in seen_ids:
|
||||
ordered_pairs.append((nid, item))
|
||||
seen_ids.add(nid)
|
||||
|
||||
return [
|
||||
_get_document(
|
||||
page_content=item["content"],
|
||||
metadata={
|
||||
**item["metadata"],
|
||||
"node_id": nid,
|
||||
"node_type": item["node_type"],
|
||||
"score": item["score"],
|
||||
},
|
||||
)
|
||||
for nid, item in ordered_pairs
|
||||
]
|
||||
|
||||
def _seed_results(self, query: str) -> List[Dict[str, Any]]:
|
||||
"""Get seed results from hybrid search or a graph keyword scan."""
|
||||
if self.hybrid is not None:
|
||||
try:
|
||||
return self.hybrid.search(query, k=self.top_k)
|
||||
except Exception as exc:
|
||||
logger.debug("hybrid search failed, falling back: %s", exc)
|
||||
# Best-effort keyword scan over graph nodes (ContextGraph.query)
|
||||
try:
|
||||
return self.graph.query(query, limit=self.top_k)
|
||||
except Exception:
|
||||
return []
|
||||
@@ -0,0 +1,133 @@
|
||||
"""
|
||||
SemanticaKGTool / SemanticaDecisionTool — LangChain ``BaseTool`` adapters
|
||||
for LangChain / LangGraph agents.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Optional, Type
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
from semantica.utils.logging import get_logger
|
||||
|
||||
logger = get_logger(__name__)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional: LangChain core
|
||||
# ---------------------------------------------------------------------------
|
||||
LANGCHAIN_AVAILABLE = False
|
||||
LANGCHAIN_IMPORT_ERROR: Optional[str] = None
|
||||
|
||||
_BaseTool: Any = object
|
||||
|
||||
|
||||
try:
|
||||
from langchain_core.tools import BaseTool as _BaseTool # type: ignore
|
||||
|
||||
LANGCHAIN_AVAILABLE = True
|
||||
except ImportError: # pragma: no cover
|
||||
LANGCHAIN_IMPORT_ERROR = (
|
||||
"langchain-core is not installed. Install with: pip install langchain-core"
|
||||
)
|
||||
logger.debug(LANGCHAIN_IMPORT_ERROR)
|
||||
|
||||
|
||||
def _json(payload: Any) -> str:
|
||||
return json.dumps(payload, default=str, ensure_ascii=False)
|
||||
|
||||
|
||||
class QueryGraphInput(BaseModel):
|
||||
query: str = Field(..., description="Natural-language or keyword graph query")
|
||||
limit: int = Field(10, description="Maximum matching nodes to return")
|
||||
|
||||
|
||||
class QueryDecisionsInput(BaseModel):
|
||||
category: str = Field(
|
||||
"",
|
||||
description="Keyword to search recorded decisions; empty returns insights",
|
||||
)
|
||||
limit: int = Field(10, description="Maximum results when searching by keyword")
|
||||
|
||||
|
||||
class SemanticaKGTool(_BaseTool): # type: ignore[misc]
|
||||
"""LangChain tool for querying a Semantica ``ContextGraph``.
|
||||
|
||||
Args:
|
||||
graph: A semantica.context.ContextGraph instance.
|
||||
|
||||
Example:
|
||||
>>> tool = SemanticaKGTool(graph)
|
||||
>>> agent = create_react_agent(model, tools=[tool])
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||
|
||||
name: str = "semantica_query_graph"
|
||||
description: str = (
|
||||
"Query Semantica's shared context graph with a natural-language "
|
||||
"keyword query. Returns matching entities and relationships."
|
||||
)
|
||||
args_schema: Type[BaseModel] = QueryGraphInput
|
||||
graph: Any = None
|
||||
|
||||
def __init__(self, graph: Any = None, **kwargs: Any) -> None:
|
||||
if LANGCHAIN_AVAILABLE:
|
||||
super().__init__(graph=graph, **kwargs)
|
||||
else:
|
||||
super().__init__()
|
||||
self.graph = graph
|
||||
|
||||
def build(self) -> Any:
|
||||
"""Return this tool, or None if langchain-core is missing."""
|
||||
return self if LANGCHAIN_AVAILABLE else None
|
||||
|
||||
def _run(self, query: str, limit: int = 10, **kwargs: Any) -> str:
|
||||
try:
|
||||
return _json(self.graph.query(query, limit=limit))
|
||||
except Exception as exc:
|
||||
return _json({"error": str(exc)})
|
||||
|
||||
async def _arun(self, query: str, limit: int = 10, **kwargs: Any) -> str:
|
||||
return self._run(query, limit=limit)
|
||||
|
||||
|
||||
class SemanticaDecisionTool(_BaseTool): # type: ignore[misc]
|
||||
"""LangChain tool for searching Semantica's recorded decision log.
|
||||
|
||||
Args:
|
||||
graph: A semantica.context.ContextGraph instance.
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||
|
||||
name: str = "semantica_query_decisions"
|
||||
description: str = (
|
||||
"Search Semantica's recorded decision log with a keyword query. "
|
||||
"Returns decisions, rationale, and context."
|
||||
)
|
||||
args_schema: Type[BaseModel] = QueryDecisionsInput
|
||||
graph: Any = None
|
||||
|
||||
def __init__(self, graph: Any = None, **kwargs: Any) -> None:
|
||||
if LANGCHAIN_AVAILABLE:
|
||||
super().__init__(graph=graph, **kwargs)
|
||||
else:
|
||||
super().__init__()
|
||||
self.graph = graph
|
||||
|
||||
def build(self) -> Any:
|
||||
"""Return this tool, or None if langchain-core is missing."""
|
||||
return self if LANGCHAIN_AVAILABLE else None
|
||||
|
||||
def _run(self, category: str = "", limit: int = 10, **kwargs: Any) -> str:
|
||||
try:
|
||||
if category:
|
||||
return _json(self.graph.query(category, limit=limit))
|
||||
return _json(self.graph.get_decision_insights())
|
||||
except Exception as exc:
|
||||
return _json({"error": str(exc)})
|
||||
|
||||
async def _arun(self, category: str = "", limit: int = 10, **kwargs: Any) -> str:
|
||||
return self._run(category=category, limit=limit)
|
||||
@@ -0,0 +1,143 @@
|
||||
"""
|
||||
SemanticaVectorStore — LangChain ``VectorStore`` adapter over Semantica's
|
||||
hybrid search (``semantica.vector_store.HybridSearch``).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from semantica.utils.logging import get_logger
|
||||
|
||||
from .retriever import _hit_content, _hit_id, _hit_score, _hit_type, _hit_layers
|
||||
|
||||
logger = get_logger(__name__)
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional: LangChain core
|
||||
# ---------------------------------------------------------------------------
|
||||
LANGCHAIN_AVAILABLE = False
|
||||
LANGCHAIN_IMPORT_ERROR: Optional[str] = None
|
||||
|
||||
_VectorStoreBase: Any = object
|
||||
_Document: Any = None
|
||||
|
||||
|
||||
def _make_document(**kwargs: Any) -> Any:
|
||||
if _Document is None: # pragma: no cover
|
||||
raise RuntimeError(LANGCHAIN_IMPORT_ERROR or "langchain-core not installed")
|
||||
return _Document(**kwargs)
|
||||
|
||||
|
||||
try:
|
||||
from langchain_core.documents import Document as _Document # type: ignore
|
||||
from langchain_core.vectorstores import (
|
||||
VectorStore as _VectorStoreBase, # type: ignore
|
||||
)
|
||||
|
||||
LANGCHAIN_AVAILABLE = True
|
||||
except ImportError: # pragma: no cover
|
||||
LANGCHAIN_IMPORT_ERROR = (
|
||||
"langchain-core is not installed. Install with: pip install langchain-core"
|
||||
)
|
||||
logger.debug(LANGCHAIN_IMPORT_ERROR)
|
||||
|
||||
|
||||
def _document_from_hit(hit: Dict[str, Any], include_score: bool = True) -> Any:
|
||||
metadata, _ = _hit_layers(hit)
|
||||
node_id = _hit_id(hit)
|
||||
doc_meta = {
|
||||
**metadata,
|
||||
"node_id": node_id,
|
||||
"node_type": _hit_type(hit),
|
||||
}
|
||||
if include_score:
|
||||
doc_meta["score"] = _hit_score(hit, default=0.0)
|
||||
return _make_document(
|
||||
page_content=_hit_content(hit),
|
||||
metadata=doc_meta,
|
||||
)
|
||||
|
||||
|
||||
class SemanticaVectorStore(_VectorStoreBase): # type: ignore[misc]
|
||||
"""Wrap Semantica hybrid search as a LangChain ``VectorStore``.
|
||||
|
||||
Args:
|
||||
hybrid: A semantica.vector_store.HybridSearch instance.
|
||||
vector_store: Optional Semantica vector store passed through to
|
||||
``HybridSearch.add_texts``.
|
||||
"""
|
||||
|
||||
hybrid: Any
|
||||
vector_store: Any = None
|
||||
|
||||
def __init__(self, hybrid: Any, vector_store: Any = None, **kwargs: Any) -> None:
|
||||
if LANGCHAIN_AVAILABLE:
|
||||
super().__init__(**kwargs)
|
||||
else:
|
||||
super().__init__()
|
||||
self.hybrid = hybrid
|
||||
self.vector_store = vector_store
|
||||
|
||||
# -- required VectorStore API ------------------------------------------
|
||||
def add_texts(
|
||||
self,
|
||||
texts: Iterable[str],
|
||||
metadatas: Optional[List[Dict[str, Any]]] = None,
|
||||
**kwargs: Any,
|
||||
) -> List[str]:
|
||||
"""Embed and store texts; return the generated IDs.
|
||||
|
||||
Delegates to the Semantica ``VectorStore.add_documents`` backing the
|
||||
HybridSearch instance (or to ``hybrid.vector_store`` if provided).
|
||||
"""
|
||||
if self.vector_store is not None:
|
||||
return self.vector_store.add_documents(
|
||||
list(texts), metadata=metadatas, **kwargs
|
||||
)
|
||||
vs = getattr(self.hybrid, "vector_store", None)
|
||||
if vs is not None and hasattr(vs, "add_documents"):
|
||||
return vs.add_documents(list(texts), metadata=metadatas, **kwargs)
|
||||
raise ValueError(
|
||||
"SemanticaVectorStore requires a Semantica vector store with "
|
||||
"add_documents (pass vector_store=... to the HybridSearch or to "
|
||||
"SemanticaVectorStore)"
|
||||
)
|
||||
|
||||
def similarity_search(self, query: str, k: int = 4, **kwargs: Any) -> List[Any]:
|
||||
"""Return documents most similar to the query."""
|
||||
return [_document_from_hit(hit) for hit in self.hybrid.search(query, k=k)]
|
||||
|
||||
def similarity_search_with_score(
|
||||
self, query: str, k: int = 4, **kwargs: Any
|
||||
) -> List[Any]:
|
||||
"""Return (document, score) pairs."""
|
||||
return [
|
||||
(
|
||||
_document_from_hit(hit, include_score=False),
|
||||
_hit_score(hit, default=0.0),
|
||||
)
|
||||
for hit in self.hybrid.search(query, k=k)
|
||||
]
|
||||
|
||||
@classmethod
|
||||
def from_texts(
|
||||
cls,
|
||||
texts: List[str],
|
||||
embedding: Any = None,
|
||||
metadatas: Optional[List[Dict[str, Any]]] = None,
|
||||
**kwargs: Any,
|
||||
) -> "SemanticaVectorStore":
|
||||
"""Build a store from a list of texts (LangChain convention).
|
||||
|
||||
Requires a pre-configured ``hybrid`` instance passed via kwargs.
|
||||
"""
|
||||
hybrid = kwargs.pop("hybrid", None)
|
||||
if hybrid is None:
|
||||
raise ValueError(
|
||||
"SemanticaVectorStore.from_texts requires a 'hybrid' "
|
||||
"HybridSearch instance as a keyword argument"
|
||||
)
|
||||
store = cls(hybrid=hybrid, **kwargs)
|
||||
store.add_texts(texts, metadatas=metadatas)
|
||||
return store
|
||||
@@ -116,7 +116,41 @@ class OpenClawKGTool:
|
||||
)
|
||||
|
||||
def __init__(self, base_url: str = "http://localhost:8000", timeout: int = 30) -> None:
|
||||
self.base_url = base_url.rstrip("/")
|
||||
# Validate base_url at construction time so callers get an immediate,
|
||||
# actionable error rather than a cryptic failure on the first request.
|
||||
# allow_private_ips=True because the documented default (localhost:8000)
|
||||
# is intentionally a local Semantica server; the scheme check and
|
||||
# URL-structure check still apply unconditionally.
|
||||
try:
|
||||
from semantica.ingest.ssrf import validate_url_for_request
|
||||
validate_url_for_request(base_url, allow_private_ips=True)
|
||||
except ImportError:
|
||||
# semantica.ingest not installed in minimal openclaw-only environments;
|
||||
# mirror the structural checks that validate_url_for_request performs
|
||||
# unconditionally (before allow_private_ips is consulted), so the
|
||||
# guarantee in the comment above — "scheme check and URL-structure check
|
||||
# still apply unconditionally" — holds in this path too.
|
||||
from urllib.parse import urlparse as _urlparse
|
||||
if not isinstance(base_url, str) or not base_url.strip():
|
||||
raise ValueError("OpenClawKGTool base_url must be a non-empty string.")
|
||||
_parsed = _urlparse(base_url.strip())
|
||||
_scheme = (_parsed.scheme or "").lower()
|
||||
if _scheme not in ("http", "https"):
|
||||
raise ValueError(
|
||||
f"OpenClawKGTool base_url scheme '{_parsed.scheme}' is not permitted. "
|
||||
"Only http and https are allowed."
|
||||
)
|
||||
if not _parsed.netloc:
|
||||
raise ValueError(
|
||||
f"Invalid OpenClawKGTool base_url '{base_url}': "
|
||||
"URL must include a netloc (domain or host)."
|
||||
)
|
||||
if not _parsed.hostname:
|
||||
raise ValueError(
|
||||
f"Invalid OpenClawKGTool base_url '{base_url}': "
|
||||
"URL must include a hostname."
|
||||
)
|
||||
self.base_url = base_url.strip().rstrip("/")
|
||||
self.timeout = timeout
|
||||
self._session: Any = None
|
||||
|
||||
|
||||
@@ -21,6 +21,17 @@ Configure in Claude Desktop, Windsurf, Cline, Continue, VS Code:
|
||||
}
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
# MCP stdio framing IS stdout: any progress bar or console renderer that writes
|
||||
# to stdout would interleave with the JSON-RPC stream and corrupt framing for
|
||||
# every client. This package is always used as an MCP stdio server, so force
|
||||
# progress tracking off for the entire process. Set before importing server /
|
||||
# tools so the Semantica progress-tracker singleton is never created with
|
||||
# output enabled (the singleton reads this variable at construction time and
|
||||
# the enabled.setter re-checks it, so later re-enable attempts are also blocked).
|
||||
os.environ["SEMANTICA_DISABLE_PROGRESS"] = "1"
|
||||
|
||||
# `semantica.__version__` is the authoritative package version — see
|
||||
# semantica/mcp_server/__init__.py for why it is used directly rather than
|
||||
# importlib.metadata.version("semantica").
|
||||
|
||||
+12
-2
@@ -93,7 +93,14 @@ def _handle_tools_call(req_id: Any, params: dict) -> dict:
|
||||
result = tool["_handler"](args)
|
||||
except Exception as exc:
|
||||
log.exception("Tool %s raised an exception", name)
|
||||
return _err(req_id, _INTERNAL_ERROR, str(exc))
|
||||
# The exception's class name (e.g. "ValidationError", "TimeoutError")
|
||||
# is safe to surface — unlike str(exc), it never carries paths,
|
||||
# connection strings, or other internal detail — and lets the
|
||||
# client distinguish failure kinds without a full message.
|
||||
return _err(
|
||||
req_id, _INTERNAL_ERROR,
|
||||
f"Tool '{name}' failed ({type(exc).__name__}). See server logs for details.",
|
||||
)
|
||||
|
||||
# MCP spec: content must be a list of content items
|
||||
return _ok(req_id, {
|
||||
@@ -171,7 +178,10 @@ class SemanticaMCPServer:
|
||||
log.exception("Unhandled error in method %s", method)
|
||||
if req_id is None:
|
||||
return None
|
||||
return _err(req_id, _INTERNAL_ERROR, str(exc))
|
||||
return _err(
|
||||
req_id, _INTERNAL_ERROR,
|
||||
f"Method '{method}' failed ({type(exc).__name__}). See server logs for details.",
|
||||
)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
def run(self) -> None:
|
||||
|
||||
+6
-1
@@ -80,7 +80,12 @@ def handle_export_graph(args: dict) -> dict:
|
||||
if rdf_fmt:
|
||||
try:
|
||||
from semantica.export import RDFExporter
|
||||
rdf_str = RDFExporter().export_to_rdf(graph, format=rdf_fmt)
|
||||
# RDFExporter.export_to_rdf() expects the canonical kg dict
|
||||
# {"entities": [...], "relationships": [...]}, not a ContextGraph
|
||||
# object. Convert before handing off; passing the raw graph
|
||||
# caused AttributeError: 'ContextGraph' object has no attribute
|
||||
# 'get' on every RDF format.
|
||||
rdf_str = RDFExporter().export_to_rdf(graph.to_kg_dict(), format=rdf_fmt)
|
||||
return {"format": rdf_fmt, "data": rdf_str}
|
||||
except Exception as exc:
|
||||
return {"error": f"RDF export failed: {exc}"}
|
||||
|
||||
@@ -53,7 +53,7 @@ plugins/
|
||||
## Prerequisites
|
||||
|
||||
```bash
|
||||
git clone https://github.com/Hawksight-AI/semantica.git
|
||||
git clone https://github.com/semantica-agi/semantica.git
|
||||
cd semantica
|
||||
pip install semantica # Python 3.10+
|
||||
```
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
{
|
||||
"name": "semantica-local",
|
||||
"owner": {
|
||||
"name": "Hawksight AI",
|
||||
"url": "https://github.com/Hawksight-AI/semantica"
|
||||
"name": "Semantica",
|
||||
"url": "https://github.com/semantica-agi/semantica"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
"author": {
|
||||
"name": "Semantica Contributors"
|
||||
},
|
||||
"homepage": "https://github.com/Hawksight-AI/semantica",
|
||||
"repository": "https://github.com/Hawksight-AI/semantica",
|
||||
"homepage": "https://github.com/semantica-agi/semantica",
|
||||
"repository": "https://github.com/semantica-agi/semantica",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"semantica",
|
||||
|
||||
@@ -6,8 +6,8 @@
|
||||
"author": {
|
||||
"name": "Semantica Contributors"
|
||||
},
|
||||
"homepage": "https://github.com/Hawksight-AI/semantica",
|
||||
"repository": "https://github.com/Hawksight-AI/semantica",
|
||||
"homepage": "https://github.com/semantica-agi/semantica",
|
||||
"repository": "https://github.com/semantica-agi/semantica",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"semantica",
|
||||
|
||||
@@ -5,8 +5,8 @@
|
||||
"author": {
|
||||
"name": "Semantica Contributors"
|
||||
},
|
||||
"homepage": "https://github.com/Hawksight-AI/semantica",
|
||||
"repository": "https://github.com/Hawksight-AI/semantica",
|
||||
"homepage": "https://github.com/semantica-agi/semantica",
|
||||
"repository": "https://github.com/semantica-agi/semantica",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"semantica",
|
||||
|
||||
@@ -6,8 +6,8 @@
|
||||
"author": {
|
||||
"name": "Semantica Contributors"
|
||||
},
|
||||
"homepage": "https://github.com/Hawksight-AI/semantica",
|
||||
"repository": "https://github.com/Hawksight-AI/semantica",
|
||||
"homepage": "https://github.com/semantica-agi/semantica",
|
||||
"repository": "https://github.com/semantica-agi/semantica",
|
||||
"license": "MIT",
|
||||
"keywords": [
|
||||
"semantica",
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user