mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-09-10 04:00:35 +00:00
Compare commits
17
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d0d6b9ab5c | ||
|
|
b177fa7556 | ||
|
|
4559ac6536 | ||
|
|
b17ce71f56 | ||
|
|
39c35549d2 | ||
|
|
d54d74c810 | ||
|
|
2086a21615 | ||
|
|
b4af22d724 | ||
|
|
6a2173027e | ||
|
|
66632a9437 | ||
|
|
a59688c6f9 | ||
|
|
40466269b8 | ||
|
|
38ae5b580b | ||
|
|
279fdbf15b | ||
|
|
45915e50a3 | ||
|
|
b7b60d4a17 | ||
|
|
f83d2a8b12 |
@@ -12,13 +12,63 @@ on:
|
|||||||
- '**/*.md'
|
- '**/*.md'
|
||||||
pull_request:
|
pull_request:
|
||||||
branches: [main]
|
branches: [main]
|
||||||
paths-ignore:
|
|
||||||
- 'docs/**'
|
|
||||||
- 'docs_check.py'
|
|
||||||
- '**/*.md'
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
# Detect whether this PR touches any source files (non-docs/non-markdown).
|
||||||
|
# The result drives the `build` job's `if:` condition so that:
|
||||||
|
# - docs-only PRs: `build` is skipped (satisfies the required check).
|
||||||
|
# - code PRs: `build` runs exactly as before.
|
||||||
|
# Push events (to main) keep their own paths-ignore above and never reach
|
||||||
|
# this job, so the push optimization is unaffected.
|
||||||
|
changes:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
# Only needed for pull_request events; push events are pre-filtered above.
|
||||||
|
if: github.event_name == 'pull_request'
|
||||||
|
outputs:
|
||||||
|
src: ${{ steps.filter.outputs.src }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||||
|
with:
|
||||||
|
# Fetch enough history to compute the merge base against the PR base.
|
||||||
|
fetch-depth: 0
|
||||||
|
- name: Check for source changes
|
||||||
|
id: filter
|
||||||
|
run: |
|
||||||
|
# List files changed in this PR relative to the true merge base.
|
||||||
|
# Using three-dot merge-base diff so changes on the base branch that
|
||||||
|
# are not part of this PR do not appear in the file list.
|
||||||
|
# If every changed file matches docs/** or *.md (any depth) or
|
||||||
|
# docs_check.py, this is a docs-only PR and src=false; otherwise
|
||||||
|
# src=true.
|
||||||
|
BASE="${{ github.event.pull_request.base.sha }}"
|
||||||
|
HEAD="${{ github.event.pull_request.head.sha }}"
|
||||||
|
MERGE_BASE=$(git merge-base "$BASE" "$HEAD")
|
||||||
|
CHANGED=$(git diff --name-only "$MERGE_BASE" "$HEAD")
|
||||||
|
echo "Changed files:"
|
||||||
|
echo "$CHANGED"
|
||||||
|
NON_DOCS=$(echo "$CHANGED" | grep -Ev '^(docs/|docs_check\.py|.*\.md$)' || true)
|
||||||
|
if [ -n "$NON_DOCS" ]; then
|
||||||
|
echo "src=true" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "src=false" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
build:
|
build:
|
||||||
|
needs: [changes]
|
||||||
|
# For pull_request events:
|
||||||
|
# - skip only when changes ran successfully and explicitly set src=false
|
||||||
|
# (i.e. a confirmed docs-only PR).
|
||||||
|
# - run when changes succeeded with src=true (source changes present).
|
||||||
|
# - run when changes failed or was cancelled (fail-closed: missing output
|
||||||
|
# must not silently skip the build).
|
||||||
|
# For push/non-PR events: changes is skipped; always() prevents the build
|
||||||
|
# from being skipped due to a skipped needs dependency.
|
||||||
|
if: >-
|
||||||
|
always() && (
|
||||||
|
github.event_name != 'pull_request' ||
|
||||||
|
needs.changes.result != 'success' ||
|
||||||
|
needs.changes.outputs.src == 'true'
|
||||||
|
)
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||||
|
|||||||
@@ -13,17 +13,65 @@ on:
|
|||||||
- '**/*.md'
|
- '**/*.md'
|
||||||
pull_request:
|
pull_request:
|
||||||
branches: [main]
|
branches: [main]
|
||||||
paths-ignore:
|
|
||||||
- 'docs/**'
|
|
||||||
- 'mkdocs.yml'
|
|
||||||
- 'requirements-docs.txt'
|
|
||||||
- '**/*.md'
|
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
# Detect whether this PR touches any source files (non-docs/non-markdown).
|
||||||
|
# The result drives the `security-scan` job's `if:` condition so that:
|
||||||
|
# - docs-only PRs: `security-scan` is skipped (satisfies the required check).
|
||||||
|
# - code PRs: the full scan runs exactly as before.
|
||||||
|
# Schedule and workflow_dispatch runs always skip this job and run the scan
|
||||||
|
# unconditionally (the security-scan job's if: accounts for that below).
|
||||||
|
# Push events (to main) keep their own paths-ignore above.
|
||||||
|
changes:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
if: github.event_name == 'pull_request'
|
||||||
|
outputs:
|
||||||
|
src: ${{ steps.filter.outputs.src }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
- name: Check for source changes
|
||||||
|
id: filter
|
||||||
|
run: |
|
||||||
|
# List files changed in this PR relative to the true merge base.
|
||||||
|
# Using three-dot merge-base diff so changes on the base branch that
|
||||||
|
# are not part of this PR do not appear in the file list.
|
||||||
|
# If every changed file matches the docs/markdown paths-ignore list
|
||||||
|
# (at any directory depth), this is a docs-only PR and src=false;
|
||||||
|
# otherwise src=true.
|
||||||
|
BASE="${{ github.event.pull_request.base.sha }}"
|
||||||
|
HEAD="${{ github.event.pull_request.head.sha }}"
|
||||||
|
MERGE_BASE=$(git merge-base "$BASE" "$HEAD")
|
||||||
|
CHANGED=$(git diff --name-only "$MERGE_BASE" "$HEAD")
|
||||||
|
echo "Changed files:"
|
||||||
|
echo "$CHANGED"
|
||||||
|
NON_DOCS=$(echo "$CHANGED" | grep -Ev '^(docs/|mkdocs\.yml$|requirements-docs\.txt$|.*\.md$)' || true)
|
||||||
|
if [ -n "$NON_DOCS" ]; then
|
||||||
|
echo "src=true" >> "$GITHUB_OUTPUT"
|
||||||
|
else
|
||||||
|
echo "src=false" >> "$GITHUB_OUTPUT"
|
||||||
|
fi
|
||||||
|
|
||||||
security-scan:
|
security-scan:
|
||||||
|
# For pull_request events:
|
||||||
|
# - skip only when changes ran successfully and explicitly set src=false
|
||||||
|
# (i.e. a confirmed docs-only PR).
|
||||||
|
# - run when changes succeeded with src=true (source changes present).
|
||||||
|
# - run when changes failed or was cancelled (fail-closed: missing output
|
||||||
|
# must not silently skip the security scan).
|
||||||
|
# For schedule/workflow_dispatch/push: changes is skipped; always() ensures
|
||||||
|
# the scan still runs unconditionally for those triggers.
|
||||||
|
needs: [changes]
|
||||||
|
if: >-
|
||||||
|
always() && (
|
||||||
|
github.event_name != 'pull_request' ||
|
||||||
|
needs.changes.result != 'success' ||
|
||||||
|
needs.changes.outputs.src == 'true'
|
||||||
|
)
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
|
|||||||
@@ -185,7 +185,7 @@ Centralized `ConfigManager` with environment variable overrides. No magic defaul
|
|||||||
| **Deduplication v2** | `blocking_v2`, `hybrid_v2`, `semantic_v2`: up to 7x faster than v1 |
|
| **Deduplication v2** | `blocking_v2`, `hybrid_v2`, `semantic_v2`: up to 7x faster than v1 |
|
||||||
| **Indexed search** | Explorer search at 0.004ms on 118k nodes (v0.5.0) |
|
| **Indexed search** | Explorer search at 0.004ms on 118k nodes (v0.5.0) |
|
||||||
|
|
||||||
- [Modules](modules) — Full module documentation with code examples.
|
- [Modules](/modules) — Full module documentation with code examples.
|
||||||
- [Learning More](learning-more) — Configuration reference, performance guide, and troubleshooting.
|
- [Learning More](/learning-more) — Configuration reference, performance guide, and troubleshooting.
|
||||||
- [Pipeline Reference](reference/pipeline) — Pipeline orchestration, workers, and retry policies.
|
- [Pipeline Reference](/reference/pipeline) — Pipeline orchestration, workers, and retry policies.
|
||||||
- [Core Reference](reference/core) — Framework lifecycle, plugin registry, and configuration.
|
- [Core Reference](/reference/core) — Framework lifecycle, plugin registry, and configuration.
|
||||||
|
|||||||
+13
-13
@@ -5,7 +5,7 @@ icon: "compass"
|
|||||||
---
|
---
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
Every module works independently — import only what you need. This page maps developer goals to starting points. The [Module Reference](modules) covers every module in depth.
|
Every module works independently — import only what you need. This page maps developer goals to starting points. The [Module Reference](/modules) covers every module in depth.
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## Quick Reference
|
## Quick Reference
|
||||||
@@ -89,7 +89,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
Pass `method="pattern"` to `NERExtractor` for zero-cost, zero-API-key extraction. Switch to `method="llm"` with any of the supported providers for higher recall.
|
Pass `method="pattern"` to `NERExtractor` for zero-cost, zero-API-key extraction. Switch to `method="llm"` with any of the supported providers for higher recall.
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
**Next:** [Quickstart →](quickstart) — full pipeline with visualization and export.
|
**Next:** [Quickstart →](/quickstart) — full pipeline with visualization and export.
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="Build GraphRAG">
|
<Tab title="Build GraphRAG">
|
||||||
@@ -122,7 +122,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
print(result["reasoning_path"]) # multi-hop trace
|
print(result["reasoning_path"]) # multi-hop trace
|
||||||
```
|
```
|
||||||
|
|
||||||
**Next:** [Context module reference →](reference/context)
|
**Next:** [Context module reference →](/reference/context)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="Add Agent Memory">
|
<Tab title="Add Agent Memory">
|
||||||
@@ -163,7 +163,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
`decision_tracking=True` is required. Without it, `record_decision()` raises `RuntimeError`.
|
`decision_tracking=True` is required. Without it, `record_decision()` raises `RuntimeError`.
|
||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
**Next:** [Context module reference →](reference/context)
|
**Next:** [Context module reference →](/reference/context)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="Track Provenance">
|
<Tab title="Track Provenance">
|
||||||
@@ -195,7 +195,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
diff = manager.diff("v1.0", "v1.1")
|
diff = manager.diff("v1.0", "v1.1")
|
||||||
```
|
```
|
||||||
|
|
||||||
**Next:** [Provenance reference →](reference/provenance) · [Change Management reference →](reference/change_management)
|
**Next:** [Provenance reference →](/reference/provenance) · [Change Management reference →](/reference/change_management)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="Export">
|
<Tab title="Export">
|
||||||
@@ -222,7 +222,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
|
|
||||||
**Formats:** Turtle · JSON-LD · N-Triples · RDF/XML · Parquet · Cypher · Arrow · OWL · CSV · ArangoDB AQL
|
**Formats:** Turtle · JSON-LD · N-Triples · RDF/XML · Parquet · Cypher · Arrow · OWL · CSV · ArangoDB AQL
|
||||||
|
|
||||||
**Next:** [Export module reference →](reference/export)
|
**Next:** [Export module reference →](/reference/export)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="MCP — Claude / Cursor">
|
<Tab title="MCP — Claude / Cursor">
|
||||||
@@ -268,7 +268,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
Set `SEMANTICA_KG_PATH` to persist your graph across restarts. Without it, all data is lost when the server process exits.
|
Set `SEMANTICA_KG_PATH` to persist your graph across restarts. Without it, all data is lost when the server process exits.
|
||||||
</Warning>
|
</Warning>
|
||||||
|
|
||||||
**Next:** [MCP Server reference →](reference/mcp_server)
|
**Next:** [MCP Server reference →](/reference/mcp_server)
|
||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
@@ -283,11 +283,11 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
|
|
||||||
Use **both together** via `AgentContext` (GraphRAG) to get grounded LLM responses where every claim traces back to a source node.
|
Use **both together** via `AgentContext` (GraphRAG) to get grounded LLM responses where every claim traces back to a source node.
|
||||||
|
|
||||||
See also: [Core Concepts](concepts)
|
See also: [Core Concepts](/concepts)
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
<Accordion title="I just want to run something quickly." icon="rocket">
|
<Accordion title="I just want to run something quickly." icon="rocket">
|
||||||
Start with the [Quickstart](quickstart). It builds a complete pipeline (ingest → parse → extract → graph → visualize → export) with no API key required.
|
Start with the [Quickstart](/quickstart). It builds a complete pipeline (ingest → parse → extract → graph → visualize → export) with no API key required.
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
<Accordion title="I'm adding Semantica to an existing agent — what's the minimum?" icon="plug">
|
<Accordion title="I'm adding Semantica to an existing agent — what's the minimum?" icon="plug">
|
||||||
@@ -304,7 +304,7 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
[Context module reference →](reference/context)
|
[Context module reference →](/reference/context)
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
<Accordion title="I need a compliance-ready pipeline — what's the minimum stack?" icon="shield-check">
|
<Accordion title="I need a compliance-ready pipeline — what's the minimum stack?" icon="shield-check">
|
||||||
@@ -322,6 +322,6 @@ Pick your goal to see the minimum imports and a working skeleton.
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
- [Quickstart](quickstart) — Full pipeline in 5 minutes.
|
- [Quickstart](/quickstart) — Full pipeline in 5 minutes.
|
||||||
- [Module Reference](modules) — Every module with examples and common chains.
|
- [Module Reference](/modules) — Every module with examples and common chains.
|
||||||
- [API Reference](reference/context) — Complete class and method documentation.
|
- [API Reference](/reference/context) — Complete class and method documentation.
|
||||||
|
|||||||
+2
-2
@@ -48,5 +48,5 @@ Published research using Semantica? [Let us know](https://github.com/semantica-a
|
|||||||
|
|
||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [License](project-license) — MIT License details.
|
- [License](/project-license) — MIT License details.
|
||||||
- [Community](community) — Connect with the Semantica community.
|
- [Community](/community) — Connect with the Semantica community.
|
||||||
|
|||||||
+9
-9
@@ -24,7 +24,7 @@ After installation the following commands are available:
|
|||||||
| `semantica-mcp` | `semantica.mcp_server:main` | MCP server (stdio) for Claude Desktop, Cursor, Windsurf, and other MCP clients |
|
| `semantica-mcp` | `semantica.mcp_server:main` | MCP server (stdio) for Claude Desktop, Cursor, Windsurf, and other MCP clients |
|
||||||
|
|
||||||
<Note>
|
<Note>
|
||||||
`semantica-explorer` requires `pip install semantica[explorer]`. Running it without that extra will immediately print an error and exit. See [Explorer Setup](explorer-setup) for the full walkthrough.
|
`semantica-explorer` requires `pip install semantica[explorer]`. Running it without that extra will immediately print an error and exit. See [Explorer Setup](/explorer-setup) for the full walkthrough.
|
||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
|
|
||||||
@@ -52,8 +52,8 @@ python -c "import semantica; print(semantica.__version__)"
|
|||||||
- **semantica** — The general-purpose CLI. Use it for one-off pipeline runs, entity extraction, and graph operations from a shell script or CI job.
|
- **semantica** — The general-purpose CLI. Use it for one-off pipeline runs, entity extraction, and graph operations from a shell script or CI job.
|
||||||
- **semantica-server** — Starts the REST API server. Binds to `0.0.0.0:8000`. Use this when another service or application needs programmatic access to Semantica over HTTP.
|
- **semantica-server** — Starts the REST API server. Binds to `0.0.0.0:8000`. Use this when another service or application needs programmatic access to Semantica over HTTP.
|
||||||
- **semantica-worker** — Background task processor. Run alongside `semantica-server` when you need async pipeline execution outside the request cycle. Start the server first, then start one or more workers pointing at the same backend.
|
- **semantica-worker** — Background task processor. Run alongside `semantica-server` when you need async pipeline execution outside the request cycle. Start the server first, then start one or more workers pointing at the same backend.
|
||||||
- **semantica-explorer** — Launches the browser dashboard. Requires `pip install semantica[explorer]`. Use this to explore a saved knowledge graph interactively. See [Explorer Setup](explorer-setup).
|
- **semantica-explorer** — Launches the browser dashboard. Requires `pip install semantica[explorer]`. Use this to explore a saved knowledge graph interactively. See [Explorer Setup](/explorer-setup).
|
||||||
- **semantica-mcp** — Runs the MCP server over stdio. Configure it in your MCP client's settings file to expose all 15 tools and 3 resources to Claude Desktop, Cursor, Windsurf, or any MCP-aware client. See [MCP Server](reference/mcp_server).
|
- **semantica-mcp** — Runs the MCP server over stdio. Configure it in your MCP client's settings file to expose all 15 tools and 3 resources to Claude Desktop, Cursor, Windsurf, or any MCP-aware client. See [MCP Server](/reference/mcp_server).
|
||||||
|
|
||||||
|
|
||||||
## Usage Examples
|
## Usage Examples
|
||||||
@@ -116,7 +116,7 @@ python -c "import semantica; print(semantica.__version__)"
|
|||||||
echo '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2024-11-05","capabilities":{},"clientInfo":{"name":"test","version":"1.0"}}}' | semantica-mcp
|
echo '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2024-11-05","capabilities":{},"clientInfo":{"name":"test","version":"1.0"}}}' | semantica-mcp
|
||||||
```
|
```
|
||||||
|
|
||||||
You should receive a JSON-RPC response. See [MCP Server](reference/mcp_server) for the full list of tools and resources.
|
You should receive a JSON-RPC response. See [MCP Server](/reference/mcp_server) for the full list of tools and resources.
|
||||||
</Tab>
|
</Tab>
|
||||||
<Tab title="Explorer">
|
<Tab title="Explorer">
|
||||||
```bash
|
```bash
|
||||||
@@ -124,7 +124,7 @@ python -c "import semantica; print(semantica.__version__)"
|
|||||||
semantica-explorer --graph my_graph.json
|
semantica-explorer --graph my_graph.json
|
||||||
```
|
```
|
||||||
|
|
||||||
See [Explorer Setup](explorer-setup) for the full walkthrough including how to build and save a graph file.
|
See [Explorer Setup](/explorer-setup) for the full walkthrough including how to build and save a graph file.
|
||||||
</Tab>
|
</Tab>
|
||||||
<Tab title="Python module form">
|
<Tab title="Python module form">
|
||||||
Every command also runs as a Python module: useful when the script directory is not on `PATH`:
|
Every command also runs as a Python module: useful when the script directory is not on `PATH`:
|
||||||
@@ -228,7 +228,7 @@ Install the [Microsoft Visual C++ Redistributable](https://aka.ms/vs/17/release/
|
|||||||
|
|
||||||
## Next Steps
|
## Next Steps
|
||||||
|
|
||||||
- [Explorer Setup](explorer-setup) — Build a graph, save it, and launch the browser dashboard.
|
- [Explorer Setup](/explorer-setup) — Build a graph, save it, and launch the browser dashboard.
|
||||||
- [MCP Server](reference/mcp_server) — All 15 tools and 3 resources exposed over the MCP protocol.
|
- [MCP Server](/reference/mcp_server) — All 15 tools and 3 resources exposed over the MCP protocol.
|
||||||
- [Installation](installation) — Virtual environments, optional extras, and platform-specific notes.
|
- [Installation](/installation) — Virtual environments, optional extras, and platform-specific notes.
|
||||||
- [Quickstart](quickstart) — End-to-end pipeline walkthrough with working code.
|
- [Quickstart](/quickstart) — End-to-end pipeline walkthrough with working code.
|
||||||
|
|||||||
@@ -109,12 +109,12 @@ def my_ingestor(source):
|
|||||||
method_registry.register("file", "my_format", my_ingestor)
|
method_registry.register("file", "my_format", my_ingestor)
|
||||||
```
|
```
|
||||||
|
|
||||||
See [Architecture](architecture#extension-points) for the full extension guide.
|
See [Architecture](/architecture#extension-points) for the full extension guide.
|
||||||
|
|
||||||
|
|
||||||
## How to Contribute
|
## How to Contribute
|
||||||
|
|
||||||
- [Contributing Guide](contributing-guide) — Submit code, documentation, tests, or cookbook notebooks.
|
- [Contributing Guide](/contributing-guide) — Submit code, documentation, tests, or cookbook notebooks.
|
||||||
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Report bugs, request features, or propose integrations.
|
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Report bugs, request features, or propose integrations.
|
||||||
- [Discord](https://discord.gg/sV34vps5hH) — Share what you're building with the community.
|
- [Discord](https://discord.gg/sV34vps5hH) — Share what you're building with the community.
|
||||||
- [GitHub Discussions](https://github.com/semantica-agi/semantica/discussions) — Long-form questions, design discussions, and ideas.
|
- [GitHub Discussions](https://github.com/semantica-agi/semantica/discussions) — Long-form questions, design discussions, and ideas.
|
||||||
|
|||||||
+5
-5
@@ -55,7 +55,7 @@ There's no single right way to contribute. Pick the path that fits your skills a
|
|||||||
- Review open pull requests
|
- Review open pull requests
|
||||||
- Share your Semantica projects in GitHub Discussions
|
- Share your Semantica projects in GitHub Discussions
|
||||||
|
|
||||||
See the [Contributing Guide](contributing-guide) for the full development workflow.
|
See the [Contributing Guide](/contributing-guide) for the full development workflow.
|
||||||
|
|
||||||
|
|
||||||
## Stay Connected
|
## Stay Connected
|
||||||
@@ -68,7 +68,7 @@ See the [Contributing Guide](contributing-guide) for the full development workfl
|
|||||||
|
|
||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Contributing Guide](contributing-guide) — Step-by-step guide for submitting PRs and setting up your dev environment.
|
- [Contributing Guide](/contributing-guide) — Step-by-step guide for submitting PRs and setting up your dev environment.
|
||||||
- [Community Projects](community-projects) — Projects and integrations built by the community.
|
- [Community Projects](/community-projects) — Projects and integrations built by the community.
|
||||||
- [FAQ](faq) — Common questions answered.
|
- [FAQ](/faq) — Common questions answered.
|
||||||
- [Governance](governance) — How the project is run and decisions are made.
|
- [Governance](/governance) — How the project is run and decisions are made.
|
||||||
|
|||||||
+7
-7
@@ -5,7 +5,7 @@ icon: "book-open"
|
|||||||
---
|
---
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
New here? Start with [Getting Started](getting-started) for hands-on examples, then return here for deeper understanding.
|
New here? Start with [Getting Started](/getting-started) for hands-on examples, then return here for deeper understanding.
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
Semantica transforms unstructured data: documents, web pages, reports, databases: into **knowledge graphs**: structured representations that AI systems can query, reason about, and trace back to sources.
|
Semantica transforms unstructured data: documents, web pages, reports, databases: into **knowledge graphs**: structured representations that AI systems can query, reason about, and trace back to sources.
|
||||||
@@ -203,7 +203,7 @@ ontology = {
|
|||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
Semantica can auto-generate ontologies from your knowledge graph or import existing OWL/RDF/Turtle ontologies. The **Ontology Hub** (v0.5.0) adds a visual editor, SHACL Studio, alignment authoring, and a live health dashboard. See the [Ontology reference](reference/ontology) for the full 6-stage generation pipeline.
|
Semantica can auto-generate ontologies from your knowledge graph or import existing OWL/RDF/Turtle ontologies. The **Ontology Hub** (v0.5.0) adds a visual editor, SHACL Studio, alignment authoring, and a live health dashboard. See the [Ontology reference](/reference/ontology) for the full 6-stage generation pipeline.
|
||||||
|
|
||||||
|
|
||||||
## Reasoning & Inference
|
## Reasoning & Inference
|
||||||
@@ -319,7 +319,7 @@ scores = calc.calculate_similarity(entity_a, entity_b)
|
|||||||
|
|
||||||
**Features:** N×N semantic distance matrices, ego-mode visualization, distance band classification (`near` / `mid` / `far`), embedding cache optimization for large graphs.
|
**Features:** N×N semantic distance matrices, ego-mode visualization, distance band classification (`near` / `mid` / `far`), embedding cache optimization for large graphs.
|
||||||
|
|
||||||
The [Visualization module](reference/visualization) renders distance matrices as interactive heatmaps and ego-mode neighborhood graphs. The [Explorer](reference/explorer) embeds distance intelligence directly in the browser dashboard.
|
The [Visualization module](/reference/visualization) renders distance matrices as interactive heatmaps and ego-mode neighborhood graphs. The [Explorer](/reference/explorer) embeds distance intelligence directly in the browser dashboard.
|
||||||
|
|
||||||
|
|
||||||
## Deduplication & Entity Resolution
|
## Deduplication & Entity Resolution
|
||||||
@@ -413,7 +413,7 @@ When multiple sources disagree on the same fact, Semantica flags and resolves th
|
|||||||
- **Majority vote**: aggregate across all sources with ≥ 2 agreeing
|
- **Majority vote**: aggregate across all sources with ≥ 2 agreeing
|
||||||
- **Manual review**: flag for human arbitration; continue pipeline without blocking
|
- **Manual review**: flag for human arbitration; continue pipeline without blocking
|
||||||
|
|
||||||
See the [Conflicts reference](reference/conflicts) for `ConflictResolver`, `SourceTracker`, and `InvestigationGuideGenerator`.
|
See the [Conflicts reference](/reference/conflicts) for `ConflictResolver`, `SourceTracker`, and `InvestigationGuideGenerator`.
|
||||||
|
|
||||||
|
|
||||||
## Custom Plugin Development
|
## Custom Plugin Development
|
||||||
@@ -482,6 +482,6 @@ Semantica is designed for extension. Any component: ingestor, extractor, graph b
|
|||||||
</Accordion>
|
</Accordion>
|
||||||
</AccordionGroup>
|
</AccordionGroup>
|
||||||
|
|
||||||
- [Quickstart Tutorial](quickstart) — Build a full pipeline with code.
|
- [Quickstart Tutorial](/quickstart) — Build a full pipeline with code.
|
||||||
- [Modules Guide](modules) — Every module explained with examples.
|
- [Modules Guide](/modules) — Every module explained with examples.
|
||||||
- [API Reference](reference/context) — Complete technical reference.
|
- [API Reference](/reference/context) — Complete technical reference.
|
||||||
|
|||||||
@@ -85,5 +85,5 @@ All contributors are expected to follow the [Contributor Covenant Code of Conduc
|
|||||||
- [GitHub Discussions](https://github.com/semantica-agi/semantica/discussions)
|
- [GitHub Discussions](https://github.com/semantica-agi/semantica/discussions)
|
||||||
- [Discord](https://discord.gg/sV34vps5hH)
|
- [Discord](https://discord.gg/sV34vps5hH)
|
||||||
|
|
||||||
- [Community](community) — Community guidelines and values.
|
- [Community](/community) — Community guidelines and values.
|
||||||
- [Governance](governance) — How decisions are made and the project is run.
|
- [Governance](/governance) — How decisions are made and the project is run.
|
||||||
|
|||||||
+1
-1
@@ -8,7 +8,7 @@ icon: "flask"
|
|||||||
**Where to start:**
|
**Where to start:**
|
||||||
- **New to Semantica**: begin with [Core Tutorials](#core-tutorials)
|
- **New to Semantica**: begin with [Core Tutorials](#core-tutorials)
|
||||||
- **Building an application**: see [Advanced Concepts](#advanced-concepts)
|
- **Building an application**: see [Advanced Concepts](#advanced-concepts)
|
||||||
- **Need installation help**: see the [Installation Guide](installation)
|
- **Need installation help**: see the [Installation Guide](/installation)
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
<Note>
|
<Note>
|
||||||
|
|||||||
@@ -6,7 +6,7 @@ icon: "map"
|
|||||||
|
|
||||||
**`semantica-explorer`** is an **interactive browser dashboard** for knowledge graph exploration. You give it a graph file, it starts a local server, and opens a browser tab where you can search nodes, find paths, inspect provenance, and run analytics: no code required after launch.
|
**`semantica-explorer`** is an **interactive browser dashboard** for knowledge graph exploration. You give it a graph file, it starts a local server, and opens a browser tab where you can search nodes, find paths, inspect provenance, and run analytics: no code required after launch.
|
||||||
|
|
||||||
This page covers everything needed to go from zero to a running Explorer. For the full REST API reference and endpoint catalogue, see [Explorer Reference](reference/explorer).
|
This page covers everything needed to go from zero to a running Explorer. For the full REST API reference and endpoint catalogue, see [Explorer Reference](/reference/explorer).
|
||||||
|
|
||||||
|
|
||||||
## Prerequisites
|
## Prerequisites
|
||||||
@@ -27,7 +27,7 @@ Verify:
|
|||||||
semantica-explorer --help
|
semantica-explorer --help
|
||||||
```
|
```
|
||||||
|
|
||||||
You should see the usage message with the four available flags. If you see `command not found`, activate your virtual environment first. See [CLI Setup](cli-setup#troubleshooting) for PATH help.
|
You should see the usage message with the four available flags. If you see `command not found`, activate your virtual environment first. See [CLI Setup](/cli-setup#troubleshooting) for PATH help.
|
||||||
|
|
||||||
|
|
||||||
## Minimal End-to-End Example
|
## Minimal End-to-End Example
|
||||||
@@ -264,7 +264,7 @@ Once running, Explorer exposes a REST API and dashboard for:
|
|||||||
|
|
||||||
The full endpoint catalogue is documented in the Swagger UI at `/docs` and in the reference page below.
|
The full endpoint catalogue is documented in the Swagger UI at `/docs` and in the reference page below.
|
||||||
|
|
||||||
- [Explorer Reference](reference/explorer) — Every REST endpoint, WebSocket events, analytics, and all supported flags.
|
- [Explorer Reference](/reference/explorer) — Every REST endpoint, WebSocket events, analytics, and all supported flags.
|
||||||
- [CLI Setup](cli-setup) — All five Semantica executables and when to use each one.
|
- [CLI Setup](/cli-setup) — All five Semantica executables and when to use each one.
|
||||||
- [Context Module](reference/context) — Full documentation for ContextGraph: build, query, save, and load.
|
- [Context Module](/reference/context) — Full documentation for ContextGraph: build, query, save, and load.
|
||||||
- [Quickstart](quickstart) — End-to-end pipeline: ingest → extract → build graph → export.
|
- [Quickstart](/quickstart) — End-to-end pipeline: ingest → extract → build graph → export.
|
||||||
|
|||||||
+3
-3
@@ -93,7 +93,7 @@ pip install --upgrade semantica
|
|||||||
pip install semantica
|
pip install semantica
|
||||||
```
|
```
|
||||||
|
|
||||||
See [Installation](installation) for virtual environment setup, optional extras (`[gpu]`, `[all]`, provider-specific), and platform-specific troubleshooting.
|
See [Installation](/installation) for virtual environment setup, optional extras (`[gpu]`, `[all]`, provider-specific), and platform-specific troubleshooting.
|
||||||
|
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
@@ -173,7 +173,7 @@ This includes PyTorch with CUDA, FAISS GPU, and CuPy.
|
|||||||
<Accordion title="How does Semantica handle large datasets?" icon="layer-group">
|
<Accordion title="How does Semantica handle large datasets?" icon="layer-group">
|
||||||
|
|
||||||
- **Batching**: process documents in configurable chunks to control memory usage
|
- **Batching**: process documents in configurable chunks to control memory usage
|
||||||
- **Parallel processing**: `Pipeline(workers=N)` runs extraction steps concurrently
|
- **Parallel processing**: the `semantica.pipeline` module can run independent, parallel-safe steps in the same dependency layer concurrently (see the [Pipeline guide](/guides/pipeline))
|
||||||
- **Delta processing**: update graphs incrementally without full recompute on new data
|
- **Delta processing**: update graphs incrementally without full recompute on new data
|
||||||
- **Persistent backends**: swap in-memory NetworkX for Neo4j, FalkorDB, or Apache AGE for large-scale production graphs
|
- **Persistent backends**: swap in-memory NetworkX for Neo4j, FalkorDB, or Apache AGE for large-scale production graphs
|
||||||
|
|
||||||
@@ -350,4 +350,4 @@ set PYTHONIOENCODING=utf-8
|
|||||||
|
|
||||||
- [Discord](https://discord.gg/sV34vps5hH) — Community chat and live support.
|
- [Discord](https://discord.gg/sV34vps5hH) — Community chat and live support.
|
||||||
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Bug reports and feature requests.
|
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Bug reports and feature requests.
|
||||||
- [Contributing](contributing-guide) — Help improve Semantica.
|
- [Contributing](/contributing-guide) — Help improve Semantica.
|
||||||
|
|||||||
+22
-22
@@ -5,7 +5,7 @@ icon: "rocket"
|
|||||||
---
|
---
|
||||||
|
|
||||||
<Tip>
|
<Tip>
|
||||||
Already installed? Jump straight to [Quickstart](quickstart). Need setup help first? See [Installation](installation).
|
Already installed? Jump straight to [Quickstart](/quickstart). Need setup help first? See [Installation](/installation).
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
## What You Can Build
|
## What You Can Build
|
||||||
@@ -52,15 +52,15 @@ icon: "rocket"
|
|||||||
|
|
||||||
| Track | You want to... | Start with |
|
| Track | You want to... | Start with |
|
||||||
| :----- | :-------------- | :--------- |
|
| :----- | :-------------- | :--------- |
|
||||||
| **Knowledge Graph** | Turn documents into structured, queryable graphs | [Quickstart → Step 1](quickstart) |
|
| **Knowledge Graph** | Turn documents into structured, queryable graphs | [Quickstart → Step 1](/quickstart) |
|
||||||
| **Agent Context** | Give your AI agent persistent memory and decision tracking | [Context reference](reference/context) |
|
| **Agent Context** | Give your AI agent persistent memory and decision tracking | [Context reference](/reference/context) |
|
||||||
| **GraphRAG** | Ground LLM answers in structured knowledge | [Concepts → GraphRAG](concepts#graphrag) |
|
| **GraphRAG** | Ground LLM answers in structured knowledge | [Concepts → GraphRAG](/concepts#graphrag) |
|
||||||
| **MCP Integration** | Use Semantica from Claude Desktop or VS Code | [MCP Server](reference/mcp_server) |
|
| **MCP Integration** | Use Semantica from Claude Desktop or VS Code | [MCP Server](/reference/mcp_server) |
|
||||||
|
|
||||||
</Step>
|
</Step>
|
||||||
|
|
||||||
<Step title="Run the pipeline">
|
<Step title="Run the pipeline">
|
||||||
The full 6-step pipeline: ingest, parse, extract, build, visualize, export: is in the [Quickstart](quickstart). Takes under 5 minutes with pattern-based extraction (no API key required).
|
The full 6-step pipeline: ingest, parse, extract, build, visualize, export: is in the [Quickstart](/quickstart). Takes under 5 minutes with pattern-based extraction (no API key required).
|
||||||
|
|
||||||
<Note>
|
<Note>
|
||||||
An LLM API key is **optional** for the quickstart. Pattern-based extraction works out of the box: upgrade to LLM extraction for higher accuracy when you're ready.
|
An LLM API key is **optional** for the quickstart. Pattern-based extraction works out of the box: upgrade to LLM extraction for higher accuracy when you're ready.
|
||||||
@@ -99,7 +99,7 @@ icon: "rocket"
|
|||||||
print(f"{len(graph['entities'])} nodes, {len(graph['relationships'])} edges")
|
print(f"{len(graph['entities'])} nodes, {len(graph['relationships'])} edges")
|
||||||
```
|
```
|
||||||
|
|
||||||
**Next:** [Full pipeline walkthrough →](quickstart)
|
**Next:** [Full pipeline walkthrough →](/quickstart)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="Agent Context">
|
<Tab title="Agent Context">
|
||||||
@@ -131,7 +131,7 @@ icon: "rocket"
|
|||||||
precedents = context.find_precedents("model selection", limit=5)
|
precedents = context.find_precedents("model selection", limit=5)
|
||||||
```
|
```
|
||||||
|
|
||||||
**Next:** [Context module reference →](reference/context)
|
**Next:** [Context module reference →](/reference/context)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="GraphRAG">
|
<Tab title="GraphRAG">
|
||||||
@@ -161,7 +161,7 @@ icon: "rocket"
|
|||||||
print(f"{claim.text} → source: {claim.source_node}")
|
print(f"{claim.text} → source: {claim.source_node}")
|
||||||
```
|
```
|
||||||
|
|
||||||
**Next:** [GraphRAG concepts →](concepts#graphrag)
|
**Next:** [GraphRAG concepts →](/concepts#graphrag)
|
||||||
</Tab>
|
</Tab>
|
||||||
|
|
||||||
<Tab title="MCP Integration">
|
<Tab title="MCP Integration">
|
||||||
@@ -185,7 +185,7 @@ icon: "rocket"
|
|||||||
|
|
||||||
15 tools available instantly: extract entities, query graph, record decisions, run reasoning, export results.
|
15 tools available instantly: extract entities, query graph, record decisions, run reasoning, export results.
|
||||||
|
|
||||||
**Next:** [MCP Server reference →](reference/mcp_server)
|
**Next:** [MCP Server reference →](/reference/mcp_server)
|
||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
@@ -194,29 +194,29 @@ icon: "rocket"
|
|||||||
|
|
||||||
Semantica uses a modular, layered architecture: import only what you need.
|
Semantica uses a modular, layered architecture: import only what you need.
|
||||||
|
|
||||||
- **[Input Layer](reference/ingest)** — Load and prepare data from any source. Modules: `ingest`, `parse`, `split`, `normalize`
|
- **[Input Layer](/reference/ingest)** — Load and prepare data from any source. Modules: `ingest`, `parse`, `split`, `normalize`
|
||||||
- **[Semantic Layer](reference/semantic_extract)** — Extract meaning from raw text. Modules: `semantic_extract`, `kg`, `ontology`, `reasoning`
|
- **[Semantic Layer](/reference/semantic_extract)** — Extract meaning from raw text. Modules: `semantic_extract`, `kg`, `ontology`, `reasoning`
|
||||||
- **[Storage Layer](reference/vector_store)** — Persist knowledge for retrieval. Modules: `embeddings`, `vector_store`, `graph_store`, `triplet_store`
|
- **[Storage Layer](/reference/vector_store)** — Persist knowledge for retrieval. Modules: `embeddings`, `vector_store`, `graph_store`, `triplet_store`
|
||||||
- **[Quality Layer](reference/deduplication)** — Validate and deduplicate. Modules: `deduplication`, `conflicts`
|
- **[Quality Layer](/reference/deduplication)** — Validate and deduplicate. Modules: `deduplication`, `conflicts`
|
||||||
- **[Context Layer](reference/context)** — Track decisions and lineage. Modules: `context`, `provenance`, `change_management`
|
- **[Context Layer](/reference/context)** — Track decisions and lineage. Modules: `context`, `provenance`, `change_management`
|
||||||
- **[Output Layer](reference/export)** — Deliver results downstream. Modules: `export`, `visualization`, `pipeline`, `explorer`
|
- **[Output Layer](/reference/export)** — Deliver results downstream. Modules: `export`, `visualization`, `pipeline`, `explorer`
|
||||||
|
|
||||||
|
|
||||||
## Which Module Do I Need?
|
## Which Module Do I Need?
|
||||||
|
|
||||||
See the [Choose the Right Module](choose-your-module) guide — it maps 35+ developer goals to the right starting point across all 27 modules, with working code for the most common paths.
|
See the [Choose the Right Module](/choose-your-module) guide — it maps 35+ developer goals to the right starting point across all 27 modules, with working code for the most common paths.
|
||||||
|
|
||||||
|
|
||||||
## Next Steps
|
## Next Steps
|
||||||
|
|
||||||
- [Core Concepts](concepts) — Knowledge graphs, ontologies, and reasoning explained in depth.
|
- [Core Concepts](/concepts) — Knowledge graphs, ontologies, and reasoning explained in depth.
|
||||||
- [Quickstart Tutorial](quickstart) — Full 6-step pipeline walkthrough with working code.
|
- [Quickstart Tutorial](/quickstart) — Full 6-step pipeline walkthrough with working code.
|
||||||
- [Module Reference](modules) — Every module, class, and common chain explained.
|
- [Module Reference](/modules) — Every module, class, and common chain explained.
|
||||||
- [API Reference](reference/context) — Complete module documentation for every class and method.
|
- [API Reference](/reference/context) — Complete module documentation for every class and method.
|
||||||
|
|
||||||
|
|
||||||
## Help
|
## Help
|
||||||
|
|
||||||
- [Discord](https://discord.gg/sV34vps5hH) — Ask questions, share projects, get community support.
|
- [Discord](https://discord.gg/sV34vps5hH) — Ask questions, share projects, get community support.
|
||||||
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Report bugs or request features.
|
- [GitHub Issues](https://github.com/semantica-agi/semantica/issues) — Report bugs or request features.
|
||||||
- [FAQ](faq) — Common questions answered.
|
- [FAQ](/faq) — Common questions answered.
|
||||||
|
|||||||
+4
-4
@@ -214,7 +214,7 @@ A vulnerability in XML parsers that allows attackers to read arbitrary files or
|
|||||||
|
|
||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Core Concepts](concepts) — Deeper explanation of key ideas with code examples.
|
- [Core Concepts](/concepts) — Deeper explanation of key ideas with code examples.
|
||||||
- [Getting Started](getting-started) — First working examples: no prior graph experience required.
|
- [Getting Started](/getting-started) — First working examples: no prior graph experience required.
|
||||||
- [Modules Guide](modules) — All 27 modules explained with code and pipeline chains.
|
- [Modules Guide](/modules) — All 27 modules explained with code and pipeline chains.
|
||||||
- [API Reference](reference/context) — Complete technical reference for every class and method.
|
- [API Reference](/reference/context) — Complete technical reference for every class and method.
|
||||||
|
|||||||
+3
-3
@@ -74,10 +74,10 @@ Semantica follows **Semantic Versioning** (`MAJOR.MINOR.PATCH`):
|
|||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT License: see [LICENSE](https://github.com/semantica-agi/semantica/blob/main/LICENSE) and the [License page](project-license).
|
MIT License: see [LICENSE](https://github.com/semantica-agi/semantica/blob/main/LICENSE) and the [License page](/project-license).
|
||||||
|
|
||||||
|
|
||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Contributing](contributing-guide) — How to submit changes.
|
- [Contributing](/contributing-guide) — How to submit changes.
|
||||||
- [Community](community) — Community guidelines and channels.
|
- [Community](/community) — Community guidelines and channels.
|
||||||
|
|||||||
@@ -46,7 +46,7 @@ Agent Memory provides persistent storage and intelligent retrieval of informatio
|
|||||||
- Simple retrieval tasks where relationships between entities don't matter
|
- Simple retrieval tasks where relationships between entities don't matter
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
This guide covers the memory layer. For graph-enriched traversal and entity linking, see [Context Graphs](context-graphs). For decision accountability — recording, auditing, and causally tracing what the agent chose — see [Decision Intelligence](decision-intelligence).
|
This guide covers the memory layer. For graph-enriched traversal and entity linking, see [Context Graphs](/guides/context-graphs). For decision accountability — recording, auditing, and causally tracing what the agent chose — see [Decision Intelligence](/guides/decision-intelligence).
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## Setting Up a Persistent Memory Context
|
## Setting Up a Persistent Memory Context
|
||||||
@@ -657,10 +657,10 @@ print("Total memories: {}".format(s.get("total_items", 0)))
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — How the underlying `ContextGraph` stores entity nodes and decision nodes; temporal interval reasoning; deduplication before node insertion; ontology from graph.
|
- [Context Graphs](/guides/context-graphs) — How the underlying `ContextGraph` stores entity nodes and decision nodes; temporal interval reasoning; deduplication before node insertion; ontology from graph.
|
||||||
- [Decision Intelligence](decision-intelligence) — Recording decisions as graph nodes with causal chains and policy gating.
|
- [Decision Intelligence](/guides/decision-intelligence) — Recording decisions as graph nodes with causal chains and policy gating.
|
||||||
- [Multi-Agent Systems](multi-agent) — Coordinating multiple agents through a shared `AgentContext` and save/load handoffs.
|
- [Multi-Agent Systems](/guides/multi-agent) — Coordinating multiple agents through a shared `AgentContext` and save/load handoffs.
|
||||||
- [LLM Integrations](llm-integrations) — Configuring the LLM provider passed to `query_with_reasoning()`.
|
- [LLM Integrations](/guides/llm-integrations) — Configuring the LLM provider passed to `query_with_reasoning()`.
|
||||||
- [Deduplication Guide](deduplication) — Full reference for `DuplicateDetector`, `EntityMerger`, similarity methods, and cluster strategies.
|
- [Deduplication Guide](deduplication) — Full reference for `DuplicateDetector`, `EntityMerger`, similarity methods, and cluster strategies.
|
||||||
- [Ontology Management](ontology) — Generate and validate OWL ontologies from the knowledge graph; export to Turtle, OWL/XML, JSON-LD.
|
- [Ontology Management](ontology) — Generate and validate OWL ontologies from the knowledge graph; export to Turtle, OWL/XML, JSON-LD.
|
||||||
- [Context Module Reference](../reference/context) — Full API: `AgentContext`, `AgentMemory`, `MemoryItem`, `ContextRetriever`.
|
- [Context Module Reference](../reference/context) — Full API: `AgentContext`, `AgentMemory`, `MemoryItem`, `ContextRetriever`.
|
||||||
|
|||||||
@@ -496,8 +496,8 @@ print("Model v1.1 verified and approved for production.")
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — `ContextGraph.to_dict()` feeds `create_snapshot()`
|
- [Context Graphs](/guides/context-graphs) — `ContextGraph.to_dict()` feeds `create_snapshot()`
|
||||||
- [Ontology Management](ontology) — pair ontology versioning with graph versioning for a complete schema + data audit trail
|
- [Ontology Management](ontology) — pair ontology versioning with graph versioning for a complete schema + data audit trail
|
||||||
- [SHACL Validation](shacl-validation) — validate graph data at each version gate before snapshotting
|
- [SHACL Validation](/guides/shacl-validation) — validate graph data at each version gate before snapshotting
|
||||||
- [Provenance](provenance) — combine change management with W3C PROV-O lineage for a full audit trail
|
- [Provenance](provenance) — combine change management with W3C PROV-O lineage for a full audit trail
|
||||||
- [Visualization](visualization) — `TemporalVisualizer.visualize_snapshot_comparison()` and `visualize_metrics_evolution()` render version diffs as interactive charts
|
- [Visualization](visualization) — `TemporalVisualizer.visualize_snapshot_comparison()` and `visualize_metrics_evolution()` render version diffs as interactive charts
|
||||||
|
|||||||
@@ -69,7 +69,7 @@ flowchart TD
|
|||||||
2. **Conflict Detection** — Call `detect_entity_conflicts()` to surface all property disagreements at once, or `detect_value_conflicts()` to target a specific property.
|
2. **Conflict Detection** — Call `detect_entity_conflicts()` to surface all property disagreements at once, or `detect_value_conflicts()` to target a specific property.
|
||||||
3. **Resolution** — For each conflict, apply a strategy (`CREDIBILITY_WEIGHTED`, `MOST_RECENT`, `VOTING`, etc.) or route it for expert review (`EXPERT_REVIEW`).
|
3. **Resolution** — For each conflict, apply a strategy (`CREDIBILITY_WEIGHTED`, `MOST_RECENT`, `VOTING`, etc.) or route it for expert review (`EXPERT_REVIEW`).
|
||||||
4. **Persist Canonical Values** — Write resolved values back to your canonical entities or graph store. See [Persisting resolved values](#persisting-resolved-values).
|
4. **Persist Canonical Values** — Write resolved values back to your canonical entities or graph store. See [Persisting resolved values](#persisting-resolved-values).
|
||||||
5. **SHACL Validation** — Enforce structural constraints on the resolved graph to confirm it satisfies your ontology. See [SHACL Validation](shacl-validation).
|
5. **SHACL Validation** — Enforce structural constraints on the resolved graph to confirm it satisfies your ontology. See [SHACL Validation](/guides/shacl-validation).
|
||||||
|
|
||||||
## Quick Start: A Beginner Example
|
## Quick Start: A Beginner Example
|
||||||
|
|
||||||
@@ -698,6 +698,6 @@ Calling `set_resolution_rule()` for every entity-property pair just to apply the
|
|||||||
|
|
||||||
- [Deduplication](deduplication) — remove duplicate nodes before running conflict detection
|
- [Deduplication](deduplication) — remove duplicate nodes before running conflict detection
|
||||||
- [Provenance](provenance) — track which source each resolved value came from, and verify the audit trail cryptographically
|
- [Provenance](provenance) — track which source each resolved value came from, and verify the audit trail cryptographically
|
||||||
- [SHACL Validation](shacl-validation) — enforce structural constraints after conflicts are resolved
|
- [SHACL Validation](/guides/shacl-validation) — enforce structural constraints after conflicts are resolved
|
||||||
- [Change Management](change-management) — snapshot the graph before and after conflict resolution runs
|
- [Change Management](/guides/change-management) — snapshot the graph before and after conflict resolution runs
|
||||||
- [Ontology Management](ontology) — align entity types to a shared vocabulary to reduce type conflicts at the schema level
|
- [Ontology Management](ontology) — align entity types to a shared vocabulary to reduce type conflicts at the schema level
|
||||||
|
|||||||
@@ -50,7 +50,7 @@ A context graph is a property graph that stores entities as **nodes** and relati
|
|||||||
- Cases where setup complexity exceeds the relationship complexity
|
- Cases where setup complexity exceeds the relationship complexity
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
ContextGraph is an **in-memory data structure**. All nodes, edges, and metadata are stored in Python dictionaries and lists. For standalone graphs, persist state with `save_to_file()`. When using `AgentContext`, call `AgentContext.save()` instead — it saves the graph, the FAISS vector index, and memory in one step. For analytical operations on top of a populated graph — centrality rankings, community detection, node embeddings, link prediction — see the [Graph Analytics guide](graph-analytics). For recording and querying decisions stored as nodes, see the [Decision Intelligence guide](decision-intelligence).
|
ContextGraph is an **in-memory data structure**. All nodes, edges, and metadata are stored in Python dictionaries and lists. For standalone graphs, persist state with `save_to_file()`. When using `AgentContext`, call `AgentContext.save()` instead — it saves the graph, the FAISS vector index, and memory in one step. For analytical operations on top of a populated graph — centrality rankings, community detection, node embeddings, link prediction — see the [Graph Analytics guide](/guides/graph-analytics). For recording and querying decisions stored as nodes, see the [Decision Intelligence guide](/guides/decision-intelligence).
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## Constructing the Graph
|
## Constructing the Graph
|
||||||
@@ -704,8 +704,8 @@ for n in stress_reach:
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Graph Analytics](graph-analytics) — centrality rankings, community detection, node embeddings, and link prediction on a populated `ContextGraph`
|
- [Graph Analytics](/guides/graph-analytics) — centrality rankings, community detection, node embeddings, and link prediction on a populated `ContextGraph`
|
||||||
- [Decision Intelligence](decision-intelligence) — recording decisions as typed nodes, causal chain analysis, precedent search, and policy enforcement
|
- [Decision Intelligence](/guides/decision-intelligence) — recording decisions as typed nodes, causal chain analysis, precedent search, and policy enforcement
|
||||||
- [Ingest](ingest) — loading data from PDFs, APIs, databases, STIX bundles, and RSS feeds into the graph
|
- [Ingest](ingest) — loading data from PDFs, APIs, databases, STIX bundles, and RSS feeds into the graph
|
||||||
- [Deduplication](deduplication) — detecting and merging near-duplicate nodes before insertion to prevent graph fragmentation
|
- [Deduplication](deduplication) — detecting and merging near-duplicate nodes before insertion to prevent graph fragmentation
|
||||||
- [Reasoning](reasoning) — temporal interval algebra (Allen relations), forward/backward chaining, and SPARQL over the knowledge graph
|
- [Reasoning](reasoning) — temporal interval algebra (Allen relations), forward/backward chaining, and SPARQL over the knowledge graph
|
||||||
|
|||||||
@@ -638,8 +638,8 @@ results = context.find_precedents("APT29 infrastructure attribution", limit=5)
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — how `ContextGraph` stores decision nodes and causal edges
|
- [Context Graphs](/guides/context-graphs) — how `ContextGraph` stores decision nodes and causal edges
|
||||||
- [Distance Intelligence](distance-intelligence) — `trace_decision_causality()` annotates causal chains with confidence decay and distance bands
|
- [Distance Intelligence](/guides/distance-intelligence) — `trace_decision_causality()` annotates causal chains with confidence decay and distance bands
|
||||||
- [Provenance](provenance) — W3C PROV-O audit trail that wraps decision records in standards-compliant provenance
|
- [Provenance](provenance) — W3C PROV-O audit trail that wraps decision records in standards-compliant provenance
|
||||||
- [MCP Server](mcp-server) — expose decision recording and precedent search to LLM agents via the `record_decision` and `find_precedents` tools
|
- [MCP Server](/guides/mcp-server) — expose decision recording and precedent search to LLM agents via the `record_decision` and `find_precedents` tools
|
||||||
- [Change Management](change-management) — checkpoint decision state with `flush_checkpoint()` for versioned snapshots
|
- [Change Management](/guides/change-management) — checkpoint decision state with `flush_checkpoint()` for versioned snapshots
|
||||||
|
|||||||
@@ -612,7 +612,7 @@ The similarity threshold controls sensitivity. Start at 0.7 and examine false po
|
|||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Ingest Anything](ingest) — multi-source ingestion creates the duplicates this module resolves
|
- [Ingest Anything](ingest) — multi-source ingestion creates the duplicates this module resolves
|
||||||
- [Context Graphs](context-graphs) — store deduplicated entities directly in the knowledge graph
|
- [Context Graphs](/guides/context-graphs) — store deduplicated entities directly in the knowledge graph
|
||||||
- [Conflict Resolution](conflict-resolution) — after merging, reconcile disagreeing property values on the canonical entity
|
- [Conflict Resolution](/guides/conflict-resolution) — after merging, reconcile disagreeing property values on the canonical entity
|
||||||
- [Provenance](provenance) — track merge lineage so every canonical entity traces back to its original sources
|
- [Provenance](provenance) — track merge lineage so every canonical entity traces back to its original sources
|
||||||
- [Pipeline](pipeline) — chain ingest, deduplicate, and store as a `PipelineBuilder` workflow
|
- [Pipeline](pipeline) — chain ingest, deduplicate, and store as a `PipelineBuilder` workflow
|
||||||
|
|||||||
@@ -557,8 +557,8 @@ for chain in chains:
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — `ContextGraph` node and edge model; `add_edge(weight=...)` feeds confidence decay
|
- [Context Graphs](/guides/context-graphs) — `ContextGraph` node and edge model; `add_edge(weight=...)` feeds confidence decay
|
||||||
- [Graph Analytics](graph-analytics) — centrality, community detection, Node2Vec embeddings, link prediction
|
- [Graph Analytics](/guides/graph-analytics) — centrality, community detection, Node2Vec embeddings, link prediction
|
||||||
- [Agent Memory](agent-memory) — proximity-blended retrieval (`proximity_weight`) integrates distance intelligence into memory search
|
- [Agent Memory](/guides/agent-memory) — proximity-blended retrieval (`proximity_weight`) integrates distance intelligence into memory search
|
||||||
- [Decision Intelligence](decision-intelligence) — `trace_decision_causality()` for causal chains with distance annotations
|
- [Decision Intelligence](/guides/decision-intelligence) — `trace_decision_causality()` for causal chains with distance annotations
|
||||||
- [Reasoning & Rules](reasoning) — `TemporalReasoningEngine` for Allen interval algebra over time-bounded graph nodes
|
- [Reasoning & Rules](reasoning) — `TemporalReasoningEngine` for Allen interval algebra over time-bounded graph nodes
|
||||||
|
|||||||
@@ -443,8 +443,8 @@ For semantic reasoning and ontology work, OWL/XML is the format — it is the on
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — the `ContextGraph` object whose `to_dict()` feeds all exports
|
- [Context Graphs](/guides/context-graphs) — the `ContextGraph` object whose `to_dict()` feeds all exports
|
||||||
- [Ontology Management](ontology) — export OWL ontologies generated from your graph
|
- [Ontology Management](ontology) — export OWL ontologies generated from your graph
|
||||||
- [Reasoning & Rules](reasoning) — reasoning results can be exported as RDF triples
|
- [Reasoning & Rules](reasoning) — reasoning results can be exported as RDF triples
|
||||||
- [Change Management](change-management) — snapshot a graph before exporting to prove the export was made from a verified state
|
- [Change Management](/guides/change-management) — snapshot a graph before exporting to prove the export was made from a verified state
|
||||||
- [Pipeline](pipeline) — chain ingest, extract, and export in a single `PipelineBuilder`
|
- [Pipeline](pipeline) — chain ingest, extract, and export in a single `PipelineBuilder`
|
||||||
|
|||||||
@@ -310,7 +310,7 @@ for node1, node2, score in predictions:
|
|||||||
A score above 0.8 is worth analyst review — these aren't random; they're edges the topology of the existing graph strongly implies. Scores below 0.5 are noise. The sweet spot for human review is 0.6–0.8: plausible but not yet confirmed.
|
A score above 0.8 is worth analyst review — these aren't random; they're edges the topology of the existing graph strongly implies. Scores below 0.5 are noise. The sweet spot for human review is 0.6–0.8: plausible but not yet confirmed.
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
Link prediction is also available on `Decision` nodes through `DecisionQuery.predict_decision_relationships(decision_id, top_k)`. See the [Decision Intelligence guide](decision-intelligence) for how to surface causal relationships between past decisions.
|
Link prediction is also available on `Decision` nodes through `DecisionQuery.predict_decision_relationships(decision_id, top_k)`. See the [Decision Intelligence guide](/guides/decision-intelligence) for how to surface causal relationships between past decisions.
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## Understanding Your Decision History
|
## Understanding Your Decision History
|
||||||
@@ -538,7 +538,7 @@ print(f"\n{len(result['communities'])} exposure clusters "
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — building and querying the underlying `ContextGraph`
|
- [Context Graphs](/guides/context-graphs) — building and querying the underlying `ContextGraph`
|
||||||
- [Visualization](visualization) — render centrality rankings and community clusters as interactive dashboards
|
- [Visualization](visualization) — render centrality rankings and community clusters as interactive dashboards
|
||||||
- [Decision Intelligence](decision-intelligence) — link prediction and structural similarity applied to decision nodes
|
- [Decision Intelligence](/guides/decision-intelligence) — link prediction and structural similarity applied to decision nodes
|
||||||
- [GraphRAG](graphrag) — using analytics results to ground LLM generation in the most contextually relevant subgraph
|
- [GraphRAG](/guides/graphrag) — using analytics results to ground LLM generation in the most contextually relevant subgraph
|
||||||
|
|||||||
@@ -576,9 +576,9 @@ The vector search and graph traversal run independently, then their scores are f
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Semantic Extraction](semantic-extraction) — build the graph from raw unstructured text
|
- [Semantic Extraction](/guides/semantic-extraction) — build the graph from raw unstructured text
|
||||||
- [Agent Memory](agent-memory) — store, retrieve, and persist agent memories
|
- [Agent Memory](/guides/agent-memory) — store, retrieve, and persist agent memories
|
||||||
- [Context Graphs](context-graphs) — build and traverse the knowledge graph directly
|
- [Context Graphs](/guides/context-graphs) — build and traverse the knowledge graph directly
|
||||||
- [Reasoning](reasoning) — derive new facts and run inference rules over the graph
|
- [Reasoning](reasoning) — derive new facts and run inference rules over the graph
|
||||||
- [Decision Intelligence](decision-intelligence) — causal chains, policy enforcement, decision tracking
|
- [Decision Intelligence](/guides/decision-intelligence) — causal chains, policy enforcement, decision tracking
|
||||||
- [LLM Integrations](llm-integrations) — connect Groq, OpenAI, Anthropic, HuggingFace, and 100+ more
|
- [LLM Integrations](/guides/llm-integrations) — connect Groq, OpenAI, Anthropic, HuggingFace, and 100+ more
|
||||||
|
|||||||
@@ -951,8 +951,8 @@ print(f"Compliance graph: {graph.stats()['node_count']} nodes, "
|
|||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Pipeline](pipeline) — chain ingest steps with `PipelineBuilder` for automated, retryable, parallelised workflows
|
- [Pipeline](pipeline) — chain ingest steps with `PipelineBuilder` for automated, retryable, parallelised workflows
|
||||||
- [Context Graphs](context-graphs) — storing and querying the entities you ingest as a typed property graph
|
- [Context Graphs](/guides/context-graphs) — storing and querying the entities you ingest as a typed property graph
|
||||||
- [Semantic Extraction](semantic-extraction) — NER, relation extraction, and triplet extraction from ingested text
|
- [Semantic Extraction](/guides/semantic-extraction) — NER, relation extraction, and triplet extraction from ingested text
|
||||||
- [Provenance](provenance) — tracking the origin document, confidence score, and ingestion timestamp for every extracted entity
|
- [Provenance](provenance) — tracking the origin document, confidence score, and ingestion timestamp for every extracted entity
|
||||||
- [Databricks Integration](../integrations/databricks) — Unity Catalog setup, PAT/OAuth M2M authentication, and lineage introspection
|
- [Databricks Integration](../integrations/databricks) — Unity Catalog setup, PAT/OAuth M2M authentication, and lineage introspection
|
||||||
- [Snowflake Integration](../integrations/snowflake) — warehouse setup and password/key-pair/OAuth authentication
|
- [Snowflake Integration](../integrations/snowflake) — warehouse setup and password/key-pair/OAuth authentication
|
||||||
|
|||||||
@@ -719,7 +719,7 @@ for src in best["sources"]:
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Agent Memory](agent-memory) — using `query_with_reasoning()` with any LLM provider for graph-grounded retrieval
|
- [Agent Memory](/guides/agent-memory) — using `query_with_reasoning()` with any LLM provider for graph-grounded retrieval
|
||||||
- [Multi-Agent Systems](multi-agent) — wiring different LLM providers to different agent tiers in a shared-graph pipeline
|
- [Multi-Agent Systems](/guides/multi-agent) — wiring different LLM providers to different agent tiers in a shared-graph pipeline
|
||||||
- [Semantic Extraction](semantic-extraction) — LLM-powered NER, relation extraction, event detection, and triplet extraction
|
- [Semantic Extraction](/guides/semantic-extraction) — LLM-powered NER, relation extraction, event detection, and triplet extraction
|
||||||
- [GraphRAG](graphrag) — multi-hop graph reasoning with `query_with_reasoning()`
|
- [GraphRAG](/guides/graphrag) — multi-hop graph reasoning with `query_with_reasoning()`
|
||||||
|
|||||||
@@ -343,7 +343,7 @@ The result is a fully auditable credit decision trail with precedent links, read
|
|||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Reasoning & Rules](reasoning) — the engine behind the `run_reasoning` tool
|
- [Reasoning & Rules](reasoning) — the engine behind the `run_reasoning` tool
|
||||||
- [Decision Intelligence](decision-intelligence) — how decisions are stored as causal graph nodes
|
- [Decision Intelligence](/guides/decision-intelligence) — how decisions are stored as causal graph nodes
|
||||||
- [Context Graphs](context-graphs) — the graph that `add_entity` and `add_relationship` write to
|
- [Context Graphs](/guides/context-graphs) — the graph that `add_entity` and `add_relationship` write to
|
||||||
- [Export & Serialization](export) — all export formats available via `export_graph`
|
- [Export & Serialization](export) — all export formats available via `export_graph`
|
||||||
- [Ontology Management](ontology) — generate OWL ontologies from the graph built via MCP
|
- [Ontology Management](ontology) — generate OWL ontologies from the graph built via MCP
|
||||||
|
|||||||
@@ -55,7 +55,7 @@ Semantica coordinates agents through shared context (memory and knowledge graphs
|
|||||||
Semantica coordinates multiple agents through a shared `ContextGraph` — agents read and write to the same graph, or hand off serialized state via `save()` and `load()`, with no message broker required. Use this pattern when splitting work across ingestion, enrichment, reasoning, and reporting roles that must share a single evidence base.
|
Semantica coordinates multiple agents through a shared `ContextGraph` — agents read and write to the same graph, or hand off serialized state via `save()` and `load()`, with no message broker required. Use this pattern when splitting work across ingestion, enrichment, reasoning, and reporting roles that must share a single evidence base.
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
This guide covers multi-agent coordination. For the memory layer each agent uses internally, see [Agent Memory](agent-memory). For graph traversal and entity linking, see [Context Graphs](context-graphs). For decision recording and precedent matching, see [Decision Intelligence](decision-intelligence).
|
This guide covers multi-agent coordination. For the memory layer each agent uses internally, see [Agent Memory](/guides/agent-memory). For graph traversal and entity linking, see [Context Graphs](/guides/context-graphs). For decision recording and precedent matching, see [Decision Intelligence](/guides/decision-intelligence).
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## The Three Coordination Patterns
|
## The Three Coordination Patterns
|
||||||
@@ -679,7 +679,7 @@ context.retrieve("...", user_id="analyst-jsmith")
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Agent Memory](agent-memory) — memory storage, retrieval, persistence, and the working memory window each agent uses internally
|
- [Agent Memory](/guides/agent-memory) — memory storage, retrieval, persistence, and the working memory window each agent uses internally
|
||||||
- [Context Graphs](context-graphs) — build and traverse the shared `ContextGraph` directly; temporal interval reasoning; entity deduplication before node insertion
|
- [Context Graphs](/guides/context-graphs) — build and traverse the shared `ContextGraph` directly; temporal interval reasoning; entity deduplication before node insertion
|
||||||
- [Decision Intelligence](decision-intelligence) — record and trace decisions across agent handoffs with causal chain analysis
|
- [Decision Intelligence](/guides/decision-intelligence) — record and trace decisions across agent handoffs with causal chain analysis
|
||||||
- [LLM Integrations](llm-integrations) — configure the LLM provider passed to `query_with_reasoning()` in each agent
|
- [LLM Integrations](/guides/llm-integrations) — configure the LLM provider passed to `query_with_reasoning()` in each agent
|
||||||
|
|||||||
@@ -297,7 +297,7 @@ export_rdf(ontology, "cyber_threat.jsonld", format="jsonld")
|
|||||||
export_rdf(ontology, "cyber_threat.nt", format="ntriples")
|
export_rdf(ontology, "cyber_threat.nt", format="ntriples")
|
||||||
```
|
```
|
||||||
|
|
||||||
The exported Turtle file is the input to Semantica's SHACL validation pipeline. See the [SHACL Validation](shacl-validation) guide for how to generate constraint shapes from this ontology and run them against live graph data.
|
The exported Turtle file is the input to Semantica's SHACL validation pipeline. See the [SHACL Validation](/guides/shacl-validation) guide for how to generate constraint shapes from this ontology and run them against live graph data.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -503,8 +503,8 @@ else:
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [SHACL Validation](shacl-validation) — generate W3C SHACL constraint shapes from your ontology and validate live graph data against them
|
- [SHACL Validation](/guides/shacl-validation) — generate W3C SHACL constraint shapes from your ontology and validate live graph data against them
|
||||||
- [Reasoning & Rules](reasoning) — apply forward/backward-chaining rules over your ontology to derive new facts
|
- [Reasoning & Rules](reasoning) — apply forward/backward-chaining rules over your ontology to derive new facts
|
||||||
- [Export & Serialization](export) — export graphs to RDF, GraphML, CSV, and Neo4j Cypher
|
- [Export & Serialization](export) — export graphs to RDF, GraphML, CSV, and Neo4j Cypher
|
||||||
- [Semantic Extraction](semantic-extraction) — extract entities and relationships that feed ontology generation
|
- [Semantic Extraction](/guides/semantic-extraction) — extract entities and relationships that feed ontology generation
|
||||||
- [Context Graphs](context-graphs) — the knowledge graph that ontology generation reads from
|
- [Context Graphs](/guides/context-graphs) — the knowledge graph that ontology generation reads from
|
||||||
|
|||||||
@@ -717,6 +717,6 @@ print(f"Compliance delta update: {result.output}")
|
|||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Ingest](ingest) — all source types for the ingest step: PDFs, APIs, databases, RSS feeds, STIX directories, and streams
|
- [Ingest](ingest) — all source types for the ingest step: PDFs, APIs, databases, RSS feeds, STIX directories, and streams
|
||||||
- [Semantic Extraction](semantic-extraction) — NER, relation extraction, triplet extraction, and event detection for the extract step
|
- [Semantic Extraction](/guides/semantic-extraction) — NER, relation extraction, triplet extraction, and event detection for the extract step
|
||||||
- [Context Graphs](context-graphs) — building and querying the `ContextGraph` that the store step populates
|
- [Context Graphs](/guides/context-graphs) — building and querying the `ContextGraph` that the store step populates
|
||||||
- [Provenance](provenance) — tracking the origin document, confidence score, and pipeline run ID for every extracted entity
|
- [Provenance](provenance) — tracking the origin document, confidence score, and pipeline run ID for every extracted entity
|
||||||
|
|||||||
@@ -662,9 +662,9 @@ print("Policy updated to v2.4.0")
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Decision Intelligence](decision-intelligence) — `record_decision()`, causal chains, and precedent search — the decisions that `check_compliance()` evaluates
|
- [Decision Intelligence](/guides/decision-intelligence) — `record_decision()`, causal chains, and precedent search — the decisions that `check_compliance()` evaluates
|
||||||
- [Reasoning & Rules](reasoning) — complement policy rules with formal inference for logical conflict detection
|
- [Reasoning & Rules](reasoning) — complement policy rules with formal inference for logical conflict detection
|
||||||
- [SHACL Validation](shacl-validation) — enforce structural constraints on policy nodes themselves
|
- [SHACL Validation](/guides/shacl-validation) — enforce structural constraints on policy nodes themselves
|
||||||
- [Change Management](change-management) — version-snapshot the policy graph alongside the knowledge graph
|
- [Change Management](/guides/change-management) — version-snapshot the policy graph alongside the knowledge graph
|
||||||
- [Provenance](provenance) — W3C PROV-O lineage for every policy decision and exception
|
- [Provenance](provenance) — W3C PROV-O lineage for every policy decision and exception
|
||||||
- [MCP Server](mcp-server) — expose `record_decision` and `find_precedents` as MCP tools for AI agents
|
- [MCP Server](/guides/mcp-server) — expose `record_decision` and `find_precedents` as MCP tools for AI agents
|
||||||
|
|||||||
@@ -659,7 +659,7 @@ Note: the banking example above passes `agent_id="credit_data_service_v2"` to `t
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Semantic Extraction](semantic-extraction) — the NER and relation extraction pipeline that auto-generates provenance entries for every extracted entity
|
- [Semantic Extraction](/guides/semantic-extraction) — the NER and relation extraction pipeline that auto-generates provenance entries for every extracted entity
|
||||||
- [Conflict Resolution](conflict-resolution) — provenance property sources feed directly into conflict detection; every resolved value is traceable to its source
|
- [Conflict Resolution](/guides/conflict-resolution) — provenance property sources feed directly into conflict detection; every resolved value is traceable to its source
|
||||||
- [Deduplication](deduplication) — merge operations are recorded in merge history; pair with provenance for a complete lineage from source to canonical entity
|
- [Deduplication](deduplication) — merge operations are recorded in merge history; pair with provenance for a complete lineage from source to canonical entity
|
||||||
- [Provenance Reference](../reference/provenance) — full storage backend API, `InMemoryStorage`, `SQLiteStorage`, and `ProvenanceEntry` schema
|
- [Provenance Reference](../reference/provenance) — full storage backend API, `InMemoryStorage`, `SQLiteStorage`, and `ProvenanceEntry` schema
|
||||||
|
|||||||
@@ -838,9 +838,9 @@ if proof:
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Semantic Extraction](semantic-extraction) — extract the entities and relationships that populate the graph facts you reason over
|
- [Semantic Extraction](/guides/semantic-extraction) — extract the entities and relationships that populate the graph facts you reason over
|
||||||
- [GraphRAG](graphrag) — retrieve graph-grounded context for LLM responses
|
- [GraphRAG](/guides/graphrag) — retrieve graph-grounded context for LLM responses
|
||||||
- [Ontology Management](ontology) — generate OWL ontologies to give your rules formal semantics
|
- [Ontology Management](ontology) — generate OWL ontologies to give your rules formal semantics
|
||||||
- [Decision Intelligence](decision-intelligence) — record and trace inferred decisions through the full causal chain
|
- [Decision Intelligence](/guides/decision-intelligence) — record and trace inferred decisions through the full causal chain
|
||||||
- [Context Graphs](context-graphs) — the knowledge graph that reasoning operates over
|
- [Context Graphs](/guides/context-graphs) — the knowledge graph that reasoning operates over
|
||||||
- [MCP Server](mcp-server) — expose `run_reasoning` as a tool for Claude and other agents
|
- [MCP Server](/guides/mcp-server) — expose `run_reasoning` as a tool for Claude and other agents
|
||||||
|
|||||||
@@ -71,7 +71,7 @@ This pipeline transforms documents like "APT29 deployed HAMMERTOSS malware targe
|
|||||||
`semantica.semantic_extract` turns unstructured text into structured graph-ready output: it identifies named entities, extracts relationships between them, detects time-anchored events, resolves coreferences, and serialises everything as RDF triplets. Use it to populate a `ContextGraph` from raw documents — intelligence reports, clinical notes, regulatory filings, or any free-text corpus.
|
`semantica.semantic_extract` turns unstructured text into structured graph-ready output: it identifies named entities, extracts relationships between them, detects time-anchored events, resolves coreferences, and serialises everything as RDF triplets. Use it to populate a `ContextGraph` from raw documents — intelligence reports, clinical notes, regulatory filings, or any free-text corpus.
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
Extracted entities and relationships feed into `ContextGraph` via `AgentContext.store()`. For how they are attributed back to source documents, see the [Provenance Guide](provenance). For how the populated graph is queried and traversed, see [Context Graphs](context-graphs).
|
Extracted entities and relationships feed into `ContextGraph` via `AgentContext.store()`. For how they are attributed back to source documents, see the [Provenance Guide](provenance). For how the populated graph is queried and traversed, see [Context Graphs](/guides/context-graphs).
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
## Step 1 — Named Entity Recognition: who and what is in the text
|
## Step 1 — Named Entity Recognition: who and what is in the text
|
||||||
@@ -664,8 +664,8 @@ The fallback behaviour is automatic: if the primary method returns an empty list
|
|||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Provenance Guide](provenance) — track every extracted entity and chunk back to its source document
|
- [Provenance Guide](provenance) — track every extracted entity and chunk back to its source document
|
||||||
- [Agent Memory Guide](agent-memory) — store extracted knowledge as searchable agent memories with graph enrichment
|
- [Agent Memory Guide](/guides/agent-memory) — store extracted knowledge as searchable agent memories with graph enrichment
|
||||||
- [Context Graphs Guide](context-graphs) — how extracted entities populate `ContextGraph` nodes and edges
|
- [Context Graphs Guide](/guides/context-graphs) — how extracted entities populate `ContextGraph` nodes and edges
|
||||||
- [GraphRAG Guide](graphrag) — retrieve facts from the populated graph to ground LLM responses
|
- [GraphRAG Guide](/guides/graphrag) — retrieve facts from the populated graph to ground LLM responses
|
||||||
- [Reasoning Guide](reasoning) — derive new facts, run SPARQL queries, and apply inference rules over the extracted graph
|
- [Reasoning Guide](reasoning) — derive new facts, run SPARQL queries, and apply inference rules over the extracted graph
|
||||||
- [Semantic Extract Reference](../reference/semantic_extract) — full API for all extractor classes, providers, and validators
|
- [Semantic Extract Reference](../reference/semantic_extract) — full API for all extractor classes, providers, and validators
|
||||||
|
|||||||
@@ -740,5 +740,5 @@ def validate_before_publish(data_graph_str: str, ontology: dict) -> None:
|
|||||||
- [Ontology Management](ontology) — generate the OWL ontology that SHACL shapes are derived from
|
- [Ontology Management](ontology) — generate the OWL ontology that SHACL shapes are derived from
|
||||||
- [Reasoning & Rules](reasoning) — complement SHACL structural constraints with logical inference rules
|
- [Reasoning & Rules](reasoning) — complement SHACL structural constraints with logical inference rules
|
||||||
- [Export & Serialization](export) — serialize graph data to Turtle/RDF/XML for `run_shacl_validation` input
|
- [Export & Serialization](export) — serialize graph data to Turtle/RDF/XML for `run_shacl_validation` input
|
||||||
- [Conflict Resolution](conflict-resolution) — detect and resolve data conflicts before SHACL validation
|
- [Conflict Resolution](/guides/conflict-resolution) — detect and resolve data conflicts before SHACL validation
|
||||||
- [Change Management](change-management) — version-gate SHACL shapes alongside ontology versions
|
- [Change Management](/guides/change-management) — version-gate SHACL shapes alongside ontology versions
|
||||||
|
|||||||
@@ -614,8 +614,8 @@ fig.write_html("out.html") # manual export
|
|||||||
|
|
||||||
## Related Guides
|
## Related Guides
|
||||||
|
|
||||||
- [Context Graphs](context-graphs) — `graph.to_dict()` is the primary input for `KGVisualizer`
|
- [Context Graphs](/guides/context-graphs) — `graph.to_dict()` is the primary input for `KGVisualizer`
|
||||||
- [Ontology Management](ontology) — `OntologyVisualizer` renders ontologies produced by `OntologyGenerator`
|
- [Ontology Management](ontology) — `OntologyVisualizer` renders ontologies produced by `OntologyGenerator`
|
||||||
- [Change Management](change-management) — `TemporalVersionManager` snapshots feed `visualize_metrics_evolution()` and `visualize_snapshot_comparison()`
|
- [Change Management](/guides/change-management) — `TemporalVersionManager` snapshots feed `visualize_metrics_evolution()` and `visualize_snapshot_comparison()`
|
||||||
- [Graph Analytics](graph-analytics) — centrality scores, community dicts, and connectivity results that feed the `AnalyticsVisualizer`
|
- [Graph Analytics](/guides/graph-analytics) — centrality scores, community dicts, and connectivity results that feed the `AnalyticsVisualizer`
|
||||||
- [Export & Serialization](export) — export the same graph to GraphML, GEXF, or DOT for Gephi and Graphviz
|
- [Export & Serialization](export) — export the same graph to GraphML, GEXF, or DOT for Gephi and Graphviz
|
||||||
|
|||||||
+12
-12
@@ -185,8 +185,8 @@ decision_id = context.record_decision(
|
|||||||
|
|
||||||
</CodeGroup>
|
</CodeGroup>
|
||||||
|
|
||||||
- [Full Quickstart](quickstart) — Step-by-step pipeline walkthrough
|
- [Full Quickstart](/quickstart) — Step-by-step pipeline walkthrough
|
||||||
- [Cookbook](cookbook) — 40+ real-world Jupyter notebooks
|
- [Cookbook](/cookbook) — 40+ real-world Jupyter notebooks
|
||||||
- [Join Discord](https://discord.gg/sV34vps5hH) — Community chat and support
|
- [Join Discord](https://discord.gg/sV34vps5hH) — Community chat and support
|
||||||
|
|
||||||
|
|
||||||
@@ -195,7 +195,7 @@ decision_id = context.record_decision(
|
|||||||
Semantica was designed for domains where every decision must be explainable and every fact must be traceable.
|
Semantica was designed for domains where every decision must be explainable and every fact must be traceable.
|
||||||
|
|
||||||
<Warning>
|
<Warning>
|
||||||
**This is system-level explainability, not foundation-model explainability.** Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. What Semantica explains is *outside* the model: the context and data fed in, the decision produced, its provenance, the relevant relationships, the policies applied, and the full execution trail. See [Core Concepts](concepts) for the full scope note.
|
**This is system-level explainability, not foundation-model explainability.** Semantica does not expose, reconstruct, or explain what happens *inside* the LLM/foundation model — its internal reasoning or chain-of-thought stays opaque, as it does for any external system. What Semantica explains is *outside* the model: the context and data fed in, the decision produced, its provenance, the relevant relationships, the policies applied, and the full execution trail. See [Core Concepts](/concepts) for the full scope note.
|
||||||
</Warning>
|
</Warning>
|
||||||
|
|
||||||
**Healthcare & Life Sciences**
|
**Healthcare & Life Sciences**
|
||||||
@@ -242,35 +242,35 @@ Semantica was designed for domains where every decision must be explainable and
|
|||||||
```bash
|
```bash
|
||||||
pip install semantica
|
pip install semantica
|
||||||
```
|
```
|
||||||
See [Installation](installation) for optional extras (`[all]`, `[neo4j]`, `[pinecone]`) and environment setup.
|
See [Installation](/installation) for optional extras (`[all]`, `[neo4j]`, `[pinecone]`) and environment setup.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Run the Quickstart">
|
<Step title="Run the Quickstart">
|
||||||
Build a complete knowledge graph pipeline in [5 minutes](quickstart):
|
Build a complete knowledge graph pipeline in [5 minutes](/quickstart):
|
||||||
- Ingest documents from any source
|
- Ingest documents from any source
|
||||||
- Extract entities and relationships
|
- Extract entities and relationships
|
||||||
- Build and query the graph
|
- Build and query the graph
|
||||||
- Record and trace a decision
|
- Record and trace a decision
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Learn the mental model">
|
<Step title="Learn the mental model">
|
||||||
[Core Concepts](concepts) covers:
|
[Core Concepts](/concepts) covers:
|
||||||
- Knowledge graphs vs. vector stores: when to use each
|
- Knowledge graphs vs. vector stores: when to use each
|
||||||
- What GraphRAG is and how Semantica implements it
|
- What GraphRAG is and how Semantica implements it
|
||||||
- How provenance and decision tracking work together
|
- How provenance and decision tracking work together
|
||||||
- The accountability layer architecture
|
- The accountability layer architecture
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Go deep on any module">
|
<Step title="Go deep on any module">
|
||||||
Every module has a dedicated [reference page](reference/context) with:
|
Every module has a dedicated [reference page](/reference/context) with:
|
||||||
- Full class and method documentation
|
- Full class and method documentation
|
||||||
- Parameter tables with types and defaults
|
- Parameter tables with types and defaults
|
||||||
- Runnable code examples for each feature
|
- Runnable code examples for each feature
|
||||||
</Step>
|
</Step>
|
||||||
</Steps>
|
</Steps>
|
||||||
|
|
||||||
- [Installation](installation) — Get Semantica installed in under a minute
|
- [Installation](/installation) — Get Semantica installed in under a minute
|
||||||
- [Quickstart](quickstart) — Build a complete knowledge graph pipeline in 5 minutes
|
- [Quickstart](/quickstart) — Build a complete knowledge graph pipeline in 5 minutes
|
||||||
- [Core Concepts](concepts) — The mental model behind the API
|
- [Core Concepts](/concepts) — The mental model behind the API
|
||||||
- [API Reference](reference/context) — Exact module, class, and method details
|
- [API Reference](/reference/context) — Exact module, class, and method details
|
||||||
- [Cookbook](cookbook) — Domain notebooks for real-world use cases
|
- [Cookbook](/cookbook) — Domain notebooks for real-world use cases
|
||||||
- [Changelog](https://github.com/semantica-agi/semantica/releases) — Release history
|
- [Changelog](https://github.com/semantica-agi/semantica/releases) — Release history
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -183,6 +183,6 @@ Install the [Microsoft Visual C++ Redistributable](https://aka.ms/vs/17/release/
|
|||||||
|
|
||||||
## Next Steps
|
## Next Steps
|
||||||
|
|
||||||
- [Getting Started](getting-started) — Understand what Semantica does before you build.
|
- [Getting Started](/getting-started) — Understand what Semantica does before you build.
|
||||||
- [Build the Pipeline](quickstart) — Follow the end-to-end workflow with code.
|
- [Build the Pipeline](/quickstart) — Follow the end-to-end workflow with code.
|
||||||
- [Browse Examples](cookbook) — See notebook examples organized by use case.
|
- [Browse Examples](/cookbook) — See notebook examples organized by use case.
|
||||||
|
|||||||
@@ -193,7 +193,7 @@ if not connector.test_connection():
|
|||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Ingest Module](../reference/ingest) — Full DatabricksIngestor and all other ingestors.
|
- [Ingest Module](../reference/ingest) — Full DatabricksIngestor and all other ingestors.
|
||||||
- [Snowflake Integration](snowflake) — Companion connector for a Snowflake + Databricks hybrid estate.
|
- [Snowflake Integration](/integrations/snowflake) — Companion connector for a Snowflake + Databricks hybrid estate.
|
||||||
- [Pipeline](../reference/pipeline) — Use Databricks ingestion as a pipeline step.
|
- [Pipeline](../reference/pipeline) — Use Databricks ingestion as a pipeline step.
|
||||||
- [Installation](../installation) — All optional dependency extras.
|
- [Installation](../installation) — All optional dependency extras.
|
||||||
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Databricks data.
|
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Databricks data.
|
||||||
|
|||||||
@@ -370,7 +370,7 @@ Common causes of authentication failures:
|
|||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Ingest Module](../reference/ingest) — Full `SalesforceIngestor` API and all other ingestors.
|
- [Ingest Module](../reference/ingest) — Full `SalesforceIngestor` API and all other ingestors.
|
||||||
- [Snowflake Integration](snowflake) — Relational warehouse connector with a similar design.
|
- [Snowflake Integration](/integrations/snowflake) — Relational warehouse connector with a similar design.
|
||||||
- [Databricks Integration](databricks) — Lakehouse connector.
|
- [Databricks Integration](/integrations/databricks) — Lakehouse connector.
|
||||||
- [Installation](../installation) — All optional dependency extras.
|
- [Installation](../installation) — All optional dependency extras.
|
||||||
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Salesforce data.
|
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Salesforce data.
|
||||||
|
|||||||
@@ -172,7 +172,7 @@ if not connector.test_connection():
|
|||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Ingest Module](../reference/ingest) — Full SnowflakeIngestor and all other ingestors.
|
- [Ingest Module](../reference/ingest) — Full SnowflakeIngestor and all other ingestors.
|
||||||
- [Databricks Integration](databricks) — Companion connector for a Snowflake + Databricks hybrid estate.
|
- [Databricks Integration](/integrations/databricks) — Companion connector for a Snowflake + Databricks hybrid estate.
|
||||||
- [Pipeline](../reference/pipeline) — Use Snowflake ingestion as a pipeline step.
|
- [Pipeline](../reference/pipeline) — Use Snowflake ingestion as a pipeline step.
|
||||||
- [Installation](../installation) — All optional dependency extras.
|
- [Installation](../installation) — All optional dependency extras.
|
||||||
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Snowflake data.
|
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Snowflake data.
|
||||||
|
|||||||
+13
-13
@@ -9,9 +9,9 @@ Whether you're running your first pipeline or deploying Semantica in production,
|
|||||||
|
|
||||||
## Learning Paths
|
## Learning Paths
|
||||||
|
|
||||||
- **Beginner (1–2 hrs)** — New to Semantica and knowledge graphs. [Start with Installation →](installation)
|
- **Beginner (1–2 hrs)** — New to Semantica and knowledge graphs. [Start with Installation →](/installation)
|
||||||
- **Intermediate (4–6 hrs)** — Comfortable with basics, building real applications. [Start with Modules →](modules)
|
- **Intermediate (4–6 hrs)** — Comfortable with basics, building real applications. [Start with Modules →](/modules)
|
||||||
- **Advanced (8+ hrs)** — Enterprise deployments, customization, and extension. [Start with Architecture →](architecture)
|
- **Advanced (8+ hrs)** — Enterprise deployments, customization, and extension. [Start with Architecture →](/architecture)
|
||||||
|
|
||||||
<Tabs>
|
<Tabs>
|
||||||
<Tab title="Beginner (1–2 hrs)">
|
<Tab title="Beginner (1–2 hrs)">
|
||||||
@@ -19,16 +19,16 @@ Whether you're running your first pipeline or deploying Semantica in production,
|
|||||||
|
|
||||||
<Steps>
|
<Steps>
|
||||||
<Step title="Set up your environment">
|
<Step title="Set up your environment">
|
||||||
[Installation Guide](installation): virtual environments, optional extras, platform-specific fixes.
|
[Installation Guide](/installation): virtual environments, optional extras, platform-specific fixes.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Understand the core ideas">
|
<Step title="Understand the core ideas">
|
||||||
[Core Concepts](concepts): what knowledge graphs are, how embeddings work, what extraction does.
|
[Core Concepts](/concepts): what knowledge graphs are, how embeddings work, what extraction does.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Run your first example">
|
<Step title="Run your first example">
|
||||||
[Getting Started](getting-started): 5-minute code walkthrough with pattern-based extraction (no API key needed).
|
[Getting Started](/getting-started): 5-minute code walkthrough with pattern-based extraction (no API key needed).
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Build your first knowledge graph">
|
<Step title="Build your first knowledge graph">
|
||||||
[Quickstart Tutorial](quickstart): full 6-step pipeline from ingestion to visualization.
|
[Quickstart Tutorial](/quickstart): full 6-step pipeline from ingestion to visualization.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Explore interactively">
|
<Step title="Explore interactively">
|
||||||
[Welcome to Semantica notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb): Jupyter walkthrough of every module.
|
[Welcome to Semantica notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/01_Welcome_to_Semantica.ipynb): Jupyter walkthrough of every module.
|
||||||
@@ -40,13 +40,13 @@ Whether you're running your first pipeline or deploying Semantica in production,
|
|||||||
|
|
||||||
<Steps>
|
<Steps>
|
||||||
<Step title="Learn every module">
|
<Step title="Learn every module">
|
||||||
[Modules Guide](modules): all 27 modules with code examples and common pipeline chains.
|
[Modules Guide](/modules): all 27 modules with code examples and common pipeline chains.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Build production knowledge graphs">
|
<Step title="Build production knowledge graphs">
|
||||||
[Building Knowledge Graphs notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb): multi-source, deduplication, conflict resolution.
|
[Building Knowledge Graphs notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb): multi-source, deduplication, conflict resolution.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Add semantic search">
|
<Step title="Add semantic search">
|
||||||
[Embeddings notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/09_Embeddings.ipynb): providers, pooling strategies, vector stores.
|
[Embedding Generation notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/12_Embedding_Generation.ipynb): generating embeddings, provider and model switching, dimensions. Then [Vector Store notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/introduction/13_Vector_Store.ipynb): storing and searching vectors for retrieval.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Multi-source integration">
|
<Step title="Multi-source integration">
|
||||||
[Multi-Source Data Integration notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb) for multi-source patterns.
|
[Multi-Source Data Integration notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb) for multi-source patterns.
|
||||||
@@ -58,7 +58,7 @@ Whether you're running your first pipeline or deploying Semantica in production,
|
|||||||
|
|
||||||
<Steps>
|
<Steps>
|
||||||
<Step title="Understand the architecture">
|
<Step title="Understand the architecture">
|
||||||
[Architecture Guide](architecture): four-layer design, extension points, and design decisions.
|
[Architecture Guide](/architecture): four-layer design, extension points, and design decisions.
|
||||||
</Step>
|
</Step>
|
||||||
<Step title="Temporal intelligence">
|
<Step title="Temporal intelligence">
|
||||||
[Temporal Graphs notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb): `valid_from`/`valid_until`, Allen interval algebra, point-in-time queries.
|
[Temporal Graphs notebook](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/10_Temporal_Knowledge_Graphs.ipynb): `valid_from`/`valid_until`, Allen interval algebra, point-in-time queries.
|
||||||
@@ -236,6 +236,6 @@ The `blocking_v2`, `hybrid_v2`, and `semantic_v2` strategies reduce O(n²) compa
|
|||||||
- **Graph exports**: encrypt sensitive exports at rest; use the v0.5.0 SSRF-safe `base_url` validation when configuring custom LLM gateways
|
- **Graph exports**: encrypt sensitive exports at rest; use the v0.5.0 SSRF-safe `base_url` validation when configuring custom LLM gateways
|
||||||
- **XML ingestion**: always use `XMLIngestor` (v0.5.0), which uses the XXE-safe lxml backend; never parse untrusted XML with the standard library parser
|
- **XML ingestion**: always use `XMLIngestor` (v0.5.0), which uses the XXE-safe lxml backend; never parse untrusted XML with the standard library parser
|
||||||
|
|
||||||
- [Cookbook](cookbook) — Interactive Jupyter notebooks from beginner to advanced.
|
- [Cookbook](/cookbook) — Interactive Jupyter notebooks from beginner to advanced.
|
||||||
- [FAQ](faq) — Common questions answered.
|
- [FAQ](/faq) — Common questions answered.
|
||||||
- [API Reference](reference/core) — Complete technical documentation.
|
- [API Reference](/reference/core) — Complete technical documentation.
|
||||||
|
|||||||
+31
-31
@@ -9,7 +9,7 @@ icon: "puzzle-piece"
|
|||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
<Tip>
|
<Tip>
|
||||||
Not sure which module to use? The [Choose the Right Module](choose-your-module) guide maps 35+ developer goals to modules with code examples — start there if you're orienting for the first time.
|
Not sure which module to use? The [Choose the Right Module](/choose-your-module) guide maps 35+ developer goals to modules with code examples — start there if you're orienting for the first time.
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
Semantica is organized into **27 modules** across six logical layers. Each module is independently importable: you never pay for what you don't use.
|
Semantica is organized into **27 modules** across six logical layers. Each module is independently importable: you never pay for what you don't use.
|
||||||
@@ -680,34 +680,34 @@ versioner.create_snapshot(kg, "2024-Q1", author="user@example.com", description=
|
|||||||
|
|
||||||
| Module | Purpose | Key Classes |
|
| Module | Purpose | Key Classes |
|
||||||
| :------ | :------- | :----------- |
|
| :------ | :------- | :----------- |
|
||||||
| [ingest](reference/ingest) | Data ingestion | `FileIngestor`, `WebIngestor`, `ParquetIngestor`, `XMLIngestor` |
|
| [ingest](/reference/ingest) | Data ingestion | `FileIngestor`, `WebIngestor`, `ParquetIngestor`, `XMLIngestor` |
|
||||||
| [parse](reference/parse) | Document parsing | `DocumentParser`, `DoclingParser` |
|
| [parse](/reference/parse) | Document parsing | `DocumentParser`, `DoclingParser` |
|
||||||
| [split](reference/split) | Text chunking | `TextSplitter` |
|
| [split](/reference/split) | Text chunking | `TextSplitter` |
|
||||||
| [normalize](reference/normalize) | Data cleaning | `TextNormalizer`, `EntityNormalizer`, `LanguageDetector` |
|
| [normalize](/reference/normalize) | Data cleaning | `TextNormalizer`, `EntityNormalizer`, `LanguageDetector` |
|
||||||
| [semantic_extract](reference/semantic_extract) | NER & relation extraction | `NERExtractor`, `RelationExtractor`, `TripletExtractor`, `SemanticAnalyzer`, `SemanticNetworkExtractor`, `ExtractionValidator` |
|
| [semantic_extract](/reference/semantic_extract) | NER & relation extraction | `NERExtractor`, `RelationExtractor`, `TripletExtractor`, `SemanticAnalyzer`, `SemanticNetworkExtractor`, `ExtractionValidator` |
|
||||||
| [kg](reference/kg) | Graph construction | `GraphBuilder`, `TemporalGraphQuery`, `SimilarityCalculator` |
|
| [kg](/reference/kg) | Graph construction | `GraphBuilder`, `TemporalGraphQuery`, `SimilarityCalculator` |
|
||||||
| [ontology](reference/ontology) | Schema management | `OntologyGenerator`, `SHACLGenerator` |
|
| [ontology](/reference/ontology) | Schema management | `OntologyGenerator`, `SHACLGenerator` |
|
||||||
| [reasoning](reference/reasoning) | Logical inference | `Reasoner`, `DatalogReasoner` |
|
| [reasoning](/reference/reasoning) | Logical inference | `Reasoner`, `DatalogReasoner` |
|
||||||
| [embeddings](reference/embeddings) | Vector embeddings | `EmbeddingGenerator` |
|
| [embeddings](/reference/embeddings) | Vector embeddings | `EmbeddingGenerator` |
|
||||||
| [vector_store](reference/vector_store) | Vector database | `VectorStore` |
|
| [vector_store](/reference/vector_store) | Vector database | `VectorStore` |
|
||||||
| [graph_store](reference/graph_store) | Graph database | `GraphStore` |
|
| [graph_store](/reference/graph_store) | Graph database | `GraphStore` |
|
||||||
| [triplet_store](reference/triplet_store) | RDF triple store | `TripletStore` |
|
| [triplet_store](/reference/triplet_store) | RDF triple store | `TripletStore` |
|
||||||
| [deduplication](reference/deduplication) | Entity resolution | `EntityResolver`, `DuplicateDetector`, `ClusterBuilder`, `MergeStrategyManager` |
|
| [deduplication](/reference/deduplication) | Entity resolution | `EntityResolver`, `DuplicateDetector`, `ClusterBuilder`, `MergeStrategyManager` |
|
||||||
| [conflicts](reference/conflicts) | Conflict resolution | `ConflictDetector` |
|
| [conflicts](/reference/conflicts) | Conflict resolution | `ConflictDetector` |
|
||||||
| [context](reference/context) | Agent context & decisions | `AgentContext`, `ContextGraph` |
|
| [context](/reference/context) | Agent context & decisions | `AgentContext`, `ContextGraph` |
|
||||||
| [provenance](reference/provenance) | W3C PROV-O lineage | `ProvenanceManager` |
|
| [provenance](/reference/provenance) | W3C PROV-O lineage | `ProvenanceManager` |
|
||||||
| [change_management](reference/change_management) | Version control | `TemporalVersionManager` |
|
| [change_management](/reference/change_management) | Version control | `TemporalVersionManager` |
|
||||||
| [export](reference/export) | Data export | `RDFExporter`, `ParquetExporter` |
|
| [export](/reference/export) | Data export | `RDFExporter`, `ParquetExporter` |
|
||||||
| [visualization](reference/visualization) | Graph visualization | `KGVisualizer` |
|
| [visualization](/reference/visualization) | Graph visualization | `KGVisualizer` |
|
||||||
| [pipeline](reference/pipeline) | Workflow orchestration | `Pipeline`, `PipelineBuilder` |
|
| [pipeline](/reference/pipeline) | Workflow orchestration | `Pipeline`, `PipelineBuilder` |
|
||||||
| [explorer](reference/explorer) | Knowledge Explorer UI | `semantica-explorer --graph <file>` |
|
| [explorer](/reference/explorer) | Knowledge Explorer UI | `semantica-explorer --graph <file>` |
|
||||||
| [llms](reference/llms) | LLM providers | `Groq`, `OpenAI`, `create_provider` |
|
| [llms](/reference/llms) | LLM providers | `Groq`, `OpenAI`, `create_provider` |
|
||||||
| [mcp_server](reference/mcp_server) | MCP stdio server | `python -m semantica.mcp_server` |
|
| [mcp_server](/reference/mcp_server) | MCP stdio server | `python -m semantica.mcp_server` |
|
||||||
| [seed](reference/seed) | KG bootstrapping from structured sources | `SeedManager` |
|
| [seed](/reference/seed) | KG bootstrapping from structured sources | `SeedManager` |
|
||||||
| [evals](reference/evals) | Quality evaluation | `KGEvaluator`, `ExtractionEvaluator`, `PipelineEvaluator`, `RegressionTracker` |
|
| [evals](/reference/evals) | Quality evaluation | `KGEvaluator`, `ExtractionEvaluator`, `PipelineEvaluator`, `RegressionTracker` |
|
||||||
| [core](reference/core) | Base classes & registry | `Semantica`, `ConfigManager`, `PluginRegistry`, `LifecycleManager` |
|
| [core](/reference/core) | Base classes & registry | `Semantica`, `ConfigManager`, `PluginRegistry`, `LifecycleManager` |
|
||||||
| [utils](reference/utils) | Shared utilities | `helpers`, `validators` |
|
| [utils](/reference/utils) | Shared utilities | `helpers`, `validators` |
|
||||||
|
|
||||||
- [Getting Started](getting-started) — Your first knowledge graph in 5 minutes.
|
- [Getting Started](/getting-started) — Your first knowledge graph in 5 minutes.
|
||||||
- [Cookbook](cookbook) — 40+ domain notebooks with real-world examples.
|
- [Cookbook](/cookbook) — 40+ domain notebooks with real-world examples.
|
||||||
- [API Reference](reference/context) — Full technical documentation.
|
- [API Reference](/reference/context) — Full technical documentation.
|
||||||
|
|||||||
@@ -76,5 +76,5 @@ By contributing to Semantica, you agree that your contributions will be licensed
|
|||||||
|
|
||||||
## See Also
|
## See Also
|
||||||
|
|
||||||
- [Contributing](contributing-guide) — How to contribute to the project.
|
- [Contributing](/contributing-guide) — How to contribute to the project.
|
||||||
- [Citation](citation) — How to cite Semantica in research.
|
- [Citation](/citation) — How to cite Semantica in research.
|
||||||
|
|||||||
+87
-64
@@ -47,36 +47,24 @@ python -c "import semantica; print(semantica.__version__)"
|
|||||||
|
|
||||||
<Step title="Ingest">
|
<Step title="Ingest">
|
||||||
|
|
||||||
Load a document from a file, directory, URL, or database.
|
Load a document from a file or directory. The rest of this walkthrough follows
|
||||||
|
the file path; other sources are shown afterwards.
|
||||||
|
|
||||||
<CodeGroup>
|
```python
|
||||||
|
|
||||||
```python File
|
|
||||||
from semantica.ingest import FileIngestor
|
from semantica.ingest import FileIngestor
|
||||||
|
|
||||||
ingestor = FileIngestor()
|
ingestor = FileIngestor()
|
||||||
sources = ingestor.ingest("data/report.pdf")
|
sources = ingestor.ingest("data/report.pdf")
|
||||||
# Also accepts: .docx, .html, .json, .csv, .xlsx, .pptx, .parquet, .xml
|
# Also accepts a directory, .docx, .html, .json, .csv, .xlsx, .pptx, .parquet, .xml
|
||||||
```
|
```
|
||||||
|
|
||||||
```python Web
|
<Tip>
|
||||||
from semantica.ingest import WebIngestor
|
**Other sources.** `WebIngestor().ingest_url(url)` returns a `WebContent` whose
|
||||||
|
`.text` you can feed straight into the Extract step (no parsing needed).
|
||||||
ingestor = WebIngestor(max_depth=2)
|
`ParquetIngestor().ingest(path)` and `XMLIngestor().ingest(path, schema_path=...)`
|
||||||
sources = ingestor.ingest("https://example.com/article")
|
return structured records rather than documents; build a graph from those with
|
||||||
```
|
`GraphBuilder().build({"entities": [...], "relationships": [...]})` directly.
|
||||||
|
</Tip>
|
||||||
```python Parquet / XML (v0.5.0)
|
|
||||||
from semantica.ingest import ParquetIngestor, XMLIngestor
|
|
||||||
|
|
||||||
# Single file or Hive-partitioned directory
|
|
||||||
sources = ParquetIngestor().ingest("data/events.parquet")
|
|
||||||
|
|
||||||
# XML with XSD schema validation
|
|
||||||
sources = XMLIngestor(validate_xsd="schema.xsd").ingest("data/records/")
|
|
||||||
```
|
|
||||||
|
|
||||||
</CodeGroup>
|
|
||||||
|
|
||||||
</Step>
|
</Step>
|
||||||
|
|
||||||
@@ -88,22 +76,24 @@ Extract structured text and layout from raw documents.
|
|||||||
from semantica.parse import DocumentParser
|
from semantica.parse import DocumentParser
|
||||||
|
|
||||||
parser = DocumentParser()
|
parser = DocumentParser()
|
||||||
parsed = parser.parse(sources[0])
|
parsed = parser.parse(sources[0].path) # parse() takes a path string
|
||||||
|
|
||||||
print(parsed.text[:200]) # extracted text
|
print(parsed["text"][:200]) # extracted text
|
||||||
print(parsed.metadata) # title, author, date, source
|
print(parsed["metadata"]) # file_path, encoding, size, and format-specific keys
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`parse()` returns a `dict` with `text`, `full_text`, and `metadata` keys.
|
||||||
|
|
||||||
<Tip>
|
<Tip>
|
||||||
For PDFs with tables, charts, or multi-column layouts, use `DoclingParser`: it applies advanced layout analysis and returns structured table data alongside text.
|
For PDFs with tables, charts, or multi-column layouts, use `DoclingParser` (`pip install semantica[parse-docling]`): it applies advanced layout analysis and returns structured table data alongside text.
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.parse import DoclingParser
|
from semantica.parse import DoclingParser
|
||||||
|
|
||||||
parser = DoclingParser()
|
parser = DoclingParser()
|
||||||
parsed = parser.parse(sources[0])
|
parsed = parser.parse(sources[0].path)
|
||||||
print(parsed.tables) # structured table objects
|
print(parsed["tables"]) # structured table data
|
||||||
```
|
```
|
||||||
|
|
||||||
</Step>
|
</Step>
|
||||||
@@ -117,26 +107,28 @@ Identify named entities and extract typed relationships between them.
|
|||||||
```python Pattern-based (fast, no API key)
|
```python Pattern-based (fast, no API key)
|
||||||
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
||||||
|
|
||||||
ner = NERExtractor(method="pattern")
|
text = parsed["text"]
|
||||||
entities = ner.extract(parsed)
|
|
||||||
# Returns: [{"text": "Apple Inc.", "type": "ORGANIZATION", "confidence": 0.98}, ...]
|
|
||||||
|
|
||||||
rel = RelationExtractor(method="rule")
|
ner = NERExtractor(method="pattern")
|
||||||
relationships = rel.extract(parsed, entities=entities)
|
entities = ner.extract(text)
|
||||||
# Returns: [{"subject": "Steve Jobs", "predicate": "founded", "object": "Apple Inc."}, ...]
|
# Returns: [Entity(text="Apple Inc.", label="ORG", start_char=0, end_char=10, confidence=0.7), ...]
|
||||||
|
|
||||||
|
rel = RelationExtractor(method="pattern")
|
||||||
|
relationships = rel.extract(text, entities=entities)
|
||||||
|
# Returns: [Relation(subject=Entity(...), predicate="founded_by", object=Entity(...), confidence=0.7), ...]
|
||||||
```
|
```
|
||||||
|
|
||||||
```python LLM-powered (higher accuracy)
|
```python LLM-powered (higher accuracy)
|
||||||
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
||||||
from semantica.llms import Groq
|
|
||||||
|
|
||||||
llm = Groq(model="llama-3.3-70b-versatile")
|
# Reads GROQ_API_KEY from the environment; provider/llm_model select the backend
|
||||||
|
text = parsed["text"]
|
||||||
|
|
||||||
ner = NERExtractor(method="llm", llm_provider=llm)
|
ner = NERExtractor(method="llm", provider="groq", llm_model="llama-3.3-70b-versatile")
|
||||||
entities = ner.extract(parsed)
|
entities = ner.extract(text)
|
||||||
|
|
||||||
rel = RelationExtractor(method="llm", llm_provider=llm)
|
rel = RelationExtractor(method="llm", provider="groq", llm_model="llama-3.3-70b-versatile")
|
||||||
relationships = rel.extract(parsed, entities=entities)
|
relationships = rel.extract(text, entities=entities)
|
||||||
```
|
```
|
||||||
|
|
||||||
</CodeGroup>
|
</CodeGroup>
|
||||||
@@ -198,16 +190,17 @@ exporter.export(graph, file_path="graph.nt", format="nt")
|
|||||||
from semantica.export import ParquetExporter
|
from semantica.export import ParquetExporter
|
||||||
|
|
||||||
exporter = ParquetExporter()
|
exporter = ParquetExporter()
|
||||||
exporter.export(graph, file_path="output/graph.parquet")
|
exporter.export(graph, file_path="output/graph")
|
||||||
# Writes nodes.parquet + edges.parquet: ready for Spark, BigQuery, Databricks
|
# Dict input writes one file per key: output/graph_entities.parquet and
|
||||||
|
# output/graph_relationships.parquet: ready for Spark, BigQuery, Databricks
|
||||||
```
|
```
|
||||||
|
|
||||||
```python ArangoDB
|
```python ArangoDB
|
||||||
from semantica.export import ArangoAQLExporter
|
from semantica.export import ArangoAQLExporter
|
||||||
|
|
||||||
exporter = ArangoAQLExporter()
|
exporter = ArangoAQLExporter()
|
||||||
aql = exporter.export(graph)
|
exporter.export(graph, file_path="graph.aql")
|
||||||
# Returns ready-to-run AQL INSERT statements
|
# Writes ready-to-run AQL INSERT statements to graph.aql
|
||||||
```
|
```
|
||||||
|
|
||||||
</CodeGroup>
|
</CodeGroup>
|
||||||
@@ -272,14 +265,21 @@ relationships = rel.extract(text, entities=entities)
|
|||||||
<Accordion title="Multi-source incremental graph build" icon="layer-group">
|
<Accordion title="Multi-source incremental graph build" icon="layer-group">
|
||||||
|
|
||||||
```python
|
```python
|
||||||
|
from semantica.ingest import FileIngestor
|
||||||
|
from semantica.parse import DocumentParser
|
||||||
|
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
||||||
from semantica.kg import GraphBuilder
|
from semantica.kg import GraphBuilder
|
||||||
|
|
||||||
builder = GraphBuilder(merge_entities=True)
|
parser = DocumentParser()
|
||||||
all_entities, all_rels = [], []
|
ner = NERExtractor(method="pattern")
|
||||||
|
rel = RelationExtractor(method="pattern")
|
||||||
|
builder = GraphBuilder(merge_entities=True)
|
||||||
|
|
||||||
for doc in parsed_docs:
|
all_entities, all_rels = [], []
|
||||||
entities = ner.extract(doc)
|
for source in FileIngestor().ingest("data/reports/"):
|
||||||
rels = rel.extract(doc, entities=entities)
|
text = parser.parse(source.path)["text"]
|
||||||
|
entities = ner.extract(text)
|
||||||
|
rels = rel.extract(text, entities=entities)
|
||||||
all_entities.extend(entities)
|
all_entities.extend(entities)
|
||||||
all_rels.extend(rels)
|
all_rels.extend(rels)
|
||||||
|
|
||||||
@@ -359,7 +359,8 @@ graph = builder.build({"entities": entities, "relationships": relationships})
|
|||||||
# Retrieve full lineage for any entity
|
# Retrieve full lineage for any entity
|
||||||
sources = prov.get_all_sources("Apple Inc.")
|
sources = prov.get_all_sources("Apple Inc.")
|
||||||
print(sources[0])
|
print(sources[0])
|
||||||
# {"source": "data/report.pdf", "location": None, "timestamp": "...", "confidence": 0.98}
|
# {"source": "data/report.pdf", "location": None, "timestamp": "...",
|
||||||
|
# "confidence": 1.0, "metadata": {"confidence": 0.98}}
|
||||||
```
|
```
|
||||||
|
|
||||||
</Accordion>
|
</Accordion>
|
||||||
@@ -373,32 +374,54 @@ print(sources[0])
|
|||||||
|
|
||||||
<Accordion title="No entities extracted" icon="magnifying-glass">
|
<Accordion title="No entities extracted" icon="magnifying-glass">
|
||||||
|
|
||||||
The document likely contains scanned images rather than machine-readable text. Enable OCR:
|
The document likely contains scanned images rather than machine-readable text. `DocumentParser` warns when a PDF has no text layer; switch to `DoclingParser` with OCR enabled:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.parse import DocumentParser
|
from semantica.parse import DoclingParser # pip install semantica[parse-docling]
|
||||||
|
|
||||||
parser = DocumentParser(ocr=True) # enables Tesseract OCR
|
parser = DoclingParser(enable_ocr=True)
|
||||||
parsed = parser.parse(sources[0])
|
parsed = parser.parse(sources[0].path)
|
||||||
```
|
```
|
||||||
|
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
<Accordion title="Slow processing on large corpora" icon="gauge">
|
<Accordion title="Slow processing on large corpora" icon="gauge">
|
||||||
|
|
||||||
Enable parallel processing and GPU acceleration:
|
Install the GPU extras so embedding and ML inference run on CUDA:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
pip install semantica[gpu]
|
pip install semantica[gpu]
|
||||||
```
|
```
|
||||||
|
|
||||||
```python
|
Scan the directory for paths first (no file contents are read), then handle one
|
||||||
from semantica.pipeline import Pipeline
|
document at a time and write to a persistent graph backend instead of the
|
||||||
|
in-memory graph:
|
||||||
|
|
||||||
pipeline = Pipeline(workers=8, batch_size=32)
|
```python
|
||||||
pipeline.run(sources)
|
from semantica.ingest import FileIngestor
|
||||||
|
from semantica.parse import DocumentParser
|
||||||
|
from semantica.semantic_extract import NERExtractor, RelationExtractor
|
||||||
|
from semantica.graph_store import GraphStore
|
||||||
|
from semantica.kg import GraphBuilder
|
||||||
|
|
||||||
|
ingestor = FileIngestor()
|
||||||
|
parser = DocumentParser()
|
||||||
|
ner = NERExtractor(method="pattern")
|
||||||
|
rel = RelationExtractor(method="pattern")
|
||||||
|
store = GraphStore(backend="neo4j", uri="bolt://localhost:7687",
|
||||||
|
user="neo4j", password="password")
|
||||||
|
builder = GraphBuilder(merge_entities=True, graph_store=store)
|
||||||
|
|
||||||
|
for info in ingestor.scan_directory("data/reports/", recursive=True):
|
||||||
|
text = parser.parse(info["path"])["text"] # one document loaded at a time
|
||||||
|
entities = ner.extract(text)
|
||||||
|
rels = rel.extract(text, entities=entities)
|
||||||
|
builder.build({"entities": entities, "relationships": rels})
|
||||||
```
|
```
|
||||||
|
|
||||||
|
For multi-step orchestration with configurable parallelism, see the
|
||||||
|
[Pipeline guide](/guides/pipeline).
|
||||||
|
|
||||||
</Accordion>
|
</Accordion>
|
||||||
|
|
||||||
<Accordion title="Memory errors on large graphs" icon="memory">
|
<Accordion title="Memory errors on large graphs" icon="memory">
|
||||||
@@ -429,7 +452,7 @@ pip install --upgrade semantica
|
|||||||
|
|
||||||
## Next Steps
|
## Next Steps
|
||||||
|
|
||||||
- [Core Concepts](concepts) — Knowledge graphs, ontologies, reasoning engines: the mental model behind Semantica.
|
- [Core Concepts](/concepts) — Knowledge graphs, ontologies, reasoning engines: the mental model behind Semantica.
|
||||||
- [Module Reference](modules) — Every module explained with key classes and common chains.
|
- [Module Reference](/modules) — Every module explained with key classes and common chains.
|
||||||
- [API Reference](reference/context) — Complete documentation for every module, class, and parameter.
|
- [API Reference](/reference/context) — Complete documentation for every module, class, and parameter.
|
||||||
- [Cookbook](cookbook) — 40+ interactive Jupyter notebooks with real-world datasets.
|
- [Cookbook](/cookbook) — 40+ interactive Jupyter notebooks with real-world datasets.
|
||||||
|
|||||||
@@ -351,6 +351,6 @@ for record in history:
|
|||||||
</AccordionGroup>
|
</AccordionGroup>
|
||||||
|
|
||||||
- [Provenance](provenance) — W3C PROV-O lineage tracking.
|
- [Provenance](provenance) — W3C PROV-O lineage tracking.
|
||||||
- [Knowledge Graph](kg) — The graph being versioned.
|
- [Knowledge Graph](/reference/kg) — The graph being versioned.
|
||||||
- [Export](export) — Export versioned snapshots.
|
- [Export](export) — Export versioned snapshots.
|
||||||
- [Conflicts](conflicts) — Detect conflicts introduced between versions.
|
- [Conflicts](/reference/conflicts) — Detect conflicts introduced between versions.
|
||||||
|
|||||||
@@ -453,4 +453,4 @@ class InvestigationStep:
|
|||||||
- [Deduplication](deduplication) — Resolve duplicate entities before conflict detection.
|
- [Deduplication](deduplication) — Resolve duplicate entities before conflict detection.
|
||||||
- [Ontology](ontology) — Logical conflicts use SHACL shapes and ontology axioms.
|
- [Ontology](ontology) — Logical conflicts use SHACL shapes and ontology axioms.
|
||||||
- [Provenance](provenance) — Track which source each conflicting fact came from.
|
- [Provenance](provenance) — Track which source each conflicting fact came from.
|
||||||
- [Knowledge Graph](kg) — The graph being checked for conflicts.
|
- [Knowledge Graph](/reference/kg) — The graph being checked for conflicts.
|
||||||
|
|||||||
@@ -449,7 +449,7 @@ print("Nodes: {}, Edges: {}".format(stats["node_count"], stats["edge_count"]))
|
|||||||
`ContextGraph` exposes a full Distance Intelligence API for exploring semantic neighborhoods and blending proximity into retrieval.
|
`ContextGraph` exposes a full Distance Intelligence API for exploring semantic neighborhoods and blending proximity into retrieval.
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
Full Distance Intelligence reference — distance matrices, API endpoints, embedding cache, Explorer UI — is covered in the dedicated [Distance Intelligence](distance) page. This section documents the context-layer API.
|
Full Distance Intelligence reference — distance matrices, API endpoints, embedding cache, Explorer UI — is covered in the dedicated [Distance Intelligence](/reference/distance) page. This section documents the context-layer API.
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
### Neighbors with Distance Metadata
|
### Neighbors with Distance Metadata
|
||||||
@@ -1087,8 +1087,8 @@ class EntityLink:
|
|||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
- [Vector Store](vector_store) — Embedding storage backend for memory retrieval.
|
- [Vector Store](/reference/vector_store) — Embedding storage backend for memory retrieval.
|
||||||
- [Knowledge Graph](kg) — Graph algorithms and analytics used inside ContextGraph.
|
- [Knowledge Graph](/reference/kg) — Graph algorithms and analytics used inside ContextGraph.
|
||||||
- [Reasoning](reasoning) — Logical inference layered on top of context.
|
- [Reasoning](reasoning) — Logical inference layered on top of context.
|
||||||
- [Provenance](provenance) — W3C PROV-O lineage for every stored fact.
|
- [Provenance](provenance) — W3C PROV-O lineage for every stored fact.
|
||||||
|
|
||||||
|
|||||||
@@ -227,6 +227,6 @@ result = build_knowledge_base(sources=["doc.pdf"], method="fast")
|
|||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
- [Pipeline](pipeline) — Pipeline execution and step orchestration.
|
- [Pipeline](pipeline) — Pipeline execution and step orchestration.
|
||||||
- [Utils](utils) — Shared utilities used by Core internally.
|
- [Utils](/reference/utils) — Shared utilities used by Core internally.
|
||||||
- [Getting Started](../getting-started) — Learn the basics before using Core.
|
- [Getting Started](../getting-started) — Learn the basics before using Core.
|
||||||
- [LLMs](llms) — Configure LLM providers via ConfigManager.
|
- [LLMs](/reference/llms) — Configure LLM providers via ConfigManager.
|
||||||
|
|||||||
@@ -437,7 +437,7 @@ result = calculate_similarity(entity_a, entity_b, method="drug_name")
|
|||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
- [Conflicts](conflicts) — Detect value conflicts between non-duplicate entities.
|
- [Conflicts](/reference/conflicts) — Detect value conflicts between non-duplicate entities.
|
||||||
- [Knowledge Graph](kg) — GraphBuilder uses deduplication during construction.
|
- [Knowledge Graph](/reference/kg) — GraphBuilder uses deduplication during construction.
|
||||||
- [Normalize](normalize) — Normalize entity names before deduplication.
|
- [Normalize](/reference/normalize) — Normalize entity names before deduplication.
|
||||||
- [Provenance](provenance) — Track merged entity lineage.
|
- [Provenance](provenance) — Track merged entity lineage.
|
||||||
|
|||||||
@@ -607,9 +607,7 @@ The Knowledge Explorer embeds Distance Intelligence directly in the browser dash
|
|||||||
The 10× cache improvement applies when the graph is unchanged between requests. In write-heavy pipelines where nodes are added continuously, cache hit rates will be lower. Use `force_refresh=False` (default) for read-heavy Explorer usage and `force_refresh=True` for batch pipeline contexts.
|
The 10× cache improvement applies when the graph is unchanged between requests. In write-heavy pipelines where nodes are added continuously, cache hit rates will be lower. Use `force_refresh=False` (default) for read-heavy Explorer usage and `force_refresh=True` for batch pipeline contexts.
|
||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
- [Context Module](context) — `ContextGraph.get_neighbors()` and proximity-blended retrieval.
|
- [Context Module](/reference/context) — `ContextGraph.get_neighbors()` and proximity-blended retrieval.
|
||||||
- [Knowledge Graph Module](kg) — `NodeEmbedder`, `SimilarityCalculator`, and graph analytics.
|
- [Knowledge Graph Module](/reference/kg) — `NodeEmbedder`, `SimilarityCalculator`, and graph analytics.
|
||||||
- [Visualization](visualization) — Programmatic distance heatmaps and ego-mode graph renders.
|
- [Visualization](visualization) — Programmatic distance heatmaps and ego-mode graph renders.
|
||||||
- [Explorer](explorer) — Knowledge Explorer with built-in Distance Intelligence dashboard.
|
- [Explorer](/reference/explorer) — Knowledge Explorer with built-in Distance Intelligence dashboard.
|
||||||
|
|
||||||
- [Distance Intelligence](https://github.com/semantica-agi/semantica/blob/main/cookbook/advanced/12_Distance_Intelligence.ipynb) — Semantic neighborhoods and distance matrices · Advanced
|
|
||||||
|
|||||||
@@ -619,7 +619,7 @@ providers = check_available_providers()
|
|||||||
# → {"sentence_transformers": True, "fastembed": True, "openai": False}
|
# → {"sentence_transformers": True, "fastembed": True, "openai": False}
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Vector Store](vector_store) — Store and search the generated embeddings.
|
- [Vector Store](/reference/vector_store) — Store and search the generated embeddings.
|
||||||
- [Split](split) — Chunk text before embedding for better retrieval quality.
|
- [Split](/reference/split) — Chunk text before embedding for better retrieval quality.
|
||||||
- [KG Module](kg) — Distance Intelligence uses graph embeddings for semantic neighbourhoods.
|
- [KG Module](/reference/kg) — Distance Intelligence uses graph embeddings for semantic neighbourhoods.
|
||||||
- [Deduplication](deduplication) — Semantic deduplication uses embedding distance for entity resolution.
|
- [Deduplication](deduplication) — Semantic deduplication uses embedding distance for entity resolution.
|
||||||
|
|||||||
+209
-49
@@ -1,64 +1,224 @@
|
|||||||
---
|
---
|
||||||
title: "Evals Module"
|
title: "Evals Module"
|
||||||
description: "Evaluation framework for measuring Knowledge Graph quality, extraction accuracy, and pipeline performance: coming soon."
|
description: "Score decision records, audit trails, and reasoning output with deterministic and model-backed evaluators plus a small run harness."
|
||||||
icon: "chart-line"
|
icon: "chart-line"
|
||||||
---
|
---
|
||||||
|
|
||||||
**`semantica.evals`** is planned as a comprehensive evaluation framework for measuring **extraction accuracy, graph quality, and pipeline performance**.
|
`semantica.evals` measures the quality of decision intelligence outputs. It takes
|
||||||
|
the decisions, audit trails, and reasoning text your pipeline produces and scores
|
||||||
|
them against expectations you define, returning a structured summary you can log,
|
||||||
|
assert on in tests, or track across runs.
|
||||||
|
|
||||||
<Warning>
|
- A registry of named evaluators, from exact string matching to ROUGE overlap and
|
||||||
**`semantica.evals` is not yet implemented.** The module is a placeholder with `__all__ = []`. No classes or functions are available for import. This page describes the planned API only.
|
LLM-as-judge
|
||||||
</Warning>
|
- `decision_scores`, a composite evaluator for `Decision` objects that checks
|
||||||
|
outcome, confidence bounds, required fields, provenance, and (optionally)
|
||||||
|
policy compliance
|
||||||
|
- A `evaluate()` runner that applies several evaluators to a list of cases and
|
||||||
|
aggregates pass / fail / error counts
|
||||||
|
- Per-evaluator **objectives** that let you override an evaluator's built-in
|
||||||
|
verdict at the run level
|
||||||
|
|
||||||
## Planned Features
|
<Note>
|
||||||
|
The module is versioned separately from the package: `semantica.evals.__version__`
|
||||||
|
is `"0.1.0"`. The public surface described here is stable, but expect additive
|
||||||
|
changes (new evaluators, new objective options) before it reaches 1.0.
|
||||||
|
</Note>
|
||||||
|
|
||||||
When released, `semantica.evals` will provide:
|
## Public API
|
||||||
|
|
||||||
| Planned Class | Role |
|
| Name | Kind | Role |
|
||||||
| :--- | :--- |
|
| :--- | :--- | :--- |
|
||||||
| `KGEvaluator` | Completeness, consistency, schema compliance, coverage, and orphan node detection |
|
| `evaluate(cases, evaluators, config=None, target_fn=None)` | function | Run named evaluators over each case, return an `EvalSummary` |
|
||||||
| `ExtractionEvaluator` | NER precision / recall / F1 and relation extraction metrics against gold datasets |
|
| `list_evaluators()` | function | Sorted names of every registered evaluator |
|
||||||
| `PipelineBenchmark` | Throughput (docs/sec), per-step latency, peak memory, and error rate |
|
| `get_evaluator(name)` | function | Look up a single evaluator function by name |
|
||||||
| `RegressionTracker` | Record runs and compare metrics across commits or config changes |
|
| `EvalMetric` | dataclass (frozen) | One evaluator's result: `score`, `passed`, `meta` |
|
||||||
| `EvalReport` | Structured report: `{scores, regressions, recommendations}` |
|
| `CaseResult` | namedtuple | One case's result: `case_id`, `status`, `metrics`, `details` |
|
||||||
| `DeduplicationEvaluator` | Merge precision, false positive / false negative rates |
|
| `EvalSummary` | dataclass | Aggregate across cases: `total`, `passed`, `failed`, `errors`, `pass_rate`, `cases` |
|
||||||
| `ReasoningEvaluator` | Inference accuracy, rule coverage, and derivation depth |
|
|
||||||
|
|
||||||
## Current Workaround
|
|
||||||
|
|
||||||
Until `semantica.evals` ships, use `semantica.ontology.OntologyEvaluator` for ontology quality metrics:
|
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.ontology import OntologyEvaluator
|
import semantica.evals as evals
|
||||||
|
from semantica.evals import evaluate, list_evaluators, get_evaluator
|
||||||
evaluator = OntologyEvaluator()
|
|
||||||
|
|
||||||
# evaluate_ontology takes the ontology dict only
|
|
||||||
result = evaluator.evaluate_ontology(ontology)
|
|
||||||
|
|
||||||
print("Coverage: ", result.coverage_score)
|
|
||||||
print("Completeness:", result.completeness_score)
|
|
||||||
print("Gaps: ", result.gaps)
|
|
||||||
print("Suggestions: ", result.suggestions)
|
|
||||||
|
|
||||||
# Full report with class granularity and relation completeness
|
|
||||||
report = evaluator.generate_report(ontology)
|
|
||||||
print("Coverage score: ", report["evaluation"]["coverage_score"])
|
|
||||||
print("Completeness score:", report["evaluation"]["completeness_score"])
|
|
||||||
print("Relation coverage: ", report["relation_completeness"]["relation_coverage"])
|
|
||||||
```
|
```
|
||||||
|
|
||||||
`EvaluationResult` fields returned by `evaluate_ontology()`:
|
## Built-in evaluators
|
||||||
|
|
||||||
| Field | Type | Description |
|
Every evaluator is a plain function `fn(actual, expected, config=None) -> EvalMetric`
|
||||||
| :----- | :---- | :----------- |
|
registered under a stable name. `list_evaluators()` returns the current set:
|
||||||
| `coverage_score` | `float` | Fraction of competency questions answerable by the ontology |
|
|
||||||
| `completeness_score` | `float` | Average of class and property completeness scores |
|
|
||||||
| `gaps` | `List[str]` | Identified gaps in coverage |
|
|
||||||
| `suggestions` | `List[str]` | Improvement suggestions |
|
|
||||||
| `metrics` | `dict` | Detailed sub-metrics |
|
|
||||||
|
|
||||||
- [Semantic Extract](semantic_extract) — Extraction module.
|
```python
|
||||||
- [Knowledge Graph](kg) — Graph quality assessment.
|
>>> list_evaluators()
|
||||||
- [Pipeline](pipeline) — Pipeline performance metrics.
|
['decision_scores', 'exact_match', 'keyword_check', 'length_range',
|
||||||
- [Ontology Evaluator](ontology) — Available now for ontology quality metrics.
|
'levenshtein', 'llm_as_judge', 'numeric_range', 'regex_match', 'rouge',
|
||||||
|
'temporal_range']
|
||||||
|
```
|
||||||
|
|
||||||
|
| Name | Passes when | Relevant `config` keys |
|
||||||
|
| :--- | :--- | :--- |
|
||||||
|
| `exact_match` | `actual == expected` | none |
|
||||||
|
| `regex_match` | `re.search(expected, actual)` matches | none |
|
||||||
|
| `keyword_check` | every required term appears in `actual` (word-boundary) | `required` (falls back to `expected`) |
|
||||||
|
| `numeric_range` | `min <= actual <= max` | `min`, `max` (both required) |
|
||||||
|
| `temporal_range` | ISO datetime `actual` falls in `[min, max]` | `min`, `max` as ISO strings (both required) |
|
||||||
|
| `length_range` | `min <= len(actual) <= max` | `min` (default 0), `max` (required) |
|
||||||
|
| `levenshtein` | normalized similarity `>= threshold` | `threshold` (default 0.8) |
|
||||||
|
| `rouge` | ROUGE-1 F1 `> 0` and `>= threshold` | `threshold` (default 0.0) |
|
||||||
|
| `llm_as_judge` | caller-supplied `judge_fn(actual, expected)` returns truthy | `judge_fn` (required callable) |
|
||||||
|
| `decision_scores` | all configured sub-checks on a `Decision` pass | see below |
|
||||||
|
|
||||||
|
An evaluator that cannot run (bad regex, unparseable datetime, no `judge_fn`) returns an
|
||||||
|
`EvalMetric` with an `"error"` key in `meta` rather than raising. Evaluators that
|
||||||
|
require numeric bounds (`numeric_range`, `length_range`) instead return a failing
|
||||||
|
metric with a `"reason"` key when the bound is missing — they do not raise and do
|
||||||
|
not set `"error"`.
|
||||||
|
|
||||||
|
### `decision_scores`
|
||||||
|
|
||||||
|
`decision_scores` accepts a `Decision` (from `semantica.context.decision_models`)
|
||||||
|
or its dict form and runs a set of field-level and governance checks. The score is
|
||||||
|
the fraction of checks that passed; `passed` is `True` only when all of them did.
|
||||||
|
|
||||||
|
| Sub-check | Controlled by |
|
||||||
|
| :--- | :--- |
|
||||||
|
| Outcome matches | `expected_outcome` in config, or the case's `expected`; **skipped** when neither is set |
|
||||||
|
| Confidence in range | `min_confidence` (default 0.0), `max_confidence` (default 1.0); always run |
|
||||||
|
| `decision_maker`, `reasoning`, `scenario` non-empty | always run |
|
||||||
|
| Provenance present in metadata | `provenance_key` (default `"provenance"`); always run |
|
||||||
|
| Policy compliance | `policy_engine` and `policy_id` both set; skipped otherwise |
|
||||||
|
|
||||||
|
Passing `causal_chain_exists` in config raises `NotImplementedError`. That key is a
|
||||||
|
reserved slot for a future release.
|
||||||
|
|
||||||
|
## Running an evaluation
|
||||||
|
|
||||||
|
`evaluate()` takes a list of cases and a list of evaluator names. A case is either
|
||||||
|
a `(expected, actual)` tuple or a dict:
|
||||||
|
|
||||||
|
```python
|
||||||
|
{
|
||||||
|
"id": "loan-001", # optional, generated if absent
|
||||||
|
"expected": ..., # optional; some evaluators read it, some don't
|
||||||
|
"actual": ..., # the value under test
|
||||||
|
"config": {...}, # optional, per-evaluator settings for this case
|
||||||
|
"target_fn": callable, # optional, called with the case to produce `actual`
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
If `actual` is missing, the runner calls the case's `target_fn` (or the
|
||||||
|
`target_fn` passed to `evaluate()`) to produce it. Per-case `config` is deep-merged
|
||||||
|
over the top-level `config`, so a case can override one evaluator's settings
|
||||||
|
without discarding the rest.
|
||||||
|
|
||||||
|
```python
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
from semantica.context.decision_models import Decision
|
||||||
|
from semantica.evals import evaluate
|
||||||
|
|
||||||
|
decision = Decision(
|
||||||
|
decision_id="d-1",
|
||||||
|
category="loan",
|
||||||
|
scenario="loan-request",
|
||||||
|
reasoning="vetted against lending policy v3",
|
||||||
|
outcome="approve",
|
||||||
|
confidence=0.87,
|
||||||
|
timestamp=datetime.now(),
|
||||||
|
decision_maker="approver-a",
|
||||||
|
metadata={"provenance": "workflow:loan/v3"},
|
||||||
|
)
|
||||||
|
|
||||||
|
cases = [
|
||||||
|
{
|
||||||
|
"id": "loan-001",
|
||||||
|
"actual": decision,
|
||||||
|
"config": {
|
||||||
|
"decision_scores": {
|
||||||
|
"expected_outcome": "approve",
|
||||||
|
"min_confidence": 0.7,
|
||||||
|
}
|
||||||
|
},
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
summary = evaluate(cases, ["decision_scores"])
|
||||||
|
print(summary.pass_rate) # 1.0
|
||||||
|
```
|
||||||
|
|
||||||
|
Evaluators run independently per case. If one raises, that case's `status` becomes
|
||||||
|
`"error"` and the exception text is captured in the metric's `meta`; the rest of
|
||||||
|
the run continues.
|
||||||
|
|
||||||
|
## Objectives
|
||||||
|
|
||||||
|
By default each evaluator decides its own pass / fail. An **objective** overrides
|
||||||
|
that verdict at the run level, keyed by evaluator name under `config`:
|
||||||
|
|
||||||
|
```python
|
||||||
|
# Raise levenshtein's bar from its default 0.8 to 0.9
|
||||||
|
evaluate(
|
||||||
|
[("apple", "aple")],
|
||||||
|
evaluators=["levenshtein"],
|
||||||
|
config={"levenshtein": {"objective": {"direction": "maximize", "threshold": 0.9}}},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Lower is better
|
||||||
|
evaluate(
|
||||||
|
[("night", "nacht")],
|
||||||
|
evaluators=["levenshtein"],
|
||||||
|
config={"levenshtein": {"objective": {"direction": "minimize", "threshold": 0.5}}},
|
||||||
|
)
|
||||||
|
|
||||||
|
# Expect the metric NOT to match
|
||||||
|
evaluate(
|
||||||
|
[("ok", "ok")],
|
||||||
|
evaluators=["exact_match"],
|
||||||
|
config={"exact_match": {"objective": {"expect": False}}},
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- `maximize` with `threshold`: pass iff `score >= threshold`. `maximize` with no
|
||||||
|
threshold is a no-op and the evaluator's own verdict stands.
|
||||||
|
- `minimize` with `threshold`: pass iff `score <= threshold`. `minimize`
|
||||||
|
**requires** a threshold; omitting it raises `ValueError`.
|
||||||
|
- `expect` (`True` / `False`): pass iff `bool(score)` equals it. Cannot be combined
|
||||||
|
with `direction` or `threshold`, and must be a real boolean.
|
||||||
|
- A metric that already carries an `"error"` in its `meta` is unaffected by any
|
||||||
|
objective.
|
||||||
|
- Invalid objective config is validated for every case before any evaluator runs,
|
||||||
|
so a bad objective fails the whole run up front rather than partway through.
|
||||||
|
|
||||||
|
## Reading the summary
|
||||||
|
|
||||||
|
```python
|
||||||
|
summary = evaluate(cases, ["decision_scores"])
|
||||||
|
|
||||||
|
summary.total, summary.passed, summary.failed, summary.errors
|
||||||
|
summary.pass_rate # passed / total, or 1.0 for an empty case list
|
||||||
|
|
||||||
|
for case in summary.cases:
|
||||||
|
print(case.case_id, case.status) # status: "pass" | "fail" | "error"
|
||||||
|
for name, metric in case.metrics.items():
|
||||||
|
print(name, metric.score, metric.passed)
|
||||||
|
print(metric.meta.get("reasons", {})) # per-sub-check failure reasons
|
||||||
|
```
|
||||||
|
|
||||||
|
`EvalMetric` is frozen (`score: float`, `passed: bool`, `meta: dict`). `CaseResult`
|
||||||
|
is a namedtuple, and `EvalSummary` is a plain dataclass, so all three are
|
||||||
|
straightforward to serialize for logging or regression tracking.
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- `llm_as_judge` needs `config["judge_fn"]`, a callable
|
||||||
|
`judge_fn(actual, expected) -> bool` you supply. No LLM backend is imported
|
||||||
|
unless you pass one in.
|
||||||
|
- `decision_scores` governance checks are opt-in: policy compliance is only
|
||||||
|
evaluated when both `policy_engine` and `policy_id` are present.
|
||||||
|
|
||||||
|
## See also
|
||||||
|
|
||||||
|
- [Decision Intelligence](/guides/decision-intelligence) — producing the `Decision` records this module scores
|
||||||
|
- [Reasoning](/reference/reasoning) — inference output that reasoning-text evaluators can measure
|
||||||
|
- [Policy Engine](/guides/policy-engine) — the `policy_engine` used by `decision_scores`
|
||||||
|
- [Ontology Evaluator](/reference/ontology) — separate tooling for ontology quality metrics
|
||||||
|
|||||||
@@ -403,7 +403,7 @@ Semantic neighborhood requires node embeddings stored in node properties (keys `
|
|||||||
**Session state lost after restart**
|
**Session state lost after restart**
|
||||||
Session state is in-memory only. Use `POST /api/export` to save a JSON snapshot before shutting down.
|
Session state is in-memory only. Use `POST /api/export` to save a JSON snapshot before shutting down.
|
||||||
|
|
||||||
- [Context](context) — Build and save the ContextGraph that Explorer loads.
|
- [Context](/reference/context) — Build and save the ContextGraph that Explorer loads.
|
||||||
- [Ontology](ontology) — Programmatic ontology management and SHACL generation.
|
- [Ontology](ontology) — Programmatic ontology management and SHACL generation.
|
||||||
- [Visualization](visualization) — Programmatic graph rendering without the Explorer server.
|
- [Visualization](visualization) — Programmatic graph rendering without the Explorer server.
|
||||||
- [Export](export) — Export to RDF, Parquet, and other formats without launching a server.
|
- [Export](export) — Export to RDF, Parquet, and other formats without launching a server.
|
||||||
|
|||||||
@@ -394,7 +394,7 @@ The `export_csv` convenience function delegates to `CSVExporter.export()`. For p
|
|||||||
**Match your export format to your consumer.** Neo4j → `cypher`; ArangoDB → `aql`; Gephi/yEd → `graphml` or `gexf`; semantic web tools → `turtle` or `json-ld`; analytics pipelines → `parquet`; zero-copy IPC → `arrow`.
|
**Match your export format to your consumer.** Neo4j → `cypher`; ArangoDB → `aql`; Gephi/yEd → `graphml` or `gexf`; semantic web tools → `turtle` or `json-ld`; analytics pipelines → `parquet`; zero-copy IPC → `arrow`.
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
- [Triplet Store](triplet_store) — Store RDF exports in a SPARQL-queryable backend.
|
- [Triplet Store](/reference/triplet_store) — Store RDF exports in a SPARQL-queryable backend.
|
||||||
- [Ontology](ontology) — Export OWL ontologies.
|
- [Ontology](ontology) — Export OWL ontologies.
|
||||||
- [Provenance](provenance) — Include provenance metadata in RDF exports.
|
- [Provenance](provenance) — Include provenance metadata in RDF exports.
|
||||||
- [Pipeline](pipeline) — Add export as a final pipeline step.
|
- [Pipeline](pipeline) — Add export as a final pipeline step.
|
||||||
|
|||||||
@@ -503,7 +503,7 @@ stats = store.get_stats()
|
|||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
- [KG Module](kg) — Build the graph before persisting it.
|
- [KG Module](/reference/kg) — Build the graph before persisting it.
|
||||||
- [Triplet Store](triplet_store) — RDF triple store for semantic web and SPARQL queries.
|
- [Triplet Store](/reference/triplet_store) — RDF triple store for semantic web and SPARQL queries.
|
||||||
- [Visualization](visualization) — Visualize graphs stored in any backend.
|
- [Visualization](visualization) — Visualize graphs stored in any backend.
|
||||||
- [Context](context) — AgentContext uses GraphStore for memory retrieval.
|
- [Context](/reference/context) — AgentContext uses GraphStore for memory retrieval.
|
||||||
|
|||||||
@@ -646,7 +646,7 @@ from semantica.ingest import ingest_file
|
|||||||
result = ingest_file("source_path", method="my_format")
|
result = ingest_file("source_path", method="my_format")
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Parse](parse) — Parse raw sources into structured text and tables.
|
- [Parse](/reference/parse) — Parse raw sources into structured text and tables.
|
||||||
- [Pipeline](pipeline) — Orchestrate ingest as the first pipeline step.
|
- [Pipeline](pipeline) — Orchestrate ingest as the first pipeline step.
|
||||||
- [Snowflake Integration](../integrations/snowflake) — Snowflake-specific setup and authentication guide.
|
- [Snowflake Integration](../integrations/snowflake) — Snowflake-specific setup and authentication guide.
|
||||||
- [Databricks Integration](../integrations/databricks) — Databricks Unity Catalog setup, authentication, and lineage guide.
|
- [Databricks Integration](../integrations/databricks) — Databricks Unity Catalog setup, authentication, and lineage guide.
|
||||||
|
|||||||
@@ -75,10 +75,10 @@ kg = builder.build({"entities": entities, "relationships": relationships})
|
|||||||
## Temporal Knowledge Graphs (v0.4.0+)
|
## Temporal Knowledge Graphs (v0.4.0+)
|
||||||
|
|
||||||
<Info>
|
<Info>
|
||||||
Full temporal reference including `BiTemporalFact`, `TemporalReasoningEngine`, Allen interval algebra, and `TemporalNormalizer` is covered in the dedicated [Temporal Intelligence](temporal) page. This section documents the KG-layer temporal API.
|
Full temporal reference including `BiTemporalFact`, `TemporalReasoningEngine`, Allen interval algebra, and `TemporalNormalizer` is covered in the dedicated [Temporal Intelligence](/reference/temporal) page. This section documents the KG-layer temporal API.
|
||||||
</Info>
|
</Info>
|
||||||
|
|
||||||
The temporal stack — see the [Temporal Intelligence](temporal) page for the full reference.
|
The temporal stack — see the [Temporal Intelligence](/reference/temporal) page for the full reference.
|
||||||
|
|
||||||
### Building a Temporal Graph
|
### Building a Temporal Graph
|
||||||
|
|
||||||
@@ -264,7 +264,7 @@ versioner.verify_checksum(past_kg)
|
|||||||
```
|
```
|
||||||
|
|
||||||
<Tip>
|
<Tip>
|
||||||
See the [Temporal Intelligence](temporal) reference for the full class API, domain examples (personnel changes, policy evolution, financial timelines), and configuration options.
|
See the [Temporal Intelligence](/reference/temporal) reference for the full class API, domain examples (personnel changes, policy evolution, financial timelines), and configuration options.
|
||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
|
|
||||||
@@ -475,10 +475,10 @@ kg:
|
|||||||
default_validity: infinite
|
default_validity: infinite
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Graph Store](graph_store) — Persist graphs in Neo4j, FalkorDB, or Apache AGE.
|
- [Graph Store](/reference/graph_store) — Persist graphs in Neo4j, FalkorDB, or Apache AGE.
|
||||||
- [Semantic Extract](semantic_extract) — Source of entities and relationships fed to GraphBuilder.
|
- [Semantic Extract](/reference/semantic_extract) — Source of entities and relationships fed to GraphBuilder.
|
||||||
- [Visualization](visualization) — Visualize knowledge graphs interactively.
|
- [Visualization](visualization) — Visualize knowledge graphs interactively.
|
||||||
- [Conflicts](conflicts) — Conflict detection and resolution.
|
- [Conflicts](/reference/conflicts) — Conflict detection and resolution.
|
||||||
|
|
||||||
### Cookbooks
|
### Cookbooks
|
||||||
|
|
||||||
|
|||||||
@@ -439,7 +439,7 @@ extractor = NERExtractor(
|
|||||||
)
|
)
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Semantic Extract](semantic_extract) — Use LLMs for NER and relation extraction.
|
- [Semantic Extract](/reference/semantic_extract) — Use LLMs for NER and relation extraction.
|
||||||
- [Agno Integration](../integrations/agno) — LLM providers in Agno multi-agent teams.
|
- [Agno Integration](../integrations/agno) — LLM providers in Agno multi-agent teams.
|
||||||
- [Reasoning](reasoning) — LLM-backed deductive and abductive reasoning.
|
- [Reasoning](reasoning) — LLM-backed deductive and abductive reasoning.
|
||||||
- [Context](context) — GraphRAG uses LLMs for reasoning over knowledge graphs.
|
- [Context](/reference/context) — GraphRAG uses LLMs for reasoning over knowledge graphs.
|
||||||
|
|||||||
@@ -45,7 +45,7 @@ python -m semantica.mcp_server
|
|||||||
- **Zero Infrastructure** — Runs over stdio: no server, no port, no Docker required. One config block to activate in any MCP client.
|
- **Zero Infrastructure** — Runs over stdio: no server, no port, no Docker required. One config block to activate in any MCP client.
|
||||||
- **Persistent Graphs** — Point `SEMANTICA_KG_PATH` at a saved graph file to reload it automatically on every server startup.
|
- **Persistent Graphs** — Point `SEMANTICA_KG_PATH` at a saved graph file to reload it automatically on every server startup.
|
||||||
- **Decision Intelligence** — Record decisions, find precedents via hybrid similarity search, and trace causal chains across agent runs.
|
- **Decision Intelligence** — Record decisions, find precedents via hybrid similarity search, and trace causal chains across agent runs.
|
||||||
- **REST Alternative** — The [Explorer](explorer) module offers a full HTTP API and browser dashboard if you prefer programmatic access.
|
- **REST Alternative** — The [Explorer](/reference/explorer) module offers a full HTTP API and browser dashboard if you prefer programmatic access.
|
||||||
|
|
||||||
## Installation
|
## Installation
|
||||||
|
|
||||||
@@ -493,7 +493,7 @@ The MCP server exposes three readable resources:
|
|||||||
| `semantica://decisions/list` | All recorded decisions (up to 50) |
|
| `semantica://decisions/list` | All recorded decisions (up to 50) |
|
||||||
| `semantica://schema/info` | Server version and available tools |
|
| `semantica://schema/info` | Server version and available tools |
|
||||||
|
|
||||||
- [Context](context) — The ContextGraph that the MCP server operates on.
|
- [Context](/reference/context) — The ContextGraph that the MCP server operates on.
|
||||||
- [Semantic Extract](semantic_extract) — NER and relation extraction powering the MCP tools.
|
- [Semantic Extract](/reference/semantic_extract) — NER and relation extraction powering the MCP tools.
|
||||||
- [Reasoning](reasoning) — Forward-chaining engine behind run_reasoning.
|
- [Reasoning](reasoning) — Forward-chaining engine behind run_reasoning.
|
||||||
- [Agno Integration](../integrations/agno) — Use Semantica inside Agno multi-agent teams.
|
- [Agno Integration](../integrations/agno) — Use Semantica inside Agno multi-agent teams.
|
||||||
|
|||||||
@@ -584,7 +584,7 @@ normalized = normalize_text("Apple Inc.", method="expand_suffixes")
|
|||||||
# → "Apple Incorporated"
|
# → "Apple Incorporated"
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Parse](parse) — Parse documents before normalization.
|
- [Parse](/reference/parse) — Parse documents before normalization.
|
||||||
- [Split](split) — Chunk normalized text for embedding.
|
- [Split](/reference/split) — Chunk normalized text for embedding.
|
||||||
- [Deduplication](deduplication) — Resolve duplicate entities after normalization.
|
- [Deduplication](deduplication) — Resolve duplicate entities after normalization.
|
||||||
- [Pipeline](pipeline) — Include normalization as a named pipeline step.
|
- [Pipeline](pipeline) — Include normalization as a named pipeline step.
|
||||||
|
|||||||
@@ -287,6 +287,6 @@ ontology_data = ingest_ontology("schema.jsonld") # JSON-LD
|
|||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
- [Reasoning](reasoning) — Apply inference rules over ontology axioms.
|
- [Reasoning](reasoning) — Apply inference rules over ontology axioms.
|
||||||
- [Knowledge Graph](kg) — The graph being modeled by the ontology.
|
- [Knowledge Graph](/reference/kg) — The graph being modeled by the ontology.
|
||||||
- [Export](export) — Export ontologies as RDF, OWL, or JSON-LD.
|
- [Export](export) — Export ontologies as RDF, OWL, or JSON-LD.
|
||||||
- [Conflicts](conflicts) — Detect ontology constraint violations.
|
- [Conflicts](/reference/conflicts) — Detect ontology constraint violations.
|
||||||
|
|||||||
@@ -298,6 +298,6 @@ for source in sources:
|
|||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
- [Ingest](ingest) — Load files before parsing.
|
- [Ingest](ingest) — Load files before parsing.
|
||||||
- [Split](split) — Chunk parsed text for embedding and extraction.
|
- [Split](/reference/split) — Chunk parsed text for embedding and extraction.
|
||||||
- [Docling Integration](../integrations/docling) — Full Docling integration setup guide.
|
- [Docling Integration](../integrations/docling) — Full Docling integration setup guide.
|
||||||
- [Semantic Extract](semantic_extract) — Extract entities and relations from parsed text.
|
- [Semantic Extract](/reference/semantic_extract) — Extract entities and relations from parsed text.
|
||||||
|
|||||||
@@ -497,7 +497,7 @@ result = engine.execute_pipeline(
|
|||||||
|
|
||||||
## SPARQL CONSTRUCT Template Steps
|
## SPARQL CONSTRUCT Template Steps
|
||||||
|
|
||||||
Use the `"construct_template"` step type to render and execute a [SPARQL CONSTRUCT template](triplet_store#sparql-construct-templates) as part of a pipeline. `store_backend` and `construct_template_registry` are execution-time resources, not step config — pass them to `execute_pipeline()`, the same way `delta_mode` steps receive `version_manager` and `triplet_store`:
|
Use the `"construct_template"` step type to render and execute a [SPARQL CONSTRUCT template](/reference/triplet_store#sparql-construct-templates) as part of a pipeline. `store_backend` and `construct_template_registry` are execution-time resources, not step config — pass them to `execute_pipeline()`, the same way `delta_mode` steps receive `version_manager` and `triplet_store`:
|
||||||
|
|
||||||
```python
|
```python
|
||||||
from semantica.pipeline import PipelineBuilder, ExecutionEngine
|
from semantica.pipeline import PipelineBuilder, ExecutionEngine
|
||||||
@@ -589,6 +589,6 @@ StepStatus.SKIPPED # Skipped due to FailureHandler "skip" strategy
|
|||||||
</AccordionGroup>
|
</AccordionGroup>
|
||||||
|
|
||||||
- [Ingest](ingest) — First step in most pipelines.
|
- [Ingest](ingest) — First step in most pipelines.
|
||||||
- [Semantic Extract](semantic_extract) — Core extraction step.
|
- [Semantic Extract](/reference/semantic_extract) — Core extraction step.
|
||||||
- [Knowledge Graph](kg) — Graph construction step.
|
- [Knowledge Graph](/reference/kg) — Graph construction step.
|
||||||
- [Export](export) — Final output step.
|
- [Export](export) — Final output step.
|
||||||
|
|||||||
@@ -522,7 +522,7 @@ Provenance tracking in Semantica produces the following audit artifacts:
|
|||||||
`ProvenanceManager` does not include built-in Turtle or JSON-LD serialization. Use `entry.to_dict()` and `get_lineage()` to retrieve provenance data, then serialize with your preferred RDF library if W3C PROV-O RDF output is required.
|
`ProvenanceManager` does not include built-in Turtle or JSON-LD serialization. Use `entry.to_dict()` and `get_lineage()` to retrieve provenance data, then serialize with your preferred RDF library if W3C PROV-O RDF output is required.
|
||||||
</Note>
|
</Note>
|
||||||
|
|
||||||
- [Change Management](change_management) — Version control and snapshot audit trails.
|
- [Change Management](/reference/change_management) — Version control and snapshot audit trails.
|
||||||
- [Ingest](ingest) — Provenance begins at the ingestion stage.
|
- [Ingest](ingest) — Provenance begins at the ingestion stage.
|
||||||
- [Export](export) — Include provenance metadata in RDF exports.
|
- [Export](export) — Include provenance metadata in RDF exports.
|
||||||
- [Context](context) — Decision provenance via AgentContext.
|
- [Context](/reference/context) — Decision provenance via AgentContext.
|
||||||
|
|||||||
@@ -482,7 +482,7 @@ step.confidence # float
|
|||||||
`GraphReasoner` requires a configured LLM provider. If the provider fails to initialize, `reason()` returns an error string instead of raising. Check `reasoner.provider is not None` before calling if you need to surface failures explicitly.
|
`GraphReasoner` requires a configured LLM provider. If the provider fails to initialize, `reason()` returns an error string instead of raising. Check `reasoner.provider is not None` before calling if you need to surface failures explicitly.
|
||||||
</Warning>
|
</Warning>
|
||||||
|
|
||||||
- [Knowledge Graph](kg) — The knowledge graph being reasoned over.
|
- [Knowledge Graph](/reference/kg) — The knowledge graph being reasoned over.
|
||||||
- [Ontology](ontology) — Ontology axioms and SHACL constraints for logical reasoning.
|
- [Ontology](ontology) — Ontology axioms and SHACL constraints for logical reasoning.
|
||||||
- [Triplet Store](triplet_store) — RDF backend for SPARQL-based reasoning.
|
- [Triplet Store](/reference/triplet_store) — RDF backend for SPARQL-based reasoning.
|
||||||
- [Context](context) — Reasoning integrated into agent decision intelligence.
|
- [Context](/reference/context) — Reasoning integrated into agent decision intelligence.
|
||||||
|
|||||||
@@ -322,6 +322,6 @@ export SEMANTICA_SEED_MERGE_STRATEGY=seed_first
|
|||||||
</Tip>
|
</Tip>
|
||||||
|
|
||||||
- [Ingest](ingest) — Load unstructured data alongside seed data.
|
- [Ingest](ingest) — Load unstructured data alongside seed data.
|
||||||
- [Knowledge Graph](kg) — The target graph that seed data populates.
|
- [Knowledge Graph](/reference/kg) — The target graph that seed data populates.
|
||||||
- [Deduplication](deduplication) — Handle duplicates during seed-extracted merge.
|
- [Deduplication](deduplication) — Handle duplicates during seed-extracted merge.
|
||||||
- [Pipeline](pipeline) — Incorporate seed loading as a named pipeline step.
|
- [Pipeline](pipeline) — Incorporate seed loading as a named pipeline step.
|
||||||
|
|||||||
@@ -410,7 +410,7 @@ triplets = trip.extract(text)
|
|||||||
| `ml` | Fast | Free | High | Limited |
|
| `ml` | Fast | Free | High | Limited |
|
||||||
| `llm` | Medium | API cost | Highest | Yes (schema) |
|
| `llm` | Medium | API cost | Highest | Yes (schema) |
|
||||||
|
|
||||||
- [LLM Providers](llms) — Configure which LLM is used for extraction.
|
- [LLM Providers](/reference/llms) — Configure which LLM is used for extraction.
|
||||||
- [Knowledge Graph](kg) — Build graphs from extracted entities and relationships.
|
- [Knowledge Graph](/reference/kg) — Build graphs from extracted entities and relationships.
|
||||||
- [Parse Module](parse) — Parse documents before extraction.
|
- [Parse Module](/reference/parse) — Parse documents before extraction.
|
||||||
- [Deduplication](deduplication) — Resolve duplicate entities after extraction.
|
- [Deduplication](deduplication) — Resolve duplicate entities after extraction.
|
||||||
|
|||||||
@@ -373,7 +373,7 @@ for chunk in chunks:
|
|||||||
|
|
||||||
For the full pipeline orchestration API, see the [Pipeline reference](pipeline).
|
For the full pipeline orchestration API, see the [Pipeline reference](pipeline).
|
||||||
|
|
||||||
- [Parse](parse) — Parse documents before chunking: produces sections and metadata.
|
- [Parse](/reference/parse) — Parse documents before chunking: produces sections and metadata.
|
||||||
- [Embeddings](embeddings) — Embed chunks for vector search and semantic chunking.
|
- [Embeddings](/reference/embeddings) — Embed chunks for vector search and semantic chunking.
|
||||||
- [Semantic Extract](semantic_extract) — Extract entities and relations from individual chunks.
|
- [Semantic Extract](/reference/semantic_extract) — Extract entities and relations from individual chunks.
|
||||||
- [Pipeline](pipeline) — Integrate splitting as a named pipeline step.
|
- [Pipeline](pipeline) — Integrate splitting as a named pipeline step.
|
||||||
|
|||||||
@@ -874,8 +874,8 @@ kg:
|
|||||||
engine: allen # allen | point_in_time_only
|
engine: allen # allen | point_in_time_only
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Knowledge Graph Module](kg) — Core graph construction, `GraphBuilder`, analytics.
|
- [Knowledge Graph Module](/reference/kg) — Core graph construction, `GraphBuilder`, analytics.
|
||||||
- [Context Module](context) — Decision temporal windows and `find_active_nodes()`.
|
- [Context Module](/reference/context) — Decision temporal windows and `find_active_nodes()`.
|
||||||
- [Provenance](provenance) — W3C PROV-O lineage stamped alongside temporal metadata.
|
- [Provenance](provenance) — W3C PROV-O lineage stamped alongside temporal metadata.
|
||||||
- [Export](export) — OWL, Turtle, JSON-LD, and Parquet export with temporal annotations.
|
- [Export](export) — OWL, Turtle, JSON-LD, and Parquet export with temporal annotations.
|
||||||
|
|
||||||
|
|||||||
@@ -564,4 +564,4 @@ for row in result.bindings:
|
|||||||
- [Export](export) — Export knowledge graphs to RDF formats.
|
- [Export](export) — Export knowledge graphs to RDF formats.
|
||||||
- [Ontology](ontology) — Load OWL ontologies and store as RDF triples.
|
- [Ontology](ontology) — Load OWL ontologies and store as RDF triples.
|
||||||
- [Reasoning](reasoning) — SPARQL-based property chain inference.
|
- [Reasoning](reasoning) — SPARQL-based property chain inference.
|
||||||
- [Graph Store](graph_store) — Property graph alternative for Cypher queries.
|
- [Graph Store](/reference/graph_store) — Property graph alternative for Cypher queries.
|
||||||
|
|||||||
@@ -222,5 +222,5 @@ from semantica.utils import read_json_file
|
|||||||
config = read_json_file("config.json")
|
config = read_json_file("config.json")
|
||||||
```
|
```
|
||||||
|
|
||||||
- [Core](core) — Framework orchestration that uses Utils internally.
|
- [Core](/reference/core) — Framework orchestration that uses Utils internally.
|
||||||
- [Pipeline](pipeline) — Uses ProgressTracker for per-step tracking.
|
- [Pipeline](pipeline) — Uses ProgressTracker for per-step tracking.
|
||||||
|
|||||||
@@ -588,7 +588,7 @@ store.create_index(index_type="pq", metric="L2", m=8)
|
|||||||
</Tab>
|
</Tab>
|
||||||
</Tabs>
|
</Tabs>
|
||||||
|
|
||||||
- [Embeddings](embeddings) — Generate the vectors stored here.
|
- [Embeddings](/reference/embeddings) — Generate the vectors stored here.
|
||||||
- [Context](context) — AgentContext uses VectorStore for memory retrieval.
|
- [Context](/reference/context) — AgentContext uses VectorStore for memory retrieval.
|
||||||
- [Split](split) — Chunk documents before embedding and storing.
|
- [Split](/reference/split) — Chunk documents before embedding and storing.
|
||||||
- [Ingest](ingest) — Ingest documents before embedding and storing.
|
- [Ingest](ingest) — Ingest documents before embedding and storing.
|
||||||
|
|||||||
@@ -288,9 +288,9 @@ For a full browser-based UI with search, path finding, and the Ontology Hub, lau
|
|||||||
semantica-explorer --graph my_graph.json
|
semantica-explorer --graph my_graph.json
|
||||||
```
|
```
|
||||||
|
|
||||||
See the [Explorer reference](explorer) for the full feature set and REST API.
|
See the [Explorer reference](/reference/explorer) for the full feature set and REST API.
|
||||||
|
|
||||||
- [Knowledge Graph](kg) — The graph being visualized.
|
- [Knowledge Graph](/reference/kg) — The graph being visualized.
|
||||||
- [Ontology](ontology) — Visualize ontology class structure.
|
- [Ontology](ontology) — Visualize ontology class structure.
|
||||||
- [Embeddings](embeddings) — Generate the embeddings visualized here.
|
- [Embeddings](/reference/embeddings) — Generate the embeddings visualized here.
|
||||||
- [Explorer](explorer) — Full interactive Knowledge Explorer UI.
|
- [Explorer](/reference/explorer) — Full interactive Knowledge Explorer UI.
|
||||||
|
|||||||
@@ -588,16 +588,15 @@ class AgentMemory:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
# Remove from vector store unless a caller is staging an atomic local update.
|
# Remove from vector store unless a caller is staging an atomic local update.
|
||||||
if not skip_vector:
|
if not skip_vector and self.vector_store:
|
||||||
if self.vector_store:
|
try:
|
||||||
try:
|
vector_ids = list(self._vector_ids.get(memory_id, [])) or [memory_id]
|
||||||
vector_ids = list(self._vector_ids.get(memory_id, [])) or [
|
self._delete_vector_ids(vector_ids)
|
||||||
memory_id
|
except Exception as e:
|
||||||
]
|
self.logger.warning(f"Failed to delete from vector store: {e}")
|
||||||
self._delete_vector_ids(vector_ids)
|
# Bookkeeping runs unconditionally: a skip_vector delete still removes the
|
||||||
except Exception as e:
|
# item, so leaving its tracked ids behind would orphan them permanently.
|
||||||
self.logger.warning(f"Failed to delete from vector store: {e}")
|
self._vector_ids.pop(memory_id, None)
|
||||||
self._vector_ids.pop(memory_id, None)
|
|
||||||
|
|
||||||
memory_item = self.memory_items[memory_id]
|
memory_item = self.memory_items[memory_id]
|
||||||
|
|
||||||
@@ -1588,12 +1587,16 @@ class AgentMemory:
|
|||||||
memory_ids.append(memory_id)
|
memory_ids.append(memory_id)
|
||||||
return memory_ids
|
return memory_ids
|
||||||
|
|
||||||
def batch_delete(self, memory_ids: List[str]) -> int:
|
def batch_delete(self, memory_ids: List[str], *, skip_vector: bool = False) -> int:
|
||||||
"""
|
"""
|
||||||
Batch delete.
|
Batch delete.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
memory_ids: List of memory IDs to delete
|
memory_ids: List of memory IDs to delete
|
||||||
|
skip_vector: If True, skip each item's own vector-store cascade
|
||||||
|
(see ``delete_memory``). A caller that is already erasing these
|
||||||
|
ids' vectors itself passes this to avoid a redundant,
|
||||||
|
best-effort delete against the vector store.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Number of memories deleted
|
Number of memories deleted
|
||||||
@@ -1603,7 +1606,7 @@ class AgentMemory:
|
|||||||
"""
|
"""
|
||||||
deleted = 0
|
deleted = 0
|
||||||
for memory_id in memory_ids:
|
for memory_id in memory_ids:
|
||||||
if self.delete_memory(memory_id):
|
if self.delete_memory(memory_id, skip_vector=skip_vector):
|
||||||
deleted += 1
|
deleted += 1
|
||||||
return deleted
|
return deleted
|
||||||
|
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ Example:
|
|||||||
'unsupported'
|
'unsupported'
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import inspect
|
||||||
from dataclasses import dataclass, field
|
from dataclasses import dataclass, field
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple, Union
|
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple, Union
|
||||||
@@ -150,6 +151,24 @@ class ErasureCoordinator:
|
|||||||
more than actually occurred. Erasing the graph last means a partial
|
more than actually occurred. Erasing the graph last means a partial
|
||||||
failure leaves the node present and the receipt incomplete, which is
|
failure leaves the node present and the receipt incomplete, which is
|
||||||
recoverable and honest.
|
recoverable and honest.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
An explicit ``vector_store=False`` also suppresses ``AgentMemory``'s
|
||||||
|
own internal vector cascade, not just the coordinator's leg (#1378).
|
||||||
|
``AgentMemory.delete_memory()`` deletes an item's vectors best-effort:
|
||||||
|
it catches a vector-store failure, logs it, and still returns ``True``,
|
||||||
|
so without this a caller who opted out of the vector leg could still
|
||||||
|
have ``memory.vector_store`` mutated underneath them while the receipt
|
||||||
|
read ``vectors: not_configured``. ``vector_store=False`` is taken to
|
||||||
|
mean "no vector activity at all", so the coordinator passes
|
||||||
|
``skip_vector=True`` through to ``memory.batch_delete()`` in that case,
|
||||||
|
and ``receipt.stores["vectors"]["status"]`` stays ``"not_configured"``
|
||||||
|
honestly -- the caller opted the vector store out entirely, rather than
|
||||||
|
the coordinator having erased it. This only applies when
|
||||||
|
``vector_store=False`` was passed explicitly; when no vector store
|
||||||
|
exists anywhere (no ``memory`` was supplied, or ``memory`` has no
|
||||||
|
``vector_store`` attribute), there is nothing to suppress and
|
||||||
|
``memory.batch_delete()`` is called as before.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
@@ -170,6 +189,12 @@ class ErasureCoordinator:
|
|||||||
|
|
||||||
self.graph = graph
|
self.graph = graph
|
||||||
self.memory = memory
|
self.memory = memory
|
||||||
|
# Distinct from `self.vector_store is None`: that's also true when no
|
||||||
|
# vector store exists anywhere (no memory, or memory with no
|
||||||
|
# vector_store attribute), where there is nothing to suppress and
|
||||||
|
# forcing skip_vector onto a duck-typed memory would break callers
|
||||||
|
# whose batch_delete() doesn't accept that kwarg.
|
||||||
|
self._vector_leg_disabled = vector_store is False
|
||||||
if vector_store is False:
|
if vector_store is False:
|
||||||
self.vector_store: Optional[Any] = None
|
self.vector_store: Optional[Any] = None
|
||||||
elif vector_store is not None:
|
elif vector_store is not None:
|
||||||
@@ -424,6 +449,20 @@ class ErasureCoordinator:
|
|||||||
return {"status": STATUS_NOT_CONFIGURED}
|
return {"status": STATUS_NOT_CONFIGURED}
|
||||||
|
|
||||||
deleted = 0
|
deleted = 0
|
||||||
|
skip_vector = self._vector_leg_disabled and _accepts_skip_vector(
|
||||||
|
self.memory.batch_delete
|
||||||
|
)
|
||||||
|
if self._vector_leg_disabled and not skip_vector:
|
||||||
|
# The class docstring only requires find_by_entity/batch_delete; a
|
||||||
|
# duck-typed adapter is not required to support skip_vector. Falling
|
||||||
|
# back to the plain call keeps the memory leg working -- the
|
||||||
|
# adapter's own cascade (if it has one) just can't be suppressed.
|
||||||
|
self.logger.warning(
|
||||||
|
"Memory adapter %r has no skip_vector support; its own vector "
|
||||||
|
"cascade (if any) could not be suppressed for %r",
|
||||||
|
type(self.memory).__name__,
|
||||||
|
entity_id,
|
||||||
|
)
|
||||||
try:
|
try:
|
||||||
# Sweep in pages until dry rather than passing one large limit:
|
# Sweep in pages until dry rather than passing one large limit:
|
||||||
# ``find_by_entity`` has historically defaulted to ``limit=10`` and
|
# ``find_by_entity`` has historically defaulted to ``limit=10`` and
|
||||||
@@ -454,7 +493,10 @@ class ErasureCoordinator:
|
|||||||
"detail": "memory items carry no 'memory_id'",
|
"detail": "memory items carry no 'memory_id'",
|
||||||
}
|
}
|
||||||
|
|
||||||
removed = self.memory.batch_delete(memory_ids)
|
if skip_vector:
|
||||||
|
removed = self.memory.batch_delete(memory_ids, skip_vector=True)
|
||||||
|
else:
|
||||||
|
removed = self.memory.batch_delete(memory_ids)
|
||||||
deleted += removed
|
deleted += removed
|
||||||
if removed == 0:
|
if removed == 0:
|
||||||
# No progress: another page would return the same items.
|
# No progress: another page would return the same items.
|
||||||
@@ -564,6 +606,25 @@ def _memory_item_id(item: Any) -> Optional[str]:
|
|||||||
return str(memory_id) if memory_id else None
|
return str(memory_id) if memory_id else None
|
||||||
|
|
||||||
|
|
||||||
|
def _accepts_skip_vector(batch_delete: Any) -> bool:
|
||||||
|
"""True when ``batch_delete`` takes a ``skip_vector`` keyword.
|
||||||
|
|
||||||
|
``skip_vector`` is an ``AgentMemory``-specific extension, not part of the
|
||||||
|
duck-typed contract the class docstring promises (``find_by_entity`` and
|
||||||
|
``batch_delete`` only). Passing it to an adapter that doesn't accept it
|
||||||
|
would raise ``TypeError`` and fail the whole memory leg, so this is
|
||||||
|
checked before ever passing the kwarg.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
signature = inspect.signature(batch_delete)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return False
|
||||||
|
for parameter in signature.parameters.values():
|
||||||
|
if parameter.name == "skip_vector" or parameter.kind == inspect.Parameter.VAR_KEYWORD:
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
#: Dict keys a backend uses to report whether a delete succeeded, and the
|
#: Dict keys a backend uses to report whether a delete succeeded, and the
|
||||||
#: values that mean it did not. Qdrant returns ``{"status": <UpdateStatus>}``
|
#: values that mean it did not. Qdrant returns ``{"status": <UpdateStatus>}``
|
||||||
#: and Pinecone ``{"deleted": True}``; neither is a bool, so a bare
|
#: and Pinecone ``{"deleted": True}``; neither is a bool, so a bare
|
||||||
|
|||||||
@@ -253,7 +253,7 @@ class GraphAnalyzer:
|
|||||||
graph,
|
graph,
|
||||||
start_time=None,
|
start_time=None,
|
||||||
end_time=None,
|
end_time=None,
|
||||||
metrics=["node_count", "edge_count", "density", "communities"],
|
metrics=None,
|
||||||
interval=None,
|
interval=None,
|
||||||
**options,
|
**options,
|
||||||
):
|
):
|
||||||
@@ -271,6 +271,8 @@ class GraphAnalyzer:
|
|||||||
Returns:
|
Returns:
|
||||||
Evolution analysis results with time series data
|
Evolution analysis results with time series data
|
||||||
"""
|
"""
|
||||||
|
if metrics is None:
|
||||||
|
metrics = ["node_count", "edge_count", "density", "communities"]
|
||||||
self.logger.info("Analyzing temporal evolution")
|
self.logger.info("Analyzing temporal evolution")
|
||||||
|
|
||||||
from .temporal_query import TemporalGraphQuery
|
from .temporal_query import TemporalGraphQuery
|
||||||
|
|||||||
@@ -375,7 +375,7 @@ class HierarchicalChunker:
|
|||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
levels: List[str] = ["section", "paragraph", "sentence"],
|
levels: Optional[List[str]] = None,
|
||||||
chunk_sizes: Optional[List[int]] = None,
|
chunk_sizes: Optional[List[int]] = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
@@ -387,6 +387,8 @@ class HierarchicalChunker:
|
|||||||
chunk_sizes: Chunk sizes for each level
|
chunk_sizes: Chunk sizes for each level
|
||||||
**kwargs: Additional options
|
**kwargs: Additional options
|
||||||
"""
|
"""
|
||||||
|
if levels is None:
|
||||||
|
levels = ["section", "paragraph", "sentence"]
|
||||||
self.levels = levels
|
self.levels = levels
|
||||||
self.chunk_sizes = chunk_sizes or [2000, 1000, 500]
|
self.chunk_sizes = chunk_sizes or [2000, 1000, 500]
|
||||||
self.options = kwargs
|
self.options = kwargs
|
||||||
|
|||||||
@@ -1402,7 +1402,7 @@ def split_embedding_semantic(
|
|||||||
|
|
||||||
def split_hierarchical(
|
def split_hierarchical(
|
||||||
text: str,
|
text: str,
|
||||||
levels: List[str] = ["section", "paragraph", "sentence"],
|
levels: Optional[List[str]] = None,
|
||||||
chunk_sizes: Optional[List[int]] = None,
|
chunk_sizes: Optional[List[int]] = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
) -> List[Chunk]:
|
) -> List[Chunk]:
|
||||||
@@ -1420,6 +1420,8 @@ def split_hierarchical(
|
|||||||
"""
|
"""
|
||||||
if chunk_sizes is None:
|
if chunk_sizes is None:
|
||||||
chunk_sizes = [2000, 1000, 500]
|
chunk_sizes = [2000, 1000, 500]
|
||||||
|
if levels is None:
|
||||||
|
levels = ["section", "paragraph", "sentence"]
|
||||||
|
|
||||||
# Start with largest level
|
# Start with largest level
|
||||||
if "section" in levels:
|
if "section" in levels:
|
||||||
|
|||||||
@@ -678,7 +678,7 @@ class TestSeparateVectorStoreHandling(unittest.TestCase):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
def test_vector_store_false_disables_vector_leg_entirely(self):
|
def test_vector_store_false_disables_vector_leg_entirely(self):
|
||||||
"""vector_store=False must disable the vector leg, not try memory.vector_store."""
|
"""vector_store=False must disable the vector leg AND memory's own cascade (#1378)."""
|
||||||
memory_store = _SelectiveDeleteStore()
|
memory_store = _SelectiveDeleteStore()
|
||||||
memory = _memory_with_embedding("customer-4471", memory_store)
|
memory = _memory_with_embedding("customer-4471", memory_store)
|
||||||
|
|
||||||
@@ -689,8 +689,91 @@ class TestSeparateVectorStoreHandling(unittest.TestCase):
|
|||||||
|
|
||||||
# Vector leg should report not_configured, not attempt deletion
|
# Vector leg should report not_configured, not attempt deletion
|
||||||
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
# Memory's own cascade still runs, but coordinator doesn't track it
|
|
||||||
self.assertTrue(receipt.complete)
|
self.assertTrue(receipt.complete)
|
||||||
|
# Memory's own internal vector cascade must be suppressed too, not just
|
||||||
|
# unreported: the embedding memory owns is left untouched, and the
|
||||||
|
# backend's delete method is never even called.
|
||||||
|
self.assertEqual(memory_store.attempts, [])
|
||||||
|
self.assertTrue(memory_store.live)
|
||||||
|
|
||||||
|
def test_vector_store_false_regression_refusing_backend_never_called(self):
|
||||||
|
"""Regression for #1378: a refusing backend must not be called at all.
|
||||||
|
|
||||||
|
Reproduces the exact bug report -- a vector store whose delete_vectors()
|
||||||
|
always returns False (refuses) bound as memory.vector_store, with the
|
||||||
|
coordinator's own vector leg disabled via vector_store=False. Before the
|
||||||
|
fix, delete_memory()'s internal cascade would still call the refusing
|
||||||
|
store, catch the failure, log a warning, and return True regardless --
|
||||||
|
so receipt.complete read True while the embedding stayed live and the
|
||||||
|
backend had in fact been asked to delete it. Pinned here so the delete
|
||||||
|
method call count can't silently regress back to nonzero.
|
||||||
|
"""
|
||||||
|
refusing_store = _SelectiveDeleteStore(refuse={"vec-0"})
|
||||||
|
memory = _memory_with_embedding("customer-4471", refusing_store)
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
memory=memory, vector_store=False
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertTrue(receipt.complete)
|
||||||
|
self.assertEqual(receipt.stores["vectors"]["status"], STATUS_NOT_CONFIGURED)
|
||||||
|
self.assertEqual(len(refusing_store.attempts), 0) # delete_calls == 0
|
||||||
|
|
||||||
|
def test_skip_vector_deletion_does_not_orphan_local_vector_id_tracking(self):
|
||||||
|
"""skip_vector=True must still pop the item's own _vector_ids entry.
|
||||||
|
|
||||||
|
Regression: delete_memory(skip_vector=True) used to leave the item's
|
||||||
|
entry in AgentMemory._vector_ids behind since the pop() lived inside
|
||||||
|
the `if not skip_vector` block alongside the actual vector-store
|
||||||
|
delete. That orphaned entry never got cleaned up and leaked into
|
||||||
|
to_dict()/from_dict() snapshots.
|
||||||
|
"""
|
||||||
|
memory = _memory_with_embedding("customer-4471", _SelectiveDeleteStore())
|
||||||
|
memory_id = next(iter(memory.memory_items))
|
||||||
|
self.assertIn(memory_id, memory._vector_ids)
|
||||||
|
|
||||||
|
ErasureCoordinator(memory=memory, vector_store=False).erase_entity(
|
||||||
|
"customer-4471"
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertNotIn(memory_id, memory.memory_items)
|
||||||
|
self.assertNotIn(memory_id, memory._vector_ids)
|
||||||
|
|
||||||
|
def test_memory_adapter_without_skip_vector_support_is_not_broken(self):
|
||||||
|
"""A duck-typed memory whose batch_delete() lacks skip_vector must still work.
|
||||||
|
|
||||||
|
The class docstring only requires find_by_entity and batch_delete; an
|
||||||
|
adapter is not obligated to support skip_vector. The coordinator must
|
||||||
|
detect that and fall back to the plain call rather than raising
|
||||||
|
TypeError and failing the whole memory leg.
|
||||||
|
"""
|
||||||
|
|
||||||
|
class _PlainAdapter:
|
||||||
|
def __init__(self):
|
||||||
|
self.items = {"m1": {"memory_id": "m1", "entities": [{"id": "customer-4471"}]}}
|
||||||
|
|
||||||
|
def find_by_entity(self, entity_id, limit=None):
|
||||||
|
return [
|
||||||
|
item
|
||||||
|
for item in self.items.values()
|
||||||
|
if any(e.get("id") == entity_id for e in item.get("entities", []))
|
||||||
|
]
|
||||||
|
|
||||||
|
def batch_delete(self, memory_ids):
|
||||||
|
removed = 0
|
||||||
|
for memory_id in memory_ids:
|
||||||
|
if self.items.pop(memory_id, None) is not None:
|
||||||
|
removed += 1
|
||||||
|
return removed
|
||||||
|
|
||||||
|
adapter = _PlainAdapter()
|
||||||
|
|
||||||
|
receipt = ErasureCoordinator(
|
||||||
|
memory=adapter, vector_store=False
|
||||||
|
).erase_entity("customer-4471")
|
||||||
|
|
||||||
|
self.assertEqual(receipt.stores["memory"]["status"], STATUS_ERASED)
|
||||||
|
self.assertEqual(adapter.items, {})
|
||||||
|
|
||||||
def test_separate_vector_store_only_handles_coordinator_store(self):
|
def test_separate_vector_store_only_handles_coordinator_store(self):
|
||||||
"""When coordinator has a different vector_store, it only handles that one.
|
"""When coordinator has a different vector_store, it only handles that one.
|
||||||
|
|||||||
@@ -242,6 +242,107 @@ class TestGraphAnalyzer(unittest.TestCase):
|
|||||||
self.mock_connectivity.analyze_connectivity.assert_called_once()
|
self.mock_connectivity.analyze_connectivity.assert_called_once()
|
||||||
mock_metrics.assert_called_once()
|
mock_metrics.assert_called_once()
|
||||||
|
|
||||||
|
class TestAnalyzeTemporalEvolutionMutableDefault(unittest.TestCase):
|
||||||
|
"""Regression tests for fix: replace mutable default argument in
|
||||||
|
GraphAnalyzer.analyze_temporal_evolution (metrics=[...] -> None).
|
||||||
|
|
||||||
|
TemporalGraphQuery is imported lazily inside the method body
|
||||||
|
(``from .temporal_query import TemporalGraphQuery``), so it is patched
|
||||||
|
at its definition site: ``semantica.kg.temporal_query.TemporalGraphQuery``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.mock_tracker_patcher = patch("semantica.kg.graph_analyzer.get_progress_tracker")
|
||||||
|
self.mock_get_tracker = self.mock_tracker_patcher.start()
|
||||||
|
self.mock_get_tracker.return_value = MagicMock()
|
||||||
|
|
||||||
|
self.mock_centrality_patcher = patch("semantica.kg.graph_analyzer.CentralityCalculator")
|
||||||
|
self.mock_centrality_patcher.start()
|
||||||
|
|
||||||
|
self.mock_community_patcher = patch("semantica.kg.graph_analyzer.CommunityDetector")
|
||||||
|
self.mock_community_patcher.start()
|
||||||
|
|
||||||
|
self.mock_connectivity_patcher = patch("semantica.kg.graph_analyzer.ConnectivityAnalyzer")
|
||||||
|
self.mock_connectivity_patcher.start()
|
||||||
|
|
||||||
|
# TemporalGraphQuery is imported *inside* the method body, so patch it
|
||||||
|
# at the definition module rather than at the caller module.
|
||||||
|
self.mock_tq_patcher = patch(
|
||||||
|
"semantica.kg.temporal_query.TemporalGraphQuery", autospec=False
|
||||||
|
)
|
||||||
|
mock_tq_cls = self.mock_tq_patcher.start()
|
||||||
|
self.mock_tq = MagicMock()
|
||||||
|
self.mock_tq.analyze_evolution.return_value = {"snapshots": []}
|
||||||
|
mock_tq_cls.return_value = self.mock_tq
|
||||||
|
|
||||||
|
def tearDown(self):
|
||||||
|
patch.stopall()
|
||||||
|
|
||||||
|
def _make_analyzer(self):
|
||||||
|
return GraphAnalyzer()
|
||||||
|
|
||||||
|
def test_default_metrics_value_is_canonical(self):
|
||||||
|
"""When metrics=None, the four canonical metric names must be used."""
|
||||||
|
analyzer = self._make_analyzer()
|
||||||
|
graph = {"entities": [], "relationships": []}
|
||||||
|
result = analyzer.analyze_temporal_evolution(graph)
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
sorted(result["metrics_tracked"]),
|
||||||
|
sorted(["node_count", "edge_count", "density", "communities"]),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_default_metrics_independent_across_calls(self):
|
||||||
|
"""Mutating the returned metrics_tracked list must not affect the next call."""
|
||||||
|
analyzer = self._make_analyzer()
|
||||||
|
graph = {"entities": [], "relationships": []}
|
||||||
|
|
||||||
|
result1 = analyzer.analyze_temporal_evolution(graph)
|
||||||
|
# Mutate the returned list in-place.
|
||||||
|
result1["metrics_tracked"].append("MUTATED")
|
||||||
|
|
||||||
|
result2 = analyzer.analyze_temporal_evolution(graph)
|
||||||
|
self.assertNotIn(
|
||||||
|
"MUTATED",
|
||||||
|
result2["metrics_tracked"],
|
||||||
|
"Mutable default leaked: 'MUTATED' appeared in the second call's metrics list",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_result_contains_metrics_tracked_key(self):
|
||||||
|
"""Return value must include 'metrics_tracked' with the default list."""
|
||||||
|
analyzer = self._make_analyzer()
|
||||||
|
graph = {"entities": [], "relationships": []}
|
||||||
|
result = analyzer.analyze_temporal_evolution(graph)
|
||||||
|
|
||||||
|
self.assertIn("metrics_tracked", result)
|
||||||
|
self.assertEqual(
|
||||||
|
sorted(result["metrics_tracked"]),
|
||||||
|
sorted(["node_count", "edge_count", "density", "communities"]),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_explicit_metrics_override_is_respected(self):
|
||||||
|
"""Explicitly passed metrics must be forwarded and reflected in the return value."""
|
||||||
|
analyzer = self._make_analyzer()
|
||||||
|
graph = {"entities": [], "relationships": []}
|
||||||
|
custom = ["node_count"]
|
||||||
|
result = analyzer.analyze_temporal_evolution(graph, metrics=custom)
|
||||||
|
|
||||||
|
self.assertEqual(result["metrics_tracked"], custom)
|
||||||
|
|
||||||
|
def test_explicit_metrics_mutation_does_not_affect_default(self):
|
||||||
|
"""Mutating the list passed as an explicit argument must not corrupt
|
||||||
|
a subsequent default call."""
|
||||||
|
analyzer = self._make_analyzer()
|
||||||
|
graph = {"entities": [], "relationships": []}
|
||||||
|
|
||||||
|
explicit = ["node_count"]
|
||||||
|
analyzer.analyze_temporal_evolution(graph, metrics=explicit)
|
||||||
|
explicit.append("MUTATED")
|
||||||
|
|
||||||
|
result = analyzer.analyze_temporal_evolution(graph)
|
||||||
|
self.assertNotIn("MUTATED", result["metrics_tracked"])
|
||||||
|
|
||||||
|
|
||||||
class TestTemporalGraphQuery(unittest.TestCase):
|
class TestTemporalGraphQuery(unittest.TestCase):
|
||||||
def setUp(self):
|
def setUp(self):
|
||||||
self.mock_tracker_patcher = patch("semantica.utils.progress_tracker.get_progress_tracker")
|
self.mock_tracker_patcher = patch("semantica.utils.progress_tracker.get_progress_tracker")
|
||||||
|
|||||||
@@ -638,3 +638,89 @@ Body paragraph under a distinct heading for separation checks.
|
|||||||
self.SAMPLE * 3, chunk_size=80, ner_method="pattern"
|
self.SAMPLE * 3, chunk_size=80, ner_method="pattern"
|
||||||
)
|
)
|
||||||
assert len(chunks) >= 1
|
assert len(chunks) >= 1
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Mutable-default regression tests (fix: replace mutable default arguments)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class TestMutableDefaultRegression:
|
||||||
|
"""Regression tests proving that mutable default arguments do not leak
|
||||||
|
between calls. Each test mutates the list returned / stored by one call
|
||||||
|
and verifies that a subsequent call still receives the *original* default
|
||||||
|
value, not the mutated one.
|
||||||
|
"""
|
||||||
|
|
||||||
|
# --- split_hierarchical -------------------------------------------------
|
||||||
|
|
||||||
|
def test_split_hierarchical_default_levels_are_independent_across_calls(self):
|
||||||
|
"""Mutating the levels list from one call must not affect the next."""
|
||||||
|
text = "Para one.\n\nPara two.\n\nPara three."
|
||||||
|
|
||||||
|
# First call – capture and mutate the levels list indirectly by
|
||||||
|
# passing explicit levels and then appending to a reference.
|
||||||
|
call1_levels: list = ["paragraph"]
|
||||||
|
chunks1 = split_hierarchical(text, levels=call1_levels, chunk_sizes=[1000])
|
||||||
|
# Mutate the list that was passed in.
|
||||||
|
call1_levels.append("MUTATED")
|
||||||
|
|
||||||
|
# Second call with default levels=None must still use the canonical default.
|
||||||
|
chunks2 = split_hierarchical(text)
|
||||||
|
# The function must succeed and produce chunks (not raise because
|
||||||
|
# "MUTATED" is not a valid level name).
|
||||||
|
assert len(chunks2) >= 1
|
||||||
|
|
||||||
|
def test_split_hierarchical_none_default_creates_fresh_list_each_call(self):
|
||||||
|
"""Two calls with levels=None must receive independent list objects."""
|
||||||
|
text = "A sentence.\n\nAnother sentence."
|
||||||
|
|
||||||
|
# Patch the body assignment so we can capture it.
|
||||||
|
captured: list = []
|
||||||
|
original_fn = split_hierarchical.__wrapped__ if hasattr(split_hierarchical, "__wrapped__") else None
|
||||||
|
|
||||||
|
# Use a simpler black-box approach: call twice and verify behaviour.
|
||||||
|
chunks_a = split_hierarchical(text)
|
||||||
|
chunks_b = split_hierarchical(text)
|
||||||
|
|
||||||
|
# Both calls should produce identical results (same default).
|
||||||
|
assert len(chunks_a) == len(chunks_b)
|
||||||
|
assert [c.text for c in chunks_a] == [c.text for c in chunks_b]
|
||||||
|
|
||||||
|
def test_split_hierarchical_default_chunk_sizes_are_independent_across_calls(self):
|
||||||
|
"""Mutating chunk_sizes in one call must not affect the next."""
|
||||||
|
text = "Para A.\n\nPara B."
|
||||||
|
mutable_sizes = [5000, 2000, 1000]
|
||||||
|
split_hierarchical(text, chunk_sizes=mutable_sizes)
|
||||||
|
# Mutate after first call.
|
||||||
|
mutable_sizes[0] = 1 # Would produce very different chunking if leaked.
|
||||||
|
|
||||||
|
# Second call with default chunk_sizes=None must still use canonical defaults.
|
||||||
|
chunks = split_hierarchical(text)
|
||||||
|
assert len(chunks) >= 1
|
||||||
|
|
||||||
|
# --- HierarchicalChunker ------------------------------------------------
|
||||||
|
|
||||||
|
def test_hierarchical_chunker_default_levels_independent_across_instances(self):
|
||||||
|
"""Mutating levels on one instance must not affect a second instance
|
||||||
|
created with the default."""
|
||||||
|
chunker_a = HierarchicalChunker()
|
||||||
|
# Mutate the instance attribute that was built from the default.
|
||||||
|
chunker_a.levels.append("MUTATED")
|
||||||
|
|
||||||
|
chunker_b = HierarchicalChunker()
|
||||||
|
assert "MUTATED" not in chunker_b.levels, (
|
||||||
|
"Mutation of chunker_a.levels leaked into chunker_b — "
|
||||||
|
"mutable default not fixed properly"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_hierarchical_chunker_default_levels_value(self):
|
||||||
|
"""Default levels must equal the canonical list."""
|
||||||
|
chunker = HierarchicalChunker()
|
||||||
|
assert chunker.levels == ["section", "paragraph", "sentence"]
|
||||||
|
|
||||||
|
def test_hierarchical_chunker_explicit_levels_preserved(self):
|
||||||
|
"""Explicitly passed levels must be stored as given."""
|
||||||
|
custom = ["document", "paragraph"]
|
||||||
|
chunker = HierarchicalChunker(levels=custom)
|
||||||
|
assert chunker.levels == custom
|
||||||
|
|||||||
Reference in New Issue
Block a user