mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-08-30 04:40:16 +00:00
Compare commits
363
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9a293bb67c | ||
|
|
44f585ffce | ||
|
|
f5332589d5 | ||
|
|
9a21ca9834 | ||
|
|
a169cf3fb9 | ||
|
|
656baa7aee | ||
|
|
e1725fd763 | ||
|
|
7ed1d49625 | ||
|
|
c94be3f9a6 | ||
|
|
142707db93 | ||
|
|
3357c14ee3 | ||
|
|
3496d62335 | ||
|
|
64f6c5cba2 | ||
|
|
ab5c12f9af | ||
|
|
fc9af2ebf8 | ||
|
|
3e9ba1b7fb | ||
|
|
51cf97765d | ||
|
|
8fa2037619 | ||
|
|
5573ab7a9f | ||
|
|
26f236923c | ||
|
|
c35899711d | ||
|
|
0a113b9702 | ||
|
|
2de6ff898a | ||
|
|
22ea189d0b | ||
|
|
30d5fef180 | ||
|
|
c85df419ae | ||
|
|
55f3ee6f84 | ||
|
|
924765b042 | ||
|
|
6f310d1d7a | ||
|
|
bd35d6031b | ||
|
|
53caefaf58 | ||
|
|
fc2083aa17 | ||
|
|
1258edfe7f | ||
|
|
5048665d35 | ||
|
|
7dce9f1b69 | ||
|
|
1b09f1ca5b | ||
|
|
03ed4b94e9 | ||
|
|
9059a44731 | ||
|
|
e90bd048e1 | ||
|
|
8e0419c864 | ||
|
|
0d51608547 | ||
|
|
aa7b7fe525 | ||
|
|
916d3974e3 | ||
|
|
f75469f472 | ||
|
|
40b81d0582 | ||
|
|
5db0adc18a | ||
|
|
50758f6f25 | ||
|
|
2756916573 | ||
|
|
f47c730f7e | ||
|
|
721a2f0e9c | ||
|
|
dd42b7fa95 | ||
|
|
248d028b09 | ||
|
|
77a2ab7b18 | ||
|
|
c8b59b47f5 | ||
|
|
c0b6a80480 | ||
|
|
36071819b5 | ||
|
|
a4dac2342b | ||
|
|
7d272f40e8 | ||
|
|
3e5d2672ad | ||
|
|
b4b10a4928 | ||
|
|
48c58a0753 | ||
|
|
6b143ef401 | ||
|
|
49f458e927 | ||
|
|
c77184bd77 | ||
|
|
fa77f5cc47 | ||
|
|
7bddee0111 | ||
|
|
5cd4407e57 | ||
|
|
1850cdd617 | ||
|
|
5f00c00be3 | ||
|
|
dee55112ef | ||
|
|
b7ac05b6f2 | ||
|
|
d9118410bc | ||
|
|
d16db085d8 | ||
|
|
cb716cec61 | ||
|
|
712a6e6d4c | ||
|
|
b52cdd5bbe | ||
|
|
d0e018a1c9 | ||
|
|
bd8d6c5913 | ||
|
|
667e69a0c1 | ||
|
|
9c7dd16126 | ||
|
|
94adcf7ad3 | ||
|
|
80de3652cf | ||
|
|
0e1b88a593 | ||
|
|
b4f820568a | ||
|
|
8d52281cdf | ||
|
|
6a0eecbe02 | ||
|
|
aa85535d47 | ||
|
|
7d936d0f7c | ||
|
|
47531d8365 | ||
|
|
86f115d200 | ||
|
|
9c5c3c4ce0 | ||
|
|
26a5c4a1fb | ||
|
|
2adc67e25e | ||
|
|
e9e05fedbd | ||
|
|
0a8330cbb0 | ||
|
|
db4361ad46 | ||
|
|
74c093facb | ||
|
|
aae4c946ea | ||
|
|
b59211ea7f | ||
|
|
c7c7250d88 | ||
|
|
21365cb0e6 | ||
|
|
76edaeb1c0 | ||
|
|
1ad00075a3 | ||
|
|
46dcbbe731 | ||
|
|
0d447560bc | ||
|
|
5094235ce1 | ||
|
|
6f6c825f3d | ||
|
|
d293ca6009 | ||
|
|
352db64b33 | ||
|
|
bc75768afe | ||
|
|
f992504227 | ||
|
|
d41530930d | ||
|
|
692260cc76 | ||
|
|
aa58b46d4d | ||
|
|
633d485045 | ||
|
|
424b63a27d | ||
|
|
6e44d98d46 | ||
|
|
d21e5e9944 | ||
|
|
67aed43997 | ||
|
|
ac64943965 | ||
|
|
4dea295f0d | ||
|
|
6c6cb3f3b5 | ||
|
|
1ae1e6d57a | ||
|
|
495e29d543 | ||
|
|
04d2a726b9 | ||
|
|
62a027d6fd | ||
|
|
7cab35bbc0 | ||
|
|
938f846dde | ||
|
|
66be1630fa | ||
|
|
a1f835c9b2 | ||
|
|
12172d03b4 | ||
|
|
60f362817f | ||
|
|
4d1e5cf37c | ||
|
|
16893c28a4 | ||
|
|
577967a549 | ||
|
|
7fb94b6528 | ||
|
|
e197977172 | ||
|
|
ecd0a26a8d | ||
|
|
f99241ca88 | ||
|
|
dabeb0e833 | ||
|
|
9458cf5b2b | ||
|
|
3db35a6871 | ||
|
|
b1daf238ca | ||
|
|
0205ecd711 | ||
|
|
9677f25d27 | ||
|
|
4a221554de | ||
|
|
7e513a12a3 | ||
|
|
39bebe7da9 | ||
|
|
9537bf17b3 | ||
|
|
e8cb337946 | ||
|
|
840629762f | ||
|
|
df6c30653c | ||
|
|
8d3c99ba30 | ||
|
|
c2079d8e92 | ||
|
|
cc864362aa | ||
|
|
d30aea79a7 | ||
|
|
2ee4da0b47 | ||
|
|
016661463e | ||
|
|
ce914b396a | ||
|
|
b105b8ea97 | ||
|
|
6ed5aea993 | ||
|
|
80bce453c3 | ||
|
|
d102584af6 | ||
|
|
4f27d3dcae | ||
|
|
35f8c0527c | ||
|
|
28fe304f76 | ||
|
|
9eea49a070 | ||
|
|
db95cedf34 | ||
|
|
3a9c7c082f | ||
|
|
4a3cf37679 | ||
|
|
dd7b090aec | ||
|
|
eaa0f823a8 | ||
|
|
7045d7b94e | ||
|
|
8430d4a56e | ||
|
|
ea71ea1823 | ||
|
|
3a1ab1a8a6 | ||
|
|
87714ec1ad | ||
|
|
1c590c622b | ||
|
|
40d6fa05d0 | ||
|
|
ac69c27b86 | ||
|
|
a16bd9600b | ||
|
|
fd84be8b66 | ||
|
|
6cc2e67b92 | ||
|
|
2dd06aadcd | ||
|
|
86db4f923d | ||
|
|
dcd936a9ab | ||
|
|
b663c6bbbf | ||
|
|
5ab21c089e | ||
|
|
84775fdea0 | ||
|
|
2a0bc7051a | ||
|
|
8beca57238 | ||
|
|
ed44260ec3 | ||
|
|
8f09d4f57b | ||
|
|
7e05f196b5 | ||
|
|
47c7f4adbe | ||
|
|
0698ba7656 | ||
|
|
33d8f806c2 | ||
|
|
530e297d17 | ||
|
|
be2f8f8cd8 | ||
|
|
a2f10dfbdd | ||
|
|
495c2a2fbd | ||
|
|
eb8156ddb3 | ||
|
|
1c3d2b949f | ||
|
|
343d2bc418 | ||
|
|
297d959f63 | ||
|
|
161d47f4d9 | ||
|
|
99b0a517fd | ||
|
|
adb46134c4 | ||
|
|
d2d38a0509 | ||
|
|
9a21e523f0 | ||
|
|
ca11a48fef | ||
|
|
aa171c16f1 | ||
|
|
cb5e93ebc7 | ||
|
|
a256a77277 | ||
|
|
e7696f462a | ||
|
|
d6c7154fa9 | ||
|
|
b473e0bd8a | ||
|
|
b1deed5857 | ||
|
|
637bfe7314 | ||
|
|
9b7a33031c | ||
|
|
443a9b78d7 | ||
|
|
36856cc92a | ||
|
|
6e4ff0c7c5 | ||
|
|
095d7d3e52 | ||
|
|
1b706e539f | ||
|
|
9c9ab6a23f | ||
|
|
47db03c72f | ||
|
|
4119c21b6e | ||
|
|
6cf5504585 | ||
|
|
f4f077f443 | ||
|
|
4a1dab8062 | ||
|
|
836eff3e55 | ||
|
|
1b87da7ce3 | ||
|
|
51953a0367 | ||
|
|
62a0a55be7 | ||
|
|
48b9cb7d33 | ||
|
|
0a05e9d936 | ||
|
|
4aab2a0248 | ||
|
|
3b3463eae3 | ||
|
|
614222ae87 | ||
|
|
a9559a6d7a | ||
|
|
e3931e0923 | ||
|
|
10e26cb570 | ||
|
|
219ebd0631 | ||
|
|
d781d052c2 | ||
|
|
d98135d9b5 | ||
|
|
77026122fc | ||
|
|
b3245613f5 | ||
|
|
a0462269db | ||
|
|
c6acd62380 | ||
|
|
a1b38efbd8 | ||
|
|
ec7979b6ca | ||
|
|
638a8c60df | ||
|
|
084fb44f05 | ||
|
|
4e973fcc30 | ||
|
|
4daa8ff3a7 | ||
|
|
cb213ee371 | ||
|
|
4f2c6c8229 | ||
|
|
c4e971c91c | ||
|
|
28b71c922f | ||
|
|
3806883093 | ||
|
|
581dbf8301 | ||
|
|
738698a75b | ||
|
|
fd7ac7465c | ||
|
|
947ecf186a | ||
|
|
18c8ba58ef | ||
|
|
bdcbaa3173 | ||
|
|
c843f09cb5 | ||
|
|
5909f23180 | ||
|
|
d71d4191aa | ||
|
|
5b357c47cc | ||
|
|
2d5bd18fa4 | ||
|
|
d74b650643 | ||
|
|
fabff5d9ec | ||
|
|
c90b7fb02b | ||
|
|
869083e0f6 | ||
|
|
e81baca5a8 | ||
|
|
eb3663e737 | ||
|
|
37c890bee2 | ||
|
|
62b079a9c3 | ||
|
|
716e47ce8f | ||
|
|
506b7060a1 | ||
|
|
de0357aec8 | ||
|
|
7d83b6744f | ||
|
|
d90929730d | ||
|
|
7455ed254c | ||
|
|
1d502d5e74 | ||
|
|
babff350ff | ||
|
|
49e5430aa3 | ||
|
|
8157fa5fd5 | ||
|
|
bbcf27a6a3 | ||
|
|
e8c9e221ef | ||
|
|
38f02956aa | ||
|
|
9aa6d14081 | ||
|
|
b3b7d8ad1d | ||
|
|
7fecdae119 | ||
|
|
b5e0529709 | ||
|
|
5b8d5c5ff6 | ||
|
|
f0828b1ff6 | ||
|
|
adb6878b00 | ||
|
|
edf3aeb90c | ||
|
|
4316f9b2fe | ||
|
|
8800c2c85a | ||
|
|
0e2dc7462c | ||
|
|
20b1455480 | ||
|
|
a765fd5a3a | ||
|
|
ada5aa7615 | ||
|
|
94c83697b0 | ||
|
|
a6db33b0fe | ||
|
|
58f3216cd4 | ||
|
|
611be57ee6 | ||
|
|
6bfb9c719c | ||
|
|
05fe7d81d7 | ||
|
|
f7821ec350 | ||
|
|
62ac59705b | ||
|
|
5e9ed6772d | ||
|
|
3ff8c11235 | ||
|
|
11836023ee | ||
|
|
9680dece2e | ||
|
|
aa2cd00f07 | ||
|
|
9094f1ed95 | ||
|
|
4011f80eff | ||
|
|
7e70508ac9 | ||
|
|
d336898f77 | ||
|
|
4cc9c8efd1 | ||
|
|
53e1769956 | ||
|
|
6e7c388242 | ||
|
|
92fb7b0826 | ||
|
|
69d61384fd | ||
|
|
12067840a5 | ||
|
|
7a4810893a | ||
|
|
5430167dbc | ||
|
|
46540df5f0 | ||
|
|
9ac3066fe1 | ||
|
|
0a793171dd | ||
|
|
fcd986d7b1 | ||
|
|
0c2e54e583 | ||
|
|
47db405737 | ||
|
|
8b2155d3d9 | ||
|
|
6491690dfe | ||
|
|
7ba05ac6b5 | ||
|
|
8cf19a6b2e | ||
|
|
6f4db5a693 | ||
|
|
8d60f68fcd | ||
|
|
6742b08743 | ||
|
|
ccb9e070b4 | ||
|
|
d243143316 | ||
|
|
0ecf43f5f6 | ||
|
|
064daca6f9 | ||
|
|
4d784d0aea | ||
|
|
dcff5e3b3d | ||
|
|
0ad1f64cc9 | ||
|
|
f6e9ef6627 | ||
|
|
9bda477847 | ||
|
|
5c512a5011 | ||
|
|
d9b24d0630 | ||
|
|
527726faa3 | ||
|
|
7f6f0c4213 | ||
|
|
76201b7587 | ||
|
|
25bee71a46 | ||
|
|
9020973498 | ||
|
|
b84441066d | ||
|
|
3126905e2d |
@@ -2,4 +2,18 @@
|
||||
# Cloud Run false-positives (CKV_K8S_21/28/30) are suppressed via per-file
|
||||
# inline checkov:skip comments in deploy/gcp/cloudrun-service.yaml rather than
|
||||
# globally here, so future real Kubernetes manifests are not silently exempted.
|
||||
#
|
||||
# The knowledge-explorer Helm chart's unconditional templates (service.yaml,
|
||||
# deployment.yaml, configmap.yaml) set metadata.namespace to .Release.Namespace,
|
||||
# which is only bound at `helm install`/`helm template` time. Checkov's helm
|
||||
# framework renders the chart without a namespace override, so it always
|
||||
# resolves to "default" and trips CKV_K8S_21 even though the chart is
|
||||
# namespace-agnostic by design. Suppressed via metadata annotations
|
||||
# (checkov.io/skip1 / runterrascan.io/skip) on each resource's metadata.annotations,
|
||||
# as both Checkov and Terrascan require K8s/Helm resource-level annotations
|
||||
# rather than file-header comments.
|
||||
# deployment.yaml additionally suppresses AC_K8S_0080 and CKV_K8S_31 (seccomp) via
|
||||
# metadata.annotations on both the Deployment resource and the pod template:
|
||||
# the seccomp profile is set correctly in values.yaml and only resolves once
|
||||
# Helm actually renders `toYaml`, which static template scanning does not do.
|
||||
skip-check: []
|
||||
|
||||
@@ -70,6 +70,13 @@ updates:
|
||||
- "dependencies"
|
||||
- "github-actions"
|
||||
- "ci"
|
||||
# All our actions are SHA-pinned with a "# vX" comment; Dependabot
|
||||
# resolves the new tag's SHA and updates both the pin and the comment
|
||||
# together, so this stays the source of truth (no separate script needed).
|
||||
groups:
|
||||
github-actions:
|
||||
patterns:
|
||||
- "*"
|
||||
|
||||
# Optional dependencies (separate schedule for stability)
|
||||
- package-ecosystem: "pip"
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
> **Before you submit:** make sure you followed the [issue workflow in CONTRIBUTING.md](https://github.com/semantica-agi/semantica/blob/main/CONTRIBUTING.md#-working-on-an-existing-issue) — comment on the issue and wait for assignment before opening a PR, to avoid duplicate work.
|
||||
|
||||
## Description
|
||||
|
||||
<!-- Provide a clear description of your changes -->
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
#!/usr/bin/env bash
|
||||
# Verifies that every third-party GitHub Action referenced in
|
||||
# .github/workflows/*.yml and .github/workflows/*.yaml is pinned to a full
|
||||
# commit SHA (not a mutable tag
|
||||
# or branch), and that any pin's trailing "# vX" comment still matches what
|
||||
# that tag resolves to today.
|
||||
#
|
||||
# Fails closed on purpose:
|
||||
# - a `uses:` line pinned to anything other than a 40-hex-char SHA is a
|
||||
# hard failure, not a skip - this is what stops a newly-added mutable
|
||||
# tag (e.g. `uses: some/action@v1`) from slipping past unnoticed.
|
||||
# - a tag that can't be resolved via the GitHub API (rate limit, deleted
|
||||
# tag, typo) is also a hard failure rather than a warning - an
|
||||
# unverifiable pin is exactly the failure mode this check exists to
|
||||
# catch, so it must not pass silently.
|
||||
set -uo pipefail
|
||||
|
||||
fail=0
|
||||
checked=0
|
||||
|
||||
# Pattern for a third-party uses: line — stored in a variable so bash's
|
||||
# [[ =~ ]] parser never sees literal \" or \' escapes, which cause a
|
||||
# "syntax error in conditional expression: unexpected token )" at runtime.
|
||||
# Semantics: optional leading quote, owner/repo, optional subpath, @ref,
|
||||
# optional trailing quote; quote chars excluded from the ref capture group.
|
||||
USES_PATTERN='uses:[[:space:]]+["'"'"']?([A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+)(/[^[:space:]@"'"'"']+)?@([^[:space:]"'"'"']+)["'"'"']?'
|
||||
|
||||
while IFS=: read -r file lineno content; do
|
||||
# Local composite actions (./x) and Docker image refs (docker://...) use a
|
||||
# different pinning mechanism and aren't in scope here.
|
||||
[[ "$content" =~ uses:\ +\./ ]] && continue
|
||||
[[ "$content" =~ uses:\ +docker:// ]] && continue
|
||||
|
||||
if [[ "$content" =~ $USES_PATTERN ]]; then
|
||||
repo="${BASH_REMATCH[1]}"
|
||||
ref="${BASH_REMATCH[3]}"
|
||||
checked=$((checked + 1))
|
||||
|
||||
if [[ ! "$ref" =~ ^[0-9a-fA-F]{40}$ ]]; then
|
||||
echo "::error file=$file,line=$lineno::$repo is pinned to '$ref', not a full commit SHA. Mutable tags/branches can be silently re-pointed (see the LiteLLM/Trivy 2026 incident) - pin to a commit SHA instead."
|
||||
fail=1
|
||||
continue
|
||||
fi
|
||||
sha="$ref"
|
||||
|
||||
if [[ "$content" =~ \#[[:space:]]*([^[:space:]]+)[[:space:]]*$ ]]; then
|
||||
tag="${BASH_REMATCH[1]}"
|
||||
else
|
||||
echo "::warning file=$file,line=$lineno::$repo@$sha has no trailing '# vX' comment recording which tag it corresponds to - add one for auditability."
|
||||
continue
|
||||
fi
|
||||
|
||||
resolved=$(gh api "repos/$repo/commits/$tag" --jq '.sha' 2>/dev/null)
|
||||
if [[ -z "$resolved" ]]; then
|
||||
echo "::error file=$file,line=$lineno::Could not resolve '$repo@$tag' via the GitHub API (rate limit, deleted tag, or typo). Treating as unverifiable = failure."
|
||||
fail=1
|
||||
continue
|
||||
fi
|
||||
|
||||
if [[ "$resolved" != "$sha" ]]; then
|
||||
echo "::error file=$file,line=$lineno::$repo is pinned to $sha but tag '$tag' now resolves to $resolved. Update the pin or the comment."
|
||||
fail=1
|
||||
else
|
||||
echo "OK $repo@$tag -> $sha ($file:$lineno)"
|
||||
fi
|
||||
fi
|
||||
done < <(grep -rHn "uses:" .github/workflows/*.yml .github/workflows/*.yaml 2>/dev/null)
|
||||
|
||||
echo "Checked $checked action reference(s)."
|
||||
exit $fail
|
||||
@@ -13,12 +13,12 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout Code
|
||||
uses: actions/checkout@v7
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Set up Python 3.12
|
||||
uses: actions/setup-python@v5
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: 'pip'
|
||||
@@ -43,7 +43,7 @@ jobs:
|
||||
# pytest-benchmark --storage file://benchmarks/results --benchmark-compare
|
||||
|
||||
- name: Upload Benchmark Results
|
||||
uses: actions/upload-artifact@v7
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
if: always()
|
||||
with:
|
||||
name: benchmark-report-${{ github.run_id }}
|
||||
|
||||
@@ -21,20 +21,27 @@ jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
with:
|
||||
node-version: '20'
|
||||
cache: 'npm'
|
||||
cache-dependency-path: explorer/package-lock.json
|
||||
- name: Build Explorer frontend
|
||||
- name: Install Explorer frontend dependencies
|
||||
working-directory: explorer
|
||||
run: npm ci
|
||||
- name: Test Explorer frontend
|
||||
working-directory: explorer
|
||||
run: |
|
||||
npm ci
|
||||
npm run build
|
||||
npm run test:graph-store
|
||||
npm run test:graph-workspace
|
||||
npm run test:plugin-registry
|
||||
- name: Build Explorer frontend
|
||||
working-directory: explorer
|
||||
run: npm run build
|
||||
- run: pip install build
|
||||
- run: python -m build
|
||||
- name: Verify Explorer frontend is packaged
|
||||
|
||||
@@ -20,20 +20,49 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v4
|
||||
# The CodeQL bundle download (github/codeql-action/init's "Setup CodeQL
|
||||
# tools" step) streams a ~1GB tarball from GitHub's release CDN and
|
||||
# does not retry on a transient connection reset (ECONNRESET) itself
|
||||
# (github/codeql-action, unresolved as of v4 / CLI 2.26.1: the HTTP
|
||||
# error is retryable but isn't retried internally). Since a `uses:`
|
||||
# step can't be wrapped by a shell-level retry action, attempt init
|
||||
# up to 3 times; each retry is a fresh download attempt with no
|
||||
# meaningful state carried over from a failed attempt.
|
||||
- name: Initialize CodeQL (attempt 1)
|
||||
id: codeql-init-1
|
||||
uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
continue-on-error: true
|
||||
with:
|
||||
languages: python
|
||||
queries: security-and-quality
|
||||
config-file: .github/codeql/codeql-config.yml
|
||||
|
||||
- name: Initialize CodeQL (attempt 2)
|
||||
id: codeql-init-2
|
||||
if: steps.codeql-init-1.outcome == 'failure'
|
||||
uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
continue-on-error: true
|
||||
with:
|
||||
languages: python
|
||||
queries: security-and-quality
|
||||
config-file: .github/codeql/codeql-config.yml
|
||||
|
||||
- name: Initialize CodeQL (attempt 3)
|
||||
id: codeql-init-3
|
||||
if: steps.codeql-init-2.outcome == 'failure'
|
||||
uses: github/codeql-action/init@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
with:
|
||||
languages: python
|
||||
queries: security-and-quality
|
||||
config-file: .github/codeql/codeql-config.yml
|
||||
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v4
|
||||
uses: github/codeql-action/autobuild@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v4
|
||||
uses: github/codeql-action/analyze@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
with:
|
||||
category: "/language:python"
|
||||
upload: false
|
||||
@@ -43,7 +72,7 @@ jobs:
|
||||
# Uploads results only when Default Setup is not active.
|
||||
# If Default Setup is still enabled, this step skips gracefully
|
||||
# instead of failing the workflow with HTTP 409.
|
||||
uses: github/codeql-action/upload-sarif@v4
|
||||
uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
with:
|
||||
sarif_file: ${{ steps.codeql.outputs.sarif-output }}
|
||||
category: "/language:python"
|
||||
|
||||
@@ -36,14 +36,14 @@ jobs:
|
||||
runs-on: windows-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-dotnet@v5
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-dotnet@a98b56852c35b8e3190ac28c8c2271da59106c68 # v6
|
||||
with:
|
||||
dotnet-version: |
|
||||
5.0.x
|
||||
6.0.x
|
||||
- name: Run Microsoft Security DevOps
|
||||
uses: microsoft/security-devops-action@v1.12.0
|
||||
uses: microsoft/security-devops-action@08976cb623803b1b36d7112d4ff9f59eae704de0 # v1.12.0
|
||||
id: msdo
|
||||
with:
|
||||
# checkov is intentionally excluded from this MSDO step.
|
||||
@@ -57,11 +57,11 @@ jobs:
|
||||
# avoiding the guardian.cmd/checkov exit-code bug in the MSDO wrapper.
|
||||
tools: eslint,templateanalyzer,terrascan
|
||||
- name: Upload results to Security tab
|
||||
uses: github/codeql-action/upload-sarif@v4
|
||||
uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
with:
|
||||
sarif_file: ${{ steps.msdo.outputs.sarifFile }}
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
@@ -82,7 +82,7 @@ jobs:
|
||||
}
|
||||
|
||||
- name: Upload Checkov results to Security tab
|
||||
uses: github/codeql-action/upload-sarif@v4
|
||||
uses: github/codeql-action/upload-sarif@5595ccaf912efad79be6eef63a5619ff05969be3 # v4
|
||||
if: always()
|
||||
with:
|
||||
sarif_file: reports/checkov.sarif
|
||||
|
||||
@@ -29,11 +29,11 @@ jobs:
|
||||
name: Validate Documentation
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
with:
|
||||
node-version: '20'
|
||||
- run: python docs_check.py
|
||||
@@ -44,9 +44,9 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
needs: validate
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
with:
|
||||
node-version: '20'
|
||||
|
||||
@@ -57,12 +57,12 @@ jobs:
|
||||
cd ..
|
||||
unzip -q export.zip -d site
|
||||
|
||||
- uses: actions/configure-pages@v6
|
||||
- uses: actions/configure-pages@45bfe0192ca1faeb007ade9deae92b16b8254a0d # v6
|
||||
|
||||
- uses: actions/upload-pages-artifact@v5
|
||||
- uses: actions/upload-pages-artifact@fc324d3547104276b827a68afc52ff2a11cc49c9 # v5
|
||||
with:
|
||||
path: ./site
|
||||
|
||||
- name: Deploy to GitHub Pages
|
||||
id: deployment
|
||||
uses: actions/deploy-pages@v5
|
||||
uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 # v5
|
||||
|
||||
@@ -5,19 +5,28 @@ on:
|
||||
tags: ['v*']
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
id-token: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
release:
|
||||
runs-on: ubuntu-latest
|
||||
environment: pypi
|
||||
concurrency:
|
||||
group: release-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
permissions:
|
||||
contents: write # for the GitHub Release
|
||||
id-token: write # for PyPI Trusted Publishing (OIDC) and attestation signing
|
||||
attestations: write # for SLSA build provenance
|
||||
# If you add another job to this workflow, give it its own explicit
|
||||
# `permissions:` block rather than relying on the workflow-level default
|
||||
# above (contents: read) - do not widen the workflow-level default.
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
- uses: actions/setup-node@v6
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
|
||||
with:
|
||||
node-version: '20'
|
||||
cache: 'npm'
|
||||
@@ -46,7 +55,11 @@ jobs:
|
||||
|
||||
print("Explorer frontend is packaged")
|
||||
PY
|
||||
- uses: softprops/action-gh-release@v3
|
||||
- name: Attest build provenance
|
||||
uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4
|
||||
with:
|
||||
subject-path: 'dist/*'
|
||||
- uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3
|
||||
with:
|
||||
files: dist/*
|
||||
- uses: pypa/gh-action-pypi-publish@release/v1
|
||||
- uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
|
||||
@@ -28,13 +28,17 @@ jobs:
|
||||
contents: read
|
||||
security-events: write
|
||||
actions: read
|
||||
|
||||
# Needed for the "Comment PR with Security Results" step below. Safe on
|
||||
# pull_request (not pull_request_target): GitHub always forces a
|
||||
# read-only token for PRs from forks regardless of this permission.
|
||||
pull-requests: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v7
|
||||
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v4
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
|
||||
@@ -42,21 +46,50 @@ jobs:
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install safety bandit semgrep jq
|
||||
|
||||
# Install the project itself (core deps + the LiteLLM provider extra)
|
||||
# so Safety scans Semantica's actual dependency tree, not just the
|
||||
# scanner tools' own dependencies.
|
||||
pip install -e ".[llm-litellm]"
|
||||
|
||||
- name: Run Safety Check (Package Vulnerabilities)
|
||||
run: |
|
||||
safety check --json --output safety-report.json || true
|
||||
# NOTE: Safety 3.x repurposed --output to select a console format
|
||||
# (json/text/screen/...), not a file path. Writing JSON to a file
|
||||
# now requires --save-json; the previous `--output safety-report.json`
|
||||
# usage was silently invalid and never produced a report.
|
||||
safety check --save-json safety-report.json || true
|
||||
|
||||
# Guard 1: fail loudly if Safety exited before writing a report at all
|
||||
# (network error, API auth failure, tool crash). Without this check a
|
||||
# missing or empty file causes jq to fall back to "0", making a broken
|
||||
# scanner indistinguishable from a clean scan.
|
||||
if [ ! -s safety-report.json ]; then
|
||||
echo "::error::Safety scan produced no report (safety-report.json is missing or empty). Treating as failure — check for network errors, API auth failures, or Safety crashes in the logs above."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Checking for package vulnerabilities..."
|
||||
|
||||
# Count vulnerabilities safely
|
||||
VULNS=$(safety check --json --output /dev/stdout 2>/dev/null | jq '.vulnerabilities | length' 2>/dev/null || echo "0")
|
||||
|
||||
|
||||
# No || echo "0" fallback: if jq fails (malformed JSON, missing key,
|
||||
# vulnerabilities:null) VULNS will be empty or "null" so guard 2 below
|
||||
# catches it rather than silently treating the broken report as zero.
|
||||
VULNS=$(jq '.vulnerabilities | length' safety-report.json 2>/dev/null)
|
||||
|
||||
# Guard 2: ensure VULNS is a non-negative integer before the -gt
|
||||
# comparison. "null" (missing/null key) or "" (jq parse failure) would
|
||||
# cause bash's -gt to throw an arithmetic error and fall through to the
|
||||
# success branch — the same silent-pass bug as a missing file.
|
||||
if ! [[ "$VULNS" =~ ^[0-9]+$ ]]; then
|
||||
echo "::error::Safety report exists but 'vulnerabilities' is missing or non-numeric (got: '${VULNS}'). The report may be malformed or Safety may have written an error-only JSON. Treating as failure."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "$VULNS" -gt 0 ]; then
|
||||
echo "❌ Security vulnerabilities found: $VULNS"
|
||||
echo "CI will fail to prevent merging of vulnerable dependencies"
|
||||
echo ""
|
||||
echo "Vulnerability details:"
|
||||
safety check || true
|
||||
jq -r '.vulnerabilities[] | "- \(.package_name)==\(.analyzed_version): \(.vulnerability_id) (\(.CVE // "no CVE assigned"))"' safety-report.json || true
|
||||
exit 1
|
||||
else
|
||||
echo "✅ No security vulnerabilities found"
|
||||
@@ -99,9 +132,10 @@ jobs:
|
||||
fi
|
||||
|
||||
- name: Upload Security Reports
|
||||
uses: actions/upload-artifact@v7
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: security-reports
|
||||
retention-days: 14
|
||||
path: |
|
||||
safety-report.json
|
||||
bandit-report.json
|
||||
@@ -109,77 +143,91 @@ jobs:
|
||||
|
||||
- name: Comment PR with Security Results
|
||||
if: github.event_name == 'pull_request'
|
||||
uses: actions/github-script@v9
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
|
||||
// Read safety report
|
||||
let safetyResults = '';
|
||||
try {
|
||||
const safetyData = JSON.parse(fs.readFileSync('safety-report.json', 'utf8'));
|
||||
if (safetyData.vulnerabilities && safetyData.vulnerabilities.length > 0) {
|
||||
safetyResults = `## Safety Vulnerabilities Found\\n`;
|
||||
safetyData.vulnerabilities.forEach(vuln => {
|
||||
safetyResults += `- **${vuln.package}**: ${vuln.advisory}\\n`;
|
||||
});
|
||||
} else {
|
||||
safetyResults = '## No Safety Vulnerabilities Found\\n';
|
||||
|
||||
// Renders one tool's findings as a section. `items` is already
|
||||
// the list of pre-formatted "- `thing` in `where`" strings; this
|
||||
// just handles the found/not-found/report-missing framing and
|
||||
// collapses long lists into a <details> block so the comment
|
||||
// doesn't turn into a wall of text.
|
||||
function renderSection(title, reportPath, parse) {
|
||||
let data;
|
||||
try {
|
||||
data = JSON.parse(fs.readFileSync(reportPath, 'utf8'));
|
||||
} catch (e) {
|
||||
return [
|
||||
`### ${title}`,
|
||||
`⚠️ No report found at \`${reportPath}\` — the scan may have failed before producing output. Check the job logs.`,
|
||||
].join('\n');
|
||||
}
|
||||
} catch (e) {
|
||||
safetyResults = '## Safety scan completed\\n';
|
||||
}
|
||||
|
||||
// Read bandit report
|
||||
let banditResults = '';
|
||||
try {
|
||||
const banditData = JSON.parse(fs.readFileSync('bandit-report.json', 'utf8'));
|
||||
if (banditData.results && banditData.results.length > 0) {
|
||||
const highIssues = banditData.results.filter(issue => issue.issue_severity === 'HIGH');
|
||||
if (highIssues.length > 0) {
|
||||
banditResults = `## High Severity Security Issues Found\\n`;
|
||||
highIssues.forEach(issue => {
|
||||
banditResults += `- **${issue.test_name}**: ${issue.filename}:${issue.line_number}\\n`;
|
||||
});
|
||||
} else {
|
||||
banditResults = '## No High Severity Security Issues Found\\n';
|
||||
}
|
||||
} else {
|
||||
banditResults = '## No Bandit Issues Found\\n';
|
||||
|
||||
const items = parse(data);
|
||||
if (items.length === 0) {
|
||||
return [`### ${title}`, `✅ No findings.`].join('\n');
|
||||
}
|
||||
} catch (e) {
|
||||
banditResults = '## Bandit scan completed\\n';
|
||||
}
|
||||
|
||||
// Read semgrep report
|
||||
let semgrepResults = '';
|
||||
try {
|
||||
const semgrepData = JSON.parse(fs.readFileSync('semgrep-report.json', 'utf8'));
|
||||
if (semgrepData.results && semgrepData.results.length > 0) {
|
||||
semgrepResults = `## Security Patterns Found\\n`;
|
||||
semgrepData.results.slice(0, 10).forEach(issue => {
|
||||
semgrepResults += `- **${issue.rule_id}**: ${issue.path}\\n`;
|
||||
});
|
||||
if (semgrepData.results.length > 10) {
|
||||
semgrepResults += `- ... and ${semgrepData.results.length - 10} more\\n`;
|
||||
}
|
||||
|
||||
const lines = [`### ${title}`, `Found **${items.length}**.`, ''];
|
||||
const shown = items.slice(0, 15);
|
||||
if (items.length > 15) {
|
||||
lines.push('<details>', '<summary>Show all findings</summary>', '');
|
||||
lines.push(...items);
|
||||
lines.push('', '</details>');
|
||||
} else {
|
||||
semgrepResults = '## No Security Patterns Found\\n';
|
||||
lines.push(...shown);
|
||||
}
|
||||
} catch (e) {
|
||||
semgrepResults = '## Semgrep scan completed\\n';
|
||||
return lines.join('\n');
|
||||
}
|
||||
|
||||
// Create summary comment
|
||||
const comment = `# 🔒 Security Scan Results\\n\\n${safetyResults}\\n\\n${banditResults}\\n\\n${semgrepResults}\\n\\n---\\n\\n*This security scan runs automatically on source-code PRs and bi-weekly (skipped for doc/markdown-only changes).*\\n\\n📊 **Security Policy**: CI fails on vulnerabilities and HIGH severity issues.`;
|
||||
|
||||
// Post comment with error handling
|
||||
|
||||
const safetySection = renderSection(
|
||||
'Safety — dependency vulnerabilities',
|
||||
'safety-report.json',
|
||||
(data) => (data.vulnerabilities || []).map(
|
||||
(v) => `- \`${v.package_name}==${v.analyzed_version}\`: ${v.vulnerability_id}` +
|
||||
(v.CVE ? ` (${v.CVE})` : '') + ` — ${v.advisory || 'no advisory text'}`
|
||||
)
|
||||
);
|
||||
|
||||
const banditSection = renderSection(
|
||||
'Bandit — HIGH-severity code issues',
|
||||
'bandit-report.json',
|
||||
(data) => (data.results || [])
|
||||
.filter((issue) => issue.issue_severity === 'HIGH')
|
||||
.map((issue) => `- \`${issue.test_name}\` in \`${issue.filename}:${issue.line_number}\``)
|
||||
);
|
||||
|
||||
const semgrepSection = renderSection(
|
||||
'Semgrep — static analysis patterns',
|
||||
'semgrep-report.json',
|
||||
(data) => (data.results || []).map(
|
||||
(issue) => `- \`${issue.check_id}\` in \`${issue.path}:${issue.start?.line ?? '?'}\``
|
||||
)
|
||||
);
|
||||
|
||||
const comment = [
|
||||
'# 🔒 Security Scan Results',
|
||||
'',
|
||||
safetySection,
|
||||
'',
|
||||
banditSection,
|
||||
'',
|
||||
semgrepSection,
|
||||
'',
|
||||
'---',
|
||||
'',
|
||||
'*This security scan runs automatically on source-code PRs and bi-weekly (skipped for doc/markdown-only changes).*',
|
||||
'',
|
||||
'📊 **Security Policy**: CI fails on Safety vulnerabilities and Bandit HIGH-severity findings. Semgrep findings above are informational and do not block merge.',
|
||||
].join('\n');
|
||||
|
||||
try {
|
||||
await github.rest.issues.createComment({
|
||||
issue_number: context.issue.number,
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
body: comment
|
||||
body: comment,
|
||||
});
|
||||
console.log('✅ Security comment posted successfully');
|
||||
} catch (error) {
|
||||
|
||||
@@ -12,8 +12,8 @@ jobs:
|
||||
audit:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
- uses: actions/setup-python@v5
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
- run: pip install pip-audit
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Verify Action Pins
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
paths:
|
||||
- '.github/workflows/**'
|
||||
- '.github/scripts/verify-action-pins.sh'
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- '.github/workflows/**'
|
||||
- '.github/scripts/verify-action-pins.sh'
|
||||
schedule:
|
||||
- cron: '0 3 * * 1' # weekly, in case an upstream tag is deliberately moved
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
- name: Verify pinned action SHAs match their tag comments
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: bash .github/scripts/verify-action-pins.sh
|
||||
BIN
Binary file not shown.
+340
@@ -9,6 +9,346 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- **Embedded Oxigraph backend for `TripletStore`** (#838, closes #834) by @Linxiushen
|
||||
- Added `OxigraphStore` (`semantica/triplet_store/oxigraph_store.py`), an in-process SPARQL 1.1 store via the optional `pyoxigraph` dependency — no external server (Blazegraph/Jena/RDF4J/Anzo) required, fixing the confusing plain connection-error failure `TripletStore` previously produced with no server running (no local Docker daemon, no Java, CI, or a fresh laptop)
|
||||
- Runs fully in memory by default, or persists to a local directory via `TripletStore(backend="oxigraph", path=...)`; reopening the same directory resumes existing data
|
||||
- Full CRUD, native batch loading (`Store.extend`), named-graph scoping (`graph=` on add/query), and SPARQL SELECT/ASK/CONSTRUCT/DESCRIBE result mapping matching the existing backend contract; reuses `sparql_escaping.py` for datatype-IRI resolution instead of reimplementing it, and preserves RDF literal datatype/language metadata across writes, reads, and query results
|
||||
- New optional `semantica[tripletstore-oxigraph]` extra (`pyoxigraph>=0.5.0`), included in the `all` extra; the import is lazy, so `TripletStore` and the rest of Semantica keep working without `pyoxigraph` installed
|
||||
- Wired into `TripletStore` (`backend="oxigraph"`, added to `SUPPORTED_BACKENDS` and `NAMED_GRAPH_CAPABLE_BACKENDS`) and exported from `semantica.triplet_store`; README, module reference, glossary, and usage guide updated with install/configuration examples
|
||||
- **Fixed along the way**: a missing `pyoxigraph` install surfaced as a generic wrapped `ProcessingError` instead of the underlying `ImportError` and its install hint, because `TripletStore._initialize_store_backend()`'s broad `except Exception` caught and rewrapped it; `ImportError` is now re-raised as-is so the `pip install "semantica[tripletstore-oxigraph]"` hint reaches the caller
|
||||
- New integration tests in `tests/triplet_store/test_oxigraph_store.py` covering persistence/reopen, named-graph isolation, SELECT/ASK/CONSTRUCT result shapes, and the missing-dependency error message; skipped automatically when `pyoxigraph` isn't installed, and not yet exercised in CI since it doesn't install the optional extra or run the Python test suite
|
||||
|
||||
- **PROV-O trust blockers and general spec completeness for `ProvenanceManager`** (#825) by @KaifAhmad1
|
||||
- **Invalidation instead of hard delete**: new `ProvenanceManager.invalidate(entity_id, agent_id, reason=None)` tombstones an entry — archives its pre-invalidation state under a stable versioned key, then appends the invalidated entry (`invalidated`, `invalidated_at_time`, `invalidated_by`, `invalidation_reason`) — instead of mutating or deleting it, so an audit can prove a fact existed, was reviewed, and was retracted. `ProvenanceManager.clear()` remains the bulk dev/test store-reset utility it always was; it was not repurposed
|
||||
- **Hash-chained integrity**: every entry now carries `sequence_id`/`previous_checksum`, chaining it to the entry immediately before it in insertion order. New `ProvenanceManager.verify_chain()` walks the chain and reports any break, including a row hard-deleted directly from the underlying table — something a lone per-row SHA-256 checksum can never detect on its own. `compute_checksum()` now also covers `agent_id`/`agent_type`, the lineage-link fields, and the invalidation fields, closing several fields that previously weren't tamper-evident
|
||||
- **Typed Agent/Activity**: `agent_id` was a dead field — no `track_*` method read it from kwargs, so it was always the `"semantica"` default regardless of what callers passed; fixed, and paired with new `AgentRecord(id, agent_type, is_automated)` / `ActivityRecord(id, activity_type, started_at_time, ended_at_time)` dataclasses (pass via `agent=`/`activity=` kwargs) so a human reviewer, an LLM call, and an automated pipeline stage are now distinguishable, and activities carry real start/end timing. Wired through all 18 `*_provenance.py` wrapper modules and `track_entity`/`track_relationship`/`track_chunk`/`track_property_source`
|
||||
- **Versioning vs. derivation split**: new `previous_version_id` ("this corrects a prior version of the same fact") and `derived_from_id` ("this was derived from a different source entity") fields, additive alongside the legacy combined `parent_entity_id` so existing readers are unaffected
|
||||
- **Downstream lineage traversal**: new `get_descendants()`/`trace_descendants()` (reverse BFS in both `InMemoryStorage` and `SQLiteStorage`), closing the gap flagged in `semantica/explorer/routes/provenance.py` where `direction="downstream"` was dead code with no reverse lookup to feed it; the Explorer's `/api/provenance` lineage response now merges both directions
|
||||
- **W3C PROV-O qualified relations**: `export_prov()` now emits `prov:qualifiedAssociation`/`hadRole` (distinguishing "approved by" from "generated by" for sign-off workflows), `qualifiedGeneration`/`Generation`, `qualifiedUsage`/`Usage`, `qualifiedDerivation`/`Derivation`, `qualifiedInvalidation`/`Invalidation`, `wasAssociatedWith` (Activity→Agent), `actedOnBehalfOf` (Agent→Agent delegation), and `wasInformedBy` (Activity→Activity, via a new `informed_by=[...]` kwarg), alongside the existing plain triples
|
||||
- **Bitemporal + Bundle support**: `revision_type`/`supersedes`/`valid_from`/`valid_until` fields (plain caller-supplied passthrough, matching the deprecated `kg.ProvenanceTracker`'s actual contract) plus new `revision_history()` and `query_recorded_between()` methods, closing the two "no direct equivalent yet" rows in `docs/migration/kg-provenance-tracker.md`; `bundle_id` emits `prov:Bundle`/`hadMember` membership triples to partition provenance by source/dataset/ingestion-run
|
||||
- **Configurable, interlinked namespace**: `export_prov(base_uri=...)` / `--base-uri` CLI flag, defaulting to a new `ProvenanceManager.DEFAULT_BASE_URI` (`https://semantica.dev/ns#`) that `RDFExporter`'s `NamespaceManager` and `OWLExporter`'s default `ontology_uri` now both reuse, so KG-exported, OWL-exported, and PROV-exported URIs for the same `entity_id` co-resolve instead of three independently-hardcoded placeholder domains
|
||||
- New CLI commands: `semantica provenance invalidate|verify-chain|descendants`
|
||||
- **Fixed along the way**: `track_entities_batch()` silently absorbed batch-level typed kwargs (`agent_id`, `entity_type`, `activity_id`) into the opaque `metadata` JSON blob instead of forwarding them, so the documented banking example in `docs/guides/provenance.md` never actually worked as written
|
||||
- **Fixed along the way**: `compute_checksum()` had to exclude `entity_id` itself from the hash — `track_entity()`'s versioning archives a prior value by copying it to a new key (`"X"` → `"X:v:<timestamp>"`), and hashing `entity_id` meant that legitimate relabel permanently orphaned any other entry that had already chained its `previous_checksum` from the pre-relabel value, surfacing as a false-positive "broken chain." Archival and invalidation are now always a pure relabel (unchanged checksum/sequence position) followed by a fresh chained append, never an in-place mutation of an already-chained entry
|
||||
- **Fixed along the way**: `InMemoryStorage.get_chain_head()` ignored the already-committed chain head whenever the current transaction had staged any entries, understating the head and corrupting the next append's chain link
|
||||
- **Fixed along the way**: several new `ProvenanceEntry` fields were initially wired into the dataclass and `export_prov()` but not into `SQLiteStorage`'s DDL/INSERT/row-mapping — `InMemoryStorage` stores the dataclass directly so it masked the gap. Added a permanent regression test (`test_all_fields_round_trip_through_sqlite`) asserting every field survives a SQLite round trip, to catch this class of bug for any future field additions
|
||||
- Flagged, not fixed (separate, pre-existing issues independent of #825): `semantica/pipeline/pipeline_provenance.py` imports a nonexistent module and wraps a `Pipeline` dataclass with no `run()` method, so `PipelineWithProvenance` has never worked; most of the 18 wrapper modules' backing classes are themselves missing or incomplete (e.g. `context.context_manager`, `deduplication.deduplicator`, `normalize.normalizer` don't exist; `EmbeddingGenerator` exists but has no `.embed()`); `kg_provenance.py` passes `entity_type` inside its `metadata={}` dict instead of as a top-level `track_entity()` kwarg across most of its ~30 call sites, so it never actually populates the real field
|
||||
- Extensive new test coverage across `tests/provenance/test_manager.py`, `test_schemas.py`, and `test_storage.py` (invalidation, hash-chain verification including a simulated hard-delete-detection case and an interleaved-chaining stress test, agent/activity typing, versioning/derivation split, downstream lineage, qualified export triples, bitemporal methods, Bundle export, and namespace interlinking)
|
||||
|
||||
- **Altair Anzo triplet store backend** (#813) by @KaifAhmad1
|
||||
- Added `AnzoStore` (`semantica/triplet_store/anzo_store.py`), a fourth peer to `BlazegraphStore`/`RDF4JStore`/`JenaStore` speaking plain SPARQL 1.1 over HTTP — no new dependency, since Anzo has no official Python SDK but needs none
|
||||
- The one structural difference from the existing backends: Anzo addresses data by a dataset/graphmart **URI** (`dataset_uri`, required) rather than a short namespace/repository name, so the endpoint path (`<endpoint>/sparql/<store_type>/<url-encoded_dataset_uri>`) percent-encodes it; `store_type` defaults to `"graphmart"` and can be set to `"dataset"`
|
||||
- Reuses the shared `sparql_escaping.py` literal-escaping, datatype-IRI resolution, and CONSTRUCT-detection helpers rather than reimplementing them, matching `BlazegraphStore`'s CONSTRUCT/bindings `execute_sparql` contract exactly
|
||||
- Wired into `TripletStore` (`backend="anzo"`, added to `SUPPORTED_BACKENDS` and `NAMED_GRAPH_CAPABLE_BACKENDS`) and `config.py` (`TRIPLET_STORE_ANZO_ENDPOINT` env var / `anzo_endpoint` config key), and exported from `semantica.triplet_store`
|
||||
- 32 new tests in `tests/triplet_store/test_anzo_store.py` (mocked HTTP, no live Anzo instance needed), including dataset-URI percent-encoding cases that don't apply to the other backends
|
||||
- Bulk loading uses SPARQL `INSERT DATA` (the same approach `BlazegraphStore` uses) rather than Anzo's separate HTTP Client Interface, keeping the `bulk_load()` contract identical across backends
|
||||
|
||||
- **Comprehensive unit and security test suite for the `/api/sparql` Explorer route** (#773) by @Sameer6305
|
||||
- Added `tests/explorer/test_sparql_route.py` (34 tests) covering the SPARQL Explorer route (`semantica/explorer/routes/sparql.py`), which executes arbitrary SPARQL queries against an in-memory rdflib projection of the live graph and previously had zero test coverage
|
||||
- Verified read-only allowlist enforcement against write and mutation queries (`INSERT DATA`, `DELETE DATA`, `DELETE WHERE`, `DROP ALL`, `CLEAR ALL`, `LOAD`, `CREATE GRAPH`, `MODIFY`, comments, and multi-statement injections like `SELECT ... ; DROP ALL`), confirming rejected queries short-circuit before any graph is built or queried
|
||||
- Verified resource-limiting behavior, confirming row capping (`_SPARQL_MAX_ROWS`) truncates results and sets `truncated: true`, query timeout (`_SPARQL_TIMEOUT_S`) returns a clean error message without crashing, and concurrency semaphore (`_SPARQL_MAX_CONCURRENT`) prevents thread starvation under load
|
||||
- Verified RDF projection fidelity for node properties and edge relationships, and error formatting for malformed SPARQL syntax with line and column extraction
|
||||
- Follow-up review fixes (#805): extracted the duplicated row-cap-and-truncate loop (previously copy-pasted between the `CONSTRUCT`/`DESCRIBE` and `SELECT` branches) into a shared `_cap_rows()` helper so the `_SPARQL_MAX_ROWS` cap is enforced identically by both; added `test_row_cap_truncates_construct_results`, since the truncation path for `CONSTRUCT`/`DESCRIBE` results had no direct test coverage even though `SELECT` truncation did
|
||||
|
||||
- **Global default persistent storage for `ProvenanceManager`, plus a working `provenance` CLI** (#795, #802) by @Sameer6305 and @KaifAhmad1
|
||||
- Every ingestion/processing module (`kg_provenance.py`, `pipeline_provenance.py`, and 20+ other call sites) instantiated its own `ProvenanceManager()` with no `storage_path`, so all of them silently fell back to `InMemoryStorage` and the SQLite audit trail was never actually written. `ProvenanceManager.set_default_storage_path(path)` now sets a class-level default that every no-arg instantiation picks up, and `Semantica.__init__` wires `config.provenance.storage_path` into it automatically during orchestrator init
|
||||
- Added the thread-safe `default_storage_path(path)` context manager (`semantica.provenance.default_storage_path`) for test isolation — it stacks nested overrides and guarantees restoration of the previous default on exit, even on exception, so tests can't leak global state into each other
|
||||
- Fixed `ProvenanceManager.__init__` raising `TypeError` on the CLI's `config=` kwarg, and implemented the four methods the CLI already called but that didn't exist on the class: `lineage()`, `audit_log()`, `export_prov()` (W3C PROV-O turtle/ntriples/jsonld via `rdflib`), and `check()` — unblocking `semantica provenance lineage|audit|export|check` end-to-end
|
||||
- Follow-up review fixes: `track_entity` no longer aliases a caller-supplied `used_entities` list (it copied the reference and later mutated it in place via `.append()`, which could corrupt a list the caller still held); removed dead fallback branches in `orchestrator.py`/`manager.py` left over from not realizing `Config.get()` already resolves dotted paths; added a `--dry-run` option to `provenance audit` to match `provenance export` (previously only the global `--dry-run` flag worked, not a local one); and `provenance check --strict` no longer prints a green "✓" success line immediately before failing — a failing check now renders as a warning before the `ClickException` is raised
|
||||
|
||||
- **Markdown round-trip export/import for `AgentMemory`** (#765, #786) by @SaurabhScripts and @Sameer6305
|
||||
- `AgentMemory.export(format="markdown")` and `import_data(format="markdown")` add a human-editable, diff-friendly alternative to the existing JSON/dict serialization: one Markdown file per memory item, with `id`, `created_at`, `updated_at`, and `type`/`kind` in required YAML frontmatter and the memory content as the Markdown body
|
||||
- Exporting without a `destination` returns a single memory as a Markdown string; exporting a set requires a destination directory and writes one stable, content-hashed filename per memory ID, so re-exporting an unchanged set is byte-for-byte idempotent
|
||||
- Importing upserts by ID: unknown IDs create new memories, known IDs replace them atomically (local state and vector store are only mutated after the whole batch validates cleanly), and unchanged re-imports are a deterministic no-op
|
||||
- Malformed frontmatter, duplicate IDs within an import batch, and duplicate YAML keys are all rejected before any memory is mutated, with actionable error messages
|
||||
- Export refuses to overwrite symbolic links and replaces files atomically; import safely compares timezone-aware and timezone-naive timestamps so retention, recency sorting, and date filters stay correct across both
|
||||
- Entities and relationships round-trip as memory-local provenance only — Markdown import intentionally does not write into `ContextGraph`, matching the MVP scope agreed on in #765
|
||||
- Documented the file contract and workflow in `docs/reference/context.md`; 43 new tests in `tests/context/test_agent_memory_markdown.py` cover round-trip losslessness, idempotency, validation errors, rollback on failure, and vector-store sync ordering
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`VectorStore.search_vectors()` returned inconsistent result shapes across backend implementations** (#853, closes #845) by @Sameer6305, reviewed by @KaifAhmad1
|
||||
- Every built-in backend (FAISS, Milvus, pgvector, Pinecone, Qdrant, SQLite-vec, Weaviate, in-memory) now returns the same canonical `SearchResult` shape (`id`, `score`, `metadata`, `vector`, `distance`), instead of some backends omitting `vector`/`metadata`/`distance` or, for Weaviate, returning a backend-specific `properties` key instead of `metadata`
|
||||
- Added a `SearchResult` `TypedDict` (`semantica/vector_store/vector_store.py`, exported from `semantica.vector_store`) documenting the contract; `metadata` now always defaults to `{}` rather than being absent, and `id` accepts `Union[str, int]` to accommodate Milvus/Qdrant's native integer IDs without casting
|
||||
- **Review fix**: the score-normalization formula added for Pinecone and Qdrant (`1.0 / (1.0 + max(0.0, 1.0 - score))`) clamped every raw score `>= 1.0` to an identical `1.0`, silently collapsing result ranking whenever the raw score could exceed 1 — which happens routinely for dot-product-metric indexes (unbounded), as opposed to cosine (bounded to `[-1, 1]`). Replaced with `(score / (1 + |score|) + 1) / 2`, which is strictly monotonic and bounded in `(0, 1)` for any real input, so ranking order is preserved regardless of metric or vector normalization
|
||||
- Added `test_qdrant_unbounded_dot_product_scores_preserve_ranking` and `test_pinecone_unbounded_dotproduct_scores_preserve_ranking` (`tests/vector_store/test_search_result_schema.py`) asserting normalized scores stay strictly ordered and bounded for raw scores well above 1.0, the case the original formula silently collapsed and the existing tests (which only used scores `< 1`) never exercised
|
||||
- Left out of scope, per the original PR: Weaviate's `similarity_search()` still isn't wired into `VectorStore.search_vectors()`'s backend dispatch; Milvus's collection schema still has no metadata column so its results always return `metadata: {}`; and `include_vectors` support (populating the `vector` field) is not yet implemented for any backend
|
||||
|
||||
- **`DecisionEmbeddingPipeline.find_similar_decisions()` crashed with `AttributeError` for any `VectorStore` backend other than `inmemory`** (#842, closes #839) by @Sameer6305
|
||||
- `_get_candidate_embeddings()` iterated `VectorStore.vectors`/`VectorStore.metadata` directly, internal dicts only populated for `backend="inmemory"`; every persistent backend (FAISS, Pinecone, Qdrant, Milvus, ...) raised `AttributeError`. It now fetches candidates via the backend-agnostic `VectorStore.search_vectors()`, reading metadata via a `res.get("metadata") or res.get("payload")` fallback for backends that key it differently
|
||||
- Backends such as FAISS don't return the raw vector for each hit; `find_similar_decisions()` and `_find_semantic_similar()` now fall back to the search-provided score (normalized from `distance` when present) as the semantic similarity for those candidates instead of computing cosine similarity against a zero placeholder vector
|
||||
- `get_decision_statistics()` had the identical bug iterating `store.metadata.values()`; it now returns a limited stats payload with an explanatory `warning` field for backends that don't expose a full in-memory metadata dict, instead of crashing
|
||||
- **Fixed along the way**: `_get_candidate_embeddings()`'s expand-and-retry loop (which widens the search pool when post-filtering leaves too few matches) discarded every candidate it had found once the pool hit its cap (`limit * 10`) without ever collecting `limit` matches or getting a short page back from the backend — the loop fell through without executing the branch that assigns results, silently returning `[]` even when matching candidates existed. It now falls back to the last batch collected instead of dropping it
|
||||
- Added end-to-end regression tests against real `inmemory` and `faiss` backends (no mocks) plus a targeted unit test for the expand-and-retry loop's fallback behavior
|
||||
|
||||
- **`QdrantStore.search_vectors()` returned results keyed by `"payload"` instead of `"metadata"`** (#841, closes #840) by @divyankshah
|
||||
- `QdrantCollection.search_points()` built its result dicts as `{"id", "score", "payload"}`, while `PineconeStore.search_vectors()` and every other backend consumed by `HybridSearch` use `"metadata"`. This silently dropped Qdrant metadata from results and made `HybridSearch.filter_by_metadata()` reject every candidate whenever a filter was applied, since it looks up `result["metadata"]` and got nothing back
|
||||
- Normalized `search_points()` to return `"metadata"` instead of `"payload"`, matching the existing convention; no other module reads the old key, so the rename is a straight fix rather than a partial one
|
||||
- Extended `tests/vector_store/test_vector_store_deepdive.py::test_qdrant_store` to assert the returned key is `"metadata"` (not `"payload"`) and that `HybridSearch.filter_by_metadata()` correctly matches against Qdrant results end-to-end
|
||||
|
||||
- **Explorer Temporal panel never rendered after clicking the toolbar button** (#830, #836) by @Sameer6305
|
||||
- The panel stayed permanently stuck on "Loading temporal…" in `npm run dev`, with repeating "Maximum update depth exceeded" errors in the browser console. Two independent render loops were responsible:
|
||||
- **Diagnostics state churn**: `handleDiagnosticsChange` unconditionally called `setGraphDiagnosticsState` on every invocation. `buildEffectAvailability` (inside `GraphCanvas`'s diagnostics `useEffect`) always returns a new object, so each call scheduled a re-render that immediately retriggered the effect. Fixed by comparing the incoming snapshot field-by-field against the last accepted value via `lastDiagnosticsRef` before calling `setState`
|
||||
- **scrubberTime churn**: React 18 concurrent mode re-ran `TimelinePanel`'s `useEffect` with a structurally-new `Date` object for the same timestamp when speculative renders discarded `useMemo` caches, causing repeated `setScrubberTime` calls that propagated into `temporalState` churn and retriggered the diagnostics effect. Fixed by deduplicating by millisecond value via `onTimeChange`/`lastScrubberMsRef`
|
||||
- **Bonus**: `temporal-overlay`'s `shouldLoad` predicate was changed to gate strictly on `panelState["temporal-panel"]`, removing the `|| temporalState?.currentTime` branch that caused eager loading on every scrubber update and continuously cancelled in-flight `load()` completions
|
||||
- **Bonus**: `temporalState` removed from the plugin-loading `useEffect` dependency array; predicates extracted into `pluginRegistryPredicates.ts` and wired through `GraphWorkspace.tsx` so regression tests exercise the production code rather than a local copy
|
||||
- The `scrubberTime`-churn fix was also applied to the equivalent (but currently unused/unmounted) `GraphWorkspaceShell.tsx`, which shares the same `TimelinePanel` integration pattern but does not have the diagnostics-churn code path
|
||||
- **Follow-up review fix**: the diagnostics dedup's `structureLayer` comparison now also covers `disabledReason`, `curveCount`, `bridgeCurveCount`, and `backboneCurveCount` (previously only `cacheKey`/`lastDrawAt`/`enabled` were compared, so a pure `disabledReason` transition could leave the dev-only diagnostics panel stale)
|
||||
- **Follow-up review fix**: `test:graph-store`, `test:graph-workspace`, and the new `test:plugin-registry` regression test are now run in CI (`.github/workflows/ci.yml`) — previously none of the Explorer frontend's `node --test` suites executed anywhere in CI, only `npm run build`, so this fix's own regression coverage (and all prior frontend test coverage) provided no protection against silent regressions
|
||||
- **`HybridSearch.search()` crashed with `AttributeError` for any `VectorStore` backend other than `inmemory`** (#833, #837) by @KaifAhmad1
|
||||
- `HybridSearch.search()` read `self.vector_store.vectors` directly, an internal dict `VectorStore` only populates for `backend="inmemory"`; every other backend (faiss, weaviate, qdrant, milvus, pinecone, pgvector, sqlite) raised `AttributeError`, making `HybridSearch` unusable against any real store. It now delegates to `VectorStore.search_vectors()` (the backend-agnostic public API) for non-inmemory backends, applies `metadata_filter` as a post-filter over the returned candidates, and normalizes results to a consistent `{id, score, distance, metadata}` shape
|
||||
- **Fixed along the way**: `vector_ids` could stay `None` when callers passed explicit `vectors`/`metadata` without `vector_ids`, crashing downstream list indexing — now defaulted to generated positional IDs
|
||||
- **Fixed along the way**: a `query_vector` passed as a plain list crashed backend stores (e.g. `FAISSStore.search_similar`) that call `.ndim` on it — now normalized to a numpy array up front
|
||||
- **Fixed along the way**: `VectorStore.store_vectors()` silently dropped metadata for FAISS (and any `add_vectors`-only backend) because it called `add_vectors(vectors, **options)` without forwarding `metadata`, even though `FAISSStore.add_vectors()` accepts it — this blocked `HybridSearch`'s metadata filtering from ever matching anything on FAISS
|
||||
- **Follow-up review fixes**: the legacy `top_k` kwarg was read but left in `options`, then forwarded via `**options` into `VectorStore.search_vectors()`, colliding with backends (sqlite, pgvector) that pass an explicit `top_k=k` to their own `search()` and raising `TypeError: got multiple values for keyword argument 'top_k'` — now popped instead of just read; `VectorStore.search_vectors()`'s dispatch only recognized backend methods named `search`/`search_similar`, so delegation still hit `NotImplementedError` for qdrant/milvus/pinecone, which name their method `search_vectors()` with a differently-named count parameter (`limit` vs `k`) — added a third dispatch branch that binds the count positionally so it works regardless of the backend's parameter name; a missing `distance` in backend-delegated results defaulted to the raw `score`, silently reusing the local path's cosine-similarity convention (`distance = 1 - score`) even for backends using unrelated metrics (L2, inner product) — now left as `None` instead of a fabricated, metric-inconsistent value
|
||||
- Verified across all 7 supported backends: `inmemory`/`faiss`/`sqlite` work live end-to-end; `pgvector`'s dispatch reaches `PgVectorStore.add()`/`.search()` (blocked only by no Postgres server in the verification sandbox); `qdrant`/`milvus`/`pinecone` now reach their real `search_vectors()` method instead of crashing, though their storage side (`store_vectors()`) still doesn't recognize `insert_vectors`/`upsert_vectors`, and `weaviate` remains entirely unwired (`add_objects`/`query_vectors`) on both sides — both are separate, pre-existing gaps independent of this fix, left for a follow-up
|
||||
|
||||
- **`VectorStore.store_vectors()` silently dropped metadata for FAISS (and any `add_vectors`-only) backend** (#832, #835) by @KaifAhmad1
|
||||
- `store_vectors()` fell into a branch that called `self._backend_store.add_vectors(vectors, **options)` without `metadata` whenever the backend exposed `add_vectors()` but neither `add()` nor `store_vectors()` — true for `FAISSStore`, the backend most real usage configures for genuine ANN search. Every caller that stores vectors with metadata (e.g. `AgentMemory._store_memory_vector()`, used internally by `AgentContext.store()`) lost that metadata once it reached FAISS, with no error or warning
|
||||
- Downstream, `ContextRetriever._retrieve_from_vector()` recovers a result's text via `metadata.get("content", "")`, which was always `""` for any vector stored this way; `_rank_and_merge()` then embedded that empty string, tripping `TextEmbedder.embed_text()`'s empty-text rejection and masking the real bug as a spurious `TextEmbedder` failure recorded by the progress tracker
|
||||
- `store_vectors()` now forwards `metadata` to `add_vectors()`, but only when the backend's `add_vectors()` signature actually accepts it (checked via `inspect.signature`, accepting either an explicit `metadata` parameter or a `**kwargs` catch-all), so a future/custom backend with a stricter signature raises no `TypeError`
|
||||
- **Follow-up review fix**: the `inspect.signature()` probe is wrapped in `try/except (ValueError, TypeError)`, consistent with the identical pattern already used in `ProvenanceManager.trace_lineage()`, so signature introspection failing on an unusual callable can no longer abort `store_vectors()` before it even attempts to call the backend
|
||||
|
||||
- **`AgnoDecisionKit.check_policy` silently treated unevaluable policy rules as compliant** (#778, #822) by @Sameer6305
|
||||
- `_eval_rule()` previously `return`ed `True` when a rule referenced a field missing from the decision payload, or when the rule string didn't match the expected `<field> <op> <value>` format — the docstring's claim that exceptions never silently return `compliant=True` didn't cover this, since neither path raised
|
||||
- Both cases now raise `ValueError` instead, which routes through `check_policy`'s existing exception handler and records a `warnings` entry (e.g. `"Could not evaluate rule 'minimum_score >= 0.9': rule references undefined field 'minimum_score'"`) instead of disappearing with no signal
|
||||
- `violations`/`compliant` are unaffected — an unevaluable rule is not counted as a violation, since it's genuinely unknown whether it would have passed; this matches the existing `compliant`/`violations`/`warnings` shape already used by `ContextGraph.enforce_decision_policy`
|
||||
- This is additive: `warnings` was already part of the return contract and populated for other exception cases, so no caller that only checks `compliant` is affected, and no existing test asserts `warnings == []` for a payload that hits either of these paths
|
||||
- **Follow-up review fix**: `check_policy` decoded `policy_rules` with `json.loads` and iterated the result without checking it was actually a list; a JSON-encoded bare string (e.g. `policy_rules='"confidence >= 0.7"'`) decodes to a `str`, so iterating it evaluated one "rule" per character — combined with the fix above, an 18-character rule string produced 17 warnings instead of being treated as the single rule it was meant to be. A decoded string is now wrapped as a single-element rule list; any other non-list shape (number, object, etc.) or non-string list element now produces exactly one `warnings` entry instead of silently misbehaving or being iterated character-by-character
|
||||
- **Follow-up review fix**: `_eval_rule` used `data.get(field) is None` to detect a missing field, which can't distinguish a genuinely absent key from a key explicitly present with a JSON `null` value — both produced the same "undefined field" warning, misdiagnosing nullable fields. Field presence is now checked with `field not in data` first; a present-but-`null` value now raises a distinct `"field {field!r} is null — cannot evaluate rule"` message instead of the misleading "undefined field" one
|
||||
- **Follow-up review fix**: `check_policy` only checked that `decision_data` was valid JSON, not that it decoded to an object. When it decoded to a list, `field not in data` silently became list-*membership* testing instead of a key check (e.g. `"confidence" not in ["confidence", 0.95]` is `False`), so a matching rule fell through to `data["confidence"]`, which raised a raw, confusing `TypeError: list indices must be integers or slices, not str` instead of any meaningful diagnostic; numbers/strings/bools produced similarly opaque `TypeError`s. `check_policy` now rejects any `decision_data` that doesn't decode to a JSON object upfront with a single clear `violations` entry, the same way it already rejects malformed JSON
|
||||
- Added 15 tests to `tests/integrations/agno/test_decision_kit.py` covering the missing-field case (the issue's traced example), the malformed-rule-string case, the bare-JSON-string `policy_rules` amplification case, non-list/non-string `policy_rules` shapes, the missing-key-vs-null-value distinction, non-object `decision_data` shapes (list/number/string/bool/null), and regression checks confirming normal rule evaluation on present fields is unchanged
|
||||
|
||||
- **No cycle detection for SKOS concepts at write time** (#774, #819) by @mikemikimike, reviewed by @Sameer6305 and @KaifAhmad1
|
||||
- Added cycle detection (`validate_skos_hierarchy`) for `skos:broader` and `skos:narrower` relationships in `ContextGraph.add_edge()` and `ContextGraph.add_edges()`, preventing direct 2-node cycles, self-loops, and multi-hop hierarchy cycles
|
||||
- Added `GraphSession.add_nodes_and_edges()` to validate SKOS hierarchy edges upfront under lock before node insertion, preventing partial-write leaks where nodes remain after a cyclic edge is rejected
|
||||
- Updated vocabulary, ontology (`/api/ontology/load`, `/api/ontology/create`), and JSON/CSV import routes to use `add_nodes_and_edges()` and return HTTP 422 with actionable error messages when a cycle is detected
|
||||
- Follow-up fix by @KaifAhmad1: `validate_skos_hierarchy()` previously re-walked *every* SKOS hierarchy edge already in the graph on each write, so one pre-existing cycle anywhere (e.g. legacy data) blocked all unrelated future writes; it now only traverses concepts touched by the edges being written, while still checking against existing edges for cycles that span old and new data
|
||||
- Follow-up fix by @KaifAhmad1: in `/api/ontology/load`, `except HTTPException: raise` was unreachable because a broader `except Exception` clause above it already matched `HTTPException`, so a 422 raised after a successful `OntologyIngestor` parse was silently swallowed and reprocessed via the fallback RDF parser; reordered the clauses so the deliberate 422 always propagates
|
||||
- Follow-up fix (#775): `/api/ontology/{uri}/refresh` was missed by the original sweep and still called `session.add_nodes()` then `session.add_edges()` as two independent operations, so a cyclic SKOS edge rejected by `add_edges()` left the nodes from the preceding `add_nodes()` call committed to the graph; switched to `session.add_nodes_and_edges()` with the same `except ValueError` → HTTP 422 handling already used by `/api/ontology/load` and `/api/ontology/create`. Audited every other `add_nodes()`/`add_edges()` pairing in the repo (`GraphStore`, `graph_builder.py`, `agent_memory.py`, `context_graph.py.load()`, `enrich.py`) — none share `GraphSession`'s SKOS-cycle-validation write path, so none were changed
|
||||
|
||||
- **Agno `_AgentScopedStore.upsert_memory` silently swallowed decision recording failures** (#779)
|
||||
- `upsert_memory()` now logs `logger.warning("[%s] record_decision failed: %s", self._role, exc, exc_info=True)` when `record_decision()` fails, matching the error-logging convention used for `store()` in the same method with traceback context preserved
|
||||
- Preserves graceful fallback behavior: `record_decision()` remains optional and `upsert_memory()` continues without propagating the exception
|
||||
- Added regression coverage in `tests/integrations/agno/test_shared_context.py` for both `store()` and `record_decision()` warning paths
|
||||
|
||||
- **`AgnoDecisionKit`/`AgnoKGToolkit` silently swallowed Agno tool registration failures** (#780, #818) by @Sameer6305 and @KaifAhmad1
|
||||
- Removed the `try/except: pass` wrapped around `self.register(fn)` in both toolkits' `__init__`; when Agno is installed, a registration failure now propagates immediately instead of leaving the toolkit half-registered with no signal to the caller
|
||||
- Graceful degradation when Agno isn't installed (`AGNO_AVAILABLE=False`) is unchanged — `_tools` is still populated so callers can introspect available tools without the package
|
||||
- Fixed a related duplicate-entry bug: `self._tools` was appended to unconditionally *before* `register()` ran, which could double-count a tool when Agno's own `Toolkit.register()` also tracks it in `self._tools`
|
||||
- This is a behavior change for callers that construct these toolkits expecting instantiation to always succeed — audited: no in-repo call site relies on the old silent-failure behavior
|
||||
- Expanded `tests/integrations/agno/test_decision_kit.py` and `test_kg_toolkit.py` with coverage for registration invocation counts, failure propagation, graceful degradation, and no-duplicate-`_tools` assertions
|
||||
|
||||
- **`ProvenanceManager` tracking methods silently swallowed failures without logging and returned fabricated entries** (#783)
|
||||
- `track_relationship()`, `track_chunk()`, and `track_property_source()` now return `Optional[ProvenanceEntry]` (`None` on storage failure, consistent with #782's `track_entity` fix) instead of a fabricated populated object
|
||||
- `_save_entry()` now always logs on any storage failure, including previously-silent per-item batch failures
|
||||
- `track_entities_batch()` and `track_chunks_batch()`'s rare block-level transaction failures are now logged too
|
||||
- `source_tracker.py`'s `track_sources_batch()` no longer counts failed tracking calls in its stats
|
||||
|
||||
- **MCP `handle_get_causal_chain` returned an empty-but-valid-looking response when both `CausalChainAnalyzer` and the graph fallback were unavailable** (#781, #817) by @Sameer6305 and @KaifAhmad1
|
||||
- Returns an explicit `{"error": "Causal chain analysis is not supported on this graph backend", "chain": []}` instead of `{"chain": [], "count": 0, "direction": ...}`, letting clients distinguish "unsupported" from a legitimately empty chain
|
||||
- The fallback path now introspects `graph.get_causal_chain`'s signature to forward `direction`/`max_depth` (or a `depth` kwarg, or nothing, depending on what the backend accepts) instead of always calling with just `decision_id`, matching the primary analyzer path's behavior
|
||||
- Hardened input handling: non-dict `args`, non-string `decision_id` (previously a latent `AttributeError` on `.strip()`), and `max_depth` clamped to `(0, 100]` with a safe default on invalid input
|
||||
- Added `tests/test_mcp_decisions_causal_chain.py` (11 tests) covering the unsupported-backend, fallback-forwarding, and validation/exception paths across multiple backend signature shapes
|
||||
- **Follow-up review fix**: the signature-detection try/except previously caught the *actual call*'s exceptions in the same block used for introspection failures, so a genuine bug inside a backend's `get_causal_chain` (raising an unrelated `TypeError`) was misread as a signature mismatch and the backend was invoked a second time with identical arguments before the real error surfaced. Signature introspection and the resulting call are now split into separate try/excepts so a successfully-introspected call is made exactly once; added `test_internal_typeerror_calls_backend_only_once` to lock this in
|
||||
|
||||
- **`ProvenanceManager.track_entity` persisted partial history and returned fabricated entries on storage failure** (#782, #816) by @Sameer6305 and @KaifAhmad1
|
||||
- `track_entity()`'s two-step write (history archive + primary update) is now atomic — if either write fails, the whole operation rolls back via the existing #807 `transaction()` mechanism, instead of silently persisting a partial state
|
||||
- `track_entity()`'s return type is now `Optional[ProvenanceEntry]`: on failure it returns a safe deep copy of the pre-failure existing entry (if one existed) or `None` (if this was a brand-new, never-successfully-tracked entity) — never a fabricated object claiming values that were never actually persisted
|
||||
- This is a behavior change for callers that inspect the return value without checking for `None` first — audited: 0 of 47 production call sites in the repo currently dereference the return value, so this is safe today, but any NEW caller must handle `None`
|
||||
- `InMemoryStorage` gained real transactional rollback (staging-buffer based) to match this guarantee — previously `transaction()` was a no-op
|
||||
|
||||
- **`ProvenanceManager` duplicated the same checksum/persist/exception-swallow block across 4 tracking methods** (#784, #815) by @Sameer6305 and @KaifAhmad1
|
||||
- Consolidated the repeated `entry.checksum = compute_checksum(entry)` / `try: self.storage.store(entry) except Exception: pass` block used by `track_entity`, `track_relationship`, `track_chunk`, and `track_property_source` into a single `ProvenanceManager._save_entry()` helper, preserving the existing graceful-failure behavior and the batch `_conn`/re-raise semantics from #807
|
||||
- Added 4 regression tests (`tests/provenance/test_manager.py`) covering storage-failure swallowing for each of the four tracking methods, none of which had coverage for this path before
|
||||
- **Follow-up review fix**: the initial refactor of `track_entity`'s exception fallback (the branch that runs when a failure happens *before* the entry is built, e.g. a retrieve error inside the atomic transaction) routed through `_save_entry()`, which made a new `self.storage.store(entry)` call outside the already-failed transaction — a real behavioral change from the original code (which only computed a checksum on that path) that could have reintroduced the exact race #807's `BEGIN IMMEDIATE` transaction serialization was meant to prevent. Reverted that branch to only compute the checksum, and added `test_track_entity_pre_build_failure_fallback_skips_store` asserting `storage.store` is never called on that path
|
||||
|
||||
- **`SQLiteStorage` and `ProvenanceManager` connection churn, non-atomic writes, and batch tracking overhead** (#807) by @Sameer6305
|
||||
- Scoped a single SQLite connection to the full duration of each public storage method call (`track_entity()`, `store()`, `retrieve_all()`, `clear()`) instead of opening independent connections per internal SQL statement, reducing connection churn by ~67% while closing the handle before the public method returns to preserve Windows filesystem unlink safety
|
||||
- Implemented the `SQLiteStorage.transaction()` context manager with Write-Ahead Logging (`PRAGMA journal_mode=WAL`), `busy_timeout=5000`, `synchronous=NORMAL`, and immediate write transactions (`BEGIN IMMEDIATE`), ensuring concurrent read-modify-write sequences (including history version ID generation) are serialized without lock contention or data loss
|
||||
- Added block-level transaction sharing to `track_entities_batch()` and `track_chunks_batch()`, reducing SQLite commit overhead by ~99.9% for large batches and deferring `tracked_count` increments until successful commit so rolled-back items are never reported as successes
|
||||
- Preserved 100% backward compatibility for custom storage backends overriding `trace_lineage(self, entity_id)` by inspecting signatures dynamically before passing `max_depth`, and optimized BFS lineage queries with batched IN-clause lookups per frontier level
|
||||
- **Follow-up fix**: `retrieve()` and `trace_lineage()` were initially routed through `transaction()` too, so plain reads took the same `BEGIN IMMEDIATE` writer lock as read-modify-write calls, serializing every read behind every other read/write and defeating the WAL concurrency this PR was meant to add. They now use a dedicated `_read_connection()` (configured, no explicit `BEGIN`) so reads no longer contend for the writer lock
|
||||
- **Follow-up fix**: `track_entity()`/`track_chunk()` caught all internal storage exceptions unconditionally, so when called from `track_entities_batch()`/`track_chunks_batch()`'s shared per-block transaction, a single item's storage failure (e.g. non-JSON-serializable metadata) was swallowed inside the call and never surfaced to the batch loop's per-item `except`, inflating `tracked_count` for entries that were never persisted. Both methods now re-raise when invoked with a shared `_conn` (batch context) while still degrading gracefully on standalone calls, so batch counts match what's actually committed
|
||||
- Added 8 dedicated regression tests in `tests/provenance/test_sqlite_storage_performance_807.py` covering PRAGMA configuration, Windows unlink safety, batch transaction sharing, BFS `max_depth`, rollback count accuracy, custom storage backward compatibility, concurrent read-modify-write serialization, and connection cleanup guards on configuration error
|
||||
|
||||
- **Closed remaining `ProvenanceManager` storage-failure test-coverage gaps identified by a #785 audit** (#785)
|
||||
- An audit of `tests/provenance/` (filed against a claim that zero tests exercised `storage.store()` failures) found #782/#783/#784/#807 had already closed most of the gap, but two residual surfaces had no test: `track_relationship()`, `track_chunk()`, and `track_property_source()`'s storage-failure-swallowing contract (returns `None`, logs, persists nothing) was only verified against `InMemoryStorage`, never `SQLiteStorage`; and `track_chunks_batch()` had no test for per-item `_save_entry` failure logging or for the block-level transaction-failure log message, even though `track_entities_batch()` had both
|
||||
- No production code changed — #782/#783/#784/#807 already implemented the correct behavior; this closes the coverage gap proving it holds on both backends
|
||||
- Added `test_track_relationship_storage_error_swallowed_sqlite`, `test_track_chunk_storage_error_swallowed_sqlite`, `test_track_property_source_storage_error_swallowed_sqlite`, `test_chunks_batch_logs_per_item_failure_memory`, and `test_track_chunks_batch_block_level_transaction_failure_logs` to `tests/provenance/test_manager.py`
|
||||
- Read-path failure coverage (`get_lineage()`/`trace_lineage()`/`get_provenance()`/`clear()` propagating a raised storage exception) remains untested and is a candidate for a follow-up issue, since none of those methods currently wrap the underlying storage call in a try/except
|
||||
|
||||
- **Explorer's Provenance UI used a naive 2-hop graph traversal instead of the audit-grade `ProvenanceManager` backend** (#792, #809) by @Sameer6305
|
||||
- `semantica/explorer/routes/provenance.py` never imported or called `ProvenanceManager` (`semantica/provenance/manager.py`); `/api/provenance` and `/api/provenance/report` built their lineage response entirely from a naive 2-hop networkx traversal over the live graph instead of querying the SQLite-backed, checksummed audit log. Both endpoints now query `session.provenance_manager.get_lineage(node_id)` first, and a new `_transform_audit_lineage()` maps the W3C PROV-O entries into the exact `{"nodes": [...], "edges": [...]}` shape `LineageDiagram.tsx` already expects — no frontend changes required
|
||||
- Falls back to the original 2-hop traversal, never a 500: no audit records for a node, a `ProvenanceManager` storage failure (corrupted DB, permissions), or a failed SHA-256 integrity check on any entry in the lineage chain all degrade cleanly to the naive path. A new `source: "audit" | "graph_traversal"` field on the response discloses which path actually served the data
|
||||
- `ProvenanceManager.get_lineage()` now returns `integrity_verified`, computed by re-verifying every entry's checksum before it's trusted; a single tampered or corrupted entry anywhere in the lineage chain now falls the *entire* response back to graph traversal rather than serving partially-verified audit data
|
||||
- Replaced an initial classmethod-based `ProvenanceManager.set_default_storage_path()` approach (caught in review before merge — it would have let any two sessions/apps in the same process silently share and overwrite each other's storage path, including across unrelated test runs) with `provenance_storage_path` threaded through `GraphSession.__init__` and `create_app(...)`, so each session's `ProvenanceManager` is independently scoped
|
||||
- Disclosed limitation: `ProvenanceManager.trace_lineage()`/`get_lineage()` only walk `parent_entity_id`/`used_entities` backward, so the audit path currently surfaces upstream lineage only — the naive fallback remains the only source for downstream/descendant relationships until `ProvenanceManager` gains a reverse lookup
|
||||
- New `tests/explorer/test_provenance_manager_wiring.py` (8 tests): the audit path via a real multi-hop `track_entity()` chain, empty-record fallback, simulated storage-failure degradation (asserts `200`, not `500`), checksum-tamper fallback, evidence-field preservation, `create_app()` storage-path wiring, and cross-session storage isolation, confirmed order-invariant across `tests/explorer/` and `tests/provenance/` in both execution orders
|
||||
|
||||
- **`POST /shacl/validate` and the `/health` SHACL dimension never ran live SHACL validation** (#772, #804) by @Sameer6305 and @KaifAhmad1
|
||||
- `/shacl/validate` had no data graph to validate submitted shapes against — only a Turtle syntax check. Added `_data_graph_turtle_for_uri()`, which serializes the loaded ontology's nodes/edges into an RDF/Turtle instance graph (CURIE resolution across owl/rdfs/skos/dct/dc, arbitrary node-property projection, typed individuals) and wires both `/shacl/validate` and the `/health` SHACL dimension to `OntologyEngine.validate_graph()` via pySHACL, returning real `conforms`/violations instead of a hardcoded `status="unavailable"` stub
|
||||
- Fixed a cross-ontology namespace leak in `_node_belongs_to_ontology`: its prefix fallback (`_extract_namespace()`) split only on the last `/`, so sibling ontologies sharing a domain (e.g. `.../onto-a` and `.../onto-b`) could match entities across ontologies that shouldn't be related; fixed by comparing against the full URI stem via the new `_ontology_namespace()` helper
|
||||
- Added resource guardrails to `/shacl/validate` to close a DoS risk flagged in review: a submitted-Turtle byte cap (`SEMANTICA_MAX_SHACL_TURTLE_BYTES`, default 256 KB), a parsed-triple cap (`SEMANTICA_MAX_SHACL_TRIPLES`, default 1,000), a validation timeout (`SEMANTICA_MAX_SHACL_TIMEOUT`, default 15s), and a global concurrency semaphore (`SEMANTICA_MAX_SHACL_CONCURRENCY`, default 4)
|
||||
- Fixed `HealthDimension.status` being set to `"error"` on a real (non-`ImportError`) validation exception, which isn't a valid value on that model — Pydantic construction raised and turned the whole `/health` endpoint into a 422 on any real bug; now reports `status="critical"` (already a valid value) with a regression test forcing this exact path
|
||||
- Follow-up review fixes: reverted an unrelated regression that had crept into this PR — `POST /api/ontology/create` had gone back to silently swallowing `OntologyEngine.from_data`/`from_text` failures into a near-empty "minimal" ontology instead of raising `HTTPException(500)`, undoing the earlier #770/#787 fix for the same endpoint (and breaking `TestOntologyCreateFailures`, which wasn't run before this PR's initial merge request); `sh:Warning`/`sh:Info`-severity pySHACL results were silently dropped from the `/shacl/validate` response — a shape using non-`Violation` severities could report `conforms=False` with an empty `violations` list and no explanation, so warnings/infos are now folded into the response's `violations` array; and `/health` was independently re-fetching and re-truncation-checking the same ontology's nodes/edges once for the generated SHACL shapes and once for the data graph — both now share a single fetch via `_fetch_analysis_graph()`
|
||||
- New regression tests: `TestOntologyCreateFailures` (pre-existing, now passing again), `test_shacl_validate_surfaces_warning_severity_results`, `test_health_dedupes_node_edge_fetch`, plus the existing 26-test `tests/explorer/test_ontology_subissue3.py` suite (28/28 passing) and the pre-existing `tests/ontology/` suite (83/83 passing)
|
||||
|
||||
- **Neptune cookbook CloudFormation stack exposed the database port to the entire internet and had no network audit trail** ([code scanning alert #28](https://github.com/semantica-agi/semantica/security/code-scanning/28), [#26](https://github.com/semantica-agi/semantica/security/code-scanning/26), [#27](https://github.com/semantica-agi/semantica/security/code-scanning/27), `AC_AWS_0276`/`AC_AWS_0369`/`AC_AWS_0148`) by @KaifAhmad1
|
||||
- `cookbook/introduction/neptune-setup.yaml`'s security group let anyone on `0.0.0.0/0` reach the Neptune Bolt/OpenCypher port (8182); it now requires a `ClientCidr` parameter (CIDR-validated, no default) so the stack can't be created without the deployer explicitly scoping access to their own IP or VPN/office range
|
||||
- Added `AWS::EC2::FlowLog` plus a dedicated CloudWatch Logs group and IAM role so all traffic in the stack's VPC is now logged
|
||||
- Left the account-wide IAM password policy check (`AC_AWS_0148`) unimplemented as a stack resource on purpose: `AWS::IAM::AccountPasswordPolicy` is an account singleton, and wiring it into a disposable per-learner tutorial stack would mean creating or deleting this stack also mutates or removes the account's real password policy — suppressed with a documented `ts:skip=AC_AWS_0148` explaining why, rather than "fixed"
|
||||
- Updated `21_Amazon_Neptune_Store.ipynb`'s `aws cloudformation create-stack` instructions, prerequisites, and cost table to match the new required `ClientCidr` parameter and flow-log line item
|
||||
|
||||
- **Follow-up to the knowledge-explorer Helm chart default-namespace/seccomp scanner findings reopening** ([code scanning alert #846](https://github.com/semantica-agi/semantica/security/code-scanning/846), [#847](https://github.com/semantica-agi/semantica/security/code-scanning/847), [#848](https://github.com/semantica-agi/semantica/security/code-scanning/848), [#68](https://github.com/semantica-agi/semantica/security/code-scanning/68), [#63](https://github.com/semantica-agi/semantica/security/code-scanning/63), `CKV_K8S_21`/`AC_K8S_0086`/`AC_K8S_0080`) by @KaifAhmad1
|
||||
- The `checkov.io/skip1` metadata annotation added previously (see the `CKV_K8S_21` entry below) evidently isn't being honored by the Microsoft Defender for DevOps scan — the same finding reopened under new alert numbers on the current `main`. Added the more standard `# checkov:skip=CKV_K8S_21` and `# ts:skip=AC_K8S_0086` inline comments at the top of `templates/deployment.yaml`, `templates/service.yaml`, and `templates/configmap.yaml` as a second suppression path (matching the convention already used in `deploy/gcp/cloudrun-service.yaml`), plus `# ts:skip=AC_K8S_0080` on `templates/deployment.yaml` for the seccomp finding, which trips for the same root cause: terrascan's static template scan never resolves `{{ toYaml .Values.podSecurityContext }}`, even though `values.yaml` sets `seccompProfile.type: RuntimeDefault` correctly
|
||||
- Confirmed the `deploy/kubernetes/*` (non-Helm) manifests already had TLS and seccomp configured correctly, so no code change was needed there for the corresponding alerts (#61 and the non-Helm seccomp finding) — expected to close on the next scan
|
||||
- Documented both suppression mechanisms and the reasoning in `.checkov.yaml`
|
||||
- Residual risk: this environment could not run checkov/terrascan locally to confirm the inline comments are actually honored during a Helm-rendered scan; if the alerts are still open after the next scan, the reliable fallback is splitting the CI checkov/terrascan invocation so `deploy/helm/` is scanned with these specific checks excluded via `--skip-check` instead of relying on in-file suppression
|
||||
|
||||
- **`react-hooks/set-state-in-effect` cascading renders across 12 Explorer workspace files** (#769, #796) by @Sameer6305 and @KaifAhmad1
|
||||
- Replaced synchronous `setState` calls inside `useEffect` bodies with React's recommended "adjust state during render" pattern (`if (x !== prevX) { setPrevX(x); ...setState... }`) across `OntologyWorkspace`, `ManageWorkspace`, `LineageWorkspace`, and `GraphWorkspace`, and inlined async data-fetching effects with `ignore` flags to prevent race conditions and stale writes after unmount
|
||||
- Fixed a regression the inlining itself introduced: `AlignmentsTab.tsx`, `KGOverviewTab.tsx`, `OntologyManager.tsx`, and `VersionsTab.tsx` each duplicated their existing fetch callback (`reload` / `fetchOverview` / `fetchRegistry` / `loadVersions`+`loadProposals`) into a second, inline copy for the mount effect, and the copy silently dropped the `setError`/`flashMsg` calls the original had — re-introducing, on the very first page load, the exact error-swallowing behavior that #767/#790 had already fixed for these same files. The inline copies now mirror the original's error handling (including `207` partial-success messages) exactly
|
||||
- Fixed `LineageDiagram.tsx` only clearing the previously-rendered nodes/edges when the new `activeId` was falsy instead of on every id change, so switching directly between two lineage views briefly kept showing the *previous* view's stale diagram instead of clearing before the new fetch resolved
|
||||
- `GraphWorkspace.tsx` and `GraphLoadingOverlay.tsx` still have unrelated `react-hooks/set-state-in-effect` violations outside this PR's 12-file scope (confirmed via `npx eslint .`); left as follow-up work rather than expanding this PR further
|
||||
|
||||
- **Checkov flagged the knowledge-explorer Helm chart for using the default Kubernetes namespace** ([code scanning alert #779](https://github.com/semantica-agi/semantica/security/code-scanning/779), [#778](https://github.com/semantica-agi/semantica/security/code-scanning/778), [#777](https://github.com/semantica-agi/semantica/security/code-scanning/777), `CKV_K8S_21`) by @KaifAhmad1
|
||||
- `templates/service.yaml`, `templates/deployment.yaml`, and `templates/configmap.yaml` all already set `metadata.namespace` to `{{ .Release.Namespace }}`, which is only bound at `helm install`/`helm template` time; Checkov's helm framework renders the chart without a namespace override, so it always resolves to `default` and trips `CKV_K8S_21` even though the chart is namespace-agnostic by design
|
||||
- Added a `checkov.io/skip1: CKV_K8S_21` metadata annotation to each of the three files to suppress the scanner artifact false-positive properly in Helm templates, and documented the reasoning in `.checkov.yaml`
|
||||
|
||||
- **No React error boundaries around lazy-loaded Explorer workspaces — a single render error crashed the whole app** (#768, #794) by @Sameer6305
|
||||
- Added an `ErrorBoundary` class component (`explorer/src/ErrorBoundary.tsx`) and wrapped each lazy-loaded workspace's `<Suspense>` block in `App.tsx` with it, keyed on the active sub-view so navigating away from and back to a crashed tab remounts it cleanly
|
||||
- Failed retries are capped at 3 before the fallback UI switches from "Try Again" to a "Reload Application" dead-end, preventing infinite retry loops on deterministic crashes; raw error/stack details are logged via `console.error` only and never rendered into the fallback UI
|
||||
- Fixed the retry counter so it resets after a retry actually succeeds and stays error-free for a few seconds, instead of never resetting (which could permanently exhaust the retry budget on unrelated, individually-recoverable transient errors) or resetting on the very next commit (which could fire prematurely while `Suspense` was still showing its fallback)
|
||||
|
||||
- **Explorer frontend workspaces silently swallowed network/server errors** (#767, #790) by @Sameer6305
|
||||
- `ShaclStudio.tsx`, `VersionsTab.tsx`, `SKOSVocabularyManager.tsx`, `EntityResolutionTab.tsx`, `LineageDiagram.tsx`, `DecisionWorkspace.tsx`, `KGOverviewTab.tsx`, `OntologyManager.tsx`, `OntologySearch.tsx`, `ReasoningWorkspace.tsx`, and `SparqlWorkspace.tsx` now render a visible error banner instead of only `console.error()`-ing failed fetches
|
||||
- Added explicit `response.status === 207` (Multi-Status) handling across these workspaces so partial backend failures surface a warning instead of reading as a full success (`response.ok` is `true` for all 2xx codes, including 207)
|
||||
- Added defensive JSON parsing so an unexpected non-JSON (e.g. HTML 500) response body no longer crashes the app with `SyntaxError: Unexpected token < in JSON`
|
||||
- Fixed `KGOverviewTab.tsx` dropping the `/api/graph/nodes` partial-success warning whenever `/api/graph/stats` also returned 207 — both warnings are now shown (appended) instead of one being silently discarded
|
||||
- Fixed `HealthTab.tsx`'s registry load still using a bare `.catch(() => {})` that swallowed errors identically to the pattern fixed elsewhere in this same folder; failures now populate the existing error banner
|
||||
- Fixed `AlignmentsTab.tsx`'s `reload()` using `Promise.allSettled` but never handling the `"rejected"` branches for the registry/alignments fetches, so both failures previously vanished with no error surfaced and no logging
|
||||
|
||||
- **`tests/explorer/test_explorer_api.py` failed with `TypeError: Client.__init__() got an unexpected keyword argument 'app'` on current httpx** (#788, #789) by @Sameer6305
|
||||
- `httpx>=0.28.0` removed the `app=` kwarg that Starlette's `TestClient` relies on to wrap a FastAPI app for testing; `httpx` wasn't pinned anywhere in `pyproject.toml`, so different environments could independently resolve an incompatible transitive version and hit the same break
|
||||
- Added an explicit `httpx<0.28.0` constraint to the main `[project.dependencies]` array (not just a dev extra), so it applies globally across production, dev, and CI installs
|
||||
- Without the pin, the full test suite fails to even complete collection (fails immediately on `tests/explorer/test_vocabulary.py` with the same `TestClient` error); with it, `tests/explorer/test_explorer_api.py` goes from 7 failed/12 passed/58 errors to 77 passed, 0 errors
|
||||
|
||||
- **Explorer backend routes returned HTTP 200 with error/empty bodies on failure, defeating frontend error handling** (#770, #787) by @Sameer6305 and @KaifAhmad1
|
||||
- `GET /api/temporal/patterns` now raises `HTTPException(500)` on a genuine computation failure instead of silently returning an empty-but-valid `TemporalPatternResponse`; the `ImportError` fallback (optional `kg` extra not installed) is unchanged and still degrades gracefully to an empty list
|
||||
- `POST /api/ontology/create` now raises `HTTPException(500)` when ontology generation fails in either the `sample_data` or `schema_text` mode, instead of silently falling back to a partial/minimal ontology with a misleading `nodes_added` count
|
||||
- `GET /api/analytics` sets `response.status_code = 207` (Multi-Status) when some, but not all, of the requested metrics fail, and raises `HTTPException(500)` when every requested metric fails — a plain 2xx (including 207) reads as success to callers that only check `response.ok`, so an all-failed request now surfaces as a hard error rather than a body full of `{"error": ...}`
|
||||
- Added regression tests covering all three failure paths (`test_patterns_failure_returns_500`, `test_analytics_partial_failure_returns_207`, `test_analytics_total_failure_returns_500`, and two `TestOntologyCreateFailures` cases)
|
||||
|
||||
### Security
|
||||
|
||||
- **CI/CD supply-chain hardening against mutable-tag Action compromise (LiteLLM/Trivy-class attack)** (#824) by @KaifAhmad1
|
||||
- Every third-party GitHub Action across all 8 workflows is now pinned to a full commit SHA instead of a mutable tag (`@v7` → `@3d3c42e... # v7`), closing the exact vector used against LiteLLM in March 2026 (a compromised Trivy Action tag stole a long-lived publishing token)
|
||||
- Added `verify-action-pins.yml` + `.github/scripts/verify-action-pins.sh`: a CI check that fails closed on any `uses:` reference that isn't a full SHA (catching a newly introduced mutable tag, not just auditing existing pins) and re-verifies every pin against the GitHub API on each workflow change, on push to `main`, and weekly; an unresolvable API lookup is treated as a failure rather than a silent skip
|
||||
- `release.yml`: scoped `permissions` to the job level (workflow default is now `contents: read`), added a `concurrency` group so simultaneous tag pushes can't race the publish job, and added SLSA build provenance attestation (`actions/attest-build-provenance`) for every released wheel
|
||||
- Created a protected `pypi` GitHub Environment (required reviewer, restricted to `v*` tag deployments) and enabled branch protection on `main` (required PR review with stale-approval dismissal, required status checks, no force-push/deletion, required conversation resolution) — PyPI publishing already used Trusted Publishing (OIDC) with no long-lived token
|
||||
- Grouped Dependabot's `github-actions` updates into a single PR
|
||||
|
||||
- **`security-scan.yml`'s Safety dependency-vulnerability check was silently non-functional** (#824) by @KaifAhmad1
|
||||
- `safety check --json --output safety-report.json` is invalid in Safety 3.x (`--output` now selects a console format, not a file path); the command errored on every run, swallowed by `|| true`, so no report was ever produced and the job always fell back to a generic "scan completed" message with the vulnerability count hardcoded to 0
|
||||
- Switched to `--save-json`, the correct flag for writing a JSON report to disk; also fixed `vuln.package` → `vuln.package_name` and Semgrep's `issue.rule_id` → `issue.check_id` (both produced `undefined` in the PR comment)
|
||||
- The job never installed Semantica's own dependencies before scanning, so Safety was auditing the scanner tools' own transitive deps, not the project's; added `pip install -e ".[llm-litellm]"` so the actual dependency tree — including the LiteLLM extra — is what gets scanned
|
||||
- Rewrote the PR-comment builder: every line previously used `\\n` inside JS template literals, which renders as the literal text `\n` rather than a newline, producing an unreadable wall of text; now builds real line arrays and collapses long finding lists into a `<details>` block
|
||||
- Added the `pull-requests: write` permission the comment-posting step was missing (silently failing via its own try/catch on every prior run)
|
||||
|
||||
- **`pypdf2==3.0.1` removed (CVE-2023-36464)** (#824) by @KaifAhmad1
|
||||
- Surfaced by the Safety fix above: PyPDF2 is a discontinued project (merged into `pypdf`) permanently frozen at the vulnerable 3.0.1 with no patched release possible. `grep -rn "import PyPDF2"` found zero real usages anywhere in the codebase — it was only referenced in docstrings describing a `PyPDF2.PdfReader()` fallback for PDF parsing that was never actually implemented (`pdfplumber` does the real work). Removed the dependency and corrected the stale docstrings in `parse/__init__.py`, `parse/methods.py`, `parse/pdf_parser.py`, and `ingest/email_ingestor.py`
|
||||
|
||||
- **10 Bandit B324 false positives suppressed (non-cryptographic MD5 use)** (#824) by @KaifAhmad1
|
||||
- Surfaced by the same Safety fix restoring a working CI gate: Bandit's HIGH-severity check was blocking on 10 pre-existing `hashlib.md5()` calls, all generating short deterministic cache keys, entity IDs, or IRI suffixes from non-secret input — none used for passwords, tokens, or verifying untrusted data
|
||||
- Bandit's own message suggests `usedforsecurity=False`, but that keyword argument needs Python 3.9+ and `pyproject.toml` declares `requires-python = ">=3.8"`; used a targeted `# nosec B324` with a one-line justification instead, which suppresses only this check with no runtime behavior change on any supported Python version
|
||||
|
||||
## [0.6.0] - 2026-07-21
|
||||
|
||||
### Added
|
||||
|
||||
- **Named-graph support for `JenaStore` via `Dataset` migration** (#756, #757) by @Sameer6305 and @KaifAhmad1
|
||||
- `JenaStore` now backs onto `rdflib.Dataset(default_union=False)` instead of `rdflib.Graph`, closing #756 and fully closing out the #754/#756 cross-backend named-graph parity effort across Blazegraph, RDF4J, and Jena
|
||||
- `default_union=False` is explicitly set so existing `execute_sparql()`/`get_triplets()` calls that don't pass `graph=` keep seeing only the default graph, not a union across all named graphs
|
||||
- `add_triplets()` accepts a `graph=` option: when supplied, triples are written to that named graph (4-tuple add via `Dataset.graph(uri)`); when omitted, behavior is unchanged (3-tuple add routes to the default graph)
|
||||
- Fixed a pre-existing bug where the remote-endpoint path instantiated the read-only rdflib `SPARQLStore` instead of `SPARQLUpdateStore`, so every `add_triplets()` call against a remote Fuseki endpoint silently failed (`TypeError` swallowed, `success=True`/`added=0` returned); also fixed a constructor bug where `self.endpoint` was always `None` regardless of how `JenaStore` was called, making the remote path unreachable in practice
|
||||
- `serialize()` now logs a warning instead of silently dropping named-graph content when the requested format (`turtle`, `xml`, `n3`, …) can only serialize the default graph; use `format="trig"` or `format="nquads"` to include all graphs
|
||||
- `create_model()`'s `triplet_count` now documented as counting across all graphs (default + named), not just the default graph, matching the `Dataset`-wide semantics
|
||||
- `delete_triplet()` remains scoped to the default graph only (named-graph parity for delete is an explicit follow-up, matching the maintainer's scoping of this migration to `add_triplets`); the removal is passed `self.graph.default_graph` explicitly as its context, since `Dataset.remove()` on a bare 3-tuple resolves to a wildcard context internally and would otherwise delete matching triples out of every named graph too — a follow-up fix to the initial PR #757 for a bug that had no test coverage
|
||||
- 9 new tests covering `Dataset` construction, `default_union=False` confirmation, named-graph write isolation, `serialize()` warning behavior, and `delete_triplet()`'s default-graph scoping
|
||||
|
||||
- **SPARQL CONSTRUCT query templates** (#752, #322, #755, #754) by @Sameer6305
|
||||
- Added parameterized, injection-safe `CONSTRUCT` templates (`ConstructTemplate`, `ParameterDescriptor`, `ConstructTemplateRegistry`)
|
||||
- Extended CONSTRUCT execution support from Blazegraph-only to the RDF4J and Jena backends (#755), closing #754
|
||||
- `RDF4JStore.execute_sparql` gains a CONSTRUCT-aware path (`Accept: text/turtle`, rdflib Turtle parsing, the same `(s, p, o, metadata)` 4-tuple contract) and named-graph writes via RDF4J's REST `context` parameter
|
||||
- `JenaStore.execute_sparql` gains the equivalent CONSTRUCT-aware path over its in-process `rdflib.Graph`
|
||||
- `_CONSTRUCT_QUERY_RE` moved to `sparql_escaping.py` as a shared, backend-agnostic constant used by all three backends
|
||||
- Added pipeline integration via the `construct_template` step type
|
||||
|
||||
- **Databricks Connector (Unity Catalog + Delta Lake ingestion)** (#747) by @KaifAhmad1
|
||||
- Added `DatabricksIngestor` (`semantica/ingest/databricks_ingestor.py`), mirroring `SnowflakeIngestor`'s structure and public API shape: a `DatabricksConnector` connection handler, a `DatabricksData` dataclass, and an optional-import guard for `databricks-sdk`/`databricks-sql-connector`
|
||||
- Supports personal access token and OAuth M2M (service principal `client_id`/`client_secret`) authentication, configurable via constructor args or `DATABRICKS_*` environment variables
|
||||
- `ingest_table()`/`ingest_query()` run against a SQL warehouse or cluster via `databricks-sql-connector`, with `where`/`order_by`/`limit`/`offset` support and the same identifier-escaping and unsafe-`ORDER BY` rejection as `SnowflakeIngestor`; each call closes the SQL connection it opened unless one is already open (e.g. via the `with DatabricksIngestor(...)` context manager), which reuses and closes it exactly once instead of leaking a second connection per call
|
||||
- `get_table_schema()`, `list_catalogs()`, `list_schemas()`, and `list_tables()` introspect Unity Catalog via `databricks-sdk`'s `WorkspaceClient`, validating both catalog and schema are resolved before calling the SDK; `get_table_lineage()` calls Unity Catalog's table-lineage REST API for upstream/downstream `Table --DEPENDS_ON--> Table` dependencies, plus an opt-in `include_column_lineage=True` that resolves per-column lineage via the column-lineage API
|
||||
- `export_as_documents()` converts ingested rows into Semantica document dicts for KG construction, matching `SnowflakeIngestor.export_as_documents()`'s shape
|
||||
- Registered as a lazy export in `semantica.ingest` (`DatabricksIngestor`, `DatabricksData`, `DatabricksConnector`) and as the `db-databricks` optional extra (`pip install "semantica[db-databricks]"`) in `pyproject.toml`, included in `db-all`
|
||||
- New `docs/integrations/databricks.md` page modeled on `docs/integrations/snowflake.md`, plus a `DatabricksIngestor` section and table row in `docs/reference/ingest.md` and cross-links between the two integration pages
|
||||
- 35 unit tests in `tests/test_databricks_ingestor.py` covering both auth methods, table/query ingestion, connection lifecycle (including reuse under the context manager), pagination, unsafe `ORDER BY` rejection, catalog/schema validation, schema/catalog/table listing, table and column lineage, document export, and the missing-dependency error path, closing #747
|
||||
|
||||
- **SQLite Vector Store Backend (`sqlite-vec`)** (#726) by @Luffy2208 and @KaifAhmad1
|
||||
- Added `SQLiteVecStore` (`semantica/vector_store/sqlite_vec_store.py`), a disk-backed local vector store using the `sqlite-vec` extension's `vec0` virtual tables, closing #240
|
||||
- Supports Cosine and L2 distance metrics, dynamic JSON metadata filtering, read-only mode, and an in-memory (`:memory:`) mode
|
||||
- Registered as the `"sqlite"` backend in `VectorStore.SUPPORTED_BACKENDS`, with `db_path`/`sqlite_path` config and a `VECTOR_STORE_SQLITE_PATH` environment variable
|
||||
- Batched `add`/`delete`/`get` and `executemany`-based `update` to avoid per-row round trips; optional `use_wal=True` enables `journal_mode=WAL` + `synchronous=NORMAL` for improved write concurrency
|
||||
- Lazy-imports `sqlite-vec` so the dependency stays fully optional (`pip install semantica[vectorstore-sqlite]`); table names and metadata filter keys are validated against a strict identifier pattern before SQL interpolation
|
||||
- Fixes `VectorStore.update_vectors`/`delete_vectors` to delegate to the active backend store instead of only mutating in-memory state, correcting existing behavior for all non-`inmemory` backends
|
||||
- 25 unit and integration tests in `tests/vector_store/test_sqlite_vec_store.py` covering init, add, search, get, update, delete, read-only mode, and stats
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`kg.ProvenanceTracker` compatibility wrapper out of sync with `ProvenanceManager`, causing 9 pre-existing test failures** (#744, #751) by @Sameer6305 and @KaifAhmad1
|
||||
- `kg.ProvenanceTracker` was a standalone in-memory implementation that never delegated to the unified `ProvenanceManager` backend; its own test suite asserted the existence of `get_lineage`, `track_relationship`, `track_entities_batch`, `get_provenance`, and `_use_unified`, none of which were ever implemented, plus a stale `get_all_sources()` assertion expecting `"timestamp"` instead of the actual `"recorded_at"` key
|
||||
- Rather than completing the abandoned compatibility layer, `kg.ProvenanceTracker` and its remaining supported methods (`track_entity`, `get_all_sources`, `query_recorded_between`, `revision_history`, `export_audit_log`) now emit `DeprecationWarning`s pointing callers to `semantica.provenance.ProvenanceManager`
|
||||
- Removed/rewrote the 9 tests that only exercised the never-implemented compatibility methods to instead verify the observable behavior of the still-supported API, and corrected the stale `get_all_sources()` assertion
|
||||
- Added the previously-missing `docs/migration/kg-provenance-tracker.md` migration guide referenced by every new deprecation warning, with a method-mapping table to `ProvenanceManager` and a before/after example, closing #744
|
||||
|
||||
- **`ProvenanceManager.track_entity` silently overrides an explicit `parent_entity_id`/`derived_from` on re-track** (#742) by @Sameer6305
|
||||
- `track_entity()` resolved `parent_id` via a documented precedence chain (`parent_entity_id` kwarg > `metadata["derived_from"]` > source-as-known-entity-id fallback), but the history-preservation block that runs afterward unconditionally overwrote that resolved value with an auto-generated `f"{entity_id}:v:{existing.last_updated}"` history pointer whenever the entity was being re-tracked, discarding whatever parent the caller had just explicitly supplied with no warning
|
||||
- `track_entity()` now records whether the precedence chain already resolved an explicit parent (`parent_entity_id` kwarg, `metadata["derived_from"]`, or the source-as-known-entity-id fallback) before the history block runs, and only falls back to the auto-generated history pointer when the caller supplied no explicit parent on that call
|
||||
- The archived history entry for the previous version is still kept reachable in `get_lineage()` via `used_entities` (BFS-traversed by `InMemoryStorage.trace_lineage()`) even when an explicit parent is supplied, so re-tracking with a new parent no longer orphans the prior version from the lineage chain; when no explicit parent is supplied, `used_entities` is left alone since `parent_entity_id` already points at the same history id, avoiding a duplicate self-reference
|
||||
- Added `test_retrack_with_explicit_parent_overrides_history_link`, `test_retrack_without_explicit_parent_still_uses_history_link`, `test_retrack_with_derived_from_overrides_history_link`, and `test_retrack_history_reachable_via_used_entities` regression tests, closing #742
|
||||
|
||||
- **`ProvenanceManager.get_lineage` does not link entities that share a source URL** (#735) by @KaifAhmad1
|
||||
- `track_entity()`'s only auto-linking logic looked up `source` as if it were an existing entity's `entity_id`, so passing the same real URL/DOI as `source` for two conceptually linked entities (e.g. a document and a decision derived from it) never produced a parent link, leaving `get_lineage()` returning a chain of length 1
|
||||
- `metadata["derived_from"]` was preserved and echoed back in the output JSON but was never consulted by any linking or traversal code, so the caller's explicit relationship was silently inert
|
||||
- `track_entity()` now treats `metadata["derived_from"]` as an explicit parent link (unless `parent_entity_id` was already passed directly), so `InMemoryStorage.trace_lineage()`'s existing BFS over `parent_entity_id` picks it up for free
|
||||
- `metadata["derived_from"]` is now recognized on any `collections.abc.Mapping`, not just a concrete `dict`, so e.g. `types.MappingProxyType` metadata still creates the parent link
|
||||
- `get_lineage()`'s metadata aggregation now applies the queried entity's own metadata last so it wins over ancestor metadata on conflicting keys, matching the documented "most recent entry's metadata takes precedence" behavior — previously `trace_lineage()`'s BFS order caused ancestor metadata (now reachable via `derived_from` chains) to silently overwrite the queried entity's own values
|
||||
- Added 9 regression/edge-case tests in `tests/provenance/test_manager.py` covering the happy path, explicit `parent_entity_id` precedence over `derived_from`, precedence over the `source`-as-known-entity-id fallback, a `derived_from` pointing at a never-tracked entity, non-string/empty-string `derived_from` values being ignored, a self-referencing `derived_from` not hanging traversal, multi-hop `derived_from` chains, metadata precedence between a queried entity and its ancestors, and non-`dict` `Mapping` metadata, closing #735
|
||||
|
||||
- **`Reasoner.add_rule` had no deduplication, doubling rules and silently emptying `forward_chain()` on rerun** (#732) by @KaifAhmad1
|
||||
- `add_rule()` unconditionally appended to `self.rules`, so re-running the same setup code on an existing `Reasoner` instance (e.g. re-executing a Jupyter cell) duplicated every rule; since `forward_chain()` only records a conclusion if it isn't already in `self.facts`, the second run's duplicated rules matched but produced no new results, with no error or warning
|
||||
- `add_rule()` now compares an incoming rule's `rule_type`, `conditions`, and `conclusion` against existing rules and returns the existing `Rule` instead of appending a duplicate, keeping repeated `add_rule()` calls with the same definition idempotent
|
||||
- Added `test_add_rule_deduplicates_identical_rule`, `test_add_rule_deduplication_is_idempotent_across_forward_chain`, and `test_add_rule_does_not_dedupe_distinct_rules` regression tests
|
||||
|
||||
- **`InferenceResult.premises` always empty from `forward_chain`/`backward_chain`** (#739) by @Sameer6305
|
||||
- `_match_rule()` discarded matched facts and returned only instantiated conclusions, so `ExplanationGenerator` always produced empty premises lists regardless of which facts actually satisfied a rule, closing #733
|
||||
- `_match_rule()` now returns `(conclusion, matched_facts)` tuples; `forward_chain()` threads those facts into `InferenceResult(premises=...)`, merging premises when the same conclusion is derived more than once within a pass
|
||||
- `_prove_goal()`'s base cases (goal already a known fact; goal matched via pattern unification) now return `premises=[goal]`/`premises=[fact]` instead of `[]`
|
||||
- Facts are matched against a `sorted()` snapshot instead of the raw `set` so rule matching and premise selection are deterministic
|
||||
- Added `test_forward_chaining_premises` regression test mirroring the existing backward-chaining premises test
|
||||
|
||||
- **Missing `shacl` optional-dependency extra** (#736) by @Sameer6305
|
||||
- `pip install semantica[shacl]` referenced no matching extra in `pyproject.toml`, so `pyshacl` was never installed despite being documented as the fix in `ontology_validator.py`'s `ImportError` message, the Explorer API, the healthcare cookbook notebook, and the changelog
|
||||
- Added `shacl = ["pyshacl>=0.25.0"]` to `[project.optional-dependencies]` and folded `shacl` into the `all` extra
|
||||
|
||||
- **`NodeEmbedder` `AttributeError` masked in `ContextGraph.analyze_graph_with_kg`** (#734) by @Sameer6305
|
||||
- `analyze_graph_with_kg()` called a non-existent `NodeEmbedder.generate_embeddings()`, and the surrounding broad `except Exception` swallowed the resulting `AttributeError`, silently returning `{"error": "Graph analysis failed due to an internal error"}` from `get_causal_chain()`'s supporting analytics and `get_decision_insights()`
|
||||
- Rewired the call site to the real `NodeEmbedder.compute_embeddings(graph_store, node_labels, relationship_types)` API, deriving `node_labels`/`relationship_types` from `self.node_type_index`/`self.edge_type_index`
|
||||
- Added a dedicated `except AttributeError` branch that logs distinctly and re-raises, so a broken internal method call surfaces as a diagnosable error instead of being indistinguishable from a legitimately empty analysis result
|
||||
|
||||
---
|
||||
|
||||
## [0.5.1] - 2026-06-29
|
||||
|
||||
@@ -19,6 +19,30 @@ Thank you for your interest in contributing! Every contribution, no matter how s
|
||||
|
||||
---
|
||||
|
||||
## 🗂️ Working on an Existing Issue
|
||||
|
||||
If you want to work on an open GitHub issue, please follow these steps to keep things coordinated and avoid duplicate effort:
|
||||
|
||||
1. **Check the issue.** Look at the issue's assignees and recent comments. If someone is already actively working on it, consider a different issue or ask in the comments whether help is welcome.
|
||||
|
||||
2. **Comment before you start.** Leave a comment on the issue saying you'd like to work on it — something like *"I'd like to take this on"* is enough. This gives maintainers the context they need to assign the issue appropriately.
|
||||
|
||||
3. **Wait for assignment.** A maintainer will review the request and assign the issue when appropriate. Please wait for this before investing significant time in implementation, as priorities and approaches can shift.
|
||||
|
||||
4. **Create a branch and implement.** Once assigned, fork the repository (if you haven't already), create a dedicated branch, and begin your work.
|
||||
|
||||
```bash
|
||||
git checkout -b fix/short-description # or feature/short-description
|
||||
```
|
||||
|
||||
5. **Open a focused PR and link the issue.** When you're ready, open a pull request and reference the issue in the description (e.g., `Closes #123`). Keep the PR scoped to the work described in the issue.
|
||||
|
||||
> **Why this matters:** Commenting before opening a PR helps maintainers track who is working on what, assign issues correctly, and prevent two contributors from solving the same problem independently. It also gives you a chance to align on the expected approach before writing code.
|
||||
|
||||
Not sure where to start? Try a [`good first issue`](https://github.com/Hawksight-AI/semantica/labels/good%20first%20issue) or ask in [Discord](https://discord.gg/sV34vps5hH).
|
||||
|
||||
---
|
||||
|
||||
## 🎯 Ways to Contribute
|
||||
|
||||
### 💻 Code
|
||||
|
||||
+2
-2
@@ -1,5 +1,5 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
FROM node:22-alpine AS frontend-builder
|
||||
FROM node:26-alpine AS frontend-builder
|
||||
|
||||
WORKDIR /app
|
||||
COPY explorer/package*.json ./explorer/
|
||||
@@ -9,7 +9,7 @@ RUN npm ci
|
||||
COPY explorer/ ./
|
||||
RUN mkdir -p /app/semantica && npm run build
|
||||
|
||||
FROM python:3.12-slim AS runtime
|
||||
FROM python:3.14-slim AS runtime
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
|
||||
+99
-4
@@ -24,7 +24,7 @@ Security vulnerabilities should be reported privately to prevent potential explo
|
||||
|
||||
### 2. Report Security Issue
|
||||
|
||||
Create a [GitHub Security Advisory](https://github.com/Hawksight-AI/semantica/security/advisories/new) or contact us through [GitHub Issues](https://github.com/Hawksight-AI/semantica/issues) with "[SECURITY]" prefix.
|
||||
Create a [GitHub Security Advisory](https://github.com/semantica-agi/semantica/security/advisories/new) or contact us via the security email listed in `SUPPORT.md`.
|
||||
|
||||
Include the following information:
|
||||
|
||||
@@ -37,7 +37,7 @@ Include the following information:
|
||||
|
||||
### 3. Response Timeline
|
||||
|
||||
- **Initial Response**: Within 48 hours
|
||||
- **Initial Response**: Within 24 hours for critical issues; within 48 hours for non-critical issues
|
||||
- **Status Update**: Within 7 days
|
||||
- **Resolution**: Depends on severity and complexity
|
||||
|
||||
@@ -112,6 +112,101 @@ We regularly update dependencies to address security vulnerabilities. However, y
|
||||
- Be cautious with external API calls
|
||||
- Implement proper authentication and authorization
|
||||
|
||||
## CI/CD Supply-Chain Security
|
||||
|
||||
Semantica's build and release pipeline is explicitly hardened against
|
||||
CI/CD supply-chain attacks — the class of attack behind the March 2026
|
||||
LiteLLM/Trivy incident, where a compromised third-party Action with a
|
||||
**mutable tag** was used to steal a long-lived publishing token, after which
|
||||
malicious packages were pushed straight to PyPI without ever touching the
|
||||
source repository. Every control below maps directly to closing one step of
|
||||
that attack chain.
|
||||
|
||||
### Immutable build inputs
|
||||
|
||||
- **Risk**: a tag (`@v4`, `@release/v1`) is re-pointed by a compromised upstream maintainer or account, silently changing what every consumer's CI runs.
|
||||
**Control**: every third-party GitHub Action in every workflow is pinned to a full 40-character commit SHA, with the human-readable tag kept only as a trailing comment (e.g. `actions/checkout@3d3c42e... # v7`).
|
||||
- **Risk**: a SHA pin drifts out of sync with its own comment over time, or is mistyped.
|
||||
**Control**: `verify-action-pins.yml` fails closed on any `uses:` reference that isn't a full commit SHA (catching a newly added mutable tag, not just auditing existing pins), resolves every pinned tag via the GitHub API on each workflow change, on every push to `main`, and weekly, and fails if the SHA no longer matches the tag it claims to be — an API lookup that can't be resolved is treated as a failure, not a silent skip.
|
||||
- **Risk**: manually re-pinning ~15 actions across 8 workflow files on every upstream release is error-prone.
|
||||
**Control**: Dependabot (`github-actions` ecosystem) opens a grouped PR that bumps the SHA *and* the tag comment together whenever an action releases — pins never require hand-editing.
|
||||
|
||||
### Publishing pipeline (highest-privilege path)
|
||||
|
||||
- **Risk**: a long-lived `PYPI_TOKEN` sitting in repo/org secrets is exfiltrated by any compromised step.
|
||||
**Control**: PyPI publishing uses Trusted Publishing (OIDC) (`id-token: write`) — there is no long-lived PyPI credential anywhere in this repository to steal.
|
||||
- **Risk**: a compromised CI run publishes to PyPI with no human in the loop.
|
||||
**Control**: the publish job runs only inside a protected `pypi` GitHub Environment with a required human reviewer — every release needs manual approval in the Actions UI before it runs.
|
||||
- **Risk**: the release job could be triggered from an arbitrary branch/ref.
|
||||
**Control**: the `pypi` environment's deployment-branch policy is restricted to `v*` tags only.
|
||||
- **Risk**: a scanner or unrelated job inherits publish-level credentials.
|
||||
**Control**: `release.yml` sets `permissions: contents: read` at the workflow level; `contents: write` / `id-token: write` / `attestations: write` are granted only to the release job, never workflow-wide.
|
||||
- **Risk**: two tag pushes race through the publish pipeline simultaneously.
|
||||
**Control**: `concurrency: group: release-${{ github.ref }}` serializes releases per tag.
|
||||
- **Risk**: a consumer can't verify a wheel on PyPI actually came from this repo's CI.
|
||||
**Control**: SLSA build provenance is attested for every release via `actions/attest-build-provenance`, producing a signed, verifiable record of the exact commit and workflow run that produced the artifact (checkable with `gh attestation verify`).
|
||||
|
||||
### Repository controls
|
||||
|
||||
- **Risk**: unreviewed or force-pushed changes land on `main`.
|
||||
**Control**: `main` requires 1 approving PR review (stale approvals dismissed on new pushes), resolved conversations, and blocks force-pushes and branch deletion.
|
||||
- **Risk**: a PR merges without its security/CI checks passing.
|
||||
**Control**: merges require the `build`, `Analyze Python` (CodeQL), and `security-scan` checks to pass, in strict mode (checks must be re-run against the latest `main`).
|
||||
- **Risk**: a compromised scanner job reaches secrets or write access.
|
||||
**Control**: scanning jobs (`CodeQL`, `security-scan.yml`, `security.yml`, `defender-for-devops.yml`) run with read-only, least-privilege permissions (typically `contents: read` + `security-events: write` only) and never share a job, environment, or secret scope with the publish job.
|
||||
- **Risk**: secrets are committed accidentally.
|
||||
**Control**: GitHub secret scanning and push protection are both enabled at the repository level, rejecting pushes that contain recognizable credential patterns before they land in history.
|
||||
|
||||
## Automated Security Scanning
|
||||
|
||||
Every scan below runs continuously in CI, not just at release time:
|
||||
|
||||
- **CodeQL** (`security-and-quality` query pack) — Python source: injection, unsafe deserialization, and other code-level vulnerability classes. Runs in `codeql.yml` on every push/PR to `main` and weekly.
|
||||
- **Bandit** — Python-specific security anti-patterns (hardcoded secrets, unsafe `eval`/`pickle`, weak crypto, etc.); CI fails on any HIGH-severity finding. Runs in `security-scan.yml` on every push/PR to `main` and twice weekly.
|
||||
- **Semgrep** (`p/security` ruleset) — cross-language static-analysis security patterns. Runs in `security-scan.yml` on every push/PR to `main` and twice weekly.
|
||||
- **Safety** — known CVEs in Semantica's own installed dependencies, including optional LLM-provider extras such as LiteLLM; CI fails on any match. Runs in `security-scan.yml` on every push/PR to `main` and twice weekly.
|
||||
- **pip-audit** — independent, PyPA-maintained vulnerability database cross-check against installed dependencies (Safety and pip-audit use different advisory sources, so both run). Runs in `security.yml` weekly.
|
||||
- **Microsoft Defender for DevOps** (`eslint`, `templateanalyzer`, `terrascan`) — JavaScript/TypeScript lint-security rules and infrastructure-as-code misconfigurations. Runs in `defender-for-devops.yml` on every push/PR to `main` and weekly.
|
||||
- **Checkov** — Kubernetes, Helm, Dockerfile, GitHub Actions, and secrets-pattern IaC scanning; results upload to the same Security tab as CodeQL. Runs in `defender-for-devops.yml` on every push/PR to `main` and weekly.
|
||||
- **GitGuardian** — secret-detection check on every pull request, installed as a GitHub App integration (not a repo-local workflow). Runs on every PR.
|
||||
- **GitHub secret scanning + push protection** — blocks known credential patterns before they're pushed, and continuously scans existing history. Platform-level, continuous.
|
||||
- **Dependabot** — version/security PRs for Python, Docker, and GitHub Actions dependencies, grouped where relevant to reduce review noise. Configured in `.github/dependabot.yml`, runs weekly for security-relevant packages and monthly for docs dependencies.
|
||||
- **`verify-action-pins.yml`** — enforces that every Action reference is a full commit SHA (failing on a newly introduced mutable tag) and confirms each SHA still matches the tag it claims to be. Runs on every workflow change, every push to `main`, and weekly.
|
||||
|
||||
All SARIF-producing scanners (CodeQL, Checkov, Microsoft Defender) publish
|
||||
findings to the repository's **Security → Code scanning alerts** tab, giving
|
||||
a single audit trail across tools rather than scattered per-tool reports.
|
||||
|
||||
### Adopting this posture in a fork or downstream deployment
|
||||
|
||||
Teams standing up their own instance of Semantica, or forking it for an
|
||||
internal/regulated deployment, can reuse this posture directly:
|
||||
|
||||
1. Keep Dependabot's `github-actions` ecosystem entry — it is what keeps
|
||||
SHA pins current without manual maintenance.
|
||||
2. Re-run `verify-action-pins.yml` after re-pointing the repository's Actions
|
||||
at your own mirrors, if you do so.
|
||||
3. If you publish your own PyPI package from a fork, configure your own
|
||||
Trusted Publishing trust relationship on PyPI (Trusted Publishing is
|
||||
scoped to a specific `owner/repo` + workflow filename) and your own
|
||||
protected environment with your own required reviewers — these are not
|
||||
transferable from this repository.
|
||||
4. Branch protection, environment protection, and repository secret
|
||||
scanning are repository *settings*, not workflow files — cloning or
|
||||
forking the repo does **not** copy them. They must be re-applied via
|
||||
the GitHub UI or API on the new repository.
|
||||
5. GitHub secret scanning and push protection are repository settings that
|
||||
don't carry over to a fork either — re-enable both under the new
|
||||
repository's Security settings, not just Dependabot.
|
||||
6. GitGuardian runs as a GitHub App installation scoped to this specific
|
||||
repository, not a workflow file — a fork gets no secret-detection
|
||||
coverage from it until the app is installed separately on the new repo.
|
||||
7. CodeQL's `upload-sarif` step in `codeql.yml` only runs meaningfully if
|
||||
Default Setup is *not* already enabled for the repository (it's designed
|
||||
to skip gracefully otherwise) — check whether Default Setup or Advanced
|
||||
Setup is active on the new repository and adjust expectations for where
|
||||
CodeQL findings show up accordingly.
|
||||
|
||||
## Dependency Security Policy
|
||||
|
||||
### Regular Updates
|
||||
@@ -156,8 +251,8 @@ We appreciate responsible disclosure. Security researchers who help us improve t
|
||||
|
||||
For security-related questions or concerns:
|
||||
|
||||
- **GitHub Issues**: [Create an issue](https://github.com/Hawksight-AI/semantica/issues) with "[SECURITY]" prefix
|
||||
- **GitHub Security Advisories**: [Report vulnerability](https://github.com/Hawksight-AI/semantica/security/advisories/new)
|
||||
- **Private Reporting**: Please do not report vulnerabilities in public issues.
|
||||
- **GitHub Security Advisories**: [Report vulnerability](https://github.com/semantica-agi/semantica/security/advisories/new)
|
||||
|
||||
## Additional Resources
|
||||
|
||||
|
||||
+8
-8
@@ -20,8 +20,8 @@ Start with our comprehensive documentation:
|
||||
|
||||
**Best for**: General questions, feature discussions, and getting help
|
||||
|
||||
- [Ask a question](https://github.com/Hawksight-AI/semantica/discussions/new?category=q-a)
|
||||
- [Browse discussions](https://github.com/Hawksight-AI/semantica/discussions)
|
||||
- [Ask a question](https://github.com/semantica-agi/semantica/discussions/new?category=q-a)
|
||||
- [Browse discussions](https://github.com/semantica-agi/semantica/discussions)
|
||||
|
||||
#### Discord
|
||||
|
||||
@@ -33,8 +33,8 @@ Start with our comprehensive documentation:
|
||||
|
||||
**Best for**: Bug reports and feature requests
|
||||
|
||||
- [Report a bug](https://github.com/Hawksight-AI/semantica/issues/new?template=bug_report.md)
|
||||
- [Request a feature](https://github.com/Hawksight-AI/semantica/issues/new?template=feature_request.md)
|
||||
- [Report a bug](https://github.com/semantica-agi/semantica/issues/new?template=bug_report.md)
|
||||
- [Request a feature](https://github.com/semantica-agi/semantica/issues/new?template=feature_request.md)
|
||||
|
||||
### Before Asking
|
||||
|
||||
@@ -47,7 +47,7 @@ Start with our comprehensive documentation:
|
||||
|
||||
### Bug Reports
|
||||
|
||||
Use our [bug report template](https://github.com/Hawksight-AI/semantica/issues/new?template=bug_report.md) to report bugs.
|
||||
Use our [bug report template](https://github.com/semantica-agi/semantica/issues/new?template=bug_report.md) to report bugs.
|
||||
|
||||
Include:
|
||||
- Clear description of the bug
|
||||
@@ -58,7 +58,7 @@ Include:
|
||||
|
||||
### Feature Requests
|
||||
|
||||
Use our [feature request template](https://github.com/Hawksight-AI/semantica/issues/new?template=feature_request.md) to suggest features.
|
||||
Use our [feature request template](https://github.com/semantica-agi/semantica/issues/new?template=feature_request.md) to suggest features.
|
||||
|
||||
Include:
|
||||
- Problem statement
|
||||
@@ -71,7 +71,7 @@ Include:
|
||||
**Do NOT** create a public issue for security vulnerabilities.
|
||||
|
||||
Instead:
|
||||
- Email: semantica-dev@users.noreply.github.com
|
||||
- Email: kaif@getsemantica.ai
|
||||
- Subject: [SECURITY] Brief description
|
||||
- See [Security Policy](SECURITY.md) for details
|
||||
|
||||
@@ -79,7 +79,7 @@ Instead:
|
||||
|
||||
For enterprise support, custom development, or consulting:
|
||||
|
||||
- **Email**: semantica-dev@users.noreply.github.com
|
||||
- **Email**: kaif@getsemantica.ai
|
||||
- **Subject**: [ENTERPRISE] Your request
|
||||
|
||||
## Response Times
|
||||
|
||||
@@ -3,81 +3,7 @@
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"# Amazon Neptune Graph Store\n",
|
||||
"\n",
|
||||
"## Overview\n",
|
||||
"\n",
|
||||
"This notebook covers the Amazon Neptune Database integration in Semantica. Amazon Neptune is a fully managed graph database service that supports both property graphs (via OpenCypher/Gremlin) and RDF graphs (via SPARQL).\n",
|
||||
"\n",
|
||||
"### Key Features\n",
|
||||
"\n",
|
||||
"- **IAM Authentication**: Secure access using AWS SigV4 signatures via AuthManager\n",
|
||||
"- **OpenCypher Support**: Query using standard OpenCypher syntax\n",
|
||||
"- **Bolt Protocol**: Uses Neo4j Bolt driver for efficient binary communication\n",
|
||||
"- **Native ~id Support**: Leverages Neptune's native element ID handling\n",
|
||||
"- **Full CRUD Operations**: Create, read, update, delete nodes and relationships\n",
|
||||
"- **Automatic Retry**: Built-in retry logic with exponential backoff for transient errors\n",
|
||||
"\n",
|
||||
"### Prerequisites\n",
|
||||
"\n",
|
||||
"- An Amazon Neptune Database cluster\n",
|
||||
"- AWS credentials configured (boto3, environment variables, or IAM role)\n",
|
||||
"- Network access to your Neptune cluster (VPC, security groups)\n",
|
||||
"\n",
|
||||
"#### Quick Setup with CloudFormation\n",
|
||||
"\n",
|
||||
"If you don't have a Neptune cluster, use the provided CloudFormation template to create one with a public endpoint and IAM authentication:\n",
|
||||
"\n",
|
||||
"```bash\n",
|
||||
"# Deploy the Neptune stack (takes ~15-20 minutes)\n",
|
||||
"aws cloudformation create-stack \\\n",
|
||||
" --stack-name semantica-neptune \\\n",
|
||||
" --template-body file://neptune-setup.yaml \\\n",
|
||||
" --capabilities CAPABILITY_NAMED_IAM\n",
|
||||
"\n",
|
||||
"# Wait for stack creation to complete\n",
|
||||
"aws cloudformation wait stack-create-complete --stack-name semantica-neptune\n",
|
||||
"\n",
|
||||
"# Get the outputs (endpoint, port, credentials)\n",
|
||||
"aws cloudformation describe-stacks --stack-name semantica-neptune \\\n",
|
||||
" --query 'Stacks[0].Outputs' --output table\n",
|
||||
"```\n",
|
||||
"\n",
|
||||
"The template creates:\n",
|
||||
"- VPC with public subnets and Internet Gateway\n",
|
||||
"- Neptune cluster (`db.t3.medium`) with IAM authentication enabled\n",
|
||||
"- IAM user with least-privilege access for OpenCypher queries\n",
|
||||
"- Security group allowing Bolt protocol (port 8182) access\n",
|
||||
"\n",
|
||||
"> ⚠️ **Security Note**: This template creates an IAM User with static access keys for simplicity in demo/test environments. For production use, we recommend IAM Roles (EC2 instance roles, ECS task roles, Lambda execution roles) which provide temporary credentials that are automatically rotated. The secret access key in the Cloudformation outputs is provided in plaintext to simplify initial setup - in production, use AWS Secrets Manager.\n",
|
||||
"\n",
|
||||
"**Outputs:**\n",
|
||||
"- `NeptuneEndpoint` - Cluster hostname (use as `NEPTUNE_ENDPOINT`)\n",
|
||||
"- `NeptunePort` - 8182 (use as `NEPTUNE_PORT`)\n",
|
||||
"- `AwsAccessKeyId` - IAM user access key (use as `AWS_ACCESS_KEY_ID`)\n",
|
||||
"- `AwsSecretAccessKey` - IAM user secret key in **plaintext** (use as `AWS_SECRET_ACCESS_KEY`)\n",
|
||||
"- `AwsRegion` - Deployment region (use as `AWS_REGION`)\n",
|
||||
"\n",
|
||||
"**Cleanup:**\n",
|
||||
"```bash\n",
|
||||
"aws cloudformation delete-stack --stack-name semantica-neptune\n",
|
||||
"```\n",
|
||||
"\n",
|
||||
"**Estimated Monthly Cost (approximately 100-105 USD/month at 100% utilization):**\n",
|
||||
"\n",
|
||||
"| Resource | Cost (USD) |\n",
|
||||
"| --- | --- |\n",
|
||||
"| Neptune db.t3.medium instance | ~96/month (0.132/hr) |\n",
|
||||
"| Storage (10 GB) | ~1/month |\n",
|
||||
"| I/O requests | ~1-5/month |\n",
|
||||
"| Public IPv4 address | ~3.60/month (0.005/hr) |\n",
|
||||
"| VPC, subnets, route tables, Internet Gateway, IAM | No Additional Charge |\n",
|
||||
"\n",
|
||||
"> **Free Tier**: New Neptune users get 30 days free (750 hours of db.t3.medium, 10M I/Os, 1 GB storage). Delete the stack when not in use to avoid charges.\n",
|
||||
"\n",
|
||||
"---"
|
||||
]
|
||||
"source": "# Amazon Neptune Graph Store\n\n## Overview\n\nThis notebook covers the Amazon Neptune Database integration in Semantica. Amazon Neptune is a fully managed graph database service that supports both property graphs (via OpenCypher/Gremlin) and RDF graphs (via SPARQL).\n\n### Key Features\n\n- **IAM Authentication**: Secure access using AWS SigV4 signatures via AuthManager\n- **OpenCypher Support**: Query using standard OpenCypher syntax\n- **Bolt Protocol**: Uses Neo4j Bolt driver for efficient binary communication\n- **Native ~id Support**: Leverages Neptune's native element ID handling\n- **Full CRUD Operations**: Create, read, update, delete nodes and relationships\n- **Automatic Retry**: Built-in retry logic with exponential backoff for transient errors\n\n### Prerequisites\n\n- An Amazon Neptune Database cluster\n- AWS credentials configured (boto3, environment variables, or IAM role)\n- Network access to your Neptune cluster (VPC, security groups)\n- Your public IP address or VPN/office CIDR (run `curl ifconfig.me` to find your public IP), used below to restrict database access\n\n#### Quick Setup with CloudFormation\n\nIf you don't have a Neptune cluster, use the provided CloudFormation template to create one with a public endpoint and IAM authentication:\n\n```bash\n# Deploy the Neptune stack (takes ~15-20 minutes)\n# Replace 203.0.113.25/32 with your own public IP (run `curl ifconfig.me` to find it)\n# or your office/VPN CIDR. This restricts who can reach the database on the\n# network level - never widen it to 0.0.0.0/0 outside of a short-lived local experiment.\naws cloudformation create-stack \\\n --stack-name semantica-neptune \\\n --template-body file://neptune-setup.yaml \\\n --parameters ParameterKey=ClientCidr,ParameterValue=203.0.113.25/32 \\\n --capabilities CAPABILITY_NAMED_IAM\n\n# Wait for stack creation to complete\naws cloudformation wait stack-create-complete --stack-name semantica-neptune\n\n# Get the outputs (endpoint, port, credentials)\naws cloudformation describe-stacks --stack-name semantica-neptune \\\n --query 'Stacks[0].Outputs' --output table\n```\n\nThe template creates:\n- VPC with public subnets, Internet Gateway, and VPC Flow Logs (to CloudWatch Logs)\n- Neptune cluster (`db.t3.medium`) with IAM authentication enabled\n- IAM user with least-privilege access for OpenCypher queries\n- Security group allowing Bolt protocol (port 8182) access only from the `ClientCidr` you specify\n\n> ⚠️ **Security Note**: This template creates an IAM User with static access keys for simplicity in demo/test environments. For production use, we recommend IAM Roles (EC2 instance roles, ECS task roles, Lambda execution roles) which provide temporary credentials that are automatically rotated. The secret access key in the Cloudformation outputs is provided in plaintext to simplify initial setup - in production, use AWS Secrets Manager. The `ClientCidr` parameter is required (no default) precisely so the database is never silently exposed to the whole internet.\n\n**Outputs:**\n- `NeptuneEndpoint` - Cluster hostname (use as `NEPTUNE_ENDPOINT`)\n- `NeptunePort` - 8182 (use as `NEPTUNE_PORT`)\n- `AwsAccessKeyId` - IAM user access key (use as `AWS_ACCESS_KEY_ID`)\n- `AwsSecretAccessKey` - IAM user secret key in **plaintext** (use as `AWS_SECRET_ACCESS_KEY`)\n- `AwsRegion` - Deployment region (use as `AWS_REGION`)\n\n**Cleanup:**\n```bash\naws cloudformation delete-stack --stack-name semantica-neptune\n```\n\n**Estimated Monthly Cost (approximately 100-105 USD/month at 100% utilization):**\n\n| Resource | Cost (USD) |\n| --- | --- |\n| Neptune db.t3.medium instance | ~96/month (0.132/hr) |\n| Storage (10 GB) | ~1/month |\n| I/O requests | ~1-5/month |\n| Public IPv4 address | ~3.60/month (0.005/hr) |\n| VPC Flow Logs (CloudWatch Logs) | ~1-2/month depending on traffic |\n| VPC, subnets, route tables, Internet Gateway, IAM | No Additional Charge |\n\n> **Free Tier**: New Neptune users get 30 days free (750 hours of db.t3.medium, 10M I/Os, 1 GB storage). Delete the stack when not in use to avoid charges.\n\n---"
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
@@ -722,4 +648,4 @@
|
||||
},
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 4
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,14 @@
|
||||
# ts:skip=AC_AWS_0148 IAM password policy is an AWS-account-wide singleton, not a
|
||||
# per-stack resource. Managing it here would mean every learner who deploys or
|
||||
# deletes this cookbook stack also mutates (or removes) their account's password
|
||||
# policy as a side effect. Account password policy should be set once, out of
|
||||
# band, by the account owner - not by a disposable tutorial stack.
|
||||
AWSTemplateFormatVersion: '2010-09-09'
|
||||
Description: >
|
||||
Amazon Neptune cluster with public endpoint, IAM authentication, and least-privilege
|
||||
IAM user for Semantica cookbook. Uses db.t3.medium (most cost-effective Neptune instance type).
|
||||
Network access to the Bolt/OpenCypher port is restricted to an operator-supplied CIDR
|
||||
(see ClientCidr) - do not widen this to 0.0.0.0/0 outside of a short-lived local experiment.
|
||||
|
||||
Parameters:
|
||||
EnvironmentName:
|
||||
@@ -9,6 +16,16 @@ Parameters:
|
||||
Default: semantica-neptune
|
||||
Description: Environment name prefix for resource naming
|
||||
|
||||
ClientCidr:
|
||||
Type: String
|
||||
Description: >-
|
||||
CIDR block allowed to reach the Neptune Bolt/OpenCypher endpoint (port 8182) - e.g. your
|
||||
workstation's public IP as "x.x.x.x/32", or your office/VPN CIDR. Required: there is no
|
||||
default, so you must explicitly choose a range. Passing 0.0.0.0/0 is possible but exposes
|
||||
the database to the entire internet and is strongly discouraged beyond a brief local test.
|
||||
AllowedPattern: '^((25[0-5]|2[0-4][0-9]|1[0-9]{2}|[1-9]?[0-9])\.){3}(25[0-5]|2[0-4][0-9]|1[0-9]{2}|[1-9]?[0-9])/(3[0-2]|[12]?[0-9])$'
|
||||
ConstraintDescription: Must be a valid IPv4 CIDR block with octets 0-255 and prefix 0-32, e.g. 203.0.113.25/32
|
||||
|
||||
Resources:
|
||||
# =============================================================================
|
||||
# VPC & NETWORKING
|
||||
@@ -87,6 +104,57 @@ Resources:
|
||||
RouteTableId: !Ref PublicRouteTable
|
||||
SubnetId: !Ref PublicSubnet2
|
||||
|
||||
# =============================================================================
|
||||
# VPC FLOW LOGS
|
||||
# =============================================================================
|
||||
|
||||
FlowLogGroup:
|
||||
Type: AWS::Logs::LogGroup
|
||||
Properties:
|
||||
LogGroupName: !Sub /aws/vpc/${EnvironmentName}-flow-logs
|
||||
RetentionInDays: 30
|
||||
|
||||
FlowLogRole:
|
||||
Type: AWS::IAM::Role
|
||||
Properties:
|
||||
RoleName: !Sub ${EnvironmentName}-flow-log-role
|
||||
AssumeRolePolicyDocument:
|
||||
Version: '2012-10-17'
|
||||
Statement:
|
||||
- Effect: Allow
|
||||
Principal:
|
||||
Service: vpc-flow-logs.amazonaws.com
|
||||
Action: sts:AssumeRole
|
||||
Policies:
|
||||
- PolicyName: flow-log-publish
|
||||
PolicyDocument:
|
||||
Version: '2012-10-17'
|
||||
Statement:
|
||||
- Effect: Allow
|
||||
Action:
|
||||
- logs:CreateLogGroup
|
||||
- logs:DescribeLogGroups
|
||||
- logs:DescribeLogStreams
|
||||
Resource: "*"
|
||||
- Effect: Allow
|
||||
Action:
|
||||
- logs:CreateLogStream
|
||||
- logs:PutLogEvents
|
||||
Resource: !GetAtt FlowLogGroup.Arn
|
||||
|
||||
VPCFlowLog:
|
||||
Type: AWS::EC2::FlowLog
|
||||
Properties:
|
||||
ResourceType: VPC
|
||||
ResourceId: !Ref VPC
|
||||
TrafficType: ALL
|
||||
LogDestinationType: cloud-watch-logs
|
||||
LogGroupName: !Ref FlowLogGroup
|
||||
DeliverLogsPermissionArn: !GetAtt FlowLogRole.Arn
|
||||
Tags:
|
||||
- Key: Name
|
||||
Value: !Sub ${EnvironmentName}-vpc-flow-log
|
||||
|
||||
# =============================================================================
|
||||
# SECURITY GROUP
|
||||
# =============================================================================
|
||||
@@ -95,14 +163,14 @@ Resources:
|
||||
Type: AWS::EC2::SecurityGroup
|
||||
Properties:
|
||||
GroupName: !Sub ${EnvironmentName}-neptune-sg
|
||||
GroupDescription: Security group for Neptune cluster - allows Bolt protocol access
|
||||
GroupDescription: Security group for Neptune cluster - allows Bolt protocol access from ClientCidr only
|
||||
VpcId: !Ref VPC
|
||||
SecurityGroupIngress:
|
||||
- IpProtocol: tcp
|
||||
FromPort: 8182
|
||||
ToPort: 8182
|
||||
CidrIp: 0.0.0.0/0
|
||||
Description: Allow Bolt protocol access from anywhere
|
||||
CidrIp: !Ref ClientCidr
|
||||
Description: Allow Bolt/OpenCypher protocol access from the operator-specified CIDR
|
||||
SecurityGroupEgress:
|
||||
- IpProtocol: -1
|
||||
CidrIp: 0.0.0.0/0
|
||||
|
||||
@@ -9,7 +9,10 @@ flyctl launch --copy-config --config deploy/fly/fly.toml --no-deploy
|
||||
# Fly.io private networking uses .internal hostnames — do not use localhost
|
||||
# unless FalkorDB is a co-located process inside the same Machine.
|
||||
flyctl secrets set FALKORDB_HOST=<falkordb-app-name>.internal FALKORDB_PORT=6379
|
||||
flyctl secrets set SEMANTICA_API_KEY=$(openssl rand -hex 32)
|
||||
flyctl deploy --config deploy/fly/fly.toml
|
||||
```
|
||||
|
||||
Change `app` in `fly.toml` before launch if the default app name is already taken.
|
||||
|
||||
Fly apps get a public `*.fly.dev` URL by default, so `SEMANTICA_API_KEY` is required — without it the Explorer refuses every protected route (503) rather than serving anonymously. Pass the same value as the `X-API-Key` header from any client that talks to the deployed API.
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# checkov:skip=CKV_K8S_21:Namespace is bound via .Release.Namespace at helm install/template time; this chart is namespace-portable by design.
|
||||
apiVersion: v1
|
||||
kind: ConfigMap
|
||||
metadata:
|
||||
@@ -5,6 +6,9 @@ metadata:
|
||||
namespace: {{ .Release.Namespace }}
|
||||
labels:
|
||||
{{- include "knowledge-explorer.labels" . | nindent 4 }}
|
||||
annotations:
|
||||
runterrascan.io/skip: '[{"rule": "AC_K8S_0086", "comment": "Namespace is bound via .Release.Namespace at helm install time"}]'
|
||||
checkov.io/skip1: CKV_K8S_21=Namespace bound via .Release.Namespace at helm install/template time
|
||||
data:
|
||||
{{- range $key, $value := .Values.env }}
|
||||
{{ $key }}: {{ $value | quote }}
|
||||
|
||||
@@ -5,6 +5,10 @@ metadata:
|
||||
namespace: {{ .Release.Namespace }}
|
||||
labels:
|
||||
{{- include "knowledge-explorer.labels" . | nindent 4 }}
|
||||
annotations:
|
||||
runterrascan.io/skip: '[{"rule": "AC_K8S_0086", "comment": "Namespace is bound via .Release.Namespace at helm install time"}, {"rule": "AC_K8S_0080", "comment": "seccompProfile RuntimeDefault is set in values.yaml (podSecurityContext)"}]'
|
||||
checkov.io/skip1: CKV_K8S_21=Namespace bound via .Release.Namespace at helm install/template time
|
||||
checkov.io/skip2: CKV_K8S_31=seccompProfile RuntimeDefault set in values.yaml
|
||||
spec:
|
||||
{{- if not .Values.autoscaling.enabled }}
|
||||
replicas: {{ .Values.replicaCount }}
|
||||
@@ -19,8 +23,10 @@ spec:
|
||||
{{- include "knowledge-explorer.selectorLabels" . | nindent 6 }}
|
||||
template:
|
||||
metadata:
|
||||
{{- with .Values.podAnnotations }}
|
||||
annotations:
|
||||
runterrascan.io/skip: '[{"rule": "AC_K8S_0080", "comment": "seccompProfile RuntimeDefault is set in values.yaml (podSecurityContext)"}]'
|
||||
checkov.io/skip1: CKV_K8S_31=seccompProfile RuntimeDefault set in values.yaml
|
||||
{{- with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
labels:
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
# checkov:skip=CKV_K8S_21:Namespace is bound via .Release.Namespace at helm install/template time; this chart is namespace-portable by design.
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
@@ -5,6 +6,9 @@ metadata:
|
||||
namespace: {{ .Release.Namespace }}
|
||||
labels:
|
||||
{{- include "knowledge-explorer.labels" . | nindent 4 }}
|
||||
annotations:
|
||||
runterrascan.io/skip: '[{"rule": "AC_K8S_0086", "comment": "Namespace is bound via .Release.Namespace at helm install time"}]'
|
||||
checkov.io/skip1: CKV_K8S_21=Namespace bound via .Release.Namespace at helm install/template time
|
||||
spec:
|
||||
type: {{ .Values.service.type }}
|
||||
ports:
|
||||
|
||||
@@ -9,7 +9,10 @@ railway add --database redis
|
||||
railway variable --set "FALKORDB_HOST=${{Redis.REDISHOST}}"
|
||||
railway variable --set "FALKORDB_PORT=${{Redis.REDISPORT}}"
|
||||
railway variable --set "ALLOWED_ORIGINS=https://${{RAILWAY_PUBLIC_DOMAIN}}"
|
||||
railway variable --set "SEMANTICA_API_KEY=$(openssl rand -hex 32)"
|
||||
railway up
|
||||
```
|
||||
|
||||
The Redis plugin variables are wired to the requested FalkorDB env names for deployment compatibility. The Explorer currently reads these settings but does not persist graph state to FalkorDB.
|
||||
|
||||
Railway exposes this service on a public domain, so `SEMANTICA_API_KEY` is required — without it the Explorer refuses every protected route (503) rather than serving anonymously. Pass the same value as the `X-API-Key` header from any client that talks to the deployed API.
|
||||
|
||||
@@ -9,3 +9,5 @@ render blueprint apply deploy/render/render.yaml
|
||||
```
|
||||
|
||||
After creation, update `ALLOWED_ORIGINS` in the Render dashboard if you attach a custom domain.
|
||||
|
||||
`SEMANTICA_API_KEY` is auto-generated by the blueprint (`generateValue: true`) since this service gets a public `onrender.com` URL — without it the Explorer refuses every protected route (503) rather than serving anonymously. Find the generated value in the Render dashboard's environment tab and pass it as the `X-API-Key` header from any client that talks to the deployed API.
|
||||
|
||||
@@ -20,6 +20,8 @@ services:
|
||||
type: keyvalue
|
||||
name: semantica-explorer-redis
|
||||
property: port
|
||||
- key: SEMANTICA_API_KEY
|
||||
generateValue: true
|
||||
|
||||
- type: keyvalue
|
||||
name: semantica-explorer-redis
|
||||
|
||||
@@ -16,6 +16,8 @@ services:
|
||||
ALLOWED_ORIGINS: http://localhost:5173,http://127.0.0.1:5173,http://localhost:8000,http://127.0.0.1:8000
|
||||
FALKORDB_HOST: falkordb
|
||||
FALKORDB_PORT: "6379"
|
||||
# Local dev only: this compose file is not for public exposure.
|
||||
SEMANTICA_ALLOW_ANONYMOUS: "true"
|
||||
volumes:
|
||||
- ./semantica:/app/semantica
|
||||
- ./pyproject.toml:/app/pyproject.toml:ro
|
||||
|
||||
@@ -8,6 +8,11 @@ services:
|
||||
FALKORDB_HOST: falkordb
|
||||
FALKORDB_PORT: "6379"
|
||||
ALLOWED_ORIGINS: ${ALLOWED_ORIGINS:-http://localhost:8000,http://127.0.0.1:8000}
|
||||
# Required for API access - the Explorer refuses all protected routes
|
||||
# (503) until this is set. Generate one with `openssl rand -hex 32`.
|
||||
SEMANTICA_API_KEY: ${SEMANTICA_API_KEY:-}
|
||||
# Trusted local-only setups only: bypasses the API key entirely.
|
||||
SEMANTICA_ALLOW_ANONYMOUS: ${SEMANTICA_ALLOW_ANONYMOUS:-false}
|
||||
depends_on:
|
||||
falkordb:
|
||||
condition: service_started
|
||||
|
||||
@@ -23,7 +23,7 @@ Loads data from any source into the pipeline as a unified `SourceDocument`.
|
||||
| Parquet | `ingest.ParquetIngestor` | PyArrow, Hive-style partitions (v0.5.0) |
|
||||
| XML | `ingest.XMLIngestor` | XXE-safe lxml, XSD/DTD validation (v0.5.0) |
|
||||
| Web pages | `ingest.WebIngestor` | Configurable depth, link filtering |
|
||||
| SQL / Snowflake | `ingest.DBIngestor` / `ingest.SnowflakeIngestor` | Custom SQL, schema introspection |
|
||||
| SQL / Snowflake / Databricks | `ingest.DBIngestor` / `ingest.SnowflakeIngestor` / `ingest.DatabricksIngestor` | Custom SQL, schema introspection, Unity Catalog lineage |
|
||||
| Kafka / streams | `ingest.StreamIngestor` | Real-time feed ingestion |
|
||||
| Email | `ingest.EmailIngestor` | IMAP/SMTP with attachment extraction |
|
||||
| Repositories | `ingest.RepoIngestor` | Git repos, code structure |
|
||||
|
||||
@@ -18,7 +18,7 @@ Find your goal below. The **Module** column is your import path; **Key class** i
|
||||
| Crawl a website | `ingest` | `WebIngestor` |
|
||||
| Load Parquet files or partitioned datasets | `ingest` | `ParquetIngestor` |
|
||||
| Ingest XML with schema validation | `ingest` | `XMLIngestor` |
|
||||
| Ingest from SQL, Snowflake, Kafka, or email | `ingest` | `DBIngestor`, `SnowflakeIngestor`, `StreamIngestor` |
|
||||
| Ingest from SQL, Snowflake, Databricks, Kafka, or email | `ingest` | `DBIngestor`, `SnowflakeIngestor`, `DatabricksIngestor`, `StreamIngestor` |
|
||||
| Extract clean text and tables from a document | `parse` | `DocumentParser` |
|
||||
| Parse complex PDFs with OCR or multi-column layout | `parse` | `DoclingParser` |
|
||||
| Chunk text for embedding or RAG | `split` | `TextSplitter` |
|
||||
|
||||
+5
-5
@@ -17,22 +17,22 @@ icon: "quote-left"
|
||||
author = {Hawksight AI},
|
||||
year = {2026},
|
||||
url = {https://github.com/semantica-agi/semantica},
|
||||
version = {0.5.1},
|
||||
version = {0.6.0},
|
||||
doi = {10.5281/zenodo.XXXXXXX}
|
||||
}
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="APA">
|
||||
Hawksight AI. (2026). *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering* (Version 0.5.1) \[Computer software\]. https://github.com/semantica-agi/semantica
|
||||
Hawksight AI. (2026). *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering* (Version 0.6.0) \[Computer software\]. https://github.com/semantica-agi/semantica
|
||||
</Tab>
|
||||
<Tab title="MLA">
|
||||
Hawksight AI. *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering*. Version 0.5.1, GitHub, 2026, https://github.com/semantica-agi/semantica.
|
||||
Hawksight AI. *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering*. Version 0.6.0, GitHub, 2026, https://github.com/semantica-agi/semantica.
|
||||
</Tab>
|
||||
<Tab title="Chicago">
|
||||
Hawksight AI. *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering*. Version 0.5.1. GitHub, 2026. https://github.com/semantica-agi/semantica.
|
||||
Hawksight AI. *Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering*. Version 0.6.0. GitHub, 2026. https://github.com/semantica-agi/semantica.
|
||||
</Tab>
|
||||
<Tab title="IEEE">
|
||||
Hawksight AI, "Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering," Version 0.5.1, GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
|
||||
Hawksight AI, "Semantica: An Open Source Framework for Semantic Layers and Knowledge Engineering," Version 0.6.0, GitHub, 2026. \[Online\]. Available: https://github.com/semantica-agi/semantica
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
|
||||
+2
-1
@@ -103,7 +103,8 @@
|
||||
"pages": [
|
||||
"integrations/agno",
|
||||
"integrations/docling",
|
||||
"integrations/snowflake"
|
||||
"integrations/snowflake",
|
||||
"integrations/databricks"
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
+2
-2
@@ -17,7 +17,7 @@ icon: "circle-question"
|
||||
| API key required? | Optional: pattern extraction works with no keys |
|
||||
| Works with LangChain / LlamaIndex? | Yes: Semantica is a layer on top, not a replacement |
|
||||
| Production-ready? | Yes: 1,000+ tests, v0.5.0 ships with 12 security fixes |
|
||||
| Latest version? | **v0.5.1** (June 2026) |
|
||||
| Latest version? | **v0.6.0** (July 2026) |
|
||||
| Local LLMs? | Yes: Ollama via LiteLLM, HuggingFaceLLM for air-gapped |
|
||||
|
||||
|
||||
@@ -129,7 +129,7 @@ If you're on an older version, install extras individually: `pip install "semant
|
||||
| :-------- | :------- |
|
||||
| **Files** | PDF, DOCX, HTML, JSON, CSV, Excel, PPTX, Parquet (v0.5.0), XML (v0.5.0), archives |
|
||||
| **Web** | `WebIngestor` crawl, RSS feeds, sitemaps |
|
||||
| **Databases** | PostgreSQL, MySQL, Snowflake via `DBIngestor` / `SnowflakeIngestor` |
|
||||
| **Databases** | PostgreSQL, MySQL, Snowflake, Databricks via `DBIngestor` / `SnowflakeIngestor` / `DatabricksIngestor` |
|
||||
| **NoSQL** | MongoDB via `MongoIngestor`, DuckDB via `DuckDBIngestor` |
|
||||
| **Streams** | Kafka, real-time ingestion via `StreamIngestor` |
|
||||
| **Protocols** | MCP (Model Context Protocol) via `MCPIngestor` |
|
||||
|
||||
@@ -42,7 +42,7 @@ icon: "rocket"
|
||||
Verify installation:
|
||||
```python
|
||||
import semantica
|
||||
print(semantica.__version__) # 0.5.1
|
||||
print(semantica.__version__) # 0.6.0
|
||||
```
|
||||
</Check>
|
||||
</Step>
|
||||
|
||||
+1
-1
@@ -149,7 +149,7 @@ A database optimized for storing and querying graph-structured data using node a
|
||||
A retrieval strategy combining vector similarity search with keyword or metadata filtering: higher accuracy than either approach alone.
|
||||
|
||||
**Triplet Store**
|
||||
A database designed specifically for storing and querying RDF `(subject, predicate, object)` triples. Semantica supports Blazegraph, Apache Jena, and RDF4J.
|
||||
A database designed specifically for storing and querying RDF `(subject, predicate, object)` triples. Semantica supports embedded Oxigraph as well as Blazegraph, Apache Jena, and RDF4J.
|
||||
|
||||
**Vector Store**
|
||||
A database optimized for storing and searching high-dimensional embedding vectors by similarity. Semantica supports FAISS, Pinecone, Weaviate, Qdrant, Milvus, and PgVector.
|
||||
|
||||
@@ -6,6 +6,45 @@ icon: "brain"
|
||||
|
||||
`AgentContext` maintains a persistent memory layer for LLM agents — storing observations as vector embeddings, retrieving them by semantic similarity, and optionally blending graph proximity into the ranking. Use it when your agent needs to recall past findings across sessions without re-reading source material on every restart.
|
||||
|
||||
## What Is Agent Memory?
|
||||
|
||||
Agent Memory provides persistent storage and intelligent retrieval of information across multiple agent sessions. `AgentContext` is the core component that orchestrates memory storage, retrieval, and management by combining three key systems:
|
||||
|
||||
**VectorStore** handles semantic search using vector embeddings. It stores text as high-dimensional vectors and retrieves similar content through cosine similarity or other distance metrics.
|
||||
|
||||
**ContextGraph** maintains structured knowledge as nodes (entities) and edges (relationships). This enables multi-hop traversal and graph-aware retrieval that follows connections between related entities.
|
||||
|
||||
**AgentContext** orchestrates both components, providing a unified interface for storing memories, retrieving relevant context, and managing conversations across sessions.
|
||||
|
||||
**Persistent memory vs stateless retrieval:** Traditional RAG systems lose context between sessions. Agent Memory persists learned information, conversation history, and accumulated knowledge across restarts, enabling long-term memory and cross-session recall.
|
||||
|
||||
## Why Use Agent Memory?
|
||||
|
||||
**Cross-session recall.** Agents remember previous interactions, findings, and decisions without re-processing source material after restarts.
|
||||
|
||||
**Long-term knowledge accumulation.** Information builds up over time as agents process more documents, creating increasingly rich knowledge bases for future queries.
|
||||
|
||||
**Conversation history.** Agents maintain context within conversations and can reference earlier parts of extended interactions or investigations.
|
||||
|
||||
**Graph-aware retrieval.** Beyond simple semantic similarity, retrieval follows entity relationships to find connected information that pure vector search would miss.
|
||||
|
||||
**Decision tracking.** Record decisions with full context and reasoning paths, enabling audit trails and precedent matching for similar future scenarios.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use Agent Memory for:**
|
||||
- Long-running agents that need to accumulate knowledge over time
|
||||
- Research assistants that build understanding across multiple sessions
|
||||
- Investigation workflows where context builds incrementally
|
||||
- Systems that must remember prior interactions and decisions
|
||||
- Scenarios requiring audit trails and decision precedents
|
||||
|
||||
**Do not use when:**
|
||||
- Building simple stateless RAG systems for one-time document queries
|
||||
- Performing one-off document searches without need for persistence
|
||||
- Running temporary experiments that don't require knowledge retention
|
||||
- Simple retrieval tasks where relationships between entities don't matter
|
||||
|
||||
<Info>
|
||||
This guide covers the memory layer. For graph-enriched traversal and entity linking, see [Context Graphs](context-graphs). For decision accountability — recording, auditing, and causally tracing what the agent chose — see [Decision Intelligence](decision-intelligence).
|
||||
</Info>
|
||||
@@ -18,11 +57,10 @@ Configure the vector store, knowledge graph, and `AgentContext` together at star
|
||||
from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
|
||||
# The FAISS index persists to disk at index_path — restart-safe
|
||||
# The VectorStore relies on explicit save()/load() for persistence
|
||||
ti_vs = VectorStore(
|
||||
backend="faiss",
|
||||
dimension=768,
|
||||
index_path="ti_agent/memory.faiss",
|
||||
)
|
||||
|
||||
# The ContextGraph holds entity nodes and their relationships
|
||||
@@ -141,7 +179,7 @@ results = ti_agent.retrieve(
|
||||
"cloud OAuth token theft campaigns",
|
||||
max_results=10,
|
||||
use_graph=True,
|
||||
anchor_node="APT29", # BFS starts from this node in the knowledge graph
|
||||
anchor_node="APT29", # Breadth-First Search (BFS) starts from this node in the knowledge graph
|
||||
max_hops=3,
|
||||
proximity_weight=0.35, # 65% semantic + 35% proximity — tune to your graph density
|
||||
min_score=0.1,
|
||||
@@ -242,7 +280,7 @@ from semantica.llms import Groq
|
||||
|
||||
ti_graph = ContextGraph(advanced_analytics=True, node_embeddings=True)
|
||||
ti_agent = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="ti_memory.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=ti_graph,
|
||||
retention_days=365,
|
||||
max_memories=50000,
|
||||
@@ -297,7 +335,7 @@ from semantica.llms import Groq
|
||||
|
||||
soc_graph = ContextGraph()
|
||||
soc_agent = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="soc_memory.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=soc_graph,
|
||||
retention_days=180,
|
||||
max_memories=100000,
|
||||
@@ -370,7 +408,7 @@ from semantica.vector_store import VectorStore
|
||||
|
||||
clinical_graph = ContextGraph(advanced_analytics=True)
|
||||
clinical_agent = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="clinical.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=clinical_graph,
|
||||
retention_days=3650, # 10-year clinical record retention
|
||||
max_memories=500000,
|
||||
@@ -446,7 +484,7 @@ from semantica.vector_store import VectorStore
|
||||
|
||||
credit_graph = ContextGraph(advanced_analytics=True)
|
||||
credit_agent = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="credit.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=credit_graph,
|
||||
retention_days=2555, # 7-year regulatory retention
|
||||
max_memories=1000000,
|
||||
@@ -546,7 +584,7 @@ from semantica.vector_store import VectorStore
|
||||
|
||||
# Create a fresh context with matching configuration
|
||||
ti_agent_restored = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="ti_memory.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=ContextGraph(advanced_analytics=True),
|
||||
retention_days=365,
|
||||
decision_tracking=True,
|
||||
@@ -605,6 +643,18 @@ s = ti_agent.stats()
|
||||
print("Total memories: {}".format(s.get("total_items", 0)))
|
||||
```
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Forgetting to persist memory before shutdown.** Agent Memory is stored in memory during execution. Without calling `save()` before process termination, all accumulated memories, graph relationships, and conversations are lost.
|
||||
|
||||
**Using the same conversation namespace for unrelated tasks.** Conversation IDs should scope related interactions. Using a single conversation for multiple unrelated investigations pollutes retrieval results and makes context less focused.
|
||||
|
||||
**Storing excessive low-value information.** Not every observation needs permanent storage. Focus on storing insights, decisions, and significant findings rather than verbose raw logs or temporary calculations.
|
||||
|
||||
**Using Agent Memory when simple retrieval would be sufficient.** For one-time document lookups or stateless queries, traditional retrieval is simpler and more efficient than setting up persistent memory infrastructure.
|
||||
|
||||
**Retrieving too much context and increasing latency.** Large `max_results`, high `max_hops`, or broad queries can retrieve excessive context, increasing LLM token usage and response latency. Start with focused retrieval parameters.
|
||||
|
||||
## Related Guides
|
||||
|
||||
- [Context Graphs](context-graphs) — How the underlying `ContextGraph` stores entity nodes and decision nodes; temporal interval reasoning; deduplication before node insertion; ontology from graph.
|
||||
|
||||
@@ -4,12 +4,99 @@ description: "Snapshot, version, diff, and migrate knowledge graphs and ontologi
|
||||
icon: "clock-rotate-left"
|
||||
---
|
||||
|
||||
Knowledge graphs change constantly — threat actors get re-attributed, CVE scores update when exploits drop, clinical trial endpoints shift between phases. `TemporalVersionManager` gives your graph a verifiable history: named snapshots before every consequential change, diffs between any two states, one-call rollback, and SHA-256 checksum verification before publishing downstream.
|
||||
## What Is Change Management & Versioning?
|
||||
|
||||
Knowledge graphs change constantly. `TemporalVersionManager` gives your graph a verifiable history by capturing complete state snapshots at specific points in time. It allows you to take named snapshots before consequential changes, generate detailed diffs between any two states, roll back to previous versions with a single call, and verify SHA-256 checksums before publishing data downstream.
|
||||
|
||||
## Storage Behavior
|
||||
|
||||
Pass `storage_path`, e.g. `TemporalVersionManager(storage_path="versions.db")`, to persist snapshots to a SQLite database on disk. Omit `storage_path` and it defaults to an in-memory store that vanishes when your script finishes.
|
||||
|
||||
## Why Use Change Management?
|
||||
|
||||
Change Management acts as your safety net and audit trail. Use it to:
|
||||
- **Safeguard Ingestion**: Take a snapshot before a large batch ingestion so you can instantly roll back if the data is corrupted.
|
||||
- **Audit Trails**: Maintain a verifiable log of when a change occurred, who authorized it, and exactly what nodes/edges were modified.
|
||||
- **Release Gating**: Compare staging and production graphs and verify checksums before signing off on a release.
|
||||
|
||||
## Which Tool Do I Need?
|
||||
|
||||
Semantica offers multiple tracking features. It is critical to choose the right one:
|
||||
- **Change Management** (this guide): Use for **whole-graph snapshots**, state diffs, and full rollbacks.
|
||||
- **Provenance**: Use for granular **source and lineage tracking**. It answers *"Which specific document did this node come from?"*
|
||||
- **Agent Memory**: Use for **conversational and context state**. It answers *"What decisions did the AI agent make during this session?"*
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
- **When to Use**: You have critical checkpoints (like daily feeds, partner merges, or regulatory submissions) where you need to freeze the entire state of the graph and potentially revert it.
|
||||
- **When NOT to Use**: You have a massive, multi-million node graph and want to track every minor edit. Because `TemporalVersionManager` snapshots the entire graph dictionary, snapshotting huge graphs too frequently will cause severe storage bloat. Use Provenance for granular tracking instead.
|
||||
|
||||
<Info>
|
||||
`TemporalVersionManager` integrates with `AgentContext.flush_checkpoint()` — agent checkpoints and manual snapshots share the same storage format, so diffs work across both.
|
||||
`TemporalVersionManager` integrates directly with `AgentContext.flush_checkpoint()` — agent checkpoints and manual snapshots share the same storage format, allowing diffs across both automated and manual workflows.
|
||||
</Info>
|
||||
|
||||
---
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
A standard change management cycle follows this progression:
|
||||
|
||||
1. **Snapshot**: Capture the baseline graph state.
|
||||
2. **Modify**: Run your ingestion, mutations, or analysis.
|
||||
3. **Compare**: Generate a diff to see what changed.
|
||||
4. **Verify**: Check the SHA-256 hash to ensure data integrity.
|
||||
5. **Tag**: Apply a human-readable tag (e.g., `approved`).
|
||||
6. **Rollback**: Revert the graph state if the modifications were incorrect.
|
||||
|
||||
---
|
||||
|
||||
## Universal Example: Employee Profile Update
|
||||
|
||||
Let's look at a universally understood example: tracking an employee's department transfer.
|
||||
|
||||
```python
|
||||
from semantica.change_management import TemporalVersionManager
|
||||
from semantica.context import ContextGraph
|
||||
|
||||
# 1. Setup Graph and Version Manager
|
||||
graph = ContextGraph()
|
||||
graph.add_node("emp-101", "Employee", "Alice")
|
||||
graph.add_node("dept-hr", "Department", "Human Resources")
|
||||
graph.add_edge("emp-101", "dept-hr", "works_in")
|
||||
|
||||
# SQLite persistence is enabled because we provided a storage_path
|
||||
vm = TemporalVersionManager(storage_path="hr_versions.db")
|
||||
|
||||
# 2. Snapshot the baseline
|
||||
snap_v1 = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "v1_baseline",
|
||||
author = "hr_system@example.com",
|
||||
description = "Initial employee graph",
|
||||
)
|
||||
|
||||
# 3. Modify the graph (Transfer Alice to Engineering)
|
||||
graph.add_node("dept-eng", "Department", "Engineering")
|
||||
graph.add_edge("emp-101", "dept-eng", "works_in")
|
||||
|
||||
# 4. Snapshot the post-change state
|
||||
snap_v2 = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "v2_transfer",
|
||||
author = "hr_admin@example.com",
|
||||
description = "Alice transferred to Engineering",
|
||||
)
|
||||
|
||||
# 5. Compare versions
|
||||
diff = vm.compare_versions("v1_baseline", "v2_transfer")
|
||||
print("Nodes added:", diff["summary"]["nodes_added"]) # 1 (Engineering)
|
||||
print("Edges added:", diff["summary"]["edges_added"]) # 1 (works_in Eng)
|
||||
```
|
||||
|
||||
Now let's explore these capabilities in more depth using domain-specific scenarios.
|
||||
|
||||
---
|
||||
|
||||
## Creating Snapshots
|
||||
|
||||
Take a snapshot before any consequential change: an ingestion sweep, a partner feed merge, or an automated enrichment run.
|
||||
@@ -28,7 +115,7 @@ vm = TemporalVersionManager(storage_path="cti_versions.db")
|
||||
snap_pre = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "q3_2025_baseline",
|
||||
author = "analyst_zhang",
|
||||
author = "analyst_zhang@example.com",
|
||||
description = "CTI baseline before Q3 OSINT sweep",
|
||||
)
|
||||
|
||||
@@ -50,7 +137,7 @@ graph.add_edge("apt40", "cve-2024-21412", "exploits", weight=0.88)
|
||||
snap_post = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "q3_2025_post_nvd_sweep",
|
||||
author = "osint_pipeline",
|
||||
author = "osint_pipeline@example.com",
|
||||
description = "After NVD weekly sweep — 2025-07-14",
|
||||
)
|
||||
```
|
||||
@@ -108,7 +195,7 @@ vm.restore_snapshot(
|
||||
vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "q3_2025_rollback",
|
||||
author = "analyst_zhang",
|
||||
author = "analyst_zhang@example.com",
|
||||
description = "Rolled back to baseline after corrupted OSINT batch",
|
||||
)
|
||||
```
|
||||
@@ -146,14 +233,14 @@ Sample output:
|
||||
Graph Change Log
|
||||
============================================================
|
||||
|
||||
[2025-07-01] q3_2025_baseline (by analyst_zhang)
|
||||
[2025-07-01] q3_2025_baseline (by analyst_zhang@example.com)
|
||||
CTI baseline before Q3 OSINT sweep
|
||||
|
||||
[2025-07-14] q3_2025_post_nvd_sweep (by osint_pipeline)
|
||||
[2025-07-14] q3_2025_post_nvd_sweep (by osint_pipeline@example.com)
|
||||
After NVD weekly sweep — 2025-07-14
|
||||
Changes: +2 nodes -0 nodes +1 edges -0 edges
|
||||
|
||||
[2025-07-14] q3_2025_rollback (by analyst_zhang)
|
||||
[2025-07-14] q3_2025_rollback (by analyst_zhang@example.com)
|
||||
Rolled back to baseline after corrupted OSINT batch
|
||||
Changes: -2 nodes +0 nodes -1 edges +0 edges
|
||||
```
|
||||
@@ -218,6 +305,18 @@ print("Decisions added :", len(diff["decisions_added"]))
|
||||
print("Relationships added:", len(diff["relationships_added"]))
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
- **Snapshotting huge graphs too frequently**: `TemporalVersionManager` snapshots the entire graph structure. Doing this on every minor edit for a massive graph will cause severe storage bloat. Use it for milestone gating, not event sourcing.
|
||||
- **Forgetting `attach_to_graph` before mutation tracking**: If you want to use `get_node_history()`, you must call `vm.attach_to_graph(graph)` *before* any mutations happen. Otherwise, the events will not be captured.
|
||||
- **Confusing provenance with versioning**: Do not use version snapshots to answer "Where did this specific node's data come from?". That is the role of the Provenance module. Versioning tracks the state of the *entire* graph at a point in time.
|
||||
- **Forgetting rollback confirmation requirements**: Calling `restore_snapshot` in automated scripts will raise a `ProcessingError` and crash your pipeline unless you explicitly pass `require_confirmation=False`.
|
||||
- **Storage growth from excessive snapshots**: Over time, SQLite databases can grow large if you never prune old snapshots or if you snapshot unnecessarily.
|
||||
|
||||
---
|
||||
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
@@ -236,7 +335,7 @@ today = datetime.date.today().isoformat()
|
||||
snap_pre = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = f"pre_nvd_{today}",
|
||||
author = "osint_pipeline",
|
||||
author = "osint_pipeline@example.com",
|
||||
description = "CTI baseline before NVD sweep",
|
||||
)
|
||||
|
||||
@@ -248,7 +347,7 @@ graph.add_edge("apt29-q3-cluster", "cve-2025-1337", "weaponizes", weight=0.91)
|
||||
snap_post = vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = f"post_nvd_{today}",
|
||||
author = "osint_pipeline",
|
||||
author = "osint_pipeline@example.com",
|
||||
description = "After NVD sweep",
|
||||
)
|
||||
|
||||
@@ -283,7 +382,7 @@ graph.add_edge("attacker-ip", "wkstn-047", "initial_access", weight=0.95)
|
||||
vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "ir042_t0_triage",
|
||||
author = "analyst_chen",
|
||||
author = "analyst_chen@example.com",
|
||||
description = "T+0 — one compromised host identified",
|
||||
)
|
||||
|
||||
@@ -296,7 +395,7 @@ graph.add_edge("svc-backup", "dc01", "lateral_move", weight=0.82)
|
||||
vm.create_snapshot(
|
||||
graph = graph.to_dict(),
|
||||
version_label = "ir042_t2h_lateral",
|
||||
author = "analyst_chen",
|
||||
author = "analyst_chen@example.com",
|
||||
description = "T+2h — lateral movement to DC01 via stolen SVC-BACKUP",
|
||||
)
|
||||
|
||||
@@ -332,11 +431,11 @@ vm = TemporalVersionManager(storage_path="trial_xr401.db")
|
||||
|
||||
vm.create_snapshot(
|
||||
graph=graph_ph2.to_dict(), version_label="phase_ii_v1.0",
|
||||
author="clinical_data_team", description="Phase II — ORR primary, NSCLC",
|
||||
author="clinical_data_team@example.com", description="Phase II — ORR primary, NSCLC",
|
||||
)
|
||||
vm.create_snapshot(
|
||||
graph=graph_ph3.to_dict(), version_label="phase_iii_v2.0",
|
||||
author="clinical_data_team", description="Phase III — PFS co-primary, Docetaxel added",
|
||||
author="clinical_data_team@example.com", description="Phase III — PFS co-primary, Docetaxel added",
|
||||
)
|
||||
|
||||
diff = vm.compare_versions("phase_ii_v1.0", "phase_iii_v2.0")
|
||||
@@ -369,7 +468,7 @@ vm = TemporalVersionManager(storage_path="credit_risk_versions.db")
|
||||
|
||||
vm.create_snapshot(
|
||||
graph=graph.to_dict(), version_label="basel_v1.0",
|
||||
author="risk_model_team", description="Basel III CRE20 initial graph",
|
||||
author="risk_model_team@example.com", description="Basel III CRE20 initial graph",
|
||||
)
|
||||
|
||||
# Regulatory update — DSCR becomes mandatory
|
||||
@@ -378,7 +477,7 @@ graph.add_edge("regulation-cre20", "metric-dscr", "requires", weight=1.0)
|
||||
|
||||
vm.create_snapshot(
|
||||
graph=graph.to_dict(), version_label="basel_v1.1",
|
||||
author="risk_model_team", description="DSCR added per EBA GL 2020/06",
|
||||
author="risk_model_team@example.com", description="DSCR added per EBA GL 2020/06",
|
||||
)
|
||||
|
||||
diff = vm.compare_versions("basel_v1.0", "basel_v1.1")
|
||||
|
||||
@@ -10,9 +10,152 @@ icon: "code-merge"
|
||||
Run conflict detection after deduplication and before SHACL validation. Deduplication removes duplicate nodes; conflict resolution reconciles disagreeing property values on the same canonical entity. Running them out of order — detecting conflicts before deduplication — will produce spurious conflicts between entities that should have been merged first.
|
||||
</Info>
|
||||
|
||||
## Detecting the disagreement
|
||||
## What Is Conflict Resolution?
|
||||
|
||||
Start by loading your multi-source records for the same entity. `ConflictDetector` groups them by entity ID, then compares the values each source reports for a given property. Any entity where two or more sources report different values for the same property produces a `Conflict` object.
|
||||
When you merge data from multiple sources, the same real-world entity — a customer, a product, a threat actor, a drug compound — often appears with contradictory property values. One database says a customer's email is `alice@example.com`; another says `alice.smith@example.com`. One security feed rates a CVE at 10.0; two others rate it 9.1 and 9.5.
|
||||
|
||||
**Conflict resolution** is the systematic process of deciding which value is most trustworthy and recording that decision with evidence, so the canonical entity ends up with one defensible, auditable value per property.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Canonical entity** — The single authoritative record for a real-world thing. After deduplication, each entity has exactly one canonical node in your graph. Conflict resolution determines which property values belong on that node.
|
||||
|
||||
**Conflicting values** — Two or more different values asserted for the same property on the same canonical entity, each reported by a different source.
|
||||
|
||||
**Credibility score** — A number between 0.0 and 1.0 you attach to each source record, indicating how reliable that source is. A government registry might carry 0.99; a scraped blog might carry 0.30. You supply these; Semantica uses them during `CREDIBILITY_WEIGHTED` resolution.
|
||||
|
||||
**Confidence score** — A number between 0.0 and 1.0 the resolver *computes* after resolution, reflecting how certain the outcome is. A unanimous vote produces high confidence; a close split among equally credible sources produces lower confidence. This appears on `ResolutionResult.confidence` and should be read as a signal, not a guarantee that the resolved value is correct.
|
||||
|
||||
**Resolution strategy** — The rule for picking the winning value: majority vote, credibility-weighted average, latest timestamp, and so on. See [Resolution strategies at a glance](#resolution-strategies-at-a-glance) for the full list.
|
||||
|
||||
**Audit trail** — The complete record of every resolution decision: conflict ID, strategy used, resolved value, sources consulted, and confidence score. Returned by `resolver.get_resolution_history()`.
|
||||
|
||||
**Provenance-aware resolution** — Resolution that records not just the winning value but which source it came from. Every `ResolutionResult` carries a `sources_used` field, so you can always trace a canonical value back to its origin — critical in regulated environments.
|
||||
|
||||
## Why Use Conflict Resolution?
|
||||
|
||||
- **Multi-source pipelines always produce disagreements.** Differences in update cadence, data-entry conventions, and source reliability are unavoidable. Without an explicit resolution step, you silently favor one source over another with no record of the choice.
|
||||
- **You get a defensible, auditable decision log.** Compliance teams, auditors, and domain experts need to know which source won and why. The audit trail provides exactly that.
|
||||
- **Easy cases are automated; hard cases are escalated.** Routine disagreements — slightly different name spellings, stale timestamps — are resolved algorithmically. Genuinely ambiguous cases — competing legal classifications, different clinical endpoints — are flagged for expert review without blocking the rest of the pipeline.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use conflict resolution when:**
|
||||
- You are merging two or more independent sources for the same entity.
|
||||
- Sources disagree on property values and you need a single canonical value.
|
||||
- You need an auditable record of every resolution decision.
|
||||
- Some conflicts require domain-expert review before they can be resolved.
|
||||
|
||||
**Skip conflict resolution when:**
|
||||
- **A single authoritative source already exists.** If one system is always correct for a given property, read from it directly. Adding resolution machinery around a single source creates complexity without benefit.
|
||||
- **All sources are always in agreement.** Verify this empirically before skipping; silent disagreements are common in practice.
|
||||
- **You want to preserve all conflicting values.** If retaining every source's assertion matters more than picking one, model provenance directly in your graph schema instead of resolving to one winner.
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[Raw Sources] --> B[Deduplication]
|
||||
B --> C[Conflict Detection]
|
||||
C --> D{Auto-resolvable?}
|
||||
D -- Yes --> E[Apply Resolution Strategy]
|
||||
D -- No --> F[Expert Review Queue]
|
||||
E --> G[Persist Canonical Values]
|
||||
F --> G
|
||||
G --> H[SHACL Validation]
|
||||
```
|
||||
|
||||
1. **Deduplication** — Merge duplicate nodes so each entity has exactly one canonical record. Conflict resolution operates on a single canonical entity; you must identify it before comparing what different sources say about it. See [Deduplication](deduplication).
|
||||
2. **Conflict Detection** — Call `detect_entity_conflicts()` to surface all property disagreements at once, or `detect_value_conflicts()` to target a specific property.
|
||||
3. **Resolution** — For each conflict, apply a strategy (`CREDIBILITY_WEIGHTED`, `MOST_RECENT`, `VOTING`, etc.) or route it for expert review (`EXPERT_REVIEW`).
|
||||
4. **Persist Canonical Values** — Write resolved values back to your canonical entities or graph store. See [Persisting resolved values](#persisting-resolved-values).
|
||||
5. **SHACL Validation** — Enforce structural constraints on the resolved graph to confirm it satisfies your ontology. See [SHACL Validation](shacl-validation).
|
||||
|
||||
## Quick Start: A Beginner Example
|
||||
|
||||
Before diving into domain-specific scenarios, here is the shortest path through the API. Three systems — a CRM, an ERP, and an LDAP directory — hold slightly different contact details for the same customer. Two of the three agree that the canonical email is `alice.smith@example.com`; the CRM has an older value.
|
||||
|
||||
```python
|
||||
from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionStrategy
|
||||
|
||||
# Same customer, three sources — only email disagrees
|
||||
customer_records = [
|
||||
{"id": "cust-001", "source": "crm", "email": "alice@example.com", "phone": "+1-555-0100"},
|
||||
{"id": "cust-001", "source": "erp", "email": "alice.smith@example.com", "phone": "+1-555-0100"},
|
||||
{"id": "cust-001", "source": "ldap", "email": "alice.smith@example.com", "phone": "+1-555-0100"},
|
||||
]
|
||||
|
||||
# Step 1: Detect all property conflicts at once — no need to name each property
|
||||
detector = ConflictDetector()
|
||||
conflicts = detector.detect_entity_conflicts(customer_records)
|
||||
|
||||
print(f"Conflicts found: {len(conflicts)}")
|
||||
for c in conflicts:
|
||||
print(f" Property : {c.property_name}")
|
||||
print(f" Values : {c.conflicting_values}")
|
||||
print(f" Severity : {c.severity}")
|
||||
|
||||
# Step 2: Resolve — two out of three sources agree, so majority vote wins
|
||||
resolver = ConflictResolver()
|
||||
results = resolver.resolve_conflicts(conflicts, strategy=ResolutionStrategy.VOTING)
|
||||
|
||||
for r in results:
|
||||
print(f"\n[{'RESOLVED' if r.resolved else 'REVIEW'}] {r.conflict_id}")
|
||||
print(f" Resolved value : {r.resolved_value}")
|
||||
print(f" Strategy : {r.resolution_strategy}")
|
||||
print(f" Confidence : {r.confidence:.0%}")
|
||||
print(f" Sources used : {r.sources_used}")
|
||||
```
|
||||
|
||||
```text
|
||||
Conflicts found: 1
|
||||
Property : email
|
||||
Values : ['alice@example.com', 'alice.smith@example.com', 'alice.smith@example.com']
|
||||
Severity : medium
|
||||
|
||||
[RESOLVED] cust-001_email_conflict
|
||||
Resolved value : alice.smith@example.com
|
||||
Strategy : voting
|
||||
Confidence : 67%
|
||||
Sources used : ['crm', 'erp', 'ldap']
|
||||
```
|
||||
|
||||
`detect_entity_conflicts()` scanned both `email` and `phone` automatically — you did not name them. Because `phone` is identical across all three records, no conflict was detected for it. The email disagreement resolves to `alice.smith@example.com` because two of three sources agree on that value.
|
||||
|
||||
When every conflict in a batch should use the same strategy, pass `strategy=` directly to `resolve_conflicts()`. Use `set_resolution_rule()` when different entity-property pairs need different strategies — explained in [Setting per-property resolution rules](#setting-per-property-resolution-rules).
|
||||
|
||||
## Detecting Conflicts
|
||||
|
||||
`ConflictDetector` provides three methods. Choose the one that fits your situation:
|
||||
|
||||
| Method | What it scans | When to use |
|
||||
| :--- | :--- | :--- |
|
||||
| `detect_entity_conflicts(entities)` | Every property on each entity at once | First pass; you do not know in advance which properties conflict |
|
||||
| `detect_value_conflicts(entities, property_name)` | One named property across all entities | Targeted check for a known hot-spot property |
|
||||
| `detect_relationship_conflicts(relationships)` | Edge types between the same node pair | Structural disagreements in graph edges |
|
||||
|
||||
### Scanning All Properties at Once — `detect_entity_conflicts`
|
||||
|
||||
`detect_entity_conflicts()` is the recommended starting point for a new pipeline. It inspects every property found on your entity records and returns a single flat list of all conflicts — without you having to enumerate properties in advance.
|
||||
|
||||
```python
|
||||
detector = ConflictDetector()
|
||||
all_conflicts = detector.detect_entity_conflicts(records)
|
||||
# Returns every conflict across every property in one call
|
||||
```
|
||||
|
||||
If you have registered conflict fields for a specific entity type, pass `entity_type` to limit detection to those fields:
|
||||
|
||||
```python
|
||||
# Limit detection to fields registered for this entity type
|
||||
all_conflicts = detector.detect_entity_conflicts(records, entity_type="vulnerability")
|
||||
```
|
||||
|
||||
Without `entity_type`, the detector checks every key found on your entity dicts (excluding bookkeeping fields such as `id`, `source`, and `metadata`). Start here to get a complete picture, then decide which conflicts need which resolution strategy.
|
||||
|
||||
### Scanning a Specific Property — `detect_value_conflicts`
|
||||
|
||||
Use `detect_value_conflicts()` when you already know which property to check, or when you want to apply different detection logic to each property. `ConflictDetector` groups the records by entity ID, then compares each source's value for that property. Any entity where two or more sources report different values produces a `Conflict` object.
|
||||
|
||||
```python
|
||||
from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionStrategy
|
||||
@@ -25,7 +168,7 @@ cve_records = [
|
||||
"cvss_score": 10.0,
|
||||
"exploit_status": "unconfirmed",
|
||||
"vector": "AV:N/AC:L/PR:N/UI:N/S:C/C:H/I:H/A:H",
|
||||
"credibility_score": 0.98,
|
||||
"metadata": {"timestamp": "2024-04-11T12:00:00Z"},
|
||||
},
|
||||
{
|
||||
"id": "cve-2024-3400",
|
||||
@@ -33,7 +176,7 @@ cve_records = [
|
||||
"cvss_score": 9.1,
|
||||
"exploit_status": "in_wild",
|
||||
"vector": "AV:N/AC:L/PR:N/UI:N/S:U/C:H/I:H/A:H",
|
||||
"credibility_score": 0.91,
|
||||
"metadata": {"timestamp": "2024-04-12T15:30:00Z"},
|
||||
},
|
||||
{
|
||||
"id": "cve-2024-3400",
|
||||
@@ -41,7 +184,7 @@ cve_records = [
|
||||
"cvss_score": 9.5,
|
||||
"exploit_status": "in_wild",
|
||||
"vector": "AV:N/AC:H/PR:N/UI:N/S:C/C:H/I:H/A:H",
|
||||
"credibility_score": 0.87,
|
||||
"metadata": {"timestamp": "2024-04-12T12:00:00Z"},
|
||||
},
|
||||
]
|
||||
|
||||
@@ -74,20 +217,40 @@ Conflict: cve-2024-3400_cvss_score_conflict
|
||||
Values : [10.0, 9.1, 9.5]
|
||||
Severity : medium
|
||||
Sources : ['nvd', 'commercial_feed', 'vendor_paloalto']
|
||||
Action : Compare source documents and use most recent or authoritative source
|
||||
Action : Multiple conflicting values detected. Manual review recommended.
|
||||
```
|
||||
|
||||
Each `Conflict` captures the full picture: which entity, which property, every disagreeing value, and which source reported each. This is already enough to build a review queue — but the goal is to resolve these automatically according to rules you set.
|
||||
|
||||
## Setting per-property resolution rules
|
||||
## Setting Per-Property Resolution Rules
|
||||
|
||||
The key method is `set_resolution_rule(entity_id, property_name, strategy)`. It takes three arguments: which entity, which property, and which `ResolutionStrategy` to apply when that combination appears in a conflict. Rules are stored in the resolver and automatically applied when you call `resolve_conflicts()` without passing an explicit strategy.
|
||||
`set_resolution_rule(entity_id, property_name, strategy)` registers a strategy for a specific entity-property combination. The resolver stores the rule under the key `entity_id.property_name` and applies it automatically when you call `resolve_conflicts()`.
|
||||
|
||||
Because rules are keyed by both entity ID and property name, `set_resolution_rule()` is entity-specific. There is no wildcard that applies a rule to all entities or all properties at once.
|
||||
|
||||
**When to use `set_resolution_rule()`:** Use it when different entity-property combinations need different strategies. For example, an entity's `legal_name` might use `CREDIBILITY_WEIGHTED` while its `last_updated` uses `MOST_RECENT`. Registering a rule per combination lets the single `resolve_conflicts()` call handle all of them correctly in one pass.
|
||||
|
||||
**When to pass `strategy=` directly to `resolve_conflicts()`:** If every conflict in a batch should use the same strategy, pass it directly to `resolve_conflicts()` instead of registering a rule for each entity-property pair:
|
||||
|
||||
```python
|
||||
# Same strategy for every conflict — no per-property rules needed
|
||||
results = resolver.resolve_conflicts(all_conflicts, strategy=ResolutionStrategy.CREDIBILITY_WEIGHTED)
|
||||
```
|
||||
|
||||
This is cleaner than calling `set_resolution_rule()` in a loop over every entity just to apply the same strategy everywhere.
|
||||
|
||||
**Per-property rules for the CVE example:**
|
||||
|
||||
```python
|
||||
resolver = ConflictResolver()
|
||||
|
||||
# Register source credibility scores so CREDIBILITY_WEIGHTED can use them
|
||||
resolver.source_tracker.set_source_credibility("nvd", 0.98)
|
||||
resolver.source_tracker.set_source_credibility("commercial_feed", 0.91)
|
||||
resolver.source_tracker.set_source_credibility("vendor_paloalto", 0.87)
|
||||
|
||||
# For this CVE, NVD is the most authoritative source on scoring.
|
||||
# CREDIBILITY_WEIGHTED will use the credibility_score field on each source record
|
||||
# CREDIBILITY_WEIGHTED uses the registered source credibility
|
||||
# to weight the vote — NVD at 0.98 will dominate over the commercial feed at 0.91.
|
||||
resolver.set_resolution_rule(
|
||||
"cve-2024-3400",
|
||||
@@ -108,9 +271,9 @@ resolver.set_resolution_rule(
|
||||
|
||||
You can set rules before or after detection — the resolver applies them lazily when `resolve_conflicts()` is called.
|
||||
|
||||
## Resolving the batch
|
||||
## Resolving the Batch
|
||||
|
||||
Pass all detected conflicts to `resolve_conflicts()`. For each conflict, the resolver looks up whether a property-specific rule is set for that entity and property combination. If one is found, it applies that strategy. If none is set, it falls back to the default strategy (voting, unless you override it in the constructor).
|
||||
Pass all detected conflicts to `resolve_conflicts()`. For each conflict, the resolver looks up whether a rule is registered for that entity-property combination. If one is found, it applies that strategy. If none is set, it falls back to the default strategy (voting, unless you override it in the constructor).
|
||||
|
||||
```python
|
||||
all_conflicts = score_conflicts + exploit_conflicts
|
||||
@@ -132,7 +295,7 @@ for r in results:
|
||||
[RESOLVED] cve-2024-3400_cvss_score_conflict
|
||||
Resolved value : 10.0
|
||||
Strategy used : credibility_weighted
|
||||
Confidence : 72%
|
||||
Confidence : 36%
|
||||
Sources used : ['nvd', 'commercial_feed', 'vendor_paloalto']
|
||||
Notes : Resolved by credibility-weighted voting (weight: 0.98)
|
||||
|
||||
@@ -146,7 +309,7 @@ for r in results:
|
||||
|
||||
NVD wins the CVSS score — its credibility weight (0.98) edges out the commercial feed (0.91) and the vendor (0.87), so 10.0 becomes the canonical score. The exploitation status resolves to `in_wild` — the commercial feed and vendor advisory are both more recent than NVD's initial triage, and both report active exploitation.
|
||||
|
||||
## Handling conflicts that need human judgment
|
||||
## Handling Conflicts That Need Human Judgment
|
||||
|
||||
Not every conflict can be auto-resolved. A disagreement about the legal classification of a financial instrument, or about a patient's current medication list, is too consequential to resolve by algorithm. Flag these for review without blocking the rest of the batch:
|
||||
|
||||
@@ -156,14 +319,11 @@ from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionSt
|
||||
# Drug trial data: efficacy agreed, primary endpoint disputed
|
||||
trial_records = [
|
||||
{"id": "dapagliflozin", "source": "declare_timi58",
|
||||
"primary_endpoint": "MACE", "hba1c_reduction_pct": 0.54,
|
||||
"credibility_score": 0.92},
|
||||
"primary_endpoint": "MACE", "hba1c_reduction_pct": 0.54},
|
||||
{"id": "dapagliflozin", "source": "dapa_hf",
|
||||
"primary_endpoint": "HF_hospitalization", "hba1c_reduction_pct": 0.48,
|
||||
"credibility_score": 0.95},
|
||||
"primary_endpoint": "HF_hospitalization", "hba1c_reduction_pct": 0.48},
|
||||
{"id": "dapagliflozin", "source": "meta_analysis",
|
||||
"primary_endpoint": "HbA1c_reduction", "hba1c_reduction_pct": 0.52,
|
||||
"credibility_score": 0.88},
|
||||
"primary_endpoint": "HbA1c_reduction", "hba1c_reduction_pct": 0.52},
|
||||
]
|
||||
|
||||
detector = ConflictDetector()
|
||||
@@ -172,6 +332,11 @@ endpoint_conflicts = detector.detect_value_conflicts(trial_records, "primary_en
|
||||
|
||||
resolver = ConflictResolver()
|
||||
|
||||
# Register source credibility scores
|
||||
resolver.source_tracker.set_source_credibility("declare_timi58", 0.92)
|
||||
resolver.source_tracker.set_source_credibility("dapa_hf", 0.95)
|
||||
resolver.source_tracker.set_source_credibility("meta_analysis", 0.88)
|
||||
|
||||
# Efficacy: credibility-weighted across trials — the meta-analysis (0.88) and
|
||||
# the two RCTs (0.92, 0.95) will produce a weighted resolution.
|
||||
resolver.set_resolution_rule(
|
||||
@@ -214,7 +379,38 @@ Expert review : 1 # primary_endpoint — EXPERT_REVIEW means resolved=False
|
||||
|
||||
`EXPERT_REVIEW` sets `resolved=False` on the result. The conflict stays in the graph unresolved, the metadata field carries `requires_expert_review: True`, and the review queue JSON gives your clinical team exactly what they need to make the call.
|
||||
|
||||
## Reviewing the full audit trail
|
||||
## Persisting Resolved Values
|
||||
|
||||
`resolve_conflicts()` returns `ResolutionResult` objects — it does not automatically write resolved values back to your graph or entity store. That step is yours to implement using whatever storage layer your pipeline uses.
|
||||
|
||||
The most direct approach is to pair each `ResolutionResult` with its original `Conflict` object — the two lists are returned in the same order — and write the winning value onto your canonical entity:
|
||||
|
||||
```python
|
||||
# canonical_entity is your authoritative record — a dict, graph node, database row, etc.
|
||||
canonical_entity = {"id": "cve-2024-3400", "cvss_score": None, "exploit_status": None}
|
||||
|
||||
for conflict, result in zip(all_conflicts, results):
|
||||
if result.resolved:
|
||||
canonical_entity[conflict.property_name] = result.resolved_value
|
||||
# Log provenance: record which source this value came from
|
||||
print(f" {conflict.property_name} = {result.resolved_value} "
|
||||
f"(from {result.sources_used}, confidence {result.confidence:.0%})")
|
||||
|
||||
# Persist canonical_entity to your graph store, database, or downstream system.
|
||||
```
|
||||
|
||||
```text
|
||||
cvss_score = 10.0 (from ['nvd', 'commercial_feed', 'vendor_paloalto'], confidence 36%)
|
||||
exploit_status = in_wild (from ['commercial_feed'], confidence 80%)
|
||||
```
|
||||
|
||||
A few things to keep in mind:
|
||||
|
||||
- **Conflicts with `resolved=False`** — flagged for expert or manual review — should not be written to the canonical record until a human has made the call. Keep them in the review queue.
|
||||
- **Confidence is a signal, not a guarantee.** A 72% confidence score means the resolver had reasonable but not unanimous evidence for its decision. Treat low-confidence results with additional scrutiny before writing them to production.
|
||||
- **Track provenance.** `result.sources_used` tells you which source's value won. Store this alongside the canonical value if your compliance requirements demand a full evidence chain.
|
||||
|
||||
## Reviewing the Full Audit Trail
|
||||
|
||||
After a resolution run, `get_resolution_history()` returns every decision made since the resolver was instantiated. This is your compliance log:
|
||||
|
||||
@@ -238,14 +434,14 @@ report = detector.get_conflict_report()
|
||||
print(f"Total conflicts detected : {report['total_conflicts']}")
|
||||
print(f"By type : {report['by_type']}")
|
||||
print(f"By severity : {report['by_severity']}")
|
||||
# Total conflicts detected : 2
|
||||
# By type : {'value_conflict': 2}
|
||||
# By severity : {'medium': 2}
|
||||
# Total conflicts detected : 6
|
||||
# By type : {'value_conflict': 6}
|
||||
# By severity : {'medium': 6}
|
||||
```
|
||||
|
||||
The report aggregates every conflict the detector has seen across its lifetime — useful for pipeline monitoring and for identifying which entity types or data sources generate the most disagreements.
|
||||
|
||||
## Detecting relationship conflicts
|
||||
## Detecting Relationship Conflicts
|
||||
|
||||
Value conflicts live on properties. Relationship conflicts live on edges — two sources asserting contradictory connections between the same node pair:
|
||||
|
||||
@@ -267,7 +463,7 @@ for c in rel_conflicts:
|
||||
|
||||
Relationship conflicts typically require expert review rather than voting, because conflicting edge types often reflect genuinely different intelligence assessments rather than data entry errors.
|
||||
|
||||
## Domain examples
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
|
||||
@@ -282,11 +478,11 @@ from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionSt
|
||||
|
||||
actor_profiles = [
|
||||
{"id": "apt29", "source": "mandiant", "nation_state": "Russia",
|
||||
"first_seen": "2008", "credibility_score": 0.95},
|
||||
"first_seen": "2008"},
|
||||
{"id": "apt29", "source": "crowdstrike", "nation_state": "Russia",
|
||||
"first_seen": "2009", "credibility_score": 0.92},
|
||||
"first_seen": "2009"},
|
||||
{"id": "apt29", "source": "oss_blog", "nation_state": "China", # wrong
|
||||
"first_seen": "2015", "credibility_score": 0.30},
|
||||
"first_seen": "2015"},
|
||||
]
|
||||
|
||||
detector = ConflictDetector()
|
||||
@@ -294,14 +490,18 @@ nation_conflicts = detector.detect_value_conflicts(actor_profiles, "nation_s
|
||||
first_seen_conflicts = detector.detect_value_conflicts(actor_profiles, "first_seen")
|
||||
|
||||
resolver = ConflictResolver()
|
||||
resolver.source_tracker.set_source_credibility("mandiant", 0.95)
|
||||
resolver.source_tracker.set_source_credibility("crowdstrike", 0.92)
|
||||
resolver.source_tracker.set_source_credibility("oss_blog", 0.30)
|
||||
|
||||
resolver.set_resolution_rule("apt29", "nation_state", ResolutionStrategy.CREDIBILITY_WEIGHTED)
|
||||
resolver.set_resolution_rule("apt29", "first_seen", ResolutionStrategy.CREDIBILITY_WEIGHTED)
|
||||
|
||||
results = resolver.resolve_conflicts(nation_conflicts + first_seen_conflicts)
|
||||
for r in results:
|
||||
print(f"{r.conflict_id}: {r.resolved_value!r} [{r.confidence:.0%} confidence]")
|
||||
# apt29_nation_state_conflict: 'Russia' [83% confidence]
|
||||
# apt29_first_seen_conflict: '2008' [73% confidence]
|
||||
# apt29_nation_state_conflict: 'Russia' [86% confidence]
|
||||
# apt29_first_seen_conflict: '2008' [44% confidence]
|
||||
# The blog's China attribution (weight 0.30) loses to Mandiant+CrowdStrike (0.95+0.92).
|
||||
|
||||
history = resolver.get_resolution_history()
|
||||
@@ -321,14 +521,11 @@ from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionSt
|
||||
|
||||
cve_records = [
|
||||
{"id": "cve-2024-3400", "source": "nvd",
|
||||
"cvss_score": 10.0, "vector": "AV:N/AC:L/PR:N/UI:N/S:C/C:H/I:H/A:H",
|
||||
"credibility_score": 0.98},
|
||||
"cvss_score": 10.0, "vector": "AV:N/AC:L/PR:N/UI:N/S:C/C:H/I:H/A:H"},
|
||||
{"id": "cve-2024-3400", "source": "mitre",
|
||||
"cvss_score": 9.8, "vector": "AV:N/AC:L/PR:N/UI:N/S:U/C:H/I:H/A:H",
|
||||
"credibility_score": 0.96},
|
||||
"cvss_score": 9.8, "vector": "AV:N/AC:L/PR:N/UI:N/S:U/C:H/I:H/A:H"},
|
||||
{"id": "cve-2024-3400", "source": "paloalto",
|
||||
"cvss_score": 9.5, "vector": "AV:N/AC:H/PR:N/UI:N/S:C/C:H/I:H/A:H",
|
||||
"credibility_score": 0.90},
|
||||
"cvss_score": 9.5, "vector": "AV:N/AC:H/PR:N/UI:N/S:C/C:H/I:H/A:H"},
|
||||
]
|
||||
|
||||
detector = ConflictDetector()
|
||||
@@ -336,6 +533,10 @@ score_conflicts = detector.detect_value_conflicts(cve_records, "cvss_score")
|
||||
vector_conflicts = detector.detect_value_conflicts(cve_records, "vector")
|
||||
|
||||
resolver = ConflictResolver()
|
||||
resolver.source_tracker.set_source_credibility("nvd", 0.98)
|
||||
resolver.source_tracker.set_source_credibility("mitre", 0.96)
|
||||
resolver.source_tracker.set_source_credibility("paloalto", 0.90)
|
||||
|
||||
resolver.set_resolution_rule(
|
||||
"cve-2024-3400", "cvss_score", ResolutionStrategy.CREDIBILITY_WEIGHTED
|
||||
)
|
||||
@@ -348,8 +549,8 @@ for r in results:
|
||||
if r.resolved:
|
||||
print(f"Canonical {r.conflict_id.split('_')[2]}: {r.resolved_value} "
|
||||
f"({r.confidence:.0%} confidence)")
|
||||
# Canonical cvss_score: 10.0 (72% confidence) — NVD wins
|
||||
# Canonical vector: AV:N/AC:L/PR:N/UI:N/S:C/C:H/I:H/A:H (54% confidence)
|
||||
# Canonical cvss_score: 10.0 (35% confidence) — NVD wins
|
||||
# Canonical vector: AV:N/AC:L/PR:N/UI:N/S:C/C:H/I:H/A:H (35% confidence)
|
||||
```
|
||||
|
||||
</Tab>
|
||||
@@ -365,14 +566,11 @@ from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionSt
|
||||
|
||||
drug_records = [
|
||||
{"id": "dapagliflozin", "source": "declare_timi58",
|
||||
"hba1c_reduction_pct": 0.54, "primary_endpoint": "MACE",
|
||||
"credibility_score": 0.92},
|
||||
"hba1c_reduction_pct": 0.54, "primary_endpoint": "MACE"},
|
||||
{"id": "dapagliflozin", "source": "dapa_hf",
|
||||
"hba1c_reduction_pct": 0.48, "primary_endpoint": "HF_hospitalization",
|
||||
"credibility_score": 0.95},
|
||||
"hba1c_reduction_pct": 0.48, "primary_endpoint": "HF_hospitalization"},
|
||||
{"id": "dapagliflozin", "source": "meta_analysis",
|
||||
"hba1c_reduction_pct": 0.52, "primary_endpoint": "HbA1c_reduction",
|
||||
"credibility_score": 0.88},
|
||||
"hba1c_reduction_pct": 0.52, "primary_endpoint": "HbA1c_reduction"},
|
||||
]
|
||||
|
||||
detector = ConflictDetector()
|
||||
@@ -380,6 +578,10 @@ efficacy_conflicts = detector.detect_value_conflicts(drug_records, "hba1c_reduct
|
||||
endpoint_conflicts = detector.detect_value_conflicts(drug_records, "primary_endpoint")
|
||||
|
||||
resolver = ConflictResolver()
|
||||
resolver.source_tracker.set_source_credibility("declare_timi58", 0.92)
|
||||
resolver.source_tracker.set_source_credibility("dapa_hf", 0.95)
|
||||
resolver.source_tracker.set_source_credibility("meta_analysis", 0.88)
|
||||
|
||||
resolver.set_resolution_rule(
|
||||
"dapagliflozin", "hba1c_reduction_pct", ResolutionStrategy.CREDIBILITY_WEIGHTED
|
||||
)
|
||||
@@ -395,7 +597,7 @@ review = [r for r in results if not r.resolved]
|
||||
print(f"Auto-resolved : {len(auto)}")
|
||||
for r in auto:
|
||||
print(f" {r.conflict_id}: {r.resolved_value} [{r.confidence:.0%}]")
|
||||
# dapagliflozin_hba1c_reduction_pct_conflict: 0.48 [38%]
|
||||
# dapagliflozin_hba1c_reduction_pct_conflict: 0.48 [35%]
|
||||
|
||||
print(f"Expert queue : {len(review)}")
|
||||
for r in review:
|
||||
@@ -416,14 +618,11 @@ from semantica.conflicts import ConflictDetector, ConflictResolver, ResolutionSt
|
||||
|
||||
client_records = [
|
||||
{"id": "corp-acme-uk", "source": "crm",
|
||||
"legal_name": "ACME UK Ltd", "sic_code": "7372",
|
||||
"credibility_score": 0.75},
|
||||
"legal_name": "ACME UK Ltd", "sic_code": "7372"},
|
||||
{"id": "corp-acme-uk", "source": "lei_registry",
|
||||
"legal_name": "ACME United Kingdom Limited", "sic_code": "7371",
|
||||
"credibility_score": 0.99}, # LEI registry is authoritative
|
||||
"legal_name": "ACME United Kingdom Limited", "sic_code": "7371"},
|
||||
{"id": "corp-acme-uk", "source": "credit_bureau",
|
||||
"legal_name": "ACME UK Ltd", "sic_code": "7372",
|
||||
"credibility_score": 0.85},
|
||||
"legal_name": "ACME UK Ltd", "sic_code": "7372"},
|
||||
]
|
||||
|
||||
detector = ConflictDetector()
|
||||
@@ -431,6 +630,10 @@ name_conflicts = detector.detect_value_conflicts(client_records, "legal_name")
|
||||
sic_conflicts = detector.detect_value_conflicts(client_records, "sic_code")
|
||||
|
||||
resolver = ConflictResolver()
|
||||
resolver.source_tracker.set_source_credibility("lei_registry", 0.99)
|
||||
resolver.source_tracker.set_source_credibility("credit_bureau", 0.50)
|
||||
resolver.source_tracker.set_source_credibility("crm", 0.40)
|
||||
|
||||
resolver.set_resolution_rule(
|
||||
"corp-acme-uk", "legal_name", ResolutionStrategy.CREDIBILITY_WEIGHTED
|
||||
)
|
||||
@@ -440,10 +643,10 @@ resolver.set_resolution_rule(
|
||||
|
||||
results = resolver.resolve_conflicts(name_conflicts + sic_conflicts)
|
||||
for r in results:
|
||||
print(f"Canonical {r.conflict_id.split('_')[2]}: {r.resolved_value!r} "
|
||||
print(f"Canonical {r.conflict_id.split('_')[1]}: {r.resolved_value!r} "
|
||||
f"[{r.confidence:.0%}]")
|
||||
# Canonical legal_name: 'ACME United Kingdom Limited' [53%] — LEI registry wins
|
||||
# Canonical sic_code: '7371' [53%] — LEI registry wins
|
||||
# Canonical legal_name: 'ACME United Kingdom Limited' [52%] — LEI registry wins
|
||||
# Canonical sic_code: '7371' [52%] — LEI registry wins
|
||||
|
||||
# Aggregate conflict statistics for the compliance report
|
||||
report = detector.get_conflict_report()
|
||||
@@ -456,7 +659,7 @@ print(f" By severity : {report['by_severity']}")
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Resolution strategies at a glance
|
||||
## Resolution Strategies at a Glance
|
||||
|
||||
| Strategy | How it decides | Best when |
|
||||
| :--- | :--- | :--- |
|
||||
@@ -468,6 +671,29 @@ print(f" By severity : {report['by_severity']}")
|
||||
| `MANUAL_REVIEW` | Flags the conflict; `resolved=False` | Low-volume, high-stakes decisions |
|
||||
| `EXPERT_REVIEW` | Flags for domain expert queue; `resolved=False` | Scientific or legal disambiguation required |
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Running conflict resolution before deduplication**
|
||||
If duplicate nodes for the same real-world entity still exist, `ConflictDetector` treats each duplicate as a separate entity disagreeing with the others — producing spurious conflicts that should never have existed. Always run deduplication first.
|
||||
|
||||
**Forgetting to persist resolved values**
|
||||
`resolve_conflicts()` returns `ResolutionResult` objects; it does not write them anywhere. Inspecting the results and moving on without updating your canonical entity means nothing has actually changed in your data. See [Persisting resolved values](#persisting-resolved-values).
|
||||
|
||||
**Scanning properties one at a time across a large entity set**
|
||||
Calling `detect_value_conflicts()` for every property in a manual loop produces redundant passes over your data. Use `detect_entity_conflicts()` instead — it handles all properties in a single call and is the recommended starting point for bulk detection.
|
||||
|
||||
**Misunderstanding credibility scores**
|
||||
Credibility scores are weights you assign based on your prior knowledge of source reliability — not ground truth. A source registered with `set_source_credibility("source", 0.99)` can still be wrong. `CREDIBILITY_WEIGHTED` resolution amplifies your beliefs about source quality; if those beliefs are miscalibrated, the resolutions will be too. Validate scores against known ground truth before relying on them in production.
|
||||
|
||||
**Treating resolved values as guaranteed truth**
|
||||
A resolved value is the most defensible answer given your sources and strategy — not necessarily the correct one. Low confidence scores and `EXPERT_REVIEW` flags are signals to scrutinize results before writing them to a canonical record or downstream system.
|
||||
|
||||
**Using conflict resolution when a single authoritative source already exists**
|
||||
If one system is always correct for a given property, read from it directly. Layering conflict resolution over a single source adds complexity, introduces unnecessary doubt, and produces an audit trail that adds no real information.
|
||||
|
||||
**Registering rules in a loop to apply one strategy uniformly**
|
||||
Calling `set_resolution_rule()` for every entity-property pair just to apply the same strategy to all of them creates O(N) setup for no benefit. Pass `strategy=` directly to `resolve_conflicts()` when one strategy covers the whole batch.
|
||||
|
||||
## Related Guides
|
||||
|
||||
- [Deduplication](deduplication) — remove duplicate nodes before running conflict detection
|
||||
|
||||
@@ -6,8 +6,61 @@ icon: "scale-balanced"
|
||||
|
||||
`AgentContext.record_decision()` stores every AI decision as a node in the knowledge graph, linked by causal edges to the decisions that preceded it and the outcomes that followed. Use it to build an auditable reasoning trail — one that lets you reconstruct, six months later, exactly which classification caused which escalation, and which policy was checked before it was recorded.
|
||||
|
||||
## What Is Decision Intelligence?
|
||||
|
||||
Decision Intelligence records and analyzes an agent's own decisions as structured data that can be queried, analyzed, and reused. Instead of decisions disappearing after execution, they become persistent graph nodes with searchable metadata, reasoning chains, and causal relationships.
|
||||
|
||||
**Decision Intelligence records decisions** by capturing the scenario, reasoning, outcome, confidence, and decision maker for each choice the agent makes. These decisions become queryable nodes in your knowledge graph.
|
||||
|
||||
**Decisions become graph nodes** that can be linked causally (Decision A caused Decision B), searched by similarity (find decisions like this scenario), and analyzed statistically (confidence trends, common outcomes).
|
||||
|
||||
**The goal is auditability, explainability, precedent search, and causal tracing.** You can trace why decisions were made, find similar past decisions for consistency, and understand the full causal chain from initial detection to final action.
|
||||
|
||||
**Decision Intelligence vs. Agent Memory:** Agent Memory stores external knowledge (documents, facts, observations). Decision Intelligence stores internal decisions (classifications, approvals, actions the agent itself made).
|
||||
|
||||
**Decision Intelligence vs. Reasoning:** Reasoning derives new facts from existing data using logical rules. Decision Intelligence records the choices and judgments the agent made during problem-solving.
|
||||
|
||||
**Decision Intelligence vs. Graph Analytics:** Graph Analytics analyzes the structural properties of your knowledge graph. Decision Intelligence focuses specifically on the decision-making process and its audit trail.
|
||||
|
||||
## Why Use Decision Intelligence?
|
||||
|
||||
**Auditable AI actions.** Every decision is recorded with reasoning, confidence, and timestamp, creating a complete audit trail for AI behavior in production systems.
|
||||
|
||||
**Explainability.** When stakeholders ask "why did the system do X?", you can trace the exact decision chain that led to that action, including intermediate reasoning steps.
|
||||
|
||||
**Precedent reuse.** Before making new decisions, agents can search for similar past scenarios and their outcomes, promoting consistency and learning from previous experience.
|
||||
|
||||
**Causal analysis.** Understand how early decisions cascade into later outcomes by following causal relationships between linked decision nodes.
|
||||
|
||||
**Governance and compliance.** Policy engines can gate decisions against compliance rules, and all policy applications are recorded for regulatory audit.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use Decision Intelligence when:**
|
||||
- Building autonomous agents that make consequential choices
|
||||
- Implementing decision workflows requiring audit trails
|
||||
- Operating under compliance requirements (financial services, healthcare, defense)
|
||||
- Building approval systems with multiple decision points
|
||||
- Working in risk-sensitive environments where decisions must be explainable
|
||||
|
||||
**Do not use when:**
|
||||
- Building stateless chatbots that only retrieve information
|
||||
- Implementing simple RAG systems without decision-making
|
||||
- Creating read-only information retrieval applications
|
||||
- Building applications that never make actionable decisions requiring audit trails
|
||||
|
||||
## API Architecture Overview
|
||||
|
||||
Decision Intelligence coordinates three main components:
|
||||
|
||||
**AgentContext** serves as the high-level orchestration layer. It provides `record_decision()`, `find_precedents()`, and causal chain methods while managing the underlying storage and retrieval systems.
|
||||
|
||||
**PolicyEngine** handles policy evaluation and compliance checking. It stores policy rules as graph nodes and validates decisions against those rules before they're recorded.
|
||||
|
||||
**DecisionRecorder** specializes in recording structured decision data, managing approval chains, and handling policy exceptions when decisions need to bypass normal rules.
|
||||
|
||||
<Info>
|
||||
Decision tracking requires both a `VectorStore` (for embedding-based precedent search) and a `ContextGraph` (for causal graph storage). Set `decision_tracking=True` on `AgentContext` — omitting either component raises `RuntimeError` at call time.
|
||||
Decision tracking requires both a `VectorStore` (for embedding-based precedent search) and a `ContextGraph` (for causal graph storage). Set `decision_tracking=True` on `AgentContext` — omitting `ContextGraph` raises a `RuntimeError` at call time. `VectorStore` is required by `AgentContext` itself: leaving the argument out raises a `TypeError` from Python's argument binding, while passing `vector_store=None` raises a `ValueError` during initialization.
|
||||
</Info>
|
||||
|
||||
## Recording the First Decision
|
||||
@@ -41,6 +94,8 @@ print("Decision recorded:", classification_id)
|
||||
# → "Decision recorded: dec_a3f2b1c4-..."
|
||||
```
|
||||
|
||||
The `decision_maker` field identifies the component, workflow, agent, or system that produced this decision. Use consistent identifiers like `"cti_pipeline_v2"`, `"analyst_chen"`, or `"risk_model_v3"` to enable filtering and analysis by decision source.
|
||||
|
||||
The `Decision` dataclass that backs this node has the following fields — these are what get stored and searched:
|
||||
|
||||
```python
|
||||
@@ -559,7 +614,7 @@ context.save("agent_state/")
|
||||
|
||||
# Start of next session
|
||||
context = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="decisions.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=ContextGraph(),
|
||||
decision_tracking=True,
|
||||
)
|
||||
@@ -569,6 +624,18 @@ context.load("agent_state/")
|
||||
results = context.find_precedents("APT29 infrastructure attribution", limit=5)
|
||||
```
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Recording decisions without linking causal relationships.** Isolated decision nodes provide less insight than connected decision chains. Use `add_causal_relationship()` to link related decisions and enable causal tracing.
|
||||
|
||||
**Creating isolated decision nodes.** Decisions gain value when connected to entities, other decisions, or outcomes in your graph. Link decisions to relevant entities using the `entities` parameter.
|
||||
|
||||
**Recording too many low-value decisions.** Not every minor choice needs permanent recording. Focus on consequential decisions that affect outcomes, require audit trails, or benefit from precedent search.
|
||||
|
||||
**Treating precedent similarity as proof.** High similarity scores indicate related scenarios, not identical situations. Use precedents as guidance while considering the specific context of each new decision.
|
||||
|
||||
**Using Decision Intelligence when simple retrieval is sufficient.** If your system only retrieves information without making actionable choices, traditional search or Agent Memory may be more appropriate than decision tracking.
|
||||
|
||||
## Related Guides
|
||||
|
||||
- [Context Graphs](context-graphs) — how `ContextGraph` stores decision nodes and causal edges
|
||||
|
||||
+190
-25
@@ -3,6 +3,96 @@ title: "Deduplication & Entity Merging"
|
||||
description: "Detect duplicate entities using multi-factor similarity, merge them with configurable strategies, and keep your knowledge graph clean at scale."
|
||||
---
|
||||
|
||||
## What Is Deduplication?
|
||||
|
||||
Deduplication is the process of identifying entities that refer to the same real-world object but appear as separate records in your data, then merging them into a single canonical representation. This process resolves aliases, spelling variations, and formatting differences that occur when data comes from multiple sources.
|
||||
|
||||
**Key deduplication concepts:**
|
||||
|
||||
**Canonical entities** are the single, authoritative representation of a real-world object after merging all duplicate records. The canonical entity becomes the node that all relationships point to in your knowledge graph.
|
||||
|
||||
**Aliases** are alternative names or identifiers for the same entity. For example, "APT29", "Cozy Bear", and "Midnight Blizzard" are all aliases for the same threat actor.
|
||||
|
||||
**Entity resolution** is the broader process of determining when different records refer to the same entity, including the similarity calculation, duplicate detection, and merging steps.
|
||||
|
||||
**Similarity algorithms:**
|
||||
- **Jaro-Winkler** measures string similarity with higher scores for shared prefixes, ideal for names with common beginnings
|
||||
- **Levenshtein** distance counts character edits needed to transform one string into another, good for catching typos and variations
|
||||
|
||||
**Clustering** groups related duplicates together using algorithms like Union-Find, ensuring that if A matches B and B matches C, all three are grouped together even if A and C don't directly match.
|
||||
|
||||
## Why Use Deduplication?
|
||||
|
||||
**Data quality and consistency.** Eliminate duplicate nodes that fragment relationships and create inconsistent query results across different names for the same entity.
|
||||
|
||||
**Accurate analytics and metrics.** Get correct counts, centrality measures, and relationship analysis when entities aren't artificially split across multiple nodes due to naming variations.
|
||||
|
||||
**Relationship consolidation.** Merge scattered relationships onto single canonical entities, enabling complete analysis of connections and patterns that would be missed with fragmented data.
|
||||
|
||||
**Source integration.** Seamlessly combine data from multiple feeds, systems, and databases where the same entities appear under different identifiers and naming conventions.
|
||||
|
||||
**Graph efficiency.** Reduce graph size and improve query performance by eliminating redundant nodes while preserving all information through proper merging strategies.
|
||||
|
||||
**Provenance preservation.** Maintain complete audit trails showing which source contributed each piece of information to the final canonical entity.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use deduplication for:**
|
||||
- Multi-source data integration where entities appear under different names or identifiers
|
||||
- Entity types prone to aliases and variations (organizations, people, products, geographic locations)
|
||||
- Knowledge graphs where relationship accuracy depends on entity consolidation
|
||||
- Data quality workflows requiring canonical entity management
|
||||
- Analytics requiring accurate entity counts and relationship metrics
|
||||
- Scenarios where the same real-world objects appear across multiple systems or databases
|
||||
|
||||
**Do NOT use deduplication for:**
|
||||
- Single-source data with consistent entity identifiers and naming conventions
|
||||
- High-throughput streaming scenarios where deduplication latency is unacceptable
|
||||
- Data with reliable primary keys where duplicates are impossible by design
|
||||
- Cases where entity variations should be preserved as separate nodes (different product versions, time-based entity states)
|
||||
- Simple exact-match scenarios where basic database constraints handle uniqueness
|
||||
|
||||
**Be cautious with:**
|
||||
- Large datasets where O(n²) pairwise comparison becomes computationally expensive
|
||||
- Fuzzy matching when deterministic primary keys (LEI, CVE-ID, ISIN) are available
|
||||
- Very low similarity thresholds that may merge genuinely different entities
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
The deduplication workflow follows a systematic process from detection through merging:
|
||||
|
||||
**1. Detect** → Use `detect_duplicates()` or `DuplicateDetector` to identify potential matches using multi-factor similarity scoring
|
||||
|
||||
**2. Group** → Apply clustering algorithms to collect transitively related duplicates into groups (A matches B, B matches C → group A,B,C)
|
||||
|
||||
**3. Select Canonical** → Choose representative entity for each group based on completeness, source authority, or confidence scores
|
||||
|
||||
**4. Merge** → Combine duplicate entities using strategies like `keep_most_complete` or `merge_all` while preserving provenance
|
||||
|
||||
**5. Validate** → Review merge results and adjust thresholds or strategies based on precision/recall analysis
|
||||
|
||||
**6. Update Graph** → Replace duplicate nodes with canonical entities and transfer all relationships
|
||||
|
||||
This pipeline transforms fragmented multi-source data into clean, consolidated knowledge graphs ready for analytics and reasoning.
|
||||
|
||||
## API Patterns: Functional vs Class-Based
|
||||
|
||||
Semantica provides both simple functional wrappers and comprehensive class APIs for different use cases:
|
||||
|
||||
**Functional wrappers for simple workflows:**
|
||||
- `detect_duplicates()` — one-shot duplicate detection with minimal configuration
|
||||
- `calculate_similarity()` — compare two entities with detailed similarity breakdown
|
||||
- `merge_entities()` — convenience wrapper around merge_duplicates() for quick merging
|
||||
|
||||
**Class APIs for complex workflows:**
|
||||
- `DuplicateDetector` — configurable duplicate detection with clustering, incremental processing, and advanced similarity options
|
||||
- `EntityMerger` — sophisticated merging with multiple strategies, provenance tracking, and merge history
|
||||
|
||||
**Usage guidelines:**
|
||||
- Use `merge_duplicates()` when you have a raw collection of entities and need automatic duplicate detection
|
||||
- Use `merge_entity_group()` when you already know which entities are duplicates and just need to merge a pre-determined group
|
||||
- Don't mix functional wrappers with class APIs in the same workflow—choose one approach and stick with it
|
||||
|
||||
The deduplication module detects duplicate entities across multi-source knowledge graphs using six complementary similarity algorithms — exact match, Levenshtein, Jaro-Winkler, cosine, property comparison, and vector embedding — then merges them into a single canonical entity while preserving full provenance. Use it to collapse alias clusters (e.g. "APT29", "Cozy Bear", "Midnight Blizzard") before running graph analytics or conflict resolution.
|
||||
|
||||
<Info>
|
||||
@@ -11,7 +101,9 @@ Run deduplication after ingestion and before conflict resolution. Deduplication
|
||||
|
||||
## Finding your duplicates: the first scan
|
||||
|
||||
Start with `detect_duplicates()`. Point it at your threat actor entities and let the pairwise algorithm compare every pair. For a dataset of a few thousand nodes this runs in seconds — the O(n²) cost only matters above ten thousand entities.
|
||||
Start with `detect_duplicates()` for straightforward duplicate detection on smaller datasets. Point it at your entities and let the pairwise algorithm compare every pair using multiple similarity signals.
|
||||
|
||||
**Scaling consideration:** For datasets of a few thousand nodes, this runs in seconds. The O(n²) pairwise comparison cost only becomes problematic above ten thousand entities—for larger sets, see the clustering section below.
|
||||
|
||||
```python
|
||||
from semantica.deduplication import detect_duplicates
|
||||
@@ -64,11 +156,11 @@ for c in candidates:
|
||||
signals: ['property'] # alias "APT29" in Midnight Blizzard record
|
||||
```
|
||||
|
||||
The scores tell a clear story. "APT29" and "APT-29" score 0.89 — the hyphen is the only difference, pure edit-distance signal. "Cozy Bear" and "The Dukes" score lower (0.61) because the names are completely dissimilar, but the property signal fires because both records carry `"APT29"` in their aliases list. "APT28" never appears in the results because it shares only the country field — not enough to cross the 0.6 threshold.
|
||||
The scores tell a clear story. "APT29" and "APT-29" score 0.89 — the hyphen is the only difference, producing strong string similarity signals. "Cozy Bear" and "The Dukes" score lower (0.61) because the names are completely dissimilar, but the property signal fires because both records carry `"APT29"` in their aliases list. "APT28" never appears in the results because it shares only the country field — not enough to cross the 0.6 threshold.
|
||||
|
||||
## Understanding the candidate object
|
||||
|
||||
Each `DuplicateCandidate` carries the two entities, their scores, and a `reasons` list explaining which signals fired. This is your audit trail for the detection decision:
|
||||
Each `DuplicateCandidate` carries the two entities, their similarity scores, and a detailed breakdown of which similarity algorithms contributed to the match. This provides full transparency for audit and threshold tuning:
|
||||
|
||||
```python
|
||||
from semantica.deduplication import calculate_similarity
|
||||
@@ -98,11 +190,11 @@ Components :
|
||||
embedding 0.78 # semantic vectors land in the same cluster
|
||||
```
|
||||
|
||||
The property component (0.94) is doing most of the work here. "Cozy Bear"'s record carries `aliases: ["APT29"]`, which creates an almost-definitive signal. When you see a pattern like this — a weak name score but a strong property score — you're looking at a real alias relationship, not a false positive.
|
||||
The property component (0.94) is doing most of the work here. "Cozy Bear"'s record carries `aliases: ["APT29"]`, which creates an almost-definitive signal that these entities refer to the same threat actor. When you see a pattern like this — weak name similarity but strong property matching — you're typically looking at a genuine alias relationship rather than a false positive.
|
||||
|
||||
## Grouping duplicates before merging
|
||||
|
||||
For a small dataset you can merge pairs directly. For a larger graph where the same entity might appear under six different names across twelve feeds, use `detect_duplicate_groups()`. It runs Union-Find clustering to collect all aliases of the same underlying entity into a single group, regardless of whether every pair individually crosses the threshold:
|
||||
For small datasets, you can merge candidate pairs directly. For larger graphs where the same entity might appear under six different names across twelve feeds, use duplicate grouping with Union-Find clustering. This ensures that if A matches B and B matches C, all three entities are grouped together even if A and C don't directly meet the similarity threshold:
|
||||
|
||||
```python
|
||||
from semantica.deduplication import DuplicateDetector, EntityMerger
|
||||
@@ -132,11 +224,11 @@ Found 2 duplicate groups:
|
||||
Representative: 'APT28'
|
||||
```
|
||||
|
||||
The group result shows the problem clearly: five separate nodes that should be one. The `representative` field is the entity the merger will use as the base — the one with the most filled properties, in this case "APT29" from the MISP feed which carries the fullest attribute set.
|
||||
The group result shows the consolidation clearly: five separate nodes that should be one canonical entity. The `representative` field identifies the entity the merger will use as the base — typically the one with the most complete attribute set, in this case "APT29" from the MISP feed.
|
||||
|
||||
## Merging: collapsing the group without losing data
|
||||
|
||||
Now merge. The `keep_most_complete` strategy keeps the entity with the highest property count as the canonical node and fills in any missing fields from the other sources. With `preserve_provenance=True`, the merge operation records which source contributed every field in the merged result:
|
||||
Once you have identified duplicate groups, the merging process consolidates them into canonical entities. The `keep_most_complete` strategy selects the entity with the highest property count as the canonical node and enriches it with any missing fields from the other sources:
|
||||
|
||||
```python
|
||||
merger = EntityMerger(preserve_provenance=True)
|
||||
@@ -145,29 +237,28 @@ for group in groups:
|
||||
if len(group.entities) < 2:
|
||||
continue
|
||||
|
||||
operations = merger.merge_duplicates(group.entities, strategy="keep_most_complete")
|
||||
# merge_entity_group() skips duplicate detection since `group.entities`
|
||||
# is already a confirmed group from detect_duplicate_groups()
|
||||
op = merger.merge_entity_group(group.entities, strategy="keep_most_complete")
|
||||
|
||||
for op in operations:
|
||||
canonical = op.merged_entity
|
||||
source_ids = [e["id"] for e in op.source_entities]
|
||||
print(f"Merged {len(op.source_entities)} entities → canonical: {canonical['name']!r}")
|
||||
print(f" Source IDs retired : {source_ids}")
|
||||
print(f" Merge strategy : {op.merge_result}")
|
||||
print(f" Timestamp : {op.timestamp}")
|
||||
canonical = op.merged_entity
|
||||
source_ids = [e["id"] for e in op.source_entities]
|
||||
print(f"Merged {len(op.source_entities)} entities → canonical: {canonical['name']!r}")
|
||||
print(f" Source IDs retired : {source_ids}")
|
||||
print(f" Merge strategy : {op.merge_result.metadata.get('strategy')}")
|
||||
```
|
||||
|
||||
```text
|
||||
Merged 5 entities → canonical: 'APT29'
|
||||
Source IDs retired : ['ta-nvd-001', 'ta-of-002', 'ta-rf-003', 'ta-sx-004', 'ta-ms-005']
|
||||
Merge strategy : MergeResult.KEPT_MOST_COMPLETE
|
||||
Timestamp : 2026-06-21T09:14:02.443Z
|
||||
Merge strategy : keep_most_complete
|
||||
```
|
||||
|
||||
The five source entities are replaced by one. Every relationship those five nodes carried — to campaigns, malware families, TTPs, infrastructure — now attaches to the canonical "APT29" node. Nothing is lost; the provenance records show exactly which feed contributed which attribute.
|
||||
The five source entities are replaced by one canonical representation. Every relationship those five nodes carried — to campaigns, malware families, TTPs, infrastructure — now attaches to the canonical "APT29" node. The merge operation preserves all information while eliminating redundancy, and the provenance records show exactly which feed contributed each attribute.
|
||||
|
||||
## Reviewing merge history for audit
|
||||
|
||||
After a batch merge, pull the full history to review every decision made:
|
||||
After batch merging operations, you can retrieve the complete history to review every decision made. This audit trail is essential for understanding merge decisions and explaining them to stakeholders:
|
||||
|
||||
```python
|
||||
history = merger.get_merge_history()
|
||||
@@ -175,14 +266,14 @@ history = merger.get_merge_history()
|
||||
print(f"Total merge operations: {len(history)}")
|
||||
for op in history:
|
||||
print(f" {op.merged_entity['name']!r} ← {len(op.source_entities)} sources")
|
||||
print(f" strategy: {op.merge_result}")
|
||||
print(f" strategy: {op.merge_result.metadata.get('strategy')}")
|
||||
```
|
||||
|
||||
This history is what you present when a feed owner asks why their entity was merged into another one. Every decision is recorded.
|
||||
This history provides complete transparency about merge decisions. When a feed owner asks why their entity was merged into another one, you have the documented evidence and reasoning for the decision.
|
||||
|
||||
## Streaming ingestion: incremental deduplication
|
||||
|
||||
When your pipeline is ingesting continuously — new STIX bundles arriving hourly — you don't want to re-run pairwise comparison over the entire graph on every batch. Use `incremental_detect()` to compare only the new entities against the existing set:
|
||||
When your pipeline processes continuous data streams — new threat intelligence arriving hourly — you don't want to re-run pairwise comparison over the entire graph on every batch. Use incremental detection to compare only new entities against the existing canonical set:
|
||||
|
||||
```python
|
||||
# Existing graph entities (already deduplicated)
|
||||
@@ -212,11 +303,13 @@ New duplicates found in this batch: 1
|
||||
score=0.67 # alias field carries "APT29" — property signal fires
|
||||
```
|
||||
|
||||
NOBELIUM goes to the merge queue. Scattered Spider scores below threshold against every existing actor and gets added to the graph as a new node.
|
||||
NOBELIUM gets queued for merging with the existing APT29 canonical entity. Scattered Spider scores below threshold against every existing actor and gets added to the graph as a new, unique node.
|
||||
|
||||
## Scaling to large entity sets
|
||||
|
||||
For graphs above ten thousand nodes, pairwise comparison becomes too slow. Use `build_clusters()` to run vectorized batch comparison, then merge each cluster:
|
||||
For graphs above ten thousand nodes, pairwise comparison becomes computationally expensive due to its O(n²) complexity. Use `build_clusters()` to run more efficient vectorized batch comparison, then merge each resulting cluster:
|
||||
|
||||
**Performance warning:** Always profile your similarity operations on representative data sizes. What works for 1,000 entities may become unacceptably slow at 10,000+ entities without appropriate scaling strategies.
|
||||
|
||||
```python
|
||||
from semantica.deduplication import build_clusters
|
||||
@@ -238,11 +331,83 @@ print(f"Quality metrics : {cluster_result.quality_metrics}")
|
||||
merger = EntityMerger(preserve_provenance=True)
|
||||
for cluster in cluster_result.clusters:
|
||||
if len(cluster.entities) > 1:
|
||||
merger.merge_duplicates(cluster.entities, strategy="keep_most_complete")
|
||||
# Use merge_entity_group() since clustering already determined these are duplicates
|
||||
merger.merge_entity_group(cluster.entities, strategy="keep_most_complete")
|
||||
```
|
||||
|
||||
For even larger sets, switch to `method="hierarchical"` which uses agglomerative bottom-up clustering and scales to hundreds of thousands of entities at the cost of some precision.
|
||||
|
||||
## A Simple Example: Customer Deduplication
|
||||
|
||||
Before exploring domain-specific cases, let's walk through a straightforward customer deduplication scenario. A company's CRM system has accumulated duplicate customer records from web signups, sales team entries, and support tickets:
|
||||
|
||||
```python
|
||||
from semantica.deduplication import detect_duplicates, merge_entities
|
||||
|
||||
customers = [
|
||||
{"id": "cust-001", "name": "John Smith", "email": "john.smith@email.com",
|
||||
"company": "Acme Corp", "source": "web_signup"},
|
||||
{"id": "cust-002", "name": "J. Smith", "email": "john.smith@email.com",
|
||||
"company": "Acme Corporation", "source": "sales_team"},
|
||||
{"id": "cust-003", "name": "John Smith", "phone": "+1-555-0123",
|
||||
"company": "Acme Corp", "source": "support_ticket"},
|
||||
{"id": "cust-004", "name": "Jane Doe", "email": "jane.doe@email.com",
|
||||
"company": "Beta Inc", "source": "web_signup"},
|
||||
]
|
||||
|
||||
# Step 1: Find potential duplicates
|
||||
candidates = detect_duplicates(
|
||||
customers,
|
||||
method="pairwise",
|
||||
similarity_threshold=0.6, # 60% similarity required
|
||||
confidence_threshold=0.5,
|
||||
)
|
||||
|
||||
print("Potential duplicates found:")
|
||||
for c in candidates:
|
||||
print(f" {c.entity1['name']} ~ {c.entity2['name']} (score: {c.similarity_score:.2f})")
|
||||
print(f" Matching signals: {c.reasons}")
|
||||
|
||||
# Expected output:
|
||||
# John Smith ~ J. Smith (score: 0.82)
|
||||
# Matching signals: ['exact', 'property'] # same email
|
||||
# John Smith ~ John Smith (score: 0.78)
|
||||
# Matching signals: ['exact', 'property'] # same name and company
|
||||
|
||||
# Step 2: Merge the duplicates
|
||||
john_smith_records = [customers[0], customers[1], customers[2]] # All John Smith variants
|
||||
merged_ops = merge_entities(john_smith_records, method="keep_most_complete")
|
||||
|
||||
for op in merged_ops:
|
||||
canonical = op.merged_entity
|
||||
print(f"\nCanonical customer: {canonical['name']}")
|
||||
print(f" Email: {canonical.get('email', 'N/A')}")
|
||||
print(f" Phone: {canonical.get('phone', 'N/A')}")
|
||||
print(f" Company: {canonical['company']}")
|
||||
print(f" Merged from {len(op.source_entities)} records")
|
||||
|
||||
# Result: One John Smith record with email, phone, and company information
|
||||
# from all three original records, with full provenance tracking
|
||||
```
|
||||
|
||||
This example demonstrates the core concepts: similarity detection finds related records, and merging consolidates them into canonical entities that preserve all available information.
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Threshold tuning without validation.** Setting thresholds too low creates false positive merges between genuinely different entities. Always manually review a sample of detected duplicates before running large-scale merging operations.
|
||||
|
||||
**Pairwise scaling problems.** The O(n²) cost of comparing every entity pair becomes prohibitive above 10,000 entities. Use clustering methods (`build_clusters`) or switch to vectorized similarity for large datasets.
|
||||
|
||||
**Using fuzzy matching when primary keys exist.** If your entities have reliable unique identifiers (LEI codes, CVE IDs, ISBN numbers), use exact matching on those fields instead of computationally expensive similarity algorithms.
|
||||
|
||||
**Mixing wrapper and class APIs inconsistently.** Don't call `detect_duplicates()` then manually instantiate `EntityMerger`—choose either the functional approach or class-based approach and use it consistently throughout your workflow.
|
||||
|
||||
**Ignoring merge strategy implications.** `keep_first` overwrites later records completely, `merge_all` can introduce conflicting values, and `keep_most_complete` may not respect source authority. Choose the strategy that matches your data quality requirements.
|
||||
|
||||
**Skipping provenance tracking.** Without `preserve_provenance=True`, you lose visibility into which source contributed each field in the canonical entity, making audit trails impossible.
|
||||
|
||||
**Inadequate similarity algorithm selection.** Pure string similarity fails for alias relationships ("APT29" vs "Cozy Bear"), while property matching may be too aggressive for entities with shared attributes but different identities.
|
||||
|
||||
## Domain examples
|
||||
|
||||
<Tabs>
|
||||
|
||||
@@ -6,13 +6,65 @@ icon: "route"
|
||||
|
||||
`ContextGraph` distance intelligence answers the structural question that pure semantic similarity cannot: given two nodes, what is their precise relationship in terms of graph topology, path weight, and inferential confidence? Use it to annotate attribution chains with hop counts and confidence decay, rank retrieval results by structural proximity to an anchor node, and surface implied connections for analyst review.
|
||||
|
||||
## What Is Distance Intelligence?
|
||||
|
||||
Distance intelligence quantifies and analyzes the structural relationships between nodes in your knowledge graph. It provides detailed metadata about graph paths including hop counts, distance bands, confidence decay, and path analysis.
|
||||
|
||||
**Distance metadata** includes hop counts (number of edges between nodes), distance bands (semantic categories like "direct", "near", "distant"), confidence decay (accumulated trust along paths), and path analysis (finding optimal routes between nodes).
|
||||
|
||||
**Hop counts** measure the number of edges you must traverse to reach one node from another. A hop count of 1 means direct connection; 3 means you traverse through 2 intermediate nodes.
|
||||
|
||||
**Distance bands** convert raw hop counts into meaningful categories: "direct" (0-1 hops), "near" (2-3 hops), "mid-range" (4-6 hops), and "distant" (7+ hops). These categories help interpret the semantic meaning of graph distances.
|
||||
|
||||
**Confidence decay** multiplies edge weights along a path to compute accumulated trust. If each edge has weight 0.8, a 3-hop path has confidence decay of 0.8³ = 0.512, indicating moderate confidence in the connection.
|
||||
|
||||
**Path analysis** finds optimal routes between nodes using algorithms like Dijkstra's shortest path or Yen's k-shortest paths algorithm.
|
||||
|
||||
**Distance intelligence vs. graph analytics:** Analytics computes statistical measures like centrality and communities across the entire graph. Distance intelligence focuses on specific paths and relationships between particular nodes.
|
||||
|
||||
**Distance intelligence vs. graph traversal:** Simple traversal follows edges to find neighbors. Distance intelligence quantifies the quality and confidence of those connections using weights, paths, and decay metrics.
|
||||
|
||||
## Why Use Distance Intelligence?
|
||||
|
||||
**Confidence-aware retrieval.** Instead of treating all graph connections equally, distance intelligence weights results by path confidence, giving higher rankings to nodes connected through stronger, more direct relationships.
|
||||
|
||||
**Relationship discovery.** Find not just whether two entities are connected, but how they're connected, through which intermediaries, and with what level of confidence across the full path.
|
||||
|
||||
**Causal analysis.** Trace cause-and-effect chains through your knowledge graph with quantified confidence at each step, essential for decision tracking and audit trails.
|
||||
|
||||
**Precedent search.** Find similar past cases by analyzing structural similarity and path patterns, not just content similarity.
|
||||
|
||||
**Graph-aware ranking.** Blend semantic similarity with graph proximity to surface contextually relevant results that pure vector search would miss.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use distance intelligence for:**
|
||||
- Multi-hop reasoning where path quality matters
|
||||
- Attribution analysis requiring confidence assessment
|
||||
- Causal chain analysis and decision tracing
|
||||
- Proximity-weighted retrieval from specific anchor nodes
|
||||
- Finding alternative connection routes for verification
|
||||
- Ranking results by both content relevance and structural proximity
|
||||
|
||||
**Simple graph traversal may be sufficient for:**
|
||||
- Finding direct neighbors of a node
|
||||
- Basic graph exploration without confidence weighting
|
||||
- Cases where all edges have equal importance
|
||||
- Simple reachability queries (can A reach B?)
|
||||
|
||||
**Distance intelligence may be unnecessary for:**
|
||||
- Single-hop neighbor lookups
|
||||
- Graphs where edge weights don't represent meaningful confidence
|
||||
- Simple existence queries rather than quality assessment
|
||||
- Scenarios where path analysis adds unnecessary complexity
|
||||
|
||||
<Info>
|
||||
Distance Intelligence feeds into proximity-blended retrieval (`proximity_weight` on `retrieve()`), causal chain analysis (`trace_decision_causality()`), and advanced precedent search (`find_precedents_hybrid()`). Enable it by passing `include_distance_metadata=True` on neighbor queries or `proximity_weight > 0` on retrieval calls.
|
||||
</Info>
|
||||
|
||||
## Distance Bands: Turning Hop Counts into Meaning
|
||||
|
||||
The first tool in distance intelligence is `classify_path_distance` — it maps any BFS depth to a human-readable band that carries semantic meaning.
|
||||
The first tool in distance intelligence is `classify_path_distance` — it maps any Breadth-First Search (BFS) depth to a human-readable band that carries semantic meaning.
|
||||
|
||||
```python
|
||||
from semantica.utils.helpers import classify_path_distance
|
||||
@@ -38,6 +90,14 @@ These bands appear automatically on every result that uses `include_distance_met
|
||||
|
||||
Each hop along a path multiplies the accumulated confidence by the edge weight. The product — `confidence_decay` — is the single most useful signal for deciding whether a multi-hop inference is trustworthy.
|
||||
|
||||
<Info>
|
||||
**Confidence Decay and Edge Weights:** Confidence decay depends directly on edge weights in your graph. Weights should represent confidence, trust, relevance, or similar domain-specific signals where higher values indicate stronger relationships. Unweighted graphs (all edges weight 1.0) produce no meaningful decay analysis.
|
||||
</Info>
|
||||
|
||||
<Info>
|
||||
**Dense Graph Warning:** Very dense graphs can make path analysis computationally expensive and results harder to interpret. Dense connectivity creates many possible paths with similar weights, making distance-based rankings less discriminating.
|
||||
</Info>
|
||||
|
||||
```python
|
||||
from semantica.context import ContextGraph
|
||||
|
||||
@@ -137,7 +197,7 @@ path = pf.bfs_shortest_path(graph, "apt29", "nato_target")
|
||||
print("Hop count:", len(path) - 1)
|
||||
```
|
||||
|
||||
**K-shortest paths — Yen's algorithm.** Use when you need alternative attribution chains, redundancy analysis, or corroboration routes. Finding the three shortest paths and showing they all converge on the same target is stronger evidence than a single path.
|
||||
**K-shortest paths — Yen's algorithm.** Yen's algorithm finds multiple alternative paths between two nodes, ranked by total path cost. Use when you need alternative attribution chains, redundancy analysis, or corroboration routes. Finding the three shortest paths and showing they all converge on the same target is stronger evidence than a single path.
|
||||
|
||||
```python
|
||||
k_paths = pf.find_k_shortest_paths(graph, "apt29", "nato_target", k=3)
|
||||
@@ -483,6 +543,18 @@ for chain in chains:
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Treating confidence decay as statistical probability.** Confidence decay is a heuristic measure based on edge weights, not a statistical probability. A decay value of 0.6 doesn't mean "60% probability" — it means the path strength based on your domain-specific weight assignments.
|
||||
|
||||
**Using unweighted graphs and expecting meaningful decay.** If all edges have weight 1.0, confidence decay will always be 1.0 regardless of path length, providing no useful discrimination between paths. Assign meaningful weights that reflect relationship strength.
|
||||
|
||||
**Excessive path exploration on dense graphs.** Dense graphs with many interconnected nodes can generate exponentially large numbers of paths. Limit `max_hops`, use `min_confidence` thresholds, and consider whether simple neighbor lookup would be sufficient.
|
||||
|
||||
**Overusing distance analysis when simple neighbor lookup is enough.** If you only need direct neighbors or one-hop connections, basic graph traversal is simpler and faster than full distance intelligence analysis.
|
||||
|
||||
**Retrieving excessive graph neighborhoods.** Large `max_hops` values can retrieve massive subgraphs that overwhelm downstream processing. Start with 2-3 hops and increase only when needed for your specific use case.
|
||||
|
||||
## Related Guides
|
||||
|
||||
- [Context Graphs](context-graphs) — `ContextGraph` node and edge model; `add_edge(weight=...)` feeds confidence decay
|
||||
|
||||
+80
-1
@@ -3,10 +3,59 @@ title: "Export & Serialization"
|
||||
description: "Export knowledge graphs to RDF (Turtle, JSON-LD, N-Triples), GraphML, Cypher (Neo4j), ArangoDB AQL, CSV, Parquet, OWL, and more."
|
||||
---
|
||||
|
||||
## What Is Export?
|
||||
|
||||
Export converts Semantica graph data into formats used by external tools and systems. Unlike internal persistence mechanisms that keep data within Semantica, export is specifically designed for interoperability with external consumers.
|
||||
|
||||
**Export vs. internal persistence:**
|
||||
- **`AgentContext.store()`** and graph persistence keep data inside Semantica for continued processing, retrieval, and reasoning
|
||||
- **Export functions** serialize graph data into standardized formats that external systems can consume directly
|
||||
|
||||
Export enables integration with analytics platforms, graph databases, RDF triple stores, semantic web systems, data warehouses, business intelligence tools, and downstream consumers that need access to your knowledge graph data in their native formats.
|
||||
|
||||
## Why Use Export?
|
||||
|
||||
**Build once, export many.** Create your knowledge graph through Semantica's extraction and reasoning workflows, then export the same graph data to multiple formats for different consumers without rebuilding or reprocessing.
|
||||
|
||||
**Interoperability with existing ecosystems.** Connect Semantica graphs to established tools and workflows in your organization, from Neo4j graph databases to Gephi visualizations to pandas data analysis pipelines.
|
||||
|
||||
**Analytics and reporting workflows.** Feed graph data into business intelligence tools, statistical analysis platforms, and machine learning pipelines that require specific data formats like CSV, Parquet, or RDF.
|
||||
|
||||
**Graph database migration and deployment.** Move graphs from Semantica's in-memory representation to production graph databases like Neo4j, ArangoDB, or triple stores for scalable query performance.
|
||||
|
||||
**RDF and semantic web integration.** Export to semantic web standards (Turtle, JSON-LD, N-Triples) for integration with ontology tools, SPARQL endpoints, and semantic reasoning systems.
|
||||
|
||||
**Data lake and warehouse integration.** Export to columnar formats like Parquet for integration with modern data stack tools including DuckDB, Apache Spark, and cloud data warehouses.
|
||||
|
||||
**Compliance and archival workflows.** Generate standardized exports for regulatory submission, long-term archival, and audit trail requirements that mandate specific data formats.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use export when:**
|
||||
- Integrating Semantica graphs with external systems and tools
|
||||
- Sharing graph data with teams using different technology stacks
|
||||
- Building analytics pipelines that consume graph data in downstream processing
|
||||
- Working with RDF and ontology workflows requiring semantic web standards
|
||||
- Creating reports, visualizations, and business intelligence dashboards
|
||||
- Migrating graphs to production databases for scalable query performance
|
||||
- Meeting compliance requirements for specific data format submissions
|
||||
|
||||
**Do not use export when:**
|
||||
- You simply want to save and reload Semantica state—use built-in persistence mechanisms instead
|
||||
- Agent persistence and memory continuity are your primary goals
|
||||
- Internal retrieval, reasoning, and graph operations are sufficient for your use case
|
||||
- Export would add unnecessary complexity to workflows that operate entirely within Semantica
|
||||
- You need real-time access to evolving graph data—export creates static snapshots
|
||||
|
||||
**Consider internal persistence instead when:**
|
||||
- Your workflow involves iterative graph building, querying, and reasoning within Semantica
|
||||
- You need to maintain agent memory, conversation history, and decision tracking
|
||||
- Graph data will continue to be processed and enriched within Semantica workflows
|
||||
|
||||
`export_rdf`, `export_graph`, `export_lpg`, and related functions serialize a `ContextGraph` to any of ten formats in a single call, preserving node types, edge weights, and metadata faithfully. Use them when downstream consumers — triple stores, graph databases, visualization tools, ML pipelines, or spreadsheet auditors — each expect a different format from the same in-memory graph.
|
||||
|
||||
<Info>
|
||||
All export functions take `graph.to_dict()` as their first argument — the same dict produced by `ContextGraph.to_dict()`. Build the graph once, export it to as many formats as you need without re-serializing.
|
||||
All export functions take `graph.to_dict()` as their first argument — the same dict produced by `ContextGraph.to_dict()`. Build the graph once, export it to as many formats as you need without re-serializing. Note that `graph.to_dict()` materializes the entire graph in memory, so very large graphs may require additional memory planning.
|
||||
</Info>
|
||||
|
||||
## Building the Graph to Export
|
||||
@@ -36,6 +85,8 @@ graph_data = graph.to_dict() # single dict, reused across all exports below
|
||||
|
||||
## RDF Formats — For Triple Stores and Semantic Reasoners
|
||||
|
||||
**RDF (Resource Description Framework)** is the foundational data model for the semantic web, representing information as subject-predicate-object triplets. RDF formats are essential for integration with semantic web technologies, ontology tools, and systems requiring formal knowledge representation.
|
||||
|
||||
When your consumers are triple stores (GraphDB, Stardog, Apache Jena) or OWL reasoners (HermiT, Pellet), you want RDF. Semantica exports to all five standard RDF serializations through a single `export_rdf` call.
|
||||
|
||||
```python
|
||||
@@ -58,6 +109,8 @@ The format to reach for depends on your consumer. Turtle is ideal for human revi
|
||||
|
||||
## Graph Formats — For Gephi, Maltego, and Network Analysis
|
||||
|
||||
**Labeled Property Graph (LPG)** formats represent networks with typed nodes and edges that carry attributes and metadata. These formats are optimized for graph visualization tools and network analysis platforms that focus on exploring relationships and structural patterns.
|
||||
|
||||
GraphML, GEXF, and DOT are the native formats of graph analysis and visualization tools. They preserve node attributes, edge weights, and type labels, so the graph you built in Semantica renders immediately in Gephi or NetworkX with full attribute data.
|
||||
|
||||
```python
|
||||
@@ -77,6 +130,8 @@ The GEXF format is worth knowing about if you use Gephi for analyst briefings
|
||||
|
||||
## Neo4j Cypher — For Graph-Pattern Threat Hunting
|
||||
|
||||
**Cypher** is Neo4j's declarative graph query language that uses pattern matching to find and manipulate graph data. Cypher exports enable teams to run complex graph queries, pattern detection, and graph analytics using Neo4j's optimized query engine.
|
||||
|
||||
When the SOC team wants to run Cypher queries against the graph — finding threat actors that share infrastructure, or tracing multi-hop attack paths — you export to Cypher and load the result into Neo4j Desktop or Memgraph with a single command.
|
||||
|
||||
```python
|
||||
@@ -116,6 +171,8 @@ The `include_collection_creation=True` flag means the AQL file is self-contained
|
||||
|
||||
## CSV — For Spreadsheet Audits and Statistical Analysis
|
||||
|
||||
**CSV (Comma-Separated Values)** is a simple tabular format universally supported by spreadsheet applications, statistical tools, and data analysis platforms. CSV export flattens graph data into rows and columns for teams that work primarily with tabular data.
|
||||
|
||||
The compliance team lives in Excel. The data science team lives in pandas. Both of them need CSV. `export_csv` writes the graph as flat rows — entities and relationships as separate files when you pass a base path.
|
||||
|
||||
```python
|
||||
@@ -136,6 +193,8 @@ The split form is more useful for downstream tools: the entities CSV feeds a piv
|
||||
|
||||
## Parquet — For Data Lakes and ML Pipelines
|
||||
|
||||
**Parquet** is a columnar storage format optimized for analytics workloads, offering efficient compression and fast query performance. Parquet files integrate seamlessly with modern data stack tools and machine learning frameworks.
|
||||
|
||||
When the data science team runs feature engineering over graph attributes in DuckDB, Spark, or a lakehouse, Parquet is the format they want. It is columnar, compressed, and readable by every major ML framework.
|
||||
|
||||
```python
|
||||
@@ -148,6 +207,10 @@ Once in Parquet, the graph entities become a DataFrame that can be joined agains
|
||||
|
||||
## OWL — For Ontology-Based Reasoning
|
||||
|
||||
**OWL (Web Ontology Language)** is a semantic web standard for representing rich ontologies with classes, properties, and logical constraints. OWL enables automated reasoning, consistency checking, and inference over formal knowledge models.
|
||||
|
||||
**OntologyGenerator** creates formal ontologies from graph data by analyzing entity types, relationships, and patterns to generate class hierarchies, property definitions, and logical constraints. This enables schema validation, automated reasoning, and integration with semantic web tools.
|
||||
|
||||
When you have generated an OWL ontology from your graph using `OntologyGenerator`, you can export it for Protégé, HermiT reasoning, or regulatory submission.
|
||||
|
||||
```python
|
||||
@@ -160,6 +223,22 @@ ontology = OntologyGenerator(base_uri="https://example.org/cti/") \
|
||||
export_owl(ontology, "cti_ontology.owl", format="owl-xml")
|
||||
```
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Confusing export with persistence.** Export creates external snapshots for interoperability, while persistence maintains Semantica's internal state. Don't use export when you need to save and reload agent memory or continue graph-based workflows—use built-in persistence mechanisms instead.
|
||||
|
||||
**Exporting stale graph data after graph changes.** Always call `graph.to_dict()` after your final graph modifications. If you store `graph_data` early in your workflow and then modify the graph, exports will reflect the outdated state, not your latest changes.
|
||||
|
||||
**Re-running expensive extraction instead of reusing existing graph data.** Build your graph once through entity extraction and relationship inference, then export to multiple formats using the same `graph_data` dict. Don't rebuild the graph for each export format.
|
||||
|
||||
**Choosing overly complex formats when CSV is sufficient.** If downstream consumers work with tabular data and don't need graph structure preservation, CSV is simpler, faster, and more universally supported than RDF or GraphML formats.
|
||||
|
||||
**Assuming provenance and history automatically appear in exports.** Standard export formats capture the current graph state but don't include provenance chains, version history, or audit trails. Use dedicated provenance export mechanisms if you need full lineage information.
|
||||
|
||||
**Ignoring downstream schema requirements.** Different systems expect different identifier formats, attribute schemas, and relationship representations. Validate that your exported data matches the expectations of consuming systems before deploying to production workflows.
|
||||
|
||||
**Exporting extremely large graphs without memory planning.** The `graph.to_dict()` operation materializes the entire graph in memory. For very large graphs, monitor memory usage and consider chunking or streaming approaches for resource-constrained environments.
|
||||
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
|
||||
+91
-16
@@ -5,6 +5,71 @@ description: "Go beyond vector search: retrieve facts, trace reasoning paths, an
|
||||
|
||||
GraphRAG combines vector similarity with knowledge graph traversal so retrieval finds structurally connected facts, not just text that sounds related. When a `ContextGraph` is attached to `AgentContext`, every retrieval call automatically blends semantic search with multi-hop graph expansion — and `query_with_reasoning()` returns an auditable reasoning path alongside the LLM answer.
|
||||
|
||||
## What Is GraphRAG?
|
||||
|
||||
GraphRAG (Graph-Augmented Retrieval-Augmented Generation) enhances traditional RAG by combining vector similarity search with knowledge graph traversal. Instead of retrieving only semantically similar text, GraphRAG follows relationships between entities to find connected evidence across multiple documents.
|
||||
|
||||
**GraphRAG vs. traditional vector-only RAG:** Vector RAG finds documents similar to your query text. GraphRAG finds documents similar to your query AND documents connected to those through entity relationships, even if they don't mention your query terms directly.
|
||||
|
||||
**The role of graph traversal:** Starting from entities found in vector-similar documents, GraphRAG expands outward through relationship edges to discover related facts. This reveals connections that pure text similarity would miss — like finding that a threat actor targets healthcare by following the path: Actor → Tool → Victim Organization → Industry Sector.
|
||||
|
||||
## Why Use GraphRAG?
|
||||
|
||||
**Multi-hop discovery.** Find facts that are 2-3 relationship steps away from your query. A question about "APT29 healthcare targeting" can surface evidence about specific hospitals by traversing: APT29 → HAMMERTOSS → LifeCare → Healthcare Sector.
|
||||
|
||||
**Connected evidence.** Instead of isolated document fragments, retrieve coherent chains of related entities and their relationships. This provides richer context for LLM responses and human analysis.
|
||||
|
||||
**Investigation workflows.** Follow evidence trails by expanding from known entities through their connections. Start with a suspicious IP and discover the full infrastructure chain, or trace a drug interaction through metabolic pathways.
|
||||
|
||||
**Richer retrieval context.** Graph expansion surfaces relevant context that keyword or semantic search alone would miss, leading to more complete and accurate LLM responses.
|
||||
|
||||
**Explainability.** GraphRAG provides audit trails showing exactly which entities and relationships led to each piece of retrieved evidence, making the retrieval process transparent and verifiable.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**GraphRAG adds value when:**
|
||||
- Your domain has rich entity relationships (threat intelligence, clinical data, regulatory documents)
|
||||
- Questions require connecting facts across multiple documents
|
||||
- Investigation workflows benefit from following entity connections
|
||||
- Explainability and audit trails are important
|
||||
- You have well-structured knowledge graphs with meaningful relationships
|
||||
|
||||
**Simple vector search may be sufficient for:**
|
||||
- Document retrieval based on topic similarity
|
||||
- Single-document question answering
|
||||
- Exploratory search where you don't know what you're looking for
|
||||
- Domains with few meaningful entity relationships
|
||||
|
||||
**Latency and complexity considerations:**
|
||||
- GraphRAG adds computational overhead from graph traversal
|
||||
- Multi-hop expansion increases retrieval time and token usage
|
||||
- Graph quality directly impacts retrieval quality
|
||||
- Setup requires entity extraction and relationship building
|
||||
|
||||
**GraphRAG may be overkill for:**
|
||||
- Simple lookup queries with known answers in specific documents
|
||||
- Real-time applications where latency is critical
|
||||
- Domains where entity relationships don't provide additional value
|
||||
|
||||
## Typical GraphRAG Workflow
|
||||
|
||||
**Ingest → Build Graph → Retrieve → Expand Context → Reason → Answer**
|
||||
|
||||
1. **Ingest** your documents using `AgentContext.store()` with entity extraction enabled
|
||||
2. **Build Graph** through Named Entity Recognition (NER) and relationship extraction to populate the `ContextGraph`
|
||||
3. **Retrieve** semantically similar documents and identify seed entities for graph expansion
|
||||
4. **Expand Context** by following entity relationships within your specified hop limit
|
||||
5. **Reason** (optional) using the expanded context with reasoning engines
|
||||
6. **Answer** by providing the enriched context to an LLM through `query_with_reasoning()`
|
||||
|
||||
<Info>
|
||||
**Graph Quality Dependency:** GraphRAG retrieval quality depends heavily on graph quality, consistent entity linking, and meaningful relationships. Poor entity extraction, duplicate entities, or weak relationships directly impact retrieval effectiveness.
|
||||
</Info>
|
||||
|
||||
<Info>
|
||||
**Context Expansion Warning:** Larger hop counts exponentially increase the amount of retrieved context, which can significantly increase LLM token usage and processing time. Start with 2-3 hops and monitor context size for your use case.
|
||||
</Info>
|
||||
|
||||
<Info>
|
||||
GraphRAG activates automatically when you pass `knowledge_graph=` to `AgentContext`. There is no separate mode to switch on. The `hybrid_alpha` parameter and `proximity_weight` argument control how much influence graph structure has relative to vector similarity.
|
||||
</Info>
|
||||
@@ -18,7 +83,7 @@ from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
|
||||
# FAISS runs locally with no external dependencies
|
||||
vs = VectorStore(backend="faiss", dimension=768, index_path="intel.faiss")
|
||||
vs = VectorStore(backend="faiss", dimension=768)
|
||||
graph = ContextGraph(advanced_analytics=True)
|
||||
|
||||
context = AgentContext(
|
||||
@@ -31,7 +96,7 @@ context = AgentContext(
|
||||
)
|
||||
```
|
||||
|
||||
Now ingest your documents. `store()` with `extract_entities=True` runs the full extraction pipeline internally — NER, relation extraction, and entity linking — and populates both the vector index and the graph simultaneously:
|
||||
Now ingest your documents. `store()` with `extract_entities=True` runs the full extraction pipeline internally — Named Entity Recognition (NER), relation extraction, and entity linking — and populates both the vector index and the graph simultaneously:
|
||||
|
||||
```python
|
||||
intel_documents = [
|
||||
@@ -82,27 +147,24 @@ With the graph populated, a plain `retrieve()` call already does more than vecto
|
||||
results = context.retrieve(
|
||||
"APT29 tactics against healthcare",
|
||||
use_graph=True,
|
||||
proximity_weight=0.5, # blend structural proximity into the final score
|
||||
max_results=10,
|
||||
expand_graph=True,
|
||||
max_hops=3,
|
||||
)
|
||||
|
||||
for r in results:
|
||||
print("[combined={:.3f} vec={:.3f} prox={:.3f}] {}".format(
|
||||
r.get("combined_score", r["score"]),
|
||||
print("[score={:.3f}] {}".format(
|
||||
r["score"],
|
||||
r.get("proximity_score", 0.0),
|
||||
r["content"][:90],
|
||||
))
|
||||
|
||||
# [combined=0.921 vec=0.884 prox=0.957] APT29 deployed HAMMERTOSS malware against NATO...
|
||||
# [combined=0.887 vec=0.701 prox=0.972] HAMMERTOSS was subsequently observed on hosts in the LifeCare...
|
||||
# [combined=0.841 vec=0.623 prox=0.961] LifeCare operates 47 acute-care hospitals...
|
||||
# [combined=0.798 vec=0.590 prox=0.907] Healthcare critical infrastructure has been a high-priority...
|
||||
# [score=0.921] APT29 deployed HAMMERTOSS malware against NATO...
|
||||
# [score=0.887] HAMMERTOSS was subsequently observed on hosts in the LifeCare...
|
||||
# [score=0.841] LifeCare operates 47 acute-care hospitals...
|
||||
# [score=0.798] Healthcare critical infrastructure has been a high-priority...
|
||||
```
|
||||
|
||||
Notice the third and fourth results: their vector scores are modest (0.623 and 0.590) — neither document mentions APT29 or TTPs. But their proximity scores are high because they are structurally adjacent to the seed nodes in the graph. Pure vector retrieval would have ranked them much lower or excluded them entirely. GraphRAG surfaces them because the graph knows they are connected.
|
||||
Notice the top results: while pure vector search might rank connected facts lower because they lack keyword overlap, GraphRAG boosts their final `score` because they are structurally adjacent to the seed nodes in the graph. The returned `score` is a transparent blend of vector relevance and graph connectivity.
|
||||
|
||||
When you know specifically which entity you want to anchor the traversal to, pass `anchor_node`:
|
||||
|
||||
@@ -299,11 +361,10 @@ print("Confidence: {:.1%}".format(triage["confidence"]))
|
||||
similar = soc_context.retrieve(
|
||||
"wmiprvse.exe encoded powershell scheduled task persistence",
|
||||
use_graph=True,
|
||||
proximity_weight=0.5,
|
||||
max_results=5,
|
||||
)
|
||||
for inc in similar:
|
||||
print("[{:.3f}] {}".format(inc.get("combined_score", inc["score"]), inc["content"][:100]))
|
||||
print("[{:.3f}] {}".format(inc["score"], inc["content"][:100]))
|
||||
```
|
||||
|
||||
</Tab>
|
||||
@@ -444,15 +505,29 @@ print(answer["reasoning_path"])
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Excessive hop counts.** Setting `max_expansion_hops` too high (>4) creates exponentially large context that overwhelms LLMs and increases costs. Start with 2-3 hops and increase only if needed.
|
||||
|
||||
**Poor graph quality.** GraphRAG amplifies graph quality issues. Duplicate entities, inconsistent naming, and weak relationships produce poor retrieval results. Clean your graph data before relying on GraphRAG for important queries.
|
||||
|
||||
**Duplicate entities.** Having "APT-29", "APT29", and "Cozy Bear" as separate nodes breaks relationship traversal. Entity linking during ingestion helps, but manual deduplication may be necessary.
|
||||
|
||||
**Using GraphRAG for simple lookup queries.** If you know the answer exists in a specific document and just need to retrieve it, traditional vector search is faster and simpler than GraphRAG.
|
||||
|
||||
**Assuming graph expansion is always beneficial.** More context isn't always better. Sometimes precise, focused retrieval outperforms broad graph expansion. Test both approaches for your specific use cases.
|
||||
|
||||
## Tuning the vector-graph balance
|
||||
|
||||
The `hybrid_alpha` parameter set in the `AgentContext` constructor establishes a default blend between vector similarity and graph influence. `0.0` is pure vector retrieval; `1.0` is pure graph traversal. The recommended starting point is `0.5`.
|
||||
|
||||
You can override this per call using `proximity_weight` in `retrieve()` without changing the constructor default:
|
||||
When targeting a specific `anchor_node`, you can apply `proximity_weight` in `retrieve()` to dynamically blend structural distance from the anchor into the final score:
|
||||
|
||||
```python
|
||||
# Exploratory query — let semantics lead, graph confirms
|
||||
results = context.retrieve(query, use_graph=True, proximity_weight=0.2)
|
||||
# Anchor node provided — let vector semantics lead, graph proximity only slightly boosts
|
||||
results = context.retrieve(
|
||||
query, use_graph=True, anchor_node="APT29", proximity_weight=0.2
|
||||
)
|
||||
|
||||
# Known-entity tracing — topology drives the retrieval
|
||||
results = context.retrieve(
|
||||
|
||||
+87
-2
@@ -50,6 +50,7 @@ Use the ingest module when your data lives outside Semantica and you need to bri
|
||||
- **Web content** — public documentation sites, regulatory publication pages, news feeds, or any URL you can crawl.
|
||||
- **REST APIs** — internal platforms (SIEM, EDR, ITSM, CRM), threat intelligence feeds, or any paginated HTTP endpoint.
|
||||
- **Databases** — existing SQL databases where relevant records can be fetched with a targeted query.
|
||||
- **Enterprise data platforms** — tables already living in a Databricks lakehouse (Unity Catalog + Delta Lake) or a Snowflake warehouse, without exporting to CSV first.
|
||||
- **Live streams** — Kafka or other message brokers where you need to process events as they arrive.
|
||||
- **Git repositories** — source code, documentation, or configuration files tracked in version control.
|
||||
|
||||
@@ -298,6 +299,89 @@ for bundle in stix_xml_files:
|
||||
print(f"{bundle.source_path}: {len(bundle.elements)} elements parsed")
|
||||
```
|
||||
|
||||
## Source 6 — Enterprise Data Platforms (Databricks & Snowflake)
|
||||
|
||||
`DatabricksIngestor` and `SnowflakeIngestor` return wrapper objects (`DatabricksData` / `SnowflakeData`) whose `.data` field is `List[Dict]` — the same list-of-dicts row shape that `DBIngestor.execute_query()` returns directly, without a wrapper. The same "transform to text, then store" pattern from Source 3 applies: pull only the tables and columns you need with a targeted query, then build a sentence per record before handing it to `AgentContext.store()`.
|
||||
|
||||
```python
|
||||
from semantica.ingest import DatabricksIngestor
|
||||
|
||||
# Unity Catalog + Delta Lake — PAT or OAuth M2M auth
|
||||
databricks = DatabricksIngestor(
|
||||
host="https://adb-xxx.azuredatabricks.net",
|
||||
token="dapi-xxxxxxxx",
|
||||
http_path="/sql/1.0/warehouses/xxxxxxxx",
|
||||
catalog="main",
|
||||
)
|
||||
|
||||
# .data is List[Dict] — one dict per row, same shape as DBIngestor.execute_query()
|
||||
customers = databricks.ingest_query(
|
||||
"SELECT customer_id, name, industry, arr FROM main.default.customers "
|
||||
"WHERE churn_risk_score > 0.7"
|
||||
)
|
||||
customer_texts = [
|
||||
f"Customer {r['customer_id']} ({r['name']}, {r['industry']}): "
|
||||
f"ARR ${r['arr']:,}, flagged high churn risk"
|
||||
for r in customers.data
|
||||
]
|
||||
|
||||
# Unity Catalog lineage — build Table --DEPENDS_ON--> Table edges directly from
|
||||
# Unity Catalog's own lineage tracking, instead of re-deriving them from query logs
|
||||
lineage = databricks.get_table_lineage("customers", catalog="main", schema="default")
|
||||
lineage_texts = [
|
||||
f"Table main.default.customers depends on {upstream}"
|
||||
for upstream in lineage["upstream"]
|
||||
]
|
||||
```
|
||||
|
||||
```python
|
||||
from semantica.ingest import SnowflakeIngestor
|
||||
|
||||
snowflake = SnowflakeIngestor(
|
||||
account="myaccount",
|
||||
user="myuser",
|
||||
password="mypassword", # or private_key=... for key-pair; use authenticator="oauth", token=... for OAuth
|
||||
warehouse="COMPUTE_WH",
|
||||
database="ANALYTICS",
|
||||
schema="PUBLIC",
|
||||
)
|
||||
|
||||
# Snowflake uppercases unquoted identifiers, so unquoted columns come back
|
||||
# as ORDER_ID, PRODUCT, etc. unless the source table quotes them lowercase
|
||||
orders = snowflake.ingest_query(
|
||||
"SELECT order_id, product, region, amount FROM orders "
|
||||
"WHERE order_date >= DATEADD(day, -30, CURRENT_DATE())"
|
||||
)
|
||||
order_texts = [
|
||||
f"Order {r['ORDER_ID']}: {r['PRODUCT']} in {r['REGION']}, ${r['AMOUNT']}"
|
||||
for r in orders.data
|
||||
]
|
||||
```
|
||||
|
||||
Feed the resulting text lists into `AgentContext.store()` exactly like any other structured source:
|
||||
|
||||
```python
|
||||
from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
|
||||
graph = ContextGraph(advanced_analytics=True)
|
||||
context = AgentContext(
|
||||
vector_store = VectorStore(backend="faiss"),
|
||||
knowledge_graph = graph,
|
||||
)
|
||||
|
||||
context.store(
|
||||
customer_texts + lineage_texts + order_texts,
|
||||
extract_entities=True,
|
||||
extract_relationships=True,
|
||||
)
|
||||
print(f"Enterprise data graph: {graph.stats()['node_count']} nodes")
|
||||
```
|
||||
|
||||
For authentication details (PAT vs. OAuth M2M for Databricks; password vs. key-pair vs. OAuth for Snowflake), schema/catalog introspection, and troubleshooting, see the dedicated [Databricks Integration](../integrations/databricks) and [Snowflake Integration](../integrations/snowflake) guides.
|
||||
|
||||
> **Security Note:** Never hardcode credentials (`token`, `password`, `private_key`) in production code; pass them via environment variables (e.g., `DATABRICKS_TOKEN`, `SNOWFLAKE_PASSWORD`) or a secrets manager.
|
||||
|
||||
## Combining All Five Sources
|
||||
|
||||
Once you have text from each source, `AgentContext.store()` accepts a flat list of strings. Semantica embeds and indexes them together — the context graph has no concept of which string came from which source unless you add metadata explicitly.
|
||||
@@ -468,8 +552,7 @@ def run_daily_ingest(since: datetime = None):
|
||||
|
||||
graph = ContextGraph(advanced_analytics=True)
|
||||
context = AgentContext(
|
||||
vector_store = VectorStore(backend="faiss", dimension=768,
|
||||
index_path="cti_index.faiss"),
|
||||
vector_store = VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph = graph,
|
||||
graph_expansion = True,
|
||||
)
|
||||
@@ -832,3 +915,5 @@ print(f"Compliance graph: {graph.stats()['node_count']} nodes, "
|
||||
- [Context Graphs](context-graphs) — storing and querying the entities you ingest as a typed property graph
|
||||
- [Semantic Extraction](semantic-extraction) — NER, relation extraction, and triplet extraction from ingested text
|
||||
- [Provenance](provenance) — tracking the origin document, confidence score, and ingestion timestamp for every extracted entity
|
||||
- [Databricks Integration](../integrations/databricks) — Unity Catalog setup, PAT/OAuth M2M authentication, and lineage introspection
|
||||
- [Snowflake Integration](../integrations/snowflake) — warehouse setup and password/key-pair/OAuth authentication
|
||||
|
||||
@@ -5,15 +5,65 @@ description: "Connect Semantica to Groq, OpenAI, Anthropic, HuggingFace, Novita
|
||||
|
||||
Semantica exposes a unified provider interface — a single `.generate()` method — across Groq, OpenAI, Anthropic Claude, HuggingFace, Novita AI, and 100+ providers via LiteLLM. Use it when you need to swap providers for latency, accuracy, cost, or data-residency reasons without touching application code.
|
||||
|
||||
## What Are LLM Integrations?
|
||||
|
||||
The `semantica.llms` module provides a unified interface for connecting to Large Language Model providers. Instead of learning different APIs for each provider, you use the same methods (`.generate()`, `.generate_structured()`) regardless of whether you're calling Groq, OpenAI, Anthropic, or local HuggingFace models.
|
||||
|
||||
**Unified interface across providers:** All LLM providers in Semantica expose identical methods, so switching from OpenAI to Anthropic requires changing only the provider constructor, not your application code.
|
||||
|
||||
**Provider wrappers vs semantic extraction provider strings:** The `semantica.llms` classes (`Groq`, `OpenAI`, `LiteLLM`, `HuggingFaceLLM`) are Python objects for text generation. The `semantica.semantic_extract` module accepts provider names as strings for entity and relationship extraction. Both approaches are covered in this guide.
|
||||
|
||||
## Why Use LLM Integrations?
|
||||
|
||||
**Provider portability.** Test with one provider, deploy with another. Switch from Groq for prototyping to Anthropic for production without code changes.
|
||||
|
||||
**Reduced vendor lock-in.** Avoid tying your application to a single LLM provider's API. If pricing changes or service availability issues arise, switching providers is straightforward.
|
||||
|
||||
**Consistent APIs.** Use the same `.generate()` and `.generate_structured()` methods across all providers instead of learning provider-specific interfaces.
|
||||
|
||||
**Multi-provider workflows.** Run fast models for initial classification and expensive frontier models for complex reasoning in the same pipeline.
|
||||
|
||||
**Local vs cloud deployment flexibility.** Use cloud providers during development and switch to local HuggingFace models for air-gapped production environments.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use LLM integrations for:**
|
||||
- Text generation, summarization, and question-answering tasks
|
||||
- Complex reasoning that requires natural language understanding
|
||||
- Structured data extraction from unstructured text
|
||||
- Multi-step analysis requiring interpretation and synthesis
|
||||
- Tasks where context, ambiguity, or domain knowledge matter
|
||||
|
||||
**Deterministic tools may be better for:**
|
||||
- Pattern matching that regular expressions can handle
|
||||
- Simple rule-based classification with clear criteria
|
||||
- Mathematical calculations or statistical analysis
|
||||
- Graph traversal and relationship queries
|
||||
- Data transformations with known logic
|
||||
|
||||
**A full LLM may be unnecessary for:**
|
||||
- Simple keyword search or exact string matching
|
||||
- Deterministic workflows with predefined decision trees
|
||||
- High-frequency, low-latency operations where inference overhead matters
|
||||
- Tasks where explainability requires transparent rule-based logic
|
||||
|
||||
<Info>
|
||||
The providers in `semantica.llms` (`Groq`, `OpenAI`, `LiteLLM`, `HuggingFaceLLM`) are for text generation and `query_with_reasoning()`. For structured entity and relation extraction, `semantica.semantic_extract` accepts provider names as strings. Both patterns are covered here.
|
||||
</Info>
|
||||
|
||||
## Choosing a Provider
|
||||
|
||||
Four factors drive provider selection. **Latency** matters most in real-time SOC triage loops where an analyst is waiting on a triage verdict — Groq's inference server typically returns 8B model responses in under 300ms. **Accuracy** matters most in high-stakes decisions: clinical contraindication checks, credit committee reasoning, and legal document analysis reward the frontier models available via `LiteLLM`. **Data residency** constraints eliminate cloud providers for classified or HIPAA-regulated workloads — `HuggingFaceLLM` with a local model path covers those cases. **Cost at scale** favors high-throughput open-model providers like Novita AI for bulk extraction pipelines where you are processing thousands of documents per hour.
|
||||
Four factors drive provider selection, each optimized for different use cases:
|
||||
|
||||
The good news: because Semantica's interface is identical across providers, you can prototype with Groq for speed, validate accuracy with Claude, and deploy to Azure OpenAI for compliance — without changing a single line of your application code. Only the provider constructor changes.
|
||||
**Latency** matters most in real-time SOC triage loops where an analyst is waiting on a triage verdict. Groq's inference infrastructure typically returns 8B model responses in under 300ms, making it ideal for interactive workflows.
|
||||
|
||||
**Accuracy** matters most in high-stakes decisions: clinical contraindication checks, credit committee reasoning, and legal document analysis. Frontier models like Claude or GPT-4 available through `LiteLLM` provide the strongest reasoning capabilities.
|
||||
|
||||
**Data residency** constraints eliminate cloud providers for classified or HIPAA-regulated workloads. `HuggingFaceLLM` with local model paths enables fully air-gapped deployments without network calls.
|
||||
|
||||
**Cost at scale** favors high-throughput providers like Novita AI for bulk extraction pipelines processing thousands of documents per hour where per-token costs accumulate quickly.
|
||||
|
||||
The unified interface means you can prototype with Groq for speed, validate accuracy with Claude, and deploy to Azure OpenAI for compliance — without changing application code.
|
||||
|
||||
## The Shared Interface
|
||||
|
||||
@@ -31,6 +81,8 @@ This means every place in Semantica that accepts an LLM — `query_with_reasonin
|
||||
|
||||
## Groq — Fast Inference for Real-Time Agents
|
||||
|
||||
**Groq** is a cloud provider that specializes in ultra-fast language model inference using custom hardware called Language Processing Units (LPUs). Their infrastructure delivers sub-300ms response times for smaller models, making them ideal for real-time applications where speed matters more than maximum reasoning capability.
|
||||
|
||||
Groq Cloud runs open models on purpose-built Language Processing Units that deliver sub-300ms latency for 8B parameter models. This makes Groq the right default for any agent loop where the LLM is in the hot path — SOC triage, real-time alert classification, conversational agents.
|
||||
|
||||
```python
|
||||
@@ -64,6 +116,8 @@ Groq model selection comes down to the speed-vs-capability tradeoff: `llama-3.1-
|
||||
|
||||
## OpenAI — Function Calling and Vision
|
||||
|
||||
**OpenAI** provides access to the GPT model family, including GPT-4o with advanced capabilities like function calling (structured tool use) and vision processing for images and documents. OpenAI models are well-suited for complex reasoning tasks that require strong language understanding and generation capabilities.
|
||||
|
||||
The `OpenAI` provider wraps the OpenAI API. Use it when you need GPT-4o's function-calling precision, vision capabilities for document screenshots, or when your team already has an OpenAI contract and wants to stay there.
|
||||
|
||||
```python
|
||||
@@ -91,6 +145,8 @@ The default model `gpt-3.5-turbo` is fine for classification and light extractio
|
||||
|
||||
## LiteLLM — One Interface, 100+ Providers
|
||||
|
||||
**LiteLLM** is a universal adapter that provides a single interface to over 100 different LLM providers, including Anthropic Claude, Azure OpenAI, AWS Bedrock, Google Vertex AI, and local Ollama instances. It acts as a translation layer, converting your unified API calls into provider-specific requests, enabling easy switching between providers without code changes.
|
||||
|
||||
`LiteLLM` is the Swiss Army knife. It wraps the `litellm` library, which speaks to every major provider using a unified completion API. The model string encodes both provider and model name: `"anthropic/claude-sonnet-4-20250514"`, `"azure/gpt-4o"`, `"bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0"`, `"ollama/llama3.2"`. Change the string, change the provider — no other code changes needed.
|
||||
|
||||
```python
|
||||
@@ -135,6 +191,8 @@ llm = LiteLLM(model=PROVIDER_MAP[env])
|
||||
|
||||
## HuggingFaceLLM — Air-Gapped and On-Premise
|
||||
|
||||
**HuggingFaceLLM** provides access to open-source models from the HuggingFace ecosystem, either downloaded from the HuggingFace Hub or loaded from local file paths. This is the only option for completely offline deployments where no network access is available during inference, such as classified environments or air-gapped systems.
|
||||
|
||||
`HuggingFaceLLM` loads a model from the HuggingFace Hub or from a local directory path. No network calls during inference. This is the only option for classified environments, HIPAA-constrained clinical deployments, and any network segment without outbound internet access.
|
||||
|
||||
```python
|
||||
@@ -302,9 +360,8 @@ from semantica.vector_store import VectorStore
|
||||
extraction_llm = HuggingFaceLLM(model="/opt/models/mistral-7b-instruct")
|
||||
reasoning_llm = HuggingFaceLLM(model="/opt/models/llama-3.1-70b-instruct")
|
||||
|
||||
# NER with local model — provider pattern still works for local paths
|
||||
# (use extract_entities_llm directly with the provider instance)
|
||||
from semantica.semantic_extract.methods import extract_entities_llm
|
||||
# The llms module wrappers can also be used directly for raw prompt generation
|
||||
# when you want to bypass the semantic extraction layer entirely
|
||||
|
||||
sigint_text = (
|
||||
"[S//NF] APT29 operator observed deploying WARPWIRE credential harvester "
|
||||
@@ -512,12 +569,26 @@ print(best["response"])
|
||||
|
||||
# Sources the answer is grounded in
|
||||
for src in best["sources"]:
|
||||
print(" - [{}] {}".format(src.get("metadata", {}).get("source", "?"), src["content"][:60]))
|
||||
print(" - [{}] {}".format(src.get("source", "?"), src["content"][:60]))
|
||||
```
|
||||
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Choosing expensive frontier models for simple extraction tasks.** GPT-4o or Claude Sonnet for basic entity extraction is overkill — Groq's Llama models handle straightforward NER and classification at a fraction of the cost and latency. Reserve frontier models for complex reasoning that requires nuanced interpretation.
|
||||
|
||||
**Ignoring latency differences between providers.** Groq typically responds in under 300ms, while Anthropic Claude can take 2-3 seconds for the same query. For real-time agents or interactive workflows, latency differences compound across multiple LLM calls. Profile your provider performance under realistic load.
|
||||
|
||||
**Using LLMs for deterministic pattern matching that regex can handle.** If your task is extracting email addresses, phone numbers, or other pattern-based entities, regular expressions are faster, cheaper, and more reliable than LLM extraction. Use LLMs when context, ambiguity, or domain knowledge matter for correct interpretation.
|
||||
|
||||
**Not validating structured outputs.** The `generate_structured()` method returns parsed JSON, but LLMs can still produce malformed or incomplete structures. Always validate the returned dictionary against your expected schema before using the data downstream.
|
||||
|
||||
**Switching providers without testing prompt behavior.** Different models respond differently to the same prompt. A prompt optimized for GPT-4 may produce poor results with Llama or Claude. When switching providers, test your prompts and adjust temperature, instructions, or examples as needed.
|
||||
|
||||
**Overusing local HuggingFace models for tasks requiring latest knowledge.** Local models have a knowledge cutoff from their training date and cannot access current information. For tasks requiring up-to-date knowledge (recent CVEs, current regulations, latest threat intelligence), cloud providers with more recent training data may be necessary.
|
||||
|
||||
## Related Guides
|
||||
|
||||
- [Agent Memory](agent-memory) — using `query_with_reasoning()` with any LLM provider for graph-grounded retrieval
|
||||
|
||||
+88
-14
@@ -4,15 +4,50 @@ description: "Connect Semantica's knowledge graph, decision intelligence, and re
|
||||
icon: "plug"
|
||||
---
|
||||
|
||||
The Semantica MCP server exposes your knowledge graph as 12 callable tools so any compatible AI client — Claude Desktop, Windsurf, VS Code extensions — can traverse the graph live, record decisions, run analytics, and export results during a conversation. Use it to give LLM agents direct, real-time access to graph data without writing custom tool wrappers.
|
||||
## What Is MCP?
|
||||
|
||||
MCP stands for the Model Context Protocol. It is an open standard that allows external AI assistants (like Claude Desktop, Cursor, or Windsurf) to securely access local tools and data sources.
|
||||
|
||||
The Semantica MCP server exposes your knowledge graph as 12 callable tools. By connecting it, any compatible AI client can traverse the graph live, record decisions, run analytics, and export results during a conversation — without you having to write custom tool wrappers.
|
||||
|
||||
<Info>
|
||||
The Semantica MCP server exposes 12 tools and 3 read-only resources. All tools accept and return JSON. No configuration beyond an optional environment variable for graph persistence is required.
|
||||
</Info>
|
||||
|
||||
## Architecture & Communication
|
||||
|
||||
It is important to understand how MCP works under the hood. **The Semantica MCP server is not a REST API.** There are no network ports, no HTTP endpoints, and no API keys required.
|
||||
|
||||
Instead, the AI client launches `semantica-mcp` locally as a subprocess. All communication between the AI and Semantica happens securely through standard input and output (`stdio`). Because the server runs locally under your user account, it inherently has your local file permissions.
|
||||
|
||||
## Why Use MCP With Semantica?
|
||||
|
||||
- **Zero-Code Integration**: Instantly connect Semantica's graph capabilities to your favorite AI IDE or desktop chat app without writing any glue code.
|
||||
- **Real-Time Graph Updates**: Chat with an AI to extract entities from documents and watch them populate your live knowledge graph instantly.
|
||||
- **Auditable AI**: Use the AI to make decisions and have it automatically record the reasoning and causal chain directly into the graph via Semantica's decision intelligence tools.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
- **When to Use**: You want to use a third-party AI interface (like Claude Desktop or Windsurf) to manipulate, query, and reason over a Semantica knowledge graph on your local machine.
|
||||
- **When NOT to Use**: You are building an autonomous Python script or backend service. If you are writing Python code to build an agent, use `semantica.context.AgentContext` natively instead of spinning up an MCP server. The MCP server does not support remote hosting over HTTP/SSE.
|
||||
|
||||
---
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
Connecting your AI client follows a standard progression:
|
||||
|
||||
1. **Install**: Install Semantica in your Python environment.
|
||||
2. **Configure Client**: Add the `semantica-mcp` command and absolute graph paths to your AI client's JSON configuration.
|
||||
3. **Start Client**: Launch Claude Desktop or Windsurf, which automatically spawns the MCP server.
|
||||
4. **Tool Calls**: Prompt the AI in natural language. The AI autonomously chains the 12 available tools.
|
||||
5. **Graph Updates**: The AI directly modifies your local graph, adding entities, edges, and decisions.
|
||||
|
||||
---
|
||||
|
||||
## Starting the Server
|
||||
|
||||
Install Semantica, then launch the MCP server. It starts in stdio mode by default — the protocol used by Claude Desktop, Windsurf, VS Code extensions, and most MCP clients.
|
||||
Install Semantica, then configure your client to launch the MCP server. The server runs using the `stdio` transport.
|
||||
|
||||
```bash
|
||||
pip install semantica
|
||||
@@ -26,14 +61,14 @@ semantica-mcp
|
||||
python -m semantica.mcp_server
|
||||
```
|
||||
|
||||
Startup info prints to stderr. Without `SEMANTICA_KG_PATH` the server initialises an empty in-memory graph — sufficient for testing. For a persistent graph that survives restarts, set the path:
|
||||
By default, the server logs at `WARNING` level and produces no startup output. Set `SEMANTICA_LOG_LEVEL=INFO` (or `DEBUG`) to see startup messages on stderr. Without `SEMANTICA_KG_PATH` the server initialises an empty in-memory graph — sufficient for testing. For a persistent graph that survives restarts, set the path:
|
||||
|
||||
```bash
|
||||
SEMANTICA_KG_PATH=/data/threat_graph.json semantica-mcp
|
||||
```
|
||||
|
||||
<Info>
|
||||
Without `SEMANTICA_KG_PATH`, the graph resets when the server process exits. Always set this path for any session whose data should survive a restart.
|
||||
Without `SEMANTICA_KG_PATH`, the graph resets when the server process exits. Always set this path using an absolute file path for any session whose data should survive a restart.
|
||||
</Info>
|
||||
|
||||
## Connecting to Claude Desktop
|
||||
@@ -46,7 +81,7 @@ Edit the Claude Desktop config file — on macOS at `~/Library/Application Suppo
|
||||
"semantica": {
|
||||
"command": "semantica-mcp",
|
||||
"env": {
|
||||
"SEMANTICA_KG_PATH": "/path/to/knowledge_graph.json",
|
||||
"SEMANTICA_KG_PATH": "/absolute/path/to/knowledge_graph.json",
|
||||
"SEMANTICA_LOG_LEVEL": "INFO"
|
||||
}
|
||||
}
|
||||
@@ -56,7 +91,7 @@ Edit the Claude Desktop config file — on macOS at `~/Library/Application Suppo
|
||||
|
||||
Restart Claude Desktop after saving. The Semantica tools appear in the tool palette automatically — Claude can now call them during any conversation.
|
||||
|
||||
If `semantica-mcp` is not on your system PATH (for example, if it is installed in a virtualenv), use the full binary path in `"command"`: `"/path/to/venv/bin/semantica-mcp"`.
|
||||
If `semantica-mcp` is not on your system PATH (for example, if it is installed in a virtualenv), use the full absolute binary path in `"command"`: `"/path/to/venv/bin/semantica-mcp"`.
|
||||
|
||||
## Connecting to Other Clients
|
||||
|
||||
@@ -66,7 +101,7 @@ If `semantica-mcp` is not on your system PATH (for example, if it is installed i
|
||||
{
|
||||
"semantica": {
|
||||
"command": "semantica-mcp",
|
||||
"env": { "SEMANTICA_KG_PATH": "/path/to/knowledge_graph.json" }
|
||||
"env": { "SEMANTICA_KG_PATH": "/absolute/path/to/knowledge_graph.json" }
|
||||
}
|
||||
}
|
||||
```
|
||||
@@ -93,7 +128,7 @@ If `semantica-mcp` is not on your system PATH (for example, if it is installed i
|
||||
```bash
|
||||
docker run --rm -i \
|
||||
-e SEMANTICA_KG_PATH=/data/kg.json \
|
||||
-v /local/path:/data \
|
||||
-v /local/absolute/path:/data \
|
||||
ghcr.io/semantica-agi/semantica-mcp:latest
|
||||
```
|
||||
|
||||
@@ -109,15 +144,42 @@ Once connected, the LLM can call any of these tools during a conversation. The a
|
||||
|
||||
**Reasoning** — `run_reasoning` applies forward-chaining IF/THEN rules over a set of facts and returns derived conclusions.
|
||||
|
||||
**Analytics and export** — `get_graph_analytics` computes PageRank centrality and community detection. `get_graph_summary` returns node count, decision count, and server status. `export_graph` serializes the current graph to Turtle, JSON-LD, N-Triples, or plain JSON.
|
||||
**Analytics and export** — `get_graph_analytics` computes PageRank centrality and community detection. `get_graph_summary` returns node count, decision count, and server status. `export_graph` serializes the current graph to Turtle (`"turtle"` / `"ttl"`), RDF/XML (`"xml"`), N-Triples (`"nt"`), JSON-LD (`"json-ld"`), or plain JSON (`"json"`).
|
||||
|
||||
## Universal Example: Employee Directory
|
||||
|
||||
Before diving into complex domain examples, here is a simple, universally understood session. An HR manager types a prompt into Claude Desktop:
|
||||
|
||||
> "Extract entities from this meeting transcript about Alice transferring to Engineering, add them to the graph, and record a promotion decision."
|
||||
|
||||
Claude chains four tool calls automatically:
|
||||
|
||||
```text
|
||||
1. extract_entities(text="Alice is transferring to Engineering...")
|
||||
→ { "entities": [{"label": "Alice", "type": "Employee"}, {"label": "Engineering", "type": "Department"}] }
|
||||
|
||||
2. add_entity(id="emp-alice", label="Alice", type="Employee")
|
||||
add_entity(id="dept-eng", label="Engineering", type="Department")
|
||||
|
||||
3. add_relationship(source="emp-alice", target="dept-eng", type="WORKS_IN")
|
||||
|
||||
4. record_decision(
|
||||
category="promotion",
|
||||
scenario="Alice transferring to Engineering",
|
||||
reasoning="Approved by Engineering Director",
|
||||
outcome="transfer_approved",
|
||||
confidence=1.0
|
||||
)
|
||||
```
|
||||
The graph is updated instantly with the new organizational structure and a fully auditable decision trail.
|
||||
|
||||
## Watching a Real Agent Session
|
||||
|
||||
Here is what happens when an analyst types a prompt into Claude Desktop and the graph is live. The prompt is:
|
||||
Here is what happens when a cybersecurity analyst types a prompt into Claude Desktop and the graph is live. The prompt is:
|
||||
|
||||
> "Extract entities and relationships from this OSINT report, add them to the knowledge graph, then record an attribution decision for APT29 with confidence 0.88 and export the full graph as Turtle."
|
||||
|
||||
Claude chains five tool calls automatically:
|
||||
Claude chains six tool calls automatically:
|
||||
|
||||
```text
|
||||
1. extract_entities(text="<report text>")
|
||||
@@ -157,7 +219,7 @@ Resources expose graph state without a tool call — the client can read them at
|
||||
|
||||
| URI | Description |
|
||||
| :-- | :---------- |
|
||||
| `semantica://graph/summary` | Node count, edge count, server status |
|
||||
| `semantica://graph/summary` | Node count, decision count, server status |
|
||||
| `semantica://decisions/list` | Up to 50 most recent recorded decisions |
|
||||
| `semantica://schema/info` | Server version, capabilities, available tool list |
|
||||
|
||||
@@ -254,11 +316,23 @@ The result is a fully auditable credit decision trail with precedent links, read
|
||||
|
||||
</Tabs>
|
||||
|
||||
---
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
- **Treating MCP as an HTTP server**: Do not try to `curl` the MCP server or look for a port number. It communicates via `stdin/stdout` and waits for JSON-RPC messages from the parent AI client.
|
||||
- **Using relative paths for `SEMANTICA_KG_PATH`**: Because the AI client spawns the server as a subprocess, the working directory can be unpredictable. Always use absolute paths (e.g., `C:\Users\Name\graph.json` or `/Users/name/graph.json`) to avoid losing your data.
|
||||
- **Virtual environment PATH issues**: If you installed Semantica inside a Python virtual environment, Claude Desktop will not automatically find `semantica-mcp` on the global system PATH. You must provide the absolute path to the binary in the `"command"` field.
|
||||
- **Expecting remote hosting support**: Stdio-based MCP servers must run on the same local machine as the AI client. Remote execution over a network is not supported.
|
||||
- **Confusing MCP integration with `AgentContext`**: If you are writing your own Python code to orchestrate an LLM, do not use the MCP server. Use the `AgentContext` class natively within your code.
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**Server does not appear in Claude Desktop** — fully quit and reopen Claude Desktop after editing the config (close the window is not enough). Verify the binary is on PATH: `which semantica-mcp` on Unix, `where semantica-mcp` on Windows. If using a virtualenv, use the absolute binary path in `"command"`. Set `SEMANTICA_LOG_LEVEL=DEBUG` and check stderr for startup errors.
|
||||
**Server does not appear in Claude Desktop** — fully quit and reopen Claude Desktop after editing the config (closing the window is not enough). Verify the binary is on PATH: `which semantica-mcp` on Unix, `where semantica-mcp` on Windows. If using a virtualenv, use the absolute binary path in `"command"`. Set `SEMANTICA_LOG_LEVEL=DEBUG` and check stderr for startup errors.
|
||||
|
||||
**Graph data not persisting between sessions** — set `SEMANTICA_KG_PATH` to an absolute file path. Without it the graph is in-memory only and resets on every server restart.
|
||||
**Graph data not persisting between sessions** — set `SEMANTICA_KG_PATH` to an absolute file path. Without it, the graph is in-memory only and resets on every server restart.
|
||||
|
||||
**Tool calls returning empty results** — `get_graph_summary` returning `"node_count": 0` means the graph is empty. Populate it via `add_entity` and `add_relationship`, or run `extract_entities` on text first and then `add_entity` for each result.
|
||||
|
||||
|
||||
+83
-11
@@ -3,6 +3,55 @@ title: "Multi-Agent Systems"
|
||||
description: "Coordinate multiple AI agents through shared memory, knowledge graphs, and decision history — without a message broker."
|
||||
---
|
||||
|
||||
## What Is Multi-Agent Coordination?
|
||||
|
||||
A multi-agent system is a software architecture where multiple autonomous agents work together to accomplish complex tasks that would be difficult or impossible for a single agent to handle effectively. Instead of building one monolithic agent that tries to do everything, developers split work across specialized agents that each focus on specific responsibilities.
|
||||
|
||||
**Why split work across multiple agents:**
|
||||
- **Separation of concerns** — each agent specializes in one domain (ingestion, analysis, reporting) rather than trying to master everything
|
||||
- **Independent reasoning** — different agents can use different models, prompts, and reasoning strategies optimized for their specific tasks
|
||||
- **Parallel processing** — multiple agents can work simultaneously on different aspects of the same problem
|
||||
- **Human-like workflow decomposition** — mimics how human teams naturally divide complex analytical work
|
||||
|
||||
**Semantica's coordination approach:**
|
||||
Semantica coordinates agents through shared context (memory and knowledge graphs) rather than message brokers or API calls between services. Agents read and write to the same underlying data structures, enabling seamless information sharing without complex middleware.
|
||||
|
||||
**Single-agent vs multi-agent architectures:**
|
||||
- **Single-agent** — one `AgentContext` handles all tasks from ingestion through final output
|
||||
- **Multi-agent** — multiple `AgentContext` instances or namespaced workflows, each responsible for specific pipeline stages or analytical roles
|
||||
|
||||
## Why Use Multi-Agent Systems?
|
||||
|
||||
**Separation of responsibilities.** Divide complex workflows into focused, manageable stages where each agent excels at its specific domain without being overwhelmed by tangential concerns.
|
||||
|
||||
**Scalability of complex workflows.** Handle sophisticated analytical pipelines that require different expertise areas, processing speeds, and reasoning approaches without creating unwieldy monolithic agents.
|
||||
|
||||
**Independent reasoning stages.** Enable different agents to use different LLMs, prompts, confidence thresholds, and reasoning strategies optimized for their specific tasks rather than compromising on a one-size-fits-all approach.
|
||||
|
||||
**Specialized agent roles.** Create agents tailored for ingestion, enrichment, analysis, synthesis, and reporting—each with role-appropriate configurations and capabilities.
|
||||
|
||||
**Shared knowledge and evidence.** Multiple agents contribute to and benefit from the same knowledge graph and memory stores, creating a cumulative evidence base that improves as more agents contribute their findings.
|
||||
|
||||
**Human-like workflow decomposition.** Mirror natural human team structures where analysts, researchers, and decision-makers each contribute specialized expertise to collaborative analytical processes.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use multi-agent systems for:**
|
||||
- Complex analytical workflows requiring multiple stages (research → analysis → synthesis → reporting)
|
||||
- Multi-stage processing pipelines with distinct phases that benefit from specialized approaches
|
||||
- Research and investigation workflows where different agents handle different information sources or analytical methods
|
||||
- Teams of specialized agents with different roles (OSINT collector, enrichment analyst, fusion officer)
|
||||
- Long-running workflows where different agents may operate at different times or schedules
|
||||
- Scenarios requiring different LLMs, reasoning approaches, or confidence thresholds for different analytical stages
|
||||
|
||||
**Do NOT use multi-agent systems for:**
|
||||
- Simple document summarization or single-step information retrieval tasks
|
||||
- Linear workflows where one agent can handle all steps effectively without specialization benefits
|
||||
- Small, straightforward tasks where the coordination overhead exceeds the complexity of the core work
|
||||
- Cases where a single agent with appropriate configuration can handle the entire workflow efficiently
|
||||
|
||||
**Important consideration:** Multi-agent systems introduce additional architectural complexity including state management, coordination patterns, and debugging challenges. Only choose multi-agent approaches when the benefits of specialization and separation of concerns outweigh this added complexity.
|
||||
|
||||
Semantica coordinates multiple agents through a shared `ContextGraph` — agents read and write to the same graph, or hand off serialized state via `save()` and `load()`, with no message broker required. Use this pattern when splitting work across ingestion, enrichment, reasoning, and reporting roles that must share a single evidence base.
|
||||
|
||||
<Info>
|
||||
@@ -13,17 +62,17 @@ Semantica coordinates multiple agents through a shared `ContextGraph` — agents
|
||||
|
||||
Before writing any code, choose the right coordination pattern for your pipeline.
|
||||
|
||||
**Shared graph** works when all agents run in the same process. They hold references to the same `ContextGraph` object — thread-safe by default — so every `store()` from one agent is immediately visible to every `retrieve()` from another. This is the lowest-latency option and the right default for in-process pipelines.
|
||||
**Shared Graph Pattern:** Multiple agents share references to the same `ContextGraph` and `VectorStore` objects within a single process. This provides the lowest latency since all agents see changes immediately, with built-in thread safety for concurrent access. Choose this when agents run simultaneously in the same application and need real-time access to each other's contributions.
|
||||
|
||||
**Save / load handoff** works when agents run in different processes, on different machines, or at different times. Agent A finishes its work, calls `context.save(path)`, and Agent B calls `context.load(path)` to pick up exactly where A left off — full memory, full graph, full vector index. This is how you implement shift handoffs, async pipelines, and cross-service orchestration.
|
||||
**Save / Load Handoff Pattern:** Agents run in different processes, containers, or at different times. The first agent completes its work and calls `context.save(path)` to serialize its complete state. The next agent calls `context.load(path)` to restore exactly where the previous agent left off, including full memory, graph data, and vector indices. Choose this for distributed systems, scheduled workflows, or when agents run on different machines that require shared storage access.
|
||||
|
||||
**Namespaced memories** works when you have a single `AgentContext` instance serving multiple logical agents, each scoping its reads and writes with a `conversation_id`. Agents are isolated by tag, not by instance — useful for lightweight role separation without the overhead of multiple contexts.
|
||||
**Namespaced Memory Pattern:** A single `AgentContext` serves multiple logical agents, with each agent scoping its reads and writes using unique `conversation_id` values. Agents remain isolated by namespace rather than by separate context instances. Choose this for lightweight role separation without the resource overhead of maintaining multiple complete contexts.
|
||||
|
||||
The pipeline in this guide uses all three.
|
||||
|
||||
## Pattern 1 — Shared Graph for Concurrent Ingestion
|
||||
|
||||
The OSINT collector and the enrichment agent run concurrently. They share a single `ContextGraph` and a single `VectorStore` — the graph's internal `RLock` makes concurrent writes safe.
|
||||
The OSINT (**Open Source Intelligence** — publicly available information) collector and the enrichment agent run concurrently. They share a single `ContextGraph` and a single `VectorStore` — the graph's internal `RLock` makes concurrent writes safe.
|
||||
|
||||
```python
|
||||
import threading
|
||||
@@ -71,7 +120,7 @@ def osint_collection():
|
||||
],
|
||||
extract_entities=True,
|
||||
extract_relationships=True,
|
||||
conversation_id="osint-pipeline",
|
||||
conversation_id="osint-pipeline", # namespace acts as agent identifier
|
||||
)
|
||||
```
|
||||
|
||||
@@ -93,7 +142,7 @@ def enrichment():
|
||||
],
|
||||
extract_entities=True,
|
||||
extract_relationships=True,
|
||||
conversation_id="enrichment-pipeline",
|
||||
conversation_id="enrichment-pipeline", # separate namespace from OSINT agent
|
||||
)
|
||||
```
|
||||
|
||||
@@ -113,6 +162,8 @@ t1.join(); t2.join()
|
||||
|
||||
The reasoning agent runs after ingestion completes. In a production pipeline this might be a separate process, a different container, or a scheduled job. The ingestion agents save their shared state; the reasoning agent loads it.
|
||||
|
||||
**Important deployment note:** When agents run in different containers or on different machines, they must have access to the same saved state location through shared storage (network file systems, cloud storage, or shared volumes).
|
||||
|
||||
```python
|
||||
# After ingestion: save the combined graph and vector index
|
||||
osint_agent.save("./pipeline/enriched_intel/")
|
||||
@@ -131,7 +182,7 @@ from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
from semantica.llms import LiteLLM
|
||||
|
||||
# Create a fresh context before loading — load() merges into the existing context
|
||||
# Create a context to load the checkpoint into — load() will overwrite existing state
|
||||
reasoning_vs = VectorStore(backend="faiss", dimension=768)
|
||||
reasoning_graph = ContextGraph(advanced_analytics=True)
|
||||
reasoning_agent = AgentContext(
|
||||
@@ -182,12 +233,17 @@ reasoning_agent.save("./pipeline/synthesis_output/")
|
||||
```
|
||||
|
||||
<Info>
|
||||
`load()` merges into the existing context — it does not wipe it first. Always create a fresh `AgentContext` before calling `load()` if you want a clean restore from a handoff checkpoint.
|
||||
`load()` overwrites the existing context — it clears current memory, graph, and vector state before loading. Any unsaved data in the context prior to calling `load()` will be lost.
|
||||
</Info>
|
||||
|
||||
## Pattern 3 — Namespaced Memories for Role Separation
|
||||
|
||||
The reporting agent does not need its own graph instance. It shares the reasoning agent's context but scopes its writes to its own namespace — the `conversation_id` acts as an agent identifier.
|
||||
The reporting agent does not need its own graph instance. It shares the reasoning agent's context but scopes its writes to its own namespace — the `conversation_id` acts as an agent identifier to separate memory streams and prevent contamination between different logical agents.
|
||||
|
||||
**Namespace isolation with conversation_id:**
|
||||
- `conversation_id` creates separate memory namespaces within the same `AgentContext`
|
||||
- Each agent's memories remain isolated unless explicitly queried across namespaces
|
||||
- Prevents accidental memory contamination when different logical agents work on related but distinct tasks
|
||||
|
||||
```python
|
||||
# The reporting agent loads the synthesis output
|
||||
@@ -215,7 +271,7 @@ for item in synthesis_items:
|
||||
# Store the final report under the reporting agent's own namespace
|
||||
reporting_agent.store(
|
||||
"\n\n".join(brief_sections),
|
||||
metadata={"type": "finished_report", "classification": "TLP:GREEN"},
|
||||
metadata={"type": "finished_report", "classification": "TLP:GREEN"}, # TLP (Traffic Light Protocol) — information sharing guidelines
|
||||
conversation_id="reporting-output", # reporting agent's namespace
|
||||
user_id="reporting_agent",
|
||||
)
|
||||
@@ -227,11 +283,27 @@ print("Pipeline produced {} traceable context items".format(len(full_trail)))
|
||||
|
||||
Each agent's contributions are retrievable individually by filtering on `conversation_id`, or collectively by querying without a filter.
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Forgetting conversation_id namespaces.** Without unique `conversation_id` values, different agents' memories mix together, making it impossible to trace which agent contributed which insights. Always use distinct, meaningful conversation IDs for each logical agent.
|
||||
|
||||
**Accidental state loss with load().** The `load()` function overwrites existing context rather than merging it. If you have unsaved state in an `AgentContext`, calling `load()` will wipe it. Always save your current state or use a fresh context before loading a checkpoint.
|
||||
|
||||
**Using Shared Graph across separate processes.** The Shared Graph pattern only works within a single process where agents share object references. For distributed agents running in different containers or machines, use the Save/Load Handoff pattern instead.
|
||||
|
||||
**Assuming save/load works without shared storage.** Agents in different processes, containers, or machines must have access to the same filesystem location for save/load handoffs. Ensure shared storage (NFS, cloud storage, shared volumes) is properly configured.
|
||||
|
||||
**Overengineering simple workflows with multiple agents.** Multi-agent systems add coordination complexity and potential failure points. For straightforward single-step tasks, a simple single-agent approach is often more reliable and easier to debug.
|
||||
|
||||
**Mixing agent responsibilities excessively.** Each agent should have a clear, focused role. Agents that try to do too many different tasks lose the benefits of specialization and become harder to optimize, debug, and maintain.
|
||||
|
||||
**Ignoring memory isolation boundaries.** When using namespaced memories, be careful about queries that span multiple `conversation_id` values. Unscoped queries can accidentally retrieve memories from other agents, breaking logical isolation.
|
||||
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
<Tab title="Defense — CTI/Threat">
|
||||
A three-agent intelligence fusion cell: an OSINT collector ingests public feeds, a HUMINT analyst loads classified summaries, and a fusion officer synthesizes both streams into a Priority Intelligence Requirement answer. The OSINT and HUMINT agents run concurrently on a shared graph; the fusion officer loads the combined state in a separate process on an air-gapped network segment.
|
||||
A three-agent intelligence fusion cell: an OSINT collector ingests public feeds, a HUMINT (**Human Intelligence** — information gathered from human sources) analyst loads classified summaries, and a fusion officer synthesizes both streams into a PIR (**Priority Intelligence Requirement** — critical information needed for decision-making) answer. The OSINT and HUMINT agents run concurrently on a shared graph; the fusion officer loads the combined state in a separate process on an **air-gapped environment** (isolated network with no internet connectivity for security).
|
||||
|
||||
```python
|
||||
import threading
|
||||
|
||||
+161
-35
@@ -4,17 +4,92 @@ description: "Define, version, and enforce governance policies over knowledge gr
|
||||
icon: "scale-balanced"
|
||||
---
|
||||
|
||||
## What Is Policy Engine?
|
||||
|
||||
Policy evaluation is the systematic checking of decisions against predefined governance rules and constraints. Unlike application enforcement that automatically blocks non-compliant actions, policy evaluation provides compliance status that can trigger different workflows—approval processes, exception handling, or audit requirements.
|
||||
|
||||
**Key policy concepts:**
|
||||
|
||||
**Policy evaluation** checks whether decisions meet defined criteria without automatically preventing actions, enabling flexible governance workflows.
|
||||
|
||||
**Governance and compliance workflows** use policy evaluation results to route decisions through appropriate approval chains, exception processes, or audit trails.
|
||||
|
||||
**Approval processes** can be triggered by policy violations, creating documented exception paths with justification and approver accountability.
|
||||
|
||||
**Difference from enforcement:** Policy evaluation returns compliance status (`True`/`False`) but does not automatically block actions. Your workflow determines what happens next—immediate approval, escalation, exception handling, or rejection.
|
||||
|
||||
## Why Use Policy Engine?
|
||||
|
||||
**Governance and accountability.** Create auditable decision workflows where every policy evaluation, exception, and approval is permanently recorded in the knowledge graph with full provenance tracking.
|
||||
|
||||
**Compliance verification.** Systematically check decisions against regulatory requirements, internal policies, and risk management rules before they are finalized or acted upon.
|
||||
|
||||
**Approval workflow orchestration.** Route non-compliant decisions through structured approval processes with documented justifications and multi-level sign-offs.
|
||||
|
||||
**Regulatory compliance.** Meet audit requirements by maintaining complete policy version histories, exception records, and compliance checking trails that regulators can inspect.
|
||||
|
||||
**Risk management.** Flag high-risk decisions for additional review while allowing routine compliant decisions to proceed with minimal friction.
|
||||
|
||||
**Policy evolution tracking.** Maintain version histories of policy changes with impact analysis, enabling evidence-based policy refinement and regulatory reporting.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use Policy Engine for:**
|
||||
- Governance workflows requiring structured approval processes and audit trails
|
||||
- Regulatory compliance where policy adherence must be documented and verifiable
|
||||
- Multi-level approval workflows for high-stakes decisions (financial approvals, security exceptions, clinical treatments)
|
||||
- Regulated environments where policy violations trigger specific escalation procedures
|
||||
- Risk management workflows where non-compliant decisions require additional oversight
|
||||
- Audit requirements demanding complete policy application and exception tracking
|
||||
|
||||
**Do NOT use Policy Engine for:**
|
||||
- Simple form validation or basic input checking—use standard validation libraries instead
|
||||
- Basic business rules that don't require audit trails or governance workflows
|
||||
- Low-stakes, high-throughput checks where policy evaluation overhead would impact performance
|
||||
- Deterministic rule checking that doesn't benefit from version tracking and approval processes
|
||||
- Real-time operational decisions where policy evaluation latency is unacceptable
|
||||
|
||||
**Warning:** Policy Engine adds governance overhead and requires careful workflow design. Only use when the benefits of structured policy management outweigh the additional complexity.
|
||||
|
||||
`PolicyEngine` enforces named policies against recorded decisions, returning `True` if the decision satisfies all policy rules. Use it to gate AI decisions at runtime — attributions requiring dual-source confirmation, escalations requiring senior approval, or any decision category where compliance must be verified before the outcome is recorded. Policies are versioned graph nodes, so every check, exception, and approval chain is part of the permanent audit trail.
|
||||
|
||||
<Info>
|
||||
The Policy Engine sits above `AgentContext` and `ContextGraph`. Policies are stored as nodes in the same graph as decisions, giving them the same causal tracing, provenance, and temporal validity as any other knowledge graph entity. `PolicyEngine` and `Policy` import from `semantica.context`. `Decision` imports from `semantica.context` (it is a dataclass defined in `semantica.context.decision_models`). `DecisionRecorder` imports from `semantica.context.decision_recorder`.
|
||||
The Policy Engine sits above `AgentContext` and `ContextGraph`. Policies are stored as nodes in the same graph as decisions, giving them the same causal tracing, provenance, and temporal validity as any other knowledge graph entity.
|
||||
|
||||
**Key objects:** `PolicyEngine` and `Policy` import from `semantica.context`. `Decision` is a dataclass with fields like `decision_id`, `category`, `scenario`, `reasoning`, `outcome`, `confidence`, `timestamp`, `decision_maker`, and `metadata`. `DecisionRecorder` imports from `semantica.context.decision_recorder` for approval workflow tracking.
|
||||
</Info>
|
||||
|
||||
## Supported Rule Types
|
||||
|
||||
The PolicyEngine implementation supports specific rule patterns that evaluate decision attributes and metadata:
|
||||
|
||||
**Confidence rules:**
|
||||
- `min_confidence: 0.85` — `decision.confidence >= 0.85`
|
||||
|
||||
**Outcome validation:**
|
||||
- `allowed_outcomes: ["approved", "approved_with_conditions"]` — `decision.outcome` must be in the list
|
||||
|
||||
**Category validation:**
|
||||
- `required_categories: ["credit_risk", "operational_risk"]` — `decision.category` must be in the list
|
||||
|
||||
**Metadata field rules:**
|
||||
- `min_*: value` — metadata field must be `>= value` (e.g., `min_credit_score: 680`)
|
||||
- `max_*: value` — metadata field must be `<= value` (e.g., `max_ltv: 0.85`)
|
||||
- `required_*: value` — metadata field must equal `value` (string) or contain all items (list)
|
||||
|
||||
**Field lookup behavior:** For rule `min_credit_score`, the engine checks `metadata["credit_score"]`, then `metadata["*_credit_score"]` (suffix match), then `decision.credit_score` attribute.
|
||||
|
||||
**Important:** The following rule types are NOT supported and will cause unexpected behavior:
|
||||
- `disallowed_outcomes` (use `allowed_outcomes` instead)
|
||||
- `mandatory_fields` (use `required_*` for specific fields)
|
||||
- `requires_mfa` (use metadata field checks like `required_mfa_verified`)
|
||||
- Complex nested conditions or operators
|
||||
|
||||
---
|
||||
|
||||
## Defining the policy
|
||||
|
||||
A `Policy` is a dataclass with a free-form `rules` dict — encode whatever your domain requires.
|
||||
A `Policy` is a dataclass with a free-form `rules` dict — encode whatever your domain requires using supported rule patterns.
|
||||
|
||||
```python
|
||||
from semantica.context import ContextGraph, PolicyEngine, Policy
|
||||
@@ -34,9 +109,11 @@ attribution_policy = Policy(
|
||||
rules = {
|
||||
"min_independent_sources": 2,
|
||||
"required_approver_role": "senior_analyst",
|
||||
"disallowed_outcomes": ["nation_state_attributed_single_source"],
|
||||
"allowed_outcomes": ["nation_state_attributed_dual_source"],
|
||||
"min_confidence": 0.85,
|
||||
"mandatory_fields": ["source_a", "source_b", "approver"],
|
||||
"required_source_a": True,
|
||||
"required_source_b": True,
|
||||
"required_approver": True,
|
||||
},
|
||||
category = "threat_attribution",
|
||||
version = "1.0.0",
|
||||
@@ -74,14 +151,23 @@ decision = Decision(
|
||||
confidence = 0.91,
|
||||
timestamp = datetime.utcnow(),
|
||||
decision_maker= "ai_threat_analyst_v3",
|
||||
metadata = {
|
||||
"independent_sources": 1, # Below min_independent_sources requirement
|
||||
"approver_role": "analyst", # Below required_approver_role
|
||||
"source_a": True, # Has first source
|
||||
# Missing source_b and approver fields
|
||||
}
|
||||
)
|
||||
|
||||
is_compliant = engine.check_compliance(decision, policy_id)
|
||||
print(f"Compliant: {is_compliant}")
|
||||
# Compliant: False
|
||||
#
|
||||
# The outcome "nation_state_attributed_single_source" is in disallowed_outcomes.
|
||||
# The policy requires min_independent_sources=2 — the decision only cited one.
|
||||
# Multiple rule violations:
|
||||
# - outcome "nation_state_attributed_single_source" not in allowed_outcomes
|
||||
# - independent_sources (1) < min_independent_sources (2)
|
||||
# - approver_role "analyst" != required_approver_role "senior_analyst"
|
||||
# - missing required_source_b and required_approver fields
|
||||
```
|
||||
|
||||
The engine returns `False`. The decision has not been rejected — it has been flagged. What happens next depends on your workflow. In some organisations, a non-compliant result simply blocks the write to the authoritative graph. In others, it triggers an exception process where a human approver reviews the evidence and signs off.
|
||||
@@ -174,7 +260,7 @@ The impact dict contains per-decision detail, not just the count. You can inspec
|
||||
The lead decides to proceed with the threshold increase. She updates the policy to version 1.1.0, recording her reason. The old version is preserved in the history.
|
||||
|
||||
```python
|
||||
updated_policy_id = engine.update_policy(
|
||||
engine.update_policy(
|
||||
policy_id = policy_id,
|
||||
rules = {**current_policy.rules, "min_confidence": 0.92},
|
||||
change_reason = "Q3 attribution quality review — raise confidence floor from 0.85 to 0.92 "
|
||||
@@ -182,8 +268,8 @@ updated_policy_id = engine.update_policy(
|
||||
new_version = "1.1.0",
|
||||
)
|
||||
|
||||
print(f"Policy updated: {updated_policy_id} -> version 1.1.0")
|
||||
# Policy updated: pol-attr-001 -> version 1.1.0
|
||||
print(f"Policy updated: {policy_id} to version 1.1.0")
|
||||
# Policy updated: pol-attr-001 to version 1.1.0
|
||||
|
||||
# Find all decisions that were evaluated under v1.0.0 —
|
||||
# these need to be re-reviewed to confirm they still meet the new standard.
|
||||
@@ -225,6 +311,24 @@ for version in history:
|
||||
|
||||
---
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Assuming failed compliance automatically blocks actions.** PolicyEngine returns compliance status but does NOT automatically prevent actions. Your workflow must check the returned boolean and decide what happens next—approval, rejection, exception handling, or escalation.
|
||||
|
||||
**Using unsupported rule keys.** The implementation only supports specific patterns: `min_*`, `max_*`, `required_*`, `min_confidence`, `allowed_outcomes`, and `required_categories`. Any other rule key falls back to a key-presence check: it passes only if that exact key exists in `decision.metadata`, regardless of its value. This means keys like `disallowed_outcomes` will silently **fail** compliance whenever that literal key is absent from metadata (the common case), and will silently **pass** — regardless of the actual outcome — if a `disallowed_outcomes` key happens to exist in metadata with any value. Neither behavior matches the intended "outcome must not be in this list" semantics — use `allowed_outcomes` instead.
|
||||
|
||||
**Treating exceptions as approvals.** Recording a policy exception with `record_exception()` does NOT automatically make a non-compliant decision compliant. Exceptions are audit trail entries—your workflow must still decide whether to proceed with the non-compliant decision.
|
||||
|
||||
**Assuming PolicyEngine modifies graph state automatically.** PolicyEngine only evaluates compliance and records policy applications, exceptions, and approval chains. It does not modify decision outcomes, metadata, or prevent actions—that is your workflow's responsibility.
|
||||
|
||||
**Using complex nested rule structures.** The implementation does not support complex conditional logic, nested operators, or arbitrary expressions. Keep rules simple: single field comparisons, list membership checks, and threshold validations only.
|
||||
|
||||
**Missing metadata for rule evaluation.** Rules like `min_credit_score` require the corresponding metadata field (`credit_score`) to be present in `decision.metadata`. Missing metadata fields cause rule evaluation to fail, making the decision non-compliant.
|
||||
|
||||
**Forgetting to check rule evaluation results.** Always handle both compliant and non-compliant cases explicitly. Non-compliant decisions that proceed without proper exception handling create audit gaps and governance risks.
|
||||
|
||||
---
|
||||
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
@@ -247,10 +351,11 @@ opsec_policy = Policy(
|
||||
name = "TLP:RED — Restricted Dissemination",
|
||||
description = "TLP:RED intelligence must not be shared outside the originating organisation",
|
||||
rules = {
|
||||
"classification": "TLP:RED",
|
||||
"disallowed_outcomes": ["shared_with_partner", "published"],
|
||||
"min_confidence": 0.95,
|
||||
"mandatory_fields": ["tlp", "classification", "authorised_recipients"],
|
||||
"required_classification": "TLP:RED",
|
||||
"allowed_outcomes": ["retained_internal", "escalated_internal"],
|
||||
"min_confidence": 0.95,
|
||||
"required_tlp": True,
|
||||
"required_authorised_recipients": True,
|
||||
},
|
||||
category = "information_sharing",
|
||||
version = "2.1.0",
|
||||
@@ -264,15 +369,20 @@ decision = Decision(
|
||||
category = "information_sharing",
|
||||
scenario = "APT29 SIGINT report TLP:RED — share with Five Eyes partners?",
|
||||
reasoning = "Tactical intelligence — partner request via UKIC liaison",
|
||||
outcome = "shared_with_partner", # violates TLP:RED policy
|
||||
confidence = 0.88,
|
||||
outcome = "shared_with_partner", # violates allowed_outcomes policy
|
||||
confidence = 0.88, # below min_confidence threshold
|
||||
timestamp = datetime.utcnow(),
|
||||
decision_maker= "analyst_rodriguez",
|
||||
metadata = {
|
||||
"classification": "TLP:RED",
|
||||
"tlp": True,
|
||||
"authorised_recipients": True,
|
||||
}
|
||||
)
|
||||
|
||||
is_compliant = engine.check_compliance(decision, "pol-opsec-001")
|
||||
print(f"Compliant: {is_compliant}")
|
||||
# Compliant: False — outcome 'shared_with_partner' is disallowed; confidence below 0.95
|
||||
# Compliant: False — outcome 'shared_with_partner' not in allowed_outcomes; confidence below 0.95
|
||||
|
||||
if not is_compliant:
|
||||
# Route to J2 for exception review — dual commander approval required
|
||||
@@ -286,7 +396,7 @@ if not is_compliant:
|
||||
recorder.record_approval_chain(
|
||||
decision_id = decision.decision_id,
|
||||
approvers = ["j2_officer_hayes", "unit_commander_brooks"],
|
||||
methods = ["secure_phone", "in_person"],
|
||||
methods = ["email", "zoom_call"],
|
||||
contexts = ["J2 tactical review", "Commander emergency approval"],
|
||||
)
|
||||
print(f"Exception recorded with dual-commander approval: {exception_id}")
|
||||
@@ -315,9 +425,9 @@ for pol in [
|
||||
name = "MFA Required — All Tier-1",
|
||||
description = "Every Tier-1 access decision must verify MFA",
|
||||
rules = {
|
||||
"requires_mfa": True,
|
||||
"disallowed_outcomes": ["access_granted_without_mfa"],
|
||||
"min_confidence": 0.90,
|
||||
"required_mfa_verified": True,
|
||||
"allowed_outcomes": ["access_granted_with_mfa"],
|
||||
"min_confidence": 0.90,
|
||||
},
|
||||
category = "access_control",
|
||||
version = "1.0.0",
|
||||
@@ -329,10 +439,10 @@ for pol in [
|
||||
name = "PAM Checkout — Privileged Accounts",
|
||||
description = "Privileged account use requires PAM session checkout",
|
||||
rules = {
|
||||
"requires_pam": True,
|
||||
"session_recording": True,
|
||||
"max_session_hours": 4,
|
||||
"disallowed_outcomes": ["privileged_access_granted_no_pam"],
|
||||
"required_pam_session": True,
|
||||
"required_session_recording": True,
|
||||
"max_session_hours": 4,
|
||||
"allowed_outcomes": ["privileged_access_granted_with_pam"],
|
||||
},
|
||||
category = "privileged_access",
|
||||
version = "1.0.0",
|
||||
@@ -352,6 +462,11 @@ decision = Decision(
|
||||
confidence = 0.78,
|
||||
timestamp = datetime.utcnow(),
|
||||
decision_maker= "soc_automation",
|
||||
metadata = {
|
||||
"pam_session": False, # PAM checkout failed
|
||||
"session_recording": True, # Manual recording in place
|
||||
"session_hours": 3, # Planned session duration
|
||||
}
|
||||
)
|
||||
|
||||
pam_compliant = engine.check_compliance(decision, "pol-zt-pam")
|
||||
@@ -396,11 +511,10 @@ safety_policy = Policy(
|
||||
name = "Metformin Absolute Contraindication — eGFR < 30",
|
||||
description = "Metformin must not be prescribed when eGFR is below 30 ml/min/1.73m²",
|
||||
rules = {
|
||||
"contraindicated_drug": "metformin",
|
||||
"contraindication_condition": {"egfr": {"operator": "<", "threshold": 30}},
|
||||
"disallowed_outcomes": ["metformin_prescribed", "metformin_continued"],
|
||||
"requires_clinician_sign_off": True,
|
||||
"mandatory_checks": ["egfr_measured_within_90_days"],
|
||||
"min_egfr": 30, # eGFR must be >= 30
|
||||
"allowed_outcomes": ["metformin_discontinued", "metformin_contraindicated", "alternative_prescribed"],
|
||||
"required_clinician_sign_off": True,
|
||||
"required_egfr_check": True,
|
||||
},
|
||||
category = "clinical_safety",
|
||||
version = "3.0.0", # aligned to BNF 2024
|
||||
@@ -420,6 +534,12 @@ decision = Decision(
|
||||
confidence = 0.97,
|
||||
timestamp = datetime.utcnow(),
|
||||
decision_maker= "cdss_v4",
|
||||
metadata = {
|
||||
"egfr": 28, # Below minimum threshold
|
||||
"clinician_sign_off": True,
|
||||
"egfr_check": True,
|
||||
"drug": "metformin",
|
||||
}
|
||||
)
|
||||
|
||||
is_compliant = engine.check_compliance(decision, "pol-clin-001")
|
||||
@@ -465,10 +585,8 @@ mortgage_policy = Policy(
|
||||
"max_ltv": 0.85,
|
||||
"max_dsti": 0.40,
|
||||
"min_credit_score": 680,
|
||||
"required_stress_test_bps": 300,
|
||||
"required_fields": ["ltv", "pd", "lgd", "dsti", "credit_score"],
|
||||
"disallowed_outcomes": ["approved_ltv_over_85", "approved_dsti_over_40"],
|
||||
"required_approvers_if_exception": ["senior_underwriter", "credit_committee"],
|
||||
"min_stress_test_bps": 300,
|
||||
"allowed_outcomes": ["approved", "approved_with_conditions"],
|
||||
},
|
||||
category = "credit_risk",
|
||||
version = "2.3.0",
|
||||
@@ -487,15 +605,23 @@ decision = Decision(
|
||||
"LTV 86% exceeds 85% cap. Stress test at +300bps passes. "
|
||||
"Credit score 710 above 680 floor. DSTI 38% within 40% limit."
|
||||
),
|
||||
outcome = "approved_ltv_over_85", # disallowed outcome — flags non-compliance
|
||||
outcome = "approved_ltv_exception", # not in allowed_outcomes — flags non-compliance
|
||||
confidence = 0.72,
|
||||
timestamp = datetime.utcnow(),
|
||||
decision_maker= "underwriting_model_v4",
|
||||
metadata = {
|
||||
"ltv": 0.86, # Exceeds max_ltv of 0.85
|
||||
"dsti": 0.38, # Within max_dsti of 0.40
|
||||
"credit_score": 710, # Above min_credit_score of 680
|
||||
"pd": 0.023, # Recorded for audit — no threshold rule in this policy
|
||||
"lgd": 0.45, # Recorded for audit — no threshold rule in this policy
|
||||
"stress_test_bps": 300,
|
||||
}
|
||||
)
|
||||
|
||||
is_compliant = engine.check_compliance(decision, "pol-credit-001")
|
||||
print(f"Compliant: {is_compliant}")
|
||||
# Compliant: False — 'approved_ltv_over_85' is in disallowed_outcomes
|
||||
# Compliant: False — ltv (0.86) > max_ltv (0.85) and outcome not in allowed_outcomes
|
||||
|
||||
if not is_compliant:
|
||||
exception_id = engine.record_exception(
|
||||
|
||||
+107
-19
@@ -4,6 +4,59 @@ description: "How Semantica tracks the origin and lineage of every entity, relat
|
||||
icon: "file-certificate"
|
||||
---
|
||||
|
||||
## What Is Provenance?
|
||||
|
||||
Provenance is the systematic recording of where data came from, how it was transformed, and who was responsible for each step in its lifecycle. Unlike ordinary graph metadata that simply describes entities, provenance creates an immutable audit trail that tracks the complete history of every piece of information in your system.
|
||||
|
||||
**Key provenance concepts:**
|
||||
|
||||
**Lineage** traces the chain of custody from original source through all transformations to the current state, showing exactly how data evolved over time.
|
||||
|
||||
**Source attribution** records the specific document, database, API call, or human input that produced each data element, enabling precise citation and verification.
|
||||
|
||||
**Integrity verification** uses cryptographic checksums to detect any unauthorized changes to provenance records after they were created.
|
||||
|
||||
**Audit trails** provide regulatory compliance by maintaining tamper-evident logs of all data operations, transformations, and decisions.
|
||||
|
||||
Provenance differs from simple metadata by creating legally defensible, cryptographically verifiable records that answer critical questions: "Where did this come from?", "Who processed it?", "When did it change?", and "Has it been tampered with?"
|
||||
|
||||
## Why Use Provenance?
|
||||
|
||||
**Compliance with regulatory requirements.** Meet FDA 21 CFR Part 11, ICH E6(R2) GCP, Basel III BCBS 239, and defense intelligence sharing agreements that mandate complete data traceability and electronic record integrity.
|
||||
|
||||
**Source attribution and citation.** Trace every entity, relationship, and property value back to its exact source document, API response, or human input for scientific reproducibility and legal defensibility.
|
||||
|
||||
**Auditability and transparency.** Provide auditors, regulators, and stakeholders with complete visibility into data processing workflows, including who performed each operation and when changes occurred.
|
||||
|
||||
**Conflict resolution and data quality.** When multiple sources provide different values for the same property, provenance records enable evidence-based conflict resolution by comparing source credibility, recency, and confidence levels.
|
||||
|
||||
**Tamper detection and forensics.** Cryptographic integrity verification detects unauthorized modifications to data records, supporting incident response and forensic analysis in security-sensitive environments.
|
||||
|
||||
**Traceability for data lineage.** Answer complex questions about data ancestry, especially in multi-stage processing pipelines where entities undergo extraction, enrichment, fusion, and analysis transformations.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use provenance tracking for:**
|
||||
- Regulated environments requiring audit trails (healthcare, finance, defense, pharmaceuticals)
|
||||
- Multi-source data fusion where conflicting information must be resolved with evidence
|
||||
- Long-lived knowledge graphs where data quality and source credibility matter
|
||||
- Production systems where data integrity and tamper detection are critical
|
||||
- Complex processing pipelines where entities undergo multiple transformations
|
||||
- Situations requiring legal defensibility of decisions based on extracted data
|
||||
|
||||
**Provenance may be unnecessary for:**
|
||||
- Simple prototypes and proof-of-concept demonstrations where compliance is not required
|
||||
- Ephemeral workflows that process data once and discard results immediately
|
||||
- Stateless applications that don't persist data across sessions
|
||||
- Internal research projects with trusted single-source data
|
||||
- High-frequency, low-latency operations where provenance overhead impacts performance
|
||||
- Scenarios where all data comes from a single, highly trusted source that never changes
|
||||
|
||||
**Consider simpler alternatives when:**
|
||||
- Basic metadata (creation timestamp, source file name) provides sufficient traceability
|
||||
- Data processing is transparent and reproducible through version control alone
|
||||
- Regulatory compliance does not require cryptographic integrity verification
|
||||
|
||||
`ProvenanceManager` records a W3C PROV-O compliant entry for every entity, relationship, document chunk, and property value — with a SHA-256 checksum for tamper detection and automatic version chaining on every `track_entity()` call. Use it when you need to answer regulatory questions about where a value came from, who wrote it, and whether it has changed since first ingestion.
|
||||
|
||||
<Info>
|
||||
@@ -30,9 +83,13 @@ prov = ProvenanceManager(storage=SQLiteStorage("audit.db"))
|
||||
|
||||
For any regulated deployment — security operations, clinical data, financial risk — use `storage_path`. A SQLite file can be backed up, versioned, and queried with standard tools without requiring a server.
|
||||
|
||||
<Note>
|
||||
`SQLiteStorage` automatically configures Write-Ahead Logging (`WAL`), `busy_timeout=5000`, and `synchronous=NORMAL`, and executes read-modify-write operations (like `track_entity()`) in atomic immediate transactions (`BEGIN IMMEDIATE`); plain reads (`retrieve()`, `trace_lineage()`) use a separate connection without an explicit write lock so they don't serialize behind writers. Furthermore, `ProvenanceManager` automatically supports custom storage backends overriding only `trace_lineage(self, entity_id)` without requiring `max_depth` in their signature.
|
||||
</Note>
|
||||
|
||||
## Recording provenance when ingesting data
|
||||
|
||||
The moment data enters your graph is the moment provenance must be recorded. `track_entity()` captures the source document, the timestamp, the operator or pipeline that ran the extraction, a verbatim quote from the source, and a confidence score. It returns a `ProvenanceEntry` with a SHA-256 checksum computed automatically.
|
||||
The moment data enters your graph is the moment provenance must be recorded. `track_entity()` captures the source document, the timestamp, the operator or pipeline that ran the extraction, a verbatim quote from the source, and a confidence score. It returns an `Optional[ProvenanceEntry]` (`ProvenanceEntry` on success, or `None` if storage fails on a brand-new entity) with a SHA-256 checksum computed automatically.
|
||||
|
||||
```python
|
||||
# Ingesting CVE-2024-3400 from NVD and a commercial feed
|
||||
@@ -51,7 +108,6 @@ entry_nvd = prov.track_entity(
|
||||
activity_id="nvd_feed_ingestion",
|
||||
source_location="CVE-2024-3400 JSON record",
|
||||
source_quote='{"cvssMetricV31":[{"cvssData":{"baseScore":10.0}}]}',
|
||||
agent_id="nvd_ingest_pipeline_v2",
|
||||
)
|
||||
|
||||
print(f"Entity tracked : {entry_nvd.entity_id}")
|
||||
@@ -83,7 +139,6 @@ entry_commercial = prov.track_entity(
|
||||
confidence=0.91,
|
||||
entity_type="vulnerability",
|
||||
activity_id="commercial_feed_ingestion",
|
||||
agent_id="threat_ingest_pipeline_v2",
|
||||
)
|
||||
|
||||
# The NVD entry is now archived as cve-2024-3400:v:2024-04-12T14:22:07
|
||||
@@ -97,6 +152,8 @@ This version chaining happens automatically. You do not need to manage history e
|
||||
|
||||
When the same property appears in multiple sources with different values — exactly the CVE score situation — use `track_property_source()` to record each attribution separately. This feeds directly into conflict detection downstream: the conflict module can compare all tracked values for a property and surface disagreements with full source metadata attached.
|
||||
|
||||
**SourceReference** is a structured metadata container that captures exactly where a piece of information came from within a document. It includes the document identifier, specific location (page, section, byte range), confidence level, and custom metadata fields for domain-specific attribution requirements.
|
||||
|
||||
```python
|
||||
from semantica.provenance.schemas import SourceReference
|
||||
|
||||
@@ -134,7 +191,7 @@ When the regulator asks "where did the 9.8 come from?", this is the answer: `com
|
||||
|
||||
## Tracing the lineage of a node
|
||||
|
||||
Six months after ingestion, run a lineage trace. `get_lineage()` returns the full version chain — every state the entity has passed through, oldest to newest — along with summary metadata:
|
||||
Once you have multiple provenance entries for an entity, you can trace its complete history to understand how it evolved over time. Six months after ingestion, run a lineage trace. `get_lineage()` returns the full version chain — every state the entity has passed through, oldest to newest — along with summary metadata:
|
||||
|
||||
```python
|
||||
lineage = prov.get_lineage("cve-2024-3400")
|
||||
@@ -161,16 +218,16 @@ Sources seen : ['NVD_feed_2024-04-12', 'commercial_feed_2024-04-12',
|
||||
'NVD_feed_2024-07-18', 'commercial_feed_2024-10-08']
|
||||
|
||||
Full version chain (oldest → newest):
|
||||
[2024-04-12T14:22:07] agent=nvd_ingest_pipeline_v2
|
||||
[2024-04-12T14:22:07] agent=semantica
|
||||
source=NVD_feed_2024-04-12
|
||||
activity=nvd_feed_ingestion
|
||||
[2024-04-12T15:18:33] agent=threat_ingest_pipeline_v2
|
||||
[2024-04-12T15:18:33] agent=semantica
|
||||
source=commercial_feed_2024-04-12
|
||||
activity=commercial_feed_ingestion
|
||||
[2024-07-18T08:04:11] agent=nvd_ingest_pipeline_v2
|
||||
[2024-07-18T08:04:11] agent=semantica
|
||||
source=NVD_feed_2024-07-18
|
||||
activity=nvd_feed_ingestion # NVD updated their score
|
||||
[2024-10-08T09:11:44] agent=threat_ingest_pipeline_v2
|
||||
[2024-10-08T09:11:44] agent=semantica
|
||||
source=commercial_feed_2024-10-08
|
||||
activity=commercial_feed_ingestion
|
||||
```
|
||||
@@ -179,7 +236,9 @@ The chain answers all three of the regulator's questions. The 9.8 came from `com
|
||||
|
||||
## Verifying integrity
|
||||
|
||||
Every `ProvenanceEntry` carries a SHA-256 checksum computed at write time. If any field is modified after the fact — by a misconfigured pipeline, a database migration, or deliberate tampering — the checksum will not match on recomputation. Run integrity checks as part of any compliance audit:
|
||||
Every `ProvenanceEntry` carries a SHA-256 checksum computed at write time. If any field is modified after the fact — by a misconfigured pipeline, a database migration, or deliberate tampering — the checksum will not match on recomputation.
|
||||
|
||||
Integrity verification is critical for regulatory compliance and forensic analysis. Run integrity checks as part of any compliance audit:
|
||||
|
||||
```python
|
||||
from semantica.provenance.integrity import compute_checksum
|
||||
@@ -206,7 +265,9 @@ A `TAMPERED` status means the stored hash does not match what would be computed
|
||||
|
||||
## Tracking document chunks and their children
|
||||
|
||||
Provenance is not just for entities. When a document is split into chunks for RAG or NLP processing, each chunk needs its own provenance record linking it to the source file and byte range. Child chunks (from recursive splitting) link to their parent via `parent_chunk_id`, which maps to `prov:wasDerivedFrom` in the W3C model:
|
||||
Provenance is not just for entities. When a document is split into chunks for retrieval-augmented generation (RAG) or natural language processing workflows, each chunk needs its own provenance record linking it to the source file and byte range.
|
||||
|
||||
Child chunks (from recursive splitting) link to their parent via `parent_chunk_id`, which maps to `prov:wasDerivedFrom` in the W3C PROV-O standard:
|
||||
|
||||
```python
|
||||
# Track the parent chunk (a section of an advisory PDF)
|
||||
@@ -260,6 +321,22 @@ Unique sources : 12
|
||||
|
||||
This summary is the starting point for a compliance attestation: you can state the total number of tracked records, the number of distinct data sources, and the breakdown by record type.
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Provenance does not guarantee truth.** Provenance records faithfully track where information came from and how it was processed, but it cannot verify that the original sources were accurate. A perfectly documented chain from a flawed or malicious source still produces unreliable data.
|
||||
|
||||
**Reusing generic source identifiers.** Using non-specific source IDs like "daily_feed" or "batch_001" makes it impossible to trace individual records back to their exact origins. Always include timestamps, version numbers, or unique batch identifiers in source document names.
|
||||
|
||||
**Bypassing provenance workflows.** Manually inserting data or using ad-hoc scripts that skip `track_entity()` calls creates gaps in the audit trail. Ensure all data entry points—automated pipelines, manual corrections, and administrative operations—record appropriate provenance.
|
||||
|
||||
**Ignoring lineage verification.** Provenance chains can become complex in multi-stage processing pipelines. Regularly verify that `get_lineage()` and `trace_lineage()` return complete, logical chains without missing links or circular references.
|
||||
|
||||
**Overusing provenance in low-value scenarios.** Recording provenance for every intermediate calculation or temporary variable creates storage overhead without compliance benefit. Focus provenance tracking on entities, relationships, and properties that have legal, regulatory, or business significance.
|
||||
|
||||
**Failing to validate integrity checksums.** Cryptographic integrity verification only works if you actually check it. Include regular `compute_checksum()` validation in audit workflows and incident response procedures.
|
||||
|
||||
**Mixing provenance granularities.** Tracking some entities at the document level and others at the sentence level creates inconsistent audit trails. Establish consistent granularity standards for each data type and processing workflow.
|
||||
|
||||
## Domain examples
|
||||
|
||||
<Tabs>
|
||||
@@ -297,7 +374,6 @@ prov.track_entity(
|
||||
entity_type="threat_actor",
|
||||
activity_id="ner_extraction",
|
||||
source_location="paragraph_3",
|
||||
agent_id="analyst_ALPHA",
|
||||
)
|
||||
|
||||
# Tier 3: Campaign relationship from all-source fusion
|
||||
@@ -307,7 +383,6 @@ prov.track_relationship(
|
||||
metadata={"type": "operates", "confidence": 0.81},
|
||||
confidence=0.81,
|
||||
activity_id="all_source_fusion",
|
||||
agent_id="fusion_cell_BRAVO",
|
||||
)
|
||||
|
||||
# Tier 4: Property from two independent INT sources
|
||||
@@ -361,7 +436,6 @@ prov.track_entity(
|
||||
confidence=0.98,
|
||||
entity_type="vulnerability",
|
||||
activity_id="nvd_feed_ingestion",
|
||||
agent_id="ingest_pipeline_v2",
|
||||
)
|
||||
|
||||
# Six weeks later: NVD revised the score after PoC publication
|
||||
@@ -372,7 +446,6 @@ prov.track_entity(
|
||||
confidence=0.98,
|
||||
entity_type="vulnerability",
|
||||
activity_id="nvd_feed_update",
|
||||
agent_id="ingest_pipeline_v2",
|
||||
)
|
||||
|
||||
# Track CISA KEV addition as a separate property source
|
||||
@@ -433,7 +506,6 @@ prov.track_entity(
|
||||
entity_type="clinical_endpoint",
|
||||
activity_id="structured_data_extraction",
|
||||
source_quote="Vaccine efficacy against COVID-19 was 95.0% (95% CI, 90.3–97.6)",
|
||||
agent_id="meddra_extraction_pipeline_v3",
|
||||
)
|
||||
|
||||
# Multi-study property tracking for meta-analysis
|
||||
@@ -533,7 +605,6 @@ prov.track_entity(
|
||||
confidence=0.89,
|
||||
entity_type="credit_decision",
|
||||
activity_id="automated_underwriting",
|
||||
agent_id="underwriting_model_v4",
|
||||
)
|
||||
|
||||
# SR 11-7 audit output
|
||||
@@ -562,12 +633,29 @@ Every `ProvenanceEntry` maps directly to W3C PROV-O terms. If your compliance te
|
||||
| :--- | :--- | :--- |
|
||||
| `prov:Entity` | `entity_id` | The tracked object — entity, chunk, relationship, or property |
|
||||
| `prov:Activity` | `activity_id` | The process that produced it — `"ner_extraction"`, `"bureau_parsing"` |
|
||||
| `prov:Agent` | `agent_id` | Who ran the activity — pipeline name, analyst ID |
|
||||
| `prov:wasDerivedFrom` | `parent_entity_id` | The previous version of this entity — enables version chaining |
|
||||
| `prov:Agent` / `prov:Person` / `prov:SoftwareAgent` / `prov:Organization` | `agent_id`, `agent_type`, `is_automated` | Who — or what — ran the activity, and whether a human was directly accountable |
|
||||
| `prov:qualifiedAssociation` + `prov:hadRole` | `role` | The agent's role for this specific entity — `"generator"` (default), `"approver"`, `"reviewer"` — for sign-off/four-eyes workflows |
|
||||
| `prov:wasDerivedFrom` | `parent_entity_id` (legacy combined field) | The previous version or source of this entity |
|
||||
| — | `previous_version_id` | This entry corrects/replaces a prior version of the *same* fact |
|
||||
| `prov:wasDerivedFrom` | `derived_from_id` | This entry was derived from a *different* source entity |
|
||||
| `prov:used` | `used_entities` | Entity IDs consumed to produce this one |
|
||||
| `prov:generatedAtTime` | `timestamp` | ISO datetime, auto-set to `datetime.utcnow()` at write time |
|
||||
| `prov:qualifiedInvalidation` | `invalidated`, `invalidated_at_time`, `invalidated_by`, `invalidation_reason` | A retraction/correction recorded as a tombstone via `ProvenanceManager.invalidate()`, never a hard delete |
|
||||
| `prov:startedAtTime` / `prov:endedAtTime` | `activity_started_at_time`, `activity_ended_at_time` | Typed Activity timing — pass an `ActivityRecord` via the `activity=` kwarg to set these together with `activity_id` |
|
||||
| `prov:qualifiedGeneration`/`Generation`, `qualifiedUsage`/`Usage`, `qualifiedDerivation`/`Derivation` | (derived from the fields above) | Additive qualified forms of `wasGeneratedBy`/`used`/`wasDerivedFrom`, emitted automatically alongside the plain triples |
|
||||
| `prov:wasAssociatedWith` | (derived from `agent_id`) | Direct Activity→Agent link, distinct from the Entity→Agent `wasAttributedTo` |
|
||||
| `prov:actedOnBehalfOf` | `acted_on_behalf_of` | Agent→Agent delegation — e.g. an automated agent acting on behalf of the human/organization that authorized it |
|
||||
| `prov:wasInformedBy` | `informed_by_activities` (pass as `informed_by=[...]`) | Chains this entry's activity to prior activities it was informed by (e.g. a pipeline stage informed by the stage before it) |
|
||||
| `prov:Bundle` + `prov:hadMember` | `bundle_id` | Groups entries by source/dataset/ingestion-run (membership triples, not true RDF named-graph partitioning) |
|
||||
| — | `valid_from`, `valid_until`, `revision_type`, `supersedes` | Bitemporal fields merged from the deprecated `kg.ProvenanceTracker` — always caller-supplied (never auto-computed), surfaced via `ProvenanceManager.revision_history()`, which falls back to timestamp-based derivation for entries that don't set them explicitly |
|
||||
|
||||
The `checksum` field is not part of the PROV-O standard — it is Semantica's tamper-detection extension. Every entry's SHA-256 is computed from its content fields at write time and can be recomputed at any time to verify the record has not been modified.
|
||||
`previous_version_id` and `derived_from_id` are additive alongside `parent_entity_id` — existing code reading `parent_entity_id` keeps working unchanged, while new code gets the two relations disambiguated.
|
||||
|
||||
The `checksum` field is not part of the PROV-O standard — it is Semantica's tamper-detection extension. Every entry's SHA-256 now also incorporates `previous_checksum` (the prior entry's checksum, by insertion order via `sequence_id`), chaining every entry to the one before it. `ProvenanceManager.verify_chain()` walks the full chain and reports any break — including a row that was hard-deleted from the underlying table, which a lone per-row checksum can't detect on its own.
|
||||
|
||||
Note: the banking example above passes `agent_id="credit_data_service_v2"` to `track_entities_batch()` — this now actually populates the entry's `agent_id` field (previously a bug caused batch-level typed kwargs like `agent_id`/`entity_type`/`activity_id` to be silently absorbed into the opaque `metadata` blob instead).
|
||||
|
||||
`export_prov()` mints entity/agent/activity URIs under `ProvenanceManager.DEFAULT_BASE_URI` (`https://semantica.dev/ns#` by default — the same namespace `RDFExporter`'s `NamespaceManager` uses for its `"semantica"` prefix, so KG-exported and PROV-exported URIs for the same `entity_id` co-resolve) unless overridden via `export_prov(base_uri=...)` or the CLI's `--base-uri` option.
|
||||
|
||||
## Related Guides
|
||||
|
||||
|
||||
@@ -4,6 +4,70 @@ description: "How Semantica extracts entities, relationships, events, and RDF tr
|
||||
icon: "magnifying-glass"
|
||||
---
|
||||
|
||||
## What Is Semantic Extraction?
|
||||
|
||||
Semantic extraction is the process of automatically identifying meaningful information from unstructured text and converting it into structured, machine-readable formats. Unlike simple keyword search or pattern matching, semantic extraction understands context, relationships, and implicit connections between concepts in natural language.
|
||||
|
||||
**Key differences from basic text processing:**
|
||||
- **Regex matching** finds exact patterns but misses contextual meaning
|
||||
- **Keyword search** locates terms but ignores relationships between them
|
||||
- **Manual annotation** captures semantic meaning but doesn't scale
|
||||
- **Semantic extraction** automatically identifies entities, relationships, and events while preserving contextual understanding
|
||||
|
||||
When you extract entities like "APT29" and "NATO" from intelligence text, semantic extraction also captures that APT29 "targets" NATO networks, creating structured knowledge that feeds directly into graph databases, reasoning systems, and retrieval workflows.
|
||||
|
||||
## Why Use Semantic Extraction?
|
||||
|
||||
**Knowledge graph population.** Transform unstructured documents into interconnected knowledge graphs where entities become nodes and relationships become edges, enabling sophisticated graph traversal and reasoning.
|
||||
|
||||
**GraphRAG preparation.** Extract structured facts from raw text so that graph-grounded retrieval can find precise, contextually relevant information instead of just similar document chunks.
|
||||
|
||||
**Turning unstructured text into structured data.** Convert intelligence reports, clinical notes, legal documents, and regulatory filings into databases, RDF triples, and JSON schemas that downstream systems can query and process.
|
||||
|
||||
**Downstream retrieval and reasoning benefits.** Enable precise entity-based search, relationship discovery, causal analysis, and multi-hop reasoning that would be impossible with document-level retrieval alone.
|
||||
|
||||
**Automated knowledge discovery.** Surface hidden connections and patterns across large document collections that human analysts would miss due to volume and complexity.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Use semantic extraction for:**
|
||||
- Converting intelligence reports, clinical notes, and regulatory documents into structured knowledge
|
||||
- Building knowledge graphs from unstructured text corpora
|
||||
- Preparing text for graph-based reasoning and GraphRAG workflows
|
||||
- Discovering relationships and connections across document collections
|
||||
- Creating structured datasets for downstream analysis and reporting
|
||||
|
||||
**Deterministic parsing may be better for:**
|
||||
- Highly structured identifiers like email addresses, UUIDs, hashes, and log IDs where regex patterns are sufficient
|
||||
- Simple data extraction from standardized formats (CSV, JSON, XML)
|
||||
- Known patterns with fixed formats that don't require contextual understanding
|
||||
- High-frequency operations where extraction speed is critical and semantic understanding unnecessary
|
||||
|
||||
**Consider simpler alternatives when:**
|
||||
- Documents are already structured and don't require natural language understanding
|
||||
- Simple keyword search or document retrieval meets your requirements
|
||||
- Text quality is too poor for reliable semantic analysis (heavily corrupted OCR, fragmentary data)
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
The semantic extraction workflow follows a structured sequence that transforms raw text into graph-ready knowledge:
|
||||
|
||||
**Ingest** → Load documents from various sources (files, databases, APIs) and prepare text for processing
|
||||
|
||||
**Extract** → Apply Named Entity Recognition (NER), relation extraction, event detection, and coreference resolution to identify meaningful information
|
||||
|
||||
**Resolve** → Consolidate entity mentions ("APT29", "the group", "they") into canonical references and disambiguate overlapping entities
|
||||
|
||||
**Relate** → Connect extracted entities through relationships, creating a web of structured connections between concepts
|
||||
|
||||
**Serialize** → Convert the extracted knowledge into RDF triplets, JSON-LD, or other structured formats
|
||||
|
||||
**Store** → Load structured output into knowledge graphs, vector databases, or agent memory systems
|
||||
|
||||
**Retrieve** → Query the structured knowledge through graph traversal, semantic search, and reasoning workflows
|
||||
|
||||
This pipeline transforms documents like "APT29 deployed HAMMERTOSS malware targeting NATO networks" into structured triplets like `(APT29, deployed, HAMMERTOSS)` and `(HAMMERTOSS, targets, NATO_networks)` that enable sophisticated downstream analysis.
|
||||
|
||||
`semantica.semantic_extract` turns unstructured text into structured graph-ready output: it identifies named entities, extracts relationships between them, detects time-anchored events, resolves coreferences, and serialises everything as RDF triplets. Use it to populate a `ContextGraph` from raw documents — intelligence reports, clinical notes, regulatory filings, or any free-text corpus.
|
||||
|
||||
<Info>
|
||||
@@ -12,6 +76,8 @@ icon: "magnifying-glass"
|
||||
|
||||
## Step 1 — Named Entity Recognition: who and what is in the text
|
||||
|
||||
**Named Entity Recognition (NER)** identifies and classifies meaningful nouns and noun phrases in text, such as people, organizations, locations, products, and domain-specific entities like threat actors or drug names. NER forms the foundation of semantic extraction by identifying the key participants and objects in your documents.
|
||||
|
||||
`NamedEntityRecognizer` extracts meaningful nouns from a document and lets you choose the underlying method depending on your latency budget and domain requirements:
|
||||
|
||||
```python
|
||||
@@ -77,6 +143,8 @@ print("High-confidence entities: {}".format(len(high_conf)))
|
||||
|
||||
## Step 2 — Relation Extraction: how the entities connect
|
||||
|
||||
**Relation Extraction** identifies semantic relationships between entities, capturing not just what entities exist in text but how they interact, influence, or connect to each other. This creates the edges that link entity nodes in your knowledge graph.
|
||||
|
||||
`RelationExtractor` produces the web of connections between entities — who deployed what, who supplied whom, which CVE targets which product:
|
||||
|
||||
```python
|
||||
@@ -111,6 +179,8 @@ The `context` field on each `Relation` stores the surrounding sentence. This let
|
||||
|
||||
## Step 3 — Event Detection: what happened, when, and to whom
|
||||
|
||||
**Event Detection** identifies discrete occurrences or actions described in text, capturing not just static relationships but dynamic processes that unfold over time. Events include participants, temporal boundaries, locations, and outcomes.
|
||||
|
||||
`EventDetector` surfaces structured time-anchored events — discrete occurrences with participants, time windows, and locations:
|
||||
|
||||
```python
|
||||
@@ -155,6 +225,8 @@ for doc_idx, doc_events in enumerate(batch_events):
|
||||
|
||||
## Step 4 — Coreference Resolution: one entity, many names
|
||||
|
||||
**Coreference Resolution** identifies when different text spans refer to the same real-world entity, consolidating mentions like "APT29", "the group", "they", and "the threat actor" into unified references. This prevents downstream processing from treating the same entity as multiple separate objects.
|
||||
|
||||
`CoreferenceResolver` collapses references like "GAMMA-7", "the group", "they", and "the threat actor" into canonical chains so downstream extraction doesn't treat them as separate entities:
|
||||
|
||||
```python
|
||||
@@ -180,6 +252,8 @@ With coreference resolved, you can now replace pronouns and aliases with canonic
|
||||
|
||||
## Step 5 — Triplet Extraction and RDF Serialisation: graph-ready output
|
||||
|
||||
**Triplet Extraction** converts semantic knowledge into subject-predicate-object triplets, the fundamental building blocks of knowledge graphs and RDF databases. This structured representation enables graph queries, reasoning, and integration with semantic web technologies.
|
||||
|
||||
`TripletExtractor` converts everything into subject-predicate-object triplets and serialises them as RDF, ready for graph ingestion and SPARQL queries:
|
||||
|
||||
```python
|
||||
@@ -313,7 +387,7 @@ def ingest_intel_report(
|
||||
# Process all 200 reports
|
||||
intel_graph = ContextGraph(advanced_analytics=True)
|
||||
intel_agent = AgentContext(
|
||||
vector_store=VectorStore(backend="faiss", dimension=768, index_path="intel.faiss"),
|
||||
vector_store=VectorStore(backend="faiss", dimension=768),
|
||||
knowledge_graph=intel_graph,
|
||||
decision_tracking=True,
|
||||
)
|
||||
@@ -559,6 +633,22 @@ jsonld = tri.serialize_triplets(valid, format="jsonld")
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Treating extraction as guaranteed truth.** Semantic extraction produces confidence scores for a reason — even high-confidence extractions can be incorrect. Always validate critical extractions, especially for high-stakes decisions in security, clinical, or financial contexts.
|
||||
|
||||
**Ignoring confidence thresholds.** Low-confidence extractions often indicate ambiguous text, poor model fit, or noisy input. Setting appropriate thresholds (typically 0.65-0.85) filters unreliable results before they pollute downstream processing.
|
||||
|
||||
**Skipping entity resolution.** Different mentions of the same entity ("NATO", "North Atlantic Treaty Organization", "the alliance") will create duplicate nodes in your knowledge graph. Always run coreference resolution and entity deduplication.
|
||||
|
||||
**Poor OCR or poor input quality.** Semantic extraction depends on readable text. Documents with OCR errors, encoding issues, or heavy redaction will produce unreliable extractions. Clean and validate input text before extraction.
|
||||
|
||||
**Using LLM extraction where regex is sufficient.** For highly structured patterns like CVE identifiers (CVE-YYYY-NNNN), IP addresses, email addresses, or UUIDs, regular expressions are faster, cheaper, and more reliable than semantic extraction.
|
||||
|
||||
**Processing too much text at once.** Very long documents (>10,000 words) can overwhelm extraction models and produce inconsistent results. Segment long documents into logical chunks (sections, paragraphs) and process them separately.
|
||||
|
||||
**Mixing incompatible extraction methods.** Different methods produce different entity label schemas. LLM extraction might return "THREAT_ACTOR" while spaCy returns "PERSON" for the same entity. Normalize labels across methods or use consistent method chains.
|
||||
|
||||
## Choosing your extraction method
|
||||
|
||||
The six extraction methods trade off speed, accuracy, and infrastructure:
|
||||
|
||||
+167
-38
@@ -4,14 +4,111 @@ description: "Generate W3C SHACL shapes from OWL ontologies, validate RDF knowle
|
||||
icon: "shield-check"
|
||||
---
|
||||
|
||||
`SHACLGenerator` produces W3C SHACL constraint shapes from an OWL ontology, and `_run_pyshacl` validates your knowledge graph against them, returning a structured violation report. Use this to gate graph data before analytics, ISAC sharing, or regulatory submission — catching missing required properties, datatype violations, and cardinality breaches before they propagate.
|
||||
## What Is SHACL Validation?
|
||||
|
||||
<Info>
|
||||
SHACL shapes are produced from the same ontology dict that `OntologyGenerator` builds. The full workflow is: graph → ontology → SHACL shapes → validation report. Each stage is one function call. `NodeShape`, `PropertyShape`, and `SHACLGraph` import from `semantica.ontology`. `SHACLValidationReport`, `SHACLViolation`, and `_run_pyshacl` import from `semantica.ontology.ontology_validator`.
|
||||
</Info>
|
||||
SHACL (Shapes Constraint Language) is a standard for validating graph-based data. While an ontology defines the conceptual *schema* (the "what" exists in your domain), SHACL defines the structural *rules and constraints* (the "how" it should be structured).
|
||||
|
||||
In Semantica, `SHACLGenerator` produces constraint rules (shapes) based on your ontology, and `_run_pyshacl` evaluates your actual data against these rules. If a node violates a rule (e.g., missing a required property or using the wrong datatype), a detailed violation report is generated.
|
||||
|
||||
## Why Use SHACL Validation?
|
||||
|
||||
Data validation is critical before running analytics, exporting data, or feeding it into production models. SHACL acts as a **data quality gate** that ensures your graph data is structurally sound. Use it to catch:
|
||||
- Missing required properties (e.g., a customer without an email address).
|
||||
- Datatype mismatches (e.g., a string where a number was expected).
|
||||
- Cardinality breaches (e.g., a person with three primary addresses).
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
- **When to Use**: You have a complex, interconnected knowledge graph and need to validate the *relationships* and structural integrity of the nodes across the graph. SHACL excels at ensuring that merged, highly connected data conforms to your business rules.
|
||||
- **When NOT to Use**: If you are simply validating a flat JSON payload or a single incoming API request. For flat data or single records, use simpler, faster libraries like Pydantic or JSONSchema.
|
||||
|
||||
---
|
||||
|
||||
## Key Terms Explained
|
||||
|
||||
Before diving in, here are a few concepts you'll encounter:
|
||||
|
||||
- **RDF (Resource Description Framework)**: A standard way of representing data as a graph. It treats information as connected "triplets" (Subject → Predicate → Object).
|
||||
- **OWL (Web Ontology Language)**: A language used to build ontologies. It defines the classes and properties that exist in your domain.
|
||||
- **SHACL Shapes**: The actual validation rules. A "Shape" targets a specific class in your data (like `Person`) and defines the constraints it must follow (like "must have one birthdate").
|
||||
- **Turtle (.ttl)**: A popular, human-readable file format for storing RDF graph data and SHACL shapes.
|
||||
|
||||
---
|
||||
|
||||
## Typical Workflow
|
||||
|
||||
A typical SHACL validation pipeline follows this lifecycle:
|
||||
|
||||
1. **Ontology**: Build an ontology representing your domain.
|
||||
2. **SHACL Shapes**: Generate shapes from that ontology.
|
||||
3. **Data Graph**: Prepare your knowledge graph.
|
||||
4. **Validation**: Validate the knowledge graph against the SHACL shapes.
|
||||
5. **Violation Report**: Analyze the report for errors.
|
||||
6. **Remediation**: Fix the data or pipeline and re-validate.
|
||||
|
||||
---
|
||||
|
||||
## Universal Example: Employee & Department
|
||||
|
||||
Let's look at a simple, universally understood example: ensuring every `Employee` belongs to a `Department` and has an `employee_id`.
|
||||
|
||||
```python
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
|
||||
# 1. Prepare your data graph
|
||||
graph = ContextGraph()
|
||||
graph.add_node("emp-1", "Employee", "Alice", employee_id="E001")
|
||||
graph.add_node("emp-2", "Employee", "Bob") # Missing employee_id, will cause a violation!
|
||||
|
||||
# 2. Build the ontology
|
||||
ontology = (
|
||||
OntologyGenerator(base_uri="https://company.example.com/ontology/", min_occurrences=1)
|
||||
.generate_from_graph(graph.to_dict(), name="CompanyOntology")
|
||||
)
|
||||
|
||||
# 3. Generate SHACL Shapes
|
||||
shacl_gen = SHACLGenerator(base_uri="https://company.example.com/shapes/", severity="Violation")
|
||||
shacl_graph = shacl_gen.generate(ontology)
|
||||
|
||||
# Inject mandatory constraints
|
||||
BASE = "https://company.example.com/ontology/"
|
||||
for ns in shacl_graph.node_shapes:
|
||||
if "Employee" in ns.target_class:
|
||||
ns.property_shapes.append(
|
||||
PropertyShape(path=f"{BASE}employee_id", min_count=1, severity="Violation")
|
||||
)
|
||||
|
||||
# Serialize shapes to Turtle
|
||||
shacl_ttl = shacl_gen.serialize(shacl_graph, format="turtle")
|
||||
|
||||
# 4. Prepare your RDF data graph
|
||||
# (For validation, serialize your graph instances to RDF. Here we use a Turtle string.)
|
||||
data_ttl = """
|
||||
@prefix ex: <https://company.example.com/ontology/> .
|
||||
|
||||
<http://example.org/emp-1> a ex:Employee ;
|
||||
ex:employee_id "E001" .
|
||||
|
||||
<http://example.org/emp-2> a ex:Employee .
|
||||
"""
|
||||
|
||||
# 5. Run Validation
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
|
||||
# 6. Analyze the Report
|
||||
print(f"Graph conforms: {report.conforms}")
|
||||
if not report.conforms:
|
||||
report.explain_violations() # Populates human-readable explanations
|
||||
for v in report.violations:
|
||||
print(f"Violation: {v.explanation}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
Now, let's explore the workflow in more depth.
|
||||
|
||||
## Step 1 — Build the ontology from your merged graph
|
||||
|
||||
SHACL shapes are derived from an ontology. If you already have one from a previous run, skip this step.
|
||||
@@ -172,15 +269,16 @@ Serialize the graph to RDF, then run `_run_pyshacl` against the shapes.
|
||||
|
||||
```python
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.export import export_rdf
|
||||
import tempfile, os
|
||||
|
||||
# Serialise the graph to a temporary Turtle file
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False, mode="w")
|
||||
export_rdf(graph.to_dict(), tmp.name, format="turtle")
|
||||
with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
os.unlink(tmp.name)
|
||||
# Prepare your RDF data string (since export_rdf primarily exports structural metadata,
|
||||
# you typically serialize your custom data graph to Turtle using rdflib or similar).
|
||||
data_ttl = """
|
||||
@prefix ex: <https://cti.example.org/ontology/> .
|
||||
|
||||
<http://example.org/malware-002> a ex:Malware .
|
||||
<http://example.org/vuln-003> a ex:Vulnerability ;
|
||||
ex:cve_id "CVE24-3400" .
|
||||
"""
|
||||
|
||||
# Run SHACL validation
|
||||
report = _run_pyshacl(
|
||||
@@ -214,13 +312,17 @@ Each `SHACLViolation` identifies the node, property path, and fix required.
|
||||
|
||||
```python
|
||||
if not report.conforms:
|
||||
# Print plain-English explanations for every violation
|
||||
# Populate plain-English explanations for every violation
|
||||
report.explain_violations()
|
||||
# Node <https://cti.example.org/data/malware-002> is missing required property
|
||||
|
||||
# Iterate and print the explanations
|
||||
for v in report.violations:
|
||||
print(v.explanation)
|
||||
# Node <http://example.org/malware-002> is missing required property
|
||||
# <https://cti.example.org/ontology/family>. At least 1 value(s) are required.
|
||||
# Node <https://cti.example.org/data/vuln-003> is missing required property
|
||||
# Node <http://example.org/vuln-003> is missing required property
|
||||
# <https://cti.example.org/ontology/cvss_score>. At least 1 value(s) are required.
|
||||
# Node <https://cti.example.org/data/vuln-003> has value 'CVE24-3400' for
|
||||
# Node <http://example.org/vuln-003> has value 'CVE24-3400' for
|
||||
# <https://cti.example.org/ontology/cve_id> which does not match the required pattern.
|
||||
|
||||
# Iterate for programmatic triage
|
||||
@@ -272,6 +374,16 @@ print(f"Violations after remediation: {report2.violation_count}")
|
||||
|
||||
---
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
- **Assuming the ontology automatically enforces data quality**: `SHACLGenerator` generates shapes based on what it observes in the data. If your data is missing a field, the generator won't know it was mandatory unless you explicitly inject the constraint (as shown in Step 3).
|
||||
- **Passing `ContextGraph` directly to SHACL validators**: The `_run_pyshacl` function expects an RDF string (like Turtle format), not a raw Python dictionary or `ContextGraph` object.
|
||||
- **Forgetting RDF serialization**: You must serialize your graph (often via a temporary file using `export_rdf`) before validating it.
|
||||
- **Treating validation as a one-time step**: Validation should be integrated as an automated step in your CI/CD pipeline or data ingestion flow, acting as a recurring gatekeeper rather than a one-off script.
|
||||
- **Ignoring validation reports**: A graph that does not conform must be remediated. Failing to review the `violation_count` and address the issues negates the purpose of SHACL validation.
|
||||
|
||||
---
|
||||
|
||||
## Domain Examples
|
||||
|
||||
<Tabs>
|
||||
@@ -285,8 +397,6 @@ from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.export import export_rdf
|
||||
import tempfile, os
|
||||
|
||||
graph = ContextGraph()
|
||||
ctx = AgentContext(
|
||||
@@ -329,11 +439,14 @@ for ns in shacl_graph.node_shapes:
|
||||
|
||||
shacl_ttl = shacl_gen.serialize(shacl_graph, format="turtle")
|
||||
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False, mode="w")
|
||||
export_rdf(graph.to_dict(), tmp.name, format="turtle")
|
||||
with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
os.unlink(tmp.name)
|
||||
# Prepare RDF data string
|
||||
data_ttl = """
|
||||
@prefix ex: <https://cti.dod.mil/ontology/> .
|
||||
|
||||
<http://example.org/apt29> a ex:ThreatActor .
|
||||
<http://example.org/cve-2024-3400> a ex:Vulnerability .
|
||||
<http://example.org/hammertoss> a ex:Malware .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
print(f"CTI graph conforms : {report.conforms}")
|
||||
@@ -342,6 +455,8 @@ print(f"Warnings : {report.warning_count}")
|
||||
|
||||
if not report.conforms:
|
||||
report.explain_violations()
|
||||
for v in report.violations:
|
||||
print(v.explanation)
|
||||
# Blocks the nightly ISAC share until violations are resolved
|
||||
```
|
||||
|
||||
@@ -355,8 +470,6 @@ A SOC team validates zero-trust policy nodes before publishing them to the polic
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.export import export_rdf
|
||||
import tempfile, os
|
||||
|
||||
graph = ContextGraph()
|
||||
graph.add_node("policy-001", "Policy", "MFA Required for Tier-1 Resources",
|
||||
@@ -392,11 +505,16 @@ for ns in shacl_graph.node_shapes:
|
||||
|
||||
shacl_ttl = shacl_gen.serialize(shacl_graph, format="turtle")
|
||||
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False, mode="w")
|
||||
export_rdf(graph.to_dict(), tmp.name, format="turtle")
|
||||
with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
os.unlink(tmp.name)
|
||||
# Prepare RDF data string
|
||||
data_ttl = """
|
||||
@prefix ex: <https://zerotrust.corp/ontology/> .
|
||||
|
||||
<http://example.org/policy-001> a ex:Policy ;
|
||||
ex:version "1.0.0" ;
|
||||
ex:effective_date "2025-01-01"^^<http://www.w3.org/2001/XMLSchema#date> .
|
||||
|
||||
<http://example.org/policy-002> a ex:Policy .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
print(f"Policy graph conforms: {report.conforms}")
|
||||
@@ -460,7 +578,9 @@ print(f"SHACL shapes generated — {len(shacl_graph.node_shapes)} node shapes")
|
||||
# SHACL shapes generated — 5 node shapes
|
||||
|
||||
# Validate trial data
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False, mode="w")
|
||||
# Serialize the ontology as data to validate against the shapes
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False)
|
||||
tmp.close()
|
||||
export_rdf(ontology, tmp.name, format="turtle")
|
||||
with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
@@ -481,8 +601,6 @@ A credit risk team validates every `LoanApplication` node against Basel III CRE2
|
||||
from semantica.context import ContextGraph
|
||||
from semantica.ontology import OntologyGenerator, SHACLGenerator, PropertyShape
|
||||
from semantica.ontology.ontology_validator import _run_pyshacl
|
||||
from semantica.export import export_rdf
|
||||
import tempfile, os
|
||||
|
||||
graph = ContextGraph()
|
||||
graph.add_node("loan-001", "LoanApplication", "Prime mortgage APP-2025-88421",
|
||||
@@ -513,11 +631,19 @@ for ns in shacl_graph.node_shapes:
|
||||
|
||||
shacl_ttl = shacl_gen.serialize(shacl_graph, format="turtle")
|
||||
|
||||
tmp = tempfile.NamedTemporaryFile(suffix=".ttl", delete=False, mode="w")
|
||||
export_rdf(graph.to_dict(), tmp.name, format="turtle")
|
||||
with open(tmp.name) as f:
|
||||
data_ttl = f.read()
|
||||
os.unlink(tmp.name)
|
||||
# Prepare RDF data string
|
||||
data_ttl = """
|
||||
@prefix ex: <https://basel.eba.eu/ontology/> .
|
||||
|
||||
<http://example.org/loan-001> a ex:LoanApplication ;
|
||||
ex:ltv "0.78" ;
|
||||
ex:pd "0.023" ;
|
||||
ex:lgd "0.45" ;
|
||||
ex:asset_class "CRE" .
|
||||
|
||||
<http://example.org/loan-002> a ex:LoanApplication ;
|
||||
ex:ltv "0.65" .
|
||||
"""
|
||||
|
||||
report = _run_pyshacl(data_ttl, shacl_ttl)
|
||||
print(f"Loan portfolio conforms: {report.conforms}")
|
||||
@@ -561,6 +687,8 @@ def validate_before_publish(data_graph_str: str, ontology: dict) -> None:
|
||||
if not report.conforms:
|
||||
print(f"Graph validation FAILED — {report.violation_count} violation(s)")
|
||||
report.explain_violations()
|
||||
for v in report.violations:
|
||||
print(v.explanation)
|
||||
sys.exit(1)
|
||||
|
||||
print(f"Graph validation PASSED ({report.warning_count} warning(s))")
|
||||
@@ -575,3 +703,4 @@ def validate_before_publish(data_graph_str: str, ontology: dict) -> None:
|
||||
- [Export & Serialization](export) — serialize graph data to Turtle/RDF/XML for `_run_pyshacl` input
|
||||
- [Conflict Resolution](conflict-resolution) — detect and resolve data conflicts before SHACL validation
|
||||
- [Change Management](change-management) — version-gate SHACL shapes alongside ontology versions
|
||||
|
||||
|
||||
@@ -6,6 +6,72 @@ icon: "chart-network"
|
||||
|
||||
`KGVisualizer`, `AnalyticsVisualizer`, `TemporalVisualizer`, and `OntologyVisualizer` turn graph dicts, analytics results, and ontologies into interactive HTML dashboards or static images in a single method call. Use them to present centrality rankings, community clusters, event timelines, and before/after snapshot diffs to stakeholders without writing any rendering code.
|
||||
|
||||
## What Is Visualization?
|
||||
|
||||
Visualization converts graph data into interactive charts, network diagrams, timelines, and other visual formats that humans can interpret. It transforms abstract graph structures and analytical results into visual representations that reveal patterns, relationships, and insights.
|
||||
|
||||
**Visualization vs. analytics:** Analytics computes numerical measures like centrality scores and community memberships. Visualization renders those measures as colored nodes, sized by importance, grouped by community.
|
||||
|
||||
**Visualization vs. reasoning:** Reasoning derives new logical facts from existing data. Visualization presents existing facts and analytical results in visual form to support human interpretation and decision-making.
|
||||
|
||||
Visualization helps humans understand graph structure, analytical results, and temporal patterns that would be difficult to interpret from raw data alone.
|
||||
|
||||
## Why Use Visualization?
|
||||
|
||||
**Visual exploration:** Interactive graphs let you pan, zoom, hover, and filter to explore large networks that would be overwhelming as text or tables.
|
||||
|
||||
**Investigation support:** Highlighting paths between entities, color-coding by entity type, and sizing nodes by importance helps analysts identify patterns and focus investigation efforts.
|
||||
|
||||
**Communication:** Visual presentations make complex graph relationships accessible to stakeholders who don't work directly with the data.
|
||||
|
||||
**Reporting:** Static visualizations provide evidence and support for written reports, presentations, and regulatory submissions.
|
||||
|
||||
## When To Use / When Not To Use
|
||||
|
||||
**Visualization is appropriate for:**
|
||||
- Presenting graph structure and analytical results to humans
|
||||
- Exploring relationships and patterns in medium-sized graphs (10-1000 nodes)
|
||||
- Creating reports and presentations for stakeholders
|
||||
- Investigating specific paths or neighborhoods within graphs
|
||||
- Communicating findings from analytics or reasoning workflows
|
||||
|
||||
**Graph traversal may be enough for:**
|
||||
- Programmatic exploration of relationships
|
||||
- Simple queries about specific paths or connections
|
||||
- Automated workflows that don't require human interpretation
|
||||
|
||||
**Analytics may be more useful for:**
|
||||
- Computing numerical measures and rankings
|
||||
- Finding communities or centrality scores programmatically
|
||||
- Quantitative comparisons that don't need visualization
|
||||
|
||||
**Reasoning may be more useful for:**
|
||||
- Deriving new facts through logical inference
|
||||
- Rule-based decision making
|
||||
- Automated policy enforcement
|
||||
|
||||
**Visualization becomes impractical when:**
|
||||
- Graphs exceed ~1000 nodes (browser performance degrades)
|
||||
- The network is too dense to interpret visually
|
||||
- You need programmatic analysis rather than human interpretation
|
||||
|
||||
## Typical Visualization Workflow
|
||||
|
||||
**Graph → Filter → Visualize → Interpret → Investigate**
|
||||
|
||||
Most effective visualization follows this pattern:
|
||||
1. **Start with your knowledge graph** from `ContextGraph` or analytics results
|
||||
2. **Filter to a meaningful subgraph** — avoid visualizing entire enterprise graphs
|
||||
3. **Choose appropriate visualization** — network, timeline, heatmap, or rankings
|
||||
4. **Interpret the visual patterns** — clusters, central nodes, temporal trends
|
||||
5. **Investigate interesting findings** — drill down on unexpected patterns or outliers
|
||||
|
||||
Always filter before visualizing. A 10,000-node enterprise graph becomes meaningful when filtered to the 50 most central nodes or the subgraph around a specific entity of interest.
|
||||
|
||||
<Info>
|
||||
**Performance Warning:** Large graphs (>1000 nodes) cause browser performance issues and become visually overwhelming. Interactive network visualizations work best with 10-1000 nodes. For larger graphs, use analytics to identify the most important subgraphs, then visualize those filtered results.
|
||||
</Info>
|
||||
|
||||
<Info>
|
||||
All visualizers accept `output="interactive"` (Plotly/pyvis HTML, shown in Jupyter or saved to file) or `output="static"` (PNG/SVG via Matplotlib). Omit `file_path` to get the figure object back for further manipulation.
|
||||
</Info>
|
||||
@@ -199,7 +265,7 @@ tv.visualize_timeline(
|
||||
|
||||
## Comparing Two Graph Snapshots Side-by-Side
|
||||
|
||||
When the question is "what changed between March 14 and April 14?", `visualize_snapshot_comparison` takes two named snapshots from `TemporalVersionManager` and renders a side-by-side diff view showing nodes and edges added or removed.
|
||||
When the question is "what changed between March 14 and April 14?", `visualize_snapshot_comparison` takes two named snapshots from `TemporalVersionManager` and renders a line chart comparing graph metrics (entities, relationships, density) across the provided snapshots.
|
||||
|
||||
```python
|
||||
from semantica.change_management import TemporalVersionManager
|
||||
@@ -393,6 +459,7 @@ from semantica.context import AgentContext, ContextGraph
|
||||
from semantica.vector_store import VectorStore
|
||||
from semantica.visualization import KGVisualizer, EmbeddingVisualizer, OntologyVisualizer
|
||||
from semantica.ontology import OntologyGenerator
|
||||
import numpy as np
|
||||
|
||||
graph = ContextGraph(advanced_analytics=True)
|
||||
ctx = AgentContext(
|
||||
@@ -426,10 +493,12 @@ ov.visualize_hierarchy(ontology, output="interactive", file_path="drug_hierarchy
|
||||
ov.visualize_structure(ontology, output="interactive", file_path="drug_ontology.html")
|
||||
|
||||
# UMAP projection and similarity heatmap for drug embeddings
|
||||
embeddings = [[0.1, 0.2, 0.3], [0.15, 0.22, 0.31], [0.8, 0.7, 0.6]]
|
||||
embeddings = np.array([[0.1, 0.2, 0.3], [0.15, 0.22, 0.31], [0.8, 0.7, 0.6]])
|
||||
labels = ["Metformin", "Dapagliflozin", "Semaglutide"]
|
||||
|
||||
ev = EmbeddingVisualizer()
|
||||
# UMAP (Uniform Manifold Approximation and Projection) reduces high-dimensional
|
||||
# embeddings to 2D while preserving local neighborhood structure
|
||||
ev.visualize_2d_projection(
|
||||
embeddings, labels, method="umap",
|
||||
output="interactive", file_path="drug_embeddings.html",
|
||||
@@ -514,6 +583,18 @@ if snap1 and snap2:
|
||||
|
||||
</Tabs>
|
||||
|
||||
## Common Pitfalls
|
||||
|
||||
**Rendering massive graphs.** Attempting to visualize graphs with thousands of nodes crashes browsers and creates uninterpretable hairballs. Always filter large graphs to meaningful subsets before visualization.
|
||||
|
||||
**Treating visual proximity as proof of relationships.** Nodes that appear close in a visualization aren't necessarily closely related in the graph structure. Visual layout algorithms optimize for readability, not semantic accuracy.
|
||||
|
||||
**Visualizing duplicate/unclean data.** Duplicate entities, inconsistent naming, and data quality issues are amplified in visualizations. Clean your graph data before creating visual presentations for stakeholders.
|
||||
|
||||
**Overloading tooltips with huge text fields.** Hovering over a node shouldn't display entire document contents. Include only essential metadata in hover tooltips — entity type, name, and key properties.
|
||||
|
||||
**Running visualizations before graph cleanup.** Visualizations reflect data quality issues directly. Entities with inconsistent names, duplicate nodes, and missing relationships create confusing and misleading visual representations.
|
||||
|
||||
## Output Modes
|
||||
|
||||
Every visualizer method accepts the same two output modes:
|
||||
|
||||
@@ -3,6 +3,10 @@ title: "Semantica"
|
||||
description: "The Accountability and Context Layer for AI: Context Graphs · Decision Intelligence · Full Provenance"
|
||||
---
|
||||
|
||||
```bash
|
||||
pip install semantica
|
||||
```
|
||||
|
||||
Your AI agent just made a decision. Now someone needs to explain it.
|
||||
|
||||
*What did it know at the time? Which facts shaped the outcome? Where did those facts come from? Has it made the same call before: and did that go well?*
|
||||
|
||||
@@ -0,0 +1,199 @@
|
||||
---
|
||||
title: "Databricks Integration"
|
||||
description: "Ingest Unity Catalog metadata and Delta Lake tables from Databricks into Semantica's KG pipeline."
|
||||
icon: "cloud"
|
||||
---
|
||||
|
||||
> Extract Delta Lake tables and Unity Catalog metadata (schemas, lineage) from Databricks into Semantica with personal access token or OAuth M2M authentication.
|
||||
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
# Install with Databricks support
|
||||
pip install "semantica[db-databricks]"
|
||||
|
||||
# Or install the connectors separately
|
||||
pip install databricks-sdk databricks-sql-connector
|
||||
```
|
||||
|
||||
|
||||
## Basic Usage
|
||||
|
||||
```python
|
||||
from semantica.ingest import DatabricksIngestor
|
||||
import os
|
||||
|
||||
ingestor = DatabricksIngestor(
|
||||
host=os.getenv("DATABRICKS_HOST"), # e.g. https://adb-xxx.azuredatabricks.net
|
||||
token=os.getenv("DATABRICKS_TOKEN"),
|
||||
http_path=os.getenv("DATABRICKS_HTTP_PATH"), # SQL warehouse or cluster HTTP path
|
||||
catalog=os.getenv("DATABRICKS_CATALOG", "main"),
|
||||
schema=os.getenv("DATABRICKS_SCHEMA", "default"),
|
||||
)
|
||||
|
||||
data = ingestor.ingest_table("customers")
|
||||
print(f"Retrieved {data.row_count} rows: columns: {data.columns}")
|
||||
```
|
||||
|
||||
<Tip>
|
||||
Use environment variables (or a `.env` file with `python-dotenv`) to keep credentials out of source code. `DatabricksIngestor()` with no arguments reads from `DATABRICKS_*` environment variables automatically.
|
||||
</Tip>
|
||||
|
||||
|
||||
## Authentication Methods
|
||||
|
||||
<Tabs>
|
||||
<Tab title="Personal Access Token">
|
||||
```python
|
||||
ingestor = DatabricksIngestor(
|
||||
host="https://adb-xxx.azuredatabricks.net",
|
||||
token="dapi-xxxxxxxx",
|
||||
http_path="/sql/1.0/warehouses/xxxxxxxx",
|
||||
)
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="OAuth M2M (Recommended)">
|
||||
```python
|
||||
ingestor = DatabricksIngestor(
|
||||
host="https://adb-xxx.azuredatabricks.net",
|
||||
client_id="your_service_principal_client_id",
|
||||
client_secret="your_service_principal_client_secret",
|
||||
http_path="/sql/1.0/warehouses/xxxxxxxx",
|
||||
)
|
||||
```
|
||||
Preferred for production: no long-lived personal token stored in config.
|
||||
</Tab>
|
||||
</Tabs>
|
||||
|
||||
<Note>
|
||||
`http_path` identifies the SQL warehouse or all-purpose cluster used for query execution. Find it in the Databricks UI under **SQL Warehouses → Connection details**. Unity Catalog metadata calls (`list_catalogs`, `get_table_schema`, `get_table_lineage`, …) only need `host` and credentials — `http_path` is not required for those.
|
||||
</Note>
|
||||
|
||||
|
||||
## Querying
|
||||
|
||||
### Ingest a table with filters
|
||||
|
||||
```python
|
||||
data = ingestor.ingest_table(
|
||||
"customers",
|
||||
catalog="main",
|
||||
schema="default",
|
||||
where="country = 'USA' AND created_date > '2024-01-01'",
|
||||
order_by="created_date DESC",
|
||||
limit=10000,
|
||||
)
|
||||
```
|
||||
|
||||
### Custom SQL
|
||||
|
||||
```python
|
||||
data = ingestor.ingest_query("""
|
||||
SELECT customer_id, SUM(amount) AS total_amount
|
||||
FROM main.default.sales
|
||||
WHERE date >= '2024-01-01'
|
||||
GROUP BY customer_id
|
||||
""")
|
||||
```
|
||||
|
||||
|
||||
## Unity Catalog Metadata
|
||||
|
||||
### Schema introspection
|
||||
|
||||
```python
|
||||
schema = ingestor.get_table_schema("customers")
|
||||
for column in schema["columns"]:
|
||||
print(f"{column['name']}: {column['type']}")
|
||||
```
|
||||
|
||||
### Catalogs, schemas, and tables
|
||||
|
||||
```python
|
||||
catalogs = ingestor.list_catalogs()
|
||||
schemas = ingestor.list_schemas(catalog="main")
|
||||
tables = ingestor.list_tables(catalog="main", schema="default")
|
||||
```
|
||||
|
||||
### Table and column lineage
|
||||
|
||||
```python
|
||||
lineage = ingestor.get_table_lineage("customers", catalog="main", schema="default")
|
||||
print(lineage["upstream"]) # tables that feed into `customers`
|
||||
print(lineage["downstream"]) # tables derived from `customers`
|
||||
```
|
||||
|
||||
Use `get_table_lineage` to build `Table --DEPENDS_ON--> Table` edges in the knowledge graph directly from Unity Catalog's lineage tracking, without re-deriving lineage from query logs.
|
||||
|
||||
<Tip>
|
||||
Pass `include_column_lineage=True` to also resolve per-column upstream/downstream references (one extra Unity Catalog request per column, so it's opt-in):
|
||||
|
||||
```python
|
||||
lineage = ingestor.get_table_lineage(
|
||||
"customers", catalog="main", schema="default", include_column_lineage=True,
|
||||
)
|
||||
print(lineage["columns"]["email"])
|
||||
# {"upstream": ["main.default.raw_customers.email_address"], "downstream": []}
|
||||
```
|
||||
|
||||
</Tip>
|
||||
|
||||
|
||||
## Export as Semantica Documents
|
||||
|
||||
```python
|
||||
documents = ingestor.export_as_documents(
|
||||
data,
|
||||
id_field="customer_id",
|
||||
text_fields=["name", "email", "notes"],
|
||||
)
|
||||
print(f"Created {len(documents)} documents for processing")
|
||||
```
|
||||
|
||||
|
||||
## Batch Processing Large Tables
|
||||
|
||||
```python
|
||||
PAGE_SIZE = 5000
|
||||
for page in range(total_pages):
|
||||
data = ingestor.ingest_table(
|
||||
"large_table",
|
||||
limit=PAGE_SIZE,
|
||||
offset=page * PAGE_SIZE,
|
||||
)
|
||||
process_batch(data)
|
||||
```
|
||||
|
||||
Or use the built-in `batch_size` parameter:
|
||||
|
||||
```python
|
||||
data = ingestor.ingest_query(
|
||||
"SELECT * FROM main.default.large_table",
|
||||
batch_size=5000,
|
||||
)
|
||||
```
|
||||
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
```python
|
||||
from semantica.ingest import DatabricksConnector
|
||||
|
||||
connector = DatabricksConnector(
|
||||
host="https://adb-xxx.azuredatabricks.net",
|
||||
token="dapi-xxxxxxxx",
|
||||
http_path="/sql/1.0/warehouses/xxxxxxxx",
|
||||
)
|
||||
if not connector.test_connection():
|
||||
print("Connection failed: check host, http_path, and credentials")
|
||||
```
|
||||
|
||||
|
||||
## See Also
|
||||
|
||||
- [Ingest Module](../reference/ingest) — Full DatabricksIngestor and all other ingestors.
|
||||
- [Snowflake Integration](snowflake) — Companion connector for a Snowflake + Databricks hybrid estate.
|
||||
- [Pipeline](../reference/pipeline) — Use Databricks ingestion as a pipeline step.
|
||||
- [Installation](../installation) — All optional dependency extras.
|
||||
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Databricks data.
|
||||
@@ -172,6 +172,7 @@ if not connector.test_connection():
|
||||
## See Also
|
||||
|
||||
- [Ingest Module](../reference/ingest) — Full SnowflakeIngestor and all other ingestors.
|
||||
- [Databricks Integration](databricks) — Companion connector for a Snowflake + Databricks hybrid estate.
|
||||
- [Pipeline](../reference/pipeline) — Use Snowflake ingestion as a pipeline step.
|
||||
- [Installation](../installation) — All optional dependency extras.
|
||||
- [Knowledge Graph](../reference/kg) — Build a KG from ingested Snowflake data.
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
---
|
||||
title: "Migrating from kg.ProvenanceTracker"
|
||||
description: "How to move from the deprecated semantica.kg.ProvenanceTracker to the unified semantica.provenance.ProvenanceManager."
|
||||
---
|
||||
|
||||
## Why migrate
|
||||
|
||||
`semantica.kg.ProvenanceTracker` is deprecated and will be removed in a future major version. It was a standalone, in-memory implementation that never delegated to the unified provenance backend — `semantica.provenance.ProvenanceManager` is that backend, and is now the supported way to track entity and relationship provenance across every Semantica module (see the [Provenance & Audit Trails guide](/guides/provenance)).
|
||||
|
||||
Every method on `kg.ProvenanceTracker` now emits a `DeprecationWarning` on use, but existing code keeps working unchanged until the class is removed — there is no forced migration deadline yet.
|
||||
|
||||
## Method mapping
|
||||
|
||||
| `kg.ProvenanceTracker` | `ProvenanceManager` equivalent | Notes |
|
||||
| --- | --- | --- |
|
||||
| `ProvenanceTracker()` | `ProvenanceManager()` | `ProvenanceManager` also accepts `storage_path=` for SQLite persistence instead of in-memory only. |
|
||||
| `track_entity(entity_id, source, metadata)` | `track_entity(entity_id, source, metadata)` | Same call shape. `ProvenanceManager` additionally auto-links each update to its prior version via `parent_entity_id`. |
|
||||
| `get_all_sources(entity_id)` | `get_all_sources(entity_id)` | Field name differs: the `kg` tracker returns each record's time under `"recorded_at"`; `ProvenanceManager` returns `"timestamp"`. |
|
||||
| `clear(entity_id=None)` | `clear()` | `ProvenanceManager.clear()` clears all provenance data; there is no per-entity clear yet. |
|
||||
| `query_recorded_between(start, end)` | `query_recorded_between(start, end)` | Same call shape; filters by `timestamp` (ISO 8601 string comparison) across all tracked entries, not just one entity. |
|
||||
| `revision_history(fact_id)` | `revision_history(fact_id)` | Same call shape and return shape (`version`, `valid_from`, `valid_until`, `recorded_at`, `author`, optional `revision_type`/`supersedes`) — walks the entity's `previous_version_id` chain rather than a flat per-entity dict. |
|
||||
| `export_audit_log(fact_ids, format)` | *No direct equivalent yet* | Build the export from `get_lineage()` output, or serialize `get_statistics()` for a summary view. |
|
||||
|
||||
Methods with no direct equivalent are not planned to be reimplemented on `kg.ProvenanceTracker` — they will need a small adapter in caller code, or a feature request against `ProvenanceManager` if you rely on them heavily.
|
||||
|
||||
## Example
|
||||
|
||||
```python
|
||||
# Before
|
||||
from semantica.kg import ProvenanceTracker
|
||||
|
||||
tracker = ProvenanceTracker()
|
||||
tracker.track_entity("entity_1", source="doc_1", metadata={"confidence": 0.9})
|
||||
sources = tracker.get_all_sources("entity_1") # [{"source": ..., "recorded_at": ..., "confidence": 0.9}]
|
||||
|
||||
# After
|
||||
from semantica.provenance import ProvenanceManager
|
||||
|
||||
prov = ProvenanceManager()
|
||||
prov.track_entity("entity_1", source="doc_1", metadata={"confidence": 0.9})
|
||||
sources = prov.get_all_sources("entity_1") # [{"source": ..., "timestamp": ..., "metadata": {...}, ...}]
|
||||
```
|
||||
|
||||
## Suppressing the warning during migration
|
||||
|
||||
If you need to keep using `kg.ProvenanceTracker` temporarily and want to silence the warning while you plan the switch:
|
||||
|
||||
```python
|
||||
import warnings
|
||||
|
||||
with warnings.catch_warnings():
|
||||
warnings.simplefilter("ignore", DeprecationWarning)
|
||||
tracker = ProvenanceTracker()
|
||||
```
|
||||
|
||||
This is a stopgap, not a fix — plan to move to `ProvenanceManager` before `kg.ProvenanceTracker` is removed.
|
||||
+14
-6
@@ -31,7 +31,7 @@ Semantica is organized into **27 modules** across six logical layers. Each modul
|
||||
Loads data from files, web, databases, and streams into a unified `SourceDocument` format.
|
||||
|
||||
```python
|
||||
from semantica.ingest import FileIngestor, WebIngestor, ParquetIngestor, XMLIngestor
|
||||
from semantica.ingest import FileIngestor, WebIngestor, ParquetIngestor, XMLIngestor, DatabricksIngestor
|
||||
|
||||
# Files: PDF, DOCX, CSV, Excel, PPTX, JSON, HTML, archives
|
||||
ingestor = FileIngestor()
|
||||
@@ -39,18 +39,26 @@ documents = ingestor.ingest_directory("data/")
|
||||
|
||||
# Web crawl
|
||||
web_ingestor = WebIngestor()
|
||||
pages = web_ingestor.ingest_urls(["https://example.com"])
|
||||
page = web_ingestor.ingest_url("https://example.com")
|
||||
|
||||
# Parquet: single file, partitioned directory, Hive-style (v0.5.0)
|
||||
parquet = ParquetIngestor()
|
||||
sources = parquet.ingest("data/events.parquet")
|
||||
|
||||
# XML with XSD/DTD validation, namespace handling (v0.5.0)
|
||||
xml = XMLIngestor(validate_xsd="schema.xsd")
|
||||
sources = xml.ingest("data/records/")
|
||||
xml = XMLIngestor()
|
||||
sources = xml.ingest("data/records/", schema_path="schema.xsd")
|
||||
|
||||
# Enterprise lakehouse/warehouse — Unity Catalog + Delta Lake, or a Snowflake warehouse
|
||||
databricks = DatabricksIngestor(host="...", token="...", http_path="...")
|
||||
customers = databricks.ingest_table("customers")
|
||||
```
|
||||
|
||||
**Available ingestors:** `FileIngestor`, `WebIngestor`, `ParquetIngestor`, `XMLIngestor`, `RESTIngestor`, `PublicAPIIngestor`, `DBIngestor`, `DuckDBIngestor`, `ElasticIngestor`, `EmailIngestor`, `FeedIngestor`, `GDriveIngestor`, `HuggingFaceIngestor`, `MCPIngestor`, `MongoIngestor`, `OntologyIngestor`, `PandasIngestor`, `RepoIngestor`, `SnowflakeIngestor`, `StreamIngestor`
|
||||
**Available ingestors:** `FileIngestor`, `WebIngestor`, `ParquetIngestor`, `XMLIngestor`, `RESTIngestor`, `PublicAPIIngestor`, `DBIngestor`, `DatabricksIngestor`, `SnowflakeIngestor`, `EmailIngestor`, `FeedIngestor`, `MCPIngestor`, `OntologyIngestor`, `RepoIngestor`, `StreamIngestor`, `ArrowIngestor`, `CloudStorageIngestor`
|
||||
|
||||
<Note>
|
||||
`DuckDBIngestor`, `ElasticIngestor`, `GDriveIngestor`, `HuggingFaceIngestor`, `MongoIngestor`, and `PandasIngestor` also ship but aren't re-exported from the top-level `semantica.ingest` namespace yet — import them directly, e.g. `from semantica.ingest.duckdb_ingestor import DuckDBIngestor`.
|
||||
</Note>
|
||||
|
||||
### Parse
|
||||
|
||||
@@ -243,7 +251,7 @@ store.add_triplets(subject, predicate, obj)
|
||||
results = store.sparql("SELECT ?s ?p ?o WHERE { ?s ?p ?o }")
|
||||
```
|
||||
|
||||
**Backends:** Blazegraph, Apache Jena, RDF4J
|
||||
**Backends:** Oxigraph (embedded), Blazegraph, Apache Jena, RDF4J
|
||||
|
||||
|
||||
## Quality Assurance
|
||||
|
||||
@@ -272,7 +272,7 @@ icon: "brain"
|
||||
</Tip>
|
||||
|
||||
<Tip>
|
||||
**Persist your vector store between runs.** Pass `index_path="context.faiss"` to `VectorStore` so the FAISS index survives process restarts.
|
||||
**Persist your context between runs.** `VectorStore` does not auto-persist — passing `index_path=` to its constructor is a no-op. Call `context.save("agent_state/")` to write memory, the vector index, and the graph to disk, and `context.load("agent_state/")` on the next process to restore them. See the "Persist & Restore" tab under [Real-World Patterns](#real-world-patterns) below.
|
||||
</Tip>
|
||||
|
||||
### Memory Methods
|
||||
@@ -586,6 +586,51 @@ history = memory.get_conversation_history(conversation_id="conv_001", max_items=
|
||||
| `max_memory_size` | `int` | `10000` | Max items before LRU eviction |
|
||||
| `retention_policy` | `str` | `"unlimited"` | `"N_days"` (e.g. `"30_days"`) or `"unlimited"` |
|
||||
|
||||
### Markdown Round Trips
|
||||
|
||||
`AgentMemory` can export human-editable Markdown and import the edited files back.
|
||||
Each file contains one memory item, with required metadata in YAML frontmatter and
|
||||
the memory content in the Markdown body:
|
||||
|
||||
```markdown
|
||||
---
|
||||
id: mem_compliance_rule
|
||||
created_at: '2026-07-22T09:00:00+00:00'
|
||||
updated_at: '2026-07-22T10:30:00+00:00'
|
||||
type: compliance
|
||||
tags:
|
||||
- trading
|
||||
- approval
|
||||
---
|
||||
|
||||
All trades must be pre-approved.
|
||||
```
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
|
||||
# A single selected memory can be returned as Markdown text.
|
||||
document = memory.export(format="markdown", type="compliance")
|
||||
|
||||
# Export a memory set as one stable Markdown file per item.
|
||||
memory.export(format="markdown", destination="memory_export/")
|
||||
|
||||
# New IDs create memories; existing IDs are updated in place.
|
||||
count = memory.import_data(Path("memory_export/"), format="markdown")
|
||||
```
|
||||
|
||||
The required frontmatter fields are `id`, `created_at`, `updated_at`, and either
|
||||
`type` or `kind`. Optional metadata can be edited at the top level. Imports reject
|
||||
malformed or duplicate fields before changing memory, and re-importing unchanged
|
||||
files is idempotent. Memory-local `entities` and `relationships` are preserved as
|
||||
provenance but are not applied to `ContextGraph` by Markdown import. Use a dedicated
|
||||
export directory: matching files are overwritten, but unrelated or stale Markdown
|
||||
files are not deleted automatically. Export refuses to overwrite symbolic links and
|
||||
uses atomic file replacement. Timestamp offsets are preserved in Markdown and
|
||||
normalized to UTC only for comparisons, so aware and local-naive records can be
|
||||
queried together safely. Vector-store writes are deferred until the in-memory import
|
||||
commits; adapter synchronization remains best-effort and logs failures.
|
||||
|
||||
|
||||
## PolicyEngine
|
||||
|
||||
|
||||
@@ -258,11 +258,16 @@ Full interactive docs at `http://localhost:8000/docs`. All endpoints accept and
|
||||
| `/api/vocabulary/hierarchy` | `GET` | Concept hierarchy tree |
|
||||
| `/api/vocabulary/import` | `POST` | Import SKOS/RDF vocabulary file |
|
||||
|
||||
SKOS hierarchy writes reject cycles in both `skos:broader` and
|
||||
`skos:narrower` relationships. Vocabulary imports validate the complete
|
||||
batch before adding nodes, while direct graph/session edge writes apply
|
||||
the same invariant at the graph storage boundary.
|
||||
|
||||
**SPARQL:**
|
||||
|
||||
| Endpoint | Method | Description |
|
||||
| :-------- | :------ | :----------- |
|
||||
| `/api/sparql` | `POST` | Execute a SPARQL SELECT or ASK query |
|
||||
| `/api/sparql` | `POST` | Execute a read-only SPARQL query (`SELECT`, `ASK`, `CONSTRUCT`, or `DESCRIBE`); `CONSTRUCT`/`DESCRIBE` return triples as `subject`, `predicate`, `object` columns, and `ASK` returns a `result` boolean column |
|
||||
|
||||
</Accordion>
|
||||
<Accordion title="Decisions, Provenance, Annotations & Export">
|
||||
|
||||
@@ -6,7 +6,7 @@ icon: "database"
|
||||
|
||||
**`semantica.ingest`** is the **universal entry point** for loading data into Semantica:
|
||||
|
||||
- 15+ ingestion adapters: files, web, SQL, Snowflake, Kafka, MCP, Git repos, email
|
||||
- 15+ ingestion adapters: files, web, SQL, Databricks, Snowflake, Kafka, MCP, Git repos, email
|
||||
- PyArrow Parquet with column selection and partitioned dataset support
|
||||
- XXE-safe lxml XML with optional XSD schema validation
|
||||
- `ingest()` unified dispatcher: auto-detects source type from path or URL
|
||||
@@ -27,7 +27,9 @@ icon: "database"
|
||||
| `RepoIngestor` | Git repositories: source files, commit history, and metadata |
|
||||
| `DBIngestor` | SQL databases via SQLAlchemy: tables, views, and custom queries |
|
||||
| `SnowflakeIngestor` | Snowflake data warehouse queries and table exports |
|
||||
| `DatabricksIngestor` | Databricks Unity Catalog metadata, Delta table queries, and lineage |
|
||||
| `ParquetIngestor` | Apache Parquet files and partitioned datasets with column selection |
|
||||
| `ArrowIngestor` | Apache Arrow IPC and Feather file processing |
|
||||
| `XMLIngestor` | XXE-safe XML parsing with optional XSD schema validation |
|
||||
| `EmailIngestor` | IMAP/POP3 email ingestion with attachment extraction |
|
||||
| `OntologyIngestor` | OWL/RDF/Turtle ontology file ingestion |
|
||||
@@ -440,6 +442,24 @@ result = ingest("ontology.ttl") # -> {"ontology": OntologyData}
|
||||
result = ingestor.ingest_query("SELECT * FROM documents")
|
||||
result = ingestor.ingest_table("documents")
|
||||
```
|
||||
|
||||
### DatabricksIngestor
|
||||
|
||||
```python
|
||||
from semantica.ingest import DatabricksIngestor
|
||||
import os
|
||||
|
||||
ingestor = DatabricksIngestor(
|
||||
host=os.getenv("DATABRICKS_HOST"),
|
||||
token=os.getenv("DATABRICKS_TOKEN"),
|
||||
http_path=os.getenv("DATABRICKS_HTTP_PATH"),
|
||||
catalog="main",
|
||||
schema="default",
|
||||
)
|
||||
result = ingestor.ingest_query("SELECT * FROM documents")
|
||||
result = ingestor.ingest_table("documents")
|
||||
lineage = ingestor.get_table_lineage("documents")
|
||||
```
|
||||
</Tab>
|
||||
<Tab title="Stream">
|
||||
### StreamIngestor
|
||||
@@ -628,4 +648,5 @@ result = ingest_file("source_path", method="my_format")
|
||||
- [Parse](parse) — Parse raw sources into structured text and tables.
|
||||
- [Pipeline](pipeline) — Orchestrate ingest as the first pipeline step.
|
||||
- [Snowflake Integration](../integrations/snowflake) — Snowflake-specific setup and authentication guide.
|
||||
- [Databricks Integration](../integrations/databricks) — Databricks Unity Catalog setup, authentication, and lineage guide.
|
||||
- [Provenance](provenance) — Track lineage from ingest through to inference.
|
||||
|
||||
@@ -495,6 +495,40 @@ result = engine.execute_pipeline(
|
||||
Delta detection uses SHA-256 checksums on source content. Only sources whose checksum differs from `base_version_id` are passed to downstream steps. For pipelines that run hourly or daily against a growing corpus, delta mode eliminates redundant re-embedding and re-extraction.
|
||||
</Note>
|
||||
|
||||
## SPARQL CONSTRUCT Template Steps
|
||||
|
||||
Use the `"construct_template"` step type to render and execute a [SPARQL CONSTRUCT template](triplet_store#sparql-construct-templates) as part of a pipeline. `store_backend` and `construct_template_registry` are execution-time resources, not step config — pass them to `execute_pipeline()`, the same way `delta_mode` steps receive `version_manager` and `triplet_store`:
|
||||
|
||||
```python
|
||||
from semantica.pipeline import PipelineBuilder, ExecutionEngine
|
||||
from semantica.triplet_store.construct_templates import construct_template_step_handler
|
||||
|
||||
builder = PipelineBuilder()
|
||||
builder.add_step(
|
||||
"apply_person_template",
|
||||
"construct_template",
|
||||
handler=construct_template_step_handler,
|
||||
template_name="person_to_foaf",
|
||||
params={"subject": "http://ex.org/p1", "name": "Alice", "age": 30},
|
||||
target_graph="http://ex.org/graphs/people",
|
||||
)
|
||||
pipeline = builder.build("person_pipeline")
|
||||
|
||||
engine = ExecutionEngine()
|
||||
result = engine.execute_pipeline(
|
||||
pipeline,
|
||||
data=None,
|
||||
store_backend=store, # required: a BlazegraphStore instance
|
||||
construct_template_registry=registry, # required: holds the registered template
|
||||
)
|
||||
|
||||
triplets = result.output # List[Triplet], already persisted via store.add_triplets
|
||||
```
|
||||
|
||||
<Note>
|
||||
`construct_template` steps raise `ProcessingError` if `store_backend` or `construct_template_registry` is missing from `execute_pipeline()`'s options, and `ValidationError` if `template_name` isn't registered.
|
||||
</Note>
|
||||
|
||||
## Schemas
|
||||
|
||||
<AccordionGroup>
|
||||
|
||||
@@ -155,13 +155,15 @@ prop_entry = manager.track_property_source(
|
||||
|
||||
### Batch Tracking
|
||||
|
||||
Batch tracking methods process items in blocks (default `batch_size=1000`) inside a shared transaction per block. Only entities or chunks that successfully commit to storage are added to the returned count, preventing rolled-back entries from inflating success counts.
|
||||
|
||||
```python
|
||||
entities = [
|
||||
{"id": "entity_1", "confidence": 0.9},
|
||||
{"id": "entity_2", "confidence": 0.85},
|
||||
]
|
||||
count = manager.track_entities_batch(entities, source="doc_1")
|
||||
# Returns the number of entities successfully tracked
|
||||
# Returns the number of entities successfully tracked and committed
|
||||
|
||||
chunks = [
|
||||
{"id": "chunk_0", "start_index": 0, "end_index": 500},
|
||||
@@ -219,10 +221,10 @@ cleared = manager.clear()
|
||||
|
||||
| Method | Returns | Description |
|
||||
| :------ | :------- | :----------- |
|
||||
| `track_entity(entity_id, source, metadata, **kwargs)` | `ProvenanceEntry` | Record entity provenance; checksum set automatically |
|
||||
| `track_relationship(relationship_id, source, metadata, **kwargs)` | `ProvenanceEntry` | Record relationship provenance |
|
||||
| `track_chunk(chunk_id, source_document, ...)` | `ProvenanceEntry` | Record chunk provenance with char offsets |
|
||||
| `track_property_source(entity_id, property_name, value, source)` | `ProvenanceEntry` | Record property-level source attribution |
|
||||
| `track_entity(entity_id, source, metadata, **kwargs)` | `Optional[ProvenanceEntry]` | Record entity provenance atomically; returns `ProvenanceEntry` on success, or `None`/existing entry on storage failure |
|
||||
| `track_relationship(relationship_id, source, metadata, **kwargs)` | `Optional[ProvenanceEntry]` | Record relationship provenance; returns `ProvenanceEntry` on success, or `None` on storage failure |
|
||||
| `track_chunk(chunk_id, source_document, ...)` | `Optional[ProvenanceEntry]` | Record chunk provenance with char offsets; returns `ProvenanceEntry` on success, or `None` on storage failure |
|
||||
| `track_property_source(entity_id, property_name, value, source)` | `Optional[ProvenanceEntry]` | Record property-level source attribution; returns `ProvenanceEntry` on success, or `None` on storage failure |
|
||||
| `track_entities_batch(entities, source)` | `int` | Batch-track entities; returns success count |
|
||||
| `track_chunks_batch(chunks, source_document)` | `int` | Batch-track chunks; returns success count |
|
||||
| `get_lineage(entity_id)` | `Dict[str, Any]` | Full lineage as aggregated dict |
|
||||
@@ -234,7 +236,7 @@ cleared = manager.clear()
|
||||
|
||||
## ProvenanceEntry Fields
|
||||
|
||||
`ProvenanceEntry` is the core dataclass. Every tracking method returns one:
|
||||
`ProvenanceEntry` is the core dataclass. Every tracking method returns one on success (or `None` on storage failure):
|
||||
|
||||
```python
|
||||
from semantica.provenance import ProvenanceEntry
|
||||
@@ -322,6 +324,9 @@ manager = ProvenanceManager(storage_path="provenance.db")
|
||||
|
||||
`SQLiteStorage` creates the database and indexes automatically on first use.
|
||||
|
||||
- **Atomicity & Concurrency**: Configures Write-Ahead Logging (`PRAGMA journal_mode=WAL`), `PRAGMA busy_timeout=5000`, and `PRAGMA synchronous=NORMAL`. Read-modify-write methods (`track_entity()`, `store()`) open a single connection and execute inside an immediate write transaction (`BEGIN IMMEDIATE`), ensuring these sequences are serialized across concurrent connections without leaving open file handles across calls. Plain reads (`retrieve()`, `trace_lineage()`) use a separate connection with no explicit write lock, so concurrent reads don't serialize behind writers or each other.
|
||||
- **Backward Compatibility**: Custom storage subclasses overriding `trace_lineage(self, entity_id)` remain backward compatible; `ProvenanceManager` inspects the override signature and automatically calls it with one argument if `max_depth` is unsupported.
|
||||
|
||||
## Tamper-Evident Checksums
|
||||
|
||||
`compute_checksum` and `verify_checksum` are auto-used by `track_entity` and all other tracking methods. You can also call them directly:
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
title: "Triplet Store Module"
|
||||
description: "RDF triple storage with SPARQL queries and bulk loading: Blazegraph, Apache Jena, and RDF4J."
|
||||
description: "Embedded and server-backed RDF storage with SPARQL queries and bulk loading."
|
||||
icon: "table"
|
||||
---
|
||||
|
||||
@@ -16,14 +16,15 @@ icon: "table"
|
||||
| `BlazegraphStore` | Blazegraph REST API: SPARQL 1.1 Update, namespace management |
|
||||
| `JenaStore` | Apache Jena: rdflib-backed, SPARQL read support via remote endpoint |
|
||||
| `RDF4JStore` | Eclipse RDF4J: REST API, transaction support |
|
||||
| `OxigraphStore` | Embedded SPARQL 1.1 store with in-memory and on-disk modes |
|
||||
|
||||
## What You Get
|
||||
|
||||
- **TripletStore** — Unified interface across Blazegraph, Apache Jena, and RDF4J: swap backends with one parameter.
|
||||
- **TripletStore** — Unified interface across embedded Oxigraph, Blazegraph, Apache Jena, and RDF4J: swap backends with one parameter.
|
||||
- **SPARQL** — Full SPARQL SELECT, ASK, CONSTRUCT, and UPDATE query support via `execute_query()`.
|
||||
- **Bulk Loading** — `add_triplets()` batches writes with configurable batch size, retry logic, and progress tracking.
|
||||
- **SKOS Vocabulary** — Built-in helpers: `add_skos_concept()` and `get_skos_concepts()` for controlled vocabulary management.
|
||||
- **Named Graphs** — Blazegraph and RDF4J support named graph scoping via `graph=` on `execute_query()`.
|
||||
- **Named Graphs** — Oxigraph, Blazegraph, and RDF4J support named graph scoping via `graph=` on `execute_query()`.
|
||||
- **Delta Computation** — `compute_delta(old_graph_uri, new_graph_uri)` returns added and removed triples between two named graph snapshots.
|
||||
|
||||
## Getting Started
|
||||
@@ -117,6 +118,25 @@ for row in result.bindings:
|
||||
## Backends
|
||||
|
||||
<Tabs>
|
||||
<Tab title="Oxigraph">
|
||||
```bash
|
||||
pip install "semantica[tripletstore-oxigraph]"
|
||||
```
|
||||
|
||||
```python
|
||||
# In-memory: no server process or files required
|
||||
store = TripletStore(backend="oxigraph")
|
||||
|
||||
# Persistent: reopen the same directory to reuse the data
|
||||
persistent_store = TripletStore(
|
||||
backend="oxigraph",
|
||||
path="./data/knowledge-graph",
|
||||
)
|
||||
```
|
||||
|
||||
**Best for:** local development, CI, desktop applications, and persistent
|
||||
single-process workloads without external infrastructure.
|
||||
</Tab>
|
||||
<Tab title="Blazegraph">
|
||||
```bash
|
||||
pip install requests
|
||||
@@ -172,6 +192,7 @@ for row in result.bindings:
|
||||
|
||||
| Backend | License | Named Graphs | Write via | Best For |
|
||||
| :------- | :------- | :------------ | :--------- | :-------- |
|
||||
| Oxigraph | Apache 2.0 / MIT | Yes | Embedded native API | Local, CI, on-disk |
|
||||
| Blazegraph | Open source | Yes | SPARQL Update REST | High triple count, SPARQL 1.1 |
|
||||
| Apache Jena | Apache 2.0 | No (rdflib backend) | rdflib in-process | Local dev, read queries |
|
||||
| RDF4J | Eclipse 1.0 | Yes | REST API N-Triples | Enterprise Java, transactions |
|
||||
@@ -180,7 +201,9 @@ for row in result.bindings:
|
||||
</Tabs>
|
||||
|
||||
<Tip>
|
||||
**Use Apache Jena for development, Blazegraph for production.** Jena initializes with rdflib in-memory: no server required for local testing. Switch to Blazegraph for high-throughput persistent workloads by changing `backend=`.
|
||||
**Use Oxigraph for zero-infrastructure development and local persistence.**
|
||||
Switch to a server-backed store for distributed production deployments by
|
||||
changing `backend=`.
|
||||
</Tip>
|
||||
|
||||
## Triplet Object
|
||||
@@ -277,6 +300,65 @@ store.execute_query("""
|
||||
**`execute_query()` returns `QueryResult`, not a list.** Iterate `result.bindings`, not `result` directly. Each binding is a dict mapping variable name → `{"value": ..., "type": ...}`.
|
||||
</Warning>
|
||||
|
||||
## SPARQL CONSTRUCT Templates
|
||||
|
||||
`semantica.triplet_store.construct_templates` provides parameterized SPARQL `CONSTRUCT` query templates: define a reusable query once, substitute typed parameters safely, and persist the resulting triples in one call. This is available for the **Blazegraph backend only** (see [Backends](#backends) above) — `BlazegraphStore.execute_sparql()` is the only backend with CONSTRUCT-aware RDF parsing.
|
||||
|
||||
```python
|
||||
from semantica.triplet_store.construct_templates import (
|
||||
ConstructTemplate,
|
||||
ParameterDescriptor,
|
||||
ConstructTemplateRegistry,
|
||||
render_construct_template,
|
||||
execute_construct_template,
|
||||
)
|
||||
from semantica.triplet_store import BlazegraphStore
|
||||
|
||||
# Define and register a template
|
||||
template = ConstructTemplate(
|
||||
name="person_to_foaf",
|
||||
description="Maps a person record subject to a foaf:name triple",
|
||||
construct_query="""
|
||||
PREFIX foaf: <http://xmlns.com/foaf/0.1/>
|
||||
CONSTRUCT { {{subject}} foaf:name {{name}} ; foaf:age {{age}} }
|
||||
WHERE { {{subject}} a <http://ex.org/Person> }
|
||||
""",
|
||||
parameters=[
|
||||
ParameterDescriptor(name="subject", type="uri", required=True),
|
||||
ParameterDescriptor(name="name", type="literal", required=True),
|
||||
ParameterDescriptor(
|
||||
name="age", type="typed-literal", required=False, default=0,
|
||||
datatype="xsd:integer",
|
||||
),
|
||||
],
|
||||
)
|
||||
|
||||
registry = ConstructTemplateRegistry()
|
||||
registry.register(template)
|
||||
|
||||
# Render only: inspect the substituted SPARQL string, no network call
|
||||
sparql = render_construct_template(
|
||||
registry.get("person_to_foaf"),
|
||||
params={"subject": "http://ex.org/p1", "name": "Alice", "age": 30},
|
||||
)
|
||||
|
||||
# Render + execute + persist in one call
|
||||
store = BlazegraphStore(endpoint="http://localhost:9999/blazegraph", namespace="kb")
|
||||
triplets = execute_construct_template(
|
||||
template=registry.get("person_to_foaf"),
|
||||
params={"subject": "http://ex.org/p1", "name": "Alice", "age": 30},
|
||||
store_backend=store,
|
||||
target_graph="http://ex.org/graphs/people",
|
||||
)
|
||||
# triplets: List[Triplet], already persisted via store.add_triplets
|
||||
```
|
||||
|
||||
Each `ParameterDescriptor.type` controls how its value is rendered: `"uri"` values are validated against an allowlist and wrapped in `<...>`, `"literal"` values are escaped and quoted, and `"typed-literal"` values require a `datatype` (e.g. `"xsd:integer"`) and render unquoted for numeric/boolean XSD types. Placeholders use `{{param}}` rather than SPARQL's own `?param` syntax so template placeholders are never confused with real SPARQL variables in the query body.
|
||||
|
||||
<Note>
|
||||
CONSTRUCT templates are Blazegraph-only. `execute_construct_template()` raises `ProcessingError` if `store_backend` does not implement both `execute_sparql()` and `add_triplets()`.
|
||||
</Note>
|
||||
|
||||
## SPARQL Result Pagination
|
||||
|
||||
For large result sets, paginate with LIMIT and OFFSET:
|
||||
@@ -305,10 +387,10 @@ while True:
|
||||
|
||||
## Named Graph Scoping
|
||||
|
||||
Blazegraph and RDF4J support named graphs. Scope `execute_query()` to a named graph with the `graph=` parameter:
|
||||
Oxigraph, Blazegraph, and RDF4J support named graphs. Scope `execute_query()` to a named graph with the `graph=` parameter:
|
||||
|
||||
```python
|
||||
# Add a triplet: named graph stored in metadata or backend-specific API
|
||||
# Add a triplet to a named graph
|
||||
from semantica.semantic_extract.types import Triplet
|
||||
|
||||
t = Triplet(
|
||||
@@ -316,7 +398,7 @@ t = Triplet(
|
||||
predicate="http://example.org/p",
|
||||
object="http://example.org/b",
|
||||
)
|
||||
store.add_triplet(t) # named graph targeting requires backend-specific API
|
||||
store.add_triplet(t, graph="http://example.org/graph1")
|
||||
|
||||
# Query a named graph via FROM clause in SPARQL
|
||||
result = store.execute_query("""
|
||||
@@ -334,11 +416,14 @@ result = store.execute_query("""
|
||||
```
|
||||
|
||||
<Note>
|
||||
Named graph support is only available for Blazegraph and RDF4J backends. The `graph=` parameter is silently ignored for the Jena backend.
|
||||
Named graph query scoping is available for Oxigraph, Blazegraph, and RDF4J.
|
||||
The `graph=` query parameter is silently ignored for the Jena backend.
|
||||
</Note>
|
||||
|
||||
<Tip>
|
||||
**Use named graphs to isolate sources.** Pass `graph="http://example.org/source_A"` to `execute_query()` to scope a query to a specific named graph. Blazegraph and RDF4J support named graphs; Jena (rdflib backend) does not.
|
||||
**Use named graphs to isolate sources.** Pass `graph="http://example.org/source_A"`
|
||||
to writes and `execute_query()` to scope both storage and retrieval. Oxigraph,
|
||||
Blazegraph, and RDF4J support named graph query scoping.
|
||||
</Tip>
|
||||
|
||||
## Bulk Loading
|
||||
|
||||
Generated
+104
-360
@@ -37,7 +37,7 @@
|
||||
"@types/react-dom": "^19.2.3",
|
||||
"@vitejs/plugin-react": "^4.3.0",
|
||||
"babel-plugin-react-compiler": "^1.0.0",
|
||||
"eslint": "^9.39.4",
|
||||
"eslint": "^10.8.0",
|
||||
"eslint-plugin-react-hooks": "^7.0.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.2",
|
||||
"globals": "^17.4.0",
|
||||
@@ -836,81 +836,44 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/config-array": {
|
||||
"version": "0.21.2",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/config-array/-/config-array-0.21.2.tgz",
|
||||
"integrity": "sha512-nJl2KGTlrf9GjLimgIru+V/mzgSK0ABCDQRvxw5BjURL7WfH5uoWmizbH7QB6MmnMBd8cIC9uceWnezL1VZWWw==",
|
||||
"version": "0.23.5",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/config-array/-/config-array-0.23.5.tgz",
|
||||
"integrity": "sha512-Y3kKLvC1dvTOT+oGlqNQ1XLqK6D1HU2YXPc52NmAlJZbMMWDzGYXMiPRJ8TYD39muD/OTjlZmNJ4ib7dvSrMBA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"@eslint/object-schema": "^2.1.7",
|
||||
"@eslint/object-schema": "^3.0.5",
|
||||
"debug": "^4.3.1",
|
||||
"minimatch": "^3.1.5"
|
||||
"minimatch": "^10.2.4"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/config-helpers": {
|
||||
"version": "0.4.2",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/config-helpers/-/config-helpers-0.4.2.tgz",
|
||||
"integrity": "sha512-gBrxN88gOIf3R7ja5K9slwNayVcZgK6SOUORm2uBzTeIEfeVaIhOpCtTox3P6R7o2jLFwLFTLnC7kU/RGcYEgw==",
|
||||
"version": "0.7.0",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/config-helpers/-/config-helpers-0.7.0.tgz",
|
||||
"integrity": "sha512-DObd/KKUsU+FaFv4PLxSRenpXfQWmPXXP3pPZ6/K1PCrMu2vQpMDMuQe/BqYeoLcz8ro0bVDF1RxOJgfVEdhUw==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"@eslint/core": "^0.17.0"
|
||||
"@eslint/core": "^1.2.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/core": {
|
||||
"version": "0.17.0",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/core/-/core-0.17.0.tgz",
|
||||
"integrity": "sha512-yL/sLrpmtDaFEiUj1osRP4TI2MDz1AddJL+jZ7KSqvBuliN4xqYY54IfdN8qD8Toa6g1iloph1fxQNkjOxrrpQ==",
|
||||
"version": "1.2.1",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/core/-/core-1.2.1.tgz",
|
||||
"integrity": "sha512-MwcE1P+AZ4C6DWlpin/OmOA54mmIZ/+xZuJiQd4SyB29oAJjN30UW9wkKNptW2ctp4cEsvhlLY/CsQ1uoHDloQ==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"@types/json-schema": "^7.0.15"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/eslintrc": {
|
||||
"version": "3.3.5",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/eslintrc/-/eslintrc-3.3.5.tgz",
|
||||
"integrity": "sha512-4IlJx0X0qftVsN5E+/vGujTRIFtwuLbNsVUe7TO6zYPDR1O6nFwvwhIKEKSrl6dZchmYBITazxKoUYOjdtjlRg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"ajv": "^6.14.0",
|
||||
"debug": "^4.3.2",
|
||||
"espree": "^10.0.1",
|
||||
"globals": "^14.0.0",
|
||||
"ignore": "^5.2.0",
|
||||
"import-fresh": "^3.2.1",
|
||||
"js-yaml": "^4.1.1",
|
||||
"minimatch": "^3.1.5",
|
||||
"strip-json-comments": "^3.1.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://opencollective.com/eslint"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/eslintrc/node_modules/globals": {
|
||||
"version": "14.0.0",
|
||||
"resolved": "https://registry.npmjs.org/globals/-/globals-14.0.0.tgz",
|
||||
"integrity": "sha512-oahGvuMGQlPw/ivIYBjVSrWAfWLBeku5tpPE2fOPLi+WHffIWbuh2tCjhyQhTBPMf5E9jDEH4FOmTYgYwbKwtQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/js": {
|
||||
@@ -927,27 +890,27 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/object-schema": {
|
||||
"version": "2.1.7",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/object-schema/-/object-schema-2.1.7.tgz",
|
||||
"integrity": "sha512-VtAOaymWVfZcmZbp6E2mympDIHvyjXs/12LqWYjVw6qjrfF+VK+fyG33kChz3nnK+SU5/NeHOqrTEHS8sXO3OA==",
|
||||
"version": "3.0.5",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/object-schema/-/object-schema-3.0.5.tgz",
|
||||
"integrity": "sha512-vqTaUEgxzm+YDSdElad6PiRoX4t8VGDjCtt05zn4nU810UIx/uNEV7/lZJ6KwFThKZOzOxzXy48da+No7HZaMw==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
}
|
||||
},
|
||||
"node_modules/@eslint/plugin-kit": {
|
||||
"version": "0.4.1",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/plugin-kit/-/plugin-kit-0.4.1.tgz",
|
||||
"integrity": "sha512-43/qtrDUokr7LJqoF2c3+RInu/t4zfrpYdoSDfYyhg52rwLV6TnOvdG4fXm7IkSB3wErkcmJS9iEhjVtOSEjjA==",
|
||||
"version": "0.7.2",
|
||||
"resolved": "https://registry.npmjs.org/@eslint/plugin-kit/-/plugin-kit-0.7.2.tgz",
|
||||
"integrity": "sha512-+CNAzxglkrpNf/kKywqQfk74QjtceuOE7Qm+AF8miRvPF/wmmK5+OJOgVh3AVTT3RP2mH3+FOaxlE5v72owk0A==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"dependencies": {
|
||||
"@eslint/core": "^0.17.0",
|
||||
"@eslint/core": "^1.2.1",
|
||||
"levn": "^0.4.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
}
|
||||
},
|
||||
"node_modules/@humanfs/core": {
|
||||
@@ -1602,6 +1565,13 @@
|
||||
"@types/d3-selection": "*"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/esrecurse": {
|
||||
"version": "4.3.1",
|
||||
"resolved": "https://registry.npmjs.org/@types/esrecurse/-/esrecurse-4.3.1.tgz",
|
||||
"integrity": "sha512-xJBAbDifo5hpffDBuHl0Y8ywswbiAp/Wi7Y/GtAgSlZyIABppyurxVueOPE8LUQOxdlgi6Zqce7uoEpqNTeiUw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/estree": {
|
||||
"version": "1.0.8",
|
||||
"resolved": "https://registry.npmjs.org/@types/estree/-/estree-1.0.8.tgz",
|
||||
@@ -1849,45 +1819,6 @@
|
||||
"typescript": ">=4.8.4 <6.1.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript-eslint/typescript-estree/node_modules/balanced-match": {
|
||||
"version": "4.0.4",
|
||||
"resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-4.0.4.tgz",
|
||||
"integrity": "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "18 || 20 || >=22"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript-eslint/typescript-estree/node_modules/brace-expansion": {
|
||||
"version": "5.0.6",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz",
|
||||
"integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"balanced-match": "^4.0.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": "18 || 20 || >=22"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript-eslint/typescript-estree/node_modules/minimatch": {
|
||||
"version": "10.2.5",
|
||||
"resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.5.tgz",
|
||||
"integrity": "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg==",
|
||||
"dev": true,
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"brace-expansion": "^5.0.5"
|
||||
},
|
||||
"engines": {
|
||||
"node": "18 || 20 || >=22"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/isaacs"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript-eslint/typescript-estree/node_modules/semver": {
|
||||
"version": "7.7.4",
|
||||
"resolved": "https://registry.npmjs.org/semver/-/semver-7.7.4.tgz",
|
||||
@@ -1943,19 +1874,6 @@
|
||||
"url": "https://opencollective.com/typescript-eslint"
|
||||
}
|
||||
},
|
||||
"node_modules/@typescript-eslint/visitor-keys/node_modules/eslint-visitor-keys": {
|
||||
"version": "5.0.1",
|
||||
"resolved": "https://registry.npmjs.org/eslint-visitor-keys/-/eslint-visitor-keys-5.0.1.tgz",
|
||||
"integrity": "sha512-tD40eHxA35h0PEIZNeIjkHoDR4YjjJp34biM0mDvplBe//mB+IHCqHDGV7pxF+7MklTvighcCPPZC7ynWyjdTA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://opencollective.com/eslint"
|
||||
}
|
||||
},
|
||||
"node_modules/@vitejs/plugin-react": {
|
||||
"version": "4.7.0",
|
||||
"resolved": "https://registry.npmjs.org/@vitejs/plugin-react/-/plugin-react-4.7.0.tgz",
|
||||
@@ -2016,9 +1934,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/acorn": {
|
||||
"version": "8.16.0",
|
||||
"resolved": "https://registry.npmjs.org/acorn/-/acorn-8.16.0.tgz",
|
||||
"integrity": "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw==",
|
||||
"version": "8.17.0",
|
||||
"resolved": "https://registry.npmjs.org/acorn/-/acorn-8.17.0.tgz",
|
||||
"integrity": "sha512-xRQbDb9BnwDafYNn6Vwl839DYVjqXYb1XVGtWAZ1kcDc6iwAL4hg3B1dZlRiuENFeO2H53gFG3in621AdERVAg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"bin": {
|
||||
@@ -2039,9 +1957,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/ajv": {
|
||||
"version": "6.14.0",
|
||||
"resolved": "https://registry.npmjs.org/ajv/-/ajv-6.14.0.tgz",
|
||||
"integrity": "sha512-IWrosm/yrn43eiKqkfkHis7QioDleaXQHdDVPKg0FSwwd/DuvyX79TZnFOnYpB7dcsFAMmtFztZuXPDvSePkFw==",
|
||||
"version": "6.15.0",
|
||||
"resolved": "https://registry.npmjs.org/ajv/-/ajv-6.15.0.tgz",
|
||||
"integrity": "sha512-fgFx7Hfoq60ytK2c7DhnF8jIvzYgOMxfugjLOSMHjLIPgenqa7S7oaagATUq99mV6IYvN2tRmC0wnTYX6iPbMw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -2055,29 +1973,6 @@
|
||||
"url": "https://github.com/sponsors/epoberezkin"
|
||||
}
|
||||
},
|
||||
"node_modules/ansi-styles": {
|
||||
"version": "4.3.0",
|
||||
"resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-4.3.0.tgz",
|
||||
"integrity": "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color-convert": "^2.0.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/chalk/ansi-styles?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/argparse": {
|
||||
"version": "2.0.1",
|
||||
"resolved": "https://registry.npmjs.org/argparse/-/argparse-2.0.1.tgz",
|
||||
"integrity": "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q==",
|
||||
"dev": true,
|
||||
"license": "Python-2.0"
|
||||
},
|
||||
"node_modules/attr-accept": {
|
||||
"version": "2.2.5",
|
||||
"resolved": "https://registry.npmjs.org/attr-accept/-/attr-accept-2.2.5.tgz",
|
||||
@@ -2098,11 +1993,14 @@
|
||||
}
|
||||
},
|
||||
"node_modules/balanced-match": {
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-1.0.2.tgz",
|
||||
"integrity": "sha512-3oSeUO0TMV67hN1AmbXsK4yaqU7tjiHlbxRDZOpH0KW9+CeX4bRAaX0Anxt0tx2MrpRpWwQaPwIlISEJhYU5Pw==",
|
||||
"version": "4.0.4",
|
||||
"resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-4.0.4.tgz",
|
||||
"integrity": "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "18 || 20 || >=22"
|
||||
}
|
||||
},
|
||||
"node_modules/baseline-browser-mapping": {
|
||||
"version": "2.10.20",
|
||||
@@ -2118,14 +2016,16 @@
|
||||
}
|
||||
},
|
||||
"node_modules/brace-expansion": {
|
||||
"version": "1.1.14",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.14.tgz",
|
||||
"integrity": "sha512-MWPGfDxnyzKU7rNOW9SP/c50vi3xrmrua/+6hfPbCS2ABNWfx24vPidzvC7krjU/RTo235sV776ymlsMtGKj8g==",
|
||||
"version": "5.0.8",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.8.tgz",
|
||||
"integrity": "sha512-JZyDyq3D4AUifKTPOB7DELf6XsB3WdPuNxCtob1vFXPsSXhdAiHBWJ/tJ8HAc9aH84BK+5JFZLNkJKx3G9kzQg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"balanced-match": "^1.0.0",
|
||||
"concat-map": "0.0.1"
|
||||
"balanced-match": "^4.0.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": "20 || >=22"
|
||||
}
|
||||
},
|
||||
"node_modules/browserslist": {
|
||||
@@ -2162,16 +2062,6 @@
|
||||
"node": "^6 || ^7 || ^8 || ^9 || ^10 || ^11 || ^12 || >=13.7"
|
||||
}
|
||||
},
|
||||
"node_modules/callsites": {
|
||||
"version": "3.1.0",
|
||||
"resolved": "https://registry.npmjs.org/callsites/-/callsites-3.1.0.tgz",
|
||||
"integrity": "sha512-P8BjAsXvZS+VIDUI11hHCQEv74YT67YUi5JJFNWIqL235sBmjX4+qx9Muvls5ivyNENctx46xQLQ3aTuE7ssaQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=6"
|
||||
}
|
||||
},
|
||||
"node_modules/caniuse-lite": {
|
||||
"version": "1.0.30001788",
|
||||
"resolved": "https://registry.npmjs.org/caniuse-lite/-/caniuse-lite-1.0.30001788.tgz",
|
||||
@@ -2193,49 +2083,12 @@
|
||||
],
|
||||
"license": "CC-BY-4.0"
|
||||
},
|
||||
"node_modules/chalk": {
|
||||
"version": "4.1.2",
|
||||
"resolved": "https://registry.npmjs.org/chalk/-/chalk-4.1.2.tgz",
|
||||
"integrity": "sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"ansi-styles": "^4.1.0",
|
||||
"supports-color": "^7.1.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=10"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/chalk/chalk?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/classcat": {
|
||||
"version": "5.0.5",
|
||||
"resolved": "https://registry.npmjs.org/classcat/-/classcat-5.0.5.tgz",
|
||||
"integrity": "sha512-JhZUT7JFcQy/EzW605k/ktHtncoo9vnyW/2GspNYwFlN1C/WmjuV/xtS04e9SOkL2sTdw0VAZ2UGCcQ9lR6p6w==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/color-convert": {
|
||||
"version": "2.0.1",
|
||||
"resolved": "https://registry.npmjs.org/color-convert/-/color-convert-2.0.1.tgz",
|
||||
"integrity": "sha512-RRECPsj7iu/xb5oKYcsFHSppFNnsj/52OVTRKb4zP5onXwVF3zVmmToNcOfGC+CRDpfK/U584fMg38ZHCaElKQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color-name": "~1.1.4"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=7.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/color-name": {
|
||||
"version": "1.1.4",
|
||||
"resolved": "https://registry.npmjs.org/color-name/-/color-name-1.1.4.tgz",
|
||||
"integrity": "sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/commander": {
|
||||
"version": "2.20.3",
|
||||
"resolved": "https://registry.npmjs.org/commander/-/commander-2.20.3.tgz",
|
||||
@@ -2253,13 +2106,6 @@
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/concat-map": {
|
||||
"version": "0.0.1",
|
||||
"resolved": "https://registry.npmjs.org/concat-map/-/concat-map-0.0.1.tgz",
|
||||
"integrity": "sha512-/Srv4dswyQNBfohGpz9o6Yb3Gz3SrUDqBH5rTuhGR7ahtlbYKnVxw2bCFMRljaA7EXHaXZ8wsHdodFvbkhKmqg==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/convert-source-map": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/convert-source-map/-/convert-source-map-2.0.0.tgz",
|
||||
@@ -2447,9 +2293,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/dompurify": {
|
||||
"version": "3.4.11",
|
||||
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.11.tgz",
|
||||
"integrity": "sha512-zhlUV12GsaRzMsf9q5M254YhA4+VuF0fG+QFqu6aYpoGlKtz+w8//jBcGVYBgQkR5GHjUomejY84AV+/uPbWdw==",
|
||||
"version": "3.4.13",
|
||||
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.13.tgz",
|
||||
"integrity": "sha512-2vmYIoqjze2d+kakP8S/nS5shfsl587kzwEjcGlTdiksUVgFHnFCsLYDVj/JNqJVOQZGSYBTmuycv0PodwmnMQ==",
|
||||
"license": "(MPL-2.0 OR Apache-2.0)",
|
||||
"peer": true,
|
||||
"optionalDependencies": {
|
||||
@@ -2529,33 +2375,33 @@
|
||||
}
|
||||
},
|
||||
"node_modules/eslint": {
|
||||
"version": "9.39.4",
|
||||
"resolved": "https://registry.npmjs.org/eslint/-/eslint-9.39.4.tgz",
|
||||
"integrity": "sha512-XoMjdBOwe/esVgEvLmNsD3IRHkm7fbKIUGvrleloJXUZgDHig2IPWNniv+GwjyJXzuNqVjlr5+4yVUZjycJwfQ==",
|
||||
"version": "10.8.0",
|
||||
"resolved": "https://registry.npmjs.org/eslint/-/eslint-10.8.0.tgz",
|
||||
"integrity": "sha512-nuKKvN+oIBO0koN7Tm7dlkmnkc21mtt0QJLwAKzjLq14y6lRTdVG36MZHJ8eQHwdJMwZbQNMlPOYedMq/oVJvQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"workspaces": [
|
||||
"packages/*"
|
||||
],
|
||||
"dependencies": {
|
||||
"@eslint-community/eslint-utils": "^4.8.0",
|
||||
"@eslint-community/regexpp": "^4.12.1",
|
||||
"@eslint/config-array": "^0.21.2",
|
||||
"@eslint/config-helpers": "^0.4.2",
|
||||
"@eslint/core": "^0.17.0",
|
||||
"@eslint/eslintrc": "^3.3.5",
|
||||
"@eslint/js": "9.39.4",
|
||||
"@eslint/plugin-kit": "^0.4.1",
|
||||
"@eslint-community/regexpp": "^4.12.2",
|
||||
"@eslint/config-array": "^0.23.5",
|
||||
"@eslint/config-helpers": "^0.7.0",
|
||||
"@eslint/core": "^1.2.1",
|
||||
"@eslint/plugin-kit": "^0.7.2",
|
||||
"@humanfs/node": "^0.16.6",
|
||||
"@humanwhocodes/module-importer": "^1.0.1",
|
||||
"@humanwhocodes/retry": "^0.4.2",
|
||||
"@types/estree": "^1.0.6",
|
||||
"ajv": "^6.14.0",
|
||||
"chalk": "^4.0.0",
|
||||
"cross-spawn": "^7.0.6",
|
||||
"debug": "^4.3.2",
|
||||
"escape-string-regexp": "^4.0.0",
|
||||
"eslint-scope": "^8.4.0",
|
||||
"eslint-visitor-keys": "^4.2.1",
|
||||
"espree": "^10.4.0",
|
||||
"esquery": "^1.5.0",
|
||||
"eslint-scope": "^9.1.2",
|
||||
"eslint-visitor-keys": "^5.0.1",
|
||||
"espree": "^11.2.0",
|
||||
"esquery": "^1.7.0",
|
||||
"esutils": "^2.0.2",
|
||||
"fast-deep-equal": "^3.1.3",
|
||||
"file-entry-cache": "^8.0.0",
|
||||
@@ -2565,8 +2411,7 @@
|
||||
"imurmurhash": "^0.1.4",
|
||||
"is-glob": "^4.0.0",
|
||||
"json-stable-stringify-without-jsonify": "^1.0.1",
|
||||
"lodash.merge": "^4.6.2",
|
||||
"minimatch": "^3.1.5",
|
||||
"minimatch": "^10.2.5",
|
||||
"natural-compare": "^1.4.0",
|
||||
"optionator": "^0.9.3"
|
||||
},
|
||||
@@ -2574,7 +2419,7 @@
|
||||
"eslint": "bin/eslint.js"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://eslint.org/donate"
|
||||
@@ -2619,48 +2464,50 @@
|
||||
}
|
||||
},
|
||||
"node_modules/eslint-scope": {
|
||||
"version": "8.4.0",
|
||||
"resolved": "https://registry.npmjs.org/eslint-scope/-/eslint-scope-8.4.0.tgz",
|
||||
"integrity": "sha512-sNXOfKCn74rt8RICKMvJS7XKV/Xk9kA7DyJr8mJik3S7Cwgy3qlkkmyS2uQB3jiJg6VNdZd/pDBJu0nvG2NlTg==",
|
||||
"version": "9.1.2",
|
||||
"resolved": "https://registry.npmjs.org/eslint-scope/-/eslint-scope-9.1.2.tgz",
|
||||
"integrity": "sha512-xS90H51cKw0jltxmvmHy2Iai1LIqrfbw57b79w/J7MfvDfkIkFZ+kj6zC3BjtUwh150HsSSdxXZcsuv72miDFQ==",
|
||||
"dev": true,
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"@types/esrecurse": "^4.3.1",
|
||||
"@types/estree": "^1.0.8",
|
||||
"esrecurse": "^4.3.0",
|
||||
"estraverse": "^5.2.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://opencollective.com/eslint"
|
||||
}
|
||||
},
|
||||
"node_modules/eslint-visitor-keys": {
|
||||
"version": "4.2.1",
|
||||
"resolved": "https://registry.npmjs.org/eslint-visitor-keys/-/eslint-visitor-keys-4.2.1.tgz",
|
||||
"integrity": "sha512-Uhdk5sfqcee/9H/rCOJikYz67o0a2Tw2hGRPOG2Y1R2dg7brRe1uG0yaNQDHu+TO/uQPF/5eCapvYSmHUjt7JQ==",
|
||||
"version": "5.0.1",
|
||||
"resolved": "https://registry.npmjs.org/eslint-visitor-keys/-/eslint-visitor-keys-5.0.1.tgz",
|
||||
"integrity": "sha512-tD40eHxA35h0PEIZNeIjkHoDR4YjjJp34biM0mDvplBe//mB+IHCqHDGV7pxF+7MklTvighcCPPZC7ynWyjdTA==",
|
||||
"dev": true,
|
||||
"license": "Apache-2.0",
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://opencollective.com/eslint"
|
||||
}
|
||||
},
|
||||
"node_modules/espree": {
|
||||
"version": "10.4.0",
|
||||
"resolved": "https://registry.npmjs.org/espree/-/espree-10.4.0.tgz",
|
||||
"integrity": "sha512-j6PAQ2uUr79PZhBjP5C5fhl8e39FmRnOjsD5lGnWrFU8i2G776tBK7+nP8KuQUTTyAZUwfQqXAgrVH5MbH9CYQ==",
|
||||
"version": "11.2.0",
|
||||
"resolved": "https://registry.npmjs.org/espree/-/espree-11.2.0.tgz",
|
||||
"integrity": "sha512-7p3DrVEIopW1B1avAGLuCSh1jubc01H2JHc8B4qqGblmg5gI9yumBgACjWo4JlIc04ufug4xJ3SQI8HkS/Rgzw==",
|
||||
"dev": true,
|
||||
"license": "BSD-2-Clause",
|
||||
"dependencies": {
|
||||
"acorn": "^8.15.0",
|
||||
"acorn": "^8.16.0",
|
||||
"acorn-jsx": "^5.3.2",
|
||||
"eslint-visitor-keys": "^4.2.1"
|
||||
"eslint-visitor-keys": "^5.0.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^18.18.0 || ^20.9.0 || >=21.1.0"
|
||||
"node": "^20.19.0 || ^22.13.0 || >=24"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://opencollective.com/eslint"
|
||||
@@ -2972,16 +2819,6 @@
|
||||
"graphology-types": ">=0.23.0"
|
||||
}
|
||||
},
|
||||
"node_modules/has-flag": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/has-flag/-/has-flag-4.0.0.tgz",
|
||||
"integrity": "sha512-EykJT/Q1KjTWctppgIAgfSO0tKVuZUjhgMr17kqTumMl6Afv3EISleU7qZUzoXDFTAHTDC4NOoG/ZxU3EvlMPQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/hermes-estree": {
|
||||
"version": "0.25.1",
|
||||
"resolved": "https://registry.npmjs.org/hermes-estree/-/hermes-estree-0.25.1.tgz",
|
||||
@@ -3018,23 +2855,6 @@
|
||||
"node": ">= 4"
|
||||
}
|
||||
},
|
||||
"node_modules/import-fresh": {
|
||||
"version": "3.3.1",
|
||||
"resolved": "https://registry.npmjs.org/import-fresh/-/import-fresh-3.3.1.tgz",
|
||||
"integrity": "sha512-TR3KfrTZTYLPB6jUjfx6MF9WcWrHL9su5TObK4ZkYgBdWKPOFoSoQIdEuTuR82pmtxH2spWG9h6etwfr1pLBqQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"parent-module": "^1.0.0",
|
||||
"resolve-from": "^4.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/imurmurhash": {
|
||||
"version": "0.1.4",
|
||||
"resolved": "https://registry.npmjs.org/imurmurhash/-/imurmurhash-0.1.4.tgz",
|
||||
@@ -3081,29 +2901,6 @@
|
||||
"integrity": "sha512-RdJUflcE3cUzKiMqQgsCu06FPu9UdIJO0beYbPhHN4k6apgJtifcoCtT9bcxOpYBtpD2kCM6Sbzg4CausW/PKQ==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz",
|
||||
"integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/puzrin"
|
||||
},
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/nodeca"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/jsesc": {
|
||||
"version": "3.1.0",
|
||||
"resolved": "https://registry.npmjs.org/jsesc/-/jsesc-3.1.0.tgz",
|
||||
@@ -3198,13 +2995,6 @@
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/lodash.merge": {
|
||||
"version": "4.6.2",
|
||||
"resolved": "https://registry.npmjs.org/lodash.merge/-/lodash.merge-4.6.2.tgz",
|
||||
"integrity": "sha512-0KpjqXRVvrYyCsX1swR/XTK0va6VQkQM6MNo7PqW77ByjAhoARA8EfrP1N4+KlKj8YS0ZUCtRT/YUuhyYDujIQ==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/loose-envify": {
|
||||
"version": "1.4.0",
|
||||
"resolved": "https://registry.npmjs.org/loose-envify/-/loose-envify-1.4.0.tgz",
|
||||
@@ -3256,16 +3046,19 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/minimatch": {
|
||||
"version": "3.1.5",
|
||||
"resolved": "https://registry.npmjs.org/minimatch/-/minimatch-3.1.5.tgz",
|
||||
"integrity": "sha512-VgjWUsnnT6n+NUk6eZq77zeFdpW2LWDzP6zFGrCbHXiYNul5Dzqk2HHQ5uFH2DNW5Xbp8+jVzaeNt94ssEEl4w==",
|
||||
"version": "10.2.5",
|
||||
"resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.5.tgz",
|
||||
"integrity": "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg==",
|
||||
"dev": true,
|
||||
"license": "ISC",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"brace-expansion": "^1.1.7"
|
||||
"brace-expansion": "^5.0.5"
|
||||
},
|
||||
"engines": {
|
||||
"node": "*"
|
||||
"node": "18 || 20 || >=22"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/isaacs"
|
||||
}
|
||||
},
|
||||
"node_modules/mnemonist": {
|
||||
@@ -3306,9 +3099,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/nanoid": {
|
||||
"version": "3.3.11",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.11.tgz",
|
||||
"integrity": "sha512-N8SpfPUnUp1bK+PMYW8qSWdl9U+wwNWI4QKxOYDy9JAro3WMX7p2OeVRF9v+347pnakNevPmiHhNmZ2HbFA76w==",
|
||||
"version": "3.3.16",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.16.tgz",
|
||||
"integrity": "sha512-bzlKTyNJ7+LdGIIwy8ijFpIqEQIvafahV7eYykJ8Cvh42EdJeODoJ6gUJXpQJvej1BddH8OqTXZNE/KfbWAu8Q==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -3412,19 +3205,6 @@
|
||||
"mnemonist": "^0.39.2"
|
||||
}
|
||||
},
|
||||
"node_modules/parent-module": {
|
||||
"version": "1.0.1",
|
||||
"resolved": "https://registry.npmjs.org/parent-module/-/parent-module-1.0.1.tgz",
|
||||
"integrity": "sha512-GQ2EWRpQV8/o+Aw8YqtfZZPfNRWZYkbidE9k5rpl/hC3vtHHBfGm2Ifi6qWV+coDGkrUKZAxE3Lot5kcsRlh+g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"callsites": "^3.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6"
|
||||
}
|
||||
},
|
||||
"node_modules/path-exists": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/path-exists/-/path-exists-4.0.0.tgz",
|
||||
@@ -3510,9 +3290,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/postcss": {
|
||||
"version": "8.5.10",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.10.tgz",
|
||||
"integrity": "sha512-pMMHxBOZKFU6HgAZ4eyGnwXF/EvPGGqUr0MnZ5+99485wwW41kW91A4LOGxSHhgugZmSChL5AlElNdwlNgcnLQ==",
|
||||
"version": "8.5.23",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.23.tgz",
|
||||
"integrity": "sha512-g50586zr4bZmwFiTlflMu8E0bDTb5I5gertgwAKmsdUlTQIhZtunzUlD1WSzwcVWPoAVpsrA6vlfCD7oXvRwgg==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -3530,7 +3310,7 @@
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"nanoid": "^3.3.11",
|
||||
"nanoid": "^3.3.16",
|
||||
"picocolors": "^1.1.1",
|
||||
"source-map-js": "^1.2.1"
|
||||
},
|
||||
@@ -3712,16 +3492,6 @@
|
||||
"integrity": "sha512-M9/ELqF6fy8FwmkpnF0S3YKOqMyoWJ4+CS5Efg2ct3oY9daQvd/Pc71FpGZsVsbl3Cpb+IIcjBDUnnyBdQbq4w==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/resolve-from": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/resolve-from/-/resolve-from-4.0.0.tgz",
|
||||
"integrity": "sha512-pb/MYmXstAkysRFx8piNI1tGFNQIFA3vkE3Gq4EuA1dF6gHp/+vgZqsCGJapvy8N3Q+4o7FwvquPJcnZ7RYy4g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=4"
|
||||
}
|
||||
},
|
||||
"node_modules/rollup": {
|
||||
"version": "4.60.2",
|
||||
"resolved": "https://registry.npmjs.org/rollup/-/rollup-4.60.2.tgz",
|
||||
@@ -3832,32 +3602,6 @@
|
||||
"integrity": "sha512-HTEHMNieakEnoe33shBYcZ7NX83ACUjCu8c40iOGEZsngj9zRnkqS9j1pqQPXwobB0ZcVTk27REb7COQ0UR59w==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/strip-json-comments": {
|
||||
"version": "3.1.1",
|
||||
"resolved": "https://registry.npmjs.org/strip-json-comments/-/strip-json-comments-3.1.1.tgz",
|
||||
"integrity": "sha512-6fPc+R4ihwqP6N/aIv2f1gMH8lOVtWQHoqC4yK6oSDVVocumAsfCqjkXnqiYMhmMwS/mEHLp7Vehlt3ql6lEig==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
},
|
||||
"funding": {
|
||||
"url": "https://github.com/sponsors/sindresorhus"
|
||||
}
|
||||
},
|
||||
"node_modules/supports-color": {
|
||||
"version": "7.2.0",
|
||||
"resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz",
|
||||
"integrity": "sha512-qpCAvRl9stuOHveKsn7HncJRvv501qIacKzQlO/+Lwxc9+0q2wLyv4Dfvt80/DPn2pqOBsJdDiogXGR9+OvwRw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"has-flag": "^4.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
}
|
||||
},
|
||||
"node_modules/tinyglobby": {
|
||||
"version": "0.2.16",
|
||||
"resolved": "https://registry.npmjs.org/tinyglobby/-/tinyglobby-0.2.16.tgz",
|
||||
|
||||
@@ -9,7 +9,8 @@
|
||||
"lint": "eslint .",
|
||||
"preview": "vite preview",
|
||||
"test:graph-store": "node --test tests/graphStore.multi-edge.test.mjs",
|
||||
"test:graph-workspace": "node --import tsx --test tests/graphSceneState.display.test.ts"
|
||||
"test:graph-workspace": "node --import tsx --test tests/graphSceneState.display.test.ts",
|
||||
"test:plugin-registry": "node --import tsx --test tests/pluginRegistry.temporal.test.mjs"
|
||||
},
|
||||
"dependencies": {
|
||||
"@monaco-editor/react": "^4.7.0",
|
||||
@@ -41,7 +42,7 @@
|
||||
"@types/react-dom": "^19.2.3",
|
||||
"@vitejs/plugin-react": "^4.3.0",
|
||||
"babel-plugin-react-compiler": "^1.0.0",
|
||||
"eslint": "^9.39.4",
|
||||
"eslint": "^10.8.0",
|
||||
"eslint-plugin-react-hooks": "^7.0.1",
|
||||
"eslint-plugin-react-refresh": "^0.5.2",
|
||||
"globals": "^17.4.0",
|
||||
|
||||
+51
-38
@@ -1,4 +1,4 @@
|
||||
import { lazy, Suspense, useEffect, useState, type ReactNode } from 'react';
|
||||
import { lazy, Suspense, useEffect, useState, type ReactNode } from 'react';
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query';
|
||||
import {
|
||||
ArrowRight,
|
||||
@@ -16,6 +16,7 @@ import {
|
||||
ShieldCheck,
|
||||
type LucideIcon,
|
||||
} from 'lucide-react';
|
||||
import { ErrorBoundary } from './ErrorBoundary';
|
||||
|
||||
const DecisionWorkspace = lazy(() => import('./workspaces/DecisionWorkspace/DecisionWorkspace').then((module) => ({ default: module.DecisionWorkspace })));
|
||||
const DiffMergeWorkspace = lazy(() => import('./workspaces/DiffMergeWorkspace/DiffMergeWorkspace').then((module) => ({ default: module.DiffMergeWorkspace })));
|
||||
@@ -1800,14 +1801,16 @@ export default function App() {
|
||||
</>
|
||||
}
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{exploreView === 'graph' ? (
|
||||
<GraphWorkspace
|
||||
externalFocusNodeId={graphFocusRequest?.nodeId}
|
||||
externalFocusToken={graphFocusRequest?.token}
|
||||
/>
|
||||
) : <VocabularyWorkspace />}
|
||||
</Suspense>
|
||||
<ErrorBoundary key={`explore-${exploreView}`}>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{exploreView === 'graph' ? (
|
||||
<GraphWorkspace
|
||||
externalFocusNodeId={graphFocusRequest?.nodeId}
|
||||
externalFocusToken={graphFocusRequest?.token}
|
||||
/>
|
||||
) : <VocabularyWorkspace />}
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
}
|
||||
@@ -1829,9 +1832,11 @@ export default function App() {
|
||||
</>
|
||||
}
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{analyzeView === 'reasoning' ? <ReasoningWorkspace /> : <SparqlWorkspace />}
|
||||
</Suspense>
|
||||
<ErrorBoundary key={`analyze-${analyzeView}`}>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{analyzeView === 'reasoning' ? <ReasoningWorkspace /> : <SparqlWorkspace />}
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
}
|
||||
@@ -1843,9 +1848,11 @@ export default function App() {
|
||||
subtitle="Inspect decision chains, causal context, and precedent matches."
|
||||
kicker="Decision Intelligence"
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
<DecisionWorkspace />
|
||||
</Suspense>
|
||||
<ErrorBoundary key="decisions">
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
<DecisionWorkspace />
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
}
|
||||
@@ -1873,12 +1880,14 @@ export default function App() {
|
||||
</>
|
||||
}
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{enrichView === 'import' ? <ImportExportWorkspace /> :
|
||||
enrichView === 'merge' ? <DiffMergeWorkspace /> :
|
||||
enrichView === 'resolve' ? <EntityResolutionTab /> :
|
||||
<RegistryTab />}
|
||||
</Suspense>
|
||||
<ErrorBoundary key={`enrich-${enrichView}`}>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{enrichView === 'import' ? <ImportExportWorkspace /> :
|
||||
enrichView === 'merge' ? <DiffMergeWorkspace /> :
|
||||
enrichView === 'resolve' ? <EntityResolutionTab /> :
|
||||
<RegistryTab />}
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
}
|
||||
@@ -1891,15 +1900,17 @@ export default function App() {
|
||||
kicker="Schema Governance"
|
||||
compact
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
<OntologyWorkspace
|
||||
onJumpToGraphNode={(nodeId: string) => {
|
||||
setGraphFocusRequest({ nodeId, token: Date.now() });
|
||||
setActiveWorkspace('explore');
|
||||
setExploreView('graph');
|
||||
}}
|
||||
/>
|
||||
</Suspense>
|
||||
<ErrorBoundary key="ontology-hub">
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
<OntologyWorkspace
|
||||
onJumpToGraphNode={(nodeId: string) => {
|
||||
setGraphFocusRequest({ nodeId, token: Date.now() });
|
||||
setActiveWorkspace('explore');
|
||||
setExploreView('graph');
|
||||
}}
|
||||
/>
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
}
|
||||
@@ -1923,14 +1934,16 @@ export default function App() {
|
||||
</>
|
||||
}
|
||||
>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{manageView === 'lineage' ? <LineageDiagram /> :
|
||||
manageView === 'kg-overview' ? <KGOverviewTab /> :
|
||||
<OntologySummaryTab onOpenVocabularyBrowser={() => {
|
||||
setActiveWorkspace('explore');
|
||||
setExploreView('vocabulary');
|
||||
}} />}
|
||||
</Suspense>
|
||||
<ErrorBoundary key={`manage-${manageView}`}>
|
||||
<Suspense fallback={<WorkspaceFallback />}>
|
||||
{manageView === 'lineage' ? <LineageDiagram /> :
|
||||
manageView === 'kg-overview' ? <KGOverviewTab /> :
|
||||
<OntologySummaryTab onOpenVocabularyBrowser={() => {
|
||||
setActiveWorkspace('explore');
|
||||
setExploreView('vocabulary');
|
||||
}} />}
|
||||
</Suspense>
|
||||
</ErrorBoundary>
|
||||
</WorkspaceShell>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
import { Component, type ErrorInfo, type ReactNode } from 'react';
|
||||
import { AlertCircle } from 'lucide-react';
|
||||
|
||||
interface ErrorBoundaryProps {
|
||||
children: ReactNode;
|
||||
}
|
||||
|
||||
interface ErrorBoundaryState {
|
||||
hasError: boolean;
|
||||
error: Error | null;
|
||||
retryCount: number;
|
||||
}
|
||||
|
||||
const RETRY_SETTLE_MS = 5000;
|
||||
|
||||
export class ErrorBoundary extends Component<ErrorBoundaryProps, ErrorBoundaryState> {
|
||||
private settleTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
|
||||
constructor(props: ErrorBoundaryProps) {
|
||||
super(props);
|
||||
this.state = { hasError: false, error: null, retryCount: 0 };
|
||||
}
|
||||
|
||||
static getDerivedStateFromError(error: Error): Partial<ErrorBoundaryState> {
|
||||
return { hasError: true, error };
|
||||
}
|
||||
|
||||
componentDidCatch(error: Error, errorInfo: ErrorInfo) {
|
||||
console.error("ErrorBoundary caught an error:", error, errorInfo);
|
||||
this.clearSettleTimer();
|
||||
}
|
||||
|
||||
componentWillUnmount() {
|
||||
this.clearSettleTimer();
|
||||
}
|
||||
|
||||
private clearSettleTimer() {
|
||||
if (this.settleTimer !== null) {
|
||||
clearTimeout(this.settleTimer);
|
||||
this.settleTimer = null;
|
||||
}
|
||||
}
|
||||
|
||||
resetErrorBoundary = () => {
|
||||
this.clearSettleTimer();
|
||||
this.setState((prev) => ({
|
||||
hasError: false,
|
||||
error: null,
|
||||
retryCount: prev.retryCount + 1
|
||||
}));
|
||||
// Only clear the retry count once the workspace has stayed error-free for a
|
||||
// sustained period, rather than on the next committed render (which can fire
|
||||
// while Suspense is still showing its fallback) or immediately on retry
|
||||
// (which would allow an unbounded number of clicks on a deterministic crash).
|
||||
this.settleTimer = setTimeout(() => {
|
||||
this.settleTimer = null;
|
||||
this.setState({ retryCount: 0 });
|
||||
}, RETRY_SETTLE_MS);
|
||||
};
|
||||
|
||||
render() {
|
||||
if (this.state.hasError) {
|
||||
const maxRetriesReached = this.state.retryCount >= 3;
|
||||
|
||||
return (
|
||||
<div
|
||||
className="workspace-loading"
|
||||
style={{
|
||||
flexDirection: 'column',
|
||||
gap: 12,
|
||||
color: 'var(--ws-red)'
|
||||
}}
|
||||
>
|
||||
<AlertCircle size={32} style={{ marginBottom: 4, opacity: 0.8 }} />
|
||||
<div style={{ fontWeight: 500, fontSize: '15px' }}>
|
||||
Something went wrong in this view.
|
||||
</div>
|
||||
<div style={{ fontSize: '13px', opacity: 0.7, maxWidth: 450, textAlign: 'center', marginBottom: 8, lineHeight: 1.5 }}>
|
||||
{maxRetriesReached
|
||||
? "This view continues to encounter a critical error. Please switch to another workspace or reload the page to restore functionality."
|
||||
: "An unexpected problem occurred while rendering this workspace. Your data is safe, but this view cannot be displayed."}
|
||||
</div>
|
||||
{!maxRetriesReached ? (
|
||||
<button
|
||||
className="ws-btn ws-btn--ghost"
|
||||
style={{
|
||||
borderColor: 'var(--ws-red-soft)',
|
||||
color: 'var(--ws-red)'
|
||||
}}
|
||||
onClick={this.resetErrorBoundary}
|
||||
>
|
||||
Try Again
|
||||
</button>
|
||||
) : (
|
||||
<button
|
||||
className="ws-btn ws-btn--ghost"
|
||||
style={{
|
||||
borderColor: 'var(--ws-border)',
|
||||
color: 'var(--ws-text)'
|
||||
}}
|
||||
onClick={() => window.location.reload()}
|
||||
>
|
||||
Reload Application
|
||||
</button>
|
||||
)}
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return this.props.children;
|
||||
}
|
||||
}
|
||||
@@ -92,6 +92,7 @@ export function DecisionWorkspace() {
|
||||
const [chainLoading, setChainLoading] = useState(false);
|
||||
const [listLoading, setListLoading] = useState(true);
|
||||
const [filter, setFilter] = useState("");
|
||||
const [error, setError] = useState("");
|
||||
|
||||
// Tracks the active chain request so stale responses from rapid selections are ignored.
|
||||
const chainCtrlRef = useRef<AbortController | null>(null);
|
||||
@@ -99,14 +100,24 @@ export function DecisionWorkspace() {
|
||||
useEffect(() => {
|
||||
const ctrl = new AbortController();
|
||||
setListLoading(true);
|
||||
setError("");
|
||||
fetch("/api/decisions", { signal: ctrl.signal })
|
||||
.then((r) => r.ok ? r.json() : Promise.reject(r.status))
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error(`HTTP ${r.status}`);
|
||||
const data = await r.json();
|
||||
if (r.status === 207) setError(data.message || "Warning: Partial success loading decisions.");
|
||||
return data;
|
||||
})
|
||||
.then((data) => {
|
||||
if (ctrl.signal.aborted) return;
|
||||
setDecisions(data);
|
||||
if (data.length > 0) void loadChain(data[0]);
|
||||
})
|
||||
.catch((e) => { if (e?.name !== "AbortError") console.error(e); })
|
||||
.catch((e) => {
|
||||
if (e?.name !== "AbortError") {
|
||||
setError(e instanceof Error ? e.message : "Failed to load decisions.");
|
||||
}
|
||||
})
|
||||
.finally(() => {
|
||||
if (!ctrl.signal.aborted) setListLoading(false);
|
||||
});
|
||||
@@ -125,13 +136,19 @@ export function DecisionWorkspace() {
|
||||
setSelected(d);
|
||||
setChainLoading(true);
|
||||
setChain([]);
|
||||
setError("");
|
||||
try {
|
||||
const res = await fetch(`/api/decisions/${encodeURIComponent(d.decision_id)}/chain`, { signal: ctrl.signal });
|
||||
if (!res.ok) throw new Error(`${res.status}`);
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const data = await res.json();
|
||||
if (res.status === 207 && !ctrl.signal.aborted) {
|
||||
setError(data.message || "Warning: Partial success loading chain.");
|
||||
}
|
||||
if (!ctrl.signal.aborted) setChain(data.chain || []);
|
||||
} catch (e) {
|
||||
if (e instanceof Error && e.name !== "AbortError") console.error(e);
|
||||
if (e instanceof Error && e.name !== "AbortError") {
|
||||
setError(e.message);
|
||||
}
|
||||
} finally {
|
||||
if (!ctrl.signal.aborted) setChainLoading(false);
|
||||
}
|
||||
@@ -213,6 +230,12 @@ export function DecisionWorkspace() {
|
||||
<div style={{ flex: 1, display: "flex", flexDirection: "column", overflow: "hidden", position: "relative" }}>
|
||||
<div style={{ position: "absolute", inset: 0, background: "radial-gradient(ellipse 60% 40% at 70% 20%, rgba(74,163,255,0.04), transparent 55%)", pointerEvents: "none" }} />
|
||||
|
||||
{error ? (
|
||||
<div style={{ padding: 12, borderRadius: 14, color: "#ffb4c2", background: "rgba(255,157,175,0.1)", border: "1px solid rgba(255,157,175,0.18)", margin: "16px 16px 0 16px", zIndex: 2, position: "relative" }}>
|
||||
{error}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{selected ? (
|
||||
<div className="ws-scroll ws-padded ws-animate-in" style={{ position: "relative", zIndex: 1 }}>
|
||||
{/* Decision header */}
|
||||
|
||||
@@ -192,6 +192,7 @@ export function EntityResolutionTab() {
|
||||
}, [threshold]);
|
||||
|
||||
const handleMerge = useCallback(async (primaryId: string, duplicateId: string) => {
|
||||
setScanError("");
|
||||
try {
|
||||
const res = await fetch("/api/enrich/merge", {
|
||||
method: "POST",
|
||||
@@ -200,6 +201,9 @@ export function EntityResolutionTab() {
|
||||
});
|
||||
if (!res.ok) throw new Error(`Merge failed (${res.status})`);
|
||||
const data = await res.json();
|
||||
if (res.status === 207) {
|
||||
setScanError(data.message || "Warning: Partial merge.");
|
||||
}
|
||||
logEvent("merge", `Merged ${duplicateId} → ${primaryId} · ${data.edges_updated ?? 0} edges redirected`, {
|
||||
primary: primaryId,
|
||||
duplicate: duplicateId,
|
||||
@@ -207,7 +211,7 @@ export function EntityResolutionTab() {
|
||||
});
|
||||
setPairs((prev) => prev.filter((p) => !(p.a.id === primaryId && p.b.id === duplicateId)));
|
||||
} catch (err) {
|
||||
console.error("[EntityResolution] merge failed", err);
|
||||
setScanError(err instanceof Error ? err.message : "Merge failed");
|
||||
}
|
||||
}, []);
|
||||
|
||||
|
||||
@@ -38,6 +38,7 @@ import {
|
||||
type GraphPluginPanelDescriptor,
|
||||
type GraphPluginToolbarItem,
|
||||
} from "./plugins";
|
||||
import { explorationEffectsShouldLoad, neighborhoodPanelShouldLoad, temporalOverlayShouldLoad } from "./pluginRegistryPredicates";
|
||||
import type { LinkPrediction, PathResponse } from "./GraphInspectorPanel";
|
||||
import type { GraphSceneHandle, GraphSceneRuntime } from "./scene";
|
||||
import type {
|
||||
@@ -126,7 +127,7 @@ type LazyPluginRegistryEntry = {
|
||||
load: () => Promise<GraphPlugin>;
|
||||
shouldLoad: (context: {
|
||||
panelState: Record<string, boolean>;
|
||||
temporalState: GraphTemporalState | null;
|
||||
temporalState?: GraphTemporalState | null;
|
||||
}) => boolean;
|
||||
};
|
||||
|
||||
@@ -1119,6 +1120,18 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
const [activeNodeCount, setActiveNodeCount] = useState<number | null>(null);
|
||||
const [temporalBounds, setTemporalBounds] = useState<TemporalBounds | null>(null);
|
||||
const [scrubberTime, setScrubberTime] = useState<Date | null>(null);
|
||||
// Deduplicates setScrubberTime calls by millisecond value so that React 18
|
||||
// concurrent-mode re-renders with a new Date object for the same timestamp
|
||||
// do not churn temporalState and retrigger the diagnostics effect (issue #830).
|
||||
const lastScrubberMsRef = useRef<number | null>(null);
|
||||
const onTimeChange = useCallback((time: Date) => {
|
||||
const ms = time.getTime();
|
||||
if (ms === lastScrubberMsRef.current) {
|
||||
return;
|
||||
}
|
||||
lastScrubberMsRef.current = ms;
|
||||
setScrubberTime(time);
|
||||
}, []);
|
||||
const [loadingProgress, setLoadingProgress] = useState<GraphLoadProgress | null>(null);
|
||||
const [pluginPanelState, setPluginPanelState] = useState<Record<string, boolean>>({
|
||||
"effects-panel": false,
|
||||
@@ -1129,6 +1142,9 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
const [pluginRuntimeVersion, setPluginRuntimeVersion] = useState(0);
|
||||
const [effectsState, setEffectsState] = useState<GraphEffectsState>(DEFAULT_EFFECTS_STATE);
|
||||
const [graphDiagnosticsState, setGraphDiagnosticsState] = useState<GraphRuntimeDiagnosticsSnapshot | null>(null);
|
||||
// Tracks the last accepted diagnostics outside React's state cycle, allowing
|
||||
// handleDiagnosticsChange to compare synchronously before calling setState.
|
||||
const lastDiagnosticsRef = useRef<GraphRuntimeDiagnosticsSnapshot | null>(null);
|
||||
const [graphAnalyticsState, setGraphAnalyticsState] = useState<GraphAnalyticsSnapshot | null>(null);
|
||||
const [loadedPlugins, setLoadedPlugins] = useState<Record<string, GraphPlugin>>({});
|
||||
|
||||
@@ -2056,7 +2072,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
title: "Open exploration effects controls",
|
||||
order: 18,
|
||||
load: loadExplorationEffectsPlugin,
|
||||
shouldLoad: ({ panelState }) => Boolean(panelState["effects-panel"]),
|
||||
shouldLoad: explorationEffectsShouldLoad,
|
||||
},
|
||||
{
|
||||
id: "neighborhood-panel",
|
||||
@@ -2065,7 +2081,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
title: "Toggle neighborhood panel",
|
||||
order: 30,
|
||||
load: loadNeighborhoodPanelPlugin,
|
||||
shouldLoad: ({ panelState }) => Boolean(panelState["neighborhood-panel"]),
|
||||
shouldLoad: neighborhoodPanelShouldLoad,
|
||||
},
|
||||
{
|
||||
id: "temporal-overlay",
|
||||
@@ -2074,7 +2090,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
title: "Toggle temporal context panel",
|
||||
order: 40,
|
||||
load: loadTemporalOverlayPlugin,
|
||||
shouldLoad: ({ panelState, temporalState }) => Boolean(panelState["temporal-panel"] || temporalState?.currentTime),
|
||||
shouldLoad: temporalOverlayShouldLoad,
|
||||
},
|
||||
],
|
||||
[],
|
||||
@@ -2092,7 +2108,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
return;
|
||||
}
|
||||
|
||||
if (!entry.shouldLoad({ panelState: pluginPanelState, temporalState })) {
|
||||
if (!entry.shouldLoad({ panelState: pluginPanelState })) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -2111,7 +2127,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, [loadedPlugins, pluginPanelState, pluginRegistry, temporalState]);
|
||||
}, [loadedPlugins, pluginPanelState, pluginRegistry]);
|
||||
|
||||
const setEffectToggle = useCallback((effect: GraphEffectToggle, enabled: boolean | ((current: boolean) => boolean)) => {
|
||||
setEffectsState((current) => {
|
||||
@@ -2274,6 +2290,55 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
if (!GRAPH_THEME.effects.diagnostics.enabledInDev) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Compare against the last accepted snapshot synchronously before calling
|
||||
// setState. buildEffectAvailability always returns a new object, so an
|
||||
// unconditional setGraphDiagnosticsState on every call created a
|
||||
// render → diagnostics effect → setState → render cycle that exceeded
|
||||
// React's max update depth in dev mode (issue #830).
|
||||
const prev = lastDiagnosticsRef.current;
|
||||
if (prev !== null) {
|
||||
const EFFECT_KEYS = [
|
||||
"pathPulse", "pathFlow", "lens", "temporalEmphasis", "semanticRegions",
|
||||
"contours", "pathfinding", "communities", "centrality", "legend", "diagnostics",
|
||||
] as const;
|
||||
const prevEA = prev.effectAvailability;
|
||||
const nextEA = diagnostics.effectAvailability;
|
||||
const availabilityChanged = EFFECT_KEYS.some((key) => {
|
||||
const p = prevEA[key];
|
||||
const n = nextEA[key];
|
||||
return (
|
||||
p.enabled !== n.enabled ||
|
||||
p.available !== n.available ||
|
||||
p.reason !== n.reason ||
|
||||
p.detail !== n.detail ||
|
||||
p.visibleSegments !== n.visibleSegments ||
|
||||
p.segmentCap !== n.segmentCap
|
||||
);
|
||||
});
|
||||
|
||||
const edgeClassesChanged =
|
||||
prev.edgeClasses?.updatedAt !== diagnostics.edgeClasses?.updatedAt;
|
||||
|
||||
const structureLayerChanged =
|
||||
prev.structureLayer?.cacheKey !== diagnostics.structureLayer?.cacheKey ||
|
||||
prev.structureLayer?.lastDrawAt !== diagnostics.structureLayer?.lastDrawAt ||
|
||||
prev.structureLayer?.enabled !== diagnostics.structureLayer?.enabled ||
|
||||
prev.structureLayer?.disabledReason !== diagnostics.structureLayer?.disabledReason ||
|
||||
prev.structureLayer?.curveCount !== diagnostics.structureLayer?.curveCount ||
|
||||
prev.structureLayer?.bridgeCurveCount !== diagnostics.structureLayer?.bridgeCurveCount ||
|
||||
prev.structureLayer?.backboneCurveCount !== diagnostics.structureLayer?.backboneCurveCount;
|
||||
|
||||
// distanceVisual is compared by reference: GraphCanvas passes the same
|
||||
// object when distances haven't changed.
|
||||
const distanceVisualChanged = prev.distanceVisual !== diagnostics.distanceVisual;
|
||||
|
||||
if (!availabilityChanged && !edgeClassesChanged && !structureLayerChanged && !distanceVisualChanged) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
lastDiagnosticsRef.current = diagnostics;
|
||||
setGraphDiagnosticsState(diagnostics);
|
||||
}, []);
|
||||
|
||||
@@ -2351,16 +2416,17 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
showPluginDock: openDockPanels.length > 0,
|
||||
};
|
||||
|
||||
useEffect(() => {
|
||||
if (!openDockPanels.length) {
|
||||
setActiveDockPanelId(null);
|
||||
return;
|
||||
}
|
||||
const openDockPanelIdsString = openDockPanels.map((p) => p.id).join("|");
|
||||
const [prevOpenDockPanelIdsString, setPrevOpenDockPanelIdsString] = useState(openDockPanelIdsString);
|
||||
|
||||
if (!activeDockPanelId || !openDockPanels.some((panel) => panel.id === activeDockPanelId)) {
|
||||
if (openDockPanelIdsString !== prevOpenDockPanelIdsString) {
|
||||
setPrevOpenDockPanelIdsString(openDockPanelIdsString);
|
||||
if (!openDockPanels.length) {
|
||||
if (activeDockPanelId !== null) setActiveDockPanelId(null);
|
||||
} else if (!activeDockPanelId || !openDockPanels.some((panel) => panel.id === activeDockPanelId)) {
|
||||
setActiveDockPanelId(openDockPanels[0].id);
|
||||
}
|
||||
}, [activeDockPanelId, openDockPanels]);
|
||||
}
|
||||
|
||||
const viewModeItems = useMemo<GraphToolbarItem[]>(() => {
|
||||
if (!hasGraphContent) {
|
||||
@@ -2921,7 +2987,7 @@ export function GraphWorkspace({ externalFocusNodeId, externalFocusToken }: Grap
|
||||
<div className="explore-scene-footer">
|
||||
<Suspense fallback={<div style={timelineFallbackStyle}>Loading timeline…</div>}>
|
||||
<LazyTimelinePanel
|
||||
onTimeChange={setScrubberTime}
|
||||
onTimeChange={onTimeChange}
|
||||
minDate={temporalBounds?.min ?? undefined}
|
||||
maxDate={temporalBounds?.max ?? undefined}
|
||||
/>
|
||||
|
||||
@@ -325,6 +325,17 @@ export function GraphWorkspaceShell() {
|
||||
const [activeNodeCount, setActiveNodeCount] = useState<number | null>(null);
|
||||
const [temporalBounds, setTemporalBounds] = useState<TemporalBounds | null>(null);
|
||||
const [scrubberTime, setScrubberTime] = useState<Date | null>(null);
|
||||
// Deduplicates setScrubberTime calls by millisecond value — same fix as
|
||||
// GraphWorkspace.tsx (issue #830).
|
||||
const lastScrubberMsRef = useRef<number | null>(null);
|
||||
const onTimeChange = useCallback((time: Date) => {
|
||||
const ms = time.getTime();
|
||||
if (ms === lastScrubberMsRef.current) {
|
||||
return;
|
||||
}
|
||||
lastScrubberMsRef.current = ms;
|
||||
setScrubberTime(time);
|
||||
}, []);
|
||||
const [loadingProgress, setLoadingProgress] = useState<GraphLoadProgress | null>(null);
|
||||
const [isGraphStageReady, setIsGraphStageReady] = useState(false);
|
||||
const [layoutStatus, setLayoutStatus] = useState<GraphLayoutStatus>({
|
||||
@@ -369,7 +380,9 @@ export function GraphWorkspaceShell() {
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
const [prevFetchedAt, setPrevFetchedAt] = useState(snapshot?.fetchedAt);
|
||||
if (snapshot?.fetchedAt !== prevFetchedAt) {
|
||||
setPrevFetchedAt(snapshot?.fetchedAt);
|
||||
if (snapshot) {
|
||||
setIsGraphStageReady(false);
|
||||
setActiveNodeCount(null);
|
||||
@@ -383,7 +396,7 @@ export function GraphWorkspaceShell() {
|
||||
stableSamples: 0,
|
||||
});
|
||||
}
|
||||
}, [snapshot?.fetchedAt]);
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
let cancelled = false;
|
||||
@@ -630,7 +643,7 @@ export function GraphWorkspaceShell() {
|
||||
|
||||
<Suspense fallback={<TimelineFallback min={temporalBounds?.min ?? null} max={temporalBounds?.max ?? null} />}>
|
||||
<TimelinePanel
|
||||
onTimeChange={setScrubberTime}
|
||||
onTimeChange={onTimeChange}
|
||||
minDate={temporalBounds?.min ?? undefined}
|
||||
maxDate={temporalBounds?.max ?? undefined}
|
||||
/>
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
/**
|
||||
* shouldLoad predicates for the GraphWorkspace lazy plugin registry.
|
||||
*
|
||||
* Extracted into a pure module so the predicates can be unit-tested without
|
||||
* importing the full GraphWorkspace React component. Each predicate gates
|
||||
* whether a plugin's module is lazily imported; none reference temporalState
|
||||
* so temporal scrubber updates never retrigger plugin loading (issue #830).
|
||||
*/
|
||||
|
||||
export type PluginShouldLoadContext = {
|
||||
panelState: Record<string, boolean>;
|
||||
};
|
||||
|
||||
export function explorationEffectsShouldLoad({ panelState }: PluginShouldLoadContext): boolean {
|
||||
return Boolean(panelState["effects-panel"]);
|
||||
}
|
||||
|
||||
export function neighborhoodPanelShouldLoad({ panelState }: PluginShouldLoadContext): boolean {
|
||||
return Boolean(panelState["neighborhood-panel"]);
|
||||
}
|
||||
|
||||
export function temporalOverlayShouldLoad({ panelState }: PluginShouldLoadContext): boolean {
|
||||
return Boolean(panelState["temporal-panel"]);
|
||||
}
|
||||
@@ -33,6 +33,7 @@ export function LineageDiagram() {
|
||||
const [edges, setEdges] = useState<any[]>([]);
|
||||
const [searchId, setSearchId] = useState("");
|
||||
const [activeId, setActiveId] = useState("");
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const downloadReport = async (format: "json" | "markdown") => {
|
||||
if (!activeId) return;
|
||||
@@ -51,12 +52,17 @@ export function LineageDiagram() {
|
||||
document.body.removeChild(anchor);
|
||||
};
|
||||
|
||||
const [prevActiveId, setPrevActiveId] = useState(activeId);
|
||||
if (activeId !== prevActiveId) {
|
||||
setPrevActiveId(activeId);
|
||||
setError("");
|
||||
setNodes([]);
|
||||
setEdges([]);
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
if (!activeId) {
|
||||
setNodes([]);
|
||||
setEdges([]);
|
||||
return;
|
||||
}
|
||||
let ignore = false;
|
||||
if (!activeId) return;
|
||||
|
||||
const xLanes = [
|
||||
{ id: "group_agent", type: "group", position: { x: 50, y: 50 }, style: { width: 800, height: 120 } },
|
||||
@@ -65,22 +71,24 @@ export function LineageDiagram() {
|
||||
];
|
||||
|
||||
const fetchLineage = async () => {
|
||||
setError("");
|
||||
try {
|
||||
const res = await fetch("/api/provenance?node_id=" + encodeURIComponent(activeId));
|
||||
|
||||
if (!res.ok) {
|
||||
const text = await res.text();
|
||||
console.error(`HTTP ${res.status}: API Route missing or failed.`, text.substring(0, 100));
|
||||
return;
|
||||
throw new Error(`HTTP ${res.status}: API Route missing or failed. ${text.substring(0, 100)}`);
|
||||
}
|
||||
|
||||
const contentType = res.headers.get("content-type");
|
||||
if (!contentType || !contentType.includes("application/json")) {
|
||||
console.error("Backend returned non-JSON response (likely an HTML fallback). Check FastAPI routing.");
|
||||
return;
|
||||
throw new Error("Backend returned non-JSON response (likely an HTML fallback).");
|
||||
}
|
||||
|
||||
const data = await res.json();
|
||||
if (res.status === 207) {
|
||||
setError(data.message || "Warning: Partial success loading lineage.");
|
||||
}
|
||||
|
||||
const counters: Record<string, number> = { "group_agent": 0, "group_activity": 0, "group_entity": 0 };
|
||||
|
||||
@@ -89,7 +97,7 @@ export function LineageDiagram() {
|
||||
counters[n.parent_id] = c + 1;
|
||||
return {
|
||||
id: n.id,
|
||||
data: { label: n.label + "\\n(" + n.prov_type + ")" },
|
||||
data: { label: n.label + "\n(" + n.prov_type + ")" },
|
||||
position: { x: 50 + c * 180, y: 30 },
|
||||
parentId: n.parent_id,
|
||||
extent: "parent",
|
||||
@@ -106,13 +114,16 @@ export function LineageDiagram() {
|
||||
style: { stroke: "#58a6ff" }
|
||||
}));
|
||||
|
||||
setNodes([...xLanes, ...mappedNodes]);
|
||||
setEdges(mappedEdges);
|
||||
if (!ignore) {
|
||||
setNodes([...xLanes, ...mappedNodes]);
|
||||
setEdges(mappedEdges);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
setError(err instanceof Error ? err.message : "Failed to load lineage.");
|
||||
}
|
||||
};
|
||||
fetchLineage();
|
||||
void fetchLineage();
|
||||
return () => { ignore = true; };
|
||||
}, [activeId]);
|
||||
|
||||
return (
|
||||
@@ -143,6 +154,12 @@ export function LineageDiagram() {
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{error ? (
|
||||
<div style={{ position: "absolute", top: 60, left: 14, right: 14, zIndex: 10, padding: 12, borderRadius: 14, color: "#ffb4c2", background: "rgba(255,157,175,0.1)", border: "1px solid rgba(255,157,175,0.18)" }}>
|
||||
{error}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{activeId ? (
|
||||
<ReactFlow nodes={nodes} edges={edges} fitView>
|
||||
<Background color="rgba(74,163,255,0.08)" gap={24} />
|
||||
|
||||
@@ -81,15 +81,79 @@ export function KGOverviewTab() {
|
||||
fetch("/api/graph/nodes?limit=500"),
|
||||
]);
|
||||
|
||||
if (statsRes.ok) {
|
||||
const statsData: KGStats = await statsRes.json();
|
||||
setStats(statsData);
|
||||
if (!statsRes.ok) throw new Error(`Stats fetch failed (${statsRes.status})`);
|
||||
if (!nodesRes.ok) throw new Error(`Nodes fetch failed (${nodesRes.status})`);
|
||||
|
||||
const statsData: KGStats = await statsRes.json();
|
||||
setStats(statsData);
|
||||
if (statsRes.status === 207) {
|
||||
setError((statsData as any).message || "Warning: Partial success loading stats.");
|
||||
}
|
||||
|
||||
if (nodesRes.ok) {
|
||||
const nodesData: NodeListResponse = await nodesRes.json();
|
||||
const nodes = nodesData.nodes ?? [];
|
||||
setNodeTypeMap(buildTypeMap(nodes, "type"));
|
||||
if (nodesRes.status === 207) {
|
||||
const nodesMessage = (nodesData as any).message || "Warning: Partial success loading nodes.";
|
||||
setError((prev) => (prev ? `${prev} ${nodesMessage}` : nodesMessage));
|
||||
}
|
||||
|
||||
// Simulate neighbor counts via edges fetch for top-N
|
||||
const edgesRes = await fetch("/api/graph/edges?limit=2000");
|
||||
if (edgesRes.ok) {
|
||||
const edgesData = await edgesRes.json();
|
||||
const edges: { source: string; target: string }[] = edgesData.edges ?? [];
|
||||
const degreeMap: Record<string, number> = {};
|
||||
for (const edge of edges) {
|
||||
degreeMap[edge.source] = (degreeMap[edge.source] ?? 0) + 1;
|
||||
degreeMap[edge.target] = (degreeMap[edge.target] ?? 0) + 1;
|
||||
}
|
||||
const sorted = nodes
|
||||
.map((n) => ({ node: n, neighborCount: degreeMap[n.id] ?? 0 }))
|
||||
.sort((a, b) => b.neighborCount - a.neighborCount)
|
||||
.slice(0, 10);
|
||||
setTopNodes(sorted);
|
||||
}
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to load graph overview. Ensure the server is running.");
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
let ignore = false;
|
||||
async function fetchInitial() {
|
||||
if (!ignore) {
|
||||
setLoading(true);
|
||||
setError("");
|
||||
}
|
||||
try {
|
||||
const [statsRes, nodesRes] = await Promise.all([
|
||||
fetch("/api/graph/stats"),
|
||||
fetch("/api/graph/nodes?limit=500"),
|
||||
]);
|
||||
|
||||
if (!statsRes.ok) throw new Error(`Stats fetch failed (${statsRes.status})`);
|
||||
if (!nodesRes.ok) throw new Error(`Nodes fetch failed (${nodesRes.status})`);
|
||||
|
||||
const statsData: KGStats = await statsRes.json();
|
||||
if (!ignore) {
|
||||
setStats(statsData);
|
||||
if (statsRes.status === 207) {
|
||||
setError((statsData as { message?: string }).message || "Warning: Partial success loading stats.");
|
||||
}
|
||||
}
|
||||
|
||||
const nodesData: NodeListResponse = await nodesRes.json();
|
||||
const nodes = nodesData.nodes ?? [];
|
||||
setNodeTypeMap(buildTypeMap(nodes, "type"));
|
||||
if (!ignore) {
|
||||
setNodeTypeMap(buildTypeMap(nodes, "type"));
|
||||
if (nodesRes.status === 207) {
|
||||
const nodesMessage = (nodesData as { message?: string }).message || "Warning: Partial success loading nodes.";
|
||||
setError((prev) => (prev ? `${prev} ${nodesMessage}` : nodesMessage));
|
||||
}
|
||||
}
|
||||
|
||||
// Simulate neighbor counts via edges fetch for top-N
|
||||
const edgesRes = await fetch("/api/graph/edges?limit=2000");
|
||||
@@ -105,20 +169,18 @@ export function KGOverviewTab() {
|
||||
.map((n) => ({ node: n, neighborCount: degreeMap[n.id] ?? 0 }))
|
||||
.sort((a, b) => b.neighborCount - a.neighborCount)
|
||||
.slice(0, 10);
|
||||
setTopNodes(sorted);
|
||||
if (!ignore) setTopNodes(sorted);
|
||||
}
|
||||
} catch (err) {
|
||||
if (!ignore) setError(err instanceof Error ? err.message : "Failed to load graph overview. Ensure the server is running.");
|
||||
} finally {
|
||||
if (!ignore) setLoading(false);
|
||||
}
|
||||
} catch {
|
||||
setError("Failed to load graph overview. Ensure the server is running.");
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
void fetchInitial();
|
||||
return () => { ignore = true; };
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
void fetchOverview();
|
||||
}, [fetchOverview]);
|
||||
|
||||
const nodeTypeEntries = Object.entries(nodeTypeMap).sort((a, b) => b[1] - a[1]);
|
||||
const edgeTypeEntries = stats?.edge_types
|
||||
? Object.entries(stats.edge_types).sort((a, b) => b[1] - a[1])
|
||||
@@ -135,6 +197,12 @@ export function KGOverviewTab() {
|
||||
|
||||
return (
|
||||
<div className="ws-page">
|
||||
{error ? (
|
||||
<div style={{ margin: "16px 22px 0 22px", padding: 12, borderRadius: 14, color: "#ffb4c2", background: "rgba(255,157,175,0.1)", border: "1px solid rgba(255,157,175,0.18)" }}>
|
||||
{error}
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{/* Header */}
|
||||
<div style={{ display: "flex", alignItems: "center", justifyContent: "space-between", padding: "16px 22px", borderBottom: "1px solid var(--ws-border)", flexShrink: 0 }}>
|
||||
<div style={{ display: "flex", alignItems: "center", gap: 10 }}>
|
||||
@@ -152,11 +220,7 @@ export function KGOverviewTab() {
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{error && (
|
||||
<div style={{ margin: "12px 22px", padding: "10px 14px", borderRadius: "var(--ws-radius-sm)", background: "var(--ws-red-soft)", border: "1px solid rgba(255,123,114,0.28)", color: "#fca5a5", fontSize: 13 }}>
|
||||
{error}
|
||||
</div>
|
||||
)}
|
||||
|
||||
|
||||
<div className="ws-scroll" style={{ flex: 1, padding: "18px 22px", display: "flex", flexDirection: "column", gap: 16 }}>
|
||||
{/* Stat cards */}
|
||||
|
||||
@@ -54,20 +54,49 @@ export function AlignmentsTab() {
|
||||
loadAlignments(),
|
||||
]);
|
||||
|
||||
const errors: string[] = [];
|
||||
if (registryResult.status === "fulfilled") {
|
||||
setRegistry(registryResult.value);
|
||||
setSourceOntology((current) => current || registryResult.value[0]?.uri || "");
|
||||
setTargetOntology((current) => current || registryResult.value[1]?.uri || registryResult.value[0]?.uri || "");
|
||||
} else {
|
||||
errors.push(registryResult.reason instanceof Error ? registryResult.reason.message : "Failed to load ontology registry.");
|
||||
}
|
||||
if (alignmentResult.status === "fulfilled") {
|
||||
setAlignments(alignmentResult.value);
|
||||
} else {
|
||||
errors.push(alignmentResult.reason instanceof Error ? alignmentResult.reason.message : "Failed to load alignments.");
|
||||
}
|
||||
|
||||
if (errors.length) setError(errors.join(" "));
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
void reload();
|
||||
}, [reload]);
|
||||
let ignore = false;
|
||||
async function fetchInitial() {
|
||||
const [registryResult, alignmentResult] = await Promise.allSettled([
|
||||
loadOntologyRegistry(),
|
||||
loadAlignments(),
|
||||
]);
|
||||
if (ignore) return;
|
||||
|
||||
const errors: string[] = [];
|
||||
if (registryResult.status === "fulfilled") {
|
||||
setRegistry(registryResult.value);
|
||||
setSourceOntology((current) => current || registryResult.value[0]?.uri || "");
|
||||
setTargetOntology((current) => current || registryResult.value[1]?.uri || registryResult.value[0]?.uri || "");
|
||||
} else {
|
||||
errors.push(registryResult.reason instanceof Error ? registryResult.reason.message : "Failed to load ontology registry.");
|
||||
}
|
||||
if (alignmentResult.status === "fulfilled") {
|
||||
setAlignments(alignmentResult.value);
|
||||
} else {
|
||||
errors.push(alignmentResult.reason instanceof Error ? alignmentResult.reason.message : "Failed to load alignments.");
|
||||
}
|
||||
if (errors.length) setError(errors.join(" "));
|
||||
}
|
||||
void fetchInitial();
|
||||
return () => { ignore = true; };
|
||||
}, []);
|
||||
|
||||
const relationCounts = useMemo(() => {
|
||||
const counts = new Map<string, number>();
|
||||
|
||||
@@ -23,29 +23,40 @@ export function HealthTab({ onFixInEditor }: HealthTabProps) {
|
||||
setRegistry(entries);
|
||||
setSelectedUri((current) => current || entries[0]?.uri || "");
|
||||
})
|
||||
.catch(() => { /* backend unavailable — leave registry empty */ });
|
||||
.catch((err) => {
|
||||
if (cancelled) return;
|
||||
setError(err instanceof Error ? err.message : "Failed to load ontology registry.");
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
const loadHealth = useCallback(async (uri: string) => {
|
||||
if (!uri) return;
|
||||
setLoading(true);
|
||||
setError("");
|
||||
try {
|
||||
setHealth(await loadOntologyHealth(uri));
|
||||
} catch {
|
||||
// Backend unavailable — show "select an ontology" placeholder, not an error
|
||||
setHealth(null);
|
||||
} finally {
|
||||
setLoading(false);
|
||||
const [prevUri, setPrevUri] = useState(selectedUri);
|
||||
if (selectedUri !== prevUri) {
|
||||
setPrevUri(selectedUri);
|
||||
if (selectedUri) {
|
||||
setLoading(true);
|
||||
setError("");
|
||||
}
|
||||
}, []);
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
void loadHealth(selectedUri);
|
||||
}, [selectedUri, loadHealth]);
|
||||
let ignore = false;
|
||||
async function fetchHealth() {
|
||||
if (!selectedUri) return;
|
||||
try {
|
||||
const data = await loadOntologyHealth(selectedUri);
|
||||
if (!ignore) setHealth(data);
|
||||
} catch {
|
||||
if (!ignore) setHealth(null);
|
||||
} finally {
|
||||
if (!ignore) setLoading(false);
|
||||
}
|
||||
}
|
||||
void fetchHealth();
|
||||
return () => { ignore = true; };
|
||||
}, [selectedUri]);
|
||||
|
||||
const exportReport = useCallback(() => {
|
||||
if (!health) return;
|
||||
|
||||
@@ -248,6 +248,18 @@ export function OntologyManager() {
|
||||
const [rightPanel, setRightPanel] = useState<RightPanel>("none");
|
||||
const [actionMsg, setActionMsg] = useState<{ type: "ok" | "err"; text: string } | null>(null);
|
||||
|
||||
const [prevSearchQ, setPrevSearchQ] = useState(searchQ);
|
||||
if (searchQ !== prevSearchQ) {
|
||||
setPrevSearchQ(searchQ);
|
||||
setLoading(true);
|
||||
setActionMsg(null);
|
||||
}
|
||||
|
||||
const flashMsg = useCallback((type: "ok" | "err", text: string) => {
|
||||
setActionMsg({ type, text });
|
||||
setTimeout(() => setActionMsg(null), 3000);
|
||||
}, []);
|
||||
|
||||
const fetchRegistry = useCallback(async () => {
|
||||
setLoading(true);
|
||||
setActionMsg(null);
|
||||
@@ -255,26 +267,42 @@ export function OntologyManager() {
|
||||
const params = new URLSearchParams();
|
||||
if (searchQ) params.set("q", searchQ);
|
||||
const res = await fetch(`/api/ontology/registry?${params}`);
|
||||
if (res.ok) {
|
||||
setEntries(await res.json());
|
||||
} else {
|
||||
setEntries([]);
|
||||
}
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const data = await res.json();
|
||||
setEntries(data);
|
||||
if (res.status === 207) flashMsg("err", data.message || "Warning: Partial success loading registry.");
|
||||
} catch {
|
||||
setEntries([]);
|
||||
flashMsg("err", "Failed to load ontology registry");
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, [searchQ, statusFilter]);
|
||||
}, [searchQ, flashMsg]);
|
||||
|
||||
useEffect(() => {
|
||||
fetchRegistry();
|
||||
}, [fetchRegistry]);
|
||||
|
||||
const flashMsg = (type: "ok" | "err", text: string) => {
|
||||
setActionMsg({ type, text });
|
||||
setTimeout(() => setActionMsg(null), 3000);
|
||||
};
|
||||
let ignore = false;
|
||||
async function fetchInitial() {
|
||||
try {
|
||||
const params = new URLSearchParams();
|
||||
if (searchQ) params.set("q", searchQ);
|
||||
const res = await fetch(`/api/ontology/registry?${params}`);
|
||||
if (!res.ok) throw new Error(`HTTP ${res.status}`);
|
||||
const data = await res.json();
|
||||
if (ignore) return;
|
||||
setEntries(data);
|
||||
if (res.status === 207) flashMsg("err", data.message || "Warning: Partial success loading registry.");
|
||||
} catch {
|
||||
if (!ignore) {
|
||||
setEntries([]);
|
||||
flashMsg("err", "Failed to load ontology registry");
|
||||
}
|
||||
} finally {
|
||||
if (!ignore) setLoading(false);
|
||||
}
|
||||
}
|
||||
void fetchInitial();
|
||||
return () => { ignore = true; };
|
||||
}, [searchQ, flashMsg]);
|
||||
|
||||
const handleToggle = useCallback(async (uri: string) => {
|
||||
try {
|
||||
|
||||
@@ -178,17 +178,33 @@ function DetailPanel({
|
||||
const [loading, setLoading] = useState(true);
|
||||
const [error, setError] = useState("");
|
||||
|
||||
useEffect(() => {
|
||||
const [prevUri, setPrevUri] = useState(uri);
|
||||
if (uri !== prevUri) {
|
||||
setPrevUri(uri);
|
||||
setLoading(true);
|
||||
setError("");
|
||||
setDetail(null);
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
let ignore = false;
|
||||
fetch(`/api/ontology/entity/${encodeURIComponent(uri)}`)
|
||||
.then((r) => {
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error("Not found");
|
||||
return r.json();
|
||||
const data = await r.json();
|
||||
if (r.status === 207) setError(data.message || "Warning: Partial success loading entity.");
|
||||
return data;
|
||||
})
|
||||
.then(setDetail)
|
||||
.catch((e) => setError(e.message))
|
||||
.finally(() => setLoading(false));
|
||||
.then((data) => {
|
||||
if (!ignore) setDetail(data);
|
||||
})
|
||||
.catch((e) => {
|
||||
if (!ignore) setError(e.message);
|
||||
})
|
||||
.finally(() => {
|
||||
if (!ignore) setLoading(false);
|
||||
});
|
||||
return () => { ignore = true; };
|
||||
}, [uri]);
|
||||
|
||||
return (
|
||||
|
||||
@@ -40,19 +40,6 @@ export function ProposalReview({ proposalId }: { proposalId: string }) {
|
||||
const [selectedElement, setSelectedElement] = useState<string | null>(null);
|
||||
const [commentText, setCommentText] = useState("");
|
||||
|
||||
const loadProposal = useCallback(async () => {
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}`);
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
setProposal(data);
|
||||
generateDiff(data);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to load proposal:", error);
|
||||
}
|
||||
}, [proposalId]);
|
||||
|
||||
const generateDiff = useCallback((prop: Proposal) => {
|
||||
const changes: DiffChange[] = [];
|
||||
|
||||
@@ -75,9 +62,38 @@ export function ProposalReview({ proposalId }: { proposalId: string }) {
|
||||
setDiff(changes);
|
||||
}, []);
|
||||
|
||||
const loadProposal = useCallback(async () => {
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}`);
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
setProposal(data);
|
||||
generateDiff(data);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to load proposal:", error);
|
||||
}
|
||||
}, [proposalId, generateDiff]);
|
||||
|
||||
useEffect(() => {
|
||||
loadProposal();
|
||||
}, [loadProposal]);
|
||||
let ignore = false;
|
||||
async function fetchInitial() {
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}`);
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
if (!ignore) {
|
||||
setProposal(data);
|
||||
generateDiff(data);
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to load proposal:", error);
|
||||
}
|
||||
}
|
||||
void fetchInitial();
|
||||
return () => { ignore = true; };
|
||||
}, [proposalId, generateDiff]);
|
||||
|
||||
const addComment = useCallback(async () => {
|
||||
if (!selectedElement || !commentText || !proposal) return;
|
||||
|
||||
@@ -77,17 +77,35 @@ function ConceptDetailPanel({
|
||||
const [loading, setLoading] = useState(true);
|
||||
const [error, setError] = useState("");
|
||||
|
||||
useEffect(() => {
|
||||
const [prevUri, setPrevUri] = useState(uri);
|
||||
if (uri !== prevUri) {
|
||||
setPrevUri(uri);
|
||||
setLoading(true);
|
||||
setError("");
|
||||
setDetail(null);
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
let ignore = false;
|
||||
fetch(`/api/ontology/skos/concept/${encodeURIComponent(uri)}`)
|
||||
.then((r) => {
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error("Concept not found");
|
||||
return r.json();
|
||||
const data = await r.json();
|
||||
if (r.status === 207) setError(data.message || "Warning: Partial success loading concept.");
|
||||
return data;
|
||||
})
|
||||
.then(setDetail)
|
||||
.catch((e) => setError(e.message))
|
||||
.finally(() => setLoading(false));
|
||||
.then((data) => {
|
||||
if (!ignore) setDetail(data);
|
||||
})
|
||||
.catch((e) => {
|
||||
if (!ignore) setError(e.message);
|
||||
})
|
||||
.finally(() => {
|
||||
if (!ignore) setLoading(false);
|
||||
});
|
||||
return () => {
|
||||
ignore = true;
|
||||
};
|
||||
}, [uri]);
|
||||
|
||||
const renderUriList = (label: string, uris: string[]) => {
|
||||
@@ -322,16 +340,48 @@ function SchemePanel({
|
||||
}) {
|
||||
const [expanded, setExpanded] = useState(true);
|
||||
const [hierarchy, setHierarchy] = useState<ConceptNode[]>([]);
|
||||
const [loading, setLoading] = useState(false);
|
||||
const [loading, setLoading] = useState(expanded);
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const [prevExpanded, setPrevExpanded] = useState(expanded);
|
||||
const [prevSchemeUri, setPrevSchemeUri] = useState(scheme.uri);
|
||||
if (expanded !== prevExpanded || scheme.uri !== prevSchemeUri) {
|
||||
setPrevExpanded(expanded);
|
||||
setPrevSchemeUri(scheme.uri);
|
||||
if (expanded) {
|
||||
setLoading(true);
|
||||
setError("");
|
||||
setHierarchy([]);
|
||||
} else {
|
||||
setLoading(false);
|
||||
}
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
let ignore = false;
|
||||
if (!expanded) return;
|
||||
setLoading(true);
|
||||
fetch(`/api/vocabulary/hierarchy?scheme=${encodeURIComponent(scheme.uri)}`)
|
||||
.then((r) => (r.ok ? r.json() : []))
|
||||
.then(setHierarchy)
|
||||
.catch(() => setHierarchy([]))
|
||||
.finally(() => setLoading(false));
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error(`HTTP ${r.status}`);
|
||||
const data = await r.json();
|
||||
if (r.status === 207 && !ignore) setError(data.message || "Warning: Partial success loading hierarchy.");
|
||||
return data;
|
||||
})
|
||||
.then((data) => {
|
||||
if (!ignore) setHierarchy(data);
|
||||
})
|
||||
.catch((err) => {
|
||||
if (!ignore) {
|
||||
setHierarchy([]);
|
||||
setError(err instanceof Error ? err.message : "Failed to load hierarchy.");
|
||||
}
|
||||
})
|
||||
.finally(() => {
|
||||
if (!ignore) setLoading(false);
|
||||
});
|
||||
return () => {
|
||||
ignore = true;
|
||||
};
|
||||
}, [scheme.uri, expanded]);
|
||||
|
||||
const totalConcepts = countConcepts(hierarchy);
|
||||
@@ -366,6 +416,7 @@ function SchemePanel({
|
||||
|
||||
{expanded && (
|
||||
<div style={{ paddingBottom: 8 }}>
|
||||
{error ? <div style={errorStyle}>{error}</div> : null}
|
||||
{loading ? (
|
||||
<div style={{ padding: "10px 20px", display: "flex", alignItems: "center", gap: 8 }}>
|
||||
<Loader2 size={12} color="#4aa3ff" style={{ animation: "spin 0.8s linear infinite" }} />
|
||||
@@ -410,9 +461,13 @@ export function SKOSVocabularyManager({ schemeUri }: Props) {
|
||||
const [selectedUri, setSelectedUri] = useState<string | null>(null);
|
||||
|
||||
useEffect(() => {
|
||||
setLoading(true);
|
||||
fetch("/api/ontology/skos/schemes")
|
||||
.then((r) => (r.ok ? r.json() : []))
|
||||
.then(async (r) => {
|
||||
if (!r.ok) throw new Error(`Failed to load schemes (${r.status})`);
|
||||
const data = await r.json();
|
||||
if (r.status === 207) setError(data.message || "Warning: Partial success loading schemes.");
|
||||
return data;
|
||||
})
|
||||
.then(setSchemes)
|
||||
.catch((e) => setError(e.message))
|
||||
.finally(() => setLoading(false));
|
||||
@@ -424,6 +479,7 @@ export function SKOSVocabularyManager({ schemeUri }: Props) {
|
||||
|
||||
return (
|
||||
<div style={managerShellStyle}>
|
||||
{error ? <div style={errorStyle}>{error}</div> : null}
|
||||
{/* Search bar */}
|
||||
<div style={skosToolbarStyle}>
|
||||
<div style={skosSearchBarStyle}>
|
||||
@@ -628,6 +684,8 @@ const navLinkStyle: React.CSSProperties = {
|
||||
textAlign: "left",
|
||||
};
|
||||
|
||||
const errorStyle: React.CSSProperties = { padding: 12, borderRadius: 14, color: "#ffb4c2", background: "rgba(255,157,175,0.1)", border: "1px solid rgba(255,157,175,0.18)", margin: "0 10px 10px 10px", fontSize: 12 };
|
||||
|
||||
const centerStyle: React.CSSProperties = {
|
||||
display: "flex",
|
||||
flexDirection: "column",
|
||||
|
||||
@@ -33,38 +33,51 @@ export function ShaclStudio({ onJumpToNode }: ShaclStudioProps) {
|
||||
setRegistry(entries);
|
||||
setSelectedUri((current) => current || entries[0]?.uri || "");
|
||||
})
|
||||
.catch(() => { /* backend unavailable — leave registry empty */ });
|
||||
.catch((err) => {
|
||||
if (cancelled) return;
|
||||
setError(err instanceof Error ? err.message : "Failed to load ontology registry.");
|
||||
});
|
||||
return () => {
|
||||
cancelled = true;
|
||||
};
|
||||
}, []);
|
||||
|
||||
const loadShapes = useCallback(async (uri: string) => {
|
||||
if (!uri) return;
|
||||
setLoading(true);
|
||||
setError("");
|
||||
try {
|
||||
const data = await loadShaclShapes(uri);
|
||||
setShapes(data.shapes);
|
||||
const turtle = data.shacl_turtle;
|
||||
setFullShacl(turtle);
|
||||
setShacl((current) => current || turtle);
|
||||
setSelectedShapeId(null);
|
||||
setValidation(null);
|
||||
} catch {
|
||||
// Shapes not yet generated or backend unavailable — show empty shape list
|
||||
setShapes([]);
|
||||
} finally {
|
||||
setLoading(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
const [prevUri, setPrevUri] = useState(selectedUri);
|
||||
if (selectedUri !== prevUri) {
|
||||
setPrevUri(selectedUri);
|
||||
setShacl("");
|
||||
setFullShacl("");
|
||||
setSelectedShapeId(null);
|
||||
void loadShapes(selectedUri);
|
||||
}, [selectedUri, loadShapes]);
|
||||
setValidation(null);
|
||||
setLoading(true);
|
||||
setError("");
|
||||
}
|
||||
|
||||
useEffect(() => {
|
||||
let ignore = false;
|
||||
async function fetchShapes() {
|
||||
if (!selectedUri) return;
|
||||
try {
|
||||
const data = await loadShaclShapes(selectedUri);
|
||||
if (!ignore) {
|
||||
setShapes(data.shapes);
|
||||
const turtle = data.shacl_turtle;
|
||||
setFullShacl(turtle);
|
||||
setShacl((current) => current || turtle);
|
||||
setSelectedShapeId(null);
|
||||
setValidation(null);
|
||||
}
|
||||
} catch {
|
||||
if (!ignore) setShapes([]);
|
||||
} finally {
|
||||
if (!ignore) setLoading(false);
|
||||
}
|
||||
}
|
||||
void fetchShapes();
|
||||
return () => {
|
||||
ignore = true;
|
||||
};
|
||||
}, [selectedUri]);
|
||||
|
||||
const handleGenerate = useCallback(async () => {
|
||||
if (!selectedUri) return;
|
||||
|
||||
@@ -47,86 +47,151 @@ export function VersionsTab() {
|
||||
const [comparePair, setComparePair] = useState<{ v1: string; v2: string } | null>(null);
|
||||
const [compareResult, setCompareResult] = useState<Record<string, any> | null>(null);
|
||||
const [isLoading, setIsLoading] = useState(false);
|
||||
const [error, setError] = useState("");
|
||||
|
||||
const loadVersions = useCallback(async () => {
|
||||
if (!ontologyUri) return;
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/versions/${encodeURIComponent(ontologyUri)}`);
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
setVersions(data);
|
||||
if (response.status === 207) setError(data.message || "Warning: Partial success loading versions.");
|
||||
} else {
|
||||
setError(`Failed to load versions (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to load versions:", error);
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to load versions.");
|
||||
}
|
||||
}, [ontologyUri]);
|
||||
|
||||
const loadProposals = useCallback(async () => {
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch("/api/ontology/proposals");
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
setProposals(data);
|
||||
if (response.status === 207) setError(data.message || "Warning: Partial success loading proposals.");
|
||||
} else {
|
||||
setError(`Failed to load proposals (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to load proposals:", error);
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to load proposals.");
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
loadVersions();
|
||||
loadProposals();
|
||||
}, [loadVersions, loadProposals]);
|
||||
let ignore = false;
|
||||
async function fetchInitial() {
|
||||
if (!ignore) setError("");
|
||||
try {
|
||||
const propRes = await fetch("/api/ontology/proposals");
|
||||
if (propRes.ok) {
|
||||
const propData = await propRes.json();
|
||||
if (!ignore) {
|
||||
setProposals(propData);
|
||||
if (propRes.status === 207) setError(propData.message || "Warning: Partial success loading proposals.");
|
||||
}
|
||||
} else if (!ignore) {
|
||||
setError(`Failed to load proposals (${propRes.status})`);
|
||||
}
|
||||
} catch (err) {
|
||||
if (!ignore) setError(err instanceof Error ? err.message : "Failed to load proposals.");
|
||||
}
|
||||
|
||||
if (!ontologyUri) return;
|
||||
try {
|
||||
const verRes = await fetch(`/api/ontology/versions/${encodeURIComponent(ontologyUri)}`);
|
||||
if (verRes.ok) {
|
||||
const verData = await verRes.json();
|
||||
if (!ignore) {
|
||||
setVersions(verData);
|
||||
if (verRes.status === 207) setError(verData.message || "Warning: Partial success loading versions.");
|
||||
}
|
||||
} else if (!ignore) {
|
||||
setError((prev) => prev || `Failed to load versions (${verRes.status})`);
|
||||
}
|
||||
} catch (err) {
|
||||
if (!ignore) setError((prev) => prev || (err instanceof Error ? err.message : "Failed to load versions."));
|
||||
}
|
||||
}
|
||||
void fetchInitial();
|
||||
return () => { ignore = true; };
|
||||
}, [ontologyUri]);
|
||||
|
||||
const approveProposal = useCallback(async (proposalId: string) => {
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}/approve`, {
|
||||
method: "POST",
|
||||
});
|
||||
if (response.ok) {
|
||||
alert("Proposal approved");
|
||||
if (response.status === 207) {
|
||||
const data = await response.json().catch(() => ({}));
|
||||
setError(data.message || "Warning: Partial success approving proposal.");
|
||||
} else {
|
||||
alert("Proposal approved");
|
||||
}
|
||||
loadProposals();
|
||||
} else {
|
||||
setError(`Failed to approve proposal (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to approve proposal:", error);
|
||||
alert("Failed to approve proposal");
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to approve proposal.");
|
||||
}
|
||||
}, [loadProposals]);
|
||||
|
||||
const rejectProposal = useCallback(async (proposalId: string) => {
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}/reject`, {
|
||||
method: "POST",
|
||||
});
|
||||
if (response.ok) {
|
||||
alert("Proposal rejected");
|
||||
if (response.status === 207) {
|
||||
const data = await response.json().catch(() => ({}));
|
||||
setError(data.message || "Warning: Partial success rejecting proposal.");
|
||||
} else {
|
||||
alert("Proposal rejected");
|
||||
}
|
||||
loadProposals();
|
||||
} else {
|
||||
setError(`Failed to reject proposal (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to reject proposal:", error);
|
||||
alert("Failed to reject proposal");
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to reject proposal.");
|
||||
}
|
||||
}, [loadProposals]);
|
||||
|
||||
const publishProposal = useCallback(async (proposalId: string) => {
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/proposals/${proposalId}/publish`, {
|
||||
method: "POST",
|
||||
});
|
||||
if (response.ok) {
|
||||
alert("Proposal published");
|
||||
if (response.status === 207) {
|
||||
const data = await response.json().catch(() => ({}));
|
||||
setError(data.message || "Warning: Partial success publishing proposal.");
|
||||
} else {
|
||||
alert("Proposal published");
|
||||
}
|
||||
loadProposals();
|
||||
loadVersions();
|
||||
} else {
|
||||
setError(`Failed to publish proposal (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to publish proposal:", error);
|
||||
alert("Failed to publish proposal");
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to publish proposal.");
|
||||
}
|
||||
}, [loadProposals, loadVersions]);
|
||||
|
||||
const runVersionComparison = useCallback(async () => {
|
||||
if (!comparePair || !ontologyUri) return;
|
||||
setIsLoading(true);
|
||||
setError("");
|
||||
try {
|
||||
const response = await fetch(`/api/ontology/versions/${encodeURIComponent(ontologyUri)}/compare`, {
|
||||
method: "POST",
|
||||
@@ -138,11 +203,13 @@ export function VersionsTab() {
|
||||
});
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
if (response.status === 207) setError(data.message || "Warning: Partial success comparing versions.");
|
||||
setCompareResult(data);
|
||||
} else {
|
||||
setError(`Failed to compare versions (${response.status})`);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error("Failed to compare versions:", error);
|
||||
alert("Failed to compare versions");
|
||||
} catch (err) {
|
||||
setError(err instanceof Error ? err.message : "Failed to compare versions.");
|
||||
} finally {
|
||||
setIsLoading(false);
|
||||
}
|
||||
@@ -265,6 +332,8 @@ export function VersionsTab() {
|
||||
marginBottom: "12px",
|
||||
};
|
||||
|
||||
const errorStyle: React.CSSProperties = { padding: "12px", borderRadius: "14px", color: "#ffb4c2", background: "rgba(255,157,175,0.1)", border: "1px solid rgba(255,157,175,0.18)", marginBottom: "16px" };
|
||||
|
||||
return (
|
||||
<div style={containerStyle}>
|
||||
<div style={headerStyle}>
|
||||
@@ -278,6 +347,8 @@ export function VersionsTab() {
|
||||
/>
|
||||
</div>
|
||||
|
||||
{error ? <div style={errorStyle}>{error}</div> : null}
|
||||
|
||||
<div style={sectionStyle}>
|
||||
<h2 style={sectionTitleStyle}>
|
||||
<Layers size={16} />
|
||||
|
||||
@@ -20,7 +20,11 @@ async function parseResponse<T>(response: Response): Promise<T> {
|
||||
}
|
||||
throw new Error(detail);
|
||||
}
|
||||
return response.json() as Promise<T>;
|
||||
const data = await response.json();
|
||||
if (response.status === 207) {
|
||||
console.warn("Partial Success:", data.message || "Warning: 207 Multi-Status");
|
||||
}
|
||||
return data as T;
|
||||
}
|
||||
|
||||
export async function loadOntologyRegistry(): Promise<OntologyEntry[]> {
|
||||
|
||||
@@ -57,6 +57,7 @@ export function ReasoningWorkspace() {
|
||||
});
|
||||
const data = await response.json();
|
||||
if (!response.ok) throw new Error(data.detail || `Status ${response.status}`);
|
||||
if (response.status === 207) setError(data.message || "Warning: Partial success reasoning.");
|
||||
setResult(data);
|
||||
if (data.mutated) queryClient.invalidateQueries({ queryKey: ["graph", "full-load"] });
|
||||
} catch (e) {
|
||||
|
||||
@@ -72,7 +72,15 @@ export function SparqlWorkspace() {
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify({ query }),
|
||||
});
|
||||
if (!res.headers.get("content-type")?.includes("application/json")) {
|
||||
const text = await res.text();
|
||||
throw new Error(`HTTP ${res.status}: ${text.substring(0, 100)}`);
|
||||
}
|
||||
const data = await res.json();
|
||||
if (res.status === 207) {
|
||||
data.error = data.message || "Warning: Partial success running query.";
|
||||
}
|
||||
|
||||
if (data.error && data.error_line && monaco && editorRef.current) {
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
(monaco as any).editor.setModelMarkers((editorRef.current as any).getModel(), "sparql", [{
|
||||
@@ -81,12 +89,14 @@ export function SparqlWorkspace() {
|
||||
endLineNumber: data.error_line,
|
||||
endColumn: 100,
|
||||
message: data.error,
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
severity: (monaco as any).MarkerSeverity.Error,
|
||||
}]);
|
||||
}
|
||||
setResult(data);
|
||||
} catch {
|
||||
setResult({ error: "Network error — could not reach the SPARQL endpoint." });
|
||||
} catch (e) {
|
||||
const msg = e instanceof Error ? e.message : "Network error — could not reach the SPARQL endpoint.";
|
||||
setResult({ error: msg });
|
||||
} finally {
|
||||
setIsLoading(false);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
/**
|
||||
* Regression tests for issue #830: plugin registry shouldLoad predicates.
|
||||
*
|
||||
* Imports the production predicates from pluginRegistryPredicates.ts so that
|
||||
* a regression in GraphWorkspace.tsx is detected here. The key invariant: no
|
||||
* predicate may read temporalState — doing so caused a render loop because
|
||||
* temporalState.currentTime is non-null from startup, which triggered eager
|
||||
* plugin loads on every scrubber update and continuously cancelled in-flight
|
||||
* load() calls before they could register the plugin.
|
||||
*/
|
||||
import test from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { createRequire } from "node:module";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
|
||||
const {
|
||||
explorationEffectsShouldLoad,
|
||||
neighborhoodPanelShouldLoad,
|
||||
temporalOverlayShouldLoad,
|
||||
} = require("../src/workspaces/GraphWorkspace/pluginRegistryPredicates.ts");
|
||||
|
||||
// ── temporal-overlay ─────────────────────────────────────────────────────────
|
||||
|
||||
test("temporal-overlay shouldLoad: false when panel is closed and no scrubber time", () => {
|
||||
assert.equal(
|
||||
temporalOverlayShouldLoad({ panelState: { "temporal-panel": false } }),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("temporal-overlay shouldLoad: false when panel is closed even if scrubber time is set", () => {
|
||||
// Before the fix, a non-null currentTime caused an eager load on every scrubber update.
|
||||
assert.equal(
|
||||
temporalOverlayShouldLoad({
|
||||
panelState: { "temporal-panel": false },
|
||||
temporalState: { currentTime: new Date() },
|
||||
}),
|
||||
false,
|
||||
);
|
||||
});
|
||||
|
||||
test("temporal-overlay shouldLoad: true only when the panel is explicitly opened", () => {
|
||||
assert.equal(
|
||||
temporalOverlayShouldLoad({ panelState: { "temporal-panel": true } }),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
test("temporal-overlay shouldLoad: true when panel opened even without a scrubber time", () => {
|
||||
assert.equal(
|
||||
temporalOverlayShouldLoad({
|
||||
panelState: { "temporal-panel": true },
|
||||
temporalState: { currentTime: null },
|
||||
}),
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
// ── other entries — confirm they also gate only on panelState ─────────────────
|
||||
|
||||
test("exploration-effects shouldLoad: gates only on effects-panel state", () => {
|
||||
assert.equal(explorationEffectsShouldLoad({ panelState: { "effects-panel": false } }), false);
|
||||
assert.equal(explorationEffectsShouldLoad({ panelState: { "effects-panel": true } }), true);
|
||||
});
|
||||
|
||||
test("neighborhood-panel shouldLoad: gates only on neighborhood-panel state", () => {
|
||||
assert.equal(neighborhoodPanelShouldLoad({ panelState: { "neighborhood-panel": false } }), false);
|
||||
assert.equal(neighborhoodPanelShouldLoad({ panelState: { "neighborhood-panel": true } }), true);
|
||||
});
|
||||
|
||||
test("all three shouldLoad conditions are consistent: none reference temporalState", () => {
|
||||
// A regressed predicate reading temporalState?.currentTime would return true
|
||||
// for a closed panel when currentTime is set — detecting the loop bug.
|
||||
const nonNullTemporalState = { currentTime: new Date(), activeNodeCount: 6 };
|
||||
|
||||
assert.equal(
|
||||
temporalOverlayShouldLoad({ panelState: { "temporal-panel": false }, temporalState: nonNullTemporalState }),
|
||||
false,
|
||||
"temporal-overlay must not load when panel is closed, regardless of scrubber time",
|
||||
);
|
||||
assert.equal(
|
||||
explorationEffectsShouldLoad({ panelState: { "effects-panel": false }, temporalState: nonNullTemporalState }),
|
||||
false,
|
||||
);
|
||||
assert.equal(
|
||||
neighborhoodPanelShouldLoad({ panelState: { "neighborhood-panel": false }, temporalState: nonNullTemporalState }),
|
||||
false,
|
||||
);
|
||||
});
|
||||
@@ -123,12 +123,10 @@ class AgnoDecisionKit(_ToolkitBase): # type: ignore[misc]
|
||||
tools_to_register.append(self.check_policy)
|
||||
|
||||
for fn in tools_to_register:
|
||||
self._tools.append(fn)
|
||||
if AGNO_AVAILABLE:
|
||||
try:
|
||||
self.register(fn)
|
||||
except Exception:
|
||||
pass
|
||||
self.register(fn)
|
||||
if fn not in self._tools:
|
||||
self._tools.append(fn)
|
||||
|
||||
logger.info("AgnoDecisionKit initialised")
|
||||
|
||||
@@ -312,18 +310,33 @@ class AgnoDecisionKit(_ToolkitBase): # type: ignore[misc]
|
||||
Rules are evaluated inline using simple comparison expressions. This
|
||||
avoids misuse of ``PolicyEngine.check_compliance`` (which requires a
|
||||
stored ``Decision`` + ``policy_id``) and ensures exceptions never
|
||||
silently return ``compliant=True``.
|
||||
silently return ``compliant=True``. A rule that references a field
|
||||
missing from ``decision_data``, a field whose value is JSON ``null``,
|
||||
or a rule that doesn't match the expected ``<field> <op> <value>``
|
||||
format, cannot be evaluated — it is recorded in ``warnings`` (not
|
||||
``violations``) since we don't know whether it would have passed or
|
||||
failed. These are reported as distinct messages (missing key vs.
|
||||
null value) so the warning is actionable.
|
||||
|
||||
Parameters
|
||||
----------
|
||||
decision_data:
|
||||
JSON string describing the decision (must include ``category``,
|
||||
``outcome``, ``confidence`` keys at minimum).
|
||||
``outcome``, ``confidence`` keys at minimum). Must decode to a
|
||||
JSON object — any other shape (list, number, string, bool) is
|
||||
rejected with a single ``violations`` entry, the same as
|
||||
malformed JSON, rather than being passed through to per-rule
|
||||
evaluation where it would produce confusing internal errors.
|
||||
policy_rules:
|
||||
JSON list of rule strings, e.g.
|
||||
``'["confidence >= 0.7", "category != \\"test\\""]'``.
|
||||
Each rule is a simple comparison: ``<field> <op> <value>``
|
||||
where op is one of ``>=``, ``<=``, ``!=``, ``==``, ``>``, ``<``.
|
||||
A JSON-encoded bare string (e.g. ``'"confidence >= 0.7"'``) is
|
||||
treated as a single rule. Any other decoded JSON shape (e.g. a
|
||||
number or object), or a non-string list element, is recorded as
|
||||
one ``warnings`` entry and otherwise ignored rather than being
|
||||
iterated character-by-character.
|
||||
|
||||
Returns
|
||||
-------
|
||||
@@ -341,16 +354,47 @@ class AgnoDecisionKit(_ToolkitBase): # type: ignore[misc]
|
||||
}
|
||||
)
|
||||
|
||||
rules: List[str] = []
|
||||
if policy_rules:
|
||||
try:
|
||||
rules = json.loads(policy_rules)
|
||||
except json.JSONDecodeError:
|
||||
rules = [r.strip() for r in policy_rules.split(",") if r.strip()]
|
||||
if not isinstance(data, dict):
|
||||
return json.dumps(
|
||||
{
|
||||
"compliant": False,
|
||||
"violations": [
|
||||
f"decision_data must decode to a JSON object, "
|
||||
f"got {type(data).__name__}: {data!r}"
|
||||
],
|
||||
"warnings": [],
|
||||
}
|
||||
)
|
||||
|
||||
violations: List[str] = []
|
||||
warnings: List[str] = []
|
||||
|
||||
rules: List[str] = []
|
||||
if policy_rules:
|
||||
try:
|
||||
parsed_rules = json.loads(policy_rules)
|
||||
except json.JSONDecodeError:
|
||||
rules = [r.strip() for r in policy_rules.split(",") if r.strip()]
|
||||
else:
|
||||
if isinstance(parsed_rules, str):
|
||||
# A single rule encoded as a bare JSON string, e.g.
|
||||
# policy_rules='"confidence >= 0.7"'. Treat it as one
|
||||
# rule rather than iterating it character-by-character.
|
||||
rules = [parsed_rules]
|
||||
elif isinstance(parsed_rules, list):
|
||||
for item in parsed_rules:
|
||||
if isinstance(item, str):
|
||||
rules.append(item)
|
||||
else:
|
||||
warnings.append(
|
||||
f"Ignoring non-string policy rule entry: {item!r}"
|
||||
)
|
||||
else:
|
||||
warnings.append(
|
||||
f"policy_rules must decode to a JSON list of rule strings, "
|
||||
f"got {type(parsed_rules).__name__}: {parsed_rules!r}"
|
||||
)
|
||||
|
||||
for rule in rules:
|
||||
try:
|
||||
if not self._eval_rule(rule, data):
|
||||
@@ -369,14 +413,23 @@ class AgnoDecisionKit(_ToolkitBase): # type: ignore[misc]
|
||||
)
|
||||
|
||||
def _eval_rule(self, rule: str, data: Dict[str, Any]) -> bool:
|
||||
"""Evaluate a simple comparison rule (``field op value``) against data."""
|
||||
"""
|
||||
Evaluate a simple comparison rule (``field op value``) against data.
|
||||
|
||||
Raises ``ValueError`` when the rule cannot be evaluated (unrecognised
|
||||
format, the referenced field is absent from ``data``, or the field's
|
||||
value is JSON ``null``) so that ``check_policy`` records it as a
|
||||
``warnings`` entry instead of silently treating it as passed.
|
||||
"""
|
||||
m = re.match(r"(\w+)\s*(>=|<=|!=|==|>|<)\s*(.+)", rule.strip())
|
||||
if not m:
|
||||
return True # unrecognised format — pass through
|
||||
raise ValueError(f"unrecognised rule format: {rule!r}")
|
||||
field, op, val_str = m.group(1), m.group(2), m.group(3).strip().strip("\"'")
|
||||
actual = data.get(field)
|
||||
if field not in data:
|
||||
raise ValueError(f"rule references undefined field {field!r}")
|
||||
actual = data[field]
|
||||
if actual is None:
|
||||
return True # field absent — cannot evaluate
|
||||
raise ValueError(f"field {field!r} is null — cannot evaluate rule")
|
||||
try:
|
||||
val: Any = type(actual)(val_str)
|
||||
except (ValueError, TypeError):
|
||||
|
||||
@@ -122,12 +122,10 @@ class AgnoKGToolkit(_ToolkitBase): # type: ignore[misc]
|
||||
self.export_subgraph,
|
||||
]
|
||||
for fn in tools_to_register:
|
||||
self._tools.append(fn)
|
||||
if AGNO_AVAILABLE:
|
||||
try:
|
||||
self.register(fn)
|
||||
except Exception:
|
||||
pass
|
||||
self.register(fn)
|
||||
if fn not in self._tools:
|
||||
self._tools.append(fn)
|
||||
|
||||
logger.info("AgnoKGToolkit initialised (backend=%s)", graph_store_backend)
|
||||
|
||||
|
||||
@@ -83,7 +83,7 @@ class _AgentScopedStore(AgnoContextStore):
|
||||
try:
|
||||
self._context.store(mem_text, conversation_id=self.session_id)
|
||||
except Exception as exc:
|
||||
logger.warning("[%s] store failed: %s", self._role, exc)
|
||||
logger.warning("[%s] store failed: %s", self._role, exc, exc_info=True)
|
||||
|
||||
if self.decision_tracking:
|
||||
try:
|
||||
@@ -94,8 +94,8 @@ class _AgentScopedStore(AgnoContextStore):
|
||||
outcome="stored",
|
||||
confidence=1.0,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
except Exception as exc:
|
||||
logger.warning("[%s] record_decision failed: %s", self._role, exc, exc_info=True)
|
||||
|
||||
if hasattr(memory, "id"):
|
||||
memory.id = mem_id
|
||||
|
||||
+80
-4
@@ -91,11 +91,19 @@ def handle_find_precedents(args: dict) -> dict:
|
||||
|
||||
def handle_get_causal_chain(args: dict) -> dict:
|
||||
"""Trace the upstream or downstream causal chain from a decision."""
|
||||
decision_id = args.get("decision_id", "").strip()
|
||||
if not isinstance(args, dict):
|
||||
return {"error": "args must be a dictionary", "chain": []}
|
||||
decision_id = str(args.get("decision_id") or "").strip()
|
||||
if not decision_id:
|
||||
return {"error": "decision_id is required", "chain": []}
|
||||
direction = args.get("direction", "downstream")
|
||||
max_depth = int(args.get("max_depth", 5))
|
||||
direction = str(args.get("direction") or "downstream").strip()
|
||||
try:
|
||||
max_depth = int(args.get("max_depth", 5))
|
||||
if max_depth <= 0:
|
||||
max_depth = 5
|
||||
max_depth = min(max_depth, 100)
|
||||
except (ValueError, TypeError):
|
||||
max_depth = 5
|
||||
try:
|
||||
graph = get_graph()
|
||||
try:
|
||||
@@ -105,7 +113,75 @@ def handle_get_causal_chain(args: dict) -> dict:
|
||||
decision_id, direction=direction, max_depth=max_depth
|
||||
)
|
||||
except (ImportError, AttributeError):
|
||||
chain = graph.get_causal_chain(decision_id) if hasattr(graph, "get_causal_chain") else []
|
||||
if hasattr(graph, "get_causal_chain"):
|
||||
import inspect
|
||||
# Introspect the signature in its own try/except: only
|
||||
# failure to introspect (ValueError/TypeError from
|
||||
# inspect.signature itself, e.g. a C-extension callable)
|
||||
# should fall through to the trial-and-error cascade below.
|
||||
# A call made after a *successful* introspection must not be
|
||||
# wrapped in that cascade's except block — otherwise a
|
||||
# genuine bug inside get_causal_chain (raising an unrelated
|
||||
# TypeError) gets misread as "wrong signature" and the
|
||||
# backend is invoked a second time with identical arguments.
|
||||
try:
|
||||
params = inspect.signature(graph.get_causal_chain).parameters
|
||||
except (ValueError, TypeError):
|
||||
params = None
|
||||
|
||||
if params is not None:
|
||||
has_var_kwargs = any(
|
||||
p.kind == inspect.Parameter.VAR_KEYWORD
|
||||
for p in params.values()
|
||||
)
|
||||
if has_var_kwargs or (
|
||||
"direction" in params and "max_depth" in params
|
||||
):
|
||||
chain = graph.get_causal_chain(
|
||||
decision_id,
|
||||
direction=direction,
|
||||
max_depth=max_depth,
|
||||
)
|
||||
elif "depth" in params:
|
||||
chain = graph.get_causal_chain(
|
||||
decision_id,
|
||||
depth=max_depth,
|
||||
)
|
||||
else:
|
||||
chain = graph.get_causal_chain(decision_id)
|
||||
else:
|
||||
try:
|
||||
chain = graph.get_causal_chain(
|
||||
decision_id,
|
||||
direction=direction,
|
||||
max_depth=max_depth,
|
||||
)
|
||||
except TypeError as exc:
|
||||
if "unexpected keyword argument" in str(
|
||||
exc
|
||||
) or "positional" in str(exc):
|
||||
try:
|
||||
chain = graph.get_causal_chain(
|
||||
decision_id,
|
||||
depth=max_depth,
|
||||
)
|
||||
except TypeError as exc2:
|
||||
if "unexpected keyword argument" in str(
|
||||
exc2
|
||||
) or "positional" in str(exc2):
|
||||
chain = graph.get_causal_chain(decision_id)
|
||||
else:
|
||||
raise
|
||||
else:
|
||||
raise
|
||||
else:
|
||||
return {
|
||||
"error": (
|
||||
"Causal chain analysis is not supported on this graph"
|
||||
" backend"
|
||||
),
|
||||
"chain": [],
|
||||
}
|
||||
result = chain if isinstance(chain, list) else list(chain)
|
||||
return {"chain": result, "count": len(result), "direction": direction}
|
||||
except Exception as exc:
|
||||
|
||||
+30
-20
@@ -4,13 +4,13 @@ build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "semantica"
|
||||
version = "0.5.1"
|
||||
version = "0.6.0"
|
||||
description = "Accountability and context layer for AI agents. Context graphs, decision intelligence, full provenance tracking, and explainable reasoning engines — every AI decision traceable, every output auditable."
|
||||
readme = "README.md"
|
||||
license = { text = "MIT" }
|
||||
|
||||
authors = [{ name = "Hawksight AI", email = "semantica-dev@users.noreply.github.com" }]
|
||||
maintainers = [{ name = "Hawksight AI", email = "semantica-dev@users.noreply.github.com" }]
|
||||
authors = [{ name = "Semantica", email = "kaif@getsemantica.ai" }]
|
||||
maintainers = [{ name = "Semantica", email = "kaif@getsemantica.ai" }]
|
||||
|
||||
requires-python = ">=3.8"
|
||||
|
||||
@@ -51,7 +51,7 @@ dependencies = [
|
||||
"umap-learn>=0.5.12",
|
||||
"spacy>=3.4.0",
|
||||
"transformers>=4.20.0",
|
||||
"torch>=1.12.0",
|
||||
"torch>=1.13.1",
|
||||
"sentence-transformers>=2.2.0",
|
||||
"rdflib>=6.2.0",
|
||||
"networkx>=2.8.0",
|
||||
@@ -62,14 +62,13 @@ dependencies = [
|
||||
"requests>=2.34.2",
|
||||
"GitPython>=3.1.50",
|
||||
"chardet>=7.4.3",
|
||||
"protobuf>=5.29.1,<7.0",
|
||||
"grpcio>=1.71.2",
|
||||
"protobuf>=5.29.1,<8.0",
|
||||
"grpcio>=1.81.1",
|
||||
"beautifulsoup4>=4.15.0",
|
||||
"lxml>=6.1.1",
|
||||
"pypdf2>=2.10.0",
|
||||
"python-docx>=1.2.0",
|
||||
"openpyxl>=3.1.5",
|
||||
"pillow>=11.3.0",
|
||||
"pillow>=12.2.0",
|
||||
"librosa>=0.9.0",
|
||||
"opencv-python>=4.13.0.92",
|
||||
"faiss-cpu>=1.7.0",
|
||||
@@ -79,13 +78,14 @@ dependencies = [
|
||||
"pydantic>=2.13.4",
|
||||
"click>=8.4.2",
|
||||
"rich>=12.5.0",
|
||||
"tqdm>=4.64.0",
|
||||
"tqdm>=4.68.3",
|
||||
"pyyaml>=6.0",
|
||||
"toml>=0.10.0",
|
||||
"python-dotenv>=1.2.1",
|
||||
"loguru>=0.7.3",
|
||||
"structlog>=22.1.0",
|
||||
"gensim>=4.4.0"
|
||||
"gensim>=4.4.0",
|
||||
"httpx<0.29.0"
|
||||
]
|
||||
|
||||
[project.urls]
|
||||
@@ -116,40 +116,49 @@ llm-all = [
|
||||
# ---- Document Parsing ----
|
||||
parse-docling = ["docling>=2.107.0"]
|
||||
|
||||
# ---- SHACL Validation ----
|
||||
shacl = ["pyshacl>=0.25.0"]
|
||||
|
||||
# ---- Database Connectors ----
|
||||
db-snowflake = ["snowflake-connector-python>=4.6.0", "cryptography>=49.0.0"]
|
||||
db-databricks = ["databricks-sdk>=0.60.0", "databricks-sql-connector>=4.0.0"]
|
||||
db-arrow = ["pyarrow>=24.0.0"]
|
||||
ingest-parquet = ["pyarrow>=24.0.0"]
|
||||
ingest-arrow = ["pyarrow>=24.0.0"]
|
||||
|
||||
db-all = [
|
||||
"semantica[db-snowflake,db-arrow]"
|
||||
"semantica[db-snowflake,db-databricks,db-arrow]"
|
||||
]
|
||||
|
||||
# ---- Embedding / Models ----
|
||||
models-huggingface = [
|
||||
"transformers>=4.20.0",
|
||||
"torch>=1.12.0"
|
||||
"torch>=1.13.1"
|
||||
]
|
||||
|
||||
# ---- Graph Backends ----
|
||||
graph-neo4j = ["neo4j>=5.0.0"]
|
||||
graph-falkordb = ["falkordb>=1.0.0", "redis>=4.3.0"]
|
||||
graph-amazon-neptune = ["boto3>=1.24.0", "neo4j>=5.0.0"]
|
||||
graph-apache-age = ["psycopg2-binary>=2.9.0"]
|
||||
|
||||
graph-all = [
|
||||
"semantica[graph-neo4j,graph-falkordb,graph-amazon-neptune]"
|
||||
"semantica[graph-neo4j,graph-falkordb,graph-amazon-neptune,graph-apache-age]"
|
||||
]
|
||||
|
||||
# ---- Triplet Store Backends ----
|
||||
tripletstore-oxigraph = ["pyoxigraph>=0.5.0"]
|
||||
|
||||
# ---- Vector Store Backends ----
|
||||
vectorstore-qdrant = ["qdrant-client>=1.0.0"]
|
||||
vectorstore-weaviate = ["weaviate-client>=4.0.0"]
|
||||
vectorstore-pinecone = ["pinecone-client>=3.0.0"]
|
||||
vectorstore-milvus = ["pymilvus>=2.0.0"]
|
||||
vectorstore-pgvector = ["psycopg[binary,pool]>=3.0.0", "pgvector>=0.2.0"]
|
||||
vectorstore-sqlite = ["sqlite-vec>=0.1.1"]
|
||||
|
||||
vectorstore-all = [
|
||||
"semantica[vectorstore-qdrant,vectorstore-weaviate,vectorstore-pinecone,vectorstore-milvus,vectorstore-pgvector]"
|
||||
"semantica[vectorstore-qdrant,vectorstore-weaviate,vectorstore-pinecone,vectorstore-milvus,vectorstore-pgvector,vectorstore-sqlite]"
|
||||
]
|
||||
|
||||
# ---- Infra / Queues / Workers ----
|
||||
@@ -173,8 +182,8 @@ monitoring = [
|
||||
"prometheus-client>=0.14.0",
|
||||
"opentelemetry-api>=1.30.0,<2.0.0",
|
||||
"opentelemetry-sdk>=1.30.0,<2.0.0",
|
||||
"opentelemetry-semantic-conventions>=0.58b0,<0.62",
|
||||
"opentelemetry-instrumentation>=0.62b1,<0.62"
|
||||
"opentelemetry-semantic-conventions>=0.58b0,<0.65",
|
||||
"opentelemetry-instrumentation>=0.62b1,<0.65"
|
||||
]
|
||||
|
||||
# ---- Visualization ----
|
||||
@@ -214,7 +223,7 @@ dev = [
|
||||
"isort>=6.1.0",
|
||||
"flake8>=4.0.0",
|
||||
"mypy>=0.971",
|
||||
"pre-commit>=2.19.0",
|
||||
"pre-commit>=4.6.0",
|
||||
"jupyter>=1.0.0",
|
||||
"ipykernel>=6.15.0"
|
||||
]
|
||||
@@ -224,7 +233,8 @@ explorer = [
|
||||
"fastapi>=0.100.0",
|
||||
"uvicorn[standard]>=0.22.0",
|
||||
"websockets>=15.0.1",
|
||||
"python-multipart>=0.0.6"
|
||||
"python-multipart>=0.0.6",
|
||||
"defusedxml>=0.7.1"
|
||||
]
|
||||
explorer-lite = [
|
||||
"streamlit>=1.25.0",
|
||||
@@ -233,8 +243,8 @@ explorer-lite = [
|
||||
|
||||
# Everything (cross-platform — gpu excluded; install semantica[gpu] separately on Linux)
|
||||
all = [
|
||||
"semantica[dev,viz,infra,cloud,monitoring,watch,llm-all,models-huggingface,split-all,graph-all,vectorstore-all,parse-docling,ingest-parquet,ingest-arrow,explorer]",
|
||||
"semantica[dev,viz,infra,cloud,monitoring,watch,llm-all,models-huggingface,split-all,graph-all,vectorstore-all,parse-docling,ingest-parquet,ingest-arrow,agno]"
|
||||
"semantica[dev,viz,infra,cloud,monitoring,watch,llm-all,models-huggingface,split-all,graph-all,tripletstore-oxigraph,vectorstore-all,parse-docling,ingest-parquet,ingest-arrow,shacl,explorer]",
|
||||
"semantica[dev,viz,infra,cloud,monitoring,watch,llm-all,models-huggingface,split-all,graph-all,tripletstore-oxigraph,vectorstore-all,parse-docling,ingest-parquet,ingest-arrow,shacl,agno]"
|
||||
]
|
||||
|
||||
# ---------------- ENTRYPOINTS ----------------
|
||||
|
||||
@@ -10,7 +10,7 @@ Main exports:
|
||||
- Config: Configuration management
|
||||
"""
|
||||
|
||||
__version__ = "0.5.1"
|
||||
__version__ = "0.6.0"
|
||||
__author__ = "Semantica Contributors"
|
||||
__license__ = "MIT"
|
||||
|
||||
|
||||
+142
-7
@@ -2523,9 +2523,11 @@ def provenance_lineage(cli_ctx: CLIContext, entity_id: str, depth: int, local_js
|
||||
default="table", show_default=True)
|
||||
@click.option("--output", default=None, type=click.Path())
|
||||
@click.option("--json", "local_json", is_flag=True, default=False)
|
||||
@click.option("--dry-run", "local_dry", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def provenance_audit(cli_ctx: CLIContext, since: Optional[str], fmt: str,
|
||||
output: Optional[str], local_json: bool) -> None:
|
||||
output: Optional[str], local_json: bool,
|
||||
local_dry: bool) -> None:
|
||||
"""Export the audit log.
|
||||
|
||||
\b
|
||||
@@ -2535,6 +2537,9 @@ def provenance_audit(cli_ctx: CLIContext, since: Optional[str], fmt: str,
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
def _action() -> None:
|
||||
if _is_dry(cli_ctx, local_dry):
|
||||
_dry(cli_ctx, "export audit log", since=since, format=fmt, output=output)
|
||||
return
|
||||
try:
|
||||
from .provenance import ProvenanceManager
|
||||
pm = ProvenanceManager(config=cli_ctx.config.to_dict())
|
||||
@@ -2554,27 +2559,31 @@ def provenance_audit(cli_ctx: CLIContext, since: Optional[str], fmt: str,
|
||||
@provenance.command("export")
|
||||
@click.option("--format", "fmt", type=click.Choice(["turtle", "ntriples", "jsonld"]),
|
||||
default="turtle", show_default=True)
|
||||
@click.option("--base-uri", "base_uri", default=None,
|
||||
help="Namespace URI entities/agents/activities are minted under "
|
||||
"(default: ProvenanceManager.DEFAULT_BASE_URI).")
|
||||
@click.option("--output", default=None, type=click.Path())
|
||||
@click.option("--dry-run", "local_dry", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def provenance_export(cli_ctx: CLIContext, fmt: str, output: Optional[str],
|
||||
local_dry: bool) -> None:
|
||||
def provenance_export(cli_ctx: CLIContext, fmt: str, base_uri: Optional[str],
|
||||
output: Optional[str], local_dry: bool) -> None:
|
||||
"""Export provenance as W3C PROV-O RDF.
|
||||
|
||||
\b
|
||||
Example:
|
||||
semantica provenance export --format turtle --output prov.ttl
|
||||
semantica provenance export --base-uri https://example.org/kg# --output prov.ttl
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
def _action() -> None:
|
||||
if _is_dry(cli_ctx, local_dry):
|
||||
_dry(cli_ctx, "export provenance", format=fmt, output=output)
|
||||
_dry(cli_ctx, "export provenance", format=fmt, base_uri=base_uri, output=output)
|
||||
return
|
||||
try:
|
||||
from .provenance import ProvenanceManager
|
||||
pm = ProvenanceManager(config=cli_ctx.config.to_dict())
|
||||
data = pm.export_prov(format=fmt)
|
||||
data = pm.export_prov(format=fmt, base_uri=base_uri)
|
||||
except ImportError as exc:
|
||||
raise click.ClickException(f"Provenance module not available: {exc}") from exc
|
||||
if output:
|
||||
@@ -2601,10 +2610,126 @@ def provenance_check(cli_ctx: CLIContext, strict: bool, local_json: bool) -> Non
|
||||
result = pm.check(strict=strict)
|
||||
except ImportError as exc:
|
||||
raise click.ClickException(f"Provenance module not available: {exc}") from exc
|
||||
is_valid = not isinstance(result, dict) or result.get("valid", True)
|
||||
if _is_json(cli_ctx, local_json):
|
||||
_jecho(result if isinstance(result, dict) else {"valid": bool(result)})
|
||||
else:
|
||||
elif is_valid:
|
||||
_ok(cli_ctx, f"Provenance check: {result}")
|
||||
else:
|
||||
_warn(cli_ctx, f"Provenance check: {result}")
|
||||
if strict and not is_valid:
|
||||
raise click.ClickException(
|
||||
f"Provenance integrity check failed: {result.get('errors')} error(s)"
|
||||
)
|
||||
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
|
||||
@provenance.command("invalidate")
|
||||
@click.argument("entity_id")
|
||||
@click.option("--by", "agent_id", required=True, help="Agent responsible for the invalidation.")
|
||||
@click.option("--reason", default=None, help="Human-readable reason for the invalidation.")
|
||||
@click.option("--json", "local_json", is_flag=True, default=False)
|
||||
@click.option("--dry-run", "local_dry", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def provenance_invalidate(cli_ctx: CLIContext, entity_id: str, agent_id: str,
|
||||
reason: Optional[str], local_json: bool,
|
||||
local_dry: bool) -> None:
|
||||
"""Mark a tracked entity as invalidated (tombstone, not a hard delete).
|
||||
|
||||
\b
|
||||
Example:
|
||||
semantica provenance invalidate entity_alice --by reviewer_jane --reason "Source retracted"
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
def _action() -> None:
|
||||
if _is_dry(cli_ctx, local_dry):
|
||||
_dry(cli_ctx, "invalidate provenance entity", entity_id=entity_id, by=agent_id, reason=reason)
|
||||
return
|
||||
try:
|
||||
from .provenance import ProvenanceManager
|
||||
pm = ProvenanceManager(config=cli_ctx.config.to_dict())
|
||||
result = pm.invalidate(entity_id, agent_id=agent_id, reason=reason)
|
||||
except ImportError as exc:
|
||||
raise click.ClickException(f"Provenance module not available: {exc}") from exc
|
||||
except ValueError as exc:
|
||||
raise click.ClickException(str(exc)) from exc
|
||||
if _is_json(cli_ctx, local_json):
|
||||
_jecho(result.to_dict())
|
||||
else:
|
||||
_ok(cli_ctx, f"Invalidated {entity_id} (by {agent_id})")
|
||||
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
|
||||
@provenance.command("verify-chain")
|
||||
@click.option("--json", "local_json", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def provenance_verify_chain(cli_ctx: CLIContext, local_json: bool) -> None:
|
||||
"""Verify the hash chain across all provenance entries.
|
||||
|
||||
Detects wholesale row deletion: per-row checksums alone only prove a
|
||||
surviving row wasn't edited in place, not that no row is missing.
|
||||
|
||||
\b
|
||||
Example:
|
||||
semantica provenance verify-chain
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
def _action() -> None:
|
||||
try:
|
||||
from .provenance import ProvenanceManager
|
||||
pm = ProvenanceManager(config=cli_ctx.config.to_dict())
|
||||
result = pm.verify_chain()
|
||||
except ImportError as exc:
|
||||
raise click.ClickException(f"Provenance module not available: {exc}") from exc
|
||||
if _is_json(cli_ctx, local_json):
|
||||
_jecho(result)
|
||||
elif result.get("valid"):
|
||||
_ok(cli_ctx, f"Chain verified: {result.get('total_entries')} entries, no breaks")
|
||||
else:
|
||||
_warn(cli_ctx, f"Chain verification failed: {len(result.get('broken_links', []))} broken link(s)")
|
||||
for link in result.get("broken_links", []):
|
||||
click.echo(f" {link}")
|
||||
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
|
||||
@provenance.command("descendants")
|
||||
@click.argument("entity_id")
|
||||
@click.option("--depth", default=None, type=int, show_default=True)
|
||||
@click.option("--json", "local_json", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def provenance_descendants(cli_ctx: CLIContext, entity_id: str, depth: Optional[int],
|
||||
local_json: bool) -> None:
|
||||
"""Show downstream descendants (reverse lineage) for an entity.
|
||||
|
||||
The counterpart to `provenance lineage`, which only traces upstream
|
||||
ancestors. Answers "entity X was wrong — what downstream facts used it?"
|
||||
|
||||
\b
|
||||
Example:
|
||||
semantica provenance descendants entity_alice --depth 3
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
def _action() -> None:
|
||||
try:
|
||||
from .provenance import ProvenanceManager
|
||||
pm = ProvenanceManager(config=cli_ctx.config.to_dict())
|
||||
if depth is not None:
|
||||
entries = [e.to_dict() for e in pm.trace_descendants(entity_id, max_depth=depth)]
|
||||
result = {"entity_id": entity_id, "depth": depth, "entries": entries}
|
||||
else:
|
||||
result = pm.get_descendants(entity_id) or {"entity_id": entity_id, "entries": []}
|
||||
except ImportError as exc:
|
||||
raise click.ClickException(f"Provenance module not available: {exc}") from exc
|
||||
if _is_json(cli_ctx, local_json):
|
||||
_jecho(result)
|
||||
else:
|
||||
_pprint(cli_ctx, result)
|
||||
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
@@ -3940,7 +4065,7 @@ def server(ctx: click.Context) -> None:
|
||||
@click.option("--port", default=8000, type=int, show_default=True)
|
||||
@click.option("--workers", default=1, type=int, show_default=True)
|
||||
@click.option("--reload", is_flag=True, default=False, help="Enable hot reload.")
|
||||
@click.option("--host", default="0.0.0.0", show_default=True)
|
||||
@click.option("--host", default="127.0.0.1", show_default=True)
|
||||
@click.pass_obj
|
||||
def server_start(cli_ctx: CLIContext, port: int, workers: int, reload: bool, host: str) -> None:
|
||||
"""Start the REST API server.
|
||||
@@ -3951,6 +4076,16 @@ def server_start(cli_ctx: CLIContext, port: int, workers: int, reload: bool, hos
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
_LOOPBACK_HOSTS = {"127.0.0.1", "::1", "localhost"}
|
||||
if host not in _LOOPBACK_HOSTS:
|
||||
console.print(
|
||||
f"[{_WARN_STY}] ⚠[/{_WARN_STY}] Binding to [cyan]{host}[/cyan] exposes "
|
||||
"the server to the network. Set SEMANTICA_API_KEY before doing this "
|
||||
"in any reachable environment — without it, protected routes refuse "
|
||||
"all requests (503), and with SEMANTICA_ALLOW_ANONYMOUS=true they are "
|
||||
"wide open."
|
||||
)
|
||||
|
||||
def _action() -> None:
|
||||
import subprocess as sp
|
||||
cmd = [
|
||||
|
||||
@@ -15,36 +15,52 @@ License: MIT
|
||||
"""
|
||||
|
||||
from typing import Optional, Dict, Any, List
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
class SourceTrackerWithUnifiedBackend:
|
||||
"""SourceTracker using unified provenance backend."""
|
||||
|
||||
def __init__(self, **config):
|
||||
def __init__(
|
||||
self,
|
||||
agent_id: Optional[str] = None,
|
||||
is_automated: bool = True,
|
||||
**config,
|
||||
):
|
||||
"""Initialize with unified backend or fallback to legacy."""
|
||||
from .source_tracker import SourceTracker
|
||||
|
||||
|
||||
self._agent_id = agent_id or self.__class__.__name__
|
||||
self._is_automated = is_automated
|
||||
|
||||
try:
|
||||
from semantica.provenance import ProvenanceManager
|
||||
self._unified_manager = ProvenanceManager()
|
||||
self._use_unified = True
|
||||
except ImportError:
|
||||
self._use_unified = False
|
||||
|
||||
|
||||
self._original_tracker = SourceTracker(**config)
|
||||
|
||||
|
||||
def track_property_source(self, entity_id: str, property_name: str, value: Any, source: Any, **metadata):
|
||||
"""Track property source with unified backend."""
|
||||
activity_started_at_time = datetime.utcnow().isoformat()
|
||||
if self._use_unified:
|
||||
from semantica.provenance import SourceReference
|
||||
|
||||
|
||||
source_ref = SourceReference(
|
||||
document=source.document if hasattr(source, 'document') else str(source),
|
||||
page=getattr(source, 'page', None),
|
||||
section=getattr(source, 'section', None),
|
||||
confidence=getattr(source, 'confidence', 1.0)
|
||||
)
|
||||
|
||||
|
||||
metadata.setdefault("agent_id", self._agent_id)
|
||||
metadata.setdefault("agent_type", "software_agent")
|
||||
metadata.setdefault("is_automated", self._is_automated)
|
||||
metadata.setdefault("activity_started_at_time", activity_started_at_time)
|
||||
metadata.setdefault("activity_ended_at_time", datetime.utcnow().isoformat())
|
||||
|
||||
self._unified_manager.track_property_source(
|
||||
entity_id=entity_id,
|
||||
property_name=property_name,
|
||||
|
||||
@@ -614,29 +614,44 @@ class SourceTracker:
|
||||
if item_type == "entity":
|
||||
entity_id = item.get("entity_id")
|
||||
if entity_id:
|
||||
self.track_entity_source(entity_id, source_ref, **metadata)
|
||||
stats["entities_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
ok = self.track_entity_source(entity_id, source_ref, **metadata)
|
||||
if ok:
|
||||
stats["entities_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
else:
|
||||
self.logger.warning(
|
||||
f"Failed to track entity source for '{entity_id}' in item {i}"
|
||||
)
|
||||
|
||||
elif item_type == "property":
|
||||
entity_id = item.get("entity_id")
|
||||
property_name = item.get("property_name")
|
||||
value = item.get("value")
|
||||
if entity_id and property_name is not None:
|
||||
self.track_property_source(
|
||||
ok = self.track_property_source(
|
||||
entity_id, property_name, value, source_ref, **metadata
|
||||
)
|
||||
stats["properties_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
if ok:
|
||||
stats["properties_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
else:
|
||||
self.logger.warning(
|
||||
f"Failed to track property source for '{entity_id}.{property_name}' in item {i}"
|
||||
)
|
||||
|
||||
elif item_type == "relationship":
|
||||
relationship_id = item.get("relationship_id")
|
||||
if relationship_id:
|
||||
self.track_relationship_source(
|
||||
ok = self.track_relationship_source(
|
||||
relationship_id, source_ref, **metadata
|
||||
)
|
||||
stats["relationships_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
if ok:
|
||||
stats["relationships_tracked"] += 1
|
||||
stats["total_tracked"] += 1
|
||||
else:
|
||||
self.logger.warning(
|
||||
f"Failed to track relationship source for '{relationship_id}' in item {i}"
|
||||
)
|
||||
|
||||
else:
|
||||
self.logger.warning(
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -117,6 +117,7 @@ import uuid
|
||||
from ..utils.logging import get_logger
|
||||
from ..utils.progress_tracker import get_progress_tracker
|
||||
from ..utils.helpers import classify_path_distance
|
||||
from ..utils.skos import is_skos_hierarchy_edge, validate_skos_hierarchy
|
||||
from .entity_linker import EntityLinker
|
||||
|
||||
# Optional imports for advanced features
|
||||
@@ -589,6 +590,14 @@ class ContextGraph:
|
||||
"""
|
||||
count = 0
|
||||
with self._lock:
|
||||
# Keep the SKOS hierarchy invariant at the lowest common write
|
||||
# layer so direct graph users cannot bypass API/session checks.
|
||||
hierarchy_edges = [edge for edge in edges if is_skos_hierarchy_edge(edge)]
|
||||
if hierarchy_edges:
|
||||
existing_edges = [
|
||||
edge for edge in self.find_edges() if is_skos_hierarchy_edge(edge)
|
||||
]
|
||||
validate_skos_hierarchy(hierarchy_edges, existing_edges)
|
||||
for raw_edge in edges:
|
||||
if not isinstance(raw_edge, dict):
|
||||
continue
|
||||
@@ -948,6 +957,12 @@ class ContextGraph:
|
||||
family_id=explicit_family_id,
|
||||
)
|
||||
with self._lock:
|
||||
candidate = {"source": source_id, "target": target_id, "type": edge_type}
|
||||
if is_skos_hierarchy_edge(candidate):
|
||||
existing_edges = [
|
||||
edge for edge in self.find_edges() if is_skos_hierarchy_edge(edge)
|
||||
]
|
||||
validate_skos_hierarchy([candidate], existing_edges)
|
||||
return self._add_internal_edge(
|
||||
ContextEdge(
|
||||
edge_id=edge_id,
|
||||
@@ -1676,7 +1691,7 @@ class ContextGraph:
|
||||
# Fallback ID generation
|
||||
import hashlib
|
||||
|
||||
entity_hash = hashlib.md5(
|
||||
entity_hash = hashlib.md5( # nosec B324 - deterministic entity ID, not security-sensitive
|
||||
f"{entity_text}_{entity_type}".encode()
|
||||
).hexdigest()[:12]
|
||||
entity_id = f"{entity_type.lower()}_{entity_hash}"
|
||||
@@ -2214,12 +2229,26 @@ class ContextGraph:
|
||||
|
||||
# Node embeddings
|
||||
if "node_embedder" in self.kg_components:
|
||||
embeddings = self.kg_components["node_embedder"].generate_embeddings(kg_graph)
|
||||
analysis["node_embeddings"] = embeddings
|
||||
node_labels = list(self.node_type_index.keys())
|
||||
relationship_types = list(self.edge_type_index.keys())
|
||||
if node_labels:
|
||||
embeddings = self.kg_components["node_embedder"].compute_embeddings(
|
||||
graph_store=self,
|
||||
node_labels=node_labels,
|
||||
relationship_types=relationship_types,
|
||||
)
|
||||
analysis["node_embeddings"] = embeddings
|
||||
|
||||
self.logger.info("Completed comprehensive graph analysis")
|
||||
return analysis
|
||||
|
||||
except AttributeError as e:
|
||||
# A broken internal method call (e.g. calling a method that doesn't
|
||||
# exist on one of the kg_components) is a programming error, not a
|
||||
# legitimate empty-analysis result. Log it distinctly and re-raise
|
||||
# rather than masking it under the generic message below.
|
||||
self.logger.error(f"Graph analysis failed due to a broken internal method call: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
self.logger.error(f"Failed to analyze graph with KG: {e}")
|
||||
return {"error": "Graph analysis failed due to an internal error"}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user