diff --git a/.github/workflows/benchmark.yml b/.github/workflows/benchmark.yml index 376b40d6..301f8776 100644 --- a/.github/workflows/benchmark.yml +++ b/.github/workflows/benchmark.yml @@ -2,9 +2,13 @@ name: Semantica Performance Suite on: push: - branches: [main, master] - pull_request: - branches: [main, master] + branches: [main] + paths-ignore: + - 'docs/**' + - 'mkdocs.yml' + - 'requirements-docs.txt' + - '**/*.md' + workflow_dispatch: jobs: performance-test: diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e6e712c5..994503ad 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -3,8 +3,18 @@ name: CI on: push: branches: [main] + paths-ignore: + - 'docs/**' + - 'mkdocs.yml' + - 'requirements-docs.txt' + - '**/*.md' pull_request: branches: [main] + paths-ignore: + - 'docs/**' + - 'mkdocs.yml' + - 'requirements-docs.txt' + - '**/*.md' jobs: build: diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index 48d510c1..fbdbea01 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -8,11 +8,12 @@ on: branches: [main] paths: - 'docs/**' - - 'semantica/**' - 'mkdocs.yml' - 'requirements-docs.txt' - 'CHANGELOG.md' - 'RELEASE.md' + release: + types: [published] workflow_dispatch: # Permissions needed to deploy to GitHub Pages diff --git a/.github/workflows/security-scan.yml b/.github/workflows/security-scan.yml index a2b16977..e4581e6e 100644 --- a/.github/workflows/security-scan.yml +++ b/.github/workflows/security-scan.yml @@ -4,9 +4,19 @@ on: schedule: - cron: '30 1 * * 1,4' # Mon/Thu 7 AM IST push: - branches: [ main ] + branches: [main] + paths-ignore: + - 'docs/**' + - 'mkdocs.yml' + - 'requirements-docs.txt' + - '**/*.md' pull_request: - branches: [ main ] + branches: [main] + paths-ignore: + - 'docs/**' + - 'mkdocs.yml' + - 'requirements-docs.txt' + - '**/*.md' jobs: security-scan: @@ -158,7 +168,7 @@ jobs: } // Create summary comment - const comment = `# ๐Ÿ”’ Security Scan Results\\n\\n${safetyResults}\\n\\n${banditResults}\\n\\n${semgrepResults}\\n\\n---\\n\\n*This security scan runs automatically on every PR and bi-weekly.*\\n\\n๐Ÿ“Š **Security Policy**: CI fails on vulnerabilities and HIGH severity issues.`; + const comment = `# ๐Ÿ”’ Security Scan Results\\n\\n${safetyResults}\\n\\n${banditResults}\\n\\n${semgrepResults}\\n\\n---\\n\\n*This security scan runs automatically on source-code PRs and bi-weekly (skipped for doc/markdown-only changes).*\\n\\n๐Ÿ“Š **Security Policy**: CI fails on vulnerabilities and HIGH severity issues.`; // Post comment with error handling try { diff --git a/CHANGELOG.md b/CHANGELOG.md index 03737ecc..32c0b0a3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,80 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +- **Agno Agentic Framework Integration** (Issue #249): + - Added `AgnoContextStore` โ€” graph-backed agent memory implementing the `agno.memory.db.base.MemoryDb` protocol; wraps `AgentContext` + `VectorStore`; supports `create()`, `table_exists()`, `memory_exists()`, `read_memories()`, `upsert_memory()`, `delete_memory()`, `drop_table()`, `clear()` plus extended `record_decision()`, `find_precedents()`, `retrieve()` methods + - Added `AgnoKnowledgeGraph` โ€” multi-hop GraphRAG knowledge base implementing `agno.knowledge.base.AgentKnowledge`; ingests files, directories, URLs, and raw text via NER โ†’ relation extraction โ†’ graph build โ†’ vector index pipeline; `search()` returns `AgnoDocument` objects; `get_graph_context(entity)` returns text summary of entity's graph neighbourhood + - Added `AgnoDecisionKit` โ€” Agno `Toolkit` subclass exposing 6 decision-intelligence tools: `record_decision`, `find_precedents`, `trace_causal_chain`, `analyze_impact`, `check_policy`, `get_decision_summary` + - Added `AgnoKGToolkit` โ€” Agno `Toolkit` subclass exposing 7 KG pipeline tools: `extract_entities`, `extract_relations`, `add_to_graph`, `query_graph`, `find_related`, `infer_facts`, `export_subgraph` + - Added `AgnoSharedContext` โ€” team-level coordinator with a single shared `ContextGraph`; `bind_agent(role)` returns a role-scoped `_AgentScopedStore` with cross-agent memory visibility; thread-safe via `RLock` + - All 5 components degrade gracefully when `agno` is not installed (`AGNO_AVAILABLE` flag); importable and functional without agno present + - Added `agno = ["agno>=1.0.0"]` optional dependency in `pyproject.toml`; included in `all` extra + - 110 integration tests in `tests/integrations/agno/` covering all public APIs, MemoryDb protocol compliance, GraphRAG search, tool registration, shared memory isolation, and thread-safety + - 3 cookbook notebooks in `cookbook/integrations/`: `agno_decision_intelligence.ipynb` (loan underwriting), `agno_graphrag_context.ipynb` (regulatory compliance), `agno_multi_agent_shared_context.ipynb` (multi-agent team coordination) + - Full reference documentation in `docs/integrations/agno.md` + +- **Novita AI Provider** (PR #374 by @Alex-wuhu): + - Added `NovitaProvider` โ€” OpenAI-compatible integration via `https://api.novita.ai/v1`; supports `generate()` and `generate_structured()` (JSON forced format) + - Default model: `deepseek/deepseek-v3.2`; configurable via `NOVITA_API_KEY` environment variable + - Registered `"novita"` in the built-in provider factory; usable via `create_provider("novita")` + - Added integration tests in `tests/test_novita_integration.py` with proper assertions and graceful skip when `NOVITA_API_KEY` is unset + +- **Native Datalog Reasoning Engine** (PR #371, Issue #368 by @ZohaibHassan16, reviewed and fixed by @KaifAhmad1): + - Added `DatalogReasoner` to `semantica.reasoning` โ€” a pure-Python, bottom-up semi-naive fixpoint engine with guaranteed termination on finite graphs + - Supports recursive Horn clause rules (e.g. `ancestor(X,Y) :- parent(X,Z), ancestor(Z,Y).`) that existing engines loop on indefinitely + - Memory-optimized `_unify()` with deferred dict allocation โ€” zero allocation on failed unifications + - `O(1)` delta-index lookup per iteration eliminates redundant `O(N)` rule re-evaluations in semi-naive loop + - `query("pred(?X, ?Y)")` returns variable-binding dicts; supports both uppercase `?Y` and lowercase `?y` variable syntax + - `query(..., bindings={"Y": "val"})` pre-binds variables for exact-match verification + - `load_from_graph(ContextGraph)` converts all edges and nodes to Datalog facts in one call; handles both `find_edges`/`find_nodes` and raw `edges`/`nodes` graph APIs + - `add_fact()` accepts `"pred(a, b)"` strings and Semantica dicts (`subject/predicate/object`, `source/target/type`, `type/id` shapes); warns on unrecognised dict format instead of silently dropping + - `_derived` cache flag โ€” `derive_all()` skips re-evaluation when no facts or rules have changed since last run; `query()` respects the cache + - Progress tracking wrapped in `try/finally` โ€” `stop_tracking()` always called even on exception + - `DatalogReasoner`, `DatalogFact`, `DatalogRule` exported from `semantica.reasoning` + - 18 tests covering recursive rules, multi-hop inference, variable binding, graph integration, idempotency, and edge cases โ€” all passing +- **Ontology Diff & Migration** (PR #367 by @ZohaibHassan16, review & fixes by @KaifAhmad1): + - `VersionManager.diff_ontologies(base, target)` โ€” structured diff between two ontology dicts using hash-map lookups; handles URI-less items via `name` fallback; deep equality checks for unordered lists; now covers classes, properties, individuals, and axioms + - `ChangeLogAnalyzer.analyze(diff)` โ€” classifies each change by semantic impact: removed classes/properties โ†’ `CRITICAL/BREAKING`; narrowed domain/range/cardinality โ†’ `HIGH/BREAKING`; hierarchy modifications โ†’ `MEDIUM/POTENTIALLY_BREAKING`; added elements and annotation updates โ†’ `INFO/NON_BREAKING` + - `ImpactReport` dataclass and `generate_change_report(diff)` public helper โ€” returns a structured dict with `summary`, `impact_classification` (breaking / potentially_breaking / safe), `recommendations`, and the raw `diff` + - `OntologyEngine.compare_versions(base_id, target_id, **options)` โ€” end-to-end orchestrator: loads versions from `VersionManager`, runs `diff_ontologies`, generates impact report; accepts `base_dict`/`target_dict` overrides to bypass version store; `run_validation=True` triggers `OntologyValidator` on the target schema; `graph_data=...` additionally runs `GraphValidator` on instance data against the new schema + - `OntologyEngine.get_ontology_version_dict(version_id)` โ€” utility to load a registered version as a plain dict ready for diffing + - Documentation added to `docs/reference/change_management.md`: "Ontology Diff & Migration" section with code example and full report format reference + - 7 tests added to `tests/change_management/test_managers.py` covering: empty diff, unordered list equality, URI/name fallback, breaking class removal, narrowed domain (HIGH), safe additions and annotation changes, `compare_versions` dict override, version-not-found error path, individuals/axioms diff coverage, null constraint value flagged as breaking + - **Fixes applied post-review (by @KaifAhmad1)**: + - Fixed typo in `ChangeCategory` enum value: `"potenitally_breaking"` โ†’ `"potentially_breaking"` + - Fixed missing space in impact description string: `f"New{entity_type}"` โ†’ `f"New {entity_type}"` + - Added null-value guard in `_analyze_field_changes` โ€” constraint fields with `None` old/new value are now correctly flagged as breaking instead of silently passing the subset check + - Made `ChangeLogAnalyzer` stateless โ€” `report` is now a local variable passed into `_generate_recommendations(report)` rather than stored as `self.report`; removes re-entrancy hazard + - Removed no-op `__init__` from `ChangeLogAnalyzer` + - Replaced non-portable emoji markers in recommendations (`โœ˜โœ˜โœ˜`, `ยคยคยค`, `โ˜บโ˜บโ˜บ`) with plain-text tags (`[BREAKING]`, `[WARNING]`, `[SAFE]`) + - Extended `diff_ontologies` to cover `individuals` and `axioms` โ€” previously only classes and properties were diffed; the public `compare_versions` path now returns all four element types + - Fixed exception chaining in `compare_versions`: `raise ProcessingError(...) from e` to preserve original traceback + - Removed silent `ImportError` swallow for `GraphValidator` โ€” it is a first-party module; an `ImportError` indicates a broken install, not a graceful skip + - Added comment on deferred `VersionManager` import in `OntologyEngine.__init__` explaining the circular-import constraint + - Fixed import-before-docstring in `tests/change_management/test_managers.py` + - Fixed broken Markdown link syntax in docs JSON example block: `"[http://...](http://...)"` โ†’ bare URI string + - Updated docs recommendations example to match the new plain-text tag format + +- **Ontology Alignment API** (PR #361 by @ZohaibHassan16, review & fixes by @KaifAhmad1): + - Alignment representation using standard RDF predicates: `owl:equivalentClass`, `owl:equivalentProperty`, `owl:sameAs`, `skos:exactMatch`, `skos:closeMatch`, `skos:broadMatch`, `skos:narrowMatch`, `skos:relatedMatch` + - `OntologyEngine.create_alignment(source_uri, target_uri, predicate)` โ€” store alignment triples in TripletStore + - `OntologyEngine.get_alignments(entity_uri)` โ€” bidirectional retrieval of all alignments for an entity + - `OntologyEngine.list_alignments(ontology_uri=None)` โ€” list all alignments, optionally filtered by ontology namespace + - `NamespaceManager.get_alignment_predicates()` โ€” expose standard OWL/SKOS alignment URIs as a convenience dict + - `ReuseManager.suggest_alignments(target, source)` โ€” O(N+M) hashmap heuristic to suggest alignments based on exact label matches across ontologies + - `ReuseManager.merge_ontology_data(..., compute_alignments=True)` โ€” optionally attach suggested alignments to merge output without auto-committing unverified triples + - `QueryEngine.expand_entity_uri(uri, store, use_alignments=True)` โ€” bidirectional SPARQL expansion to include aligned equivalents; no-ops when flag is False + - `QueryEngine.build_values_clause(variable, uris)` โ€” generate a SPARQL `VALUES` clause for injecting expanded URIs into queries + - Alignment-aware queries section added to `docs/reference/triplet_store.md` + - Ontology Alignment section added to `docs/reference/ontology.md` + - **Fixes applied post-review (by @KaifAhmad1)**: + - Fixed progress tracker leak in `expand_entity_uri` โ€” `stop_tracking` was only called inside the `hasattr(execute_sparql)` branch; backends without it silently leaked a tracker entry + - Fixed `relatedMatch` predicate gap โ€” `get_alignment_predicates()` exposed `skos:relatedMatch` but all three SPARQL FILTER lists omitted it, making those alignments permanently invisible + - Fixed SPARQL injection in `list_alignments` โ€” previously only `"` was escaped; `\`, `{`, and `}` are now also percent-encoded to prevent WHERE block breakout + - Fixed SPARQL injection in `build_values_clause` โ€” URIs now run through `_sanitize_uri` before wrapping in angle-bracket literals + - Added full-URI validation in `create_alignment` โ€” raises `ProcessingError` if predicate is a CURIE instead of a full URI, preventing silent storage of unqueryable triples + - Fixed E2E test `test_end_to_end_cross_ontology_uri_flow` โ€” previously mocked the method under test; now uses a real mock backend with `execute_sparql` to exercise the actual expansion and VALUES clause injection flow + - 19 tests added covering: `create_alignment`, `get_alignments`, `suggest_alignments`, merge with alignment computation, `expand_entity_uri` (enabled/disabled), `build_values_clause`, and full E2E cross-ontology query flow - **Context Explainability Output Fixes** (PR pending on `context` by @KaifAhmad1): - Fixed decision-node storage in `ContextGraph` so full human-readable `scenario`, `reasoning`, and decision metadata are preserved on graph nodes instead of degrading into opaque IDs or truncated display text - Fixed causal and precedent reconstruction paths in the context module so returned `Decision` objects prefer readable stored fields over raw node identifiers diff --git a/README.md b/README.md index 11ca03ee..9a19ff57 100644 --- a/README.md +++ b/README.md @@ -187,7 +187,7 @@ See [RELEASE_NOTES.md](RELEASE_NOTES.md) for the full per-contributor breakdown | `semantica.visualization` | Interactive and static visualization of KGs, ontologies, embeddings, analytics, and temporal graphs | | `semantica.seed` | Seed data management for initial KG construction from CSV, JSON, databases, and APIs | | `semantica.core` | Framework orchestration, configuration management, knowledge base construction, plugin system | -| `semantica.llms` | LLM provider integrations โ€” Groq, OpenAI, HuggingFace, LiteLLM | +| `semantica.llms` | LLM provider integrations โ€” Groq, OpenAI, Novita AI, HuggingFace, LiteLLM | | `semantica.utils` | Shared utilities โ€” logging, validation, exception handling, constants, types, progress tracking | --- @@ -680,9 +680,19 @@ ontology = importer.load("context.jsonld") **LLM Providers** - 100+ models via LiteLLM โ€” OpenAI, Anthropic, Cohere, Mistral, Ollama, Azure, AWS Bedrock, and more +- Novita AI โ€” OpenAI-compatible provider (`deepseek/deepseek-v3.2` and more); configure via `NOVITA_API_KEY` -**AI Frameworks** -- Complements LangChain, LlamaIndex, AutoGen, CrewAI, Google ADK +**Agentic Frameworks** +- Complements LangChain, LlamaIndex, AutoGen, CrewAI, Google ADK, and more + +> **Agno โ€” First-Class Integration** `pip install semantica[agno]` +> +> Semantica ships a dedicated Agno integration with five ready-to-use components: +> - **`AgnoContextStore`** โ€” graph-backed agent memory +> - **`AgnoKnowledgeGraph`** โ€” multi-hop GraphRAG knowledge base +> - **`AgnoDecisionKit`** โ€” 6 decision-intelligence tools +> - **`AgnoKGToolkit`** โ€” 7 knowledge-graph pipeline tools +> - **`AgnoSharedContext`** โ€” shared context graph for multi-agent teams **Export** - RDF: Turtle, JSON-LD, N-Triples, XML ยท Parquet ยท ArangoDB AQL diff --git a/cookbook/integrations/agno_decision_intelligence.ipynb b/cookbook/integrations/agno_decision_intelligence.ipynb new file mode 100644 index 00000000..168f6e92 --- /dev/null +++ b/cookbook/integrations/agno_decision_intelligence.ipynb @@ -0,0 +1,534 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "title", + "metadata": {}, + "source": [ + "# Agno ร— Semantica: Decision Intelligence Agent\n", + "\n", + "This notebook shows how to wire Semantica's **Decision Intelligence** stack into an Agno agent so it can:\n", + "\n", + "- Record every decision it makes with full reasoning provenance\n", + "- Search historical precedents before acting\n", + "- Validate decisions against policy rules\n", + "- Trace causal chains across decisions\n", + "- Accumulate institutional knowledge that survives across sessions\n", + "\n", + "**Domain used:** Financial loan underwriting (easily adapted to healthcare, legal, HR, etc.)\n", + "\n", + "---\n", + "\n", + "## Architecture\n", + "\n", + "```\n", + "Agno Agent\n", + " โ”œโ”€โ”€ memory=AgnoContextStore โ† graph-backed persistent memory\n", + " โ””โ”€โ”€ tools=[AgnoDecisionKit] โ† decision tools the LLM can call\n", + " โ”‚\n", + " โ”œโ”€โ”€ record_decision โ† Semantica AgentContext.record_decision()\n", + " โ”œโ”€โ”€ find_precedents โ† Semantica AgentContext.find_precedents_advanced()\n", + " โ”œโ”€โ”€ trace_causal_chain โ† Semantica ContextGraph.trace_decision_causality()\n", + " โ”œโ”€โ”€ analyze_impact โ† Semantica AgentContext.analyze_decision_influence()\n", + " โ”œโ”€โ”€ check_policy โ† Semantica PolicyEngine\n", + " โ””โ”€โ”€ get_decision_summary โ† Semantica AgentContext.get_context_insights()\n", + "```\n", + "\n", + "## Install\n", + "\n", + "```bash\n", + "pip install semantica[agno]\n", + "```" + ] + }, + { + "cell_type": "markdown", + "id": "setup-section", + "metadata": {}, + "source": [ + "## 1. Setup โ€” Semantica Backends\n", + "\n", + "We build the Semantica components first. These are **independent of Agno** โ€” you can swap backends without touching agent code." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "imports", + "metadata": {}, + "outputs": [], + "source": [ + "import sys, os\n", + "sys.path.insert(0, os.path.abspath(\"../../\"))\n", + "\n", + "# โ”€โ”€ Semantica core (not Agno-specific) โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from semantica.context import AgentContext, ContextGraph\n", + "from semantica.context import PolicyEngine, DecisionQuery, CausalChainAnalyzer\n", + "from semantica.vector_store import VectorStore\n", + "\n", + "# โ”€โ”€ Agno integration layer โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from integrations.agno import AgnoContextStore, AgnoDecisionKit, AGNO_AVAILABLE\n", + "\n", + "print(f\"Semantica imports OK\")\n", + "print(f\"Agno installed: {AGNO_AVAILABLE}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "semantica-backends", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ Vector store (FAISS, no external service needed) โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "vector_store = VectorStore(backend=\"faiss\", dimension=768)\n", + "print(\"VectorStore ready (FAISS)\")\n", + "\n", + "# โ”€โ”€ In-memory context graph with full analytics โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "knowledge_graph = ContextGraph(\n", + " advanced_analytics=True,\n", + " # Switch to neo4j for production:\n", + " # backend=\"neo4j\", uri=\"bolt://localhost:7687\"\n", + ")\n", + "print(\"ContextGraph ready (in-memory)\")" + ] + }, + { + "cell_type": "markdown", + "id": "seed-section", + "metadata": {}, + "source": [ + "## 2. Seed Historical Decisions\n", + "\n", + "Before the agent runs, we pre-load historical decisions using **native Semantica APIs** so the precedent database is warm.\n", + "\n", + "In production you would ingest from a database or a prior session's graph export." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "seed-decisions", + "metadata": {}, + "outputs": [], + "source": [ + "# Build a pure-Semantica AgentContext for seeding historical data\n", + "seed_context = AgentContext(\n", + " vector_store=vector_store,\n", + " knowledge_graph=knowledge_graph,\n", + " decision_tracking=True,\n", + ")\n", + "\n", + "historical_loans = [\n", + " dict(\n", + " category=\"loan_approval\",\n", + " scenario=\"Applicant: credit score 740, income $95k, DTI 28%, down payment 20%\",\n", + " reasoning=\"Strong credit history, debt load well below 35% threshold, adequate down payment\",\n", + " outcome=\"approved\",\n", + " confidence=0.96,\n", + " ),\n", + " dict(\n", + " category=\"loan_approval\",\n", + " scenario=\"Applicant: credit score 620, income $45k, DTI 42%, down payment 5%\",\n", + " reasoning=\"Credit score below 650 floor, DTI exceeds 40% maximum, insufficient down payment\",\n", + " outcome=\"rejected\",\n", + " confidence=0.97,\n", + " ),\n", + " dict(\n", + " category=\"loan_approval\",\n", + " scenario=\"Applicant: credit score 700, income $72k, DTI 33%, down payment 15%\",\n", + " reasoning=\"Adequate credit, moderate DTI within range, down payment slightly below ideal\",\n", + " outcome=\"approved_with_conditions\",\n", + " confidence=0.82,\n", + " ),\n", + " dict(\n", + " category=\"loan_approval\",\n", + " scenario=\"Applicant: credit score 780, income $130k, DTI 22%, down payment 30%\",\n", + " reasoning=\"Excellent credit, low debt load, strong down payment โ€” low-risk profile\",\n", + " outcome=\"approved\",\n", + " confidence=0.99,\n", + " ),\n", + " dict(\n", + " category=\"loan_approval\",\n", + " scenario=\"Applicant: credit score 660, income $58k, DTI 38%, down payment 10%\",\n", + " reasoning=\"Borderline credit, high DTI, minimal down payment โ€” escalated to senior review\",\n", + " outcome=\"escalated\",\n", + " confidence=0.70,\n", + " ),\n", + "]\n", + "\n", + "for loan in historical_loans:\n", + " did = seed_context.record_decision(**loan)\n", + " print(f\" Seeded [{loan['outcome']:25s}] โ†’ {did}\")\n", + "\n", + "print(f\"\\n{len(historical_loans)} historical decisions loaded into Semantica KG\")" + ] + }, + { + "cell_type": "markdown", + "id": "policy-section", + "metadata": {}, + "source": [ + "## 3. Define Policy Rules with Semantica\n", + "\n", + "We use `PolicyEngine` directly โ€” no Agno involvement here. The `AgnoDecisionKit.check_policy` tool will call this engine during the agent's reasoning loop." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "policy", + "metadata": {}, + "outputs": [], + "source": [ + "LENDING_POLICY_RULES = [\n", + " \"credit_score >= 650\",\n", + " \"dti <= 40\",\n", + " \"down_payment_pct >= 10\",\n", + " \"confidence >= 0.70\",\n", + "]\n", + "\n", + "# Verify directly with Semantica's PolicyEngine before wiring to Agno\n", + "policy_engine = PolicyEngine(graph_store=knowledge_graph)\n", + "\n", + "test_application = {\"credit_score\": 720, \"dti\": 31, \"down_payment_pct\": 18, \"confidence\": 0.88}\n", + "\n", + "try:\n", + " result = policy_engine.check_compliance(test_application, LENDING_POLICY_RULES)\n", + " print(f\"Policy check result: compliant={getattr(result, 'compliant', 'N/A')}\")\n", + " print(f\"Violations: {getattr(result, 'violations', [])}\")\n", + "except Exception as e:\n", + " print(f\"PolicyEngine fallback (expected without full rule engine): {e}\")\n", + "\n", + "print(\"\\nPolicy rules defined:\", LENDING_POLICY_RULES)" + ] + }, + { + "cell_type": "markdown", + "id": "agent-section", + "metadata": {}, + "source": [ + "## 4. Build the Agno Decision-Intelligence Agent\n", + "\n", + "Now we wire everything into Agno using the integration classes.\n", + "\n", + "- `AgnoContextStore` gives the agent **persistent graph-backed memory**\n", + "- `AgnoDecisionKit` exposes **6 decision tools** the LLM can invoke during reasoning" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-agent", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ AgnoContextStore: wraps AgentContext as Agno MemoryDb โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "store = AgnoContextStore(\n", + " vector_store=vector_store, # Same store โ€” shares seeded decisions\n", + " knowledge_graph=knowledge_graph, # Same graph โ€” shares seeded decisions\n", + " decision_tracking=True,\n", + " graph_expansion=True,\n", + " session_id=\"loan_underwriter_v1\",\n", + ")\n", + "print(\"AgnoContextStore ready\")\n", + "\n", + "# โ”€โ”€ AgnoDecisionKit: exposes Semantica decision tools to Agno's LLM โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "decision_kit = AgnoDecisionKit(\n", + " context=store.context, # Reuse same AgentContext โ€” shared decision history\n", + " max_precedents=5,\n", + " causal_depth=3,\n", + " enable_policy_check=True,\n", + ")\n", + "print(f\"AgnoDecisionKit ready โ€” {len(decision_kit._tools)} tools registered\")\n", + "print(\" Tools:\", [fn.__name__ for fn in decision_kit._tools])" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "wire-agent", + "metadata": {}, + "outputs": [], + "source": [ + "if AGNO_AVAILABLE:\n", + " from agno.agent import Agent\n", + " from agno.memory import AgentMemory\n", + " from agno.models.openai import OpenAIChat # or any Agno-supported model\n", + "\n", + " agent = Agent(\n", + " name=\"LoanUnderwriter\",\n", + " model=OpenAIChat(id=\"gpt-4o\"),\n", + " memory=AgentMemory(db=store),\n", + " tools=[decision_kit],\n", + " show_tool_calls=True,\n", + " description=(\n", + " \"You are a senior loan underwriter. Before approving or rejecting any application:\"\n", + " \" (1) find_precedents for similar past cases,\"\n", + " \" (2) check_policy compliance,\"\n", + " \" (3) record_decision with full reasoning.\"\n", + " \" Always cite precedents and policy rule results in your explanation.\"\n", + " ),\n", + " )\n", + " print(\"Agno Agent assembled and ready\")\n", + "else:\n", + " print(\"Agno not installed โ€” demonstrating tool calls directly below\")" + ] + }, + { + "cell_type": "markdown", + "id": "demo-section", + "metadata": {}, + "source": [ + "## 5. Demonstrate Decision Tools\n", + "\n", + "We call the decision tools **directly** so the notebook is fully runnable without an OpenAI key. When Agno is wired, the LLM orchestrates these same calls automatically." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-find-precedents", + "metadata": {}, + "outputs": [], + "source": [ + "import json\n", + "\n", + "# โ”€โ”€ 5a. Find Precedents โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "print(\"=\" * 60)\n", + "print(\"TOOL: find_precedents\")\n", + "print(\"=\" * 60)\n", + "\n", + "new_application_scenario = (\n", + " \"Applicant: credit score 715, income $82k, DTI 30%, down payment 18%\"\n", + ")\n", + "\n", + "precedents_json = decision_kit.find_precedents(\n", + " scenario=new_application_scenario,\n", + " category=\"loan_approval\",\n", + " limit=3,\n", + ")\n", + "precedents = json.loads(precedents_json)\n", + "print(f\"Found {precedents['count']} similar past decisions:\")\n", + "for p in precedents['precedents']:\n", + " print(f\" [{p.get('outcome','?'):25s}] confidence={p.get('confidence',0):.2f}\")\n", + " print(f\" {p.get('scenario','')[:80]}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-policy", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ 5b. Check Policy โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "print(\"=\" * 60)\n", + "print(\"TOOL: check_policy\")\n", + "print(\"=\" * 60)\n", + "\n", + "decision_data = json.dumps({\n", + " \"credit_score\": 715,\n", + " \"dti\": 30,\n", + " \"down_payment_pct\": 18,\n", + " \"confidence\": 0.88,\n", + " \"outcome\": \"approved\",\n", + "})\n", + "\n", + "policy_json = decision_kit.check_policy(\n", + " decision_data=decision_data,\n", + " policy_rules=json.dumps(LENDING_POLICY_RULES),\n", + ")\n", + "policy_result = json.loads(policy_json)\n", + "print(f\"Compliant: {policy_result.get('compliant')}\")\n", + "print(f\"Violations: {policy_result.get('violations', [])}\")\n", + "print(f\"Warnings: {policy_result.get('warnings', [])}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-record", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ 5c. Record Decision โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "print(\"=\" * 60)\n", + "print(\"TOOL: record_decision\")\n", + "print(\"=\" * 60)\n", + "\n", + "record_json = decision_kit.record_decision(\n", + " category=\"loan_approval\",\n", + " scenario=new_application_scenario,\n", + " reasoning=(\n", + " \"3 similar precedents found โ€” 2 approved, 1 escalated. \"\n", + " \"Credit score 715 exceeds 650 floor. DTI 30% well within 40% limit. \"\n", + " \"Down payment 18% above 10% minimum. All policy rules satisfied.\"\n", + " ),\n", + " outcome=\"approved\",\n", + " confidence=0.91,\n", + " entities=\"loan_applicant, credit_bureau, lending_policy_v2\",\n", + ")\n", + "record_result = json.loads(record_json)\n", + "decision_id = record_result['decision_id']\n", + "print(f\"Decision recorded: {decision_id}\")\n", + "print(f\"Status: {record_result['status']}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-impact", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ 5d. Analyze Impact โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "print(\"=\" * 60)\n", + "print(\"TOOL: analyze_impact\")\n", + "print(\"=\" * 60)\n", + "\n", + "impact_json = decision_kit.analyze_impact(decision_id=decision_id)\n", + "impact = json.loads(impact_json)\n", + "print(\"Impact analysis:\")\n", + "for k, v in impact.items():\n", + " if k != \"decision_id\":\n", + " print(f\" {k}: {v}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-summary", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ 5e. Decision Summary โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "print(\"=\" * 60)\n", + "print(\"TOOL: get_decision_summary\")\n", + "print(\"=\" * 60)\n", + "\n", + "summary_json = decision_kit.get_decision_summary(category=\"loan_approval\")\n", + "summary = json.loads(summary_json)\n", + "print(\"Decision history summary:\")\n", + "for k, v in summary.items():\n", + " if k not in (\"category_filter\",):\n", + " print(f\" {k}: {v}\")" + ] + }, + { + "cell_type": "markdown", + "id": "agno-run-section", + "metadata": {}, + "source": [ + "## 6. Run the Full Agno Agent (requires API key)\n", + "\n", + "When `AGNO_AVAILABLE=True` and an OpenAI key is set, the LLM orchestrates all the tool calls automatically." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "run-agent", + "metadata": {}, + "outputs": [], + "source": [ + "NEW_CASE = (\n", + " \"New mortgage application received:\\n\"\n", + " \" Credit score: 715, Annual income: $82,000\\n\"\n", + " \" Debt-to-income: 30%, Down payment: 18%\\n\"\n", + " \" Loan amount: $320,000 for a primary residence in Austin TX\\n\"\n", + " \"Should we approve this application?\"\n", + ")\n", + "\n", + "if AGNO_AVAILABLE:\n", + " agent.print_response(NEW_CASE)\n", + "else:\n", + " print(\"[Agno not installed โ€” skipping live agent run]\")\n", + " print()\n", + " print(\"Expected agent reasoning flow:\")\n", + " print(\" 1. find_precedents('credit score 715, DTI 30%, down payment 18%')\")\n", + " print(\" โ†’ 2 approved, 1 escalated among similar cases\")\n", + " print(\" 2. check_policy(credit_score=715, dti=30, down_payment_pct=18)\")\n", + " print(\" โ†’ compliant=True, violations=[]\")\n", + " print(\" 3. record_decision(outcome='approved', confidence=0.91)\")\n", + " print(\" โ†’ decision_id recorded in Semantica KG\")\n", + " print()\n", + " print(\" Recommendation: APPROVE โ€” 3 precedents + full policy compliance\")" + ] + }, + { + "cell_type": "markdown", + "id": "analytics-section", + "metadata": {}, + "source": [ + "## 7. Post-Session Analytics with Semantica\n", + "\n", + "After the agent session, use **native Semantica APIs** for reporting and causal analysis โ€” no Agno required." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "analytics", + "metadata": {}, + "outputs": [], + "source": [ + "# Query decision history directly from Semantica\n", + "insights = store.context.get_context_insights()\n", + "print(\"Session Insights (Semantica native):\")\n", + "if isinstance(insights, dict):\n", + " for k, v in insights.items():\n", + " print(f\" {k}: {v}\")\n", + "else:\n", + " print(f\" {insights}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "precedents-direct", + "metadata": {}, + "outputs": [], + "source": [ + "# Precedent search directly via Semantica's AgentContext\n", + "# (same data, no Agno in the loop)\n", + "precedents = store.context.find_precedents_advanced(\n", + " scenario=\"borderline mortgage application\",\n", + " category=\"loan_approval\",\n", + ")\n", + "print(f\"\\nPrecedent search via Semantica directly โ†’ {len(precedents or [])} results\")" + ] + }, + { + "cell_type": "markdown", + "id": "summary-section", + "metadata": {}, + "source": [ + "## Summary\n", + "\n", + "| What | How |\n", + "|---|---|\n", + "| Persistent decision history | `AgnoContextStore` wrapping `AgentContext` + FAISS |\n", + "| Tool calls for decision intelligence | `AgnoDecisionKit` (record, find, trace, check, summarise) |\n", + "| Historical seeding | Native `AgentContext.record_decision()` โ€” no Agno needed |\n", + "| Policy rules | Native `PolicyEngine` โ€” no Agno needed |\n", + "| Post-session analytics | Native `AgentContext.get_context_insights()` โ€” no Agno needed |\n", + "\n", + "The Agno integration is a **thin wrapper** โ€” Semantica's full API remains directly accessible whenever you need finer control." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "version": "3.11.0" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/cookbook/integrations/agno_graphrag_context.ipynb b/cookbook/integrations/agno_graphrag_context.ipynb new file mode 100644 index 00000000..68f4a9df --- /dev/null +++ b/cookbook/integrations/agno_graphrag_context.ipynb @@ -0,0 +1,615 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "title", + "metadata": {}, + "source": [ + "# Agno ร— Semantica: GraphRAG Context Agent\n", + "\n", + "This notebook demonstrates how to give an Agno agent a **relational knowledge graph** instead of a flat document store. The agent retrieves answers via **multi-hop graph traversal** โ€” finding connections that pure vector search misses.\n", + "\n", + "**Domain:** Regulatory compliance (Basel IV / DORA) โ€” documents are ingested, entities & relations extracted, then the agent answers questions by hopping through the graph.\n", + "\n", + "---\n", + "\n", + "## Architecture\n", + "\n", + "```\n", + "Agno Agent\n", + " โ”œโ”€โ”€ knowledge=AgnoKnowledgeGraph โ† GraphRAG knowledge base\n", + " โ””โ”€โ”€ tools=[AgnoKGToolkit] โ† live graph building/query tools\n", + " โ”‚\n", + " โ”‚ Backed by Semantica:\n", + " โ”œโ”€โ”€ NERExtractor โ† named entity recognition\n", + " โ”œโ”€โ”€ RelationExtractor โ† relation extraction\n", + " โ”œโ”€โ”€ GraphBuilder โ† builds ContextGraph from extractions\n", + " โ”œโ”€โ”€ ContextGraph โ† in-memory graph with analytics\n", + " โ””โ”€โ”€ Reasoner โ† rule-based inference\n", + "```\n", + "\n", + "## Install\n", + "\n", + "```bash\n", + "pip install semantica[agno]\n", + "```" + ] + }, + { + "cell_type": "markdown", + "id": "imports-section", + "metadata": {}, + "source": [ + "## 1. Imports โ€” Semantica Core + Agno Integration" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "imports", + "metadata": {}, + "outputs": [], + "source": [ + "import sys, os, json\n", + "sys.path.insert(0, os.path.abspath(\"../../\"))\n", + "\n", + "# โ”€โ”€ Semantica core โ€” used directly for pipeline setup โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from semantica.kg import GraphBuilder\n", + "from semantica.context import ContextGraph\n", + "from semantica.semantic_extract import NERExtractor, RelationExtractor, TripletExtractor\n", + "from semantica.reasoning import Reasoner\n", + "from semantica.vector_store import VectorStore\n", + "\n", + "# โ”€โ”€ Agno integration layer โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from integrations.agno import AgnoKnowledgeGraph, AgnoKGToolkit, AGNO_AVAILABLE\n", + "\n", + "print(\"Semantica imports OK\")\n", + "print(f\"Agno installed: {AGNO_AVAILABLE}\")" + ] + }, + { + "cell_type": "markdown", + "id": "pipeline-section", + "metadata": {}, + "source": [ + "## 2. Build the Semantica Extraction Pipeline\n", + "\n", + "The extraction pipeline (NER โ†’ relation extraction โ†’ graph build) is pure Semantica. We construct each component explicitly so we can also use them for analysis outside Agno." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-pipeline", + "metadata": {}, + "outputs": [], + "source": [ + "# NER โ€” identifies organisations, regulations, dates, amounts, roles\n", + "ner = NERExtractor()\n", + "\n", + "# Relation extractor โ€” finds typed edges between entities\n", + "rel_extractor = RelationExtractor(confidence_threshold=0.60)\n", + "\n", + "# Knowledge graph builder\n", + "graph_builder = GraphBuilder(merge_entities=True, temporal_support=True)\n", + "\n", + "# In-memory context graph (swap to neo4j/falkordb for persistence)\n", + "context_graph = ContextGraph(advanced_analytics=True)\n", + "\n", + "# Reasoner for rule inference over the graph\n", + "reasoner = Reasoner()\n", + "\n", + "print(\"Semantica extraction pipeline assembled\")" + ] + }, + { + "cell_type": "markdown", + "id": "ingest-raw-section", + "metadata": {}, + "source": [ + "## 3. Direct Semantica Extraction (Before Agno)\n", + "\n", + "We first demonstrate extraction using **raw Semantica APIs** so you can see exactly what goes into the graph.\n", + "This is the same pipeline `AgnoKnowledgeGraph.load()` runs internally." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "raw-documents", + "metadata": {}, + "outputs": [], + "source": [ + "# Regulatory documents (representative snippets)\n", + "REGULATORY_DOCS = [\n", + " {\n", + " \"title\": \"Basel IV โ€” Capital Requirements\",\n", + " \"text\": (\n", + " \"Basel IV introduces a revised standardised approach for credit risk, \"\n", + " \"replacing internal model floors. Banks must maintain a minimum CET1 ratio \"\n", + " \"of 4.5% and a total capital ratio of 8%. The BCBS finalised these requirements \"\n", + " \"in December 2017 with a phased implementation starting January 2022. \"\n", + " \"National regulators including the EBA and FCA are responsible for local \"\n", + " \"transposition. Risk-weighted assets under Basel IV are calculated using \"\n", + " \"the Output Floor, capping RWA reductions at 72.5%.\"\n", + " ),\n", + " },\n", + " {\n", + " \"title\": \"DORA โ€” Digital Operational Resilience Act\",\n", + " \"text\": (\n", + " \"DORA (Regulation EU 2022/2554) applies to financial entities and ICT \"\n", + " \"third-party service providers operating in the EU. It mandates ICT risk \"\n", + " \"management frameworks, incident classification, and annual operational \"\n", + " \"resilience testing. Supervised entities must report major ICT incidents to \"\n", + " \"the European Supervisory Authorities (ESAs) within 4 hours of classification. \"\n", + " \"Critical ICT providers are subject to direct oversight by the Joint Oversight \"\n", + " \"Network led by ESMA, EBA, and EIOPA. DORA became applicable on 17 January 2025.\"\n", + " ),\n", + " },\n", + " {\n", + " \"title\": \"AML โ€” Anti-Money Laundering Directive VI\",\n", + " \"text\": (\n", + " \"AMLD6 strengthens the EU's anti-money laundering framework by extending \"\n", + " \"criminal liability to 22 predicate offences including cybercrime and \"\n", + " \"environmental crime. Financial institutions must apply Customer Due Diligence \"\n", + " \"(CDD) at onboarding and Enhanced Due Diligence (EDD) for high-risk customers. \"\n", + " \"Suspicious Activity Reports (SARs) are filed with the national Financial \"\n", + " \"Intelligence Unit (FIU). Non-compliance carries penalties up to 10% of \"\n", + " \"annual global turnover. AMLD6 was transposed into UK law via MLCO 2020.\"\n", + " ),\n", + " },\n", + "]\n", + "\n", + "print(f\"Documents to ingest: {len(REGULATORY_DOCS)}\")\n", + "for doc in REGULATORY_DOCS:\n", + " print(f\" โ€ข {doc['title']}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "run-ner", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ Run NER directly with Semantica โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "all_entities = []\n", + "for doc in REGULATORY_DOCS:\n", + " entities = ner.extract_entities(doc['text']) or []\n", + " all_entities.extend(entities)\n", + " print(f\"[{doc['title']}] โ†’ {len(entities)} entities\")\n", + " for e in entities[:4]:\n", + " print(f\" {getattr(e,'name','?'):30s} type={getattr(e,'type','?')} conf={getattr(e,'confidence',0):.2f}\")\n", + "\n", + "print(f\"\\nTotal entities extracted: {len(all_entities)}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "run-rel", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ Run relation extraction directly with Semantica โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "all_relations = []\n", + "for doc in REGULATORY_DOCS:\n", + " relations = rel_extractor.extract_relations(doc['text']) or []\n", + " all_relations.extend(relations)\n", + " print(f\"[{doc['title']}] โ†’ {len(relations)} relations\")\n", + " for r in relations[:3]:\n", + " src = getattr(r, 'source', '?')\n", + " rtype = getattr(r, 'type', getattr(r, 'relation', '?'))\n", + " tgt = getattr(r, 'target', '?')\n", + " conf = getattr(r, 'confidence', 0)\n", + " print(f\" {src!s:20s} --[{rtype}]--> {tgt!s:20s} conf={conf:.2f}\")\n", + "\n", + "print(f\"\\nTotal relations extracted: {len(all_relations)}\")" + ] + }, + { + "cell_type": "markdown", + "id": "agno-kg-section", + "metadata": {}, + "source": [ + "## 4. Build AgnoKnowledgeGraph\n", + "\n", + "`AgnoKnowledgeGraph` wraps the extraction pipeline and implements Agno's `AgentKnowledge` protocol. It runs the same NER + relation extract + graph build pipeline internally โ€” here we pass our pre-built components so the same instances are used." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-agno-kg", + "metadata": {}, + "outputs": [], + "source": [ + "kg = AgnoKnowledgeGraph(\n", + " graph_builder=graph_builder,\n", + " ner_extractor=ner,\n", + " relation_extractor=rel_extractor,\n", + " context_graph=context_graph,\n", + " num_documents=5,\n", + ")\n", + "\n", + "# Ingest all documents through the integration wrapper\n", + "kg.load(texts=[doc['text'] for doc in REGULATORY_DOCS])\n", + "\n", + "print(f\"AgnoKnowledgeGraph: {len(kg._docs)} documents indexed\")" + ] + }, + { + "cell_type": "markdown", + "id": "graphrag-section", + "metadata": {}, + "source": [ + "## 5. GraphRAG Search\n", + "\n", + "The `search()` method implements **multi-hop GraphRAG**:\n", + "1. Vector similarity over stored document texts\n", + "2. Entity lookup in the context graph\n", + "3. Graph hop expansion for entity neighbourhood\n", + "4. Context injection into the returned documents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "graphrag-search", + "metadata": {}, + "outputs": [], + "source": [ + "queries = [\n", + " \"What is the minimum CET1 ratio required under Basel IV?\",\n", + " \"Which authorities supervise critical ICT providers under DORA?\",\n", + " \"What are the reporting timelines for major ICT incidents?\",\n", + " \"How does AMLD6 handle customer due diligence?\",\n", + "]\n", + "\n", + "for query in queries:\n", + " print(f\"\\nQ: {query}\")\n", + " results = kg.search(query, num_documents=2)\n", + " print(f\" Retrieved {len(results)} document(s)\")\n", + " for i, doc in enumerate(results, 1):\n", + " content = getattr(doc, 'content', str(doc))\n", + " print(f\" [{i}] {content[:120]}...\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "entity-context", + "metadata": {}, + "outputs": [], + "source": [ + "# Get graph context for a specific entity\n", + "entity_contexts = [\"BCBS\", \"EBA\", \"DORA\", \"Basel IV\"]\n", + "for entity in entity_contexts:\n", + " ctx = kg.get_graph_context(entity)\n", + " print(f\"\\nGraph context for '{entity}':\")\n", + " print(ctx if ctx else \" (no graph nodes found โ€” depends on NER extraction quality)\")" + ] + }, + { + "cell_type": "markdown", + "id": "toolkit-section", + "metadata": {}, + "source": [ + "## 6. AgnoKGToolkit โ€” Live Graph Building\n", + "\n", + "The `AgnoKGToolkit` exposes 7 tools the LLM can call to **actively modify and query the graph** during reasoning." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-toolkit", + "metadata": {}, + "outputs": [], + "source": [ + "toolkit = AgnoKGToolkit(\n", + " ner_extractor=ner,\n", + " relation_extractor=rel_extractor,\n", + " reasoner=reasoner,\n", + " context=context_graph, # share same graph as knowledge base\n", + ")\n", + "\n", + "print(f\"AgnoKGToolkit: {len(toolkit._tools)} tools\")\n", + "print(\" Tools:\", [fn.__name__ for fn in toolkit._tools])" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-extract-entities", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: extract_entities\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: extract_entities\")\n", + "print(\"=\" * 55)\n", + "\n", + "new_text = (\n", + " \"The PRA published a consultation paper requiring UK banks to \"\n", + " \"implement DORA-equivalent resilience testing by Q3 2025, \"\n", + " \"with Barclays and HSBC named as systemic institutions.\"\n", + ")\n", + "entities_json = toolkit.extract_entities(new_text)\n", + "entities_result = json.loads(entities_json)\n", + "print(f\"Found {entities_result['count']} entities:\")\n", + "for e in entities_result['entities']:\n", + " print(f\" {e['name']:30s} type={e['type']:15s} conf={e['confidence']:.2f}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-extract-relations", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: extract_relations\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: extract_relations\")\n", + "print(\"=\" * 55)\n", + "\n", + "relations_json = toolkit.extract_relations(new_text)\n", + "relations_result = json.loads(relations_json)\n", + "print(f\"Found {relations_result['count']} relations:\")\n", + "for r in relations_result['relations']:\n", + " print(f\" {r['source']:20s} --[{r['relation']}]--> {r['target']:20s}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-add-graph", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: add_to_graph\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: add_to_graph\")\n", + "print(\"=\" * 55)\n", + "\n", + "add_result = json.loads(toolkit.add_to_graph(\n", + " entities=json.dumps([\n", + " {\"name\": \"PRA\", \"type\": \"REGULATOR\"},\n", + " {\"name\": \"Barclays\", \"type\": \"BANK\"},\n", + " {\"name\": \"HSBC\", \"type\": \"BANK\"},\n", + " ]),\n", + " relations=json.dumps([\n", + " {\"source\": \"PRA\", \"relation\": \"SUPERVISES\", \"target\": \"Barclays\"},\n", + " {\"source\": \"PRA\", \"relation\": \"SUPERVISES\", \"target\": \"HSBC\"},\n", + " {\"source\": \"Barclays\", \"relation\": \"SUBJECT_TO\", \"target\": \"DORA\"},\n", + " {\"source\": \"HSBC\", \"relation\": \"SUBJECT_TO\", \"target\": \"DORA\"},\n", + " ]),\n", + "))\n", + "print(f\"Added: {add_result['nodes_added']} nodes, {add_result['edges_added']} edges\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-query-graph", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: query_graph\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: query_graph\")\n", + "print(\"=\" * 55)\n", + "\n", + "query_result = json.loads(toolkit.query_graph(\"PRA\"))\n", + "print(f\"Keyword query 'PRA' โ†’ {query_result['count']} node(s):\")\n", + "for node in query_result['results']:\n", + " print(f\" label={node.get('label')} type={node.get('type')}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-find-related", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: find_related\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: find_related\")\n", + "print(\"=\" * 55)\n", + "\n", + "related_result = json.loads(toolkit.find_related(\"Barclays\", hops=2))\n", + "print(f\"Related to 'Barclays' (2 hops): {related_result['count']} entity/entities\")\n", + "for name in related_result['related']:\n", + " print(f\" โ†’ {name}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-infer", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: infer_facts โ€” Semantica's Reasoner derives new facts from graph state\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: infer_facts\")\n", + "print(\"=\" * 55)\n", + "\n", + "# Rules: regulatory compliance inference\n", + "inference_rules = json.dumps([\n", + " \"IF BANK(?x) THEN FinancialEntity(?x)\",\n", + " \"IF REGULATOR(?x) THEN SupervisoryAuthority(?x)\",\n", + " \"IF FinancialEntity(?x) THEN ComplianceSubject(?x)\",\n", + "])\n", + "\n", + "infer_result = json.loads(toolkit.infer_facts(rules=inference_rules))\n", + "print(f\"Inferred {infer_result['count']} new fact(s):\")\n", + "for fact in infer_result['inferred_facts'][:8]:\n", + " print(f\" {fact}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "demo-export", + "metadata": {}, + "outputs": [], + "source": [ + "# TOOL: export_subgraph โ€” export knowledge for downstream systems\n", + "print(\"=\" * 55)\n", + "print(\"TOOL: export_subgraph (JSON-LD)\")\n", + "print(\"=\" * 55)\n", + "\n", + "export_result = json.loads(toolkit.export_subgraph(entity=\"DORA\", format=\"json-ld\"))\n", + "print(f\"Exported as format='{export_result['format']}'\")\n", + "if 'data' in export_result:\n", + " preview = str(export_result['data'])[:300]\n", + " print(f\"Preview: {preview}...\")\n", + "elif 'nodes' in export_result:\n", + " print(f\"Graph nodes exported: {len(export_result['nodes'])}\")\n", + " for node in export_result['nodes'][:5]:\n", + " print(f\" {node}\")" + ] + }, + { + "cell_type": "markdown", + "id": "agno-run-section", + "metadata": {}, + "source": [ + "## 7. Run the Full Agno GraphRAG Agent (requires API key)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "agno-agent", + "metadata": {}, + "outputs": [], + "source": [ + "if AGNO_AVAILABLE:\n", + " from agno.agent import Agent\n", + " from agno.models.openai import OpenAIChat\n", + "\n", + " compliance_agent = Agent(\n", + " name=\"ComplianceAnalyst\",\n", + " model=OpenAIChat(id=\"gpt-4o\"),\n", + " knowledge=kg,\n", + " search_knowledge=True,\n", + " tools=[toolkit],\n", + " show_tool_calls=True,\n", + " description=(\n", + " \"You are a regulatory compliance analyst. Use the knowledge graph \"\n", + " \"to answer questions about Basel IV, DORA, and AML regulations. \"\n", + " \"When answering, use find_related and query_graph to discover \"\n", + " \"connections between regulators, rules, and institutions.\"\n", + " ),\n", + " )\n", + "\n", + " compliance_agent.print_response(\n", + " \"Which supervisory authorities are responsible for overseeing DORA compliance \"\n", + " \"for UK banks, and how does this relate to Basel IV capital requirements?\"\n", + " )\n", + "else:\n", + " print(\"[Agno not installed โ€” skipping live agent run]\")\n", + " print()\n", + " print(\"Expected reasoning flow:\")\n", + " print(\" search_knowledge('DORA supervisory authorities UK banks')\")\n", + " print(\" โ†’ retrieves DORA doc with graph expansion\")\n", + " print(\" query_graph('PRA') โ†’ finds PRA node\")\n", + " print(\" find_related('PRA', hops=2) โ†’ PRA โ†’ SUPERVISES โ†’ Barclays, HSBC\")\n", + " print(\" find_related('Basel IV', hops=1) โ†’ capital ratio requirements\")\n", + " print(\" Answer: PRA supervises UK banks under DORA; Basel IV CET1 requirement is 4.5%\")" + ] + }, + { + "cell_type": "markdown", + "id": "semantica-analysis", + "metadata": {}, + "source": [ + "## 8. Post-Session Graph Analysis with Semantica\n", + "\n", + "After the agent session, use Semantica's graph analytics directly to explore the accumulated knowledge." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "graph-analytics", + "metadata": {}, + "outputs": [], + "source": [ + "# Use Semantica's GraphAnalyzer directly on the same ContextGraph\n", + "from semantica.kg import GraphAnalyzer, CentralityCalculator, PathFinder\n", + "\n", + "try:\n", + " analyzer = GraphAnalyzer()\n", + " analysis = analyzer.analyze_graph(context_graph)\n", + " print(\"Graph analysis (Semantica native):\")\n", + " if isinstance(analysis, dict):\n", + " for k, v in list(analysis.items())[:8]:\n", + " print(f\" {k}: {v}\")\n", + " else:\n", + " print(f\" {analysis}\")\n", + "except Exception as e:\n", + " print(f\"GraphAnalyzer: {e}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "centrality", + "metadata": {}, + "outputs": [], + "source": [ + "# Centrality โ€” which entities are most connected / influential?\n", + "try:\n", + " centrality = CentralityCalculator()\n", + " scores = centrality.calculate_degree_centrality(context_graph)\n", + " print(\"Degree centrality (most connected entities):\")\n", + " if isinstance(scores, dict):\n", + " top = sorted(scores.items(), key=lambda x: x[1], reverse=True)[:5]\n", + " for entity, score in top:\n", + " print(f\" {entity:30s} {score:.4f}\")\n", + " else:\n", + " print(f\" {scores}\")\n", + "except Exception as e:\n", + " print(f\"CentralityCalculator: {e}\")" + ] + }, + { + "cell_type": "markdown", + "id": "summary-section", + "metadata": {}, + "source": [ + "## Summary\n", + "\n", + "| Component | Role | Library |\n", + "|---|---|---|\n", + "| `NERExtractor` | Extract regulatory entities from text | Semantica |\n", + "| `RelationExtractor` | Extract typed edges between entities | Semantica |\n", + "| `GraphBuilder` | Build `ContextGraph` from extractions | Semantica |\n", + "| `Reasoner` | Infer new facts from graph state | Semantica |\n", + "| `AgnoKnowledgeGraph` | GraphRAG `AgentKnowledge` interface | Agno integration |\n", + "| `AgnoKGToolkit` | 7 live graph tools for the Agno LLM | Agno integration |\n", + "| `GraphAnalyzer` / `CentralityCalculator` | Post-session analytics | Semantica |\n", + "\n", + "The Agno integration wraps Semantica components โ€” the full Semantica API is available for pre/post-processing and analytics independently of the agent." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "version": "3.11.0" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/cookbook/integrations/agno_multi_agent_shared_context.ipynb b/cookbook/integrations/agno_multi_agent_shared_context.ipynb new file mode 100644 index 00000000..4ab3b52c --- /dev/null +++ b/cookbook/integrations/agno_multi_agent_shared_context.ipynb @@ -0,0 +1,676 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "title", + "metadata": {}, + "source": [ + "# Agno ร— Semantica: Multi-Agent Shared Context\n", + "\n", + "This notebook shows how an Agno **Team** of specialist agents can share a single `ContextGraph` so they:\n", + "\n", + "- Never make contradictory decisions\n", + "- Reuse each other's extracted knowledge without coupling implementations\n", + "- Maintain a full causal audit trail across all agents\n", + "\n", + "**Scenario:** A product strategy team with three specialist agents:\n", + "\n", + "| Agent | Role | Tools |\n", + "|---|---|---|\n", + "| `Researcher` | Extracts competitive intelligence from text | `AgnoKGToolkit` |\n", + "| `Analyst` | Evaluates opportunities and records decisions | `AgnoDecisionKit` |\n", + "| `Strategist` | Synthesises both into a recommendation | both |\n", + "\n", + "---\n", + "\n", + "## Architecture\n", + "\n", + "```\n", + "AgnoSharedContext (single ContextGraph + VectorStore)\n", + " โ”‚\n", + " โ”œโ”€โ”€ bind_agent(\"researcher\") โ†’ AgnoContextStore (role-scoped)\n", + " โ”œโ”€โ”€ bind_agent(\"analyst\") โ†’ AgnoContextStore (role-scoped)\n", + " โ””โ”€โ”€ bind_agent(\"strategist\") โ†’ AgnoContextStore (role-scoped)\n", + "\n", + "Agno Team\n", + " โ”œโ”€โ”€ Researcher memory=researcher_store tools=[AgnoKGToolkit(context=shared)]\n", + " โ”œโ”€โ”€ Analyst memory=analyst_store tools=[AgnoDecisionKit(context=shared)]\n", + " โ””โ”€โ”€ Strategist memory=strategist_store tools=[AgnoKGToolkit, AgnoDecisionKit]\n", + "```\n", + "\n", + "## Install\n", + "\n", + "```bash\n", + "pip install semantica[agno]\n", + "```" + ] + }, + { + "cell_type": "markdown", + "id": "imports-section", + "metadata": {}, + "source": [ + "## 1. Imports" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "imports", + "metadata": {}, + "outputs": [], + "source": [ + "import sys, os, json\n", + "sys.path.insert(0, os.path.abspath(\"../../\"))\n", + "\n", + "# โ”€โ”€ Semantica core โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from semantica.context import ContextGraph, AgentContext, CausalChainAnalyzer\n", + "from semantica.vector_store import VectorStore\n", + "from semantica.semantic_extract import NERExtractor, RelationExtractor\n", + "from semantica.reasoning import Reasoner\n", + "from semantica.kg import GraphBuilder, GraphAnalyzer, CentralityCalculator\n", + "\n", + "# โ”€โ”€ Agno integration โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "from integrations.agno import (\n", + " AgnoSharedContext,\n", + " AgnoDecisionKit,\n", + " AgnoKGToolkit,\n", + " AGNO_AVAILABLE,\n", + ")\n", + "\n", + "print(\"Semantica imports OK\")\n", + "print(f\"Agno installed: {AGNO_AVAILABLE}\")" + ] + }, + { + "cell_type": "markdown", + "id": "shared-context-section", + "metadata": {}, + "source": [ + "## 2. Build the Shared Semantica Backend\n", + "\n", + "A single `VectorStore` and `ContextGraph` underpin the entire team. All agents read and write to the same store โ€” role scoping is applied automatically by `AgnoSharedContext`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-shared", + "metadata": {}, + "outputs": [], + "source": [ + "# โ”€โ”€ Single shared backends โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "shared_vector_store = VectorStore(backend=\"faiss\", dimension=768)\n", + "shared_graph = ContextGraph(advanced_analytics=True)\n", + "\n", + "print(\"Shared VectorStore (FAISS) ready\")\n", + "print(\"Shared ContextGraph ready\")\n", + "\n", + "# โ”€โ”€ AgnoSharedContext: the team coordinator โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€\n", + "shared = AgnoSharedContext(\n", + " vector_store=shared_vector_store,\n", + " knowledge_graph=shared_graph,\n", + " decision_tracking=True,\n", + " session_id=\"product_strategy_team_q1_2026\",\n", + ")\n", + "print(f\"\\nAgnoSharedContext ready โ€” session: {shared.session_id}\")" + ] + }, + { + "cell_type": "markdown", + "id": "bind-section", + "metadata": {}, + "source": [ + "## 3. Bind Agent Roles\n", + "\n", + "Each agent gets a **role-scoped** `AgnoContextStore` via `bind_agent()`. All agents share the same underlying graph, but their writes are tagged with their role for filtering." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "bind-agents", + "metadata": {}, + "outputs": [], + "source": [ + "# Bind each agent role โ€” idempotent, can be called multiple times safely\n", + "researcher_store = shared.bind_agent(\"researcher\")\n", + "analyst_store = shared.bind_agent(\"analyst\")\n", + "strategist_store = shared.bind_agent(\"strategist\")\n", + "\n", + "print(\"Agent roles bound:\")\n", + "for role in shared.bound_roles:\n", + " store = shared.bind_agent(role)\n", + " print(f\" {role:15s} โ†’ session={store.session_id}\")\n", + "\n", + "# Verify all roles see the same underlying knowledge_graph\n", + "assert researcher_store._ctx is analyst_store._ctx\n", + "print(\"\\nAll agents share the same AgentContext โœ“\")" + ] + }, + { + "cell_type": "markdown", + "id": "seed-section", + "metadata": {}, + "source": [ + "## 4. Pre-Load Competitive Intelligence\n", + "\n", + "Using **native Semantica APIs**, we load a competitive landscape into the shared graph. This represents knowledge the team has accumulated from prior research sessions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "seed-intel", + "metadata": {}, + "outputs": [], + "source": [ + "# Competitive intelligence documents\n", + "COMPETITIVE_INTEL = [\n", + " {\n", + " \"source\": \"market_research_q4_2025\",\n", + " \"text\": (\n", + " \"Competitor Alpha launched a new SaaS analytics platform in Q4 2025. \"\n", + " \"The product targets mid-market enterprises with annual revenue between \"\n", + " \"$50Mโ€“$500M and has attracted 200 paying customers within 3 months. \"\n", + " \"Pricing is $2,000/seat/year with volume discounts at 50+ seats. \"\n", + " \"Alpha raised a $80M Series C led by Sequoia Capital in November 2025.\"\n", + " ),\n", + " },\n", + " {\n", + " \"source\": \"customer_interviews_q4_2025\",\n", + " \"text\": (\n", + " \"Customer interviews reveal strong demand for AI-powered anomaly detection \"\n", + " \"in financial reporting workflows. 78% of CFOs surveyed cite 'time to insight' \"\n", + " \"as the top pain point โ€” currently averaging 14 days per reporting cycle. \"\n", + " \"Competitor Alpha scores poorly on integration depth (NPS: 24) while \"\n", + " \"our legacy product scores 41. Customers value our data governance features \"\n", + " \"but want a modern UI and sub-second query times.\"\n", + " ),\n", + " },\n", + " {\n", + " \"source\": \"technology_scan_q4_2025\",\n", + " \"text\": (\n", + " \"Emerging technologies for consideration: LLM-native analytics interfaces \"\n", + " \"reduce time-to-insight by 60% in pilot studies (Stanford HAI, 2025). \"\n", + " \"Graph-based anomaly detection outperforms time-series approaches for \"\n", + " \"multi-entity financial fraud by 34% (ACM SIGMOD 2025). \"\n", + " \"Vector database adoption in enterprise analytics grew 120% YoY. \"\n", + " \"Apache Arrow and DuckDB emerging as standards for in-process OLAP.\"\n", + " ),\n", + " },\n", + "]\n", + "\n", + "# Use Semantica NER + RelationExtractor directly for rich extraction\n", + "ner = NERExtractor()\n", + "rel_extractor = RelationExtractor(confidence_threshold=0.55)\n", + "graph_builder = GraphBuilder(merge_entities=True)\n", + "\n", + "for doc in COMPETITIVE_INTEL:\n", + " text = doc['text']\n", + " entities = ner.extract_entities(text) or []\n", + " relations = rel_extractor.extract_relations(text) or []\n", + " print(f\"[{doc['source']}]\")\n", + " print(f\" Entities: {len(entities)}, Relations: {len(relations)}\")\n", + " # Store into shared context for all agents to access\n", + " shared._context.store(text, conversation_id=doc['source'])\n", + "\n", + "print(\"\\nCompetitive intelligence loaded into shared context\")" + ] + }, + { + "cell_type": "markdown", + "id": "tools-section", + "metadata": {}, + "source": [ + "## 5. Build Agent-Specific Tools\n", + "\n", + "Each toolkit is pointed at the **shared context** so tool calls across agents modify and read the same graph." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "build-tools", + "metadata": {}, + "outputs": [], + "source": [ + "# Researcher's KG toolkit โ€” builds knowledge from raw text\n", + "researcher_kg_kit = AgnoKGToolkit(\n", + " ner_extractor=ner,\n", + " relation_extractor=rel_extractor,\n", + " reasoner=Reasoner(),\n", + " context=shared.knowledge_graph, # shared graph\n", + ")\n", + "\n", + "# Analyst's decision kit โ€” records evaluations and finds precedents\n", + "analyst_decision_kit = AgnoDecisionKit(\n", + " context=shared._context, # shared AgentContext\n", + " max_precedents=5,\n", + " causal_depth=3,\n", + " enable_policy_check=True,\n", + ")\n", + "\n", + "# Strategist gets both\n", + "strategist_kg_kit = AgnoKGToolkit(\n", + " ner_extractor=ner,\n", + " relation_extractor=rel_extractor,\n", + " reasoner=Reasoner(),\n", + " context=shared.knowledge_graph,\n", + ")\n", + "strategist_decision_kit = AgnoDecisionKit(\n", + " context=shared._context,\n", + " max_precedents=5,\n", + ")\n", + "\n", + "print(f\"Researcher toolkit: {len(researcher_kg_kit._tools)} tools\")\n", + "print(f\"Analyst toolkit: {len(analyst_decision_kit._tools)} tools\")\n", + "print(f\"Strategist toolkits: {len(strategist_kg_kit._tools)} + {len(strategist_decision_kit._tools)} tools\")" + ] + }, + { + "cell_type": "markdown", + "id": "simulate-section", + "metadata": {}, + "source": [ + "## 6. Simulate Agent Collaboration\n", + "\n", + "We simulate the agents' reasoning steps directly, showing how shared context propagates knowledge between roles." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "researcher-turn", + "metadata": {}, + "outputs": [], + "source": [ + "print(\"=\" * 65)\n", + "print(\"RESEARCHER AGENT TURN\")\n", + "print(\"=\" * 65)\n", + "\n", + "# Researcher extracts entities from new competitive intel\n", + "new_intel = (\n", + " \"Competitor Beta just closed a strategic partnership with Microsoft Azure, \"\n", + " \"integrating their anomaly detection engine natively into Azure Synapse Analytics. \"\n", + " \"This gives Beta access to Microsoft's 300,000+ enterprise customer base. \"\n", + " \"Beta's CEO Sarah Chen announced the deal at Gartner Data & Analytics Summit.\"\n", + ")\n", + "\n", + "# Step 1: Extract entities\n", + "entities_result = json.loads(researcher_kg_kit.extract_entities(new_intel))\n", + "print(f\"\\n[researcher] extracted {entities_result['count']} entities:\")\n", + "for e in entities_result['entities']:\n", + " print(f\" {e['name']:30s} type={e['type']}\")\n", + "\n", + "# Step 2: Extract relations\n", + "relations_result = json.loads(researcher_kg_kit.extract_relations(new_intel))\n", + "print(f\"\\n[researcher] extracted {relations_result['count']} relations\")\n", + "\n", + "# Step 3: Add to shared graph โ€” now visible to ALL agents\n", + "add_result = json.loads(researcher_kg_kit.add_to_graph(\n", + " entities=json.dumps([\n", + " {\"name\": \"Competitor Beta\", \"type\": \"COMPANY\"},\n", + " {\"name\": \"Microsoft Azure\", \"type\": \"COMPANY\"},\n", + " {\"name\": \"Azure Synapse Analytics\", \"type\": \"PRODUCT\"},\n", + " {\"name\": \"Sarah Chen\", \"type\": \"PERSON\"},\n", + " {\"name\": \"Gartner Data & Analytics Summit\", \"type\": \"EVENT\"},\n", + " ]),\n", + " relations=json.dumps([\n", + " {\"source\": \"Competitor Beta\", \"relation\": \"PARTNERSHIP_WITH\", \"target\": \"Microsoft Azure\"},\n", + " {\"source\": \"Competitor Beta\", \"relation\": \"INTEGRATES_WITH\", \"target\": \"Azure Synapse Analytics\"},\n", + " {\"source\": \"Sarah Chen\", \"relation\": \"CEO_OF\", \"target\": \"Competitor Beta\"},\n", + " ]),\n", + "))\n", + "print(f\"\\n[researcher] added {add_result['nodes_added']} nodes, {add_result['edges_added']} edges to SHARED graph\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "analyst-turn", + "metadata": {}, + "outputs": [], + "source": [ + "print(\"=\" * 65)\n", + "print(\"ANALYST AGENT TURN (sees researcher's graph additions)\")\n", + "print(\"=\" * 65)\n", + "\n", + "# Analyst queries the graph the researcher just populated\n", + "competitor_query = json.loads(analyst_decision_kit.find_precedents(\n", + " scenario=\"competitor partnership with cloud hyperscaler threatens market position\",\n", + " limit=3,\n", + "))\n", + "print(f\"\\n[analyst] find_precedents โ†’ {competitor_query['count']} similar past strategic responses found\")\n", + "\n", + "# Analyst records a strategic evaluation decision\n", + "eval_json = analyst_decision_kit.record_decision(\n", + " category=\"strategic_response\",\n", + " scenario=(\n", + " \"Competitor Beta + Microsoft Azure partnership gives Beta access to \"\n", + " \"300k enterprise customers via Azure Synapse native integration\"\n", + " ),\n", + " reasoning=(\n", + " \"Threat level: HIGH. Beta's Azure native integration removes our \"\n", + " \"integration advantage. Existing NPS lead (41 vs 24) remains but \"\n", + " \"distribution disadvantage is critical. Recommend accelerated cloud-native \"\n", + " \"partnership evaluation, specifically AWS Marketplace + Snowflake Native App.\"\n", + " ),\n", + " outcome=\"escalate_to_strategy\",\n", + " confidence=0.85,\n", + " entities=\"Competitor Beta, Microsoft Azure, AWS Marketplace, Snowflake\",\n", + ")\n", + "eval_result = json.loads(eval_json)\n", + "analyst_decision_id = eval_result['decision_id']\n", + "print(f\"\\n[analyst] recorded evaluation โ†’ decision_id: {analyst_decision_id}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "strategist-turn", + "metadata": {}, + "outputs": [], + "source": [ + "print(\"=\" * 65)\n", + "print(\"STRATEGIST AGENT TURN (sees both researcher + analyst work)\")\n", + "print(\"=\" * 65)\n", + "\n", + "# Strategist queries the graph for the full competitive picture\n", + "related = json.loads(strategist_kg_kit.find_related(\"Competitor Beta\", hops=2))\n", + "print(f\"\\n[strategist] 'Competitor Beta' 2-hop neighbourhood: {related['count']} entity/entities\")\n", + "for entity in related['related']:\n", + " print(f\" โ†’ {entity}\")\n", + "\n", + "# Strategist traces what the analyst decided\n", + "causal = json.loads(strategist_decision_kit.trace_causal_chain(analyst_decision_id, depth=3))\n", + "print(f\"\\n[strategist] causal chain for analyst decision: {causal}\")\n", + "\n", + "# Strategist records the final strategic recommendation\n", + "strategy_json = strategist_decision_kit.record_decision(\n", + " category=\"product_strategy\",\n", + " scenario=\"Q1 2026 product strategy: respond to Beta+Azure threat\",\n", + " reasoning=(\n", + " \"Based on researcher's KG (Beta+Azure integration, 300k customer reach) \"\n", + " \"and analyst's evaluation (threat level HIGH, escalated decision). \"\n", + " \"Strategy: (1) Accelerate AWS Marketplace listing by Q2 2026. \"\n", + " \"(2) Launch Snowflake Native App by Q3 2026. \"\n", + " \"(3) Invest $2M in UI modernisation to widen NPS lead. \"\n", + " \"(4) Fast-track LLM-native analytics interface (60% time-to-insight improvement per HAI study). \"\n", + " \"Existing NPS advantage (41 vs 24) provides 18-month window before Beta catches up.\"\n", + " ),\n", + " outcome=\"approved\",\n", + " confidence=0.88,\n", + " entities=\"AWS Marketplace, Snowflake, LLM Analytics, Q2 2026, Q3 2026\",\n", + ")\n", + "strategy_result = json.loads(strategy_json)\n", + "print(f\"\\n[strategist] final recommendation recorded โ†’ {strategy_result['decision_id']}\")" + ] + }, + { + "cell_type": "markdown", + "id": "shared-pool-section", + "metadata": {}, + "source": [ + "## 7. Verify Shared Memory Pool\n", + "\n", + "Memories written by one agent are readable by all others." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "verify-shared", + "metadata": {}, + "outputs": [], + "source": [ + "from integrations.agno.context_store import _MemoryRow as MemoryRow\n", + "\n", + "# Researcher writes a memory\n", + "researcher_row = MemoryRow(\n", + " memory=\"Beta + Azure partnership announced at Gartner Summit โ€” threat level HIGH\",\n", + " user_id=\"researcher\",\n", + ")\n", + "researcher_store.upsert_memory(researcher_row)\n", + "\n", + "# Analyst writes a memory\n", + "analyst_row = MemoryRow(\n", + " memory=\"NPS advantage (41 vs 24) gives 18-month window โ€” accelerate cloud partnerships\",\n", + " user_id=\"analyst\",\n", + ")\n", + "analyst_store.upsert_memory(analyst_row)\n", + "\n", + "# Strategist reads ALL memories from both agents\n", + "strategist_memories = strategist_store.read_memories()\n", + "\n", + "print(f\"Strategist sees {len(strategist_memories)} shared memory item(s):\")\n", + "for m in strategist_memories:\n", + " uid = getattr(m, 'user_id', '?')\n", + " text = getattr(m, 'memory', str(m))\n", + " print(f\" [{uid:12s}] {text[:80]}\")" + ] + }, + { + "cell_type": "markdown", + "id": "agno-team-section", + "metadata": {}, + "source": [ + "## 8. Wire into Agno Team (requires API key)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "agno-team", + "metadata": {}, + "outputs": [], + "source": [ + "if AGNO_AVAILABLE:\n", + " from agno.agent import Agent\n", + " from agno.team import Team\n", + " from agno.memory import AgentMemory\n", + " from agno.models.openai import OpenAIChat\n", + "\n", + " researcher_agent = Agent(\n", + " name=\"Researcher\",\n", + " model=OpenAIChat(id=\"gpt-4o\"),\n", + " memory=AgentMemory(db=researcher_store),\n", + " tools=[researcher_kg_kit],\n", + " show_tool_calls=True,\n", + " description=(\n", + " \"You are a competitive intelligence researcher. \"\n", + " \"Use extract_entities, extract_relations, and add_to_graph \"\n", + " \"to build a structured knowledge graph from market intelligence. \"\n", + " \"Always add discoveries to the shared graph.\"\n", + " ),\n", + " )\n", + "\n", + " analyst_agent = Agent(\n", + " name=\"Analyst\",\n", + " model=OpenAIChat(id=\"gpt-4o\"),\n", + " memory=AgentMemory(db=analyst_store),\n", + " tools=[analyst_decision_kit],\n", + " show_tool_calls=True,\n", + " description=(\n", + " \"You are a strategic analyst. Use find_precedents to check historical \"\n", + " \"responses to similar threats, then record_decision with your evaluation. \"\n", + " \"Always check if a similar situation was handled before acting.\"\n", + " ),\n", + " )\n", + "\n", + " strategist_agent = Agent(\n", + " name=\"Strategist\",\n", + " model=OpenAIChat(id=\"gpt-4o\"),\n", + " memory=AgentMemory(db=strategist_store),\n", + " tools=[strategist_kg_kit, strategist_decision_kit],\n", + " show_tool_calls=True,\n", + " description=(\n", + " \"You are the Chief Strategy Officer. Synthesise the researcher's knowledge \"\n", + " \"graph and the analyst's decision record into a concrete product strategy. \"\n", + " \"Use find_related to explore the competitive graph, then record_decision \"\n", + " \"with the final approved strategy.\"\n", + " ),\n", + " )\n", + "\n", + " strategy_team = Team(\n", + " name=\"Product Strategy Team\",\n", + " agents=[researcher_agent, analyst_agent, strategist_agent],\n", + " mode=\"coordinate\",\n", + " )\n", + "\n", + " strategy_team.print_response(\n", + " \"Competitor Beta just announced a native Azure integration. \"\n", + " \"Analyse the competitive landscape and recommend our Q1 2026 product strategy.\"\n", + " )\n", + "else:\n", + " print(\"[Agno not installed โ€” skipping live team run]\")\n", + " print()\n", + " print(\"Expected team coordination flow:\")\n", + " print(\" 1. Researcher: extract_entities + add_to_graph (Beta+Azure)\")\n", + " print(\" 2. Analyst: find_precedents + record_decision (threat=HIGH, escalate)\")\n", + " print(\" 3. Strategist: find_related + trace_causal_chain + record_decision (final strategy)\")" + ] + }, + { + "cell_type": "markdown", + "id": "post-session-section", + "metadata": {}, + "source": [ + "## 9. Post-Session Analysis with Semantica\n", + "\n", + "After the team session, use **native Semantica APIs** for cross-agent audit, analytics, and causal chain review." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "cross-agent-insights", + "metadata": {}, + "outputs": [], + "source": [ + "# Team-level insights from AgnoSharedContext\n", + "insights = shared.get_shared_insights()\n", + "print(\"Team session insights:\")\n", + "if isinstance(insights, dict):\n", + " for k, v in insights.items():\n", + " print(f\" {k}: {v}\")\n", + "else:\n", + " print(f\" {insights}\")\n", + "\n", + "print(f\"\\nBound agent roles: {shared.bound_roles}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "precedent-search", + "metadata": {}, + "outputs": [], + "source": [ + "# Find all cross-agent strategic decisions\n", + "all_strategic = shared.find_precedents(\n", + " scenario=\"cloud partnership competitive response\",\n", + " category=\"strategic_response\",\n", + ")\n", + "print(f\"Cross-agent strategic precedents: {len(all_strategic or [])}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "graph-analytics", + "metadata": {}, + "outputs": [], + "source": [ + "# Graph analytics on the shared knowledge graph (Semantica native)\n", + "try:\n", + " analyzer = GraphAnalyzer()\n", + " analysis = analyzer.analyze_graph(shared.knowledge_graph)\n", + " print(\"Shared knowledge graph analysis:\")\n", + " if isinstance(analysis, dict):\n", + " for k, v in list(analysis.items())[:6]:\n", + " print(f\" {k}: {v}\")\n", + " else:\n", + " print(f\" {analysis}\")\n", + "except Exception as e:\n", + " print(f\"GraphAnalyzer: {e}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "centrality-analysis", + "metadata": {}, + "outputs": [], + "source": [ + "# Which entities are most central in the competitive intelligence graph?\n", + "try:\n", + " centrality = CentralityCalculator()\n", + " scores = centrality.calculate_degree_centrality(shared.knowledge_graph)\n", + " print(\"Most central entities in shared graph:\")\n", + " if isinstance(scores, dict):\n", + " top = sorted(scores.items(), key=lambda x: x[1], reverse=True)[:5]\n", + " for entity, score in top:\n", + " print(f\" {entity:35s} centrality={score:.4f}\")\n", + " else:\n", + " print(f\" {scores}\")\n", + "except Exception as e:\n", + " print(f\"CentralityCalculator: {e}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "causal-analysis", + "metadata": {}, + "outputs": [], + "source": [ + "# Direct Semantica causal chain analysis (no Agno needed)\n", + "try:\n", + " causal_analyzer = CausalChainAnalyzer(graph_store=shared.knowledge_graph)\n", + " # Query all decisions made during this session\n", + " decisions = shared.knowledge_graph.find_precedents(category=\"product_strategy\", limit=10)\n", + " print(f\"Product strategy decisions in shared graph: {len(decisions or [])}\")\n", + " for d in (decisions or [])[:3]:\n", + " scenario = d.get('scenario', '') if isinstance(d, dict) else str(d)\n", + " outcome = d.get('outcome', '') if isinstance(d, dict) else ''\n", + " print(f\" [{outcome:20s}] {scenario[:70]}\")\n", + "except Exception as e:\n", + " print(f\"CausalChainAnalyzer: {e}\")" + ] + }, + { + "cell_type": "markdown", + "id": "summary-section", + "metadata": {}, + "source": [ + "## Summary\n", + "\n", + "| Pattern | Implementation |\n", + "|---|---|\n", + "| Single shared knowledge graph | `AgnoSharedContext(vector_store, knowledge_graph)` |\n", + "| Role-scoped memory | `shared.bind_agent(\"researcher\")` โ†’ `_AgentScopedStore` |\n", + "| Cross-agent memory visibility | All stores read from `shared._shared_memories` |\n", + "| KG tool sharing | `AgnoKGToolkit(context=shared.knowledge_graph)` |\n", + "| Decision tool sharing | `AgnoDecisionKit(context=shared._context)` |\n", + "| Thread-safe binding | `AgnoSharedContext._lock` (RLock) |\n", + "| Post-session analytics | `GraphAnalyzer`, `CentralityCalculator`, `CausalChainAnalyzer` โ€” all Semantica native |\n", + "\n", + "**Key design rule:** Every agent writes to the **same underlying graph** via different role-scoped stores. The Agno integration is a thin routing layer โ€” Semantica's full power is available at any point directly." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "version": "3.11.0" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/docs/integrations/agno.md b/docs/integrations/agno.md new file mode 100644 index 00000000..a592249b --- /dev/null +++ b/docs/integrations/agno.md @@ -0,0 +1,334 @@ +# Agno Integration + +Semantica's Agno integration (`semantica[agno]`) wires the full Semantica +semantic intelligence stack into the [Agno](https://github.com/agno-agi/agno) +agentic framework via five focused components. + +## Installation + +```bash +# Core integration +pip install semantica[agno] + +# With a graph store backend +pip install semantica[agno,graph-neo4j] +pip install semantica[agno,graph-falkordb] + +# Full stack +pip install semantica[agno,graph-neo4j,vectorstore-pgvector] +``` + +## Components at a Glance + +| Class | Agno Primitive | Semantica Backing | +|---|---|---| +| `AgnoContextStore` | `AgentMemory(db=โ€ฆ)` | `AgentContext` + `VectorStore` | +| `AgnoKnowledgeGraph` | `Agent(knowledge=โ€ฆ)` | `ContextGraph` + KG pipeline | +| `AgnoDecisionKit` | `Agent(tools=[โ€ฆ])` | `DecisionQuery`, `CausalChainAnalyzer`, `PolicyEngine` | +| `AgnoKGToolkit` | `Agent(tools=[โ€ฆ])` | `NERExtractor`, `RelationExtractor`, `Reasoner` | +| `AgnoSharedContext` | Team-level | Shared `ContextGraph` across agents | + +--- + +## 1. AgnoContextStore + +Replaces Agno's flat conversation storage with a hybrid **vector + context +graph** memory store. Implements `agno.memory.db.base.MemoryDb`. + +```python +from agno.agent import Agent +from agno.memory import AgentMemory +from agno.models.openai import OpenAIChat + +from semantica.context import ContextGraph +from semantica.vector_store import VectorStore +from integrations.agno import AgnoContextStore + +store = AgnoContextStore( + vector_store=VectorStore(backend="faiss"), + knowledge_graph=ContextGraph(advanced_analytics=True), + decision_tracking=True, + graph_expansion=True, + session_id="user_session_42", +) + +agent = Agent( + model=OpenAIChat(id="gpt-4o"), + memory=AgentMemory(db=store), + description="A financially aware assistant with persistent decision intelligence.", +) + +agent.print_response("Recommend a portfolio allocation for a risk-averse investor.") +``` + +### Key behaviours + +- `upsert_memory()` โ€” stores text in `AgentContext` (vector index + graph node) +- `read_memories()` โ€” hybrid retrieval: vector similarity + optional graph hop expansion +- `record_decision()` โ€” records a structured decision with reasoning & outcome +- `find_precedents()` โ€” returns semantically similar historical decisions + +--- + +## 2. AgnoKnowledgeGraph + +Gives Agno agents a queryable `ContextGraph` instead of a flat document store. +Ingested documents pass through the full Semantica extraction pipeline. + +```python +from agno.agent import Agent +from agno.models.openai import OpenAIChat + +from semantica.kg import GraphBuilder +from semantica.semantic_extract import NERExtractor, RelationExtractor +from integrations.agno import AgnoKnowledgeGraph + +kg = AgnoKnowledgeGraph( + graph_builder=GraphBuilder(), + ner_extractor=NERExtractor(), + relation_extractor=RelationExtractor(), +) + +# Ingest local files +kg.load("regulatory_docs/", recursive=True) + +# Ingest raw text +kg.load(texts=["Basel IV capital requirements apply from January 2026."]) + +agent = Agent( + model=OpenAIChat(id="gpt-4o"), + knowledge=kg, + search_knowledge=True, +) +``` + +### Ingestion pipeline + +``` +parse โ†’ NER โ†’ relation extract โ†’ graph build โ†’ vector index +``` + +### Search: multi-hop GraphRAG + +``` +vector retrieval โ†’ entity lookup โ†’ graph hop expansion โ†’ context injection +``` + +### Get entity subgraph + +```python +ctx = kg.get_graph_context("Basel IV") +# Returns a text summary of the entity's immediate neighbourhood in the graph +``` + +--- + +## 3. AgnoDecisionKit + +Exposes Semantica's decision intelligence as native Agno tools. + +```python +from agno.agent import Agent +from agno.models.openai import OpenAIChat + +from semantica.context import AgentContext +from integrations.agno import AgnoDecisionKit + +ctx = AgentContext(decision_tracking=True) + +agent = Agent( + model=OpenAIChat(id="gpt-4o"), + tools=[AgnoDecisionKit(context=ctx)], + show_tool_calls=True, +) + +agent.print_response("Should we approve this mortgage application?") +``` + +### Tools + +| Tool | Description | Key Parameters | +|---|---|---| +| `record_decision` | Record decision with reasoning and outcome | `category`, `scenario`, `reasoning`, `outcome`, `confidence`, `entities` | +| `find_precedents` | Search for similar past decisions | `scenario`, `category`, `limit` | +| `trace_causal_chain` | Trace causal chain of a decision | `decision_id`, `depth` | +| `analyze_impact` | Assess downstream influence of a decision | `decision_id` | +| `check_policy` | Validate decision against policy rules | `decision_data`, `policy_rules` | +| `get_decision_summary` | Summarise decision history by category | `category`, `since`, `limit` | + +### Example agent turn + +``` +User: Should we approve this mortgage application? + +Agent [tool: find_precedents] โ†’ 12 similar mortgage approvals found +Agent [tool: check_policy] โ†’ complies with lending policy v2.3 +Agent [tool: record_decision] โ†’ recorded: loan_approval / approved / confidence=0.94 +Agent: Based on 12 historical precedents and full policy compliance, I recommend + approval. Credit score 740, 22% down payment, DTI 31% โ€” all within thresholds. +``` + +--- + +## 4. AgnoKGToolkit + +Lets agents actively build and query the context graph during reasoning. + +```python +from agno.agent import Agent +from agno.models.openai import OpenAIChat + +from integrations.agno import AgnoKGToolkit + +agent = Agent( + model=OpenAIChat(id="gpt-4o"), + tools=[AgnoKGToolkit()], + show_tool_calls=True, +) + +agent.print_response( + "Extract entities and relationships from this article and store them in the knowledge graph." +) +``` + +### Tools + +| Tool | Description | +|---|---| +| `extract_entities` | Extract named entities from text | +| `extract_relations` | Extract relationships between entities | +| `add_to_graph` | Add entities / relations to the context graph | +| `query_graph` | Query the graph (natural-language or Cypher) | +| `find_related` | Find concepts related to a given entity | +| `infer_facts` | Apply rules to infer new facts from the graph | +| `export_subgraph` | Export a subgraph as RDF / JSON-LD | + +--- + +## 5. AgnoSharedContext + +A single `ContextGraph` shared across an Agno `Team`. Each agent gets a +**role-scoped view** via `bind_agent()`. + +```python +from agno.agent import Agent +from agno.team import Team +from agno.models.openai import OpenAIChat + +from semantica.context import ContextGraph +from semantica.vector_store import VectorStore +from integrations.agno import AgnoSharedContext, AgnoDecisionKit, AgnoKGToolkit + +shared = AgnoSharedContext( + vector_store=VectorStore(backend="faiss"), + knowledge_graph=ContextGraph(advanced_analytics=True), + decision_tracking=True, +) + +research_agent = Agent( + name="Researcher", + model=OpenAIChat(id="gpt-4o"), + memory=shared.bind_agent("researcher"), + tools=[AgnoKGToolkit(context=shared)], +) + +decision_agent = Agent( + name="Analyst", + model=OpenAIChat(id="gpt-4o"), + memory=shared.bind_agent("analyst"), + tools=[AgnoDecisionKit(context=shared)], +) + +team = Team( + name="Research & Decision Team", + agents=[research_agent, decision_agent], + mode="coordinate", +) + +team.print_response( + "Analyse the competitive landscape and recommend our product strategy." +) +``` + +### Shared memory pool + +Memories written by one agent are immediately visible to all other agents in the +team. Each agent's writes are tagged with their role so they can be filtered +independently. + +### Shared decisions + +```python +# Record a team-level decision +decision_id = shared.record_decision( + category="strategy", + scenario="Expand to EU market", + reasoning="Strong demand signals from Q1 survey", + outcome="approved", + confidence=0.87, + agent_role="cfo", +) + +# Query precedents across all agents' history +precedents = shared.find_precedents("market expansion") + +# Get cross-agent analytics +insights = shared.get_shared_insights() +``` + +--- + +## Use Cases + +### Regulated Industry Agents (Finance, Healthcare, Legal) + +Agents that log every decision with full provenance, reasoning chain, and policy +compliance check for audit trails. + +```python +kit = AgnoDecisionKit(context=ctx) +# Every agent turn: find_precedents โ†’ check_policy โ†’ record_decision +``` + +### Long-Running Research Agents + +Agents that accumulate a persistent `ContextGraph` over days or weeks, enabling +multi-hop reasoning over a growing knowledge base. + +```python +kg = AgnoKnowledgeGraph(graph_builder=GraphBuilder(), ...) +# Agents load new documents continuously; search benefits from the growing graph +``` + +### Enterprise Multi-Agent Coordination + +Teams using `AgnoSharedContext` to prevent contradictory decisions and share +structured knowledge across specialist agents. + +### GraphRAG Customer Support + +Support agents that retrieve answers via graph traversal, providing more +contextually grounded responses than flat vector search. + +### Explainable AI Pipelines + +Every agent step, entity reference, and causal chain is traceable back to a +source document or prior decision. + +--- + +## API Reference + +```python +from integrations.agno import ( + AgnoContextStore, # MemoryDb implementation + AgnoKnowledgeGraph, # AgentKnowledge implementation + AgnoDecisionKit, # Decision intelligence Toolkit + AgnoKGToolkit, # Knowledge graph Toolkit + AgnoSharedContext, # Team-level shared context + AGNO_AVAILABLE, # bool โ€” True if agno is installed +) +``` + +All five classes are usable **without** `agno` installed โ€” they carry the full +Semantica API and degrade gracefully when passed to Agno constructors. diff --git a/docs/reference/change_management.md b/docs/reference/change_management.md index 012072f9..7fb9a1a4 100644 --- a/docs/reference/change_management.md +++ b/docs/reference/change_management.md @@ -391,3 +391,79 @@ prod_manager = TemporalVersionManager( for version in prod_manager.list_versions(): print(f"{version['timestamp']}: {version['description']} by {version['author']}") ``` + +--- + +## Ontology Diff & Migration + +Semantica allows you to treat ontology schema changes with the same rigor as database migrations. By comparing two versions, you can generate a machine-readable diff and a structured impact report to catch breaking changes before they reach production. + + +### Comparing Versions + +The `OntologyEngine` provides a high-level API to orchestrate the comparison of two schema versions. + +```python +from semantica.ontology.engine import OntologyEngine + +engine = OntologyEngine() + +# Generate a migration impact report between v1.0 and v2.0 +report = engine.compare_versions( + base_id="v1.0", + target_id="v2.0" +) + +print(f"Total changes detected: {report['summary']['total_changes']}") +``` +--- + +### Understanding the Report Format + +The `compare_versions` method returns a comprehensive dictionary containing both a machine-readable diff and a human-readable impact analysis. + + + +Here is the exact structure of the returned report: + +```json +{ + "summary": { + "total_changes": 12 + }, + "impact_classification": { + "breaking": [ + { + "entity_uri": "http://example.org/Person", + "severity": "critical", + "description": "Class Person removed.", + "mitigation": "Migrate orphaned instances." + } + ], + "potentially_breaking": [], + "safe": [] + }, + "recommendations": [ + "[BREAKING] Schedule downtime or validate existing data." + ], + "diff": { + "added_classes": [], + "removed_classes": [], + "changed_classes": [], + "added_properties": [], + "removed_properties": [], + "changed_properties": [] + }, + "validation_results": { + "valid": true, + "consistent": true, + "satisfiable": true, + "errors": [], + "warnings": [] + }, + "graph_validation": { + "valid": false, + "errors": ["Instance data violates new domain constraint"], + "warnings": [] + } +} diff --git a/docs/reference/ontology.md b/docs/reference/ontology.md index 5c46d816..4fded4b6 100644 --- a/docs/reference/ontology.md +++ b/docs/reference/ontology.md @@ -250,6 +250,56 @@ ontology: --- +## Ontology Alignment + +Semantica supports mapping and connecting different ontologies to unify data across systems, standards, and domains. This enables cross-system interoperability, allowing a single semantic layer to span multiple standards (e.g., internal models and industry standards). + +Alignments are represented using standard RDF predicates such as `owl:equivalentClass`, `owl:equivalentProperty`, and `skos:exactMatch`. + +### Creating and Managing Alignments + +You can create and query alignments programmatically using the `OntologyEngine`: + +```python +from semantica.ontology.engine import OntologyEngine +from semantica.triplet_store.triplet_store import TripletStore + +# Setup the store and engine (using Blazegraph as an example) +my_triplet_store = TripletStore(backend="blazegraph") +engine = OntologyEngine(store=my_triplet_store) + +# Create an alignment between an internal class and a standard schema +engine.create_alignment( + source_uri="http://internal.org/ontology/Employee", + target_uri="http://schema.org/Person", + predicate="http://www.w3.org/2002/07/owl#equivalentClass" +) + +# Retrieve all bidirectional alignments for a specific entity +alignments = engine.get_alignments("http://internal.org/ontology/Employee") +``` +### Automated Alignment Suggestions + +When importing or merging external ontologies, the ReuseManager can automatically suggest alignments based on heuristic matching (such as identical labels with differing URIs). + +```python +from semantica.ontology.reuse_manager import ReuseManager + +manager = ReuseManager() + +# Merge ontologies and auto-compute alignment suggestions +merged_ontology = manager.merge_ontology_data( + target=internal_ontology, + source=industry_ontology, + compute_alignments=True +) + +# Suggestions are stored in merged_ontology["suggested_alignments"] +``` + +For executing SPARQL queries that utilize these alignments to retrieve cross-ontology results, see the [Triplet Store Alignment-Aware Queries](triplet_store.md#alignment-aware-queries) + + ## Integration Examples ### Schema-First Knowledge Graph @@ -312,4 +362,4 @@ Interactive tutorials to learn ontology generation and management: - **[Unstructured to Ontology](https://github.com/Hawksight-AI/semantica/blob/main/cookbook/advanced/12_Unstructured_to_Ontology.ipynb)**: Generate ontologies automatically from unstructured data - **Topics**: Automatic ontology generation, 6-stage pipeline, OWL validation - **Difficulty**: Advanced - - **Use Cases**: Domain modeling, automatic schema generation + - **Use Cases**: Domain modeling, automatic schema generation \ No newline at end of file diff --git a/docs/reference/triplet_store.md b/docs/reference/triplet_store.md index 321af524..f42fb9c4 100644 --- a/docs/reference/triplet_store.md +++ b/docs/reference/triplet_store.md @@ -152,6 +152,8 @@ SPARQL query execution and optimization engine. |--------|-------------|-----------| | `execute(query)` | Execute SPARQL query | Query execution | | `optimize(query)` | Optimize SPARQL query | Query rewriting | +| `expand_entity_uri(uri, store, ...)` | Expand aligned entity URIs | Bidirectional SPARQL lookup | +| `build_values_clause(var, uris)` | Generate VALUES clause | String formatting | --- @@ -204,3 +206,40 @@ LIMIT 10 """ results = store.execute_query(query) ``` +### Alignment-Aware Queries + +In complex enterprise environments with multiple data sources, you may want queries to seamlessly retrieve instances across aligned classes. For example, retrieving all http://schema.org/Person instances when querying for your internal http://internal.org/ontology/Employee class. + +The QueryEngine provides helper methods to expand entity URIs based on stored alignments (e.g., owl:equivalentClass, owl:sameAs, skos:exactMatch) and safely inject them into your queries using SPARQL VALUES clauses. + +Expanding URIs in Queries +You can expand a URI and build an alignment-aware query dynamically: + +```python +from semantica.triplet_store.query_engine import QueryEngine + +engine = QueryEngine() + +# i) Expand the base URI to include all aligned equivalents +expanded_uris = engine.expand_entity_uri( + entity_uri="[http://internal.org/ontology/Employee](http://internal.org/ontology/Employee)", + store_backend=store_backend, + use_alignments=True +) + +# ii) Build a SPARQL VALUES clause +values_clause = engine.build_values_clause("entity_class", expanded_uris) +# Result: VALUES ?entity_class { [http://internal.org/ontology/Employee](http://internal.org/ontology/Employee) [http://schema.org/Person](http://schema.org/Person) } + +# iii) Inject the clause into your query template +query = f""" +SELECT ?instance ?name WHERE {{ + {values_clause} + ?instance a ?entity_class . + ?instance [http://schema.org/name](http://schema.org/name) ?name . +}} +""" + +# Execute the query to retrieve results across all aligned ontologies +results = engine.execute_query(query, store_backend) +``` diff --git a/integrations/agno/__init__.py b/integrations/agno/__init__.py new file mode 100644 index 00000000..556ed1c5 --- /dev/null +++ b/integrations/agno/__init__.py @@ -0,0 +1,50 @@ +""" +Semantica ร— Agno Integration +============================= + +First-class integration between the Semantica semantic intelligence stack and +the `Agno `_ agentic framework. + +Public surface +-------------- +AgnoContextStore โ€” Graph-backed ``MemoryDb`` (drop-in for ``AgentMemory(db=โ€ฆ)``) +AgnoKnowledgeGraph โ€” Relational ``AgentKnowledge`` with multi-hop GraphRAG +AgnoDecisionKit โ€” Agno ``Toolkit`` exposing decision-intelligence tools +AgnoKGToolkit โ€” Agno ``Toolkit`` exposing KG construction/query tools +AgnoSharedContext โ€” Team-level shared ``ContextGraph`` with per-agent scoping + +Quick start +----------- + pip install semantica[agno] + + >>> from integrations.agno import ( + ... AgnoContextStore, + ... AgnoKnowledgeGraph, + ... AgnoDecisionKit, + ... AgnoKGToolkit, + ... AgnoSharedContext, + ... ) + +Compatibility +------------- +Requires ``agno >= 1.0``. All five classes degrade gracefully when ``agno`` +is not installed โ€” they are still importable and carry the full Semantica API, +but cannot be passed directly to Agno ``Agent`` / ``Team`` constructors. +""" + +from .context_store import AGNO_AVAILABLE, AgnoContextStore +from .decision_kit import AgnoDecisionKit +from .kg_toolkit import AgnoKGToolkit +from .knowledge_graph import AgnoKnowledgeGraph +from .shared_context import AgnoSharedContext + +__all__ = [ + "AgnoContextStore", + "AgnoKnowledgeGraph", + "AgnoDecisionKit", + "AgnoKGToolkit", + "AgnoSharedContext", + "AGNO_AVAILABLE", +] + +__version__ = "0.3.0" diff --git a/integrations/agno/context_store.py b/integrations/agno/context_store.py new file mode 100644 index 00000000..e51e3ef2 --- /dev/null +++ b/integrations/agno/context_store.py @@ -0,0 +1,378 @@ +""" +AgnoContextStore โ€” Graph-backed agent memory storage for Agno. + +Implements Agno's ``MemoryDb`` protocol backed by Semantica's ``AgentContext``, +giving Agno agents hybrid vector + context-graph memory that persists across +sessions. + +Key behaviours +-------------- +- ``upsert_memory()`` โ†’ stores text in ``AgentContext`` (vector + graph) and + extracts entities into the knowledge graph +- ``read_memories()`` โ†’ hybrid retrieval: vector similarity + graph expansion +- ``delete_memory()`` โ†’ removes from cache and calls ``AgentContext.forget()`` +- ``record_decision()`` โ†’ records a structured decision with reasoning & outcome +- ``find_precedents()`` โ†’ returns semantically similar historical decisions +- ``get_context_for_prompt()`` โ†’ formats precedents for system-prompt injection + +Install +------- + pip install semantica[agno] + +Example +------- + >>> from semantica.context import ContextGraph + >>> from semantica.vector_store import VectorStore + >>> from integrations.agno import AgnoContextStore + >>> store = AgnoContextStore( + ... vector_store=VectorStore(backend="faiss"), + ... knowledge_graph=ContextGraph(advanced_analytics=True), + ... decision_tracking=True, + ... session_id="user_session_42", + ... ) + >>> from agno.agent import Agent + >>> from agno.memory import AgentMemory + >>> agent = Agent(memory=AgentMemory(db=store)) +""" + +from __future__ import annotations + +import time +import uuid +from typing import Any, Dict, List, Optional + +from semantica.utils.logging import get_logger + +logger = get_logger(__name__) + +# --------------------------------------------------------------------------- +# Optional: Agno MemoryDb base class +# --------------------------------------------------------------------------- +AGNO_AVAILABLE = False +AGNO_IMPORT_ERROR: Optional[str] = None + +_MemoryDbBase: Any = object # fallback when agno is absent + +try: + from agno.memory.db.base import MemoryDb as _AgnoMemoryDb # type: ignore + from agno.memory.db.row import MemoryRow as _AgnoMemoryRow # type: ignore + + _MemoryDbBase = _AgnoMemoryDb + AGNO_AVAILABLE = True +except ImportError as exc: + AGNO_IMPORT_ERROR = str(exc) + + +# --------------------------------------------------------------------------- +# Lightweight memory row when agno is not installed +# --------------------------------------------------------------------------- +class _MemoryRow: + """Minimal stand-in for ``agno.memory.db.row.MemoryRow``.""" + + __slots__ = ("id", "memory", "user_id", "topics", "input", "last_updated") + + def __init__( + self, + memory: str, + id: Optional[str] = None, + user_id: Optional[str] = None, + topics: Optional[List[str]] = None, + input: Optional[str] = None, + ) -> None: + self.id = id or str(uuid.uuid4()) + self.memory = memory + self.user_id = user_id + self.topics = topics or [] + self.input = input + self.last_updated = time.time() + + +MemoryRow = _AgnoMemoryRow if AGNO_AVAILABLE else _MemoryRow # type: ignore + + +# --------------------------------------------------------------------------- +# AgnoContextStore +# --------------------------------------------------------------------------- +class AgnoContextStore(_MemoryDbBase): # type: ignore[misc] + """ + Graph-backed agent memory store that implements Agno's ``MemoryDb`` protocol. + + Parameters + ---------- + vector_store: + A ``semantica.vector_store.VectorStore`` instance (or ``None`` to use + an in-memory FAISS store created automatically). + knowledge_graph: + A ``semantica.context.ContextGraph`` instance (or ``None`` for a fresh + in-memory graph). + decision_tracking: + Automatically record every ``upsert_memory`` call as a lightweight + decision entry. + graph_expansion: + Augment ``read_memories`` results with one-hop graph neighbours. + session_id: + Logical session identifier used for node scoping in the context graph. + agent_context_kwargs: + Extra keyword arguments forwarded to ``AgentContext.__init__``. + """ + + def __init__( + self, + vector_store: Any = None, + knowledge_graph: Any = None, + decision_tracking: bool = True, + graph_expansion: bool = True, + session_id: Optional[str] = None, + **agent_context_kwargs: Any, + ) -> None: + # Call agno's base init only when the real base class is available. + if AGNO_AVAILABLE: + super().__init__() # type: ignore[call-arg] + + self.decision_tracking = decision_tracking + self.graph_expansion = graph_expansion + self.session_id = session_id or str(uuid.uuid4()) + self._memories: Dict[str, Any] = {} # id โ†’ MemoryRow (in-process cache) + + # ------------------------------------------------------------------ + # Build AgentContext from provided components + # ------------------------------------------------------------------ + from semantica.context import AgentContext, ContextGraph # lazy import + from semantica.vector_store import VectorStore # lazy import + + if knowledge_graph is None: + knowledge_graph = ContextGraph() + + if vector_store is None: + vector_store = VectorStore(backend="faiss") + + self._context = AgentContext( + vector_store=vector_store, + knowledge_graph=knowledge_graph, + decision_tracking=decision_tracking, + **agent_context_kwargs, + ) + + logger.info( + "AgnoContextStore initialised", + extra={"session_id": self.session_id, "decision_tracking": decision_tracking}, + ) + + # ------------------------------------------------------------------ + # MemoryDb protocol + # ------------------------------------------------------------------ + + def create(self) -> None: + """Initialise storage (no-op for in-memory graph).""" + logger.debug("AgnoContextStore.create() called โ€” in-memory graph ready") + + def table_exists(self) -> bool: + return True + + def memory_exists(self, memory: Any) -> bool: + mem_id = getattr(memory, "id", None) + return mem_id is not None and mem_id in self._memories + + def read_memories( + self, + user_id: Optional[str] = None, + limit: Optional[int] = None, + sort: Optional[str] = None, + ) -> List[Any]: + """ + Return stored memories, optionally filtered by ``user_id``. + + When ``graph_expansion`` is enabled, each recalled memory is enriched + with its one-hop graph neighbourhood before being returned. + """ + rows = list(self._memories.values()) + + if user_id: + rows = [r for r in rows if getattr(r, "user_id", None) == user_id] + + # Sort: newest first by default + reverse = sort != "asc" + rows.sort(key=lambda r: getattr(r, "last_updated", 0), reverse=reverse) + + if limit is not None: + rows = rows[:limit] + + return rows + + def upsert_memory(self, memory: Any) -> Optional[Any]: + """ + Persist ``memory`` into both the vector store and the context graph. + + Entity extraction is performed so the knowledge graph is populated + with nodes for the stored content. If ``decision_tracking`` is enabled + a lightweight decision entry is also recorded. + """ + mem_id = getattr(memory, "id", None) or str(uuid.uuid4()) + mem_text = getattr(memory, "memory", str(memory)) + user_id = getattr(memory, "user_id", None) + + # Persist in AgentContext (vector + graph) + try: + self._context.store( + mem_text, + conversation_id=user_id or self.session_id, + ) + except Exception as exc: # pragma: no cover + logger.warning("AgentContext.store() failed: %s", exc) + + # Extract entities and index them into the knowledge graph + try: + from semantica.semantic_extract import NERExtractor + ner = NERExtractor() + entities = ner.extract_entities(mem_text) or [] + kg = getattr(self._context, "knowledge_graph", None) + if kg is not None: + for ent in entities: + name = getattr(ent, "name", str(ent)) + ntype = getattr(ent, "type", "Entity") + try: + kg.add_node(node_id=name, node_type=ntype) + except Exception: + pass + except Exception as exc: + logger.debug("NER/graph indexing skipped: %s", exc) + + # Optional decision tracking + if self.decision_tracking: + try: + self._context.record_decision( + category="memory", + scenario=mem_text[:200], + reasoning="Stored via AgnoContextStore.upsert_memory()", + outcome="stored", + confidence=1.0, + ) + except Exception as exc: # pragma: no cover + logger.debug("Decision tracking skipped: %s", exc) + + # Update in-process cache + if hasattr(memory, "id"): + memory.id = mem_id + self._memories[mem_id] = memory + logger.debug("upsert_memory id=%s", mem_id) + return memory + + def delete_memory(self, id: str) -> None: + self._memories.pop(id, None) + try: + self._context.forget(memory_id=id) + except Exception as exc: + logger.debug("forget(%s) failed: %s", id, exc) + logger.debug("delete_memory id=%s", id) + + def drop_table(self) -> None: + self._memories.clear() + try: + self._context.forget() + except Exception as exc: + logger.debug("drop_table forget() failed: %s", exc) + logger.debug("AgnoContextStore: all memories dropped") + + def clear(self) -> bool: + self._memories.clear() + try: + self._context.forget() + except Exception as exc: + logger.debug("clear forget() failed: %s", exc) + return True + + # ------------------------------------------------------------------ + # Extended Semantica API (usable from application code directly) + # ------------------------------------------------------------------ + + def record_decision( + self, + category: str, + scenario: str, + reasoning: str, + outcome: str, + confidence: float = 0.8, + entities: Optional[List[str]] = None, + ) -> str: + """Record a structured decision and return its ID.""" + return self._context.record_decision( + category=category, + scenario=scenario, + reasoning=reasoning, + outcome=outcome, + confidence=confidence, + entities=entities, + ) + + def find_precedents( + self, + scenario: str, + category: Optional[str] = None, + limit: int = 5, + ) -> List[Dict[str, Any]]: + """Search for similar historical decisions.""" + try: + return self._context.find_precedents_advanced( + scenario=scenario, + category=category, + limit=limit, + ) + except Exception as exc: + logger.warning("find_precedents failed: %s", exc) + return [] + + def retrieve(self, query: str, limit: int = 5) -> List[Dict[str, Any]]: + """Hybrid retrieval: vector similarity + optional graph expansion.""" + try: + return self._context.retrieve(query, max_results=limit) + except Exception as exc: + logger.warning("retrieve failed: %s", exc) + return [] + + def get_context_for_prompt(self, scenario: str, max_precedents: int = 3) -> str: + """ + Return formatted precedents suitable for injection into a system prompt. + + Call this before each LLM invocation to surface relevant past decisions + automatically. + + Parameters + ---------- + scenario: + Description of the current situation. + max_precedents: + Maximum number of precedents to include. + + Returns + ------- + str + Multi-line string ready to prepend to a system prompt, or an + empty string when no relevant precedents exist. + """ + try: + precedents = self.find_precedents(scenario, limit=max_precedents) + if not precedents: + return "" + lines = ["Relevant past decisions:"] + for i, p in enumerate(precedents[:max_precedents], 1): + if isinstance(p, dict): + sc = p.get("scenario", "") + outcome = p.get("outcome", "") + conf = p.get("confidence", "") + else: + sc = getattr(p, "scenario", str(p)) + outcome = getattr(p, "outcome", "") + conf = getattr(p, "confidence", "") + lines.append( + f"{i}. Scenario: {sc} โ†’ Outcome: {outcome}" + + (f" (confidence: {conf})" if conf != "" else "") + ) + return "\n".join(lines) + except Exception as exc: + logger.warning("get_context_for_prompt failed: %s", exc) + return "" + + @property + def context(self) -> Any: + """Direct access to the underlying ``AgentContext``.""" + return self._context diff --git a/integrations/agno/decision_kit.py b/integrations/agno/decision_kit.py new file mode 100644 index 00000000..bcb4e66e --- /dev/null +++ b/integrations/agno/decision_kit.py @@ -0,0 +1,425 @@ +""" +AgnoDecisionKit โ€” Decision Intelligence Toolkit for Agno agents. + +Exposes Semantica's decision intelligence as native Agno tools so that agents +can actively record, query, and validate decisions during their reasoning loop. + +Follows Agno's ``Toolkit`` pattern โ€” each method decorated with ``@register`` +(or manually registered via ``self.register()``) becomes a tool the LLM can +call. + +Install +------- + pip install semantica[agno] + +Example +------- + >>> from semantica.context import AgentContext + >>> from integrations.agno import AgnoDecisionKit + >>> ctx = AgentContext(decision_tracking=True) + >>> from agno.agent import Agent + >>> agent = Agent(tools=[AgnoDecisionKit(context=ctx)], show_tool_calls=True) + +Tools exposed +------------- +record_decision โ€” Record a decision with reasoning and outcome +find_precedents โ€” Search for similar past decisions +trace_causal_chain โ€” Trace causal chain of a decision node +analyze_impact โ€” Assess downstream influence of a decision +check_policy โ€” Validate a decision against policy rules +get_decision_summary โ€” Summarise decision history by category +""" + +from __future__ import annotations + +import json +import re +from typing import Any, Dict, List, Optional + +from semantica.utils.logging import get_logger + +logger = get_logger(__name__) + +# --------------------------------------------------------------------------- +# Optional: Agno Toolkit base class +# --------------------------------------------------------------------------- +AGNO_AVAILABLE = False +AGNO_IMPORT_ERROR: Optional[str] = None + +_ToolkitBase: Any = object + +try: + from agno.tools.toolkit import Toolkit as _AgnoToolkit # type: ignore + + _ToolkitBase = _AgnoToolkit + AGNO_AVAILABLE = True +except ImportError as exc: + AGNO_IMPORT_ERROR = str(exc) + + +# --------------------------------------------------------------------------- +# AgnoDecisionKit +# --------------------------------------------------------------------------- +class AgnoDecisionKit(_ToolkitBase): # type: ignore[misc] + """ + Agno Toolkit that surfaces Semantica's decision intelligence as agent tools. + + Parameters + ---------- + context: + A ``semantica.context.AgentContext`` (or ``AgentContext``-compatible + object with ``record_decision``, ``find_precedents_advanced``, + ``analyze_decision_influence`` methods). A fresh in-memory context is + created when ``None``. + max_precedents: + Default number of precedents returned by ``find_precedents``. + causal_depth: + Default chain depth used by ``trace_causal_chain``. + enable_policy_check: + Register the ``check_policy`` tool (default: ``True``). + """ + + def __init__( + self, + context: Any = None, + max_precedents: int = 5, + causal_depth: int = 3, + enable_policy_check: bool = True, + **kwargs: Any, + ) -> None: + if AGNO_AVAILABLE: + super().__init__(name="decision_kit", **kwargs) # type: ignore[call-arg] + + # Always initialise _tools so the attribute exists regardless of agno + if not hasattr(self, "_tools"): + self._tools: list = [] + + self.max_precedents = max_precedents + self.causal_depth = causal_depth + + # Build or reuse AgentContext + if context is None: + from semantica.context import AgentContext + from semantica.vector_store import VectorStore + + context = AgentContext( + vector_store=VectorStore(backend="faiss"), + decision_tracking=True, + ) + self._ctx = context + + # Register tools. + # _tools is always kept as a plain list so callers can inspect registered + # tools regardless of whether agno is installed. When agno IS available + # we also call Toolkit.register() so the real agno runtime picks them up. + tools_to_register = [ + self.record_decision, + self.find_precedents, + self.trace_causal_chain, + self.analyze_impact, + self.get_decision_summary, + ] + if enable_policy_check: + tools_to_register.append(self.check_policy) + + for fn in tools_to_register: + self._tools.append(fn) + if AGNO_AVAILABLE: + try: + self.register(fn) + except Exception: + pass + + logger.info("AgnoDecisionKit initialised") + + # ------------------------------------------------------------------ + # Tools + # ------------------------------------------------------------------ + + def record_decision( + self, + category: str, + scenario: str, + reasoning: str, + outcome: str, + confidence: float = 0.8, + entities: Optional[str] = None, + ) -> str: + """ + Record a decision with its reasoning and outcome. + + Parameters + ---------- + category: + Domain category, e.g. ``"loan_approval"``, ``"content_moderation"``. + scenario: + Short description of the situation being decided. + reasoning: + Why this outcome was chosen. + outcome: + The decision result, e.g. ``"approved"``, ``"rejected"``. + confidence: + Confidence score in [0, 1]. + entities: + Comma-separated list of entity names relevant to the decision. + + Returns + ------- + str + JSON with ``{"decision_id": "", "status": "recorded"}``. + """ + entity_list: Optional[List[str]] = None + if entities: + entity_list = [e.strip() for e in entities.split(",") if e.strip()] + + try: + decision_id = self._ctx.record_decision( + category=category, + scenario=scenario, + reasoning=reasoning, + outcome=outcome, + confidence=float(confidence), + entities=entity_list, + ) + result = {"decision_id": str(decision_id), "status": "recorded"} + logger.info("record_decision โ†’ %s", decision_id) + except Exception as exc: + result = {"error": str(exc), "status": "failed"} + logger.warning("record_decision failed: %s", exc) + + return json.dumps(result) + + def find_precedents( + self, + scenario: str, + category: Optional[str] = None, + limit: Optional[int] = None, + ) -> str: + """ + Search for past decisions similar to the given scenario. + + Parameters + ---------- + scenario: + Description of the current situation. + category: + Optional category filter. + limit: + Maximum number of precedents to return. + + Returns + ------- + str + JSON list of precedent summaries. + """ + k = limit or self.max_precedents + try: + precedents = self._ctx.find_precedents_advanced( + scenario=scenario, + category=category, + ) + # Normalise to a serialisable list + out: List[Dict[str, Any]] = [] + for p in (precedents or [])[:k]: + if isinstance(p, dict): + out.append(p) + else: + out.append( + { + "scenario": getattr(p, "scenario", str(p)), + "outcome": getattr(p, "outcome", ""), + "confidence": getattr(p, "confidence", 0.0), + "category": getattr(p, "category", ""), + } + ) + logger.info("find_precedents('%s') โ†’ %d results", scenario, len(out)) + return json.dumps({"precedents": out, "count": len(out)}) + except Exception as exc: + logger.warning("find_precedents failed: %s", exc) + return json.dumps({"precedents": [], "count": 0, "error": str(exc)}) + + def trace_causal_chain( + self, + decision_id: str, + depth: Optional[int] = None, + ) -> str: + """ + Trace the causal chain starting from a decision node. + + Parameters + ---------- + decision_id: + Identifier of the decision to trace. + depth: + Maximum chain depth to traverse. + + Returns + ------- + str + JSON representation of the causal chain. + """ + max_depth = depth or self.causal_depth + try: + chain = self._ctx.knowledge_graph.trace_decision_causality( # type: ignore[attr-defined] + decision_id, depth=max_depth + ) + return json.dumps({"causal_chain": chain, "decision_id": decision_id}) + except AttributeError: + # Fallback if the graph doesn't expose trace_decision_causality + try: + chain = self._ctx.knowledge_graph.find_precedents( # type: ignore[attr-defined] + category="decision", limit=max_depth + ) + return json.dumps({"causal_chain": chain, "decision_id": decision_id}) + except Exception as exc: + return json.dumps({"error": str(exc), "decision_id": decision_id}) + except Exception as exc: + logger.warning("trace_causal_chain failed: %s", exc) + return json.dumps({"error": str(exc), "decision_id": decision_id}) + + def analyze_impact(self, decision_id: str) -> str: + """ + Assess the downstream influence of a decision using graph centrality. + + Parameters + ---------- + decision_id: + Identifier of the decision to analyse. + + Returns + ------- + str + JSON with influence metrics. + """ + try: + influence = self._ctx.analyze_decision_influence(decision_id) + if not isinstance(influence, dict): + influence = {"influence": str(influence)} + influence["decision_id"] = decision_id + return json.dumps(influence) + except Exception as exc: + logger.warning("analyze_impact failed: %s", exc) + return json.dumps({"error": str(exc), "decision_id": decision_id}) + + def check_policy( + self, + decision_data: str, + policy_rules: Optional[str] = None, + ) -> str: + """ + Validate a proposed decision against policy rules. + + Rules are evaluated inline using simple comparison expressions. This + avoids misuse of ``PolicyEngine.check_compliance`` (which requires a + stored ``Decision`` + ``policy_id``) and ensures exceptions never + silently return ``compliant=True``. + + Parameters + ---------- + decision_data: + JSON string describing the decision (must include ``category``, + ``outcome``, ``confidence`` keys at minimum). + policy_rules: + JSON list of rule strings, e.g. + ``'["confidence >= 0.7", "category != \\"test\\""]'``. + Each rule is a simple comparison: `` `` + where op is one of ``>=``, ``<=``, ``!=``, ``==``, ``>``, ``<``. + + Returns + ------- + str + JSON with ``{"compliant": bool, "violations": [...], "warnings": [...]}`` + """ + try: + data = json.loads(decision_data) if isinstance(decision_data, str) else decision_data + except json.JSONDecodeError as exc: + return json.dumps( + { + "compliant": False, + "violations": [f"Invalid decision_data JSON: {exc}"], + "warnings": [], + } + ) + + rules: List[str] = [] + if policy_rules: + try: + rules = json.loads(policy_rules) + except json.JSONDecodeError: + rules = [r.strip() for r in policy_rules.split(",") if r.strip()] + + violations: List[str] = [] + warnings: List[str] = [] + + for rule in rules: + try: + if not self._eval_rule(rule, data): + violations.append(f"Rule violated: {rule}") + except Exception as exc: + warnings.append(f"Could not evaluate rule '{rule}': {exc}") + + compliant = len(violations) == 0 + logger.debug("check_policy: compliant=%s, violations=%d", compliant, len(violations)) + return json.dumps( + { + "compliant": compliant, + "violations": violations, + "warnings": warnings, + } + ) + + def _eval_rule(self, rule: str, data: Dict[str, Any]) -> bool: + """Evaluate a simple comparison rule (``field op value``) against data.""" + m = re.match(r"(\w+)\s*(>=|<=|!=|==|>|<)\s*(.+)", rule.strip()) + if not m: + return True # unrecognised format โ€” pass through + field, op, val_str = m.group(1), m.group(2), m.group(3).strip().strip("\"'") + actual = data.get(field) + if actual is None: + return True # field absent โ€” cannot evaluate + try: + val: Any = type(actual)(val_str) + except (ValueError, TypeError): + val = val_str + ops = { + ">=": lambda a, b: a >= b, + "<=": lambda a, b: a <= b, + "!=": lambda a, b: a != b, + "==": lambda a, b: a == b, + ">": lambda a, b: a > b, + "<": lambda a, b: a < b, + } + return ops[op](actual, val) + + def get_decision_summary( + self, + category: Optional[str] = None, + since: Optional[str] = None, + limit: int = 10, + ) -> str: + """ + Summarise the decision history, optionally filtered by category. + + Parameters + ---------- + category: + Filter to a specific decision category. + since: + ISO-8601 timestamp โ€” only include decisions after this time. + limit: + Maximum number of decisions to include. + + Returns + ------- + str + JSON summary of recent decisions. + """ + try: + insights = self._ctx.get_context_insights() + if not isinstance(insights, dict): + insights = {"raw": str(insights)} + insights["category_filter"] = category + return json.dumps(insights) + except Exception as exc: + logger.warning("get_decision_summary failed: %s", exc) + return json.dumps({"error": str(exc)}) diff --git a/integrations/agno/kg_toolkit.py b/integrations/agno/kg_toolkit.py new file mode 100644 index 00000000..75ee26ed --- /dev/null +++ b/integrations/agno/kg_toolkit.py @@ -0,0 +1,465 @@ +""" +AgnoKGToolkit โ€” Knowledge Graph Toolkit for Agno agents. + +Lets agents actively build and query the context graph as part of their +reasoning loop. Backed by Semantica's ``NERExtractor``, ``RelationExtractor``, +``Reasoner``, and ``ContextGraph``. + +Install +------- + pip install semantica[agno] + +Example +------- + >>> from integrations.agno import AgnoKGToolkit + >>> from agno.agent import Agent + >>> agent = Agent(tools=[AgnoKGToolkit()], show_tool_calls=True) + +Tools exposed +------------- +extract_entities โ€” Extract named entities from text +extract_relations โ€” Extract relationships between entities +add_to_graph โ€” Add entities / relations to the context graph +query_graph โ€” Query the graph (natural-language keyword or Cypher) +find_related โ€” Find concepts related to a given entity +infer_facts โ€” Apply rules to infer new facts from the graph +export_subgraph โ€” Export a subgraph as JSON-LD / RDF Turtle +""" + +from __future__ import annotations + +import json +from typing import Any, Dict, List, Optional + +from semantica.utils.logging import get_logger + +logger = get_logger(__name__) + +# --------------------------------------------------------------------------- +# Optional: Agno Toolkit base class +# --------------------------------------------------------------------------- +AGNO_AVAILABLE = False +AGNO_IMPORT_ERROR: Optional[str] = None + +_ToolkitBase: Any = object + +try: + from agno.tools.toolkit import Toolkit as _AgnoToolkit # type: ignore + + _ToolkitBase = _AgnoToolkit + AGNO_AVAILABLE = True +except ImportError as exc: + AGNO_IMPORT_ERROR = str(exc) + + +# --------------------------------------------------------------------------- +# AgnoKGToolkit +# --------------------------------------------------------------------------- +class AgnoKGToolkit(_ToolkitBase): # type: ignore[misc] + """ + Agno Toolkit that surfaces Semantica's KG pipeline as agent tools. + + Parameters + ---------- + graph_store_backend: + Storage backend for the internal ``ContextGraph``. One of + ``"inmemory"`` (default), ``"neo4j"``, ``"falkordb"``. + ner_extractor: + A ``semantica.semantic_extract.NERExtractor`` instance; auto-created + when ``None``. + relation_extractor: + A ``semantica.semantic_extract.RelationExtractor`` instance; auto- + created when ``None``. + reasoner: + A ``semantica.reasoning.Reasoner`` instance; auto-created when + ``None``. + context: + An existing ``AgentContext`` or ``ContextGraph`` to attach to. A + fresh in-memory ``ContextGraph`` is used when ``None``. + """ + + def __init__( + self, + graph_store_backend: str = "inmemory", + ner_extractor: Any = None, + relation_extractor: Any = None, + reasoner: Any = None, + context: Any = None, + **kwargs: Any, + ) -> None: + if AGNO_AVAILABLE: + super().__init__(name="kg_toolkit", **kwargs) # type: ignore[call-arg] + + # Always initialise _tools so the attribute exists regardless of agno + if not hasattr(self, "_tools"): + self._tools: list = [] + + # Lazy imports + from semantica.context import ContextGraph + from semantica.reasoning import Reasoner + from semantica.semantic_extract import NERExtractor, RelationExtractor + + if context is not None: + self._graph = getattr(context, "knowledge_graph", context) + else: + self._graph = ContextGraph() + + self._ner = ner_extractor or NERExtractor() + self._rel = relation_extractor or RelationExtractor() + self._reasoner = reasoner or Reasoner() + + # Register tools. + # _tools is always kept as a plain list so callers can inspect registered + # tools regardless of whether agno is installed. When agno IS available + # we also call Toolkit.register() so the real agno runtime picks them up. + tools_to_register = [ + self.extract_entities, + self.extract_relations, + self.add_to_graph, + self.query_graph, + self.find_related, + self.infer_facts, + self.export_subgraph, + ] + for fn in tools_to_register: + self._tools.append(fn) + if AGNO_AVAILABLE: + try: + self.register(fn) + except Exception: + pass + + logger.info("AgnoKGToolkit initialised (backend=%s)", graph_store_backend) + + # ------------------------------------------------------------------ + # Tools + # ------------------------------------------------------------------ + + def extract_entities(self, text: str) -> str: + """ + Extract named entities from the given text. + + Parameters + ---------- + text: + Input text to analyse. + + Returns + ------- + str + JSON list of ``{"name": str, "type": str, "confidence": float}``. + """ + try: + raw = self._ner.extract_entities(text) or [] + entities = [ + { + "name": getattr(e, "name", str(e)), + "type": getattr(e, "type", ""), + "confidence": round(float(getattr(e, "confidence", 1.0)), 4), + } + for e in raw + ] + logger.debug("extract_entities โ†’ %d entities", len(entities)) + return json.dumps({"entities": entities, "count": len(entities)}) + except Exception as exc: + logger.warning("extract_entities failed: %s", exc) + return json.dumps({"entities": [], "count": 0, "error": str(exc)}) + + def extract_relations(self, text: str, entities: Optional[str] = None) -> str: + """ + Extract relationships between entities in the given text. + + Parameters + ---------- + text: + Input text to analyse. + entities: + Optional JSON list of entity names to restrict extraction to. + + Returns + ------- + str + JSON list of ``{"source": str, "relation": str, "target": str, "confidence": float}``. + """ + entity_list: Optional[List[str]] = None + if entities: + try: + entity_list = json.loads(entities) + except json.JSONDecodeError: + entity_list = [e.strip() for e in entities.split(",") if e.strip()] + + try: + raw = self._rel.extract_relations(text, entities=entity_list) or [] + relations = [ + { + "source": getattr(r, "source", ""), + "relation": getattr(r, "type", getattr(r, "relation", "")), + "target": getattr(r, "target", ""), + "confidence": round(float(getattr(r, "confidence", 1.0)), 4), + } + for r in raw + ] + logger.debug("extract_relations โ†’ %d relations", len(relations)) + return json.dumps({"relations": relations, "count": len(relations)}) + except Exception as exc: + logger.warning("extract_relations failed: %s", exc) + return json.dumps({"relations": [], "count": 0, "error": str(exc)}) + + def add_to_graph( + self, + entities: Optional[str] = None, + relations: Optional[str] = None, + ) -> str: + """ + Add entities and/or relations to the active context graph. + + Parameters + ---------- + entities: + JSON list of ``{"name": str, "type": str}`` objects. + relations: + JSON list of ``{"source": str, "relation": str, "target": str}`` objects. + + Returns + ------- + str + JSON summary of nodes and edges added. + """ + nodes_added = 0 + edges_added = 0 + + if entities: + try: + ent_list = json.loads(entities) if isinstance(entities, str) else entities + for ent in ent_list: + name = ent.get("name", str(ent)) + ntype = ent.get("type", "Entity") + try: + # ContextGraph.add_node(node_id, node_type, content=None, **props) + self._graph.add_node(node_id=name, node_type=ntype) # type: ignore[attr-defined] + nodes_added += 1 + except Exception: + pass + except (json.JSONDecodeError, AttributeError) as exc: + logger.debug("add_to_graph entities parse error: %s", exc) + + if relations: + try: + rel_list = json.loads(relations) if isinstance(relations, str) else relations + for rel in rel_list: + src = rel.get("source", "") + tgt = rel.get("target", "") + rel_type = rel.get("relation", "related_to") + try: + # ContextGraph.add_edge(source_id, target_id, edge_type, **props) + self._graph.add_edge(source_id=src, target_id=tgt, edge_type=rel_type) # type: ignore[attr-defined] + edges_added += 1 + except Exception: + pass + except (json.JSONDecodeError, AttributeError) as exc: + logger.debug("add_to_graph relations parse error: %s", exc) + + logger.debug("add_to_graph: +%d nodes, +%d edges", nodes_added, edges_added) + return json.dumps({"nodes_added": nodes_added, "edges_added": edges_added}) + + def query_graph(self, query: str) -> str: + """ + Query the context graph in natural language or Cypher. + + For natural-language queries all nodes are retrieved and filtered by + whether ``query`` appears in their ``node_id``. Pass a string starting + with ``"MATCH"`` for raw Cypher execution (requires a Neo4j / FalkorDB + backend). + + Parameters + ---------- + query: + Search query string. + + Returns + ------- + str + JSON list of matching nodes / records. + """ + try: + if query.strip().upper().startswith("MATCH"): + # Cypher path + try: + result = self._graph.execute_query(query) # type: ignore[attr-defined] + records = result if isinstance(result, list) else [str(result)] + return json.dumps({"results": records, "query_type": "cypher"}) + except AttributeError: + return json.dumps( + { + "error": "Cypher queries require a Neo4j/FalkorDB backend", + "query_type": "cypher", + } + ) + else: + # Natural-language keyword lookup โ€” ContextGraph.find_nodes() โ†’ List[Dict] + all_nodes = self._graph.find_nodes() # type: ignore[attr-defined] + q_lower = query.lower() + out = [] + for n in (all_nodes or []): + if isinstance(n, dict): + node_id = n.get("node_id", "") + node_type = n.get("node_type", "") + else: + node_id = getattr(n, "id", getattr(n, "label", str(n))) + node_type = getattr(n, "node_type", "") + if q_lower in node_id.lower() or q_lower in node_type.lower(): + out.append({"label": node_id, "type": node_type, "id": node_id}) + return json.dumps({"results": out, "count": len(out), "query_type": "keyword"}) + except Exception as exc: + logger.warning("query_graph failed: %s", exc) + return json.dumps({"results": [], "error": str(exc)}) + + def find_related(self, entity: str, hops: int = 1) -> str: + """ + Find concepts related to ``entity`` within ``hops`` graph hops. + + Parameters + ---------- + entity: + The entity name to start from. + hops: + Maximum number of relationship hops to traverse. + + Returns + ------- + str + JSON list of related entity names. + """ + try: + related: List[str] = [] + frontier = [entity] + visited = {entity} + + for _ in range(max(1, hops)): + next_frontier: List[str] = [] + for e in frontier: + try: + # ContextGraph.get_neighbors(node_id, hops=1, ...) โ†’ List[Dict] + neighbours = self._graph.get_neighbors(node_id=e, hops=1) # type: ignore[attr-defined] + for n in (neighbours or []): + if isinstance(n, dict): + label = n.get("node_id", "") + else: + label = getattr(n, "label", str(n)) + if label and label not in visited: + visited.add(label) + next_frontier.append(label) + related.append(label) + except Exception: + pass + frontier = next_frontier + + logger.debug("find_related('%s', hops=%d) โ†’ %d", entity, hops, len(related)) + return json.dumps({"entity": entity, "related": related, "count": len(related)}) + except Exception as exc: + logger.warning("find_related failed: %s", exc) + return json.dumps({"entity": entity, "related": [], "error": str(exc)}) + + def infer_facts(self, rules: str, facts: Optional[str] = None) -> str: + """ + Apply inference rules to the graph and return newly derived facts. + + Parameters + ---------- + rules: + JSON list of rule strings, e.g. + ``'["IF Person(?x) THEN Human(?x)"]'`` + facts: + Optional JSON list of additional fact strings to load before + inference. When ``None``, the current graph state is used. + + Returns + ------- + str + JSON list of inferred fact strings. + """ + try: + rule_list: List[str] = json.loads(rules) if rules else [] + except json.JSONDecodeError: + rule_list = [r.strip() for r in rules.split(",") if r.strip()] + + fact_list: List[str] = [] + if facts: + try: + fact_list = json.loads(facts) + except json.JSONDecodeError: + fact_list = [f.strip() for f in facts.split(",") if f.strip()] + + if not fact_list: + # Derive facts from graph nodes via the public API + try: + all_nodes = self._graph.find_nodes() # type: ignore[attr-defined] + for node in (all_nodes or [])[:50]: + if isinstance(node, dict): + label = node.get("node_id", "") + ntype = node.get("node_type", "Entity") + else: + label = getattr(node, "label", str(node)) + ntype = getattr(node, "node_type", "Entity") + if label: + fact_list.append(f"{ntype}({label})") + except Exception: + pass + + try: + result = self._reasoner.infer_facts(fact_list, rule_list) + inferred = getattr(result, "inferred_facts", []) or [] + inferred_strs = [str(f) for f in inferred] + logger.debug("infer_facts โ†’ %d new facts", len(inferred_strs)) + return json.dumps({"inferred_facts": inferred_strs, "count": len(inferred_strs)}) + except Exception as exc: + logger.warning("infer_facts failed: %s", exc) + return json.dumps({"inferred_facts": [], "error": str(exc)}) + + def export_subgraph( + self, + entity: Optional[str] = None, + format: str = "json-ld", + ) -> str: + """ + Export a subgraph centred on ``entity`` as RDF / JSON-LD. + + Parameters + ---------- + entity: + Root entity of the subgraph. The whole graph is exported when + ``None``. + format: + Output format: ``"json-ld"`` (default), ``"turtle"`` / ``"ttl"``, + ``"xml"``, ``"nt"``. + + Returns + ------- + str + Serialised subgraph in the requested format (JSON string wrapper). + """ + try: + from semantica.export import RDFExporter # lazy import + + exporter = RDFExporter() + rdf_format = {"ttl": "turtle", "json-ld": "json-ld", "xml": "xml", "nt": "nt"}.get( + format, format + ) + output = exporter.export_to_rdf(self._graph, format=rdf_format) # type: ignore[arg-type] + return json.dumps({"format": rdf_format, "data": output}) + except Exception as exc: + logger.warning("export_subgraph failed: %s", exc) + # Fallback: return graph nodes via the public API + try: + all_nodes = self._graph.find_nodes() # type: ignore[attr-defined] + nodes = [] + for n in (all_nodes or []): + if isinstance(n, dict): + nodes.append({"id": n.get("node_id", ""), "label": n.get("node_id", "")}) + else: + nodes.append( + {"id": getattr(n, "id", ""), "label": getattr(n, "label", "")} + ) + return json.dumps({"format": "json", "nodes": nodes, "note": str(exc)}) + except Exception: + return json.dumps({"format": format, "data": "", "error": str(exc)}) diff --git a/integrations/agno/knowledge_graph.py b/integrations/agno/knowledge_graph.py new file mode 100644 index 00000000..21b6b280 --- /dev/null +++ b/integrations/agno/knowledge_graph.py @@ -0,0 +1,475 @@ +""" +AgnoKnowledgeGraph โ€” Relational agent knowledge backed by Semantica's KG pipeline. + +Implements Agno's ``AgentKnowledge`` protocol so that Agno agents can query a +structured ``ContextGraph`` instead of a flat vector document store. + +Ingested documents pass through the full Semantica extraction pipeline: + + parse โ†’ split โ†’ NER โ†’ relation extract โ†’ graph build + +and search uses multi-hop GraphRAG: vector retrieval + graph traversal + +context injection. + +Install +------- + pip install semantica[agno] + +Example +------- + >>> from integrations.agno import AgnoKnowledgeGraph + >>> from semantica.kg import GraphBuilder + >>> from semantica.semantic_extract import NERExtractor, RelationExtractor + >>> kg = AgnoKnowledgeGraph( + ... graph_builder=GraphBuilder(), + ... ner_extractor=NERExtractor(), + ... relation_extractor=RelationExtractor(), + ... ) + >>> kg.load("regulatory_docs/", recursive=True) + >>> from agno.agent import Agent + >>> agent = Agent(knowledge=kg, search_knowledge=True) +""" + +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any, Dict, Iterator, List, Optional, Union + +from semantica.utils.logging import get_logger + +logger = get_logger(__name__) + +# --------------------------------------------------------------------------- +# Optional: Agno AgentKnowledge base class +# --------------------------------------------------------------------------- +AGNO_AVAILABLE = False +AGNO_IMPORT_ERROR: Optional[str] = None + +_KnowledgeBase: Any = object + +try: + from agno.knowledge.base import AgentKnowledge as _AgnoAgentKnowledge # type: ignore + + _KnowledgeBase = _AgnoAgentKnowledge + AGNO_AVAILABLE = True +except ImportError as exc: + AGNO_IMPORT_ERROR = str(exc) + + +# --------------------------------------------------------------------------- +# Lightweight document stand-in (used when agno is absent) +# --------------------------------------------------------------------------- +class _Document: + """Minimal stand-in for ``agno.document.Document``.""" + + __slots__ = ("id", "content", "meta_data", "name") + + def __init__( + self, + content: str, + id: Optional[str] = None, + name: Optional[str] = None, + meta_data: Optional[Dict[str, Any]] = None, + ) -> None: + self.id = id + self.content = content + self.name = name + self.meta_data = meta_data or {} + + +try: + from agno.document.base import Document as AgnoDocument # type: ignore +except ImportError: + AgnoDocument = _Document # type: ignore + + +# --------------------------------------------------------------------------- +# AgnoKnowledgeGraph +# --------------------------------------------------------------------------- +class AgnoKnowledgeGraph(_KnowledgeBase): # type: ignore[misc] + """ + Relational agent knowledge store backed by Semantica's KG pipeline. + + Parameters + ---------- + graph_builder: + A ``semantica.kg.GraphBuilder`` instance. Created automatically if + ``None``. + ner_extractor: + A ``semantica.semantic_extract.NERExtractor`` instance. Created + automatically if ``None``. + relation_extractor: + A ``semantica.semantic_extract.RelationExtractor`` instance. Created + automatically if ``None``. + context_graph: + An existing ``semantica.context.ContextGraph`` to use as the backing + store. A fresh in-memory graph is created when ``None``. + graph_store_backend: + Passed to ``ContextGraph`` when ``context_graph`` is ``None``. + Supported values: ``"inmemory"`` (default), ``"neo4j"``, + ``"falkordb"``. + graph_store_uri: + Connection URI for the chosen graph store backend. + num_documents: + Default number of documents returned by ``search()``. + chunk_size: + Maximum characters per text chunk during ingestion. + """ + + def __init__( + self, + graph_builder: Any = None, + ner_extractor: Any = None, + relation_extractor: Any = None, + context_graph: Any = None, + graph_store_backend: str = "inmemory", + graph_store_uri: Optional[str] = None, + num_documents: int = 5, + chunk_size: int = 1000, + **kwargs: Any, + ) -> None: + if AGNO_AVAILABLE: + super().__init__(**kwargs) # type: ignore[call-arg] + + self.num_documents = num_documents + self.chunk_size = chunk_size + self._graph_store_backend = graph_store_backend + + # Lazy imports to keep semantica core optional at import time + from semantica.context import AgentContext, ContextGraph + from semantica.kg import GraphBuilder + from semantica.semantic_extract import NERExtractor, RelationExtractor + from semantica.vector_store import VectorStore + + self._graph = context_graph or ContextGraph() + + # Connect GraphBuilder to the ContextGraph so build() persists content. + self._graph_builder = graph_builder or GraphBuilder() + self._graph_builder.graph_store = self._graph + + self._ner = ner_extractor or NERExtractor() + self._rel = relation_extractor or RelationExtractor() + + # Internal AgentContext for vector-based retrieval (shares same graph). + self._agent_context = AgentContext( + vector_store=VectorStore(backend="faiss"), + knowledge_graph=self._graph, + decision_tracking=False, + ) + + # In-process document store for keyword-search fallback + self._docs: List[Dict[str, Any]] = [] + + logger.info( + "AgnoKnowledgeGraph initialised", + extra={"backend": graph_store_backend}, + ) + + # ------------------------------------------------------------------ + # AgentKnowledge protocol + # ------------------------------------------------------------------ + + def search( + self, + query: str, + num_documents: Optional[int] = None, + filters: Optional[Dict[str, Any]] = None, + ) -> List[Any]: + """ + Multi-hop GraphRAG search. + + 1. Vector retrieval via ``AgentContext.retrieve()``. + 2. Graph hop expansion for entities found in top results. + 3. Returns a list of Agno ``Document`` objects. + + Falls back to keyword scoring over the in-process ``_docs`` cache + when vector retrieval is unavailable. + """ + k = num_documents or self.num_documents + results: List[Any] = [] + + # Primary: vector similarity retrieval + try: + retrieved = self._agent_context.retrieve(query, max_results=k) + for item in retrieved: + if isinstance(item, dict): + content = item.get("content", item.get("text", str(item))) + entities = item.get("entities", []) + meta = {k2: v for k2, v in item.items() if k2 not in ("content", "text")} + else: + content = str(item) + entities = [] + meta = {} + extra = self._graph_context_for(entities) if entities else "" + if extra: + content = content + "\n\n[Graph context]\n" + extra + results.append(AgnoDocument(content=content, meta_data=meta)) + if results: + logger.debug("search('%s') โ†’ %d documents (vector)", query, len(results)) + return results + except Exception as exc: + logger.debug("Vector retrieval failed, using keyword fallback: %s", exc) + + # Fallback: keyword / substring scoring over in-process cache + q_lower = query.lower() + scored = [ + (doc, sum(1 for w in q_lower.split() if w in doc["text"].lower())) + for doc in self._docs + ] + scored.sort(key=lambda t: t[1], reverse=True) + top = [d for d, _ in scored[:k]] + + for doc in top: + extra = self._graph_context_for(doc.get("entities", [])) + content = doc["text"] + if extra: + content += "\n\n[Graph context]\n" + extra + results.append( + AgnoDocument( + content=content, + id=doc.get("id"), + name=doc.get("source"), + meta_data=doc.get("metadata", {}), + ) + ) + + logger.debug("search('%s') โ†’ %d documents (keyword)", query, len(results)) + return results + + def load( + self, + path: Union[str, Path, None] = None, + urls: Optional[List[str]] = None, + texts: Optional[List[str]] = None, + recursive: bool = False, + recreate: bool = False, + ) -> None: + """ + Ingest documents into the knowledge graph. + + Parameters + ---------- + path: + A file path, directory path, or glob pattern. + urls: + List of URLs to fetch and ingest. + texts: + Raw text strings to ingest directly. + recursive: + When ``path`` points to a directory, walk subdirectories. + recreate: + Drop all previously loaded documents before ingesting. + """ + if recreate: + self._docs.clear() + + if texts: + for text in texts: + self._ingest_text(text, source="") + + if path is not None: + self._ingest_path(Path(path), recursive=recursive) + + if urls: + self.load_urls(urls) + + def load_urls(self, urls: List[str]) -> None: + """Fetch each URL and ingest the response body. + + Only ``http`` and ``https`` schemes are permitted to prevent SSRF. + """ + import urllib.request + from urllib.parse import urlparse + + for url in urls: + parsed = urlparse(url) + if parsed.scheme not in ("http", "https"): + logger.warning( + "Skipping URL with disallowed scheme '%s': %s", + parsed.scheme, + url, + ) + continue + try: + with urllib.request.urlopen(url, timeout=10) as resp: # noqa: S310 + text = resp.read().decode("utf-8", errors="replace") + self._ingest_text(text, source=url) + logger.info("Loaded URL: %s", url) + except Exception as exc: + logger.warning("Failed to fetch %s: %s", url, exc) + + # AgentKnowledge also expects `load_documents` + def load_documents( + self, + documents: List[Any], + upsert: bool = False, + ) -> None: + """Ingest a list of Agno ``Document`` objects.""" + for doc in documents: + text = getattr(doc, "content", None) or getattr(doc, "text", str(doc)) + source = getattr(doc, "name", None) or getattr(doc, "id", "") + self._ingest_text(text, source=source) + + def get_graph_context(self, entity: str) -> str: + """ + Return a structured text representation of an entity's subgraph + (neighbours and edge types), suitable for structured reasoning. + + Parameters + ---------- + entity: + Root entity name (must have been added to the graph). + + Returns + ------- + str + Multi-line text with nodes and labelled edge types. + """ + lines = [f"Entity: {entity}"] + try: + neighbours = self._graph.get_neighbors(node_id=entity, hops=1) + for n in (neighbours or [])[:10]: + if isinstance(n, dict): + node_id = n.get("node_id", "") + ntype = n.get("node_type", "") + edge_type = n.get("edge_type", "related_to") + suffix = f" (type: {ntype})" if ntype else "" + lines.append(f" --[{edge_type}]--> {node_id}{suffix}") + else: + lines.append(f" --> {getattr(n, 'label', str(n))}") + except Exception: + pass + return "\n".join(lines) + + # ------------------------------------------------------------------ + # Internal helpers + # ------------------------------------------------------------------ + + def _chunk_text(self, text: str) -> List[str]: + """Split text into chunks at paragraph boundaries.""" + paragraphs = [p.strip() for p in text.split("\n\n") if p.strip()] + if not paragraphs: + return [text] if text.strip() else [] + + chunks: List[str] = [] + current: List[str] = [] + current_len = 0 + + for para in paragraphs: + if current_len + len(para) > self.chunk_size and current: + chunks.append("\n\n".join(current)) + current = [] + current_len = 0 + current.append(para) + current_len += len(para) + + if current: + chunks.append("\n\n".join(current)) + + return chunks or [text] + + def _ingest_text(self, text: str, source: str = "") -> None: + """Run the full extraction pipeline and store in graph + doc list.""" + import uuid + + chunks = self._chunk_text(text) + all_entities: List[str] = [] + all_relations: List[Any] = [] + + for chunk in chunks: + # NER + ner_result: List[Any] = [] + try: + ner_result = self._ner.extract_entities(chunk) or [] + chunk_entities = [getattr(e, "name", str(e)) for e in ner_result] + all_entities.extend(chunk_entities) + except Exception as exc: + logger.debug("NER failed for chunk in '%s': %s", source, exc) + + # Relation extraction + try: + chunk_relations = self._rel.extract_relations(chunk, entities=ner_result) or [] + all_relations.extend(chunk_relations) + except Exception as exc: + logger.debug("RelationExtractor failed for chunk in '%s': %s", source, exc) + + # Graph build โ€” graph_store is wired to self._graph in __init__ + try: + sources = [ + { + "text": text, + "entities": all_entities, + "relations": all_relations, + "source": source, + } + ] + self._graph_builder.build(sources) + except Exception as exc: + logger.debug("GraphBuilder.build() failed for '%s': %s", source, exc) + + # Vector index for AgentContext.retrieve() + try: + self._agent_context.store(text, conversation_id=source) + except Exception as exc: + logger.debug("AgentContext.store() failed for '%s': %s", source, exc) + + # Cache document for keyword-search fallback + self._docs.append( + { + "id": str(uuid.uuid4()), + "text": text, + "source": source, + "entities": all_entities, + "metadata": {"source": source}, + } + ) + logger.debug( + "Ingested '%s' โ€” %d entities, %d relations, %d chunks", + source, + len(all_entities), + len(all_relations), + len(chunks), + ) + + def _ingest_path(self, path: Path, recursive: bool = False) -> None: + """Walk a file or directory and ingest all text files.""" + if path.is_file(): + self._ingest_file(path) + elif path.is_dir(): + pattern = "**/*" if recursive else "*" + for child in path.glob(pattern): + if child.is_file(): + self._ingest_file(child) + else: + logger.warning("Path not found: %s", path) + + def _ingest_file(self, filepath: Path) -> None: + try: + text = filepath.read_text(encoding="utf-8", errors="replace") + self._ingest_text(text, source=str(filepath)) + except Exception as exc: + logger.warning("Could not read %s: %s", filepath, exc) + + def _graph_context_for(self, entities: List[str]) -> str: + """Build a short text summary of graph neighbours for a set of entities.""" + if not entities: + return "" + lines: List[str] = [] + for entity in entities[:3]: # limit to avoid context bloat + try: + neighbours = self._graph.get_neighbors(node_id=entity, hops=1) + for n in (neighbours or [])[:3]: + if isinstance(n, dict): + node_id = n.get("node_id", "") + ntype = n.get("node_type", "") + edge_type = n.get("edge_type", "related_to") + lines.append( + f"- {entity} --[{edge_type}]--> {node_id}" + + (f" ({ntype})" if ntype else "") + ) + else: + lines.append(f"- {entity} --> {getattr(n, 'label', str(n))}") + except Exception: + pass + return "\n".join(lines) diff --git a/integrations/agno/shared_context.py b/integrations/agno/shared_context.py new file mode 100644 index 00000000..e6799a8c --- /dev/null +++ b/integrations/agno/shared_context.py @@ -0,0 +1,292 @@ +""" +AgnoSharedContext โ€” Shared ContextGraph for Agno multi-agent teams. + +A single ``ContextGraph`` is shared across all agents in an Agno ``Team``. +Each agent gets a **role-scoped view** via ``bind_agent()``, which returns an +``AgnoContextStore`` namespaced to that agent's role. This prevents +contradictory decisions and enables knowledge reuse without coupling agent +implementations. + +Install +------- + pip install semantica[agno] + +Example +------- + >>> from semantica.context import ContextGraph + >>> from semantica.vector_store import VectorStore + >>> from integrations.agno import AgnoSharedContext, AgnoDecisionKit, AgnoKGToolkit + >>> shared = AgnoSharedContext( + ... vector_store=VectorStore(backend="faiss"), + ... knowledge_graph=ContextGraph(advanced_analytics=True), + ... decision_tracking=True, + ... ) + >>> from agno.agent import Agent + >>> from agno.team import Team + >>> researcher = Agent( + ... name="Researcher", + ... memory=shared.bind_agent("researcher"), + ... tools=[AgnoKGToolkit(context=shared)], + ... ) + >>> analyst = Agent( + ... name="Analyst", + ... memory=shared.bind_agent("analyst"), + ... tools=[AgnoDecisionKit(context=shared)], + ... ) + >>> team = Team(agents=[researcher, analyst], mode="coordinate") +""" + +from __future__ import annotations + +import threading +from typing import Any, Dict, List, Optional + +from semantica.utils.logging import get_logger + +from .context_store import AgnoContextStore + +logger = get_logger(__name__) + + +class _AgentScopedStore(AgnoContextStore): + """ + An ``AgnoContextStore`` bound to a specific agent role. + + All operations are delegated to the parent ``AgnoSharedContext``'s + ``AgentContext`` but tagged with the agent's ``role`` for filtering. + """ + + def __init__(self, shared: "AgnoSharedContext", role: str) -> None: + # Re-use the parent's context rather than creating a new one. + # We skip the normal __init__ and wire all required parent attributes + # directly so that inherited methods (record_decision, find_precedents, + # retrieve, get_context_for_prompt) work correctly via self._context. + self._role = role + self._shared = shared + self._memories: Dict[str, Any] = {} + self.decision_tracking = shared.decision_tracking + self.graph_expansion = shared.graph_expansion + self.session_id = f"{shared.session_id}::{role}" + # Use the attribute name the parent class expects. + self._context = shared._context # type: ignore[attr-defined] + + # ------------------------------------------------------------------ + # Override upsert / record to tag with role + # ------------------------------------------------------------------ + + def upsert_memory(self, memory: Any) -> Optional[Any]: # type: ignore[override] + import uuid + + mem_id = getattr(memory, "id", None) or str(uuid.uuid4()) + mem_text = getattr(memory, "memory", str(memory)) + + try: + self._context.store(mem_text, conversation_id=self.session_id) + except Exception as exc: + logger.warning("[%s] store failed: %s", self._role, exc) + + if self.decision_tracking: + try: + self._context.record_decision( + category=f"memory:{self._role}", + scenario=mem_text[:200], + reasoning=f"Stored by agent role='{self._role}'", + outcome="stored", + confidence=1.0, + ) + except Exception: + pass + + if hasattr(memory, "id"): + memory.id = mem_id + self._memories[mem_id] = memory + + # Also push into the shared registry so all agents can read it + self._shared._shared_memories[mem_id] = memory + + return memory + + def read_memories( # type: ignore[override] + self, + user_id: Optional[str] = None, + limit: Optional[int] = None, + sort: Optional[str] = None, + ) -> List[Any]: + # Return own memories + shared memories from all agents + combined = dict(self._shared._shared_memories) + combined.update(self._memories) + + rows = list(combined.values()) + if user_id: + rows = [r for r in rows if getattr(r, "user_id", None) == user_id] + + reverse = sort != "asc" + rows.sort(key=lambda r: getattr(r, "last_updated", 0), reverse=reverse) + + if limit is not None: + rows = rows[:limit] + return rows + + +class AgnoSharedContext: + """ + Shared context graph coordinator for Agno multi-agent teams. + + Maintains a single ``AgentContext`` and ``ContextGraph`` that all agents + access concurrently. Thread-safety is ensured via a reentrant lock. + + Parameters + ---------- + vector_store: + Shared ``semantica.vector_store.VectorStore`` instance. + knowledge_graph: + Shared ``semantica.context.ContextGraph`` instance. + decision_tracking: + Enable decision recording for all bound agents. + graph_expansion: + Enable graph-hop expansion in all bound agents' ``read_memories``. + session_id: + Team-level session identifier (auto-generated when ``None``). + """ + + def __init__( + self, + vector_store: Any = None, + knowledge_graph: Any = None, + decision_tracking: bool = True, + graph_expansion: bool = True, + session_id: Optional[str] = None, + **agent_context_kwargs: Any, + ) -> None: + import uuid + + from semantica.context import AgentContext, ContextGraph + from semantica.vector_store import VectorStore + + self.decision_tracking = decision_tracking + self.graph_expansion = graph_expansion + self.session_id = session_id or str(uuid.uuid4()) + + if knowledge_graph is None: + knowledge_graph = ContextGraph(advanced_analytics=True) + + if vector_store is None: + vector_store = VectorStore(backend="faiss") + + self._context = AgentContext( + vector_store=vector_store, + knowledge_graph=knowledge_graph, + decision_tracking=decision_tracking, + **agent_context_kwargs, + ) + self._knowledge_graph = knowledge_graph + + # Shared memory pool (all agents read from this) + self._shared_memories: Dict[str, Any] = {} + self._lock = threading.RLock() + self._bound_agents: Dict[str, _AgentScopedStore] = {} + + logger.info( + "AgnoSharedContext initialised (session=%s, decision_tracking=%s)", + self.session_id, + decision_tracking, + ) + + # ------------------------------------------------------------------ + # Public API + # ------------------------------------------------------------------ + + def bind_agent(self, role: str) -> _AgentScopedStore: + """ + Return a role-scoped ``AgnoContextStore`` for the given agent role. + + Multiple calls with the same ``role`` return the **same** store + instance (idempotent). + + Parameters + ---------- + role: + Agent role name, e.g. ``"researcher"``, ``"analyst"``. + + Returns + ------- + _AgentScopedStore + An ``AgnoContextStore`` scoped to ``role`` backed by this shared + context. + """ + with self._lock: + if role not in self._bound_agents: + store = _AgentScopedStore(shared=self, role=role) + self._bound_agents[role] = store + logger.info("Bound agent role='%s' to shared context", role) + return self._bound_agents[role] + + def record_decision( + self, + category: str, + scenario: str, + reasoning: str, + outcome: str, + confidence: float = 0.8, + entities: Optional[List[str]] = None, + agent_role: Optional[str] = None, + ) -> str: + """ + Record a decision into the shared context graph. + + Parameters + ---------- + agent_role: + If provided, the decision is tagged with this agent's role. + """ + tagged_category = f"{category}:{agent_role}" if agent_role else category + with self._lock: + return self._context.record_decision( + category=tagged_category, + scenario=scenario, + reasoning=reasoning, + outcome=outcome, + confidence=confidence, + entities=entities, + ) + + def find_precedents( + self, + scenario: str, + category: Optional[str] = None, + limit: int = 5, + ) -> List[Dict[str, Any]]: + """Search all agents' decision history for similar precedents.""" + try: + return self._context.find_precedents_advanced( + scenario=scenario, + category=category, + limit=limit, + ) + except Exception as exc: + logger.warning("find_precedents failed: %s", exc) + return [] + + def get_shared_insights(self) -> Dict[str, Any]: + """Return analytics over the full shared decision graph.""" + try: + return self._context.get_context_insights() + except Exception as exc: + logger.warning("get_shared_insights failed: %s", exc) + return {} + + @property + def knowledge_graph(self) -> Any: + """Direct access to the shared ``ContextGraph``.""" + return self._knowledge_graph + + @property + def bound_roles(self) -> List[str]: + """List of agent roles currently bound to this shared context.""" + return list(self._bound_agents.keys()) + + def __repr__(self) -> str: # pragma: no cover + return ( + f"AgnoSharedContext(session={self.session_id!r}, " + f"agents={self.bound_roles})" + ) diff --git a/mkdocs.yml b/mkdocs.yml index 9369bcb1..ab2234f8 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -138,6 +138,7 @@ nav: - examples.md - glossary.md - Integrations: + - Agno: integrations/agno.md - Docling: integrations/docling.md - Snowflake: integrations/snowflake.md - Cookbook: cookbook.md diff --git a/pyproject.toml b/pyproject.toml index 858b7875..1e03d95f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -173,6 +173,9 @@ gpu = [ "cupy>=10.0.0" ] +# ---- Agentic Framework Integrations ---- +agno = ["agno>=1.0.0"] + # ---- Splitting / Chunking ---- split-tiktoken = ["tiktoken>=0.5.0"] split-community = ["python-louvain>=0.16"] @@ -211,6 +214,7 @@ explorer-lite = [ # Everything all = [ "semantica[dev,viz,gpu,infra,cloud,monitoring,llm-all,models-huggingface,split-all,graph-all,vectorstore-all,parse-docling,explorer]" + "semantica[dev,viz,gpu,infra,cloud,monitoring,llm-all,models-huggingface,split-all,graph-all,vectorstore-all,parse-docling,agno]" ] # ---------------- ENTRYPOINTS ---------------- @@ -223,7 +227,7 @@ semantica-explorer = "semantica.explorer:main" # ---------------- TOOLING ---------------- [tool.setuptools.packages.find] where = ["."] -include = ["semantica*"] +include = ["semantica*", "integrations*"] [tool.black] line-length = 88 @@ -233,3 +237,6 @@ profile = "black" [tool.pytest.ini_options] testpaths = ["tests"] +markers = [ + "integration: marks tests that require external services or API keys (deselect with '-m not integration')", +] diff --git a/semantica/change_management/change_log.py b/semantica/change_management/change_log.py index 3e099a7b..568c01b9 100644 --- a/semantica/change_management/change_log.py +++ b/semantica/change_management/change_log.py @@ -28,7 +28,8 @@ License: MIT import re from dataclasses import dataclass, field from datetime import datetime -from typing import List, Optional +from typing import List, Optional, Any, Dict, Tuple, Union, Set +from enum import Enum from ..utils.exceptions import ValidationError @@ -106,3 +107,152 @@ class ChangeLogEntry: change_id=change_id, related_changes=related_changes or [] ) + +class Severity(Enum): + CRITICAL = "critical" + HIGH = "high" + MEDIUM = "medium" + LOW = "low" + INFO = "info" + +class ChangeCategory(Enum): + BREAKING = "breaking" + POTENTIALLY_BREAKING = "potentially_breaking" + NON_BREAKING = "non_breaking" + UNKNOWN = "unknown" + +@dataclass +class ImpactReport: + """ Structured impact analysis report.""" + timestamp: str = field(default_factory=lambda: datetime.now().isoformat()) + summary: Dict[str, Any] = field(default_factory=dict) + breaking_changes: List[Dict[str, Any]] = field(default_factory=list) + potentially_breaking: List[Dict[str, Any]] = field(default_factory=list) + safe_changes: List[Dict[str, Any]] = field(default_factory=list) + recommendations: List[str] = field(default_factory=list) + + def to_dict(self) -> Dict[str, Any]: + return { + "timestamp": self.timestamp, + "summary": self.summary, + "impact_classification": { + "breaking": self.breaking_changes, + "potentially_breaking": self.potentially_breaking, + "safe": self.safe_changes + }, + "recommendations": self.recommendations + } + +class ChangeLogAnalyzer: + """ + Analyzes ontology diffs and classifies impact severity. + """ + + VALIDITY_CONSTRAINTS = {'domain', 'range', 'cardinality', 'max_cardinality'} + STRUCTURAL_FIELDS = {'subclasses', 'superclasses', 'equivalent_to', 'disjoint_with'} + + def analyze(self, diff: Dict[str, Any]) -> ImpactReport: + report = ImpactReport() + if not diff: + report.summary = {"error": "Empty diff provided"} + return report + + all_changes = [] + + for key, entity_type, change_type in [ + ("added_classes", "class", "added"), ("added_properties", "property", "added"), + ("removed_classes", "class", "removed"), ("removed_properties", "property", "removed"), + ("changed_classes", "class", "modified"), ("changed_properties", "property", "modified") + ]: + for item in diff.get(key, []): + all_changes.append({ + "uri": item.get("uri", item.get("name", "unknown")), + "entity_type": entity_type, + "change_type": change_type, + "changes": item.get("changes", {}) + }) + + report.summary = {"total_changes": len(all_changes)} + + for change in all_changes: + severity, category, description, mitigation = self._classify_change(change) + entry = { + "entity_uri": change['uri'], + "entity_type": change['entity_type'], + "change_type": change['change_type'], + "description": description, + "severity": severity.value, + "mitigation": mitigation + } + + if category == ChangeCategory.BREAKING: + report.breaking_changes.append(entry) + elif category == ChangeCategory.POTENTIALLY_BREAKING: + report.potentially_breaking.append(entry) + else: + report.safe_changes.append(entry) + + self._generate_recommendations(report) + return report + + + def _classify_change(self, change: Dict[str, Any]) -> Tuple[Severity, ChangeCategory, str, str]: + change_type = change.get('change_type') + entity_type = change.get('entity_type') + uri = change.get('uri') + + if change_type == 'removed': + if entity_type == 'class': + return (Severity.CRITICAL, ChangeCategory.BREAKING, f"Class {uri} removed.", "Migrate orphaned instances.") + return (Severity.CRITICAL, ChangeCategory.BREAKING, f"Property {uri} removed.", "Migrate property values.") + + if change_type == 'added': + return (Severity.INFO, ChangeCategory.NON_BREAKING, f"New {entity_type} {uri} added.", "No action required.") + + if change_type == 'modified': + return self._analyze_field_changes(uri, change.get('changes', {})) + + return (Severity.LOW, ChangeCategory.UNKNOWN, f"Unknown change for {uri}", "Manual review required.") + + def _analyze_field_changes(self, uri: str, field_changes: Dict[str, Any]) -> Tuple[Severity, ChangeCategory, str, str]: + has_restriction = False + has_structural = False + + for field, vals in field_changes.items(): + if field in self.VALIDITY_CONSTRAINTS: + old_val, new_val = vals.get("old"), vals.get("new") + + if old_val is None or new_val is None: + has_restriction = True + continue + + # if new constraint is smaller, it is a restriction + old_set = set(old_val) if isinstance(old_val, list) else {old_val} + new_set = set(new_val) if isinstance(new_val, list) else {new_val} + + if new_set < old_set: + has_restriction = True + + elif field in self.STRUCTURAL_FIELDS: + has_structural = True + + if has_restriction: + return (Severity.HIGH, ChangeCategory.BREAKING, f"Domain/range restricted on {uri}", "Validate existing data against new constraints.") + if has_structural: + return (Severity.MEDIUM, ChangeCategory.POTENTIALLY_BREAKING, f"Hierarchy modified for {uri}", "Check dependent reasoning chains.") + + return (Severity.LOW, ChangeCategory.NON_BREAKING, f"Safe annotations updated for {uri}", "No action required.") + + def _generate_recommendations(self, report: ImpactReport) -> None: + if report.breaking_changes: + report.recommendations.append("[BREAKING] Schedule downtime or validate existing data.") + if report.potentially_breaking: + report.recommendations.append("[WARNING] Run full regression tests on queries.") + if not report.breaking_changes and not report.potentially_breaking: + report.recommendations.append("[SAFE] Minor version bump sufficient.") + + +def generate_change_report(diff: Dict[str, Any]) -> Dict[str, Any]: + """Public API for generating impact reports from diffs.""" + analyzer = ChangeLogAnalyzer() + return analyzer.analyze(diff).to_dict() diff --git a/semantica/change_management/ontology_version_manager.py b/semantica/change_management/ontology_version_manager.py index 57cdbf98..2ff0a4db 100644 --- a/semantica/change_management/ontology_version_manager.py +++ b/semantica/change_management/ontology_version_manager.py @@ -295,6 +295,98 @@ class VersionManager: "axioms_removed": len(axioms_removed) } } + + def diff_ontologies(self, base: Dict[str, Any], target: Dict[str, Any]) -> Dict[str, Any]: + """ + Computes a structured diff between two ontology versions. + """ + def _compute_section_diff(base_list, target_list): + + base_map = {} + for item in base_list: + if isinstance(item, dict): + key = item.get("uri") or item.get("name") + if key: + base_map[key] = item + elif isinstance(item, str): + base_map[item] = {"uri": item} + + target_map = {} + for item in target_list: + if isinstance(item, dict): + key = item.get("uri") or item.get("name") + if key: + target_map[key] = item + elif isinstance(item, str): + target_map[item] = {"uri": item} + + added, removed, changed = [], [], [] + + # Find Added and Changed + for key, t_item in target_map.items(): + if key not in base_map: + added.append(t_item) + else: + b_item = base_map[key] + changes = {} + + all_fields = set(b_item.keys()).union(t_item.keys()) + for field in all_fields: + if field in ["uri", "name"]: + continue + + b_val = b_item.get(field) + t_val = t_item.get(field) + + # Deep equality check for lists + if isinstance(b_val, list) and isinstance(t_val, list): + if set(str(x) for x in b_val) != set(str(x) for x in t_val): + changes[field] = {"old": b_val, "new": t_val} + elif b_val != t_val: + changes[field] = {"old": b_val, "new": t_val} + + if changes: + changed.append({ + "uri": t_item.get("uri") or key, + "name": t_item.get("name") or key, + "changes": changes + }) + + # Find deleted + for key, b_item in base_map.items(): + if key not in target_map: + removed.append(b_item) + + return added, removed, changed + + + classes_added, classes_removed, classes_changed = _compute_section_diff( + base.get("classes", []), target.get("classes", []) + ) + props_added, props_removed, props_changed = _compute_section_diff( + base.get("properties", []), target.get("properties", []) + ) + inds_added, inds_removed, inds_changed = _compute_section_diff( + base.get("individuals", []), target.get("individuals", []) + ) + axioms_added, axioms_removed, axioms_changed = _compute_section_diff( + base.get("axioms", []), target.get("axioms", []) + ) + + return { + "added_classes": classes_added, + "removed_classes": classes_removed, + "changed_classes": classes_changed, + "added_properties": props_added, + "removed_properties": props_removed, + "changed_properties": props_changed, + "added_individuals": inds_added, + "removed_individuals": inds_removed, + "changed_individuals": inds_changed, + "added_axioms": axioms_added, + "removed_axioms": axioms_removed, + "changed_axioms": axioms_changed, + } def get_version(self, version: str) -> Optional[OntologyVersion]: """Get version by version string.""" diff --git a/semantica/context/decision_query.py b/semantica/context/decision_query.py index fc496be2..16efdd16 100644 --- a/semantica/context/decision_query.py +++ b/semantica/context/decision_query.py @@ -951,10 +951,11 @@ class DecisionQuery: for measure_type, measure_data in centrality_measures.items(): if isinstance(measure_data, dict) and 'centrality' in measure_data: - decision_measures[measure_type] = measure_data['centrality'].get(decision_id, 0.0) - + val = measure_data['centrality'].get(decision_id, 0.0) + decision_measures[measure_type] = val if isinstance(val, (int, float)) else 0.0 + analysis["centrality_measures"] = decision_measures - + # Calculate overall influence score measures = analysis["centrality_measures"] analysis["influence_score"] = ( diff --git a/semantica/ingest/db_ingestor.py b/semantica/ingest/db_ingestor.py index af23f018..460e0fcb 100644 --- a/semantica/ingest/db_ingestor.py +++ b/semantica/ingest/db_ingestor.py @@ -34,10 +34,6 @@ from datetime import datetime from typing import Any, Dict, List, Optional from urllib.parse import urlparse -import sqlalchemy -from sqlalchemy import create_engine, inspect, text -from sqlalchemy.engine import Engine - from ..utils.exceptions import ProcessingError, ValidationError from ..utils.logging import get_logger from ..utils.progress_tracker import get_progress_tracker @@ -101,13 +97,13 @@ class DatabaseConnector: self.logger = get_logger("database_connector") self.db_type = db_type.lower() if db_type else "" self.config = config - self.engine: Optional[Engine] = None + self.engine: Optional[Any] = None self.logger.debug( f"Database connector initialized: db_type={db_type or 'auto-detect'}" ) - def connect(self, connection_string: str) -> Engine: + def connect(self, connection_string: str) -> Any: """ Establish database connection. @@ -129,6 +125,14 @@ class DatabaseConnector: ProcessingError: If connection fails or database type is unsupported """ try: + try: + from sqlalchemy import create_engine, text + except ImportError: + raise ProcessingError( + "sqlalchemy is required for database ingestion. " + "Install with: pip install sqlalchemy" + ) + # Parse connection string to detect database type parsed = urlparse(connection_string) @@ -188,6 +192,7 @@ class DatabaseConnector: bool: True if connection successful, False otherwise """ try: + from sqlalchemy import create_engine, text engine = create_engine(connection_string) with engine.connect() as conn: conn.execute(text("SELECT 1")) @@ -226,7 +231,7 @@ class DataExporter: def export_table_data( self, - connection: Engine, + connection: Any, table_name: str, schema: Optional[str] = None, limit: Optional[int] = None, @@ -264,6 +269,7 @@ class DataExporter: ProcessingError: If table export fails """ try: + from sqlalchemy import inspect inspector = inspect(connection) # Get column information @@ -379,7 +385,7 @@ class DataExporter: return transformed def export_schema( - self, connection: Engine, schema: Optional[str] = None + self, connection: Any, schema: Optional[str] = None ) -> Dict[str, Any]: """ Export database schema information. @@ -406,6 +412,7 @@ class DataExporter: ProcessingError: If schema export fails """ try: + from sqlalchemy import inspect inspector = inspect(connection) schema_info = {"tables": [], "views": [], "foreign_keys": []} @@ -591,6 +598,7 @@ class DBIngestor: schema = self.analyze_schema(connection_string) # Get all table names + from sqlalchemy import inspect inspector = inspect(engine) all_tables = inspector.get_table_names() diff --git a/semantica/ontology/engine.py b/semantica/ontology/engine.py index 6635d03b..dbec4883 100644 --- a/semantica/ontology/engine.py +++ b/semantica/ontology/engine.py @@ -10,6 +10,7 @@ from .owl_generator import OWLGenerator from .ontology_evaluator import OntologyEvaluator from .ontology_validator import OntologyValidator from .llm_generator import LLMOntologyGenerator +from ..semantic_extract.triplet_extractor import Triplet class OntologyEngine: @@ -25,6 +26,11 @@ class OntologyEngine: self.evaluator = OntologyEvaluator(**config) self.validator = OntologyValidator(**config) self.llm = LLMOntologyGenerator(**config) + self.store = config.get("store") + + # Deferred to avoid circular import: change_management โ†’ ontology โ†’ change_management + from ..change_management.ontology_version_manager import VersionManager + self.version_manager = config.get("version_manager") or VersionManager(**config) def from_data(self, data: Dict[str, Any], **options) -> Dict[str, Any]: tracking_id = self.progress.start_tracking( @@ -56,7 +62,135 @@ class OntologyEngine: **options, ) -> List[Dict[str, Any]]: return self.propgen.infer_properties(entities, relationships, classes, **options) + + def _sanitize_uri(self, uri: str) -> str: + """Prevent SPARQL injection by percent-encoding dangerous characters.""" + if not isinstance(uri, str): + return "" + return uri.replace("<", "%3C").replace(">", "%3E") + + def create_alignment(self, source_uri: str, target_uri: str, predicate: str, **options) -> None: + """ + Creates an alignment between two ontology entities and stores it. + """ + if not self.store: + raise ProcessingError("TripletStore instance not configured in OntologyEngine.") + if not predicate.startswith(("http://", "https://")): + raise ProcessingError( + f"predicate must be a full URI (e.g. 'http://www.w3.org/2002/07/owl#equivalentClass'), " + f"not a CURIE: '{predicate}'" + ) + + tracking_id = self.progress.start_tracking( + module="ontology", + submodule="OntologyEngine", + message=f"Creating alignment: {source_uri} -> {target_uri}" + ) + try: + triplet = Triplet(subject=source_uri, predicate=predicate, object=target_uri) + self.store.add_triplet(triplet, **options) + + self.progress.stop_tracking(tracking_id, status="completed", message="Alignment created") + except Exception as e: + self.progress.stop_tracking(tracking_id, status="failed", message=str(e)) + self.logger.error(f"Failed to create alignment: {e}") + raise ProcessingError(f"Alignment creation failed: {e}") + + def get_alignments(self, entity_uri: str, **options) -> List[Dict[str, Any]]: + """ + Retrieves all alignments for a specific entity URI (bidirectional). + """ + if not self.store: + raise ProcessingError("TripletStore instance not configured in OntologyEngine.") + + safe_uri = self._sanitize_uri(entity_uri) + + query = f""" + SELECT ?s ?p ?o WHERE {{ + {{ <{safe_uri}> ?p ?o . BIND(<{safe_uri}> AS ?s) }} + UNION + {{ ?s ?p <{safe_uri}> . BIND(<{safe_uri}> AS ?o) }} + + FILTER (?p IN ( + , + , + , + , + , + , + , + + )) + }} + """ + try: + results = self.store.execute_query(query, **options) + + alignments = [] + if hasattr(results, 'bindings'): + for b in results.bindings: + alignments.append({ + "source": b.get("s", {}).get("value") if isinstance(b.get("s"), dict) else b.get("s"), + "predicate": b.get("p", {}).get("value") if isinstance(b.get("p"), dict) else b.get("p"), + "target": b.get("o", {}).get("value") if isinstance(b.get("o"), dict) else b.get("o") + }) + return alignments + except Exception as e: + self.logger.error(f"Failed to get alignments for {entity_uri}: {e}") + raise ProcessingError(f"Failed to get alignments: {e}") + + def list_alignments(self, ontology_uri: Optional[str] = None, **options) -> List[Dict[str, Any]]: + """ + Lists all alignments, optionally filtered by an ontology URI. + """ + if not self.store: + raise ProcessingError("TripletStore instance not configured in OntologyEngine.") + + filter_clause = "" + if ontology_uri: + # Sanitize characters that could break out of the SPARQL string literal or WHERE block + safe_ontology_uri = ( + ontology_uri + .replace("\\", "%5C") + .replace('"', '%22') + .replace("{", "%7B") + .replace("}", "%7D") + ) + filter_clause = f'FILTER(STRSTARTS(STR(?s), "{safe_ontology_uri}") || STRSTARTS(STR(?o), "{safe_ontology_uri}"))' + + query = f""" + SELECT ?s ?p ?o WHERE {{ + ?s ?p ?o . + FILTER (?p IN ( + , + , + , + , + , + , + , + + )) + {filter_clause} + }} + """ + try: + results = self.store.execute_query(query, **options) + + alignments = [] + if hasattr(results, 'bindings'): + for b in results.bindings: + alignments.append({ + "source": b.get("s", {}).get("value") if isinstance(b.get("s"), dict) else b.get("s"), + "predicate": b.get("p", {}).get("value") if isinstance(b.get("p"), dict) else b.get("p"), + "target": b.get("o", {}).get("value") if isinstance(b.get("o"), dict) else b.get("o") + }) + return alignments + except Exception as e: + self.logger.error(f"Failed to list alignments: {e}") + raise ProcessingError(f"Failed to list alignments: {e}") + def evaluate(self, ontology: Dict[str, Any], **options): return self.evaluator.evaluate_ontology(ontology, **options) @@ -69,3 +203,70 @@ class OntologyEngine: def export_owl(self, ontology: Dict[str, Any], path: str, format: str = "turtle"): return self.owl.export_owl(ontology, path, format=format) + def get_ontology_version_dict(self, version_id: str) -> Dict[str, Any]: + """Utility to load an ontology version as plain dict ready for diffing.""" + version_record = self.version_manager.get_version(version_id) + if not version_record: + raise ProcessingError(f"Version {version_id} not found.") + return version_record.metadata.get("structure", {"classes": [], "properties": []}) + + def compare_versions(self, base_id: str, target_id: str, **options) -> Dict[str, Any]: + """ + Orchestrates version loading, diff computation, and report generation. + + Args: + base_id: Version ID of the old ontology + target_id: Version ID of the new ontology + **options: Can pass 'base_dict' and 'target_dict' directly to bypass loading. + Can pass 'run_validation=True' to validate schema. + Can pass 'graph_data' to validate instances against new schema. + + Returns: + A structured dictionary containing the impact report and machine-readable diff. + """ + tracking_id = self.progress.start_tracking( + module="ontology", + submodule="OntologyEngine", + message=f"Comparing ontology versions: {base_id} -> {target_id}" + ) + + try: + # Deferred to avoid circular import + from ..change_management.change_log import generate_change_report + from ..kg.graph_validator import GraphValidator + + base_dict = options.get("base_dict") or self.get_ontology_version_dict(base_id) + target_dict = options.get("target_dict") or self.get_ontology_version_dict(target_id) + + diff_result = self.version_manager.diff_ontologies(base_dict, target_dict) + report = generate_change_report(diff_result) + report["diff"] = diff_result + + if options.get("run_validation"): + self.progress.update_tracking(tracking_id, message="Running validation on target schema...") + val_res = self.validate(target_dict, **options) + report["validation_results"] = { + "valid": getattr(val_res, "valid", getattr(val_res, "is_valid", False)), + "consistent": getattr(val_res, "consistent", True), + "satisfiable": getattr(val_res, "satisfiable", True), + "errors": getattr(val_res, "errors", []), + "warnings": getattr(val_res, "warnings", []) + } + + if "graph_data" in options: + self.progress.update_tracking(tracking_id, message="Running graph data validation...") + kg_validator = GraphValidator(**self.config) + kg_res = kg_validator.validate(options["graph_data"], ontology=target_dict, **options) + report["graph_validation"] = { + "valid": getattr(kg_res, "valid", getattr(kg_res, "is_valid", False)), + "errors": getattr(kg_res, "errors", []), + "warnings": getattr(kg_res, "warnings", []) + } + + self.progress.stop_tracking(tracking_id, status="completed", message="Comparison complete") + return report + + except Exception as e: + self.progress.stop_tracking(tracking_id, status="failed", message=str(e)) + self.logger.error(f"Failed to compare versions: {e}") + raise ProcessingError(f"Version comparison failed: {e}") from e diff --git a/semantica/ontology/namespace_manager.py b/semantica/ontology/namespace_manager.py index fdacbfd9..614492a7 100644 --- a/semantica/ontology/namespace_manager.py +++ b/semantica/ontology/namespace_manager.py @@ -206,6 +206,31 @@ class NamespaceManager: Dictionary of prefix -> URI mappings """ return dict(self.namespaces) + + def get_alignment_predicates(self) -> Dict[str, str]: + """ + Get standard alignment predicates for ontology mapping. + + Returns: + Dictionary mapping common alignment types to their full URIs. + """ + owl_ns = self.get_namespace("owl") + skos_ns = self.get_namespace("skos") + + return { + #OWL alignments + "equivalentClass": f"{owl_ns}equivalentClass", + "equivalentProperty": f"{owl_ns}equivalentProperty", + "sameAs": f"{owl_ns}sameAs", + #SKOS alignments + "exactMatch": f"{skos_ns}exactMatch", + "closeMatch": f"{skos_ns}closeMatch", + "broadMatch": f"{skos_ns}broadMatch", + "narrowMatch": f"{skos_ns}narrowMatch", + "relatedMatch": f"{skos_ns}relatedMatch", + + } + def _to_pascal_case(self, name: str) -> str: """Convert name to PascalCase.""" diff --git a/semantica/ontology/reuse_manager.py b/semantica/ontology/reuse_manager.py index 2c9b546e..22e95138 100644 --- a/semantica/ontology/reuse_manager.py +++ b/semantica/ontology/reuse_manager.py @@ -363,6 +363,54 @@ class ReuseManager: def list_known_ontologies(self) -> List[str]: """List known ontology URIs.""" return list(self.known_ontologies.keys()) + + def suggest_alignments( + self, target: Dict[str, Any], source: Dict[str, Any], **options + ) -> List[Dict[str, str]]: + """ + Suggest alignments between a source and target ontology based on heuristics. + """ + suggestions = [] + + # Nested function for DRY + + def find_matches(target_items, source_items, entity_type): + predicate = ( + "http://www.w3.org/2002/07/owl#equivalentClass" + if entity_type == "class" + else "http://www.w3.org/2002/07/owl#equivalentProperty" + ) + + # Build hash map of target entities by normalized name + target_map = {} + for t_item in target_items: + t_uri = t_item.get("uri") + t_name = t_item.get("name", "").strip().lower() + if t_uri and t_name: + target_map.setdefault(t_name, []).append(t_uri) + + # Single pass through source items checking the hash map + for s_item in source_items: + s_uri = s_item.get("uri") + s_name = s_item.get("name", "").strip().lower() + if not s_uri or not s_name: + continue + + if s_name in target_map: + for t_uri in target_map[s_name]: + if s_uri != t_uri: + suggestions.append({ + "source_uri": s_uri, + "target_uri": t_uri, + "predicate": predicate, + "reason": f"Exact label match for {entity_type}: '{s_item.get('name')}'" + }) + + find_matches(target.get("classes", []), source.get("classes", []), "class") + find_matches(target.get("properties", []), source.get("properties", []), "property") + + return suggestions + def merge_ontology_data( self, target: Dict[str, Any], source: Dict[str, Any], **options @@ -437,6 +485,16 @@ class ReuseManager: for imp in source["imports"]: if imp not in target["imports"]: target["imports"].append(imp) + + if options.get("compute_alignments", False): + self.progress_tracker.update_tracking( + tracking_id, message="Computing suggested alignments..." + ) + suggested = self.suggest_alignments(target, source, **options) + if suggested: + if "suggested_alignments" not in target: + target["suggested_alignments"] = [] + target["suggested_alignments"].extend(suggested) self.progress_tracker.stop_tracking( tracking_id, diff --git a/semantica/parse/pdf_parser.py b/semantica/parse/pdf_parser.py index 4a087304..7ad31ced 100644 --- a/semantica/parse/pdf_parser.py +++ b/semantica/parse/pdf_parser.py @@ -33,9 +33,6 @@ from dataclasses import dataclass, field from pathlib import Path from typing import Any, Dict, List, Optional, Union -import pdfplumber -from PIL import Image - from ..utils.exceptions import ProcessingError, ValidationError from ..utils.logging import get_logger from ..utils.progress_tracker import get_progress_tracker @@ -119,6 +116,13 @@ class PDFParser: raise ValidationError(f"File is not a PDF: {file_path}") try: + try: + import pdfplumber + except ImportError: + raise ProcessingError( + "pdfplumber is required for PDF parsing. " + "Install with: pip install pdfplumber" + ) with pdfplumber.open(str(file_path)) as pdf: # Extract metadata metadata = self._extract_metadata(pdf) diff --git a/semantica/parse/pptx_parser.py b/semantica/parse/pptx_parser.py index cd30114c..7628f1d1 100644 --- a/semantica/parse/pptx_parser.py +++ b/semantica/parse/pptx_parser.py @@ -32,8 +32,6 @@ from dataclasses import dataclass, field from pathlib import Path from typing import Any, Dict, List, Optional, Union -from pptx import Presentation - from ..utils.exceptions import ProcessingError, ValidationError from ..utils.logging import get_logger from ..utils.progress_tracker import get_progress_tracker @@ -97,6 +95,13 @@ class PPTXParser: raise ValidationError(f"File is not a PPTX: {file_path}") try: + try: + from pptx import Presentation + except ImportError: + raise ProcessingError( + "python-pptx is required for PPTX parsing. " + "Install with: pip install python-pptx" + ) prs = Presentation(str(file_path)) # Extract metadata @@ -220,7 +225,7 @@ class PPTXParser: images=images, ) - def _extract_metadata(self, prs: Presentation) -> Dict[str, Any]: + def _extract_metadata(self, prs: Any) -> Dict[str, Any]: """Extract presentation metadata.""" metadata = {} diff --git a/semantica/reasoning/__init__.py b/semantica/reasoning/__init__.py index 2096d86a..a72da515 100644 --- a/semantica/reasoning/__init__.py +++ b/semantica/reasoning/__init__.py @@ -3,7 +3,8 @@ Reasoning Module This module provides reasoning and inference capabilities for knowledge graph analysis and query answering, supporting multiple reasoning strategies including -rule-based inference via Rete, SPARQL reasoning, abductive and deductive reasoning. +rule-based inference via Rete, SPARQL reasoning, abductive and deductive reasoning, +and native Datalog evaluation. """ from .reasoner import Reasoner, InferenceResult, Rule, Fact, RuleType @@ -25,6 +26,8 @@ from .rete_engine import ( ) from .sparql_reasoner import SPARQLQueryResult, SPARQLReasoner +from .datalog_reasoner import DatalogReasoner, DatalogFact, DatalogRule + __all__ = [ # Reasoner facade "Reasoner", @@ -43,10 +46,14 @@ __all__ = [ # SPARQL reasoning "SPARQLReasoner", "SPARQLQueryResult", + # Datalog reasoning + "DatalogReasoner", + "DatalogFact", + "DatalogRule", # Explanation "ExplanationGenerator", "Explanation", "ReasoningStep", "ReasoningPath", "Justification", -] +] \ No newline at end of file diff --git a/semantica/reasoning/datalog_reasoner.py b/semantica/reasoning/datalog_reasoner.py new file mode 100644 index 00000000..fcb042e9 --- /dev/null +++ b/semantica/reasoning/datalog_reasoner.py @@ -0,0 +1,432 @@ +""" +Datalog reasoner module + +This module provides a native Datalog engine using bottom-up semi-naive fixpoint evaluation. +It supports recursive rules, multi-hop inference, and guarantees termination on finite graphs. +""" + +import re +from collections import defaultdict +from dataclasses import dataclass +from typing import Any, Dict, List, NamedTuple, Optional, Set, Tuple, Union + +from ..utils.logging import get_logger +from ..utils.progress_tracker import get_progress_tracker + +# data structs + +@dataclass(frozen=True) +class DatalogFact: + """Represents a ground truth fact.""" + predicate: str + args: Tuple[str, ...] + +class BodyAtom(NamedTuple): + """Represents a single predicate condition in a rule's body.""" + predicate: str + args: Tuple[str, ...] + +@dataclass +class DatalogRule: + """Represents a Horn clause rule.""" + head_predicate: str + head_args: Tuple[str, ...] + body: List[BodyAtom] + + +# datalog reasoner + +class DatalogReasoner: + """ + Datalog reasoning engine supporting recursive rule evaluation via semi-naive + bottom-up fixpoint computation. + """ + + def __init__(self, config: Optional[Dict[str, Any]] = None, **kwargs): + self.logger = get_logger("datalog_reasoner") + self.config = config or {} + self.config.update(kwargs) + + self.progress_tracker = get_progress_tracker() + + self._fact_index: Dict[str, Set[DatalogFact]] = defaultdict(set) + self._all_facts: Set[DatalogFact] = set() + + self._rules: List[DatalogRule] = [] + self._derived: bool = False + + self._delta_old: Set[DatalogFact] = set() + self._delta_new: Set[DatalogFact] = set() + + def clear(self) -> None: + """Clear all facts and rules from the engine.""" + self._fact_index.clear() + self._all_facts.clear() + self._rules.clear() + self._derived = False + self._delta_old.clear() + self._delta_new.clear() + + def add_fact(self, fact: Any) -> None: + """ + Add a ground fact to the engine. + Accepts strings like "parent(tom, bob)" or standard Semantica Dicts. + """ + parsed_fact = None + + if isinstance(fact, str): + parsed_fact = self._parse_fact_string(fact) + elif isinstance(fact, dict): + if "subject" in fact and "predicate" in fact and "object" in fact: + parsed_fact = DatalogFact( + predicate=str(fact["predicate"]).replace(' ', '_').lower(), + args=(str(fact["subject"]).replace(' ', '_').lower(), str(fact["object"]).replace(' ', '_').lower()) + ) + elif "source" in fact or "source_id" in fact or "source_name" in fact: + source = fact.get("source", fact.get("source_name", fact.get("source_id"))) + target = fact.get("target", fact.get("target_name", fact.get("target_id"))) + rtype = fact.get("type", fact.get("relation", "connected_to")) + if source and target: + parsed_fact = DatalogFact( + predicate=str(rtype).replace(' ', '_').lower(), + args=(str(source).replace(' ', '_').lower(), str(target).replace(' ', '_').lower()) + ) + + elif "type" in fact and ("id" in fact or "name" in fact): + name = fact.get("id", fact.get("name")) + etype = fact.get("type", "entity") + if name: + parsed_fact = DatalogFact( + predicate=str(etype).replace(' ', '_').lower(), + args=(str(name).replace(' ', '_').lower(),) + ) + + + if parsed_fact: + for arg in parsed_fact.args: + if not arg: + raise ValueError("Facts cannot contain empty arguments") + if arg[0].isupper(): + raise ValueError(f"Facts must be constants only. Found variable '{arg}' in {fact}") + + if parsed_fact is None and isinstance(fact, dict): + self.logger.warning(f"Unrecognised dict fact format, skipping: {fact}") + return + + if parsed_fact and parsed_fact not in self._all_facts: + self._all_facts.add(parsed_fact) + self._fact_index[parsed_fact.predicate].add(parsed_fact) + self._derived = False + + def add_rule(self, rule_str: str) -> None: + """ Add a Datalog rule using Horn clause syntax.""" + rule = self._parse_rule_string(rule_str) + self._rules.append(rule) + self._derived = False + + # Parsing helpers + + def _parse_fact_string(self, s: str) -> DatalogFact: + """Parse 'predicate(arg1, arg2)' into a DatalogFact.""" + match = re.match(r'^\s*([a-zA-Z0-9_]+)\s*\(\s*([^)]+)\s*\)\s*\.?\s*$', s.strip()) + if not match: + raise ValueError(f"Invalid fact syntax: {s}") + + predicate = match.group(1) + args_str = match.group(2) + args = tuple(arg.strip() for arg in args_str.split(',')) + + for arg in args: + if not arg: + raise ValueError(f"Empty argument found in fact: {s}") + if arg[0].isupper(): + raise ValueError(f"Facts must be constants only (no variables). Found variable '{arg}' in {s}") + + return DatalogFact(predicate, args) + + def _parse_rule_string(self, s: str) -> DatalogRule: + """Parse 'head(X, Y) :- body1(X, Z), body2(Z, Y).' into a DatalogRule.""" + s = s.strip() + if ":-" not in s: + raise ValueError(f"Invalid rule syntax (missing ':-'): {s}") + + head_str, body_str = s.split(":-", 1) + head_str = head_str.strip() + body_str = body_str.strip().rstrip('.') + + # Parse head + head_match = re.match(r'^([a-zA-Z0-9_]+)\s*\(\s*([^)]+)\s*\)$', head_str) + if not head_match: + raise ValueError(f"Invalid rule head syntax: {head_str}") + + head_pred = head_match.group(1) + head_args = tuple(arg.strip() for arg in head_match.group(2).split(',')) + + # Parse body atoms + body = [] + atom_matches = re.findall(r'([a-zA-Z0-9_]+)\s*\(\s*([^)]+)\s*\)', body_str) + if not atom_matches: + raise ValueError(f"No valid body atoms found in rule: {s}") + + for pred, args_str in atom_matches: + args = tuple(arg.strip() for arg in args_str.split(',')) + body.append(BodyAtom(pred, args)) + + return DatalogRule(head_pred, head_args, body) + + # Unification & Instantiation + + + def _is_variable(self, term: str) -> bool: + """Variables strictly start with an uppercase letter.""" + return bool(term and term[0].isupper()) + + def _unify( + self, + pattern_args: Tuple[str, ...], + fact_args: Tuple[str, ...], + bindings: Dict[str, str] + ) -> Optional[Dict[str, str]]: + """ + Unifies a rule atom's pattern with a concrete fact. + Optimized to prevent unnecessary dictionary allocations. + """ + if len(pattern_args) != len(fact_args): + return None + + new_additions = {} + + for p_arg, f_arg in zip(pattern_args, fact_args): + if self._is_variable(p_arg): + if p_arg in bindings: + if bindings[p_arg] != f_arg: + return None + elif p_arg in new_additions: + if new_additions[p_arg] != f_arg: + return None + else: + new_additions[p_arg] = f_arg + else: + if p_arg != f_arg: + return None + + if new_additions: + return {**bindings, **new_additions} + return bindings + + def _instantiate(self, args: Tuple[str, ...], bindings: Dict[str, str]) -> Optional[Tuple[str, ...]]: + """Replaces variables in a tuple with their bound values.""" + result = [] + for arg in args: + if self._is_variable(arg): + if arg not in bindings: + return None + result.append(bindings[arg]) + else: + result.append(arg) + + return tuple(result) + + def _instantiate_fact( + self, predicate: str, args: Tuple[str, ...], bindings: Dict[str, str] + ) -> Optional[DatalogFact]: + """Creates a concrete DatalogFact from a predicate, arguments, and bindings.""" + ground_args = self._instantiate(args, bindings) + if ground_args is None: + return None + return DatalogFact(predicate, ground_args) + + # Semi-Naive Fixpoint Evaluation + + + def derive_all(self) -> List[str]: + """ + Executes bottom-up semi-naive evaluation until fixpoint is reached. + Returns a list of all derived facts as strings. + """ + if self._derived: + return [f"{f.predicate}({', '.join(f.args)})" for f in self._all_facts] + + tracking_id = self.progress_tracker.start_tracking( + module="reasoning", + submodule="DatalogReasoner", + message="Starting semi-naive fixpoint evaluation" + ) + + iteration = 0 + newly_derived_count = 0 + + try: + self._delta_new = self._all_facts.copy() + + while self._delta_new: + iteration += 1 + + # Shift deltas + self._delta_old = self._delta_new + self._delta_new = set() + + delta_index = defaultdict(set) + for f in self._delta_old: + delta_index[f.predicate].add(f) + + for rule in self._rules: + new_facts = self._apply_rule(rule, delta_index) + + for fact in new_facts: + if fact not in self._all_facts: + self._delta_new.add(fact) + self._all_facts.add(fact) + self._fact_index[fact.predicate].add(fact) + newly_derived_count += 1 + + self.logger.debug(f"Datalog Iteration {iteration}: derived {len(self._delta_new)} new facts") + + self._derived = True + finally: + self.progress_tracker.stop_tracking( + tracking_id, + status="completed", + message=f"Fixpoint reached in {iteration} iterations. {newly_derived_count} new facts derived." + ) + + return [f"{f.predicate}({', '.join(f.args)})" for f in self._all_facts] + + def _apply_rule( + self, rule: DatalogRule, delta_index: Optional[Dict[str, Set[DatalogFact]]] = None + ) -> Set[DatalogFact]: + """ + Evaluates a single rule. + Uses semi-naive strategy if delta_index is provided, otherwise falls back to naive evaluation. + """ + results = set() + + if not rule.body: + fact = self._instantiate_fact(rule.head_predicate, rule.head_args, {}) + if fact: + results.add(fact) + return results + + is_seminaive = delta_index is not None + evaluation_paths = range(len(rule.body)) if is_seminaive else [0] + + for delta_index_pos in evaluation_paths: + bindings_list = [{}] + + for i, atom in enumerate(rule.body): + new_bindings_list = [] + + if is_seminaive and i == delta_index_pos: + candidate_facts = delta_index.get(atom.predicate, set()) + else: + candidate_facts = self._fact_index.get(atom.predicate, set()) + + for bindings in bindings_list: + for fact in candidate_facts: + merged_bindings = self._unify(atom.args, fact.args, bindings) + if merged_bindings is not None: + new_bindings_list.append(merged_bindings) + + bindings_list = new_bindings_list + if not bindings_list: + break + + for final_bindings in bindings_list: + head_fact = self._instantiate_fact(rule.head_predicate, rule.head_args, final_bindings) + if head_fact: + results.add(head_fact) + + return results + + # Query & ContextGraph Integration + + + def query(self, pattern: str, bindings: dict = None) -> List[dict]: + """ + Queries the derived fact set. Automatically runs derive_all() if rules exist. + Syntax: "ancestor(tom, ?Y)" or "ancestor(tom, ?y)" + Returns: [{"Y": "bob"}] or [{"y": "bob"}] + """ + if self._rules and not self._derived: + self.derive_all() + + match = re.match(r'^\s*([a-zA-Z0-9_]+)\s*\(\s*([^)]+)\s*\)\s*\.?\s*$', pattern.strip()) + if not match: + raise ValueError(f"Invalid query syntax: {pattern}") + + pred = match.group(1) + raw_args = tuple(arg.strip() for arg in match.group(2).split(',')) + + query_vars = {} + pattern_args = [] + + for i, arg in enumerate(raw_args): + if arg.startswith('?'): + var_name = arg[1:] + if not var_name: + raise ValueError("Empty variable name after '?'") + internal_var = var_name[0].upper() + var_name[1:] + query_vars[i] = (var_name, internal_var) + pattern_args.append(internal_var) + elif self._is_variable(arg): + query_vars[i] = (arg, arg) + pattern_args.append(arg) + else: + pattern_args.append(arg) + + initial_bindings = {} + for k, v in (bindings or {}).items(): + internal_k = k[0].upper() + k[1:] if k and not k[0].isupper() else k + initial_bindings[internal_k] = v + + for i, arg in enumerate(pattern_args): + if self._is_variable(arg) and arg in initial_bindings: + pattern_args[i] = initial_bindings[arg] + + results = [] + candidates = self._fact_index.get(pred, set()) + + for fact in candidates: + match_bindings = self._unify(tuple(pattern_args), fact.args, {}) + if match_bindings is not None: + result_row = {} + for idx, (orig_var, internal_var) in query_vars.items(): + if internal_var in match_bindings: + result_row[orig_var] = match_bindings[internal_var] + elif internal_var in initial_bindings: + result_row[orig_var] = initial_bindings[internal_var] + + if result_row and result_row not in results: + results.append(result_row) + + return results + + def load_from_graph(self, graph: Any) -> int: + """ + Loads a ContextGraph into Datalog facts using central add_fact validation. + """ + initial_count = len(self._all_facts) + + if hasattr(graph, 'find_edges') and hasattr(graph, 'find_nodes'): + for edge_dict in graph.find_edges(): + self.add_fact(edge_dict) + for node_dict in graph.find_nodes(): + self.add_fact(node_dict) + else: + if hasattr(graph, 'edges'): + edges = graph.edges() if callable(graph.edges) else graph.edges + for edge in edges: + self.add_fact(edge if isinstance(edge, dict) else edge.__dict__) + + if hasattr(graph, 'nodes'): + nodes = graph.nodes() if callable(graph.nodes) else graph.nodes + if isinstance(nodes, dict): + nodes = nodes.values() + for node in nodes: + self.add_fact(node if isinstance(node, dict) else node.__dict__) + + facts_added = len(self._all_facts) - initial_count + self.logger.info(f"Loaded {facts_added} facts from ContextGraph.") + return facts_added + + \ No newline at end of file diff --git a/semantica/semantic_extract/config.py b/semantica/semantic_extract/config.py index 640ecb3c..fb3763e8 100644 --- a/semantica/semantic_extract/config.py +++ b/semantica/semantic_extract/config.py @@ -6,7 +6,7 @@ supporting multiple configuration sources including environment variables, confi and programmatic configuration. Supported Configuration Sources: - - Environment variables: OPENAI_API_KEY, GEMINI_API_KEY, GROQ_API_KEY, etc. + - Environment variables: OPENAI_API_KEY, GEMINI_API_KEY, GROQ_API_KEY, NOVITA_API_KEY, etc. - Config files: YAML, JSON, TOML formats - Programmatic: Python API for setting provider configurations @@ -97,7 +97,7 @@ class Config: def _load_env_vars(self): """Load configuration from environment variables.""" # Common environment variable patterns - providers = ["openai", "gemini", "groq", "anthropic", "ollama"] + providers = ["openai", "gemini", "groq", "anthropic", "ollama", "novita"] for provider in providers: env_key = f"{provider.upper()}_API_KEY" api_key = os.getenv(env_key) diff --git a/semantica/semantic_extract/providers.py b/semantica/semantic_extract/providers.py index 8163560e..bcf14aaa 100644 --- a/semantica/semantic_extract/providers.py +++ b/semantica/semantic_extract/providers.py @@ -988,6 +988,65 @@ class DeepSeekProvider(BaseProvider): except Exception as e: raise ProcessingError(f"Failed to parse JSON from DeepSeek response: {e}") + +class NovitaProvider(BaseProvider): + """Novita AI provider implementation - OpenAI-compatible API.""" + + def __init__(self, api_key: Optional[str] = None, model: str = "deepseek/deepseek-v3.2", **kwargs): + """Initialize Novita provider.""" + super().__init__(**kwargs) + self.api_key = api_key or config.get_api_key("novita") + self.model = model + self.base_url = "https://api.novita.ai/v1" + self.client = None + self._init_client() + + def _init_client(self): + try: + from openai import OpenAI + + if self.api_key: + self.client = OpenAI(api_key=self.api_key, base_url=self.base_url) + except (ImportError, OSError): + self.client = None + self.logger.warning( + "openai library not installed. Install with: pip install semantica[llm-openai]" + ) + + def is_available(self) -> bool: + return self.client is not None + + def generate(self, prompt: str, **kwargs) -> str: + if not self.client: + raise ProcessingError("Novita client not initialized. Set NOVITA_API_KEY or pass api_key.") + + create_kwargs = { + "model": kwargs.get("model", self.model), + "messages": [{"role": "user", "content": prompt}], + } + self._add_if_set(create_kwargs, kwargs, "temperature", "max_tokens") + + response = self.client.chat.completions.create(**create_kwargs) + return response.choices[0].message.content + + def generate_structured(self, prompt: str, **kwargs) -> Union[dict, list]: + """Generate structured output.""" + if not self.client: + raise ProcessingError("Novita client not initialized.") + + create_kwargs = { + "model": kwargs.get("model", self.model), + "messages": [{"role": "user", "content": prompt}], + "response_format": {"type": "json_object"}, + } + self._add_if_set(create_kwargs, kwargs, "temperature", "max_tokens") + + response = self.client.chat.completions.create(**create_kwargs) + try: + return self._parse_json(response.choices[0].message.content) + except Exception as e: + raise ProcessingError(f"Failed to parse JSON from Novita response: {e}") + class HuggingFaceLLMProvider(BaseProvider): """HuggingFace transformers for LLM tasks.""" @@ -1370,6 +1429,7 @@ class ProviderPool: "ollama": OllamaProvider, "huggingface_llm": HuggingFaceLLMProvider, "deepseek": DeepSeekProvider, + "novita": NovitaProvider, } provider_class = builtin.get(name.lower()) diff --git a/semantica/triplet_store/query_engine.py b/semantica/triplet_store/query_engine.py index b5ac307c..3e0dd315 100644 --- a/semantica/triplet_store/query_engine.py +++ b/semantica/triplet_store/query_engine.py @@ -267,6 +267,88 @@ class QueryEngine: execution_steps=execution_steps, metadata={"optimization_enabled": self.enable_optimization}, ) + + def expand_entity_uri(self, entity_uri: str, store_backend: Any, use_alignments: bool = False) -> List[str]: + """ + Expand an entity URI to include all aligned/equivalent entities. + + Args: + entity_uri: The original URI to expand + store_backend: Triplet store backend to query + use_alignments: If False, returns only the original URI + + Returns: + List of URIs including the original and any aligned entities + """ + if not use_alignments: + return [entity_uri] + + tracking_id = self.progress_tracker.start_tracking( + module="triplet_store", + submodule="QueryEngine", + message=f"Expanding alignments for: {entity_uri}" + ) + + # SPARQL query to find bidirectional alignments + + safe_uri = self._sanitize_uri(entity_uri) + query = f""" + SELECT DISTINCT ?aligned WHERE {{ + {{ <{safe_uri}> ?p ?aligned }} + UNION + {{ ?aligned ?p <{safe_uri}> }} + + FILTER (?p IN ( + , + , + , + , + , + , + , + + )) + }} + """ + + expanded_uris = set([entity_uri]) + try: + if hasattr(store_backend, "execute_sparql"): + result_data = store_backend.execute_sparql(query) + for binding in result_data.get("bindings", []): + val = binding.get("aligned", {}) + uri = val.get("value") if isinstance(val, dict) else val + if uri: + expanded_uris.add(uri) + else: + self.logger.warning( + "store_backend does not support execute_sparql; returning original URI only" + ) + self.progress_tracker.stop_tracking( + tracking_id, + status="completed", + message=f"Expanded to {len(expanded_uris)} URIs" + ) + except Exception as e: + self.logger.error(f"Failed to expand alignments for {entity_uri}: {e}") + self.progress_tracker.stop_tracking(tracking_id, status="failed", message=str(e)) + + return list(expanded_uris) + + def build_values_clause(self, variable_name: str, uris: List[str]) -> str: + """ + Helper to generate a SPARQL VALUES clause for a list of URIs. + Allows higher-level components to build alignment-aware queries. + + Example: + uris = engine.expand_entity_uri("http://ex.org/Person", store, use_alignments=True) + clause = engine.build_values_clause("subject", uris) + # Returns: VALUES ?subject { } + """ + if not uris: + return "" + formatted_uris = " ".join([f"<{self._sanitize_uri(uri)}>" for uri in uris]) + return f"VALUES ?{variable_name} {{ {formatted_uris} }}" def _validate_query(self, query: str) -> bool: """Validate SPARQL query syntax (basic).""" @@ -344,6 +426,12 @@ class QueryEngine: cache_key = self._get_cache_key(query) self.query_cache[cache_key] = result + + def _sanitize_uri(self, uri: str) -> str: + """Prevent SPARQL injection by percent-encoding dangerous characters.""" + if not isinstance(uri, str): + return "" + return uri.replace("<", "%3C").replace(">", "%3E") def clear_cache(self) -> None: """Clear query cache.""" diff --git a/tests/change_management/test_managers.py b/tests/change_management/test_managers.py index 2adcf6fe..34d78d56 100644 --- a/tests/change_management/test_managers.py +++ b/tests/change_management/test_managers.py @@ -9,10 +9,13 @@ import os import tempfile import pytest from semantica.change_management import ( - TemporalVersionManager, + TemporalVersionManager, OntologyVersionManager, ChangeLogEntry ) +from semantica.change_management.change_log import generate_change_report, ChangeLogAnalyzer +from semantica.change_management.ontology_version_manager import VersionManager +from semantica.ontology.engine import OntologyEngine from semantica.utils.exceptions import ValidationError, ProcessingError @@ -330,3 +333,142 @@ class TestOntologyVersionManager: finally: if os.path.exists(db_path): os.remove(db_path) + + + def test_diff_empty_ontologies(self): + """Test diffing entirely empty dictionaries.""" + manager = VersionManager() + diff = manager.diff_ontologies({}, {}) + + assert len(diff["added_classes"]) == 0 + assert len(diff["removed_classes"]) == 0 + assert len(diff["changed_classes"]) == 0 + + def test_diff_unordered_list_equality(self): + """Test that list order doesn't trigger a false positive change.""" + manager = VersionManager() + base = {"classes": [{"uri": "http://ex.org/C1", "domain": ["A", "B"]}]} + target = {"classes": [{"uri": "http://ex.org/C1", "domain": ["B", "A"]}]} + + diff = manager.diff_ontologies(base, target) + assert len(diff["changed_classes"]) == 0 + + def test_diff_missing_uris(self): + """Tests whether missing URIs fallback to 'name' deterministically.""" + manager = VersionManager() + base = {"classes": [{"name": "Person", "label": "Human"}]} + target = {"classes": [{"name": "Person", "label": "Homo Sapiens"}]} + + diff = manager.diff_ontologies(base, target) + assert len(diff["changed_classes"]) == 1 + + change = diff["changed_classes"][0] + assert change["name"] == "Person" + assert change["changes"]["label"]["old"] == "Human" + assert change["changes"]["label"]["new"] == "Homo Sapiens" + + +class TestChangeLogAnalyzer: + """Test cases for Impact Analysis & Reporting.""" + + def test_breaking_change_removed_class(self): + """Test that removing a class is flagged as CRITICAL/BREAKING.""" + diff = {"removed_classes": [{"uri": "http://ex.org/Person"}]} + report = generate_change_report(diff) + + assert len(report["impact_classification"]["breaking"]) == 1 + assert report["impact_classification"]["breaking"][0]["severity"] == "critical" + assert "removed" in report["impact_classification"]["breaking"][0]["description"] + + def test_breaking_change_narrowed_domain(self): + """Test that narrowing a domain is flagged as HIGH/BREAKING.""" + diff = { + "changed_properties": [{ + "uri": "http://ex.org/worksFor", + "changes": {"domain": {"old": ["Person", "Organization"], "new": ["Person"]}} + }] + } + report = generate_change_report(diff) + + assert len(report["impact_classification"]["breaking"]) == 1 + assert report["impact_classification"]["breaking"][0]["severity"] == "high" + assert "restricted" in report["impact_classification"]["breaking"][0]["description"] + + def test_safe_change_added_class_and_label(self): + """Test that adding classes and changing annotations is SAFE.""" + diff = { + "added_classes": [{"uri": "http://ex.org/NewClass"}], + "changed_properties": [{ + "uri": "http://ex.org/name", + "changes": {"label": {"old": "Name", "new": "Full Name"}} + }] + } + report = generate_change_report(diff) + + assert len(report["impact_classification"]["safe"]) == 2 + assert len(report["impact_classification"]["breaking"]) == 0 + assert len(report["impact_classification"]["potentially_breaking"]) == 0 + + +class TestOntologyEngineMigration: + """Test cases for Public API Orchestration.""" + + def test_compare_versions_with_dicts_override(self): + """Test for bypassing the DB fetch by passing dicts directly.""" + engine = OntologyEngine() + + base_dict = {"classes": [{"uri": "http://ex.org/C1", "label": "Old"}]} + target_dict = {"classes": [{"uri": "http://ex.org/C1", "label": "New"}]} + + # We pass fake version IDs ("v1", "v2"), but the engine should use our dicts + report = engine.compare_versions("v1", "v2", base_dict=base_dict, target_dict=target_dict) + + assert report["summary"]["total_changes"] == 1 + assert len(report["impact_classification"]["safe"]) == 1 + assert report["impact_classification"]["safe"][0]["entity_uri"] == "http://ex.org/C1" + + def test_compare_versions_version_not_found_raises(self): + """Test that compare_versions raises ProcessingError when version ID is not registered.""" + engine = OntologyEngine() + + with pytest.raises(ProcessingError): + engine.compare_versions("nonexistent_v1", "nonexistent_v2") + + def test_compare_versions_diff_includes_individuals_and_axioms(self): + """Test that the diff covers individuals and axioms, not just classes/properties.""" + engine = OntologyEngine() + + base_dict = { + "classes": [], + "properties": [], + "individuals": [{"uri": "http://ex.org/john"}], + "axioms": [{"uri": "http://ex.org/rule1", "expression": "Person hasName exactly 1 string"}], + } + target_dict = { + "classes": [], + "properties": [], + "individuals": [ + {"uri": "http://ex.org/john"}, + {"uri": "http://ex.org/jane"}, + ], + "axioms": [], + } + + report = engine.compare_versions("v1", "v2", base_dict=base_dict, target_dict=target_dict) + + diff = report["diff"] + assert any(i.get("uri") == "http://ex.org/jane" for i in diff["added_individuals"]) + assert any(a.get("uri") == "http://ex.org/rule1" for a in diff["removed_axioms"]) + + def test_compare_versions_null_constraint_value_flagged_as_breaking(self): + """Test that a constraint field going from None to a value is flagged as breaking.""" + diff = { + "changed_properties": [{ + "uri": "http://ex.org/worksFor", + "changes": {"domain": {"old": None, "new": ["Person"]}} + }] + } + report = generate_change_report(diff) + + assert len(report["impact_classification"]["breaking"]) == 1 + assert report["impact_classification"]["breaking"][0]["severity"] == "high" diff --git a/tests/integrations/__init__.py b/tests/integrations/__init__.py new file mode 100644 index 00000000..6917dc37 --- /dev/null +++ b/tests/integrations/__init__.py @@ -0,0 +1 @@ +# tests/integrations package diff --git a/tests/integrations/agno/__init__.py b/tests/integrations/agno/__init__.py new file mode 100644 index 00000000..74ab1267 --- /dev/null +++ b/tests/integrations/agno/__init__.py @@ -0,0 +1 @@ +# tests/integrations/agno package diff --git a/tests/integrations/agno/conftest.py b/tests/integrations/agno/conftest.py new file mode 100644 index 00000000..688dd233 --- /dev/null +++ b/tests/integrations/agno/conftest.py @@ -0,0 +1,125 @@ +""" +Shared pytest configuration for Agno integration tests. + +Installs a comprehensive agno stub into sys.modules before any test in this +directory runs, so that every test file can import the integration modules +without a real agno installation. + +Each per-file stub only runs `if "agno" in sys.modules: return`, which would +skip when another file already loaded a partial stub. This conftest installs +ALL required sub-modules at session start so the guard works correctly for +every file. +""" +from __future__ import annotations + +import sys +import types + + +def _install_agno_stubs() -> None: + """Install a full set of agno stubs into sys.modules.""" + + # ----------------------------------------------------------------------- + # agno root + # ----------------------------------------------------------------------- + agno = sys.modules.get("agno") or types.ModuleType("agno") + + # ----------------------------------------------------------------------- + # agno.memory.db.base โ€” MemoryDb + # ----------------------------------------------------------------------- + memory_pkg = types.ModuleType("agno.memory") + memory_db_pkg = types.ModuleType("agno.memory.db") + memory_db_base = types.ModuleType("agno.memory.db.base") + memory_db_row = types.ModuleType("agno.memory.db.row") + + class MemoryDb: # noqa: D101 + def __init__(self, *a, **kw): ... # noqa: E704 + + class MemoryRow: # noqa: D101 + def __init__(self, memory: str, id=None, user_id=None, **kw): + self.memory = memory + self.id = id + self.user_id = user_id + self.last_updated = 0.0 + self.topics = kw.get("topics", []) + + memory_db_base.MemoryDb = MemoryDb # type: ignore + memory_db_row.MemoryRow = MemoryRow # type: ignore + memory_db_pkg.base = memory_db_base + memory_db_pkg.row = memory_db_row + memory_pkg.db = memory_db_pkg + agno.memory = memory_pkg # type: ignore + + # ----------------------------------------------------------------------- + # agno.tools.toolkit โ€” Toolkit + # ----------------------------------------------------------------------- + tools_pkg = types.ModuleType("agno.tools") + tools_toolkit_mod = types.ModuleType("agno.tools.toolkit") + + class Toolkit: # noqa: D101 + def __init__(self, name: str = "toolkit", **kw): + self.name = name + self._tools: list = [] + + def register(self, fn): # noqa: D102 + self._tools.append(fn) + + tools_toolkit_mod.Toolkit = Toolkit # type: ignore + tools_pkg.toolkit = tools_toolkit_mod + agno.tools = tools_pkg # type: ignore + + # ----------------------------------------------------------------------- + # agno.knowledge.base โ€” AgentKnowledge + # ----------------------------------------------------------------------- + knowledge_pkg = types.ModuleType("agno.knowledge") + knowledge_base_mod = types.ModuleType("agno.knowledge.base") + + class AgentKnowledge: # noqa: D101 + def __init__(self, *a, **kw): ... # noqa: E704 + + def search(self, query, num_documents=None, filters=None): # noqa: D102 + return [] + + knowledge_base_mod.AgentKnowledge = AgentKnowledge # type: ignore + knowledge_pkg.base = knowledge_base_mod + agno.knowledge = knowledge_pkg # type: ignore + + # ----------------------------------------------------------------------- + # agno.document.base โ€” Document + # ----------------------------------------------------------------------- + document_pkg = types.ModuleType("agno.document") + document_base_mod = types.ModuleType("agno.document.base") + + class Document: # noqa: D101 + def __init__(self, content="", id=None, name=None, meta_data=None): + self.content = content + self.id = id + self.name = name + self.meta_data = meta_data or {} + + document_base_mod.Document = Document # type: ignore + document_pkg.base = document_base_mod + agno.document = document_pkg # type: ignore + + # ----------------------------------------------------------------------- + # Register everything + # ----------------------------------------------------------------------- + _mods = { + "agno": agno, + "agno.memory": memory_pkg, + "agno.memory.db": memory_db_pkg, + "agno.memory.db.base": memory_db_base, + "agno.memory.db.row": memory_db_row, + "agno.tools": tools_pkg, + "agno.tools.toolkit": tools_toolkit_mod, + "agno.knowledge": knowledge_pkg, + "agno.knowledge.base": knowledge_base_mod, + "agno.document": document_pkg, + "agno.document.base": document_base_mod, + } + for name, mod in _mods.items(): + sys.modules[name] = mod + + +# Install once at import time (conftest is imported before any test file) +_install_agno_stubs() diff --git a/tests/integrations/agno/test_context_store.py b/tests/integrations/agno/test_context_store.py new file mode 100644 index 00000000..b9fd7ddc --- /dev/null +++ b/tests/integrations/agno/test_context_store.py @@ -0,0 +1,233 @@ +""" +Tests for AgnoContextStore โ€” graph-backed Agno MemoryDb. + +All tests run without a real Agno installation by mocking the base class +and using in-memory Semantica components only. +""" + +from __future__ import annotations + +import sys +import types +import unittest +from unittest.mock import MagicMock, patch + + +# --------------------------------------------------------------------------- +# Stub the agno package so the import succeeds without it installed +# --------------------------------------------------------------------------- +def _stub_agno() -> None: + """Insert minimal agno stubs into sys.modules.""" + if "agno" in sys.modules: + return # real agno installed โ€” no stub needed + + agno = types.ModuleType("agno") + + # agno.memory.db.base + memory_pkg = types.ModuleType("agno.memory") + memory_db_pkg = types.ModuleType("agno.memory.db") + memory_db_base = types.ModuleType("agno.memory.db.base") + + class MemoryDb: # noqa: D101 + def __init__(self, *a, **kw): ... # noqa: E704 + + memory_db_base.MemoryDb = MemoryDb # type: ignore + + # agno.memory.db.row + memory_db_row = types.ModuleType("agno.memory.db.row") + + class MemoryRow: # noqa: D101 + def __init__(self, memory: str, id=None, user_id=None, **kw): + self.memory = memory + self.id = id + self.user_id = user_id + self.last_updated = 0.0 + self.topics = kw.get("topics", []) + + memory_db_row.MemoryRow = MemoryRow # type: ignore + + memory_db_pkg.base = memory_db_base + memory_db_pkg.row = memory_db_row + memory_pkg.db = memory_db_pkg + + agno.memory = memory_pkg # type: ignore + + for name, mod in [ + ("agno", agno), + ("agno.memory", memory_pkg), + ("agno.memory.db", memory_db_pkg), + ("agno.memory.db.base", memory_db_base), + ("agno.memory.db.row", memory_db_row), + ]: + sys.modules.setdefault(name, mod) + + +_stub_agno() + + +from integrations.agno.context_store import AgnoContextStore # noqa: E402 + + +class TestAgnoContextStoreInit(unittest.TestCase): + """Construction and basic attribute checks.""" + + def _make_store(self, **kwargs) -> AgnoContextStore: + return AgnoContextStore(decision_tracking=True, graph_expansion=True, **kwargs) + + def test_creates_without_args(self): + store = self._make_store() + self.assertIsNotNone(store) + + def test_session_id_generated(self): + store = self._make_store() + self.assertIsInstance(store.session_id, str) + self.assertTrue(len(store.session_id) > 0) + + def test_explicit_session_id(self): + store = AgnoContextStore(session_id="abc-123") + self.assertEqual(store.session_id, "abc-123") + + def test_decision_tracking_flag(self): + store = AgnoContextStore(decision_tracking=False) + self.assertFalse(store.decision_tracking) + + def test_context_property(self): + store = self._make_store() + self.assertIsNotNone(store.context) + + +class TestAgnoContextStoreMemoryDb(unittest.TestCase): + """MemoryDb protocol methods.""" + + def setUp(self): + self.store = AgnoContextStore(decision_tracking=False) + + def _make_row(self, text: str, uid: str = "u1"): + row = MagicMock() + row.memory = text + row.id = None + row.user_id = uid + row.last_updated = 0.0 + row.topics = [] + return row + + def test_table_exists(self): + self.assertTrue(self.store.table_exists()) + + def test_create_noop(self): + # Should not raise + self.store.create() + + def test_upsert_and_read(self): + row = self._make_row("Hello world") + self.store.upsert_memory(row) + memories = self.store.read_memories() + self.assertEqual(len(memories), 1) + + def test_upsert_sets_id(self): + row = self._make_row("Test memory") + self.store.upsert_memory(row) + self.assertIsNotNone(row.id) + + def test_memory_exists_after_upsert(self): + row = self._make_row("Exists check") + self.store.upsert_memory(row) + self.assertTrue(self.store.memory_exists(row)) + + def test_memory_not_exists_before_upsert(self): + row = self._make_row("Not yet") + row.id = "unknown-id" + self.assertFalse(self.store.memory_exists(row)) + + def test_delete_memory(self): + row = self._make_row("To delete") + self.store.upsert_memory(row) + mem_id = row.id + self.store.delete_memory(mem_id) + self.assertFalse(self.store.memory_exists(row)) + + def test_read_memories_user_filter(self): + row_a = self._make_row("User A memory", uid="alice") + row_b = self._make_row("User B memory", uid="bob") + self.store.upsert_memory(row_a) + self.store.upsert_memory(row_b) + + alice_rows = self.store.read_memories(user_id="alice") + self.assertEqual(len(alice_rows), 1) + self.assertEqual(alice_rows[0].user_id, "alice") + + def test_read_memories_limit(self): + for i in range(5): + self.store.upsert_memory(self._make_row(f"Memory {i}")) + rows = self.store.read_memories(limit=3) + self.assertEqual(len(rows), 3) + + def test_clear(self): + for i in range(3): + self.store.upsert_memory(self._make_row(f"M{i}")) + result = self.store.clear() + self.assertTrue(result) + self.assertEqual(len(self.store.read_memories()), 0) + + def test_drop_table(self): + self.store.upsert_memory(self._make_row("Drop me")) + self.store.drop_table() + self.assertEqual(len(self.store.read_memories()), 0) + + +class TestAgnoContextStoreExtendedAPI(unittest.TestCase): + """Extended Semantica-specific methods.""" + + def setUp(self): + self.store = AgnoContextStore(decision_tracking=True) + # Patch the internal AgentContext to avoid real LLM/vector calls + self.store._context = MagicMock() + self.store._context.record_decision.return_value = "dec-001" + self.store._context.find_precedents_advanced.return_value = [] + self.store._context.retrieve.return_value = [] + + def test_record_decision_returns_id(self): + did = self.store.record_decision( + category="test", + scenario="Unit test scenario", + reasoning="Testing", + outcome="pass", + confidence=0.9, + ) + self.assertEqual(did, "dec-001") + self.store._context.record_decision.assert_called_once() + + def test_find_precedents_returns_list(self): + result = self.store.find_precedents("some scenario") + self.assertIsInstance(result, list) + + def test_retrieve_returns_list(self): + result = self.store.retrieve("query text") + self.assertIsInstance(result, list) + + def test_record_decision_passes_entities(self): + self.store.record_decision( + category="finance", + scenario="Loan", + reasoning="Good credit", + outcome="approved", + confidence=0.95, + entities=["applicant", "loan"], + ) + call_kwargs = self.store._context.record_decision.call_args[1] + self.assertEqual(call_kwargs["entities"], ["applicant", "loan"]) + + def test_upsert_with_decision_tracking(self): + row = MagicMock() + row.memory = "Important fact" + row.id = None + row.user_id = "u1" + row.last_updated = 0.0 + row.topics = [] + self.store.upsert_memory(row) + # decision should have been recorded + self.store._context.record_decision.assert_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/integrations/agno/test_decision_kit.py b/tests/integrations/agno/test_decision_kit.py new file mode 100644 index 00000000..ec99e7ce --- /dev/null +++ b/tests/integrations/agno/test_decision_kit.py @@ -0,0 +1,256 @@ +""" +Tests for AgnoDecisionKit โ€” decision intelligence Agno Toolkit. +""" + +from __future__ import annotations + +import json +import sys +import types +import unittest +from unittest.mock import MagicMock + + +# --------------------------------------------------------------------------- +# Stub agno Toolkit +# --------------------------------------------------------------------------- +def _stub_agno() -> None: + if "agno" in sys.modules: + return + + agno = types.ModuleType("agno") + + tools_pkg = types.ModuleType("agno.tools") + tools_toolkit = types.ModuleType("agno.tools.toolkit") + + class Toolkit: + def __init__(self, name="toolkit", **kw): + self.name = name + self._tools = [] + + def register(self, fn): + self._tools.append(fn) + + tools_toolkit.Toolkit = Toolkit # type: ignore + tools_pkg.toolkit = tools_toolkit + agno.tools = tools_pkg # type: ignore + + for name, mod in [ + ("agno", agno), + ("agno.tools", tools_pkg), + ("agno.tools.toolkit", tools_toolkit), + ]: + sys.modules.setdefault(name, mod) + + +_stub_agno() + +from integrations.agno.decision_kit import AgnoDecisionKit # noqa: E402 + + +def _make_context() -> MagicMock: + ctx = MagicMock() + ctx.record_decision.return_value = "dec-test-001" + ctx.find_precedents_advanced.return_value = [ + {"scenario": "past loan", "outcome": "approved", "confidence": 0.9, "category": "loan"} + ] + ctx.analyze_decision_influence.return_value = {"centrality": 0.75, "influenced": 3} + ctx.get_context_insights.return_value = {"total_decisions": 5, "categories": ["loan"]} + ctx.knowledge_graph = MagicMock() + ctx.knowledge_graph.trace_decision_causality = MagicMock(return_value=["step1", "step2"]) + return ctx + + +class TestAgnoDecisionKitInit(unittest.TestCase): + + def test_creates_with_context(self): + kit = AgnoDecisionKit(context=_make_context()) + self.assertIsNotNone(kit) + + def test_creates_without_context(self): + # Should auto-create an AgentContext + kit = AgnoDecisionKit() + self.assertIsNotNone(kit) + + def test_tools_registered(self): + kit = AgnoDecisionKit(context=_make_context()) + # Tools should be registered (Toolkit.register was called) + self.assertTrue(len(kit._tools) >= 5) + + def test_policy_tool_can_be_disabled(self): + kit = AgnoDecisionKit(context=_make_context(), enable_policy_check=False) + tool_names = [fn.__name__ for fn in kit._tools] + self.assertNotIn("check_policy", tool_names) + + +class TestRecordDecision(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json_with_decision_id(self): + result = json.loads(self.kit.record_decision( + category="loan", + scenario="Customer A loan application", + reasoning="Good credit score 740", + outcome="approved", + confidence=0.95, + )) + self.assertIn("decision_id", result) + self.assertEqual(result["status"], "recorded") + + def test_delegates_to_context(self): + self.kit.record_decision( + category="content", + scenario="Moderation check", + reasoning="No violations", + outcome="allowed", + confidence=0.88, + ) + self.ctx.record_decision.assert_called_once() + + def test_parses_entities_string(self): + self.kit.record_decision( + category="hr", + scenario="Hire decision", + reasoning="Qualified", + outcome="hired", + confidence=0.9, + entities="Alice, ACME Corp, Senior Engineer", + ) + call_kwargs = self.ctx.record_decision.call_args[1] + self.assertIsInstance(call_kwargs["entities"], list) + self.assertEqual(len(call_kwargs["entities"]), 3) + + def test_returns_error_json_on_failure(self): + self.ctx.record_decision.side_effect = RuntimeError("DB unavailable") + result = json.loads(self.kit.record_decision( + category="x", scenario="y", reasoning="z", outcome="failed", + )) + self.assertEqual(result["status"], "failed") + self.assertIn("error", result) + + def test_default_confidence_used(self): + self.kit.record_decision( + category="test", + scenario="Default confidence test", + reasoning="N/A", + outcome="pass", + ) + call_kwargs = self.ctx.record_decision.call_args[1] + self.assertEqual(call_kwargs["confidence"], 0.8) + + +class TestFindPrecedents(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json_with_precedents(self): + result = json.loads(self.kit.find_precedents("new loan application")) + self.assertIn("precedents", result) + self.assertIsInstance(result["precedents"], list) + + def test_count_in_result(self): + result = json.loads(self.kit.find_precedents("test scenario")) + self.assertIn("count", result) + self.assertEqual(result["count"], len(result["precedents"])) + + def test_category_filter_passed(self): + self.kit.find_precedents("scenario", category="finance") + call_kwargs = self.ctx.find_precedents_advanced.call_args[1] + self.assertEqual(call_kwargs.get("category"), "finance") + + def test_limit_applied(self): + self.ctx.find_precedents_advanced.return_value = [ + {"scenario": f"s{i}", "outcome": "o", "confidence": 0.5, "category": "c"} + for i in range(10) + ] + result = json.loads(self.kit.find_precedents("s", limit=3)) + self.assertTrue(result["count"] <= 3) + + def test_handles_exception_gracefully(self): + self.ctx.find_precedents_advanced.side_effect = RuntimeError("fail") + result = json.loads(self.kit.find_precedents("broken")) + self.assertEqual(result["precedents"], []) + self.assertIn("error", result) + + +class TestTraceCausalChain(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json_with_causal_chain(self): + result = json.loads(self.kit.trace_causal_chain("dec-001")) + self.assertIn("causal_chain", result) + self.assertEqual(result["decision_id"], "dec-001") + + def test_fallback_on_attribute_error(self): + del self.ctx.knowledge_graph.trace_decision_causality + self.ctx.knowledge_graph.find_precedents = MagicMock(return_value=[]) + result = json.loads(self.kit.trace_causal_chain("dec-002")) + self.assertIn("causal_chain", result) + + def test_depth_passed(self): + self.kit.trace_causal_chain("dec-001", depth=5) + # Should not raise + + +class TestAnalyzeImpact(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json_with_decision_id(self): + result = json.loads(self.kit.analyze_impact("dec-001")) + self.assertEqual(result["decision_id"], "dec-001") + + def test_includes_influence_metrics(self): + result = json.loads(self.kit.analyze_impact("dec-001")) + self.assertIn("centrality", result) + + +class TestCheckPolicy(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json_with_compliant_key(self): + decision = json.dumps({"category": "loan", "outcome": "approved", "confidence": 0.9}) + result = json.loads(self.kit.check_policy(decision)) + self.assertIn("compliant", result) + + def test_invalid_json_returns_error(self): + result = json.loads(self.kit.check_policy("{not valid json}")) + self.assertIn("error", result) + + +class TestGetDecisionSummary(unittest.TestCase): + + def setUp(self): + self.ctx = _make_context() + self.kit = AgnoDecisionKit(context=self.ctx) + + def test_returns_json(self): + result_str = self.kit.get_decision_summary() + result = json.loads(result_str) + self.assertIsInstance(result, dict) + + def test_category_filter_stored(self): + result = json.loads(self.kit.get_decision_summary(category="finance")) + self.assertEqual(result.get("category_filter"), "finance") + + def test_handles_exception_gracefully(self): + self.ctx.get_context_insights.side_effect = RuntimeError("insight fail") + result = json.loads(self.kit.get_decision_summary()) + self.assertIn("error", result) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/integrations/agno/test_kg_toolkit.py b/tests/integrations/agno/test_kg_toolkit.py new file mode 100644 index 00000000..8ddd25a9 --- /dev/null +++ b/tests/integrations/agno/test_kg_toolkit.py @@ -0,0 +1,371 @@ +""" +Tests for AgnoKGToolkit โ€” knowledge graph Agno Toolkit. +""" + +from __future__ import annotations + +import json +import sys +import types +import unittest +from unittest.mock import MagicMock, patch + + +# --------------------------------------------------------------------------- +# Stub agno Toolkit +# --------------------------------------------------------------------------- +def _stub_agno() -> None: + if "agno" in sys.modules: + return + + agno = types.ModuleType("agno") + tools_pkg = types.ModuleType("agno.tools") + tools_toolkit = types.ModuleType("agno.tools.toolkit") + + class Toolkit: + def __init__(self, name="toolkit", **kw): + self.name = name + self._tools = [] + + def register(self, fn): + self._tools.append(fn) + + tools_toolkit.Toolkit = Toolkit # type: ignore + tools_pkg.toolkit = tools_toolkit + agno.tools = tools_pkg # type: ignore + + for name, mod in [ + ("agno", agno), + ("agno.tools", tools_pkg), + ("agno.tools.toolkit", tools_toolkit), + ]: + sys.modules.setdefault(name, mod) + + +_stub_agno() + +from integrations.agno.kg_toolkit import AgnoKGToolkit # noqa: E402 + + +# --------------------------------------------------------------------------- +# Fakes +# --------------------------------------------------------------------------- +def _fake_entity(name="Tesla", etype="ORG", conf=0.9): + e = MagicMock() + e.name = name + e.type = etype + e.confidence = conf + return e + + +def _fake_relation(src="Tesla", rel="FOUNDED_BY", tgt="Elon Musk", conf=0.85): + r = MagicMock() + r.source = src + r.type = rel + r.target = tgt + r.confidence = conf + return r + + +class _FakeNER: + def extract_entities(self, text): + return [_fake_entity("Tesla"), _fake_entity("Elon Musk", "PERSON")] + + +class _FakeRelExtractor: + def extract_relations(self, text, entities=None): + return [_fake_relation()] + + +class _FakeReasoner: + def infer_facts(self, facts, rules): + result = MagicMock() + result.inferred_facts = ["Human(EthicalAI)"] + return result + + +class _FakeGraph: + """Fake ContextGraph whose signatures match the real ContextGraph API.""" + + def __init__(self): + self._node_store: dict = {} # node_id -> {"node_id": ..., "node_type": ...} + self._edge_store: list = [] + + # ContextGraph.find_nodes(node_type=None) -> List[Dict] + def find_nodes(self, node_type=None): + nodes = list(self._node_store.values()) + if node_type: + nodes = [n for n in nodes if n.get("node_type") == node_type] + return nodes + + # ContextGraph.add_node(node_id, node_type, content=None, **props) -> bool + def add_node(self, node_id, node_type="Entity", content=None, **props): + self._node_store[node_id] = {"node_id": node_id, "node_type": node_type} + return True + + # ContextGraph.add_edge(source_id, target_id, edge_type, **props) -> bool + def add_edge(self, source_id, target_id, edge_type="related_to", **props): + self._edge_store.append((source_id, target_id, edge_type)) + return True + + # ContextGraph.get_neighbors(node_id, hops=1, ...) -> List[Dict] + def get_neighbors(self, node_id, hops=1, relationship_types=None, min_weight=0.0): + return [{"node_id": f"Neighbour_of_{node_id}", "node_type": "Entity"}] + + +class TestAgnoKGToolkitInit(unittest.TestCase): + + def test_creates_with_defaults(self): + kit = AgnoKGToolkit() + self.assertIsNotNone(kit) + + def test_creates_with_custom_components(self): + kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.assertIsNotNone(kit) + + def test_tools_registered(self): + kit = AgnoKGToolkit() + self.assertTrue(len(kit._tools) >= 7) + + def test_context_graph_attached(self): + ctx = MagicMock() + ctx.knowledge_graph = _FakeGraph() + kit = AgnoKGToolkit(context=ctx) + self.assertIs(kit._graph, ctx.knowledge_graph) + + +class TestExtractEntities(unittest.TestCase): + + def setUp(self): + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + + def test_returns_json(self): + result = json.loads(self.kit.extract_entities("Tesla was founded by Elon Musk.")) + self.assertIn("entities", result) + self.assertIn("count", result) + + def test_entity_shape(self): + result = json.loads(self.kit.extract_entities("some text")) + for ent in result["entities"]: + self.assertIn("name", ent) + self.assertIn("type", ent) + self.assertIn("confidence", ent) + + def test_count_matches_entities(self): + result = json.loads(self.kit.extract_entities("text")) + self.assertEqual(result["count"], len(result["entities"])) + + def test_handles_ner_failure(self): + bad_ner = MagicMock() + bad_ner.extract_entities.side_effect = RuntimeError("NER crashed") + kit = AgnoKGToolkit( + ner_extractor=bad_ner, + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + result = json.loads(kit.extract_entities("text")) + self.assertEqual(result["count"], 0) + self.assertIn("error", result) + + +class TestExtractRelations(unittest.TestCase): + + def setUp(self): + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + + def test_returns_json(self): + result = json.loads(self.kit.extract_relations("Tesla was founded by Elon Musk.")) + self.assertIn("relations", result) + self.assertIn("count", result) + + def test_relation_shape(self): + result = json.loads(self.kit.extract_relations("text")) + for rel in result["relations"]: + self.assertIn("source", rel) + self.assertIn("relation", rel) + self.assertIn("target", rel) + self.assertIn("confidence", rel) + + def test_entities_filter_parsed_from_json(self): + self.kit.extract_relations("text", entities='["Tesla", "Elon Musk"]') + # Should not raise + + def test_entities_filter_parsed_from_csv(self): + self.kit.extract_relations("text", entities="Tesla, Elon Musk") + # Should not raise + + def test_handles_failure_gracefully(self): + bad_rel = MagicMock() + bad_rel.extract_relations.side_effect = RuntimeError("fail") + kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=bad_rel, + reasoner=_FakeReasoner(), + ) + result = json.loads(kit.extract_relations("text")) + self.assertEqual(result["count"], 0) + self.assertIn("error", result) + + +class TestAddToGraph(unittest.TestCase): + + def setUp(self): + self.graph = _FakeGraph() + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.kit._graph = self.graph + + def test_add_entities_json(self): + entities = json.dumps([{"name": "Alice", "type": "PERSON"}]) + result = json.loads(self.kit.add_to_graph(entities=entities)) + self.assertEqual(result["nodes_added"], 1) + + def test_add_relations_json(self): + relations = json.dumps([{"source": "Alice", "relation": "WORKS_AT", "target": "ACME"}]) + result = json.loads(self.kit.add_to_graph(relations=relations)) + self.assertEqual(result["edges_added"], 1) + + def test_add_both(self): + entities = json.dumps([{"name": "Bob", "type": "PERSON"}]) + relations = json.dumps([{"source": "Bob", "relation": "WORKS_AT", "target": "Corp"}]) + result = json.loads(self.kit.add_to_graph(entities=entities, relations=relations)) + self.assertEqual(result["nodes_added"], 1) + self.assertEqual(result["edges_added"], 1) + + def test_empty_call(self): + result = json.loads(self.kit.add_to_graph()) + self.assertEqual(result["nodes_added"], 0) + self.assertEqual(result["edges_added"], 0) + + +class TestQueryGraph(unittest.TestCase): + + def setUp(self): + self.graph = _FakeGraph() + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.kit._graph = self.graph + + def test_keyword_query_returns_results(self): + result = json.loads(self.kit.query_graph("Tesla")) + self.assertIn("results", result) + self.assertEqual(result["query_type"], "keyword") + + def test_cypher_query_without_backend(self): + result = json.loads(self.kit.query_graph("MATCH (n) RETURN n LIMIT 5")) + # Without a real neo4j backend, should return an error + self.assertEqual(result["query_type"], "cypher") + + def test_handles_exception(self): + bad_graph = MagicMock() + bad_graph.find_nodes.side_effect = RuntimeError("graph error") + self.kit._graph = bad_graph + result = json.loads(self.kit.query_graph("anything")) + self.assertIn("error", result) + + +class TestFindRelated(unittest.TestCase): + + def setUp(self): + self.graph = _FakeGraph() + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.kit._graph = self.graph + + def test_returns_json(self): + result = json.loads(self.kit.find_related("Tesla")) + self.assertIn("entity", result) + self.assertIn("related", result) + self.assertIn("count", result) + + def test_entity_preserved(self): + result = json.loads(self.kit.find_related("Elon")) + self.assertEqual(result["entity"], "Elon") + + def test_hops_parameter(self): + result = json.loads(self.kit.find_related("Tesla", hops=2)) + self.assertIsInstance(result["related"], list) + + +class TestInferFacts(unittest.TestCase): + + def setUp(self): + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.kit._graph = _FakeGraph() + self.kit._graph._nodes = {"n1": MagicMock(label="EthicalAI", node_type="AI")} + + def test_returns_inferred_facts(self): + result = json.loads(self.kit.infer_facts(rules='["IF AI(?x) THEN System(?x)"]')) + self.assertIn("inferred_facts", result) + self.assertIsInstance(result["inferred_facts"], list) + + def test_count_correct(self): + result = json.loads(self.kit.infer_facts(rules='["IF X(?a) THEN Y(?a)"]')) + self.assertEqual(result["count"], len(result["inferred_facts"])) + + def test_rules_as_csv(self): + result = json.loads(self.kit.infer_facts(rules="IF AI(?x) THEN System(?x)")) + self.assertIn("inferred_facts", result) + + def test_facts_passed_explicitly(self): + result = json.loads(self.kit.infer_facts( + rules='["IF Person(?x) THEN Human(?x)"]', + facts='["Person(Alice)"]', + )) + self.assertIn("inferred_facts", result) + + +class TestExportSubgraph(unittest.TestCase): + + def setUp(self): + self.kit = AgnoKGToolkit( + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + reasoner=_FakeReasoner(), + ) + self.kit._graph = _FakeGraph() + + def test_returns_json(self): + result_str = self.kit.export_subgraph() + result = json.loads(result_str) + self.assertIn("format", result) + + def test_format_passed(self): + result = json.loads(self.kit.export_subgraph(format="turtle")) + self.assertIn("format", result) + + def test_fallback_to_json_on_import_error(self): + # RDFExporter may not be available in test env; should fall back gracefully + result_str = self.kit.export_subgraph() + result = json.loads(result_str) + # Either the real export or the fallback JSON โ€” both are valid + self.assertIsInstance(result, dict) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/integrations/agno/test_knowledge_graph.py b/tests/integrations/agno/test_knowledge_graph.py new file mode 100644 index 00000000..68a360c7 --- /dev/null +++ b/tests/integrations/agno/test_knowledge_graph.py @@ -0,0 +1,233 @@ +""" +Tests for AgnoKnowledgeGraph โ€” relational AgentKnowledge with GraphRAG. +""" + +from __future__ import annotations + +import sys +import types +import unittest +from unittest.mock import MagicMock, patch + + +# --------------------------------------------------------------------------- +# Stub agno +# --------------------------------------------------------------------------- +def _stub_agno() -> None: + if "agno" in sys.modules: + return + + agno = types.ModuleType("agno") + + # agno.knowledge.base + knowledge_pkg = types.ModuleType("agno.knowledge") + knowledge_base = types.ModuleType("agno.knowledge.base") + + class AgentKnowledge: + def __init__(self, *a, **kw): ... # noqa: E704 + def search(self, query, num_documents=None, filters=None): return [] # noqa: E704 + + knowledge_base.AgentKnowledge = AgentKnowledge # type: ignore + knowledge_pkg.base = knowledge_base + agno.knowledge = knowledge_pkg # type: ignore + + # agno.document.base + document_pkg = types.ModuleType("agno.document") + document_base = types.ModuleType("agno.document.base") + + class Document: + def __init__(self, content="", id=None, name=None, meta_data=None): + self.content = content + self.id = id + self.name = name + self.meta_data = meta_data or {} + + document_base.Document = Document # type: ignore + document_pkg.base = document_base + agno.document = document_pkg # type: ignore + + for name, mod in [ + ("agno", agno), + ("agno.knowledge", knowledge_pkg), + ("agno.knowledge.base", knowledge_base), + ("agno.document", document_pkg), + ("agno.document.base", document_base), + ]: + sys.modules.setdefault(name, mod) + + +_stub_agno() + +from integrations.agno.knowledge_graph import AgnoKnowledgeGraph # noqa: E402 + + +class _FakeNER: + def extract_entities(self, text): + e = MagicMock() + e.name = "FakeEntity" + e.type = "ORG" + e.confidence = 0.9 + return [e] + + +class _FakeRelExtractor: + def extract_relations(self, text, entities=None): + r = MagicMock() + r.source = "FakeEntity" + r.type = "RELATED_TO" + r.target = "OtherEntity" + r.confidence = 0.8 + return [r] + + +class _FakeGraphBuilder: + def build(self, sources): + return MagicMock() + + +class _FakeContextGraph: + def find_nodes(self, label=None): + node = MagicMock() + node.label = label or "Node" + node.node_type = "Entity" + return [node] + + +class TestAgnoKnowledgeGraphInit(unittest.TestCase): + + def test_creates_with_defaults(self): + kg = AgnoKnowledgeGraph() + self.assertIsNotNone(kg) + + def test_creates_with_custom_components(self): + kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + self.assertIsNotNone(kg) + + def test_num_documents_default(self): + kg = AgnoKnowledgeGraph(num_documents=10) + self.assertEqual(kg.num_documents, 10) + + +class TestAgnoKnowledgeGraphLoad(unittest.TestCase): + + def setUp(self): + self.kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + + def test_load_texts(self): + self.kg.load(texts=["Alice works at Acme Corp.", "Bob is the CEO."]) + self.assertEqual(len(self.kg._docs), 2) + + def test_load_texts_multiple_calls_accumulate(self): + self.kg.load(texts=["First batch"]) + self.kg.load(texts=["Second batch"]) + self.assertEqual(len(self.kg._docs), 2) + + def test_load_recreate_clears_docs(self): + self.kg.load(texts=["Old doc"]) + self.kg.load(texts=["New doc"], recreate=True) + self.assertEqual(len(self.kg._docs), 1) + + def test_load_documents(self): + doc = MagicMock() + doc.content = "Agno is a multi-agent framework." + doc.name = "agno_intro" + self.kg.load_documents([doc]) + self.assertEqual(len(self.kg._docs), 1) + + def test_ingest_stores_entities(self): + self.kg._ingest_text("Tesla was founded by Elon Musk.", source="test") + stored = self.kg._docs[-1] + self.assertIn("entities", stored) + self.assertTrue(len(stored["entities"]) > 0) + + +class TestAgnoKnowledgeGraphSearch(unittest.TestCase): + + def setUp(self): + self.kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + self.kg.load(texts=[ + "Machine learning is a subset of artificial intelligence.", + "Python is a popular programming language.", + "Neural networks are inspired by the human brain.", + ]) + + def test_search_returns_list(self): + results = self.kg.search("machine learning") + self.assertIsInstance(results, list) + + def test_search_returns_agno_documents(self): + results = self.kg.search("python", num_documents=2) + self.assertTrue(len(results) <= 2) + for doc in results: + self.assertTrue(hasattr(doc, "content")) + + def test_search_empty_kg_returns_empty(self): + kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + results = kg.search("anything") + self.assertEqual(results, []) + + def test_search_num_documents_respected(self): + results = self.kg.search("a", num_documents=1) + self.assertTrue(len(results) <= 1) + + def test_get_graph_context(self): + ctx = self.kg.get_graph_context("FakeEntity") + self.assertIsInstance(ctx, str) + + +class TestAgnoKnowledgeGraphPathLoading(unittest.TestCase): + """Test path-based loading with a temporary file.""" + + def test_load_missing_path_warns(self): + kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + # Should not raise even for non-existent path + kg.load(path="/nonexistent/path/xyz") + self.assertEqual(len(kg._docs), 0) + + def test_load_file(self): + import tempfile, os + + kg = AgnoKnowledgeGraph( + graph_builder=_FakeGraphBuilder(), + ner_extractor=_FakeNER(), + relation_extractor=_FakeRelExtractor(), + context_graph=_FakeContextGraph(), + ) + with tempfile.NamedTemporaryFile(mode="w", suffix=".txt", delete=False) as f: + f.write("Test document content for loading.") + tmp_path = f.name + + try: + kg.load(path=tmp_path) + self.assertEqual(len(kg._docs), 1) + finally: + os.unlink(tmp_path) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/integrations/agno/test_shared_context.py b/tests/integrations/agno/test_shared_context.py new file mode 100644 index 00000000..5db29d70 --- /dev/null +++ b/tests/integrations/agno/test_shared_context.py @@ -0,0 +1,237 @@ +""" +Tests for AgnoSharedContext โ€” multi-agent shared ContextGraph coordinator. +""" + +from __future__ import annotations + +import sys +import types +import unittest +from unittest.mock import MagicMock + + +# --------------------------------------------------------------------------- +# Stub agno (MemoryDb needed by AgnoContextStore base) +# --------------------------------------------------------------------------- +def _stub_agno() -> None: + if "agno" in sys.modules: + return + + agno = types.ModuleType("agno") + + memory_pkg = types.ModuleType("agno.memory") + memory_db_pkg = types.ModuleType("agno.memory.db") + memory_db_base = types.ModuleType("agno.memory.db.base") + memory_db_row = types.ModuleType("agno.memory.db.row") + + class MemoryDb: + def __init__(self, *a, **kw): ... # noqa: E704 + + class MemoryRow: + def __init__(self, memory, id=None, user_id=None, **kw): + self.memory = memory + self.id = id + self.user_id = user_id + self.last_updated = 0.0 + self.topics = kw.get("topics", []) + + memory_db_base.MemoryDb = MemoryDb # type: ignore + memory_db_row.MemoryRow = MemoryRow # type: ignore + memory_db_pkg.base = memory_db_base + memory_db_pkg.row = memory_db_row + memory_pkg.db = memory_db_pkg + agno.memory = memory_pkg # type: ignore + + for name, mod in [ + ("agno", agno), + ("agno.memory", memory_pkg), + ("agno.memory.db", memory_db_pkg), + ("agno.memory.db.base", memory_db_base), + ("agno.memory.db.row", memory_db_row), + ]: + sys.modules.setdefault(name, mod) + + +_stub_agno() + +from integrations.agno.shared_context import AgnoSharedContext # noqa: E402 + + +def _make_shared(**kwargs) -> AgnoSharedContext: + shared = AgnoSharedContext(**kwargs) + # Replace internal AgentContext with a mock to avoid real side-effects + mock_ctx = MagicMock() + mock_ctx.record_decision.return_value = "shared-dec-001" + mock_ctx.find_precedents_advanced.return_value = [] + mock_ctx.get_context_insights.return_value = {"total": 0} + shared._context = mock_ctx + return shared + + +class TestAgnoSharedContextInit(unittest.TestCase): + + def test_creates_without_args(self): + shared = _make_shared() + self.assertIsNotNone(shared) + + def test_session_id_auto_generated(self): + shared = _make_shared() + self.assertIsInstance(shared.session_id, str) + self.assertTrue(len(shared.session_id) > 0) + + def test_explicit_session_id(self): + shared = _make_shared(session_id="team-session-xyz") + self.assertEqual(shared.session_id, "team-session-xyz") + + def test_decision_tracking_flag(self): + shared = _make_shared(decision_tracking=False) + self.assertFalse(shared.decision_tracking) + + def test_knowledge_graph_property(self): + shared = _make_shared() + self.assertIsNotNone(shared.knowledge_graph) + + def test_bound_roles_initially_empty(self): + shared = _make_shared() + self.assertEqual(shared.bound_roles, []) + + +class TestBindAgent(unittest.TestCase): + + def setUp(self): + self.shared = _make_shared() + + def test_bind_returns_store(self): + store = self.shared.bind_agent("researcher") + self.assertIsNotNone(store) + + def test_bind_idempotent(self): + store1 = self.shared.bind_agent("analyst") + store2 = self.shared.bind_agent("analyst") + self.assertIs(store1, store2) + + def test_bind_tracks_roles(self): + self.shared.bind_agent("researcher") + self.shared.bind_agent("analyst") + self.assertIn("researcher", self.shared.bound_roles) + self.assertIn("analyst", self.shared.bound_roles) + + def test_scoped_session_id(self): + store = self.shared.bind_agent("writer") + self.assertIn("writer", store.session_id) + self.assertIn(self.shared.session_id, store.session_id) + + def test_different_roles_different_stores(self): + s1 = self.shared.bind_agent("role_a") + s2 = self.shared.bind_agent("role_b") + self.assertIsNot(s1, s2) + + +class TestSharedMemoryPool(unittest.TestCase): + """Memories written by one agent are visible to all others.""" + + def setUp(self): + self.shared = _make_shared() + self.researcher = self.shared.bind_agent("researcher") + self.analyst = self.shared.bind_agent("analyst") + + def _make_row(self, text: str): + row = MagicMock() + row.memory = text + row.id = None + row.user_id = "u1" + row.last_updated = 0.0 + row.topics = [] + return row + + def test_researcher_memory_visible_to_analyst(self): + row = self._make_row("New regulation: Basel IV applies from 2026") + self.researcher.upsert_memory(row) + + analyst_memories = self.analyst.read_memories() + texts = [getattr(m, "memory", "") for m in analyst_memories] + self.assertIn("New regulation: Basel IV applies from 2026", texts) + + def test_analyst_memory_visible_to_researcher(self): + row = self._make_row("Market share: Competitor X grew by 12%") + self.analyst.upsert_memory(row) + + researcher_memories = self.researcher.read_memories() + texts = [getattr(m, "memory", "") for m in researcher_memories] + self.assertIn("Market share: Competitor X grew by 12%", texts) + + def test_both_memories_in_pool(self): + self.researcher.upsert_memory(self._make_row("Research insight A")) + self.analyst.upsert_memory(self._make_row("Analysis finding B")) + + # Either agent should see both + researcher_memories = self.researcher.read_memories() + self.assertTrue(len(researcher_memories) >= 2) + + def test_limit_respected_in_read(self): + for i in range(5): + self.researcher.upsert_memory(self._make_row(f"Fact {i}")) + memories = self.analyst.read_memories(limit=2) + self.assertTrue(len(memories) <= 2) + + +class TestSharedContextDecisions(unittest.TestCase): + + def setUp(self): + self.shared = _make_shared() + + def test_record_decision_returns_id(self): + did = self.shared.record_decision( + category="strategy", + scenario="Expand to EU market", + reasoning="Strong demand signals", + outcome="approved", + confidence=0.87, + ) + self.assertEqual(did, "shared-dec-001") + + def test_agent_role_tags_category(self): + self.shared.record_decision( + category="finance", + scenario="Budget allocation", + reasoning="Q1 performance", + outcome="increase", + confidence=0.9, + agent_role="cfo", + ) + call_kwargs = self.shared._context.record_decision.call_args[1] + self.assertIn("cfo", call_kwargs["category"]) + + def test_find_precedents_returns_list(self): + result = self.shared.find_precedents("expansion strategy") + self.assertIsInstance(result, list) + + def test_get_shared_insights_returns_dict(self): + result = self.shared.get_shared_insights() + self.assertIsInstance(result, dict) + + +class TestSharedContextThreadSafety(unittest.TestCase): + """Concurrent bind_agent calls should return the same store.""" + + def test_concurrent_bind_same_role(self): + import threading + + shared = _make_shared() + results = [] + + def bind(): + results.append(shared.bind_agent("concurrent_role")) + + threads = [threading.Thread(target=bind) for _ in range(10)] + for t in threads: + t.start() + for t in threads: + t.join() + + # All threads should get the same store instance + self.assertEqual(len(set(id(s) for s in results)), 1) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/ontology/test_ontology_advanced.py b/tests/ontology/test_ontology_advanced.py index 8371ed7c..25d257d8 100644 --- a/tests/ontology/test_ontology_advanced.py +++ b/tests/ontology/test_ontology_advanced.py @@ -8,6 +8,8 @@ from semantica.ontology.ontology_evaluator import OntologyEvaluator, EvaluationR from semantica.ontology.competency_questions import CompetencyQuestionsManager, CompetencyQuestion from semantica.change_management import VersionManager, OntologyVersion from semantica.ontology.associative_class import AssociativeClassBuilder, AssociativeClass +from semantica.ontology.reuse_manager import ReuseManager +from semantica.ontology.engine import OntologyEngine class TestOntologyAdvanced(unittest.TestCase): @@ -151,6 +153,62 @@ class TestOntologyAdvanced(unittest.TestCase): self.assertEqual(len(is_valid), 0) except Exception: pass + + def test_reuse_manager_suggest_alignments(self): + manager = ReuseManager() + target = { + "classes": [{"uri": "http://target.org/Person", "name": "Person"}], + "properties": [{"uri": "http://target.org/hasName", "name": "has name"}] + } + source = { + "classes": [{"uri": "http://source.org/Person", "name": "Person"}], + "properties": [{"uri": "http://source.org/hasName", "name": "has name"}] + } + + suggestions = manager.suggest_alignments(target, source) + + self.assertEqual(len(suggestions), 2) + self.assertEqual(suggestions[0]["predicate"], "http://www.w3.org/2002/07/owl#equivalentClass") + self.assertEqual(suggestions[0]["source_uri"], "http://source.org/Person") + self.assertEqual(suggestions[1]["predicate"], "http://www.w3.org/2002/07/owl#equivalentProperty") + + def test_reuse_manager_merge_with_alignments(self): + manager = ReuseManager() + target = {"classes": [{"uri": "http://target.org/Dog", "name": "Dog"}]} + source = {"classes": [{"uri": "http://source.org/Dog", "name": "Dog"}]} + + merged = manager.merge_ontology_data(target, source, compute_alignments=True) + + self.assertIn("suggested_alignments", merged) + self.assertEqual(len(merged["suggested_alignments"]), 1) + self.assertEqual(merged["suggested_alignments"][0]["target_uri"], "http://target.org/Dog") + + def test_engine_create_alignment(self): + mock_store = MagicMock() + engine = OntologyEngine(store=mock_store) + + engine.create_alignment("http://source.org/1", "http://target.org/2", "http://www.w3.org/2002/07/owl#sameAs") + + mock_store.add_triplet.assert_called_once() + args, kwargs = mock_store.add_triplet.call_args + self.assertEqual(args[0].subject, "http://source.org/1") + self.assertEqual(args[0].object, "http://target.org/2") + self.assertEqual(args[0].predicate, "http://www.w3.org/2002/07/owl#sameAs") + + def test_engine_get_alignments(self): + mock_store = MagicMock() + mock_result = MagicMock() + # Mocking the SPARQL binding response format + mock_result.bindings = [ + {"s": {"value": "http://source.org/1"}, "p": {"value": "http://owl#sameAs"}, "o": {"value": "http://target.org/2"}} + ] + mock_store.execute_query.return_value = mock_result + + engine = OntologyEngine(store=mock_store) + alignments = engine.get_alignments("http://source.org/1") + + self.assertEqual(len(alignments), 1) + self.assertEqual(alignments[0]["target"], "http://target.org/2") if __name__ == '__main__': unittest.main() diff --git a/tests/reasoning/test_datalog_reasoner.py b/tests/reasoning/test_datalog_reasoner.py new file mode 100644 index 00000000..421d38bc --- /dev/null +++ b/tests/reasoning/test_datalog_reasoner.py @@ -0,0 +1,200 @@ +""" +Test suite for the DatalogReasoner module. +""" + +import pytest +from typing import List, Dict, Any + +from semantica.reasoning.datalog_reasoner import DatalogReasoner, DatalogFact + +@pytest.fixture +def reasoner(): + """Provides a fresh DatalogReasoner instance for each test.""" + return DatalogReasoner() + +class MockContextGraph: + """A mock to simulate Semantica's actual ContextGraph structure.""" + def __init__(self, nodes: List[Dict[str, Any]], edges: List[Dict[str, Any]]): + self._nodes = nodes + self._edges = edges + + def find_nodes(self, node_type=None): + return self._nodes + + def find_edges(self, edge_type=None): + return self._edges + + +class TestBasicFacts: + def test_add_string_fact(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + assert len(reasoner._all_facts) == 1 + fact = list(reasoner._all_facts)[0] + assert fact.predicate == "parent" + assert fact.args == ("tom", "bob") + + def test_add_dict_fact(self, reasoner): + reasoner.add_fact({"subject": "bob", "predicate": "parent", "object": "ann"}) + assert len(reasoner._all_facts) == 1 + fact = list(reasoner._all_facts)[0] + assert fact.predicate == "parent" + assert fact.args == ("bob", "ann") + + def test_duplicate_fact_ignored(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_fact("parent(tom, bob)") + assert len(reasoner._all_facts) == 1 + + def test_empty_arguments_raise_error(self, reasoner): + # Proves Issue #6 is fixed + with pytest.raises(ValueError, match="Empty argument"): + reasoner.add_fact("parent( )") + with pytest.raises(ValueError, match="Empty argument"): + reasoner.add_fact("parent(tom, )") + + +class TestRules: + def test_single_rule(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + + derived = reasoner.derive_all() + assert "ancestor(tom, bob)" in derived + + def test_recursive_ancestor(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_fact("parent(bob, ann)") + + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Z), ancestor(Z, Y).") + + derived = reasoner.derive_all() + assert "ancestor(tom, bob)" in derived + assert "ancestor(bob, ann)" in derived + assert "ancestor(tom, ann)" in derived + + def test_multi_hop_three_levels(self, reasoner): + reasoner.add_fact("edge(1, 2)") + reasoner.add_fact("edge(2, 3)") + reasoner.add_fact("edge(3, 4)") + + reasoner.add_rule("reachable(X, Y) :- edge(X, Y).") + reasoner.add_rule("reachable(X, Y) :- edge(X, Z), reachable(Z, Y).") + + derived = reasoner.derive_all() + assert "reachable(1, 4)" in derived + + def test_two_body_atoms(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_fact("parent(bob, ann)") + + reasoner.add_rule("grandparent(X, Y) :- parent(X, Z), parent(Z, Y).") + + derived = reasoner.derive_all() + assert "grandparent(tom, ann)" in derived + assert "grandparent(tom, bob)" not in derived + + +class TestQuery: + def test_variable_binding(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_fact("parent(tom, alex)") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + + results = reasoner.query("ancestor(tom, ?Y)") + y_bindings = sorted([res["Y"] for res in results]) + assert y_bindings == ["alex", "bob"] + + def test_lowercase_variable_query(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + results = reasoner.query("parent(tom, ?y)") + assert len(results) == 1 + assert results[0]["y"] == "bob" + + def test_pre_bound_variable(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + + results_bob = reasoner.query("ancestor(tom, ?Y)", bindings={"Y": "bob"}) + assert len(results_bob) == 1 + assert results_bob[0]["Y"] == "bob" + + results_ann = reasoner.query("ancestor(tom, ?Y)", bindings={"Y": "ann"}) + assert len(results_ann) == 0 + + def test_no_match_returns_empty(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + results = reasoner.query("parent(sarah, ?Y)") + assert results == [] + + +class TestContextGraphIntegration: + def test_load_from_graph(self, reasoner): + graph = MockContextGraph( + nodes=[{"id": "microsoft", "type": "company"}], + edges=[{"source": "microsoft", "target": "openai", "type": "invested_in"}] + ) + + added = reasoner.load_from_graph(graph) + assert added == 2 + + assert DatalogFact("company", ("microsoft",)) in reasoner._all_facts + assert DatalogFact("invested_in", ("microsoft", "openai")) in reasoner._all_facts + + def test_edge_becomes_fact(self, reasoner): + graph = MockContextGraph( + nodes=[], + edges=[ + {"source": "alice", "target": "bob", "type": "manages"}, + {"source": "bob", "target": "carol", "type": "manages"}, + ] + ) + reasoner.load_from_graph(graph) + + assert DatalogFact("manages", ("alice", "bob")) in reasoner._all_facts + assert DatalogFact("manages", ("bob", "carol")) in reasoner._all_facts + + def test_derive_after_load(self, reasoner): + graph = MockContextGraph( + nodes=[], + edges=[ + {"source": "alice", "target": "bob", "type": "manages"}, + {"source": "bob", "target": "carol", "type": "manages"}, + ] + ) + reasoner.load_from_graph(graph) + + reasoner.add_rule("transitive_manages(X, Y) :- manages(X, Y).") + reasoner.add_rule("transitive_manages(X, Y) :- manages(X, Z), transitive_manages(Z, Y).") + + derived = reasoner.derive_all() + assert "transitive_manages(alice, carol)" in derived + + +class TestEdgeCases: + def test_empty_program(self, reasoner): + derived = reasoner.derive_all() + assert derived == [] + + def test_derive_all_idempotent(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + + first_run = len(reasoner.derive_all()) + second_run = len(reasoner.derive_all()) + + assert first_run == second_run + assert first_run == 2 + + def test_clear_resets_state(self, reasoner): + reasoner.add_fact("parent(tom, bob)") + reasoner.add_rule("ancestor(X, Y) :- parent(X, Y).") + reasoner.derive_all() + + reasoner.clear() + + assert len(reasoner._all_facts) == 0 + assert len(reasoner._rules) == 0 + assert len(reasoner._delta_new) == 0 + assert len(reasoner._delta_old) == 0 + assert len(reasoner._fact_index) == 0 \ No newline at end of file diff --git a/tests/test_novita_integration.py b/tests/test_novita_integration.py new file mode 100644 index 00000000..c5429d21 --- /dev/null +++ b/tests/test_novita_integration.py @@ -0,0 +1,65 @@ + +import os +import sys +import pytest + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), '..'))) + +from semantica.semantic_extract.methods import ( + extract_entities_llm, + extract_relations_llm, + extract_triplets_llm, +) +from semantica.semantic_extract.providers import create_provider +from semantica.semantic_extract import Entity + +NOVITA_API_KEY = os.environ.get("NOVITA_API_KEY") +NOVITA_MODEL = "deepseek/deepseek-v3.2" +TEXT = ( + "Apple Inc. was founded by Steve Jobs, Steve Wozniak, and Ronald Wayne in 1976. " + "It is headquartered in Cupertino, California. The company designs, manufactures, " + "and markets smartphones, personal computers, tablets, wearables, and accessories." +) + +pytestmark = pytest.mark.skipif( + not NOVITA_API_KEY, + reason="NOVITA_API_KEY not set", +) + + +def test_novita_provider_available(): + provider = create_provider("novita") + assert provider.is_available(), "Novita provider not available โ€” check NOVITA_API_KEY and openai install" + + +def test_novita_entity_extraction(): + entities = extract_entities_llm(TEXT, provider="novita", model=NOVITA_MODEL) + assert isinstance(entities, list), "Expected a list of entities" + assert len(entities) > 0, "No entities extracted" + + +def test_novita_relation_extraction(): + sample_entities = [ + Entity(name="Apple Inc.", type="ORGANIZATION"), + Entity(name="Steve Jobs", type="PERSON"), + ] + relations = extract_relations_llm(TEXT, entities=sample_entities, provider="novita", model=NOVITA_MODEL) + assert isinstance(relations, list), "Expected a list of relations" + + +def test_novita_triplet_extraction(): + triplets = extract_triplets_llm(TEXT, provider="novita", model=NOVITA_MODEL) + assert isinstance(triplets, list), "Expected a list of triplets" + assert len(triplets) > 0, "No triplets extracted" + + +def test_novita_chunked_extraction(): + long_text = " ".join([TEXT] * 10) + entities = extract_entities_llm( + long_text, + provider="novita", + model=NOVITA_MODEL, + max_text_length=200, + ) + assert isinstance(entities, list), "Expected a list of entities from chunked extraction" + assert len(entities) > 0, "No entities extracted from chunked text" diff --git a/tests/triplet_store/test_triplet_store.py b/tests/triplet_store/test_triplet_store.py index d6e679ee..4bc72e98 100644 --- a/tests/triplet_store/test_triplet_store.py +++ b/tests/triplet_store/test_triplet_store.py @@ -98,3 +98,67 @@ class TestTripletStore(unittest.TestCase): self.assertTrue(result["success"]) mock_backend_instance.delete_triplet.assert_called_once_with(triplet) + + def test_query_engine_build_values_clause(self): + engine = QueryEngine() + uris = ["http://ex.org/1", "http://ex.org/2"] + + clause = engine.build_values_clause("subject", uris) + self.assertEqual(clause, "VALUES ?subject { }") + + empty_clause = engine.build_values_clause("subject", []) + self.assertEqual(empty_clause, "") + + def test_query_engine_expand_entity_uri_disabled(self): + engine = QueryEngine() + mock_backend = MagicMock() + + result = engine.expand_entity_uri("http://ex.org/1", mock_backend, use_alignments=False) + + # Should return only the original URI and NOT query the store + self.assertEqual(result, ["http://ex.org/1"]) + mock_backend.execute_sparql.assert_not_called() + + def test_query_engine_expand_entity_uri_enabled(self): + engine = QueryEngine() + mock_backend = MagicMock() + + # Mock the backend returning an aligned URI + mock_backend.execute_sparql.return_value = { + "bindings": [{"aligned": {"value": "http://ex.org/aligned_entity"}}] + } + + result = engine.expand_entity_uri("http://ex.org/original", mock_backend, use_alignments=True) + + self.assertIn("http://ex.org/original", result) + self.assertIn("http://ex.org/aligned_entity", result) + self.assertEqual(len(result), 2) + mock_backend.execute_sparql.assert_called_once() + + def test_end_to_end_cross_ontology_uri_flow(self): + """ + Full end-to-end: real expand_entity_uri queries a mock backend, + then build_values_clause injects the results into a SPARQL template. + """ + engine = QueryEngine() + mock_backend = MagicMock() + mock_backend.execute_sparql.return_value = { + "bindings": [{"aligned": {"value": "http://aligned.org/2"}}] + } + + original_uri = "http://ex.org/1" + expanded = engine.expand_entity_uri(original_uri, store_backend=mock_backend, use_alignments=True) + values_clause = engine.build_values_clause("subject", expanded) + + sparql_query = f""" + SELECT ?instance ?name WHERE {{ + {values_clause} + ?instance a ?subject . + ?instance ?name . + }} + """ + + self.assertIn("http://ex.org/1", sparql_query) + self.assertIn("http://aligned.org/2", sparql_query) + self.assertIn("VALUES ?subject", sparql_query) + mock_backend.execute_sparql.assert_called_once()