From 1fceb634aef9bb04db6cf8bafee5a7a19d1bdaed Mon Sep 17 00:00:00 2001 From: KaifAhmad1 Date: Sat, 20 Dec 2025 20:18:14 +0530 Subject: [PATCH] chore: remove 07_Pipeline_Orchestration notebook and all references --- README.md | 1225 ++++++++--------- .../advanced/07_Pipeline_Orchestration.ipynb | 242 ---- docs/cookbook.md | 9 - docs/reference/pipeline.md | 3 - tests/pipeline/test_notebook_07.py | 154 --- tests/test_pipeline_orchestration.py | 4 +- 6 files changed, 614 insertions(+), 1023 deletions(-) delete mode 100644 cookbook/advanced/07_Pipeline_Orchestration.ipynb delete mode 100644 tests/pipeline/test_notebook_07.py diff --git a/README.md b/README.md index a516e1e1..7cf78e66 100644 --- a/README.md +++ b/README.md @@ -1,613 +1,612 @@ -
- -Semantica Logo - -# 🧠 Semantica - -[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/) -[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -[![PyPI version](https://badge.fury.io/py/semantica.svg)](https://pypi.org/project/semantica/0.0.1/) -[![Downloads](https://pepy.tech/badge/semantica)](https://pepy.tech/project/semantica) -[![Discord](https://img.shields.io/discord/semantica?color=7289da&label=discord)](https://discord.gg/semantica) -[![CI](https://github.com/Hawksight-AI/semantica/workflows/CI/badge.svg)](https://github.com/Hawksight-AI/semantica/actions) - -

- - Give a Star - -    - - Support Project - -

- -**Open Source Framework for Semantic Layer & Knowledge Engineering** - -> **Transform chaotic data into intelligent knowledge.** - -*The missing fabric between raw data and AI engineering. A comprehensive open-source framework for building semantic layers and knowledge engineering systems that transform unstructured data into AI-ready knowledge β€” powering Knowledge Graph-Powered RAG (GraphRAG), AI Agents, Multi-Agent Systems, and AI applications with structured semantic knowledge.* - -**100% Open Source** β€’ **MIT Licensed** β€’ **Production Ready** β€’ **Community Driven** - -[**Discord**](https://discord.gg/semantica) β€’ [**GitHub**](https://github.com/Hawksight-AI/semantica) - -
- -## What is Semantica? - -Semantica bridges the gap between raw data chaos and AI-ready knowledge. It's a **semantic intelligence platform** that transforms unstructured data into structured, queryable knowledge graphs powering GraphRAG, AI agents, and multi-agent systems. - -### What Makes Semantica Different? - -Unlike traditional approaches that process isolated documents and extract text into vectors, Semantica understands **semantic relationships across all content**, provides **automated ontology generation**, and builds a **unified semantic layer** with **production-grade QA**. - -| **Traditional Approaches** | **Semantica's Approach** | -|:---------------------------|:-------------------------| -| Process data as isolated documents | Understands semantic relationships across all content | -| Extract text and store vectors | Builds knowledge graphs with meaningful connections | -| Generic entity recognition | General-purpose ontology generation and validation | -| Manual schema definition | Automatic semantic modeling from content patterns | -| Disconnected data silos | Unified semantic layer across all data sources | -| Basic quality checks | Production-grade QA with conflict detection & resolution | - ---- - -## 🎯 The Problem We Solve - -### The Semantic Gap - -Organizations today face a **fundamental mismatch** between how data exists and how AI systems need it. - -#### The Semantic Gap: Problem vs. Solution - -Organizations have **unstructured data** (PDFs, emails, logs), **messy data** (inconsistent formats, duplicates, conflicts), and **disconnected silos** (no shared context, missing relationships). AI systems need **clear rules** (formal ontologies), **structured entities** (validated, consistent), and **relationships** (semantic connections, context-aware reasoning). - -| **What Organizations Have** | **What AI Systems Require** | -|:------------------------------|:------------------------------| -| **Unstructured Data** | **Clear Rules** | -| PDFs, emails, logs | Formal ontologies | -| Mixed schemas | Graphs & Networks | -| Conflicting facts | | -| **Messy, Noisy Data** | **Structured Entities** | -| Inconsistent formats | Validated entities | -| Duplicate records | Domain Knowledge | -| Missing relationships | | -| **Disconnected, Siloed Data** | **Relationships** | -| Data in separate systems | Semantic connections | -| No shared context | Context-Aware Reasoning | -| Isolated knowledge | | - -### **SEMANTICA FRAMEWORK** - -Semantica operates through three integrated layers that transform raw data into AI-ready knowledge: - -**Input Layer** β€” Universal ingestion from 50+ data formats (PDFs, DOCX, HTML, JSON, CSV, databases, live feeds, APIs, streams, archives, multi-modal content) into a unified pipeline. - -**Semantic Layer** β€” Core intelligence engine performing entity extraction, relationship mapping, ontology generation, context engineering, and quality assurance. Includes **advanced entity deduplication** (Jaro-Winkler, disjoint property handling) to ensure a clean single source of truth. - -**Output Layer** β€” Production-ready knowledge graphs, vector embeddings, and validated ontologies that power GraphRAG systems, AI agents, and multi-agent systems. - -**Powers: GraphRAG, AI Agents, Multi-Agent Systems** - -#### Semantica Processing Flow - -
-View Interactive Flowchart - -```mermaid -flowchart TD - A[Raw Data Sources
PDFs, Emails, Logs, Databases
50+ Formats] --> B[Input Layer
Universal Data Ingestion] - B --> C[Format Detection
& Parsing] - C --> D[Normalization
& Preprocessing] - D --> E[Semantic Layer
Core Intelligence] - - E --> F[Entity Extraction
NER + LLM Enhancement] - E --> G[Relationship Mapping
Triplet Generation] - E --> H[Ontology Generation
6-Stage Pipeline] - E --> I[Context Engineering
Semantic Enrichment] - E --> J[Quality Assurance
Conflict Detection] - - F --> K[Output Layer] - G --> K - H --> K - I --> K - J --> K - - K --> L[Knowledge Graphs
Production-Ready] - K --> M[Vector Embeddings
Semantic Search] - K --> N[Ontologies
OWL Validated] - - L --> O[Application Layer] - M --> O - N --> O - - O --> P[GraphRAG Engine
91% Accuracy] - O --> Q[AI Agents
Persistent Memory] - O --> R[Multi-Agent Systems
Shared Models] - O --> S[Analytics & BI
Graph Insights] - - style A fill:#e1f5ff - style E fill:#fff4e1 - style K fill:#e8f5e9 - style O fill:#f3e5f5 -``` - -
- - -### What Happens Without Semantics? - -**They Break** β€” Systems crash due to inconsistent formats and missing structure. - -**They Hallucinate** β€” AI models generate false information without semantic context to validate outputs. - -**They Fail Silently** β€” Systems return wrong answers without warnings, leading to bad decisions. - -**Why?** Systems have data β€” not semantics. They can't connect concepts, understand relationships, validate against domain rules, or detect conflicts. - ---- - -## πŸ’‘ The Semantica Solution - -**Semantica** is an **open-source framework** that closes the semantic gap between real-world messy data and the structured semantic layers required by advanced AI systems β€” GraphRAG, agents, multi-agent systems, reasoning models, and more. - -### How Semantica Solves These Problems - -**Efficient Embeddings** β€” Uses **FastEmbed** by default for high-performance, lightweight local embedding generation (faster than sentence-transformers). - -**Universal Data Ingestion** β€” Handles 50+ formats (PDF, DOCX, HTML, JSON, CSV, databases, APIs, streams) with unified pipeline, no custom parsers needed. - -**Automated Semantic Extraction** β€” NER, relationship extraction, and triplet generation with LLM enhancement discovers entities and relationships automatically. - -**Knowledge Graph Construction** β€” Production-ready graphs with entity resolution, temporal support, and graph analytics. Queryable knowledge ready for AI applications. - -**GraphRAG Engine** β€” Hybrid vector + graph retrieval achieves 91% accuracy (30% improvement) via semantic search + graph traversal for multi-hop reasoning. - -**AI Agent Context Engineering** β€” Persistent memory with RAG + knowledge graphs enables context maintenance, action validation, and structured knowledge access. - -**Automated Ontology Generation** β€” 6-stage LLM pipeline generates validated OWL ontologies with HermiT/Pellet validation, eliminating manual engineering. - -**Production-Grade QA** β€” Conflict detection, deduplication, quality scoring, and provenance tracking ensure trusted, production-ready knowledge graphs. - -**Pipeline Orchestration** β€” Flexible pipeline builder with parallel execution enables scalable processing via orchestrator-worker pattern. - -### Core Features at a Glance - -| **Feature Category** | **Capabilities** | **Key Benefits** | -|:---------------------|:-----------------|:------------------| -| **Data Ingestion** | 50+ formats (PDF, DOCX, HTML, JSON, CSV, databases, APIs, streams, archives) | Universal ingestion, no custom parsers needed | -| **Semantic Extraction** | NER, relationship extraction, triplet generation, LLM enhancement | Automated discovery of entities and relationships | -| **Knowledge Graphs** | Entity resolution, temporal support, graph analytics, query interface | Production-ready, queryable knowledge structures | -| **Ontology Generation** | 6-stage LLM pipeline, OWL generation, HermiT/Pellet validation | Automated ontology creation from documents | -| **GraphRAG** | Hybrid vector + graph retrieval, multi-hop reasoning | 91% accuracy, 30% improvement over vector-only | -| **Agent Memory** | Persistent memory (Save/Load), Hybrid Retrieval (Vector+Graph), FastEmbed support | Context-aware agents with semantic understanding | -| **Pipeline Orchestration** | Parallel execution, custom steps, orchestrator-worker pattern | Scalable, flexible data processing | -| **Quality Assurance** | Conflict detection, deduplication, quality scoring, provenance | Trusted knowledge graphs ready for production | - ---- - -## πŸ‘₯ Who Is This For? - -Semantica is designed for **developers, data engineers, and organizations** building the next generation of AI applications that require semantic understanding and knowledge graphs. - -### Who Uses Semantica - -**AI/ML Engineers & Data Scientists** β€” Build GraphRAG systems, AI agents, and multi-agent systems. - -**Data Engineers** β€” Build scalable pipelines with semantic enrichment. - -**Knowledge Engineers & Ontologists** β€” Create knowledge graphs and ontologies with automated pipelines. - -**Enterprise Data Teams** β€” Unify semantic layers, improve data quality, resolve conflicts. - -**Software & DevOps Engineers** β€” Build semantic APIs and infrastructure with production-ready SDK. - -**Analysts & Researchers** β€” Transform data into queryable knowledge graphs for insights. - -**Security & Compliance Teams** β€” Threat intelligence, regulatory reporting, audit trails. - -**Product Teams & Startups** β€” Rapid prototyping of AI products and semantic features. - ---- - -## πŸ“¦ Installation - -**Prerequisites:** Python 3.8+ (3.9+ recommended) β€’ pip (latest version) - -### Install from PyPI (Recommended) - -```bash -# Install latest version from PyPI -pip install semantica - -# Or install with optional dependencies -pip install semantica[all] - -# Verify installation -python -c "import semantica; print(semantica.__version__)" -``` - -**Current Version:** [![PyPI version](https://badge.fury.io/py/semantica.svg)](https://pypi.org/project/semantica/0.0.1/) β€’ [View on PyPI](https://pypi.org/project/semantica/0.0.1/) - -### Install from Source (Development) - -```bash -# Clone and install in editable mode -git clone https://github.com/Hawksight-AI/semantica.git -cd semantica -pip install -e . - -# Or with all optional dependencies -pip install -e ".[all]" - -# Development setup -pip install -e ".[dev]" -``` - -## πŸ“š Resources - -> **New to Semantica?** Check out the [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) for hands-on examples! - -- [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) - 50+ interactive notebooks - - [Introduction](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction) - Getting started tutorials - - [Advanced](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced) - Advanced techniques - - [Use Cases](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases) - Real-world applications - -## ✨ Core Capabilities - -| **Data Ingestion** | **Semantic Extract** | **Knowledge Graphs** | **Ontology** | -|:--------------------:|:----------------------:|:----------------------:|:--------------:| -| [50+ Formats](#universal-data-ingestion) | [Entity & Relations](#semantic-intelligence-engine) | [Graph Analytics](#knowledge-graph-construction) | [Auto Generation](#ontology-generation--management) | -| **Context** | **GraphRAG** | **Pipeline** | **QA** | -| [Agent Memory](#context-engineering-for-ai-agents) | [Hybrid RAG](#knowledge-graph-powered-rag-graphrag) | [Parallel Workers](#pipeline-orchestration--parallel-processing) | [Conflict Resolution](#production-ready-quality-assurance) | - ---- - -### Universal Data Ingestion - -> **50+ file formats** β€’ PDF, DOCX, HTML, JSON, CSV, databases, feeds, archives - -```python -from semantica.ingest import FileIngestor, WebIngestor, DBIngestor - -file_ingestor = FileIngestor(recursive=True) -web_ingestor = WebIngestor(max_depth=3) -db_ingestor = DBIngestor(connection_string="postgresql://...") - -sources = [] -sources.extend(file_ingestor.ingest("documents/")) -sources.extend(web_ingestor.ingest("https://example.com")) -sources.extend(db_ingestor.ingest(query="SELECT * FROM articles")) - -print(f" Ingested {len(sources)} sources") -``` - -[**Cookbook: Data Ingestion**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/02_Data_Ingestion.ipynb) β€’ [**Document Parsing**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/03_Document_Parsing.ipynb) β€’ [**Data Normalization**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/04_Data_Normalization.ipynb) β€’ [**Chunking & Splitting**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb) - -### Semantic Intelligence Engine - -> **Entity & Relation Extraction** β€’ NER, Relationships, Events, Triplets with LLM Enhancement - -```python -from semantica.core import Semantica - -text = "Apple Inc., founded by Steve Jobs in 1976, acquired Beats Electronics for $3 billion." - -core = Semantica(ner_model="transformer", relation_strategy="hybrid") -results = core.extract_semantics(text) - -print(f"Entities: {len(results.entities)}, Relationships: {len(results.relationships)}") -``` - -[**Cookbook: Entity Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/05_Entity_Extraction.ipynb) β€’ [**Relation Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/06_Relation_Extraction.ipynb) β€’ [**Advanced Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/01_Advanced_Extraction.ipynb) - -### Knowledge Graph Construction - -> **Production-Ready KGs** β€’ Entity Resolution β€’ Temporal Support β€’ Graph Analytics - -```python -from semantica.core import Semantica -from semantica.kg import GraphAnalyzer - -documents = ["doc1.txt", "doc2.txt", "doc3.txt"] -core = Semantica(graph_db="neo4j", merge_entities=True) -kg = core.build_knowledge_graph(documents, generate_embeddings=True) - -analyzer = GraphAnalyzer() -pagerank = analyzer.compute_centrality(kg, method="pagerank") -communities = analyzer.detect_communities(kg, method="louvain") - -result = kg.query("Who founded the company?", return_format="structured") -print(f"Nodes: {kg.node_count}, Answer: {result.answer}") -``` - -[**Cookbook: Building Knowledge Graphs**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb) β€’ [**Graph Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/09_Graph_Store.ipynb) β€’ [**Triplet Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/20_Triplet_Store.ipynb) β€’ [**Visualization**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/16_Visualization.ipynb) - -[**Graph Analytics**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/10_Graph_Analytics.ipynb) β€’ [**Advanced Graph Analytics**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/02_Advanced_Graph_Analytics.ipynb) - -### Triplet Store Integration - -> **SPARQL Support** β€’ **Blazegraph, Jena, RDF4J** β€’ **Reasoning & Inference** - -```python -from semantica.triplet_store import TripletStore - -# Initialize store (Blazegraph, Jena, or RDF4J) -store = TripletStore(backend="blazegraph", endpoint="http://localhost:9999/blazegraph") - -# Add triplets and execute SPARQL queries -store.add_triplet({ - "subject": "http://example.org/Alice", - "predicate": "http://example.org/knows", - "object": "http://example.org/Bob" -}) - -results = store.execute_query("SELECT ?s ?p ?o WHERE { ?s ?p ?o } LIMIT 10") -``` - -[**Cookbook: Triplet Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/20_Triplet_Store.ipynb) - -### Ontology Generation & Management - -> **6-Stage LLM Pipeline** β€’ Automatic OWL Generation β€’ HermiT/Pellet Validation - -```python -from semantica.ontology import OntologyGenerator - -generator = OntologyGenerator(llm_provider="openai", model="gpt-4") -ontology = generator.generate_from_documents(sources=["domain_docs/"]) - -print(f"Classes: {len(ontology.classes)}") -``` - -[**Cookbook: Ontology**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/14_Ontology.ipynb) - -### Context Engineering & Memory Systems - -> **Persistent Memory** β€’ **Hybrid Retrieval (Vector + Graph)** β€’ **Hierarchical Storage** β€’ **Entity Linking** - -```python -from semantica.context import AgentContext -from semantica.vector_store import VectorStore - -# Initialize Context with Hybrid Retrieval (Graph + Vector) -context = AgentContext( - vector_store=VectorStore(backend="faiss"), - hybrid_alpha=0.75 # 75% weight to Knowledge Graph, 25% to Vector -) - -# Store memory with automatic entity linking -context.store( - "User is building a RAG system with Semantica", - metadata={"priority": "high", "topic": "rag"} -) - -# Retrieve with context expansion -results = context.retrieve("What is the user building?", use_graph_expansion=True) -``` - -**Core Notebooks:** -- [**Context Module Introduction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/19_Context_Module.ipynb) - Basic memory and storage. -- [**Advanced Context Engineering**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/11_Advanced_Context_Engineering.ipynb) - Hybrid retrieval, graph builders, and custom memory policies. - -**Related Components:** -[**Vector Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/13_Vector_Store.ipynb) β€’ [**Embedding Generation**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/12_Embedding_Generation.ipynb) β€’ [**Advanced Vector Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb) - -### Knowledge Graph-Powered RAG (GraphRAG) - -> **30% Accuracy Improvement** β€’ Vector + Graph Hybrid Search β€’ 91% Accuracy - -```python -from semantica.qa_rag import GraphRAGEngine -from semantica.vector_store import VectorStore - -graphrag = GraphRAGEngine( - vector_store=VectorStore(backend="faiss"), - knowledge_graph=kg -) -result = graphrag.query("Who founded the company?", top_k=5, expand_graph=True) -print(f"Answer: {result.answer} (Confidence: {result.confidence:.2f})") -``` - -[**Cookbook: GraphRAG**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases/advanced_rag/01_GraphRAG_Complete.ipynb) - -### Pipeline Orchestration & Parallel Processing - -> **Orchestrator-Worker Pattern** β€’ Parallel Execution β€’ Scalable Processing - -```python -from semantica.pipeline import PipelineBuilder, ExecutionEngine - -pipeline = PipelineBuilder() \ - .add_step("ingest", "custom", func=ingest_data) \ - .add_step("extract", "custom", func=extract_entities) \ - .add_step("build", "custom", func=build_graph) \ - .build() - -result = ExecutionEngine().execute_pipeline(pipeline, parallel=True) -``` - -[**Cookbook: Pipeline Orchestration**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/07_Pipeline_Orchestration.ipynb) - -### Production-Ready Quality Assurance - -> **Enterprise-Grade QA** β€’ Conflict Detection β€’ Deduplication - -```python -from semantica.deduplication import DuplicateDetector -from semantica.conflicts import ConflictDetector - -entities = kg.get("entities", []) -conflicts = ConflictDetector().detect_conflicts(entities) -duplicates = DuplicateDetector(similarity_threshold=0.85).detect_duplicates(entities) - -print(f"Conflicts: {len(conflicts)} | Duplicates: {len(duplicates)}") -``` - -[**Cookbook: Conflict Detection & Resolution**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/17_Conflict_Detection_and_Resolution.ipynb) β€’ [**Deduplication**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/18_Deduplication.ipynb) - -### Export & Integration - -> **Multi-Format Export** β€’ JSON, CSV, RDF, GraphML - -```python -from semantica.export import GraphExporter - -exporter = GraphExporter(kg) -exporter.export("graph.json", format="json") -exporter.export("graph.ttl", format="turtle") -``` - -[**Cookbook: Export**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/15_Export.ipynb) β€’ [**Multi-Format Export**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/05_Multi_Format_Export.ipynb) β€’ [**Multi-Source Integration**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb) - -## πŸš€ Quick Start - -> **For comprehensive examples, see the [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) with 50+ interactive notebooks!** - -```python -from semantica.core import Semantica - -# Initialize and build knowledge graph -core = Semantica(ner_model="transformer", relation_strategy="hybrid") -documents = ["doc1.txt", "doc2.txt", "doc3.txt"] -kg = core.build_knowledge_graph(documents, merge_entities=True) - -# Query the graph -result = kg.query("Who founded the company?", return_format="structured") -print(f"Answer: {result.answer} | Nodes: {kg.node_count}, Edges: {kg.edge_count}") -``` - -[**Cookbook: Your First Knowledge Graph**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb) - -## 🎯 Use Cases - -**Enterprise Knowledge Engineering** β€” Unify data sources into knowledge graphs, breaking down silos. - -**AI Agents & Autonomous Systems** β€” Build agents with persistent memory and semantic understanding. - -**Multi-Format Document Processing** β€” Process 50+ formats through a unified pipeline. - -**Data Pipeline Processing** β€” Build scalable pipelines with parallel execution. - -**Intelligence & Security** β€” Analyze networks, threat intelligence, forensic analysis. - -**Finance & Trading** β€” Fraud detection, market intelligence, risk assessment. - -**Healthcare & Biomedical** β€” Clinical reports, drug discovery, medical literature analysis. - -[**Explore Use Case Examples**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases) β€” See real-world implementations in finance, healthcare, cybersecurity, trading, and more. - -## πŸ”¬ Advanced Features - -**Incremental Updates** β€” Real-time stream processing with Kafka, RabbitMQ, Kinesis for live updates. - -**Multi-Language Support** β€” Process 50+ languages with automatic detection. - -**Custom Ontology Import** β€” Import and extend Schema.org and custom ontologies. - -**Advanced Reasoning** β€” Deductive, inductive, abductive reasoning with HermiT/Pellet. - -**Graph Analytics** β€” Centrality, community detection, path finding, temporal analysis. - -**Custom Pipelines** β€” Build custom pipelines with parallel execution. - -**API Integration** β€” Integrate external APIs for entity enrichment. - -[**See Advanced Examples**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced) β€” Advanced extraction, graph analytics, reasoning, and more. - -## πŸ—ΊοΈ Roadmap - -### Q1 2026 -- [x] Core framework (v1.0) -- [x] GraphRAG engine -- [x] 6-stage ontology pipeline -- [ ] Quality assurance features and Quality Assurance module -- [ ] Enhanced multi-language support -- [ ] Real-time streaming improvements -- [ ] Advanced reasoning v2 - -### Q2 2026 -- [ ] Multi-modal processing - ---- - -## 🀝 Community & Support - -### Join Our Community - -| **Channel** | **Purpose** | -|:-----------:|:-----------| -| [**Discord**](https://discord.gg/semantica) | Real-time help, showcases | -| [**GitHub Discussions**](https://github.com/Hawksight-AI/semantica/discussions) | Q&A, feature requests | - -### Learning Resources - - -### Enterprise Support - -| **Tier** | **Features** | **SLA** | **Price** | -|:--------:|:-----------|:-------:|:--------:| -| **Community** | Public support | Best effort | Free | -| **Professional** | Email support | 48h | Contact | -| **Enterprise** | 24/7 support | 4h | Contact | -| **Premium** | Phone, custom dev | 1h | Contact | - -**Contact:** [GitHub Issues](https://github.com/Hawksight-AI/semantica/issues) with "[Enterprise]" prefix - -## 🀝 Contributing - -### How to Contribute - -```bash -# Fork and clone -git clone https://github.com/your-username/semantica.git -cd semantica - -# Create branch -git checkout -b feature/your-feature - -# Install dev dependencies -pip install -e ".[dev,test]" - -# Make changes and test -pytest tests/ -black semantica/ -flake8 semantica/ - -# Commit and push -git commit -m "Add feature" -git push origin feature/your-feature -``` - -### Contribution Types - -1. **Code** - New features, bug fixes -2. **Documentation** - Improvements, tutorials -3. **Bug Reports** - [Create issue](https://github.com/Hawksight-AI/semantica/issues/new) -4. **Feature Requests** - [Request feature](https://github.com/Hawksight-AI/semantica/issues/new) - -### Recognition - -Contributors receive: -- Recognition in [CONTRIBUTORS.md](https://github.com/Hawksight-AI/semantica/blob/main/CONTRIBUTORS.md) -- GitHub badges -- Semantica swag -- Featured showcases - -## πŸ† Contributors - - - Contributors - - -## πŸ“œ License - -Semantica is licensed under the **MIT License** - see the [LICENSE](https://github.com/Hawksight-AI/semantica/blob/main/LICENSE) file for details. - -
- -**Built by the Semantica Community** - -[GitHub](https://github.com/Hawksight-AI/semantica) β€’ [Discord](https://discord.gg/semantica) - -
+
+ +Semantica Logo + +# 🧠 Semantica + +[![Python 3.8+](https://img.shields.io/badge/python-3.8+-blue.svg)](https://www.python.org/downloads/) +[![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) +[![PyPI version](https://badge.fury.io/py/semantica.svg)](https://pypi.org/project/semantica/0.0.1/) +[![Downloads](https://pepy.tech/badge/semantica)](https://pepy.tech/project/semantica) +[![Discord](https://img.shields.io/discord/semantica?color=7289da&label=discord)](https://discord.gg/semantica) +[![CI](https://github.com/Hawksight-AI/semantica/workflows/CI/badge.svg)](https://github.com/Hawksight-AI/semantica/actions) + +

+ + Give a Star + +    + + Support Project + +

+ +**Open Source Framework for Semantic Layer & Knowledge Engineering** + +> **Transform chaotic data into intelligent knowledge.** + +*The missing fabric between raw data and AI engineering. A comprehensive open-source framework for building semantic layers and knowledge engineering systems that transform unstructured data into AI-ready knowledge β€” powering Knowledge Graph-Powered RAG (GraphRAG), AI Agents, Multi-Agent Systems, and AI applications with structured semantic knowledge.* + +**100% Open Source** β€’ **MIT Licensed** β€’ **Production Ready** β€’ **Community Driven** + +[**Discord**](https://discord.gg/semantica) β€’ [**GitHub**](https://github.com/Hawksight-AI/semantica) + +
+ +## What is Semantica? + +Semantica bridges the gap between raw data chaos and AI-ready knowledge. It's a **semantic intelligence platform** that transforms unstructured data into structured, queryable knowledge graphs powering GraphRAG, AI agents, and multi-agent systems. + +### What Makes Semantica Different? + +Unlike traditional approaches that process isolated documents and extract text into vectors, Semantica understands **semantic relationships across all content**, provides **automated ontology generation**, and builds a **unified semantic layer** with **production-grade QA**. + +| **Traditional Approaches** | **Semantica's Approach** | +|:---------------------------|:-------------------------| +| Process data as isolated documents | Understands semantic relationships across all content | +| Extract text and store vectors | Builds knowledge graphs with meaningful connections | +| Generic entity recognition | General-purpose ontology generation and validation | +| Manual schema definition | Automatic semantic modeling from content patterns | +| Disconnected data silos | Unified semantic layer across all data sources | +| Basic quality checks | Production-grade QA with conflict detection & resolution | + +--- + +## 🎯 The Problem We Solve + +### The Semantic Gap + +Organizations today face a **fundamental mismatch** between how data exists and how AI systems need it. + +#### The Semantic Gap: Problem vs. Solution + +Organizations have **unstructured data** (PDFs, emails, logs), **messy data** (inconsistent formats, duplicates, conflicts), and **disconnected silos** (no shared context, missing relationships). AI systems need **clear rules** (formal ontologies), **structured entities** (validated, consistent), and **relationships** (semantic connections, context-aware reasoning). + +| **What Organizations Have** | **What AI Systems Require** | +|:------------------------------|:------------------------------| +| **Unstructured Data** | **Clear Rules** | +| PDFs, emails, logs | Formal ontologies | +| Mixed schemas | Graphs & Networks | +| Conflicting facts | | +| **Messy, Noisy Data** | **Structured Entities** | +| Inconsistent formats | Validated entities | +| Duplicate records | Domain Knowledge | +| Missing relationships | | +| **Disconnected, Siloed Data** | **Relationships** | +| Data in separate systems | Semantic connections | +| No shared context | Context-Aware Reasoning | +| Isolated knowledge | | + +### **SEMANTICA FRAMEWORK** + +Semantica operates through three integrated layers that transform raw data into AI-ready knowledge: + +**Input Layer** β€” Universal ingestion from 50+ data formats (PDFs, DOCX, HTML, JSON, CSV, databases, live feeds, APIs, streams, archives, multi-modal content) into a unified pipeline. + +**Semantic Layer** β€” Core intelligence engine performing entity extraction, relationship mapping, ontology generation, context engineering, and quality assurance. Includes **advanced entity deduplication** (Jaro-Winkler, disjoint property handling) to ensure a clean single source of truth. + +**Output Layer** β€” Production-ready knowledge graphs, vector embeddings, and validated ontologies that power GraphRAG systems, AI agents, and multi-agent systems. + +**Powers: GraphRAG, AI Agents, Multi-Agent Systems** + +#### Semantica Processing Flow + +
+View Interactive Flowchart + +```mermaid +flowchart TD + A[Raw Data Sources
PDFs, Emails, Logs, Databases
50+ Formats] --> B[Input Layer
Universal Data Ingestion] + B --> C[Format Detection
& Parsing] + C --> D[Normalization
& Preprocessing] + D --> E[Semantic Layer
Core Intelligence] + + E --> F[Entity Extraction
NER + LLM Enhancement] + E --> G[Relationship Mapping
Triplet Generation] + E --> H[Ontology Generation
6-Stage Pipeline] + E --> I[Context Engineering
Semantic Enrichment] + E --> J[Quality Assurance
Conflict Detection] + + F --> K[Output Layer] + G --> K + H --> K + I --> K + J --> K + + K --> L[Knowledge Graphs
Production-Ready] + K --> M[Vector Embeddings
Semantic Search] + K --> N[Ontologies
OWL Validated] + + L --> O[Application Layer] + M --> O + N --> O + + O --> P[GraphRAG Engine
91% Accuracy] + O --> Q[AI Agents
Persistent Memory] + O --> R[Multi-Agent Systems
Shared Models] + O --> S[Analytics & BI
Graph Insights] + + style A fill:#e1f5ff + style E fill:#fff4e1 + style K fill:#e8f5e9 + style O fill:#f3e5f5 +``` + +
+ + +### What Happens Without Semantics? + +**They Break** β€” Systems crash due to inconsistent formats and missing structure. + +**They Hallucinate** β€” AI models generate false information without semantic context to validate outputs. + +**They Fail Silently** β€” Systems return wrong answers without warnings, leading to bad decisions. + +**Why?** Systems have data β€” not semantics. They can't connect concepts, understand relationships, validate against domain rules, or detect conflicts. + +--- + +## πŸ’‘ The Semantica Solution + +**Semantica** is an **open-source framework** that closes the semantic gap between real-world messy data and the structured semantic layers required by advanced AI systems β€” GraphRAG, agents, multi-agent systems, reasoning models, and more. + +### How Semantica Solves These Problems + +**Efficient Embeddings** β€” Uses **FastEmbed** by default for high-performance, lightweight local embedding generation (faster than sentence-transformers). + +**Universal Data Ingestion** β€” Handles 50+ formats (PDF, DOCX, HTML, JSON, CSV, databases, APIs, streams) with unified pipeline, no custom parsers needed. + +**Automated Semantic Extraction** β€” NER, relationship extraction, and triplet generation with LLM enhancement discovers entities and relationships automatically. + +**Knowledge Graph Construction** β€” Production-ready graphs with entity resolution, temporal support, and graph analytics. Queryable knowledge ready for AI applications. + +**GraphRAG Engine** β€” Hybrid vector + graph retrieval achieves 91% accuracy (30% improvement) via semantic search + graph traversal for multi-hop reasoning. + +**AI Agent Context Engineering** β€” Persistent memory with RAG + knowledge graphs enables context maintenance, action validation, and structured knowledge access. + +**Automated Ontology Generation** β€” 6-stage LLM pipeline generates validated OWL ontologies with HermiT/Pellet validation, eliminating manual engineering. + +**Production-Grade QA** β€” Conflict detection, deduplication, quality scoring, and provenance tracking ensure trusted, production-ready knowledge graphs. + +**Pipeline Orchestration** β€” Flexible pipeline builder with parallel execution enables scalable processing via orchestrator-worker pattern. + +### Core Features at a Glance + +| **Feature Category** | **Capabilities** | **Key Benefits** | +|:---------------------|:-----------------|:------------------| +| **Data Ingestion** | 50+ formats (PDF, DOCX, HTML, JSON, CSV, databases, APIs, streams, archives) | Universal ingestion, no custom parsers needed | +| **Semantic Extraction** | NER, relationship extraction, triplet generation, LLM enhancement | Automated discovery of entities and relationships | +| **Knowledge Graphs** | Entity resolution, temporal support, graph analytics, query interface | Production-ready, queryable knowledge structures | +| **Ontology Generation** | 6-stage LLM pipeline, OWL generation, HermiT/Pellet validation | Automated ontology creation from documents | +| **GraphRAG** | Hybrid vector + graph retrieval, multi-hop reasoning | 91% accuracy, 30% improvement over vector-only | +| **Agent Memory** | Persistent memory (Save/Load), Hybrid Retrieval (Vector+Graph), FastEmbed support | Context-aware agents with semantic understanding | +| **Pipeline Orchestration** | Parallel execution, custom steps, orchestrator-worker pattern | Scalable, flexible data processing | +| **Quality Assurance** | Conflict detection, deduplication, quality scoring, provenance | Trusted knowledge graphs ready for production | + +--- + +## πŸ‘₯ Who Is This For? + +Semantica is designed for **developers, data engineers, and organizations** building the next generation of AI applications that require semantic understanding and knowledge graphs. + +### Who Uses Semantica + +**AI/ML Engineers & Data Scientists** β€” Build GraphRAG systems, AI agents, and multi-agent systems. + +**Data Engineers** β€” Build scalable pipelines with semantic enrichment. + +**Knowledge Engineers & Ontologists** β€” Create knowledge graphs and ontologies with automated pipelines. + +**Enterprise Data Teams** β€” Unify semantic layers, improve data quality, resolve conflicts. + +**Software & DevOps Engineers** β€” Build semantic APIs and infrastructure with production-ready SDK. + +**Analysts & Researchers** β€” Transform data into queryable knowledge graphs for insights. + +**Security & Compliance Teams** β€” Threat intelligence, regulatory reporting, audit trails. + +**Product Teams & Startups** β€” Rapid prototyping of AI products and semantic features. + +--- + +## πŸ“¦ Installation + +**Prerequisites:** Python 3.8+ (3.9+ recommended) β€’ pip (latest version) + +### Install from PyPI (Recommended) + +```bash +# Install latest version from PyPI +pip install semantica + +# Or install with optional dependencies +pip install semantica[all] + +# Verify installation +python -c "import semantica; print(semantica.__version__)" +``` + +**Current Version:** [![PyPI version](https://badge.fury.io/py/semantica.svg)](https://pypi.org/project/semantica/0.0.1/) β€’ [View on PyPI](https://pypi.org/project/semantica/0.0.1/) + +### Install from Source (Development) + +```bash +# Clone and install in editable mode +git clone https://github.com/Hawksight-AI/semantica.git +cd semantica +pip install -e . + +# Or with all optional dependencies +pip install -e ".[all]" + +# Development setup +pip install -e ".[dev]" +``` + +## πŸ“š Resources + +> **New to Semantica?** Check out the [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) for hands-on examples! + +- [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) - 50+ interactive notebooks + - [Introduction](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction) - Getting started tutorials + - [Advanced](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced) - Advanced techniques + - [Use Cases](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases) - Real-world applications + +## ✨ Core Capabilities + +| **Data Ingestion** | **Semantic Extract** | **Knowledge Graphs** | **Ontology** | +|:--------------------:|:----------------------:|:----------------------:|:--------------:| +| [50+ Formats](#universal-data-ingestion) | [Entity & Relations](#semantic-intelligence-engine) | [Graph Analytics](#knowledge-graph-construction) | [Auto Generation](#ontology-generation--management) | +| **Context** | **GraphRAG** | **Pipeline** | **QA** | +| [Agent Memory](#context-engineering-for-ai-agents) | [Hybrid RAG](#knowledge-graph-powered-rag-graphrag) | [Parallel Workers](#pipeline-orchestration--parallel-processing) | [Conflict Resolution](#production-ready-quality-assurance) | + +--- + +### Universal Data Ingestion + +> **50+ file formats** β€’ PDF, DOCX, HTML, JSON, CSV, databases, feeds, archives + +```python +from semantica.ingest import FileIngestor, WebIngestor, DBIngestor + +file_ingestor = FileIngestor(recursive=True) +web_ingestor = WebIngestor(max_depth=3) +db_ingestor = DBIngestor(connection_string="postgresql://...") + +sources = [] +sources.extend(file_ingestor.ingest("documents/")) +sources.extend(web_ingestor.ingest("https://example.com")) +sources.extend(db_ingestor.ingest(query="SELECT * FROM articles")) + +print(f" Ingested {len(sources)} sources") +``` + +[**Cookbook: Data Ingestion**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/02_Data_Ingestion.ipynb) β€’ [**Document Parsing**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/03_Document_Parsing.ipynb) β€’ [**Data Normalization**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/04_Data_Normalization.ipynb) β€’ [**Chunking & Splitting**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/11_Chunking_and_Splitting.ipynb) + +### Semantic Intelligence Engine + +> **Entity & Relation Extraction** β€’ NER, Relationships, Events, Triplets with LLM Enhancement + +```python +from semantica.core import Semantica + +text = "Apple Inc., founded by Steve Jobs in 1976, acquired Beats Electronics for $3 billion." + +core = Semantica(ner_model="transformer", relation_strategy="hybrid") +results = core.extract_semantics(text) + +print(f"Entities: {len(results.entities)}, Relationships: {len(results.relationships)}") +``` + +[**Cookbook: Entity Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/05_Entity_Extraction.ipynb) β€’ [**Relation Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/06_Relation_Extraction.ipynb) β€’ [**Advanced Extraction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/01_Advanced_Extraction.ipynb) + +### Knowledge Graph Construction + +> **Production-Ready KGs** β€’ Entity Resolution β€’ Temporal Support β€’ Graph Analytics + +```python +from semantica.core import Semantica +from semantica.kg import GraphAnalyzer + +documents = ["doc1.txt", "doc2.txt", "doc3.txt"] +core = Semantica(graph_db="neo4j", merge_entities=True) +kg = core.build_knowledge_graph(documents, generate_embeddings=True) + +analyzer = GraphAnalyzer() +pagerank = analyzer.compute_centrality(kg, method="pagerank") +communities = analyzer.detect_communities(kg, method="louvain") + +result = kg.query("Who founded the company?", return_format="structured") +print(f"Nodes: {kg.node_count}, Answer: {result.answer}") +``` + +[**Cookbook: Building Knowledge Graphs**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/07_Building_Knowledge_Graphs.ipynb) β€’ [**Graph Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/09_Graph_Store.ipynb) β€’ [**Triplet Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/20_Triplet_Store.ipynb) β€’ [**Visualization**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/16_Visualization.ipynb) + +[**Graph Analytics**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/10_Graph_Analytics.ipynb) β€’ [**Advanced Graph Analytics**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/02_Advanced_Graph_Analytics.ipynb) + +### Triplet Store Integration + +> **SPARQL Support** β€’ **Blazegraph, Jena, RDF4J** β€’ **Reasoning & Inference** + +```python +from semantica.triplet_store import TripletStore + +# Initialize store (Blazegraph, Jena, or RDF4J) +store = TripletStore(backend="blazegraph", endpoint="http://localhost:9999/blazegraph") + +# Add triplets and execute SPARQL queries +store.add_triplet({ + "subject": "http://example.org/Alice", + "predicate": "http://example.org/knows", + "object": "http://example.org/Bob" +}) + +results = store.execute_query("SELECT ?s ?p ?o WHERE { ?s ?p ?o } LIMIT 10") +``` + +[**Cookbook: Triplet Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/20_Triplet_Store.ipynb) + +### Ontology Generation & Management + +> **6-Stage LLM Pipeline** β€’ Automatic OWL Generation β€’ HermiT/Pellet Validation + +```python +from semantica.ontology import OntologyGenerator + +generator = OntologyGenerator(llm_provider="openai", model="gpt-4") +ontology = generator.generate_from_documents(sources=["domain_docs/"]) + +print(f"Classes: {len(ontology.classes)}") +``` + +[**Cookbook: Ontology**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/14_Ontology.ipynb) + +### Context Engineering & Memory Systems + +> **Persistent Memory** β€’ **Hybrid Retrieval (Vector + Graph)** β€’ **Hierarchical Storage** β€’ **Entity Linking** + +```python +from semantica.context import AgentContext +from semantica.vector_store import VectorStore + +# Initialize Context with Hybrid Retrieval (Graph + Vector) +context = AgentContext( + vector_store=VectorStore(backend="faiss"), + hybrid_alpha=0.75 # 75% weight to Knowledge Graph, 25% to Vector +) + +# Store memory with automatic entity linking +context.store( + "User is building a RAG system with Semantica", + metadata={"priority": "high", "topic": "rag"} +) + +# Retrieve with context expansion +results = context.retrieve("What is the user building?", use_graph_expansion=True) +``` + +**Core Notebooks:** +- [**Context Module Introduction**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/19_Context_Module.ipynb) - Basic memory and storage. +- [**Advanced Context Engineering**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/11_Advanced_Context_Engineering.ipynb) - Hybrid retrieval, graph builders, and custom memory policies. + +**Related Components:** +[**Vector Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/13_Vector_Store.ipynb) β€’ [**Embedding Generation**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/12_Embedding_Generation.ipynb) β€’ [**Advanced Vector Store**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/Advanced_Vector_Store_and_Search.ipynb) + +### Knowledge Graph-Powered RAG (GraphRAG) + +> **30% Accuracy Improvement** β€’ Vector + Graph Hybrid Search β€’ 91% Accuracy + +```python +from semantica.qa_rag import GraphRAGEngine +from semantica.vector_store import VectorStore + +graphrag = GraphRAGEngine( + vector_store=VectorStore(backend="faiss"), + knowledge_graph=kg +) +result = graphrag.query("Who founded the company?", top_k=5, expand_graph=True) +print(f"Answer: {result.answer} (Confidence: {result.confidence:.2f})") +``` + +[**Cookbook: GraphRAG**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases/advanced_rag/01_GraphRAG_Complete.ipynb) + +### Pipeline Orchestration & Parallel Processing + +> **Orchestrator-Worker Pattern** β€’ Parallel Execution β€’ Scalable Processing + +```python +from semantica.pipeline import PipelineBuilder, ExecutionEngine + +pipeline = PipelineBuilder() \ + .add_step("ingest", "custom", func=ingest_data) \ + .add_step("extract", "custom", func=extract_entities) \ + .add_step("build", "custom", func=build_graph) \ + .build() + +result = ExecutionEngine().execute_pipeline(pipeline, parallel=True) +``` + + +### Production-Ready Quality Assurance + +> **Enterprise-Grade QA** β€’ Conflict Detection β€’ Deduplication + +```python +from semantica.deduplication import DuplicateDetector +from semantica.conflicts import ConflictDetector + +entities = kg.get("entities", []) +conflicts = ConflictDetector().detect_conflicts(entities) +duplicates = DuplicateDetector(similarity_threshold=0.85).detect_duplicates(entities) + +print(f"Conflicts: {len(conflicts)} | Duplicates: {len(duplicates)}") +``` + +[**Cookbook: Conflict Detection & Resolution**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/17_Conflict_Detection_and_Resolution.ipynb) β€’ [**Deduplication**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/18_Deduplication.ipynb) + +### Export & Integration + +> **Multi-Format Export** β€’ JSON, CSV, RDF, GraphML + +```python +from semantica.export import GraphExporter + +exporter = GraphExporter(kg) +exporter.export("graph.json", format="json") +exporter.export("graph.ttl", format="turtle") +``` + +[**Cookbook: Export**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/15_Export.ipynb) β€’ [**Multi-Format Export**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/05_Multi_Format_Export.ipynb) β€’ [**Multi-Source Integration**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb) + +## πŸš€ Quick Start + +> **For comprehensive examples, see the [**Cookbook**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook) with 50+ interactive notebooks!** + +```python +from semantica.core import Semantica + +# Initialize and build knowledge graph +core = Semantica(ner_model="transformer", relation_strategy="hybrid") +documents = ["doc1.txt", "doc2.txt", "doc3.txt"] +kg = core.build_knowledge_graph(documents, merge_entities=True) + +# Query the graph +result = kg.query("Who founded the company?", return_format="structured") +print(f"Answer: {result.answer} | Nodes: {kg.node_count}, Edges: {kg.edge_count}") +``` + +[**Cookbook: Your First Knowledge Graph**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/introduction/08_Your_First_Knowledge_Graph.ipynb) + +## 🎯 Use Cases + +**Enterprise Knowledge Engineering** β€” Unify data sources into knowledge graphs, breaking down silos. + +**AI Agents & Autonomous Systems** β€” Build agents with persistent memory and semantic understanding. + +**Multi-Format Document Processing** β€” Process 50+ formats through a unified pipeline. + +**Data Pipeline Processing** β€” Build scalable pipelines with parallel execution. + +**Intelligence & Security** β€” Analyze networks, threat intelligence, forensic analysis. + +**Finance & Trading** β€” Fraud detection, market intelligence, risk assessment. + +**Healthcare & Biomedical** β€” Clinical reports, drug discovery, medical literature analysis. + +[**Explore Use Case Examples**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/use_cases) β€” See real-world implementations in finance, healthcare, cybersecurity, trading, and more. + +## πŸ”¬ Advanced Features + +**Incremental Updates** β€” Real-time stream processing with Kafka, RabbitMQ, Kinesis for live updates. + +**Multi-Language Support** β€” Process 50+ languages with automatic detection. + +**Custom Ontology Import** β€” Import and extend Schema.org and custom ontologies. + +**Advanced Reasoning** β€” Deductive, inductive, abductive reasoning with HermiT/Pellet. + +**Graph Analytics** β€” Centrality, community detection, path finding, temporal analysis. + +**Custom Pipelines** β€” Build custom pipelines with parallel execution. + +**API Integration** β€” Integrate external APIs for entity enrichment. + +[**See Advanced Examples**](https://github.com/Hawksight-AI/semantica/tree/main/cookbook/advanced) β€” Advanced extraction, graph analytics, reasoning, and more. + +## πŸ—ΊοΈ Roadmap + +### Q1 2026 +- [x] Core framework (v1.0) +- [x] GraphRAG engine +- [x] 6-stage ontology pipeline +- [ ] Quality assurance features and Quality Assurance module +- [ ] Enhanced multi-language support +- [ ] Real-time streaming improvements +- [ ] Advanced reasoning v2 + +### Q2 2026 +- [ ] Multi-modal processing + +--- + +## 🀝 Community & Support + +### Join Our Community + +| **Channel** | **Purpose** | +|:-----------:|:-----------| +| [**Discord**](https://discord.gg/semantica) | Real-time help, showcases | +| [**GitHub Discussions**](https://github.com/Hawksight-AI/semantica/discussions) | Q&A, feature requests | + +### Learning Resources + + +### Enterprise Support + +| **Tier** | **Features** | **SLA** | **Price** | +|:--------:|:-----------|:-------:|:--------:| +| **Community** | Public support | Best effort | Free | +| **Professional** | Email support | 48h | Contact | +| **Enterprise** | 24/7 support | 4h | Contact | +| **Premium** | Phone, custom dev | 1h | Contact | + +**Contact:** [GitHub Issues](https://github.com/Hawksight-AI/semantica/issues) with "[Enterprise]" prefix + +## 🀝 Contributing + +### How to Contribute + +```bash +# Fork and clone +git clone https://github.com/your-username/semantica.git +cd semantica + +# Create branch +git checkout -b feature/your-feature + +# Install dev dependencies +pip install -e ".[dev,test]" + +# Make changes and test +pytest tests/ +black semantica/ +flake8 semantica/ + +# Commit and push +git commit -m "Add feature" +git push origin feature/your-feature +``` + +### Contribution Types + +1. **Code** - New features, bug fixes +2. **Documentation** - Improvements, tutorials +3. **Bug Reports** - [Create issue](https://github.com/Hawksight-AI/semantica/issues/new) +4. **Feature Requests** - [Request feature](https://github.com/Hawksight-AI/semantica/issues/new) + +### Recognition + +Contributors receive: +- Recognition in [CONTRIBUTORS.md](https://github.com/Hawksight-AI/semantica/blob/main/CONTRIBUTORS.md) +- GitHub badges +- Semantica swag +- Featured showcases + +## πŸ† Contributors + + + Contributors + + +## πŸ“œ License + +Semantica is licensed under the **MIT License** - see the [LICENSE](https://github.com/Hawksight-AI/semantica/blob/main/LICENSE) file for details. + +
+ +**Built by the Semantica Community** + +[GitHub](https://github.com/Hawksight-AI/semantica) β€’ [Discord](https://discord.gg/semantica) + +
diff --git a/cookbook/advanced/07_Pipeline_Orchestration.ipynb b/cookbook/advanced/07_Pipeline_Orchestration.ipynb deleted file mode 100644 index 225f4d04..00000000 --- a/cookbook/advanced/07_Pipeline_Orchestration.ipynb +++ /dev/null @@ -1,242 +0,0 @@ -{ - "cells": [ - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "[![Open In Colab](https://colab.research.google.com/assets/colab-badge.svg)](https://colab.research.google.com/github/Hawksight-AI/semantica/blob/main/cookbook/advanced/07_Pipeline_Orchestration.ipynb)\n", - "\n", - "# Pipeline Orchestration\n", - "\n", - "## Overview\n", - "\n", - "Build complex pipelines, execute them, handle failures, enable parallel processing, and monitor execution.\n", - "\n", - "\n", - "**Documentation**: [API Reference](https://semantica.readthedocs.io/reference/pipeline/)\n", - "\n", - "## Installation\n", - "\n", - "Install Semantica from PyPI:\n", - "\n", - "```bash\n", - "pip install semantica\n", - "# Or with all optional dependencies:\n", - "pip install semantica[all]\n", - "```\n", - "\n", - "## Workflow: Build Pipelines \u2192 Execute \u2192 Handle Failures \u2192 Parallel Processing \u2192 Monitor\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "!pip install semantica\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "from semantica.pipeline import (\n", - " PipelineBuilder,\n", - " ExecutionEngine,\n", - " FailureHandler,\n", - " ParallelismManager,\n", - " RetryPolicy,\n", - " RetryStrategy\n", - ")\n", - "from semantica.ingest import FileIngestor\n", - "from semantica.parse import DocumentParser\n", - "from semantica.semantic_extract import NERExtractor\n", - "from semantica.kg import GraphBuilder\n", - "import time\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Step 1: Build Complex Pipelines\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "builder = PipelineBuilder()\n", - "\n", - "file_ingestor = FileIngestor()\n", - "document_parser = DocumentParser()\n", - "ner_extractor = NERExtractor()\n", - "graph_builder = GraphBuilder()\n", - "\n", - "# Define handlers for each pipeline step\n", - "def ingest_handler(data, **config):\n", - " files = data.get(\"files\", [])\n", - " if files:\n", - " # Ingest first file as example\n", - " file_obj = file_ingestor.ingest_file(files[0], read_content=True)\n", - " return {**data, \"file\": file_obj}\n", - " return data\n", - "\n", - "def parse_handler(data, **config):\n", - " # If a file was ingested, try parsing; otherwise pass text through\n", - " file_obj = data.get(\"file\")\n", - " if file_obj and getattr(file_obj, \"path\", None):\n", - " parsed = document_parser.parse_document(file_obj.path)\n", - " text = parsed.get(\"text\") if isinstance(parsed, dict) else None\n", - " return {**data, \"text\": text or data.get(\"text\")}\n", - " return data\n", - "\n", - "def extract_handler(data, **config):\n", - " text = data.get(\"text\", \"\")\n", - " entities = ner_extractor.extract_entities(text)\n", - " # Normalize to dict list for graph builder\n", - " entity_dicts = [\n", - " {\"id\": f\"e{i}\", \"name\": e.text, \"type\": e.label} for i, e in enumerate(entities)\n", - " ]\n", - " return {**data, \"entities\": entity_dicts}\n", - "\n", - "def build_graph_handler(data, **config):\n", - " entities = data.get(\"entities\", [])\n", - " graph = graph_builder.build({\"entities\": entities})\n", - " return {**data, \"graph\": graph}\n", - "\n", - "# Build pipeline with proper handlers and dependencies\n", - "pipeline = (\n", - " builder\n", - " .add_step(\"ingest\", \"ingest\", handler=ingest_handler)\n", - " .add_step(\"parse\", \"parse\", dependencies=[\"ingest\"], handler=parse_handler)\n", - " .add_step(\"extract\", \"extract\", dependencies=[\"parse\"], handler=extract_handler)\n", - " .add_step(\"build_graph\", \"build_graph\", dependencies=[\"extract\"], handler=build_graph_handler)\n", - ").build()\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Step 2: Execute Pipeline\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "engine = ExecutionEngine()\n", - "\n", - "input_data = {\n", - " \"text\": \"Alice works at Tech Corp. Bob is a friend of Alice.\",\n", - " \"files\": []\n", - "}\n", - "\n", - "start_time = time.time()\n", - "result = engine.execute_pipeline(pipeline, input_data)\n", - "execution_time = result.metrics.get(\"execution_time\", time.time() - start_time)\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Step 3: Handle Failures\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Configure retry policy for the 'extract' step type\n", - "engine.failure_handler.set_retry_policy(\n", - " \"extract\",\n", - " RetryPolicy(max_retries=3, backoff_factor=2.0, strategy=RetryStrategy.EXPONENTIAL)\n", - ")\n", - "\n", - "result = engine.execute_pipeline(pipeline, input_data)\n", - "print(\"Pipeline executed with retry policy configured\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Step 4: Parallel Processing\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "parallelism = ParallelismManager(max_workers=4)\n", - "\n", - "# Identify groups of steps that can run in parallel\n", - "groups = parallelism.identify_parallelizable_steps(pipeline)\n", - "\n", - "# Execute first parallelizable group as a demonstration\n", - "start_time = time.time()\n", - "parallel_results = []\n", - "for group in groups:\n", - " parallel_results.extend(parallelism.execute_pipeline_steps_parallel(group, input_data, max_workers=4))\n", - "parallel_time = time.time() - start_time\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Step 5: Monitor Pipeline Execution\n" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# Metrics from execution engine\n", - "metrics = result.metrics\n", - "progress = engine.get_progress(pipeline.name)\n", - "\n", - "print(f\"Duration: {metrics.get('execution_time', 0):.2f} seconds\")\n", - "print(f\"Steps Executed: {metrics.get('steps_executed', 0)}\")\n", - "print(f\"Steps Failed: {metrics.get('steps_failed', 0)}\")\n", - "print(f\"Progress: {progress.get('progress_percentage', 0):.1f}% (status: {progress.get('status')})\")\n" - ] - }, - { - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Summary\n", - "\n", - "Pipeline orchestration workflow:\n", - "- Complex Pipeline Built\n", - "- Pipeline Executed\n", - "- Failure Handling Configured\n", - "- Parallel Processing Enabled\n", - "- Full Monitoring and Observability\n" - ] - } - ], - "metadata": { - "language_info": { - "name": "python" - } - }, - "nbformat": 4, - "nbformat_minor": 2 -} \ No newline at end of file diff --git a/docs/cookbook.md b/docs/cookbook.md index 99ed22dc..64f4ea21 100644 --- a/docs/cookbook.md +++ b/docs/cookbook.md @@ -242,15 +242,6 @@ Deep dive into advanced features, customization, and complex workflows. [Open Notebook](https://github.com/Hawksight-AI/semantica/blob/main/cookbook/advanced/06_Multi_Source_Data_Integration.ipynb) -- :material-pipe: **Pipeline Orchestration** - --- - Building robust, automated data processing pipelines. - - **Topics**: Workflows, Automation, Error Handling - - **Difficulty**: Advanced - - [Open Notebook](https://github.com/Hawksight-AI/semantica/blob/main/cookbook/advanced/07_Pipeline_Orchestration.ipynb) - :material-brain: **Reasoning and Inference** --- diff --git a/docs/reference/pipeline.md b/docs/reference/pipeline.md index c492fa08..5df06913 100644 --- a/docs/reference/pipeline.md +++ b/docs/reference/pipeline.md @@ -381,6 +381,3 @@ result = engine.execute_pipeline(pipeline, data={"path": "document.pdf"}) - [Split Module](split.md) - Common processing step - [Vector Store Module](vector_store.md) - Common sink step -## Cookbook - -- [Pipeline Orchestration](https://github.com/Hawksight-AI/semantica/blob/main/cookbook/advanced/07_Pipeline_Orchestration.ipynb) diff --git a/tests/pipeline/test_notebook_07.py b/tests/pipeline/test_notebook_07.py deleted file mode 100644 index 73f92afe..00000000 --- a/tests/pipeline/test_notebook_07.py +++ /dev/null @@ -1,154 +0,0 @@ -import unittest -from unittest.mock import MagicMock, patch -import time - -import pytest - -from semantica.pipeline import ( - PipelineBuilder, - ExecutionEngine, - FailureHandler, - ParallelismManager, - RetryPolicy, - RetryStrategy -) - -pytestmark = pytest.mark.integration - -class TestNotebook07(unittest.TestCase): - - def setUp(self): - # Mock external dependencies used in the notebook - self.mock_file_ingestor = MagicMock() - self.mock_document_parser = MagicMock() - self.mock_ner_extractor = MagicMock() - self.mock_graph_builder = MagicMock() - - # Setup return values - self.mock_file_ingestor.ingest_file.return_value = MagicMock(path="test.txt") - self.mock_document_parser.parse_document.return_value = {"text": "Alice works at Tech Corp."} - - # Mock NER entities - mock_entity = MagicMock() - mock_entity.text = "Alice" - mock_entity.label = "PERSON" - self.mock_ner_extractor.extract_entities.return_value = [mock_entity] - - self.mock_graph_builder.build.return_value = {"nodes": [], "edges": []} - - def test_pipeline_orchestration_workflow(self): - """Replicates the workflow in 07_Pipeline_Orchestration.ipynb""" - - builder = PipelineBuilder() - - # Define handlers (logic copied from notebook) - def ingest_handler(data, **config): - files = data.get("files", []) - if files: - # Ingest first file as example - file_obj = self.mock_file_ingestor.ingest_file(files[0], read_content=True) - return {**data, "file": file_obj} - return data - - def parse_handler(data, **config): - # If a file was ingested, try parsing; otherwise pass text through - file_obj = data.get("file") - # Mock object path check - if file_obj and getattr(file_obj, "path", None): - parsed = self.mock_document_parser.parse_document(file_obj.path) - text = parsed.get("text") if isinstance(parsed, dict) else None - return {**data, "text": text or data.get("text")} - return data - - def extract_handler(data, **config): - text = data.get("text", "") - entities = self.mock_ner_extractor.extract_entities(text) - # Normalize to dict list for graph builder - entity_dicts = [ - {"id": f"e{i}", "name": e.text, "type": e.label} for i, e in enumerate(entities) - ] - return {**data, "entities": entity_dicts} - - def build_graph_handler(data, **config): - entities = data.get("entities", []) - graph = self.mock_graph_builder.build({"entities": entities}) - return {**data, "graph": graph} - - # Build pipeline - pipeline = ( - builder - .add_step("ingest", "ingest", handler=ingest_handler) - .add_step("parse", "parse", dependencies=["ingest"], handler=parse_handler) - .add_step("extract", "extract", dependencies=["parse"], handler=extract_handler) - .add_step("build_graph", "build_graph", dependencies=["extract"], handler=build_graph_handler) - ).build() - - # Step 2: Execute Pipeline - engine = ExecutionEngine() - input_data = { - "text": "Alice works at Tech Corp. Bob is a friend of Alice.", - "files": ["sample.txt"] - } - - result = engine.execute_pipeline(pipeline, input_data) - - self.assertTrue(result.success) - self.assertIn("graph", result.output) - - # Verify mocks called - self.mock_file_ingestor.ingest_file.assert_called() - self.mock_document_parser.parse_document.assert_called() - self.mock_ner_extractor.extract_entities.assert_called() - self.mock_graph_builder.build.assert_called() - - # Step 3: Handle Failures - # Configure retry policy - engine.failure_handler.set_retry_policy( - "extract", - RetryPolicy(max_retries=3, backoff_factor=1.0, strategy=RetryStrategy.LINEAR) - ) - - # Execute again (should still pass) - result_retry = engine.execute_pipeline(pipeline, input_data) - self.assertTrue(result_retry.success) - - # Step 4: Parallel Processing - parallelism = ParallelismManager(max_workers=4) - groups = parallelism.identify_parallelizable_steps(pipeline) - - # The pipeline is sequential (ingest->parse->extract->build_graph), so groups should be single steps - # [[ingest], [parse], [extract], [build_graph]] - self.assertEqual(len(groups), 4) - - # Execute parallel steps (simulated) - parallel_results = [] - for group in groups: - # We mock the execution here or just call the manager's method - # Since execute_pipeline_steps_parallel needs Task objects or similar logic, - # and the notebook uses it slightly differently (it seems to assume integration with engine). - # Let's check how the notebook uses it: - # parallel_results.extend(parallelism.execute_pipeline_steps_parallel(group, input_data, max_workers=4)) - - # The ParallelismManager.execute_pipeline_steps_parallel likely takes PipelineStep objects and data - # We need to ensure input_data flows correctly. In a real pipeline, output of one step is input to next. - # The notebook example simplifies this by passing `input_data` to all, which works if steps are independent or data is static. - # But here steps depend on previous output. - # So we'll just verify the method runs without error. - try: - parallelism.execute_pipeline_steps_parallel(group, input_data, max_workers=2) - except Exception as e: - # It might fail if handlers expect data from previous steps which is not in 'input_data' - # For this test, we accept that or catch it. - # Actually, let's just verify `identify_parallelizable_steps` works as expected. - pass - - # Step 5: Monitor - metrics = result.metrics - progress = engine.get_progress(pipeline.name) - - self.assertIn("execution_time", metrics) - self.assertEqual(metrics.get("steps_failed", 0), 0) - # Progress might be cleared or 100% depending on implementation - -if __name__ == '__main__': - unittest.main() diff --git a/tests/test_pipeline_orchestration.py b/tests/test_pipeline_orchestration.py index bfce1631..85de22cc 100644 --- a/tests/test_pipeline_orchestration.py +++ b/tests/test_pipeline_orchestration.py @@ -260,10 +260,10 @@ def test_parallelism_manager_identify_parallelizable_steps(parallelism_manager, assert "s2" in names assert "s3" in names -# --- End-to-End Notebook Simulation --- +# --- End-to-End Pipeline Orchestration Test --- def test_end_to_end_pipeline_orchestration(pipeline_builder, execution_engine): - # This simulates the logic in 07_Pipeline_Orchestration.ipynb + # This simulates a complete pipeline orchestration workflow # Mocks for actual components to avoid file I/O and heavy processing file_ingestor_mock = MagicMock()