mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-09-15 04:00:33 +00:00
refactor(semantic_extract): remove build function and enhance documentation
BREAKING CHANGE: Removed build() convenience function from semantic_extract module - Removed build() function from semantic_extract/__init__.py - Updated __all__ exports to remove 'build' - Resolved merge conflicts in named_entity_recognizer.py, relation_extractor.py, triple_extractor.py - Updated semantic_extract_usage.md with class-based examples - Updated docs/reference/semantic_extract.md with detailed parameter documentation - Fixed 01_GraphRAG_Complete.ipynb to use individual extractor classes - Enhanced 05_Entity_Extraction.ipynb with comprehensive examples (9 sections) - Enhanced 06_Relation_Extraction.ipynb with complete pipeline examples (9 sections) Users should now use individual classes (NERExtractor, RelationExtractor, TripleExtractor, etc.) instead of the build() function for better control and flexibility. Migration guide available in documentation.
This commit is contained in:
@@ -83,7 +83,6 @@
|
||||
### NamedEntityRecognizer
|
||||
|
||||
Coordinator for entity extraction.
|
||||
<<<<<<< HEAD
|
||||
|
||||
**Parameters:**
|
||||
|
||||
@@ -93,8 +92,6 @@ Coordinator for entity extraction.
|
||||
| `confidence_threshold` | float | `0.5` | Minimum confidence score |
|
||||
| `merge_overlapping` | bool | `True` | Merge overlapping entities |
|
||||
| `include_standard_types` | bool | `True` | Include Person, Org, Location |
|
||||
=======
|
||||
>>>>>>> origin/main
|
||||
|
||||
**Methods:**
|
||||
|
||||
@@ -108,7 +105,6 @@ Coordinator for entity extraction.
|
||||
```python
|
||||
from semantica.semantic_extract import NamedEntityRecognizer
|
||||
|
||||
<<<<<<< HEAD
|
||||
# Basic usage
|
||||
ner = NamedEntityRecognizer()
|
||||
entities = ner.extract_entities("Elon Musk leads SpaceX.")
|
||||
@@ -121,17 +117,11 @@ ner = NamedEntityRecognizer(
|
||||
merge_overlapping=True
|
||||
)
|
||||
entities = ner.extract_entities("Apple Inc. was founded in 1976.")
|
||||
=======
|
||||
ner = NamedEntityRecognizer()
|
||||
entities = ner.extract_entities("Elon Musk leads SpaceX.")
|
||||
# [Entity(text="Elon Musk", label="PERSON"), Entity(text="SpaceX", label="ORG")]
|
||||
>>>>>>> origin/main
|
||||
```
|
||||
|
||||
### RelationExtractor
|
||||
|
||||
Extracts relationships between entities.
|
||||
<<<<<<< HEAD
|
||||
|
||||
**Parameters:**
|
||||
|
||||
@@ -141,8 +131,6 @@ Extracts relationships between entities.
|
||||
| `bidirectional` | bool | `False` | Extract bidirectional relations |
|
||||
| `confidence_threshold` | float | `0.6` | Minimum confidence score |
|
||||
| `max_distance` | int | `50` | Max token distance between entities |
|
||||
=======
|
||||
>>>>>>> origin/main
|
||||
|
||||
**Methods:**
|
||||
|
||||
@@ -155,7 +143,6 @@ Extracts relationships between entities.
|
||||
```python
|
||||
from semantica.semantic_extract import RelationExtractor, NamedEntityRecognizer
|
||||
|
||||
<<<<<<< HEAD
|
||||
# First extract entities
|
||||
ner = NamedEntityRecognizer()
|
||||
text = "Elon Musk founded SpaceX in 2002."
|
||||
@@ -173,16 +160,10 @@ rel_extractor = RelationExtractor(
|
||||
bidirectional=False
|
||||
)
|
||||
relations = rel_extractor.extract_relations(text, entities=entities)
|
||||
=======
|
||||
re = RelationExtractor()
|
||||
relations = re.extract_relations(text, entities)
|
||||
# [Relation(source="Elon Musk", target="SpaceX", type="leads")]
|
||||
>>>>>>> origin/main
|
||||
```
|
||||
|
||||
### EventDetector
|
||||
|
||||
<<<<<<< HEAD
|
||||
Identifies events with temporal information and participants.
|
||||
|
||||
**Parameters:**
|
||||
@@ -193,9 +174,6 @@ Identifies events with temporal information and participants.
|
||||
| `extract_participants` | bool | `True` | Extract event participants |
|
||||
| `extract_location` | bool | `True` | Extract event locations |
|
||||
| `extract_time` | bool | `True` | Extract temporal information |
|
||||
=======
|
||||
Identifies events.
|
||||
>>>>>>> origin/main
|
||||
|
||||
**Methods:**
|
||||
|
||||
@@ -203,7 +181,6 @@ Identifies events.
|
||||
|--------|-------------|
|
||||
| `detect_events(text)` | Find events |
|
||||
|
||||
<<<<<<< HEAD
|
||||
**Example:**
|
||||
|
||||
```python
|
||||
@@ -227,18 +204,12 @@ Extracts RDF triples (Subject-Predicate-Object).
|
||||
|-----------|------|---------|-------------|
|
||||
| `include_temporal` | bool | `False` | Include time information |
|
||||
| `include_provenance` | bool | `False` | Track source sentences |
|
||||
=======
|
||||
### TripleExtractor
|
||||
|
||||
Extracts RDF triples.
|
||||
>>>>>>> origin/main
|
||||
|
||||
**Methods:**
|
||||
|
||||
| Method | Description |
|
||||
|--------|-------------|
|
||||
| `extract_triples(text)` | Get (S, P, O) tuples |
|
||||
<<<<<<< HEAD
|
||||
|
||||
**Example:**
|
||||
|
||||
@@ -252,15 +223,12 @@ extractor = TripleExtractor(
|
||||
triples = extractor.extract_triples("Steve Jobs founded Apple in 1976.")
|
||||
# [Triple(subject="Steve Jobs", predicate="founded", object="Apple", temporal="1976")]
|
||||
```
|
||||
=======
|
||||
>>>>>>> origin/main
|
||||
|
||||
---
|
||||
|
||||
## Convenience Functions
|
||||
## Usage Examples
|
||||
|
||||
```python
|
||||
<<<<<<< HEAD
|
||||
from semantica.semantic_extract import (
|
||||
NamedEntityRecognizer,
|
||||
RelationExtractor,
|
||||
@@ -295,19 +263,6 @@ print(f"Entities: {len(entities)}")
|
||||
print(f"Relations: {len(relations)}")
|
||||
print(f"Triples: {len(triples)}")
|
||||
print(f"Events: {len(events)}")
|
||||
=======
|
||||
from semantica.semantic_extract import build
|
||||
|
||||
# All-in-one extraction
|
||||
result = build(
|
||||
"Apple released the iPhone in 2007.",
|
||||
extract_entities=True,
|
||||
extract_relations=True,
|
||||
extract_events=True
|
||||
)
|
||||
|
||||
print(result['triples'])
|
||||
>>>>>>> origin/main
|
||||
```
|
||||
|
||||
---
|
||||
@@ -344,7 +299,6 @@ semantic_extract:
|
||||
### KG Population Pipeline
|
||||
|
||||
```python
|
||||
<<<<<<< HEAD
|
||||
from semantica.semantic_extract import NamedEntityRecognizer, RelationExtractor, TripleExtractor
|
||||
from semantica.kg import GraphBuilder
|
||||
|
||||
@@ -362,23 +316,6 @@ sources = [{
|
||||
"relationships": [{"source": t.subject, "target": t.object, "type": t.predicate} for t in triples]
|
||||
}]
|
||||
kg = builder.build(sources)
|
||||
=======
|
||||
from semantica.semantic_extract import build
|
||||
from semantica.kg import KnowledgeGraph
|
||||
|
||||
# 1. Extract
|
||||
text = "Google was founded by Larry Page and Sergey Brin."
|
||||
data = build(text, extract_triples=True)
|
||||
|
||||
# 2. Populate KG
|
||||
kg = KnowledgeGraph()
|
||||
for triple in data['triples']:
|
||||
kg.add_triple(
|
||||
subject=triple.subject,
|
||||
predicate=triple.predicate,
|
||||
object=triple.object
|
||||
)
|
||||
>>>>>>> origin/main
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
Reference in New Issue
Block a user