From 2f04bc01a32552a2310363cc03edc66651572ba8 Mon Sep 17 00:00:00 2001 From: Accute9 Date: Sat, 15 Aug 2026 20:37:56 -0400 Subject: [PATCH 1/7] route spaCy model loads through process cache --- semantica/split/methods.py | 3 ++- semantica/split/semantic_chunker.py | 5 ++++- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/semantica/split/methods.py b/semantica/split/methods.py index 8c338dc6..f2b3f525 100644 --- a/semantica/split/methods.py +++ b/semantica/split/methods.py @@ -93,6 +93,7 @@ from ..utils.exceptions import ProcessingError from ..utils.helpers import safe_import from ..utils.logging import get_logger from .semantic_chunker import Chunk +from ..semantic_extract.methods import load_spacy_model logger = get_logger("split_methods") @@ -336,7 +337,7 @@ def split_by_sentences( # Try spaCy first if SPACY_AVAILABLE and kwargs.get("use_spacy", True): try: - nlp = spacy.load("en_core_web_sm") + nlp = load_spacy_model("en_core_web_sm") doc = nlp(text) sentences = [sent.text for sent in doc.sents] except Exception: diff --git a/semantica/split/semantic_chunker.py b/semantica/split/semantic_chunker.py index 079ba976..5d712627 100644 --- a/semantica/split/semantic_chunker.py +++ b/semantica/split/semantic_chunker.py @@ -35,6 +35,8 @@ from ..utils.exceptions import ProcessingError from ..utils.helpers import safe_import from ..utils.logging import get_logger from ..utils.progress_tracker import get_progress_tracker +from ..semantic_extract.methods import load_spacy_model + spacy, SPACY_AVAILABLE = safe_import("spacy") @@ -79,7 +81,8 @@ class SemanticChunker: if SPACY_AVAILABLE: model_name = config.get("model", "en_core_web_sm") try: - self.nlp = spacy.load(model_name) + # self.nlp = spacy.load(model_name) + self.nlp = load_spacy_model(model_name) except OSError: self.logger.warning( f"spaCy model {model_name} not found. Using fallback chunking." From 83649f68219ef069f1a12f58424a99abb4f2639d Mon Sep 17 00:00:00 2001 From: Accute9 Date: Sat, 15 Aug 2026 20:56:39 -0400 Subject: [PATCH 2/7] forgot to remove comment --- semantica/split/semantic_chunker.py | 1 - 1 file changed, 1 deletion(-) diff --git a/semantica/split/semantic_chunker.py b/semantica/split/semantic_chunker.py index 5d712627..d7f72726 100644 --- a/semantica/split/semantic_chunker.py +++ b/semantica/split/semantic_chunker.py @@ -81,7 +81,6 @@ class SemanticChunker: if SPACY_AVAILABLE: model_name = config.get("model", "en_core_web_sm") try: - # self.nlp = spacy.load(model_name) self.nlp = load_spacy_model(model_name) except OSError: self.logger.warning( From 893b6db3c3d2abf2c0656baeb3549fb24a840f13 Mon Sep 17 00:00:00 2001 From: Accute9 Date: Sun, 16 Aug 2026 16:07:45 -0400 Subject: [PATCH 3/7] regression tests added and tested for routing spaCy model loads through cache --- tests/split/test_spacy_model_cache.py | 169 ++++++++++++++++++++++++++ 1 file changed, 169 insertions(+) create mode 100644 tests/split/test_spacy_model_cache.py diff --git a/tests/split/test_spacy_model_cache.py b/tests/split/test_spacy_model_cache.py new file mode 100644 index 00000000..d3b27148 --- /dev/null +++ b/tests/split/test_spacy_model_cache.py @@ -0,0 +1,169 @@ +from types import SimpleNamespace +from unittest.mock import MagicMock + +import pytest + +from semantica.semantic_extract import methods as se_methods +from semantica.split import methods as split_methods +from semantica.split import semantic_chunker + + +@pytest.fixture(autouse=True) +def clear_cache(): + se_methods.clear_spacy_model_cache() + yield + se_methods.clear_spacy_model_cache() + + +@pytest.fixture(autouse=True) +def force_spacy_available(monkeypatch): + # split.methods and split.semantic_chunker each compute their own + # SPACY_AVAILABLE flag from the real environment at import time; force + # both true so these tests exercise the spaCy branch regardless of + # whether spaCy is actually installed where they run. + monkeypatch.setattr(split_methods, "SPACY_AVAILABLE", True) + monkeypatch.setattr(semantic_chunker, "SPACY_AVAILABLE", True) + + +def _fake_spacy(load): + return SimpleNamespace(load=load, util=SimpleNamespace(is_package=lambda _name: True)) + + +def _nlp_mock(sentences=("Hello world.",)): + """A stand-in spaCy Language object: callable, returns a doc with .sents.""" + nlp = MagicMock() + nlp.return_value = SimpleNamespace( + sents=[SimpleNamespace(text=s) for s in sentences] + ) + return nlp + + +class TestSpacyModelCache: + """split.methods and split.semantic_chunker must share the cached model + defined in semantic_extract.methods instead of each calling spacy.load() + independently. + """ + + def test_split_by_sentences_reuses_cached_model(self, monkeypatch): + calls = [] + + def fake_load(name, **kwargs): + calls.append((name, kwargs)) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + split_methods.split_by_sentences("Hello world. Bye world.") + split_methods.split_by_sentences("Another sentence here.") + split_methods.split_by_sentences("A third call.") + + assert len(calls) == 1, "spacy.load should run once, not once per call" + assert calls[0][0] == "en_core_web_sm" + + def test_semantic_chunker_reuses_cached_model_across_instances(self, monkeypatch): + calls = [] + + def fake_load(name, **kwargs): + calls.append((name, kwargs)) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + chunker1 = semantic_chunker.SemanticChunker() + chunker2 = semantic_chunker.SemanticChunker() + + assert len(calls) == 1, "each new SemanticChunker should not reload the model" + assert chunker1.nlp is chunker2.nlp + + def test_split_methods_and_semantic_chunker_share_the_cache(self, monkeypatch): + calls = [] + + def fake_load(name, **kwargs): + calls.append((name, kwargs)) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + split_methods.split_by_sentences("Test sentence for split.methods.") + semantic_chunker.SemanticChunker() + + assert len(calls) == 1, ( + "split.methods and split.semantic_chunker must share one cached " + "model instead of each loading their own" + ) + + def test_distinct_model_names_load_separately(self, monkeypatch): + calls = [] + + def fake_load(name, **kwargs): + calls.append((name, kwargs)) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + sm_chunker = semantic_chunker.SemanticChunker(model="en_core_web_sm") + lg_chunker = semantic_chunker.SemanticChunker(model="en_core_web_lg") + sm_chunker_again = semantic_chunker.SemanticChunker(model="en_core_web_sm") + + assert [name for name, _ in calls] == ["en_core_web_sm", "en_core_web_lg"] + assert sm_chunker.nlp is sm_chunker_again.nlp + assert sm_chunker.nlp is not lg_chunker.nlp + + def test_no_disable_kwarg_requested(self, monkeypatch): + """split.methods and split.semantic_chunker both want the full + pipeline (they need .sents, which requires the parser/senter). If + either one later starts requesting a trimmed pipeline (e.g. + disable=["ner"]), the name-only cache key in load_spacy_model would + silently hand back a cached model built for a different config -- + this test should catch that the moment it happens. + """ + calls = [] + + def fake_load(_name, **kwargs): + calls.append(kwargs) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + split_methods.split_by_sentences("Hello world.") + se_methods.clear_spacy_model_cache() + semantic_chunker.SemanticChunker() + + assert calls == [{}, {}], "neither caller should request a partial pipeline" + + def test_missing_model_falls_back_without_poisoning_cache(self, monkeypatch): + attempts = [] + + def failing_load(name, **_kwargs): + attempts.append(name) + raise OSError(f"Can't find model '{name}'") + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(failing_load)) + + # split_by_sentences should fall back to regex splitting, not raise + chunks = split_methods.split_by_sentences("Hello world. Bye world.") + assert chunks, "fallback splitting should still produce chunks" + + # SemanticChunker should leave .nlp as None rather than propagate + chunker = semantic_chunker.SemanticChunker() + assert chunker.nlp is None + + assert len(attempts) == 2, "a failed load must not be cached" + + # Once the model is available, both callers should now get it, and + # share a single successful load. + def working_load(name, **_kwargs): + attempts.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(working_load)) + + chunker2 = semantic_chunker.SemanticChunker() + split_methods.split_by_sentences("One more sentence.") + + assert len(attempts) == 3, "the model should load once after it becomes available" + assert chunker2.nlp is not None + + +if __name__ == "__main__": + pytest.main([__file__]) From 0b77e5fe9476fab239f5ec2705f939456bb1d2cb Mon Sep 17 00:00:00 2001 From: Aneesh Mandapati <93799543+Accute9@users.noreply.github.com> Date: Sun, 16 Aug 2026 20:31:06 -0400 Subject: [PATCH 4/7] Refactor for flake8 max line length (88) issue Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- tests/split/test_spacy_model_cache.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/split/test_spacy_model_cache.py b/tests/split/test_spacy_model_cache.py index d3b27148..551b82f9 100644 --- a/tests/split/test_spacy_model_cache.py +++ b/tests/split/test_spacy_model_cache.py @@ -26,7 +26,10 @@ def force_spacy_available(monkeypatch): def _fake_spacy(load): - return SimpleNamespace(load=load, util=SimpleNamespace(is_package=lambda _name: True)) + return SimpleNamespace( + load=load, + util=SimpleNamespace(is_package=lambda _name: True), + ) def _nlp_mock(sentences=("Hello world.",)): From 0f252ab355b60df015937cb75b51484c2f90bc34 Mon Sep 17 00:00:00 2001 From: Accute9 Date: Sun, 16 Aug 2026 21:04:57 -0400 Subject: [PATCH 5/7] Fixed max line length (88) issues and eager imports --- semantica/split/methods.py | 4 ++-- semantica/split/semantic_chunker.py | 4 ++-- tests/split/test_spacy_model_cache.py | 16 +++++++++++++--- 3 files changed, 17 insertions(+), 7 deletions(-) diff --git a/semantica/split/methods.py b/semantica/split/methods.py index f2b3f525..61b67ee0 100644 --- a/semantica/split/methods.py +++ b/semantica/split/methods.py @@ -93,12 +93,11 @@ from ..utils.exceptions import ProcessingError from ..utils.helpers import safe_import from ..utils.logging import get_logger from .semantic_chunker import Chunk -from ..semantic_extract.methods import load_spacy_model logger = get_logger("split_methods") # Try to import optional dependencies -spacy, SPACY_AVAILABLE = safe_import("spacy") +_, SPACY_AVAILABLE = safe_import("spacy") nltk, NLTK_AVAILABLE = safe_import("nltk") tiktoken, TIKTOKEN_AVAILABLE = safe_import("tiktoken") @@ -337,6 +336,7 @@ def split_by_sentences( # Try spaCy first if SPACY_AVAILABLE and kwargs.get("use_spacy", True): try: + from ..semantic_extract.methods import load_spacy_model nlp = load_spacy_model("en_core_web_sm") doc = nlp(text) sentences = [sent.text for sent in doc.sents] diff --git a/semantica/split/semantic_chunker.py b/semantica/split/semantic_chunker.py index d7f72726..2945bbd5 100644 --- a/semantica/split/semantic_chunker.py +++ b/semantica/split/semantic_chunker.py @@ -35,10 +35,9 @@ from ..utils.exceptions import ProcessingError from ..utils.helpers import safe_import from ..utils.logging import get_logger from ..utils.progress_tracker import get_progress_tracker -from ..semantic_extract.methods import load_spacy_model -spacy, SPACY_AVAILABLE = safe_import("spacy") +_, SPACY_AVAILABLE = safe_import("spacy") @dataclass @@ -81,6 +80,7 @@ class SemanticChunker: if SPACY_AVAILABLE: model_name = config.get("model", "en_core_web_sm") try: + from ..semantic_extract.methods import load_spacy_model self.nlp = load_spacy_model(model_name) except OSError: self.logger.warning( diff --git a/tests/split/test_spacy_model_cache.py b/tests/split/test_spacy_model_cache.py index d3b27148..97a4ad87 100644 --- a/tests/split/test_spacy_model_cache.py +++ b/tests/split/test_spacy_model_cache.py @@ -26,7 +26,12 @@ def force_spacy_available(monkeypatch): def _fake_spacy(load): - return SimpleNamespace(load=load, util=SimpleNamespace(is_package=lambda _name: True)) + return SimpleNamespace( + load=load, + util=SimpleNamespace( + is_package=lambda _name: True + ), + ) def _nlp_mock(sentences=("Hello world.",)): @@ -129,7 +134,10 @@ class TestSpacyModelCache: se_methods.clear_spacy_model_cache() semantic_chunker.SemanticChunker() - assert calls == [{}, {}], "neither caller should request a partial pipeline" + assert len(calls) == 2 + assert all("disable" not in kwargs for kwargs in calls), ( + "neither caller should request a partial pipeline" + ) def test_missing_model_falls_back_without_poisoning_cache(self, monkeypatch): attempts = [] @@ -161,7 +169,9 @@ class TestSpacyModelCache: chunker2 = semantic_chunker.SemanticChunker() split_methods.split_by_sentences("One more sentence.") - assert len(attempts) == 3, "the model should load once after it becomes available" + assert len(attempts) == 3, ( + "the model should load once after it becomes available" + ) assert chunker2.nlp is not None From c7415f2e92434c65246d564184292064f9c42224 Mon Sep 17 00:00:00 2001 From: Sameer6305 Date: Mon, 17 Aug 2026 12:40:10 +0530 Subject: [PATCH 6/7] fix: complete spaCy model cache integration --- semantica/semantic_extract/ner_extractor.py | 7 +- tests/split/test_spacy_model_cache.py | 138 +++++++++++++++++++- tests/split/test_splitter.py | 16 +-- tests/test_ner_configurations.py | 29 ++-- 4 files changed, 166 insertions(+), 24 deletions(-) diff --git a/semantica/semantic_extract/ner_extractor.py b/semantica/semantic_extract/ner_extractor.py index e8b57bcd..a920efe1 100644 --- a/semantica/semantic_extract/ner_extractor.py +++ b/semantica/semantic_extract/ner_extractor.py @@ -144,7 +144,12 @@ class NERExtractor: self._ml_runtime_usable = True if "ml" in self.method and SPACY_AVAILABLE: try: - self.nlp = spacy.load(self.model_name) + # Deferred import: keeps semantic_extract.methods out of the + # module-level import graph and routes loading through the + # process-level cache so repeated NERExtractor constructions + # never pay the ~120 ms spacy.load() cost more than once. + from .methods import load_spacy_model + self.nlp = load_spacy_model(self.model_name) except OSError: self.logger.warning( f"spaCy model {self.model_name} not found. ML method will fallback." diff --git a/tests/split/test_spacy_model_cache.py b/tests/split/test_spacy_model_cache.py index d8f38117..de21e433 100644 --- a/tests/split/test_spacy_model_cache.py +++ b/tests/split/test_spacy_model_cache.py @@ -6,6 +6,8 @@ import pytest from semantica.semantic_extract import methods as se_methods from semantica.split import methods as split_methods from semantica.split import semantic_chunker +from semantica.semantic_extract import ner_extractor as ner_extractor_module +from semantica.semantic_extract.ner_extractor import NERExtractor @pytest.fixture(autouse=True) @@ -17,12 +19,13 @@ def clear_cache(): @pytest.fixture(autouse=True) def force_spacy_available(monkeypatch): - # split.methods and split.semantic_chunker each compute their own - # SPACY_AVAILABLE flag from the real environment at import time; force - # both true so these tests exercise the spaCy branch regardless of + # split.methods, split.semantic_chunker, and ner_extractor each compute + # their own SPACY_AVAILABLE flag from the real environment at import time; + # force all true so these tests exercise the spaCy branch regardless of # whether spaCy is actually installed where they run. monkeypatch.setattr(split_methods, "SPACY_AVAILABLE", True) monkeypatch.setattr(semantic_chunker, "SPACY_AVAILABLE", True) + monkeypatch.setattr(ner_extractor_module, "SPACY_AVAILABLE", True) def _fake_spacy(load): @@ -133,9 +136,11 @@ class TestSpacyModelCache: semantic_chunker.SemanticChunker() assert len(calls) == 2 - assert all("disable" not in kwargs for kwargs in calls), ( - "neither caller should request a partial pipeline" - ) + assert all(kwargs == {} for kwargs in calls), ( + "neither caller should pass any pipeline-configuration kwargs; " + "the name-only cache key in load_spacy_model cannot distinguish " + "models loaded with different component configs" + ) def test_missing_model_falls_back_without_poisoning_cache(self, monkeypatch): attempts = [] @@ -173,5 +178,126 @@ class TestSpacyModelCache: assert chunker2.nlp is not None +class TestNERExtractorSpacyModelCache: + """NERExtractor(method="ml") must reuse the centralized cache in + semantic_extract.methods, not call spacy.load() on every construction. + + These tests mirror TestSpacyModelCache but focus on the NERExtractor path, + confirming that all three callers (split_by_sentences, SemanticChunker, and + NERExtractor) draw from the same process-level cache. + """ + + def test_ner_extractor_reuses_cached_model_across_instances(self, monkeypatch): + """Two NERExtractor(method='ml') constructions with the same model name + must cause exactly one underlying spacy.load() call.""" + calls = [] + + def fake_load(name, **kwargs): + calls.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + e1 = NERExtractor(method="ml") + e2 = NERExtractor(method="ml") + e3 = NERExtractor(method="ml", model="en_core_web_sm") + + assert len(calls) == 1, ( + "repeated NERExtractor constructions should not reload the model" + ) + assert e1.nlp is e2.nlp is e3.nlp + + def test_ner_extractor_and_split_callers_share_one_cached_model(self, monkeypatch): + """NERExtractor, SemanticChunker, and split_by_sentences must all use + the same cached Language object for the same model name.""" + calls = [] + + def fake_load(name, **kwargs): + calls.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + split_methods.split_by_sentences("First sentence.") + semantic_chunker.SemanticChunker() + NERExtractor(method="ml") + + assert len(calls) == 1, ( + "split_by_sentences, SemanticChunker, and NERExtractor must share " + "one cached model instead of each loading their own" + ) + + def test_ner_extractor_distinct_model_names_load_separately(self, monkeypatch): + """Different model names must produce separate cache entries.""" + calls = [] + + def fake_load(name, **kwargs): + calls.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + sm = NERExtractor(method="ml", model="en_core_web_sm") + lg = NERExtractor(method="ml", model="en_core_web_lg") + sm_again = NERExtractor(method="ml", model="en_core_web_sm") + + assert calls == ["en_core_web_sm", "en_core_web_lg"] + assert sm.nlp is sm_again.nlp + assert sm.nlp is not lg.nlp + + def test_ner_extractor_failed_load_not_cached_and_retried(self, monkeypatch): + """A missing model must not poison the cache. A subsequent construction + after the model becomes available must succeed and share the loaded model.""" + attempts = [] + + def failing_load(name, **_kwargs): + attempts.append(name) + raise OSError(f"Can't find model '{name}'") + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(failing_load)) + + # Construction with missing model: nlp must remain None, no crash + extractor1 = NERExtractor(method="ml") + assert extractor1.nlp is None + assert len(attempts) == 1, "one load attempt expected for the missing model" + + # Second construction: must retry (cache must not hold the failure) + extractor2 = NERExtractor(method="ml") + assert extractor2.nlp is None + assert len(attempts) == 2, "a failed load must not be cached" + + # Now install a working model and verify recovery + def working_load(name, **_kwargs): + attempts.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(working_load)) + + extractor3 = NERExtractor(method="ml") + extractor4 = NERExtractor(method="ml") + + assert extractor3.nlp is not None + assert extractor3.nlp is extractor4.nlp + assert len(attempts) == 3, ( + "exactly one successful load expected after the model becomes available" + ) + + def test_ner_extractor_non_ml_method_does_not_load_model(self, monkeypatch): + """NERExtractor with a non-ml method must not touch the spaCy cache.""" + calls = [] + + def fake_load(name, **kwargs): + calls.append(name) + return _nlp_mock() + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(fake_load)) + + NERExtractor(method="pattern") + NERExtractor(method="llm") + NERExtractor(method="regex") + + assert calls == [], "non-ml methods must not trigger any spacy.load()" + + if __name__ == "__main__": pytest.main([__file__]) diff --git a/tests/split/test_splitter.py b/tests/split/test_splitter.py index 76cc872f..725b959a 100644 --- a/tests/split/test_splitter.py +++ b/tests/split/test_splitter.py @@ -30,18 +30,16 @@ class TestSplitter(unittest.TestCase): splitter = TextSplitter(method=["recursive", "token"]) self.assertEqual(splitter.methods, ["recursive", "token"]) - @patch('semantica.split.semantic_chunker.spacy') + @patch('semantica.semantic_extract.methods.spacy') def test_semantic_chunker_initialization(self, mock_spacy): - # Mock spacy.load to return a mock nlp object + # SemanticChunker now loads spaCy through the centralized + # load_spacy_model() in semantic_extract.methods, so we patch + # methods.spacy rather than the removed semantic_chunker.spacy binding. mock_nlp = MagicMock() mock_spacy.load.return_value = mock_nlp - - # We need to ensure SPACY_AVAILABLE is True for this test context if possible, - # but it is imported at module level. - # If spacy is not installed, it sets SPACY_AVAILABLE = False. - # We might need to patch the module attribute or just test fallback if spacy missing. - - chunker = SemanticChunker(chunk_size=100) + + with patch('semantica.split.semantic_chunker.SPACY_AVAILABLE', True): + chunker = SemanticChunker(chunk_size=100) self.assertEqual(chunker.chunk_size, 100) def test_chunk_dataclass(self): diff --git a/tests/test_ner_configurations.py b/tests/test_ner_configurations.py index 2fead463..15c2568a 100644 --- a/tests/test_ner_configurations.py +++ b/tests/test_ner_configurations.py @@ -101,9 +101,15 @@ class TestNERConfigurations(unittest.TestCase): self.assertEqual(entities[0].metadata["extraction_method"], "ml") self.assertEqual(entities[0].metadata["model"], "en_core_web_trf") - @patch('semantica.semantic_extract.ner_extractor.spacy') + @patch('semantica.semantic_extract.methods.spacy') def test_ner_ml_init_falls_back_when_spacy_runtime_is_broken(self, mock_spacy): - """Test NER init does not crash when spaCy is installed but unusable at runtime.""" + """Test NER init does not crash when spaCy is installed but unusable at runtime. + + The model load now goes through load_spacy_model() in semantic_extract.methods, + so we patch methods.spacy (not ner_extractor.spacy) to inject the failure. + """ + from semantica.semantic_extract.methods import clear_spacy_model_cache + clear_spacy_model_cache() mock_spacy.load.side_effect = RuntimeError("ConfigSchemaNlp is not fully defined") with patch('semantica.semantic_extract.ner_extractor.SPACY_AVAILABLE', True): @@ -112,17 +118,23 @@ class TestNERConfigurations(unittest.TestCase): self.assertIsNone(extractor.nlp) self.assertFalse(extractor._ml_runtime_usable) - @patch('semantica.semantic_extract.ner_extractor.spacy') @patch('semantica.semantic_extract.methods.get_entity_method') @patch('semantica.semantic_extract.methods.spacy') def test_ner_ml_runtime_failure_disables_repeated_ml_load_attempts( self, mock_methods_spacy, mock_get_method, - mock_init_spacy, ): - """Test degraded ML mode skips repeated spaCy load attempts after init failure.""" - mock_init_spacy.load.side_effect = RuntimeError("ConfigSchemaNlp is not fully defined") + """Test degraded ML mode skips repeated spaCy load attempts after init failure. + + The model load at construction time now goes through load_spacy_model() in + semantic_extract.methods, so methods.spacy is the single mock target for the + init-time failure. After the RuntimeError is raised, _ml_runtime_usable is + False and no further spacy.load (or extract_entities_ml) calls are made. + """ + from semantica.semantic_extract.methods import clear_spacy_model_cache + clear_spacy_model_cache() + mock_methods_spacy.load.side_effect = RuntimeError("ConfigSchemaNlp is not fully defined") mock_ml_method = MagicMock(return_value=[]) mock_get_method.side_effect = lambda name: mock_ml_method if name == "ml" else (lambda *_args, **_kwargs: []) @@ -132,8 +144,9 @@ class TestNERConfigurations(unittest.TestCase): entities = extractor.extract_entities(self.text) self.assertFalse(extractor._ml_runtime_usable) - self.assertEqual(mock_init_spacy.load.call_count, 1) - self.assertEqual(mock_methods_spacy.load.call_count, 0) + # methods.spacy.load called once during __init__ (the RuntimeError); not again + # during extract_entities because _filter_unusable_methods removes "ml". + self.assertEqual(mock_methods_spacy.load.call_count, 1) self.assertEqual(mock_ml_method.call_count, 0) self.assertIsInstance(entities, list) From a8194dfc60a17de99e926f153bc7c8fa3f3a8598 Mon Sep 17 00:00:00 2001 From: KaifAhmad1 Date: Mon, 17 Aug 2026 13:16:17 +0530 Subject: [PATCH 7/7] fix(split): catch broken-runtime spaCy failures in SemanticChunker SemanticChunker.__init__ only caught OSError around load_spacy_model(), while NERExtractor's identical call (fixed earlier in this PR) also catches generic Exception for a model that is installed but fails at runtime. Bring SemanticChunker in line so a broken spaCy config degrades to fallback chunking instead of crashing __init__. Adds a regression test mirroring the existing NERExtractor case, and a CHANGELOG entry for #998/#1042. --- CHANGELOG.md | 8 ++++++++ semantica/split/semantic_chunker.py | 7 +++++++ tests/split/test_spacy_model_cache.py | 18 ++++++++++++++++++ 3 files changed, 33 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 78212679..f670cb22 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -73,6 +73,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **`split`/chunking paths bypassed the centralized spaCy model cache, reloading the model on every call** (#1042, closes #998) by @Accute9, reviewed by @Sameer6305 + - `semantica/split/methods.py`'s `split_by_sentences()` and `semantica/split/semantic_chunker.py`'s `SemanticChunker.__init__` each called `spacy.load()` directly instead of reusing the process-level cache added in #889/`semantic_extract/methods.py`'s `load_spacy_model()` — every call/construction re-paid the ~120ms model-load cost independently of `NERExtractor`, which already used the cache + - Both now route through `load_spacy_model()`, sharing one cached `Language` instance per model name across `split_by_sentences()`, `SemanticChunker`, and `NERExtractor`; a missing model still falls back to regex/paragraph chunking without poisoning the cache for a later successful load + - **Fixed during review** (@Sameer6305): `NERExtractor.__init__()` still had a direct `spacy.load()` call site with the same cache-bypass issue, outside the two files named in #998 but sharing the same root cause; routed through the cache alongside stale test patch targets and a strengthened cache-configuration assertion + - **Fixed during review** (@KaifAhmad1): `SemanticChunker.__init__` only caught `OSError` around `load_spacy_model()`, while the sibling fix to `NERExtractor` in this same PR added a broader `except Exception` for a model that is installed but fails at runtime (e.g. a config incompatible with the installed spaCy version). A broken-but-present model crashed `SemanticChunker()` outright instead of degrading to fallback chunking like every other path in this PR. Added the matching `except Exception` branch, leaving `self.nlp` as `None`; new `test_semantic_chunker_falls_back_when_spacy_runtime_is_broken` mirrors the existing `NERExtractor` regression test for the same scenario + - New `tests/split/test_spacy_model_cache.py`: cache reuse across repeated calls/instances, shared cache between `split_by_sentences()`/`SemanticChunker`/`NERExtractor`, distinct model names loading separately, missing-model fallback without poisoning the cache, and the broken-runtime fallback added above + - `pytest tests/split/test_spacy_model_cache.py tests/split/test_splitter.py tests/split/test_chunkers.py`: all passing (3 pre-existing, unrelated `tests/test_ner_configurations.py` failures confirmed present on `main` before this PR) + - **`export_yaml` raised a raw `AttributeError` on list input, silently wrote empty exports for unrecognized dict keys, and graph payloads were reconciled differently by every exporter** (#958, closes #956, #952, #953) by @pravit-amp, reviewed by @Sameer6305 - Graph payloads circulate under two vocabularies, `entities`/`relationships` and `nodes`/`edges`, and each exporter reconciled them locally with a different idiom — `LPGExporter` in particular dropped every entity whenever `nodes` was present but empty, the exact shape `JSONExporter` emits. A new `normalize_graph_payload()` in `utils/helpers.py` centralizes that decision once, adopted by `LPGExporter`, `ArangoAQLExporter`, `Neo4jCSVExporter`, and both YAML exporters; `ContextGraph.to_dict()` now round-trips through YAML correctly as a result - `export_yaml(records, path)` on a bare list previously failed with `AttributeError` from inside the exporter; it and the other YAML methods now reject non-mapping input with an actionable `ProcessingError` naming the expected keys, since these formats distinguish entities/relationships/triplets and guessing which one a list represents would mislabel the records diff --git a/semantica/split/semantic_chunker.py b/semantica/split/semantic_chunker.py index 2945bbd5..fc6fa6aa 100644 --- a/semantica/split/semantic_chunker.py +++ b/semantica/split/semantic_chunker.py @@ -86,6 +86,13 @@ class SemanticChunker: self.logger.warning( f"spaCy model {model_name} not found. Using fallback chunking." ) + except Exception: + self.logger.warning( + "spaCy model %s failed to initialize and will be disabled " + "for this chunker instance. Using fallback chunking.", + model_name, + exc_info=True, + ) def chunk(self, text: str, **options) -> List[Chunk]: """ diff --git a/tests/split/test_spacy_model_cache.py b/tests/split/test_spacy_model_cache.py index de21e433..00012030 100644 --- a/tests/split/test_spacy_model_cache.py +++ b/tests/split/test_spacy_model_cache.py @@ -177,6 +177,24 @@ class TestSpacyModelCache: ) assert chunker2.nlp is not None + def test_semantic_chunker_falls_back_when_spacy_runtime_is_broken( + self, monkeypatch + ): + """A spaCy model that is installed but unusable at runtime (e.g. a + config incompatible with the installed spaCy version) must degrade + SemanticChunker to fallback chunking, not crash __init__ -- mirrors + TestNERExtractorSpacyModelCache's equivalent broken-runtime test. + """ + + def broken_load(name, **_kwargs): + raise RuntimeError("ConfigSchemaNlp is not fully defined") + + monkeypatch.setattr(se_methods, "spacy", _fake_spacy(broken_load)) + + chunker = semantic_chunker.SemanticChunker() + + assert chunker.nlp is None + class TestNERExtractorSpacyModelCache: """NERExtractor(method="ml") must reuse the centralized cache in