Compare commits

...
Author SHA1 Message Date
Zohaib Hassnain 14760f7b36 docs(vector_store): note the milvus schema assumption 2026-08-31 13:44:57 +05:00
Zohaib Hassnain 54a387fdca feat(vector_store): add milvus iter_all 2026-08-31 12:21:15 +05:00
Zohaib Hassnain ec9e63e16f fix(vector_store): drop qdrant migrate wiring, keep iter_all only
VectorStore cannot actually migrate to or from qdrant yet. _init_backend_store constructs QdrantStore without connecting or selecting a collection, so reads raise a Collection not initialized error, and the facade store_vectors dispatches only to add/add_vectors while QdrantStore exposes insert_vectors, so writes raise NotImplementedError.

Both are pre-existing facade gaps that nothing had exposed, since migrate previously only allowed faiss/sqlite/pgvector. Adding qdrant to the allowlist claimed support that does not work end to end, so it is removed along with the dimension inference that only fires for backends missing a .dimension attribute. Tracked separately; this PR keeps just the iter_all primitive.
2026-08-31 03:00:56 +05:00
Zohaib Hassnain 274d5d1195 feat(vector_store): add iter_all enumeration and wire up qdrant migration 2026-08-31 02:34:26 +05:00
6 changed files with 575 additions and 1 deletions
+69
View File
@@ -680,6 +680,75 @@ class MilvusStore:
self.logger.warning(f"Failed to query Milvus vectors by metadata expression: {e}")
return []
def iter_all(self, batch_size: int = 500):
"""
Iterate over every stored entity using Milvus's query iterator.
Paginates by primary-key cursor rather than row offset, which is why
this exists instead of scan_vectors(offset, limit). query(offset=...)
is capped by the 16384 result window and would truncate anything
larger.
Assumes the schema create_collection() builds: a VARCHAR `id` primary
key plus vector and metadata fields, as get_vector() and
filter_by_metadata() already do. get_collection() does not validate
schema, so a collection with an integer key or no metadata field fails
here.
Args:
batch_size: Entities to request per iterator batch
Yields:
Result dicts with 'id', 'metadata', and 'vector', in cursor order
Raises:
ProcessingError: If the collection is not initialized, or the
installed pymilvus does not expose query_iterator().
"""
if self.collection is None or not MILVUS_AVAILABLE:
raise ProcessingError(
"Collection not initialized. Call create_collection() or get_collection() first."
)
query_iterator = getattr(self.collection.collection, "query_iterator", None)
if not callable(query_iterator):
raise ProcessingError(
"This pymilvus version does not expose Collection.query_iterator(), "
"which full enumeration requires. Falling back to query(offset=...) "
"is not safe here: it is capped by the 16384 result window and would "
"silently truncate a larger collection."
)
# Query operations need a loaded collection. Idempotent, and once per
# scan rather than per batch.
self.collection.load()
# Milvus rejects an empty expression; this match-all form is what
# filter_by_metadata() already uses.
iterator = query_iterator(
batch_size=batch_size,
expr="id != ''",
output_fields=["id", "vector", "metadata"],
)
try:
while True:
batch = iterator.next()
if not batch:
return
for item in batch:
vec = item.get("vector")
yield {
"id": str(item.get("id")),
"metadata": item.get("metadata") or {},
"vector": np.array(vec) if vec is not None else None,
}
finally:
# Release the server-side iterator even if the consumer stops early.
close = getattr(iterator, "close", None)
if callable(close):
close()
def get_stats(self, collection_name: Optional[str] = None) -> Dict[str, Any]:
"""Get collection statistics."""
if self.collection is None and collection_name:
+56
View File
@@ -604,6 +604,62 @@ class QdrantStore:
self.logger.warning(f"Failed to scroll Qdrant points by metadata filter: {e}")
return []
def iter_all(self, batch_size: int = 500):
"""
Iterate over every stored point using Qdrant's native scroll cursor.
Qdrant paginates by point-ID cursor, not by row offset, so this is
exposed instead of scan_vectors(offset, limit). An integer passed to
scroll()'s offset is a point ID rather than a rank, so there is no way
to seek to "the Nth record" without walking from the start.
VectorStore.iter_vectors() prefers this method when it is present.
Assumes a single unnamed vector per point, matching how insert_vectors()
writes them and how get_vector() reads them back. Collections configured
with named or multi-vectors are not handled here.
Args:
batch_size: Points to request per scroll call
Yields:
Result dicts with 'id', 'metadata', and 'vector', in scroll order
Raises:
ProcessingError: If the collection or client is not initialized.
Errors are raised rather than swallowed because a scan that
silently yields nothing is indistinguishable from an empty
source, which would let a caller such as `store migrate`
report success having copied nothing (issue #1083).
"""
if self.collection is None or self.client is None or not QDRANT_AVAILABLE:
raise ProcessingError(
"Collection not initialized. Call create_collection() or get_collection() first."
)
next_offset = None
while True:
records, next_offset = self.client.scroll(
collection_name=self.collection.collection_name,
limit=batch_size,
offset=next_offset,
with_payload=True,
with_vectors=True,
)
for rec in records:
yield {
"id": str(rec.id),
"metadata": rec.payload or {},
"vector": np.array(rec.vector) if rec.vector is not None else None,
}
# The final page can carry records while already reporting no next
# cursor, so those records are yielded above before stopping here.
# Calling scroll() again with offset=None would restart from the
# beginning rather than continue past the end.
if next_offset is None or not records:
return
def delete_vectors(
self, point_ids: List[Union[str, int]], **options
) -> Dict[str, Any]:
+14 -1
View File
@@ -867,12 +867,25 @@ class VectorStore:
"""
Iterate over every stored vector, one page at a time.
Backends whose native pagination is cursor based (Qdrant, Pinecone,
Milvus, Weaviate) cannot honestly implement the positional
scan_vectors(offset, limit) contract, so they expose iter_all()
instead and it is preferred here when present. Backends with real
positional access (inmemory, FAISS, SQLite-vec, PgVector) fall
through to the offset loop below.
Args:
batch_size: Number of vectors to fetch per underlying scan_vectors() call
batch_size: Number of vectors to fetch per underlying call
Yields:
Result dicts with 'id', 'metadata', and 'vector', in scan order
"""
if self.backend != "inmemory" and self._backend_store is not None:
iter_all = getattr(self._backend_store, "iter_all", None)
if callable(iter_all):
yield from iter_all(batch_size=batch_size)
return
offset = 0
while True:
page = self.scan_vectors(offset=offset, limit=batch_size)
+176
View File
@@ -0,0 +1,176 @@
"""Tests for MilvusStore.iter_all() query-iterator enumeration.
pymilvus is not installed in this environment, so these drive the real
MilvusStore against MagicMocks, following the pattern already used for milvus
in test_backend_metadata_filtering.py.
"""
from unittest.mock import MagicMock, patch
import numpy as np
import pytest
from semantica.utils.exceptions import ProcessingError
from semantica.vector_store.milvus_store import MilvusStore
def _store_with_batches(*batches):
"""MilvusStore whose query_iterator yields the given batches then stops.
The attribute path is doubled here: the pymilvus Collection sits at
wrapper.collection.
"""
store = MilvusStore()
wrapper = MagicMock()
inner = MagicMock()
iterator = MagicMock()
iterator.next.side_effect = list(batches)
inner.query_iterator.return_value = iterator
wrapper.collection = inner
store.collection = wrapper
return store, wrapper, inner, iterator
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_yields_batches_until_exhausted():
"""Exhaustion is an empty list, not StopIteration."""
store, _, _, iterator = _store_with_batches(
[{"id": 1, "vector": [0.1], "metadata": {}}],
[{"id": 2, "vector": [0.2], "metadata": {}}],
[],
)
result = list(store.iter_all(batch_size=1))
assert [item["id"] for item in result] == ["1", "2"]
assert iterator.next.call_count == 3
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_requests_the_fields_needed_for_the_result_shape():
store, _, inner, _ = _store_with_batches([])
list(store.iter_all(batch_size=64))
kwargs = inner.query_iterator.call_args[1]
assert kwargs["batch_size"] == 64
assert kwargs["output_fields"] == ["id", "vector", "metadata"]
# Milvus rejects an empty expression, so a match-all form is required.
assert kwargs["expr"] == "id != ''"
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_loads_the_collection_before_querying():
"""Milvus requires a loaded collection for query operations."""
store, wrapper, _, _ = _store_with_batches([])
list(store.iter_all())
assert wrapper.load.called
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_closes_the_iterator_on_exhaustion():
store, _, _, iterator = _store_with_batches([])
list(store.iter_all())
assert iterator.close.called
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_closes_the_iterator_when_consumer_stops_early():
"""Abandoning the generator early must still release the iterator."""
store, _, _, iterator = _store_with_batches(
[{"id": 1, "vector": [0.1], "metadata": {}}],
[{"id": 2, "vector": [0.2], "metadata": {}}],
[],
)
generator = store.iter_all(batch_size=1)
next(generator)
assert not iterator.close.called
generator.close()
assert iterator.close.called
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_converts_entities_to_the_shared_result_shape():
store, _, _, _ = _store_with_batches(
[{"id": 7, "vector": [0.1, 0.2, 0.3], "metadata": {"tag": "x"}}], []
)
item = list(store.iter_all())[0]
assert item["id"] == "7"
assert item["metadata"] == {"tag": "x"}
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2, 0.3]))
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_handles_missing_vector_and_metadata():
store, _, _, _ = _store_with_batches([{"id": 1, "vector": None, "metadata": None}], [])
item = list(store.iter_all())[0]
assert item["metadata"] == {}
assert item["vector"] is None
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_empty_collection_yields_nothing():
store, _, _, _ = _store_with_batches([])
assert list(store.iter_all()) == []
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_raises_when_query_iterator_is_unavailable():
"""Older pymilvus lacks query_iterator; falling back to query(offset=...)
would truncate at the 16384 window."""
store = MilvusStore()
wrapper = MagicMock()
wrapper.collection = MagicMock(spec=["query"])
store.collection = wrapper
with pytest.raises(ProcessingError, match="query_iterator"):
list(store.iter_all())
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_raises_when_collection_not_initialized():
"""Must fail loudly: an empty scan reads the same as an empty source."""
store = MilvusStore()
with pytest.raises(ProcessingError, match="Collection not initialized"):
list(store.iter_all())
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", False)
def test_iter_all_raises_when_milvus_unavailable():
store = MilvusStore()
store.collection = MagicMock()
with pytest.raises(ProcessingError):
list(store.iter_all())
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_propagates_iterator_errors():
store, _, _, iterator = _store_with_batches()
iterator.next.side_effect = RuntimeError("connection reset")
with pytest.raises(RuntimeError, match="connection reset"):
list(store.iter_all())
@patch("semantica.vector_store.milvus_store.MILVUS_AVAILABLE", True)
def test_iter_all_closes_the_iterator_when_a_batch_fails():
store, _, _, iterator = _store_with_batches()
iterator.next.side_effect = RuntimeError("connection reset")
with pytest.raises(RuntimeError):
list(store.iter_all())
assert iterator.close.called
+152
View File
@@ -0,0 +1,152 @@
"""Tests for QdrantStore.iter_all() cursor enumeration.
Qdrant is not installed in this environment, so these drive the real
QdrantStore against a MagicMock standing in for the qdrant_client, following
the pattern already used for qdrant in test_backend_metadata_filtering.py.
"""
from unittest.mock import MagicMock, patch
import numpy as np
import pytest
from semantica.utils.exceptions import ProcessingError
from semantica.vector_store.qdrant_store import QdrantStore
def _record(point_id, payload=None, vector=None):
"""Build a stand-in for a qdrant_client Record."""
rec = MagicMock()
rec.id = point_id
rec.payload = payload
rec.vector = vector
return rec
def _store_with_scroll(*pages):
"""QdrantStore whose client.scroll() returns the given (records, cursor) pages."""
store = QdrantStore()
store.client = MagicMock()
store.client.scroll.side_effect = list(pages)
store.collection = MagicMock()
store.collection.collection_name = "test_collection"
return store
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_threads_cursor_across_pages():
"""The next call must continue from the previous page's next_page_offset."""
store = _store_with_scroll(
([_record(1), _record(2)], "cursor-1"),
([_record(3)], None),
)
result = list(store.iter_all(batch_size=2))
assert [item["id"] for item in result] == ["1", "2", "3"]
calls = store.client.scroll.call_args_list
assert len(calls) == 2
assert calls[0][1]["offset"] is None
assert calls[0][1]["limit"] == 2
assert calls[1][1]["offset"] == "cursor-1"
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_yields_final_page_that_reports_no_next_cursor():
"""Qdrant can return records and a null cursor on the same page.
Those records must still be yielded. Treating a null cursor as "stop
before this page" would silently drop the tail of every scan.
"""
store = _store_with_scroll(([_record(1), _record(2)], None))
result = list(store.iter_all(batch_size=10))
assert [item["id"] for item in result] == ["1", "2"]
assert store.client.scroll.call_count == 1
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_converts_records_to_the_shared_result_shape():
store = _store_with_scroll(
([_record(7, payload={"tag": "x"}, vector=[0.1, 0.2, 0.3])], None),
)
item = list(store.iter_all())[0]
assert item["id"] == "7"
assert item["metadata"] == {"tag": "x"}
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2, 0.3]))
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_handles_missing_payload_and_vector():
store = _store_with_scroll(([_record(1, payload=None, vector=None)], None))
item = list(store.iter_all())[0]
assert item["metadata"] == {}
assert item["vector"] is None
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_empty_collection_yields_nothing():
store = _store_with_scroll(([], None))
assert list(store.iter_all()) == []
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_stops_on_empty_page_even_with_a_cursor():
"""Defensive: an empty page ends the scan rather than looping forever."""
store = _store_with_scroll(([], "cursor-that-never-clears"))
assert list(store.iter_all()) == []
assert store.client.scroll.call_count == 1
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_raises_when_collection_not_initialized():
"""Must fail loudly, not yield nothing.
An empty scan is indistinguishable from an empty source, which would let
`store migrate` report success having copied nothing (issue #1083).
"""
store = QdrantStore()
with pytest.raises(ProcessingError, match="Collection not initialized"):
list(store.iter_all())
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", False)
def test_iter_all_raises_when_qdrant_unavailable():
store = QdrantStore()
store.client = MagicMock()
store.collection = MagicMock()
with pytest.raises(ProcessingError):
list(store.iter_all())
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_propagates_scroll_errors():
store = QdrantStore()
store.client = MagicMock()
store.client.scroll.side_effect = RuntimeError("connection reset")
store.collection = MagicMock()
store.collection.collection_name = "test_collection"
with pytest.raises(RuntimeError, match="connection reset"):
list(store.iter_all())
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
def test_iter_all_requests_payload_and_vectors():
store = _store_with_scroll(([], None))
list(store.iter_all())
kwargs = store.client.scroll.call_args[1]
assert kwargs["with_payload"] is True
assert kwargs["with_vectors"] is True
assert kwargs["collection_name"] == "test_collection"
@@ -26,6 +26,7 @@ from unittest.mock import MagicMock, patch
import numpy as np
from semantica.utils.exceptions import ProcessingError
from semantica.vector_store.vector_store import VectorStore, VectorManager
@@ -138,6 +139,38 @@ class _NonScanningBackendStore:
"""Fake persistent backend store without any scan capability."""
class _IterAllBackendStore:
"""Fake cursor-based backend store exposing iter_all() but not scan_vectors().
Mirrors qdrant/pinecone/milvus/weaviate, which cannot honour a positional
offset and therefore expose native iteration instead.
"""
def __init__(self, items):
self._items = items
self.batch_sizes = []
def iter_all(self, batch_size=500):
self.batch_sizes.append(batch_size)
for item in self._items:
yield item
def scan_vectors(self, offset=0, limit=100):
raise AssertionError("scan_vectors() must not be called when iter_all() exists")
class _MisShapedIterAllBackendStore:
"""Backend store whose ``iter_all`` attribute is not callable."""
iter_all = 42 # plain attribute, not a method
def __init__(self, items):
self._items = items
def scan_vectors(self, offset=0, limit=100):
return self._items[offset:offset + limit]
class VectorStoreScanVectorsTests(unittest.TestCase):
"""VectorStore.scan_vectors() / iter_vectors() backend-agnostic accessors."""
@@ -192,6 +225,81 @@ class VectorStoreScanVectorsTests(unittest.TestCase):
self.assertEqual(list(store.iter_vectors(batch_size=2)), [])
# ---------------------------------------------------------------------------
# VectorStore.iter_vectors() preference for a native iter_all()
# ---------------------------------------------------------------------------
class VectorStoreIterAllDispatchTests(unittest.TestCase):
"""iter_vectors() prefers a backend's native iter_all() when present.
Cursor-based backends cannot implement scan_vectors(offset, limit)
honestly, so they expose iter_all() instead and iter_vectors() routes to
it rather than walking offsets.
"""
def _persistent_store(self, backend_store, backend_name="qdrant"):
store = VectorStore(backend="inmemory", dimension=2)
store.backend = backend_name
store._backend_store = backend_store
return store
def test_iter_vectors_uses_iter_all_when_available(self):
items = [
{"id": "a", "vector": None, "metadata": {"n": 1}},
{"id": "b", "vector": None, "metadata": {"n": 2}},
]
backend = _IterAllBackendStore(items)
store = self._persistent_store(backend)
self.assertEqual(list(store.iter_vectors(batch_size=7)), items)
def test_iter_vectors_forwards_batch_size_to_iter_all(self):
backend = _IterAllBackendStore([])
store = self._persistent_store(backend)
list(store.iter_vectors(batch_size=32))
self.assertEqual(backend.batch_sizes, [32])
def test_iter_vectors_falls_back_to_scan_vectors_without_iter_all(self):
items = [{"id": "a", "vector": None, "metadata": {}}]
store = self._persistent_store(_ScanningBackendStore(items))
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
def test_iter_vectors_falls_back_when_iter_all_not_callable(self):
# A mis-shaped adapter exposing a non-callable ``iter_all`` must not be
# invoked; the offset path still has to work. Mirrors the count()
# precedent in _MisShapedBackendStore.
items = [{"id": "a", "vector": None, "metadata": {}}]
store = self._persistent_store(_MisShapedIterAllBackendStore(items))
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
def test_iter_vectors_inmemory_ignores_iter_all(self):
store = VectorStore(backend="inmemory", dimension=2)
store.store_vectors([np.array([1.0, 0.0])], [{"type": "a"}])
store._backend_store = _IterAllBackendStore([{"id": "wrong"}])
collected = list(store.iter_vectors(batch_size=2))
self.assertEqual([item["metadata"] for item in collected], [{"type": "a"}])
def test_iter_vectors_propagates_iter_all_errors(self):
# A scan that silently yields nothing is indistinguishable from an
# empty source, which would let `store migrate` report success having
# copied nothing (issue #1083).
class _FailingIterAll:
def iter_all(self, batch_size=500):
raise ProcessingError("backend unreachable")
yield # pragma: no cover - makes this a generator
store = self._persistent_store(_FailingIterAll())
with self.assertRaises(ProcessingError):
list(store.iter_vectors(batch_size=2))
# ---------------------------------------------------------------------------
# VectorManager tests — inmemory backend
# ---------------------------------------------------------------------------