mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-09-01 04:00:28 +00:00
Compare commits
5
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8d4748274d | ||
|
|
4a2aa1b3ed | ||
|
|
8a1db9e2e2 | ||
|
|
ec9e63e16f | ||
|
|
274d5d1195 |
@@ -6,7 +6,6 @@
|
||||
!README.md
|
||||
!LICENSE
|
||||
!MANIFEST.in
|
||||
!requirements-ci.txt
|
||||
!semantica/
|
||||
!semantica/**
|
||||
!integrations/
|
||||
|
||||
@@ -72,7 +72,7 @@ jobs:
|
||||
diff \
|
||||
<(grep -E '^[a-zA-Z0-9._-]+==' requirements-ci.txt | sed 's/ \\$//') \
|
||||
<(grep -E '^[a-zA-Z0-9._-]+==' /tmp/requirements-ci-check.txt)
|
||||
- run: pip install build==1.6.0
|
||||
- run: pip install build
|
||||
# wheel is build-time only (not in requirements-ci.txt) — install the
|
||||
# same pinned version [build-system] declares so --no-isolation works.
|
||||
- run: pip install wheel==0.48.0
|
||||
|
||||
@@ -12,7 +12,6 @@ on:
|
||||
- 'README.md'
|
||||
- 'LICENSE'
|
||||
- 'MANIFEST.in'
|
||||
- 'requirements-ci.txt'
|
||||
- 'semantica/**'
|
||||
- 'integrations/**'
|
||||
- 'explorer/**'
|
||||
|
||||
@@ -16,7 +16,7 @@ jobs:
|
||||
cancel-in-progress: false
|
||||
permissions:
|
||||
contents: write # for the GitHub Release
|
||||
id-token: write # for PyPI Trusted Publishing (OIDC), attestation signing, and Sigstore
|
||||
id-token: write # for PyPI Trusted Publishing (OIDC) and attestation signing
|
||||
attestations: write # for SLSA build provenance
|
||||
# If you add another job to this workflow, give it its own explicit
|
||||
# `permissions:` block rather than relying on the workflow-level default
|
||||
@@ -40,7 +40,7 @@ jobs:
|
||||
# build runs against the same versions CI tests against.
|
||||
- name: Install pinned build dependencies
|
||||
run: pip install -r requirements-ci.txt
|
||||
- run: pip install build==1.6.0
|
||||
- run: pip install build
|
||||
# wheel is build-time only (not in requirements-ci.txt) — install the
|
||||
# same pinned version [build-system] declares so --no-isolation works.
|
||||
- run: pip install wheel==0.48.0
|
||||
@@ -71,21 +71,7 @@ jobs:
|
||||
uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4
|
||||
with:
|
||||
subject-path: 'dist/*'
|
||||
# attest-build-provenance publishes to the GH attestations API only, which
|
||||
# OpenSSF Scorecard's Signed-Releases check does not inspect - it looks for
|
||||
# signature files attached as release assets. Sign here too so
|
||||
# `dist/*.sigstore.json` bundles ship alongside the wheel/sdist on the
|
||||
# GitHub Release itself.
|
||||
- name: Sign artifacts with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@790bc6befb9d733738f18d8f895854b453640ec9 # v3.5.0
|
||||
with:
|
||||
inputs: |
|
||||
dist/*.whl
|
||||
dist/*.tar.gz
|
||||
- uses: softprops/action-gh-release@efb35369e0ad2afab669f228072c1b0d510eae64 # v3.0.3
|
||||
with:
|
||||
files: |
|
||||
dist/*.whl
|
||||
dist/*.tar.gz
|
||||
dist/*.sigstore.json
|
||||
files: dist/*
|
||||
- uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
|
||||
@@ -52,7 +52,7 @@ jobs:
|
||||
# Tooling AFTER the pinned set: installing safety/bandit/semgrep/jq
|
||||
# first lets the pinned requirements overwrite their transitive deps
|
||||
# (e.g. rich), which breaks the safety CLI at runtime.
|
||||
pip install safety==3.8.1 bandit==1.9.4 semgrep==1.175.0 jq==1.12.0
|
||||
pip install safety bandit semgrep jq
|
||||
|
||||
- name: Run Safety Check (Package Vulnerabilities)
|
||||
run: |
|
||||
|
||||
@@ -37,6 +37,6 @@ jobs:
|
||||
# pyproject.toml changes under review. The schedule/workflow_dispatch
|
||||
# runs stay non-blocking until a full pass over pre-existing findings
|
||||
# across the whole [all] tree has been done.
|
||||
- run: pip install pip-audit==2.10.1
|
||||
- run: pip install pip-audit
|
||||
- run: pip-audit -r requirements-ci.txt
|
||||
continue-on-error: ${{ github.event_name != 'pull_request' }}
|
||||
|
||||
+4
-30
@@ -1,5 +1,5 @@
|
||||
# syntax=docker/dockerfile:1
|
||||
FROM node:26-alpine@sha256:2d984a15c9b54fd0aeb608b8e0d0d83529eb34d2966db27a1fb4f1edc3d298a3 AS frontend-builder
|
||||
FROM node:26-alpine AS frontend-builder
|
||||
|
||||
WORKDIR /app
|
||||
COPY explorer/package*.json ./explorer/
|
||||
@@ -9,18 +9,7 @@ RUN npm ci
|
||||
COPY explorer/ ./
|
||||
RUN mkdir -p /app/semantica && npm run build
|
||||
|
||||
# CVE-2026-14456 (OpenSSL QUIC-server DoS, flagged against this base image's
|
||||
# openssl/libssl3t64/openssl-provider-legacy): the Debian fix
|
||||
# (3.5.7-1~deb13u2) is only in trixie-proposed-updates as of this writing,
|
||||
# not yet promoted to trixie-security, so there's no package to pin here
|
||||
# today. Deliberately NOT running `apt-get upgrade` to chase it - that
|
||||
# breaks build reproducibility (terrascan AC_DOCKER_0052) and still
|
||||
# wouldn't reach a proposed-updates-only package. Once Debian ships the fix
|
||||
# and rebuilds this tag, the docker Dependabot ecosystem in
|
||||
# .github/dependabot.yml opens a PR bumping the digest pin above. Also: this
|
||||
# image only serves plain HTTP via uvicorn and never opens a QUIC listener,
|
||||
# so the bug isn't reachable here regardless.
|
||||
FROM python:3.13-slim@sha256:7ce4b6dfe35e55397b7cda544f8a13f191b7ae28dc5aad71fe664dbc9bc2623f AS runtime
|
||||
FROM python:3.13-slim AS runtime
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
@@ -33,27 +22,12 @@ WORKDIR /app
|
||||
RUN groupadd --system semantica \
|
||||
&& useradd --system --gid semantica --home-dir /app --shell /usr/sbin/nologin semantica
|
||||
|
||||
COPY pyproject.toml README.md LICENSE MANIFEST.in requirements-ci.txt ./
|
||||
COPY pyproject.toml README.md LICENSE MANIFEST.in ./
|
||||
COPY semantica/ ./semantica/
|
||||
COPY integrations/ ./integrations/
|
||||
COPY --from=frontend-builder /app/semantica/static ./semantica/static
|
||||
|
||||
# The base image ships an outdated setuptools (CVE-2025-47273); upgrade it
|
||||
# explicitly since nothing in our own dependency tree otherwise pulls a
|
||||
# newer copy. Pinned to the exact version requirements-ci.txt/pyproject.toml
|
||||
# already build against, rather than a floor, per terrascan AC_DOCKER_0010.
|
||||
# requirements-ci.txt itself carries the audited, CVE-checked pins for every
|
||||
# transitive dependency (see security-scan.yml / security.yml) - feed them
|
||||
# in as an unhashed constraints file (pip's hash-checking mode rejects the
|
||||
# unhashable local source directory this installs) so the image lands on
|
||||
# the same patched versions CI verified, e.g. msgpack>=1.2.1, rather than
|
||||
# letting pip freely re-resolve and pick up an unpatched transitive version.
|
||||
# (Extracted with Python's re module rather than sed/grep so there's no
|
||||
# line-continuation-backslash stripping to get subtly wrong.)
|
||||
RUN pip install --no-cache-dir "setuptools==84.0.0" \
|
||||
&& python -c "import re, pathlib; pins = re.findall(r'^([A-Za-z0-9._-]+==\S+)', pathlib.Path('requirements-ci.txt').read_text(), re.M); pathlib.Path('/tmp/constraints.txt').write_text('\n'.join(pins))" \
|
||||
&& pip install --no-cache-dir -c /tmp/constraints.txt ".[explorer]" \
|
||||
&& rm -f /tmp/constraints.txt requirements-ci.txt \
|
||||
RUN pip install --no-cache-dir ".[explorer]" \
|
||||
&& chown -R semantica:semantica /app
|
||||
|
||||
USER semantica
|
||||
|
||||
+1
-8
@@ -49,14 +49,7 @@ dependencies = [
|
||||
"scipy>=1.13.1",
|
||||
"scikit-learn>=1.7.2",
|
||||
"umap-learn>=0.5.12",
|
||||
# thinc (spacy's core dep) dropped Python 3.9 wheels at 8.3.10, and later
|
||||
# spacy patch releases (3.8.8+) require thinc>=8.3.9-only-on-3.10+ ranges,
|
||||
# which forces a source build that fails outright on 3.9 (see Install
|
||||
# Matrix run history). Capping both keeps 3.9 on the last wheel-compatible
|
||||
# pair; 3.10+ is left unconstrained to always get the latest spacy/thinc.
|
||||
"spacy>=3.4.0,<3.8.8; python_version < '3.10'",
|
||||
"spacy>=3.4.0; python_version >= '3.10'",
|
||||
"thinc<8.3.5; python_version < '3.10'",
|
||||
"spacy>=3.4.0",
|
||||
"transformers>=4.20.0",
|
||||
"torch>=1.13.1",
|
||||
"sentence-transformers>=2.2.0",
|
||||
|
||||
@@ -299,6 +299,44 @@ class PineconeSearch:
|
||||
)
|
||||
|
||||
|
||||
def _pinecone_listed_ids(response: Any) -> List[str]:
|
||||
"""Extract vector IDs from a list_paginated() response.
|
||||
|
||||
Accepts record objects, bare id strings and dicts, since what listing
|
||||
returns has changed across pinecone SDK major versions.
|
||||
"""
|
||||
records = getattr(response, "vectors", None)
|
||||
if records is None and isinstance(response, dict):
|
||||
records = response.get("vectors")
|
||||
|
||||
ids: List[str] = []
|
||||
for record in records or []:
|
||||
if isinstance(record, str):
|
||||
ids.append(record)
|
||||
elif isinstance(record, dict):
|
||||
if record.get("id") is not None:
|
||||
ids.append(record["id"])
|
||||
else:
|
||||
record_id = getattr(record, "id", None)
|
||||
if record_id is not None:
|
||||
ids.append(record_id)
|
||||
return ids
|
||||
|
||||
|
||||
def _pinecone_next_token(response: Any) -> Optional[str]:
|
||||
"""Return the continuation token, or None when the listing is exhausted."""
|
||||
pagination = getattr(response, "pagination", None)
|
||||
if pagination is None and isinstance(response, dict):
|
||||
pagination = response.get("pagination")
|
||||
if pagination is None:
|
||||
return None
|
||||
|
||||
token = getattr(pagination, "next", None)
|
||||
if token is None and isinstance(pagination, dict):
|
||||
token = pagination.get("next")
|
||||
return token or None
|
||||
|
||||
|
||||
class PineconeStore:
|
||||
"""
|
||||
Pinecone store for vector storage and similarity search.
|
||||
@@ -735,6 +773,84 @@ class PineconeStore:
|
||||
self.logger.warning(f"Failed to filter Pinecone vectors by metadata: {e}")
|
||||
return []
|
||||
|
||||
def iter_all(self, batch_size: int = 500, namespace: str = ""):
|
||||
"""
|
||||
Iterate over every stored vector by listing IDs then fetching them.
|
||||
|
||||
Paginates with an opaque continuation token, which is why this exists
|
||||
instead of scan_vectors(offset, limit): the token for page N cannot be
|
||||
constructed without walking there.
|
||||
|
||||
Needs two calls per page, unlike the other backends, because listing
|
||||
returns IDs only. Both calls are namespace scoped and must agree, and
|
||||
listing covers one namespace rather than the whole index.
|
||||
|
||||
Args:
|
||||
batch_size: IDs to request per list_paginated() call
|
||||
namespace: Namespace to enumerate (default: the default namespace)
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in listing order
|
||||
|
||||
Raises:
|
||||
ProcessingError: If the index is not initialized, if the installed
|
||||
SDK does not expose list_paginated(), or if the listing stops
|
||||
advancing.
|
||||
"""
|
||||
if self.index is None or not PINECONE_AVAILABLE:
|
||||
raise ProcessingError(
|
||||
"Index not initialized. Call create_index() or get_index() first."
|
||||
)
|
||||
|
||||
# list_paginated() rather than list(): list() is an auto-paging
|
||||
# iterator in current SDKs but reads as plain id lists in older
|
||||
# examples. Threading the token explicitly is version-agnostic.
|
||||
list_paginated = getattr(self.index.index, "list_paginated", None)
|
||||
if not callable(list_paginated):
|
||||
raise ProcessingError(
|
||||
"This pinecone SDK version does not expose Index.list_paginated(), "
|
||||
"which full enumeration requires."
|
||||
)
|
||||
|
||||
token = None
|
||||
while True:
|
||||
kwargs: Dict[str, Any] = {"limit": batch_size, "namespace": namespace}
|
||||
if token is not None:
|
||||
kwargs["pagination_token"] = token
|
||||
|
||||
response = list_paginated(**kwargs)
|
||||
vector_ids = _pinecone_listed_ids(response)
|
||||
if not vector_ids:
|
||||
return
|
||||
|
||||
fetched = self.index.fetch_vectors(vector_ids, namespace=namespace)
|
||||
vectors = fetched.get("vectors") or {}
|
||||
|
||||
for vector_id in vector_ids:
|
||||
entry = vectors.get(vector_id)
|
||||
if entry is None:
|
||||
# fetch() omits ids it cannot find: deleted since listing.
|
||||
continue
|
||||
values = entry.get("values")
|
||||
yield {
|
||||
"id": vector_id,
|
||||
"metadata": entry.get("metadata") or {},
|
||||
"vector": np.array(values) if values is not None else None,
|
||||
}
|
||||
|
||||
next_token = _pinecone_next_token(response)
|
||||
if not next_token:
|
||||
return
|
||||
if next_token == token:
|
||||
# Distinct from exhaustion above: a partial scan here would be
|
||||
# indistinguishable from a complete one.
|
||||
raise ProcessingError(
|
||||
"Pinecone returned the same pagination token twice, so the "
|
||||
"listing is not advancing. Refusing to return a truncated "
|
||||
"scan."
|
||||
)
|
||||
token = next_token
|
||||
|
||||
def fetch_vectors(
|
||||
self, vector_ids: List[str], namespace: str = "", **options
|
||||
) -> Dict[str, Any]:
|
||||
|
||||
@@ -604,6 +604,62 @@ class QdrantStore:
|
||||
self.logger.warning(f"Failed to scroll Qdrant points by metadata filter: {e}")
|
||||
return []
|
||||
|
||||
def iter_all(self, batch_size: int = 500):
|
||||
"""
|
||||
Iterate over every stored point using Qdrant's native scroll cursor.
|
||||
|
||||
Qdrant paginates by point-ID cursor, not by row offset, so this is
|
||||
exposed instead of scan_vectors(offset, limit). An integer passed to
|
||||
scroll()'s offset is a point ID rather than a rank, so there is no way
|
||||
to seek to "the Nth record" without walking from the start.
|
||||
VectorStore.iter_vectors() prefers this method when it is present.
|
||||
|
||||
Assumes a single unnamed vector per point, matching how insert_vectors()
|
||||
writes them and how get_vector() reads them back. Collections configured
|
||||
with named or multi-vectors are not handled here.
|
||||
|
||||
Args:
|
||||
batch_size: Points to request per scroll call
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in scroll order
|
||||
|
||||
Raises:
|
||||
ProcessingError: If the collection or client is not initialized.
|
||||
Errors are raised rather than swallowed because a scan that
|
||||
silently yields nothing is indistinguishable from an empty
|
||||
source, which would let a caller such as `store migrate`
|
||||
report success having copied nothing (issue #1083).
|
||||
"""
|
||||
if self.collection is None or self.client is None or not QDRANT_AVAILABLE:
|
||||
raise ProcessingError(
|
||||
"Collection not initialized. Call create_collection() or get_collection() first."
|
||||
)
|
||||
|
||||
next_offset = None
|
||||
while True:
|
||||
records, next_offset = self.client.scroll(
|
||||
collection_name=self.collection.collection_name,
|
||||
limit=batch_size,
|
||||
offset=next_offset,
|
||||
with_payload=True,
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
for rec in records:
|
||||
yield {
|
||||
"id": str(rec.id),
|
||||
"metadata": rec.payload or {},
|
||||
"vector": np.array(rec.vector) if rec.vector is not None else None,
|
||||
}
|
||||
|
||||
# The final page can carry records while already reporting no next
|
||||
# cursor, so those records are yielded above before stopping here.
|
||||
# Calling scroll() again with offset=None would restart from the
|
||||
# beginning rather than continue past the end.
|
||||
if next_offset is None or not records:
|
||||
return
|
||||
|
||||
def delete_vectors(
|
||||
self, point_ids: List[Union[str, int]], **options
|
||||
) -> Dict[str, Any]:
|
||||
|
||||
@@ -867,12 +867,25 @@ class VectorStore:
|
||||
"""
|
||||
Iterate over every stored vector, one page at a time.
|
||||
|
||||
Backends whose native pagination is cursor based (Qdrant, Pinecone,
|
||||
Milvus, Weaviate) cannot honestly implement the positional
|
||||
scan_vectors(offset, limit) contract, so they expose iter_all()
|
||||
instead and it is preferred here when present. Backends with real
|
||||
positional access (inmemory, FAISS, SQLite-vec, PgVector) fall
|
||||
through to the offset loop below.
|
||||
|
||||
Args:
|
||||
batch_size: Number of vectors to fetch per underlying scan_vectors() call
|
||||
batch_size: Number of vectors to fetch per underlying call
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in scan order
|
||||
"""
|
||||
if self.backend != "inmemory" and self._backend_store is not None:
|
||||
iter_all = getattr(self._backend_store, "iter_all", None)
|
||||
if callable(iter_all):
|
||||
yield from iter_all(batch_size=batch_size)
|
||||
return
|
||||
|
||||
offset = 0
|
||||
while True:
|
||||
page = self.scan_vectors(offset=offset, limit=batch_size)
|
||||
|
||||
@@ -249,6 +249,166 @@ class TestPineconeIndex(unittest.TestCase):
|
||||
mock_index.query.assert_called_once()
|
||||
|
||||
|
||||
class TestPineconeIterAll(unittest.TestCase):
|
||||
"""PineconeStore.iter_all() list-then-fetch enumeration."""
|
||||
|
||||
def _page(self, ids, next_token):
|
||||
"""Stand-in for a list_paginated() response."""
|
||||
response = MagicMock()
|
||||
response.vectors = [MagicMock(id=vector_id) for vector_id in ids]
|
||||
response.pagination = MagicMock(next=next_token)
|
||||
return response
|
||||
|
||||
def _store(self, pages, fetch_results):
|
||||
store = PineconeStore()
|
||||
wrapper = MagicMock()
|
||||
raw_index = MagicMock()
|
||||
raw_index.list_paginated.side_effect = list(pages)
|
||||
wrapper.index = raw_index
|
||||
wrapper.fetch_vectors.side_effect = list(fetch_results)
|
||||
store.index = wrapper
|
||||
return store, wrapper, raw_index
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_threads_pagination_token_across_pages(self):
|
||||
store, _, raw_index = self._store(
|
||||
[self._page(["a", "b"], "token-1"), self._page(["c"], None)],
|
||||
[
|
||||
{"vectors": {"a": {"values": [0.1], "metadata": {}},
|
||||
"b": {"values": [0.2], "metadata": {}}}},
|
||||
{"vectors": {"c": {"values": [0.3], "metadata": {}}}},
|
||||
],
|
||||
)
|
||||
|
||||
result = list(store.iter_all(batch_size=2))
|
||||
|
||||
self.assertEqual([item["id"] for item in result], ["a", "b", "c"])
|
||||
calls = raw_index.list_paginated.call_args_list
|
||||
self.assertNotIn("pagination_token", calls[0][1])
|
||||
self.assertEqual(calls[1][1]["pagination_token"], "token-1")
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_hydrates_listed_ids_with_a_fetch(self):
|
||||
"""Listing returns ids only, so each page needs a fetch()."""
|
||||
store, wrapper, _ = self._store(
|
||||
[self._page(["a"], None)],
|
||||
[{"vectors": {"a": {"values": [0.1, 0.2], "metadata": {"tag": "x"}}}}],
|
||||
)
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
self.assertEqual(item["id"], "a")
|
||||
self.assertEqual(item["metadata"], {"tag": "x"})
|
||||
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2]))
|
||||
wrapper.fetch_vectors.assert_called_once_with(["a"], namespace="")
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_list_and_fetch_use_the_same_namespace(self):
|
||||
store, wrapper, raw_index = self._store(
|
||||
[self._page(["a"], None)],
|
||||
[{"vectors": {"a": {"values": [0.1], "metadata": {}}}}],
|
||||
)
|
||||
|
||||
list(store.iter_all(namespace="prod"))
|
||||
|
||||
self.assertEqual(raw_index.list_paginated.call_args[1]["namespace"], "prod")
|
||||
wrapper.fetch_vectors.assert_called_once_with(["a"], namespace="prod")
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_skips_ids_deleted_between_list_and_fetch(self):
|
||||
"""fetch() omits ids it cannot find rather than returning blanks."""
|
||||
store, _, _ = self._store(
|
||||
[self._page(["a", "gone"], None)],
|
||||
[{"vectors": {"a": {"values": [0.1], "metadata": {}}}}],
|
||||
)
|
||||
|
||||
result = list(store.iter_all())
|
||||
|
||||
self.assertEqual([item["id"] for item in result], ["a"])
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_raises_when_pagination_token_repeats(self):
|
||||
"""A stalled token must not loop forever, nor quietly return a partial
|
||||
scan that reads as a complete one."""
|
||||
store, _, raw_index = self._store(
|
||||
[self._page(["a"], "same"), self._page(["b"], "same")],
|
||||
[
|
||||
{"vectors": {"a": {"values": [0.1], "metadata": {}}}},
|
||||
{"vectors": {"b": {"values": [0.2], "metadata": {}}}},
|
||||
],
|
||||
)
|
||||
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
self.assertEqual(raw_index.list_paginated.call_count, 2)
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_empty_listing_yields_nothing_without_fetching(self):
|
||||
store, wrapper, _ = self._store([self._page([], None)], [])
|
||||
|
||||
self.assertEqual(list(store.iter_all()), [])
|
||||
wrapper.fetch_vectors.assert_not_called()
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_accepts_plain_string_ids_from_listing(self):
|
||||
"""SDK generations differ on what listing yields."""
|
||||
store, _, _ = self._store(
|
||||
[self._page([], None)],
|
||||
[{"vectors": {"a": {"values": [0.1], "metadata": {}}}}],
|
||||
)
|
||||
response = MagicMock()
|
||||
response.vectors = ["a"]
|
||||
response.pagination = MagicMock(next=None)
|
||||
store.index.index.list_paginated.side_effect = [response]
|
||||
|
||||
self.assertEqual([item["id"] for item in store.iter_all()], ["a"])
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_handles_missing_values_and_metadata(self):
|
||||
store, _, _ = self._store(
|
||||
[self._page(["a"], None)],
|
||||
[{"vectors": {"a": {"values": None, "metadata": None}}}],
|
||||
)
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
self.assertIsNone(item["vector"])
|
||||
self.assertEqual(item["metadata"], {})
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_raises_when_list_paginated_unavailable(self):
|
||||
store = PineconeStore()
|
||||
wrapper = MagicMock()
|
||||
wrapper.index = MagicMock(spec=["query", "fetch"])
|
||||
store.index = wrapper
|
||||
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_raises_when_index_not_initialized(self):
|
||||
"""Must fail loudly: an empty scan reads the same as an empty source."""
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(PineconeStore().iter_all())
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', False)
|
||||
def test_raises_when_pinecone_unavailable(self):
|
||||
store = PineconeStore()
|
||||
store.index = MagicMock()
|
||||
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
@patch('semantica.vector_store.pinecone_store.PINECONE_AVAILABLE', True)
|
||||
def test_propagates_listing_errors(self):
|
||||
store, _, raw_index = self._store([], [])
|
||||
raw_index.list_paginated.side_effect = RuntimeError("connection reset")
|
||||
|
||||
with self.assertRaises(RuntimeError):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print("DEBUG: Starting unittest.main()")
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Tests for QdrantStore.iter_all() cursor enumeration.
|
||||
|
||||
Qdrant is not installed in this environment, so these drive the real
|
||||
QdrantStore against a MagicMock standing in for the qdrant_client, following
|
||||
the pattern already used for qdrant in test_backend_metadata_filtering.py.
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from semantica.utils.exceptions import ProcessingError
|
||||
from semantica.vector_store.qdrant_store import QdrantStore
|
||||
|
||||
|
||||
def _record(point_id, payload=None, vector=None):
|
||||
"""Build a stand-in for a qdrant_client Record."""
|
||||
rec = MagicMock()
|
||||
rec.id = point_id
|
||||
rec.payload = payload
|
||||
rec.vector = vector
|
||||
return rec
|
||||
|
||||
|
||||
def _store_with_scroll(*pages):
|
||||
"""QdrantStore whose client.scroll() returns the given (records, cursor) pages."""
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.client.scroll.side_effect = list(pages)
|
||||
store.collection = MagicMock()
|
||||
store.collection.collection_name = "test_collection"
|
||||
return store
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_threads_cursor_across_pages():
|
||||
"""The next call must continue from the previous page's next_page_offset."""
|
||||
store = _store_with_scroll(
|
||||
([_record(1), _record(2)], "cursor-1"),
|
||||
([_record(3)], None),
|
||||
)
|
||||
|
||||
result = list(store.iter_all(batch_size=2))
|
||||
|
||||
assert [item["id"] for item in result] == ["1", "2", "3"]
|
||||
calls = store.client.scroll.call_args_list
|
||||
assert len(calls) == 2
|
||||
assert calls[0][1]["offset"] is None
|
||||
assert calls[0][1]["limit"] == 2
|
||||
assert calls[1][1]["offset"] == "cursor-1"
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_yields_final_page_that_reports_no_next_cursor():
|
||||
"""Qdrant can return records and a null cursor on the same page.
|
||||
|
||||
Those records must still be yielded. Treating a null cursor as "stop
|
||||
before this page" would silently drop the tail of every scan.
|
||||
"""
|
||||
store = _store_with_scroll(([_record(1), _record(2)], None))
|
||||
|
||||
result = list(store.iter_all(batch_size=10))
|
||||
|
||||
assert [item["id"] for item in result] == ["1", "2"]
|
||||
assert store.client.scroll.call_count == 1
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_converts_records_to_the_shared_result_shape():
|
||||
store = _store_with_scroll(
|
||||
([_record(7, payload={"tag": "x"}, vector=[0.1, 0.2, 0.3])], None),
|
||||
)
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["id"] == "7"
|
||||
assert item["metadata"] == {"tag": "x"}
|
||||
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2, 0.3]))
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_handles_missing_payload_and_vector():
|
||||
store = _store_with_scroll(([_record(1, payload=None, vector=None)], None))
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["metadata"] == {}
|
||||
assert item["vector"] is None
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_empty_collection_yields_nothing():
|
||||
store = _store_with_scroll(([], None))
|
||||
|
||||
assert list(store.iter_all()) == []
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_stops_on_empty_page_even_with_a_cursor():
|
||||
"""Defensive: an empty page ends the scan rather than looping forever."""
|
||||
store = _store_with_scroll(([], "cursor-that-never-clears"))
|
||||
|
||||
assert list(store.iter_all()) == []
|
||||
assert store.client.scroll.call_count == 1
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_raises_when_collection_not_initialized():
|
||||
"""Must fail loudly, not yield nothing.
|
||||
|
||||
An empty scan is indistinguishable from an empty source, which would let
|
||||
`store migrate` report success having copied nothing (issue #1083).
|
||||
"""
|
||||
store = QdrantStore()
|
||||
|
||||
with pytest.raises(ProcessingError, match="Collection not initialized"):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", False)
|
||||
def test_iter_all_raises_when_qdrant_unavailable():
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.collection = MagicMock()
|
||||
|
||||
with pytest.raises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_propagates_scroll_errors():
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.client.scroll.side_effect = RuntimeError("connection reset")
|
||||
store.collection = MagicMock()
|
||||
store.collection.collection_name = "test_collection"
|
||||
|
||||
with pytest.raises(RuntimeError, match="connection reset"):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_requests_payload_and_vectors():
|
||||
store = _store_with_scroll(([], None))
|
||||
|
||||
list(store.iter_all())
|
||||
|
||||
kwargs = store.client.scroll.call_args[1]
|
||||
assert kwargs["with_payload"] is True
|
||||
assert kwargs["with_vectors"] is True
|
||||
assert kwargs["collection_name"] == "test_collection"
|
||||
@@ -26,6 +26,7 @@ from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
|
||||
from semantica.utils.exceptions import ProcessingError
|
||||
from semantica.vector_store.vector_store import VectorStore, VectorManager
|
||||
|
||||
|
||||
@@ -138,6 +139,38 @@ class _NonScanningBackendStore:
|
||||
"""Fake persistent backend store without any scan capability."""
|
||||
|
||||
|
||||
class _IterAllBackendStore:
|
||||
"""Fake cursor-based backend store exposing iter_all() but not scan_vectors().
|
||||
|
||||
Mirrors qdrant/pinecone/milvus/weaviate, which cannot honour a positional
|
||||
offset and therefore expose native iteration instead.
|
||||
"""
|
||||
|
||||
def __init__(self, items):
|
||||
self._items = items
|
||||
self.batch_sizes = []
|
||||
|
||||
def iter_all(self, batch_size=500):
|
||||
self.batch_sizes.append(batch_size)
|
||||
for item in self._items:
|
||||
yield item
|
||||
|
||||
def scan_vectors(self, offset=0, limit=100):
|
||||
raise AssertionError("scan_vectors() must not be called when iter_all() exists")
|
||||
|
||||
|
||||
class _MisShapedIterAllBackendStore:
|
||||
"""Backend store whose ``iter_all`` attribute is not callable."""
|
||||
|
||||
iter_all = 42 # plain attribute, not a method
|
||||
|
||||
def __init__(self, items):
|
||||
self._items = items
|
||||
|
||||
def scan_vectors(self, offset=0, limit=100):
|
||||
return self._items[offset:offset + limit]
|
||||
|
||||
|
||||
class VectorStoreScanVectorsTests(unittest.TestCase):
|
||||
"""VectorStore.scan_vectors() / iter_vectors() backend-agnostic accessors."""
|
||||
|
||||
@@ -192,6 +225,81 @@ class VectorStoreScanVectorsTests(unittest.TestCase):
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), [])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VectorStore.iter_vectors() preference for a native iter_all()
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class VectorStoreIterAllDispatchTests(unittest.TestCase):
|
||||
"""iter_vectors() prefers a backend's native iter_all() when present.
|
||||
|
||||
Cursor-based backends cannot implement scan_vectors(offset, limit)
|
||||
honestly, so they expose iter_all() instead and iter_vectors() routes to
|
||||
it rather than walking offsets.
|
||||
"""
|
||||
|
||||
def _persistent_store(self, backend_store, backend_name="qdrant"):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.backend = backend_name
|
||||
store._backend_store = backend_store
|
||||
return store
|
||||
|
||||
def test_iter_vectors_uses_iter_all_when_available(self):
|
||||
items = [
|
||||
{"id": "a", "vector": None, "metadata": {"n": 1}},
|
||||
{"id": "b", "vector": None, "metadata": {"n": 2}},
|
||||
]
|
||||
backend = _IterAllBackendStore(items)
|
||||
store = self._persistent_store(backend)
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=7)), items)
|
||||
|
||||
def test_iter_vectors_forwards_batch_size_to_iter_all(self):
|
||||
backend = _IterAllBackendStore([])
|
||||
store = self._persistent_store(backend)
|
||||
|
||||
list(store.iter_vectors(batch_size=32))
|
||||
|
||||
self.assertEqual(backend.batch_sizes, [32])
|
||||
|
||||
def test_iter_vectors_falls_back_to_scan_vectors_without_iter_all(self):
|
||||
items = [{"id": "a", "vector": None, "metadata": {}}]
|
||||
store = self._persistent_store(_ScanningBackendStore(items))
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
|
||||
|
||||
def test_iter_vectors_falls_back_when_iter_all_not_callable(self):
|
||||
# A mis-shaped adapter exposing a non-callable ``iter_all`` must not be
|
||||
# invoked; the offset path still has to work. Mirrors the count()
|
||||
# precedent in _MisShapedBackendStore.
|
||||
items = [{"id": "a", "vector": None, "metadata": {}}]
|
||||
store = self._persistent_store(_MisShapedIterAllBackendStore(items))
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
|
||||
|
||||
def test_iter_vectors_inmemory_ignores_iter_all(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.store_vectors([np.array([1.0, 0.0])], [{"type": "a"}])
|
||||
store._backend_store = _IterAllBackendStore([{"id": "wrong"}])
|
||||
|
||||
collected = list(store.iter_vectors(batch_size=2))
|
||||
|
||||
self.assertEqual([item["metadata"] for item in collected], [{"type": "a"}])
|
||||
|
||||
def test_iter_vectors_propagates_iter_all_errors(self):
|
||||
# A scan that silently yields nothing is indistinguishable from an
|
||||
# empty source, which would let `store migrate` report success having
|
||||
# copied nothing (issue #1083).
|
||||
class _FailingIterAll:
|
||||
def iter_all(self, batch_size=500):
|
||||
raise ProcessingError("backend unreachable")
|
||||
yield # pragma: no cover - makes this a generator
|
||||
|
||||
store = self._persistent_store(_FailingIterAll())
|
||||
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(store.iter_vectors(batch_size=2))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VectorManager tests — inmemory backend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user