mirror of
https://github.com/semantica-agi/semantica.git
synced 2026-09-01 04:00:28 +00:00
Compare commits
8
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bd1ba24b24 | ||
|
|
e8ff36f088 | ||
|
|
ec9e63e16f | ||
|
|
274d5d1195 | ||
|
|
fa87a1a9be | ||
|
|
08c78bfb40 | ||
|
|
56b174781f | ||
|
|
dfda4c561a |
@@ -0,0 +1,56 @@
|
||||
name: 'Setup Semantica'
|
||||
description: 'Install Python, cache pip, and install the semantica package into a workflow'
|
||||
author: 'Semantica'
|
||||
|
||||
inputs:
|
||||
python-version:
|
||||
description: 'Python version to set up'
|
||||
required: false
|
||||
default: '3.11'
|
||||
version:
|
||||
description: 'Version constraint to append to the pip spec, e.g. "==0.6.7" or ">=0.6,<0.7". Leave empty for the latest release.'
|
||||
required: false
|
||||
default: ''
|
||||
extras:
|
||||
description: 'Comma-separated extras to install, e.g. "explorer,all"'
|
||||
required: false
|
||||
default: ''
|
||||
cache:
|
||||
description: 'Pip cache mode passed straight to actions/setup-python ("pip" to enable). Left empty (disabled) by default because this action is meant to run standalone in any caller repo, and actions/setup-python errors out if it cannot find a requirements.txt/pyproject.toml/setup.py/poetry.lock to key the cache on. Opt in only when the caller repo has one of those files.'
|
||||
required: false
|
||||
default: ''
|
||||
|
||||
outputs:
|
||||
version:
|
||||
description: 'The installed semantica version'
|
||||
value: ${{ steps.verify.outputs.version }}
|
||||
|
||||
runs:
|
||||
using: 'composite'
|
||||
steps:
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: ${{ inputs.python-version }}
|
||||
cache: ${{ inputs.cache }}
|
||||
|
||||
- name: Install semantica
|
||||
shell: bash
|
||||
env:
|
||||
SEMANTICA_EXTRAS: ${{ inputs.extras }}
|
||||
SEMANTICA_VERSION: ${{ inputs.version }}
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
if [ -n "$SEMANTICA_EXTRAS" ]; then
|
||||
spec="semantica[$SEMANTICA_EXTRAS]$SEMANTICA_VERSION"
|
||||
else
|
||||
spec="semantica$SEMANTICA_VERSION"
|
||||
fi
|
||||
python -m pip install -- "$spec"
|
||||
|
||||
- name: Verify install
|
||||
id: verify
|
||||
shell: bash
|
||||
run: |
|
||||
VERSION=$(python -c "import semantica; print(semantica.__version__)")
|
||||
echo "Installed semantica $VERSION"
|
||||
echo "version=$VERSION" >> "$GITHUB_OUTPUT"
|
||||
@@ -101,6 +101,29 @@ updates:
|
||||
allow:
|
||||
- dependency-type: "production"
|
||||
|
||||
# Explorer frontend (npm)
|
||||
- package-ecosystem: "npm"
|
||||
directory: "/explorer"
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
day: "monday"
|
||||
time: "03:30" # 3:30 AM UTC (9:00 AM IST)
|
||||
open-pull-requests-limit: 10
|
||||
reviewers:
|
||||
- "KaifAhmad1"
|
||||
assignees:
|
||||
- "KaifAhmad1"
|
||||
commit-message:
|
||||
prefix: "security"
|
||||
include: "scope"
|
||||
labels:
|
||||
- "dependencies"
|
||||
- "javascript"
|
||||
- "security"
|
||||
allow:
|
||||
- dependency-type: "production"
|
||||
- dependency-type: "development"
|
||||
|
||||
# Docker dependencies (if you use Docker)
|
||||
- package-ecosystem: "docker"
|
||||
directory: "/"
|
||||
|
||||
@@ -10,13 +10,15 @@ on:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
security-events: write
|
||||
actions: read
|
||||
|
||||
jobs:
|
||||
analyze:
|
||||
name: Analyze Python
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
security-events: write # for github/codeql-action/upload-sarif below
|
||||
actions: read # for github/codeql-action/init's CodeQL bundle cache lookup
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
name: Container Security Scan
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
# Mirrors .dockerignore's opt-in list exactly - anything not listed there
|
||||
# can't reach the build context, so it can't change the built image.
|
||||
paths:
|
||||
- 'Dockerfile'
|
||||
- '.dockerignore'
|
||||
- 'pyproject.toml'
|
||||
- 'README.md'
|
||||
- 'LICENSE'
|
||||
- 'MANIFEST.in'
|
||||
- 'semantica/**'
|
||||
- 'integrations/**'
|
||||
- 'explorer/**'
|
||||
- '.github/workflows/container-scan.yml'
|
||||
schedule:
|
||||
- cron: '30 2 * * 1' # weekly, catches new CVEs published against the base image between pushes
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
scan:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
security-events: write # for github/codeql-action/upload-sarif below
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- name: Build image
|
||||
run: docker build -t semantica:scan .
|
||||
|
||||
# Run Trivy as a digest-pinned image rather than the aquasecurity/trivy-action
|
||||
# marketplace wrapper: the aquasecurity GitHub org has an IP allow list on its
|
||||
# API that 403s verify-action-pins.sh's live tag->SHA check from Actions-runner
|
||||
# IPs, and this repo already treats Trivy's action pin as a known past target
|
||||
# for tag-repointing (see the LiteLLM/Trivy 2026 incident note above). Pulling
|
||||
# by sha256 digest from Docker Hub is immutable and verifiable independently of
|
||||
# GitHub's API, so it sidesteps both problems at once instead of carving a skip
|
||||
# exception into the pin verifier for an org already flagged as higher-risk.
|
||||
#
|
||||
# Report-only for now: this is Trivy's first run against this image, so we
|
||||
# don't yet know the CRITICAL/HIGH baseline. Findings still land in the
|
||||
# Security tab either way. Once triaged, add `--exit-code 1` (like
|
||||
# Safety/Bandit-HIGH in security-scan.yml) to make it a hard gate.
|
||||
- name: Scan image for vulnerabilities (Trivy)
|
||||
run: |
|
||||
docker run --rm \
|
||||
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||
-v "$PWD:/output" \
|
||||
aquasec/trivy@sha256:62b1e65e8869bc4b4c6aa4fa2b21595256c7c2f6018a9d9ad61caf87187c1969 \
|
||||
image --format sarif --output /output/trivy-results.sarif \
|
||||
--severity CRITICAL,HIGH --ignore-unfixed semantica:scan
|
||||
|
||||
- name: Upload Trivy SARIF
|
||||
if: always()
|
||||
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-container
|
||||
|
||||
- name: Generate SBOM (Syft)
|
||||
if: always()
|
||||
uses: anchore/sbom-action@3ad7283483fc7af8ff2b4ea19663c2d5ca935e26 # v0.24.2
|
||||
with:
|
||||
image: semantica:scan
|
||||
format: spdx-json
|
||||
output-file: semantica-sbom.spdx.json
|
||||
@@ -28,12 +28,14 @@ on:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
security-events: write
|
||||
|
||||
jobs:
|
||||
MSDO:
|
||||
# currently only windows-latest is supported
|
||||
runs-on: windows-latest
|
||||
permissions:
|
||||
contents: read
|
||||
security-events: write # for github/codeql-action/upload-sarif below
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
name: Install Matrix
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: '0 6 * * 1' # weekly, catches upstream dependency breakage between releases
|
||||
workflow_run:
|
||||
# The Release workflow publishes the GitHub release *before* it uploads to
|
||||
# PyPI (see release.yml), so triggering on `release: published` would race
|
||||
# the PyPI upload and could pass by silently installing the prior version.
|
||||
# workflow_run fires only after the whole Release workflow - including the
|
||||
# PyPI publish step - has finished.
|
||||
workflows: ['Release']
|
||||
types: [completed]
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
verify-install:
|
||||
if: github.event_name != 'workflow_run' || github.event.workflow_run.conclusion == 'success'
|
||||
name: pip install semantica (${{ matrix.os }}, py${{ matrix.python-version }})
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
python-version: ['3.9', '3.10', '3.11', '3.12']
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- name: Pin expected version for release-triggered runs
|
||||
id: expected-version
|
||||
if: github.event_name == 'workflow_run'
|
||||
shell: bash
|
||||
env:
|
||||
EXPECTED_TAG: ${{ github.event.workflow_run.head_branch }}
|
||||
run: |
|
||||
expected="${EXPECTED_TAG#v}"
|
||||
if [ -z "$expected" ]; then
|
||||
echo "::error::Could not determine a release tag from the triggering workflow run (head_branch was empty)."
|
||||
exit 1
|
||||
fi
|
||||
echo "constraint===$expected" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- id: setup-semantica
|
||||
uses: ./.github/actions/setup-semantica
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
cache: 'pip'
|
||||
version: ${{ steps.expected-version.outputs.constraint }}
|
||||
|
||||
- name: Smoke test import
|
||||
shell: bash
|
||||
run: |
|
||||
python -c "
|
||||
import semantica
|
||||
print('semantica', semantica.__version__, 'installed and importable')
|
||||
"
|
||||
@@ -63,11 +63,15 @@ jobs:
|
||||
|
||||
print("Explorer frontend is packaged")
|
||||
PY
|
||||
- name: Verify PyPI long-description will render
|
||||
run: |
|
||||
pip install twine==7.0.0
|
||||
twine check dist/*
|
||||
- name: Attest build provenance
|
||||
uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4
|
||||
with:
|
||||
subject-path: 'dist/*'
|
||||
- uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v3
|
||||
- uses: softprops/action-gh-release@efb35369e0ad2afab669f228072c1b0d510eae64 # v3.0.3
|
||||
with:
|
||||
files: dist/*
|
||||
- uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # release/v1
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
name: Scorecard supply-chain security
|
||||
|
||||
permissions: read-all
|
||||
|
||||
on:
|
||||
branch_protection_rule:
|
||||
schedule:
|
||||
- cron: '30 1 * * 6' # weekly
|
||||
push:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
analysis:
|
||||
name: Scorecard analysis
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
security-events: write # to upload SARIF results
|
||||
id-token: write # to publish results and get a badge
|
||||
contents: read
|
||||
actions: read # to detect GitHub Actions workflows
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Run analysis
|
||||
uses: ossf/scorecard-action@2d1146689b8cda280b9bc96326124645441f03bc # v2.4.4
|
||||
with:
|
||||
results_file: results.sarif
|
||||
results_format: sarif
|
||||
publish_results: true
|
||||
|
||||
- name: Upload artifact
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
|
||||
with:
|
||||
name: SARIF file
|
||||
path: results.sarif
|
||||
retention-days: 5
|
||||
|
||||
- name: Upload to code-scanning
|
||||
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4
|
||||
with:
|
||||
sarif_file: results.sarif
|
||||
@@ -0,0 +1,20 @@
|
||||
cff-version: 1.2.0
|
||||
message: "If you use this software, please cite it as below."
|
||||
title: "Semantica: Graph-Native Infrastructure for Context and Accountable AI Systems"
|
||||
type: software
|
||||
authors:
|
||||
- name: "Semantica"
|
||||
repository-code: "https://github.com/semantica-agi/semantica"
|
||||
url: "https://getsemantica.ai"
|
||||
license: MIT
|
||||
version: 0.6.7
|
||||
date-released: 2026-08-28
|
||||
keywords:
|
||||
- knowledge-graph
|
||||
- context-graph
|
||||
- ai-agents
|
||||
- llm
|
||||
- decision-intelligence
|
||||
- provenance
|
||||
- explainability
|
||||
- graph-rag
|
||||
@@ -0,0 +1,131 @@
|
||||
# Growth & Distribution Playbook
|
||||
|
||||
North star: **10,000 developers who actually use Semantica in real projects**, not a raw PyPI download number. Downloads are a lagging indicator of distribution, not a target to optimize directly.
|
||||
|
||||
```
|
||||
GitHub stars → Website visitors → PyPI installs → Weekly active users → Production deployments → Enterprise customers
|
||||
```
|
||||
The last two matter far more than the download count.
|
||||
|
||||
## Guardrails — do not do this
|
||||
|
||||
- No fake/looping CI jobs that repeatedly `pip install semantica` purely to inflate the graph. It's detectable, it produces zero real users, and it damages credibility with anyone doing diligence (investors, enterprise buyers, security reviewers).
|
||||
- No package-splitting purely to multiply install counts — only split into `semantica-*` packages when there's a real architectural reason.
|
||||
- No meaningless Docker pulls or notebook launches with no real content behind them.
|
||||
- Every item below should get someone from "installed it" to "used it for something real." If a channel can't do that, it's not worth building.
|
||||
|
||||
## 30-day priority sprint
|
||||
|
||||
Ordered by leverage-to-effort ratio; do these first.
|
||||
|
||||
| # | Initiative | Target |
|
||||
| - | ---------- | ------ |
|
||||
| 1 | ✅ GitHub Actions example + reusable `setup-semantica` composite action + install-matrix badge | done |
|
||||
| 2 | Google Colab notebooks | 10 |
|
||||
| 3 | Docker images (RAG, Graph, Agent, API) | 4-5 |
|
||||
| 4 | Hugging Face Spaces demos | 3-4 |
|
||||
| 5 | LangChain integration + example | 1 |
|
||||
| 6 | LlamaIndex integration + example | 1 |
|
||||
| 7 | Vector/graph DB integrations (Qdrant, Weaviate, Neo4j) | 3 |
|
||||
| 8 | MCP server + example | 1 (already have `mcp/` — package as a distributable example) |
|
||||
| 9 | Production-quality starter repos (FastAPI, Streamlit, Gradio) | 3 |
|
||||
| 10 | `awesome-rag` / `awesome-llm` / `awesome-knowledge-graph` list submissions | 3+ PRs |
|
||||
|
||||
Push everything through: GitHub → Discord (`sV34vps5hH`) → X (`@BuildSemantica`) → GitHub Discussions → Reddit → Hacker News → relevant newsletters.
|
||||
|
||||
## Full channel checklist
|
||||
|
||||
### CI/CD (highest-intent distribution — installs tied to real pipelines)
|
||||
|
||||
- [x] GitHub Actions example in `examples/ci/github-actions.yml`
|
||||
- [x] Reusable composite GitHub Action — [`.github/actions/setup-semantica`](.github/actions/setup-semantica/action.yml), modeled on `actions/setup-python`; usable by any repo as `uses: semantica-agi/semantica/.github/actions/setup-semantica@main`
|
||||
- [x] "pip install" status badge in the README, backed by [`.github/workflows/install-matrix.yml`](.github/workflows/install-matrix.yml) — verifies the *published* package installs cleanly on Ubuntu/macOS/Windows across Python 3.9-3.12, weekly + on every release
|
||||
- [x] GitLab CI template — `examples/ci/gitlab-ci.yml`
|
||||
- [x] CircleCI template — `examples/ci/circleci-config.yml`
|
||||
- [ ] Jenkins, Azure DevOps, Bitbucket Pipelines, Buildkite, Travis CI equivalents
|
||||
|
||||
### Release pipeline hardening (already had Trusted Publishing/OIDC + SLSA attestation — this rounds it out to match top-tier OSS release practice)
|
||||
|
||||
- [x] `twine check` gate in `.github/workflows/release.yml` before publish — catches a broken PyPI long-description render before it goes live instead of after (a malformed README on the live PyPI page is a silent conversion killer)
|
||||
- [x] `CITATION.cff` (see Academic & research below)
|
||||
- [x] OpenSSF Scorecard (see Discoverability below)
|
||||
- [ ] Considered and deliberately skipped: Release Drafter / auto-generated changelogs — this repo hand-curates `CHANGELOG.md` with far more detail (PR numbers, contributors, phase-1 limitations) than a bot would produce. Don't introduce this without checking with maintainers first.
|
||||
- [ ] Renovate / Dependabot config templates that auto-bump the `semantica` version in downstream repos — real recurring CI runs on real adopters
|
||||
- [ ] Nightly scheduled workflow template that tests a downstream project against `semantica@latest`
|
||||
|
||||
### Containers & dev environments
|
||||
|
||||
- [ ] Official Docker images: RAG, Graph, Agent, API, `+Postgres`, `+Neo4j`, `+Qdrant`
|
||||
- [ ] `docker-compose` examples (repo already has `docker-compose.dev.yml` / `docker-compose.yml` as a base)
|
||||
- [ ] `.devcontainer/devcontainer.json` for one-click "Reopen in Container"
|
||||
- [ ] GitHub Codespaces-ready config
|
||||
- [ ] Gitpod config
|
||||
- [ ] "Use this template" GitHub repo button so new projects start with `semantica` in `requirements.txt`
|
||||
|
||||
### Notebooks & hosted demos
|
||||
|
||||
- [ ] 10-20 Google Colab notebooks (Graph RAG, agent memory, entity resolution, semantic search, document intelligence)
|
||||
- [ ] Kaggle Notebooks/Kernels
|
||||
- [ ] Binder / mybinder.org config for instant repo launch
|
||||
- [ ] SageMaker Studio Lab / Databricks Community Edition / Paperspace Gradient examples
|
||||
- [ ] Hugging Face Spaces (Streamlit/Gradio) demos with `semantica` in `requirements.txt`
|
||||
- [ ] Public hosted playground (source on GitHub, install visible)
|
||||
|
||||
### Framework & data-store integrations
|
||||
|
||||
- [x] LangChain integration — `integrations/langchain/` (`SemanticaRetriever`, `SemanticaVectorStore`, `SemanticaKGTool`/`SemanticaDecisionTool`), `pip install semantica[langchain]`, shipped in 0.6.7
|
||||
- [ ] LlamaIndex integration + example
|
||||
- [ ] LangGraph example
|
||||
- [ ] Neo4j integration/example (docs already list it as a supported graph store — turn into a runnable example repo)
|
||||
- [ ] Vector DB examples: Qdrant, Weaviate, Milvus, Pinecone, Chroma, FAISS, pgvector, OpenSearch/Elasticsearch (FAISS/Pinecone/Weaviate/Qdrant/Milvus/PgVector already supported per `docs/community-projects.md` — package each as a standalone example)
|
||||
- [ ] LLM provider quickstarts: OpenAI, Anthropic, Gemini, Groq, Ollama, HuggingFace, DeepSeek, LiteLLM (already-supported providers per docs — each gets its own copy-paste quickstart)
|
||||
- [ ] CrewAI / Agno integration examples (already documented under `docs/integrations/`) — promote as standalone repos, not just docs pages
|
||||
|
||||
### Package managers & installers
|
||||
|
||||
- [ ] conda-forge feedstock
|
||||
- [ ] Homebrew formula for the CLI
|
||||
- [ ] Nix/nixpkgs packaging
|
||||
- [ ] Chocolatey / Scoop (Windows)
|
||||
- [ ] Document `uv add semantica` and `poetry add semantica` explicitly alongside `pip install`
|
||||
|
||||
### Downstream packages & CLI
|
||||
|
||||
- [ ] Genuinely useful `semantica-*` packages only where warranted (e.g. `semantica-rag`, `semantica-connectors`) — each pulls `semantica` as a real dependency
|
||||
- [ ] Make sure `semantica init / ingest / index / query / serve` CLI flows are the default onboarding path in every tutorial
|
||||
- [ ] VS Code extension wrapping the CLI (scaffold + run commands from the command palette)
|
||||
- [ ] JetBrains plugin equivalent
|
||||
|
||||
### Templates & starters
|
||||
|
||||
- [ ] Cookiecutter templates: `cookiecutter-semantic-rag`, `cookiecutter-ai-agent`, `cookiecutter-enterprise-rag`
|
||||
- [ ] Starter repos: FastAPI, Streamlit, Gradio, Next.js frontend + Semantica backend
|
||||
- [ ] Cloud deploy templates: AWS, GCP, Azure, Modal, Railway, Render, Fly.io (repo already has `deploy/azure`, `deploy/gcp`, `deploy/fly`, `deploy/railway`, `deploy/render`, `deploy/kubernetes`, `deploy/helm` — link these prominently from the README/quickstart, they're already-built distribution surface)
|
||||
- [ ] Terraform / Pulumi / Helm modules published to their respective registries
|
||||
|
||||
### Discoverability & curation
|
||||
|
||||
- [ ] Submit to `awesome-rag`, `awesome-llm`, `awesome-knowledge-graph`, `awesome-python`
|
||||
- [ ] Pitch newsletters with engaged Python/AI audiences (Python Weekly, Import AI, TLDR AI, etc.)
|
||||
- [x] PyPI trove classifiers/keywords and `project.urls` (Homepage/Docs/Repository/Changelog/Bug Tracker) — already complete in `pyproject.toml`
|
||||
- [ ] Get listed on Papers With Code for any retrieval/graph-RAG benchmark work
|
||||
- [x] [OpenSSF Scorecard](https://scorecard.dev/viewer/?uri=github.com/semantica-agi/semantica) badge + weekly workflow (`.github/workflows/scorecard.yml`) — a concrete trust signal security/procurement teams check before greenlighting adoption, which gates real (non-CI-bot) install growth at enterprises
|
||||
|
||||
### Academic & research
|
||||
|
||||
- [x] `CITATION.cff` at repo root — enables GitHub's native "Cite this repository" button, feeds Google Scholar/academic tooling; complements `docs/citation.md` (still needs a real Zenodo DOI to replace the `XXXXXXX` placeholder in both places once one is minted)
|
||||
- [ ] arXiv paper if there's real architectural novelty to describe
|
||||
- [ ] Zenodo DOI for citability (`docs/citation.md` already exists — make sure it points to a real DOI)
|
||||
- [ ] Workshop/tutorial sessions at PyData/ODSC-style events with hands-on install steps
|
||||
- [ ] University course material / bootcamp adoption outreach
|
||||
|
||||
### Content
|
||||
|
||||
- [ ] Reproducible benchmark repos (Graph RAG vs vector RAG, retrieval@k, enterprise-scale retrieval) with `pip install semantica && python benchmark.py`
|
||||
- [ ] 20-30 real-world example applications (RAG, enterprise document intelligence, financial entity graphs, code knowledge graphs, research discovery, agent memory)
|
||||
- [ ] Blog/tutorial posts on Dev.to, Medium, personal blogs — always with runnable code, not just prose
|
||||
- [ ] Contribute integrations/PRs to other projects building RAG/agents/knowledge graphs — "I implemented Semantica support" beats "please use Semantica"
|
||||
|
||||
## Tracking
|
||||
|
||||
Don't just watch the raw PyPI number — use download analytics (e.g. PePy) to separate CI/bot traffic from real installs, and track the funnel above end-to-end where possible (stars → site visits → installs → weekly actives).
|
||||
@@ -26,7 +26,7 @@
|
||||
|
||||
#### Built for High-Stakes, Regulated Domains
|
||||
|
||||
[](https://github.com/semantica-agi/semantica) [](https://github.com/semantica-agi/semantica/network/members) [](https://github.com/semantica-agi/semantica/graphs/contributors) [](https://pypi.org/project/semantica/) [](https://pepy.tech/project/semantica) [](https://www.python.org/) [](https://opensource.org/licenses/MIT) [](https://github.com/semantica-agi/semantica/actions) [](https://deepwiki.com/semantica-agi/semantica)
|
||||
[](https://github.com/semantica-agi/semantica) [](https://github.com/semantica-agi/semantica/network/members) [](https://github.com/semantica-agi/semantica/graphs/contributors) [](https://pypi.org/project/semantica/) [](https://pepy.tech/project/semantica) [](https://www.python.org/) [](https://opensource.org/licenses/MIT) [](https://github.com/semantica-agi/semantica/actions) [](https://github.com/semantica-agi/semantica/actions/workflows/install-matrix.yml) [](https://scorecard.dev/viewer/?uri=github.com/semantica-agi/semantica) [](https://deepwiki.com/semantica-agi/semantica)
|
||||
|
||||
[](https://getsemantica.ai/) [](https://docs.getsemantica.ai/) [](https://discord.gg/sV34vps5hH) [](https://x.com/BuildSemantica) [](https://www.youtube.com/watch?v=QfnNZg4-dZA) [](CHANGELOG.md)
|
||||
|
||||
@@ -1534,6 +1534,20 @@ git clone https://github.com/semantica-agi/semantica.git
|
||||
cd semantica && pip install -e ".[dev]" && pytest tests/
|
||||
```
|
||||
|
||||
### CI & Deployment
|
||||
|
||||
Wiring `semantica` into your own CI is a two-minute job. On GitHub Actions, use the reusable composite action:
|
||||
|
||||
```yaml
|
||||
- uses: semantica-agi/semantica/.github/actions/setup-semantica@main
|
||||
with:
|
||||
python-version: '3.11'
|
||||
```
|
||||
|
||||
Copy-paste starting templates for GitHub Actions, GitLab CI, and CircleCI live in [examples/ci/](examples/ci/). The published package itself is verified installable across Ubuntu/macOS/Windows and Python 3.9-3.12 every week by the [Install Matrix workflow](.github/workflows/install-matrix.yml).
|
||||
|
||||
Ready-made deployment configs for AWS, GCP, Azure, Fly.io, Railway, Render, Kubernetes, and Helm are in [deploy/](deploy/).
|
||||
|
||||
---
|
||||
|
||||
## Enterprise
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
# CI templates
|
||||
|
||||
Copy-paste starting points for wiring `semantica` into your own project's CI. Each file is a
|
||||
complete, working config — rename it into your project (see the comment at the top of each file
|
||||
for the target path) and swap the smoke-test / test step for whatever your project does with
|
||||
Semantica. Each template installs `semantica` unconditionally and your own project's dependencies
|
||||
only if a `requirements.txt` is present; if your project uses `pyproject.toml`, Poetry, or Pipenv
|
||||
instead, adjust the marked install line (each file calls it out inline).
|
||||
|
||||
| File | Target path in your repo |
|
||||
| ---- | ------------------------- |
|
||||
| [`github-actions.yml`](github-actions.yml) | `.github/workflows/semantica.yml` |
|
||||
| [`gitlab-ci.yml`](gitlab-ci.yml) | `.gitlab-ci.yml` |
|
||||
| [`circleci-config.yml`](circleci-config.yml) | `.circleci/config.yml` |
|
||||
|
||||
If your own project is hosted on GitHub, you can skip the setup boilerplate entirely and use
|
||||
Semantica's reusable composite action instead:
|
||||
|
||||
```yaml
|
||||
- uses: semantica-agi/semantica/.github/actions/setup-semantica@main
|
||||
with:
|
||||
python-version: '3.11'
|
||||
# extras: 'explorer,all' # optional
|
||||
# version: '==0.6.7' # optional, pin an exact release
|
||||
# cache: 'pip' # optional, only if your repo has a requirements.txt/pyproject.toml/etc.
|
||||
```
|
||||
|
||||
`@main` always tracks this repo's default branch, which is convenient but — like any mutable
|
||||
ref — can change out from under you between runs. For production CI, pin it to a commit SHA
|
||||
instead (find one via `git rev-parse` against a tagged release, or the commit history for
|
||||
[`.github/actions/setup-semantica/`](../../.github/actions/setup-semantica/)) and update the pin
|
||||
deliberately when you want to pick up changes, the same way this repo's own workflows are pinned
|
||||
(see [`verify-action-pins.yml`](../../.github/workflows/verify-action-pins.yml)).
|
||||
|
||||
It installs Python, installs `semantica`, and verifies the import (pip caching is opt-in via `cache: 'pip'`, since not every caller repo has a requirements file to key the cache on) — see
|
||||
[`.github/actions/setup-semantica/action.yml`](../../.github/actions/setup-semantica/action.yml).
|
||||
@@ -0,0 +1,40 @@
|
||||
# Drop this in as .circleci/config.yml in your own project.
|
||||
version: 2.1
|
||||
|
||||
jobs:
|
||||
test:
|
||||
docker:
|
||||
- image: cimg/python:3.11
|
||||
steps:
|
||||
- checkout
|
||||
# A content-hashed cache key (e.g. `{{ checksum "requirements.txt" }}`)
|
||||
# is more precise but breaks if that exact file doesn't exist in your
|
||||
# project - swap in one matched to however you declare dependencies
|
||||
# once you've adjusted the install step below.
|
||||
- restore_cache:
|
||||
keys:
|
||||
- pip-cache-v1
|
||||
- run:
|
||||
name: Install dependencies
|
||||
command: |
|
||||
pip install --upgrade pip
|
||||
pip install semantica
|
||||
# Install your own project's dependencies however your project
|
||||
# declares them - adjust this to match, e.g. `pip install -e .`
|
||||
# for pyproject.toml / setup.cfg, or `poetry install`.
|
||||
if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
||||
- save_cache:
|
||||
key: pip-cache-v1
|
||||
paths:
|
||||
- ~/.cache/pip
|
||||
- run:
|
||||
name: Smoke test
|
||||
command: python -c "import semantica; print('semantica', semantica.__version__)"
|
||||
- run:
|
||||
name: Run tests
|
||||
command: pytest
|
||||
|
||||
workflows:
|
||||
test:
|
||||
jobs:
|
||||
- test
|
||||
@@ -0,0 +1,44 @@
|
||||
# Drop this in as .github/workflows/semantica.yml in your own project.
|
||||
#
|
||||
# Installs Semantica and runs a smoke import + your test suite. Swap the
|
||||
# smoke-test step for whatever your project actually does with Semantica
|
||||
# (build a context graph, run an ingest pipeline, etc.).
|
||||
#
|
||||
# Third-party actions below are pinned to a commit SHA rather than a mutable
|
||||
# tag - a moved tag can silently swap in different code. Update the pin (and
|
||||
# the trailing "# vX" comment) deliberately when you want a newer version;
|
||||
# see semantica-agi/semantica's own .github/workflows/verify-action-pins.yml
|
||||
# for one way to keep pins honest automatically.
|
||||
name: Semantica
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
||||
|
||||
- uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7
|
||||
with:
|
||||
python-version: '3.11'
|
||||
cache: 'pip'
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install semantica
|
||||
# Install your own project's dependencies however your project
|
||||
# declares them - adjust this to match. Examples:
|
||||
# pip install -r requirements.txt
|
||||
# pip install -e . # pyproject.toml / setup.cfg
|
||||
# pip install -e ".[dev]"
|
||||
# poetry install
|
||||
if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
||||
|
||||
- name: Run tests
|
||||
run: pytest
|
||||
@@ -0,0 +1,20 @@
|
||||
# Drop this in as .gitlab-ci.yml in your own project.
|
||||
semantica-test:
|
||||
image: python:3.11-slim
|
||||
cache:
|
||||
paths:
|
||||
- .cache/pip
|
||||
variables:
|
||||
PIP_CACHE_DIR: "$CI_PROJECT_DIR/.cache/pip"
|
||||
script:
|
||||
- pip install --upgrade pip
|
||||
- pip install semantica
|
||||
# Install your own project's dependencies however your project declares
|
||||
# them - adjust this to match, e.g. `pip install -e .` for pyproject.toml
|
||||
# / setup.cfg, or `poetry install`.
|
||||
- if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
||||
- python -c "import semantica; print('semantica', semantica.__version__)"
|
||||
- pytest
|
||||
rules:
|
||||
- if: '$CI_PIPELINE_SOURCE == "merge_request_event"'
|
||||
- if: '$CI_COMMIT_BRANCH == "main"'
|
||||
Generated
+6
-6
@@ -2083,9 +2083,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/brace-expansion": {
|
||||
"version": "5.0.8",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.8.tgz",
|
||||
"integrity": "sha512-JZyDyq3D4AUifKTPOB7DELf6XsB3WdPuNxCtob1vFXPsSXhdAiHBWJ/tJ8HAc9aH84BK+5JFZLNkJKx3G9kzQg==",
|
||||
"version": "5.0.9",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.9.tgz",
|
||||
"integrity": "sha512-ScQ4IuvIEF1TMlP7Zt+vjJ//9zlPb2SDcxWxM3bk8s6t6GGdJ7KO1dCcTidOPJKePW30LE/2cT7wCyPho9/Wxg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -4250,9 +4250,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/nanoid": {
|
||||
"version": "3.3.16",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.16.tgz",
|
||||
"integrity": "sha512-bzlKTyNJ7+LdGIIwy8ijFpIqEQIvafahV7eYykJ8Cvh42EdJeODoJ6gUJXpQJvej1BddH8OqTXZNE/KfbWAu8Q==",
|
||||
"version": "3.3.18",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.18.tgz",
|
||||
"integrity": "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
|
||||
+114
-9
@@ -3714,19 +3714,61 @@ def store_stats(cli_ctx: CLIContext, backend: str, fmt: str, local_json: bool) -
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
|
||||
_MIGRATE_SUPPORTED_BACKENDS = {"faiss", "sqlite", "pgvector"}
|
||||
_MIGRATE_BATCH_SIZE = 500
|
||||
|
||||
|
||||
def _migrate_backend_config(vs_cfg: Dict[str, Any], backend: str) -> Dict[str, Any]:
|
||||
"""Resolve per-backend config out of the vector_store config section.
|
||||
|
||||
Supports both a per-backend nested shape (``vector_store.faiss.dimension``)
|
||||
and the common flat single-backend shape (``vector_store.backend`` +
|
||||
sibling keys), since either can appear depending on how many backends a
|
||||
user has configured.
|
||||
"""
|
||||
nested = vs_cfg.get(backend)
|
||||
if isinstance(nested, dict):
|
||||
return dict(nested)
|
||||
if vs_cfg.get("backend") == backend:
|
||||
return {k: v for k, v in vs_cfg.items() if k != "backend"}
|
||||
return {}
|
||||
|
||||
|
||||
def _require_faiss_index_path(cfg: Dict[str, Any], role: str) -> str:
|
||||
"""FAISS has no server to hold state between commands: a fresh FAISSStore
|
||||
starts empty and nothing outside the process persists it, so migration
|
||||
needs an explicit on-disk index to read from or write to."""
|
||||
index_path = cfg.get("index_path")
|
||||
if not index_path:
|
||||
raise click.ClickException(
|
||||
f"faiss as migration {role} requires 'index_path' in the vector_store "
|
||||
f"config (vector_store.faiss.index_path or vector_store.index_path "
|
||||
f"when faiss is the configured backend)."
|
||||
)
|
||||
return index_path
|
||||
|
||||
|
||||
@store.command("migrate")
|
||||
@click.option("--from", "from_backend", required=True)
|
||||
@click.option("--to", "to_backend", required=True)
|
||||
@click.option("--namespace", default=None)
|
||||
@click.option("--dry-run", "local_dry", is_flag=True, default=False)
|
||||
@click.option("--json", "local_json", is_flag=True, default=False)
|
||||
@click.pass_obj
|
||||
def store_migrate(cli_ctx: CLIContext, from_backend: str, to_backend: str,
|
||||
namespace: Optional[str], local_dry: bool) -> None:
|
||||
namespace: Optional[str], local_dry: bool, local_json: bool) -> None:
|
||||
"""Migrate data between backends.
|
||||
|
||||
Direct migration is only wired up between faiss, sqlite, and pgvector -
|
||||
these are the backends whose storage contract supports paging through
|
||||
every stored vector. Migrating to or from qdrant, pinecone, milvus, or
|
||||
weaviate still needs the export/reindex workaround below, since each of
|
||||
those needs its own enumeration design (Qdrant scroll, Pinecone list,
|
||||
etc.) that hasn't been built yet.
|
||||
|
||||
\b
|
||||
Example:
|
||||
semantica store migrate --from faiss --to qdrant --namespace production --dry-run
|
||||
semantica store migrate --from faiss --to sqlite --namespace production --dry-run
|
||||
"""
|
||||
cli_ctx = _require_ctx(cli_ctx)
|
||||
|
||||
@@ -3734,13 +3776,76 @@ def store_migrate(cli_ctx: CLIContext, from_backend: str, to_backend: str,
|
||||
if _is_dry(cli_ctx, local_dry):
|
||||
_dry(cli_ctx, "migrate", from_backend=from_backend, to_backend=to_backend)
|
||||
return
|
||||
raise click.ClickException(
|
||||
f"Direct backend migration ({from_backend} → {to_backend}) is not yet supported "
|
||||
"by the vector store layer. To migrate, export your data first:\n"
|
||||
" semantica export --format parquet --output dump.parquet\n"
|
||||
f" semantica embed index dump.parquet --store {to_backend}"
|
||||
+ (f" --namespace {namespace}" if namespace else "")
|
||||
)
|
||||
|
||||
if from_backend not in _MIGRATE_SUPPORTED_BACKENDS or to_backend not in _MIGRATE_SUPPORTED_BACKENDS:
|
||||
raise click.ClickException(
|
||||
f"Direct backend migration ({from_backend} → {to_backend}) is only supported "
|
||||
f"between {', '.join(sorted(_MIGRATE_SUPPORTED_BACKENDS))}. To migrate involving "
|
||||
"another backend, export your data first:\n"
|
||||
" semantica export --format parquet --output dump.parquet\n"
|
||||
f" semantica embed index dump.parquet --store {to_backend}"
|
||||
+ (f" --namespace {namespace}" if namespace else "")
|
||||
)
|
||||
|
||||
from .vector_store import VectorStore
|
||||
|
||||
vs_cfg = cli_ctx.config.to_dict().get("vector_store", {}) or {}
|
||||
source_cfg = _migrate_backend_config(vs_cfg, from_backend)
|
||||
dest_cfg = _migrate_backend_config(vs_cfg, to_backend)
|
||||
|
||||
source_index_path = None
|
||||
if from_backend == "faiss":
|
||||
source_index_path = _require_faiss_index_path(source_cfg, "source")
|
||||
dest_index_path = None
|
||||
if to_backend == "faiss":
|
||||
dest_index_path = _require_faiss_index_path(dest_cfg, "destination")
|
||||
|
||||
source = VectorStore(backend=from_backend, config=source_cfg)
|
||||
if source_index_path:
|
||||
source._backend_store.load_index(source_index_path)
|
||||
|
||||
source_dimension = getattr(source._backend_store, "dimension", None)
|
||||
if source_dimension and "dimension" not in dest_cfg:
|
||||
dest_cfg["dimension"] = source_dimension
|
||||
|
||||
dest = VectorStore(backend=to_backend, config=dest_cfg)
|
||||
if dest_index_path and Path(dest_index_path).exists():
|
||||
dest._backend_store.load_index(dest_index_path)
|
||||
|
||||
migrated = 0
|
||||
vectors_batch: List[Any] = []
|
||||
metadata_batch: List[Dict[str, Any]] = []
|
||||
ids_batch: List[str] = []
|
||||
|
||||
def _flush() -> None:
|
||||
nonlocal migrated
|
||||
if not vectors_batch:
|
||||
return
|
||||
dest.store_vectors(list(vectors_batch), list(metadata_batch), ids=list(ids_batch))
|
||||
migrated += len(vectors_batch)
|
||||
vectors_batch.clear()
|
||||
metadata_batch.clear()
|
||||
ids_batch.clear()
|
||||
|
||||
for item in source.iter_vectors(batch_size=_MIGRATE_BATCH_SIZE):
|
||||
meta = dict(item.get("metadata") or {})
|
||||
if namespace and "namespace" not in meta:
|
||||
meta["namespace"] = namespace
|
||||
vectors_batch.append(item["vector"])
|
||||
metadata_batch.append(meta)
|
||||
ids_batch.append(item["id"])
|
||||
if len(vectors_batch) >= _MIGRATE_BATCH_SIZE:
|
||||
_flush()
|
||||
_flush()
|
||||
|
||||
if dest_index_path and migrated:
|
||||
dest._backend_store.save_index(dest_index_path)
|
||||
|
||||
result = {"from": from_backend, "to": to_backend, "migrated": migrated}
|
||||
if _is_json(cli_ctx, local_json):
|
||||
_jecho(result)
|
||||
else:
|
||||
_ok(cli_ctx, f"Migrated {migrated} vectors from {from_backend} to {to_backend}")
|
||||
|
||||
_run_with_error_handling(_action)
|
||||
|
||||
|
||||
@@ -66,12 +66,32 @@ class FAISSIndex:
|
||||
self.metadata: Dict[str, Dict[str, Any]] = {}
|
||||
|
||||
def add_vectors(self, vectors: np.ndarray, ids: Optional[List[str]] = None):
|
||||
"""Add vectors to index."""
|
||||
"""
|
||||
Add vectors to index.
|
||||
|
||||
Skips any id already present in vector_ids rather than appending a
|
||||
second physical vector under the same id. FAISS indices here don't
|
||||
support removing or replacing a single vector in place, so an
|
||||
"update" isn't possible; without this check, re-running an add for
|
||||
ids that already exist (e.g. retrying an interrupted migration)
|
||||
would silently duplicate vectors under the same id on every retry.
|
||||
"""
|
||||
if ids is None:
|
||||
ids = [f"vec_{i}" for i in range(len(vectors))]
|
||||
|
||||
self.index.add(vectors.astype(np.float32))
|
||||
self.vector_ids.extend(ids)
|
||||
new_rows = []
|
||||
new_ids = []
|
||||
existing = set(self.vector_ids)
|
||||
for row, vec_id in zip(vectors, ids):
|
||||
if vec_id in existing:
|
||||
continue
|
||||
new_rows.append(row)
|
||||
new_ids.append(vec_id)
|
||||
existing.add(vec_id)
|
||||
|
||||
if new_rows:
|
||||
self.index.add(np.array(new_rows, dtype=np.float32))
|
||||
self.vector_ids.extend(new_ids)
|
||||
|
||||
def search(
|
||||
self, query_vectors: np.ndarray, k: int = 10
|
||||
@@ -305,6 +325,12 @@ class FAISSStore:
|
||||
"""
|
||||
Add vectors to index.
|
||||
|
||||
Any id that already exists in the index is skipped rather than
|
||||
stored as a second physical vector under the same id (see
|
||||
FAISSIndex.add_vectors), so calling this again with ids from a
|
||||
previous call is safe and doesn't accumulate duplicates. Metadata
|
||||
for those ids is still updated.
|
||||
|
||||
Args:
|
||||
vectors: List of vectors or numpy array
|
||||
ids: Vector IDs
|
||||
@@ -312,7 +338,8 @@ class FAISSStore:
|
||||
**options: Additional options
|
||||
|
||||
Returns:
|
||||
List of vector IDs
|
||||
List of vector IDs (including ids that were already present
|
||||
and therefore not re-added as new vectors)
|
||||
"""
|
||||
num_vectors = len(vectors) if isinstance(vectors, (list, np.ndarray)) else 1
|
||||
tracking_id = self.progress_tracker.start_tracking(
|
||||
@@ -526,6 +553,30 @@ class FAISSStore:
|
||||
|
||||
return results
|
||||
|
||||
def scan_vectors(self, offset: int = 0, limit: int = 100) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Page through stored vectors in insertion order.
|
||||
|
||||
Args:
|
||||
offset: Number of vectors to skip
|
||||
limit: Maximum number of vectors to return
|
||||
|
||||
Returns:
|
||||
List of result dicts with 'id', 'metadata', and 'vector'
|
||||
"""
|
||||
if self.index is None or limit <= 0:
|
||||
return []
|
||||
|
||||
ids_page = self.index.vector_ids[offset:offset + limit]
|
||||
return [
|
||||
{
|
||||
"id": vector_id,
|
||||
"metadata": self.get_metadata(vector_id) or {},
|
||||
"vector": self.get_vector(vector_id),
|
||||
}
|
||||
for vector_id in ids_page
|
||||
]
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""Get index statistics."""
|
||||
if self.index is None:
|
||||
|
||||
@@ -656,6 +656,52 @@ class PgVectorStore:
|
||||
self.logger.warning(f"Failed to get metadata for {vector_id}: {e}")
|
||||
return None
|
||||
|
||||
def scan_vectors(self, offset: int = 0, limit: int = 100) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Page through stored vectors ordered by id.
|
||||
|
||||
Args:
|
||||
offset: Number of rows to skip
|
||||
limit: Maximum number of rows to return
|
||||
|
||||
Returns:
|
||||
List of result dicts with 'id', 'metadata', and 'vector'
|
||||
"""
|
||||
if not PSYCOPG3_AVAILABLE and not PSYCOPG2_AVAILABLE:
|
||||
raise ProcessingError(
|
||||
"Neither psycopg3 nor psycopg2 is available. "
|
||||
"Install with: pip install psycopg[binary] or psycopg2-binary"
|
||||
)
|
||||
|
||||
if limit <= 0:
|
||||
return []
|
||||
|
||||
scan_sql = psycopg_sql.SQL("""
|
||||
SELECT id, vector, metadata
|
||||
FROM {}
|
||||
ORDER BY id
|
||||
LIMIT %s OFFSET %s
|
||||
""").format(psycopg_sql.Identifier(self.table_name))
|
||||
|
||||
with self._get_connection() as conn:
|
||||
try:
|
||||
cur = conn.cursor()
|
||||
cur.execute(scan_sql, (limit, offset))
|
||||
rows = cur.fetchall()
|
||||
cur.close()
|
||||
|
||||
results = []
|
||||
for row in rows:
|
||||
vec_id, vec, meta = row
|
||||
results.append({
|
||||
"id": vec_id,
|
||||
"metadata": meta if isinstance(meta, dict) else json.loads(meta) if meta else {},
|
||||
"vector": np.array(vec) if vec is not None else None,
|
||||
})
|
||||
return results
|
||||
except Exception as e:
|
||||
raise ProcessingError(f"Failed to scan vectors: {str(e)}") from e
|
||||
|
||||
def filter_by_metadata(
|
||||
self, filters: Dict[str, Any], limit: int = 10
|
||||
) -> List[Dict[str, Any]]:
|
||||
|
||||
@@ -604,6 +604,62 @@ class QdrantStore:
|
||||
self.logger.warning(f"Failed to scroll Qdrant points by metadata filter: {e}")
|
||||
return []
|
||||
|
||||
def iter_all(self, batch_size: int = 500):
|
||||
"""
|
||||
Iterate over every stored point using Qdrant's native scroll cursor.
|
||||
|
||||
Qdrant paginates by point-ID cursor, not by row offset, so this is
|
||||
exposed instead of scan_vectors(offset, limit). An integer passed to
|
||||
scroll()'s offset is a point ID rather than a rank, so there is no way
|
||||
to seek to "the Nth record" without walking from the start.
|
||||
VectorStore.iter_vectors() prefers this method when it is present.
|
||||
|
||||
Assumes a single unnamed vector per point, matching how insert_vectors()
|
||||
writes them and how get_vector() reads them back. Collections configured
|
||||
with named or multi-vectors are not handled here.
|
||||
|
||||
Args:
|
||||
batch_size: Points to request per scroll call
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in scroll order
|
||||
|
||||
Raises:
|
||||
ProcessingError: If the collection or client is not initialized.
|
||||
Errors are raised rather than swallowed because a scan that
|
||||
silently yields nothing is indistinguishable from an empty
|
||||
source, which would let a caller such as `store migrate`
|
||||
report success having copied nothing (issue #1083).
|
||||
"""
|
||||
if self.collection is None or self.client is None or not QDRANT_AVAILABLE:
|
||||
raise ProcessingError(
|
||||
"Collection not initialized. Call create_collection() or get_collection() first."
|
||||
)
|
||||
|
||||
next_offset = None
|
||||
while True:
|
||||
records, next_offset = self.client.scroll(
|
||||
collection_name=self.collection.collection_name,
|
||||
limit=batch_size,
|
||||
offset=next_offset,
|
||||
with_payload=True,
|
||||
with_vectors=True,
|
||||
)
|
||||
|
||||
for rec in records:
|
||||
yield {
|
||||
"id": str(rec.id),
|
||||
"metadata": rec.payload or {},
|
||||
"vector": np.array(rec.vector) if rec.vector is not None else None,
|
||||
}
|
||||
|
||||
# The final page can carry records while already reporting no next
|
||||
# cursor, so those records are yielded above before stopping here.
|
||||
# Calling scroll() again with offset=None would restart from the
|
||||
# beginning rather than continue past the end.
|
||||
if next_offset is None or not records:
|
||||
return
|
||||
|
||||
def delete_vectors(
|
||||
self, point_ids: List[Union[str, int]], **options
|
||||
) -> Dict[str, Any]:
|
||||
|
||||
@@ -616,6 +616,49 @@ class SQLiteVecStore:
|
||||
self.logger.warning(f"Failed to get metadata for {vector_id}: {e}")
|
||||
return None
|
||||
|
||||
def scan_vectors(self, offset: int = 0, limit: int = 100) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Page through stored vectors ordered by id.
|
||||
|
||||
Args:
|
||||
offset: Number of rows to skip
|
||||
limit: Maximum number of rows to return
|
||||
|
||||
Returns:
|
||||
List of result dicts with 'id', 'metadata', and 'vector'
|
||||
"""
|
||||
if limit <= 0:
|
||||
return []
|
||||
|
||||
query_sql = f"""
|
||||
SELECT id, embedding, metadata
|
||||
FROM {self.table_name}
|
||||
ORDER BY id
|
||||
LIMIT ? OFFSET ?
|
||||
"""
|
||||
|
||||
with self._lock, self._get_connection() as conn:
|
||||
try:
|
||||
cur = conn.cursor()
|
||||
cur.execute(query_sql, (limit, offset))
|
||||
rows = cur.fetchall()
|
||||
cur.close()
|
||||
|
||||
results = []
|
||||
for row in rows:
|
||||
vec_id, embedding_blob, meta_json = row
|
||||
vec = None
|
||||
if embedding_blob:
|
||||
vec = np.frombuffer(embedding_blob, dtype=np.float32).copy()
|
||||
results.append({
|
||||
"id": vec_id,
|
||||
"metadata": json.loads(meta_json) if meta_json else {},
|
||||
"vector": vec,
|
||||
})
|
||||
return results
|
||||
except Exception as e:
|
||||
raise ProcessingError(f"Failed to scan vectors: {str(e)}") from e
|
||||
|
||||
def filter_by_metadata(
|
||||
self, filters: Dict[str, Any], limit: int = 10
|
||||
) -> List[Dict[str, Any]]:
|
||||
|
||||
@@ -824,6 +824,77 @@ class VectorStore:
|
||||
else:
|
||||
raise NotImplementedError(f"Backend store {type(self._backend_store).__name__} does not implement get_metadata")
|
||||
|
||||
def scan_vectors(self, offset: int = 0, limit: int = 100) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Page through stored vectors, backend-agnostic.
|
||||
|
||||
Follows the get_vector()/get_metadata() precedent (#843): the inmemory
|
||||
backend pages its local dict directly, a persistent backend delegates
|
||||
to a scan_vectors() on the wrapped store when available, and one that
|
||||
cannot enumerate its contents raises NotImplementedError rather than
|
||||
silently returning an empty page.
|
||||
|
||||
Args:
|
||||
offset: Number of vectors to skip
|
||||
limit: Maximum number of vectors to return
|
||||
|
||||
Returns:
|
||||
List of result dicts with 'id', 'metadata', and 'vector'
|
||||
"""
|
||||
if limit <= 0:
|
||||
return []
|
||||
|
||||
if self.backend == "inmemory":
|
||||
ids_page = list(self.vectors.keys())[offset:offset + limit]
|
||||
return [
|
||||
{
|
||||
"id": vec_id,
|
||||
"metadata": self.metadata.get(vec_id, {}),
|
||||
"vector": self.vectors.get(vec_id),
|
||||
}
|
||||
for vec_id in ids_page
|
||||
]
|
||||
elif self._backend_store and hasattr(self._backend_store, "scan_vectors"):
|
||||
return self._backend_store.scan_vectors(offset=offset, limit=limit)
|
||||
else:
|
||||
raise NotImplementedError(
|
||||
f"Backend store {type(self._backend_store).__name__} does not "
|
||||
"implement scan_vectors(). Add a scan_vectors() method to the "
|
||||
"backend store adapter to enable enumeration for this backend."
|
||||
)
|
||||
|
||||
def iter_vectors(self, batch_size: int = 500):
|
||||
"""
|
||||
Iterate over every stored vector, one page at a time.
|
||||
|
||||
Backends whose native pagination is cursor based (Qdrant, Pinecone,
|
||||
Milvus, Weaviate) cannot honestly implement the positional
|
||||
scan_vectors(offset, limit) contract, so they expose iter_all()
|
||||
instead and it is preferred here when present. Backends with real
|
||||
positional access (inmemory, FAISS, SQLite-vec, PgVector) fall
|
||||
through to the offset loop below.
|
||||
|
||||
Args:
|
||||
batch_size: Number of vectors to fetch per underlying call
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in scan order
|
||||
"""
|
||||
if self.backend != "inmemory" and self._backend_store is not None:
|
||||
iter_all = getattr(self._backend_store, "iter_all", None)
|
||||
if callable(iter_all):
|
||||
yield from iter_all(batch_size=batch_size)
|
||||
return
|
||||
|
||||
offset = 0
|
||||
while True:
|
||||
page = self.scan_vectors(offset=offset, limit=batch_size)
|
||||
if not page:
|
||||
return
|
||||
for item in page:
|
||||
yield item
|
||||
offset += len(page)
|
||||
|
||||
def count(self) -> int:
|
||||
"""Return the number of vectors in the store, backend-agnostic.
|
||||
|
||||
|
||||
@@ -611,6 +611,115 @@ class WeaviateStore:
|
||||
self.logger.warning(f"Failed to fetch Weaviate objects by metadata filter: {e}")
|
||||
return results if results else []
|
||||
|
||||
def iter_all(self, batch_size: int = 500):
|
||||
"""
|
||||
Iterate over every stored object using Weaviate's UUID cursor.
|
||||
|
||||
Paginates by the last object's UUID rather than a row offset, which is
|
||||
why this exists instead of scan_vectors(offset, limit).
|
||||
|
||||
Assumes a single unnamed vector per object, as get_vector() and
|
||||
filter_by_metadata() already do. Named-vector collections return a
|
||||
mapping and are not handled.
|
||||
|
||||
Args:
|
||||
batch_size: Objects to request per fetch_objects() call
|
||||
|
||||
Yields:
|
||||
Result dicts with 'id', 'metadata', and 'vector', in cursor order
|
||||
|
||||
Raises:
|
||||
ProcessingError: If the collection is not initialized, or if the
|
||||
scan cannot advance past a full page.
|
||||
"""
|
||||
if self.collection is None or not WEAVIATE_AVAILABLE:
|
||||
raise ProcessingError(
|
||||
"Collection not initialized. Call get_collection() first."
|
||||
)
|
||||
|
||||
after_cursor = None
|
||||
scanned_count = 0
|
||||
# Degrades cursor -> offset -> single_page as the client rejects each
|
||||
# form. Tracked across iterations, not just inside the except branch,
|
||||
# or later pages go out with no pagination argument at all.
|
||||
mode = "cursor"
|
||||
|
||||
while True:
|
||||
kwargs = {"limit": batch_size, "include_vector": True}
|
||||
if mode == "cursor" and after_cursor is not None:
|
||||
kwargs["after"] = after_cursor
|
||||
elif mode == "offset":
|
||||
kwargs["offset"] = scanned_count
|
||||
|
||||
try:
|
||||
objs = self.collection.query.fetch_objects(**kwargs)
|
||||
except TypeError:
|
||||
if mode == "cursor" and "after" in kwargs:
|
||||
mode = "offset"
|
||||
kwargs.pop("after", None)
|
||||
kwargs["offset"] = scanned_count
|
||||
try:
|
||||
objs = self.collection.query.fetch_objects(**kwargs)
|
||||
except TypeError:
|
||||
mode = "single_page"
|
||||
kwargs.pop("offset", None)
|
||||
objs = self.collection.query.fetch_objects(**kwargs)
|
||||
elif mode == "offset":
|
||||
mode = "single_page"
|
||||
kwargs.pop("offset", None)
|
||||
objs = self.collection.query.fetch_objects(**kwargs)
|
||||
else:
|
||||
raise
|
||||
|
||||
batch_objects = getattr(objs, "objects", None) if objs else None
|
||||
if not batch_objects:
|
||||
return
|
||||
|
||||
for obj in batch_objects:
|
||||
obj_uuid = getattr(obj, "uuid", None)
|
||||
raw_vector = getattr(obj, "vector", None)
|
||||
yield {
|
||||
"id": str(obj_uuid) if obj_uuid is not None else None,
|
||||
"metadata": getattr(obj, "properties", None) or {},
|
||||
"vector": (
|
||||
np.array(raw_vector)
|
||||
if raw_vector is not None and len(raw_vector) > 0
|
||||
else None
|
||||
),
|
||||
}
|
||||
|
||||
scanned_count += len(batch_objects)
|
||||
|
||||
# Past this point the page was full, so failing to advance is
|
||||
# truncation rather than completion.
|
||||
if len(batch_objects) < batch_size:
|
||||
return
|
||||
|
||||
if mode == "single_page":
|
||||
raise ProcessingError(
|
||||
"This Weaviate client accepts neither an `after` cursor nor a "
|
||||
"numeric offset, so the scan cannot advance past the first "
|
||||
"page. Refusing to return a truncated scan."
|
||||
)
|
||||
|
||||
if mode == "offset":
|
||||
continue
|
||||
|
||||
last_uuid = getattr(batch_objects[-1], "uuid", None)
|
||||
if last_uuid is None:
|
||||
raise ProcessingError(
|
||||
"The last object of a full Weaviate page has no uuid, so the "
|
||||
"cursor cannot advance. Refusing to return a truncated scan."
|
||||
)
|
||||
|
||||
next_cursor = str(last_uuid)
|
||||
if next_cursor == after_cursor:
|
||||
raise ProcessingError(
|
||||
"The Weaviate cursor stopped advancing, so the listing is "
|
||||
"repeating a page. Refusing to return a truncated scan."
|
||||
)
|
||||
after_cursor = next_cursor
|
||||
|
||||
|
||||
def query_vectors(
|
||||
self,
|
||||
|
||||
@@ -1359,6 +1359,101 @@ class TestStore:
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate", "--from", "faiss"])
|
||||
assert result.exit_code != 0
|
||||
|
||||
def test_migrate_refuses_unsupported_backend_pair(self, runner):
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "faiss", "--to", "qdrant"])
|
||||
assert result.exit_code != 0
|
||||
assert "faiss, pgvector, sqlite" in result.output
|
||||
|
||||
def _fake_migrate_store_module(self, source_items, stored, dest_configs=None):
|
||||
class _FakeBackendStore:
|
||||
def __init__(self, dimension=None):
|
||||
self.dimension = dimension
|
||||
|
||||
class _FakeStore:
|
||||
def __init__(self, backend, config=None, **kw):
|
||||
self.backend = backend
|
||||
self._config = config or {}
|
||||
dim = self._config.get("dimension")
|
||||
self._backend_store = _FakeBackendStore(dimension=dim)
|
||||
if dest_configs is not None:
|
||||
dest_configs[backend] = dict(self._config)
|
||||
|
||||
def iter_vectors(self, batch_size=500):
|
||||
if self.backend == "sqlite":
|
||||
yield from source_items
|
||||
return
|
||||
return
|
||||
yield # pragma: no cover - makes this a generator for other backends
|
||||
|
||||
def store_vectors(self, vectors, metadata, ids=None):
|
||||
for vec_id, meta in zip(ids, metadata):
|
||||
stored[vec_id] = meta
|
||||
|
||||
return _fake_module(VectorStore=_FakeStore)
|
||||
|
||||
def test_migrate_runs_between_supported_backends(self, runner, monkeypatch):
|
||||
source_items = [
|
||||
{"id": "a", "vector": [0.1, 0.2], "metadata": {"tag": "x"}},
|
||||
{"id": "b", "vector": [0.3, 0.4], "metadata": {}},
|
||||
]
|
||||
stored = {}
|
||||
fake_vs = self._fake_migrate_store_module(source_items, stored)
|
||||
monkeypatch.setitem(__import__("sys").modules, "semantica.vector_store", fake_vs)
|
||||
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "sqlite", "--to", "pgvector",
|
||||
"--namespace", "prod", "--json"])
|
||||
_ok(result)
|
||||
data = _json_output(result)
|
||||
assert data == {"from": "sqlite", "to": "pgvector", "migrated": 2}
|
||||
assert stored == {"a": {"tag": "x", "namespace": "prod"}, "b": {"namespace": "prod"}}
|
||||
|
||||
def test_migrate_reports_zero_for_empty_source(self, runner, monkeypatch):
|
||||
stored = {}
|
||||
fake_vs = self._fake_migrate_store_module([], stored)
|
||||
monkeypatch.setitem(__import__("sys").modules, "semantica.vector_store", fake_vs)
|
||||
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "sqlite", "--to", "pgvector", "--json"])
|
||||
_ok(result)
|
||||
assert _json_output(result)["migrated"] == 0
|
||||
assert stored == {}
|
||||
|
||||
def test_migrate_inherits_source_dimension_into_dest(self, runner, monkeypatch):
|
||||
source_items = [{"id": "a", "vector": [0.1, 0.2, 0.3], "metadata": {}}]
|
||||
stored = {}
|
||||
dest_configs: dict = {}
|
||||
fake_vs = self._fake_migrate_store_module(source_items, stored, dest_configs)
|
||||
monkeypatch.setitem(__import__("sys").modules, "semantica.vector_store", fake_vs)
|
||||
monkeypatch.setattr(
|
||||
cli_module.Config, "to_dict",
|
||||
lambda self: {"vector_store": {"sqlite": {"dimension": 3}, "pgvector": {}}},
|
||||
)
|
||||
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "sqlite", "--to", "pgvector", "--json"])
|
||||
_ok(result)
|
||||
assert dest_configs["pgvector"].get("dimension") == 3
|
||||
|
||||
def test_migrate_faiss_source_requires_index_path(self, runner, monkeypatch):
|
||||
fake_vs = _fake_module(VectorStore=lambda **kw: MagicMock())
|
||||
monkeypatch.setitem(__import__("sys").modules, "semantica.vector_store", fake_vs)
|
||||
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "faiss", "--to", "sqlite"])
|
||||
assert result.exit_code != 0
|
||||
assert "index_path" in result.output
|
||||
|
||||
def test_migrate_faiss_dest_requires_index_path(self, runner, monkeypatch):
|
||||
fake_vs = _fake_module(VectorStore=lambda **kw: MagicMock())
|
||||
monkeypatch.setitem(__import__("sys").modules, "semantica.vector_store", fake_vs)
|
||||
|
||||
result = runner.invoke(cli_module.main, ["store", "migrate",
|
||||
"--from", "sqlite", "--to", "faiss"])
|
||||
assert result.exit_code != 0
|
||||
assert "index_path" in result.output
|
||||
|
||||
def test_flush_requires_confirm(self, runner):
|
||||
result = runner.invoke(cli_module.main, ["store", "flush"])
|
||||
assert result.exit_code != 0
|
||||
|
||||
@@ -3,7 +3,7 @@ from unittest.mock import MagicMock
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from semantica.vector_store.faiss_store import FAISSIndex
|
||||
from semantica.vector_store.faiss_store import FAISSIndex, FAISSStore
|
||||
|
||||
|
||||
def test_get_vector_reconstructs_from_flat_l2_index():
|
||||
@@ -120,3 +120,90 @@ def test_get_vector_reconstructs_from_real_ivfflat_index_without_prior_direct_ma
|
||||
result = index.get_vector("vec_target")
|
||||
|
||||
np.testing.assert_allclose(result, vectors[3], atol=1e-6)
|
||||
|
||||
|
||||
def _store_with_fake_index(ids, metadata_by_id=None):
|
||||
backend_index = MagicMock()
|
||||
backend_index.reconstruct.side_effect = lambda idx: [float(idx)] * 3
|
||||
index = FAISSIndex(backend_index, dimension=3)
|
||||
index.vector_ids = list(ids)
|
||||
index.metadata = dict(metadata_by_id or {})
|
||||
|
||||
store = FAISSStore(dimension=3)
|
||||
store.index = index
|
||||
return store
|
||||
|
||||
|
||||
def test_scan_vectors_returns_all_across_pages():
|
||||
store = _store_with_fake_index(["a", "b", "c", "d", "e"])
|
||||
|
||||
seen_ids = []
|
||||
offset = 0
|
||||
while True:
|
||||
page = store.scan_vectors(offset=offset, limit=2)
|
||||
if not page:
|
||||
break
|
||||
seen_ids.extend(p["id"] for p in page)
|
||||
offset += len(page)
|
||||
|
||||
assert seen_ids == ["a", "b", "c", "d", "e"]
|
||||
|
||||
|
||||
def test_scan_vectors_includes_vector_and_metadata():
|
||||
store = _store_with_fake_index(["a"], {"a": {"tag": "only"}})
|
||||
|
||||
page = store.scan_vectors(offset=0, limit=10)
|
||||
|
||||
assert len(page) == 1
|
||||
assert page[0]["id"] == "a"
|
||||
assert page[0]["metadata"] == {"tag": "only"}
|
||||
np.testing.assert_array_equal(page[0]["vector"], np.array([0.0, 0.0, 0.0], dtype=np.float32))
|
||||
|
||||
|
||||
def test_scan_vectors_no_index_returns_empty_list():
|
||||
store = FAISSStore(dimension=3)
|
||||
assert store.scan_vectors(offset=0, limit=10) == []
|
||||
|
||||
|
||||
def test_scan_vectors_zero_limit_returns_empty_list():
|
||||
store = _store_with_fake_index(["a"])
|
||||
assert store.scan_vectors(offset=0, limit=0) == []
|
||||
|
||||
|
||||
def test_scan_vectors_offset_past_end_returns_empty_list():
|
||||
store = _store_with_fake_index(["a"])
|
||||
assert store.scan_vectors(offset=100, limit=10) == []
|
||||
|
||||
|
||||
def test_add_vectors_retry_with_same_ids_does_not_duplicate():
|
||||
"""Re-running add_vectors with ids already in the index (e.g. retrying
|
||||
an interrupted migration) must not create a second physical vector
|
||||
under the same id."""
|
||||
backend_index = MagicMock()
|
||||
store = FAISSStore(dimension=3)
|
||||
store.index = FAISSIndex(backend_index, dimension=3)
|
||||
|
||||
vectors = np.array([[1, 2, 3], [4, 5, 6], [7, 8, 9], [10, 11, 12]], dtype=np.float32)
|
||||
ids = ["a", "b", "c", "d"]
|
||||
|
||||
store.add_vectors(vectors, ids=ids, metadata=[{"i": i} for i in range(4)])
|
||||
assert store.count() == 4
|
||||
|
||||
store.add_vectors(vectors, ids=ids, metadata=[{"i": i} for i in range(4)])
|
||||
|
||||
assert store.count() == 4
|
||||
assert store.index.vector_ids == ids
|
||||
|
||||
|
||||
def test_add_vectors_retry_with_partial_overlap_only_adds_new_ids():
|
||||
backend_index = MagicMock()
|
||||
store = FAISSStore(dimension=3)
|
||||
store.index = FAISSIndex(backend_index, dimension=3)
|
||||
|
||||
store.add_vectors(np.array([[1, 2, 3], [4, 5, 6]], dtype=np.float32), ids=["a", "b"])
|
||||
store.add_vectors(np.array([[1, 2, 3], [7, 8, 9]], dtype=np.float32), ids=["a", "c"])
|
||||
|
||||
assert store.index.vector_ids == ["a", "b", "c"]
|
||||
second_call_vectors = backend_index.add.call_args[0][0]
|
||||
assert second_call_vectors.shape[0] == 1
|
||||
np.testing.assert_array_equal(second_call_vectors[0], np.array([7, 8, 9], dtype=np.float32))
|
||||
|
||||
@@ -467,6 +467,48 @@ class TestPgVectorStoreDelete:
|
||||
assert success is True
|
||||
|
||||
|
||||
class TestPgVectorStoreScan:
|
||||
"""Test scan_vectors pagination."""
|
||||
|
||||
def test_scan_returns_all_vectors_across_pages(self, store):
|
||||
vectors = [np.random.rand(128).astype(np.float32) for _ in range(5)]
|
||||
ids = store.add(vectors, [{"index": i} for i in range(5)])
|
||||
|
||||
seen_ids = []
|
||||
offset = 0
|
||||
while True:
|
||||
page = store.scan_vectors(offset=offset, limit=2)
|
||||
if not page:
|
||||
break
|
||||
seen_ids.extend(p["id"] for p in page)
|
||||
offset += len(page)
|
||||
|
||||
assert set(seen_ids) == set(ids)
|
||||
assert len(seen_ids) == 5
|
||||
|
||||
def test_scan_page_includes_vector_and_metadata(self, store):
|
||||
vectors = [np.random.rand(128).astype(np.float32)]
|
||||
ids = store.add(vectors, [{"tag": "only"}])
|
||||
|
||||
page = store.scan_vectors(offset=0, limit=10)
|
||||
|
||||
assert len(page) == 1
|
||||
assert page[0]["id"] == ids[0]
|
||||
assert page[0]["metadata"] == {"tag": "only"}
|
||||
assert page[0]["vector"] is not None
|
||||
|
||||
def test_scan_empty_store_returns_empty_list(self, store):
|
||||
assert store.scan_vectors(offset=0, limit=10) == []
|
||||
|
||||
def test_scan_zero_limit_returns_empty_list(self, store):
|
||||
store.add([np.random.rand(128).astype(np.float32)])
|
||||
assert store.scan_vectors(offset=0, limit=0) == []
|
||||
|
||||
def test_scan_offset_past_end_returns_empty_list(self, store):
|
||||
store.add([np.random.rand(128).astype(np.float32)])
|
||||
assert store.scan_vectors(offset=100, limit=10) == []
|
||||
|
||||
|
||||
class TestPgVectorStoreIndex:
|
||||
"""Test index creation operations."""
|
||||
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Tests for QdrantStore.iter_all() cursor enumeration.
|
||||
|
||||
Qdrant is not installed in this environment, so these drive the real
|
||||
QdrantStore against a MagicMock standing in for the qdrant_client, following
|
||||
the pattern already used for qdrant in test_backend_metadata_filtering.py.
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from semantica.utils.exceptions import ProcessingError
|
||||
from semantica.vector_store.qdrant_store import QdrantStore
|
||||
|
||||
|
||||
def _record(point_id, payload=None, vector=None):
|
||||
"""Build a stand-in for a qdrant_client Record."""
|
||||
rec = MagicMock()
|
||||
rec.id = point_id
|
||||
rec.payload = payload
|
||||
rec.vector = vector
|
||||
return rec
|
||||
|
||||
|
||||
def _store_with_scroll(*pages):
|
||||
"""QdrantStore whose client.scroll() returns the given (records, cursor) pages."""
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.client.scroll.side_effect = list(pages)
|
||||
store.collection = MagicMock()
|
||||
store.collection.collection_name = "test_collection"
|
||||
return store
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_threads_cursor_across_pages():
|
||||
"""The next call must continue from the previous page's next_page_offset."""
|
||||
store = _store_with_scroll(
|
||||
([_record(1), _record(2)], "cursor-1"),
|
||||
([_record(3)], None),
|
||||
)
|
||||
|
||||
result = list(store.iter_all(batch_size=2))
|
||||
|
||||
assert [item["id"] for item in result] == ["1", "2", "3"]
|
||||
calls = store.client.scroll.call_args_list
|
||||
assert len(calls) == 2
|
||||
assert calls[0][1]["offset"] is None
|
||||
assert calls[0][1]["limit"] == 2
|
||||
assert calls[1][1]["offset"] == "cursor-1"
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_yields_final_page_that_reports_no_next_cursor():
|
||||
"""Qdrant can return records and a null cursor on the same page.
|
||||
|
||||
Those records must still be yielded. Treating a null cursor as "stop
|
||||
before this page" would silently drop the tail of every scan.
|
||||
"""
|
||||
store = _store_with_scroll(([_record(1), _record(2)], None))
|
||||
|
||||
result = list(store.iter_all(batch_size=10))
|
||||
|
||||
assert [item["id"] for item in result] == ["1", "2"]
|
||||
assert store.client.scroll.call_count == 1
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_converts_records_to_the_shared_result_shape():
|
||||
store = _store_with_scroll(
|
||||
([_record(7, payload={"tag": "x"}, vector=[0.1, 0.2, 0.3])], None),
|
||||
)
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["id"] == "7"
|
||||
assert item["metadata"] == {"tag": "x"}
|
||||
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2, 0.3]))
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_handles_missing_payload_and_vector():
|
||||
store = _store_with_scroll(([_record(1, payload=None, vector=None)], None))
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["metadata"] == {}
|
||||
assert item["vector"] is None
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_empty_collection_yields_nothing():
|
||||
store = _store_with_scroll(([], None))
|
||||
|
||||
assert list(store.iter_all()) == []
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_stops_on_empty_page_even_with_a_cursor():
|
||||
"""Defensive: an empty page ends the scan rather than looping forever."""
|
||||
store = _store_with_scroll(([], "cursor-that-never-clears"))
|
||||
|
||||
assert list(store.iter_all()) == []
|
||||
assert store.client.scroll.call_count == 1
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_raises_when_collection_not_initialized():
|
||||
"""Must fail loudly, not yield nothing.
|
||||
|
||||
An empty scan is indistinguishable from an empty source, which would let
|
||||
`store migrate` report success having copied nothing (issue #1083).
|
||||
"""
|
||||
store = QdrantStore()
|
||||
|
||||
with pytest.raises(ProcessingError, match="Collection not initialized"):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", False)
|
||||
def test_iter_all_raises_when_qdrant_unavailable():
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.collection = MagicMock()
|
||||
|
||||
with pytest.raises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_propagates_scroll_errors():
|
||||
store = QdrantStore()
|
||||
store.client = MagicMock()
|
||||
store.client.scroll.side_effect = RuntimeError("connection reset")
|
||||
store.collection = MagicMock()
|
||||
store.collection.collection_name = "test_collection"
|
||||
|
||||
with pytest.raises(RuntimeError, match="connection reset"):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.qdrant_store.QDRANT_AVAILABLE", True)
|
||||
def test_iter_all_requests_payload_and_vectors():
|
||||
store = _store_with_scroll(([], None))
|
||||
|
||||
list(store.iter_all())
|
||||
|
||||
kwargs = store.client.scroll.call_args[1]
|
||||
assert kwargs["with_payload"] is True
|
||||
assert kwargs["with_vectors"] is True
|
||||
assert kwargs["collection_name"] == "test_collection"
|
||||
@@ -415,6 +415,48 @@ class TestSQLiteVecStoreStats:
|
||||
assert stats["vector_count"] == 4
|
||||
|
||||
|
||||
class TestSQLiteVecStoreScan:
|
||||
"""Test scan_vectors pagination."""
|
||||
|
||||
def test_scan_returns_all_vectors_across_pages(self, store):
|
||||
vectors = [np.random.rand(128).astype(np.float32) for _ in range(5)]
|
||||
ids = store.add(vectors, [{"index": i} for i in range(5)])
|
||||
|
||||
seen_ids = []
|
||||
offset = 0
|
||||
while True:
|
||||
page = store.scan_vectors(offset=offset, limit=2)
|
||||
if not page:
|
||||
break
|
||||
seen_ids.extend(p["id"] for p in page)
|
||||
offset += len(page)
|
||||
|
||||
assert set(seen_ids) == set(ids)
|
||||
assert len(seen_ids) == 5
|
||||
|
||||
def test_scan_page_includes_vector_and_metadata(self, store):
|
||||
vectors = [np.random.rand(128).astype(np.float32)]
|
||||
ids = store.add(vectors, [{"tag": "only"}])
|
||||
|
||||
page = store.scan_vectors(offset=0, limit=10)
|
||||
|
||||
assert len(page) == 1
|
||||
assert page[0]["id"] == ids[0]
|
||||
assert page[0]["metadata"] == {"tag": "only"}
|
||||
assert page[0]["vector"] is not None
|
||||
|
||||
def test_scan_empty_store_returns_empty_list(self, store):
|
||||
assert store.scan_vectors(offset=0, limit=10) == []
|
||||
|
||||
def test_scan_zero_limit_returns_empty_list(self, store):
|
||||
store.add([np.random.rand(128).astype(np.float32)])
|
||||
assert store.scan_vectors(offset=0, limit=0) == []
|
||||
|
||||
def test_scan_offset_past_end_returns_empty_list(self, store):
|
||||
store.add([np.random.rand(128).astype(np.float32)])
|
||||
assert store.scan_vectors(offset=100, limit=10) == []
|
||||
|
||||
|
||||
class TestSQLiteVecStoreFilterByMetadata:
|
||||
"""Test filter_by_metadata, including list-valued metadata handling."""
|
||||
|
||||
|
||||
@@ -26,6 +26,7 @@ from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
|
||||
from semantica.utils.exceptions import ProcessingError
|
||||
from semantica.vector_store.vector_store import VectorStore, VectorManager
|
||||
|
||||
|
||||
@@ -120,6 +121,185 @@ class VectorStoreCountTests(unittest.TestCase):
|
||||
self.assertIn("count()", msg)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VectorStore.scan_vectors() / iter_vectors() dispatch tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class _ScanningBackendStore:
|
||||
"""Fake persistent backend store that supports scan_vectors()."""
|
||||
|
||||
def __init__(self, items):
|
||||
self._items = items
|
||||
|
||||
def scan_vectors(self, offset=0, limit=100):
|
||||
return self._items[offset:offset + limit]
|
||||
|
||||
|
||||
class _NonScanningBackendStore:
|
||||
"""Fake persistent backend store without any scan capability."""
|
||||
|
||||
|
||||
class _IterAllBackendStore:
|
||||
"""Fake cursor-based backend store exposing iter_all() but not scan_vectors().
|
||||
|
||||
Mirrors qdrant/pinecone/milvus/weaviate, which cannot honour a positional
|
||||
offset and therefore expose native iteration instead.
|
||||
"""
|
||||
|
||||
def __init__(self, items):
|
||||
self._items = items
|
||||
self.batch_sizes = []
|
||||
|
||||
def iter_all(self, batch_size=500):
|
||||
self.batch_sizes.append(batch_size)
|
||||
for item in self._items:
|
||||
yield item
|
||||
|
||||
def scan_vectors(self, offset=0, limit=100):
|
||||
raise AssertionError("scan_vectors() must not be called when iter_all() exists")
|
||||
|
||||
|
||||
class _MisShapedIterAllBackendStore:
|
||||
"""Backend store whose ``iter_all`` attribute is not callable."""
|
||||
|
||||
iter_all = 42 # plain attribute, not a method
|
||||
|
||||
def __init__(self, items):
|
||||
self._items = items
|
||||
|
||||
def scan_vectors(self, offset=0, limit=100):
|
||||
return self._items[offset:offset + limit]
|
||||
|
||||
|
||||
class VectorStoreScanVectorsTests(unittest.TestCase):
|
||||
"""VectorStore.scan_vectors() / iter_vectors() backend-agnostic accessors."""
|
||||
|
||||
def setUp(self):
|
||||
self.vectors = [np.array([1.0, 0.0]), np.array([0.0, 1.0]), np.array([1.0, 1.0])]
|
||||
self.metadata = [{"type": "a"}, {"type": "b"}, {"type": "c"}]
|
||||
|
||||
def test_scan_inmemory_pages_through_all_vectors(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
ids = store.store_vectors(self.vectors, self.metadata)
|
||||
|
||||
page1 = store.scan_vectors(offset=0, limit=2)
|
||||
page2 = store.scan_vectors(offset=2, limit=2)
|
||||
|
||||
self.assertEqual([p["id"] for p in page1], ids[:2])
|
||||
self.assertEqual([p["id"] for p in page2], ids[2:])
|
||||
self.assertEqual(page2[0]["metadata"], {"type": "c"})
|
||||
|
||||
def test_scan_inmemory_empty_store(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
self.assertEqual(store.scan_vectors(offset=0, limit=10), [])
|
||||
|
||||
def test_scan_zero_limit_returns_empty_list(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.store_vectors(self.vectors, self.metadata)
|
||||
self.assertEqual(store.scan_vectors(offset=0, limit=0), [])
|
||||
|
||||
def test_scan_delegates_to_backend_store(self):
|
||||
items = [{"id": "a", "metadata": {}, "vector": None}]
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.backend = "faiss"
|
||||
store._backend_store = _ScanningBackendStore(items)
|
||||
self.assertEqual(store.scan_vectors(offset=0, limit=10), items)
|
||||
|
||||
def test_scan_raises_not_implemented_without_backend_support(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.backend = "faiss"
|
||||
store._backend_store = _NonScanningBackendStore()
|
||||
with self.assertRaises(NotImplementedError):
|
||||
store.scan_vectors(offset=0, limit=10)
|
||||
|
||||
def test_iter_vectors_walks_every_page(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
ids = store.store_vectors(self.vectors, self.metadata)
|
||||
|
||||
collected = list(store.iter_vectors(batch_size=2))
|
||||
|
||||
self.assertEqual([item["id"] for item in collected], ids)
|
||||
|
||||
def test_iter_vectors_empty_store_yields_nothing(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), [])
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VectorStore.iter_vectors() preference for a native iter_all()
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class VectorStoreIterAllDispatchTests(unittest.TestCase):
|
||||
"""iter_vectors() prefers a backend's native iter_all() when present.
|
||||
|
||||
Cursor-based backends cannot implement scan_vectors(offset, limit)
|
||||
honestly, so they expose iter_all() instead and iter_vectors() routes to
|
||||
it rather than walking offsets.
|
||||
"""
|
||||
|
||||
def _persistent_store(self, backend_store, backend_name="qdrant"):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.backend = backend_name
|
||||
store._backend_store = backend_store
|
||||
return store
|
||||
|
||||
def test_iter_vectors_uses_iter_all_when_available(self):
|
||||
items = [
|
||||
{"id": "a", "vector": None, "metadata": {"n": 1}},
|
||||
{"id": "b", "vector": None, "metadata": {"n": 2}},
|
||||
]
|
||||
backend = _IterAllBackendStore(items)
|
||||
store = self._persistent_store(backend)
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=7)), items)
|
||||
|
||||
def test_iter_vectors_forwards_batch_size_to_iter_all(self):
|
||||
backend = _IterAllBackendStore([])
|
||||
store = self._persistent_store(backend)
|
||||
|
||||
list(store.iter_vectors(batch_size=32))
|
||||
|
||||
self.assertEqual(backend.batch_sizes, [32])
|
||||
|
||||
def test_iter_vectors_falls_back_to_scan_vectors_without_iter_all(self):
|
||||
items = [{"id": "a", "vector": None, "metadata": {}}]
|
||||
store = self._persistent_store(_ScanningBackendStore(items))
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
|
||||
|
||||
def test_iter_vectors_falls_back_when_iter_all_not_callable(self):
|
||||
# A mis-shaped adapter exposing a non-callable ``iter_all`` must not be
|
||||
# invoked; the offset path still has to work. Mirrors the count()
|
||||
# precedent in _MisShapedBackendStore.
|
||||
items = [{"id": "a", "vector": None, "metadata": {}}]
|
||||
store = self._persistent_store(_MisShapedIterAllBackendStore(items))
|
||||
|
||||
self.assertEqual(list(store.iter_vectors(batch_size=2)), items)
|
||||
|
||||
def test_iter_vectors_inmemory_ignores_iter_all(self):
|
||||
store = VectorStore(backend="inmemory", dimension=2)
|
||||
store.store_vectors([np.array([1.0, 0.0])], [{"type": "a"}])
|
||||
store._backend_store = _IterAllBackendStore([{"id": "wrong"}])
|
||||
|
||||
collected = list(store.iter_vectors(batch_size=2))
|
||||
|
||||
self.assertEqual([item["metadata"] for item in collected], [{"type": "a"}])
|
||||
|
||||
def test_iter_vectors_propagates_iter_all_errors(self):
|
||||
# A scan that silently yields nothing is indistinguishable from an
|
||||
# empty source, which would let `store migrate` report success having
|
||||
# copied nothing (issue #1083).
|
||||
class _FailingIterAll:
|
||||
def iter_all(self, batch_size=500):
|
||||
raise ProcessingError("backend unreachable")
|
||||
yield # pragma: no cover - makes this a generator
|
||||
|
||||
store = self._persistent_store(_FailingIterAll())
|
||||
|
||||
with self.assertRaises(ProcessingError):
|
||||
list(store.iter_vectors(batch_size=2))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# VectorManager tests — inmemory backend
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
"""Tests for WeaviateStore.iter_all() cursor enumeration.
|
||||
|
||||
weaviate-client is not installed in this environment, so these drive the real
|
||||
WeaviateStore against MagicMocks, following the pattern already used for
|
||||
weaviate in test_backend_metadata_filtering.py.
|
||||
"""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
from semantica.utils.exceptions import ProcessingError
|
||||
from semantica.vector_store.weaviate_store import WeaviateStore
|
||||
|
||||
|
||||
def _obj(uuid, properties=None, vector=None):
|
||||
"""Stand-in for a weaviate v4 returned object."""
|
||||
obj = MagicMock()
|
||||
obj.uuid = uuid
|
||||
obj.properties = properties
|
||||
obj.vector = vector
|
||||
return obj
|
||||
|
||||
|
||||
def _page(objects):
|
||||
"""Stand-in for a fetch_objects() response."""
|
||||
response = MagicMock()
|
||||
response.objects = objects
|
||||
return response
|
||||
|
||||
|
||||
def _store_with_pages(*pages):
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
store.collection.query.fetch_objects.side_effect = list(pages)
|
||||
return store
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_threads_uuid_cursor_across_pages():
|
||||
"""The next page must continue after the last object's UUID."""
|
||||
store = _store_with_pages(
|
||||
_page([_obj("uuid-1"), _obj("uuid-2")]),
|
||||
_page([_obj("uuid-3")]),
|
||||
)
|
||||
|
||||
result = list(store.iter_all(batch_size=2))
|
||||
|
||||
assert [item["id"] for item in result] == ["uuid-1", "uuid-2", "uuid-3"]
|
||||
calls = store.collection.query.fetch_objects.call_args_list
|
||||
assert "after" not in calls[0][1]
|
||||
assert calls[1][1]["after"] == "uuid-2"
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_stops_on_short_page():
|
||||
"""A page smaller than batch_size means the collection is exhausted."""
|
||||
store = _store_with_pages(_page([_obj("uuid-1")]))
|
||||
|
||||
result = list(store.iter_all(batch_size=5))
|
||||
|
||||
assert [item["id"] for item in result] == ["uuid-1"]
|
||||
assert store.collection.query.fetch_objects.call_count == 1
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_raises_when_cursor_stops_advancing():
|
||||
"""A stalled cursor must terminate, but not quietly: a partial scan reads
|
||||
as a complete one."""
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
store.collection.query.fetch_objects.return_value = _page(
|
||||
[_obj("same-uuid"), _obj("same-uuid")]
|
||||
)
|
||||
|
||||
with pytest.raises(ProcessingError, match="stopped advancing"):
|
||||
list(store.iter_all(batch_size=2))
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_offset_fallback_advances_across_pages():
|
||||
"""Regression: the offset was only set inside the except branch, so pages
|
||||
after the fallback went out with no pagination at all and the scan
|
||||
restarted from page one."""
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
calls = []
|
||||
|
||||
def _fetch(**kwargs):
|
||||
calls.append(dict(kwargs))
|
||||
if "after" in kwargs:
|
||||
raise TypeError("unexpected keyword argument 'after'")
|
||||
page_number = len(calls)
|
||||
if page_number < 4:
|
||||
return _page([_obj(f"u{page_number}a"), _obj(f"u{page_number}b")])
|
||||
return _page([_obj("last")])
|
||||
|
||||
store.collection.query.fetch_objects.side_effect = _fetch
|
||||
|
||||
ids = [item["id"] for item in store.iter_all(batch_size=2)]
|
||||
|
||||
assert len(set(ids)) == len(ids), f"duplicate ids means the scan restarted: {ids}"
|
||||
assert [c.get("offset") for c in calls] == [None, None, 2, 4]
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_raises_when_no_pagination_is_supported():
|
||||
"""A client rejecting both `after` and `offset` cannot page past the first
|
||||
result."""
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
|
||||
def _fetch(**kwargs):
|
||||
if "after" in kwargs or "offset" in kwargs:
|
||||
raise TypeError("unsupported")
|
||||
return _page([_obj("a"), _obj("b")])
|
||||
|
||||
store.collection.query.fetch_objects.side_effect = _fetch
|
||||
|
||||
with pytest.raises(ProcessingError, match="neither an .after. cursor nor a"):
|
||||
list(store.iter_all(batch_size=2))
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_empty_collection_yields_nothing():
|
||||
store = _store_with_pages(_page([]))
|
||||
|
||||
assert list(store.iter_all()) == []
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_converts_objects_to_the_shared_result_shape():
|
||||
store = _store_with_pages(
|
||||
_page([_obj("uuid-7", properties={"tag": "x"}, vector=[0.1, 0.2, 0.3])]),
|
||||
)
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["id"] == "uuid-7"
|
||||
assert item["metadata"] == {"tag": "x"}
|
||||
np.testing.assert_allclose(item["vector"], np.array([0.1, 0.2, 0.3]))
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_handles_missing_properties_and_vector():
|
||||
store = _store_with_pages(_page([_obj("uuid-1", properties=None, vector=None)]))
|
||||
|
||||
item = list(store.iter_all())[0]
|
||||
|
||||
assert item["metadata"] == {}
|
||||
assert item["vector"] is None
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_treats_empty_vector_as_none():
|
||||
store = _store_with_pages(_page([_obj("uuid-1", vector=[])]))
|
||||
|
||||
assert list(store.iter_all())[0]["vector"] is None
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_requests_vectors():
|
||||
"""Weaviate omits vectors unless include_vector is set."""
|
||||
store = _store_with_pages(_page([]))
|
||||
|
||||
list(store.iter_all(batch_size=64))
|
||||
|
||||
kwargs = store.collection.query.fetch_objects.call_args[1]
|
||||
assert kwargs["include_vector"] is True
|
||||
assert kwargs["limit"] == 64
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_falls_back_to_offset_when_after_unsupported():
|
||||
"""Older clients reject `after`; the scan degrades to numeric offset."""
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
seen = {"calls": 0}
|
||||
|
||||
def _fetch(**kwargs):
|
||||
if "after" in kwargs:
|
||||
raise TypeError("unexpected keyword argument 'after'")
|
||||
seen["calls"] += 1
|
||||
if seen["calls"] == 1:
|
||||
return _page([_obj("uuid-1"), _obj("uuid-2")])
|
||||
return _page([_obj("uuid-3")])
|
||||
|
||||
store.collection.query.fetch_objects.side_effect = _fetch
|
||||
|
||||
result = list(store.iter_all(batch_size=2))
|
||||
|
||||
assert [item["id"] for item in result] == ["uuid-1", "uuid-2", "uuid-3"]
|
||||
offsets = [
|
||||
c[1]["offset"]
|
||||
for c in store.collection.query.fetch_objects.call_args_list
|
||||
if "offset" in c[1]
|
||||
]
|
||||
assert offsets == [2]
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_raises_when_collection_not_initialized():
|
||||
"""Must fail loudly, not yield nothing.
|
||||
|
||||
An empty scan is indistinguishable from an empty source, which would let
|
||||
`store migrate` report success having copied nothing (issue #1083).
|
||||
"""
|
||||
store = WeaviateStore()
|
||||
|
||||
with pytest.raises(ProcessingError, match="Collection not initialized"):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", False)
|
||||
def test_iter_all_raises_when_weaviate_unavailable():
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
|
||||
with pytest.raises(ProcessingError):
|
||||
list(store.iter_all())
|
||||
|
||||
|
||||
@patch("semantica.vector_store.weaviate_store.WEAVIATE_AVAILABLE", True)
|
||||
def test_iter_all_propagates_fetch_errors():
|
||||
store = WeaviateStore()
|
||||
store.collection = MagicMock()
|
||||
store.collection.query.fetch_objects.side_effect = RuntimeError("connection reset")
|
||||
|
||||
with pytest.raises(RuntimeError, match="connection reset"):
|
||||
list(store.iter_all())
|
||||
Reference in New Issue
Block a user