From af932fc40e3da0b1615916f2a29fc709e11f3376 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Mon, 28 Sep 2026 14:51:02 +0900 Subject: [PATCH 1/7] feat: add get_by_ids, reading documents back by item ID Envector.get_by_ids(ids, /, *, partition_name=None) returns a Document for every id that names a live item, in request order. Non-item ids, unknown ids and deleted ids are left out rather than raised, as LangChain's contract asks. It uses pyenvector's Index.get_by_ids, which reads stored metadata by item_id without a search, so a document is readable as soon as add_texts returns and stops being readable as soon as delete returns. Item ids are unique within a partition only: a document added under a named partition must be read with that partition_name, otherwise the same number in the default partition is a different document. The search result parsing is shared with get_by_ids through _stored_document. has_get_by_ids is now True; three of the four standard tests pass and test_add_documents_with_existing_ids is xfail because a caller-chosen id cannot be created. Requires a pyenvector with Index.get_by_ids; the version pin is raised once that SDK release exists. Co-Authored-By: Claude Fable 5.1 --- README.md | 12 +- .../langchain_envector/vectorstore.py | 84 ++++++++++--- tests/conftest.py | 33 ++++- tests/integration_tests/test_get_by_ids.py | 112 +++++++++++++++++ tests/integration_tests/test_vectorstore.py | 10 +- tests/test_vectorstore.py | 117 ++++++++++++++++++ 6 files changed, 343 insertions(+), 25 deletions(-) create mode 100644 tests/integration_tests/test_get_by_ids.py diff --git a/README.md b/README.md index 83960ee..d5e8f55 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ Encrypted vector search for LangChain using Envector, powered by homomorphic enc - LangChain `VectorStore` interface with `similarity_search`, `from_texts`, etc. - Optional `VectorStoreRetriever` helper for quick RAG integrations. - Client-side encryption handled transparently by the SDK, including score thresholds and filtering. -- In-place `delete`, `update_documents` and `upsert_documents` by item ID, plus named partitions. +- In-place `delete`, `update_documents` and `upsert_documents` by item ID, `get_by_ids` to read documents back, plus named partitions. Requires `pyenvector >= 1.6.2`. @@ -42,7 +42,8 @@ Key dataclasses live in `libs/envector/config.py`: ## Limitations - Item IDs are issued by the server (positive integers, returned as strings). Pass them back to `delete`, `update_documents`, `upsert_documents`, or as `ids` to `add_documents` to update in place — any numeric id is taken to be one of them. Other IDs cannot be created; such rows get server-issued IDs and a `UserWarning`. -- Fetch-by-ID (`get_by_ids`) is unsupported, and so is LangChain's `indexing` API, which depends on its own IDs. +- LangChain's `indexing` API is unsupported, since it depends on its own IDs. +- Item IDs are unique within a partition only; pass `partition_name` to `get_by_ids` for documents added to a named partition. - Embeddings must be unit norm: scores are inner products computed under encryption, and vectors with components outside [-1, 1] rank incorrectly. - A row deleted moments ago can still take a top-k slot briefly, so a search right after `delete` may return fewer than `k`; pass `fetch_k` to over-fetch. - Filtering happens client-side after the server returns `k` hits, so filtered results can be fewer than `k`; set `fetch_k` (or `IndexSettings.fetch_k`) to over-fetch. @@ -177,6 +178,13 @@ result = store.upsert_documents( print(result["inserted_item_ids"]) # IDs issued for the ID-less entries ``` +### Fetch by ID + +```python +docs = store.get_by_ids(ids) # Documents for the IDs that exist; missing IDs are skipped +docs = store.get_by_ids(ids, partition_name="tenant_a") # rows in a named partition +``` + ### Delete ```python diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index a8b64ea..88ef13a 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -1,7 +1,7 @@ from __future__ import annotations import warnings -from typing import Any, Dict, Iterable, List, Optional, Tuple +from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple from langchain_core.documents import Document from langchain_core.vectorstores import VectorStore @@ -90,6 +90,30 @@ def _chunked(items: List[Any], size: int) -> Iterable[List[Any]]: yield items[start : start + size] +def _stored_document(item: Dict[str, Any]) -> Tuple[Document, bool]: + """Turn an SDK result dict (search hit or ``get_by_ids`` entry) into a Document. + + The payload is the JSON envelope ``{"text": ..., "metadata": {...}}`` that + `add_texts` stores; anything else is taken as the document text. Returns the + Document and whether the payload carried any content. + """ + # Metadata encryption/decryption is handled by the SDK. Envector stores a + # single string per item; `unpack_metadata` also accepts the dict the SDK + # returns once it has decrypted and parsed that string. + md_obj = unpack_metadata(item.get("metadata")) + text = md_obj.get("text", "") if "_raw" not in md_obj else md_obj["_raw"] + metadata = md_obj.get("metadata", {}) if "_raw" not in md_obj else {} + if text is None: + text = "" + doc_id = item.get("id") + doc = Document( + page_content=text, + metadata=metadata, + id=str(doc_id) if doc_id is not None else None, + ) + return doc, bool(text or metadata) + + def _is_empty_shard_list_error(exc: Exception) -> bool: """True for the backend's "index has no shards" answer to a search. @@ -394,6 +418,41 @@ def delete( ) return True + def get_by_ids( + self, ids: Sequence[str], /, *, partition_name: Optional[str] = None + ) -> List[Document]: + """Read documents by item ID, without a search. + + Takes the IDs `add_texts` / `add_documents` return (or a search + result's ``Document.id``), as ``str`` or ``int``. Every live item comes + back as a ``Document`` whose ``id`` is its item ID, in the order of + ``ids``; repeated IDs are read once. IDs that match no live row — never + issued, deleted, or not enVector item IDs at all — are left out rather + than raised, as LangChain's contract asks. + + A document is readable as soon as `add_texts` returns and stops being + readable as soon as `delete` returns; neither waits for a merge. + + Item IDs are unique within a partition only: pass the ``partition_name`` + a document was added under. Without it the default partition is read, + where the same ID may be a different document. + """ + item_ids = list( + dict.fromkeys(i for i in _split_caller_ids(list(ids))[0] if i is not None) + ) + if not item_ids: + return [] + index = self.client.index + docs: List[Document] = [] + for chunk in _chunked(item_ids, MAX_MUTATION_ITEMS_PER_CALL): + for item in index.get_by_ids( + chunk, + output_fields=self.config.index.output_fields, + partition_name=partition_name, + ): + docs.append(_stored_document(item)[0]) + return docs + # ------------------------------- # In-place mutation (pyenvector >= 1.6.0) # ------------------------------- @@ -729,38 +788,23 @@ def _similarity_search_with_scores( for item in result: # item = {"id": ..., "score": float, "metadata": [str] or {...}} score = float(item.get("score", 0.0)) - md_obj_raw = item.get("metadata") - if md_obj_raw in (None, "", [], {}): + if item.get("metadata") in (None, "", [], {}): # Skip placeholder/empty hits returned by the backend. continue - - # Metadata encryption/decryption is handled by the SDK. - # Envector currently supports a single associated data field (string). - # Convention: if the string is JSON like {"text": str, "metadata": {...}}, - # we unpack it; otherwise, we treat the raw string as the document text. - md_obj = unpack_metadata(md_obj_raw) - - text = md_obj.get("text", "") if "_raw" not in md_obj else md_obj["_raw"] - metadata = md_obj.get("metadata", {}) if "_raw" not in md_obj else {} - if not text and not metadata: + doc, has_content = _stored_document(item) + if not has_content: # Treat empty text+metadata as no result. continue # client-side filter if filter: # simple dict-equality filter on top-level user metadata - matched = all(metadata.get(k) == v for k, v in filter.items()) + matched = all(doc.metadata.get(k) == v for k, v in filter.items()) if not matched: continue if score_threshold is not None and score < score_threshold: continue - doc_id = item.get("id") - doc = Document( - page_content=text, - metadata=metadata, - id=str(doc_id) if doc_id is not None else None, - ) docs_with_scores.append((doc, score)) # Trim to k after filtering diff --git a/tests/conftest.py b/tests/conftest.py index 2695286..d6789a6 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -38,6 +38,9 @@ class FakeIndex: load_calls: int = 0 next_item_id: int = 1 row_count: int = 0 + # (partition_name, item_id) -> stored metadata string, for get_by_ids. + stored: Dict[Any, str] = field(default_factory=dict) + fetched: List[Dict[str, Any]] = field(default_factory=list) def load(self): self.load_calls += 1 @@ -87,7 +90,10 @@ def insert( if request_ids is not None: request_ids.append(f"req-ins-{len(self.inserted)}") self.row_count += len(metadata) - return self._issue_ids(len(metadata)) + ids = self._issue_ids(len(metadata)) + for i, m in zip(ids, metadata): + self.stored[(partition_name, i)] = m + return ids def wait_for_insert_stage( self, @@ -124,8 +130,33 @@ def delete( } ) self.row_count = max(0, self.row_count - len(item_ids)) + for i in item_ids: + self.stored.pop((partition_name, i), None) return f"req-del-{len(self.deleted)}" + def get_by_ids( + self, + item_ids: List[int], + output_fields: Optional[List[str]] = None, + partition_name: Optional[str] = None, + ) -> List[Dict[str, Any]]: + self.fetched.append( + { + "item_ids": list(item_ids), + "output_fields": output_fields, + "partition_name": partition_name, + } + ) + return [ + { + "id": i, + "metadata": self.stored[(partition_name, i)] if output_fields else "", + "partition_name": partition_name or "", + } + for i in item_ids + if (partition_name, i) in self.stored + ] + def update( self, items: List[Any], diff --git a/tests/integration_tests/test_get_by_ids.py b/tests/integration_tests/test_get_by_ids.py new file mode 100644 index 0000000..a1bfc57 --- /dev/null +++ b/tests/integration_tests/test_get_by_ids.py @@ -0,0 +1,112 @@ +"""Integration coverage for `get_by_ids` against a live server. + +The LangChain standard tests cover the plain read-back. This adds what they do +not: reading right after an un-awaited insert, a delete becoming invisible as +soon as it returns, named partitions, and an index with metadata encryption. + +Run with: + ENVECTOR_ADDRESS=host:port ENVECTOR_KEY_PATH=./keys ENVECTOR_KEY_ID=my_key \\ + pytest -q -m integration tests/integration_tests/test_get_by_ids.py +""" + +from __future__ import annotations + +import os +import secrets +from typing import Generator, List + +import pytest + +from langchain_envector.config import ( + ConnectionConfig, + EnvectorConfig, + IndexSettings, + KeyConfig, + WriteSettings, +) +from langchain_envector.vectorstore import Document, Envector + +pytestmark = pytest.mark.integration + +DIM = 32 + + +def _require_env(name: str) -> str: + value = os.environ.get(name) + if not value: + pytest.skip(f"Set {name} to enable integration test") + return value + + +def _unit_vector(pos: int) -> List[float]: + vec = [0.0] * DIM + vec[pos % DIM] = 1.0 + return vec + + +@pytest.fixture(params=[False, True], ids=["plain-metadata", "encrypted-metadata"]) +def store(request) -> Generator[Envector, None, None]: + name = f"lc_gbi_{secrets.token_hex(4)}" + cfg = EnvectorConfig( + connection=ConnectionConfig(address=_require_env("ENVECTOR_ADDRESS")), + key=KeyConfig( + key_path=_require_env("ENVECTOR_KEY_PATH"), + key_id=_require_env("ENVECTOR_KEY_ID"), + ), + index=IndexSettings( + index_name=name, dim=DIM, metadata_encryption=request.param + ), + # Nothing here waits for a merge: get_by_ids must not need one. + write=WriteSettings(await_insert=False, await_delete=False), + create_if_missing=True, + ) + store = Envector(config=cfg) + try: + yield store + finally: + try: + store.client.ev.delete_index(name) + except Exception: + pass + + +def _add(store: Envector, texts: List[str], **kwargs) -> List[str]: + return store.add_texts( + texts, + metadatas=[{"n": i} for i in range(len(texts))], + vectors=[_unit_vector(i) for i in range(len(texts))], + **kwargs, + ) + + +def test_readable_right_after_insert_and_gone_right_after_delete( + store: Envector, +) -> None: + ids = _add(store, ["a", "b", "c"]) + + docs = store.get_by_ids(ids) + assert docs == [ + Document(page_content=t, metadata={"n": i}, id=ids[i]) + for i, t in enumerate("abc") + ] + + # Missing and foreign ids are left out; order follows the request. + got = store.get_by_ids([ids[2], "999999", "not-an-id", ids[0]]) + assert [d.id for d in got] == [ids[2], ids[0]] + + store.delete([ids[1]]) + assert [d.id for d in store.get_by_ids(ids)] == [ids[0], ids[2]] + + +def test_named_partition_needs_partition_name(store: Envector) -> None: + default_ids = _add(store, ["default doc"]) + store.create_partition("tenant_a") + tenant_ids = _add(store, ["tenant doc"], partition_name="tenant_a") + + in_tenant = store.get_by_ids(tenant_ids, partition_name="tenant_a") + assert [d.page_content for d in in_tenant] == ["tenant doc"] + + # Item ids restart per partition: the same number in the default partition + # is a different document, which is why partition_name matters. + assert tenant_ids == default_ids + assert [d.page_content for d in store.get_by_ids(tenant_ids)] == ["default doc"] diff --git a/tests/integration_tests/test_vectorstore.py b/tests/integration_tests/test_vectorstore.py index 6c4aede..41b9143 100644 --- a/tests/integration_tests/test_vectorstore.py +++ b/tests/integration_tests/test_vectorstore.py @@ -64,8 +64,7 @@ def has_async(self) -> bool: @property def has_get_by_ids(self) -> bool: - # Envector does not yet support get by IDs. - return False + return True @pytest.fixture() def vectorstore(self) -> Generator[VectorStore, None, None]: # type: ignore[override] @@ -117,5 +116,12 @@ def test_deleting_documents(self, vectorstore: VectorStore) -> None: def test_deleting_bulk_documents(self, vectorstore: VectorStore) -> None: super().test_deleting_bulk_documents(vectorstore) + @pytest.mark.xfail( + reason="enVector issues item IDs; a caller-chosen id such as 'foo' " + "cannot be created, so it is not among the returned ids." + ) + def test_add_documents_with_existing_ids(self, vectorstore: VectorStore) -> None: + super().test_add_documents_with_existing_ids(vectorstore) + # Async standard tests are not overridden: has_async=False makes the base # class skip them. diff --git a/tests/test_vectorstore.py b/tests/test_vectorstore.py index 29a21e5..113e822 100644 --- a/tests/test_vectorstore.py +++ b/tests/test_vectorstore.py @@ -1269,3 +1269,120 @@ async def test_async_relevance_scores_match_sync(): assert [(d.page_content, s) for d, s in async_pairs] == [ (d.page_content, s) for d, s in sync_pairs ] + + +def test_get_by_ids_reads_documents_back_in_request_order(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + ids = store.add_texts(["a", "b", "c"], metadatas=[{"k": 1}, {"k": 2}, {"k": 3}]) + + docs = store.get_by_ids([ids[2], ids[0]]) + + assert docs == [ + LC_Document(page_content="c", metadata={"k": 3}, id=ids[2]), + LC_Document(page_content="a", metadata={"k": 1}, id=ids[0]), + ] + assert client.index.fetched[-1]["output_fields"] == ["metadata"] + + +def test_get_by_ids_leaves_out_ids_it_cannot_find_without_raising(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + ids = store.add_texts(["a"]) + + docs = store.get_by_ids(["uuid-like", "0", "-3", "999", ids[0], ids[0], 1]) + + assert [d.id for d in docs] == [ids[0]] + # Non-item IDs never reach the SDK, and repeats are sent once. + assert client.index.fetched[-1]["item_ids"] == [999, 1] + + +def test_get_by_ids_empty_or_foreign_only_makes_no_call(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + assert store.get_by_ids([]) == [] + assert store.get_by_ids(["foo", "bar"]) == [] + assert client.index.fetched == [] + + +def test_get_by_ids_does_not_see_deleted_documents(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + ids = store.add_texts(["a", "b"]) + store.delete([ids[0]]) + assert [d.id for d in store.get_by_ids(ids)] == [ids[1]] + + +def test_get_by_ids_reads_the_named_partition(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + ids = store.add_texts(["tenant doc"], partition_name="tenant_a") + + assert store.get_by_ids(ids) == [] + docs = store.get_by_ids(ids, partition_name="tenant_a") + assert [d.page_content for d in docs] == ["tenant doc"] + assert client.index.fetched[-1]["partition_name"] == "tenant_a" + + +def test_get_by_ids_keeps_a_live_row_with_no_stored_content(): + client = FakeClient() + index = client.index + index.stored[(None, 7)] = "" + index.stored[(None, 8)] = "not an envelope" + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + + docs = store.get_by_ids(["7", "8"]) + + assert docs == [ + LC_Document(page_content="", metadata={}, id="7"), + LC_Document(page_content="not an envelope", metadata={}, id="8"), + ] + + +def test_get_by_ids_accepts_already_decrypted_payloads(): + # With metadata encryption on, the SDK hands back the parsed envelope. + client = FakeClient() + client.index.get_by_ids = lambda item_ids, **kw: [ + {"id": 5, "metadata": {"text": "t", "metadata": {"m": 1}}, "partition_name": ""} + ] + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + assert store.get_by_ids(["5"]) == [ + LC_Document(page_content="t", metadata={"m": 1}, id="5") + ] + + +def test_get_by_ids_splits_above_the_per_call_cap(monkeypatch): + from langchain_envector import vectorstore as vs + + monkeypatch.setattr(vs, "MAX_MUTATION_ITEMS_PER_CALL", 2) + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + ids = store.add_texts(["a", "b", "c", "d", "e"]) + + docs = store.get_by_ids(ids) + + assert [d.id for d in docs] == ids + assert [f["item_ids"] for f in client.index.fetched] == [[1, 2], [3, 4], [5]] + + +def test_get_by_ids_does_not_load_the_index(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + client.index.stored[(None, 1)] = '{"text": "x", "metadata": {}}' + store.get_by_ids(["1"]) + assert client.index.load_calls == 0 + + +def test_stored_null_text_reads_as_empty_document(): + # An envelope whose text is JSON null (a foreign writer) yields an empty page, + # not a pydantic error, in both search and get_by_ids. + client = FakeClient() + client.index.stored[(None, 1)] = '{"text": null, "metadata": {"k": 1}}' + client.index.search_payload = [ + [{"id": 1, "score": 0.5, "metadata": client.index.stored[(None, 1)]}] + ] + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + + expected = [LC_Document(page_content="", metadata={"k": 1}, id="1")] + assert store.get_by_ids(["1"]) == expected + assert store.similarity_search("q", k=1) == expected From f38186abaf57bf3eec056d31caa9ec5b1968f2b2 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Tue, 29 Sep 2026 13:54:49 +0900 Subject: [PATCH 2/7] refactor: let the SDK split get_by_ids requests Index.get_by_ids already splits the ids at the server's per-call cap, so the wrapper's own 10,000-id loop did nothing. get_by_ids now passes all ids in one call, and the unit test for the wrapper-side split is removed; the SDK's tests cover the split. Co-Authored-By: Claude Opus 5.5 (1M context) --- libs/envector/langchain_envector/vectorstore.py | 17 +++++++---------- tests/test_vectorstore.py | 14 -------------- 2 files changed, 7 insertions(+), 24 deletions(-) diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index 88ef13a..9a52c13 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -442,16 +442,13 @@ def get_by_ids( ) if not item_ids: return [] - index = self.client.index - docs: List[Document] = [] - for chunk in _chunked(item_ids, MAX_MUTATION_ITEMS_PER_CALL): - for item in index.get_by_ids( - chunk, - output_fields=self.config.index.output_fields, - partition_name=partition_name, - ): - docs.append(_stored_document(item)[0]) - return docs + # The SDK splits the request at the server's per-call cap. + items = self.client.index.get_by_ids( + item_ids, + output_fields=self.config.index.output_fields, + partition_name=partition_name, + ) + return [_stored_document(item)[0] for item in items] # ------------------------------- # In-place mutation (pyenvector >= 1.6.0) diff --git a/tests/test_vectorstore.py b/tests/test_vectorstore.py index 113e822..4cbbb49 100644 --- a/tests/test_vectorstore.py +++ b/tests/test_vectorstore.py @@ -1351,20 +1351,6 @@ def test_get_by_ids_accepts_already_decrypted_payloads(): ] -def test_get_by_ids_splits_above_the_per_call_cap(monkeypatch): - from langchain_envector import vectorstore as vs - - monkeypatch.setattr(vs, "MAX_MUTATION_ITEMS_PER_CALL", 2) - client = FakeClient() - store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) - ids = store.add_texts(["a", "b", "c", "d", "e"]) - - docs = store.get_by_ids(ids) - - assert [d.id for d in docs] == ids - assert [f["item_ids"] for f in client.index.fetched] == [[1, 2], [3, 4], [5]] - - def test_get_by_ids_does_not_load_the_index(): client = FakeClient() store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) From a64d1911a3a20a8eab6078c33e245a170c1b52c9 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Tue, 29 Sep 2026 16:23:10 +0900 Subject: [PATCH 3/7] docs: say how get_by_ids liveness differs from search visibility Review on envector-msa #2565 asked that the difference be written where users read it: a document is readable by id before a merge makes it searchable, and right after an update the new content is readable while search may still score the old vector. Both are expected. Co-Authored-By: Claude Fable 5.1 --- libs/envector/langchain_envector/vectorstore.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index 9a52c13..ce6d8c4 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -430,8 +430,12 @@ def get_by_ids( issued, deleted, or not enVector item IDs at all — are left out rather than raised, as LangChain's contract asks. - A document is readable as soon as `add_texts` returns and stops being - readable as soon as `delete` returns; neither waits for a merge. + Liveness here is the row's own state, not search visibility. A document + is readable as soon as `add_texts` returns, before any merge, so it can + come back from ``get_by_ids`` while ``similarity_search`` does not find + it yet; right after `update_documents` this returns the new content + while search may still score the old vector. A deleted document stops + being readable as soon as `delete` returns. These differences are expected. Item IDs are unique within a partition only: pass the ``partition_name`` a document was added under. Without it the default partition is read, From 652e876702bc4717c802dcd8867b2901b6b6a86d Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Thu, 1 Oct 2026 14:13:11 +0900 Subject: [PATCH 4/7] feat: raise NotImplementedError from get_by_ids on a pyenvector without it Index.get_by_ids ships in a pyenvector release that is not out yet, while this package accepts pyenvector>=1.6.2. On 1.6.2 get_by_ids failed with "'Index' object has no attribute 'get_by_ids'". Check the installed SDK first and raise NotImplementedError, LangChain's usual answer for an unsupported get_by_ids, saying to upgrade pyenvector. SDK_HAS_GET_BY_IDS drives has_get_by_ids in the standard tests and skips tests/integration_tests/test_get_by_ids.py, so the suites skip instead of failing on an older SDK. README Limitations gets one line. The pin and the README "Requires" line move once the release exists. Co-Authored-By: Claude Opus 5.5 (1M context) --- README.md | 1 + libs/envector/langchain_envector/vectorstore.py | 12 +++++++++++- tests/integration_tests/test_get_by_ids.py | 11 ++++++++--- tests/integration_tests/test_vectorstore.py | 5 +++-- tests/test_vectorstore.py | 14 ++++++++++++++ 5 files changed, 37 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index d5e8f55..b810f17 100644 --- a/README.md +++ b/README.md @@ -44,6 +44,7 @@ Key dataclasses live in `libs/envector/config.py`: - Item IDs are issued by the server (positive integers, returned as strings). Pass them back to `delete`, `update_documents`, `upsert_documents`, or as `ids` to `add_documents` to update in place — any numeric id is taken to be one of them. Other IDs cannot be created; such rows get server-issued IDs and a `UserWarning`. - LangChain's `indexing` API is unsupported, since it depends on its own IDs. - Item IDs are unique within a partition only; pass `partition_name` to `get_by_ids` for documents added to a named partition. +- `get_by_ids` needs a pyenvector release that provides `Index.get_by_ids`; with an earlier pyenvector it raises `NotImplementedError`. - Embeddings must be unit norm: scores are inner products computed under encryption, and vectors with components outside [-1, 1] rank incorrectly. - A row deleted moments ago can still take a top-k slot briefly, so a search right after `delete` may return fewer than `k`; pass `fetch_k` to over-fetch. - Filtering happens client-side after the server returns `k` hits, so filtered results can be fewer than `k`; set `fetch_k` (or `IndexSettings.fetch_k`) to over-fetch. diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index ce6d8c4..b3c829a 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -7,11 +7,15 @@ from langchain_core.vectorstores import VectorStore from pyenvector import UpdateItem, UpsertItem from pyenvector.index.index import MAX_MUTATION_ITEMS_PER_CALL +from pyenvector.index.index import Index as _SdkIndex from .config import EnvectorConfig from .client import EnvectorClient from .types import Embeddings, as_embeddings, pack_metadata, unpack_metadata +# Index.get_by_ids is newer than the oldest pyenvector this package accepts. +SDK_HAS_GET_BY_IDS = hasattr(_SdkIndex, "get_by_ids") + def _mutation_items( item_ids: List[Any], label: str, *, dedupe: bool = False @@ -441,13 +445,19 @@ def get_by_ids( a document was added under. Without it the default partition is read, where the same ID may be a different document. """ + index = self.client.index + if not hasattr(index, "get_by_ids"): + raise NotImplementedError( + "Envector.get_by_ids needs a pyenvector release that provides " + "Index.get_by_ids; upgrade pyenvector." + ) item_ids = list( dict.fromkeys(i for i in _split_caller_ids(list(ids))[0] if i is not None) ) if not item_ids: return [] # The SDK splits the request at the server's per-call cap. - items = self.client.index.get_by_ids( + items = index.get_by_ids( item_ids, output_fields=self.config.index.output_fields, partition_name=partition_name, diff --git a/tests/integration_tests/test_get_by_ids.py b/tests/integration_tests/test_get_by_ids.py index a1bfc57..73946ab 100644 --- a/tests/integration_tests/test_get_by_ids.py +++ b/tests/integration_tests/test_get_by_ids.py @@ -24,9 +24,14 @@ KeyConfig, WriteSettings, ) -from langchain_envector.vectorstore import Document, Envector - -pytestmark = pytest.mark.integration +from langchain_envector.vectorstore import SDK_HAS_GET_BY_IDS, Document, Envector + +pytestmark = [ + pytest.mark.integration, + pytest.mark.skipif( + not SDK_HAS_GET_BY_IDS, reason="installed pyenvector has no Index.get_by_ids" + ), +] DIM = 32 diff --git a/tests/integration_tests/test_vectorstore.py b/tests/integration_tests/test_vectorstore.py index 41b9143..92f9e72 100644 --- a/tests/integration_tests/test_vectorstore.py +++ b/tests/integration_tests/test_vectorstore.py @@ -17,7 +17,7 @@ IndexSettings, KeyConfig, ) -from langchain_envector.vectorstore import Envector +from langchain_envector.vectorstore import SDK_HAS_GET_BY_IDS, Envector pytestmark = pytest.mark.integration @@ -64,7 +64,8 @@ def has_async(self) -> bool: @property def has_get_by_ids(self) -> bool: - return True + # Follows the installed SDK: skipped where pyenvector lacks Index.get_by_ids. + return SDK_HAS_GET_BY_IDS @pytest.fixture() def vectorstore(self) -> Generator[VectorStore, None, None]: # type: ignore[override] diff --git a/tests/test_vectorstore.py b/tests/test_vectorstore.py index 4cbbb49..4631df9 100644 --- a/tests/test_vectorstore.py +++ b/tests/test_vectorstore.py @@ -1372,3 +1372,17 @@ def test_stored_null_text_reads_as_empty_document(): expected = [LC_Document(page_content="", metadata={"k": 1}, id="1")] assert store.get_by_ids(["1"]) == expected assert store.similarity_search("q", k=1) == expected + + +def test_get_by_ids_needs_an_sdk_that_has_it(): + # A pyenvector release without Index.get_by_ids gets a clear + # NotImplementedError, LangChain's usual answer, not an AttributeError. + client = FakeClient() + + class _OldIndex(FakeIndex): + get_by_ids = property(lambda self: (_ for _ in ()).throw(AttributeError)) + + client._index = _OldIndex() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + with pytest.raises(NotImplementedError, match="upgrade pyenvector"): + store.get_by_ids(["1"]) From 298a173f7b670239cedade60aeb17ae7a4ad69c9 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Thu, 1 Oct 2026 15:21:44 +0900 Subject: [PATCH 5/7] fix: get_by_ids reads only the item IDs the caller named MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit get_by_ids turned ids into int(...) through the add_texts helper, so True read item 1 and 3.9 or 3.0 read item 3 — a document the caller did not ask for. pyenvector's own check rejects bool and float, but it only ever saw the already-converted int. _readable_item_id accepts a positive int or its decimal string and nothing else; any other value names no item and is left out, as LangChain's get_by_ids contract asks. Unit tests cover bool, float, "3.0", negatives, zero, blanks, None, bytes and lists. The same conversion in delete / update_* / upsert_documents and in the ids of add_texts predates this PR and is fixed in the next one. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../langchain_envector/vectorstore.py | 27 +++++++++++++++++-- tests/test_vectorstore.py | 24 +++++++++++++++++ 2 files changed, 49 insertions(+), 2 deletions(-) diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index b3c829a..c6b5f7f 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -70,6 +70,26 @@ def _split_caller_ids(ids: List[Any]) -> Tuple[List[Optional[int]], List[Any]]: return item_ids, foreign +def _readable_item_id(value: Any) -> Optional[int]: + """The item ID ``value`` names exactly, or ``None`` when it names none. + + For `get_by_ids`, which must never read an item the caller did not name: + only a positive ``int`` or a decimal string of one counts. ``bool`` and + ``float`` are not item IDs — ``int(True)`` is 1 and ``int(3.9)`` is 3, so + coercing them would return a different document. + """ + if isinstance(value, bool): + return None + if isinstance(value, int): + return value if value > 0 else None + if isinstance(value, str): + text = value.strip() + if text.isdecimal(): + item_id = int(text) + return item_id if item_id > 0 else None + return None + + def _one_embedding_arg(embedding: Any, embeddings: Any) -> Any: """Resolve the standard positional ``embedding`` and our older ``embeddings=`` keyword into one value, rejecting conflicting pairs.""" @@ -428,7 +448,8 @@ def get_by_ids( """Read documents by item ID, without a search. Takes the IDs `add_texts` / `add_documents` return (or a search - result's ``Document.id``), as ``str`` or ``int``. Every live item comes + result's ``Document.id``), as ``str`` or ``int``; any other value, + ``bool`` and ``float`` included, names no item and is left out. Every live item comes back as a ``Document`` whose ``id`` is its item ID, in the order of ``ids``; repeated IDs are read once. IDs that match no live row — never issued, deleted, or not enVector item IDs at all — are left out rather @@ -452,7 +473,9 @@ def get_by_ids( "Index.get_by_ids; upgrade pyenvector." ) item_ids = list( - dict.fromkeys(i for i in _split_caller_ids(list(ids))[0] if i is not None) + dict.fromkeys( + i for i in (_readable_item_id(x) for x in ids) if i is not None + ) ) if not item_ids: return [] diff --git a/tests/test_vectorstore.py b/tests/test_vectorstore.py index 4631df9..9bbedbd 100644 --- a/tests/test_vectorstore.py +++ b/tests/test_vectorstore.py @@ -1386,3 +1386,27 @@ class _OldIndex(FakeIndex): store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) with pytest.raises(NotImplementedError, match="upgrade pyenvector"): store.get_by_ids(["1"]) + + +@pytest.mark.parametrize( + "not_an_id", [True, False, 3.9, 3.0, "3.0", "-3", "0", " ", None, b"3", [3]] +) +def test_get_by_ids_never_reads_an_item_the_caller_did_not_name(not_an_id): + # int(True) is 1 and int(3.9) is 3: coercing would return another document. + # Such values name no item, so they are left out, as LangChain asks. + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + store.add_texts(["a", "b", "c"]) + + assert store.get_by_ids([not_an_id]) == [] + assert client.index.fetched == [] + + +def test_get_by_ids_accepts_ints_and_decimal_strings(): + client = FakeClient() + store = Envector(config=_cfg(), embeddings=FakeEmbeddings(dim=4), client=client) + store.add_texts(["a", "b", "c"]) + + docs = store.get_by_ids([3, " 2 ", "1", 3.9, True]) + assert [d.page_content for d in docs] == ["c", "b", "a"] + assert client.index.fetched[-1]["item_ids"] == [3, 2, 1] From 5de00b1c23fa5b96836505d77945c8a92a14fcd0 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Thu, 1 Oct 2026 15:21:44 +0900 Subject: [PATCH 6/7] docs: list get_by_ids among the methods that take item IDs The Limitations line on item IDs listed delete / update / upsert / add_documents but not get_by_ids, and said "any numeric id is taken to be one of them", which get_by_ids no longer does for 3.0 or 3.9. Name the accepted forms (the returned strings, or ints) and say that a non-item ID such as a UUID is never found by get_by_ids. Co-Authored-By: Claude Opus 5.5 (1M context) --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index b810f17..9b378b1 100644 --- a/README.md +++ b/README.md @@ -41,7 +41,7 @@ Key dataclasses live in `libs/envector/config.py`: - Client-side filtering requires the JSON envelope to include an object under `metadata`. ## Limitations -- Item IDs are issued by the server (positive integers, returned as strings). Pass them back to `delete`, `update_documents`, `upsert_documents`, or as `ids` to `add_documents` to update in place — any numeric id is taken to be one of them. Other IDs cannot be created; such rows get server-issued IDs and a `UserWarning`. +- Item IDs are issued by the server (positive integers, returned as strings). Pass them back — as those strings or as ints — to `get_by_ids`, `delete`, `update_documents`, `upsert_documents`, or as `ids` to `add_documents` to update in place. Other IDs, such as UUIDs, cannot be created; such rows get server-issued IDs and a `UserWarning`, and `get_by_ids` never finds them. - LangChain's `indexing` API is unsupported, since it depends on its own IDs. - Item IDs are unique within a partition only; pass `partition_name` to `get_by_ids` for documents added to a named partition. - `get_by_ids` needs a pyenvector release that provides `Index.get_by_ids`; with an earlier pyenvector it raises `NotImplementedError`. From 35bfa33a829ca27e13fe70b8a60793d952707703 Mon Sep 17 00:00:00 2001 From: Minseok Park Date: Thu, 1 Oct 2026 16:12:14 +0900 Subject: [PATCH 7/7] docs: correct the get_by_ids docstring on updates and trim the ID sentence The docstring said that right after update_documents search "may still score the old vector". It does not: a vector update deprecates the old shard slot in the same transaction that writes the new metadata, so until the new vector is searchable search leaves the document out rather than ranking it by the old vector. update_documents waits for that by default (await_update), so the gap exists only when it is called with await_completion=False. Say that instead. Also drop the list of accepted ID types from the first sentence: the IDs callers pass back are the ones add_texts and search return, and the next sentence already says anything that is not an item ID is left out. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../langchain_envector/vectorstore.py | 21 ++++++++++--------- 1 file changed, 11 insertions(+), 10 deletions(-) diff --git a/libs/envector/langchain_envector/vectorstore.py b/libs/envector/langchain_envector/vectorstore.py index c6b5f7f..6e5f7c7 100644 --- a/libs/envector/langchain_envector/vectorstore.py +++ b/libs/envector/langchain_envector/vectorstore.py @@ -447,20 +447,21 @@ def get_by_ids( ) -> List[Document]: """Read documents by item ID, without a search. - Takes the IDs `add_texts` / `add_documents` return (or a search - result's ``Document.id``), as ``str`` or ``int``; any other value, - ``bool`` and ``float`` included, names no item and is left out. Every live item comes - back as a ``Document`` whose ``id`` is its item ID, in the order of - ``ids``; repeated IDs are read once. IDs that match no live row — never - issued, deleted, or not enVector item IDs at all — are left out rather - than raised, as LangChain's contract asks. + Takes the IDs `add_texts` / `add_documents` return, or a search + result's ``Document.id``. Every live item comes back as a ``Document`` + whose ``id`` is its item ID, in the order of ``ids``; repeated IDs are + read once. IDs that match no live row — never issued, deleted, or not + enVector item IDs at all — are left out rather than raised, as + LangChain's contract asks. Liveness here is the row's own state, not search visibility. A document is readable as soon as `add_texts` returns, before any merge, so it can come back from ``get_by_ids`` while ``similarity_search`` does not find - it yet; right after `update_documents` this returns the new content - while search may still score the old vector. A deleted document stops - being readable as soon as `delete` returns. These differences are expected. + it yet. A deleted document stops being readable as soon as `delete` + returns. After an `update_documents` that replaces the vector and is + not awaited (``await_completion=False``), this returns the new content + while search may leave the document out until the new vector is + searchable. These differences are expected. Item IDs are unique within a partition only: pass the ``partition_name`` a document was added under. Without it the default partition is read,