From d523a39b34b98bd7994a492b63b2799244ab53d9 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 15:02:34 +0000 Subject: [PATCH] Add client.literature (public /v1 API), API-key login and an MCP server - OmniBioAI(api_key="omni_sk_...") (or OMNIBIOAI_API_KEY) sends the key as the bearer token; no refresh applies. access_token and api_key are mutually exclusive; a refresh_token with an API key is rejected. - client.literature.ask(question, study, top_k, mode, hybrid_search, idempotency_key) -> POST {gateway}/v1/literature/answers, always with an Idempotency-Key (generated unless given) so a retry is never billed twice. client.literature.studies() -> GET /v1/literature/studies. - New exceptions: QuotaExceededError (402) and RateLimitError (429, .retry_after from Retry-After), both subclasses of ValidationError so existing handlers keep catching them. The /v1 error envelope's message is used as the exception text. - omnibioai-mcp (optional extra "mcp", mcp>=2,<3): stdio MCP server with answer_with_citations, find_pmids and list_studies tools, configured by OMNIBIOAI_API_KEY and OMNIBIOAI_BASE_URL. `import omnibioai` still needs only requests. Coverage stays at 100% (112 tests). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_016nxi4bp7DP1ancERNUUfzj --- README.md | 34 ++++++++++++ omnibioai/__init__.py | 3 +- omnibioai/_base.py | 22 ++++++++ omnibioai/client.py | 22 +++++++- omnibioai/exceptions.py | 17 ++++++ omnibioai/literature/__init__.py | 3 ++ omnibioai/literature/client.py | 51 ++++++++++++++++++ omnibioai/mcp_server.py | 62 +++++++++++++++++++++ pyproject.toml | 9 ++++ tests/test_literature_client.py | 93 ++++++++++++++++++++++++++++++++ tests/test_mcp_server.py | 56 +++++++++++++++++++ 11 files changed, 370 insertions(+), 2 deletions(-) create mode 100644 omnibioai/literature/__init__.py create mode 100644 omnibioai/literature/client.py create mode 100644 omnibioai/mcp_server.py create mode 100644 tests/test_literature_client.py create mode 100644 tests/test_mcp_server.py diff --git a/README.md b/README.md index f03e910..b6ff6d9 100644 --- a/README.md +++ b/README.md @@ -106,6 +106,38 @@ print(run) > Never point the SDK directly at individual services (auth-service, > workbench etc.) in production. +### Literature AI with an API key + +The public, billable Literature AI API (`/v1/literature` on the gateway) +accepts an `omni_sk_` API key in place of a login token: + +```python +from omnibioai import OmniBioAI + +client = OmniBioAI(api_key="omni_sk_...", base_url="https://") +# or set OMNIBIOAI_API_KEY and call OmniBioAI(base_url=...) + +answer = client.literature.ask("Which genes drive hypereosinophilic syndrome?", top_k=8) +studies = client.literature.studies() # free +``` + +Each `ask()` sends an `Idempotency-Key`; pass your own +`idempotency_key=` when retrying a call whose result you did not see, and +the gateway returns the stored answer without billing it again. Over the +per-key rate limit you get `RateLimitError` (with `retry_after` seconds); +once your organization's included answers run out, `QuotaExceededError`. + +### MCP server (Claude, ChatGPT and other MCP clients) + +```bash +pip install "omnibioai-sdk[mcp]" +OMNIBIOAI_API_KEY=omni_sk_... OMNIBIOAI_BASE_URL=https:// omnibioai-mcp +``` + +Tools: `answer_with_citations`, `find_pmids` and `list_studies` (free). For +a desktop MCP client, register the `omnibioai-mcp` command with those two +environment variables. Answer tools are billed like `literature.ask()`. + ### Getting a token Obtain the access/refresh token pair through the OmniBioAI Auth login or SSO @@ -215,6 +247,8 @@ c = OmniClient() from omnibioai.exceptions import ( AuthenticationError, PermissionDeniedError, + QuotaExceededError, # 402: /v1 API, included usage used up + RateLimitError, # 429: /v1 API, .retry_after seconds ServiceUnavailableError, ) diff --git a/omnibioai/__init__.py b/omnibioai/__init__.py index 9b4bbdc..8a8ec6e 100644 --- a/omnibioai/__init__.py +++ b/omnibioai/__init__.py @@ -10,9 +10,10 @@ """ from .client import OmniBioAI from .legacy import OmniClient +from .literature import LiteratureClient from .models import ModelsClient from .rag import RAGClient from .tes import TESClient from .workflows import WorkflowsClient -__all__ = ["OmniBioAI", "OmniClient", "RAGClient", "ModelsClient", "TESClient", "WorkflowsClient"] +__all__ = ["OmniBioAI", "OmniClient", "LiteratureClient", "RAGClient", "ModelsClient", "TESClient", "WorkflowsClient"] diff --git a/omnibioai/_base.py b/omnibioai/_base.py index 917cf0c..264cce9 100644 --- a/omnibioai/_base.py +++ b/omnibioai/_base.py @@ -10,6 +10,8 @@ AuthenticationError, GatewayError, PermissionDeniedError, + QuotaExceededError, + RateLimitError, ResourceNotFoundError, ServiceUnavailableError, ValidationError, @@ -64,6 +66,16 @@ def _parse(self, response: requests.Response) -> Any: raise AuthenticationError(message, status_code=401, response_body=body, trace_id=trace_id) if response.status_code == 403: raise PermissionDeniedError(message, status_code=403, response_body=body, trace_id=trace_id) + if response.status_code == 402: + raise QuotaExceededError(message, status_code=402, response_body=body, trace_id=trace_id) + if response.status_code == 429: + raise RateLimitError( + message, + retry_after=_retry_after(response), + status_code=429, + response_body=body, + trace_id=trace_id, + ) if response.status_code == 404: raise ResourceNotFoundError(message, status_code=404, response_body=body, trace_id=trace_id) if response.status_code >= 500: @@ -84,12 +96,20 @@ def _safe_json(response: requests.Response) -> Any: return None +def _retry_after(response: requests.Response) -> Optional[int]: + try: + return int(response.headers.get("Retry-After", "")) + except ValueError: + return None + + def _extract_message(body: Any) -> Optional[str]: """Normalizes the ecosystem's several observed error-body shapes into one human-readable message string: - API Gateway: {"error": "...", "reason": "..."} - FastAPI default: {"detail": "..."} (RAG, control-center, TES, most of Model Registry) - Model Registry (DB): {"detail": {"ok": false, "error": "..."}} + - Public /v1 API: {"error": {"type": "...", "message": "...", "request_id": "..."}} See the Phase 1 findings report's Cross-Service Consistency section for the full survey this is built from. Returns None (falls back to response.reason at the call site) if the body doesn't match any @@ -104,6 +124,8 @@ def _extract_message(body: Any) -> Optional[str]: if nested: return str(nested) error = body.get("error") + if isinstance(error, dict): + return str(error.get("message") or error.get("type") or error) if error: reason = body.get("reason") return f"{error}: {reason}" if reason else str(error) diff --git a/omnibioai/client.py b/omnibioai/client.py index 9b9ab80..e25b1f1 100644 --- a/omnibioai/client.py +++ b/omnibioai/client.py @@ -1,10 +1,12 @@ """omnibioai/client.py""" from __future__ import annotations +import os from typing import Optional from .auth.session import AuthenticatedSession from .auth.tokens import TokenPair +from .literature.client import LiteratureClient from .models.client import ModelsClient from .rag.client import RAGClient from .tes.client import TESClient @@ -51,13 +53,29 @@ class OmniBioAI: def __init__( self, - access_token: str, + access_token: Optional[str] = None, refresh_token: Optional[str] = None, base_url: str = DEFAULT_BASE_URL, auth_url: str = DEFAULT_AUTH_URL, workflows_url: Optional[str] = None, timeout: float = 60, + *, + api_key: Optional[str] = None, ) -> None: + # An omni_sk_ API key is sent as the bearer token exactly like an + # access token; the gateway recognises it. Keys never expire into a + # refresh, so no refresh_token applies. Falls back to the + # OMNIBIOAI_API_KEY environment variable when neither is given. + if access_token and api_key: + raise ValueError("Pass either access_token or api_key, not both") + if api_key is None and access_token is None: + api_key = os.environ.get("OMNIBIOAI_API_KEY") + if api_key is not None: + if refresh_token: + raise ValueError("refresh_token does not apply to an API key") + access_token = api_key + if not access_token: + raise ValueError("An access_token or api_key (or OMNIBIOAI_API_KEY) is required") self.base_url = base_url.rstrip("/") self.timeout = timeout self.tokens = TokenPair(access_token=access_token, refresh_token=refresh_token) @@ -65,6 +83,8 @@ def __init__( tokens=self.tokens, auth_url=auth_url, timeout=timeout, ) self.rag = RAGClient(base_url=f"{self.base_url}/rag", session=self.session) + # Public, billable /v1 API -- see LiteratureClient's docstring. + self.literature = LiteratureClient(base_url=f"{self.base_url}/v1/literature", session=self.session) self.models = ModelsClient(base_url=f"{self.base_url}/model-registry", session=self.session) self.tes = TESClient(base_url=f"{self.base_url}/tes", session=self.session) # workflows_url is a SEPARATE override from base_url, unlike diff --git a/omnibioai/exceptions.py b/omnibioai/exceptions.py index e852fba..65ff10f 100644 --- a/omnibioai/exceptions.py +++ b/omnibioai/exceptions.py @@ -80,3 +80,20 @@ class GatewayError(OmniBioAIError): distinguishing it from ServiceUnavailableError/ValidationError lets a caller tell "the gateway itself doesn't know this route" apart from "the target service rejected the request".""" + + +class QuotaExceededError(ValidationError): + """402 from the public /v1 API: the organization has used its included + units (or prepaid credit). Add a payment method or upgrade the plan. + A subclass of ValidationError so existing `except ValidationError` + handlers keep catching it.""" + + +class RateLimitError(ValidationError): + """429 from the public /v1 API: too many requests for this API key in + the current one-minute window. `retry_after` is the number of seconds + the gateway said to wait (from Retry-After), or None if it sent none.""" + + def __init__(self, message: str, *, retry_after: Optional[int] = None, **kwargs: Any) -> None: + super().__init__(message, **kwargs) + self.retry_after = retry_after diff --git a/omnibioai/literature/__init__.py b/omnibioai/literature/__init__.py new file mode 100644 index 0000000..caff82e --- /dev/null +++ b/omnibioai/literature/__init__.py @@ -0,0 +1,3 @@ +from .client import LiteratureClient + +__all__ = ["LiteratureClient"] diff --git a/omnibioai/literature/client.py b/omnibioai/literature/client.py new file mode 100644 index 0000000..7fb1ac3 --- /dev/null +++ b/omnibioai/literature/client.py @@ -0,0 +1,51 @@ +"""omnibioai/literature/client.py""" +from __future__ import annotations + +import uuid +from typing import Literal, Optional + +from .._base import BaseServiceClient + + +class LiteratureClient(BaseServiceClient): + """The public, billable Literature AI API at `{gateway}/v1/literature`. + + Unlike `.rag` (the internal service route), this is the stable, + versioned contract meant for API-key users: it is rate-limited per key + (RateLimitError on 429), billed per answer (QuotaExceededError on 402 + once an organization's included answers run out), and every `ask()` + sends an Idempotency-Key, so a retried call is answered from the + gateway's stored result and never billed twice. + + Request fields are omnibioai-rag's POST /v1/query contract, passed + through unchanged by the gateway. + """ + + def ask( + self, + question: str, + *, + study: str = "default", + top_k: Optional[int] = None, + mode: Literal["rag", "pmids_only", "structured"] = "rag", + hybrid_search: bool = False, + idempotency_key: Optional[str] = None, + ) -> dict: + """Ask a biomedical question; returns the answer with its PubMed + citations. Pass the same `idempotency_key` when retrying a call + whose outcome you did not see (one is generated otherwise). `top_k` + is omitted unless given, so the server's own default applies.""" + payload: dict = { + "query": question, + "study": study, + "mode": mode, + "hybrid_search": hybrid_search, + } + if top_k is not None: + payload["top_k"] = top_k + headers = {"Idempotency-Key": idempotency_key or str(uuid.uuid4())} + return self._request("POST", "/answers", json=payload, headers=headers) + + def studies(self) -> dict: + """The queryable studies/research domains. Free, still rate-limited.""" + return self._request("GET", "/studies") diff --git a/omnibioai/mcp_server.py b/omnibioai/mcp_server.py new file mode 100644 index 0000000..93c1442 --- /dev/null +++ b/omnibioai/mcp_server.py @@ -0,0 +1,62 @@ +"""omnibioai/mcp_server.py + +An MCP (Model Context Protocol) server exposing the public Literature AI +API as tools, so Claude, ChatGPT and other MCP clients can ask OmniBioAI +for PubMed-cited answers. Every tool call goes through the same metered +/v1/literature API as the SDK (one billable answer per answer tool call), +authenticated with an omni_sk_ API key. + +Run it (stdio, the transport desktop MCP clients launch): + + pip install "omnibioai-sdk[mcp]" + OMNIBIOAI_API_KEY=omni_sk_... OMNIBIOAI_BASE_URL=https:// omnibioai-mcp + +The `mcp` package is an optional extra and only imported here, so +`import omnibioai` never requires it. +""" +from __future__ import annotations + +import os +from typing import Any, Optional + +from .client import DEFAULT_BASE_URL, OmniBioAI + +SERVER_NAME = "omnibioai" +INSTRUCTIONS = ( + "Biomedical literature tools backed by PubMed. Use answer_with_citations for " + "questions that need an evidence-based answer; every claim carries its PubMed ID. " + "Use find_pmids when only the relevant papers are needed, and list_studies to see " + "which research domains can be searched." +) + + +def build_server(client: OmniBioAI) -> Any: + """The MCP server, with its tools bound to `client`.""" + from mcp.server.mcpserver import MCPServer + + server = MCPServer(SERVER_NAME, instructions=INSTRUCTIONS) + + @server.tool(description="Answer a biomedical question from PubMed abstracts, citing PMIDs. Billed per answer.") + def answer_with_citations(question: str, study: str = "default", top_k: Optional[int] = None) -> dict: + return client.literature.ask(question, study=study, top_k=top_k, mode="rag") + + @server.tool(description="Return the PubMed IDs most relevant to a question, without an answer. Billed per call.") + def find_pmids(question: str, study: str = "default", top_k: Optional[int] = None) -> dict: + return client.literature.ask(question, study=study, top_k=top_k, mode="pmids_only") + + @server.tool(description="List the research domains (studies) that can be searched. Free.") + def list_studies() -> dict: + return client.literature.studies() + + return server + + +def client_from_env() -> OmniBioAI: + api_key = os.environ.get("OMNIBIOAI_API_KEY") + if not api_key: + raise SystemExit("Set OMNIBIOAI_API_KEY to an omni_sk_ API key.") + return OmniBioAI(api_key=api_key, base_url=os.environ.get("OMNIBIOAI_BASE_URL", DEFAULT_BASE_URL)) + + +def main() -> None: # pragma: no cover - process entry point + build_server(client_from_env()).run(transport=os.environ.get("OMNIBIOAI_MCP_TRANSPORT", "stdio")) diff --git a/pyproject.toml b/pyproject.toml index 8ae88e9..b749a19 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -25,7 +25,13 @@ classifiers = [ ] [project.optional-dependencies] +# MCP server (omnibioai-mcp): Literature AI as tools for Claude, ChatGPT and +# other MCP clients. Optional so the core SDK stays requests-only. +mcp = [ + "mcp>=2,<3" +] dev = [ + "mcp>=2,<3", "pytest>=8", "pytest-cov>=5.0", "responses>=0.25", @@ -34,6 +40,9 @@ dev = [ "ruff>=0.6" ] +[project.scripts] +omnibioai-mcp = "omnibioai.mcp_server:main" + [project.urls] Homepage = "https://github.com/man4ish/omnibioai_sdk" Source = "https://github.com/man4ish/omnibioai_sdk" diff --git a/tests/test_literature_client.py b/tests/test_literature_client.py new file mode 100644 index 0000000..bece5ca --- /dev/null +++ b/tests/test_literature_client.py @@ -0,0 +1,93 @@ +"""LiteratureClient (public /v1/literature API) and API-key construction: +request shape and Idempotency-Key handling for ask()/studies(), the v1 error +envelope, 402 -> QuotaExceededError, 429 -> RateLimitError with retry_after, +and OmniBioAI's api_key / OMNIBIOAI_API_KEY / argument-validation rules. +""" +from __future__ import annotations + +import json +import uuid + +import pytest +import responses + +from omnibioai import LiteratureClient, OmniBioAI +from omnibioai.exceptions import QuotaExceededError, RateLimitError, ValidationError + +BASE = "https://gateway.example.com" +KEY = "omni_sk_" + "a" * 40 +URL = f"{BASE}/v1/literature" + + +def _client(): + return OmniBioAI(api_key=KEY, base_url=BASE) + + +@responses.activate +def test_ask_sends_payload_key_and_generated_idempotency_key(): + responses.add(responses.POST, f"{URL}/answers", json={"answer": "x"}, status=200) + assert _client().literature.ask("What does TP53 do?") == {"answer": "x"} + req = responses.calls[0].request + assert json.loads(req.body) == {"query": "What does TP53 do?", "study": "default", "mode": "rag", + "hybrid_search": False} + assert req.headers["Authorization"] == f"Bearer {KEY}" + uuid.UUID(req.headers["Idempotency-Key"]) + + +@responses.activate +def test_ask_options_and_explicit_idempotency_key(): + responses.add(responses.POST, f"{URL}/answers", json={}, status=200) + _client().literature.ask("q", study="Oncology", top_k=5, mode="pmids_only", hybrid_search=True, + idempotency_key="retry-1") + req = responses.calls[0].request + assert json.loads(req.body)["top_k"] == 5 + assert json.loads(req.body)["study"] == "Oncology" + assert req.headers["Idempotency-Key"] == "retry-1" + + +@responses.activate +def test_studies(): + responses.add(responses.GET, f"{URL}/studies", json={"studies": ["Oncology"]}, status=200) + assert _client().literature.studies() == {"studies": ["Oncology"]} + + +@responses.activate +def test_quota_exceeded(): + responses.add(responses.POST, f"{URL}/answers", status=402, json={ + "error": {"type": "quota_exceeded", "message": "Your organization has used its included answers.", + "request_id": "r1"}}) + with pytest.raises(QuotaExceededError) as exc: + _client().literature.ask("q") + assert "included answers" in str(exc.value) + assert isinstance(exc.value, ValidationError) and exc.value.status_code == 402 + + +@responses.activate +def test_rate_limited_with_and_without_retry_after(): + responses.add(responses.POST, f"{URL}/answers", status=429, headers={"Retry-After": "17"}, + json={"error": {"type": "rate_limit_exceeded", "message": "slow down", "request_id": "r"}}) + responses.add(responses.POST, f"{URL}/answers", status=429, json={"error": {"type": "rate_limit_exceeded"}}) + with pytest.raises(RateLimitError) as first: + _client().literature.ask("q") + assert first.value.retry_after == 17 and str(first.value) == "slow down" + with pytest.raises(RateLimitError) as second: + _client().literature.ask("q") + assert second.value.retry_after is None and str(second.value) == "rate_limit_exceeded" + + +def test_api_key_from_environment(monkeypatch): + monkeypatch.setenv("OMNIBIOAI_API_KEY", KEY) + client = OmniBioAI(base_url=BASE) + assert client.access_token == KEY and client.refresh_token is None + assert isinstance(client.literature, LiteratureClient) + + +def test_constructor_argument_rules(monkeypatch): + monkeypatch.delenv("OMNIBIOAI_API_KEY", raising=False) + with pytest.raises(ValueError, match="either"): + OmniBioAI(access_token="jwt", api_key=KEY) + with pytest.raises(ValueError, match="refresh_token"): + OmniBioAI(api_key=KEY, refresh_token="r") + with pytest.raises(ValueError, match="required"): + OmniBioAI() + assert OmniBioAI(access_token="jwt").access_token == "jwt" diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py new file mode 100644 index 0000000..70ef91e --- /dev/null +++ b/tests/test_mcp_server.py @@ -0,0 +1,56 @@ +"""MCP server (omnibioai.mcp_server): the three tools are registered with +their input schemas, each delegates to the matching LiteratureClient call, +and client_from_env reads OMNIBIOAI_API_KEY / OMNIBIOAI_BASE_URL. +""" +from __future__ import annotations + +import asyncio +import json +from unittest.mock import MagicMock + +import pytest + +from omnibioai.client import DEFAULT_BASE_URL +from omnibioai.mcp_server import build_server, client_from_env + + +@pytest.fixture +def server(): + client = MagicMock() + client.literature.ask.return_value = {"answer": "TP53 [PMID:1]"} + client.literature.studies.return_value = {"studies": ["Oncology"]} + return build_server(client), client + + +def test_tools_are_listed_with_schemas(server): + srv, _ = server + tools = {t.name: t for t in asyncio.run(srv.list_tools())} + assert set(tools) == {"answer_with_citations", "find_pmids", "list_studies"} + assert tools["answer_with_citations"].input_schema["required"] == ["question"] + + +def test_answer_with_citations(server): + srv, client = server + result = asyncio.run(srv.call_tool("answer_with_citations", {"question": "q", "study": "Oncology", "top_k": 3})) + assert not result.is_error + assert json.loads(result.content[0].text) == {"answer": "TP53 [PMID:1]"} + client.literature.ask.assert_called_once_with("q", study="Oncology", top_k=3, mode="rag") + + +def test_find_pmids_and_list_studies(server): + srv, client = server + asyncio.run(srv.call_tool("find_pmids", {"question": "q"})) + client.literature.ask.assert_called_once_with("q", study="default", top_k=None, mode="pmids_only") + result = asyncio.run(srv.call_tool("list_studies", {})) + assert json.loads(result.content[0].text) == {"studies": ["Oncology"]} + + +def test_client_from_env(monkeypatch): + monkeypatch.delenv("OMNIBIOAI_API_KEY", raising=False) + with pytest.raises(SystemExit): + client_from_env() + monkeypatch.setenv("OMNIBIOAI_API_KEY", "omni_sk_" + "a" * 40) + monkeypatch.delenv("OMNIBIOAI_BASE_URL", raising=False) + assert client_from_env().base_url == DEFAULT_BASE_URL + monkeypatch.setenv("OMNIBIOAI_BASE_URL", "https://api.example.com/") + assert client_from_env().base_url == "https://api.example.com"