Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
34 changes: 34 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,38 @@ print(run)
> Never point the SDK directly at individual services (auth-service,
> workbench etc.) in production.

### Literature AI with an API key

The public, billable Literature AI API (`/v1/literature` on the gateway)
accepts an `omni_sk_` API key in place of a login token:

```python
from omnibioai import OmniBioAI

client = OmniBioAI(api_key="omni_sk_...", base_url="https://<gateway>")
# or set OMNIBIOAI_API_KEY and call OmniBioAI(base_url=...)

answer = client.literature.ask("Which genes drive hypereosinophilic syndrome?", top_k=8)
studies = client.literature.studies() # free
```

Each `ask()` sends an `Idempotency-Key`; pass your own
`idempotency_key=` when retrying a call whose result you did not see, and
the gateway returns the stored answer without billing it again. Over the
per-key rate limit you get `RateLimitError` (with `retry_after` seconds);
once your organization's included answers run out, `QuotaExceededError`.

### MCP server (Claude, ChatGPT and other MCP clients)

```bash
pip install "omnibioai-sdk[mcp]"
OMNIBIOAI_API_KEY=omni_sk_... OMNIBIOAI_BASE_URL=https://<gateway> omnibioai-mcp
```

Tools: `answer_with_citations`, `find_pmids` and `list_studies` (free). For
a desktop MCP client, register the `omnibioai-mcp` command with those two
environment variables. Answer tools are billed like `literature.ask()`.

### Getting a token

Obtain the access/refresh token pair through the OmniBioAI Auth login or SSO
Expand Down Expand Up @@ -215,6 +247,8 @@ c = OmniClient()
from omnibioai.exceptions import (
AuthenticationError,
PermissionDeniedError,
QuotaExceededError, # 402: /v1 API, included usage used up
RateLimitError, # 429: /v1 API, .retry_after seconds
ServiceUnavailableError,
)

Expand Down
3 changes: 2 additions & 1 deletion omnibioai/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,9 +10,10 @@
"""
from .client import OmniBioAI
from .legacy import OmniClient
from .literature import LiteratureClient
from .models import ModelsClient
from .rag import RAGClient
from .tes import TESClient
from .workflows import WorkflowsClient

__all__ = ["OmniBioAI", "OmniClient", "RAGClient", "ModelsClient", "TESClient", "WorkflowsClient"]
__all__ = ["OmniBioAI", "OmniClient", "LiteratureClient", "RAGClient", "ModelsClient", "TESClient", "WorkflowsClient"]
22 changes: 22 additions & 0 deletions omnibioai/_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,8 @@
AuthenticationError,
GatewayError,
PermissionDeniedError,
QuotaExceededError,
RateLimitError,
ResourceNotFoundError,
ServiceUnavailableError,
ValidationError,
Expand Down Expand Up @@ -64,6 +66,16 @@ def _parse(self, response: requests.Response) -> Any:
raise AuthenticationError(message, status_code=401, response_body=body, trace_id=trace_id)
if response.status_code == 403:
raise PermissionDeniedError(message, status_code=403, response_body=body, trace_id=trace_id)
if response.status_code == 402:
raise QuotaExceededError(message, status_code=402, response_body=body, trace_id=trace_id)
if response.status_code == 429:
raise RateLimitError(
message,
retry_after=_retry_after(response),
status_code=429,
response_body=body,
trace_id=trace_id,
)
if response.status_code == 404:
raise ResourceNotFoundError(message, status_code=404, response_body=body, trace_id=trace_id)
if response.status_code >= 500:
Expand All @@ -84,12 +96,20 @@ def _safe_json(response: requests.Response) -> Any:
return None


def _retry_after(response: requests.Response) -> Optional[int]:
try:
return int(response.headers.get("Retry-After", ""))
except ValueError:
return None


def _extract_message(body: Any) -> Optional[str]:
"""Normalizes the ecosystem's several observed error-body shapes into
one human-readable message string:
- API Gateway: {"error": "...", "reason": "..."}
- FastAPI default: {"detail": "..."} (RAG, control-center, TES, most of Model Registry)
- Model Registry (DB): {"detail": {"ok": false, "error": "..."}}
- Public /v1 API: {"error": {"type": "...", "message": "...", "request_id": "..."}}
See the Phase 1 findings report's Cross-Service Consistency section
for the full survey this is built from. Returns None (falls back to
response.reason at the call site) if the body doesn't match any
Expand All @@ -104,6 +124,8 @@ def _extract_message(body: Any) -> Optional[str]:
if nested:
return str(nested)
error = body.get("error")
if isinstance(error, dict):
return str(error.get("message") or error.get("type") or error)
if error:
reason = body.get("reason")
return f"{error}: {reason}" if reason else str(error)
Expand Down
22 changes: 21 additions & 1 deletion omnibioai/client.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,12 @@
"""omnibioai/client.py"""
from __future__ import annotations

import os
from typing import Optional

from .auth.session import AuthenticatedSession
from .auth.tokens import TokenPair
from .literature.client import LiteratureClient
from .models.client import ModelsClient
from .rag.client import RAGClient
from .tes.client import TESClient
Expand Down Expand Up @@ -51,20 +53,38 @@ class OmniBioAI:

def __init__(
self,
access_token: str,
access_token: Optional[str] = None,
refresh_token: Optional[str] = None,
base_url: str = DEFAULT_BASE_URL,
auth_url: str = DEFAULT_AUTH_URL,
workflows_url: Optional[str] = None,
timeout: float = 60,
*,
api_key: Optional[str] = None,
) -> None:
# An omni_sk_ API key is sent as the bearer token exactly like an
# access token; the gateway recognises it. Keys never expire into a
# refresh, so no refresh_token applies. Falls back to the
# OMNIBIOAI_API_KEY environment variable when neither is given.
if access_token and api_key:
raise ValueError("Pass either access_token or api_key, not both")
if api_key is None and access_token is None:
api_key = os.environ.get("OMNIBIOAI_API_KEY")
if api_key is not None:
if refresh_token:
raise ValueError("refresh_token does not apply to an API key")
access_token = api_key
if not access_token:
raise ValueError("An access_token or api_key (or OMNIBIOAI_API_KEY) is required")
self.base_url = base_url.rstrip("/")
self.timeout = timeout
self.tokens = TokenPair(access_token=access_token, refresh_token=refresh_token)
self.session = AuthenticatedSession(
tokens=self.tokens, auth_url=auth_url, timeout=timeout,
)
self.rag = RAGClient(base_url=f"{self.base_url}/rag", session=self.session)
# Public, billable /v1 API -- see LiteratureClient's docstring.
self.literature = LiteratureClient(base_url=f"{self.base_url}/v1/literature", session=self.session)
self.models = ModelsClient(base_url=f"{self.base_url}/model-registry", session=self.session)
self.tes = TESClient(base_url=f"{self.base_url}/tes", session=self.session)
# workflows_url is a SEPARATE override from base_url, unlike
Expand Down
17 changes: 17 additions & 0 deletions omnibioai/exceptions.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,3 +80,20 @@ class GatewayError(OmniBioAIError):
distinguishing it from ServiceUnavailableError/ValidationError lets a
caller tell "the gateway itself doesn't know this route" apart from
"the target service rejected the request"."""


class QuotaExceededError(ValidationError):
"""402 from the public /v1 API: the organization has used its included
units (or prepaid credit). Add a payment method or upgrade the plan.
A subclass of ValidationError so existing `except ValidationError`
handlers keep catching it."""


class RateLimitError(ValidationError):
"""429 from the public /v1 API: too many requests for this API key in
the current one-minute window. `retry_after` is the number of seconds
the gateway said to wait (from Retry-After), or None if it sent none."""

def __init__(self, message: str, *, retry_after: Optional[int] = None, **kwargs: Any) -> None:
super().__init__(message, **kwargs)
self.retry_after = retry_after
3 changes: 3 additions & 0 deletions omnibioai/literature/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,3 @@
from .client import LiteratureClient

__all__ = ["LiteratureClient"]
51 changes: 51 additions & 0 deletions omnibioai/literature/client.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
"""omnibioai/literature/client.py"""
from __future__ import annotations

import uuid
from typing import Literal, Optional

from .._base import BaseServiceClient


class LiteratureClient(BaseServiceClient):
"""The public, billable Literature AI API at `{gateway}/v1/literature`.

Unlike `.rag` (the internal service route), this is the stable,
versioned contract meant for API-key users: it is rate-limited per key
(RateLimitError on 429), billed per answer (QuotaExceededError on 402
once an organization's included answers run out), and every `ask()`
sends an Idempotency-Key, so a retried call is answered from the
gateway's stored result and never billed twice.

Request fields are omnibioai-rag's POST /v1/query contract, passed
through unchanged by the gateway.
"""

def ask(
self,
question: str,
*,
study: str = "default",
top_k: Optional[int] = None,
mode: Literal["rag", "pmids_only", "structured"] = "rag",
hybrid_search: bool = False,
idempotency_key: Optional[str] = None,
) -> dict:
"""Ask a biomedical question; returns the answer with its PubMed
citations. Pass the same `idempotency_key` when retrying a call
whose outcome you did not see (one is generated otherwise). `top_k`
is omitted unless given, so the server's own default applies."""
payload: dict = {
"query": question,
"study": study,
"mode": mode,
"hybrid_search": hybrid_search,
}
if top_k is not None:
payload["top_k"] = top_k
headers = {"Idempotency-Key": idempotency_key or str(uuid.uuid4())}
return self._request("POST", "/answers", json=payload, headers=headers)

def studies(self) -> dict:
"""The queryable studies/research domains. Free, still rate-limited."""
return self._request("GET", "/studies")
62 changes: 62 additions & 0 deletions omnibioai/mcp_server.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,62 @@
"""omnibioai/mcp_server.py

An MCP (Model Context Protocol) server exposing the public Literature AI
API as tools, so Claude, ChatGPT and other MCP clients can ask OmniBioAI
for PubMed-cited answers. Every tool call goes through the same metered
/v1/literature API as the SDK (one billable answer per answer tool call),
authenticated with an omni_sk_ API key.

Run it (stdio, the transport desktop MCP clients launch):

pip install "omnibioai-sdk[mcp]"
OMNIBIOAI_API_KEY=omni_sk_... OMNIBIOAI_BASE_URL=https://<gateway> omnibioai-mcp

The `mcp` package is an optional extra and only imported here, so
`import omnibioai` never requires it.
"""
from __future__ import annotations

import os
from typing import Any, Optional

from .client import DEFAULT_BASE_URL, OmniBioAI

SERVER_NAME = "omnibioai"
INSTRUCTIONS = (
"Biomedical literature tools backed by PubMed. Use answer_with_citations for "
"questions that need an evidence-based answer; every claim carries its PubMed ID. "
"Use find_pmids when only the relevant papers are needed, and list_studies to see "
"which research domains can be searched."
)


def build_server(client: OmniBioAI) -> Any:
"""The MCP server, with its tools bound to `client`."""
from mcp.server.mcpserver import MCPServer

server = MCPServer(SERVER_NAME, instructions=INSTRUCTIONS)

@server.tool(description="Answer a biomedical question from PubMed abstracts, citing PMIDs. Billed per answer.")
def answer_with_citations(question: str, study: str = "default", top_k: Optional[int] = None) -> dict:
return client.literature.ask(question, study=study, top_k=top_k, mode="rag")

@server.tool(description="Return the PubMed IDs most relevant to a question, without an answer. Billed per call.")
def find_pmids(question: str, study: str = "default", top_k: Optional[int] = None) -> dict:
return client.literature.ask(question, study=study, top_k=top_k, mode="pmids_only")

@server.tool(description="List the research domains (studies) that can be searched. Free.")
def list_studies() -> dict:
return client.literature.studies()

return server


def client_from_env() -> OmniBioAI:
api_key = os.environ.get("OMNIBIOAI_API_KEY")
if not api_key:
raise SystemExit("Set OMNIBIOAI_API_KEY to an omni_sk_ API key.")
return OmniBioAI(api_key=api_key, base_url=os.environ.get("OMNIBIOAI_BASE_URL", DEFAULT_BASE_URL))


def main() -> None: # pragma: no cover - process entry point
build_server(client_from_env()).run(transport=os.environ.get("OMNIBIOAI_MCP_TRANSPORT", "stdio"))
9 changes: 9 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,13 @@ classifiers = [
]

[project.optional-dependencies]
# MCP server (omnibioai-mcp): Literature AI as tools for Claude, ChatGPT and
# other MCP clients. Optional so the core SDK stays requests-only.
mcp = [
"mcp>=2,<3"
]
dev = [
"mcp>=2,<3",
"pytest>=8",
"pytest-cov>=5.0",
"responses>=0.25",
Expand All @@ -34,6 +40,9 @@ dev = [
"ruff>=0.6"
]

[project.scripts]
omnibioai-mcp = "omnibioai.mcp_server:main"

[project.urls]
Homepage = "https://github.com/man4ish/omnibioai_sdk"
Source = "https://github.com/man4ish/omnibioai_sdk"
Expand Down
Loading
Loading