Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
7 changes: 7 additions & 0 deletions app/core/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,13 @@ class Config:
# Upper bound on how long one exchange is reused; the minted token's
# own expiry (minus a safety margin) caps it further.
API_KEY_CACHE_TTL = int(os.getenv("API_KEY_CACHE_TTL", "60"))
# M16 (BYOK provider routing, design audit gap #4): same shared-
# secret shape as API_KEY_EXCHANGE_SECRET above, for omnibioai-auth's
# POST /internal/organizations/{id}/provider-keys/{provider}/reveal
# -- called immediately before a BYOK-routed /v1/literature/answers
# call is forwarded to RAG. Empty = BYOK routing rejected (503), not
# a silent fallback to the platform's own model.
PROVIDER_KEY_REVEAL_SECRET = os.getenv("PROVIDER_KEY_REVEAL_SECRET", "")

# Public /v1 API (app/routes/v1.py). Rate-limit, idempotency and quota
# state lives under gateway:v1:* in the Redis at V1_REDIS_URL, which
Expand Down
114 changes: 99 additions & 15 deletions app/routes/v1.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,12 @@

ANSWER_RESOURCE = "literature.answer"
SEARCH_RESOURCE = "literature.search"
# M16 (design audit gap #4): emitted alongside ANSWER_RESOURCE, only for
# a BYOK-routed call that got a real token count back from RAG -- never
# fabricated, same "omit rather than invent" rule the rest of this gap
# has already followed for model/price fields.
TOKEN_INPUT_RESOURCE = "llm.tokens.input"
TOKEN_OUTPUT_RESOURCE = "llm.tokens.output"
_IDEMPOTENCY_KEY = re.compile(r"^[\x21-\x7e]{1,255}$")


Expand Down Expand Up @@ -157,6 +163,29 @@ def _upstream_error(status: int, response, request_id: str, headers: dict):
"The literature service rejected the request.", request_id, headers, detail=response)


async def _reveal_provider_key(org_id: str, provider: str):
"""M16 (BYOK provider routing, design audit gap #4): resolves the
organization's own decrypted key for `provider` from omnibioai-auth
immediately before a BYOK-routed call is forwarded to RAG -- never
stored or logged here, attached to this one RAG request body and
nowhere else. Returns (api_key, None) on success, or (None,
error_response) on failure. Uses the shared-secret header, never
the caller's own forwarded bearer token -- this is a service-to-
service call, not something the caller's own identity authorizes.
"""
secret = Config.PROVIDER_KEY_REVEAL_SECRET
if not secret:
return None, (503, {"detail": "Provider key reveal is not configured"})
status, response = await proxy.forward(
url=f"{resolve_service('auth')}/internal/organizations/{org_id}/provider-keys/{provider}/reveal",
method="POST",
headers={"X-Provider-Key-Reveal-Secret": secret},
)
if status != 200:
return None, (status, response)
return response.get("api_key"), None


async def _handle_billable_literature_call(
request: Request, *, resource: str, build_rag_body, build_public_response, build_test_response,
quota_exceeded_message: str, max_concurrent_answers: int | None = None,
Expand Down Expand Up @@ -220,6 +249,33 @@ async def _handle_billable_literature_call(
return _error(409, "idempotency_in_progress",
"A request with this Idempotency-Key is still running.", request_id, headers)

# M16 (BYOK provider routing): resolved before the concurrency slot
# and quota are touched -- a request that can't get a key can't
# succeed regardless, so it shouldn't consume either. Skipped
# entirely for test_mode: a test key never reaches RAG at all, so
# there is nothing here for it to route through.
byok_model = rag_body.get("model") if not test_mode else None
if byok_model:
provider_api_key, reveal_error = await _reveal_provider_key(org_id, byok_model)
if reveal_error is not None:
reveal_status, reveal_response = reveal_error
if idempotency_key is not None:
await store.idempotency_finish(subject, idempotency_key, fingerprint, 400, None)
if reveal_status == 404:
return _error(
400, "provider_key_not_configured",
f"Your organization has no {byok_model} key configured. "
f"Set one with PUT /v1/provider-keys/{byok_model} first.",
request_id, headers,
)
if reveal_status == 503:
return _error(503, "provider_key_reveal_unavailable",
"Provider key routing is not available.", request_id, headers)
return _error(502, "upstream_error",
"The identity service failed to resolve your provider key.", request_id, headers,
detail=reveal_response if reveal_status < 500 else None)
rag_body["provider_api_key"] = provider_api_key

# A test key never touches real RAG capacity, so it never needs (or
# holds) a concurrency slot -- acquiring one here, only to release it
# a few lines down having done no real work, would just be unearned
Expand Down Expand Up @@ -273,19 +329,41 @@ async def _handle_billable_literature_call(
if idempotency_key is not None:
await store.idempotency_finish(subject, idempotency_key, fingerprint, status, public_response)

usage_dedup_key = f"{subject}:{idempotency_key}" if idempotency_key else request_id
await store.emit_usage(
org_id=org_id,
user_id=user_id,
resource=resource,
trace_id=request_id,
dedup_key=f"{subject}:{idempotency_key}" if idempotency_key else request_id,
dedup_key=usage_dedup_key,
metadata={
"request_id": request_id,
"client_id": identity.get("client_id"),
"token_type": identity.get("token_type"),
"idempotency_key_sha256": request_fingerprint(idempotency_key) if idempotency_key else None,
},
)
# M16 (design audit gap #4): a BYOK-routed call that got real
# token counts back from RAG also emits the two token-usage
# events -- never fabricated for the default (non-BYOK) path,
# where these are always None (Ollama's own response carries no
# such count at all). Distinct dedup_key suffixes: emit_usage's
# own event_id is derived from dedup_key, and these are two
# different resources on the same request_id, not duplicates of
# each other or of the literature.answer event above.
token_usage = public_response.get("usage") or {}
if token_usage.get("input_tokens") is not None:
await store.emit_usage(
org_id=org_id, user_id=user_id, resource=TOKEN_INPUT_RESOURCE, trace_id=request_id,
dedup_key=f"{usage_dedup_key}:tokens:input", quantity=token_usage["input_tokens"], unit="tokens",
metadata={"request_id": request_id, "model": public_response.get("model_source")},
)
if token_usage.get("output_tokens") is not None:
await store.emit_usage(
org_id=org_id, user_id=user_id, resource=TOKEN_OUTPUT_RESOURCE, trace_id=request_id,
dedup_key=f"{usage_dedup_key}:tokens:output", quantity=token_usage["output_tokens"], unit="tokens",
metadata={"request_id": request_id, "model": public_response.get("model_source")},
)
return JSONResponse(public_response, status_code=status, headers={**headers, "X-Request-Id": request_id})
finally:
# Released regardless of how the try block above exited (a quota
Expand Down Expand Up @@ -423,20 +501,24 @@ async def literature_usage(request: Request):
async def literature_models(request: Request):
"""Free: the model catalog -- answer/embedding models currently
served, with source and price. Rate-limited like every /v1 call,
never billed. No upstream call: there is exactly one model path
today, RAG's own GPU-hosted default (see the design's "largest
gaps" #4 -- Claude/OpenAI routing and bring-your-own-key do not
exist yet, and build_rag_query already rejects any request that
would need one).

price is null, not a placeholder dollar figure: the design doc's own
pricing section says to measure real GPU cost per answer first
(milestone M0, 1,000 representative questions) before setting
prices, and that measurement has not been run. The model actually
used for a given answer is already reported per-call in
/v1/literature/answers' response (its `model` field is the real
value RAG used; "default" here is the stable identifier a caller
passes back as this endpoint's own `model` request field).
never billed. No upstream call.

price is null, not a placeholder dollar figure, for every entry: the
design doc's own pricing section says to measure real cost per
answer first (milestone M0, 1,000 representative questions) before
setting prices, and that measurement has not been run -- true for
the default model and for claude/openai alike (BYOK means an org
pays its own provider directly; this platform still doesn't charge
its own per-unit fee on top). The model actually used for a given
answer is already reported per-call in /v1/literature/answers'
response (its `model` field is the real value used; "default" here
is the stable identifier a caller passes back as this endpoint's own
`model` request field).

claude/openai (M16, design audit gap #4) require use_own_key: true
on the actual /v1/literature/answers call -- there is no platform-
wide key for either, only an organization's own BYOK key (see
PUT /v1/provider-keys/{provider}).
"""
request_id = getattr(request.state, "trace_id", "")
subject, org_id, _ = _caller(request)
Expand All @@ -445,6 +527,8 @@ async def literature_models(request: Request):
return limited
models = [
{"model": "default", "source": "omnibioai_gpu", "billed_by": "query", "price": None},
{"model": "claude", "source": "claude", "billed_by": "query", "price": None},
{"model": "openai", "source": "openai", "billed_by": "query", "price": None},
]
return JSONResponse({"models": models}, status_code=200, headers={**headers, "X-Request-Id": request_id})

Expand Down
60 changes: 43 additions & 17 deletions app/services/literature_contract.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,32 +29,56 @@ def __init__(self, field: str, message: str):

# The only two spellings "no model override" takes today: the field is
# absent/None, or the caller explicitly names RAG's actual current
# default via this sentinel-free "default" shorthand. Any other value
# is a model this gateway has no route for yet.
# default via this sentinel-free "default" shorthand.
_NO_MODEL_OVERRIDE = (None, "", "default")

# M16 (design audit gap #4): the only two models BYOK routing can select
# -- there is no platform-wide Claude/OpenAI key, only an organization's
# own (see app/routes/v1.py's reveal-and-forward step), so either of
# these requires use_own_key: true. Any other non-default value is a
# model this gateway still has no route for at all.
_BYOK_MODELS = ("claude", "openai")


def build_rag_query(body: dict) -> dict:
"""Translate a public /v1/literature/answers request body into RAG's
POST /v1/query body. Raises UnsupportedRequestError for a field this
gateway cannot yet honor."""
gateway cannot yet honor.

model="claude"/"openai" with use_own_key=true passes `model` through
to RAG's own body (see QueryRequest in omnibioai-rag's
app/api/server.py) -- app/routes/v1.py is what actually resolves and
attaches the organization's decrypted key before forwarding; this
function only validates the request shape, never touches a key.
"""
if not isinstance(body, dict):
raise UnsupportedRequestError("body", "Request body must be a JSON object.")

question = body.get("question")
if not isinstance(question, str) or not question.strip():
raise UnsupportedRequestError("question", "\"question\" is required and must be a non-empty string.")

if body.get("model") not in _NO_MODEL_OVERRIDE:
model = body.get("model")
use_own_key = bool(body.get("use_own_key"))
if model not in _NO_MODEL_OVERRIDE and model not in _BYOK_MODELS:
raise UnsupportedRequestError(
"model", f"Model {model!r} is not yet supported; omit this field to use the default model.",
)
if model in _BYOK_MODELS and not use_own_key:
raise UnsupportedRequestError(
"model", f"Model {body.get('model')!r} is not yet supported; omit this field to use the default model.",
"use_own_key",
f"Routing to {model!r} requires use_own_key: true -- there is no platform-wide key for this provider.",
)
if use_own_key and model not in _BYOK_MODELS:
raise UnsupportedRequestError(
"model", "use_own_key: true requires model to be \"claude\" or \"openai\".",
)
if body.get("use_own_key"):
raise UnsupportedRequestError("use_own_key", "Bring-your-own-key is not yet supported.")
if body.get("stream"):
raise UnsupportedRequestError("stream", "Streaming responses are not yet supported.")

query: dict = {"query": question, "study": body.get("domain") or "default"}
if model in _BYOK_MODELS:
query["model"] = model
max_citations = body.get("max_citations")
if isinstance(max_citations, (int, float)) and not isinstance(max_citations, bool) and max_citations > 0:
query["top_k"] = int(max_citations)
Expand Down Expand Up @@ -136,20 +160,22 @@ def build_public_answer(rag_result: dict, *, domain, request_id: str, latency_ms
"answer": summary.get("text", ""),
"citations": citations,
"model": summary.get("model"),
# Only one model source exists today -- RAG's own GPU-hosted
# model. Claude/OpenAI routing and BYOK (design "largest gaps"
# #4) would each need their own model_source value; until they
# exist, build_rag_query above already rejects any request that
# would need one.
"model_source": "omnibioai_gpu",
# M16: "omnibioai_gpu" when RAG used its own default model,
# "claude"/"openai" when this call was BYOK-routed -- RAG's own
# summary.model_source already reflects which one actually
# happened (see that repo's resolve_chat_model), so this just
# relays it rather than hardcoding the pre-M16 default.
"model_source": summary.get("model_source", "omnibioai_gpu"),
"domain": rag_result.get("study", domain),
"usage": {
"queries": 1,
# No per-provider token accounting exists yet (same gap #4)
# -- reporting a fabricated count here would be worse than
# M16: real counts for a BYOK-routed call (RAG's own
# provider client reports them); still None for the default
# path, since Ollama's own response carries no token count
# at all -- reporting a fabricated one would be worse than
# omitting it.
"input_tokens": None,
"output_tokens": None,
"input_tokens": summary.get("input_tokens"),
"output_tokens": summary.get("output_tokens"),
"billed_by": "query",
"latency_ms": latency_ms,
},
Expand Down
14 changes: 11 additions & 3 deletions app/services/v1_store.py
Original file line number Diff line number Diff line change
Expand Up @@ -342,11 +342,19 @@ async def _xadd_usage_event(self, event: dict) -> bool:
return False

async def emit_usage(self, *, org_id: str, user_id: str, resource: str, trace_id: str,
dedup_key: str, metadata: dict) -> bool:
dedup_key: str, metadata: dict, quantity: int = 1, unit: str = "requests") -> bool:
"""XADD one billable usage event in omnibioai-usage-client's wire
format. event_id is derived from dedup_key, so a replayed or
re-emitted request maps to the same id and billing counts it once.

quantity/unit default to 1/"requests" (unchanged from before
M16) -- a BYOK-routed call's llm.tokens.input/output events (see
app/routes/v1.py) pass the real token count and unit="tokens"
instead; dedup_key must be distinct per resource in that case
(e.g. f"{request_id}:tokens:input"), since this method's own
event_id is derived from it and two different resources sharing
one dedup_key would collide.

A successful answer has already been returned to the caller by
the time this runs -- a lost event here is lost revenue, never
an overcharge, so an XADD failure writes to the local outbox
Expand All @@ -361,8 +369,8 @@ async def emit_usage(self, *, org_id: str, user_id: str, resource: str, trace_id
"service": "api",
"resource": resource,
"action": "completed",
"quantity": 1,
"unit": "requests",
"quantity": quantity,
"unit": unit,
"user_id": str(user_id) if user_id else None,
"trace_id": trace_id or None,
"metadata": {**metadata, "billable": True},
Expand Down
55 changes: 50 additions & 5 deletions tests/test_literature_contract.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,20 +60,47 @@ def test_no_model_override_variants_are_accepted(model):
build_rag_query({"question": "q", "model": model}) # must not raise


def test_model_override_is_rejected():
def test_unsupported_model_override_is_rejected():
with pytest.raises(UnsupportedRequestError) as exc:
build_rag_query({"question": "q", "model": "claude"})
build_rag_query({"question": "q", "model": "llama-4"})
assert exc.value.field == "model"


def test_use_own_key_is_rejected():
def test_use_own_key_false_is_accepted():
build_rag_query({"question": "q", "use_own_key": False}) # must not raise


# ---------------------------------------------------------------------------
# M16 (design audit gap #4): BYOK model routing -- claude/openai require
# use_own_key: true together, since there is no platform-wide key for
# either provider.
# ---------------------------------------------------------------------------


def test_use_own_key_alone_without_a_byok_model_is_rejected():
with pytest.raises(UnsupportedRequestError) as exc:
build_rag_query({"question": "q", "use_own_key": True})
assert exc.value.field == "model"


@pytest.mark.parametrize("model", ["claude", "openai"])
def test_byok_model_without_use_own_key_is_rejected(model):
with pytest.raises(UnsupportedRequestError) as exc:
build_rag_query({"question": "q", "model": model})
assert exc.value.field == "use_own_key"


def test_use_own_key_false_is_accepted():
build_rag_query({"question": "q", "use_own_key": False}) # must not raise
@pytest.mark.parametrize("model", ["claude", "openai"])
def test_byok_model_with_use_own_key_is_accepted_and_passed_through(model):
query = build_rag_query({"question": "q", "model": model, "use_own_key": True})
assert query["model"] == model


def test_default_model_request_has_no_model_key_in_rag_query():
"""The non-BYOK path's RAG body is unchanged from before M16 --
no "model" key at all, not even None."""
query = build_rag_query({"question": "q"})
assert "model" not in query


def test_stream_is_rejected():
Expand Down Expand Up @@ -107,6 +134,24 @@ def test_builds_full_public_shape_from_rag_result():
}


def test_byok_summary_fields_flow_through_to_the_public_response():
"""M16: a BYOK-routed RAG result's model_source/input_tokens/
output_tokens are relayed as-is, not overridden by the pre-M16
hardcoded defaults."""
rag_result = {
"study": "Oncology",
"summary": {
"text": "TP53 is a tumor suppressor.", "model": "claude-3-5-sonnet-20241022",
"model_source": "claude", "input_tokens": 120, "output_tokens": 15,
},
"documents": [],
}
answer = build_public_answer(rag_result, domain="Oncology", request_id="req-1", latency_ms=42)
assert answer["model_source"] == "claude"
assert answer["usage"]["input_tokens"] == 120
assert answer["usage"]["output_tokens"] == 15


def test_falls_back_to_similarity_score_when_citation_confidence_absent():
rag_result = {"documents": [{"pmid": "1", "similarity_score": 0.5}]}
answer = build_public_answer(rag_result, domain=None, request_id="r", latency_ms=1)
Expand Down
Loading
Loading