diff --git a/app/core/config.py b/app/core/config.py index 81999cb..093967b 100644 --- a/app/core/config.py +++ b/app/core/config.py @@ -32,6 +32,13 @@ class Config: # Upper bound on how long one exchange is reused; the minted token's # own expiry (minus a safety margin) caps it further. API_KEY_CACHE_TTL = int(os.getenv("API_KEY_CACHE_TTL", "60")) + # M16 (BYOK provider routing, design audit gap #4): same shared- + # secret shape as API_KEY_EXCHANGE_SECRET above, for omnibioai-auth's + # POST /internal/organizations/{id}/provider-keys/{provider}/reveal + # -- called immediately before a BYOK-routed /v1/literature/answers + # call is forwarded to RAG. Empty = BYOK routing rejected (503), not + # a silent fallback to the platform's own model. + PROVIDER_KEY_REVEAL_SECRET = os.getenv("PROVIDER_KEY_REVEAL_SECRET", "") # Public /v1 API (app/routes/v1.py). Rate-limit, idempotency and quota # state lives under gateway:v1:* in the Redis at V1_REDIS_URL, which diff --git a/app/routes/v1.py b/app/routes/v1.py index e4f787b..9dba4ee 100644 --- a/app/routes/v1.py +++ b/app/routes/v1.py @@ -52,6 +52,12 @@ ANSWER_RESOURCE = "literature.answer" SEARCH_RESOURCE = "literature.search" +# M16 (design audit gap #4): emitted alongside ANSWER_RESOURCE, only for +# a BYOK-routed call that got a real token count back from RAG -- never +# fabricated, same "omit rather than invent" rule the rest of this gap +# has already followed for model/price fields. +TOKEN_INPUT_RESOURCE = "llm.tokens.input" +TOKEN_OUTPUT_RESOURCE = "llm.tokens.output" _IDEMPOTENCY_KEY = re.compile(r"^[\x21-\x7e]{1,255}$") @@ -157,6 +163,29 @@ def _upstream_error(status: int, response, request_id: str, headers: dict): "The literature service rejected the request.", request_id, headers, detail=response) +async def _reveal_provider_key(org_id: str, provider: str): + """M16 (BYOK provider routing, design audit gap #4): resolves the + organization's own decrypted key for `provider` from omnibioai-auth + immediately before a BYOK-routed call is forwarded to RAG -- never + stored or logged here, attached to this one RAG request body and + nowhere else. Returns (api_key, None) on success, or (None, + error_response) on failure. Uses the shared-secret header, never + the caller's own forwarded bearer token -- this is a service-to- + service call, not something the caller's own identity authorizes. + """ + secret = Config.PROVIDER_KEY_REVEAL_SECRET + if not secret: + return None, (503, {"detail": "Provider key reveal is not configured"}) + status, response = await proxy.forward( + url=f"{resolve_service('auth')}/internal/organizations/{org_id}/provider-keys/{provider}/reveal", + method="POST", + headers={"X-Provider-Key-Reveal-Secret": secret}, + ) + if status != 200: + return None, (status, response) + return response.get("api_key"), None + + async def _handle_billable_literature_call( request: Request, *, resource: str, build_rag_body, build_public_response, build_test_response, quota_exceeded_message: str, max_concurrent_answers: int | None = None, @@ -220,6 +249,33 @@ async def _handle_billable_literature_call( return _error(409, "idempotency_in_progress", "A request with this Idempotency-Key is still running.", request_id, headers) + # M16 (BYOK provider routing): resolved before the concurrency slot + # and quota are touched -- a request that can't get a key can't + # succeed regardless, so it shouldn't consume either. Skipped + # entirely for test_mode: a test key never reaches RAG at all, so + # there is nothing here for it to route through. + byok_model = rag_body.get("model") if not test_mode else None + if byok_model: + provider_api_key, reveal_error = await _reveal_provider_key(org_id, byok_model) + if reveal_error is not None: + reveal_status, reveal_response = reveal_error + if idempotency_key is not None: + await store.idempotency_finish(subject, idempotency_key, fingerprint, 400, None) + if reveal_status == 404: + return _error( + 400, "provider_key_not_configured", + f"Your organization has no {byok_model} key configured. " + f"Set one with PUT /v1/provider-keys/{byok_model} first.", + request_id, headers, + ) + if reveal_status == 503: + return _error(503, "provider_key_reveal_unavailable", + "Provider key routing is not available.", request_id, headers) + return _error(502, "upstream_error", + "The identity service failed to resolve your provider key.", request_id, headers, + detail=reveal_response if reveal_status < 500 else None) + rag_body["provider_api_key"] = provider_api_key + # A test key never touches real RAG capacity, so it never needs (or # holds) a concurrency slot -- acquiring one here, only to release it # a few lines down having done no real work, would just be unearned @@ -273,12 +329,13 @@ async def _handle_billable_literature_call( if idempotency_key is not None: await store.idempotency_finish(subject, idempotency_key, fingerprint, status, public_response) + usage_dedup_key = f"{subject}:{idempotency_key}" if idempotency_key else request_id await store.emit_usage( org_id=org_id, user_id=user_id, resource=resource, trace_id=request_id, - dedup_key=f"{subject}:{idempotency_key}" if idempotency_key else request_id, + dedup_key=usage_dedup_key, metadata={ "request_id": request_id, "client_id": identity.get("client_id"), @@ -286,6 +343,27 @@ async def _handle_billable_literature_call( "idempotency_key_sha256": request_fingerprint(idempotency_key) if idempotency_key else None, }, ) + # M16 (design audit gap #4): a BYOK-routed call that got real + # token counts back from RAG also emits the two token-usage + # events -- never fabricated for the default (non-BYOK) path, + # where these are always None (Ollama's own response carries no + # such count at all). Distinct dedup_key suffixes: emit_usage's + # own event_id is derived from dedup_key, and these are two + # different resources on the same request_id, not duplicates of + # each other or of the literature.answer event above. + token_usage = public_response.get("usage") or {} + if token_usage.get("input_tokens") is not None: + await store.emit_usage( + org_id=org_id, user_id=user_id, resource=TOKEN_INPUT_RESOURCE, trace_id=request_id, + dedup_key=f"{usage_dedup_key}:tokens:input", quantity=token_usage["input_tokens"], unit="tokens", + metadata={"request_id": request_id, "model": public_response.get("model_source")}, + ) + if token_usage.get("output_tokens") is not None: + await store.emit_usage( + org_id=org_id, user_id=user_id, resource=TOKEN_OUTPUT_RESOURCE, trace_id=request_id, + dedup_key=f"{usage_dedup_key}:tokens:output", quantity=token_usage["output_tokens"], unit="tokens", + metadata={"request_id": request_id, "model": public_response.get("model_source")}, + ) return JSONResponse(public_response, status_code=status, headers={**headers, "X-Request-Id": request_id}) finally: # Released regardless of how the try block above exited (a quota @@ -423,20 +501,24 @@ async def literature_usage(request: Request): async def literature_models(request: Request): """Free: the model catalog -- answer/embedding models currently served, with source and price. Rate-limited like every /v1 call, - never billed. No upstream call: there is exactly one model path - today, RAG's own GPU-hosted default (see the design's "largest - gaps" #4 -- Claude/OpenAI routing and bring-your-own-key do not - exist yet, and build_rag_query already rejects any request that - would need one). - - price is null, not a placeholder dollar figure: the design doc's own - pricing section says to measure real GPU cost per answer first - (milestone M0, 1,000 representative questions) before setting - prices, and that measurement has not been run. The model actually - used for a given answer is already reported per-call in - /v1/literature/answers' response (its `model` field is the real - value RAG used; "default" here is the stable identifier a caller - passes back as this endpoint's own `model` request field). + never billed. No upstream call. + + price is null, not a placeholder dollar figure, for every entry: the + design doc's own pricing section says to measure real cost per + answer first (milestone M0, 1,000 representative questions) before + setting prices, and that measurement has not been run -- true for + the default model and for claude/openai alike (BYOK means an org + pays its own provider directly; this platform still doesn't charge + its own per-unit fee on top). The model actually used for a given + answer is already reported per-call in /v1/literature/answers' + response (its `model` field is the real value used; "default" here + is the stable identifier a caller passes back as this endpoint's own + `model` request field). + + claude/openai (M16, design audit gap #4) require use_own_key: true + on the actual /v1/literature/answers call -- there is no platform- + wide key for either, only an organization's own BYOK key (see + PUT /v1/provider-keys/{provider}). """ request_id = getattr(request.state, "trace_id", "") subject, org_id, _ = _caller(request) @@ -445,6 +527,8 @@ async def literature_models(request: Request): return limited models = [ {"model": "default", "source": "omnibioai_gpu", "billed_by": "query", "price": None}, + {"model": "claude", "source": "claude", "billed_by": "query", "price": None}, + {"model": "openai", "source": "openai", "billed_by": "query", "price": None}, ] return JSONResponse({"models": models}, status_code=200, headers={**headers, "X-Request-Id": request_id}) diff --git a/app/services/literature_contract.py b/app/services/literature_contract.py index 564592f..3f07b47 100644 --- a/app/services/literature_contract.py +++ b/app/services/literature_contract.py @@ -29,15 +29,28 @@ def __init__(self, field: str, message: str): # The only two spellings "no model override" takes today: the field is # absent/None, or the caller explicitly names RAG's actual current -# default via this sentinel-free "default" shorthand. Any other value -# is a model this gateway has no route for yet. +# default via this sentinel-free "default" shorthand. _NO_MODEL_OVERRIDE = (None, "", "default") +# M16 (design audit gap #4): the only two models BYOK routing can select +# -- there is no platform-wide Claude/OpenAI key, only an organization's +# own (see app/routes/v1.py's reveal-and-forward step), so either of +# these requires use_own_key: true. Any other non-default value is a +# model this gateway still has no route for at all. +_BYOK_MODELS = ("claude", "openai") + def build_rag_query(body: dict) -> dict: """Translate a public /v1/literature/answers request body into RAG's POST /v1/query body. Raises UnsupportedRequestError for a field this - gateway cannot yet honor.""" + gateway cannot yet honor. + + model="claude"/"openai" with use_own_key=true passes `model` through + to RAG's own body (see QueryRequest in omnibioai-rag's + app/api/server.py) -- app/routes/v1.py is what actually resolves and + attaches the organization's decrypted key before forwarding; this + function only validates the request shape, never touches a key. + """ if not isinstance(body, dict): raise UnsupportedRequestError("body", "Request body must be a JSON object.") @@ -45,16 +58,27 @@ def build_rag_query(body: dict) -> dict: if not isinstance(question, str) or not question.strip(): raise UnsupportedRequestError("question", "\"question\" is required and must be a non-empty string.") - if body.get("model") not in _NO_MODEL_OVERRIDE: + model = body.get("model") + use_own_key = bool(body.get("use_own_key")) + if model not in _NO_MODEL_OVERRIDE and model not in _BYOK_MODELS: + raise UnsupportedRequestError( + "model", f"Model {model!r} is not yet supported; omit this field to use the default model.", + ) + if model in _BYOK_MODELS and not use_own_key: raise UnsupportedRequestError( - "model", f"Model {body.get('model')!r} is not yet supported; omit this field to use the default model.", + "use_own_key", + f"Routing to {model!r} requires use_own_key: true -- there is no platform-wide key for this provider.", + ) + if use_own_key and model not in _BYOK_MODELS: + raise UnsupportedRequestError( + "model", "use_own_key: true requires model to be \"claude\" or \"openai\".", ) - if body.get("use_own_key"): - raise UnsupportedRequestError("use_own_key", "Bring-your-own-key is not yet supported.") if body.get("stream"): raise UnsupportedRequestError("stream", "Streaming responses are not yet supported.") query: dict = {"query": question, "study": body.get("domain") or "default"} + if model in _BYOK_MODELS: + query["model"] = model max_citations = body.get("max_citations") if isinstance(max_citations, (int, float)) and not isinstance(max_citations, bool) and max_citations > 0: query["top_k"] = int(max_citations) @@ -136,20 +160,22 @@ def build_public_answer(rag_result: dict, *, domain, request_id: str, latency_ms "answer": summary.get("text", ""), "citations": citations, "model": summary.get("model"), - # Only one model source exists today -- RAG's own GPU-hosted - # model. Claude/OpenAI routing and BYOK (design "largest gaps" - # #4) would each need their own model_source value; until they - # exist, build_rag_query above already rejects any request that - # would need one. - "model_source": "omnibioai_gpu", + # M16: "omnibioai_gpu" when RAG used its own default model, + # "claude"/"openai" when this call was BYOK-routed -- RAG's own + # summary.model_source already reflects which one actually + # happened (see that repo's resolve_chat_model), so this just + # relays it rather than hardcoding the pre-M16 default. + "model_source": summary.get("model_source", "omnibioai_gpu"), "domain": rag_result.get("study", domain), "usage": { "queries": 1, - # No per-provider token accounting exists yet (same gap #4) - # -- reporting a fabricated count here would be worse than + # M16: real counts for a BYOK-routed call (RAG's own + # provider client reports them); still None for the default + # path, since Ollama's own response carries no token count + # at all -- reporting a fabricated one would be worse than # omitting it. - "input_tokens": None, - "output_tokens": None, + "input_tokens": summary.get("input_tokens"), + "output_tokens": summary.get("output_tokens"), "billed_by": "query", "latency_ms": latency_ms, }, diff --git a/app/services/v1_store.py b/app/services/v1_store.py index 0882b90..42945db 100644 --- a/app/services/v1_store.py +++ b/app/services/v1_store.py @@ -342,11 +342,19 @@ async def _xadd_usage_event(self, event: dict) -> bool: return False async def emit_usage(self, *, org_id: str, user_id: str, resource: str, trace_id: str, - dedup_key: str, metadata: dict) -> bool: + dedup_key: str, metadata: dict, quantity: int = 1, unit: str = "requests") -> bool: """XADD one billable usage event in omnibioai-usage-client's wire format. event_id is derived from dedup_key, so a replayed or re-emitted request maps to the same id and billing counts it once. + quantity/unit default to 1/"requests" (unchanged from before + M16) -- a BYOK-routed call's llm.tokens.input/output events (see + app/routes/v1.py) pass the real token count and unit="tokens" + instead; dedup_key must be distinct per resource in that case + (e.g. f"{request_id}:tokens:input"), since this method's own + event_id is derived from it and two different resources sharing + one dedup_key would collide. + A successful answer has already been returned to the caller by the time this runs -- a lost event here is lost revenue, never an overcharge, so an XADD failure writes to the local outbox @@ -361,8 +369,8 @@ async def emit_usage(self, *, org_id: str, user_id: str, resource: str, trace_id "service": "api", "resource": resource, "action": "completed", - "quantity": 1, - "unit": "requests", + "quantity": quantity, + "unit": unit, "user_id": str(user_id) if user_id else None, "trace_id": trace_id or None, "metadata": {**metadata, "billable": True}, diff --git a/tests/test_literature_contract.py b/tests/test_literature_contract.py index 29dd6a7..04d4a3b 100644 --- a/tests/test_literature_contract.py +++ b/tests/test_literature_contract.py @@ -60,20 +60,47 @@ def test_no_model_override_variants_are_accepted(model): build_rag_query({"question": "q", "model": model}) # must not raise -def test_model_override_is_rejected(): +def test_unsupported_model_override_is_rejected(): with pytest.raises(UnsupportedRequestError) as exc: - build_rag_query({"question": "q", "model": "claude"}) + build_rag_query({"question": "q", "model": "llama-4"}) assert exc.value.field == "model" -def test_use_own_key_is_rejected(): +def test_use_own_key_false_is_accepted(): + build_rag_query({"question": "q", "use_own_key": False}) # must not raise + + +# --------------------------------------------------------------------------- +# M16 (design audit gap #4): BYOK model routing -- claude/openai require +# use_own_key: true together, since there is no platform-wide key for +# either provider. +# --------------------------------------------------------------------------- + + +def test_use_own_key_alone_without_a_byok_model_is_rejected(): with pytest.raises(UnsupportedRequestError) as exc: build_rag_query({"question": "q", "use_own_key": True}) + assert exc.value.field == "model" + + +@pytest.mark.parametrize("model", ["claude", "openai"]) +def test_byok_model_without_use_own_key_is_rejected(model): + with pytest.raises(UnsupportedRequestError) as exc: + build_rag_query({"question": "q", "model": model}) assert exc.value.field == "use_own_key" -def test_use_own_key_false_is_accepted(): - build_rag_query({"question": "q", "use_own_key": False}) # must not raise +@pytest.mark.parametrize("model", ["claude", "openai"]) +def test_byok_model_with_use_own_key_is_accepted_and_passed_through(model): + query = build_rag_query({"question": "q", "model": model, "use_own_key": True}) + assert query["model"] == model + + +def test_default_model_request_has_no_model_key_in_rag_query(): + """The non-BYOK path's RAG body is unchanged from before M16 -- + no "model" key at all, not even None.""" + query = build_rag_query({"question": "q"}) + assert "model" not in query def test_stream_is_rejected(): @@ -107,6 +134,24 @@ def test_builds_full_public_shape_from_rag_result(): } +def test_byok_summary_fields_flow_through_to_the_public_response(): + """M16: a BYOK-routed RAG result's model_source/input_tokens/ + output_tokens are relayed as-is, not overridden by the pre-M16 + hardcoded defaults.""" + rag_result = { + "study": "Oncology", + "summary": { + "text": "TP53 is a tumor suppressor.", "model": "claude-3-5-sonnet-20241022", + "model_source": "claude", "input_tokens": 120, "output_tokens": 15, + }, + "documents": [], + } + answer = build_public_answer(rag_result, domain="Oncology", request_id="req-1", latency_ms=42) + assert answer["model_source"] == "claude" + assert answer["usage"]["input_tokens"] == 120 + assert answer["usage"]["output_tokens"] == 15 + + def test_falls_back_to_similarity_score_when_citation_confidence_absent(): rag_result = {"documents": [{"pmid": "1", "similarity_score": 0.5}]} answer = build_public_answer(rag_result, domain=None, request_id="r", latency_ms=1) diff --git a/tests/test_v1_literature.py b/tests/test_v1_literature.py index 93df7d2..cb3ef86 100644 --- a/tests/test_v1_literature.py +++ b/tests/test_v1_literature.py @@ -186,8 +186,8 @@ def test_rejects_missing_question(client, redis, upstream): @pytest.mark.parametrize("field,body", [ - ("model", {"question": "q", "model": "claude"}), - ("use_own_key", {"question": "q", "use_own_key": True}), + ("model", {"question": "q", "model": "llama-4"}), + ("use_own_key", {"question": "q", "model": "claude"}), # claude without use_own_key: true ("stream", {"question": "q", "stream": True}), ]) def test_rejects_not_yet_supported_fields_without_calling_upstream_or_billing(client, redis, upstream, field, body): @@ -509,6 +509,113 @@ def test_test_mode_key_supports_idempotency_replay(client, redis, upstream): upstream.assert_not_called() +# --------------------------------------------------------------------------- +# M16 (design audit gap #4): BYOK provider routing. "upstream" patches +# proxy.forward globally, so these tests distinguish the reveal call +# (to omnibioai-auth) from the RAG forward by URL via a custom side_effect. +# --------------------------------------------------------------------------- + +BYOK_BODY = {"question": "What does TP53 do?", "model": "claude", "use_own_key": True} +RAG_RESPONSE_WITH_USAGE = { + "study": "default", + "summary": { + "text": "TP53 [PMID:1]", "model": "claude-3-5-sonnet-20241022", + "model_source": "claude", "input_tokens": 120, "output_tokens": 15, + }, + "documents": [{"pmid": "1", "title": "TP53 review", "year": 2021, "citation_confidence": 0.9}], +} + + +def _byok_forward(reveal_response=(200, {"provider": "claude", "api_key": "sk-ant-real-key"}), + rag_response=(200, RAG_RESPONSE_WITH_USAGE)): + def fake_forward(url, method, headers=None, body=None): + if "provider-keys" in url: + return reveal_response + return rag_response + return fake_forward + + +def test_byok_reveals_key_and_forwards_it_with_the_rag_request(client, redis, upstream, monkeypatch): + monkeypatch.setattr(Config, "PROVIDER_KEY_REVEAL_SECRET", "test-reveal-secret") + upstream.side_effect = _byok_forward() + + resp = _post(client, body=BYOK_BODY) + assert resp.status_code == 200 + assert resp.json()["model_source"] == "claude" + + reveal_call, rag_call = upstream.call_args_list + assert reveal_call.kwargs["url"] == "http://omnibioai-auth:8000/internal/organizations/42/provider-keys/claude/reveal" + assert reveal_call.kwargs["method"] == "POST" + assert reveal_call.kwargs["headers"] == {"X-Provider-Key-Reveal-Secret": "test-reveal-secret"} + + assert rag_call.kwargs["body"]["model"] == "claude" + assert rag_call.kwargs["body"]["provider_api_key"] == "sk-ant-real-key" + # The key must never appear in the headers forwarded to RAG (that's + # still the minted JWT, exactly like every other /v1 call). + assert "sk-ant-real-key" not in str(rag_call.kwargs["headers"]) + + +def test_byok_without_configured_reveal_secret_returns_503(client, redis, upstream, monkeypatch): + monkeypatch.setattr(Config, "PROVIDER_KEY_REVEAL_SECRET", "") + resp = _post(client, body=BYOK_BODY) + assert resp.status_code == 503 + upstream.assert_not_called() + + +def test_byok_no_key_configured_returns_400_without_calling_rag(client, redis, upstream, monkeypatch): + monkeypatch.setattr(Config, "PROVIDER_KEY_REVEAL_SECRET", "test-reveal-secret") + upstream.side_effect = _byok_forward(reveal_response=(404, {"detail": "No claude key is configured"})) + + resp = _post(client, body=BYOK_BODY) + assert resp.status_code == 400 + assert resp.json()["error"]["type"] == "provider_key_not_configured" + assert upstream.call_count == 1 # reveal only -- RAG never called + assert _usage(redis) == [] + + +def test_byok_reveal_5xx_returns_502_without_calling_rag(client, redis, upstream, monkeypatch): + monkeypatch.setattr(Config, "PROVIDER_KEY_REVEAL_SECRET", "test-reveal-secret") + upstream.side_effect = _byok_forward(reveal_response=(500, {"detail": "CONFIG_ENCRYPTION_KEY is not set"})) + + resp = _post(client, body=BYOK_BODY) + assert resp.status_code == 502 + assert upstream.call_count == 1 + + +def test_byok_emits_token_usage_events_alongside_the_answer_event(client, redis, upstream, monkeypatch): + monkeypatch.setattr(Config, "PROVIDER_KEY_REVEAL_SECRET", "test-reveal-secret") + upstream.side_effect = _byok_forward() + + resp = _post(client, body=BYOK_BODY) + assert resp.status_code == 200 + + events = _usage(redis) + by_resource = {e["resource"]: e for e in events} + assert set(by_resource) == {"literature.answer", "llm.tokens.input", "llm.tokens.output"} + assert by_resource["llm.tokens.input"]["quantity"] == 120 and by_resource["llm.tokens.input"]["unit"] == "tokens" + assert by_resource["llm.tokens.output"]["quantity"] == 15 and by_resource["llm.tokens.output"]["unit"] == "tokens" + assert by_resource["literature.answer"]["quantity"] == 1 and by_resource["literature.answer"]["unit"] == "requests" + + +def test_default_path_never_emits_token_usage_events(client, redis, upstream): + """RAG_RESPONSE (the default fixture) has no input_tokens/output_tokens + at all -- the non-BYOK path must not emit token events.""" + resp = _post(client) + assert resp.status_code == 200 + resources = {e["resource"] for e in _usage(redis)} + assert resources == {"literature.answer"} + + +def test_byok_search_is_rejected_regardless_of_use_own_key(client, redis, upstream): + """build_rag_search_query never validates model/use_own_key at all + (search is retrieval-only) -- confirms that stays true after M16.""" + headers = {"Authorization": f"Bearer {KEY}"} + resp = client.post("/v1/literature/search", json={"question": "q", "model": "claude", "use_own_key": True}, + headers=headers) + assert resp.status_code == 200 # accepted; search never routes through a provider at all + upstream.assert_called_once() # only the retrieval-only RAG call, no reveal + + def test_quota_exhausted_returns_402_without_calling_upstream(client, redis, upstream): redis.kv["gateway:v1:quota:42:literature.answer"] = 0 resp = _post(client, idem="q-1") @@ -735,8 +842,11 @@ def test_usage_upstream_5xx_is_502(client, redis, upstream): def test_models_is_free_static_and_never_calls_upstream(client, redis, upstream): resp = client.get("/v1/models", headers={"Authorization": f"Bearer {KEY}"}) assert resp.status_code == 200 - assert resp.json() == {"models": [{"model": "default", "source": "omnibioai_gpu", - "billed_by": "query", "price": None}]} + assert resp.json() == {"models": [ + {"model": "default", "source": "omnibioai_gpu", "billed_by": "query", "price": None}, + {"model": "claude", "source": "claude", "billed_by": "query", "price": None}, + {"model": "openai", "source": "openai", "billed_by": "query", "price": None}, + ]} upstream.assert_not_called() assert _usage(redis) == []