From 75c8da918fa8ae7218c664125c44040e3dfa79fb Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Sun, 27 Sep 2026 23:08:04 +0800 Subject: [PATCH 01/14] [Deps] Require laya 0.3.9 and drop the local weight-init skip laya 0.3.9 builds the encoder under transformers' no_init_weights inside laya.load and loads the checkpoint with strict=True, so the without_weight_init() wrapper from #22 and its test stubs are no longer needed. The lock moves laya from 0.3.5 to 0.3.9, the first release with the skip; later releases change MPS precision and are left for their own bump. The system_one config-error test pins the version lookup, so it passes with the laya extra installed as well as on the core install. Closes #28 --- CHANGELOG.md | 4 +- pyproject.toml | 2 +- s1a/decision_models/laya.py | 18 +------- tests/test_decision_models_factory.py | 5 +-- tests/test_decision_models_laya.py | 64 ++++----------------------- uv.lock | 8 ++-- 6 files changed, 17 insertions(+), 84 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7b1eca6..fd92ef6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,8 +20,8 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver ### Changed -- `--model laya` loads in about 3 s instead of about 35 s: the encoder is built with transformers' weight init - off, since the checkpoint replaces every weight. Weights and answers are unchanged. +- `--model laya` loads in about 3 s instead of about 35 s: the `laya` extra now needs laya 0.3.9 or later, which + builds the encoder with transformers' weight init off, since the checkpoint replaces every weight. - `--model` picks the model on every agent, on `decide` and on `probe`: `jev`, `laya`, `cua`, `llm`, `random` or `rule`. The results table's column, the replay page's badge data and a browser run's `answer.json` name it `model` as well; the replay still reads the `slot` key of records written by 0.1.0. diff --git a/pyproject.toml b/pyproject.toml index 682ddda..89e0e3c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -40,7 +40,7 @@ alfworld-visual = [ "torchvision>=0.15", ] report = ["pillow>=10", "playwright>=1.45"] # evals.replay: pages, GIFs; never imported by a runner -laya = ["laya>=0.3.4"] # the in-process decision model behind --model laya; pulls torch and transformers +laya = ["laya>=0.3.9"] # the in-process decision model behind --model laya; pulls torch and transformers cua = [ # Cua-S1 Nano behind --model cua; pinned to the Cua PR that ships the checkpoints (trycua/cua#4023), pulls torch "cua-s1 @ git+https://github.com/trycua/cua.git@aea61b6eb97e2d8c0f6f71eb804e5769fe910af4#subdirectory=libs/cua-s1/python", "huggingface-hub>=0.24", diff --git a/s1a/decision_models/laya.py b/s1a/decision_models/laya.py index b4178a2..1fb786d 100644 --- a/s1a/decision_models/laya.py +++ b/s1a/decision_models/laya.py @@ -7,10 +7,8 @@ from __future__ import annotations import asyncio -import importlib import os import time -from contextlib import AbstractContextManager, nullcontext from importlib import metadata from typing import Any @@ -34,19 +32,6 @@ def laya_question(question: Question) -> Json: return {"type": "noul", "instructions": question.question, **criteria} -def without_weight_init() -> AbstractContextManager[Any]: - """transformers' ``no_init_weights``. ``laya.load`` builds the encoder from its config, which draws every weight - at random (about 30 s of the load on CPU), then loads the checkpoint over all of them with ``strict=True``, so - the draw is thrown away. The helper sits in ``transformers.initialization`` from 5.0 and in - ``transformers.modeling_utils`` before; without either the load runs as it is.""" - for module in ("transformers.initialization", "transformers.modeling_utils"): - try: - return importlib.import_module(module).no_init_weights() - except (ImportError, AttributeError): - continue - return nullcontext() - - class LayaModel(DecisionModel): """Laya's ``Agent`` (or anything with ``system_one(state, questions)`` and a ``cfg``) behind the interface.""" @@ -113,8 +98,7 @@ def from_env(cls) -> "LayaModel": ) from exc model = os.getenv("LAYA_MODEL") or LAYA_DEFAULT_MODEL subfolder = os.getenv("LAYA_SUBFOLDER") or None - with without_weight_init(): - agent = laya.load(model, device=os.getenv("LAYA_DEVICE") or None, subfolder=subfolder) + agent = laya.load(model, device=os.getenv("LAYA_DEVICE") or None, subfolder=subfolder) if not callable(getattr(agent, "system_one", None)): try: version = metadata.version("laya") diff --git a/tests/test_decision_models_factory.py b/tests/test_decision_models_factory.py index 3c85f6d..d38866e 100644 --- a/tests/test_decision_models_factory.py +++ b/tests/test_decision_models_factory.py @@ -5,7 +5,6 @@ import os import sys -from contextlib import nullcontext from types import SimpleNamespace from unittest import TestCase from unittest.mock import patch @@ -32,9 +31,7 @@ def test_every_name_builds_its_class(self) -> None: fake_laya = SimpleNamespace( load=lambda *a, **k: SimpleNamespace(cfg={}, system_one=lambda state, questions: {}) ) - # transformers' helper stubbed so torch is not imported inside patch.dict: see TestFromEnv in the laya tests. - modules = {"laya": fake_laya, "transformers.initialization": SimpleNamespace(no_init_weights=nullcontext)} - with patch.dict(sys.modules, modules), patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}): + with patch.dict(sys.modules, {"laya": fake_laya}), patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}): self.assertIsInstance(build_model("laya"), LayaModel) self.assertIsInstance(build_model("random", seed=3), RandomModel) rule = build_model("rule", rule=("always-inc", lambda state, options: "inc")) diff --git a/tests/test_decision_models_laya.py b/tests/test_decision_models_laya.py index 992924d..6c91122 100644 --- a/tests/test_decision_models_laya.py +++ b/tests/test_decision_models_laya.py @@ -1,14 +1,12 @@ # coding: utf-8 """``LayaModel`` over a fake ``laya.Agent`` (no torch): the contract, the question mapping, the error wrap, -the filled-window error, and ``from_env`` with and without the extra and with the weight init off.""" +the filled-window error, and ``from_env`` with and without the extra.""" from __future__ import annotations import os import sys import time -from collections.abc import Iterator -from contextlib import contextmanager, nullcontext from types import SimpleNamespace from typing import Any from unittest import IsolatedAsyncioTestCase, TestCase @@ -163,14 +161,6 @@ async def test_the_window_scales_with_the_number_of_questions(self) -> None: class TestFromEnv(TestCase): - def setUp(self) -> None: - # A stand-in for transformers' helper, so no test imports torch: patch.dict drops a torch imported inside it - # from sys.modules, and importing torch a second time in one process crashes it. - helper = {"transformers.initialization": SimpleNamespace(no_init_weights=nullcontext)} - stub = patch.dict(sys.modules, helper) - stub.start() - self.addCleanup(stub.stop) - def test_without_the_extra_it_is_a_config_error_naming_the_extra(self) -> None: with patch.dict(sys.modules, {"laya": None}): with self.assertRaises(BaseError) as caught: @@ -180,12 +170,17 @@ def test_without_the_extra_it_is_a_config_error_naming_the_extra(self) -> None: def test_an_agent_without_system_one_is_a_config_error_naming_the_method(self) -> None: fake_laya = SimpleNamespace(load=lambda *a, **k: SimpleNamespace(cfg={})) - with patch.dict(sys.modules, {"laya": fake_laya}), patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}): + # The installed version is pinned so the message reads the same with and without the extra. + with ( + patch.dict(sys.modules, {"laya": fake_laya}), + patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}), + patch.object(laya_module.metadata, "version", return_value="0.3.0"), + ): with self.assertRaises(BaseError) as caught: LayaModel.from_env() self.assertEqual(caught.exception.status, StatusCode.MODEL_SERVICE_CONFIG_ERROR) self.assertIn("system_one", str(caught.exception)) - self.assertIn("laya unknown", str(caught.exception)) + self.assertIn("laya 0.3.0", str(caught.exception)) def test_the_env_names_the_checkpoint_and_overrides_the_window(self) -> None: loads: list[tuple[Any, ...]] = [] @@ -207,49 +202,6 @@ def load(model: str, device: Any = None, token: Any = None, subfolder: Any = Non self.assertEqual(decision_model.model, "convaiinnovations/laya/multilingual") self.assertEqual(decision_model._agent.cfg, {"max_len": 1024, "head_max_len": 512}) - def test_the_checkpoint_loads_with_the_weight_init_off(self) -> None: - events: list[str] = [] - - @contextmanager - def no_init_weights() -> Iterator[None]: - events.append("off") - yield - events.append("on") - - def load(*args: Any, **kwargs: Any) -> FakeLayaAgent: - events.append("load") - return FakeLayaAgent() - - modules = { - "laya": SimpleNamespace(load=load), - "transformers.initialization": SimpleNamespace(no_init_weights=no_init_weights), - } - with patch.dict(sys.modules, modules), patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}): - LayaModel.from_env() - self.assertEqual(events, ["off", "load", "on"]) - - def test_the_4x_location_of_the_helper_is_used_when_the_5x_one_is_missing(self) -> None: - @contextmanager - def no_init_weights() -> Iterator[str]: - yield "4.x" - - modules = { - "transformers.initialization": None, - "transformers.modeling_utils": SimpleNamespace(no_init_weights=no_init_weights), - } - with patch.dict(sys.modules, modules), laya_module.without_weight_init() as entered: - self.assertEqual(entered, "4.x") - - def test_without_the_helper_the_checkpoint_still_loads(self) -> None: - modules = { - "laya": SimpleNamespace(load=lambda *a, **k: FakeLayaAgent()), - "transformers.initialization": None, - "transformers.modeling_utils": SimpleNamespace(), - } - with patch.dict(sys.modules, modules), patch.dict(os.environ, {"LAYA_SUBFOLDER": ""}): - decision_model = LayaModel.from_env() - self.assertIsInstance(decision_model._agent, FakeLayaAgent) - def test_the_defaults_when_the_env_is_empty(self) -> None: env = {"LAYA_MODEL": "", "LAYA_SUBFOLDER": "", "LAYA_DEVICE": "", "LAYA_MAX_LEN": "", "LAYA_HEAD_MAX_LEN": ""} with patch.dict(sys.modules, {"laya": SimpleNamespace(load=lambda *a, **k: FakeLayaAgent())}): diff --git a/uv.lock b/uv.lock index 52ec63f..d46522c 100644 --- a/uv.lock +++ b/uv.lock @@ -2287,7 +2287,7 @@ wheels = [ [[package]] name = "laya" -version = "0.3.5" +version = "0.3.9" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "huggingface-hub" }, @@ -2297,9 +2297,9 @@ dependencies = [ { name = "torch" }, { name = "transformers" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/f7/f1/12459ecda42123f93da36f9f1e0c9b7b9191648fa68e7e0b09d292cc6dce/laya-0.3.5.tar.gz", hash = "sha256:5e8a4c2b38dbddc0febe7443f26f74fd9d7571fb172648b1481830c667d59219", size = 69616, upload-time = "2026-09-21T18:20:10.029Z" } +sdist = { url = "https://files.pythonhosted.org/packages/4f/89/6eec1f50cbc421fa5b1733d166387eae6bc0dc222a8928ac875c271c8cf1/laya-0.3.9.tar.gz", hash = "sha256:3d255a778a1c70e2ed1fb4129147ccbb7fa96f3d0a0c910b2f41ddacb49ea5f6", size = 201267, upload-time = "2026-09-23T14:55:16.694Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/ec/1c/a9903d5c3c51579f45f9656f733c8a2d985b16a3cb7594edf45369d1f995/laya-0.3.5-py3-none-any.whl", hash = "sha256:4c57f64cbaf893bb5c7b4affddc2bf21a819f55df51941689f11868583be2903", size = 41658, upload-time = "2026-09-21T18:20:08.641Z" }, + { url = "https://files.pythonhosted.org/packages/d0/cc/75f6b030f26b78a0ac0165c455866ab897eee251f7edb115733d4675fcf6/laya-0.3.9-py3-none-any.whl", hash = "sha256:8080d99792867096c970b1b24464e888f3c37d902f2c947be1f8baa3677c9717", size = 93224, upload-time = "2026-09-23T14:55:15.288Z" }, ] [[package]] @@ -5200,7 +5200,7 @@ requires-dist = [ { name = "cua-s1", marker = "extra == 'cua'", git = "https://github.com/trycua/cua.git?subdirectory=libs%2Fcua-s1%2Fpython&rev=aea61b6eb97e2d8c0f6f71eb804e5769fe910af4" }, { name = "httpx", specifier = ">=0.28" }, { name = "huggingface-hub", marker = "extra == 'cua'", specifier = ">=0.24" }, - { name = "laya", marker = "extra == 'laya'", specifier = ">=0.3.4" }, + { name = "laya", marker = "extra == 'laya'", specifier = ">=0.3.9" }, { name = "mcp", specifier = ">=1.26" }, { name = "opencv-python-headless", marker = "extra == 'alfworld-visual'", specifier = ">=4.10" }, { name = "openjiuwen", git = "https://github.com/ThinkFlowLab/agent-core?rev=jj-0.1.0" }, From 4cccf98ca2c9be6f28e79c79bd604e29d3c36cdf Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:37:22 +0800 Subject: [PATCH 02/14] [Deps] Move laya to 0.3.20, the version the served worker runs system1-omni's Laya worker pins laya[serve]==0.3.20. Comparing in-process Laya with the served one (#20) needs the same library on both sides, so the in-process extra moves from 0.3.9 to 0.3.20. 0.3.20 autocasts to fp16 on MPS for requests with five or more questions; CPU runs and smaller requests are unchanged. --- pyproject.toml | 2 +- uv.lock | 72 +++++++++++++++++++++++++------------------------- 2 files changed, 37 insertions(+), 37 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 89e0e3c..083c2b3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -40,7 +40,7 @@ alfworld-visual = [ "torchvision>=0.15", ] report = ["pillow>=10", "playwright>=1.45"] # evals.replay: pages, GIFs; never imported by a runner -laya = ["laya>=0.3.9"] # the in-process decision model behind --model laya; pulls torch and transformers +laya = ["laya>=0.3.20"] # the in-process decision model behind --model laya; pulls torch and transformers cua = [ # Cua-S1 Nano behind --model cua; pinned to the Cua PR that ships the checkpoints (trycua/cua#4023), pulls torch "cua-s1 @ git+https://github.com/trycua/cua.git@aea61b6eb97e2d8c0f6f71eb804e5769fe910af4#subdirectory=libs/cua-s1/python", "huggingface-hub>=0.24", diff --git a/uv.lock b/uv.lock index d46522c..c8be294 100644 --- a/uv.lock +++ b/uv.lock @@ -739,7 +739,7 @@ resolution-markers = [ "python_full_version < '3.12' and sys_platform != 'emscripten' and sys_platform != 'win32'", ] dependencies = [ - { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" } }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.12'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/58/01/1253e6698a07380cd31a736d248a3f2a50a7c88779a1813da27503cadc2a/contourpy-1.3.3.tar.gz", hash = "sha256:083e12155b210502d0bca491432bb04d56dc3432f95a979b429f2848c3dbe880", size = 13466174, upload-time = "2025-07-26T12:03:12.549Z" } wheels = [ @@ -829,7 +829,7 @@ resolution-markers = [ "python_full_version >= '3.12' and python_full_version < '3.14' and sys_platform != 'emscripten' and sys_platform != 'win32'", ] dependencies = [ - { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" } }, + { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/83/5a/a55177dd22553a277388e8a1b3220e92de91bacb28356cdc73caa240121d/contourpy-1.4.0.tar.gz", hash = "sha256:20156f5a1ac4f8ce02656e39a61e82164a3d359796dc8026f75b062783d500e1", size = 13323726, upload-time = "2026-09-11T19:05:05.808Z" } wheels = [ @@ -990,7 +990,7 @@ name = "cuda-bindings" version = "13.4.2" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "cuda-pathfinder" }, + { name = "cuda-pathfinder", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/19/6f/e00ffcebcad6405a2326a52612eff98240a67c637dfed516e61b881e2db3/cuda_bindings-13.4.2-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fc0a18f65b26459cd3c49c0250a44b419cdf790cffcb096ff4d46a1973f6f4b7", size = 6484069, upload-time = "2026-09-17T22:05:52.594Z" }, @@ -1023,43 +1023,43 @@ wheels = [ [package.optional-dependencies] cublas = [ - { name = "nvidia-cublas", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-cuda-nvrtc", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cublas", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-cuda-nvrtc", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cudart = [ - { name = "nvidia-cuda-runtime", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cuda-runtime", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cufft = [ - { name = "nvidia-cufft", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-nvjitlink", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cufft", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-nvjitlink", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cufile = [ - { name = "nvidia-cufile", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cufile", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cupti = [ - { name = "nvidia-cuda-cupti", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cuda-cupti", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] curand = [ - { name = "nvidia-curand", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-curand", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cusolver = [ - { name = "nvidia-cublas", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-cusolver", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-cusparse", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-nvjitlink", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cublas", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-cusolver", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-cusparse", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-nvjitlink", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] cusparse = [ - { name = "nvidia-cusparse", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, - { name = "nvidia-nvjitlink", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cusparse", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, + { name = "nvidia-nvjitlink", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] nvjitlink = [ - { name = "nvidia-nvjitlink", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-nvjitlink", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] nvrtc = [ - { name = "nvidia-cuda-nvrtc", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-cuda-nvrtc", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] nvtx = [ - { name = "nvidia-nvtx", marker = "platform_machine == 'aarch64' or platform_machine == 'x86_64'" }, + { name = "nvidia-nvtx", marker = "(platform_machine == 'aarch64' and sys_platform == 'linux') or (platform_machine == 'x86_64' and sys_platform == 'linux')" }, ] [[package]] @@ -1746,8 +1746,8 @@ name = "httpcore2" version = "2.13.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "h11" }, - { name = "truststore" }, + { name = "h11", marker = "sys_platform != 'emscripten'" }, + { name = "truststore", marker = "sys_platform != 'emscripten'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/15/8c/e925b1c92018abb3a1863ce1549d76d2381e334d21d65d4ac8f65dabd78a/httpcore2-2.13.0.tar.gz", hash = "sha256:2adc8be4fb285fbcd6d894298db3b52c177e74b6674eda3a76bd36be3292a3db", size = 67740, upload-time = "2026-09-14T14:18:04.717Z" } wheels = [ @@ -1838,7 +1838,7 @@ name = "importlib-metadata" version = "9.0.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "zipp" }, + { name = "zipp", marker = "python_full_version < '3.12'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/6f/7e/1e7e8dc30634b93ebb3d58a3dea569ad146e656218d3960ab04f62047b29/importlib_metadata-9.0.1.tar.gz", hash = "sha256:ab830580bc0ef3db61ce8fae716389e5462b67e033018bab6d8f80ef17172f99", size = 59124, upload-time = "2026-08-28T15:30:34.646Z" } wheels = [ @@ -2287,7 +2287,7 @@ wheels = [ [[package]] name = "laya" -version = "0.3.9" +version = "0.3.20" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "huggingface-hub" }, @@ -2297,9 +2297,9 @@ dependencies = [ { name = "torch" }, { name = "transformers" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/4f/89/6eec1f50cbc421fa5b1733d166387eae6bc0dc222a8928ac875c271c8cf1/laya-0.3.9.tar.gz", hash = "sha256:3d255a778a1c70e2ed1fb4129147ccbb7fa96f3d0a0c910b2f41ddacb49ea5f6", size = 201267, upload-time = "2026-09-23T14:55:16.694Z" } +sdist = { url = "https://files.pythonhosted.org/packages/31/82/0964e13e1a67ae4ac2824c3115470d31cd149a065d971a5dec6e4049ca23/laya-0.3.20.tar.gz", hash = "sha256:692de1346cf0239bb7bbcffdf0cfbd538c24834f8bc61e9b60f4106465c033c9", size = 275028, upload-time = "2026-09-24T05:41:07.653Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/d0/cc/75f6b030f26b78a0ac0165c455866ab897eee251f7edb115733d4675fcf6/laya-0.3.9-py3-none-any.whl", hash = "sha256:8080d99792867096c970b1b24464e888f3c37d902f2c947be1f8baa3677c9717", size = 93224, upload-time = "2026-09-23T14:55:15.288Z" }, + { url = "https://files.pythonhosted.org/packages/34/0c/80982f9ad6f52ca9cb146cfc00fac79e7b7f4388f00f4197575a3170be48/laya-0.3.20-py3-none-any.whl", hash = "sha256:6039e802fa5effb8dd492061cd7ad39a43087beadc4a4fa4a649614e77eb83d4", size = 118520, upload-time = "2026-09-24T05:41:06.273Z" }, ] [[package]] @@ -3123,7 +3123,7 @@ name = "nvidia-cublas" version = "13.1.1.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cuda-nvrtc" }, + { name = "nvidia-cuda-nvrtc", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/a7/a1/0bd24ee8c8d03adac032fd2909426a00c88f8c57961b1277ded97f91119f/nvidia_cublas-13.1.1.3-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:b7a210458267ac818974c53038fbec2e969d5c99f305ab15c72522fa9f001dd5", size = 542848918, upload-time = "2026-04-08T18:46:22.985Z" }, @@ -3162,7 +3162,7 @@ name = "nvidia-cudnn-cu13" version = "9.24.0.43" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas" }, + { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/ca/30/7c257e3d5cb4fecb147b93895c66e29c93f8e76d74b45bb418ff0587c4ec/nvidia_cudnn_cu13-9.24.0.43-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:a6812a554a1ff0413e9c52b84c26c050380649ab9615f9c16bded368ce9f421f", size = 650976863, upload-time = "2026-07-02T16:23:39.248Z" }, @@ -3174,7 +3174,7 @@ name = "nvidia-cufft" version = "12.0.0.61" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink" }, + { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/8b/ae/f417a75c0259e85c1d2f83ca4e960289a5f814ed0cea74d18c353d3e989d/nvidia_cufft-12.0.0.61-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:2708c852ef8cd89d1d2068bdbece0aa188813a0c934db3779b9b1faa8442e5f5", size = 214053554, upload-time = "2025-09-04T08:31:38.196Z" }, @@ -3204,9 +3204,9 @@ name = "nvidia-cusolver" version = "12.0.4.66" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-cublas" }, - { name = "nvidia-cusparse" }, - { name = "nvidia-nvjitlink" }, + { name = "nvidia-cublas", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-cusparse", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/c8/c3/b30c9e935fc01e3da443ec0116ed1b2a009bb867f5324d3f2d7e533e776b/nvidia_cusolver-12.0.4.66-py3-none-manylinux_2_27_aarch64.whl", hash = "sha256:02c2457eaa9e39de20f880f4bd8820e6a1cfb9f9a34f820eb12a155aa5bc92d2", size = 223467760, upload-time = "2025-09-04T08:33:04.222Z" }, @@ -3218,7 +3218,7 @@ name = "nvidia-cusparse" version = "12.6.3.3" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "nvidia-nvjitlink" }, + { name = "nvidia-nvjitlink", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] wheels = [ { url = "https://files.pythonhosted.org/packages/f8/94/5c26f33738ae35276672f12615a64bd008ed5be6d1ebcb23579285d960a9/nvidia_cusparse-12.6.3.3-py3-none-manylinux2014_aarch64.manylinux_2_17_aarch64.whl", hash = "sha256:80bcc4662f23f1054ee334a15c72b8940402975e0eab63178fc7e670aa59472c", size = 162155568, upload-time = "2025-09-04T08:33:42.864Z" }, @@ -4845,8 +4845,8 @@ name = "secretstorage" version = "3.5.0" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "cryptography" }, - { name = "jeepney" }, + { name = "cryptography", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, + { name = "jeepney", marker = "sys_platform != 'emscripten' and sys_platform != 'win32'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/1c/03/e834bcd866f2f8a49a85eaff47340affa3bfa391ee9912a952a1faa68c7b/secretstorage-3.5.0.tar.gz", hash = "sha256:f04b8e4689cbce351744d5537bf6b1329c6fc68f91fa666f60a380edddcd11be", size = 19884, upload-time = "2025-11-23T19:02:53.191Z" } wheels = [ @@ -5200,7 +5200,7 @@ requires-dist = [ { name = "cua-s1", marker = "extra == 'cua'", git = "https://github.com/trycua/cua.git?subdirectory=libs%2Fcua-s1%2Fpython&rev=aea61b6eb97e2d8c0f6f71eb804e5769fe910af4" }, { name = "httpx", specifier = ">=0.28" }, { name = "huggingface-hub", marker = "extra == 'cua'", specifier = ">=0.24" }, - { name = "laya", marker = "extra == 'laya'", specifier = ">=0.3.9" }, + { name = "laya", marker = "extra == 'laya'", specifier = ">=0.3.20" }, { name = "mcp", specifier = ">=1.26" }, { name = "opencv-python-headless", marker = "extra == 'alfworld-visual'", specifier = ">=4.10" }, { name = "openjiuwen", git = "https://github.com/ThinkFlowLab/agent-core?rev=jj-0.1.0" }, From 162a37dff6cb6e486c6a1162118e9639d985b895 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:22:22 +0800 Subject: [PATCH 03/14] [Docs] Design and API spec for served Laya (#20) docs/served-laya.md designs the served Laya decision model: requirements, the target interface, what today's server lacks and where each part gets built, the client (--model laya-served), error handling and trade-offs. docs/api/laya-systemone.openapi.yaml is the target interface as OpenAPI 3.1 (RFC 9457 errors with stable codes, request ids, Server-Timing, /livez and /readyz, 503 with Retry-After, served_by per response); each item is marked implemented or planned. laya-systemone.current.openapi.yaml specifies today's server (laya-serve 0.3.20 behind the system1-omni worker and frontend), checked against traffic from a running worker. --- README.md | 1 + docs/api/laya-systemone.current.openapi.yaml | 451 ++++++++++++++++ docs/api/laya-systemone.openapi.yaml | 508 +++++++++++++++++++ docs/served-laya.md | 112 ++++ 4 files changed, 1072 insertions(+) create mode 100644 docs/api/laya-systemone.current.openapi.yaml create mode 100644 docs/api/laya-systemone.openapi.yaml create mode 100644 docs/served-laya.md diff --git a/README.md b/README.md index 1df99dd..83b17a3 100644 --- a/README.md +++ b/README.md @@ -141,6 +141,7 @@ interface fits: [docs/architecture.md](docs/architecture.md), [docs/decision-mod - [docs/agents.md](docs/agents.md): every agent with its flags, run command and extra. - [docs/architecture.md](docs/architecture.md) and [docs/decision-models.md](docs/decision-models.md): the fronts, the model slot, the model interface, adding a backend. - [docs/browser-front.md](docs/browser-front.md): the browser policy, decision by decision. +- [docs/served-laya.md](docs/served-laya.md): Laya served by system1-omni as a decision model over HTTP, with its [API spec](docs/api/laya-systemone.openapi.yaml). - [docs/configuration.md](docs/configuration.md): environment variables, defaults and reader subsystems in one table. - [docs/glossary.md](docs/glossary.md): terms the documentation glosses on first mention. - [docs/why.md](docs/why.md): the problem, the philosophy, the precedents. diff --git a/docs/api/laya-systemone.current.openapi.yaml b/docs/api/laya-systemone.current.openapi.yaml new file mode 100644 index 0000000..2df7db4 --- /dev/null +++ b/docs/api/laya-systemone.current.openapi.yaml @@ -0,0 +1,451 @@ +openapi: 3.1.0 +info: + title: Laya decisions (System 1) — /v1/systemone, as implemented today + version: 0.1.0 + summary: Typed decisions from a Laya checkpoint over HTTP. + description: | + The interface a client such as system1-agents uses to get decisions from Laya served by + system1-omni: the Laya worker (`src/models/laya/worker.py`, laya-serve 0.3.20 underneath), either + directly or behind the Rust frontend (`omni-jev`), which forwards requests and responses unchanged. + + This document describes the behaviour of laya-serve 0.3.20, the worker and the frontend as they + are, captured from a running worker. Differences from common API practice are listed under + `x-known-gaps` rather than described as if they existed. + + **Idempotency.** A decision has no side effects and the same request returns the same answers + (Laya is deterministic for a given checkpoint, device and precision). Clients may retry any + request whose outcome they did not receive. + + **Retries.** Neither the frontend nor the worker retries. The client owns retries and their + deadline. + + **Concurrency.** The worker runs one forward pass at a time; concurrent requests queue, so latency + grows with concurrency while throughput stays flat. + license: + name: Apache-2.0 +servers: + - url: http://127.0.0.1:8000 + description: Laya worker directly (LAYA_HOST / LAYA_PORT) + - url: http://127.0.0.1:8080 + description: Rust frontend in front of the worker (OMNI_JEV_BIND) +security: + - {} + - bearerAuth: [] +tags: + - name: decisions + - name: health +paths: + /v1/systemone: + post: + tags: [decisions] + operationId: decide + summary: Answer typed questions about one state + description: | + All questions are answered in one forward pass. Question ids are chosen by the client and + echoed as the keys of `answers`. + security: + - {} + - bearerAuth: [] + requestBody: + required: true + description: At most 2 MiB (2,097,152 bytes); larger bodies get 413 before they are parsed. + content: + application/json: + schema: { $ref: '#/components/schemas/DecisionRequest' } + examples: + combined: + summary: One question of each type + value: + model: english + state: I was charged twice for my order. Please refund the duplicate today. + questions: + department: + type: choice + instructions: Which team should handle this? + criteria: { billing: Charges and refunds, technical: Software problems } + urgency: + type: score + instructions: How urgent is the request? + criteria: [Not urgent, Needs attention soon, Needs attention immediately] + refund: + type: noul + instructions: Does the customer ask for a refund? + responses: + '200': + description: Answers, one per question id. + content: + application/json: + schema: { $ref: '#/components/schemas/DecisionResponse' } + examples: + combined: + summary: Response to the combined request (captured, M1 Pro, MPS, fp16 weights) + value: + model: laya-rl-agent + answers: + department: + type: choice + choice: billing + probabilities: { billing: 0.952, technical: 0.048 } + confidence: 0.7222 + answer_confidence: 0.952 + action: { act_probability: 1.0 } + urgency: + type: score + score: 1.7841 + legend: { '0': Not urgent, '1': Needs attention soon, '2': Needs attention immediately } + probabilities: { '0': 0.0241, '1': 0.1677, '2': 0.8082 } + confidence: 0.4891 + answer_confidence: 0.8082 + action: { act_probability: 1.0 } + refund: + type: noul + noul: 0.9161 + confidence: 0.9161 + answer_confidence: 0.9161 + action: { act_probability: 1.0 } + usage: { input_tokens: 137, output_tokens: 0 } + routing: + model: english + repo: convaiinnovations/laya + reason: explicit model='english' + detection: null + workflow: null + '400': + description: The body is not valid JSON, is not an object, lacks `questions`, or `questions` is not an object. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + examples: + malformed: { value: { detail: request body must be valid JSON } } + noQuestions: { value: { detail: "request body must be an object with a 'questions' field" } } + '401': + description: The worker was started with `LAYA_API_KEY` and the bearer token is missing or wrong. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + example: { detail: invalid or missing bearer token } + '413': + description: Body over 2 MiB, more than 64 questions, or a state over 50,000 characters. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + example: { detail: too many questions (65 > 64) } + '422': + description: | + A question is invalid: unknown type, no `instructions`, wrongly shaped or empty `criteria`, + `labels` on a non-noul question, or options longer than the checkpoint's option window. + `detail` names the question id. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + example: { detail: "question 'q': unknown type 'bogus'; use one of ['choice', 'noul', 'score']" } + '500': + description: The forward pass failed. `detail` carries no internal information. + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + example: { detail: inference failed } + '502': + description: Frontend only. The worker could not be reached. + content: + text/plain: + schema: { type: string } + example: "backend unavailable\n" + '504': + description: Frontend only. The worker did not answer within the frontend's 60 s timeout. + content: + text/plain: + schema: { type: string } + example: "backend timed out\n" + /health: + get: + tags: [health] + operationId: health + summary: Readiness and what the worker is running + description: | + The worker listens only after it has loaded and warmed up every model, so any response means + ready; a refused connection means not ready yet. Through the frontend, the worker's response is + forwarded (502 or 504 if the worker is unreachable). No authentication. + + Plain laya-serve (without the worker) answers `{status, loaded, device}` as soon as it binds, + before any warmup, and `device` is the configured `LAYA_DEVICE`, not the device in use. + security: + - {} + responses: + '200': + description: Ready. + content: + application/json: + schema: { $ref: '#/components/schemas/Health' } + example: + status: ok + ready: true + loaded: [english] + device: mps + requested_device: mps + device_mismatch: false + weights_dtype: torch.float16 + autocast_dtype: torch.float16 + mps_amp_min_rows: 5 + checkpoint: convaiinnovations/laya + revision: 55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851 + warmup_ms: 34426.6 + models: + english: + device: mps + requested_device: mps + device_mismatch: false + weights_dtype: torch.float16 + autocast_dtype: torch.float16 + mps_amp_min_rows: 5 + checkpoint: convaiinnovations/laya + revision: 55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851 + warmup_ms: 34426.6 + compile: { mode: 'on', graphs_at_ready: 3, graphs_now: 3, recompiled_after_ready: false } + '502': + description: Frontend only. The worker could not be reached. + content: + text/plain: + schema: { type: string } + '504': + description: Frontend only. The worker did not answer in time. + content: + text/plain: + schema: { type: string } +components: + securitySchemes: + bearerAuth: + type: http + scheme: bearer + description: Required only when the worker is started with `LAYA_API_KEY`. The frontend forwards the header. + schemas: + DecisionRequest: + type: object + required: [questions] + properties: + model: + type: string + description: | + Checkpoint: `english`, `multilingual`, `typed-decisions`, an alias (`en`, `laya`, `default`), + or a published Hugging Face id. Omitted or unrecognised (e.g. a Jev model id), the worker + picks a checkpoint from the state's language. Only loaded checkpoints are served warm. + examples: [english] + state: + description: What the questions are about. At most 50,000 characters (as text or as serialised JSON). + oneOf: + - type: string + - type: object + - type: array + questions: + type: object + description: Question id → question. At least 1 and at most 64. + minProperties: 1 + maxProperties: 64 + additionalProperties: { $ref: '#/components/schemas/Question' } + Question: + oneOf: + - $ref: '#/components/schemas/ChoiceQuestion' + - $ref: '#/components/schemas/ScoreQuestion' + - $ref: '#/components/schemas/NoulQuestion' + discriminator: + propertyName: type + mapping: + choice: '#/components/schemas/ChoiceQuestion' + score: '#/components/schemas/ScoreQuestion' + noul: '#/components/schemas/NoulQuestion' + Instructions: + description: What to decide. An object is serialised to JSON before the model reads it. + oneOf: + - type: string + - type: object + ChoiceQuestion: + type: object + required: [type, instructions, criteria] + properties: + type: { const: choice } + instructions: { $ref: '#/components/schemas/Instructions' } + criteria: + description: Options, as label → description or as a list of labels. At least one. + oneOf: + - type: object + minProperties: 1 + additionalProperties: {} + - type: array + minItems: 1 + items: { type: string } + ScoreQuestion: + type: object + required: [type, instructions, criteria] + properties: + type: { const: score } + instructions: { $ref: '#/components/schemas/Instructions' } + criteria: + type: array + description: Level descriptions, level 0 first. At least one. + minItems: 1 + items: {} + NoulQuestion: + type: object + required: [type, instructions] + properties: + type: { const: noul } + instructions: + $ref: '#/components/schemas/Instructions' + description: A statement or yes/no question; a plain string is what Laya was trained on. + criteria: + type: object + description: Optional texts for the two outcomes, keyed only `true` and/or `false`. + propertyNames: { enum: ['true', 'false', 'True', 'False', 'TRUE', 'FALSE'] } + labels: + type: object + description: Optional wording of the two outcomes shown to the model; exactly `false` and `true`, distinct and non-empty. + required: ['false', 'true'] + additionalProperties: false + properties: + 'false': { type: string, minLength: 1 } + 'true': { type: string, minLength: 1 } + DecisionResponse: + type: object + required: [model, answers, usage] + properties: + model: + type: string + description: Always `laya-rl-agent`, whichever checkpoint answered. Use `routing.repo` and `/health` for identity. + answers: + type: object + description: Question id → answer, one per question in the request. + additionalProperties: { $ref: '#/components/schemas/Answer' } + usage: { $ref: '#/components/schemas/Usage' } + routing: { $ref: '#/components/schemas/Routing' } + Answer: + oneOf: + - $ref: '#/components/schemas/ChoiceAnswer' + - $ref: '#/components/schemas/ScoreAnswer' + - $ref: '#/components/schemas/NoulAnswer' + discriminator: + propertyName: type + mapping: + choice: '#/components/schemas/ChoiceAnswer' + score: '#/components/schemas/ScoreAnswer' + noul: '#/components/schemas/NoulAnswer' + Probability: + type: number + minimum: 0 + maximum: 1 + description: Rounded to 4 decimals. + AnswerCommon: + type: object + required: [type, confidence, answer_confidence, action] + properties: + confidence: + $ref: '#/components/schemas/Probability' + description: Calibrated confidence in the answer. + answer_confidence: { $ref: '#/components/schemas/Probability' } + action: + type: object + required: [act_probability] + properties: + act_probability: { $ref: '#/components/schemas/Probability' } + ChoiceAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, choice, probabilities] + properties: + type: { const: choice } + choice: { type: string, description: The chosen label. } + probabilities: + type: object + description: Label → probability, summing to 1 (within rounding). + additionalProperties: { $ref: '#/components/schemas/Probability' } + ScoreAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, score, legend, probabilities] + properties: + type: { const: score } + score: { type: number, minimum: 0, description: Expected level index. } + legend: + type: object + description: Level index (as a string) → level description. + additionalProperties: {} + probabilities: + type: object + description: Level index (as a string) → probability, summing to 1 (within rounding). + additionalProperties: { $ref: '#/components/schemas/Probability' } + NoulAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, noul] + properties: + type: { const: noul } + noul: + $ref: '#/components/schemas/Probability' + description: Probability that the statement is true. + Usage: + type: object + required: [input_tokens, output_tokens] + properties: + input_tokens: { type: integer, minimum: 0, description: Tokens read across all questions. } + output_tokens: { type: integer, const: 0, description: Laya generates no text. } + Routing: + type: object + description: Which checkpoint answered and why. + properties: + model: { type: string, examples: [english] } + repo: { type: string, examples: [convaiinnovations/laya] } + reason: { type: string } + detection: + type: [object, 'null'] + description: Language detection details when the checkpoint was chosen automatically. + workflow: { type: [object, 'null'] } + Error: + type: object + required: [detail] + properties: + detail: { type: string, description: Human-readable; not meant for parsing. } + ModelHealth: + type: object + required: [device, requested_device, device_mismatch] + properties: + device: { type: string, description: 'Where the model actually runs, e.g. mps, cpu, cuda:0.' } + requested_device: { type: string, description: LAYA_DEVICE, or `auto`. } + device_mismatch: { type: boolean } + weights_dtype: { type: [string, 'null'] } + autocast_dtype: { type: string } + mps_amp_min_rows: { type: [integer, 'null'] } + checkpoint: { type: [string, 'null'] } + revision: + type: [string, 'null'] + description: Commit the weights were downloaded from; null for a local checkpoint. + warmup_ms: { type: number } + Health: + allOf: + - $ref: '#/components/schemas/ModelHealth' + - type: object + required: [status, ready, loaded, models, compile] + description: Top-level model fields describe LAYA_WORKER_MODEL; `models` has every loaded model. + properties: + status: { const: ok } + ready: { const: true } + loaded: { type: array, items: { type: string } } + models: + type: object + additionalProperties: { $ref: '#/components/schemas/ModelHealth' } + compile: + type: object + required: [mode] + properties: + mode: { enum: ['off', 'on'] } + graphs_at_ready: { type: integer } + graphs_now: { type: integer } + recompiled_after_ready: { type: boolean } +x-known-gaps: + - Errors use FastAPI's `{"detail": string}`, not RFC 9457 problem details; the frontend's 502/504 are text/plain. Clients should branch on the status code only. + - No request id or trace header is accepted or returned. + - Overload has no 429/503 with Retry-After; requests queue behind the single inference thread. + - The response `model` field does not identify the checkpoint (always `laya-rl-agent`). + - The response carries no server-side inference time; only the client round trip can be measured. + - '`score` answers are returned by the server; system1-agents does not ask score questions yet.' diff --git a/docs/api/laya-systemone.openapi.yaml b/docs/api/laya-systemone.openapi.yaml new file mode 100644 index 0000000..cf4a65a --- /dev/null +++ b/docs/api/laya-systemone.openapi.yaml @@ -0,0 +1,508 @@ +openapi: 3.1.0 +info: + title: Laya decisions (System 1) — /v1/systemone + version: 1.0.0-draft + summary: Target interface for typed decisions from Laya over HTTP. + description: | + Design of the interface between decision clients (system1-agents) and Laya served by + system1-omni: the Laya worker (`src/models/laya/worker.py`), optionally behind the Rust + frontend (`omni-jev`). + + Every operation, header and field carries `x-status`: `implemented` where the server already + behaves this way, `planned` where it does not yet. `laya-systemone.current.openapi.yaml` + describes today's behaviour exactly; `../served-laya.md` says where each planned item is built. + + **Semantics.** A decision has no side effects, and the same request returns the same answers + for a given checkpoint, device and precision. Any request may be retried; no idempotency key is + needed. + + **Retries.** The client owns retries and their deadline. Servers do not retry. + + **Compatibility.** Within `/v1`, changes are additive: new optional request fields, new response + fields and headers. Clients ignore fields they do not know. A breaking change gets a new major + path (`/v2`) and the old one announces its end with `Deprecation` and `Sunset` headers. + license: + name: Apache-2.0 +servers: + - url: http://127.0.0.1:8000 + description: Laya worker directly + - url: http://127.0.0.1:8080 + description: Rust frontend in front of the worker +security: + - {} + - bearerAuth: [] +tags: + - name: decisions + - name: health +paths: + /v1/systemone: + post: + tags: [decisions] + operationId: decide + x-status: implemented + summary: Answer typed questions about one state + description: | + All questions are answered in one forward pass. Question ids are chosen by the client and + echoed as the keys of `answers`. + parameters: + - $ref: '#/components/parameters/RequestId' + - $ref: '#/components/parameters/Traceparent' + requestBody: + required: true + description: '`application/json`, at most 2 MiB.' + content: + application/json: + schema: { $ref: '#/components/schemas/DecisionRequest' } + examples: + combined: + value: + model: english + state: I was charged twice for my order. Please refund the duplicate today. + questions: + department: + type: choice + instructions: Which team should handle this? + criteria: { billing: Charges and refunds, technical: Software problems } + urgency: + type: score + instructions: How urgent is the request? + criteria: [Not urgent, Needs attention soon, Needs attention immediately] + refund: + type: noul + instructions: Does the customer ask for a refund? + responses: + '200': + description: One answer per question id. + headers: + X-Request-Id: { $ref: '#/components/headers/X-Request-Id' } + Server-Timing: { $ref: '#/components/headers/Server-Timing' } + content: + application/json: + schema: { $ref: '#/components/schemas/DecisionResponse' } + examples: + combined: + value: + model: english + served_by: + checkpoint: convaiinnovations/laya + revision: 55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851 + device: mps + weights_dtype: float16 + compile: 'on' + answers: + department: + type: choice + choice: billing + probabilities: { billing: 0.952, technical: 0.048 } + confidence: 0.7222 + answer_confidence: 0.952 + action: { act_probability: 1.0 } + urgency: + type: score + score: 1.7841 + legend: { '0': Not urgent, '1': Needs attention soon, '2': Needs attention immediately } + probabilities: { '0': 0.0241, '1': 0.1677, '2': 0.8082 } + confidence: 0.4891 + answer_confidence: 0.8082 + action: { act_probability: 1.0 } + refund: + type: noul + noul: 0.9161 + confidence: 0.9161 + answer_confidence: 0.9161 + action: { act_probability: 1.0 } + usage: { input_tokens: 137, output_tokens: 0 } + routing: { model: english, repo: convaiinnovations/laya, reason: explicit model='english', detection: null, workflow: null } + '400': { $ref: '#/components/responses/BadRequest' } + '401': { $ref: '#/components/responses/Unauthorized' } + '413': { $ref: '#/components/responses/TooLarge' } + '415': { $ref: '#/components/responses/UnsupportedMediaType' } + '422': { $ref: '#/components/responses/InvalidQuestion' } + '500': { $ref: '#/components/responses/InferenceFailed' } + '502': { $ref: '#/components/responses/BackendUnavailable' } + '503': { $ref: '#/components/responses/Overloaded' } + '504': { $ref: '#/components/responses/BackendTimeout' } + /livez: + get: + tags: [health] + operationId: livez + x-status: planned + summary: The process is up + description: 200 as soon as the server accepts connections, before loading. For supervisors deciding whether to restart the process. + security: [{}] + responses: + '200': + description: Alive. + content: + application/json: + schema: + type: object + required: [status] + properties: + status: { const: ok } + /readyz: + get: + tags: [health] + operationId: readyz + x-status: planned + summary: Ready to serve decisions + description: | + 200 once every model is loaded and warmed up; 503 with `Retry-After` before that. Clients and + load balancers send decisions only after 200. The body is the same as `/health`. + security: [{}] + responses: + '200': + description: Ready. + content: + application/json: + schema: { $ref: '#/components/schemas/Health' } + '503': + description: Starting, loading or warming up. + headers: + Retry-After: { $ref: '#/components/headers/Retry-After' } + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:not-ready', title: Not ready, status: 503, code: not_ready, detail: 'warming up: 2 of 5 shapes done' } + /health: + get: + tags: [health] + operationId: health + x-status: implemented + summary: What the server is running (kept as an alias of /readyz) + description: Today the worker listens only after warmup, so any response means ready. Planned — same body and status codes as `/readyz`. + security: [{}] + responses: + '200': + description: Ready. + content: + application/json: + schema: { $ref: '#/components/schemas/Health' } +components: + securitySchemes: + bearerAuth: + type: http + scheme: bearer + x-status: implemented + description: Required when the worker is started with `LAYA_API_KEY`; the frontend forwards the header. + parameters: + RequestId: + name: X-Request-Id + in: header + required: false + x-status: planned + description: Client-chosen id for this request (1–128 visible ASCII characters). Echoed in the response and in problem `instance`; the server generates one if absent. The frontend forwards it. + schema: { type: string, minLength: 1, maxLength: 128, pattern: '^[\x21-\x7E]+$' } + Traceparent: + name: traceparent + in: header + required: false + x-status: planned + description: W3C Trace Context. Propagated through the frontend to the worker. + schema: { type: string } + headers: + X-Request-Id: + x-status: planned + description: The request id, as sent by the client or generated by the server. + schema: { type: string } + Server-Timing: + x-status: planned + description: | + W3C Server-Timing with the server's own durations in ms: `queue` (waiting for the inference + thread), `infer` (tokenisation, forward pass, decoding). The frontend appends `proxy`. + Example: `queue;dur=0.2, infer;dur=27.4, proxy;dur=0.3`. + schema: { type: string } + Retry-After: + x-status: planned + description: Seconds to wait before retrying. + schema: { type: integer, minimum: 1 } + responses: + BadRequest: + description: The body is not a JSON object with a `questions` object. + headers: { X-Request-Id: { $ref: '#/components/headers/X-Request-Id' } } + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:malformed-request', title: Malformed request, status: 400, code: malformed_request, detail: request body must be valid JSON, instance: 'urn:request:7f3a' } + Unauthorized: + description: Missing or wrong bearer token. + headers: + WWW-Authenticate: { description: '`Bearer`', schema: { type: string } } + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:unauthorized', title: Unauthorized, status: 401, code: unauthorized } + TooLarge: + description: Body over 2 MiB, more than 64 questions, or a state over 50,000 characters. + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:too-large', title: Request too large, status: 413, code: too_many_questions, detail: too many questions (65 > 64), limit: 64 } + UnsupportedMediaType: + description: The body is not `application/json`. + x-status: planned + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + InvalidQuestion: + description: A question is invalid. `errors` lists every invalid question, not only the first. + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: + type: 'urn:laya:problem:invalid-question' + title: Invalid question + status: 422 + code: invalid_question + errors: + - { question: q, field: type, detail: "unknown type 'bogus'; use one of choice, noul, score" } + InferenceFailed: + description: The forward pass failed. Retrying the same request fails the same way. + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:inference-failed', title: Inference failed, status: 500, code: inference_failed } + BackendUnavailable: + description: Frontend only — the worker could not be reached. Safe to retry. + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:backend-unavailable', title: Backend unavailable, status: 502, code: backend_unavailable } + Overloaded: + description: | + The worker's queue is full (it runs one forward pass at a time), or it is not ready yet. + Retry after `Retry-After` seconds. + x-status: planned + headers: + Retry-After: { $ref: '#/components/headers/Retry-After' } + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:overloaded', title: Overloaded, status: 503, code: overloaded, queue_depth: 32 } + BackendTimeout: + description: Frontend only — the worker did not answer within the frontend's timeout (60 s). Safe to retry. + content: + application/problem+json: + schema: { $ref: '#/components/schemas/Problem' } + example: { type: 'urn:laya:problem:backend-timeout', title: Backend timeout, status: 504, code: backend_timeout } + schemas: + Problem: + description: RFC 9457 problem details. Clients branch on `code`; `detail` is for people. + x-status: planned + type: object + required: [type, title, status, code] + properties: + type: { type: string, format: uri-reference } + title: { type: string } + status: { type: integer, minimum: 400, maximum: 599 } + code: + type: string + description: Stable machine-readable reason. + enum: [malformed_request, unauthorized, too_large, too_many_questions, state_too_large, + unsupported_media_type, invalid_question, inference_failed, backend_unavailable, + backend_timeout, overloaded, not_ready] + detail: { type: string } + instance: { type: string, description: The request id. } + errors: + type: array + description: For `invalid_question`, one entry per invalid question. + items: + type: object + required: [question, detail] + properties: + question: { type: string } + field: { type: string } + detail: { type: string } + additionalProperties: true + DecisionRequest: + type: object + required: [questions] + description: Unknown fields are ignored, so fields added later in /v1 stay compatible with older servers. + properties: + model: + type: string + description: Checkpoint name (`english`, `multilingual`, `typed-decisions`) or alias. Omitted, the server picks one from the state's language. + state: + description: What the questions are about. At most 50,000 characters as text or serialised JSON. + oneOf: [{ type: string }, { type: object }, { type: array }] + questions: + type: object + description: Question id → question; 1 to 64. + minProperties: 1 + maxProperties: 64 + propertyNames: { minLength: 1, maxLength: 128 } + additionalProperties: { $ref: '#/components/schemas/Question' } + Question: + oneOf: + - $ref: '#/components/schemas/ChoiceQuestion' + - $ref: '#/components/schemas/ScoreQuestion' + - $ref: '#/components/schemas/NoulQuestion' + discriminator: + propertyName: type + mapping: + choice: '#/components/schemas/ChoiceQuestion' + score: '#/components/schemas/ScoreQuestion' + noul: '#/components/schemas/NoulQuestion' + Instructions: + oneOf: [{ type: string }, { type: object }] + ChoiceQuestion: + type: object + required: [type, instructions, criteria] + properties: + type: { const: choice } + instructions: { $ref: '#/components/schemas/Instructions' } + criteria: + oneOf: + - { type: object, minProperties: 1 } + - { type: array, minItems: 1, items: { type: string } } + ScoreQuestion: + type: object + required: [type, instructions, criteria] + properties: + type: { const: score } + instructions: { $ref: '#/components/schemas/Instructions' } + criteria: { type: array, minItems: 1, description: Level descriptions, level 0 first. } + NoulQuestion: + type: object + required: [type, instructions] + properties: + type: { const: noul } + instructions: { $ref: '#/components/schemas/Instructions' } + criteria: + type: object + description: Optional texts for the outcomes, keyed `true` and/or `false`. + propertyNames: { enum: ['true', 'false'] } + labels: + type: object + required: ['false', 'true'] + additionalProperties: false + properties: + 'false': { type: string, minLength: 1 } + 'true': { type: string, minLength: 1 } + DecisionResponse: + type: object + required: [model, served_by, answers, usage] + properties: + model: + type: string + description: The checkpoint that answered (`english`, …). + x-status: planned + served_by: { $ref: '#/components/schemas/ServedBy' } + answers: + type: object + additionalProperties: { $ref: '#/components/schemas/Answer' } + usage: { $ref: '#/components/schemas/Usage' } + routing: { $ref: '#/components/schemas/Routing' } + ServedBy: + type: object + x-status: planned + description: What produced these answers, per response, so a record needs no separate /health call. + required: [checkpoint, revision, device, weights_dtype] + properties: + checkpoint: { type: string } + revision: { type: [string, 'null'], description: Null for a local checkpoint. } + device: { type: string } + weights_dtype: { enum: [float32, float16, bfloat16] } + compile: { enum: ['off', 'on'] } + Answer: + oneOf: + - $ref: '#/components/schemas/ChoiceAnswer' + - $ref: '#/components/schemas/ScoreAnswer' + - $ref: '#/components/schemas/NoulAnswer' + discriminator: + propertyName: type + mapping: + choice: '#/components/schemas/ChoiceAnswer' + score: '#/components/schemas/ScoreAnswer' + noul: '#/components/schemas/NoulAnswer' + Probability: { type: number, minimum: 0, maximum: 1 } + AnswerCommon: + type: object + required: [type, confidence, answer_confidence, action] + properties: + confidence: { $ref: '#/components/schemas/Probability' } + answer_confidence: { $ref: '#/components/schemas/Probability' } + action: + type: object + required: [act_probability] + properties: + act_probability: { $ref: '#/components/schemas/Probability' } + ChoiceAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, choice, probabilities] + properties: + type: { const: choice } + choice: { type: string } + probabilities: { type: object, additionalProperties: { $ref: '#/components/schemas/Probability' } } + ScoreAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, score, legend, probabilities] + properties: + type: { const: score } + score: { type: number, minimum: 0 } + legend: { type: object } + probabilities: { type: object, additionalProperties: { $ref: '#/components/schemas/Probability' } } + NoulAnswer: + allOf: + - $ref: '#/components/schemas/AnswerCommon' + - type: object + required: [type, noul] + properties: + type: { const: noul } + noul: { $ref: '#/components/schemas/Probability' } + Usage: + type: object + required: [input_tokens, output_tokens] + properties: + input_tokens: { type: integer, minimum: 0 } + output_tokens: { type: integer, const: 0 } + Routing: + type: object + properties: + model: { type: string } + repo: { type: string } + reason: { type: string } + detection: { type: [object, 'null'] } + workflow: { type: [object, 'null'] } + ModelHealth: + type: object + required: [device, requested_device, device_mismatch] + properties: + device: { type: string } + requested_device: { type: string } + device_mismatch: { type: boolean } + weights_dtype: { type: [string, 'null'] } + autocast_dtype: { type: string } + checkpoint: { type: [string, 'null'] } + revision: { type: [string, 'null'] } + warmup_ms: { type: number } + Health: + allOf: + - $ref: '#/components/schemas/ModelHealth' + - type: object + required: [status, ready, loaded, models, compile] + properties: + status: { const: ok } + ready: { const: true } + loaded: { type: array, items: { type: string } } + models: { type: object, additionalProperties: { $ref: '#/components/schemas/ModelHealth' } } + compile: + type: object + required: [mode] + properties: + mode: { enum: ['off', 'on'] } + graphs_at_ready: { type: integer } + graphs_now: { type: integer } + recompiled_after_ready: { type: boolean } + limits: + x-status: planned + type: object + description: Request limits, so clients can check before sending. + properties: + max_questions: { const: 64 } + max_state_chars: { const: 50000 } + max_body_bytes: { const: 2097152 } diff --git a/docs/served-laya.md b/docs/served-laya.md new file mode 100644 index 0000000..aff156d --- /dev/null +++ b/docs/served-laya.md @@ -0,0 +1,112 @@ +# Served Laya: a decision model over HTTP + +## 1. Requirements + +Functional (from #20): +- An agent can use a Laya model served by system1-omni instead of loading it in process, from the CLI + and from MCP, with no cloud key; the in-process `--model laya` stays. +- `choice` and `noul`, one or several questions per request, through the existing answer validation. +- Runs record which model, checkpoint and serving backend answered. + +Non-functional: +- Latency: a warm decision on MPS takes 25–160 ms server-side; the client adds little on localhost. +- Failure: a slow or absent server fails one decision within a bounded deadline, with a clear error. +- Contract: one written interface both repositories test against (OpenAPI 3.1, [api/laya-systemone.openapi.yaml](api/laya-systemone.openapi.yaml)). + +Constraints: today's server is laya-serve 0.3.20 behind the system1-omni worker and optionally the Rust +frontend. The design sets the target interface; section 4 says where each part that is not there yet +gets built, and how the client works with both in the meantime. + +## 2. High level + +``` +agent step ──► ServedLayaModel (s1a) ──HTTP──► [omni-jev frontend :8080] ──► Laya worker :8000 ──► laya (MPS/CPU) + │ laya_question() forwards unchanged warmup before listen + │ answer validation 502/504 if worker down /health: device, revision, compile + └─ run record: /health snapshot + client round trip +``` + +## 3. Interface (target; full spec in [api/laya-systemone.openapi.yaml](api/laya-systemone.openapi.yaml)) + +| | | +|---|---| +| `POST /v1/systemone` | `{model?, state, questions}` → `{model, served_by, answers, usage, routing}` | +| `GET /livez`, `/readyz` | process up; ready to serve (503 + `Retry-After` until warm). `/health` stays as an alias | +| Errors | RFC 9457 `application/problem+json` with a stable `code`; 422 lists every invalid question | +| Tracing | `X-Request-Id` (or W3C `traceparent`) in, echoed out, used as problem `instance` | +| Timing | `Server-Timing: queue, infer` from the worker, `proxy` added by the frontend | +| Overload | bounded queue; 503 + `Retry-After` when full | +| Auth | optional bearer token; 401 with `WWW-Authenticate: Bearer` | +| Limits | 1–64 questions, state ≤ 50,000 chars, body ≤ 2 MiB; advertised in `/readyz` | +| Evolution | additive within `/v1`; unknown request fields ignored; breaking change → `/v2`, `Deprecation`/`Sunset` | +| Semantics | deterministic and side-effect free: any request may be retried, no idempotency key | + +## 4. From today to the target + +Today's behaviour is specified in [api/laya-systemone.current.openapi.yaml](api/laya-systemone.current.openapi.yaml), checked against +traffic captured from a running worker. Validating that traffic against the target spec: every error response, the missing +`served_by`, and the absent `X-Request-Id` / `Server-Timing` headers are the gap; `/health` already +matches. Most of it sits in the worker, which wraps laya-serve's app, so laya itself needs no change. + +| item | today | built in | +|---|---|---| +| problem+json errors with `code` | `{"detail"}` (worker), text/plain (frontend 502/504) | worker: exception handler over laya-serve's app; frontend: its two error bodies | +| 422 lists every invalid question | first invalid question only | worker: validate all questions before calling laya | +| `X-Request-Id`, `traceparent` | not handled | worker middleware; frontend forwards the headers (it already forwards the rest) | +| `Server-Timing` | none | worker middleware around the inference call; frontend appends `proxy` | +| `served_by`, `model` = checkpoint | `model` is always `laya-rl-agent` | worker: add from the loaded agent (it already reports these in `/health`) | +| `/livez`, `/readyz` | worker binds after warmup, `/health` only | worker: bind first, gate `/readyz` on warmup | +| 503 + `Retry-After` on overload | requests queue without bound | worker: bounded queue in front of laya-serve's single inference thread | +| 415 on non-JSON bodies | parsed regardless of Content-Type | worker middleware | +| limits in `/readyz` | not advertised | worker | + +Client compatibility during the change: branch on the status code, read `code` when the body is +problem+json and fall back to `detail`; read identity from `served_by` when present, else from the +`/health` snapshot taken at warm-up. + +## 5. Client design (system1-agents) + +- **Selection.** `--model laya-served`, a new name so run records say served Laya, not Jev or in-process Laya. +- **Configuration.** `LAYA_SERVED_URL` (required), `LAYA_SERVED_MODEL` (default `english`), + `LAYA_SERVED_API_KEY` (optional), `LAYA_SERVED_TIMEOUT_S` (default 5, one deadline per decision, + retries included). +- **Request.** Questions serialised with the existing `laya_question()`, which keeps Laya's own `noul` + shape (a plain-string instruction). `score` is not sent until an agent needs it. +- **Identity.** Each response's `served_by` goes into the run record; until servers send it, `warm()` + reads `/health` once and records checkpoint, revision, device, dtypes and compile mode. +- **Errors → agent errors.** + + | outcome | handling | + |---|---| + | connection refused / reset | retry once within the deadline (worker may be starting or restarting) | + | 502, 504 | retry once within the deadline | + | 503 (overloaded or not ready) | wait `Retry-After` if it fits the deadline, then retry once | + | 400, 413, 422 | fail at once: the request is wrong, a retry returns the same | + | 401 | fail at once as a configuration error | + | 500 | fail at once: the same request fails the same way | + | deadline passed | fail with a timeout error naming the URL | + +- **Timing.** The record keeps the client round trip per decision and, when present, `Server-Timing`'s + `queue` and `infer`, so network, queueing and model time separate. +- **Tracing.** The client sends an `X-Request-Id` per decision and stores it with the step. + +## 6. Trade-offs + +| decision | chosen | alternative | why | +|---|---|---|---| +| spec | hand-written OpenAPI 3.1 target, plus an as-implemented spec checked against captured traffic | generate from FastAPI | laya-serve reads the raw body, so FastAPI's generated schema has no request or response shape | +| errors | RFC 9457 with a stable `code` | keep `{"detail"}` | clients need a machine-readable reason; `detail` wording changes between laya versions | +| retries | client owns them; servers never retry | retries in the frontend | the client knows its step deadline; a retrying proxy multiplies load when the worker is saturated | +| identity | per-response `served_by` | only `/health` | a record then stays correct across worker restarts and multi-model routing | +| overload | 503 + `Retry-After` from a bounded queue | 429 | the limit is server capacity, not a per-client quota | +| readiness | `/livez` + `/readyz` | bind only after warmup (today) | a supervisor can tell a slow start from a dead process | +| timing | `Server-Timing` header | a field in the body | standard, visible in tooling, keeps the Jev-compatible body unchanged | +| model name | `laya-served` | reuse `laya` with a URL switch | in-process and served runs stay distinguishable in records and evaluations | + +## 7. Revisit when + +- Several agents share one worker: per-client quotas (429) on top of the capacity limit. +- Throughput matters more than single-request latency: batching concurrent requests in the worker. +- A second model family (ThinkFlowLab/system1-omni#9) reuses `/v1/systemone`: move `served_by` and the problem codes into a + shared contract instead of the Laya spec. +- `score` becomes useful to an agent: extend the client; the server already answers it. From 61d6de95efbd79f633067e4dfeb239f2497747a0 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:36:15 +0800 Subject: [PATCH 04/14] [Feat] Add the laya-served decision model (#20) ServedLayaModel asks a Laya served over HTTP (system1-omni's worker, its omni-jev frontend, or plain laya-serve) through POST /v1/systemone, with the body the in-process model builds (laya_question). The client holds one deadline per decision, retries once on a dropped connection, 502 or 504, and after Retry-After on 503, sends one X-Request-Id per decision, and maps every other status to MODEL_CALL_FAILED or MODEL_SERVICE_CONFIG_ERROR. Decision.model is the served checkpoint and revision, read from the response's served_by when a server sends it, else from /health (refreshed after 30 s, since the worker reports a CPU fallback there), else from routing.repo against plain laya-serve. The window check is shared with LayaModel (check_window). Tests run over httpx.MockTransport: the shared contract, the error and retry mapping, identity, and responses recorded from a real worker, the frontend and laya-serve, which are checked against the as-implemented OpenAPI spec (jsonschema, pyyaml and referencing join the dev extra). --- pyproject.toml | 2 +- s1a/decision_models/laya.py | 37 +- s1a/decision_models/served.py | 354 ++++++++++++ tests/data/served_laya/README.md | 17 + tests/data/served_laya/capture.py | 116 ++++ .../frontend-down.unreachable.json | 15 + tests/data/served_laya/frontend.choice.json | 47 ++ tests/data/served_laya/frontend.combined.json | 88 +++ tests/data/served_laya/frontend.health.json | 39 ++ .../frontend.invalid_question.json | 17 + .../served_laya/frontend.malformed_json.json | 7 + .../served_laya/frontend.no_questions.json | 11 + tests/data/served_laya/frontend.noul.json | 39 ++ .../frontend.too_many_questions.json | 7 + .../served_laya/frontend.unauthorized.json | 17 + tests/data/served_laya/laya-serve.choice.json | 47 ++ .../data/served_laya/laya-serve.combined.json | 88 +++ tests/data/served_laya/laya-serve.health.json | 11 + .../laya-serve.invalid_question.json | 17 + .../laya-serve.malformed_json.json | 7 + .../served_laya/laya-serve.no_questions.json | 11 + tests/data/served_laya/laya-serve.noul.json | 39 ++ .../laya-serve.too_many_questions.json | 7 + tests/data/served_laya/worker.choice.json | 47 ++ tests/data/served_laya/worker.combined.json | 88 +++ tests/data/served_laya/worker.health.json | 39 ++ .../served_laya/worker.invalid_question.json | 17 + .../served_laya/worker.malformed_json.json | 7 + .../data/served_laya/worker.no_questions.json | 11 + tests/data/served_laya/worker.noul.json | 39 ++ .../worker.too_many_questions.json | 7 + .../data/served_laya/worker.unauthorized.json | 17 + tests/test_decision_models_served.py | 510 ++++++++++++++++++ tests/test_served_laya_fixtures.py | 58 ++ uv.lock | 6 + 35 files changed, 1871 insertions(+), 15 deletions(-) create mode 100644 s1a/decision_models/served.py create mode 100644 tests/data/served_laya/README.md create mode 100644 tests/data/served_laya/capture.py create mode 100644 tests/data/served_laya/frontend-down.unreachable.json create mode 100644 tests/data/served_laya/frontend.choice.json create mode 100644 tests/data/served_laya/frontend.combined.json create mode 100644 tests/data/served_laya/frontend.health.json create mode 100644 tests/data/served_laya/frontend.invalid_question.json create mode 100644 tests/data/served_laya/frontend.malformed_json.json create mode 100644 tests/data/served_laya/frontend.no_questions.json create mode 100644 tests/data/served_laya/frontend.noul.json create mode 100644 tests/data/served_laya/frontend.too_many_questions.json create mode 100644 tests/data/served_laya/frontend.unauthorized.json create mode 100644 tests/data/served_laya/laya-serve.choice.json create mode 100644 tests/data/served_laya/laya-serve.combined.json create mode 100644 tests/data/served_laya/laya-serve.health.json create mode 100644 tests/data/served_laya/laya-serve.invalid_question.json create mode 100644 tests/data/served_laya/laya-serve.malformed_json.json create mode 100644 tests/data/served_laya/laya-serve.no_questions.json create mode 100644 tests/data/served_laya/laya-serve.noul.json create mode 100644 tests/data/served_laya/laya-serve.too_many_questions.json create mode 100644 tests/data/served_laya/worker.choice.json create mode 100644 tests/data/served_laya/worker.combined.json create mode 100644 tests/data/served_laya/worker.health.json create mode 100644 tests/data/served_laya/worker.invalid_question.json create mode 100644 tests/data/served_laya/worker.malformed_json.json create mode 100644 tests/data/served_laya/worker.no_questions.json create mode 100644 tests/data/served_laya/worker.noul.json create mode 100644 tests/data/served_laya/worker.too_many_questions.json create mode 100644 tests/data/served_laya/worker.unauthorized.json create mode 100644 tests/test_decision_models_served.py create mode 100644 tests/test_served_laya_fixtures.py diff --git a/pyproject.toml b/pyproject.toml index 083c2b3..3dd00f9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -45,7 +45,7 @@ cua = [ # Cua-S1 Nano behind --model cua; pinned to the Cua PR that ships the c "cua-s1 @ git+https://github.com/trycua/cua.git@aea61b6eb97e2d8c0f6f71eb804e5769fe910af4#subdirectory=libs/cua-s1/python", "huggingface-hub>=0.24", ] -dev = ["pytest>=8", "pytest-asyncio>=0.24", "ruff>=0.6", "ty>=0.0.83"] +dev = ["pytest>=8", "pytest-asyncio>=0.24", "ruff>=0.6", "ty>=0.0.83", "jsonschema>=4.18", "pyyaml>=6", "referencing>=0.30"] # the last three check tests/data/served_laya against docs/api [build-system] requires = ["hatchling"] diff --git a/s1a/decision_models/laya.py b/s1a/decision_models/laya.py index 1fb786d..34b637c 100644 --- a/s1a/decision_models/laya.py +++ b/s1a/decision_models/laya.py @@ -32,6 +32,23 @@ def laya_question(question: Question) -> Json: return {"type": "noul", "instructions": question.question, **criteria} +def check_window(usage: Usage, questions: int, max_len: int, hint: str) -> None: + """Laya cuts each option to 48 tokens, shrinks every option when the head overflows, and cuts the state to + what is left, all silently. A filled window raises: a decision over a cut state is a guess. + ``input_tokens`` is the attention-mask sum over every question's row and a row is at most ``max_len`` long, + so the sum reaches ``questions * max_len`` only when every row hit the window. Exact for one question; with + several, a cut on the widest head alone goes unseen. Shared by the in-process and the served Laya.""" + window = max_len * questions + if usage.input_tokens >= window: + raise build_error( + StatusCode.MODEL_SERVICE_CONFIG_ERROR, + error_msg=( + f"the laya {window}-token window filled ({usage.input_tokens} tokens over {questions} question(s)): " + f"the state or the options were cut; {hint}" + ), + ) + + class LayaModel(DecisionModel): """Laya's ``Agent`` (or anything with ``system_one(state, questions)`` and a ``cfg``) behind the interface.""" @@ -68,20 +85,12 @@ async def _decide(self, observation: Observation, questions: dict[str, Question] ) def _check_the_window(self, usage: Usage, questions: int) -> None: - """Laya cuts each option to 48 tokens, shrinks every option when the head overflows, and cuts the state to - what is left, all silently. A filled window raises: a decision over a cut state is a guess. - ``input_tokens`` is the attention-mask sum over every question's row and a row is at most ``max_len`` long, - so the sum reaches ``questions * max_len`` only when every row hit the window. Exact for one question; with - several, a cut on the widest head alone goes unseen.""" - window = int(self._agent.cfg.get("max_len", LAYA_DEFAULT_MAX_LEN)) * questions - if usage.input_tokens >= window: - raise build_error( - StatusCode.MODEL_SERVICE_CONFIG_ERROR, - error_msg=( - f"the laya {window}-token window filled ({usage.input_tokens} tokens over {questions} question(s)): " - "the state or the options were cut; raise LAYA_MAX_LEN / LAYA_HEAD_MAX_LEN or shorten the state" - ), - ) + check_window( + usage, + questions, + int(self._agent.cfg.get("max_len", LAYA_DEFAULT_MAX_LEN)), + "raise LAYA_MAX_LEN / LAYA_HEAD_MAX_LEN or shorten the state", + ) # ponytail: warm() with one tiny forward pass so CUDA kernels compile before the first real turn diff --git a/s1a/decision_models/served.py b/s1a/decision_models/served.py new file mode 100644 index 0000000..09b4803 --- /dev/null +++ b/s1a/decision_models/served.py @@ -0,0 +1,354 @@ +# coding: utf-8 +"""Laya served over HTTP (system1-omni's worker, or plain laya-serve), behind the decision-model interface. + +The interface is specified in ``docs/api/laya-systemone.openapi.yaml`` (target) and +``docs/api/laya-systemone.current.openapi.yaml`` (today's servers); ``docs/served-laya.md`` has the design. +``ServedLayaClient`` owns the connection, the deadline, the retries and the error mapping and reads no answer; +``ServedLayaModel`` builds the body the in-process Laya would read and records who answered. +""" + +from __future__ import annotations + +import asyncio +import logging +import os +import time +import uuid +from collections.abc import Awaitable, Callable +from datetime import datetime, timezone +from typing import Any + +import httpx + +from openjiuwen.core.common.exception.codes import StatusCode +from openjiuwen.core.common.exception.errors import build_error + +from s1a.decision_models.base import DecisionModel +from s1a.decision_models.laya import LAYA_DEFAULT_MAX_LEN, check_window, laya_question +from s1a.decision_models.types import Json, Observation, Question, Reply, Usage + +logger = logging.getLogger(__name__) + +SERVED_TIMEOUT_S = 5.0 # one decision, retries included; a warm Laya answers in 25-160 ms on MPS +HEALTH_TIMEOUT_S = 2.0 +HEALTH_MAX_AGE_S = 30.0 # the worker reports the live device; laya moves a model to the CPU on a GPU OOM +DEFAULT_SERVED_MODEL = "english" +DEFAULT_RETRY_AFTER_S = 0.5 +_RETRIED_TRANSPORT = (httpx.ConnectError, httpx.ConnectTimeout, httpx.RemoteProtocolError, httpx.ReadError) +_RETRIED_STATUSES = frozenset({502, 504}) # the frontend could not reach the worker, or it was too slow +_REQUEST_ERRORS = frozenset({400, 413, 422}) +_NOT_UP = ( + "the system1-omni worker listens only after loading and warming up (35-39 s with LAYA_WORKER_COMPILE=on " + "and fp16 weights on an M1 Pro); start it per system1-omni's recipe/laya/apple-silicon.md" +) + + +def _reason(response: httpx.Response) -> str: + """The server's reason: ``code`` and ``detail`` from problem+json, ``detail`` from JSON, else the body's start.""" + try: + body = response.json() + except ValueError: + return response.text[:200].strip() + if isinstance(body, dict): + code, detail = body.get("code"), body.get("detail") + if code and detail: + return f"{code}: {detail}" + if code or detail: + return str(code or detail) + return response.text[:200].strip() + + +def _retry_after_s(response: httpx.Response) -> float: + try: + return max(0.0, float(response.headers.get("retry-after", DEFAULT_RETRY_AFTER_S))) + except ValueError: + return DEFAULT_RETRY_AFTER_S + + +def parse_server_timing(header: str | None) -> dict[str, float]: + """``queue;dur=0.2, infer;dur=27.4`` -> ``{"queue": 0.2, "infer": 27.4}``; entries without a duration are dropped.""" + timings: dict[str, float] = {} + for entry in (header or "").split(","): + name, *params = [part.strip() for part in entry.split(";")] + for param in params: + key, _, value = param.partition("=") + if name and key.strip().lower() == "dur": + try: + timings[name] = float(value.strip().strip('"')) + except ValueError: + pass + return timings + + +class ServedLayaClient: + """HTTP to one ``/v1/systemone`` server: one deadline per decision, one retry, the error mapping.""" + + def __init__( + self, + *, + url: str, + api_key: str | None = None, + timeout_s: float = SERVED_TIMEOUT_S, + transport: httpx.AsyncBaseTransport | None = None, + clock: Callable[[], float] = time.monotonic, + sleep: Callable[[float], Awaitable[None]] = asyncio.sleep, + ) -> None: + self.url = url.rstrip("/") + self._timeout_s = timeout_s + self._clock = clock + self._sleep = sleep + headers = {"Authorization": f"Bearer {api_key}"} if api_key else {} + self._client = httpx.AsyncClient(timeout=timeout_s, headers=headers, transport=transport) + + async def health(self) -> Json: + """``GET /health`` once. Refused or unreachable raises; any other failure returns ``{}`` with a warning.""" + try: + response = await self._client.get(f"{self.url}/health", timeout=HEALTH_TIMEOUT_S) + except (httpx.ConnectError, httpx.ConnectTimeout) as exc: + raise build_error( + StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"no served Laya at {self.url}: {_NOT_UP}" + ) from exc + except httpx.HTTPError as exc: + logger.warning("[laya-served] /health at %s failed: %s", self.url, exc) + return {} + if response.status_code != 200: + logger.warning("[laya-served] /health at %s returned HTTP %s", self.url, response.status_code) + return {} + try: + body = response.json() + except ValueError: + logger.warning("[laya-served] /health at %s returned a body that is not JSON", self.url) + return {} + return body if isinstance(body, dict) else {} + + async def decide(self, body: Json, request_id: str) -> tuple[Json, dict[str, str], int]: + """One decision: the payload, the response headers and the last attempt's round trip in ms. Every attempt + sends the same ``X-Request-Id``, so the server's logs tie a retry to its first try.""" + deadline = self._clock() + self._timeout_s + retried = False + while True: + remaining = deadline - self._clock() + if remaining <= 0: + raise build_error( + StatusCode.MODEL_CALL_FAILED, + error_msg=f"no answer from served Laya at {self.url} within {self._timeout_s:g} s", + ) + started = time.perf_counter() + try: + response = await self._client.post( + f"{self.url}/v1/systemone", json=body, headers={"X-Request-Id": request_id}, timeout=remaining + ) + except httpx.TimeoutException as exc: + if isinstance(exc, httpx.ConnectTimeout) and not retried: + retried = True + continue + raise build_error( + StatusCode.MODEL_CALL_FAILED, + cause=exc, + error_msg=f"no answer from served Laya at {self.url} within {self._timeout_s:g} s", + ) from exc + except _RETRIED_TRANSPORT as exc: + if not retried: + retried = True + continue + raise build_error( + StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"served Laya unreachable at {self.url}: {exc}" + ) from exc + except httpx.HTTPError as exc: + raise build_error( + StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"served Laya call failed at {self.url}: {exc}" + ) from exc + ms = round((time.perf_counter() - started) * 1000) + status = response.status_code + if status in _RETRIED_STATUSES and not retried: + retried = True + continue + if status == 503 and not retried: + wait = _retry_after_s(response) + if self._clock() + wait < deadline: + retried = True + await self._sleep(wait) + continue + if status == 401: + raise build_error( + StatusCode.MODEL_SERVICE_CONFIG_ERROR, + error_msg=f"served Laya at {self.url} refused the token: set LAYA_SERVED_API_KEY to the worker's LAYA_API_KEY", + ) + if status in _REQUEST_ERRORS: + raise build_error( + StatusCode.MODEL_CALL_FAILED, + error_msg=f"served Laya rejected the request (HTTP {status}): {_reason(response)}", + ) + if status == 503: + raise build_error( + StatusCode.MODEL_CALL_FAILED, + error_msg=f"served Laya at {self.url} is overloaded: {_reason(response)}", + ) + if status in _RETRIED_STATUSES: + raise build_error( + StatusCode.MODEL_CALL_FAILED, + error_msg=f"served Laya unreachable behind {self.url} (HTTP {status}): {_reason(response)}", + ) + if response.is_error: + raise build_error( + StatusCode.MODEL_CALL_FAILED, + error_msg=f"served Laya failed (HTTP {status}): {_reason(response)}", + ) + try: + payload = response.json() + except ValueError as exc: + raise build_error( + StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg="served Laya returned a malformed body" + ) from exc + if not isinstance(payload, dict) or not isinstance(payload.get("answers"), dict): + raise build_error(StatusCode.MODEL_CALL_FAILED, error_msg="served Laya returned no answers object") + return payload, dict(response.headers), ms + + async def close(self) -> None: + await self._client.aclose() + + +def served_by_from_health(health: Json, routing: Any, read_at: str | None) -> Json: + """Who answered, from a ``/health`` reading: the answering model's entry on the system1-omni worker, its + top-level fields otherwise; plain laya-serve names no checkpoint, so only the response's ``routing.repo``.""" + routing = routing if isinstance(routing, dict) else {} + raw_models = health.get("models") + models: dict[str, Any] = raw_models if isinstance(raw_models, dict) else {} + entry = models.get(routing.get("model")) if routing.get("model") in models else None + if entry is None and "checkpoint" in health: + entry = health + raw_compile = health.get("compile") + compile_state: dict[str, Any] = raw_compile if isinstance(raw_compile, dict) else {} + if not isinstance(entry, dict): # plain laya-serve: its /health device is the configured one, not a fact + return {"checkpoint": routing.get("repo"), "revision": None, "device": None, "source": "routing"} + return { + "checkpoint": entry.get("checkpoint") or routing.get("repo"), + "revision": entry.get("revision"), + "device": entry.get("device"), + "weights_dtype": entry.get("weights_dtype"), + "autocast_dtype": entry.get("autocast_dtype"), + "compile": compile_state.get("mode"), + "source": "health", + "read_at": read_at, + } + + +def identity(served_by: Json, fallback: str) -> str: + """``checkpoint@revision`` (12 characters of it), the checkpoint alone, or the configured model name.""" + checkpoint, revision = served_by.get("checkpoint"), served_by.get("revision") + if checkpoint and revision: + return f"{checkpoint}@{str(revision)[:12]}" + return str(checkpoint or fallback) + + +class ServedLayaModel(DecisionModel): + """Laya behind ``/v1/systemone``: the in-process Laya's questions over HTTP, with who answered in every record.""" + + name = "laya-served" + deterministic = True + + def __init__( + self, + client: ServedLayaClient, + *, + model: str = DEFAULT_SERVED_MODEL, + max_len: int = LAYA_DEFAULT_MAX_LEN, + clock: Callable[[], float] = time.monotonic, + ) -> None: + self._client = client + self._model = model + self._max_len = max_len + self._clock = clock + self._health: Json = {} + self._health_read_at: str | None = None + self._health_tried: float | None = None + + @property + def model(self) -> str: + return self._model + + async def _read_health(self, *, strict: bool) -> None: + self._health_tried = self._clock() + try: + health = await self._client.health() + except Exception: + if strict: + raise + logger.warning("[laya-served] kept the previous /health reading from %s", self._health_read_at) + return + if health: + self._health = health + self._health_read_at = datetime.now(timezone.utc).isoformat(timespec="seconds") + + async def warm(self) -> None: + """Read ``/health`` once: fails early when no server is up, and gives the first identity reading.""" + await self._read_health(strict=True) + + async def _decide(self, observation: Observation, questions: dict[str, Question]) -> Reply: + if self._health_tried is None or self._clock() - self._health_tried >= HEALTH_MAX_AGE_S: + await self._read_health(strict=False) + body = { + "model": self._model, + "state": observation.state, + "questions": {name: laya_question(question) for name, question in questions.items()}, + } + request_id = uuid.uuid4().hex + payload, headers, ms = await self._client.decide(body, request_id) + usage = Usage.from_payload(payload.get("usage")) + check_window( + usage, + len(questions), + self._max_len, + "shorten the state, or raise the worker's LAYA_MAX_LEN and LAYA_SERVED_MAX_LEN together", + ) + sent = payload.get("served_by") + if isinstance(sent, dict): + served_by = {**sent, "source": "response"} + else: + served_by = served_by_from_health(self._health, payload.get("routing"), self._health_read_at) + answers = payload["answers"] + lower = {key.lower(): value for key, value in headers.items()} + return Reply( + answers={name: answers[name] for name in questions if name in answers}, + latency_ms=ms, + usage=usage, + model=identity(served_by, self._model), + raw={ + **payload, + "served_by": served_by, + "url": self._client.url, + "request_id": request_id, + "server_timing": parse_server_timing(lower.get("server-timing")), + }, + ) + + async def close(self) -> None: + await self._client.close() + + @classmethod + def from_env(cls) -> "ServedLayaModel": + """``LAYA_SERVED_URL`` (required), ``LAYA_SERVED_MODEL``, ``LAYA_SERVED_API_KEY``, ``LAYA_SERVED_TIMEOUT_S``, + ``LAYA_SERVED_MAX_LEN``. No cloud key is read.""" + url = os.getenv("LAYA_SERVED_URL") + if not url: + raise build_error( + StatusCode.MODEL_SERVICE_CONFIG_ERROR, + error_msg="--model laya-served needs LAYA_SERVED_URL, e.g. http://127.0.0.1:8000 (the worker) " + "or http://127.0.0.1:8080 (the frontend)", + ) + try: + timeout_s = float(os.getenv("LAYA_SERVED_TIMEOUT_S") or SERVED_TIMEOUT_S) + max_len = int(os.getenv("LAYA_SERVED_MAX_LEN") or LAYA_DEFAULT_MAX_LEN) + except ValueError as exc: + raise build_error( + StatusCode.MODEL_SERVICE_CONFIG_ERROR, + cause=exc, + error_msg="LAYA_SERVED_TIMEOUT_S must be a number and LAYA_SERVED_MAX_LEN an integer", + ) from exc + if timeout_s <= 0 or max_len <= 0: + raise build_error( + StatusCode.MODEL_SERVICE_CONFIG_ERROR, + error_msg="LAYA_SERVED_TIMEOUT_S and LAYA_SERVED_MAX_LEN must be positive", + ) + client = ServedLayaClient(url=url, api_key=os.getenv("LAYA_SERVED_API_KEY") or None, timeout_s=timeout_s) + return cls(client, model=os.getenv("LAYA_SERVED_MODEL") or DEFAULT_SERVED_MODEL, max_len=max_len) diff --git a/tests/data/served_laya/README.md b/tests/data/served_laya/README.md new file mode 100644 index 0000000..5261e74 --- /dev/null +++ b/tests/data/served_laya/README.md @@ -0,0 +1,17 @@ +# Served Laya fixtures + +Responses recorded from running servers on 2026-09-30, one JSON file per case: `status`, +`content_type`, `body` and, where it was sent, the `request`. + +| prefix | server | +|---|---| +| `worker.` | system1-omni's Laya worker at `6311ae8` (PR #30), `LAYA_WORKER_COMPILE=on`, `LAYA_WORKER_WEIGHTS=fp16`, `LAYA_API_KEY=fixture-token`, MPS | +| `frontend.` | the same worker behind `omni-jev` (system1-omni#2) | +| `frontend-down.` | `omni-jev` with the worker stopped | +| `laya-serve.` | plain `laya-serve`, no API key, MPS | + +laya 0.3.20, torch 2.14.0, checkpoint `convaiinnovations/laya` at `55cf4c4`, M1 Pro. +`tests/test_served_laya_fixtures.py` checks them against `docs/api/laya-systemone.current.openapi.yaml`; +`tests/test_decision_models_served.py` replays them through a mock transport. + +To record again, start a server and run `python capture.py [token]`. diff --git a/tests/data/served_laya/capture.py b/tests/data/served_laya/capture.py new file mode 100644 index 0000000..6a6e262 --- /dev/null +++ b/tests/data/served_laya/capture.py @@ -0,0 +1,116 @@ +"""Record /v1/systemone and /health responses from running servers into JSON fixtures.""" + +import http.client +import json +import sys +from pathlib import Path + +out = Path(sys.argv[1]) +server = sys.argv[2] +port = int(sys.argv[3]) +token = sys.argv[4] if len(sys.argv) > 4 else "" +out.mkdir(parents=True, exist_ok=True) +Q = { + "choice": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": {"billing": "Charges and refunds", "technical": "Software problems"}, + }, + "score": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"], + }, + "noul": {"type": "noul", "instructions": "Does the customer ask for a refund?"}, +} +ST = "I was charged twice for my order. Please refund the duplicate today." + + +def req(method, path, body=None, auth=True, raw=False): + c = http.client.HTTPConnection("127.0.0.1", port, timeout=60) + h = {"Content-Type": "application/json"} + if auth and token: + h["Authorization"] = f"Bearer {token}" + data = body if raw else (json.dumps(body).encode() if body is not None else None) + c.request(method, path, body=data, headers=h) + r = c.getresponse() + b = r.read() + ctype = r.getheader("content-type", "") + try: + parsed = json.loads(b) + except ValueError: + parsed = b.decode(errors="replace") + return {"status": r.status, "content_type": ctype.split(";")[0], "body": parsed} + + +cases = {"health": ("GET", "/health", None, True, False)} +if server != "frontend-down": + cases.update( + { + "choice": ( + "POST", + "/v1/systemone", + {"model": "english", "state": ST, "questions": {"department": Q["choice"]}}, + True, + False, + ), + "noul": ( + "POST", + "/v1/systemone", + {"model": "english", "state": ST, "questions": {"refund": Q["noul"]}}, + True, + False, + ), + "combined": ( + "POST", + "/v1/systemone", + { + "model": "english", + "state": ST, + "questions": {"department": Q["choice"], "urgency": Q["score"], "refund": Q["noul"]}, + }, + True, + False, + ), + "malformed_json": ("POST", "/v1/systemone", b"{not json", True, True), + "no_questions": ("POST", "/v1/systemone", {"model": "english", "state": ST}, True, False), + "invalid_question": ( + "POST", + "/v1/systemone", + {"model": "english", "state": ST, "questions": {"q": {"type": "bogus", "instructions": "?"}}}, + True, + False, + ), + "too_many_questions": ( + "POST", + "/v1/systemone", + {"model": "english", "state": "x", "questions": {f"q{i}": Q["noul"] for i in range(65)}}, + True, + False, + ), + } + ) + if token: + cases["unauthorized"] = ( + "POST", + "/v1/systemone", + {"model": "english", "state": ST, "questions": {"refund": Q["noul"]}}, + False, + False, + ) +else: + cases = { + "unreachable": ( + "POST", + "/v1/systemone", + {"model": "english", "state": ST, "questions": {"refund": Q["noul"]}}, + True, + False, + ) + } +for name, (m, p, b, a, raw) in cases.items(): + rec = req(m, p, b, a, raw) + if not raw and b is not None and name != "too_many_questions": + rec["request"] = b + (out / f"{server}.{name}.json").write_text(json.dumps(rec, indent=1, ensure_ascii=False) + "\n") + print(f"{server:14} {name:18} {rec['status']} {rec['content_type']}") diff --git a/tests/data/served_laya/frontend-down.unreachable.json b/tests/data/served_laya/frontend-down.unreachable.json new file mode 100644 index 0000000..12c0bf0 --- /dev/null +++ b/tests/data/served_laya/frontend-down.unreachable.json @@ -0,0 +1,15 @@ +{ + "status": 502, + "content_type": "text/plain", + "body": "backend unavailable\n", + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/frontend.choice.json b/tests/data/served_laya/frontend.choice.json new file mode 100644 index 0000000..b4eb617 --- /dev/null +++ b/tests/data/served_laya/frontend.choice.json @@ -0,0 +1,47 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.9519, + "technical": 0.0481 + }, + "confidence": 0.7217, + "answer_confidence": 0.9519, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 40, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + } + } + } +} diff --git a/tests/data/served_laya/frontend.combined.json b/tests/data/served_laya/frontend.combined.json new file mode 100644 index 0000000..a3d9c9a --- /dev/null +++ b/tests/data/served_laya/frontend.combined.json @@ -0,0 +1,88 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.952, + "technical": 0.048 + }, + "confidence": 0.7222, + "answer_confidence": 0.952, + "action": { + "act_probability": 1.0 + } + }, + "urgency": { + "type": "score", + "score": 1.7841, + "legend": { + "0": "Not urgent", + "1": "Needs attention soon", + "2": "Needs attention immediately" + }, + "probabilities": { + "0": 0.0241, + "1": 0.1677, + "2": 0.8082 + }, + "confidence": 0.4891, + "answer_confidence": 0.8082, + "action": { + "act_probability": 1.0 + } + }, + "refund": { + "type": "noul", + "noul": 0.9161, + "confidence": 0.9161, + "answer_confidence": 0.9161, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 137, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + }, + "urgency": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": [ + "Not urgent", + "Needs attention soon", + "Needs attention immediately" + ] + }, + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/frontend.health.json b/tests/data/served_laya/frontend.health.json new file mode 100644 index 0000000..8825dba --- /dev/null +++ b/tests/data/served_laya/frontend.health.json @@ -0,0 +1,39 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "status": "ok", + "ready": true, + "loaded": [ + "english" + ], + "device": "mps", + "requested_device": "mps", + "device_mismatch": false, + "weights_dtype": "torch.float16", + "autocast_dtype": "torch.float16", + "mps_amp_min_rows": 5, + "checkpoint": "convaiinnovations/laya", + "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", + "warmup_ms": 29202.8, + "models": { + "english": { + "device": "mps", + "requested_device": "mps", + "device_mismatch": false, + "weights_dtype": "torch.float16", + "autocast_dtype": "torch.float16", + "mps_amp_min_rows": 5, + "checkpoint": "convaiinnovations/laya", + "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", + "warmup_ms": 29202.8 + } + }, + "compile": { + "mode": "on", + "graphs_at_ready": 3, + "graphs_now": 3, + "recompiled_after_ready": false + } + } +} diff --git a/tests/data/served_laya/frontend.invalid_question.json b/tests/data/served_laya/frontend.invalid_question.json new file mode 100644 index 0000000..8c53247 --- /dev/null +++ b/tests/data/served_laya/frontend.invalid_question.json @@ -0,0 +1,17 @@ +{ + "status": 422, + "content_type": "application/json", + "body": { + "detail": "question 'q': unknown type 'bogus'; use one of ['choice', 'noul', 'score']" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "q": { + "type": "bogus", + "instructions": "?" + } + } + } +} diff --git a/tests/data/served_laya/frontend.malformed_json.json b/tests/data/served_laya/frontend.malformed_json.json new file mode 100644 index 0000000..8465324 --- /dev/null +++ b/tests/data/served_laya/frontend.malformed_json.json @@ -0,0 +1,7 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be valid JSON" + } +} diff --git a/tests/data/served_laya/frontend.no_questions.json b/tests/data/served_laya/frontend.no_questions.json new file mode 100644 index 0000000..e59b775 --- /dev/null +++ b/tests/data/served_laya/frontend.no_questions.json @@ -0,0 +1,11 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be an object with a 'questions' field" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today." + } +} diff --git a/tests/data/served_laya/frontend.noul.json b/tests/data/served_laya/frontend.noul.json new file mode 100644 index 0000000..7b2a96b --- /dev/null +++ b/tests/data/served_laya/frontend.noul.json @@ -0,0 +1,39 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "refund": { + "type": "noul", + "noul": 0.9162, + "confidence": 0.9162, + "answer_confidence": 0.9162, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 48, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/frontend.too_many_questions.json b/tests/data/served_laya/frontend.too_many_questions.json new file mode 100644 index 0000000..89d8d0f --- /dev/null +++ b/tests/data/served_laya/frontend.too_many_questions.json @@ -0,0 +1,7 @@ +{ + "status": 413, + "content_type": "application/json", + "body": { + "detail": "too many questions (65 > 64)" + } +} diff --git a/tests/data/served_laya/frontend.unauthorized.json b/tests/data/served_laya/frontend.unauthorized.json new file mode 100644 index 0000000..6f59214 --- /dev/null +++ b/tests/data/served_laya/frontend.unauthorized.json @@ -0,0 +1,17 @@ +{ + "status": 401, + "content_type": "application/json", + "body": { + "detail": "invalid or missing bearer token" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/laya-serve.choice.json b/tests/data/served_laya/laya-serve.choice.json new file mode 100644 index 0000000..3a167b8 --- /dev/null +++ b/tests/data/served_laya/laya-serve.choice.json @@ -0,0 +1,47 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.952, + "technical": 0.048 + }, + "confidence": 0.722, + "answer_confidence": 0.952, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 40, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + } + } + } +} diff --git a/tests/data/served_laya/laya-serve.combined.json b/tests/data/served_laya/laya-serve.combined.json new file mode 100644 index 0000000..af589f8 --- /dev/null +++ b/tests/data/served_laya/laya-serve.combined.json @@ -0,0 +1,88 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.952, + "technical": 0.048 + }, + "confidence": 0.722, + "answer_confidence": 0.952, + "action": { + "act_probability": 1.0 + } + }, + "urgency": { + "type": "score", + "score": 1.7837, + "legend": { + "0": "Not urgent", + "1": "Needs attention soon", + "2": "Needs attention immediately" + }, + "probabilities": { + "0": 0.0241, + "1": 0.1681, + "2": 0.8078 + }, + "confidence": 0.4885, + "answer_confidence": 0.8078, + "action": { + "act_probability": 1.0 + } + }, + "refund": { + "type": "noul", + "noul": 0.916, + "confidence": 0.916, + "answer_confidence": 0.916, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 137, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + }, + "urgency": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": [ + "Not urgent", + "Needs attention soon", + "Needs attention immediately" + ] + }, + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/laya-serve.health.json b/tests/data/served_laya/laya-serve.health.json new file mode 100644 index 0000000..6485454 --- /dev/null +++ b/tests/data/served_laya/laya-serve.health.json @@ -0,0 +1,11 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "status": "ok", + "loaded": [ + "english" + ], + "device": "mps" + } +} diff --git a/tests/data/served_laya/laya-serve.invalid_question.json b/tests/data/served_laya/laya-serve.invalid_question.json new file mode 100644 index 0000000..8c53247 --- /dev/null +++ b/tests/data/served_laya/laya-serve.invalid_question.json @@ -0,0 +1,17 @@ +{ + "status": 422, + "content_type": "application/json", + "body": { + "detail": "question 'q': unknown type 'bogus'; use one of ['choice', 'noul', 'score']" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "q": { + "type": "bogus", + "instructions": "?" + } + } + } +} diff --git a/tests/data/served_laya/laya-serve.malformed_json.json b/tests/data/served_laya/laya-serve.malformed_json.json new file mode 100644 index 0000000..8465324 --- /dev/null +++ b/tests/data/served_laya/laya-serve.malformed_json.json @@ -0,0 +1,7 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be valid JSON" + } +} diff --git a/tests/data/served_laya/laya-serve.no_questions.json b/tests/data/served_laya/laya-serve.no_questions.json new file mode 100644 index 0000000..e59b775 --- /dev/null +++ b/tests/data/served_laya/laya-serve.no_questions.json @@ -0,0 +1,11 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be an object with a 'questions' field" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today." + } +} diff --git a/tests/data/served_laya/laya-serve.noul.json b/tests/data/served_laya/laya-serve.noul.json new file mode 100644 index 0000000..0ce42ee --- /dev/null +++ b/tests/data/served_laya/laya-serve.noul.json @@ -0,0 +1,39 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "refund": { + "type": "noul", + "noul": 0.916, + "confidence": 0.916, + "answer_confidence": 0.916, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 48, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/laya-serve.too_many_questions.json b/tests/data/served_laya/laya-serve.too_many_questions.json new file mode 100644 index 0000000..89d8d0f --- /dev/null +++ b/tests/data/served_laya/laya-serve.too_many_questions.json @@ -0,0 +1,7 @@ +{ + "status": 413, + "content_type": "application/json", + "body": { + "detail": "too many questions (65 > 64)" + } +} diff --git a/tests/data/served_laya/worker.choice.json b/tests/data/served_laya/worker.choice.json new file mode 100644 index 0000000..b4eb617 --- /dev/null +++ b/tests/data/served_laya/worker.choice.json @@ -0,0 +1,47 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.9519, + "technical": 0.0481 + }, + "confidence": 0.7217, + "answer_confidence": 0.9519, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 40, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + } + } + } +} diff --git a/tests/data/served_laya/worker.combined.json b/tests/data/served_laya/worker.combined.json new file mode 100644 index 0000000..a3d9c9a --- /dev/null +++ b/tests/data/served_laya/worker.combined.json @@ -0,0 +1,88 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "department": { + "type": "choice", + "choice": "billing", + "probabilities": { + "billing": 0.952, + "technical": 0.048 + }, + "confidence": 0.7222, + "answer_confidence": 0.952, + "action": { + "act_probability": 1.0 + } + }, + "urgency": { + "type": "score", + "score": 1.7841, + "legend": { + "0": "Not urgent", + "1": "Needs attention soon", + "2": "Needs attention immediately" + }, + "probabilities": { + "0": 0.0241, + "1": 0.1677, + "2": 0.8082 + }, + "confidence": 0.4891, + "answer_confidence": 0.8082, + "action": { + "act_probability": 1.0 + } + }, + "refund": { + "type": "noul", + "noul": 0.9161, + "confidence": 0.9161, + "answer_confidence": 0.9161, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 137, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "department": { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": { + "billing": "Charges and refunds", + "technical": "Software problems" + } + }, + "urgency": { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": [ + "Not urgent", + "Needs attention soon", + "Needs attention immediately" + ] + }, + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/worker.health.json b/tests/data/served_laya/worker.health.json new file mode 100644 index 0000000..8825dba --- /dev/null +++ b/tests/data/served_laya/worker.health.json @@ -0,0 +1,39 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "status": "ok", + "ready": true, + "loaded": [ + "english" + ], + "device": "mps", + "requested_device": "mps", + "device_mismatch": false, + "weights_dtype": "torch.float16", + "autocast_dtype": "torch.float16", + "mps_amp_min_rows": 5, + "checkpoint": "convaiinnovations/laya", + "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", + "warmup_ms": 29202.8, + "models": { + "english": { + "device": "mps", + "requested_device": "mps", + "device_mismatch": false, + "weights_dtype": "torch.float16", + "autocast_dtype": "torch.float16", + "mps_amp_min_rows": 5, + "checkpoint": "convaiinnovations/laya", + "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", + "warmup_ms": 29202.8 + } + }, + "compile": { + "mode": "on", + "graphs_at_ready": 3, + "graphs_now": 3, + "recompiled_after_ready": false + } + } +} diff --git a/tests/data/served_laya/worker.invalid_question.json b/tests/data/served_laya/worker.invalid_question.json new file mode 100644 index 0000000..8c53247 --- /dev/null +++ b/tests/data/served_laya/worker.invalid_question.json @@ -0,0 +1,17 @@ +{ + "status": 422, + "content_type": "application/json", + "body": { + "detail": "question 'q': unknown type 'bogus'; use one of ['choice', 'noul', 'score']" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "q": { + "type": "bogus", + "instructions": "?" + } + } + } +} diff --git a/tests/data/served_laya/worker.malformed_json.json b/tests/data/served_laya/worker.malformed_json.json new file mode 100644 index 0000000..8465324 --- /dev/null +++ b/tests/data/served_laya/worker.malformed_json.json @@ -0,0 +1,7 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be valid JSON" + } +} diff --git a/tests/data/served_laya/worker.no_questions.json b/tests/data/served_laya/worker.no_questions.json new file mode 100644 index 0000000..e59b775 --- /dev/null +++ b/tests/data/served_laya/worker.no_questions.json @@ -0,0 +1,11 @@ +{ + "status": 400, + "content_type": "application/json", + "body": { + "detail": "request body must be an object with a 'questions' field" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today." + } +} diff --git a/tests/data/served_laya/worker.noul.json b/tests/data/served_laya/worker.noul.json new file mode 100644 index 0000000..7b2a96b --- /dev/null +++ b/tests/data/served_laya/worker.noul.json @@ -0,0 +1,39 @@ +{ + "status": 200, + "content_type": "application/json", + "body": { + "model": "laya-rl-agent", + "answers": { + "refund": { + "type": "noul", + "noul": 0.9162, + "confidence": 0.9162, + "answer_confidence": 0.9162, + "action": { + "act_probability": 1.0 + } + } + }, + "usage": { + "input_tokens": 48, + "output_tokens": 0 + }, + "routing": { + "model": "english", + "repo": "convaiinnovations/laya", + "reason": "explicit model='english'", + "detection": null, + "workflow": null + } + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/data/served_laya/worker.too_many_questions.json b/tests/data/served_laya/worker.too_many_questions.json new file mode 100644 index 0000000..89d8d0f --- /dev/null +++ b/tests/data/served_laya/worker.too_many_questions.json @@ -0,0 +1,7 @@ +{ + "status": 413, + "content_type": "application/json", + "body": { + "detail": "too many questions (65 > 64)" + } +} diff --git a/tests/data/served_laya/worker.unauthorized.json b/tests/data/served_laya/worker.unauthorized.json new file mode 100644 index 0000000..6f59214 --- /dev/null +++ b/tests/data/served_laya/worker.unauthorized.json @@ -0,0 +1,17 @@ +{ + "status": 401, + "content_type": "application/json", + "body": { + "detail": "invalid or missing bearer token" + }, + "request": { + "model": "english", + "state": "I was charged twice for my order. Please refund the duplicate today.", + "questions": { + "refund": { + "type": "noul", + "instructions": "Does the customer ask for a refund?" + } + } + } +} diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py new file mode 100644 index 0000000..d60feb1 --- /dev/null +++ b/tests/test_decision_models_served.py @@ -0,0 +1,510 @@ +# coding: utf-8 +"""``laya-served``: the shared contract, the error mapping, the window check and who answered. + +A ``httpx.MockTransport`` stands in for the server; its answers follow the responses recorded in +``tests/data/served_laya/``. No torch, laya or network. +""" + +from __future__ import annotations + +import json +import os +from pathlib import Path +from typing import Any +from unittest import IsolatedAsyncioTestCase +from unittest.mock import patch + +import httpx +from openjiuwen.core.common.exception.codes import StatusCode +from openjiuwen.core.common.exception.errors import BaseError + +from s1a.decision_models import ChoiceQuestion, NoulQuestion, Observation +from s1a.decision_models.laya import laya_question +from s1a.decision_models.served import ( + HEALTH_MAX_AGE_S, + ServedLayaClient, + ServedLayaModel, + parse_server_timing, +) +from tests.decision_model_contract import DecisionModelContract + +FIXTURES = Path(__file__).resolve().parent / "data" / "served_laya" +URL = "http://laya.test" +OBSERVATION = Observation({"ticket": "I was charged twice. Please refund the duplicate."}) +PICK = ChoiceQuestion({"billing": "Charges and refunds", "technical": "Software problems"}, rules="route it") +CHECK = NoulQuestion("Does the customer ask for a refund?") + + +def fixture(name: str) -> dict[str, Any]: + return json.loads((FIXTURES / f"{name}.json").read_text()) + + +WORKER_HEALTH = fixture("worker.health")["body"] +LAYA_SERVE_HEALTH = fixture("laya-serve.health")["body"] +ROUTING = {"model": "english", "repo": "convaiinnovations/laya", "reason": "explicit model='english'"} + + +def laya_answer(question: dict[str, Any]) -> dict[str, Any]: + """An answer the way laya-serve shapes it: the first option wins.""" + if question["type"] == "noul": + return { + "type": "noul", + "noul": 0.7, + "confidence": 0.7, + "answer_confidence": 0.7, + "action": {"act_probability": 1.0}, + } + keys = list(question["criteria"]) + rest = 0.2 / max(1, len(keys) - 1) + probabilities = {key: (0.8 if i == 0 else rest) for i, key in enumerate(keys)} if len(keys) > 1 else {keys[0]: 1.0} + return { + "type": "choice", + "choice": keys[0], + "probabilities": probabilities, + "confidence": 0.8, + "answer_confidence": 0.8, + "action": {"act_probability": 1.0}, + } + + +def ok(payload: dict[str, Any], headers: dict[str, str] | None = None) -> httpx.Response: + return httpx.Response(200, json=payload, headers=headers) + + +class Server: + """A scripted server: ``/health`` returns ``health``; each decision takes the next entry of ``script`` + (a response, an exception, or a callable of the request), then answers like laya-serve.""" + + def __init__(self, *, health: Any = WORKER_HEALTH, script: list[Any] | None = None, usage: int = 40) -> None: + self.health = health + self.script = list(script or []) + self.usage = usage + self.decisions: list[httpx.Request] = [] + self.health_calls = 0 + + def answer(self, request: httpx.Request) -> httpx.Response: + body = json.loads(request.content) + return ok( + { + "model": "laya-rl-agent", + "answers": {name: laya_answer(q) for name, q in body["questions"].items()}, + "usage": {"input_tokens": self.usage, "output_tokens": 0}, + "routing": ROUTING, + } + ) + + def handler(self, request: httpx.Request) -> httpx.Response: + if request.url.path == "/health": + self.health_calls += 1 + if isinstance(self.health, Exception): + raise self.health + if isinstance(self.health, httpx.Response): + return self.health + return ok(self.health) + self.decisions.append(request) + if self.script: + step = self.script.pop(0) + if isinstance(step, Exception): + raise step + if callable(step): + return step(request) + return step + return self.answer(request) + + +class Clock: + def __init__(self) -> None: + self.now = 1000.0 + + def __call__(self) -> float: + return self.now + + +def make_model( + server: Server, + *, + api_key: str | None = None, + timeout_s: float = 5.0, + max_len: int = 512, + clock: Clock | None = None, +) -> tuple[ServedLayaModel, list[float]]: + clock = clock or Clock() + slept: list[float] = [] + + async def sleep(seconds: float) -> None: + slept.append(seconds) + clock.now += seconds + + client = ServedLayaClient( + url=URL + "/", + api_key=api_key, + timeout_s=timeout_s, + transport=httpx.MockTransport(server.handler), + clock=clock, + sleep=sleep, + ) + return ServedLayaModel(client, max_len=max_len, clock=clock), slept + + +class ServedContract(DecisionModelContract, IsolatedAsyncioTestCase): + def make(self) -> ServedLayaModel: + return make_model(Server())[0] + + def make_scripted(self, answers: list[dict[str, Any]]) -> ServedLayaModel: + script = [ + ok({"model": "laya-rl-agent", "answers": a, "usage": {"input_tokens": 30}, "routing": ROUTING}) + for a in answers + ] + return make_model(Server(script=script))[0] + + +class RequestTests(IsolatedAsyncioTestCase): + async def test_the_body_is_what_in_process_laya_reads(self) -> None: + server = Server() + model, _ = make_model(server) + await model.decide_many(OBSERVATION, {"pick": PICK, "check": CHECK}) + request = server.decisions[0] + self.assertEqual(request.url, httpx.URL(URL + "/v1/systemone")) + self.assertEqual( + json.loads(request.content), + { + "model": "english", + "state": OBSERVATION.state, + "questions": {"pick": laya_question(PICK), "check": laya_question(CHECK)}, + }, + ) + self.assertIsInstance(json.loads(request.content)["questions"]["check"]["instructions"], str) + + async def test_the_bearer_token_is_sent_only_when_set(self) -> None: + server = Server() + await make_model(server, api_key="secret")[0].decide_many(OBSERVATION, {"pick": PICK}) + await make_model(server)[0].decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(server.decisions[0].headers.get("authorization"), "Bearer secret") + self.assertNotIn("authorization", server.decisions[1].headers) + + async def test_only_the_asked_questions_come_back(self) -> None: + extra = lambda request: ok( # noqa: E731 + { + "answers": {"pick": laya_answer(laya_question(PICK)), "stray": {"noul": 0.1}}, + "usage": {"input_tokens": 9}, + } + ) + decision = await make_model(Server(script=[extra]))[0].decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(list(decision.answers), ["pick"]) + + +class ErrorTests(IsolatedAsyncioTestCase): + async def assert_fails(self, server: Server, status: StatusCode, text: str, **kwargs: Any) -> BaseError: + model, _ = make_model(server, **kwargs) + with self.assertRaises(BaseError) as caught: + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(caught.exception.status, status) + self.assertIn(text, str(caught.exception)) + return caught.exception + + async def test_a_dropped_connection_is_retried_once(self) -> None: + server = Server(script=[httpx.ConnectError("refused")]) + decision = await make_model(server)[0].decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual((len(server.decisions), decision.choice("pick").key), (2, "billing")) + + async def test_one_request_id_per_decision_kept_across_the_retry(self) -> None: + server = Server(script=[httpx.ConnectError("refused")]) + model = make_model(server)[0] + first = await model.decide_many(OBSERVATION, {"pick": PICK}) + await model.decide_many(OBSERVATION, {"pick": PICK}) + sent = [request.headers["X-Request-Id"] for request in server.decisions] + self.assertEqual((sent[0], sent[1], len(set(sent))), (first.raw["request_id"], first.raw["request_id"], 2)) + + async def test_a_second_dropped_connection_fails(self) -> None: + server = Server(script=[httpx.ConnectError("refused"), httpx.ConnectError("refused")]) + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "unreachable at http://laya.test") + self.assertEqual(len(server.decisions), 2) + + async def test_502_and_504_from_the_frontend_are_retried_once(self) -> None: + for status in (502, 504): + server = Server(script=[httpx.Response(status, text="backend unavailable\n")]) + await make_model(server)[0].decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(len(server.decisions), 2, status) + failing = Server(script=[httpx.Response(status, text="backend timed out\n")] * 2) + await self.assert_fails(failing, StatusCode.MODEL_CALL_FAILED, f"HTTP {status}") + + async def test_503_waits_retry_after_then_retries(self) -> None: + server = Server(script=[httpx.Response(503, headers={"Retry-After": "1"}, json={"detail": "busy"})]) + model, slept = make_model(server) + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual((slept, len(server.decisions)), ([1.0], 2)) + + async def test_503_past_the_deadline_fails_without_waiting(self) -> None: + server = Server(script=[httpx.Response(503, headers={"Retry-After": "30"}, json={"detail": "busy"})]) + model, slept = make_model(server) + with self.assertRaises(BaseError) as caught: + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertIn("overloaded", str(caught.exception)) + self.assertEqual((slept, len(server.decisions)), ([], 1)) + + async def test_request_errors_fail_at_once_with_the_servers_reason(self) -> None: + for name in ( + "worker.malformed_json", + "worker.no_questions", + "worker.invalid_question", + "worker.too_many_questions", + ): + record = fixture(name) + server = Server(script=[httpx.Response(record["status"], json=record["body"])]) + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, record["body"]["detail"]) + self.assertEqual(len(server.decisions), 1, name) + + async def test_problem_json_reasons_carry_the_code(self) -> None: + problem = { + "type": "urn:laya:problem:invalid-question", + "title": "Invalid question", + "status": 422, + "code": "invalid_question", + "detail": "question 'q': unknown type", + } + response = httpx.Response(422, json=problem, headers={"content-type": "application/problem+json"}) + await self.assert_fails( + Server(script=[response]), StatusCode.MODEL_CALL_FAILED, "invalid_question: question 'q'" + ) + + async def test_401_is_a_configuration_error(self) -> None: + record = fixture("worker.unauthorized") + server = Server(script=[httpx.Response(401, json=record["body"])]) + await self.assert_fails(server, StatusCode.MODEL_SERVICE_CONFIG_ERROR, "LAYA_SERVED_API_KEY") + + async def test_500_fails_at_once(self) -> None: + server = Server(script=[httpx.Response(500, json={"detail": "inference failed"})]) + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "inference failed") + self.assertEqual(len(server.decisions), 1) + + async def test_a_read_timeout_fails_with_the_deadline(self) -> None: + server = Server(script=[httpx.ReadTimeout("slow")]) + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "within 5 s") + + async def test_the_deadline_covers_the_retry(self) -> None: + clock = Clock() + + def slow_refusal(request: httpx.Request) -> httpx.Response: + clock.now += 6.0 # the first attempt used the whole deadline + raise httpx.ConnectError("refused") + + server = Server(script=[slow_refusal]) + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "within 5 s", clock=clock) + self.assertEqual(len(server.decisions), 1) + + async def test_a_body_without_answers_is_malformed(self) -> None: + for response in (httpx.Response(200, text="not json"), ok({"model": "x"}), httpx.Response(200, json=[1])): + await self.assert_fails(Server(script=[response]), StatusCode.MODEL_CALL_FAILED, "served Laya returned") + + +class WindowTests(IsolatedAsyncioTestCase): + async def test_a_filled_window_is_an_error(self) -> None: + model, _ = make_model(Server(usage=512), max_len=512) + with self.assertRaises(BaseError) as caught: + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(caught.exception.status, StatusCode.MODEL_SERVICE_CONFIG_ERROR) + self.assertIn("LAYA_SERVED_MAX_LEN", str(caught.exception)) + + async def test_the_window_scales_with_the_questions(self) -> None: + model, _ = make_model(Server(usage=600), max_len=512) + await model.decide_many(OBSERVATION, {"pick": PICK, "check": CHECK}) # 600 < 2 * 512 + + +class WarmTests(IsolatedAsyncioTestCase): + async def test_warm_fails_early_when_no_server_listens(self) -> None: + model, _ = make_model(Server(health=httpx.ConnectError("refused"))) + with self.assertRaises(BaseError) as caught: + await model.warm() + self.assertEqual(caught.exception.status, StatusCode.MODEL_CALL_FAILED) + self.assertIn("listens only after loading and warming up", str(caught.exception)) + + async def test_an_unusable_health_is_a_warning_not_an_error(self) -> None: + for health in (httpx.Response(404), httpx.Response(200, text="ok")): + model, _ = make_model(Server(health=health)) + with self.assertLogs("s1a.decision_models.served", level="WARNING"): + await model.warm() + decision = await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(decision.model, "convaiinnovations/laya") # from the response's routing + + +class IdentityTests(IsolatedAsyncioTestCase): + async def raw(self, server: Server, clock: Clock | None = None) -> tuple[Any, dict[str, Any]]: + model, _ = make_model(server, clock=clock) + await model.warm() + decision = await model.decide_many(OBSERVATION, {"pick": PICK}) + return decision, decision.raw + + async def test_the_worker_names_checkpoint_revision_and_device(self) -> None: + decision, raw = await self.raw(Server()) + self.assertEqual(decision.model, "convaiinnovations/laya@55cf4c4ebb4e") + served_by = raw["served_by"] + self.assertEqual( + {k: served_by[k] for k in ("checkpoint", "device", "weights_dtype", "compile", "source")}, + { + "checkpoint": "convaiinnovations/laya", + "device": "mps", + "weights_dtype": "torch.float16", + "compile": "on", + "source": "health", + }, + ) + self.assertTrue(served_by["read_at"]) + self.assertEqual(raw["url"], URL) + + async def test_plain_laya_serve_falls_back_to_the_routed_repo(self) -> None: + decision, raw = await self.raw(Server(health=LAYA_SERVE_HEALTH)) + self.assertEqual(decision.model, "convaiinnovations/laya") + self.assertEqual( + raw["served_by"], + {"checkpoint": "convaiinnovations/laya", "revision": None, "device": None, "source": "routing"}, + ) + + async def test_the_answering_model_is_looked_up_by_routing(self) -> None: + health = json.loads(json.dumps(WORKER_HEALTH)) + health["models"]["multilingual"] = { + **health["models"]["english"], + "device": "cpu", + "checkpoint": "convaiinnovations/laya-multilingual", + } + routed = lambda request: ok( # noqa: E731 + { + "answers": {"pick": laya_answer(laya_question(PICK))}, + "usage": {"input_tokens": 9}, + "routing": {"model": "multilingual", "repo": "convaiinnovations/laya"}, + } + ) + _, raw = await self.raw(Server(health=health, script=[routed])) + self.assertEqual( + (raw["served_by"]["device"], raw["served_by"]["checkpoint"]), ("cpu", "convaiinnovations/laya-multilingual") + ) + + async def test_served_by_in_the_response_wins(self) -> None: + sent = { + "checkpoint": "convaiinnovations/laya", + "revision": "abcdef1234567890", + "device": "mps", + "weights_dtype": "float16", + } + with_served_by = lambda request: ok( # noqa: E731 + { + "model": "english", + "served_by": sent, + "answers": {"pick": laya_answer(laya_question(PICK))}, + "usage": {"input_tokens": 9}, + } + ) + decision, raw = await self.raw(Server(script=[with_served_by])) + self.assertEqual( + (decision.model, raw["served_by"]["source"]), ("convaiinnovations/laya@abcdef123456", "response") + ) + + async def test_server_timing_is_kept(self) -> None: + timed = lambda request: ok( # noqa: E731 + {"answers": {"pick": laya_answer(laya_question(PICK))}, "usage": {"input_tokens": 9}}, + headers={"Server-Timing": "queue;dur=0.2, infer;dur=27.4"}, + ) + _, raw = await self.raw(Server(script=[timed])) + self.assertEqual(raw["server_timing"], {"queue": 0.2, "infer": 27.4}) + + async def test_a_reading_older_than_the_limit_is_refreshed(self) -> None: + clock = Clock() + server = Server() + model, _ = make_model(server, clock=clock) + await model.warm() + await model.decide_many(OBSERVATION, {"pick": PICK}) + clock.now += HEALTH_MAX_AGE_S - 1 + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(server.health_calls, 1) + server.health = { + **WORKER_HEALTH, + "models": {"english": {**WORKER_HEALTH["models"]["english"], "device": "cpu"}}, + } + clock.now += 2 + decision = await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual((server.health_calls, decision.raw["served_by"]["device"]), (2, "cpu")) + + async def test_a_failed_refresh_keeps_the_previous_reading(self) -> None: + clock = Clock() + server = Server() + model, _ = make_model(server, clock=clock) + await model.warm() + server.health = httpx.ReadTimeout("slow") + clock.now += HEALTH_MAX_AGE_S + 1 + with self.assertLogs("s1a.decision_models.served", level="WARNING"): + decision = await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(decision.raw["served_by"]["device"], "mps") + + +class RecordedResponseTests(IsolatedAsyncioTestCase): + async def test_recorded_worker_and_laya_serve_answers_validate(self) -> None: + for server_name in ("worker", "frontend", "laya-serve"): + record = fixture(f"{server_name}.combined") + request = record["request"] + questions = { + "department": ChoiceQuestion(request["questions"]["department"]["criteria"], rules="route"), + "refund": NoulQuestion(request["questions"]["refund"]["instructions"]), + } + payload = dict(record["body"]) + payload["answers"] = {k: payload["answers"][k] for k in questions} + model, _ = make_model(Server(script=[ok(payload)])) + decision = await model.decide_many(Observation(request["state"]), questions) + self.assertEqual(decision.choice("department").key, "billing", server_name) + self.assertGreater(decision.noul("refund").p, 0.5, server_name) + + +class ConfigTests(IsolatedAsyncioTestCase): + def env(self, **values: str) -> Any: + keys = ( + "LAYA_SERVED_URL", + "LAYA_SERVED_MODEL", + "LAYA_SERVED_API_KEY", + "LAYA_SERVED_TIMEOUT_S", + "LAYA_SERVED_MAX_LEN", + ) + return patch.dict(os.environ, {**{k: "" for k in keys}, **values}) + + async def test_the_url_is_required(self) -> None: + with self.env(), self.assertRaises(BaseError) as caught: + ServedLayaModel.from_env() + self.assertEqual(caught.exception.status, StatusCode.MODEL_SERVICE_CONFIG_ERROR) + self.assertIn("LAYA_SERVED_URL", str(caught.exception)) + + async def test_defaults_and_overrides(self) -> None: + with self.env(LAYA_SERVED_URL="http://127.0.0.1:8000"): + model = ServedLayaModel.from_env() + self.assertEqual( + (model.model, model._max_len, model._client.url), ("english", 512, "http://127.0.0.1:8000") + ) + await model.close() + with self.env( + LAYA_SERVED_URL="http://h:8080/", + LAYA_SERVED_MODEL="multilingual", + LAYA_SERVED_MAX_LEN="1024", + LAYA_SERVED_TIMEOUT_S="2.5", + LAYA_SERVED_API_KEY="k", + ): + model = ServedLayaModel.from_env() + self.assertEqual((model.model, model._max_len, model._client.url), ("multilingual", 1024, "http://h:8080")) + self.assertEqual(model._client._client.headers["authorization"], "Bearer k") + await model.close() + + async def test_bad_numbers_are_configuration_errors(self) -> None: + for values in ({"LAYA_SERVED_TIMEOUT_S": "soon"}, {"LAYA_SERVED_MAX_LEN": "0"}): + with self.env(LAYA_SERVED_URL="http://h", **values), self.assertRaises(BaseError) as caught: + ServedLayaModel.from_env() + self.assertEqual(caught.exception.status, StatusCode.MODEL_SERVICE_CONFIG_ERROR) + + async def test_no_cloud_key_is_read(self) -> None: + with ( + self.env(LAYA_SERVED_URL="http://h"), + patch.dict(os.environ, {"TYPESAFE_API_KEY": "", "OPENROUTER_API_KEY": ""}), + ): + await ServedLayaModel.from_env().close() + + +def test_server_timing_parsing() -> None: + assert parse_server_timing(None) == {} + assert parse_server_timing('queue;dur=0.2, infer;desc="x";dur=27.4, proxy, bad;dur=x') == { + "queue": 0.2, + "infer": 27.4, + } diff --git a/tests/test_served_laya_fixtures.py b/tests/test_served_laya_fixtures.py new file mode 100644 index 0000000..2656967 --- /dev/null +++ b/tests/test_served_laya_fixtures.py @@ -0,0 +1,58 @@ +# coding: utf-8 +"""The recorded served-Laya responses match the as-implemented spec, so the mock server in +``test_decision_models_served`` answers the way the real ones do.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest +import yaml +from jsonschema import Draft202012Validator +from referencing import Registry, Resource +from referencing.jsonschema import DRAFT202012 + +ROOT = Path(__file__).resolve().parents[1] +FIXTURES = sorted((ROOT / "tests" / "data" / "served_laya").glob("*.json")) +SPEC = yaml.safe_load((ROOT / "docs" / "api" / "laya-systemone.current.openapi.yaml").read_text()) +REGISTRY = Registry().with_resource("urn:spec", Resource.from_contents(SPEC, default_specification=DRAFT202012)) + + +def errors(schema: str, instance: object) -> list[str]: + validator = Draft202012Validator({"$ref": f"urn:spec#/components/schemas/{schema}"}, registry=REGISTRY) + return [error.message for error in validator.iter_errors(instance)] + + +def test_there_are_fixtures_from_every_server() -> None: + prefixes = {path.name.split(".")[0] for path in FIXTURES} + assert prefixes == {"worker", "frontend", "frontend-down", "laya-serve"} + + +@pytest.mark.parametrize("path", FIXTURES, ids=[path.stem for path in FIXTURES]) +def test_fixture_matches_the_spec(path: Path) -> None: + record = json.loads(path.read_text()) + server, case = path.stem.split(".", 1) + status, body = record["status"], record["body"] + if status in (502, 504): # the frontend's own errors are plain text + assert record["content_type"] == "text/plain" + return + assert record["content_type"] == "application/json" + if case == "health": + if server == "laya-serve": # answers before warmup, with the configured device only + assert set(body) == {"status", "loaded", "device"} + else: + assert errors("Health", body) == [] + return + schema = "DecisionResponse" if status == 200 else "Error" + assert errors(schema, body) == [] + + +@pytest.mark.parametrize("path", [p for p in FIXTURES if "request" in json.loads(p.read_text())], ids=lambda p: p.stem) +def test_sent_requests_match_the_spec_unless_meant_to_fail(path: Path) -> None: + record = json.loads(path.read_text()) + found = errors("DecisionRequest", record["request"]) + if record["status"] in (400, 422): + assert found, "a request the server rejected should not pass the schema" + else: + assert found == [] diff --git a/uv.lock b/uv.lock index c8be294..3f547a5 100644 --- a/uv.lock +++ b/uv.lock @@ -5180,8 +5180,11 @@ cua = [ { name = "huggingface-hub" }, ] dev = [ + { name = "jsonschema" }, { name = "pytest" }, { name = "pytest-asyncio" }, + { name = "pyyaml" }, + { name = "referencing" }, { name = "ruff" }, { name = "ty" }, ] @@ -5200,6 +5203,7 @@ requires-dist = [ { name = "cua-s1", marker = "extra == 'cua'", git = "https://github.com/trycua/cua.git?subdirectory=libs%2Fcua-s1%2Fpython&rev=aea61b6eb97e2d8c0f6f71eb804e5769fe910af4" }, { name = "httpx", specifier = ">=0.28" }, { name = "huggingface-hub", marker = "extra == 'cua'", specifier = ">=0.24" }, + { name = "jsonschema", marker = "extra == 'dev'", specifier = ">=4.18" }, { name = "laya", marker = "extra == 'laya'", specifier = ">=0.3.20" }, { name = "mcp", specifier = ">=1.26" }, { name = "opencv-python-headless", marker = "extra == 'alfworld-visual'", specifier = ">=4.10" }, @@ -5209,6 +5213,8 @@ requires-dist = [ { name = "pytest", marker = "extra == 'dev'", specifier = ">=8" }, { name = "pytest-asyncio", marker = "extra == 'dev'", specifier = ">=0.24" }, { name = "python-dotenv", specifier = ">=1.0" }, + { name = "pyyaml", marker = "extra == 'dev'", specifier = ">=6" }, + { name = "referencing", marker = "extra == 'dev'", specifier = ">=0.30" }, { name = "rlcard", marker = "extra == 'blackjack'", specifier = ">=1.2" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.6" }, { name = "system1-agents", extras = ["alfworld"], marker = "extra == 'alfworld-visual'" }, From d873bc239d535eed237544312997c419d2e14d60 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:36:15 +0800 Subject: [PATCH 05/14] [Feat] Offer laya-served wherever laya is offered (#20) --model laya-served on every agent, on decide, on the rails and in MCP decide. A test finds any name list, match or Literal that has laya without laya-served, so a new front cannot miss it. --- s1a/browser/browse.py | 5 ++- s1a/cli.py | 11 ++++- s1a/decision_models/factory.py | 6 ++- s1a/mcp_server.py | 12 +++-- s1a/rails.py | 7 ++- s1a/tool/loop.py | 4 +- s1a/tool/series.py | 2 +- tests/test_browse.py | 2 +- tests/test_decision_models_factory.py | 2 +- tests/test_mcp_server.py | 6 +-- tests/test_served_laya_registration.py | 62 ++++++++++++++++++++++++++ 11 files changed, 100 insertions(+), 19 deletions(-) create mode 100644 tests/test_served_laya_registration.py diff --git a/s1a/browser/browse.py b/s1a/browser/browse.py index b9be304..db7150a 100644 --- a/s1a/browser/browse.py +++ b/s1a/browser/browse.py @@ -37,9 +37,10 @@ BROWSER_MODEL_NAMES = ( "jev", "laya", + "laya-served", "cua", "llm", -) # a decision model (Jev over HTTP, Laya or Cua-S1 in process) or the chat model +) # a decision model (Jev or Laya over HTTP, Laya or Cua-S1 in process) or the chat model RUNS_DIR = HOME / "runs" / "browser" @@ -164,7 +165,7 @@ async def browse( workspace = str(logs_dir / "workspace") # the harness scaffolds SOUL.md, memory/ and friends here, not in the cwd instance = BrowserInstanceConfig(launch_args=browser_launch_args(headless)) match model_name: - case "jev" | "laya" | "cua": + case "jev" | "laya" | "laya-served" | "cua": if decision_model is None: raise RuntimeError(f"--model {model_name} needs a decision model") slot_model = BrowserDecisionModel(spec, policy, counted, decision_model=decision_model, value_model=None) diff --git a/s1a/cli.py b/s1a/cli.py index d049de0..0169d18 100644 --- a/s1a/cli.py +++ b/s1a/cli.py @@ -22,6 +22,7 @@ DECIDE_MODEL_NAMES = ( "jev", "laya", + "laya-served", "cua", ) # the decision models that answer one question on their own: no env, no rule, no chance @@ -45,7 +46,10 @@ def parser() -> argparse.ArgumentParser: ) decide.add_argument("--rules", required=True, help="the facts the model applies when it picks") decide.add_argument( - "--model", choices=DECIDE_MODEL_NAMES, default="jev", help="who answers: jev, or laya and cua in process" + "--model", + choices=DECIDE_MODEL_NAMES, + default="jev", + help="who answers: jev or laya-served over HTTP, laya and cua in process", ) fit = commands.add_parser("probe", help="the fit probe: hand-written choice cases from a JSONL file") fit.add_argument( @@ -54,7 +58,10 @@ def parser() -> argparse.ArgumentParser: help="JSONL, one case per line: state (object), options (key to text), rules, accept (list of right keys), note", ) fit.add_argument( - "--model", choices=DECIDE_MODEL_NAMES, default="jev", help="who answers: jev, or laya and cua in process" + "--model", + choices=DECIDE_MODEL_NAMES, + default="jev", + help="who answers: jev or laya-served over HTTP, laya and cua in process", ) return build diff --git a/s1a/decision_models/factory.py b/s1a/decision_models/factory.py index d81ae45..a73d6b0 100644 --- a/s1a/decision_models/factory.py +++ b/s1a/decision_models/factory.py @@ -8,10 +8,12 @@ from s1a.decision_models.cua import CuaS1Model from s1a.decision_models.jev import JevModel from s1a.decision_models.laya import LayaModel +from s1a.decision_models.served import ServedLayaModel DECISION_MODEL_NAMES = ( "jev", "laya", + "laya-served", "cua", "random", "rule", @@ -19,12 +21,14 @@ def build_model(model_name: str, *, seed: int = 0, rule: tuple[str, Rule] | None = None) -> DecisionModel: - """``jev``, ``laya`` and ``cua`` from the environment, ``random`` from the seed, ``rule`` from the agent's baseline.""" + """``jev``, ``laya``, ``laya-served`` and ``cua`` from the environment, ``random`` from the seed, ``rule`` from the agent's baseline.""" match model_name: case "jev": return JevModel.from_env() case "laya": return LayaModel.from_env() # the laya import happens inside + case "laya-served": + return ServedLayaModel.from_env() # Laya over HTTP; no torch, no cloud key case "cua": return CuaS1Model.from_env() # the cua_s1 import happens inside case "random": diff --git a/s1a/mcp_server.py b/s1a/mcp_server.py index aa35f03..af24dd5 100644 --- a/s1a/mcp_server.py +++ b/s1a/mcp_server.py @@ -25,8 +25,8 @@ "constraint puzzles or free-text generation. list_agents gives every agent's flags: a browser agent takes " "--model jev --goal '...' and needs a chat-model key (OPENAI_API_KEY or LLM_API_KEY plus MODEL_NAME), a Jev key " "(TYPESAFE_API_KEY or OPENROUTER_API_KEY) and Node for @playwright/mcp; a tool agent takes --model, --rethink and " - "--episodes; a rail takes --labelled-set. decide accepts model jev (default), laya or cua; local models need " - "their optional extra and no Jev key. Runs and decisions go one at a time per server." + "--episodes; a rail takes --labelled-set. decide accepts model jev (default), laya, cua or laya-served; local models " + "need their optional extra and no Jev key, laya-served needs LAYA_SERVED_URL. Runs and decisions go one at a time per server." ) @@ -91,11 +91,15 @@ async def run_agent(name: str, flags: list[str]) -> dict[str, Any]: @server.tool() async def decide( - state: dict[str, Any], options: dict[str, str], rules: str, model: Literal["jev", "laya", "cua"] = "jev" + state: dict[str, Any], + options: dict[str, str], + rules: str, + model: Literal["jev", "laya", "laya-served", "cua"] = "jev", ) -> dict[str, Any]: """One choice question: the chosen key, probabilities, confidence and decision latency in ms. - Use jev over HTTP (default), or laya/cua locally after installing the matching extra. Local model loading + Use jev over HTTP (default), laya/cua locally after installing the matching extra, or laya-served against a + running Laya server (LAYA_SERVED_URL). Local model loading is excluded from the reported latency. """ async with _ONE_RUN: diff --git a/s1a/rails.py b/s1a/rails.py index b8941da..ec2d154 100644 --- a/s1a/rails.py +++ b/s1a/rails.py @@ -29,7 +29,7 @@ from s1a.spec import Json, RailSpec, Verdict QUESTION = "check" -RAIL_MODEL_NAMES = ("jev", "laya") +RAIL_MODEL_NAMES = ("jev", "laya", "laya-served") def question(spec: RailSpec) -> Question: @@ -171,7 +171,10 @@ def parser(spec: RailSpec) -> argparse.ArgumentParser: help="JSONL records with a state and a boolean label; the spec's set when it names one", ) build.add_argument( - "--model", choices=RAIL_MODEL_NAMES, default="jev", help="who answers the question: jev, or laya in process" + "--model", + choices=RAIL_MODEL_NAMES, + default="jev", + help="who answers the question: jev or laya-served over HTTP, laya in process", ) return build diff --git a/s1a/tool/loop.py b/s1a/tool/loop.py index 2d71ab1..cdae413 100644 --- a/s1a/tool/loop.py +++ b/s1a/tool/loop.py @@ -26,7 +26,7 @@ from s1a.tool.rethink import RethinkRail from s1a.tool.models import ACT_TOOL, OBSERVE_TOOL, EvalState, ToolDecisionModel -MODEL_NAMES = ("jev", "llm", "random", "rule", "laya", "cua") +MODEL_NAMES = ("jev", "llm", "random", "rule", "laya", "laya-served", "cua") EVAL_PROMPT = ( "You play a game through two tools. Call observe first. Then call act with exactly one of the candidate keys the " "last tool result offered, one act per turn, until done is true. Then reply with one line: the final score." @@ -142,7 +142,7 @@ def build_slot_model( if chat is None: raise RuntimeError("--model llm needs the chat model: OPENAI_API_KEY or LLM_API_KEY, and MODEL_NAME") return chat - case "jev" | "laya" | "cua" | "random" | "rule": + case "jev" | "laya" | "laya-served" | "cua" | "random" | "rule": if decision_model is None: raise RuntimeError(f"--model {model_name} needs a decision model") return ToolDecisionModel(env, state, rules=rules, decision_model=decision_model, fallback=chat) diff --git a/s1a/tool/series.py b/s1a/tool/series.py index 17aed74..6f6166b 100644 --- a/s1a/tool/series.py +++ b/s1a/tool/series.py @@ -83,7 +83,7 @@ async def play(spec: ToolAgentSpec, args: argparse.Namespace, *, results_dir: Pa raise RuntimeError("--model llm needs the chat model: OPENAI_API_KEY or LLM_API_KEY, and MODEL_NAME") if args.rethink == "on" and spec.budget.stall_after > 0 and chat is None: raise RuntimeError("--rethink on needs the chat model for plans: OPENAI_API_KEY or LLM_API_KEY, and MODEL_NAME") - shared = build_model(args.model) if args.model in ("jev", "laya", "cua") else None + shared = build_model(args.model) if args.model in ("jev", "laya", "laya-served", "cua") else None run = await asyncio.to_thread(spec.series, args) # question fetches, game file parsing: seconds of blocking I/O if args.model == "rule": shared = build_model("rule", rule=run.baseline) diff --git a/tests/test_browse.py b/tests/test_browse.py index 71f9c20..0d3676c 100644 --- a/tests/test_browse.py +++ b/tests/test_browse.py @@ -56,7 +56,7 @@ def test_a_timeout_or_max_steps_at_or_below_zero_is_a_usage_error(self) -> None: self.assertEqual(caught.exception.code, 2) def test_the_model_flag_takes_a_decision_model_or_the_chat_model(self) -> None: - self.assertEqual(browse.BROWSER_MODEL_NAMES, ("jev", "laya", "cua", "llm")) + self.assertEqual(browse.BROWSER_MODEL_NAMES, ("jev", "laya", "laya-served", "cua", "llm")) for model_name in browse.BROWSER_MODEL_NAMES: self.assertEqual(browse.parser(SPEC).parse_args(["--model", model_name, "--goal", "x"]).model, model_name) with self.assertRaises(SystemExit): diff --git a/tests/test_decision_models_factory.py b/tests/test_decision_models_factory.py index d38866e..d94a4b3 100644 --- a/tests/test_decision_models_factory.py +++ b/tests/test_decision_models_factory.py @@ -37,7 +37,7 @@ def test_every_name_builds_its_class(self) -> None: rule = build_model("rule", rule=("always-inc", lambda state, options: "inc")) self.assertIsInstance(rule, RuleModel) self.assertEqual(rule.name, "always-inc") - self.assertEqual(DECISION_MODEL_NAMES, ("jev", "laya", "cua", "random", "rule")) + self.assertEqual(DECISION_MODEL_NAMES, ("jev", "laya", "laya-served", "cua", "random", "rule")) def test_the_errors(self) -> None: with self.assertRaises(RuntimeError): diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py index 737db84..0a58280 100644 --- a/tests/test_mcp_server.py +++ b/tests/test_mcp_server.py @@ -151,12 +151,12 @@ async def test_decide_advertises_optional_model_choices(self) -> None: async with create_connected_server_and_client_session(mcp_server.server) as session: tools = await session.list_tools() schema = next(tool.inputSchema for tool in tools.tools if tool.name == "decide") - self.assertEqual(schema["properties"]["model"]["enum"], ["jev", "laya", "cua"]) + self.assertEqual(schema["properties"]["model"]["enum"], ["jev", "laya", "laya-served", "cua"]) self.assertEqual(schema["properties"]["model"]["default"], "jev") self.assertNotIn("model", schema["required"]) async def test_decide_selects_and_closes_each_local_model_without_api_keys(self) -> None: - for model_name in ("laya", "cua"): + for model_name in ("laya", "laya-served", "cua"): with self.subTest(model=model_name): decision_model = ScriptedModel(choose="inc") with ( @@ -248,7 +248,7 @@ async def test_list_agents_names_every_front_with_its_flags(self) -> None: flags = {name: {row["flag"]: row for row in rows[name]["flags"]} for name in ("game2048", "flights")} self.assertEqual( (flags["game2048"]["--model"]["required"], flags["game2048"]["--model"]["choices"]), - (True, ["jev", "llm", "random", "rule", "laya", "cua"]), + (True, ["jev", "llm", "random", "rule", "laya", "laya-served", "cua"]), ) max_steps = str(agents.load("game2048").budget.max_steps) self.assertEqual( diff --git a/tests/test_served_laya_registration.py b/tests/test_served_laya_registration.py new file mode 100644 index 0000000..2b80697 --- /dev/null +++ b/tests/test_served_laya_registration.py @@ -0,0 +1,62 @@ +# coding: utf-8 +"""``laya-served`` is offered wherever ``laya`` is: every tuple, list or set of model names, every ``match`` over +them and every ``Literal`` that lists ``laya`` lists ``laya-served`` too. A place added later without it fails here.""" + +from __future__ import annotations + +import ast +import typing +from pathlib import Path + +from s1a import mcp_server +from s1a.cli import DECIDE_MODEL_NAMES +from s1a.decision_models import DECISION_MODEL_NAMES +from s1a.rails import RAIL_MODEL_NAMES +from s1a.tool.loop import MODEL_NAMES + +SOURCE = Path(__file__).resolve().parents[1] / "s1a" + + +def constants(node: ast.AST) -> set[str]: + return {n.value for n in ast.walk(node) if isinstance(n, ast.Constant) and isinstance(n.value, str)} + + +def places_missing_served() -> list[str]: + missing = [] + for path in sorted(SOURCE.rglob("*.py")): + tree = ast.parse(path.read_text()) + for node in ast.walk(tree): + if isinstance(node, (ast.Tuple, ast.List, ast.Set)): + values = {e.value for e in node.elts if isinstance(e, ast.Constant)} + elif isinstance(node, ast.Match): + values = set().union(*(constants(case.pattern) for case in node.cases)) + elif ( + isinstance(node, ast.Subscript) + and getattr(node.value, "id", getattr(node.value, "attr", "")) == "Literal" + ): + values = constants(node.slice) + else: + continue + if "laya" in values and "laya-served" not in values: + missing.append(f"{path.relative_to(SOURCE.parent)}:{node.lineno}") + return missing + + +def test_every_place_that_offers_laya_offers_laya_served() -> None: + assert places_missing_served() == [] + + +def test_the_named_lists() -> None: + for names in (DECISION_MODEL_NAMES, DECIDE_MODEL_NAMES, RAIL_MODEL_NAMES, MODEL_NAMES): + assert "laya-served" in names + decide_model = typing.get_type_hints( + mcp_server.decide.__wrapped__ if hasattr(mcp_server.decide, "__wrapped__") else mcp_server.decide + )["model"] + assert "laya-served" in typing.get_args(decide_model) + + +def test_the_check_would_catch_a_missing_place(tmp_path: Path, monkeypatch) -> None: + (tmp_path / "s1a").mkdir() + (tmp_path / "s1a" / "new_front.py").write_text('NAMES = ("jev", "laya")\n') + monkeypatch.setattr(__name__ + ".SOURCE", tmp_path / "s1a") + assert places_missing_served() == ["s1a/new_front.py:1"] From d82f5ceb8fac752f88b6f4b455fa94f6c3cb4724 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:36:16 +0800 Subject: [PATCH 06/14] [Feat] Record the answering model and served provenance in ticks (#20) A tool-front tick keeps Decision.model as model and, from a served model, served_by, url, request_id and server_timing (Decision.provenance), so a run's artifacts show the checkpoint, revision and device of each step. --- s1a/decision_models/types.py | 6 ++++++ s1a/tool/models.py | 2 ++ tests/test_tool_models.py | 34 ++++++++++++++++++++++++++++++++++ 3 files changed, 42 insertions(+) diff --git a/s1a/decision_models/types.py b/s1a/decision_models/types.py index 9656e2f..dde0f69 100644 --- a/s1a/decision_models/types.py +++ b/s1a/decision_models/types.py @@ -10,6 +10,7 @@ Json = dict[str, Any] QuestionType = Literal["choice", "noul"] +PROVENANCE_KEYS = ("served_by", "url", "request_id", "server_timing") # the raw fields a step record keeps @dataclass(frozen=True) @@ -173,3 +174,8 @@ def noul(self, name: str) -> Noul: if not isinstance(answer, Noul): raise TypeError(f"{name!r} is a choice answer, not a noul") return answer + + @property + def provenance(self) -> Json: + """Where a served answer came from, for step records: the server's identity, URL and timings; empty in process.""" + return {key: self.raw[key] for key in PROVENANCE_KEYS if key in self.raw} diff --git a/s1a/tool/models.py b/s1a/tool/models.py index 1e76e33..4d881de 100644 --- a/s1a/tool/models.py +++ b/s1a/tool/models.py @@ -157,6 +157,8 @@ async def _decide(self) -> AssistantMessage: "plan": bool(state.plan), "blocked": sorted(state.blocked), "source": self.name, + "model": decision.model, + **decision.provenance, } ) state.blocked = set() # a block, the notices and the plan last one turn diff --git a/tests/test_tool_models.py b/tests/test_tool_models.py index 3ebe221..a8e002a 100644 --- a/tests/test_tool_models.py +++ b/tests/test_tool_models.py @@ -16,6 +16,9 @@ from s1a.decision_models import ( DecisionModel, JevModel, + Observation, + Question, + Reply, RandomModel, RuleModel, ScriptedTransport, @@ -188,7 +191,38 @@ async def test_stream_yields_the_act_call_as_one_chunk(self) -> None: self.assertEqual(chunks[0].finish_reason, "tool_calls") +SERVED_BY = {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4", "device": "mps", "source": "health"} + + +class ServedStub(DecisionModel): + """Answers the way the served model does: an identity in ``model``, the server's facts in ``raw``.""" + + name = "laya-served" + + @property + def model(self) -> str: + return "convaiinnovations/laya@55cf4c4" + + async def _decide(self, observation: Observation, questions: dict[str, Question]) -> Reply: + answers = {"pick": {"choice": "inc", "confidence": 0.8, "probabilities": {"inc": 0.9, "noop": 0.1}}} + raw = {"answers": answers, "served_by": SERVED_BY, "url": "http://127.0.0.1:8000", "server_timing": {}} + return Reply(answers, latency_ms=7, model=self.model, raw=raw) + + class TestOtherModels(IsolatedAsyncioTestCase): + async def test_a_tick_names_the_answering_model_and_where_a_served_answer_came_from(self) -> None: + state = EvalState() + await _model(CountingEnv(), state, ServedStub()).invoke([], tools=TOOLS) + tick = state.ticks[0] + self.assertEqual((tick["source"], tick["model"]), ("laya-served", "convaiinnovations/laya@55cf4c4")) + self.assertEqual( + (tick["served_by"], tick["url"], tick["server_timing"]), (SERVED_BY, "http://127.0.0.1:8000", {}) + ) + self.assertNotIn("answers", tick) + in_process = EvalState() + await _model(CountingEnv(), in_process, RandomModel(1)).invoke([], tools=TOOLS) + self.assertNotIn("served_by", in_process.ticks[0]) + async def test_random_picks_an_offered_key_with_a_uniform_distribution(self) -> None: state = EvalState() message = await _model(CountingEnv(), state, RandomModel(1)).invoke([], tools=TOOLS) From 0b3d23c395e8b2b45d41328313348dfc7d2c4ff9 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:36:16 +0800 Subject: [PATCH 07/14] [Docs] Document laya-served and how to run it (#20) decision-models.md and configuration.md list laya-served and its variables; served-laya.md gains a run section (start the worker once, use it from the CLI and MCP) and matches system1-omni#30 on /health freshness and GPU-only worker options. --- .env.example | 7 ++++ CHANGELOG.md | 6 +++ docs/api/laya-systemone.openapi.yaml | 2 +- docs/configuration.md | 5 +++ docs/decision-models.md | 14 +++++-- docs/served-laya.md | 56 +++++++++++++++++++++++++--- 6 files changed, 80 insertions(+), 10 deletions(-) diff --git a/.env.example b/.env.example index 58acac1..61ca83b 100644 --- a/.env.example +++ b/.env.example @@ -34,6 +34,13 @@ MODEL_NAME=google/gemini-2.5-flash # LAYA_MAX_LEN=1024 # LAYA_HEAD_MAX_LEN=512 # raise for choice questions with many options +# ---- Served Laya (behind --model laya-served; a system1-omni worker or laya-serve, no extra needed) ---- +# LAYA_SERVED_URL=http://127.0.0.1:8000 # the worker; :8080 for the omni-jev frontend +# LAYA_SERVED_MODEL=english +# LAYA_SERVED_API_KEY= # the worker's LAYA_API_KEY, when it sets one +# LAYA_SERVED_TIMEOUT_S=5 # per decision, the one retry included +# LAYA_SERVED_MAX_LEN=512 # the worker's LAYA_MAX_LEN + # ---- Cua-S1 Nano (the in-process option scorer behind --model cua; needs `uv sync --extra cua`) ---- # CUA_S1_CHECKPOINT=cua-ai/cua-s1-nano-0.1 # a Hugging Face id, or a local directory holding / # CUA_S1_SUBFOLDER=text # the text-only checkpoint; the window is 256 bytes of state diff --git a/CHANGELOG.md b/CHANGELOG.md index fd92ef6..4140577 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,12 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver ### Added +- `--model laya-served`: Laya served over HTTP by system1-omni's worker (or plain laya-serve), on every front that + takes `laya`, in `decide` and in MCP `decide`. Configured by `LAYA_SERVED_URL` and optional `LAYA_SERVED_*` + variables; no cloud key. Run records name the served checkpoint, revision and device. Design, API spec and the + run steps: `docs/served-laya.md`, `docs/api/`. +- Tool-front ticks keep the answering model as `model`, and a served model's `served_by`, `url`, `request_id` and + `server_timing`. - The MCP `decide` tool accepts `model="jev"|"laya"|"cua"`, defaulting to `jev`. Local backends use their optional extras and need no Jev API key. - `docs/benchmarks.md`: the Google Flights driver comparison rerun on 2026-09-23 from Poland, every arm three times on diff --git a/docs/api/laya-systemone.openapi.yaml b/docs/api/laya-systemone.openapi.yaml index cf4a65a..120baf1 100644 --- a/docs/api/laya-systemone.openapi.yaml +++ b/docs/api/laya-systemone.openapi.yaml @@ -396,7 +396,7 @@ components: ServedBy: type: object x-status: planned - description: What produced these answers, per response, so a record needs no separate /health call. + description: What produced these answers, per response, so a record needs no separate /health call and stays exact when laya moves a model to the CPU mid-run. required: [checkpoint, revision, device, weights_dtype] properties: checkpoint: { type: string } diff --git a/docs/configuration.md b/docs/configuration.md index 7c71622..99543cf 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -32,6 +32,11 @@ Variables can be exported in your shell or placed in a `.env` file at the root o | `LAYA_DEVICE` | `laya` model | `(library default)` | PyTorch device for Laya model evaluation; passes None so the library selects CUDA, MPS, or CPU. | | `LAYA_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya state representation; overrides checkpoint window only when set. | | `LAYA_HEAD_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya decision head options; overrides checkpoint window only when set. | +| `LAYA_SERVED_URL` | `laya-served` model | *(unset, required)* | Base URL of a served Laya: the system1-omni worker (`http://127.0.0.1:8000`), its `omni-jev` frontend (`:8080`) or plain laya-serve. | +| `LAYA_SERVED_MODEL` | `laya-served` model | `english` | Name of the served model to ask, one the server loaded. | +| `LAYA_SERVED_API_KEY` | `laya-served` model | *(unset)* | Bearer token, the server's `LAYA_API_KEY` when it sets one. | +| `LAYA_SERVED_TIMEOUT_S` | `laya-served` model | `5` | Deadline per decision in seconds, the one retry included. | +| `LAYA_SERVED_MAX_LEN` | `laya-served` model | `512` | The server's token window per question (its `LAYA_MAX_LEN`); a request that fills it raises. | | `CUA_S1_CHECKPOINT` | `cua` model | `cua-ai/cua-s1-nano-0.1` | Hugging Face checkpoint ID or local directory for Cua-S1 Nano option scorer. | | `CUA_S1_SUBFOLDER` | `cua` model | `text` | Subfolder within checkpoint directory containing text option scoring weights. | | `CUA_S1_DEVICE` | `cua` model | `auto` | PyTorch device used for Cua-S1 Nano evaluation (`auto`, `cpu`, `cuda`, or `mps`). | diff --git a/docs/decision-models.md b/docs/decision-models.md index 89b2e0d..62d9960 100644 --- a/docs/decision-models.md +++ b/docs/decision-models.md @@ -8,7 +8,8 @@ Every front that asks "which one" (the tool loop, the browser policy, the rails, talks only to `DecisionModel`; the wire client is private to `s1a/decision_models/`. A decision model is a classifier over options the caller enumerates. It reads a state and returns a distribution over the offered keys. Hugging Face writes "System 1 decision model" and TypeSafe "System One model". This repository uses the terms interchangeably. `--model` picks the backend: `jev` (TypeSafe Jev over HTTP), -`laya` (in process, behind `uv sync --extra laya`), `cua` (Cua-S1 Nano in process, behind `uv sync --extra cua`), +`laya` (in process, behind `uv sync --extra laya`), `laya-served` (Laya served over HTTP by system1-omni, see +[served-laya.md](served-laya.md)), `cua` (Cua-S1 Nano in process, behind `uv sync --extra cua`), `random` and `rule` (the tool front's baselines). `build_model(model_name, seed=, rule=)` builds one from the environment; `llm` names the chat model, which `build_model` does not build. @@ -44,11 +45,14 @@ shorthands; `warm()` and `close()` open and release the backend. |---|---|---|---| | `jev` | `JevModel(transport)` | `jev` | the request body every front sent before the layer existed, byte for byte; `from_env` picks TypeSafe or the OpenRouter proxy (see [configuration.md](configuration.md)) | | `laya` | `LayaModel(agent, model=)` | `laya` | one forward pass per call on a thread; `MODEL_SERVICE_CONFIG_ERROR` when `input_tokens` fills the window (Laya cuts the state silently; see `LAYA_MAX_LEN` and `LAYA_HEAD_MAX_LEN` in [configuration.md](configuration.md)); `ValueError` and `RuntimeError` from the library become `MODEL_CALL_FAILED` | +| `laya-served` | `ServedLayaModel(client, model=, max_len=)` | `laya-served` | one `POST /v1/systemone` per request to `LAYA_SERVED_URL`, body built with `laya_question()` as for `laya`; one retry on a dropped connection, 502 or 504, and after `Retry-After` on 503, all within `LAYA_SERVED_TIMEOUT_S`; the same window check as `laya`; `model` is the served checkpoint and revision, and `raw` keeps `served_by`, `url`, `request_id` and `server_timing` (see [served-laya.md](served-laya.md)) | | `cua` | `CuaS1Model(scorer, collator, model=, context_bytes=, option_bytes=)` | `cua` | Cua-S1 Nano, one `score_elements` pass per request on a thread; choice questions only, text only, deterministic; the context is header, state and rules; the checkpoint reads its first 256 bytes, and the first overflowing request logs one warning; `from_env` reads `CUA_S1_*` (see [configuration.md](configuration.md)) | | `random` | `RandomModel(seed)` | `random` | uniform over the offered keys, confidence 0, one seeded stream per episode; choice questions only | | `rule` | `RuleModel(name, rule)` | the rule's name | one-hot, confidence 1; a key outside the menu raises `RuntimeError`, a bug in the rule | `name` lands in every tick's `source` and in `Episode.policy`; the eval table's columns take their labels from it. +A tool-front tick also keeps `Decision.model` as `model` and, for a served model, `served_by`, `url`, `request_id` +and `server_timing` (`Decision.provenance`). TypeSafe Jev answers `--model jev`. One request holds a `state` and one or more questions over options the caller enumerates; the answer holds one option per question, a probability per option and a confidence, from one forward @@ -75,7 +79,9 @@ Three oracle tests in `tests/test_decision_models_jev.py` pin the tool, rail and `ScriptedTransport` fakes the wire under `JevModel` (the adapter's body building and payload reading run for real; `bodies` records every request). `ScriptedModel` fakes the interface for front tests that need no wire. -`FakeLayaAgent` in `tests/test_decision_models_laya.py` stands in for the library. Nothing patches `httpx`. +`FakeLayaAgent` in `tests/test_decision_models_laya.py` stands in for the library. The served model runs over an +`httpx.MockTransport` scripted in `tests/test_decision_models_served.py`, answering like the responses recorded from +real servers in `tests/data/served_laya/`. Nothing patches `httpx`. ## Adding a backend @@ -84,7 +90,9 @@ real; `bodies` records every request). `ScriptedModel` fakes the interface for f questions and returns a `Reply` whose `answers` are the backend's own dicts; `decide_many` validates them into a `Decision`. Keep any heavy import inside `from_env()`. 2. A `case` in `factory.build_model` and the name in `DECISION_MODEL_NAMES`, `tool/loop.py::MODEL_NAMES`, - `browser/browse.py::BROWSER_MODEL_NAMES`, `rails.RAIL_MODEL_NAMES` and `cli.DECIDE_MODEL_NAMES`. + `browser/browse.py::BROWSER_MODEL_NAMES`, `rails.RAIL_MODEL_NAMES`, `cli.DECIDE_MODEL_NAMES` and the + `mcp_server.decide` `Literal`, plus the `match` in `tool/loop.py` and `browser/browse.py` when the backend is + a decision model; `tests/test_served_laya_registration.py` finds a list that has `laya` without `laya-served`. 3. `tests/test_decision_models_.py` with `TestContract(DecisionModelContract, IsolatedAsyncioTestCase)` plus the backend's mapping tests; a fake for its SDK lives in that file. 4. An optional extra in `pyproject.toml` and an env block in `.env.example` when it needs a dependency. diff --git a/docs/served-laya.md b/docs/served-laya.md index aff156d..edba71f 100644 --- a/docs/served-laya.md +++ b/docs/served-laya.md @@ -23,7 +23,7 @@ gets built, and how the client works with both in the meantime. agent step ──► ServedLayaModel (s1a) ──HTTP──► [omni-jev frontend :8080] ──► Laya worker :8000 ──► laya (MPS/CPU) │ laya_question() forwards unchanged warmup before listen │ answer validation 502/504 if worker down /health: device, revision, compile - └─ run record: /health snapshot + client round trip + └─ run record: identity (served_by) + client round trip ``` ## 3. Interface (target; full spec in [api/laya-systemone.openapi.yaml](api/laya-systemone.openapi.yaml)) @@ -69,11 +69,20 @@ problem+json and fall back to `detail`; read identity from `served_by` when pres - **Selection.** `--model laya-served`, a new name so run records say served Laya, not Jev or in-process Laya. - **Configuration.** `LAYA_SERVED_URL` (required), `LAYA_SERVED_MODEL` (default `english`), `LAYA_SERVED_API_KEY` (optional), `LAYA_SERVED_TIMEOUT_S` (default 5, one deadline per decision, - retries included). + retries included), `LAYA_SERVED_MAX_LEN` (default 512, the server's window per question, for the + same full-window check as in-process Laya). - **Request.** Questions serialised with the existing `laya_question()`, which keeps Laya's own `noul` shape (a plain-string instruction). `score` is not sent until an agent needs it. -- **Identity.** Each response's `served_by` goes into the run record; until servers send it, `warm()` - reads `/health` once and records checkpoint, revision, device, dtypes and compile mode. +- **Identity.** Each response's `served_by` goes into the run record. Until servers send it, the client + reads `/health` at warm-up and again whenever its reading is older than 30 s (the worker reports the + live device, and laya moves a model to the CPU on a GPU out-of-memory error), records the reading's + time as `read_at`, and each decision takes the entry for the model that answered + (`models[routing.model]` on the system1-omni worker, else its top-level fields). Plain laya-serve + reports no checkpoint or revision, so the record falls back to the response's `routing.repo`. +- **Servers.** The system1-omni worker is the recommended server; plain laya-serve works with reduced + identity. On MPS the worker's fast setting is `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`. + Both apply on the GPU only: on the CPU, including after a fallback, the worker runs Laya's fp32 + model uncompiled. - **Errors → agent errors.** | outcome | handling | @@ -87,8 +96,10 @@ problem+json and fall back to `detail`; read identity from `served_by` when pres | deadline passed | fail with a timeout error naming the URL | - **Timing.** The record keeps the client round trip per decision and, when present, `Server-Timing`'s - `queue` and `infer`, so network, queueing and model time separate. -- **Tracing.** The client sends an `X-Request-Id` per decision and stores it with the step. + `queue` and `infer`, so network, queueing and model time separate. Today's servers send no + `Server-Timing`, so `server_timing` is `{}` until the worker adds it. +- **Tracing.** The client sends an `X-Request-Id` per decision, the same on its retry, and stores it + with the step. ## 6. Trade-offs @@ -110,3 +121,36 @@ problem+json and fall back to `detail`; read identity from `served_by` when pres - A second model family (ThinkFlowLab/system1-omni#9) reuses `/v1/systemone`: move `served_by` and the problem codes into a shared contract instead of the Laya spec. - `score` becomes useful to an agent: extend the client; the server already answers it. + +## 8. Run it + +Start the server once, from a system1-omni checkout, with its +[Apple Silicon recipe](https://github.com/cacheline999/system1-omni/blob/laya-apple-silicon/recipe/laya/apple-silicon.md) +(ThinkFlowLab/system1-omni#30, until it merges): + +```sh +LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ +LAYA_MODELS=english LAYA_REQUIRE_DEVICE=1 \ + .venv/bin/python src/models/laya/worker.py +``` + +The worker listens once it is warm, after about 40 s on an M1 Pro with these options; until then a +decision fails with "not up or still warming". On a Mac without MPS, or on Linux, drop the two +`LAYA_WORKER_*` options and set `LAYA_DEVICE=cpu`. The Rust frontend (`omni-jev`, port 8080) can sit in +front of it; point `LAYA_SERVED_URL` at whichever you call. + +Then, from this repository, with no extra installed: + +```sh +export LAYA_SERVED_URL=http://127.0.0.1:8000 +uv run s1a decide --model laya-served --state '{"ticket": "I was charged twice"}' \ + --option billing='a payment problem' --option technical='a product fault' --rules 'route the ticket' +uv run s1a run ticket_router --model laya-served --rethink off --seed 0 --episodes 1 +``` + +Over MCP, the `decide` tool takes `model="laya-served"`. `s1a-mcp` reads `LAYA_SERVED_URL` from its own +environment: set it in the host's MCP server entry, or in `.env` at the repository root. + +Each tool-front step records `source: laya-served`, `model` as `@` and +`served_by` with the device, dtypes, compile mode and the time of the `/health` reading it came from. +Against plain laya-serve, `served_by` has the checkpoint only. From fb682ed6226307c1c59647207db06a170140e589 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:48:40 +0800 Subject: [PATCH 08/14] [Feat] Say when a served Laya is not up yet (#20) A refused connection after the retry now says the worker listens only once warm and points to system1-omni's recipe; a first /health read that fails no longer logs about a previous reading that does not exist. The --model help of every front names laya-served. --- s1a/browser/__init__.py | 2 +- s1a/browser/browse.py | 2 +- s1a/cli.py | 3 ++- s1a/decision_models/served.py | 7 ++++++- s1a/tool/series.py | 2 +- tests/test_decision_models_served.py | 4 +++- 6 files changed, 14 insertions(+), 6 deletions(-) diff --git a/s1a/browser/__init__.py b/s1a/browser/__init__.py index c45f07e..6a49338 100644 --- a/s1a/browser/__init__.py +++ b/s1a/browser/__init__.py @@ -3,7 +3,7 @@ # Modifications Copyright 2026 ThinkFlowLab # SPDX-License-Identifier: Apache-2.0 -"""The browser policy: a decision model (Jev over HTTP, Laya or Cua-S1 in process) in the browser subagent's model slot. +"""The browser policy: a decision model (Jev or served Laya over HTTP, Laya or Cua-S1 in process) in the browser subagent's model slot. The observe, decide, act tick follows the design of browser-use/jev-ultrafast (MIT). """ diff --git a/s1a/browser/browse.py b/s1a/browser/browse.py index db7150a..17ec087 100644 --- a/s1a/browser/browse.py +++ b/s1a/browser/browse.py @@ -225,7 +225,7 @@ def parser(spec: BrowserAgentSpec) -> argparse.ArgumentParser: "--model", choices=BROWSER_MODEL_NAMES, required=True, - help="who decides each browser step: jev (over HTTP), laya or cua (in process), or llm (the chat model in MODEL_NAME)", + help="who decides each browser step: jev or laya-served (over HTTP), laya or cua (in process), or llm (the chat model in MODEL_NAME)", ) build.add_argument( "--goal", default=spec.goal, required=spec.goal is None, help="the task; the spec's goal when it has one" diff --git a/s1a/cli.py b/s1a/cli.py index 0169d18..c5d9793 100644 --- a/s1a/cli.py +++ b/s1a/cli.py @@ -38,7 +38,8 @@ def parser() -> argparse.ArgumentParser: run.add_argument("agent", help=f"one of: {', '.join(agents.names())}") run.add_argument("flags", nargs=argparse.REMAINDER) decide = commands.add_parser( - "decide", help="one choice question to a decision model: jev over HTTP, or laya and cua in process" + "decide", + help="one choice question to a decision model: jev or laya-served over HTTP, or laya and cua in process", ) decide.add_argument("--state", required=True, help="a JSON object, or @path to a file holding one") decide.add_argument( diff --git a/s1a/decision_models/served.py b/s1a/decision_models/served.py index 09b4803..21bc0ae 100644 --- a/s1a/decision_models/served.py +++ b/s1a/decision_models/served.py @@ -151,6 +151,10 @@ async def decide(self, body: Json, request_id: str) -> tuple[Json, dict[str, str if not retried: retried = True continue + if isinstance(exc, httpx.ConnectError): + raise build_error( + StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"no served Laya at {self.url}: {_NOT_UP}" + ) from exc raise build_error( StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"served Laya unreachable at {self.url}: {exc}" ) from exc @@ -274,7 +278,8 @@ async def _read_health(self, *, strict: bool) -> None: except Exception: if strict: raise - logger.warning("[laya-served] kept the previous /health reading from %s", self._health_read_at) + if self._health_read_at is not None: # with no reading yet, the decision itself reports the failure + logger.warning("[laya-served] kept the previous /health reading from %s", self._health_read_at) return if health: self._health = health diff --git a/s1a/tool/series.py b/s1a/tool/series.py index 6f6166b..1694e53 100644 --- a/s1a/tool/series.py +++ b/s1a/tool/series.py @@ -22,7 +22,7 @@ def parser(spec: ToolAgentSpec) -> argparse.ArgumentParser: "--model", choices=MODEL_NAMES, required=True, - help="who decides: jev (over HTTP), laya or cua (in process), llm (the chat model in MODEL_NAME), random, or rule (the agent's baseline)", + help="who decides: jev or laya-served (over HTTP), laya or cua (in process), llm (the chat model in MODEL_NAME), random, or rule (the agent's baseline)", ) build.add_argument( "--rethink", diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py index d60feb1..271c91c 100644 --- a/tests/test_decision_models_served.py +++ b/tests/test_decision_models_served.py @@ -217,8 +217,10 @@ async def test_one_request_id_per_decision_kept_across_the_retry(self) -> None: async def test_a_second_dropped_connection_fails(self) -> None: server = Server(script=[httpx.ConnectError("refused"), httpx.ConnectError("refused")]) - await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "unreachable at http://laya.test") + await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "no served Laya at http://laya.test") self.assertEqual(len(server.decisions), 2) + dropped = Server(script=[httpx.ReadError("reset"), httpx.ReadError("reset")]) + await self.assert_fails(dropped, StatusCode.MODEL_CALL_FAILED, "unreachable at http://laya.test") async def test_502_and_504_from_the_frontend_are_retried_once(self) -> None: for status in (502, 504): From 073f2d465b3f5c27f1675f0bdf34e9370e1e2240 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:48:40 +0800 Subject: [PATCH 09/14] [Docs] Ticket router: in-process against served Laya (#20) 90 decisions per configuration on an M1 Pro: in-process Laya, the system1-omni worker (compile + fp16) and the same worker behind omni-jev routed every (seed, ticket) pair the same and scored 63/90 each; p50 101, 77 and 82 ms. compare_served.py builds the table from the job dirs and keeps every decision in served_laya_records.json. --- docs/served-laya.md | 3 + evals/ticket_router/SERVED_LAYA.md | 59 ++++ evals/ticket_router/compare_served.py | 129 +++++++++ evals/ticket_router/served_laya_records.json | 278 +++++++++++++++++++ 4 files changed, 469 insertions(+) create mode 100644 evals/ticket_router/SERVED_LAYA.md create mode 100644 evals/ticket_router/compare_served.py create mode 100644 evals/ticket_router/served_laya_records.json diff --git a/docs/served-laya.md b/docs/served-laya.md index edba71f..093a43d 100644 --- a/docs/served-laya.md +++ b/docs/served-laya.md @@ -154,3 +154,6 @@ environment: set it in the host's MCP server entry, or in `.env` at the reposito Each tool-front step records `source: laya-served`, `model` as `@` and `served_by` with the device, dtypes, compile mode and the time of the `/health` reading it came from. Against plain laya-serve, `served_by` has the checkpoint only. + +In-process and served Laya routed all 90 ticket-router decisions the same on an M1 Pro; the numbers are in +[evals/ticket_router/SERVED_LAYA.md](../evals/ticket_router/SERVED_LAYA.md). diff --git a/evals/ticket_router/SERVED_LAYA.md b/evals/ticket_router/SERVED_LAYA.md new file mode 100644 index 0000000..d067b92 --- /dev/null +++ b/evals/ticket_router/SERVED_LAYA.md @@ -0,0 +1,59 @@ +# Ticket router: in-process Laya against served Laya + +Date: 2026-09-30. The plan below was fixed before the runs. + +## Setup + +| | how | Laya | +|---|---|---| +| C-in | `--model laya`, `LAYA_DEVICE=mps` | in process: fp32 weights, not compiled | +| C-direct | `--model laya-served`, `LAYA_SERVED_URL=http://127.0.0.1:8000` | system1-omni worker, `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, MPS | +| C-front | `--model laya-served`, `LAYA_SERVED_URL=http://127.0.0.1:8080` | the same worker process behind `omni-jev` | + +- Checkpoint `convaiinnovations/laya` at `55cf4c4`, laya 0.3.20, torch 2.14.0. +- Hardware: M1 Pro (16 GB), macOS 26.1, on AC power. +- system1-omni at `9ad04e3`, the head of ThinkFlowLab/system1-omni#30 on 2026-09-30. +- system1-agents on branch `served-laya` at `0b3d23c`. The commits after it change only error wording, help text and this folder. +- Each configuration ran `s1a run ticket_router --model --rethink off --seed 0 --episodes 3`. Seeds 0, 1 and 2 shuffle the same 30 tickets, giving 90 decisions per configuration. +- Order: the worker started once and was ready after 37 s. C-direct and C-front ran against it. The worker was then stopped and C-in ran, so no two models shared the GPU. +- The one-minute load average was 5.3–6.0 at the start of each configuration, from other work on the machine. + +## Results + +| config | correct | p50 ms | p95 ms | episodes s | model recorded | served on | +|---|---:|---:|---:|---:|---|---| +| C-in | 63/90 | 101 | 125 | 10.2 | `laya-rl-agent` | in process | +| C-direct | 63/90 | 77 | 94 | 7.8 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled | +| C-front | 63/90 | 82 | 87 | 8.8 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled, via `omni-jev` | + +- Latency is per decision: the client round trip for the served configurations, the forward pass on a thread for C-in. +- Episode time is the sum over the three episodes. It leaves out process start and model load. + +Routing agreement: +- All three configurations routed every one of the 90 (seed, ticket) pairs the same. +- C-in and C-direct differed by at most 0.001 in any probability. +- C-direct and C-front gave identical probabilities. +- The closest call in C-in had 0.0126 between its top two queues. + +Both plan expectations held: +- C-direct and C-front agree everywhere. +- No ticket flips between the fp32 in-process model and the fp16 worker, so there are no differences to list. + +C-in's lower speed comes from how Laya ran, fp32 and not compiled, against the worker's compiled fp16 model. It says nothing about HTTP cost. The frontend added about 5 ms at p50. + +In-process runs record `laya-rl-agent`, the name laya reports. Served runs record the checkpoint and revision, plus `served_by` (device, dtypes, compile mode, time of the `/health` reading) in every tick. + +## Reproduce + +Start the worker and the frontend as in [docs/served-laya.md](../../docs/served-laya.md#8-run-it). Then, from this repository: + +```sh +LAYA_SERVED_URL=http://127.0.0.1:8000 uv run s1a run ticket_router --model laya-served --rethink off --seed 0 --episodes 3 +LAYA_SERVED_URL=http://127.0.0.1:8080 uv run s1a run ticket_router --model laya-served --rethink off --seed 0 --episodes 3 +# stop the worker, then +LAYA_DEVICE=mps uv run s1a run ticket_router --model laya --rethink off --seed 0 --episodes 3 +uv run python evals/ticket_router/compare_served.py C-in= C-direct= C-front= \ + --json evals/ticket_router/served_laya_records.json +``` + +Job directories stay local, as for the other results here. [served_laya_records.json](served_laya_records.json) keeps every decision of the three runs: ticket, expected and predicted queue, probabilities, ms and the recorded model. diff --git a/evals/ticket_router/compare_served.py b/evals/ticket_router/compare_served.py new file mode 100644 index 0000000..5d4b456 --- /dev/null +++ b/evals/ticket_router/compare_served.py @@ -0,0 +1,129 @@ +# coding: utf-8 +"""Compare ticket-router jobs that ran the same seeds with different decision models (in-process vs served Laya). + + uv run python evals/ticket_router/compare_served.py C-in= C-direct= C-front= \ + [--json records.json] + +Prints one markdown table (correct routes, per-decision latency p50/p95, total episode time, the answering model and +where it ran), how many (seed, ticket) pairs each two configurations routed the same, and every pair routed +differently with each configuration's top two probabilities. A job dir is what ``s1a run ticket_router`` prints as +``job_dir``; episode time excludes process start and model load. ``--json`` also writes every decision (seed, +ticket, expected and predicted queue, probabilities, ms, model, served_by) per configuration, since job dirs stay local. +""" + +from __future__ import annotations + +import json +import statistics +import sys +from dataclasses import dataclass +from pathlib import Path +from typing import Any + + +@dataclass +class Trial: + seed: int + router: dict[str, Any] # the episode's ticket_router extra: ticket_ids, routes, correct, total + ticks: list[dict[str, Any]] + elapsed_s: float + + +def load(job_dir: Path) -> list[Trial]: + trials = [] + for trial_dir in sorted(path.parent for path in job_dir.glob("*/result.json")): + episode = json.loads((trial_dir / "agent" / "episode.json").read_text()) + result = json.loads((trial_dir / "result.json").read_text()) + router = episode["extra"]["ticket_router"] + elapsed = float(result["agent_result"]["metadata"]["elapsed_s"]) + trials.append(Trial(router["seed"], router, episode["decisions"], elapsed)) + return sorted(trials, key=lambda trial: trial.seed) + + +def percentile(values: list[float], q: float) -> float: + ordered = sorted(values) + return ordered[min(len(ordered) - 1, round(q * (len(ordered) - 1)))] + + +def routes(trials: list[Trial]) -> dict[tuple[int, str], tuple[str, dict[str, float]]]: + """(seed, ticket id) -> (predicted queue, probabilities); ticks follow the batch's ticket order.""" + return { + (trial.seed, ticket_id): (tick["key"], tick["probabilities"]) + for trial in trials + for ticket_id, tick in zip(trial.router["ticket_ids"], trial.ticks, strict=True) + } + + +def row(name: str, trials: list[Trial]) -> str: + ticks = [tick for trial in trials for tick in trial.ticks] + ms = [tick["ms"] for tick in ticks] + correct = sum(trial.router["correct"] for trial in trials) + total = sum(trial.router["total"] for trial in trials) + served_by = ticks[0].get("served_by") or {} + where = "in process" + if served_by: + dtype = (served_by.get("weights_dtype") or "").removeprefix("torch.") + parts = [str(part) for part in (served_by.get("device"), dtype, served_by.get("compile")) if part] + where = f"{' '.join(parts) or 'device not reported'} via {ticks[0]['url']}" + return ( + f"| {name} | {correct}/{total} | {statistics.median(ms):.0f} | {percentile(ms, 0.95):.0f} | " + f"{sum(trial.elapsed_s for trial in trials):.1f} | {ticks[0].get('model') or ticks[0]['source']} | {where} |" + ) + + +def records(trials: list[Trial]) -> list[dict[str, Any]]: + """One row per decision; ``served_by`` and ``url`` only on the first row and where they change.""" + rows, last = [], None + for trial in trials: + for route, tick in zip(trial.router["routes"], trial.ticks, strict=True): + row = {"seed": trial.seed, "ticket": route["id"], "expected": route["expected"]} + row |= {k: tick.get(k) for k in ("key", "probabilities", "ms", "input_tokens", "model")} + source = {k: tick[k] for k in ("served_by", "url") if k in tick} + if source and source != last: + row |= source + rows.append(row) + last = source + return rows + + +def main(arguments: list[str]) -> None: + json_path = None + if "--json" in arguments: + at = arguments.index("--json") + json_path, arguments = Path(arguments[at + 1]), arguments[:at] + arguments[at + 2 :] + configs = {name: load(Path(path)) for name, _, path in (argument.partition("=") for argument in arguments)} + if json_path is not None: + blocks = [ + json.dumps(name) + ": [\n" + ",\n".join(json.dumps(row) for row in records(trials)) + "\n]" + for name, trials in configs.items() + ] + json_path.write_text("{\n" + ",\n".join(blocks) + "\n}\n") # one decision per line, for diffs + print("| config | correct | p50 ms | p95 ms | episodes s | model | served on |") + print("|---|---:|---:|---:|---:|---|---|") + for name, trials in configs.items(): + print(row(name, trials)) + picked = {name: routes(trials) for name, trials in configs.items()} + names = list(picked) + keys = sorted(set.intersection(*(set(found) for found in picked.values()))) + print() + for i, a in enumerate(names): + for b in names[i + 1 :]: + same = sum(picked[a][key][0] == picked[b][key][0] for key in keys) + print(f"- {a} and {b} route {same} of {len(keys)} (seed, ticket) pairs the same") + differing = [key for key in keys if len({picked[name][key][0] for name in names}) > 1] + if not differing: + return + print() + print("| seed | ticket | " + " | ".join(names) + " |") + print("|---|---|" + "---|" * len(names)) + for seed, ticket in differing: + cells = [] + for name in names: + key, probabilities = picked[name][(seed, ticket)] + top = sorted(probabilities.items(), key=lambda item: -item[1])[:2] + cells.append(f"{key} ({', '.join(f'{k} {p:.3f}' for k, p in top)})") + print(f"| {seed} | {ticket} | " + " | ".join(cells) + " |") + + +if __name__ == "__main__": + main(sys.argv[1:]) diff --git a/evals/ticket_router/served_laya_records.json b/evals/ticket_router/served_laya_records.json new file mode 100644 index 0000000..6420da4 --- /dev/null +++ b/evals/ticket_router/served_laya_records.json @@ -0,0 +1,278 @@ +{ +"C-in": [ +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 577, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 122, "input_tokens": 274, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 115, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 117, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 118, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 120, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 103, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 111, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 125, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 106, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 101, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 109, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 102, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 100, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 106, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 115, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 106, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 101, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 105, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 105, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 103, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 110, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 104, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 106, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 114, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 110, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 111, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 109, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 104, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 105, "input_tokens": 252, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 132, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 115, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 114, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 115, "input_tokens": 274, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 107, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 108, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 110, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 106, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 101, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 95, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 106, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 109, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 107, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 111, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 99, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 101, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 86, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 84, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 87, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 122, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 139, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 95, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 91, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 93, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 86, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 89, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 80, "input_tokens": 252, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 84, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 90, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 110, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 147, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 87, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 88, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 89, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 83, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 85, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 88, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 83, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 89, "input_tokens": 274, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 91, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 85, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 91, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 92, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 91, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 92, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 89, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 88, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 82, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 85, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 78, "input_tokens": 252, "model": "laya-rl-agent"} +], +"C-direct": [ +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 283, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compile": "on", "source": "health", "read_at": "2026-09-30T14:45:44+00:00"}, "url": "http://127.0.0.1:8000"}, +{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 94, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 86, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 93, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 95, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 100, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 76, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 92, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 91, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 77, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 75, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 91, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 79, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 83, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 94, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 90, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 76, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 78, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 75, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 79, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 89, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 77, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 74, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 75, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 76, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 90, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 76, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 81, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 82, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 76, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 76, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 74, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 78, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 75, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 74, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 77, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 75, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 75, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 73, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 76, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 76, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 75, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 80, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 78, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 74, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 76, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 80, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 75, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 78, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 69, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 72, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 77, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 75, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 75, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 76, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 79, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 80, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 76, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 77, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 75, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 75, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 76, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 74, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 78, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 80, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 75, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 76, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 77, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 73, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 75, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 77, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 67, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} +], +"C-front": [ +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 980, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compile": "on", "source": "health", "read_at": "2026-09-30T14:45:56+00:00"}, "url": "http://127.0.0.1:8080"}, +{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 83, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 86, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 86, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 82, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 82, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 81, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 87, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 79, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 78, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 82, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 80, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 85, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 87, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 80, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 81, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 83, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 79, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 78, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 77, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 99, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 86, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 84, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 84, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 80, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 84, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 80, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 83, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 84, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 84, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 85, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 78, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 80, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 83, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 84, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 88, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 87, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 85, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 87, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 78, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 84, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 84, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 81, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 85, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 81, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 82, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 79, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 83, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 82, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 83, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 84, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 80, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 77, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 83, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 84, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 83, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 83, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 76, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} +] +} From 9811906e12b5cc97df6b3fc8fe0dbec272511118 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 09:07:02 +0800 Subject: [PATCH 10/14] [Docs] List the dev extra's spec-check packages in CONTRIBUTING (#20) --- CONTRIBUTING.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 9af4d53..64e9272 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -71,7 +71,7 @@ Everything outside `openjiuwen` is an extra. An agent whose extra is missing say | `report` | pillow, playwright | `python -m evals.replay`, the showcase pages and GIFs; `--gif` also needs `uv run playwright install chromium` | | `laya` | laya (torch, transformers) | `--model laya` on every agent and on `decide` and `probe`: Laya in process, no Jev key; the checkpoint downloads into the Hugging Face cache (`HF_HOME`) on first use | | `cua` | cua-s1 (torch), huggingface-hub | `--model cua` on tool and browser agents and on `decide` and `probe`: Cua-S1 Nano in process; the 3 MB checkpoint downloads into the Hugging Face cache (`HF_HOME`) on first use | -| `dev` | pytest, pytest-asyncio, ruff, ty | the test suite, `scripts/smoke.sh` and the lint and type checks | +| `dev` | pytest, pytest-asyncio, ruff, ty, jsonschema, pyyaml, referencing | the test suite, `scripts/smoke.sh` and the lint and type checks; the last three check the served-Laya fixtures against `docs/api/` | `uv sync --all-extras` installs all seven. The CLI runs from a checkout; a wheel install (`uv tool install`, `pip install`) is unsupported, because the data folders (`evals/2048`, `evals/millionaire`, `evals/labelled`) sit From ff39eded1cbfd254a61c52a4f4903dde33c79f1a Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 10:14:29 +0800 Subject: [PATCH 11/14] [Feat] Follow system1-omni#30's final worker (#20) The worker now starts as `python -m frontend.laya_mps --compile --weights fp16` and its /health reports `compile.enabled` instead of `compile.mode`. served_by records `compiled` (true/false); both specs, the run section and the not-up message follow. Worker fixtures were recorded again at 3d6cb57; the other responses came out byte for byte the same. The ticket-router comparison was rerun at 3d6cb57: 63/90 in every configuration, all 90 decisions routed the same, p50 88 ms in process and 72 ms served, direct and through omni-jev. --- docs/api/laya-systemone.current.openapi.yaml | 10 +- docs/api/laya-systemone.openapi.yaml | 10 +- docs/served-laya.md | 15 +- evals/ticket_router/SERVED_LAYA.md | 22 +- evals/ticket_router/compare_served.py | 6 +- evals/ticket_router/served_laya_records.json | 528 +++++++++---------- s1a/decision_models/served.py | 4 +- tests/data/served_laya/README.md | 4 +- tests/data/served_laya/frontend.health.json | 6 +- tests/data/served_laya/worker.health.json | 6 +- tests/test_decision_models_served.py | 4 +- 11 files changed, 310 insertions(+), 305 deletions(-) diff --git a/docs/api/laya-systemone.current.openapi.yaml b/docs/api/laya-systemone.current.openapi.yaml index 2df7db4..8776df9 100644 --- a/docs/api/laya-systemone.current.openapi.yaml +++ b/docs/api/laya-systemone.current.openapi.yaml @@ -5,7 +5,7 @@ info: summary: Typed decisions from a Laya checkpoint over HTTP. description: | The interface a client such as system1-agents uses to get decisions from Laya served by - system1-omni: the Laya worker (`src/models/laya/worker.py`, laya-serve 0.3.20 underneath), either + system1-omni: the Laya worker (`python -m frontend.laya_mps`, laya-serve 0.3.20 underneath), either directly or behind the Rust frontend (`omni-jev`), which forwards requests and responses unchanged. This document describes the behaviour of laya-serve 0.3.20, the worker and the frontend as they @@ -201,7 +201,7 @@ paths: checkpoint: convaiinnovations/laya revision: 55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851 warmup_ms: 34426.6 - compile: { mode: 'on', graphs_at_ready: 3, graphs_now: 3, recompiled_after_ready: false } + compile: { enabled: true, graphs_at_ready: 3, graphs_now: 3, recompiled_after_ready: false } '502': description: Frontend only. The worker could not be reached. content: @@ -426,7 +426,7 @@ components: - $ref: '#/components/schemas/ModelHealth' - type: object required: [status, ready, loaded, models, compile] - description: Top-level model fields describe LAYA_WORKER_MODEL; `models` has every loaded model. + description: Top-level model fields describe the `--model` the worker serves; `models` has every loaded model. properties: status: { const: ok } ready: { const: true } @@ -436,9 +436,9 @@ components: additionalProperties: { $ref: '#/components/schemas/ModelHealth' } compile: type: object - required: [mode] + required: [enabled] properties: - mode: { enum: ['off', 'on'] } + enabled: { type: boolean, description: "Started with --compile." } graphs_at_ready: { type: integer } graphs_now: { type: integer } recompiled_after_ready: { type: boolean } diff --git a/docs/api/laya-systemone.openapi.yaml b/docs/api/laya-systemone.openapi.yaml index 120baf1..31b3569 100644 --- a/docs/api/laya-systemone.openapi.yaml +++ b/docs/api/laya-systemone.openapi.yaml @@ -5,7 +5,7 @@ info: summary: Target interface for typed decisions from Laya over HTTP. description: | Design of the interface between decision clients (system1-agents) and Laya served by - system1-omni: the Laya worker (`src/models/laya/worker.py`), optionally behind the Rust + system1-omni: the Laya worker (`python -m frontend.laya_mps`), optionally behind the Rust frontend (`omni-jev`). Every operation, header and field carries `x-status`: `implemented` where the server already @@ -88,7 +88,7 @@ paths: revision: 55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851 device: mps weights_dtype: float16 - compile: 'on' + compiled: true answers: department: type: choice @@ -403,7 +403,7 @@ components: revision: { type: [string, 'null'], description: Null for a local checkpoint. } device: { type: string } weights_dtype: { enum: [float32, float16, bfloat16] } - compile: { enum: ['off', 'on'] } + compiled: { type: boolean, description: The answer came from the worker's compiled model. } Answer: oneOf: - $ref: '#/components/schemas/ChoiceAnswer' @@ -492,9 +492,9 @@ components: models: { type: object, additionalProperties: { $ref: '#/components/schemas/ModelHealth' } } compile: type: object - required: [mode] + required: [enabled] properties: - mode: { enum: ['off', 'on'] } + enabled: { type: boolean } graphs_at_ready: { type: integer } graphs_now: { type: integer } recompiled_after_ready: { type: boolean } diff --git a/docs/served-laya.md b/docs/served-laya.md index 093a43d..3c6253f 100644 --- a/docs/served-laya.md +++ b/docs/served-laya.md @@ -80,7 +80,7 @@ problem+json and fall back to `detail`; read identity from `served_by` when pres (`models[routing.model]` on the system1-omni worker, else its top-level fields). Plain laya-serve reports no checkpoint or revision, so the record falls back to the response's `routing.repo`. - **Servers.** The system1-omni worker is the recommended server; plain laya-serve works with reduced - identity. On MPS the worker's fast setting is `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`. + identity. On MPS the worker's fast setting is `--compile --weights fp16`. Both apply on the GPU only: on the CPU, including after a fallback, the worker runs Laya's fp32 model uncompiled. - **Errors → agent errors.** @@ -129,14 +129,14 @@ Start the server once, from a system1-omni checkout, with its (ThinkFlowLab/system1-omni#30, until it merges): ```sh -LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ -LAYA_MODELS=english LAYA_REQUIRE_DEVICE=1 \ - .venv/bin/python src/models/laya/worker.py +PYTHONPATH=src .venv/bin/python -m frontend.laya_mps --device mps --model english --require-device \ + --compile --weights fp16 --port 8000 ``` The worker listens once it is warm, after about 40 s on an M1 Pro with these options; until then a -decision fails with "not up or still warming". On a Mac without MPS, or on Linux, drop the two -`LAYA_WORKER_*` options and set `LAYA_DEVICE=cpu`. The Rust frontend (`omni-jev`, port 8080) can sit in +decision fails with "not up or still warming". On a Mac without MPS, or on Linux, drop `--compile` +and `--weights fp16` and use `--device cpu`. laya-serve's `LAYA_API_KEY` still turns on bearer auth; set the +same value in `LAYA_SERVED_API_KEY`. The Rust frontend (`omni-jev`, port 8080) can sit in front of it; point `LAYA_SERVED_URL` at whichever you call. Then, from this repository, with no extra installed: @@ -152,7 +152,8 @@ Over MCP, the `decide` tool takes `model="laya-served"`. `s1a-mcp` reads `LAYA_S environment: set it in the host's MCP server entry, or in `.env` at the repository root. Each tool-front step records `source: laya-served`, `model` as `@` and -`served_by` with the device, dtypes, compile mode and the time of the `/health` reading it came from. +`served_by` with the device, dtypes, whether the model was compiled and the time of the `/health` reading it +came from. Against plain laya-serve, `served_by` has the checkpoint only. In-process and served Laya routed all 90 ticket-router decisions the same on an M1 Pro; the numbers are in diff --git a/evals/ticket_router/SERVED_LAYA.md b/evals/ticket_router/SERVED_LAYA.md index d067b92..95e6099 100644 --- a/evals/ticket_router/SERVED_LAYA.md +++ b/evals/ticket_router/SERVED_LAYA.md @@ -1,30 +1,30 @@ # Ticket router: in-process Laya against served Laya -Date: 2026-09-30. The plan below was fixed before the runs. +Date: 2026-10-01. The plan below was fixed before the runs. ## Setup | | how | Laya | |---|---|---| | C-in | `--model laya`, `LAYA_DEVICE=mps` | in process: fp32 weights, not compiled | -| C-direct | `--model laya-served`, `LAYA_SERVED_URL=http://127.0.0.1:8000` | system1-omni worker, `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, MPS | +| C-direct | `--model laya-served`, `LAYA_SERVED_URL=http://127.0.0.1:8000` | system1-omni worker, `--compile --weights fp16`, MPS | | C-front | `--model laya-served`, `LAYA_SERVED_URL=http://127.0.0.1:8080` | the same worker process behind `omni-jev` | - Checkpoint `convaiinnovations/laya` at `55cf4c4`, laya 0.3.20, torch 2.14.0. - Hardware: M1 Pro (16 GB), macOS 26.1, on AC power. -- system1-omni at `9ad04e3`, the head of ThinkFlowLab/system1-omni#30 on 2026-09-30. -- system1-agents on branch `served-laya` at `0b3d23c`. The commits after it change only error wording, help text and this folder. +- system1-omni at `3d6cb57`, the head of ThinkFlowLab/system1-omni#30 on 2026-10-01. +- system1-agents on branch `served-laya`, the commit that adds this file. - Each configuration ran `s1a run ticket_router --model --rethink off --seed 0 --episodes 3`. Seeds 0, 1 and 2 shuffle the same 30 tickets, giving 90 decisions per configuration. -- Order: the worker started once and was ready after 37 s. C-direct and C-front ran against it. The worker was then stopped and C-in ran, so no two models shared the GPU. -- The one-minute load average was 5.3–6.0 at the start of each configuration, from other work on the machine. +- Order: the worker started once and was ready after 38 s. C-direct and C-front ran against it. The worker was then stopped and C-in ran, so no two models shared the GPU. +- The one-minute load average was 7.6–8.1 at the start of each configuration, from other work on the machine. ## Results | config | correct | p50 ms | p95 ms | episodes s | model recorded | served on | |---|---:|---:|---:|---:|---|---| -| C-in | 63/90 | 101 | 125 | 10.2 | `laya-rl-agent` | in process | -| C-direct | 63/90 | 77 | 94 | 7.8 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled | -| C-front | 63/90 | 82 | 87 | 8.8 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled, via `omni-jev` | +| C-in | 63/90 | 88 | 101 | 9.2 | `laya-rl-agent` | in process | +| C-direct | 63/90 | 72 | 88 | 7.5 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled | +| C-front | 63/90 | 72 | 76 | 8.3 | `convaiinnovations/laya@55cf4c4ebb4e` | mps, float16, compiled, via `omni-jev` | - Latency is per decision: the client round trip for the served configurations, the forward pass on a thread for C-in. - Episode time is the sum over the three episodes. It leaves out process start and model load. @@ -39,9 +39,9 @@ Both plan expectations held: - C-direct and C-front agree everywhere. - No ticket flips between the fp32 in-process model and the fp16 worker, so there are no differences to list. -C-in's lower speed comes from how Laya ran, fp32 and not compiled, against the worker's compiled fp16 model. It says nothing about HTTP cost. The frontend added about 5 ms at p50. +C-in's lower speed comes from how Laya ran, fp32 and not compiled, against the worker's compiled fp16 model. It says nothing about HTTP cost. Direct and through the frontend had the same p50, 72 ms. -In-process runs record `laya-rl-agent`, the name laya reports. Served runs record the checkpoint and revision, plus `served_by` (device, dtypes, compile mode, time of the `/health` reading) in every tick. +In-process runs record `laya-rl-agent`, the name laya reports. Served runs record the checkpoint and revision, plus `served_by` (device, dtypes, whether it was compiled, time of the `/health` reading) in every tick. ## Reproduce diff --git a/evals/ticket_router/compare_served.py b/evals/ticket_router/compare_served.py index 5d4b456..ef369e0 100644 --- a/evals/ticket_router/compare_served.py +++ b/evals/ticket_router/compare_served.py @@ -63,7 +63,11 @@ def row(name: str, trials: list[Trial]) -> str: where = "in process" if served_by: dtype = (served_by.get("weights_dtype") or "").removeprefix("torch.") - parts = [str(part) for part in (served_by.get("device"), dtype, served_by.get("compile")) if part] + parts = [ + str(part) + for part in (served_by.get("device"), dtype, "compiled" if served_by.get("compiled") else "") + if part + ] where = f"{' '.join(parts) or 'device not reported'} via {ticks[0]['url']}" return ( f"| {name} | {correct}/{total} | {statistics.median(ms):.0f} | {percentile(ms, 0.95):.0f} | " diff --git a/evals/ticket_router/served_laya_records.json b/evals/ticket_router/served_laya_records.json index 6420da4..61f5f0f 100644 --- a/evals/ticket_router/served_laya_records.json +++ b/evals/ticket_router/served_laya_records.json @@ -1,278 +1,278 @@ { "C-in": [ -{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 577, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 122, "input_tokens": 274, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 115, "input_tokens": 262, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 117, "input_tokens": 281, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 118, "input_tokens": 279, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 120, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 103, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 111, "input_tokens": 259, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 125, "input_tokens": 265, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 106, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 101, "input_tokens": 262, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 109, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 102, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 100, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 106, "input_tokens": 258, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 115, "input_tokens": 280, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 106, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 101, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 105, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 105, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 103, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 110, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 104, "input_tokens": 279, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 106, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 114, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 110, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 111, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 109, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 104, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 105, "input_tokens": 252, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 132, "input_tokens": 259, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 115, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 114, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 115, "input_tokens": 274, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 107, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 108, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 110, "input_tokens": 279, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 106, "input_tokens": 265, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 101, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 95, "input_tokens": 262, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 106, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 109, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 107, "input_tokens": 281, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 111, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 99, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 101, "input_tokens": 258, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 86, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 84, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 87, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 122, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 139, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 95, "input_tokens": 279, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 91, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 93, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 701, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 101, "input_tokens": 274, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 94, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 102, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 99, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 96, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 90, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 96, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 94, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 84, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 94, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 87, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 88, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 92, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 97, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 94, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 86, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 87, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 88, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 88, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 94, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 87, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 85, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 96, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 85, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 88, "input_tokens": 252, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 101, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 87, "input_tokens": 274, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 85, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 90, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 85, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 85, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 82, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 89, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 88, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 88, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 86, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 87, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 86, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 89, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 85, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 88, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 88, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 88, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 89, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 88, "input_tokens": 277, "model": "laya-rl-agent"}, {"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 86, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 89, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 80, "input_tokens": 252, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 84, "input_tokens": 262, "model": "laya-rl-agent"}, -{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 90, "input_tokens": 280, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 110, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 147, "input_tokens": 281, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 79, "input_tokens": 252, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 83, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 89, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7733, "payment": 0.0682, "returns": 0.0616, "account": 0.0329, "human": 0.064}, "ms": 89, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0187, "payment": 0.0147, "returns": 0.0362, "account": 0.9073, "human": 0.0231}, "ms": 89, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7175, "payment": 0.0254, "returns": 0.1784, "account": 0.0236, "human": 0.0552}, "ms": 88, "input_tokens": 281, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1644, "payment": 0.1207, "returns": 0.5976, "account": 0.0599, "human": 0.0574}, "ms": 88, "input_tokens": 279, "model": "laya-rl-agent"}, {"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9463, "payment": 0.0092, "returns": 0.0139, "account": 0.0093, "human": 0.0212}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 87, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 88, "input_tokens": 280, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 86, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0269, "payment": 0.1069, "returns": 0.0288, "account": 0.8051, "human": 0.0323}, "ms": 93, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8232, "payment": 0.0217, "returns": 0.085, "account": 0.0216, "human": 0.0486}, "ms": 90, "input_tokens": 280, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0524, "payment": 0.7587, "returns": 0.0837, "account": 0.0455, "human": 0.0598}, "ms": 88, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3384, "payment": 0.048, "returns": 0.4657, "account": 0.0444, "human": 0.1034}, "ms": 88, "input_tokens": 271, "model": "laya-rl-agent"}, {"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0216, "payment": 0.8293, "returns": 0.1041, "account": 0.0191, "human": 0.0259}, "ms": 89, "input_tokens": 273, "model": "laya-rl-agent"}, {"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.0209, "account": 0.9116, "human": 0.039}, "ms": 83, "input_tokens": 262, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 85, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 88, "input_tokens": 266, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 83, "input_tokens": 262, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.0229, "returns": 0.0697, "account": 0.0191, "human": 0.0409}, "ms": 86, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4706, "payment": 0.0385, "returns": 0.3826, "account": 0.0347, "human": 0.0736}, "ms": 85, "input_tokens": 266, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0082, "account": 0.979, "human": 0.0062}, "ms": 85, "input_tokens": 262, "model": "laya-rl-agent"}, {"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1021, "payment": 0.6139, "returns": 0.1456, "account": 0.0606, "human": 0.0778}, "ms": 89, "input_tokens": 274, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 91, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 85, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 91, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 92, "input_tokens": 272, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 91, "input_tokens": 258, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 92, "input_tokens": 275, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 89, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 88, "input_tokens": 271, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 89, "input_tokens": 279, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 82, "input_tokens": 259, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 90, "input_tokens": 277, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 88, "input_tokens": 273, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 85, "input_tokens": 265, "model": "laya-rl-agent"}, -{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 78, "input_tokens": 252, "model": "laya-rl-agent"} +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.106, "payment": 0.4177, "returns": 0.3168, "account": 0.0566, "human": 0.1029}, "ms": 87, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9276, "human": 0.0168}, "ms": 86, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.8318, "returns": 0.0428, "account": 0.0179, "human": 0.0842}, "ms": 92, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9369, "returns": 0.0322, "account": 0.0079, "human": 0.0127}, "ms": 91, "input_tokens": 272, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0237, "payment": 0.01, "returns": 0.0193, "account": 0.9311, "human": 0.0158}, "ms": 87, "input_tokens": 258, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.777, "returns": 0.1024, "account": 0.025, "human": 0.0505}, "ms": 95, "input_tokens": 275, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0438, "account": 0.0238, "human": 0.0252}, "ms": 101, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0145, "payment": 0.0119, "returns": 0.1124, "account": 0.844, "human": 0.0172}, "ms": 92, "input_tokens": 271, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8481, "payment": 0.0265, "returns": 0.0491, "account": 0.0273, "human": 0.0491}, "ms": 95, "input_tokens": 279, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3833, "payment": 0.3707, "returns": 0.0946, "account": 0.0639, "human": 0.0875}, "ms": 101, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3609, "payment": 0.0539, "returns": 0.1383, "account": 0.0465, "human": 0.4004}, "ms": 91, "input_tokens": 259, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0332, "account": 0.0133, "human": 0.0309}, "ms": 100, "input_tokens": 277, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7355, "payment": 0.0342, "returns": 0.1004, "account": 0.034, "human": 0.096}, "ms": 100, "input_tokens": 273, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0171, "payment": 0.0147, "returns": 0.0362, "account": 0.8091, "human": 0.1228}, "ms": 93, "input_tokens": 265, "model": "laya-rl-agent"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2552, "payment": 0.0673, "returns": 0.1241, "account": 0.0584, "human": 0.4949}, "ms": 86, "input_tokens": 252, "model": "laya-rl-agent"} ], "C-direct": [ -{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 283, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compile": "on", "source": "health", "read_at": "2026-09-30T14:45:44+00:00"}, "url": "http://127.0.0.1:8000"}, -{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 94, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 86, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 93, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 95, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 100, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 76, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 92, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 91, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 77, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 75, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 91, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 79, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 83, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 94, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 90, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 76, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 78, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 75, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 79, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 89, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 77, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 74, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 75, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 76, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 90, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 76, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 81, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 82, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 76, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 76, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 74, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 78, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 75, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 74, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 77, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 75, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 75, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 73, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 76, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 76, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 75, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 80, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 78, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 74, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 76, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 80, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 75, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 78, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 69, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 72, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 77, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 77, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 75, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 75, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 76, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 79, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 80, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 76, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 77, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 75, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 75, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 76, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 74, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 78, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 80, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 75, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 76, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 77, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 77, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 73, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 78, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 75, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 77, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 67, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} -], -"C-front": [ -{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 980, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compile": "on", "source": "health", "read_at": "2026-09-30T14:45:56+00:00"}, "url": "http://127.0.0.1:8080"}, +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 195, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compiled": true, "source": "health", "read_at": "2026-10-01T02:11:51+00:00"}, "url": "http://127.0.0.1:8000"}, {"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 83, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 86, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 86, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 82, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 84, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 88, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 87, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, {"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 82, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 81, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 87, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 79, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 78, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 82, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 80, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 85, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 86, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 71, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 72, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 88, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 71, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 86, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 87, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 84, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, {"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 87, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 80, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 81, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 83, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 79, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 78, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 77, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 99, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 86, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 84, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 79, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 84, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 80, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 84, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 80, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 83, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 84, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 84, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 85, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 78, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 80, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 81, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 83, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 84, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 88, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 87, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 85, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 87, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 78, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 84, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 84, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 81, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 85, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 81, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 82, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 79, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 80, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 83, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 82, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 83, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 79, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 84, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 82, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 78, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 84, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 80, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 77, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 83, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 82, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 84, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 82, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 83, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 83, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 85, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 83, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, -{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 76, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 73, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 87, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 71, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 79, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 95, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 73, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 71, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 72, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 72, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 71, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 73, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 69, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 72, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 72, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 68, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 70, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 73, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 74, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 70, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 64, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 68, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 71, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 71, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 72, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 75, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 73, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 72, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 69, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 96, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 97, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 71, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 73, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 72, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 70, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 72, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 70, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 71, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 71, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 71, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 65, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} +], +"C-front": [ +{"seed": 0, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 1181, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e", "served_by": {"checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "device": "mps", "weights_dtype": "torch.float16", "autocast_dtype": "torch.float16", "compiled": true, "source": "health", "read_at": "2026-10-01T02:12:02+00:00"}, "url": "http://127.0.0.1:8080"}, +{"seed": 0, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 76, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 74, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 73, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 74, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 71, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 81, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 73, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 75, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 70, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 75, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 70, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 73, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 71, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 75, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 75, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 72, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 75, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 72, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 72, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 0, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 68, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 89, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 71, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 73, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 73, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 70, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 72, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 72, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 72, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 69, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 73, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 71, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 72, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 72, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 72, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 73, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 65, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 1, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 74, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_a5a229edd810", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.7728, "payment": 0.0683, "returns": 0.0616, "account": 0.033, "human": 0.0643}, "ms": 71, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_9f4c2e9e9fc4", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0188, "payment": 0.0148, "returns": 0.0363, "account": 0.9068, "human": 0.0233}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6892a5546a24", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.7176, "payment": 0.0253, "returns": 0.1784, "account": 0.0236, "human": 0.0551}, "ms": 72, "input_tokens": 281, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_2ff652a83e3b", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.1642, "payment": 0.1205, "returns": 0.5981, "account": 0.0598, "human": 0.0574}, "ms": 72, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_52586863f51e", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9462, "payment": 0.0092, "returns": 0.014, "account": 0.0093, "human": 0.0212}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_d6e0fd840d60", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0272, "payment": 0.1066, "returns": 0.0289, "account": 0.8048, "human": 0.0325}, "ms": 73, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_e4351833c8c8", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8236, "payment": 0.0216, "returns": 0.0848, "account": 0.0215, "human": 0.0485}, "ms": 73, "input_tokens": 280, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_1a6847ba8b66", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0522, "payment": 0.7592, "returns": 0.0834, "account": 0.0454, "human": 0.0598}, "ms": 72, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f21a8be4a071", "expected": "returns", "key": "returns", "probabilities": {"logistics": 0.3374, "payment": 0.0481, "returns": 0.4665, "account": 0.0444, "human": 0.1036}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ec027c12e4da", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.0217, "payment": 0.829, "returns": 0.1042, "account": 0.0191, "human": 0.0259}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_507d761b520a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.011, "payment": 0.0175, "returns": 0.021, "account": 0.9115, "human": 0.039}, "ms": 77, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_892cab266e67", "expected": "human", "key": "logistics", "probabilities": {"logistics": 0.8474, "payment": 0.023, "returns": 0.0696, "account": 0.0191, "human": 0.0409}, "ms": 71, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_afe6633569aa", "expected": "returns", "key": "logistics", "probabilities": {"logistics": 0.4704, "payment": 0.0385, "returns": 0.3827, "account": 0.0347, "human": 0.0737}, "ms": 72, "input_tokens": 266, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_55ac1dd8d045", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0034, "payment": 0.0033, "returns": 0.0083, "account": 0.9789, "human": 0.0062}, "ms": 69, "input_tokens": 262, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c85ca34669b1", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.1022, "payment": 0.614, "returns": 0.1455, "account": 0.0605, "human": 0.0777}, "ms": 73, "input_tokens": 274, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_273060d164f5", "expected": "returns", "key": "payment", "probabilities": {"logistics": 0.1062, "payment": 0.4176, "returns": 0.3168, "account": 0.0565, "human": 0.1029}, "ms": 70, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_5b12b66a4428", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0151, "payment": 0.0128, "returns": 0.0278, "account": 0.9275, "human": 0.0168}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c72f978e4f83", "expected": "human", "key": "payment", "probabilities": {"logistics": 0.0233, "payment": 0.832, "returns": 0.0427, "account": 0.0179, "human": 0.0842}, "ms": 73, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_07772aa5f822", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0103, "payment": 0.9371, "returns": 0.0321, "account": 0.0079, "human": 0.0127}, "ms": 71, "input_tokens": 272, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_f5251390fb88", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0236, "payment": 0.0101, "returns": 0.0193, "account": 0.9308, "human": 0.0161}, "ms": 69, "input_tokens": 258, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_609c2a02ab84", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0451, "payment": 0.7771, "returns": 0.1024, "account": 0.0249, "human": 0.0504}, "ms": 74, "input_tokens": 275, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c8fd45b69015", "expected": "payment", "key": "payment", "probabilities": {"logistics": 0.0189, "payment": 0.8883, "returns": 0.0437, "account": 0.0239, "human": 0.0252}, "ms": 73, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_46f559df489a", "expected": "account", "key": "account", "probabilities": {"logistics": 0.0146, "payment": 0.0119, "returns": 0.1125, "account": 0.8438, "human": 0.0172}, "ms": 70, "input_tokens": 271, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_97415027853c", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.8479, "payment": 0.0265, "returns": 0.0492, "account": 0.0273, "human": 0.0491}, "ms": 72, "input_tokens": 279, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_ace2461b953a", "expected": "payment", "key": "logistics", "probabilities": {"logistics": 0.3834, "payment": 0.3707, "returns": 0.0946, "account": 0.064, "human": 0.0874}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_fac102ef66cc", "expected": "human", "key": "human", "probabilities": {"logistics": 0.3611, "payment": 0.0539, "returns": 0.138, "account": 0.0468, "human": 0.4002}, "ms": 70, "input_tokens": 259, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_8f0ec563a91a", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.9098, "payment": 0.0128, "returns": 0.0331, "account": 0.0133, "human": 0.031}, "ms": 72, "input_tokens": 277, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_6cb98186dde0", "expected": "logistics", "key": "logistics", "probabilities": {"logistics": 0.735, "payment": 0.0342, "returns": 0.1007, "account": 0.034, "human": 0.0961}, "ms": 72, "input_tokens": 273, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_30a6dedb9828", "expected": "human", "key": "account", "probabilities": {"logistics": 0.0173, "payment": 0.0148, "returns": 0.0363, "account": 0.809, "human": 0.1226}, "ms": 70, "input_tokens": 265, "model": "convaiinnovations/laya@55cf4c4ebb4e"}, +{"seed": 2, "ticket": "t_c5314933eb53", "expected": "human", "key": "human", "probabilities": {"logistics": 0.2557, "payment": 0.0672, "returns": 0.1239, "account": 0.0584, "human": 0.4948}, "ms": 65, "input_tokens": 252, "model": "convaiinnovations/laya@55cf4c4ebb4e"} ] } diff --git a/s1a/decision_models/served.py b/s1a/decision_models/served.py index 21bc0ae..3af82a5 100644 --- a/s1a/decision_models/served.py +++ b/s1a/decision_models/served.py @@ -38,7 +38,7 @@ _RETRIED_STATUSES = frozenset({502, 504}) # the frontend could not reach the worker, or it was too slow _REQUEST_ERRORS = frozenset({400, 413, 422}) _NOT_UP = ( - "the system1-omni worker listens only after loading and warming up (35-39 s with LAYA_WORKER_COMPILE=on " + "the system1-omni worker listens only after loading and warming up (35-39 s with --compile " "and fp16 weights on an M1 Pro); start it per system1-omni's recipe/laya/apple-silicon.md" ) @@ -231,7 +231,7 @@ def served_by_from_health(health: Json, routing: Any, read_at: str | None) -> Js "device": entry.get("device"), "weights_dtype": entry.get("weights_dtype"), "autocast_dtype": entry.get("autocast_dtype"), - "compile": compile_state.get("mode"), + "compiled": compile_state.get("enabled"), "source": "health", "read_at": read_at, } diff --git a/tests/data/served_laya/README.md b/tests/data/served_laya/README.md index 5261e74..ed608d5 100644 --- a/tests/data/served_laya/README.md +++ b/tests/data/served_laya/README.md @@ -1,11 +1,11 @@ # Served Laya fixtures -Responses recorded from running servers on 2026-09-30, one JSON file per case: `status`, +Responses recorded from running servers on 2026-10-01, one JSON file per case: `status`, `content_type`, `body` and, where it was sent, the `request`. | prefix | server | |---|---| -| `worker.` | system1-omni's Laya worker at `6311ae8` (PR #30), `LAYA_WORKER_COMPILE=on`, `LAYA_WORKER_WEIGHTS=fp16`, `LAYA_API_KEY=fixture-token`, MPS | +| `worker.` | system1-omni's Laya worker at `3d6cb57` (PR #30): `python -m frontend.laya_mps --device mps --model english --require-device --compile --weights fp16`, `LAYA_API_KEY=fixture-token` | | `frontend.` | the same worker behind `omni-jev` (system1-omni#2) | | `frontend-down.` | `omni-jev` with the worker stopped | | `laya-serve.` | plain `laya-serve`, no API key, MPS | diff --git a/tests/data/served_laya/frontend.health.json b/tests/data/served_laya/frontend.health.json index 8825dba..3769286 100644 --- a/tests/data/served_laya/frontend.health.json +++ b/tests/data/served_laya/frontend.health.json @@ -15,7 +15,7 @@ "mps_amp_min_rows": 5, "checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", - "warmup_ms": 29202.8, + "warmup_ms": 32915.6, "models": { "english": { "device": "mps", @@ -26,11 +26,11 @@ "mps_amp_min_rows": 5, "checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", - "warmup_ms": 29202.8 + "warmup_ms": 32915.6 } }, "compile": { - "mode": "on", + "enabled": true, "graphs_at_ready": 3, "graphs_now": 3, "recompiled_after_ready": false diff --git a/tests/data/served_laya/worker.health.json b/tests/data/served_laya/worker.health.json index 8825dba..3769286 100644 --- a/tests/data/served_laya/worker.health.json +++ b/tests/data/served_laya/worker.health.json @@ -15,7 +15,7 @@ "mps_amp_min_rows": 5, "checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", - "warmup_ms": 29202.8, + "warmup_ms": 32915.6, "models": { "english": { "device": "mps", @@ -26,11 +26,11 @@ "mps_amp_min_rows": 5, "checkpoint": "convaiinnovations/laya", "revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", - "warmup_ms": 29202.8 + "warmup_ms": 32915.6 } }, "compile": { - "mode": "on", + "enabled": true, "graphs_at_ready": 3, "graphs_now": 3, "recompiled_after_ready": false diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py index 271c91c..687762c 100644 --- a/tests/test_decision_models_served.py +++ b/tests/test_decision_models_served.py @@ -341,12 +341,12 @@ async def test_the_worker_names_checkpoint_revision_and_device(self) -> None: self.assertEqual(decision.model, "convaiinnovations/laya@55cf4c4ebb4e") served_by = raw["served_by"] self.assertEqual( - {k: served_by[k] for k in ("checkpoint", "device", "weights_dtype", "compile", "source")}, + {k: served_by[k] for k in ("checkpoint", "device", "weights_dtype", "compiled", "source")}, { "checkpoint": "convaiinnovations/laya", "device": "mps", "weights_dtype": "torch.float16", - "compile": "on", + "compiled": True, "source": "health", }, ) From b67e822dd6ce4267f2555a800a1ce27797bad9a0 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 11:29:18 +0800 Subject: [PATCH 12/14] [Fix] Read and write the served-Laya files as UTF-8 on Windows (#20) The registration check read s1a's sources with the platform encoding, cp1252 on Windows, and reported paths with backslashes there. Every new text read and write now names utf-8, and paths are reported in POSIX form. --- evals/ticket_router/compare_served.py | 6 +++--- tests/data/served_laya/capture.py | 2 +- tests/test_decision_models_served.py | 2 +- tests/test_served_laya_fixtures.py | 10 ++++++---- tests/test_served_laya_registration.py | 6 +++--- 5 files changed, 14 insertions(+), 12 deletions(-) diff --git a/evals/ticket_router/compare_served.py b/evals/ticket_router/compare_served.py index ef369e0..12a56b4 100644 --- a/evals/ticket_router/compare_served.py +++ b/evals/ticket_router/compare_served.py @@ -32,8 +32,8 @@ class Trial: def load(job_dir: Path) -> list[Trial]: trials = [] for trial_dir in sorted(path.parent for path in job_dir.glob("*/result.json")): - episode = json.loads((trial_dir / "agent" / "episode.json").read_text()) - result = json.loads((trial_dir / "result.json").read_text()) + episode = json.loads((trial_dir / "agent" / "episode.json").read_text(encoding="utf-8")) + result = json.loads((trial_dir / "result.json").read_text(encoding="utf-8")) router = episode["extra"]["ticket_router"] elapsed = float(result["agent_result"]["metadata"]["elapsed_s"]) trials.append(Trial(router["seed"], router, episode["decisions"], elapsed)) @@ -101,7 +101,7 @@ def main(arguments: list[str]) -> None: json.dumps(name) + ": [\n" + ",\n".join(json.dumps(row) for row in records(trials)) + "\n]" for name, trials in configs.items() ] - json_path.write_text("{\n" + ",\n".join(blocks) + "\n}\n") # one decision per line, for diffs + json_path.write_text("{\n" + ",\n".join(blocks) + "\n}\n", encoding="utf-8") # one decision per line, for diffs print("| config | correct | p50 ms | p95 ms | episodes s | model | served on |") print("|---|---:|---:|---:|---:|---|---|") for name, trials in configs.items(): diff --git a/tests/data/served_laya/capture.py b/tests/data/served_laya/capture.py index 6a6e262..f8e1940 100644 --- a/tests/data/served_laya/capture.py +++ b/tests/data/served_laya/capture.py @@ -112,5 +112,5 @@ def req(method, path, body=None, auth=True, raw=False): rec = req(m, p, b, a, raw) if not raw and b is not None and name != "too_many_questions": rec["request"] = b - (out / f"{server}.{name}.json").write_text(json.dumps(rec, indent=1, ensure_ascii=False) + "\n") + (out / f"{server}.{name}.json").write_text(json.dumps(rec, indent=1, ensure_ascii=False) + "\n", encoding="utf-8") print(f"{server:14} {name:18} {rec['status']} {rec['content_type']}") diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py index 687762c..fc4f4b7 100644 --- a/tests/test_decision_models_served.py +++ b/tests/test_decision_models_served.py @@ -36,7 +36,7 @@ def fixture(name: str) -> dict[str, Any]: - return json.loads((FIXTURES / f"{name}.json").read_text()) + return json.loads((FIXTURES / f"{name}.json").read_text(encoding="utf-8")) WORKER_HEALTH = fixture("worker.health")["body"] diff --git a/tests/test_served_laya_fixtures.py b/tests/test_served_laya_fixtures.py index 2656967..7ec2b34 100644 --- a/tests/test_served_laya_fixtures.py +++ b/tests/test_served_laya_fixtures.py @@ -15,7 +15,7 @@ ROOT = Path(__file__).resolve().parents[1] FIXTURES = sorted((ROOT / "tests" / "data" / "served_laya").glob("*.json")) -SPEC = yaml.safe_load((ROOT / "docs" / "api" / "laya-systemone.current.openapi.yaml").read_text()) +SPEC = yaml.safe_load((ROOT / "docs" / "api" / "laya-systemone.current.openapi.yaml").read_text(encoding="utf-8")) REGISTRY = Registry().with_resource("urn:spec", Resource.from_contents(SPEC, default_specification=DRAFT202012)) @@ -31,7 +31,7 @@ def test_there_are_fixtures_from_every_server() -> None: @pytest.mark.parametrize("path", FIXTURES, ids=[path.stem for path in FIXTURES]) def test_fixture_matches_the_spec(path: Path) -> None: - record = json.loads(path.read_text()) + record = json.loads(path.read_text(encoding="utf-8")) server, case = path.stem.split(".", 1) status, body = record["status"], record["body"] if status in (502, 504): # the frontend's own errors are plain text @@ -48,9 +48,11 @@ def test_fixture_matches_the_spec(path: Path) -> None: assert errors(schema, body) == [] -@pytest.mark.parametrize("path", [p for p in FIXTURES if "request" in json.loads(p.read_text())], ids=lambda p: p.stem) +@pytest.mark.parametrize( + "path", [p for p in FIXTURES if "request" in json.loads(p.read_text(encoding="utf-8"))], ids=lambda p: p.stem +) def test_sent_requests_match_the_spec_unless_meant_to_fail(path: Path) -> None: - record = json.loads(path.read_text()) + record = json.loads(path.read_text(encoding="utf-8")) found = errors("DecisionRequest", record["request"]) if record["status"] in (400, 422): assert found, "a request the server rejected should not pass the schema" diff --git a/tests/test_served_laya_registration.py b/tests/test_served_laya_registration.py index 2b80697..94e2c6c 100644 --- a/tests/test_served_laya_registration.py +++ b/tests/test_served_laya_registration.py @@ -24,7 +24,7 @@ def constants(node: ast.AST) -> set[str]: def places_missing_served() -> list[str]: missing = [] for path in sorted(SOURCE.rglob("*.py")): - tree = ast.parse(path.read_text()) + tree = ast.parse(path.read_text(encoding="utf-8")) for node in ast.walk(tree): if isinstance(node, (ast.Tuple, ast.List, ast.Set)): values = {e.value for e in node.elts if isinstance(e, ast.Constant)} @@ -38,7 +38,7 @@ def places_missing_served() -> list[str]: else: continue if "laya" in values and "laya-served" not in values: - missing.append(f"{path.relative_to(SOURCE.parent)}:{node.lineno}") + missing.append(f"{path.relative_to(SOURCE.parent).as_posix()}:{node.lineno}") return missing @@ -57,6 +57,6 @@ def test_the_named_lists() -> None: def test_the_check_would_catch_a_missing_place(tmp_path: Path, monkeypatch) -> None: (tmp_path / "s1a").mkdir() - (tmp_path / "s1a" / "new_front.py").write_text('NAMES = ("jev", "laya")\n') + (tmp_path / "s1a" / "new_front.py").write_text('NAMES = ("jev", "laya")\n', encoding="utf-8") monkeypatch.setattr(__name__ + ".SOURCE", tmp_path / "s1a") assert places_missing_served() == ["s1a/new_front.py:1"] From 39afcc0554bc71cda0d57c797f2af1780c90ee5a Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 12:18:46 +0800 Subject: [PATCH 13/14] [Fix] Keep the /health refresh inside the decision deadline; compare stopped episodes (#20) The identity refresh every 30 s ran before the decision's deadline began, so a stalled /health could stretch one decision to LAYA_SERVED_TIMEOUT_S plus 2 s. The refresh now shares the decision's deadline and takes at most half of what is left. compare_served.py paired the batch's ticket ids with the ticks, so a job with an episode that stopped early raised in zip(). It now pairs the processed routes with the ticks and checks that each pair agrees. --- .env.example | 2 +- docs/configuration.md | 2 +- docs/served-laya.md | 3 +- evals/ticket_router/compare_served.py | 19 +++++++-- s1a/decision_models/served.py | 34 +++++++++++----- tests/test_compare_served.py | 57 +++++++++++++++++++++++++++ tests/test_decision_models_served.py | 20 ++++++++++ 7 files changed, 120 insertions(+), 17 deletions(-) create mode 100644 tests/test_compare_served.py diff --git a/.env.example b/.env.example index 61ca83b..f428328 100644 --- a/.env.example +++ b/.env.example @@ -38,7 +38,7 @@ MODEL_NAME=google/gemini-2.5-flash # LAYA_SERVED_URL=http://127.0.0.1:8000 # the worker; :8080 for the omni-jev frontend # LAYA_SERVED_MODEL=english # LAYA_SERVED_API_KEY= # the worker's LAYA_API_KEY, when it sets one -# LAYA_SERVED_TIMEOUT_S=5 # per decision, the one retry included +# LAYA_SERVED_TIMEOUT_S=5 # per decision, the retry and any /health refresh included # LAYA_SERVED_MAX_LEN=512 # the worker's LAYA_MAX_LEN # ---- Cua-S1 Nano (the in-process option scorer behind --model cua; needs `uv sync --extra cua`) ---- diff --git a/docs/configuration.md b/docs/configuration.md index 99543cf..7178bc0 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -35,7 +35,7 @@ Variables can be exported in your shell or placed in a `.env` file at the root o | `LAYA_SERVED_URL` | `laya-served` model | *(unset, required)* | Base URL of a served Laya: the system1-omni worker (`http://127.0.0.1:8000`), its `omni-jev` frontend (`:8080`) or plain laya-serve. | | `LAYA_SERVED_MODEL` | `laya-served` model | `english` | Name of the served model to ask, one the server loaded. | | `LAYA_SERVED_API_KEY` | `laya-served` model | *(unset)* | Bearer token, the server's `LAYA_API_KEY` when it sets one. | -| `LAYA_SERVED_TIMEOUT_S` | `laya-served` model | `5` | Deadline per decision in seconds, the one retry included. | +| `LAYA_SERVED_TIMEOUT_S` | `laya-served` model | `5` | Deadline per decision in seconds, the one retry and any `/health` refresh included. | | `LAYA_SERVED_MAX_LEN` | `laya-served` model | `512` | The server's token window per question (its `LAYA_MAX_LEN`); a request that fills it raises. | | `CUA_S1_CHECKPOINT` | `cua` model | `cua-ai/cua-s1-nano-0.1` | Hugging Face checkpoint ID or local directory for Cua-S1 Nano option scorer. | | `CUA_S1_SUBFOLDER` | `cua` model | `text` | Subfolder within checkpoint directory containing text option scoring weights. | diff --git a/docs/served-laya.md b/docs/served-laya.md index 3c6253f..4ad7432 100644 --- a/docs/served-laya.md +++ b/docs/served-laya.md @@ -74,7 +74,8 @@ problem+json and fall back to `detail`; read identity from `served_by` when pres - **Request.** Questions serialised with the existing `laya_question()`, which keeps Laya's own `noul` shape (a plain-string instruction). `score` is not sent until an agent needs it. - **Identity.** Each response's `served_by` goes into the run record. Until servers send it, the client - reads `/health` at warm-up and again whenever its reading is older than 30 s (the worker reports the + reads `/health` at warm-up and again whenever its reading is older than 30 s, inside the decision's + deadline and with at most half of it (the worker reports the live device, and laya moves a model to the CPU on a GPU out-of-memory error), records the reading's time as `read_at`, and each decision takes the entry for the model that answered (`models[routing.model]` on the system1-omni worker, else its top-level fields). Plain laya-serve diff --git a/evals/ticket_router/compare_served.py b/evals/ticket_router/compare_served.py index 12a56b4..65906ac 100644 --- a/evals/ticket_router/compare_served.py +++ b/evals/ticket_router/compare_served.py @@ -45,12 +45,23 @@ def percentile(values: list[float], q: float) -> float: return ordered[min(len(ordered) - 1, round(q * (len(ordered) - 1)))] +def decided(trial: Trial) -> list[tuple[dict[str, Any], dict[str, Any]]]: + """(route, tick) per processed ticket. An episode that stopped early has fewer of both than its batch.""" + pairs = list(zip(trial.router["routes"], trial.ticks, strict=True)) + for route, tick in pairs: + if route["predicted"] != tick["key"]: + raise ValueError( + f"seed {trial.seed}: {route['id']} routed {route['predicted']}, its tick says {tick['key']}" + ) + return pairs + + def routes(trials: list[Trial]) -> dict[tuple[int, str], tuple[str, dict[str, float]]]: - """(seed, ticket id) -> (predicted queue, probabilities); ticks follow the batch's ticket order.""" + """(seed, ticket id) -> (predicted queue, probabilities) for every processed ticket.""" return { - (trial.seed, ticket_id): (tick["key"], tick["probabilities"]) + (trial.seed, route["id"]): (tick["key"], tick["probabilities"]) for trial in trials - for ticket_id, tick in zip(trial.router["ticket_ids"], trial.ticks, strict=True) + for route, tick in decided(trial) } @@ -79,7 +90,7 @@ def records(trials: list[Trial]) -> list[dict[str, Any]]: """One row per decision; ``served_by`` and ``url`` only on the first row and where they change.""" rows, last = [], None for trial in trials: - for route, tick in zip(trial.router["routes"], trial.ticks, strict=True): + for route, tick in decided(trial): row = {"seed": trial.seed, "ticket": route["id"], "expected": route["expected"]} row |= {k: tick.get(k) for k in ("key", "probabilities", "ms", "input_tokens", "model")} source = {k: tick[k] for k in ("served_by", "url") if k in tick} diff --git a/s1a/decision_models/served.py b/s1a/decision_models/served.py index 3af82a5..ec8a82c 100644 --- a/s1a/decision_models/served.py +++ b/s1a/decision_models/served.py @@ -100,10 +100,20 @@ def __init__( headers = {"Authorization": f"Bearer {api_key}"} if api_key else {} self._client = httpx.AsyncClient(timeout=timeout_s, headers=headers, transport=transport) - async def health(self) -> Json: - """``GET /health`` once. Refused or unreachable raises; any other failure returns ``{}`` with a warning.""" + def deadline(self) -> float: + """The end of one decision's time budget, on this client's clock.""" + return self._clock() + self._timeout_s + + async def health(self, deadline: float | None = None) -> Json: + """``GET /health`` once. Refused or unreachable raises; any other failure returns ``{}`` with a warning. + Within a decision's ``deadline`` it takes at most half of what is left, so the decision keeps the rest.""" + timeout = HEALTH_TIMEOUT_S + if deadline is not None: + timeout = min(timeout, (deadline - self._clock()) / 2) + if timeout <= 0: + return {} try: - response = await self._client.get(f"{self.url}/health", timeout=HEALTH_TIMEOUT_S) + response = await self._client.get(f"{self.url}/health", timeout=timeout) except (httpx.ConnectError, httpx.ConnectTimeout) as exc: raise build_error( StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"no served Laya at {self.url}: {_NOT_UP}" @@ -121,10 +131,13 @@ async def health(self) -> Json: return {} return body if isinstance(body, dict) else {} - async def decide(self, body: Json, request_id: str) -> tuple[Json, dict[str, str], int]: + async def decide( + self, body: Json, request_id: str, deadline: float | None = None + ) -> tuple[Json, dict[str, str], int]: """One decision: the payload, the response headers and the last attempt's round trip in ms. Every attempt - sends the same ``X-Request-Id``, so the server's logs tie a retry to its first try.""" - deadline = self._clock() + self._timeout_s + sends the same ``X-Request-Id``, so the server's logs tie a retry to its first try. ``deadline`` (from + ``deadline()``) is shared with a ``/health`` read made for the same decision; a fresh one starts otherwise.""" + deadline = self.deadline() if deadline is None else deadline retried = False while True: remaining = deadline - self._clock() @@ -271,10 +284,10 @@ def __init__( def model(self) -> str: return self._model - async def _read_health(self, *, strict: bool) -> None: + async def _read_health(self, *, strict: bool, deadline: float | None = None) -> None: self._health_tried = self._clock() try: - health = await self._client.health() + health = await self._client.health(deadline) except Exception: if strict: raise @@ -290,15 +303,16 @@ async def warm(self) -> None: await self._read_health(strict=True) async def _decide(self, observation: Observation, questions: dict[str, Question]) -> Reply: + deadline = self._client.deadline() # one budget for the identity refresh and the decision if self._health_tried is None or self._clock() - self._health_tried >= HEALTH_MAX_AGE_S: - await self._read_health(strict=False) + await self._read_health(strict=False, deadline=deadline) body = { "model": self._model, "state": observation.state, "questions": {name: laya_question(question) for name, question in questions.items()}, } request_id = uuid.uuid4().hex - payload, headers, ms = await self._client.decide(body, request_id) + payload, headers, ms = await self._client.decide(body, request_id, deadline) usage = Usage.from_payload(payload.get("usage")) check_window( usage, diff --git a/tests/test_compare_served.py b/tests/test_compare_served.py new file mode 100644 index 0000000..72de1f3 --- /dev/null +++ b/tests/test_compare_served.py @@ -0,0 +1,57 @@ +# coding: utf-8 +"""evals/ticket_router/compare_served.py: the table and the agreement over jobs, including one that stopped early.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from evals.ticket_router import compare_served + +TICKETS = [("t1", "payment"), ("t2", "returns"), ("t3", "account")] + + +def write_job(root: Path, name: str, predicted: list[str]) -> Path: + """One seed-0 trial that decided the first ``len(predicted)`` of three tickets.""" + trial = root / name / "ticket_router--0__x" + (trial / "agent").mkdir(parents=True) + routes = [ + {"id": tid, "expected": expected, "predicted": key, "correct": key == expected} + for (tid, expected), key in zip(TICKETS, predicted) + ] + ticks = [ + {"key": key, "probabilities": {key: 0.9, "human": 0.1}, "ms": 50 + i, "source": name} + for i, key in enumerate(predicted) + ] + router = { + "seed": 0, + "total": 3, + "correct": sum(r["correct"] for r in routes), + "ticket_ids": [t for t, _ in TICKETS], + "routes": routes, + } + episode = {"decisions": ticks, "extra": {"ticket_router": router}} + (trial / "agent" / "episode.json").write_text(json.dumps(episode), encoding="utf-8") + (trial / "result.json").write_text(json.dumps({"agent_result": {"metadata": {"elapsed_s": 1.0}}}), encoding="utf-8") + return root / name + + +def test_a_job_that_stopped_early_is_compared_on_what_it_decided(tmp_path: Path, capsys) -> None: + full = write_job(tmp_path, "full", ["payment", "returns", "human"]) + early = write_job(tmp_path, "early", ["payment"]) # the episode stopped after one decision + compare_served.main([f"A={full}", f"B={early}"]) + out = capsys.readouterr().out + assert "| A | 2/3 |" in out and "| B | 1/3 |" in out + assert "- A and B route 1 of 1 (seed, ticket) pairs the same" in out + + +def test_a_tick_that_does_not_match_its_route_is_an_error(tmp_path: Path) -> None: + job = write_job(tmp_path, "job", ["payment", "returns"]) + episode_path = next(job.glob("*/agent/episode.json")) + episode = json.loads(episode_path.read_text(encoding="utf-8")) + episode["decisions"][1]["key"] = "account" + episode_path.write_text(json.dumps(episode), encoding="utf-8") + with pytest.raises(ValueError, match="t2 routed returns"): + compare_served.routes(compare_served.load(job)) diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py index fc4f4b7..09eff7a 100644 --- a/tests/test_decision_models_served.py +++ b/tests/test_decision_models_served.py @@ -98,6 +98,8 @@ def handler(self, request: httpx.Request) -> httpx.Response: self.health_calls += 1 if isinstance(self.health, Exception): raise self.health + if callable(self.health): + return self.health(request) if isinstance(self.health, httpx.Response): return self.health return ok(self.health) @@ -294,6 +296,24 @@ def slow_refusal(request: httpx.Request) -> httpx.Response: await self.assert_fails(server, StatusCode.MODEL_CALL_FAILED, "within 5 s", clock=clock) self.assertEqual(len(server.decisions), 1) + async def test_a_stalled_identity_refresh_stays_inside_the_deadline(self) -> None: + clock = Clock() + given: dict[str, float] = {} + + def stalled_health(request: httpx.Request) -> httpx.Response: + given["health"] = request.extensions["timeout"]["read"] + clock.now += given["health"] # waits out its whole share + raise httpx.ReadTimeout("stalled") + + def answer(request: httpx.Request) -> httpx.Response: + given["decide"] = request.extensions["timeout"]["read"] + return server.answer(request) + + server = Server(health=stalled_health, script=[answer]) + model, _ = make_model(server, timeout_s=3.0, clock=clock) + await model.decide_many(OBSERVATION, {"pick": PICK}) + self.assertEqual(given, {"health": 1.5, "decide": 1.5}) # half the budget each, 3 s in all + async def test_a_body_without_answers_is_malformed(self) -> None: for response in (httpx.Response(200, text="not json"), ok({"model": "x"}), httpx.Response(200, json=[1])): await self.assert_fails(Server(script=[response]), StatusCode.MODEL_CALL_FAILED, "served Laya returned") From 3a1712bbf7047eebbcab45f6ae72d5e8b84ecdfe Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:59:06 +0800 Subject: [PATCH 14/14] [Docs] Keep benchmark figures out of served-Laya code comments and errors (#20) The not-up error quoted the M1 Pro ready time with --compile and fp16, which misleads on a CPU worker or another Mac; it now says the worker may still be starting and points to the recipe. The timeout comment gives its reason instead of a measured latency range, and the browser comment names served Laya as the HTTP one. The run section quotes the real message. --- docs/served-laya.md | 2 +- s1a/browser/browse.py | 2 +- s1a/decision_models/served.py | 8 +++++--- tests/test_decision_models_served.py | 2 +- 4 files changed, 8 insertions(+), 6 deletions(-) diff --git a/docs/served-laya.md b/docs/served-laya.md index 4ad7432..9f08ef6 100644 --- a/docs/served-laya.md +++ b/docs/served-laya.md @@ -135,7 +135,7 @@ PYTHONPATH=src .venv/bin/python -m frontend.laya_mps --device mps --model englis ``` The worker listens once it is warm, after about 40 s on an M1 Pro with these options; until then a -decision fails with "not up or still warming". On a Mac without MPS, or on Linux, drop `--compile` +decision fails with "no served Laya at …", saying it may still be starting. On a Mac without MPS, or on Linux, drop `--compile` and `--weights fp16` and use `--device cpu`. laya-serve's `LAYA_API_KEY` still turns on bearer auth; set the same value in `LAYA_SERVED_API_KEY`. The Rust frontend (`omni-jev`, port 8080) can sit in front of it; point `LAYA_SERVED_URL` at whichever you call. diff --git a/s1a/browser/browse.py b/s1a/browser/browse.py index 17ec087..aab204a 100644 --- a/s1a/browser/browse.py +++ b/s1a/browser/browse.py @@ -40,7 +40,7 @@ "laya-served", "cua", "llm", -) # a decision model (Jev or Laya over HTTP, Laya or Cua-S1 in process) or the chat model +) # a decision model (Jev or served Laya over HTTP, Laya or Cua-S1 in process) or the chat model RUNS_DIR = HOME / "runs" / "browser" diff --git a/s1a/decision_models/served.py b/s1a/decision_models/served.py index ec8a82c..7d1d242 100644 --- a/s1a/decision_models/served.py +++ b/s1a/decision_models/served.py @@ -29,7 +29,9 @@ logger = logging.getLogger(__name__) -SERVED_TIMEOUT_S = 5.0 # one decision, retries included; a warm Laya answers in 25-160 ms on MPS +SERVED_TIMEOUT_S = ( + 5.0 # one decision, retry included: room for a slow answer and one retry, short enough to fail a step +) HEALTH_TIMEOUT_S = 2.0 HEALTH_MAX_AGE_S = 30.0 # the worker reports the live device; laya moves a model to the CPU on a GPU OOM DEFAULT_SERVED_MODEL = "english" @@ -38,8 +40,8 @@ _RETRIED_STATUSES = frozenset({502, 504}) # the frontend could not reach the worker, or it was too slow _REQUEST_ERRORS = frozenset({400, 413, 422}) _NOT_UP = ( - "the system1-omni worker listens only after loading and warming up (35-39 s with --compile " - "and fp16 weights on an M1 Pro); start it per system1-omni's recipe/laya/apple-silicon.md" + "the system1-omni worker listens only once it has loaded and warmed up, so it may still be starting; " + "start it per system1-omni's recipe/laya/apple-silicon.md" ) diff --git a/tests/test_decision_models_served.py b/tests/test_decision_models_served.py index 09eff7a..0392af5 100644 --- a/tests/test_decision_models_served.py +++ b/tests/test_decision_models_served.py @@ -338,7 +338,7 @@ async def test_warm_fails_early_when_no_server_listens(self) -> None: with self.assertRaises(BaseError) as caught: await model.warm() self.assertEqual(caught.exception.status, StatusCode.MODEL_CALL_FAILED) - self.assertIn("listens only after loading and warming up", str(caught.exception)) + self.assertIn("listens only once it has loaded and warmed up", str(caught.exception)) async def test_an_unusable_health_is_a_warning_not_an_error(self) -> None: for health in (httpx.Response(404), httpx.Response(200, text="ok")):