From 2a4735808cbe93ff10965e42debb5ca8ee8c9a59 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Mon, 28 Sep 2026 17:50:37 +0800 Subject: [PATCH 01/21] Serve Laya on Apple Silicon: worker, benchmarks and recipe Add a Laya worker (src/models/laya/worker.py) that runs laya-serve unchanged except for startup and /health: it binds only after a warmup, and /health reports the device, dtypes, checkpoint and revision the model actually uses. LAYA_REQUIRE_DEVICE=1 exits instead of serving on the wrong device, and LAYA_WORKER_COMPILE=single compiles the one-question path. Add benchmarks/laya: fixed inputs, in-process and HTTP benchmarks (laya-serve directly and behind the Rust frontend), output parity against CPU with tolerances fixed in advance, an MPS profile and a report built from raw JSONL. Add recipe/laya/apple-silicon.md with setup, frontend, tests and benchmark commands on a Mac. --- benchmarks/laya/README.md | 42 ++++ benchmarks/laya/bench_http.py | 291 +++++++++++++++++++++++++ benchmarks/laya/bench_inproc.py | 154 +++++++++++++ benchmarks/laya/check_workloads.py | 27 +++ benchmarks/laya/env.py | 159 ++++++++++++++ benchmarks/laya/parity.py | 112 ++++++++++ benchmarks/laya/profile_mps.py | 257 ++++++++++++++++++++++ benchmarks/laya/report.py | 189 ++++++++++++++++ benchmarks/laya/results/.gitignore | 5 + benchmarks/laya/workloads.jsonl | 11 + benchmarks/laya/workloads.src.py | 90 ++++++++ recipe/README.md | 2 + recipe/laya/README.md | 3 +- recipe/laya/apple-silicon.md | 123 +++++++++++ src/models/laya/README.md | 26 ++- src/models/laya/tests/test_contract.py | 166 ++++++++++++++ src/models/laya/tests/test_worker.py | 196 +++++++++++++++++ src/models/laya/worker.py | 206 +++++++++++++++++ 18 files changed, 2057 insertions(+), 2 deletions(-) create mode 100644 benchmarks/laya/README.md create mode 100644 benchmarks/laya/bench_http.py create mode 100644 benchmarks/laya/bench_inproc.py create mode 100644 benchmarks/laya/check_workloads.py create mode 100644 benchmarks/laya/env.py create mode 100644 benchmarks/laya/parity.py create mode 100644 benchmarks/laya/profile_mps.py create mode 100644 benchmarks/laya/report.py create mode 100644 benchmarks/laya/results/.gitignore create mode 100644 benchmarks/laya/workloads.jsonl create mode 100644 benchmarks/laya/workloads.src.py create mode 100644 recipe/laya/apple-silicon.md create mode 100644 src/models/laya/tests/test_contract.py create mode 100644 src/models/laya/tests/test_worker.py create mode 100644 src/models/laya/worker.py diff --git a/benchmarks/laya/README.md b/benchmarks/laya/README.md new file mode 100644 index 0000000..96cfe5c --- /dev/null +++ b/benchmarks/laya/README.md @@ -0,0 +1,42 @@ +# Laya benchmarks + +Measures Laya on CPU and Apple Silicon (MPS) for [#3](https://github.com/ThinkFlowLab/system1-omni/issues/3). +Every script writes raw JSONL; `report.py` is the only place numbers are computed. + +| file | purpose | +| --- | --- | +| `workloads.src.py` → `workloads.jsonl` | fixed inputs: W1–W6 timed, P* parity-only | +| `check_workloads.py` | tokens per row with Laya's tokenizer, ±10% of each target | +| `bench_inproc.py` | in-process: import, load, warmup, first request per workload, warm latency, memory | +| `bench_http.py` | against a `/v1/systemone` worker: process-to-ready (with `--spawn`), first request per workload, warm latency and throughput at each `--concurrency` | +| `profile_mps.py` | where a request's time goes on MPS: length sweep with a fixed-cost fit, stage split (encode, dispatch, GPU wait, copy back, decode) and host operator counts | +| `parity.py` | answers of every run vs a reference run, tolerances fixed in advance (fp32 1e-3, fp16 1e-2) | +| `env.py` | run header: SHAs, versions, checkpoint revision, hardware, power, load | +| `report.py` | JSONL → tables, including the run-to-run gate | + +## Run + +Use the same environment as the Laya recipe (`laya[serve]==0.3.20`, Python 3.12). + +```sh +python benchmarks/laya/check_workloads.py +python benchmarks/laya/bench_inproc.py --device cpu --config C1 --run feasibility +python benchmarks/laya/bench_inproc.py --device mps --config C2 --run feasibility +python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m1 +python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m2 +python benchmarks/laya/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve +python benchmarks/laya/bench_http.py --config C4 --run feasibility --url http://127.0.0.1:8080 +python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl +python benchmarks/laya/parity.py benchmarks/laya/results/*.jsonl --ref C1 +``` + +Measured runs (any `--run` other than `feasibility`) refuse to start on battery power or when the +1-minute load average is above `--max-load` (default 2). Two measured runs of a config pass when +their p50s differ by at most 10% for every workload. + +Timing: wall clock around `Agent.system_one` followed by `torch.mps.synchronize()`, so it includes +tokenization and post-processing. Workload order is shuffled per run with `--seed`. + +Memory is the physical footprint of the process running Laya (`proc_pid_rusage`, the same number as +Activity Monitor's "Memory" and `footprint -p`). On Apple silicon it includes Metal allocations, so +in-process and worker numbers are comparable and MPS tensors are counted. diff --git a/benchmarks/laya/bench_http.py b/benchmarks/laya/bench_http.py new file mode 100644 index 0000000..8b706e8 --- /dev/null +++ b/benchmarks/laya/bench_http.py @@ -0,0 +1,291 @@ +"""HTTP benchmark against a /v1/systemone worker (C3 = laya-serve, C4 = laya-serve behind the Rust frontend). + +With --spawn the script starts the worker itself and times process start to the first successful +/health, then the first request per workload after that. Without it, it attaches to --url. With +--frontend as well, the worker listens on --backend-port and the Rust frontend binary is started on +--url in front of it; readiness is then the frontend's /health, which proxies the worker's. Each +workload runs at every --concurrency level; each client thread keeps one keep-alive connection. + + python benchmarks/laya/bench_http.py --config C3 --run m1 --spawn .venv-laya/bin/laya-serve + python benchmarks/laya/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ + --frontend target/release/omni-jev --spawn .venv-laya/bin/laya-serve +""" + +import argparse +import http.client +import json +import os +import random +import subprocess +import sys +import threading +import time +from pathlib import Path +from urllib.parse import urlsplit + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE)) +from env import footprint_mb, header, noise_problems + +CHECKPOINT = "convaiinnovations/laya" # what laya-serve's "english" model resolves to (laya/router.py) + + +class Client: + """One keep-alive connection. Not thread-safe: one per thread.""" + + def __init__(self, url, token=None): + parts = urlsplit(url) + self.conn = http.client.HTTPConnection(parts.hostname, parts.port or 80, timeout=120) + self.headers = {"Content-Type": "application/json"} + if token: + self.headers["Authorization"] = f"Bearer {token}" + + def request(self, method, path, body=None, retry=False): + """Timed calls pass retry=False so a dropped connection shows up as an error, not a slow request. + Untimed calls retry once: uvicorn closes keep-alive connections idle for 5 s.""" + try: + started = time.perf_counter() + self.conn.request(method, path, body=body, headers=self.headers) + response = self.conn.getresponse() + data = response.read() + return (time.perf_counter() - started) * 1000, response.status, data + except (http.client.RemoteDisconnected, BrokenPipeError, ConnectionResetError): + self.conn.close() + if not retry: + raise + return self.request(method, path, body) + + def close(self): + self.conn.close() + + +def wait_ready(url, processes, timeout_s): + """Poll /health every 50 ms; return seconds from now until it answers 200, and the body.""" + started = time.perf_counter() + while time.perf_counter() - started < timeout_s: + for name, process in processes.items(): + if process.poll() is not None: + sys.exit(f"{name} exited with {process.returncode} before it was ready") + client = Client(url) + try: + _, status, body = client.request("GET", "/health") + if status == 200: + return time.perf_counter() - started, json.loads(body) + except OSError: + pass + finally: + client.close() + time.sleep(0.05) + sys.exit(f"worker not ready after {timeout_s} s") + + +def body_for(workload, model): + return json.dumps({"model": model, "state": workload["state"], "questions": workload["questions"]}).encode() + + +def run_level(url, token, body, n, concurrency): + """n requests split over `concurrency` threads. Returns [(thread, ms, status)], elapsed seconds.""" + per_thread = [n // concurrency + (i < n % concurrency) for i in range(concurrency)] + results, lock = [], threading.Lock() + barrier = threading.Barrier(concurrency + 1, timeout=120) # a thread that fails to connect breaks it + + def worker(index, count): + client = Client(url, token) + client.request("POST", "/v1/systemone", body, retry=True) # connect outside the timed window + barrier.wait() + mine = [] + for _ in range(count): + try: + ms, status, _ = client.request("POST", "/v1/systemone", body) + except (OSError, http.client.HTTPException): + ms, status = None, 0 # counted as an error, excluded from latency + mine.append((index, ms, status)) + client.close() + with lock: + results.extend(mine) + + threads = [threading.Thread(target=worker, args=(i, c)) for i, c in enumerate(per_thread)] + for t in threads: + t.start() + barrier.wait() + started = time.perf_counter() + for t in threads: + t.join() + return results, time.perf_counter() - started + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--config", required=True, help="label, e.g. C3 or C4") + parser.add_argument("--run", required=True, help="feasibility, m1, m2, ...") + parser.add_argument("--url", default="http://127.0.0.1:8000") + parser.add_argument("--model", default="english", help="model name the worker serves") + parser.add_argument("--frontend", help="Rust frontend binary to start on --url in front of the spawned worker") + parser.add_argument("--backend-port", type=int, default=8000, help="worker port when --frontend is used") + parser.add_argument("--spawn", nargs=argparse.REMAINDER, help="start this worker command, then benchmark it") + parser.add_argument("--device", default="mps", help="LAYA_DEVICE for a spawned worker") + parser.add_argument("--ready-timeout", type=float, default=600) + parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) + parser.add_argument("--only", nargs="*", help="bench workload ids to run (default: all)") + parser.add_argument("-n", type=int, default=300, help="timed requests per workload and concurrency level") + parser.add_argument("--discard", type=int, default=20, help="warmup requests per workload") + parser.add_argument("--concurrency", type=int, nargs="+", default=[1, 4]) + parser.add_argument("--seed", type=int, default=0, help="workload order seed") + parser.add_argument("--out", default=str(HERE / "results")) + parser.add_argument("--max-load", type=float, default=2.0, help="1-min load average allowed for measured runs") + args = parser.parse_args() + token = os.environ.get("OMNI_JEV_TEST_TOKEN") + if args.frontend and not args.spawn: + parser.error("--frontend needs --spawn: the frontend is started in front of a spawned worker") + if args.frontend and urlsplit(args.url).port == args.backend_port: + parser.error("--url and --backend-port must differ when --frontend is used") + + problems = noise_problems(args.max_load) + if problems and args.run != "feasibility": + sys.exit("refusing a measured run: " + "; ".join(problems)) + for problem in problems: + print(f"warning: {problem}", file=sys.stderr) + + with open(args.workloads) as f: + workloads = [json.loads(line) for line in f if line.strip()] + bench = [w for w in workloads if w["kind"] == "bench" and (not args.only or w["id"] in args.only)] + parity = [w for w in workloads if w["kind"] == "parity"] + + Path(args.out).mkdir(parents=True, exist_ok=True) + processes = {} + if args.spawn: + port = args.backend_port if args.frontend else urlsplit(args.url).port + env = { + **os.environ, + "LAYA_HOST": "127.0.0.1", + "LAYA_PORT": str(port), + "LAYA_DEVICE": args.device, + "LAYA_MODELS": args.model, + "LAYA_PRELOAD": "1", + "LAYA_LOG_LEVEL": "warning", + } + spawn_log = open(Path(args.out) / f"http_{args.config}_{args.run}.worker.log", "w") # noqa: SIM115 + processes["worker"] = subprocess.Popen(args.spawn, env=env, stdout=spawn_log, stderr=subprocess.STDOUT) + if args.frontend: + parts = urlsplit(args.url) + env = { + **os.environ, + "OMNI_JEV_BIND": f"{parts.hostname}:{parts.port}", + "OMNI_JEV_BACKEND_URL": f"http://127.0.0.1:{args.backend_port}", + } + frontend_log = open(Path(args.out) / f"http_{args.config}_{args.run}.frontend.log", "w") # noqa: SIM115 + processes["frontend"] = subprocess.Popen( + [args.frontend], env=env, stdout=frontend_log, stderr=subprocess.STDOUT + ) + process = processes.get("worker") + frontend = processes.get("frontend") + + def memory(): + mem = footprint_mb(process.pid) if process else {} + if frontend: + mem["frontend_footprint_mb"] = footprint_mb(frontend.pid).get("footprint_mb") + return mem + + out = Path(args.out) / f"http_{args.config}_{args.run}.jsonl" + common = {"config": args.config, "run": args.run} + try: + ready_s, health = wait_ready(args.url, processes, args.ready_timeout) + with open(out, "w") as f: + + def emit(record): + f.write(json.dumps({**common, **record}) + "\n") + + # /health's device is what the worker reports; laya-serve 0.3.20 echoes LAYA_DEVICE (issue #3, G2). + emit( + header( + CHECKPOINT, + n=args.n, + discard=args.discard, + seed=args.seed, + url=args.url, + spawned=args.spawn, + frontend=args.frontend, + health=health, + device_actual=health.get("device"), + # The worker reports these in /health; laya-serve does not, so its values are assumed. + mps_amp_min_rows=health.get("mps_amp_min_rows", int(os.environ.get("LAYA_MPS_AMP_MIN_ROWS", "5"))), + amp_dtype=health.get("autocast_dtype") + or ("torch.float16" if health.get("device") == "mps" else "torch.float32"), + weights_dtype=health.get("weights_dtype") or "torch.float32", + dtype_source=None if "weights_dtype" in health else "assumed: laya 0.3.20 defaults", + ) + ) + + client = Client(args.url, token) + started = time.perf_counter() + first_ms, routing = {}, None + for w in bench: + ms, status, data = client.request("POST", "/v1/systemone", body_for(w, args.model), retry=True) + if status != 200: + sys.exit(f"{w['id']}: status {status}: {data[:200]!r}") + first_ms[w["id"]] = ms + routing = json.loads(data).get("routing") + for _ in range(args.discard - 1): + client.request("POST", "/v1/systemone", body_for(w, args.model), retry=True) + warmup_s = time.perf_counter() - started + emit( + { + "type": "phase", + "process_to_ready_s": round(ready_s, 3) if process else None, + "warmup_s": round(warmup_s, 3), + "first_ms": {k: round(v, 2) for k, v in first_ms.items()}, + "routing": routing, + "noise": problems, + **memory(), + } + ) + + order = bench[:] + random.Random(args.seed).shuffle(order) + for w in order: + body = body_for(w, args.model) + _, _, data = client.request("POST", "/v1/systemone", body, retry=True) + emit({"type": "answers", "workload": w["id"], "answers": json.loads(data)["answers"]}) + for concurrency in args.concurrency: + results, elapsed = run_level(args.url, token, body, args.n, concurrency) + bad = [s for _, _, s in results if s != 200] + for i, (thread, ms, status) in enumerate(results): + emit( + { + "type": "req", + "workload": w["id"], + "concurrency": concurrency, + "i": i, + "thread": thread, + "wall_ms": None if ms is None else round(ms, 3), + "status": status, + "rows": len(w["questions"]), + } + ) + emit( + { + "type": "throughput", + "workload": w["id"], + "concurrency": concurrency, + "n": len(results), + "errors": len(bad), + "elapsed_s": round(elapsed, 3), + "rps": round(len(results) / elapsed, 2), + } + ) + + for w in parity: + _, _, data = client.request("POST", "/v1/systemone", body_for(w, args.model), retry=True) + emit({"type": "answers", "workload": w["id"], "answers": json.loads(data)["answers"]}) + _, _, health_end = client.request("GET", "/health", retry=True) + client.close() + emit({"type": "end", "health": json.loads(health_end), **memory()}) + finally: + for proc in processes.values(): + proc.terminate() + proc.wait(timeout=30) + print(out) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/bench_inproc.py b/benchmarks/laya/bench_inproc.py new file mode 100644 index 0000000..3d3eb59 --- /dev/null +++ b/benchmarks/laya/bench_inproc.py @@ -0,0 +1,154 @@ +"""In-process Laya benchmark (configs C1 = CPU, C2 = MPS). + +Phases are timed separately: import, load, warmup, then warm requests. Every request is one line of +JSONL; report.py turns the file into tables. Run from the repository root or this directory: + + python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m1 +""" + +import argparse +import json +import random +import sys +import time +import warnings +from pathlib import Path + +T_START = time.perf_counter() +warnings.filterwarnings("ignore") +import laya +import torch + +T_IMPORT = time.perf_counter() - T_START + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE)) +from env import footprint_mb, header, noise_problems + + +def load_workloads(path): + with open(path) as f: + return [json.loads(line) for line in f if line.strip()] + + +def sync(device): + if device.type == "mps": + torch.mps.synchronize() + + +def timed_call(agent, workload): + started = time.perf_counter() + result = agent.system_one(workload["state"], workload["questions"]) + sync(agent.device) + return (time.perf_counter() - started) * 1000, result + + +def memory(device): + mem = footprint_mb() + if device.type == "mps": + mem["mps_driver_mb"] = round(torch.mps.driver_allocated_memory() / 2**20) + return mem + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--device", required=True, choices=["cpu", "mps"]) + parser.add_argument("--config", required=True, help="label, e.g. C1 or C2") + parser.add_argument("--run", required=True, help="feasibility, m1, m2, ...") + parser.add_argument("--checkpoint", default="convaiinnovations/laya") + parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) + parser.add_argument("--only", nargs="*", help="bench workload ids to run (default: all)") + parser.add_argument("-n", type=int, default=300, help="timed requests per workload") + parser.add_argument("--discard", type=int, default=20, help="warmup requests per workload") + parser.add_argument("--seed", type=int, default=0, help="workload order seed") + parser.add_argument("--out", default=str(HERE / "results")) + parser.add_argument("--max-load", type=float, default=2.0, help="1-min load average allowed for measured runs") + args = parser.parse_args() + + problems = noise_problems(args.max_load) + if problems and args.run != "feasibility": + sys.exit("refusing a measured run: " + "; ".join(problems)) + for problem in problems: + print(f"warning: {problem}", file=sys.stderr) + + workloads = load_workloads(args.workloads) + bench = [w for w in workloads if w["kind"] == "bench" and (not args.only or w["id"] in args.only)] + parity = [w for w in workloads if w["kind"] == "parity"] + + out = Path(args.out) / f"inproc_{args.config}_{args.device}_{args.run}.jsonl" + out.parent.mkdir(parents=True, exist_ok=True) + common = {"config": args.config, "device_requested": args.device, "run": args.run} + + with open(out, "w") as f: + + def emit(record): + f.write(json.dumps({**common, **record}) + "\n") + + started = time.perf_counter() + agent = laya.load(args.checkpoint, device=args.device) + load_s = time.perf_counter() - started + + emit( + header( + args.checkpoint, + n=args.n, + discard=args.discard, + seed=args.seed, + device_actual=str(agent.device), + weights_dtype=str(next(agent.model.parameters()).dtype), + amp_dtype=str(agent.dtype), + mps_amp_min_rows=getattr(agent, "mps_amp_min_rows", None), + ) + ) + if agent.device.type != args.device: + print(f"warning: asked for {args.device}, laya is on {agent.device}", file=sys.stderr) + + # Warmup: the first call per workload is kept apart, it is the first-shape cost. + started = time.perf_counter() + first_ms = {} + for w in bench: + first_ms[w["id"]], _ = timed_call(agent, w) + for _ in range(args.discard - 1): + timed_call(agent, w) + warmup_s = time.perf_counter() - started + emit( + { + "type": "phase", + "import_s": round(T_IMPORT, 3), + "load_s": round(load_s, 3), + "warmup_s": round(warmup_s, 3), + "first_ms": {k: round(v, 2) for k, v in first_ms.items()}, + "noise": problems, + **memory(agent.device), + } + ) + + order = bench[:] + random.Random(args.seed).shuffle(order) + for w in order: + rows = len(w["questions"]) + for i in range(args.n): + ms, result = timed_call(agent, w) + emit( + { + "type": "req", + "workload": w["id"], + "i": i, + "wall_ms": round(ms, 3), + "tokens": result["usage"]["input_tokens"], + "rows": rows, + } + ) + if i == 0: + emit({"type": "answers", "workload": w["id"], "answers": result["answers"]}) + + for w in parity: + _, result = timed_call(agent, w) + emit({"type": "answers", "workload": w["id"], "answers": result["answers"]}) + + emit({"type": "end", "total_s": round(time.perf_counter() - T_START, 1), **memory(agent.device)}) + print(out) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/check_workloads.py b/benchmarks/laya/check_workloads.py new file mode 100644 index 0000000..f36e8b8 --- /dev/null +++ b/benchmarks/laya/check_workloads.py @@ -0,0 +1,27 @@ +"""T1 check: tokens per row of every workload, measured with Laya's own tokenizer (usage.input_tokens). + +Each question is sent alone so its row length is exact; bench workloads must land within ±10% of target. +""" + +import json +import sys +import warnings +from pathlib import Path + +warnings.filterwarnings("ignore") +import laya + +agent = laya.load("convaiinnovations/laya", device="cpu") +max_len = agent.cfg.get("max_len", 512) +failed = 0 +with open(Path(__file__).resolve().parent / "workloads.jsonl") as f: + workloads = [json.loads(line) for line in f] +for w in workloads: + rows = [agent.system_one(w["state"], {qid: q})["usage"]["input_tokens"] for qid, q in w["questions"].items()] + mean = sum(rows) / len(rows) + target = w["target_tokens_per_row"] + ok = target is None or abs(mean - target) <= 0.1 * target + ok = ok and max(rows) < max_len # a full row means Laya cut the state + failed += not ok + print(f"{'ok ' if ok else 'BAD'} {w['id']:5} rows={len(rows)} tokens/row={rows} mean={mean:.0f} target={target}") +sys.exit(1 if failed else 0) diff --git a/benchmarks/laya/env.py b/benchmarks/laya/env.py new file mode 100644 index 0000000..758e771 --- /dev/null +++ b/benchmarks/laya/env.py @@ -0,0 +1,159 @@ +"""Run header: everything needed to tell whether two result files are comparable. + +Standard library plus whatever the benchmark already imported; every probe degrades to None +instead of failing the run. +""" + +import importlib.metadata +import os +import platform +import subprocess +import sys +from datetime import datetime, timezone +from pathlib import Path + +REPO = Path(__file__).resolve().parents[2] + + +def _run(*cmd): + try: + return subprocess.run(cmd, capture_output=True, text=True, timeout=10, check=True).stdout.strip() + except (OSError, subprocess.SubprocessError): + return None + + +def _version(package): + try: + return importlib.metadata.version(package) + except importlib.metadata.PackageNotFoundError: + return None + + +def _checkpoint_revision(repo_id, ref="main"): + """The commit the cached ref points at. Read directly: laya fetches only the files it needs, so + the snapshot is partial and snapshot_download(local_files_only=True) refuses it.""" + try: + from huggingface_hub.constants import HF_HUB_CACHE + + ref_file = Path(HF_HUB_CACHE) / f"models--{repo_id.replace('/', '--')}" / "refs" / ref + return ref_file.read_text().strip() + except (ImportError, OSError): + return None + + +def _power_source(): + out = _run("pmset", "-g", "batt") + if not out: + return None + first = out.splitlines()[0] + return first.split("'")[1] if "'" in first else first + + +def _gpu_cores(): + out = _run("system_profiler", "SPDisplaysDataType") + for line in (out or "").splitlines(): + if "Total Number of Cores" in line: + return int(line.split(":")[1]) + return None + + +def noise_problems(max_load): + """Reasons this machine is not fit for a measured run, empty when it is.""" + problems = [] + power = _power_source() + if power and power != "AC Power": + problems.append(f"on {power}") + load = os.getloadavg()[0] + if load > max_load: + top = _run("ps", "-Ao", "pcpu=,comm=", "-r") or "" + busiest = "; ".join(" ".join(line.split()[:1] + [line.split("/")[-1]]) for line in top.splitlines()[:3]) + problems.append(f"1-min load {load:.1f} > {max_load} (busiest: {busiest})") + return problems + + +def header(checkpoint, **extra): + status = _run("git", "-C", str(REPO), "status", "--porcelain", "--", ".", ":!benchmarks/laya/results") + return { + "type": "env", + "utc": datetime.now(timezone.utc).isoformat(timespec="seconds"), + "omni_sha": _run("git", "-C", str(REPO), "rev-parse", "HEAD"), + "omni_dirty": bool(status), + "checkpoint": checkpoint, + "checkpoint_revision": _checkpoint_revision(checkpoint), + "laya": _version("laya"), + "torch": _version("torch"), + "transformers": _version("transformers"), + "python": platform.python_version(), + "os": f"macOS {platform.mac_ver()[0]}" if sys.platform == "darwin" else platform.platform(), + "chip": _run("sysctl", "-n", "machdep.cpu.brand_string"), + "cpu_perf_cores": _run("sysctl", "-n", "hw.perflevel0.physicalcpu"), + "cpu_eff_cores": _run("sysctl", "-n", "hw.perflevel1.physicalcpu"), + "gpu_cores": _gpu_cores(), + "mem_gb": round(int(_run("sysctl", "-n", "hw.memsize") or 0) / 2**30), + "power": _power_source(), + "loadavg_1m": round(os.getloadavg()[0], 2), + "argv": sys.argv, + **extra, + } + + +def footprint_mb(pid=None): + """Physical footprint of a process, now and its lifetime peak (MB), from proc_pid_rusage. + + This is Activity Monitor's "Memory" column. On Apple silicon it includes Metal allocations, so it + covers MPS tensors that RSS misses, and it reads the same way for this process and a worker's pid. + """ + import ctypes + + class RusageInfoV4(ctypes.Structure): # , rusage_info_v4 + _fields_ = [("ri_uuid", ctypes.c_uint8 * 16)] + [ + (name, ctypes.c_uint64) + for name in [ + "user_time", + "system_time", + "pkg_idle_wkups", + "interrupt_wkups", + "pageins", + "wired_size", + "resident_size", + "phys_footprint", + "proc_start_abstime", + "proc_exit_abstime", + "child_user_time", + "child_system_time", + "child_pkg_idle_wkups", + "child_interrupt_wkups", + "child_pageins", + "child_elapsed_abstime", + "diskio_bytesread", + "diskio_byteswritten", + "cpu_time_qos_default", + "cpu_time_qos_maintenance", + "cpu_time_qos_background", + "cpu_time_qos_utility", + "cpu_time_qos_legacy", + "cpu_time_qos_user_initiated", + "cpu_time_qos_user_interactive", + "billed_system_time", + "serviced_system_time", + "logical_writes", + "lifetime_max_phys_footprint", + "instructions", + "cycles", + "billed_energy", + "serviced_energy", + "interval_max_phys_footprint", + "runnable_time", + ] + ] + + if sys.platform != "darwin": + return {} + info = RusageInfoV4() + libc = ctypes.CDLL("/usr/lib/libSystem.B.dylib", use_errno=True) + if libc.proc_pid_rusage(pid or os.getpid(), 4, ctypes.byref(info)) != 0: # RUSAGE_INFO_V4 + return {} + return { + "footprint_mb": round(info.phys_footprint / 2**20), + "footprint_peak_mb": round(info.lifetime_max_phys_footprint / 2**20), + } diff --git a/benchmarks/laya/parity.py b/benchmarks/laya/parity.py new file mode 100644 index 0000000..4fb998f --- /dev/null +++ b/benchmarks/laya/parity.py @@ -0,0 +1,112 @@ +"""Compare every run's answers with a reference run, using the tolerances declared in advance. + + python benchmarks/laya/parity.py benchmarks/laya/results/*.jsonl --ref C1 + +Per question: the decision must match (choice: chosen option; score: most likely level; noul: side of +0.5) and the largest absolute probability difference must stay within tolerance: 1e-3 when the request +ran in fp32, 1e-2 when it ran under fp16 autocast (MPS and rows >= the worker's amp threshold). +A flipped decision is reported with the reference margin between its top two outcomes. +Exits 1 when any question fails. +""" + +import argparse +import json +import sys + +TOLERANCE = {"fp32": 1e-3, "fp16": 1e-2} + + +def read(paths): + envs, answers = {}, {} + for path in paths: + with open(path) as f: + for line in f: + if not line.strip(): + continue + r = json.loads(line) + key = (r["config"], r["run"]) + if r["type"] == "env": + envs[key] = r + elif r["type"] == "answers": + answers.setdefault(key, {})[r["workload"]] = r["answers"] + return envs, answers + + +def outcome(answer): + """(decision, {outcome: probability}) for one question's answer.""" + kind = answer["type"] + if kind == "noul": + p = answer["noul"] + return p >= 0.5, {"yes": p} + probs = answer["probabilities"] + if kind == "choice": + return answer["choice"], probs + return max(probs, key=probs.get), probs # score: most likely level + + +def margin(probs): + if len(probs) == 1: # noul: distance from the 0.5 boundary + return abs(next(iter(probs.values())) - 0.5) + top = sorted(probs.values(), reverse=True) + return top[0] - top[1] + + +def path(env, rows): + fp16 = ( + env.get("device_actual") == "mps" + and env.get("amp_dtype") == "torch.float16" + and rows >= (env.get("mps_amp_min_rows") or 10**9) + ) + return "fp16" if fp16 else "fp32" + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("files", nargs="+") + parser.add_argument("--ref", default="C1", help="reference config") + parser.add_argument("--ref-run", help="reference run label (default: the first one found)") + args = parser.parse_args() + envs, answers = read(args.files) + + refs = sorted(k for k in answers if k[0] == args.ref and (not args.ref_run or k[1] == args.ref_run)) + if not refs: + sys.exit(f"no answers for reference config {args.ref}") + ref_key = refs[0] + reference = answers[ref_key] + + print(f"Reference: {ref_key[0]} / {ref_key[1]}. Tolerances: fp32 {TOLERANCE['fp32']}, fp16 {TOLERANCE['fp16']}.\n") + print( + "| config | run | workload | question | type | path | decision ref → run | max abs Δp | ref margin | result |" + ) + print("|---|---|---|---|---|---|---|---|---|---|") + failed = total = 0 + for key in sorted(answers): + if key == ref_key: + continue + env = envs.get(key, {}) + for workload, questions in sorted(reference.items()): + got = answers[key].get(workload) + if got is None: + print(f"| {key[0]} | {key[1]} | {workload} | | | | missing | | | FAIL |") + failed += 1 + total += 1 + continue + precision = path(env, len(questions)) + for qid, ref_answer in sorted(questions.items()): + total += 1 + ref_decision, ref_probs = outcome(ref_answer) + decision, probs = outcome(got[qid]) + delta = max(abs(ref_probs.get(o, 0.0) - probs.get(o, 0.0)) for o in ref_probs.keys() | probs.keys()) + ok = decision == ref_decision and delta <= TOLERANCE[precision] + failed += not ok + flip = f"{ref_decision} → {decision}" if decision != ref_decision else f"{decision}" + print( + f"| {key[0]} | {key[1]} | {workload} | {qid} | {ref_answer['type']} | {precision} | {flip} " + f"| {delta:.4f} | {margin(ref_probs):.4f} | {'PASS' if ok else 'FAIL'} |" + ) + print(f"\n{total - failed}/{total} questions within tolerance.") + sys.exit(1 if failed else 0) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/profile_mps.py b/benchmarks/laya/profile_mps.py new file mode 100644 index 0000000..28d2183 --- /dev/null +++ b/benchmarks/laya/profile_mps.py @@ -0,0 +1,257 @@ +"""Where a Laya request's time goes on MPS (spec R2). Wraps laya's stages on the loaded instance; laya +itself is not modified. + +Three measurements: + +1. sweep: one choice question, state length swept; fits wall = a + b * tokens. A large `a` relative + to a short request means fixed per-request cost (dispatch, Python, sync) dominates. +2. stages: per request, time in encode (tokenize and build sequences), collate, host dispatch of the + forward (the call returns once kernels are queued), waiting for the GPU after dispatch, copy back, + and decode; GPU execution time from MPS events. Every stage runs synchronously in order, so the + stages add up to the request; "other" is what the wrappers do not cover. +3. ops: torch.profiler CPU trace of the forward: operator calls per request and the top operators by + self CPU time, i.e. the host cost of issuing the forward. + + python benchmarks/laya/profile_mps.py --run feasibility +""" + +import argparse +import json +import statistics +import sys +import time +import warnings +from collections import defaultdict +from pathlib import Path + +warnings.filterwarnings("ignore") +import laya +import laya.agent +import torch + +HERE = Path(__file__).resolve().parent +sys.path.insert(0, str(HERE)) +from env import header, noise_problems + +STAGES = ["encode", "collate", "dispatch", "gpu_wait", "copy_back", "decode", "other"] + + +class StageTimer: + """Times laya's request stages by wrapping them on one Agent instance.""" + + def __init__(self, agent): + self.agent = agent + self.current = None + self.use_events = agent.device.type == "mps" + self._install() + + def _add(self, stage, ms): + if self.current is not None: + self.current[stage] = self.current.get(stage, 0.0) + ms + + def _timed(self, stage, fn): + def wrapper(*args, **kwargs): + started = time.perf_counter() + try: + return fn(*args, **kwargs) + finally: + self._add(stage, (time.perf_counter() - started) * 1000) + + return wrapper + + def _install(self): + agent = self.agent + agent._encode_state = self._timed("encode", agent._encode_state) + agent._decode_answers = self._timed("decode", agent._decode_answers) + laya.agent.collate_items = self._timed("collate", laya.agent.collate_items) + infer = agent._infer + + def forward(b): + # Replaces Agent._forward: same result, with dispatch, GPU wait and copy back split apart. + start_event = end_event = None + if self.use_events: + start_event = torch.mps.Event(enable_timing=True) + end_event = torch.mps.Event(enable_timing=True) + start_event.record() + t0 = time.perf_counter() + logits, act = infer(b) + if self.use_events: + end_event.record() + t1 = time.perf_counter() + if agent.device.type == "mps": + torch.mps.synchronize() + t2 = time.perf_counter() + out = logits.float().cpu().numpy(), torch.softmax(act.float(), -1).cpu().numpy() + t3 = time.perf_counter() + self._add("dispatch", (t1 - t0) * 1000) + self._add("gpu_wait", (t2 - t1) * 1000) + self._add("copy_back", (t3 - t2) * 1000) + if self.use_events: + self._add("gpu_exec", start_event.elapsed_time(end_event)) + return out + + agent._forward = forward + + def request(self, state, questions): + self.current = {} + started = time.perf_counter() + result = self.agent.system_one(state, questions) + if self.agent.device.type == "mps": + torch.mps.synchronize() + stages, self.current = self.current, None + stages["wall"] = (time.perf_counter() - started) * 1000 + stages["other"] = stages["wall"] - sum(stages.get(s, 0.0) for s in STAGES if s != "other") + return stages, result + + +def fit(xs, ys): + """Least squares y = a + b x, with R^2.""" + mx, my = statistics.fmean(xs), statistics.fmean(ys) + sxx = sum((x - mx) ** 2 for x in xs) + b = sum((x - mx) * (y - my) for x, y in zip(xs, ys)) / sxx + a = my - b * mx + ss_res = sum((y - (a + b * x)) ** 2 for x, y in zip(xs, ys)) + ss_tot = sum((y - my) ** 2 for y in ys) + return a, b, 1 - ss_res / ss_tot if ss_tot else 1.0 + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--run", required=True) + parser.add_argument("--device", default="mps", choices=["mps", "cpu"]) + parser.add_argument("--checkpoint", default="convaiinnovations/laya") + parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) + parser.add_argument("--stage-workloads", nargs="+", default=["W1", "W3", "W5"]) + parser.add_argument("-n", type=int, default=100, help="timed requests per point") + parser.add_argument("--discard", type=int, default=10) + parser.add_argument("--sweep-words", type=int, nargs="+", default=[1, 8, 24, 56, 120, 250, 380]) + parser.add_argument("--ops-requests", type=int, default=20) + parser.add_argument("--out", default=str(HERE / "results")) + parser.add_argument("--max-load", type=float, default=2.0) + args = parser.parse_args() + + problems = noise_problems(args.max_load) + if problems and args.run != "feasibility": + sys.exit("refusing a measured run: " + "; ".join(problems)) + for problem in problems: + print(f"warning: {problem}", file=sys.stderr) + + with open(args.workloads) as f: + workloads = {w["id"]: w for w in (json.loads(line) for line in f if line.strip())} + route = workloads["W1"]["questions"] + + agent = laya.load(args.checkpoint, device=args.device) + timer = StageTimer(agent) + out = Path(args.out) / f"profile_{args.device}_{args.run}.jsonl" + out.parent.mkdir(parents=True, exist_ok=True) + common = {"config": f"profile-{args.device}", "run": args.run} + + with open(out, "w") as f: + + def emit(record): + f.write(json.dumps({**common, **record}) + "\n") + + emit( + header( + args.checkpoint, + device_actual=str(agent.device), + noise=problems, + weights_dtype=str(next(agent.model.parameters()).dtype), + amp_dtype=str(agent.dtype), + mps_amp_min_rows=getattr(agent, "mps_amp_min_rows", None), + ) + ) + + # 1. Length sweep. + print("## Length sweep (1 choice question, 5 options)\n") + print("| words | tokens | p50 ms |\n|---|---|---|") + points = [] + for words in args.sweep_words: + state = " ".join(["delivery"] * words) + for _ in range(args.discard): + timer.request(state, route) + walls, tokens = [], None + for _ in range(args.n): + stages, result = timer.request(state, route) + walls.append(stages["wall"]) + tokens = result["usage"]["input_tokens"] + p50 = statistics.median(walls) + points.append((tokens, p50)) + emit( + { + "type": "sweep", + "words": words, + "tokens": tokens, + "p50_ms": round(p50, 3), + "wall_ms": [round(w, 3) for w in walls], + } + ) + print(f"| {words} | {tokens} | {p50:.1f} |") + a, b, r2 = fit([t for t, _ in points], [p for _, p in points]) + emit({"type": "fit", "a_ms": round(a, 3), "b_ms_per_token": round(b, 5), "r2": round(r2, 4)}) + print(f"\nwall ≈ {a:.1f} ms + {b:.3f} ms/token × tokens (R² {r2:.3f})") + + # 2. Stage split. + print("\n## Stages (median ms per request)\n") + columns = ["wall", *STAGES, "gpu_exec"] + print("| workload | tokens | rows | " + " | ".join(columns) + " |\n|" + "---|" * (len(columns) + 3)) + for wid in args.stage_workloads: + w = workloads[wid] + for _ in range(args.discard): + timer.request(w["state"], w["questions"]) + per_stage, tokens = defaultdict(list), None + for _ in range(args.n): + stages, result = timer.request(w["state"], w["questions"]) + tokens = result["usage"]["input_tokens"] + for k, v in stages.items(): + per_stage[k].append(v) + medians = {k: statistics.median(v) for k, v in per_stage.items()} + emit( + { + "type": "stages", + "workload": wid, + "tokens": tokens, + "rows": len(w["questions"]), + "median_ms": {k: round(v, 3) for k, v in medians.items()}, + "samples": {k: [round(x, 3) for x in v] for k, v in per_stage.items()}, + } + ) + cells = " | ".join(f"{medians[c]:.1f}" if c in medians else "" for c in columns) + print(f"| {wid} | {tokens} | {len(w['questions'])} | {cells} |") + + # 3. Operators on the host. + print("\n## Host operators for W1 (torch.profiler, CPU)\n") + w = workloads["W1"] + with torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CPU]) as prof: + for _ in range(args.ops_requests): + timer.request(w["state"], w["questions"]) + events = [e for e in prof.key_averages() if e.key.startswith("aten::")] + calls = sum(e.count for e in events) / args.ops_requests + top = sorted(events, key=lambda e: e.self_cpu_time_total, reverse=True)[:12] + emit( + { + "type": "ops", + "workload": "W1", + "requests": args.ops_requests, + "aten_calls_per_request": calls, + "top": [ + { + "op": e.key, + "calls_per_request": e.count / args.ops_requests, + "self_cpu_ms_per_request": e.self_cpu_time_total / 1000 / args.ops_requests, + } + for e in top + ], + } + ) + print(f"aten calls per request: {calls:.0f}\n") + print("| op | calls/request | self CPU ms/request |\n|---|---|---|") + for e in top: + print( + f"| {e.key} | {e.count / args.ops_requests:.0f} | {e.self_cpu_time_total / 1000 / args.ops_requests:.2f} |" + ) + print(f"\n{out}") + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/report.py b/benchmarks/laya/report.py new file mode 100644 index 0000000..cb1c9ee --- /dev/null +++ b/benchmarks/laya/report.py @@ -0,0 +1,189 @@ +"""Turn benchmark JSONL into markdown tables. The only place numbers are computed from raw data. + + python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl + +Percentiles are nearest-rank. The run-to-run gate compares p50 across measured runs (every run whose +label is not "feasibility") of the same config, workload and concurrency: (max - min) / min <= 10%. +In-process results have concurrency 1. +""" + +import argparse +import json +import math +import statistics +from collections import defaultdict + +GATE = 0.10 +PHASES = ["import_s", "load_s", "process_to_ready_s", "warmup_s"] # whichever a result file has + + +def percentile(sorted_values, p): + return sorted_values[max(0, math.ceil(p * len(sorted_values)) - 1)] + + +def read(paths): + records = [] + for path in paths: + with open(path) as f: + records.extend(json.loads(line) for line in f if line.strip()) + return records + + +def table(headers, rows): + lines = ["| " + " | ".join(headers) + " |", "|" + "---|" * len(headers)] + lines += ["| " + " | ".join("" if c is None else str(c) for c in row) + " |" for row in rows] + return "\n".join(lines) + + +def short(sha, dirty=False): + return (sha or "?")[:7] + ("+dirty" if dirty else "") + + +def environment(envs): + rows = [] + for k, e in sorted(envs.items()): + dtype = f"{e['weights_dtype']} (amp {e['amp_dtype']}, >= {e['mps_amp_min_rows']} rows)" + if e.get("dtype_source"): + dtype += f" [{e['dtype_source']}]" + rows.append( + [ + *k, + e["device_actual"], + dtype, + e["chip"], + e["os"], + e["power"], + e["loadavg_1m"], + f"laya {e['laya']} / torch {e['torch']}", + short(e["checkpoint_revision"]), + short(e["omni_sha"], e["omni_dirty"]), + ] + ) + headers = [ + "config", + "run", + "device", + "weights (autocast)", + "chip", + "os", + "power", + "load 1m", + "versions", + "ckpt", + "omni", + ] + return table(headers, rows) + + +def phase_table(phases): + present = [p for p in PHASES if any(ph.get(p) is not None for ph in phases.values())] + workload_ids = sorted({w for ph in phases.values() for w in ph["first_ms"]}) + rows = [ + [*k, *(ph.get(p) for p in present), *(ph["first_ms"].get(w) for w in workload_ids)] + for k, ph in sorted(phases.items()) + ] + headers = ["config", "run", *(p.removesuffix("_s") for p in present), *(f"first {w}" for w in workload_ids)] + return table(headers, rows) + + +def latency(records): + samples = defaultdict(list) + for r in records: + if r["type"] == "req" and r.get("status", 200) == 200: + samples[(r["config"], r["workload"], r.get("concurrency", 1), r["run"])].append(r["wall_ms"]) + rows, p50s = [], defaultdict(dict) + for (config, workload, concurrency, run), values in sorted(samples.items()): + values.sort() + mean = statistics.fmean(values) + cv = statistics.stdev(values) / mean if len(values) > 1 else 0.0 + p50 = percentile(values, 0.50) + if run != "feasibility": + p50s[(config, workload, concurrency)][run] = p50 + rows.append( + [ + config, + workload, + concurrency, + run, + len(values), + round(p50, 2), + round(percentile(values, 0.95), 2), + round(mean, 2), + f"{cv:.1%}", + ] + ) + return table(["config", "workload", "conc", "run", "n", "p50", "p95", "mean", "CV"], rows), p50s + + +def gate(p50s): + rows = [] + for (config, workload, concurrency), by_run in sorted(p50s.items()): + if len(by_run) < 2: + rows.append([config, workload, concurrency, len(by_run), "", "needs 2 measured runs"]) + continue + spread = (max(by_run.values()) - min(by_run.values())) / min(by_run.values()) + rows.append([config, workload, concurrency, len(by_run), f"{spread:.1%}", "PASS" if spread <= GATE else "FAIL"]) + return table(["config", "workload", "conc", "runs", "spread", "gate"], rows) + + +def throughput(records): + rows = [ + [r["config"], r["workload"], r["concurrency"], r["run"], r["n"], r["errors"], r["elapsed_s"], r["rps"]] + for r in records + if r["type"] == "throughput" + ] + return table(["config", "workload", "conc", "run", "n", "errors", "elapsed s", "req/s"], sorted(rows)) + + +def memory(phases, ends): + """Physical footprint (Activity Monitor's "Memory", includes Metal allocations) of the process running + Laya: the benchmark itself in-process, the worker over HTTP. Peak is the process lifetime maximum.""" + rows = [] + for k in sorted(ends): + ph, e = phases.get(k, {}), ends[k] + rows.append( + [ + *k, + ph.get("footprint_mb"), + e.get("footprint_mb"), + e.get("footprint_peak_mb"), + ph.get("mps_driver_mb"), + e.get("mps_driver_mb"), + e.get("frontend_footprint_mb"), + ] + ) + headers = [ + "config", + "run", + "footprint after warmup", + "footprint end", + "footprint peak", + "MPS driver after warmup", + "MPS driver end", + "frontend footprint", + ] + return table(headers, rows) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("files", nargs="+") + args = parser.parse_args() + records = read(args.files) + + def by_run(kind): + return {(r["config"], r["run"]): r for r in records if r["type"] == kind} + + envs, phases, ends = by_run("env"), by_run("phase"), by_run("end") + latency_table, p50s = latency(records) + print("## Environment\n\n" + environment(envs)) + print("\n## Phases (s) and first request per workload (ms)\n\n" + phase_table(phases)) + print("\n## Warm latency (ms)\n\n" + latency_table) + print(f"\n## Run-to-run gate (p50 spread across measured runs <= {GATE:.0%})\n\n" + gate(p50s)) + if any(r["type"] == "throughput" for r in records): + print("\n## Throughput\n\n" + throughput(records)) + print("\n## Memory (MB)\n\n" + memory(phases, ends)) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/results/.gitignore b/benchmarks/laya/results/.gitignore new file mode 100644 index 0000000..1c54c95 --- /dev/null +++ b/benchmarks/laya/results/.gitignore @@ -0,0 +1,5 @@ +# Committed: measured runs (m1, m2, ...) and the reports built from them. +# Not committed: feasibility runs, worker/frontend logs, runner state. +*feasibility* +*.log +done_* diff --git a/benchmarks/laya/workloads.jsonl b/benchmarks/laya/workloads.jsonl new file mode 100644 index 0000000..12f444b --- /dev/null +++ b/benchmarks/laya/workloads.jsonl @@ -0,0 +1,11 @@ +{"id": "W1", "kind": "bench", "target_tokens_per_row": 68, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "Shipping, delivery and tracking", "payment": "Charges, invoices and refunds", "returns": "Returns and exchanges", "account": "Login, password and profile", "human": "Anything else"}}}} +{"id": "W2", "kind": "bench", "target_tokens_per_row": 200, "state": "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "Shipping, delivery and tracking", "payment": "Charges, invoices and refunds", "returns": "Returns and exchanges", "account": "Login, password and profile", "human": "Anything else"}}}} +{"id": "W3", "kind": "bench", "target_tokens_per_row": 480, "state": "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened. Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened. Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "Shipping, delivery and tracking", "payment": "Charges, invoices and refunds", "returns": "Returns and exchanges", "account": "Login, password and profile", "human": "Anything else"}}}} +{"id": "W4", "kind": "bench", "target_tokens_per_row": 54, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "Shipping, delivery and tracking", "payment": "Charges, invoices and refunds", "returns": "Returns and exchanges", "account": "Login, password and profile", "human": "Anything else"}}, "urgency": {"type": "score", "instructions": "How urgent is the ticket?", "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"]}, "refund": {"type": "noul", "instructions": "Does the customer ask for a refund?"}}} +{"id": "W5", "kind": "bench", "target_tokens_per_row": 49, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "Shipping, delivery and tracking", "payment": "Charges, invoices and refunds", "returns": "Returns and exchanges", "account": "Login, password and profile", "human": "Anything else"}}, "urgency": {"type": "score", "instructions": "How urgent is the ticket?", "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"]}, "refund": {"type": "noul", "instructions": "Does the customer ask for a refund?"}, "angry": {"type": "noul", "instructions": "Is the customer angry?"}, "cancel": {"type": "noul", "instructions": "Does the customer want to cancel the order?"}, "lang": {"type": "choice", "instructions": "Which language is the ticket written in?", "criteria": {"en": "English", "de": "German", "fr": "French"}}}} +{"id": "W6", "kind": "bench", "target_tokens_per_row": 47, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"refund": {"type": "noul", "instructions": "Does the customer ask for a refund?"}}} +{"id": "P2", "kind": "parity", "target_tokens_per_row": null, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "The logistics team", "payment": "The payment team"}}}} +{"id": "P5", "kind": "parity", "target_tokens_per_row": null, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "The logistics team", "payment": "The payment team", "returns": "The returns team", "account": "The account team", "human": "The human team"}}}} +{"id": "P10", "kind": "parity", "target_tokens_per_row": null, "state": "My package never arrived and tracking has not updated in ten days.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "The logistics team", "payment": "The payment team", "returns": "The returns team", "account": "The account team", "human": "The human team", "billing": "The billing team", "technical": "The technical team", "sales": "The sales team", "legal": "The legal team", "security": "The security team"}}}} +{"id": "P2L", "kind": "parity", "target_tokens_per_row": null, "state": "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "The logistics team", "payment": "The payment team"}}, "urgency": {"type": "score", "instructions": "How urgent is the ticket?", "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"]}, "refund": {"type": "noul", "instructions": "Does the customer ask for a refund?"}}} +{"id": "P10L", "kind": "parity", "target_tokens_per_row": null, "state": "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. The confirmation email said the parcel would arrive within two business days, but the tracking page has shown 'label created' ever since. I contacted the courier and they told me they never received the parcel from your warehouse. In the meantime I was charged twice on my credit card, once for the original amount and once for a slightly different amount that I do not recognise. I need the shoes for a race next weekend, so please either ship them today with a tracking number that actually works or cancel the order and refund both charges. I have been a customer for years and this is the first time something like this has happened.", "questions": {"route": {"type": "choice", "instructions": "Route the ticket to the queue that owns it.", "criteria": {"logistics": "The logistics team", "payment": "The payment team", "returns": "The returns team", "account": "The account team", "human": "The human team", "billing": "The billing team", "technical": "The technical team", "sales": "The sales team", "legal": "The legal team", "security": "The security team"}}, "urgency": {"type": "score", "instructions": "How urgent is the ticket?", "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"]}, "refund": {"type": "noul", "instructions": "Does the customer ask for a refund?"}}} diff --git a/benchmarks/laya/workloads.src.py b/benchmarks/laya/workloads.src.py new file mode 100644 index 0000000..92f5981 --- /dev/null +++ b/benchmarks/laya/workloads.src.py @@ -0,0 +1,90 @@ +"""Writes workloads.jsonl. Edit here, not the JSONL: the text is fixed so runs stay comparable.""" + +import json +from pathlib import Path + +ROUTE = { + "type": "choice", + "instructions": "Route the ticket to the queue that owns it.", + "criteria": { + "logistics": "Shipping, delivery and tracking", + "payment": "Charges, invoices and refunds", + "returns": "Returns and exchanges", + "account": "Login, password and profile", + "human": "Anything else", + }, +} +URGENCY = { + "type": "score", + "instructions": "How urgent is the ticket?", + "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"], +} +REFUND = {"type": "noul", "instructions": "Does the customer ask for a refund?"} +ANGRY = {"type": "noul", "instructions": "Is the customer angry?"} +CANCEL = {"type": "noul", "instructions": "Does the customer want to cancel the order?"} +LANG = { + "type": "choice", + "instructions": "Which language is the ticket written in?", + "criteria": {"en": "English", "de": "German", "fr": "French"}, +} + +SHORT = "My package never arrived and tracking has not updated in ten days." +LONG = ( + "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. " + "The confirmation email said the parcel would arrive within two business days, but the tracking " + "page has shown 'label created' ever since. I contacted the courier and they told me they never " + "received the parcel from your warehouse. In the meantime I was charged twice on my credit card, " + "once for the original amount and once for a slightly different amount that I do not recognise. " + "I need the shoes for a race next weekend, so please either ship them today with a tracking number " + "that actually works or cancel the order and refund both charges. I have been a customer for years " + "and this is the first time something like this has happened." +) +NEAR = " ".join([LONG] * 3) + + +def options(n): + names = [ + "logistics", + "payment", + "returns", + "account", + "human", + "billing", + "technical", + "sales", + "legal", + "security", + ] + return { + "type": "choice", + "instructions": ROUTE["instructions"], + "criteria": {k: f"The {k} team" for k in names[:n]}, + } + + +W = [ + # id, kind, state, questions, target tokens per row, measured with check_workloads.py + ("W1", "bench", SHORT, {"route": ROUTE}, 68), + ("W2", "bench", LONG, {"route": ROUTE}, 200), + ("W3", "bench", NEAR, {"route": ROUTE}, 480), + ("W4", "bench", SHORT, {"route": ROUTE, "urgency": URGENCY, "refund": REFUND}, 54), + ( + "W5", + "bench", + SHORT, + {"route": ROUTE, "urgency": URGENCY, "refund": REFUND, "angry": ANGRY, "cancel": CANCEL, "lang": LANG}, + 49, + ), + ("W6", "bench", SHORT, {"refund": REFUND}, 47), + ("P2", "parity", SHORT, {"route": options(2)}, None), + ("P5", "parity", SHORT, {"route": options(5)}, None), + ("P10", "parity", SHORT, {"route": options(10)}, None), + ("P2L", "parity", LONG, {"route": options(2), "urgency": URGENCY, "refund": REFUND}, None), + ("P10L", "parity", LONG, {"route": options(10), "urgency": URGENCY, "refund": REFUND}, None), +] + +with open(Path(__file__).resolve().parent / "workloads.jsonl", "w") as f: + f.writelines( + json.dumps({"id": wid, "kind": kind, "target_tokens_per_row": target, "state": state, "questions": qs}) + "\n" + for wid, kind, state, qs, target in W + ) diff --git a/recipe/README.md b/recipe/README.md index 4d0bf6c..762b191 100644 --- a/recipe/README.md +++ b/recipe/README.md @@ -2,6 +2,8 @@ - [Laya text worker](laya/README.md): start the external Python worker, connect the Rust frontend and compare direct and proxied responses. +- [Laya on Apple Silicon](laya/apple-silicon.md): serve Laya on the Mac GPU with the Laya + worker, put the frontend in front of it and run the benchmarks. Recipes contain setup, launch commands and examples. Reusable implementation code belongs under `src/`. diff --git a/recipe/laya/README.md b/recipe/laya/README.md index b4f0497..254c12c 100644 --- a/recipe/laya/README.md +++ b/recipe/laya/README.md @@ -3,7 +3,8 @@ This recipe runs the external Laya Python package behind the Rust frontend. It validates text decisions; image, audio and video inference are not covered. -Run all commands from the repository root. +Run all commands from the repository root. To serve on the GPU of an Apple Silicon Mac, see +[Laya on Apple Silicon](apple-silicon.md). ## Start the worker diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md new file mode 100644 index 0000000..dd21bb5 --- /dev/null +++ b/recipe/laya/apple-silicon.md @@ -0,0 +1,123 @@ +# Laya on Apple Silicon + +This recipe serves Laya on the GPU of an Apple Silicon Mac (PyTorch MPS) with the Laya worker +from [`src/models/laya/`](../../src/models/laya/), puts the Rust frontend in front of it and runs +the benchmark suite. The [Laya text worker](README.md) recipe covers the plain CPU setup. + +Validated on an M1 Pro (16 GB, 16-core GPU), macOS 26.1, Python 3.12, `laya[serve]==0.3.20`, +torch 2.14.0 and the `english` checkpoint (`convaiinnovations/laya` at `55cf4c4`). Other M-series +Macs have not been tested. + +Run all commands from the repository root. + +## Install + +Use Python 3.12. If `python3.12` is not on your `PATH`, install it with `brew install python@3.12` +or `uv python install 3.12`. + +```sh +python3.12 -m venv .venv +.venv/bin/python -m pip install 'laya[serve]==0.3.20' pytest +.venv/bin/python -c "import torch; print(torch.backends.mps.is_available())" +``` + +The last command must print `True`. The standard macOS arm64 wheel of torch includes MPS. + +## Start the worker + +```sh +LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps LAYA_MODELS=english \ +LAYA_REQUIRE_DEVICE=1 \ + .venv/bin/python src/models/laya/worker.py +``` + +First startup downloads the checkpoint (about 850 MB). The worker loads the model, runs a warmup +over short, long and multi-question requests, and only then listens on port 8000, so the first +request it accepts is already warm. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently +serving on the CPU when the model cannot be placed on MPS. + +Check what it is running on: + +```sh +curl -s http://127.0.0.1:8000/health +``` + +`device` must be `mps` and `device_mismatch` `false`. The response also names the checkpoint and +revision, the weight dtype (`torch.float32`; Laya upcasts the fp16 checkpoint on MPS), the autocast +dtype Laya uses for requests with at least `mps_amp_min_rows` questions, and the warmup time. + +### Compiled one-question path + +```sh +LAYA_WORKER_COMPILE=single LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ +LAYA_MODELS=english LAYA_REQUIRE_DEVICE=1 \ + .venv/bin/python src/models/laya/worker.py +``` + +`single` sends requests with one question through a `torch.compile` graph and runs the rest eagerly. +In two measured runs on the M1 Pro it cut warm p50 for a 68-token one-question request from about +46 ms to 33 ms (−29%) and for a 47-token one from about 39 ms to 26 ms, left three- and six-question +requests within about 3%, and gave the same answers as CPU within the benchmark tolerances. The price is +startup: the worker was ready after about 22 s instead of 8 s while the warmup compiles, and the +gain shrinks with length (−7% to −13% at 484 tokens). + +`/health` reports under `compile` how many graphs existed when the worker became ready and how many +exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. `all` +compiles every path; in feasibility runs it made multi-question requests up to 65% slower and took +over a minute to start, so it is not recommended. + +## Start the frontend + +In another terminal: + +```sh +cargo build --release --locked +OMNI_JEV_BIND=127.0.0.1:8080 \ +OMNI_JEV_BACKEND_URL=http://127.0.0.1:8000 \ + ./target/release/omni-jev +``` + +## Send a request + +```sh +curl http://127.0.0.1:8080/v1/systemone \ + -H 'Content-Type: application/json' \ + -d '{"model":"english","state":"Please refund the duplicate charge.","questions":{"refund":{"type":"noul","instructions":"Does the customer ask for a refund?"}}}' +``` + +The frontend forwards the worker's response unchanged; `compare_with_backend.py` from the +[Laya text worker](README.md#compare-responses) recipe checks that against this setup as well. + +## Test + +```sh +.venv/bin/python -m pytest src/models/laya/tests # unit tests, no model +LAYA_CONTRACT=1 .venv/bin/python -m pytest src/models/laya/tests # plus contract tests against a CPU worker +``` + +## Benchmark + +Stop the worker and frontend first; the benchmark starts its own. The suite and its measurement +rules are described in [`benchmarks/laya/`](../../benchmarks/laya/README.md). A first pass that +checks everything runs: + +```sh +.venv/bin/python benchmarks/laya/check_workloads.py +.venv/bin/python benchmarks/laya/bench_inproc.py --device mps --config C2 --run feasibility +.venv/bin/python benchmarks/laya/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve +.venv/bin/python benchmarks/laya/bench_http.py --config C4 --run feasibility \ + --url http://127.0.0.1:8080 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve +.venv/bin/python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl +``` + +Runs labelled anything other than `feasibility` refuse to start on battery power or when the +1-minute load average is above 2, so close other heavy applications and plug the Mac in first. + +## Troubleshooting + +- `device_mismatch: true`, or the worker exits with `asked for mps, model is on cpu`: MPS is not + available to this Python. Check the `torch.backends.mps.is_available()` line above; an x86_64 + Python running under Rosetta cannot use MPS. +- The worker process uses about 4 GB (Activity Monitor's Memory column, which counts MPS + allocations). On a 16 GB Mac, close other large applications before benchmarking. +- `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/src/models/laya/README.md b/src/models/laya/README.md index ec10b58..2599277 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -4,4 +4,28 @@ LAYA is the first planned System1-Omni model. This directory owns its complete r GPU operations and kernel implementations belong in [`backends/cuda/`](../../backends/cuda/) and [`backends/metal/`](../../backends/metal/). Setup and usage examples belong in the top-level [`recipe/`](../../../recipe/) directory. -Status: planned; no model implementation or validated GPU backend support yet. +Status: the Python worker below serves LAYA through laya-serve on CPU and Apple Silicon (PyTorch MPS, +validated on an M1 Pro). No native CUDA or Metal backend yet. + +## Worker + +`worker.py` runs laya-serve (`laya[serve]==0.3.20`) with two changes: + +- It binds only after a warmup over short, long and multi-question requests, so `/health` never + answers for a worker that has not run a forward pass. Without it the first request after `/health` + returned 200 took ~230 ms against ~45 ms warm on an M1 Pro (MPS); with it, ~73 ms. +- `/health` reports the device, weight and autocast dtypes of the loaded model, the checkpoint and + revision, and `device_mismatch` when the model is not on the device `LAYA_DEVICE` asked for. + laya-serve reports `LAYA_DEVICE` as configured, and laya falls back to CPU with only a printed + warning. `LAYA_REQUIRE_DEVICE=1` makes the worker exit instead. + +Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, `LAYA_API_KEY`, ...), +plus `LAYA_WORKER_COMPILE=off|single|all`: `single` sends one-question requests through a +`torch.compile(dynamic=True)` graph compiled during warmup and runs the rest eagerly; `/health` reports +compiled graphs at readiness and now. See the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). + +```sh +LAYA_DEVICE=mps LAYA_MODELS=english python src/models/laya/worker.py +python -m pytest src/models/laya/tests # unit tests, no model +LAYA_CONTRACT=1 python -m pytest src/models/laya/tests # plus contract tests on CPU, loads the checkpoint +``` diff --git a/src/models/laya/tests/test_contract.py b/src/models/laya/tests/test_contract.py new file mode 100644 index 0000000..cd01468 --- /dev/null +++ b/src/models/laya/tests/test_contract.py @@ -0,0 +1,166 @@ +"""Contract tests against a real worker process on CPU. They load the Laya checkpoint, so they only run +with LAYA_CONTRACT=1: + + LAYA_CONTRACT=1 python -m pytest src/models/laya/tests/test_contract.py +""" + +import http.client +import json +import os +import socket +import statistics +import subprocess +import sys +import time +from pathlib import Path + +import pytest + +pytestmark = pytest.mark.skipif(os.environ.get("LAYA_CONTRACT") != "1", reason="set LAYA_CONTRACT=1") + +WORKER = Path(__file__).resolve().parents[1] / "worker.py" +TOKEN = "contract-test-token" +STATE = "I was charged twice for my order. Please refund the duplicate today." +CHOICE = { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": {"billing": "Charges and refunds", "technical": "Software problems"}, +} +SCORE = { + "type": "score", + "instructions": "How urgent is the request?", + "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"], +} +NOUL = {"type": "noul", "instructions": "Does the customer ask for a refund?"} + + +def free_port(): + with socket.socket() as s: + s.bind(("127.0.0.1", 0)) + return s.getsockname()[1] + + +def call(port, method, path, body=None, token=TOKEN, raw=False): + conn = http.client.HTTPConnection("127.0.0.1", port, timeout=120) + headers = {"Content-Type": "application/json"} + if token: + headers["Authorization"] = f"Bearer {token}" + data = body if raw or body is None else json.dumps(body).encode() + started = time.perf_counter() + conn.request(method, path, body=data, headers=headers) + response = conn.getresponse() + payload = response.read() + ms = (time.perf_counter() - started) * 1000 + conn.close() + return response.status, payload, ms + + +def decide(port, questions, **kwargs): + return call(port, "POST", "/v1/systemone", {"model": "english", "state": STATE, "questions": questions}, **kwargs) + + +@pytest.fixture(scope="module") +def worker(): + port = free_port() + env = { + **os.environ, + "LAYA_HOST": "127.0.0.1", + "LAYA_PORT": str(port), + "LAYA_DEVICE": "cpu", + "LAYA_MODELS": "english", + "LAYA_API_KEY": TOKEN, + "LAYA_LOG_LEVEL": "warning", + } + process = subprocess.Popen([sys.executable, str(WORKER)], env=env) + deadline = time.monotonic() + 600 + while time.monotonic() < deadline: + assert process.poll() is None, f"worker exited with {process.returncode}" + try: + status, _, _ = call(port, "GET", "/health") + if status == 200: + break + except OSError: + time.sleep(0.1) + else: + pytest.fail("worker not ready in 600 s") + # The first decision after readiness, before any other test warms anything. + first = decide(port, {"q": CHOICE}) + yield port, first + process.terminate() + process.wait(timeout=30) + + +def test_health_reports_the_loaded_model(worker): + port, _ = worker + status, body, _ = call(port, "GET", "/health") + health = json.loads(body) + assert status == 200 + assert health["ready"] is True + assert health["device"] == "cpu" + assert health["device_mismatch"] is False + assert health["checkpoint"] == "convaiinnovations/laya" + assert health["loaded"] == ["english"] + + +def test_first_request_after_ready_is_warm(worker): + port, (status, _, first_ms) = worker + assert status == 200 + warm = [decide(port, {"q": CHOICE})[2] for _ in range(20)] + assert first_ms <= 2 * statistics.median(warm), f"first {first_ms:.0f} ms, warm p50 {statistics.median(warm):.0f}" + + +@pytest.mark.parametrize( + ("questions", "kinds"), + [ + ({"q": CHOICE}, {"q": "choice"}), + ({"q": SCORE}, {"q": "score"}), + ({"q": NOUL}, {"q": "noul"}), + ({"a": CHOICE, "b": SCORE, "c": NOUL}, {"a": "choice", "b": "score", "c": "noul"}), + ], + ids=["choice", "score", "noul", "combined"], +) +def test_decisions(worker, questions, kinds): + port, _ = worker + status, body, _ = decide(port, questions) + assert status == 200 + result = json.loads(body) + assert set(result["answers"]) == set(kinds) + assert result["usage"]["input_tokens"] > 0 + for qid, kind in kinds.items(): + answer = result["answers"][qid] + assert answer["type"] == kind + if kind == "noul": + assert 0.0 <= answer["noul"] <= 1.0 + else: + assert sum(answer["probabilities"].values()) == pytest.approx(1.0, abs=1e-3) + if kind == "choice": + assert answer["choice"] in CHOICE["criteria"] + + +def test_same_request_same_answer(worker): + port, _ = worker + first, second = (json.loads(decide(port, {"a": CHOICE, "b": NOUL})[1])["answers"] for _ in range(2)) + assert first == second + + +@pytest.mark.parametrize( + ("body", "raw", "token", "expected"), + [ + (b"{not json", True, TOKEN, 400), + ({"model": "english", "state": STATE}, False, TOKEN, 400), + ( + {"model": "english", "state": STATE, "questions": {"q": {"type": "bogus", "instructions": "?"}}}, + False, + TOKEN, + 422, + ), + (b"x" * (2 * 1024 * 1024 + 1), True, TOKEN, 413), + ({"model": "english", "state": STATE, "questions": {"q": NOUL}}, False, "wrong", 401), + ({"model": "english", "state": STATE, "questions": {"q": NOUL}}, False, None, 401), + ], + ids=["malformed-json", "no-questions", "bad-question", "too-large", "wrong-token", "no-token"], +) +def test_errors(worker, body, raw, token, expected): + port, _ = worker + status, payload, _ = call(port, "POST", "/v1/systemone", body, token=token, raw=raw) + assert status == expected, payload[:200] diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py new file mode 100644 index 0000000..aa92469 --- /dev/null +++ b/src/models/laya/tests/test_worker.py @@ -0,0 +1,196 @@ +"""Unit tests for the Laya worker. A fake Router stands in for laya's; no model is loaded. + +python -m pytest src/models/laya/tests +""" + +import sys +from pathlib import Path + +import pytest +from fastapi.testclient import TestClient + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import worker + +ANSWER = {"type": "noul", "noul": 0.9, "confidence": 0.9} + + +class FakeAgent: + def __init__(self, device="mps", dtype="torch.float16"): + self.device = device + self.dtype = dtype + self.mps_amp_min_rows = 5 + + +class FakeRouter: + """The part of laya.router.Router the worker and laya.serve.create_app use.""" + + def __init__(self, agent=None, fail_on_call=None): + self.agent = agent or FakeAgent() + self.calls = [] + self.fail_on_call = fail_on_call + + @property + def loaded(self): + return ["english"] + + def load(self, name): + return self.agent + + def predict(self, state, questions, model=None): + self.calls.append((state, questions, model)) + if self.fail_on_call is not None and len(self.calls) == self.fail_on_call: + raise RuntimeError("MPS backend out of memory") + return { + "model": "laya-rl-agent", + "answers": {qid: ANSWER for qid in questions}, + "usage": {"input_tokens": 10, "output_tokens": 0}, + "routing": {"model": model, "repo": "convaiinnovations/laya"}, + } + + +def test_warmup_covers_short_long_and_fp16_multi_question_shapes(): + router = FakeRouter() + worker.warmup(router, "english") + words = {len(state.split()) for state, _, _ in router.calls} + rows = {len(questions) for _, questions, _ in router.calls} + assert min(words) <= 20 and max(words) >= 400 + assert max(rows) >= router.agent.mps_amp_min_rows # crosses laya's fp16 autocast threshold on MPS + assert {q["type"] for _, questions, _ in router.calls for q in questions.values()} == {"choice", "score", "noul"} + assert len(router.calls) == len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + assert all(model == "english" for _, _, model in router.calls) + + +def test_warmup_runs_before_the_app_exists(): + router = FakeRouter() + worker.create_worker_app(router, "english", "mps") + assert len(router.calls) == len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + + +def test_warmup_failure_raises_and_no_app_is_built(): + with pytest.raises(RuntimeError, match="out of memory"): + worker.create_worker_app(FakeRouter(fail_on_call=3), "english", "mps") + + +def test_health_reports_the_agent_device_not_the_requested_one(): + router = FakeRouter(FakeAgent(device="cpu", dtype="torch.float32")) + health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() + assert health["device"] == "cpu" + assert health["requested_device"] == "mps" + assert health["device_mismatch"] is True + assert health["ready"] is True + + +def test_health_on_the_requested_device(): + health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() + assert health["device"] == "mps" + assert health["device_mismatch"] is False + assert health["autocast_dtype"] == "torch.float16" + assert health["checkpoint"] == "convaiinnovations/laya" + assert health["warmup_ms"] >= 0 + + +def test_device_index_is_not_a_mismatch(): + router = FakeRouter(FakeAgent(device="cuda:0")) + assert ( + TestClient(worker.create_worker_app(router, "english", "cuda")).get("/health").json()["device_mismatch"] + is False + ) + + +def test_auto_device_is_never_a_mismatch(): + router = FakeRouter(FakeAgent(device="cpu")) + health = TestClient(worker.create_worker_app(router, "english", None)).get("/health").json() + assert health["requested_device"] == "auto" + assert health["device_mismatch"] is False + + +def test_require_device_refuses_to_serve_on_another_device(): + router = FakeRouter(FakeAgent(device="cpu")) + with pytest.raises(RuntimeError, match="asked for mps, model is on cpu"): + worker.create_worker_app(router, "english", "mps", require_device=True) + + +def test_only_one_health_route_remains(): + app = worker.create_worker_app(FakeRouter(), "english", "mps") + assert [r.path for r in app.router.routes if getattr(r, "path", None) == "/health"] == ["/health"] + + +def test_decisions_still_go_through_laya_serve(): + router = FakeRouter() + client = TestClient(worker.create_worker_app(router, "english", "mps")) + before = len(router.calls) + response = client.post( + "/v1/systemone", + json={"model": "english", "state": "refund me", "questions": {"r": {"type": "noul", "instructions": "?"}}}, + ) + assert response.status_code == 200 + assert response.json()["answers"]["r"]["noul"] == 0.9 + assert len(router.calls) == before + 1 + + +def test_main_exits_non_zero_when_warmup_fails(monkeypatch): + import laya.serve + + monkeypatch.setattr(laya.serve, "build_router", lambda: FakeRouter(fail_on_call=1)) + monkeypatch.setattr("uvicorn.run", lambda *a, **k: pytest.fail("must not bind")) + with pytest.raises(SystemExit, match="not starting"): + worker.main() + + +def test_compile_wraps_the_model_before_warmup(monkeypatch): + router = FakeRouter() + order = [] + monkeypatch.setattr(worker, "compile_agent", lambda agent, mode: order.append((mode, len(router.calls)))) + worker.create_worker_app(router, "english", "mps", compile="single", graph_counter=lambda: 3) + assert order == [("single", 0)] # before the first warmup request + + +def test_health_reports_compile_off_by_default(): + health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() + assert health["compile"] == {"mode": "off"} + + +def test_health_flags_graphs_compiled_after_ready(monkeypatch): + monkeypatch.setattr(worker, "compile_agent", lambda agent, mode: None) + graphs = iter([4, 4, 5]) # at readiness, first /health, second /health after a new shape compiled + client = TestClient( + worker.create_worker_app(FakeRouter(), "english", "mps", compile="all", graph_counter=lambda: next(graphs)) + ) + first = client.get("/health").json()["compile"] + assert first == {"mode": "all", "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} + assert client.get("/health").json()["compile"]["recompiled_after_ready"] is True + + +def test_compile_failure_means_no_app(monkeypatch): + def broken(agent, mode): + raise RuntimeError("inductor: unsupported op on mps") + + monkeypatch.setattr(worker, "compile_agent", broken) + with pytest.raises(RuntimeError, match="unsupported op"): + worker.create_worker_app(FakeRouter(), "english", "mps", compile="all") + + +def test_unknown_compile_mode_is_refused(): + with pytest.raises(ValueError, match="off, all or single"): + worker.create_worker_app(FakeRouter(), "english", "mps", compile="invalid") + + +def test_single_row_batches_use_the_compiled_model(monkeypatch): + import torch + + class Echo(torch.nn.Module): + def __init__(self): + super().__init__() + self.weight = torch.nn.Parameter(torch.ones(1)) + + def forward(self, input_ids): + return "eager" + + monkeypatch.setattr(torch, "compile", lambda model, dynamic: lambda input_ids: "compiled") + agent = FakeAgent() + agent.model = Echo() + worker.compile_agent(agent, "single") + assert agent.model(torch.zeros(1, 7)) == "compiled" + assert agent.model(torch.zeros(3, 7)) == "eager" + assert [p.shape for p in agent.model.parameters()] == [torch.Size([1])] # one set of weights diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py new file mode 100644 index 0000000..37315d8 --- /dev/null +++ b/src/models/laya/worker.py @@ -0,0 +1,206 @@ +"""Laya worker: laya-serve with warmup before readiness and the actual device in /health. + +laya-serve (laya 0.3.20) answers /health as soon as it binds, before any forward pass, and reports +LAYA_DEVICE as configured rather than where the model ended up. This worker reuses laya's app and +request handling unchanged and fixes both: + +- It binds only after a warmup that covers short, long and multi-question requests (the last one + crosses laya's fp16 autocast threshold on MPS), so a reachable worker is a warm one. +- /health reports the device, weight and autocast dtypes of the loaded agent, the checkpoint and + revision it serves, and whether the device differs from the one requested. + +Configuration is laya-serve's (LAYA_HOST, LAYA_PORT, LAYA_DEVICE, LAYA_MODELS, LAYA_API_KEY, ...) plus: + + LAYA_WORKER_MODEL model to warm up and describe english + LAYA_REQUIRE_DEVICE exit instead of serving on another 0 + device than LAYA_DEVICE asked for + LAYA_WORKER_COMPILE off, all, or single: torch.compile off + (dynamic=True) the model before warmup, + for every batch or one-row batches only + +The warmup also compiles every shape class it sends through the compiled model: in measured runs on an +M1 Pro the worker with `single` was ready after about 22 s instead of 8 s (`all` took over a minute). +/health counts compiled graphs at readiness and now; `recompiled_after_ready` means a request hit a +shape class the warmup did not cover. `single` exists because on MPS compiling cut one-question +latency by about 29% while compiling everything made multi-question requests slower (benchmarks/laya). +""" + +import logging +import os +import sys +import time +from pathlib import Path +from typing import Any + +log = logging.getLogger("laya-worker") + +# (words of state, questions): each shape runs twice. Short, mid-length and near-window states, then +# 3 and 6 questions (6 is at or above laya's MPS fp16 autocast threshold of 5 rows). +_CHOICE = { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": {"billing": "Charges and refunds", "technical": "Software problems", "other": "Anything else"}, +} +_SCORE = {"type": "score", "instructions": "How urgent is it?", "criteria": ["Low", "Medium", "High"]} +_NOUL = {"type": "noul", "instructions": "Does the customer ask for a refund?"} +WARMUP_SHAPES = [ + (10, {"q": _CHOICE}), + (150, {"q": _CHOICE}), + (400, {"q": _CHOICE}), + (10, {"a": _CHOICE, "b": _SCORE, "c": _NOUL}), + (10, {f"q{i}": q for i, q in enumerate([_CHOICE, _SCORE, _NOUL, _CHOICE, _SCORE, _NOUL])}), +] +WARMUP_REPEATS = 2 + + +def _env_bool(name: str) -> bool: + return os.environ.get(name, "").strip().lower() in ("1", "true", "yes", "on") + + +def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_REPEATS) -> dict[str, Any]: + """Run every shape `repeats` times. Any failure propagates: a worker that cannot answer must not bind.""" + started = time.perf_counter() + routing = None + for words, questions in shapes: + state = " ".join(["refund"] * words) + for _ in range(repeats): + result = router.predict(state, questions, model=model) + routing = result.get("routing") or routing + return {"warmup_ms": round((time.perf_counter() - started) * 1000, 1), "routing": routing} + + +def compiled_graphs() -> int: + """Graphs torch.compile has produced in this process.""" + from torch._dynamo.utils import counters + + return int(counters["stats"]["unique_graphs"]) + + +COMPILE_MODES = {"0": "off", "off": "off", "": "off", "1": "all", "all": "all", "single": "single"} + + +def compile_agent(agent: Any, mode: str) -> None: + """`all`: every forward goes through the compiled model. `single`: batches of one row (one question) + do, everything else runs eager. Both paths share the same parameters.""" + import torch + + eager = agent.model + compiled = torch.compile(eager, dynamic=True) + if mode == "all": + agent.model = compiled + return + + class SingleRowCompiled(torch.nn.Module): + def __init__(self): + super().__init__() + self.eager = eager + + def forward(self, input_ids, *args, **kwargs): + model = compiled if input_ids.shape[0] == 1 else self.eager + return model(input_ids, *args, **kwargs) + + agent.model = SingleRowCompiled() + + +def _cached_revision(repo: str | None, ref: str = "main") -> str | None: + if not repo: + return None + try: + from huggingface_hub.constants import HF_HUB_CACHE + + return (Path(HF_HUB_CACHE) / f"models--{repo.replace('/', '--')}" / "refs" / ref).read_text().strip() + except (ImportError, OSError): + return None + + +def describe(agent: Any, requested: str | None, routing: dict[str, Any] | None) -> dict[str, Any]: + """What /health reports about the loaded agent.""" + device = str(getattr(agent, "device", "unknown")) + model = getattr(agent, "model", None) + try: + weights = str(next(model.parameters()).dtype) if model is not None else None + except (AttributeError, StopIteration, TypeError): + weights = None + repo = (routing or {}).get("repo") + requested_type = requested.split(":")[0] if requested else None + return { + "device": device, + "requested_device": requested or "auto", + "device_mismatch": bool(requested_type) and device.split(":")[0] != requested_type, + "weights_dtype": weights, + "autocast_dtype": str(getattr(agent, "dtype", None)), + "mps_amp_min_rows": getattr(agent, "mps_amp_min_rows", None), + "checkpoint": repo, + "revision": _cached_revision(repo), + } + + +def create_worker_app( + router: Any, + model: str, + requested: str | None, + require_device: bool = False, + compile: str = "off", + graph_counter=compiled_graphs, +): + """Optionally compile, warm the router up, then return laya's app with /health replaced. + Raises if compiling or warmup fails.""" + from laya.serve import create_app + + if compile not in ("off", "all", "single"): + raise ValueError(f"compile mode must be off, all or single, not {compile!r}") + if compile != "off": + compile_agent(router.load(model), compile) + warm = warmup(router, model) + info = describe(router.load(model), requested, warm["routing"]) + info["warmup_ms"] = warm["warmup_ms"] + graphs_at_ready = graph_counter() if compile != "off" else None + if info["device_mismatch"]: + message = f"asked for {info['requested_device']}, model is on {info['device']}" + if require_device: + raise RuntimeError(message) + log.warning(message) + + app = create_app(router) + app.router.routes[:] = [r for r in app.router.routes if getattr(r, "path", None) != "/health"] + + @app.get("/health") + def health() -> dict[str, Any]: + compiled = {"mode": compile} + if compile != "off": + now = graph_counter() + compiled.update( + graphs_at_ready=graphs_at_ready, graphs_now=now, recompiled_after_ready=now > graphs_at_ready + ) + return {"status": "ok", "ready": True, "loaded": router.loaded, **info, "compile": compiled} + + return app + + +def main() -> None: + import uvicorn + from laya.serve import _resolve_port, build_router + + logging.basicConfig(level=logging.INFO, format="%(name)s: %(message)s") + model = os.environ.get("LAYA_WORKER_MODEL", "english") + requested = os.environ.get("LAYA_DEVICE") or None + try: + app = create_worker_app( + build_router(), + model, + requested, + require_device=_env_bool("LAYA_REQUIRE_DEVICE"), + compile=COMPILE_MODES.get(os.environ.get("LAYA_WORKER_COMPILE", "").strip().lower(), "invalid"), + ) + except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever + sys.exit(f"laya-worker: not starting: {exc}") + uvicorn.run( + app, + host=os.environ.get("LAYA_HOST", "127.0.0.1"), + port=_resolve_port(), + log_level=os.environ.get("LAYA_LOG_LEVEL", "info"), + ) + + +if __name__ == "__main__": + main() From 326838ac237a89fb166dbc14c67aceda877a04b7 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:43:47 +0800 Subject: [PATCH 02/21] Add measured Laya reports and a paired frontend-overhead probe Reports built from the measured runs on an M1 Pro (CPU and MPS in-process, laya-serve, laya-serve behind the Rust frontend, the worker with and without LAYA_WORKER_COMPILE=single): latency tables, output parity and the paired frontend overhead. The raw JSONL is published as a release asset of the fork; benchmarks/laya/README.md has the link and checksum. frontend_overhead.py sends each request directly and through the frontend back to back, so background load that shifts separate runs cancels out. --- benchmarks/laya/README.md | 14 + benchmarks/laya/frontend_overhead.py | 80 ++++ benchmarks/laya/results/.gitignore | 6 +- .../laya/results/frontend_overhead_m1.md | 10 + benchmarks/laya/results/measured-parity.md | 336 ++++++++++++++ benchmarks/laya/results/measured-report.md | 431 ++++++++++++++++++ 6 files changed, 875 insertions(+), 2 deletions(-) create mode 100644 benchmarks/laya/frontend_overhead.py create mode 100644 benchmarks/laya/results/frontend_overhead_m1.md create mode 100644 benchmarks/laya/results/measured-parity.md create mode 100644 benchmarks/laya/results/measured-report.md diff --git a/benchmarks/laya/README.md b/benchmarks/laya/README.md index 96cfe5c..e1e686c 100644 --- a/benchmarks/laya/README.md +++ b/benchmarks/laya/README.md @@ -10,6 +10,7 @@ Every script writes raw JSONL; `report.py` is the only place numbers are compute | `bench_inproc.py` | in-process: import, load, warmup, first request per workload, warm latency, memory | | `bench_http.py` | against a `/v1/systemone` worker: process-to-ready (with `--spawn`), first request per workload, warm latency and throughput at each `--concurrency` | | `profile_mps.py` | where a request's time goes on MPS: length sweep with a fixed-cost fit, stage split (encode, dispatch, GPU wait, copy back, decode) and host operator counts | +| `frontend_overhead.py` | frontend cost, paired: each request direct and through the frontend back to back, so background load cancels | | `parity.py` | answers of every run vs a reference run, tolerances fixed in advance (fp32 1e-3, fp16 1e-2) | | `env.py` | run header: SHAs, versions, checkpoint revision, hardware, power, load | | `report.py` | JSONL → tables, including the run-to-run gate | @@ -40,3 +41,16 @@ tokenization and post-processing. Workload order is shuffled per run with `--see Memory is the physical footprint of the process running Laya (`proc_pid_rusage`, the same number as Activity Monitor's "Memory" and `footprint -p`). On Apple silicon it includes Metal allocations, so in-process and worker numbers are comparable and MPS tensors are counted. + +## Results + +`results/` holds the reports built from the measured runs on an M1 Pro: `measured-report.md` +(`report.py`), `measured-parity.md` (`parity.py --ref C1`) and `frontend_overhead_m1.md`. The raw +JSONL they were built from (7 MB, 16 files) is published as a release asset rather than committed: + +```sh +curl -LO https://github.com/cacheline999/system1-omni/releases/download/laya-mps-results-2026-09-28/laya-mps-results-2026-09-28.tar.gz +shasum -a 256 laya-mps-results-2026-09-28.tar.gz # 611ed30707ac8c98875b5aa5382360b5a7d760da166d61c626eb07ebe1ee6404 +tar xzf laya-mps-results-2026-09-28.tar.gz -C benchmarks/laya/results +python benchmarks/laya/report.py benchmarks/laya/results/*_m[0-9].jsonl +``` diff --git a/benchmarks/laya/frontend_overhead.py b/benchmarks/laya/frontend_overhead.py new file mode 100644 index 0000000..4d3f753 --- /dev/null +++ b/benchmarks/laya/frontend_overhead.py @@ -0,0 +1,80 @@ +"""Frontend overhead, paired: each request goes to the worker directly and through the frontend, one right +after the other, alternating which goes first. Pairing cancels background load that shifts separate +runs, so the per-request difference is the frontend's cost. + +Start a worker and the frontend first (recipe/laya/apple-silicon.md), then: + + python benchmarks/laya/frontend_overhead.py --direct http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 +""" + +import argparse +import http.client +import json +import statistics +import time +from pathlib import Path +from urllib.parse import urlsplit + +HERE = Path(__file__).resolve().parent + + +def connect(url): + parts = urlsplit(url) + return http.client.HTTPConnection(parts.hostname, parts.port or 80, timeout=120) + + +def quantile(sorted_values, p): + return sorted_values[min(len(sorted_values) - 1, int(p * len(sorted_values)))] + + +def call(conn, body): + started = time.perf_counter() + conn.request("POST", "/v1/systemone", body=body, headers={"Content-Type": "application/json"}) + response = conn.getresponse() + response.read() + if response.status != 200: + raise SystemExit(f"status {response.status}") + return (time.perf_counter() - started) * 1000 + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--direct", default="http://127.0.0.1:8000") + parser.add_argument("--frontend", default="http://127.0.0.1:8080") + parser.add_argument("--model", default="english") + parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) + parser.add_argument("--only", nargs="*", help="bench workload ids (default: all)") + parser.add_argument("-n", type=int, default=120, help="pairs per workload") + parser.add_argument("--discard", type=int, default=10) + args = parser.parse_args() + + with open(args.workloads) as f: + workloads = [json.loads(line) for line in f if line.strip()] + bench = [w for w in workloads if w["kind"] == "bench" and (not args.only or w["id"] in args.only)] + direct, frontend = connect(args.direct), connect(args.frontend) + + print("| workload | body bytes | direct p50 | frontend p50 | Δ p10 | Δ p50 | Δ p90 | Δ > 10 ms |") + print("|---|---|---|---|---|---|---|---|") + for w in bench: + body = json.dumps({"model": args.model, "state": w["state"], "questions": w["questions"]}).encode() + for _ in range(args.discard): + call(direct, body) + call(frontend, body) + d_ms, f_ms, delta = [], [], [] + for i in range(args.n): + if i % 2 == 0: + a, b = call(direct, body), call(frontend, body) + else: + b, a = call(frontend, body), call(direct, body) + d_ms.append(a) + f_ms.append(b) + delta.append(b - a) + delta.sort() + print( + f"| {w['id']} | {len(body)} | {statistics.median(d_ms):.1f} | {statistics.median(f_ms):.1f} " + f"| {quantile(delta, 0.1):+.1f} | {statistics.median(delta):+.1f} | {quantile(delta, 0.9):+.1f} | {sum(x > 10 for x in delta)}/{args.n} |" + ) + + +if __name__ == "__main__": + main() diff --git a/benchmarks/laya/results/.gitignore b/benchmarks/laya/results/.gitignore index 1c54c95..a16a40b 100644 --- a/benchmarks/laya/results/.gitignore +++ b/benchmarks/laya/results/.gitignore @@ -1,5 +1,7 @@ -# Committed: measured runs (m1, m2, ...) and the reports built from them. -# Not committed: feasibility runs, worker/frontend logs, runner state. +# Committed: reports built from measured runs. +# Not committed: raw JSONL (published as a release asset, see ../README.md), feasibility runs, +# worker/frontend logs, runner state. +*.jsonl *feasibility* *.log done_* diff --git a/benchmarks/laya/results/frontend_overhead_m1.md b/benchmarks/laya/results/frontend_overhead_m1.md new file mode 100644 index 0000000..be3b463 --- /dev/null +++ b/benchmarks/laya/results/frontend_overhead_m1.md @@ -0,0 +1,10 @@ +Paired frontend overhead, 2026-09-28T13:40Z, Apple M1 Pro, AC Power, 1-min load 4.80, omni 2a47358, laya-serve on MPS behind target/release/omni-jev + +| workload | body bytes | direct p50 | frontend p50 | Δ p10 | Δ p50 | Δ p90 | Δ > 10 ms | +|---|---|---|---|---|---|---|---| +| W1 | 416 | 45.3 | 45.6 | -0.9 | +0.3 | +1.7 | 0/120 | +| W2 | 1075 | 68.5 | 68.8 | -0.8 | +0.4 | +1.8 | 0/120 | +| W3 | 2527 | 151.5 | 151.8 | -3.7 | +0.5 | +3.5 | 0/120 | +| W4 | 657 | 71.0 | 71.3 | -0.8 | +0.3 | +1.7 | 0/120 | +| W5 | 968 | 138.9 | 139.0 | -2.8 | +0.1 | +3.6 | 1/120 | +| W6 | 197 | 39.0 | 39.2 | -0.9 | +0.2 | +1.6 | 0/120 | diff --git a/benchmarks/laya/results/measured-parity.md b/benchmarks/laya/results/measured-parity.md new file mode 100644 index 0000000..32f0b4d --- /dev/null +++ b/benchmarks/laya/results/measured-parity.md @@ -0,0 +1,336 @@ +Reference: C1 / m1. Tolerances: fp32 0.001, fp16 0.01. + +| config | run | workload | question | type | path | decision ref → run | max abs Δp | ref margin | result | +|---|---|---|---|---|---|---|---|---|---| +| C1 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C1 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C1 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C1 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C1 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C1 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C1 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C1 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C1 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C1 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C1 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C1 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C1 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C1 | m2 | W5 | angry | noul | fp32 | True | 0.0000 | 0.0231 | PASS | +| C1 | m2 | W5 | cancel | noul | fp32 | False | 0.0000 | 0.4696 | PASS | +| C1 | m2 | W5 | lang | choice | fp32 | en | 0.0000 | 0.1674 | PASS | +| C1 | m2 | W5 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C1 | m2 | W5 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m2 | W5 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C1 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C1 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C1 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C1 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C1 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C1 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C1 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C1 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C1 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C1 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C1 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C1 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C1 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C1 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C1 | m3 | W5 | angry | noul | fp32 | True | 0.0000 | 0.0231 | PASS | +| C1 | m3 | W5 | cancel | noul | fp32 | False | 0.0000 | 0.4696 | PASS | +| C1 | m3 | W5 | lang | choice | fp32 | en | 0.0000 | 0.1674 | PASS | +| C1 | m3 | W5 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C1 | m3 | W5 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C1 | m3 | W5 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C1 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C2 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C2 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C2 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C2 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C2 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C2 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C2 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C2 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C2 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C2 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C2 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C2 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C2 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C2 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C2 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C2 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C2 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C2 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C2 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C2 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C2 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C2 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C2 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C2 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C2 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C2 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C2 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C2 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C2 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C2 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C2 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C2 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C2 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C2 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C2 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C2 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C2 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C2 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C2 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C2 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C2 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C2 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C2 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C2 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3 | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3 | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3 | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3 | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3 | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3 | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3s | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3s | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3s | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3s | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3s | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3s | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3s | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3s | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3s | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3s | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3s | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3s | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3s | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3s | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3s | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3s | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3s | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3s | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3s | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3s | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3s | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3s | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3s | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3s | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3s | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3s | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3s | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3s | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3s | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3s | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3s | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3s | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3s | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3s | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3s | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3s | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3s | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3s | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3s | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3s | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3s | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3s | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3s | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3s | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3w | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3w | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3w | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3w | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3w | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3w | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3w | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3w | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3w | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3w | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3w | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3w | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3w | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3w | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3w | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3w | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3w | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3w | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3w | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3w | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3w | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3w | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3w | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3w | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3w | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3w | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3w | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3w | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C3w | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C3w | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C3w | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C3w | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C3w | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C3w | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C3w | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C3w | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C3w | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C3w | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C3w | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C3w | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C3w | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C3w | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C3w | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C3w | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C3w | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C3w | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C4 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C4 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C4 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C4 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C4 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C4 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C4 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C4 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C4 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C4 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C4 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C4 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C4 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C4 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C4 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C4 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C4 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C4 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C4 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C4 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C4 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C4 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C4 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C4 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C4 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C4 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C4 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C4 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | +| C4 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | +| C4 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | +| C4 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | +| C4 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | +| C4 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | +| C4 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | +| C4 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | +| C4 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | +| C4 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | +| C4 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | +| C4 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | +| C4 | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | +| C4 | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | +| C4 | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | +| C4 | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | +| C4 | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | +| C4 | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | +| C4 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | + +330/330 questions within tolerance. diff --git a/benchmarks/laya/results/measured-report.md b/benchmarks/laya/results/measured-report.md new file mode 100644 index 0000000..7974689 --- /dev/null +++ b/benchmarks/laya/results/measured-report.md @@ -0,0 +1,431 @@ +## Environment + +| config | run | device | weights (autocast) | chip | os | power | load 1m | versions | ckpt | omni | +|---|---|---|---|---|---|---|---|---|---|---| +| C1 | m1 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.45 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C1 | m2 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.69 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C1 | m3 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.62 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C2 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.72 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C2 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.04 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 5.33 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.89 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3 | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 5.22 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | +| C3s | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.44 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3s | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.47 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3w | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.63 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3w | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.57 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C3w | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.32 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | +| C4 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.03 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C4 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 3.89 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | +| C4 | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.62 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | + +## Phases (s) and first request per workload (ms) + +| config | run | import | load | process_to_ready | warmup | first W1 | first W2 | first W3 | first W4 | first W5 | first W6 | +|---|---|---|---|---|---|---|---|---|---|---|---| +| C1 | m1 | 1.477 | 5.511 | | 32.22 | 851.59 | 233.17 | 396.66 | 229.58 | 288.89 | 218.59 | +| C1 | m2 | 1.337 | 4.902 | | 28.264 | 676.67 | 249.27 | 509.78 | 213.0 | 298.91 | 120.46 | +| C1 | m3 | 1.425 | 5.501 | | 28.847 | 562.08 | 228.06 | 447.66 | 210.57 | 308.64 | 115.43 | +| C2 | m1 | 1.309 | 6.023 | | 11.54 | 999.09 | 77.8 | 157.55 | 86.45 | 269.89 | 39.37 | +| C2 | m2 | 0.917 | 5.562 | | 13.615 | 3241.64 | 73.52 | 161.09 | 86.16 | 326.02 | 42.86 | +| C3 | m1 | | | 7.478 | 11.876 | 1134.76 | 76.18 | 160.33 | 84.95 | 356.31 | 42.91 | +| C3 | m2 | | | 7.312 | 11.13 | 697.05 | 73.98 | 151.34 | 90.97 | 393.92 | 42.52 | +| C3 | m3 | | | 8.71 | 11.668 | 895.33 | 82.71 | 156.29 | 93.78 | 363.58 | 65.97 | +| C3s | m1 | | | 22.862 | 9.407 | 62.37 | 100.8 | 146.01 | 79.25 | 142.37 | 45.97 | +| C3s | m2 | | | 22.08 | 9.652 | 61.59 | 93.57 | 143.69 | 83.71 | 146.98 | 42.47 | +| C3w | m1 | | | 7.92 | 10.323 | 75.4 | 104.47 | 156.66 | 75.58 | 142.32 | 46.31 | +| C3w | m2 | | | 8.482 | 10.728 | 81.0 | 142.28 | 152.95 | 90.44 | 145.9 | 41.09 | +| C3w | m3 | | | 9.256 | 10.407 | 78.41 | 211.06 | 152.64 | 82.69 | 147.66 | 42.33 | +| C4 | m1 | | | 7.56 | 11.452 | 749.71 | 80.02 | 148.21 | 87.17 | 334.73 | 50.21 | +| C4 | m2 | | | 7.512 | 11.501 | 976.06 | 76.8 | 159.66 | 81.79 | 335.96 | 42.85 | +| C4 | m3 | | | 8.129 | 12.079 | 724.84 | 82.86 | 173.12 | 85.07 | 380.83 | 49.62 | + +## Warm latency (ms) + +| config | workload | conc | run | n | p50 | p95 | mean | CV | +|---|---|---|---|---|---|---|---|---| +| C1 | W1 | 1 | m1 | 300 | 136.16 | 152.83 | 139.3 | 13.3% | +| C1 | W1 | 1 | m2 | 300 | 134.18 | 174.19 | 140.56 | 13.5% | +| C1 | W1 | 1 | m3 | 300 | 143.71 | 163.08 | 147.56 | 8.8% | +| C1 | W2 | 1 | m1 | 300 | 214.92 | 256.75 | 222.03 | 12.4% | +| C1 | W2 | 1 | m2 | 300 | 211.7 | 277.15 | 222.45 | 17.3% | +| C1 | W2 | 1 | m3 | 300 | 238.15 | 364.68 | 261.25 | 36.7% | +| C1 | W3 | 1 | m1 | 300 | 423.07 | 489.32 | 429.02 | 10.0% | +| C1 | W3 | 1 | m2 | 300 | 381.17 | 443.43 | 393.26 | 11.1% | +| C1 | W3 | 1 | m3 | 300 | 427.47 | 661.61 | 469.78 | 33.3% | +| C1 | W4 | 1 | m1 | 300 | 206.81 | 230.14 | 209.84 | 7.8% | +| C1 | W4 | 1 | m2 | 300 | 197.91 | 218.13 | 201.89 | 9.6% | +| C1 | W4 | 1 | m3 | 300 | 225.78 | 273.31 | 243.74 | 42.7% | +| C1 | W5 | 1 | m1 | 300 | 399.96 | 623.75 | 432.71 | 29.0% | +| C1 | W5 | 1 | m2 | 300 | 289.94 | 319.87 | 293.91 | 5.2% | +| C1 | W5 | 1 | m3 | 300 | 316.47 | 362.75 | 323.07 | 6.8% | +| C1 | W6 | 1 | m1 | 300 | 112.68 | 120.81 | 113.63 | 5.4% | +| C1 | W6 | 1 | m2 | 300 | 108.29 | 128.27 | 111.36 | 9.5% | +| C1 | W6 | 1 | m3 | 300 | 122.7 | 163.07 | 130.51 | 24.1% | +| C2 | W1 | 1 | m1 | 300 | 45.4 | 47.88 | 45.7 | 4.6% | +| C2 | W1 | 1 | m2 | 300 | 44.15 | 46.5 | 44.51 | 4.3% | +| C2 | W2 | 1 | m1 | 300 | 69.01 | 74.36 | 69.4 | 3.2% | +| C2 | W2 | 1 | m2 | 300 | 68.26 | 73.05 | 68.72 | 3.0% | +| C2 | W3 | 1 | m1 | 300 | 150.23 | 168.83 | 152.71 | 5.5% | +| C2 | W3 | 1 | m2 | 300 | 144.64 | 155.04 | 146.47 | 5.3% | +| C2 | W4 | 1 | m1 | 300 | 71.87 | 77.93 | 73.6 | 25.9% | +| C2 | W4 | 1 | m2 | 300 | 71.14 | 78.44 | 73.13 | 23.2% | +| C2 | W5 | 1 | m1 | 300 | 144.57 | 153.01 | 145.58 | 5.4% | +| C2 | W5 | 1 | m2 | 300 | 139.31 | 146.83 | 140.19 | 3.9% | +| C2 | W6 | 1 | m1 | 300 | 37.48 | 40.59 | 37.88 | 3.8% | +| C2 | W6 | 1 | m2 | 300 | 37.53 | 41.02 | 37.67 | 5.3% | +| C3 | W1 | 1 | m1 | 300 | 46.43 | 51.36 | 47.18 | 6.7% | +| C3 | W1 | 1 | m2 | 300 | 45.16 | 47.45 | 45.4 | 2.9% | +| C3 | W1 | 1 | m3 | 300 | 46.21 | 51.04 | 47.02 | 8.1% | +| C3 | W1 | 4 | m1 | 300 | 183.78 | 193.71 | 183.85 | 5.6% | +| C3 | W1 | 4 | m2 | 300 | 179.8 | 184.82 | 179.29 | 5.6% | +| C3 | W1 | 4 | m3 | 300 | 182.29 | 195.7 | 182.58 | 6.5% | +| C3 | W2 | 1 | m1 | 300 | 68.15 | 69.76 | 68.29 | 1.4% | +| C3 | W2 | 1 | m2 | 300 | 69.29 | 73.9 | 69.89 | 3.3% | +| C3 | W2 | 1 | m3 | 300 | 72.34 | 79.75 | 73.32 | 6.4% | +| C3 | W2 | 4 | m1 | 300 | 270.14 | 274.17 | 269.09 | 5.4% | +| C3 | W2 | 4 | m2 | 300 | 277.11 | 288.8 | 276.67 | 5.6% | +| C3 | W2 | 4 | m3 | 300 | 285.48 | 446.69 | 314.42 | 25.5% | +| C3 | W3 | 1 | m1 | 300 | 147.18 | 164.15 | 149.63 | 4.6% | +| C3 | W3 | 1 | m2 | 300 | 156.25 | 180.33 | 158.78 | 7.3% | +| C3 | W3 | 1 | m3 | 300 | 155.7 | 171.08 | 158.16 | 5.0% | +| C3 | W3 | 4 | m1 | 300 | 576.78 | 646.04 | 582.0 | 6.9% | +| C3 | W3 | 4 | m2 | 300 | 604.59 | 636.61 | 603.6 | 6.6% | +| C3 | W3 | 4 | m3 | 300 | 606.93 | 696.18 | 618.08 | 7.9% | +| C3 | W4 | 1 | m1 | 300 | 76.26 | 103.39 | 82.23 | 38.3% | +| C3 | W4 | 1 | m2 | 300 | 72.09 | 76.13 | 72.49 | 3.8% | +| C3 | W4 | 1 | m3 | 300 | 71.6 | 76.35 | 72.3 | 3.4% | +| C3 | W4 | 4 | m1 | 300 | 283.54 | 289.94 | 282.96 | 5.2% | +| C3 | W4 | 4 | m2 | 300 | 290.06 | 299.55 | 288.61 | 5.7% | +| C3 | W4 | 4 | m3 | 300 | 295.12 | 341.81 | 299.56 | 8.5% | +| C3 | W5 | 1 | m1 | 300 | 138.54 | 144.85 | 139.84 | 4.6% | +| C3 | W5 | 1 | m2 | 300 | 140.16 | 145.2 | 140.5 | 2.5% | +| C3 | W5 | 1 | m3 | 300 | 141.0 | 146.11 | 142.02 | 6.4% | +| C3 | W5 | 4 | m1 | 300 | 549.68 | 556.64 | 547.68 | 5.4% | +| C3 | W5 | 4 | m2 | 300 | 563.58 | 588.67 | 564.85 | 5.9% | +| C3 | W5 | 4 | m3 | 300 | 564.72 | 605.95 | 565.53 | 5.9% | +| C3 | W6 | 1 | m1 | 300 | 41.71 | 46.22 | 42.23 | 9.1% | +| C3 | W6 | 1 | m2 | 300 | 36.99 | 42.08 | 37.79 | 5.5% | +| C3 | W6 | 1 | m3 | 300 | 38.85 | 44.38 | 39.89 | 13.9% | +| C3 | W6 | 4 | m1 | 300 | 162.26 | 181.65 | 163.49 | 7.7% | +| C3 | W6 | 4 | m2 | 300 | 147.05 | 154.68 | 146.98 | 5.9% | +| C3 | W6 | 4 | m3 | 300 | 150.12 | 168.94 | 150.83 | 7.5% | +| C3s | W1 | 1 | m1 | 300 | 32.99 | 35.83 | 33.38 | 4.8% | +| C3s | W1 | 1 | m2 | 300 | 32.55 | 33.89 | 32.68 | 2.2% | +| C3s | W1 | 4 | m1 | 300 | 129.91 | 135.95 | 130.38 | 6.2% | +| C3s | W1 | 4 | m2 | 300 | 128.91 | 132.43 | 129.01 | 6.0% | +| C3s | W2 | 1 | m1 | 300 | 63.03 | 83.07 | 66.36 | 19.3% | +| C3s | W2 | 1 | m2 | 300 | 60.73 | 64.65 | 61.12 | 2.9% | +| C3s | W2 | 4 | m1 | 300 | 259.71 | 306.33 | 266.76 | 10.1% | +| C3s | W2 | 4 | m2 | 300 | 238.37 | 248.95 | 238.59 | 5.8% | +| C3s | W3 | 1 | m1 | 300 | 134.16 | 146.03 | 135.84 | 4.2% | +| C3s | W3 | 1 | m2 | 300 | 138.18 | 150.78 | 139.85 | 4.4% | +| C3s | W3 | 4 | m1 | 300 | 544.89 | 583.88 | 547.89 | 6.6% | +| C3s | W3 | 4 | m2 | 300 | 543.35 | 564.9 | 545.68 | 6.8% | +| C3s | W4 | 1 | m1 | 300 | 72.6 | 88.29 | 75.21 | 8.7% | +| C3s | W4 | 1 | m2 | 300 | 72.83 | 77.13 | 73.2 | 3.3% | +| C3s | W4 | 4 | m1 | 300 | 295.51 | 377.02 | 308.8 | 12.5% | +| C3s | W4 | 4 | m2 | 300 | 288.58 | 299.58 | 287.73 | 5.7% | +| C3s | W5 | 1 | m1 | 300 | 138.74 | 143.5 | 139.51 | 4.5% | +| C3s | W5 | 1 | m2 | 300 | 140.03 | 144.33 | 140.38 | 2.1% | +| C3s | W5 | 4 | m1 | 300 | 555.12 | 568.42 | 553.25 | 5.5% | +| C3s | W5 | 4 | m2 | 300 | 560.2 | 571.87 | 558.3 | 5.5% | +| C3s | W6 | 1 | m1 | 300 | 26.04 | 27.02 | 26.15 | 2.0% | +| C3s | W6 | 1 | m2 | 300 | 26.43 | 30.05 | 27.01 | 9.4% | +| C3s | W6 | 4 | m1 | 300 | 103.97 | 118.23 | 105.48 | 8.7% | +| C3s | W6 | 4 | m2 | 300 | 102.84 | 106.2 | 102.83 | 5.7% | +| C3w | W1 | 1 | m1 | 300 | 46.82 | 52.47 | 47.43 | 4.8% | +| C3w | W1 | 1 | m2 | 300 | 45.76 | 48.19 | 45.99 | 2.4% | +| C3w | W1 | 1 | m3 | 300 | 45.53 | 47.85 | 45.74 | 2.7% | +| C3w | W1 | 4 | m1 | 300 | 185.89 | 220.62 | 189.55 | 8.9% | +| C3w | W1 | 4 | m2 | 300 | 181.79 | 187.96 | 181.7 | 5.7% | +| C3w | W1 | 4 | m3 | 300 | 180.96 | 184.56 | 180.41 | 5.4% | +| C3w | W2 | 1 | m1 | 300 | 68.45 | 71.76 | 68.78 | 2.2% | +| C3w | W2 | 1 | m2 | 300 | 72.4 | 79.95 | 72.95 | 5.7% | +| C3w | W2 | 1 | m3 | 300 | 68.69 | 70.34 | 68.74 | 1.3% | +| C3w | W2 | 4 | m1 | 300 | 274.76 | 310.3 | 278.37 | 7.5% | +| C3w | W2 | 4 | m2 | 300 | 283.5 | 311.64 | 284.33 | 6.9% | +| C3w | W2 | 4 | m3 | 300 | 271.99 | 275.56 | 270.72 | 5.5% | +| C3w | W3 | 1 | m1 | 300 | 144.67 | 150.49 | 145.15 | 1.9% | +| C3w | W3 | 1 | m2 | 300 | 158.77 | 185.47 | 160.35 | 7.9% | +| C3w | W3 | 1 | m3 | 300 | 146.14 | 154.61 | 147.05 | 2.8% | +| C3w | W3 | 4 | m1 | 300 | 587.08 | 621.88 | 589.07 | 5.8% | +| C3w | W3 | 4 | m2 | 300 | 630.84 | 692.64 | 630.62 | 7.2% | +| C3w | W3 | 4 | m3 | 300 | 617.72 | 658.0 | 619.19 | 6.2% | +| C3w | W4 | 1 | m1 | 300 | 70.34 | 72.85 | 70.55 | 1.7% | +| C3w | W4 | 1 | m2 | 300 | 74.18 | 79.15 | 74.51 | 3.6% | +| C3w | W4 | 1 | m3 | 300 | 70.9 | 72.72 | 71.02 | 1.6% | +| C3w | W4 | 4 | m1 | 300 | 281.56 | 297.73 | 282.03 | 6.2% | +| C3w | W4 | 4 | m2 | 300 | 292.1 | 302.51 | 290.43 | 5.8% | +| C3w | W4 | 4 | m3 | 300 | 282.83 | 287.0 | 281.54 | 5.4% | +| C3w | W5 | 1 | m1 | 300 | 138.91 | 144.43 | 139.59 | 2.9% | +| C3w | W5 | 1 | m2 | 300 | 139.93 | 143.23 | 140.23 | 2.4% | +| C3w | W5 | 1 | m3 | 300 | 138.91 | 141.94 | 139.23 | 2.1% | +| C3w | W5 | 4 | m1 | 300 | 555.48 | 573.79 | 555.16 | 5.9% | +| C3w | W5 | 4 | m2 | 300 | 559.9 | 575.41 | 557.86 | 5.5% | +| C3w | W5 | 4 | m3 | 300 | 555.77 | 562.55 | 553.06 | 5.5% | +| C3w | W6 | 1 | m1 | 300 | 40.15 | 43.83 | 40.52 | 5.2% | +| C3w | W6 | 1 | m2 | 300 | 37.95 | 42.76 | 38.56 | 5.3% | +| C3w | W6 | 1 | m3 | 300 | 38.91 | 43.52 | 39.21 | 5.1% | +| C3w | W6 | 4 | m1 | 300 | 159.07 | 174.04 | 160.42 | 7.0% | +| C3w | W6 | 4 | m2 | 300 | 148.98 | 160.82 | 149.96 | 7.0% | +| C3w | W6 | 4 | m3 | 300 | 156.96 | 173.82 | 157.85 | 7.2% | +| C4 | W1 | 1 | m1 | 300 | 51.34 | 53.4 | 51.54 | 3.9% | +| C4 | W1 | 1 | m2 | 300 | 46.31 | 51.61 | 47.32 | 13.3% | +| C4 | W1 | 1 | m3 | 300 | 47.42 | 49.16 | 47.47 | 2.4% | +| C4 | W1 | 4 | m1 | 300 | 201.31 | 204.91 | 200.77 | 5.5% | +| C4 | W1 | 4 | m2 | 300 | 182.06 | 187.56 | 181.81 | 6.2% | +| C4 | W1 | 4 | m3 | 300 | 187.8 | 192.08 | 186.87 | 5.6% | +| C4 | W2 | 1 | m1 | 300 | 87.24 | 93.59 | 88.08 | 3.8% | +| C4 | W2 | 1 | m2 | 300 | 83.78 | 90.77 | 84.15 | 5.0% | +| C4 | W2 | 1 | m3 | 300 | 71.73 | 74.42 | 71.65 | 2.2% | +| C4 | W2 | 4 | m1 | 300 | 347.52 | 354.59 | 345.44 | 5.6% | +| C4 | W2 | 4 | m2 | 300 | 293.51 | 315.78 | 292.63 | 6.6% | +| C4 | W2 | 4 | m3 | 300 | 286.32 | 313.29 | 289.29 | 8.1% | +| C4 | W3 | 1 | m1 | 300 | 153.71 | 191.37 | 158.66 | 9.3% | +| C4 | W3 | 1 | m2 | 300 | 190.42 | 203.53 | 186.58 | 7.2% | +| C4 | W3 | 1 | m3 | 300 | 168.01 | 251.87 | 184.08 | 32.3% | +| C4 | W3 | 4 | m1 | 300 | 771.69 | 962.56 | 795.76 | 11.8% | +| C4 | W3 | 4 | m2 | 300 | 759.95 | 795.4 | 749.15 | 6.8% | +| C4 | W3 | 4 | m3 | 300 | 632.37 | 697.57 | 636.88 | 7.5% | +| C4 | W4 | 1 | m1 | 300 | 90.42 | 96.35 | 90.93 | 3.3% | +| C4 | W4 | 1 | m2 | 300 | 72.31 | 76.18 | 72.74 | 3.0% | +| C4 | W4 | 1 | m3 | 300 | 74.47 | 77.65 | 74.55 | 2.6% | +| C4 | W4 | 4 | m1 | 300 | 360.37 | 373.47 | 361.38 | 8.2% | +| C4 | W4 | 4 | m2 | 300 | 290.97 | 299.16 | 289.11 | 5.8% | +| C4 | W4 | 4 | m3 | 300 | 295.8 | 301.27 | 294.11 | 5.6% | +| C4 | W5 | 1 | m1 | 300 | 138.98 | 145.95 | 139.77 | 3.0% | +| C4 | W5 | 1 | m2 | 300 | 140.01 | 145.13 | 140.46 | 2.6% | +| C4 | W5 | 1 | m3 | 300 | 150.35 | 166.48 | 151.86 | 5.5% | +| C4 | W5 | 4 | m1 | 300 | 559.72 | 572.81 | 557.28 | 5.7% | +| C4 | W5 | 4 | m2 | 300 | 562.18 | 574.27 | 559.79 | 5.7% | +| C4 | W5 | 4 | m3 | 300 | 606.57 | 622.48 | 604.12 | 5.7% | +| C4 | W6 | 1 | m1 | 300 | 41.73 | 43.16 | 41.86 | 2.2% | +| C4 | W6 | 1 | m2 | 300 | 38.31 | 42.38 | 38.86 | 4.7% | +| C4 | W6 | 1 | m3 | 300 | 39.21 | 43.03 | 39.48 | 3.8% | +| C4 | W6 | 4 | m1 | 300 | 166.55 | 169.36 | 166.38 | 5.9% | +| C4 | W6 | 4 | m2 | 300 | 147.78 | 159.22 | 148.24 | 5.7% | +| C4 | W6 | 4 | m3 | 300 | 154.91 | 169.31 | 156.36 | 6.5% | + +## Run-to-run gate (p50 spread across measured runs <= 10%) + +| config | workload | conc | runs | spread | gate | +|---|---|---|---|---|---| +| C1 | W1 | 1 | 3 | 7.1% | PASS | +| C1 | W2 | 1 | 3 | 12.5% | FAIL | +| C1 | W3 | 1 | 3 | 12.1% | FAIL | +| C1 | W4 | 1 | 3 | 14.1% | FAIL | +| C1 | W5 | 1 | 3 | 37.9% | FAIL | +| C1 | W6 | 1 | 3 | 13.3% | FAIL | +| C2 | W1 | 1 | 2 | 2.8% | PASS | +| C2 | W2 | 1 | 2 | 1.1% | PASS | +| C2 | W3 | 1 | 2 | 3.9% | PASS | +| C2 | W4 | 1 | 2 | 1.0% | PASS | +| C2 | W5 | 1 | 2 | 3.8% | PASS | +| C2 | W6 | 1 | 2 | 0.1% | PASS | +| C3 | W1 | 1 | 3 | 2.8% | PASS | +| C3 | W1 | 4 | 3 | 2.2% | PASS | +| C3 | W2 | 1 | 3 | 6.2% | PASS | +| C3 | W2 | 4 | 3 | 5.7% | PASS | +| C3 | W3 | 1 | 3 | 6.2% | PASS | +| C3 | W3 | 4 | 3 | 5.2% | PASS | +| C3 | W4 | 1 | 3 | 6.5% | PASS | +| C3 | W4 | 4 | 3 | 4.1% | PASS | +| C3 | W5 | 1 | 3 | 1.8% | PASS | +| C3 | W5 | 4 | 3 | 2.7% | PASS | +| C3 | W6 | 1 | 3 | 12.7% | FAIL | +| C3 | W6 | 4 | 3 | 10.3% | FAIL | +| C3s | W1 | 1 | 2 | 1.4% | PASS | +| C3s | W1 | 4 | 2 | 0.8% | PASS | +| C3s | W2 | 1 | 2 | 3.8% | PASS | +| C3s | W2 | 4 | 2 | 9.0% | PASS | +| C3s | W3 | 1 | 2 | 3.0% | PASS | +| C3s | W3 | 4 | 2 | 0.3% | PASS | +| C3s | W4 | 1 | 2 | 0.3% | PASS | +| C3s | W4 | 4 | 2 | 2.4% | PASS | +| C3s | W5 | 1 | 2 | 0.9% | PASS | +| C3s | W5 | 4 | 2 | 0.9% | PASS | +| C3s | W6 | 1 | 2 | 1.5% | PASS | +| C3s | W6 | 4 | 2 | 1.1% | PASS | +| C3w | W1 | 1 | 3 | 2.8% | PASS | +| C3w | W1 | 4 | 3 | 2.7% | PASS | +| C3w | W2 | 1 | 3 | 5.8% | PASS | +| C3w | W2 | 4 | 3 | 4.2% | PASS | +| C3w | W3 | 1 | 3 | 9.7% | PASS | +| C3w | W3 | 4 | 3 | 7.5% | PASS | +| C3w | W4 | 1 | 3 | 5.5% | PASS | +| C3w | W4 | 4 | 3 | 3.7% | PASS | +| C3w | W5 | 1 | 3 | 0.7% | PASS | +| C3w | W5 | 4 | 3 | 0.8% | PASS | +| C3w | W6 | 1 | 3 | 5.8% | PASS | +| C3w | W6 | 4 | 3 | 6.8% | PASS | +| C4 | W1 | 1 | 3 | 10.9% | FAIL | +| C4 | W1 | 4 | 3 | 10.6% | FAIL | +| C4 | W2 | 1 | 3 | 21.6% | FAIL | +| C4 | W2 | 4 | 3 | 21.4% | FAIL | +| C4 | W3 | 1 | 3 | 23.9% | FAIL | +| C4 | W3 | 4 | 3 | 22.0% | FAIL | +| C4 | W4 | 1 | 3 | 25.0% | FAIL | +| C4 | W4 | 4 | 3 | 23.9% | FAIL | +| C4 | W5 | 1 | 3 | 8.2% | PASS | +| C4 | W5 | 4 | 3 | 8.4% | PASS | +| C4 | W6 | 1 | 3 | 8.9% | PASS | +| C4 | W6 | 4 | 3 | 12.7% | FAIL | + +## Throughput + +| config | workload | conc | run | n | errors | elapsed s | req/s | +|---|---|---|---|---|---|---|---| +| C3 | W1 | 1 | m1 | 300 | 0 | 14.154 | 21.2 | +| C3 | W1 | 1 | m2 | 300 | 0 | 13.621 | 22.03 | +| C3 | W1 | 1 | m3 | 300 | 0 | 14.107 | 21.27 | +| C3 | W1 | 4 | m1 | 300 | 0 | 13.857 | 21.65 | +| C3 | W1 | 4 | m2 | 300 | 0 | 13.513 | 22.2 | +| C3 | W1 | 4 | m3 | 300 | 0 | 13.765 | 21.79 | +| C3 | W2 | 1 | m1 | 300 | 0 | 20.488 | 14.64 | +| C3 | W2 | 1 | m2 | 300 | 0 | 20.969 | 14.31 | +| C3 | W2 | 1 | m3 | 300 | 0 | 21.998 | 13.64 | +| C3 | W2 | 4 | m1 | 300 | 0 | 20.286 | 14.79 | +| C3 | W2 | 4 | m2 | 300 | 0 | 20.857 | 14.38 | +| C3 | W2 | 4 | m3 | 300 | 0 | 23.684 | 12.67 | +| C3 | W3 | 1 | m1 | 300 | 0 | 44.889 | 6.68 | +| C3 | W3 | 1 | m2 | 300 | 0 | 47.636 | 6.3 | +| C3 | W3 | 1 | m3 | 300 | 0 | 47.45 | 6.32 | +| C3 | W3 | 4 | m1 | 300 | 0 | 43.916 | 6.83 | +| C3 | W3 | 4 | m2 | 300 | 0 | 45.497 | 6.59 | +| C3 | W3 | 4 | m3 | 300 | 0 | 46.581 | 6.44 | +| C3 | W4 | 1 | m1 | 300 | 0 | 24.672 | 12.16 | +| C3 | W4 | 1 | m2 | 300 | 0 | 21.748 | 13.79 | +| C3 | W4 | 1 | m3 | 300 | 0 | 21.691 | 13.83 | +| C3 | W4 | 4 | m1 | 300 | 0 | 21.328 | 14.07 | +| C3 | W4 | 4 | m2 | 300 | 0 | 21.756 | 13.79 | +| C3 | W4 | 4 | m3 | 300 | 0 | 22.575 | 13.29 | +| C3 | W5 | 1 | m1 | 300 | 0 | 41.952 | 7.15 | +| C3 | W5 | 1 | m2 | 300 | 0 | 42.153 | 7.12 | +| C3 | W5 | 1 | m3 | 300 | 0 | 42.608 | 7.04 | +| C3 | W5 | 4 | m1 | 300 | 0 | 41.283 | 7.27 | +| C3 | W5 | 4 | m2 | 300 | 0 | 42.581 | 7.05 | +| C3 | W5 | 4 | m3 | 300 | 0 | 42.643 | 7.04 | +| C3 | W6 | 1 | m1 | 300 | 0 | 12.669 | 23.68 | +| C3 | W6 | 1 | m2 | 300 | 0 | 11.336 | 26.46 | +| C3 | W6 | 1 | m3 | 300 | 0 | 11.969 | 25.07 | +| C3 | W6 | 4 | m1 | 300 | 0 | 12.321 | 24.35 | +| C3 | W6 | 4 | m2 | 300 | 0 | 11.077 | 27.08 | +| C3 | W6 | 4 | m3 | 300 | 0 | 11.366 | 26.39 | +| C3s | W1 | 1 | m1 | 300 | 0 | 10.014 | 29.96 | +| C3s | W1 | 1 | m2 | 300 | 0 | 9.805 | 30.6 | +| C3s | W1 | 4 | m1 | 300 | 0 | 9.827 | 30.53 | +| C3s | W1 | 4 | m2 | 300 | 0 | 9.724 | 30.85 | +| C3s | W2 | 1 | m1 | 300 | 0 | 19.908 | 15.07 | +| C3s | W2 | 1 | m2 | 300 | 0 | 18.337 | 16.36 | +| C3s | W2 | 4 | m1 | 300 | 0 | 20.102 | 14.92 | +| C3s | W2 | 4 | m2 | 300 | 0 | 17.989 | 16.68 | +| C3s | W3 | 1 | m1 | 300 | 0 | 40.752 | 7.36 | +| C3s | W3 | 1 | m2 | 300 | 0 | 41.956 | 7.15 | +| C3s | W3 | 4 | m1 | 300 | 0 | 41.293 | 7.27 | +| C3s | W3 | 4 | m2 | 300 | 0 | 41.135 | 7.29 | +| C3s | W4 | 1 | m1 | 300 | 0 | 22.564 | 13.3 | +| C3s | W4 | 1 | m2 | 300 | 0 | 21.96 | 13.66 | +| C3s | W4 | 4 | m1 | 300 | 0 | 23.276 | 12.89 | +| C3s | W4 | 4 | m2 | 300 | 0 | 21.691 | 13.83 | +| C3s | W5 | 1 | m1 | 300 | 0 | 41.854 | 7.17 | +| C3s | W5 | 1 | m2 | 300 | 0 | 42.115 | 7.12 | +| C3s | W5 | 4 | m1 | 300 | 0 | 41.699 | 7.19 | +| C3s | W5 | 4 | m2 | 300 | 0 | 42.082 | 7.13 | +| C3s | W6 | 1 | m1 | 300 | 0 | 7.844 | 38.24 | +| C3s | W6 | 1 | m2 | 300 | 0 | 8.104 | 37.02 | +| C3s | W6 | 4 | m1 | 300 | 0 | 7.951 | 37.73 | +| C3s | W6 | 4 | m2 | 300 | 0 | 7.75 | 38.71 | +| C3w | W1 | 1 | m1 | 300 | 0 | 14.23 | 21.08 | +| C3w | W1 | 1 | m2 | 300 | 0 | 13.797 | 21.74 | +| C3w | W1 | 1 | m3 | 300 | 0 | 13.723 | 21.86 | +| C3w | W1 | 4 | m1 | 300 | 0 | 14.283 | 21.0 | +| C3w | W1 | 4 | m2 | 300 | 0 | 13.695 | 21.91 | +| C3w | W1 | 4 | m3 | 300 | 0 | 13.598 | 22.06 | +| C3w | W2 | 1 | m1 | 300 | 0 | 20.634 | 14.54 | +| C3w | W2 | 1 | m2 | 300 | 0 | 21.885 | 13.71 | +| C3w | W2 | 1 | m3 | 300 | 0 | 20.623 | 14.55 | +| C3w | W2 | 4 | m1 | 300 | 0 | 20.979 | 14.3 | +| C3w | W2 | 4 | m2 | 300 | 0 | 21.438 | 13.99 | +| C3w | W2 | 4 | m3 | 300 | 0 | 20.407 | 14.7 | +| C3w | W3 | 1 | m1 | 300 | 0 | 43.546 | 6.89 | +| C3w | W3 | 1 | m2 | 300 | 0 | 48.108 | 6.24 | +| C3w | W3 | 1 | m3 | 300 | 0 | 44.116 | 6.8 | +| C3w | W3 | 4 | m1 | 300 | 0 | 44.403 | 6.76 | +| C3w | W3 | 4 | m2 | 300 | 0 | 47.512 | 6.31 | +| C3w | W3 | 4 | m3 | 300 | 0 | 46.673 | 6.43 | +| C3w | W4 | 1 | m1 | 300 | 0 | 21.166 | 14.17 | +| C3w | W4 | 1 | m2 | 300 | 0 | 22.353 | 13.42 | +| C3w | W4 | 1 | m3 | 300 | 0 | 21.306 | 14.08 | +| C3w | W4 | 4 | m1 | 300 | 0 | 21.26 | 14.11 | +| C3w | W4 | 4 | m2 | 300 | 0 | 21.892 | 13.7 | +| C3w | W4 | 4 | m3 | 300 | 0 | 21.221 | 14.14 | +| C3w | W5 | 1 | m1 | 300 | 0 | 41.88 | 7.16 | +| C3w | W5 | 1 | m2 | 300 | 0 | 42.071 | 7.13 | +| C3w | W5 | 1 | m3 | 300 | 0 | 41.77 | 7.18 | +| C3w | W5 | 4 | m1 | 300 | 0 | 41.844 | 7.17 | +| C3w | W5 | 4 | m2 | 300 | 0 | 42.047 | 7.13 | +| C3w | W5 | 4 | m3 | 300 | 0 | 41.688 | 7.2 | +| C3w | W6 | 1 | m1 | 300 | 0 | 12.158 | 24.68 | +| C3w | W6 | 1 | m2 | 300 | 0 | 11.569 | 25.93 | +| C3w | W6 | 1 | m3 | 300 | 0 | 11.765 | 25.5 | +| C3w | W6 | 4 | m1 | 300 | 0 | 12.095 | 24.8 | +| C3w | W6 | 4 | m2 | 300 | 0 | 11.303 | 26.54 | +| C3w | W6 | 4 | m3 | 300 | 0 | 11.899 | 25.21 | +| C4 | W1 | 1 | m1 | 300 | 0 | 15.462 | 19.4 | +| C4 | W1 | 1 | m2 | 300 | 0 | 14.197 | 21.13 | +| C4 | W1 | 1 | m3 | 300 | 0 | 14.242 | 21.06 | +| C4 | W1 | 4 | m1 | 300 | 0 | 15.132 | 19.83 | +| C4 | W1 | 4 | m2 | 300 | 0 | 13.707 | 21.89 | +| C4 | W1 | 4 | m3 | 300 | 0 | 14.085 | 21.3 | +| C4 | W2 | 1 | m1 | 300 | 0 | 26.424 | 11.35 | +| C4 | W2 | 1 | m2 | 300 | 0 | 25.247 | 11.88 | +| C4 | W2 | 1 | m3 | 300 | 0 | 21.495 | 13.96 | +| C4 | W2 | 4 | m1 | 300 | 0 | 26.038 | 11.52 | +| C4 | W2 | 4 | m2 | 300 | 0 | 22.054 | 13.6 | +| C4 | W2 | 4 | m3 | 300 | 0 | 21.819 | 13.75 | +| C4 | W3 | 1 | m1 | 300 | 0 | 47.598 | 6.3 | +| C4 | W3 | 1 | m2 | 300 | 0 | 55.974 | 5.36 | +| C4 | W3 | 1 | m3 | 300 | 0 | 55.228 | 5.43 | +| C4 | W3 | 4 | m1 | 300 | 0 | 59.984 | 5.0 | +| C4 | W3 | 4 | m2 | 300 | 0 | 56.483 | 5.31 | +| C4 | W3 | 4 | m3 | 300 | 0 | 48.002 | 6.25 | +| C4 | W4 | 1 | m1 | 300 | 0 | 27.28 | 11.0 | +| C4 | W4 | 1 | m2 | 300 | 0 | 21.849 | 13.73 | +| C4 | W4 | 1 | m3 | 300 | 0 | 22.367 | 13.41 | +| C4 | W4 | 4 | m1 | 300 | 0 | 27.241 | 11.01 | +| C4 | W4 | 4 | m2 | 300 | 0 | 21.794 | 13.77 | +| C4 | W4 | 4 | m3 | 300 | 0 | 22.164 | 13.54 | +| C4 | W5 | 1 | m1 | 300 | 0 | 41.931 | 7.15 | +| C4 | W5 | 1 | m2 | 300 | 0 | 42.141 | 7.12 | +| C4 | W5 | 1 | m3 | 300 | 0 | 45.56 | 6.58 | +| C4 | W5 | 4 | m1 | 300 | 0 | 42.006 | 7.14 | +| C4 | W5 | 4 | m2 | 300 | 0 | 42.196 | 7.11 | +| C4 | W5 | 4 | m3 | 300 | 0 | 45.537 | 6.59 | +| C4 | W6 | 1 | m1 | 300 | 0 | 12.559 | 23.89 | +| C4 | W6 | 1 | m2 | 300 | 0 | 11.658 | 25.73 | +| C4 | W6 | 1 | m3 | 300 | 0 | 11.846 | 25.33 | +| C4 | W6 | 4 | m1 | 300 | 0 | 12.541 | 23.92 | +| C4 | W6 | 4 | m2 | 300 | 0 | 11.173 | 26.85 | +| C4 | W6 | 4 | m3 | 300 | 0 | 11.785 | 25.46 | + +## Memory (MB) + +| config | run | footprint after warmup | footprint end | footprint peak | MPS driver after warmup | MPS driver end | frontend footprint | +|---|---|---|---|---|---|---|---| +| C1 | m1 | 2030 | 2014 | 2030 | | | | +| C1 | m2 | 2025 | 2003 | 2029 | | | | +| C1 | m3 | 2022 | 2013 | 2023 | | | | +| C2 | m1 | 4169 | 3490 | 4174 | 3085 | 3102 | | +| C2 | m2 | 4170 | 3489 | 4174 | 3085 | 3102 | | +| C3 | m1 | 4184 | 3513 | 4188 | | | | +| C3 | m2 | 4182 | 3497 | 4187 | | | | +| C3 | m3 | 4181 | 3513 | 4185 | | | | +| C3s | m1 | 3927 | 3664 | 4239 | | | | +| C3s | m2 | 4086 | 3728 | 4431 | | | | +| C3w | m1 | 4198 | 3519 | 4203 | | | | +| C3w | m2 | 4193 | 3519 | 4198 | | | | +| C3w | m3 | 4199 | 3519 | 4204 | | | | +| C4 | m1 | 4184 | 3497 | 4189 | | | 3 | +| C4 | m2 | 4181 | 3513 | 4185 | | | 3 | +| C4 | m3 | 4185 | 3512 | 4189 | | | 3 | From a2b8b298d9c4305da03c9602d97be3e62cdcce2c Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:32:19 +0800 Subject: [PATCH 03/21] Warm every loaded Laya model and harden the HTTP benchmark Worker: warm up, compile and describe every model the router has loaded, not only LAYA_WORKER_MODEL. With LAYA_MODELS unset laya-serve preloads all checkpoints, and a request auto-routed to one of the others was served cold. A model that was not preloaded is no longer loaded just for warmup. /health lists each model under `models`; device_mismatch is set if any model is off the requested device. Worker: report the revision the weights were actually downloaded from, recorded from snapshot_download's path, instead of reading refs/main from the cache, which can move or not apply to a local checkpoint. bench_http: close the connection on any failed request so the next one reconnects (a timeout used to turn every later request on that thread into CannotSendRequest), record a failed answers request instead of aborting the run, and count only successful requests in req/s. --- benchmarks/laya/bench_http.py | 40 +++++++++++--- benchmarks/laya/report.py | 2 +- src/models/laya/README.md | 9 ++-- src/models/laya/tests/test_worker.py | 63 ++++++++++++++++++++-- src/models/laya/worker.py | 79 ++++++++++++++++++++-------- 5 files changed, 154 insertions(+), 39 deletions(-) diff --git a/benchmarks/laya/bench_http.py b/benchmarks/laya/bench_http.py index 8b706e8..54aa60e 100644 --- a/benchmarks/laya/bench_http.py +++ b/benchmarks/laya/bench_http.py @@ -42,7 +42,8 @@ def __init__(self, url, token=None): def request(self, method, path, body=None, retry=False): """Timed calls pass retry=False so a dropped connection shows up as an error, not a slow request. - Untimed calls retry once: uvicorn closes keep-alive connections idle for 5 s.""" + Untimed calls retry once: uvicorn closes keep-alive connections idle for 5 s. Any failure closes + the connection, so the next call reconnects instead of failing on a half-finished exchange.""" try: started = time.perf_counter() self.conn.request(method, path, body=body, headers=self.headers) @@ -54,6 +55,9 @@ def request(self, method, path, body=None, retry=False): if not retry: raise return self.request(method, path, body) + except BaseException: + self.conn.close() + raise def close(self): self.conn.close() @@ -79,6 +83,20 @@ def wait_ready(url, processes, timeout_s): sys.exit(f"worker not ready after {timeout_s} s") +def fetch_answers(client, body): + """(answers, None) or (None, error record fields) for one untimed request.""" + try: + _, status, data = client.request("POST", "/v1/systemone", body, retry=True) + except (OSError, http.client.HTTPException) as exc: + return None, {"status": 0, "detail": repr(exc)} + if status != 200: + return None, {"status": status, "detail": data[:200].decode(errors="replace")} + try: + return json.loads(data)["answers"], None + except (ValueError, KeyError) as exc: + return None, {"status": status, "detail": f"no answers in response: {exc!r}"} + + def body_for(workload, model): return json.dumps({"model": model, "state": workload["state"], "questions": workload["questions"]}).encode() @@ -244,11 +262,14 @@ def emit(record): random.Random(args.seed).shuffle(order) for w in order: body = body_for(w, args.model) - _, _, data = client.request("POST", "/v1/systemone", body, retry=True) - emit({"type": "answers", "workload": w["id"], "answers": json.loads(data)["answers"]}) + answers, error = fetch_answers(client, body) + if error: # parity.py reports the workload as missing + emit({"type": "answers_error", "workload": w["id"], **error}) + else: + emit({"type": "answers", "workload": w["id"], "answers": answers}) for concurrency in args.concurrency: results, elapsed = run_level(args.url, token, body, args.n, concurrency) - bad = [s for _, _, s in results if s != 200] + ok = sum(s == 200 for _, _, s in results) for i, (thread, ms, status) in enumerate(results): emit( { @@ -268,15 +289,18 @@ def emit(record): "workload": w["id"], "concurrency": concurrency, "n": len(results), - "errors": len(bad), + "errors": len(results) - ok, "elapsed_s": round(elapsed, 3), - "rps": round(len(results) / elapsed, 2), + "rps": round(ok / elapsed, 2), # successful requests only } ) for w in parity: - _, _, data = client.request("POST", "/v1/systemone", body_for(w, args.model), retry=True) - emit({"type": "answers", "workload": w["id"], "answers": json.loads(data)["answers"]}) + answers, error = fetch_answers(client, body_for(w, args.model)) + if error: + emit({"type": "answers_error", "workload": w["id"], **error}) + else: + emit({"type": "answers", "workload": w["id"], "answers": answers}) _, _, health_end = client.request("GET", "/health", retry=True) client.close() emit({"type": "end", "health": json.loads(health_end), **memory()}) diff --git a/benchmarks/laya/report.py b/benchmarks/laya/report.py index cb1c9ee..3a5cd30 100644 --- a/benchmarks/laya/report.py +++ b/benchmarks/laya/report.py @@ -132,7 +132,7 @@ def throughput(records): for r in records if r["type"] == "throughput" ] - return table(["config", "workload", "conc", "run", "n", "errors", "elapsed s", "req/s"], sorted(rows)) + return table(["config", "workload", "conc", "run", "n", "errors", "elapsed s", "successful req/s"], sorted(rows)) def memory(phases, ends): diff --git a/src/models/laya/README.md b/src/models/laya/README.md index 2599277..be7d5d4 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -11,11 +11,12 @@ validated on an M1 Pro). No native CUDA or Metal backend yet. `worker.py` runs laya-serve (`laya[serve]==0.3.20`) with two changes: -- It binds only after a warmup over short, long and multi-question requests, so `/health` never - answers for a worker that has not run a forward pass. Without it the first request after `/health` +- It binds only after a warmup of every loaded model over short, long and multi-question requests, + so `/health` never answers for a worker that has not run a forward pass. Without it the first request after `/health` returned 200 took ~230 ms against ~45 ms warm on an M1 Pro (MPS); with it, ~73 ms. -- `/health` reports the device, weight and autocast dtypes of the loaded model, the checkpoint and - revision, and `device_mismatch` when the model is not on the device `LAYA_DEVICE` asked for. +- `/health` reports, for each loaded model under `models` and for `LAYA_WORKER_MODEL` at the top level, + the device, weight and autocast dtypes, the checkpoint and the revision its weights were downloaded + from, and `device_mismatch` when a model is not on the device `LAYA_DEVICE` asked for. laya-serve reports `LAYA_DEVICE` as configured, and laya falls back to CPU with only a printed warning. `LAYA_REQUIRE_DEVICE=1` makes the worker exit instead. diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index aa92469..c9f2da1 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -25,17 +25,20 @@ def __init__(self, device="mps", dtype="torch.float16"): class FakeRouter: """The part of laya.router.Router the worker and laya.serve.create_app use.""" - def __init__(self, agent=None, fail_on_call=None): + def __init__(self, agent=None, fail_on_call=None, agents=None): self.agent = agent or FakeAgent() + self.agents = agents if agents is not None else {"english": self.agent} self.calls = [] + self.loads = [] self.fail_on_call = fail_on_call @property def loaded(self): - return ["english"] + return list(self.agents) def load(self, name): - return self.agent + self.loads.append(name) + return self.agents.setdefault(name, self.agent) def predict(self, state, questions, model=None): self.calls.append((state, questions, model)) @@ -107,7 +110,7 @@ def test_auto_device_is_never_a_mismatch(): def test_require_device_refuses_to_serve_on_another_device(): router = FakeRouter(FakeAgent(device="cpu")) - with pytest.raises(RuntimeError, match="asked for mps, model is on cpu"): + with pytest.raises(RuntimeError, match="asked for mps, english is on cpu"): worker.create_worker_app(router, "english", "mps", require_device=True) @@ -194,3 +197,55 @@ def forward(self, input_ids): assert agent.model(torch.zeros(1, 7)) == "compiled" assert agent.model(torch.zeros(3, 7)) == "eager" assert [p.shape for p in agent.model.parameters()] == [torch.Size([1])] # one set of weights + + +def test_every_loaded_model_is_warmed_and_described(): + agents = {"english": FakeAgent(), "multilingual": FakeAgent(device="cpu", dtype="torch.float32")} + router = FakeRouter(agents=agents) + health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() + per_model = len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + assert [m for _, _, m in router.calls].count("multilingual") == per_model + assert [m for _, _, m in router.calls].count("english") == per_model + assert set(health["models"]) == {"english", "multilingual"} + assert health["device"] == "mps" # the top level summarises LAYA_WORKER_MODEL + assert health["models"]["multilingual"]["device"] == "cpu" + assert health["device_mismatch"] is True # one model off the requested device is enough + + +def test_a_model_that_is_not_preloaded_is_not_loaded_for_warmup(): + router = FakeRouter(agents={"multilingual": FakeAgent()}) + health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() + assert "english" not in router.loads + assert {m for _, _, m in router.calls} == {"multilingual"} + assert set(health["models"]) == {"multilingual"} + + +def test_nothing_preloaded_warms_the_worker_model(): + router = FakeRouter(agents={}) + worker.create_worker_app(router, "english", "mps") + assert {m for _, _, m in router.calls} == {"english"} + + +def test_revision_comes_from_the_loaded_snapshot_not_a_guess(): + revisions = {"convaiinnovations/laya": "55cf4c4"} + health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps", revisions=revisions)).get("/health") + assert health.json()["revision"] == "55cf4c4" + unknown = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() + assert unknown["revision"] is None + + +def test_record_snapshot_revisions_reads_the_downloaded_path(monkeypatch): + import huggingface_hub + + paths = { + "convaiinnovations/laya": "/cache/models--convaiinnovations--laya/snapshots/55cf4c4abc/multilingual", + "/local/checkpoint": "/local/checkpoint", + } + monkeypatch.setattr(huggingface_hub, "snapshot_download", lambda repo_id, **kwargs: paths[repo_id]) + revisions = worker.record_snapshot_revisions() + assert ( + huggingface_hub.snapshot_download("convaiinnovations/laya", allow_patterns=["*"]) + == paths["convaiinnovations/laya"] + ) + huggingface_hub.snapshot_download("/local/checkpoint") + assert revisions == {"convaiinnovations/laya": "55cf4c4abc"} diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 37315d8..d4992c8 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -4,14 +4,17 @@ LAYA_DEVICE as configured rather than where the model ended up. This worker reuses laya's app and request handling unchanged and fixes both: -- It binds only after a warmup that covers short, long and multi-question requests (the last one - crosses laya's fp16 autocast threshold on MPS), so a reachable worker is a warm one. -- /health reports the device, weight and autocast dtypes of the loaded agent, the checkpoint and - revision it serves, and whether the device differs from the one requested. +- It binds only after a warmup of every loaded model that covers short, long and multi-question + requests (the last one crosses laya's fp16 autocast threshold on MPS), so a reachable worker is a + warm one whichever model a request is routed to. +- /health reports, per loaded model, the device, weight and autocast dtypes, the checkpoint and the + revision its weights were downloaded from, and whether the device differs from the one requested. Configuration is laya-serve's (LAYA_HOST, LAYA_PORT, LAYA_DEVICE, LAYA_MODELS, LAYA_API_KEY, ...) plus: - LAYA_WORKER_MODEL model to warm up and describe english + LAYA_WORKER_MODEL model summarised at the top of /health; english + also the one loaded when nothing is + preloaded (LAYA_PRELOAD=0) LAYA_REQUIRE_DEVICE exit instead of serving on another 0 device than LAYA_DEVICE asked for LAYA_WORKER_COMPILE off, all, or single: torch.compile off @@ -102,19 +105,33 @@ def forward(self, input_ids, *args, **kwargs): agent.model = SingleRowCompiled() -def _cached_revision(repo: str | None, ref: str = "main") -> str | None: - if not repo: - return None - try: - from huggingface_hub.constants import HF_HUB_CACHE +def record_snapshot_revisions() -> dict[str, str]: + """Record the commit each Hugging Face checkpoint is loaded from, keyed by repo id. + + laya calls huggingface_hub.snapshot_download while loading and keeps only the repo id; the + returned path (.../snapshots//...) is the only place the loaded revision appears. Call this + before the router loads anything. A checkpoint loaded from a local path records nothing. + """ + import huggingface_hub + + revisions: dict[str, str] = {} + original = huggingface_hub.snapshot_download - return (Path(HF_HUB_CACHE) / f"models--{repo.replace('/', '--')}" / "refs" / ref).read_text().strip() - except (ImportError, OSError): - return None + def recording(repo_id, *args, **kwargs): + path = original(repo_id, *args, **kwargs) + parts = Path(path).parts + if "snapshots" in parts[:-1]: + revisions[repo_id] = parts[parts.index("snapshots") + 1] + return path + huggingface_hub.snapshot_download = recording + return revisions -def describe(agent: Any, requested: str | None, routing: dict[str, Any] | None) -> dict[str, Any]: - """What /health reports about the loaded agent.""" + +def describe( + agent: Any, requested: str | None, routing: dict[str, Any] | None, revisions: dict[str, str] | None = None +) -> dict[str, Any]: + """What /health reports about one loaded agent.""" device = str(getattr(agent, "device", "unknown")) model = getattr(agent, "model", None) try: @@ -131,7 +148,7 @@ def describe(agent: Any, requested: str | None, routing: dict[str, Any] | None) "autocast_dtype": str(getattr(agent, "dtype", None)), "mps_amp_min_rows": getattr(agent, "mps_amp_min_rows", None), "checkpoint": repo, - "revision": _cached_revision(repo), + "revision": (revisions or {}).get(repo), } @@ -142,21 +159,37 @@ def create_worker_app( require_device: bool = False, compile: str = "off", graph_counter=compiled_graphs, + revisions: dict[str, str] | None = None, ): - """Optionally compile, warm the router up, then return laya's app with /health replaced. + """Optionally compile, warm up every loaded model, then return laya's app with /health replaced. + `model` is the one summarised at the top of /health, and the one loaded if nothing is preloaded. Raises if compiling or warmup fails.""" from laya.serve import create_app if compile not in ("off", "all", "single"): raise ValueError(f"compile mode must be off, all or single, not {compile!r}") + names = list(router.loaded) or [model] # never load a model the worker was not asked to serve if compile != "off": - compile_agent(router.load(model), compile) - warm = warmup(router, model) - info = describe(router.load(model), requested, warm["routing"]) - info["warmup_ms"] = warm["warmup_ms"] + for name in names: + compile_agent(router.load(name), compile) + models = {} + for name in names: + warm = warmup(router, name) + models[name] = { + **describe(router.load(name), requested, warm["routing"], revisions), + "warmup_ms": warm["warmup_ms"], + } + primary = model if model in models else names[0] + info = { + **models[primary], + "device_mismatch": any(m["device_mismatch"] for m in models.values()), + "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), + "models": models, + } graphs_at_ready = graph_counter() if compile != "off" else None if info["device_mismatch"]: - message = f"asked for {info['requested_device']}, model is on {info['device']}" + wrong = ", ".join(f"{n} is on {m['device']}" for n, m in models.items() if m["device_mismatch"]) + message = f"asked for {info['requested_device']}, {wrong}" if require_device: raise RuntimeError(message) log.warning(message) @@ -184,6 +217,7 @@ def main() -> None: logging.basicConfig(level=logging.INFO, format="%(name)s: %(message)s") model = os.environ.get("LAYA_WORKER_MODEL", "english") requested = os.environ.get("LAYA_DEVICE") or None + revisions = record_snapshot_revisions() try: app = create_worker_app( build_router(), @@ -191,6 +225,7 @@ def main() -> None: requested, require_device=_env_bool("LAYA_REQUIRE_DEVICE"), compile=COMPILE_MODES.get(os.environ.get("LAYA_WORKER_COMPILE", "").strip().lower(), "invalid"), + revisions=revisions, ) except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever sys.exit(f"laya-worker: not starting: {exc}") From d44d3fe31834b00f517f386e0081f3eea3cb84bc Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Mon, 28 Sep 2026 23:41:18 +0800 Subject: [PATCH 04/21] Move the Laya benchmark scripts under recipe/laya/bench Keep the repository to the src/ and recipe/ layout: the scripts that reproduce the recipe's numbers now live next to it, like the Cua-S1 benchmarks under recipe/cua_s1/. The README describes only how to run them and where the results are. --- benchmarks/laya/README.md | 56 ------------------- recipe/laya/apple-silicon.md | 17 +++--- recipe/laya/bench/README.md | 51 +++++++++++++++++ .../laya => recipe/laya/bench}/bench_http.py | 4 +- .../laya/bench}/bench_inproc.py | 2 +- .../laya/bench}/check_workloads.py | 0 {benchmarks/laya => recipe/laya/bench}/env.py | 4 +- .../laya/bench}/frontend_overhead.py | 2 +- .../laya => recipe/laya/bench}/parity.py | 2 +- .../laya => recipe/laya/bench}/profile_mps.py | 2 +- .../laya => recipe/laya/bench}/report.py | 2 +- .../laya/bench}/results/.gitignore | 0 .../bench}/results/frontend_overhead_m1.md | 0 .../laya/bench}/results/measured-parity.md | 0 .../laya/bench}/results/measured-report.md | 2 +- .../laya/bench}/workloads.jsonl | 0 .../laya/bench}/workloads.src.py | 0 src/models/laya/worker.py | 2 +- 18 files changed, 70 insertions(+), 76 deletions(-) delete mode 100644 benchmarks/laya/README.md create mode 100644 recipe/laya/bench/README.md rename {benchmarks/laya => recipe/laya/bench}/bench_http.py (98%) rename {benchmarks/laya => recipe/laya/bench}/bench_inproc.py (98%) rename {benchmarks/laya => recipe/laya/bench}/check_workloads.py (100%) rename {benchmarks/laya => recipe/laya/bench}/env.py (98%) rename {benchmarks/laya => recipe/laya/bench}/frontend_overhead.py (96%) rename {benchmarks/laya => recipe/laya/bench}/parity.py (98%) rename {benchmarks/laya => recipe/laya/bench}/profile_mps.py (99%) rename {benchmarks/laya => recipe/laya/bench}/report.py (98%) rename {benchmarks/laya => recipe/laya/bench}/results/.gitignore (100%) rename {benchmarks/laya => recipe/laya/bench}/results/frontend_overhead_m1.md (100%) rename {benchmarks/laya => recipe/laya/bench}/results/measured-parity.md (100%) rename {benchmarks/laya => recipe/laya/bench}/results/measured-report.md (99%) rename {benchmarks/laya => recipe/laya/bench}/workloads.jsonl (100%) rename {benchmarks/laya => recipe/laya/bench}/workloads.src.py (100%) diff --git a/benchmarks/laya/README.md b/benchmarks/laya/README.md deleted file mode 100644 index e1e686c..0000000 --- a/benchmarks/laya/README.md +++ /dev/null @@ -1,56 +0,0 @@ -# Laya benchmarks - -Measures Laya on CPU and Apple Silicon (MPS) for [#3](https://github.com/ThinkFlowLab/system1-omni/issues/3). -Every script writes raw JSONL; `report.py` is the only place numbers are computed. - -| file | purpose | -| --- | --- | -| `workloads.src.py` → `workloads.jsonl` | fixed inputs: W1–W6 timed, P* parity-only | -| `check_workloads.py` | tokens per row with Laya's tokenizer, ±10% of each target | -| `bench_inproc.py` | in-process: import, load, warmup, first request per workload, warm latency, memory | -| `bench_http.py` | against a `/v1/systemone` worker: process-to-ready (with `--spawn`), first request per workload, warm latency and throughput at each `--concurrency` | -| `profile_mps.py` | where a request's time goes on MPS: length sweep with a fixed-cost fit, stage split (encode, dispatch, GPU wait, copy back, decode) and host operator counts | -| `frontend_overhead.py` | frontend cost, paired: each request direct and through the frontend back to back, so background load cancels | -| `parity.py` | answers of every run vs a reference run, tolerances fixed in advance (fp32 1e-3, fp16 1e-2) | -| `env.py` | run header: SHAs, versions, checkpoint revision, hardware, power, load | -| `report.py` | JSONL → tables, including the run-to-run gate | - -## Run - -Use the same environment as the Laya recipe (`laya[serve]==0.3.20`, Python 3.12). - -```sh -python benchmarks/laya/check_workloads.py -python benchmarks/laya/bench_inproc.py --device cpu --config C1 --run feasibility -python benchmarks/laya/bench_inproc.py --device mps --config C2 --run feasibility -python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m1 -python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m2 -python benchmarks/laya/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve -python benchmarks/laya/bench_http.py --config C4 --run feasibility --url http://127.0.0.1:8080 -python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl -python benchmarks/laya/parity.py benchmarks/laya/results/*.jsonl --ref C1 -``` - -Measured runs (any `--run` other than `feasibility`) refuse to start on battery power or when the -1-minute load average is above `--max-load` (default 2). Two measured runs of a config pass when -their p50s differ by at most 10% for every workload. - -Timing: wall clock around `Agent.system_one` followed by `torch.mps.synchronize()`, so it includes -tokenization and post-processing. Workload order is shuffled per run with `--seed`. - -Memory is the physical footprint of the process running Laya (`proc_pid_rusage`, the same number as -Activity Monitor's "Memory" and `footprint -p`). On Apple silicon it includes Metal allocations, so -in-process and worker numbers are comparable and MPS tensors are counted. - -## Results - -`results/` holds the reports built from the measured runs on an M1 Pro: `measured-report.md` -(`report.py`), `measured-parity.md` (`parity.py --ref C1`) and `frontend_overhead_m1.md`. The raw -JSONL they were built from (7 MB, 16 files) is published as a release asset rather than committed: - -```sh -curl -LO https://github.com/cacheline999/system1-omni/releases/download/laya-mps-results-2026-09-28/laya-mps-results-2026-09-28.tar.gz -shasum -a 256 laya-mps-results-2026-09-28.tar.gz # 611ed30707ac8c98875b5aa5382360b5a7d760da166d61c626eb07ebe1ee6404 -tar xzf laya-mps-results-2026-09-28.tar.gz -C benchmarks/laya/results -python benchmarks/laya/report.py benchmarks/laya/results/*_m[0-9].jsonl -``` diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index dd21bb5..ef82605 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -97,17 +97,16 @@ LAYA_CONTRACT=1 .venv/bin/python -m pytest src/models/laya/tests # plus contr ## Benchmark -Stop the worker and frontend first; the benchmark starts its own. The suite and its measurement -rules are described in [`benchmarks/laya/`](../../benchmarks/laya/README.md). A first pass that -checks everything runs: +Stop the worker and frontend first; the benchmark starts its own. The scripts are listed in +[`bench/`](bench/README.md). A first pass that checks everything runs: ```sh -.venv/bin/python benchmarks/laya/check_workloads.py -.venv/bin/python benchmarks/laya/bench_inproc.py --device mps --config C2 --run feasibility -.venv/bin/python benchmarks/laya/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve -.venv/bin/python benchmarks/laya/bench_http.py --config C4 --run feasibility \ +.venv/bin/python recipe/laya/bench/check_workloads.py +.venv/bin/python recipe/laya/bench/bench_inproc.py --device mps --config C2 --run feasibility +.venv/bin/python recipe/laya/bench/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve +.venv/bin/python recipe/laya/bench/bench_http.py --config C4 --run feasibility \ --url http://127.0.0.1:8080 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve -.venv/bin/python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl +.venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*.jsonl ``` Runs labelled anything other than `feasibility` refuse to start on battery power or when the @@ -115,7 +114,7 @@ Runs labelled anything other than `feasibility` refuse to start on battery power ## Troubleshooting -- `device_mismatch: true`, or the worker exits with `asked for mps, model is on cpu`: MPS is not +- `device_mismatch: true`, or the worker exits with `asked for mps, english is on cpu`: MPS is not available to this Python. Check the `torch.backends.mps.is_available()` line above; an x86_64 Python running under Rosetta cannot use MPS. - The worker process uses about 4 GB (Activity Monitor's Memory column, which counts MPS diff --git a/recipe/laya/bench/README.md b/recipe/laya/bench/README.md new file mode 100644 index 0000000..df9ed0c --- /dev/null +++ b/recipe/laya/bench/README.md @@ -0,0 +1,51 @@ +# Laya benchmark scripts + +Scripts behind the numbers in the [Apple Silicon recipe](../apple-silicon.md). Each run writes raw +JSONL to `results/`; `report.py` and `parity.py` build the tables from it. + +| file | purpose | +| --- | --- | +| `workloads.src.py` → `workloads.jsonl` | fixed inputs: W1–W6 timed, P* parity only | +| `check_workloads.py` | token count of each input with Laya's tokenizer | +| `bench_inproc.py` | Laya in-process: load, warmup, first request, warm latency, memory | +| `bench_http.py` | a `/v1/systemone` worker, optionally behind the frontend: time to ready, first request, warm latency, throughput | +| `frontend_overhead.py` | frontend cost, each request sent directly and through the frontend back to back | +| `profile_mps.py` | where a request's time goes on MPS | +| `parity.py` | answers of every run against a reference run | +| `report.py` | tables from the JSONL | +| `env.py` | versions, checkpoint, hardware and load recorded with each run | + +## Run + +From the repository root, in the environment of the recipe: + +```sh +python recipe/laya/bench/check_workloads.py +python recipe/laya/bench/bench_inproc.py --device cpu --config C1 --run m1 +python recipe/laya/bench/bench_inproc.py --device mps --config C2 --run m1 +python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv/bin/laya-serve +python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ + --frontend target/release/omni-jev --spawn .venv/bin/laya-serve +LAYA_WORKER_COMPILE=off python recipe/laya/bench/bench_http.py --config C3w --run m1 \ + --spawn .venv/bin/python src/models/laya/worker.py +LAYA_WORKER_COMPILE=single python recipe/laya/bench/bench_http.py --config C3s --run m1 \ + --spawn .venv/bin/python src/models/laya/worker.py +python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl +python recipe/laya/bench/parity.py recipe/laya/bench/results/*_m[0-9].jsonl --ref C1 +``` + +Repeat with `--run m2` for a second measured run. Runs refuse to start on battery power or above a +1-minute load average of `--max-load` (default 2) unless labelled `--run feasibility`. Memory is the +process's physical footprint, which on Apple Silicon includes MPS allocations. + +## Results + +`results/` holds the reports from the measured runs on an M1 Pro: `measured-report.md`, +`measured-parity.md` and `frontend_overhead_m1.md`. The raw JSONL (7 MB) is a release asset: + +```sh +curl -LO https://github.com/cacheline999/system1-omni/releases/download/laya-mps-results-2026-09-28/laya-mps-results-2026-09-28.tar.gz +shasum -a 256 laya-mps-results-2026-09-28.tar.gz # 611ed30707ac8c98875b5aa5382360b5a7d760da166d61c626eb07ebe1ee6404 +tar xzf laya-mps-results-2026-09-28.tar.gz -C recipe/laya/bench/results +python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl +``` diff --git a/benchmarks/laya/bench_http.py b/recipe/laya/bench/bench_http.py similarity index 98% rename from benchmarks/laya/bench_http.py rename to recipe/laya/bench/bench_http.py index 54aa60e..67215b7 100644 --- a/benchmarks/laya/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -6,8 +6,8 @@ --url in front of it; readiness is then the frontend's /health, which proxies the worker's. Each workload runs at every --concurrency level; each client thread keeps one keep-alive connection. - python benchmarks/laya/bench_http.py --config C3 --run m1 --spawn .venv-laya/bin/laya-serve - python benchmarks/laya/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ + python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv-laya/bin/laya-serve + python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ --frontend target/release/omni-jev --spawn .venv-laya/bin/laya-serve """ diff --git a/benchmarks/laya/bench_inproc.py b/recipe/laya/bench/bench_inproc.py similarity index 98% rename from benchmarks/laya/bench_inproc.py rename to recipe/laya/bench/bench_inproc.py index 3d3eb59..5eb8e09 100644 --- a/benchmarks/laya/bench_inproc.py +++ b/recipe/laya/bench/bench_inproc.py @@ -3,7 +3,7 @@ Phases are timed separately: import, load, warmup, then warm requests. Every request is one line of JSONL; report.py turns the file into tables. Run from the repository root or this directory: - python benchmarks/laya/bench_inproc.py --device mps --config C2 --run m1 + python recipe/laya/bench/bench_inproc.py --device mps --config C2 --run m1 """ import argparse diff --git a/benchmarks/laya/check_workloads.py b/recipe/laya/bench/check_workloads.py similarity index 100% rename from benchmarks/laya/check_workloads.py rename to recipe/laya/bench/check_workloads.py diff --git a/benchmarks/laya/env.py b/recipe/laya/bench/env.py similarity index 98% rename from benchmarks/laya/env.py rename to recipe/laya/bench/env.py index 758e771..edee840 100644 --- a/benchmarks/laya/env.py +++ b/recipe/laya/bench/env.py @@ -12,7 +12,7 @@ from datetime import datetime, timezone from pathlib import Path -REPO = Path(__file__).resolve().parents[2] +REPO = Path(__file__).resolve().parents[3] def _run(*cmd): @@ -72,7 +72,7 @@ def noise_problems(max_load): def header(checkpoint, **extra): - status = _run("git", "-C", str(REPO), "status", "--porcelain", "--", ".", ":!benchmarks/laya/results") + status = _run("git", "-C", str(REPO), "status", "--porcelain", "--", ".", ":!recipe/laya/bench/results") return { "type": "env", "utc": datetime.now(timezone.utc).isoformat(timespec="seconds"), diff --git a/benchmarks/laya/frontend_overhead.py b/recipe/laya/bench/frontend_overhead.py similarity index 96% rename from benchmarks/laya/frontend_overhead.py rename to recipe/laya/bench/frontend_overhead.py index 4d3f753..dea614c 100644 --- a/benchmarks/laya/frontend_overhead.py +++ b/recipe/laya/bench/frontend_overhead.py @@ -4,7 +4,7 @@ Start a worker and the frontend first (recipe/laya/apple-silicon.md), then: - python benchmarks/laya/frontend_overhead.py --direct http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 + python recipe/laya/bench/frontend_overhead.py --direct http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 """ import argparse diff --git a/benchmarks/laya/parity.py b/recipe/laya/bench/parity.py similarity index 98% rename from benchmarks/laya/parity.py rename to recipe/laya/bench/parity.py index 4fb998f..944d8c7 100644 --- a/benchmarks/laya/parity.py +++ b/recipe/laya/bench/parity.py @@ -1,6 +1,6 @@ """Compare every run's answers with a reference run, using the tolerances declared in advance. - python benchmarks/laya/parity.py benchmarks/laya/results/*.jsonl --ref C1 + python recipe/laya/bench/parity.py recipe/laya/bench/results/*.jsonl --ref C1 Per question: the decision must match (choice: chosen option; score: most likely level; noul: side of 0.5) and the largest absolute probability difference must stay within tolerance: 1e-3 when the request diff --git a/benchmarks/laya/profile_mps.py b/recipe/laya/bench/profile_mps.py similarity index 99% rename from benchmarks/laya/profile_mps.py rename to recipe/laya/bench/profile_mps.py index 28d2183..4a9298d 100644 --- a/benchmarks/laya/profile_mps.py +++ b/recipe/laya/bench/profile_mps.py @@ -12,7 +12,7 @@ 3. ops: torch.profiler CPU trace of the forward: operator calls per request and the top operators by self CPU time, i.e. the host cost of issuing the forward. - python benchmarks/laya/profile_mps.py --run feasibility + python recipe/laya/bench/profile_mps.py --run feasibility """ import argparse diff --git a/benchmarks/laya/report.py b/recipe/laya/bench/report.py similarity index 98% rename from benchmarks/laya/report.py rename to recipe/laya/bench/report.py index 3a5cd30..23bb3d9 100644 --- a/benchmarks/laya/report.py +++ b/recipe/laya/bench/report.py @@ -1,6 +1,6 @@ """Turn benchmark JSONL into markdown tables. The only place numbers are computed from raw data. - python benchmarks/laya/report.py benchmarks/laya/results/*.jsonl + python recipe/laya/bench/report.py recipe/laya/bench/results/*.jsonl Percentiles are nearest-rank. The run-to-run gate compares p50 across measured runs (every run whose label is not "feasibility") of the same config, workload and concurrency: (max - min) / min <= 10%. diff --git a/benchmarks/laya/results/.gitignore b/recipe/laya/bench/results/.gitignore similarity index 100% rename from benchmarks/laya/results/.gitignore rename to recipe/laya/bench/results/.gitignore diff --git a/benchmarks/laya/results/frontend_overhead_m1.md b/recipe/laya/bench/results/frontend_overhead_m1.md similarity index 100% rename from benchmarks/laya/results/frontend_overhead_m1.md rename to recipe/laya/bench/results/frontend_overhead_m1.md diff --git a/benchmarks/laya/results/measured-parity.md b/recipe/laya/bench/results/measured-parity.md similarity index 100% rename from benchmarks/laya/results/measured-parity.md rename to recipe/laya/bench/results/measured-parity.md diff --git a/benchmarks/laya/results/measured-report.md b/recipe/laya/bench/results/measured-report.md similarity index 99% rename from benchmarks/laya/results/measured-report.md rename to recipe/laya/bench/results/measured-report.md index 7974689..3889ce5 100644 --- a/benchmarks/laya/results/measured-report.md +++ b/recipe/laya/bench/results/measured-report.md @@ -274,7 +274,7 @@ ## Throughput -| config | workload | conc | run | n | errors | elapsed s | req/s | +| config | workload | conc | run | n | errors | elapsed s | successful req/s | |---|---|---|---|---|---|---|---| | C3 | W1 | 1 | m1 | 300 | 0 | 14.154 | 21.2 | | C3 | W1 | 1 | m2 | 300 | 0 | 13.621 | 22.03 | diff --git a/benchmarks/laya/workloads.jsonl b/recipe/laya/bench/workloads.jsonl similarity index 100% rename from benchmarks/laya/workloads.jsonl rename to recipe/laya/bench/workloads.jsonl diff --git a/benchmarks/laya/workloads.src.py b/recipe/laya/bench/workloads.src.py similarity index 100% rename from benchmarks/laya/workloads.src.py rename to recipe/laya/bench/workloads.src.py diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index d4992c8..799afd9 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -25,7 +25,7 @@ M1 Pro the worker with `single` was ready after about 22 s instead of 8 s (`all` took over a minute). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means a request hit a shape class the warmup did not cover. `single` exists because on MPS compiling cut one-question -latency by about 29% while compiling everything made multi-question requests slower (benchmarks/laya). +latency by about 29% while compiling everything made multi-question requests slower (recipe/laya/bench). """ import logging From d3d0817b55f058508fc431c3f9bcbc77cffb8e91 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 09:47:10 +0800 Subject: [PATCH 05/21] Add LAYA_WORKER_WEIGHTS=fp16 for the compiled worker Keep the checkpoint's fp16 weights instead of laya's fp32 upcast on MPS (act_head stays fp32, since laya feeds it fp32 features). With LAYA_WORKER_COMPILE=single, two paired runs with both workers alive lowered warm p50 by about 14% for one-question requests (median ratio 0.855-0.857 at 68 tokens), 9-11% for longer and six-question requests, raised it 4% for three questions, and cut the worker's memory from 3.6 GB to 2.7 GB. Answers stayed within 0.0031 of the fp32 worker's. paired.py runs two worker configurations at once and sends each request to both back to back; separate runs could not resolve a 10% difference under background load. parity.py now applies the fp16 tolerance to every request of a worker whose weights are fp16. --- recipe/laya/apple-silicon.md | 9 + recipe/laya/bench/README.md | 17 +- recipe/laya/bench/paired.py | 211 +++++++++++++++++++++++ recipe/laya/bench/parity.py | 7 +- recipe/laya/bench/results/paired-fp16.md | 30 ++++ src/models/laya/README.md | 4 +- src/models/laya/tests/test_worker.py | 30 ++++ src/models/laya/worker.py | 22 +++ 8 files changed, 325 insertions(+), 5 deletions(-) create mode 100644 recipe/laya/bench/paired.py create mode 100644 recipe/laya/bench/results/paired-fp16.md diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index ef82605..c180d10 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -66,6 +66,15 @@ exist now; `recompiled_after_ready: true` means a request shape was not covered compiles every path; in feasibility runs it made multi-question requests up to 65% slower and took over a minute to start, so it is not recommended. +### fp16 weights + +Add `LAYA_WORKER_WEIGHTS=fp16` to the command above to keep the checkpoint's fp16 weights instead of +Laya's fp32 upcast on MPS. With `single`, in two paired runs (both workers alive, every request sent to +each back to back) it lowered warm p50 by about 14% for one-question requests (median ratio 0.855–0.857 +at 68 tokens), about 10% at 198–484 tokens and 11% for six questions, and raised it by 4% for three +questions. The worker's memory dropped from 3.6 GB to 2.7 GB. Answers stayed within 0.0031 of the +fp32 worker's. Without compile, fp16 weights did not make one-question requests faster. + ## Start the frontend In another terminal: diff --git a/recipe/laya/bench/README.md b/recipe/laya/bench/README.md index df9ed0c..f1fbda7 100644 --- a/recipe/laya/bench/README.md +++ b/recipe/laya/bench/README.md @@ -9,6 +9,7 @@ JSONL to `results/`; `report.py` and `parity.py` build the tables from it. | `check_workloads.py` | token count of each input with Laya's tokenizer | | `bench_inproc.py` | Laya in-process: load, warmup, first request, warm latency, memory | | `bench_http.py` | a `/v1/systemone` worker, optionally behind the frontend: time to ready, first request, warm latency, throughput | +| `paired.py` | two worker configurations alive at once, each request sent to both back to back; median ratio with a bootstrap interval | | `frontend_overhead.py` | frontend cost, each request sent directly and through the frontend back to back | | `profile_mps.py` | where a request's time goes on MPS | | `parity.py` | answers of every run against a reference run | @@ -38,10 +39,20 @@ Repeat with `--run m2` for a second measured run. Runs refuse to start on batter 1-minute load average of `--max-load` (default 2) unless labelled `--run feasibility`. Memory is the process's physical footprint, which on Apple Silicon includes MPS allocations. +Two worker configurations can also be compared request by request, which holds up under background +load better than separate runs: + +```sh +python recipe/laya/bench/paired.py --run p1 --a "LAYA_WORKER_COMPILE=single" \ + --b "LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16" +python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_p1.jsonl +``` + ## Results `results/` holds the reports from the measured runs on an M1 Pro: `measured-report.md`, -`measured-parity.md` and `frontend_overhead_m1.md`. The raw JSONL (7 MB) is a release asset: +`measured-parity.md`, `frontend_overhead_m1.md` and `paired-fp16.md`. The raw JSONL is published as +release assets: ```sh curl -LO https://github.com/cacheline999/system1-omni/releases/download/laya-mps-results-2026-09-28/laya-mps-results-2026-09-28.tar.gz @@ -49,3 +60,7 @@ shasum -a 256 laya-mps-results-2026-09-28.tar.gz # 611ed30707ac8c98875b5aa5382 tar xzf laya-mps-results-2026-09-28.tar.gz -C recipe/laya/bench/results python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl ``` + +The paired fp16 runs (`paired_e4a.jsonl`, `paired_e4b.jsonl`) are in +`laya-mps-paired-fp16-2026-09-30.tar.gz` on the same release (sha256 `cc0d6f5bda6e3e0ee1f40c6966f84429e902a2f585b8e1ee33658a9be139326e`); rebuild the summary with +`paired.py --summarize`. diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py new file mode 100644 index 0000000..796feae --- /dev/null +++ b/recipe/laya/bench/paired.py @@ -0,0 +1,211 @@ +"""Paired comparison of two worker configurations: both run at once, and every request goes to A and to B +back to back, alternating which goes first, so background load that shifts both cancels out. + + python recipe/laya/bench/paired.py --run e4a \ + --a "LAYA_WORKER_COMPILE=single" --b "LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16" + python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_e4*.jsonl + +Each side is src/models/laya/worker.py started with the given environment. The summary reports, per +input, the median of the per-pair ratio B/A with a 95% bootstrap interval, and B's answers against A's. +""" + +import argparse +import json +import os +import random +import statistics +import subprocess +import sys +from pathlib import Path + +HERE = Path(__file__).resolve().parent +REPO = HERE.parents[2] +sys.path.insert(0, str(HERE)) +from bench_http import Client, body_for, fetch_answers, wait_ready +from env import footprint_mb, header, noise_problems + +CHECKPOINT = "convaiinnovations/laya" + + +def spawn(env_spec, port, python, log_path): + env = { + **os.environ, + "LAYA_HOST": "127.0.0.1", + "LAYA_PORT": str(port), + "LAYA_DEVICE": "mps", + "LAYA_MODELS": "english", + "LAYA_LOG_LEVEL": "warning", + } + env.update(pair.split("=", 1) for pair in env_spec.split()) + log = open(log_path, "w") # noqa: SIM115 + return subprocess.Popen( + [python, str(REPO / "src/models/laya/worker.py")], env=env, stdout=log, stderr=subprocess.STDOUT + ) + + +def run(args): + problems = noise_problems(args.max_load) + if problems and args.run != "feasibility": + sys.exit("refusing a measured run: " + "; ".join(problems)) + with open(args.workloads) as f: + workloads = [json.loads(line) for line in f if line.strip()] + bench = [w for w in workloads if w["kind"] == "bench"] + parity = [w for w in workloads if w["kind"] == "parity"] + out_dir = Path(args.out) + out_dir.mkdir(parents=True, exist_ok=True) + sides = {"A": (args.a, args.port_a), "B": (args.b, args.port_b)} + procs = { + s: spawn(spec, port, args.python, out_dir / f"paired_{args.run}_{s}.log") for s, (spec, port) in sides.items() + } + urls = {s: f"http://127.0.0.1:{port}" for s, (_, port) in sides.items()} + try: + health = {s: wait_ready(urls[s], {s: procs[s]}, args.ready_timeout)[1] for s in sides} + with open(out_dir / f"paired_{args.run}.jsonl", "w") as f: + + def emit(record): + f.write(json.dumps({"run": args.run, **record}) + "\n") + + emit( + header( + CHECKPOINT, + a=args.a, + b=args.b, + n=args.n, + discard=args.discard, + seed=args.seed, + health=health, + noise=problems, + ) + ) + clients = {s: Client(urls[s]) for s in sides} + for w in bench + parity: + body = body_for(w, args.model) + for s in sides: + answers, error = fetch_answers(clients[s], body) + emit({"type": "answers", "side": s, "workload": w["id"], "answers": answers, "error": error}) + order = bench[:] + random.Random(args.seed).shuffle(order) + for w in order: + body = body_for(w, args.model) + for i in range(args.discard + args.n): + first, second = ("A", "B") if i % 2 == 0 else ("B", "A") + ms = {} + for s in (first, second): + t, status, _ = clients[s].request("POST", "/v1/systemone", body, retry=True) + ms[s] = t if status == 200 else None + if i >= args.discard: + emit( + { + "type": "pair", + "workload": w["id"], + "i": i - args.discard, + "first": first, + "a_ms": ms["A"], + "b_ms": ms["B"], + "rows": len(w["questions"]), + } + ) + end_health = {s: json.loads(clients[s].request("GET", "/health", retry=True)[2]) for s in sides} + emit( + { + "type": "end", + "health": end_health, + "footprint_mb": {s: footprint_mb(procs[s].pid).get("footprint_mb") for s in sides}, + } + ) + finally: + for p in procs.values(): + p.terminate() + p.wait(timeout=30) + print(out_dir / f"paired_{args.run}.jsonl") + + +def median_interval(ratios, seed=0, resamples=2000): + rng = random.Random(seed) + meds = sorted(statistics.median(rng.choices(ratios, k=len(ratios))) for _ in range(resamples)) + return statistics.median(ratios), meds[int(0.025 * resamples)], meds[int(0.975 * resamples) - 1] + + +def flat(answer): + return answer.get("probabilities", {"p": answer.get("noul")}) + + +def decision(answer): + if answer["type"] == "choice": + return answer["choice"] + if answer["type"] == "noul": + return answer["noul"] >= 0.5 + return max(answer["probabilities"], key=answer["probabilities"].get) + + +def margin(answer): + p = sorted(flat(answer).values(), reverse=True) + return abs(p[0] - 0.5) if len(p) == 1 else p[0] - p[1] + + +def summarize(paths): + for path in paths: + with open(path) as f: + records = [json.loads(line) for line in f if line.strip()] + env = next(r for r in records if r["type"] == "env") + print(f"## {env['run']}: A = `{env['a']}`, B = `{env['b']}`, load at start {env['loadavg_1m']}\n") + print("| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval |\n|---|---|---|---|---|---|") + pairs = {} + for r in records: + if r["type"] == "pair" and r["a_ms"] and r["b_ms"]: + pairs.setdefault(r["workload"], []).append(r) + for wid in sorted(pairs): + ps = pairs[wid] + med, lo, hi = median_interval([p["b_ms"] / p["a_ms"] for p in ps]) + print( + f"| {wid} | {len(ps)} | {statistics.median(p['a_ms'] for p in ps):.1f} | " + f"{statistics.median(p['b_ms'] for p in ps):.1f} | {med:.3f} | {lo:.3f}–{hi:.3f} |" + ) + answers = {} + for r in records: + if r["type"] == "answers": + answers.setdefault(r["workload"], {})[r["side"]] = r + worst, flips, errors = 0.0, [], [] + for wid, sides in answers.items(): + if sides["A"]["error"] or sides["B"]["error"]: + errors.append(wid) + continue + for q, a in sides["A"]["answers"].items(): + b = sides["B"]["answers"][q] + worst = max(worst, max(abs(flat(a)[k] - flat(b)[k]) for k in flat(a))) + if decision(a) != decision(b): + flips.append((wid, q, round(margin(a), 4))) + end = next(r for r in records if r["type"] == "end") + compile_state = {s: h.get("compile", {}).get("recompiled_after_ready") for s, h in end["health"].items()} + print(f"\nB vs A answers: max |Δp| {worst:.4f}, flips {flips}, errors {errors}") + print(f"recompiled after ready: {compile_state}; footprint MB: {end['footprint_mb']}\n") + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--summarize", nargs="+", metavar="JSONL") + parser.add_argument("--run") + parser.add_argument("--a", default="LAYA_WORKER_COMPILE=single", help="environment of side A") + parser.add_argument("--b", default="LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16") + parser.add_argument("--port-a", type=int, default=8000) + parser.add_argument("--port-b", type=int, default=8001) + parser.add_argument("--python", default=sys.executable) + parser.add_argument("--model", default="english") + parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) + parser.add_argument("-n", type=int, default=300, help="timed pairs per input") + parser.add_argument("--discard", type=int, default=20) + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--ready-timeout", type=float, default=900) + parser.add_argument("--max-load", type=float, default=2.0) + parser.add_argument("--out", default=str(HERE / "results")) + args = parser.parse_args() + if args.summarize: + summarize(args.summarize) + elif args.run: + run(args) + else: + parser.error("give --run or --summarize") + + +if __name__ == "__main__": + main() diff --git a/recipe/laya/bench/parity.py b/recipe/laya/bench/parity.py index 944d8c7..a7bb794 100644 --- a/recipe/laya/bench/parity.py +++ b/recipe/laya/bench/parity.py @@ -4,7 +4,8 @@ Per question: the decision must match (choice: chosen option; score: most likely level; noul: side of 0.5) and the largest absolute probability difference must stay within tolerance: 1e-3 when the request -ran in fp32, 1e-2 when it ran under fp16 autocast (MPS and rows >= the worker's amp threshold). +ran in fp32, 1e-2 when it ran in fp16: fp16 weights (every request), or fp16 autocast on MPS for +rows >= the worker's amp threshold. A flipped decision is reported with the reference margin between its top two outcomes. Exits 1 when any question fails. """ @@ -52,12 +53,12 @@ def margin(probs): def path(env, rows): - fp16 = ( + autocast = ( env.get("device_actual") == "mps" and env.get("amp_dtype") == "torch.float16" and rows >= (env.get("mps_amp_min_rows") or 10**9) ) - return "fp16" if fp16 else "fp32" + return "fp16" if autocast or env.get("weights_dtype") == "torch.float16" else "fp32" def main(): diff --git a/recipe/laya/bench/results/paired-fp16.md b/recipe/laya/bench/results/paired-fp16.md new file mode 100644 index 0000000..37c47ac --- /dev/null +++ b/recipe/laya/bench/results/paired-fp16.md @@ -0,0 +1,30 @@ +Paired comparison of the worker with LAYA_WORKER_COMPILE=single, fp32 weights (A) and fp16 weights (B), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. + +## e4a: A = `LAYA_WORKER_COMPILE=single`, B = `LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16`, load at start 31.32 + +| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | +|---|---|---|---|---|---| +| W1 | 300 | 33.2 | 28.5 | 0.857 | 0.855–0.861 | +| W2 | 300 | 62.8 | 56.9 | 0.910 | 0.907–0.913 | +| W3 | 300 | 152.8 | 137.0 | 0.900 | 0.897–0.905 | +| W4 | 300 | 71.6 | 74.6 | 1.042 | 1.040–1.045 | +| W5 | 300 | 164.0 | 145.9 | 0.894 | 0.891–0.899 | +| W6 | 300 | 26.5 | 22.7 | 0.855 | 0.853–0.859 | + +B vs A answers: max |Δp| 0.0031, flips [], errors [] +recompiled after ready: {'A': False, 'B': False}; footprint MB: {'A': 3647, 'B': 2717} + +## e4b: A = `LAYA_WORKER_COMPILE=single`, B = `LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16`, load at start 18.75 + +| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | +|---|---|---|---|---|---| +| W1 | 300 | 34.9 | 29.7 | 0.855 | 0.851–0.861 | +| W2 | 300 | 68.3 | 61.8 | 0.910 | 0.908–0.913 | +| W3 | 300 | 158.8 | 143.3 | 0.902 | 0.897–0.908 | +| W4 | 300 | 90.0 | 93.6 | 1.040 | 1.032–1.047 | +| W5 | 300 | 161.7 | 143.2 | 0.891 | 0.889–0.896 | +| W6 | 300 | 32.7 | 28.0 | 0.850 | 0.839–0.861 | + +B vs A answers: max |Δp| 0.0031, flips [], errors [] +recompiled after ready: {'A': False, 'B': False}; footprint MB: {'A': 3646, 'B': 2721} + diff --git a/src/models/laya/README.md b/src/models/laya/README.md index be7d5d4..e9db01c 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -23,7 +23,9 @@ validated on an M1 Pro). No native CUDA or Metal backend yet. Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, `LAYA_API_KEY`, ...), plus `LAYA_WORKER_COMPILE=off|single|all`: `single` sends one-question requests through a `torch.compile(dynamic=True)` graph compiled during warmup and runs the rest eagerly; `/health` reports -compiled graphs at readiness and now. See the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). +compiled graphs at readiness and now, and `LAYA_WORKER_WEIGHTS=fp32|fp16`: `fp16` keeps the checkpoint's +fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast. See the +[Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). ```sh LAYA_DEVICE=mps LAYA_MODELS=english python src/models/laya/worker.py diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index c9f2da1..cee2916 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -249,3 +249,33 @@ def test_record_snapshot_revisions_reads_the_downloaded_path(monkeypatch): ) huggingface_hub.snapshot_download("/local/checkpoint") assert revisions == {"convaiinnovations/laya": "55cf4c4abc"} + + +def test_fp16_weights_keep_act_head_in_fp32(): + import torch + + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + self.encoder = torch.nn.Linear(4, 4) + self.act_head = torch.nn.Linear(4, 2) + + agent = FakeAgent() + agent.model = Model() + worker.use_fp16_weights(agent) + assert agent.model.encoder.weight.dtype == torch.float16 + assert agent.model.act_head.weight.dtype == torch.float32 + + +def test_fp16_weights_are_applied_to_every_loaded_model_before_warmup(monkeypatch): + order = [] + agents = {"english": FakeAgent(), "multilingual": FakeAgent()} + router = FakeRouter(agents=agents) + monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: order.append((agent, len(router.calls)))) + worker.create_worker_app(router, "english", "mps", weights="fp16") + assert order == [(agents["english"], 0), (agents["multilingual"], 0)] + + +def test_unknown_weights_mode_is_refused(): + with pytest.raises(ValueError, match="fp32 or fp16"): + worker.create_worker_app(FakeRouter(), "english", "mps", weights="int8") diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 799afd9..19279d5 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -20,6 +20,9 @@ LAYA_WORKER_COMPILE off, all, or single: torch.compile off (dynamic=True) the model before warmup, for every batch or one-row batches only + LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 + fp16 weights instead of laya's fp32 + upcast on MPS and CPU The warmup also compiles every shape class it sends through the compiled model: in measured runs on an M1 Pro the worker with `single` was ready after about 22 s instead of 8 s (`all` took over a minute). @@ -82,6 +85,18 @@ def compiled_graphs() -> int: COMPILE_MODES = {"0": "off", "off": "off", "": "off", "1": "all", "all": "all", "single": "single"} +WEIGHT_MODES = {"": "fp32", "fp32": "fp32", "fp16": "fp16"} + + +def use_fp16_weights(agent: Any) -> None: + """Keep the weights in fp16, the checkpoint's own precision, so the conversion is exact. laya 0.3.20 + upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features.""" + agent.model.half() + act_head = getattr(agent.model, "act_head", None) + if act_head is not None: + act_head.float() + + def compile_agent(agent: Any, mode: str) -> None: """`all`: every forward goes through the compiled model. `single`: batches of one row (one question) do, everything else runs eager. Both paths share the same parameters.""" @@ -160,6 +175,7 @@ def create_worker_app( compile: str = "off", graph_counter=compiled_graphs, revisions: dict[str, str] | None = None, + weights: str = "fp32", ): """Optionally compile, warm up every loaded model, then return laya's app with /health replaced. `model` is the one summarised at the top of /health, and the one loaded if nothing is preloaded. @@ -168,7 +184,12 @@ def create_worker_app( if compile not in ("off", "all", "single"): raise ValueError(f"compile mode must be off, all or single, not {compile!r}") + if weights not in ("fp32", "fp16"): + raise ValueError(f"weights must be fp32 or fp16, not {weights!r}") names = list(router.loaded) or [model] # never load a model the worker was not asked to serve + if weights == "fp16": + for name in names: + use_fp16_weights(router.load(name)) if compile != "off": for name in names: compile_agent(router.load(name), compile) @@ -226,6 +247,7 @@ def main() -> None: require_device=_env_bool("LAYA_REQUIRE_DEVICE"), compile=COMPILE_MODES.get(os.environ.get("LAYA_WORKER_COMPILE", "").strip().lower(), "invalid"), revisions=revisions, + weights=WEIGHT_MODES.get(os.environ.get("LAYA_WORKER_WEIGHTS", "").strip().lower(), "invalid"), ) except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever sys.exit(f"laya-worker: not starting: {exc}") From e10cf06744fff4297a0bc1818b8c5099c78a8eab Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 11:54:08 +0800 Subject: [PATCH 06/21] Compile the encoder for multi-question batches; make compile on/off LAYA_WORKER_COMPILE is now off or on. On compiles one-question batches end to end, as before, and for several questions compiles only the encoder: Laya's decision head is two nn.TransformerEncoderLayer with a padding mask, which lose PyTorch's fused path when compiled and were about 50 ms slower on padded batches. The mode that compiled every batch is removed. With LAYA_WORKER_COMPILE=on and LAYA_WORKER_WEIGHTS=fp16 against the worker with neither, in two paired runs: median latency ratio 0.62 for a 68-token one-question request, 0.80-0.83 at 198-484 tokens, 0.86 for three questions and 0.82 for six; no input slower; answers within 0.0031; worker memory 3.5 GB -> 2.8 GB. --- recipe/laya/apple-silicon.md | 32 ++++------- recipe/laya/bench/README.md | 13 +++-- recipe/laya/bench/paired.py | 8 +-- .../bench/results/paired-all-optimizations.md | 30 ++++++++++ recipe/laya/bench/results/paired-fp16.md | 2 +- src/models/laya/README.md | 6 +- src/models/laya/tests/test_worker.py | 57 ++++++++++++------- src/models/laya/worker.py | 48 +++++++++------- 8 files changed, 123 insertions(+), 73 deletions(-) create mode 100644 recipe/laya/bench/results/paired-all-optimizations.md diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index c180d10..7cbb888 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -46,34 +46,26 @@ curl -s http://127.0.0.1:8000/health revision, the weight dtype (`torch.float32`; Laya upcasts the fp16 checkpoint on MPS), the autocast dtype Laya uses for requests with at least `mps_amp_min_rows` questions, and the warmup time. -### Compiled one-question path +### Faster: compile and fp16 weights ```sh -LAYA_WORKER_COMPILE=single LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ +LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ LAYA_MODELS=english LAYA_REQUIRE_DEVICE=1 \ .venv/bin/python src/models/laya/worker.py ``` -`single` sends requests with one question through a `torch.compile` graph and runs the rest eagerly. -In two measured runs on the M1 Pro it cut warm p50 for a 68-token one-question request from about -46 ms to 33 ms (−29%) and for a 47-token one from about 39 ms to 26 ms, left three- and six-question -requests within about 3%, and gave the same answers as CPU within the benchmark tolerances. The price is -startup: the worker was ready after about 22 s instead of 8 s while the warmup compiles, and the -gain shrinks with length (−7% to −13% at 484 tokens). +`LAYA_WORKER_COMPILE=on` compiles the model during warmup: one-question requests run the whole model +compiled, requests with several questions run the encoder compiled and Laya's decision head as it is. +`LAYA_WORKER_WEIGHTS=fp16` keeps the checkpoint's fp16 weights instead of Laya's fp32 upcast on MPS. + +On the M1 Pro, with both workers running and every request sent to each back to back, the two options +together lowered warm p50 against the worker without them by 37–38% for a 68-token one-question +request (about 57 → 35 ms in those runs), 17–20% at 198–484 tokens, 14% for three questions and 18% for +six, and cut the worker's memory from 3.5 GB to 2.8 GB. Answers stayed within 0.0031 of the fp32 +worker's. The price is startup: the worker becomes ready after 20–30 s instead of about 8 s. `/health` reports under `compile` how many graphs existed when the worker became ready and how many -exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. `all` -compiles every path; in feasibility runs it made multi-question requests up to 65% slower and took -over a minute to start, so it is not recommended. - -### fp16 weights - -Add `LAYA_WORKER_WEIGHTS=fp16` to the command above to keep the checkpoint's fp16 weights instead of -Laya's fp32 upcast on MPS. With `single`, in two paired runs (both workers alive, every request sent to -each back to back) it lowered warm p50 by about 14% for one-question requests (median ratio 0.855–0.857 -at 68 tokens), about 10% at 198–484 tokens and 11% for six questions, and raised it by 4% for three -questions. The worker's memory dropped from 3.6 GB to 2.7 GB. Answers stayed within 0.0031 of the -fp32 worker's. Without compile, fp16 weights did not make one-question requests faster. +exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. ## Start the frontend diff --git a/recipe/laya/bench/README.md b/recipe/laya/bench/README.md index f1fbda7..77e8a1a 100644 --- a/recipe/laya/bench/README.md +++ b/recipe/laya/bench/README.md @@ -29,7 +29,7 @@ python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve LAYA_WORKER_COMPILE=off python recipe/laya/bench/bench_http.py --config C3w --run m1 \ --spawn .venv/bin/python src/models/laya/worker.py -LAYA_WORKER_COMPILE=single python recipe/laya/bench/bench_http.py --config C3s --run m1 \ +LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 python recipe/laya/bench/bench_http.py --config C3o --run m1 \ --spawn .venv/bin/python src/models/laya/worker.py python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl python recipe/laya/bench/parity.py recipe/laya/bench/results/*_m[0-9].jsonl --ref C1 @@ -43,15 +43,17 @@ Two worker configurations can also be compared request by request, which holds u load better than separate runs: ```sh -python recipe/laya/bench/paired.py --run p1 --a "LAYA_WORKER_COMPILE=single" \ - --b "LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16" +python recipe/laya/bench/paired.py --run p1 --a "LAYA_WORKER_COMPILE=off" \ + --b "LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16" python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_p1.jsonl ``` ## Results `results/` holds the reports from the measured runs on an M1 Pro: `measured-report.md`, -`measured-parity.md`, `frontend_overhead_m1.md` and `paired-fp16.md`. The raw JSONL is published as +`measured-parity.md`, `frontend_overhead_m1.md`, `paired-fp16.md` and `paired-all-optimizations.md`. +`C3s` in `measured-report.md` and side B of `paired-fp16.md` ran an earlier compile mode that compiled +one-question requests only; `on` compiles them the same way and adds the encoder for several questions. The raw JSONL is published as release assets: ```sh @@ -64,3 +66,6 @@ python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl The paired fp16 runs (`paired_e4a.jsonl`, `paired_e4b.jsonl`) are in `laya-mps-paired-fp16-2026-09-30.tar.gz` on the same release (sha256 `cc0d6f5bda6e3e0ee1f40c6966f84429e902a2f585b8e1ee33658a9be139326e`); rebuild the summary with `paired.py --summarize`. + +The runs behind `paired-all-optimizations.md` (`paired_e5a.jsonl`, `paired_e5b.jsonl`) are in +`laya-mps-paired-all-2026-09-30.tar.gz` on the same release (sha256 `25d1b7bc9d6dff7173f9b972ebd0204f8fb4e2089fe2d27924d48ba6e589780c`). diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index 796feae..377fbaa 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -1,8 +1,8 @@ """Paired comparison of two worker configurations: both run at once, and every request goes to A and to B back to back, alternating which goes first, so background load that shifts both cancels out. - python recipe/laya/bench/paired.py --run e4a \ - --a "LAYA_WORKER_COMPILE=single" --b "LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16" + python recipe/laya/bench/paired.py --run p1 \ + --a "LAYA_WORKER_COMPILE=off" --b "LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16" python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_e4*.jsonl Each side is src/models/laya/worker.py started with the given environment. The summary reports, per @@ -185,8 +185,8 @@ def main(): parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) parser.add_argument("--summarize", nargs="+", metavar="JSONL") parser.add_argument("--run") - parser.add_argument("--a", default="LAYA_WORKER_COMPILE=single", help="environment of side A") - parser.add_argument("--b", default="LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16") + parser.add_argument("--a", default="LAYA_WORKER_COMPILE=off", help="environment of side A") + parser.add_argument("--b", default="LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16", help="environment of side B") parser.add_argument("--port-a", type=int, default=8000) parser.add_argument("--port-b", type=int, default=8001) parser.add_argument("--python", default=sys.executable) diff --git a/recipe/laya/bench/results/paired-all-optimizations.md b/recipe/laya/bench/results/paired-all-optimizations.md new file mode 100644 index 0000000..44cb564 --- /dev/null +++ b/recipe/laya/bench/results/paired-all-optimizations.md @@ -0,0 +1,30 @@ +Paired comparison of the worker with every optimization (B: LAYA_WORKER_COMPILE=on, LAYA_WORKER_WEIGHTS=fp16) against the worker with none (A: compile off, fp32 weights), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. + +## e5a: A = `LAYA_WORKER_COMPILE=off`, B = `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, load at start 10.84 + +| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | +|---|---|---|---|---|---| +| W1 | 300 | 55.3 | 34.7 | 0.626 | 0.620–0.633 | +| W2 | 300 | 79.6 | 63.4 | 0.798 | 0.792–0.803 | +| W3 | 300 | 184.5 | 153.2 | 0.833 | 0.826–0.842 | +| W4 | 300 | 95.3 | 82.0 | 0.859 | 0.853–0.868 | +| W5 | 300 | 181.4 | 148.7 | 0.818 | 0.808–0.825 | +| W6 | 300 | 44.8 | 27.6 | 0.616 | 0.608–0.625 | + +B vs A answers: max |Δp| 0.0031, flips [], errors [] +recompiled after ready: {'A': None, 'B': False}; footprint MB: {'A': 3501, 'B': 2849} + +## e5b: A = `LAYA_WORKER_COMPILE=off`, B = `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, load at start 15.12 + +| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | +|---|---|---|---|---|---| +| W1 | 300 | 59.3 | 35.9 | 0.616 | 0.609–0.628 | +| W2 | 300 | 137.0 | 108.2 | 0.795 | 0.778–0.814 | +| W3 | 300 | 199.1 | 166.7 | 0.832 | 0.825–0.838 | +| W4 | 300 | 101.2 | 87.1 | 0.863 | 0.857–0.868 | +| W5 | 300 | 187.8 | 156.2 | 0.824 | 0.818–0.834 | +| W6 | 300 | 64.7 | 36.8 | 0.584 | 0.569–0.600 | + +B vs A answers: max |Δp| 0.0031, flips [], errors [] +recompiled after ready: {'A': None, 'B': False}; footprint MB: {'A': 3533, 'B': 2783} + diff --git a/recipe/laya/bench/results/paired-fp16.md b/recipe/laya/bench/results/paired-fp16.md index 37c47ac..f748c29 100644 --- a/recipe/laya/bench/results/paired-fp16.md +++ b/recipe/laya/bench/results/paired-fp16.md @@ -1,4 +1,4 @@ -Paired comparison of the worker with LAYA_WORKER_COMPILE=single, fp32 weights (A) and fp16 weights (B), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. +Paired comparison of fp32 (A) and fp16 (B) weights with the one-question path compiled (then `LAYA_WORKER_COMPILE=single`, now the one-question part of `on`), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. ## e4a: A = `LAYA_WORKER_COMPILE=single`, B = `LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16`, load at start 31.32 diff --git a/src/models/laya/README.md b/src/models/laya/README.md index e9db01c..bb60ff3 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -21,9 +21,9 @@ validated on an M1 Pro). No native CUDA or Metal backend yet. warning. `LAYA_REQUIRE_DEVICE=1` makes the worker exit instead. Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, `LAYA_API_KEY`, ...), -plus `LAYA_WORKER_COMPILE=off|single|all`: `single` sends one-question requests through a -`torch.compile(dynamic=True)` graph compiled during warmup and runs the rest eagerly; `/health` reports -compiled graphs at readiness and now, and `LAYA_WORKER_WEIGHTS=fp32|fp16`: `fp16` keeps the checkpoint's +plus `LAYA_WORKER_COMPILE=off|on` and `LAYA_WORKER_WEIGHTS=fp32|fp16`. `on` compiles one-question +requests end to end and, for several questions, only the encoder (Laya's decision head is slower +compiled on MPS); `/health` reports compiled graphs at readiness and now. `fp16` keeps the checkpoint's fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast. See the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index cee2916..da3b276 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -144,9 +144,9 @@ def test_main_exits_non_zero_when_warmup_fails(monkeypatch): def test_compile_wraps_the_model_before_warmup(monkeypatch): router = FakeRouter() order = [] - monkeypatch.setattr(worker, "compile_agent", lambda agent, mode: order.append((mode, len(router.calls)))) - worker.create_worker_app(router, "english", "mps", compile="single", graph_counter=lambda: 3) - assert order == [("single", 0)] # before the first warmup request + monkeypatch.setattr(worker, "compile_agent", lambda agent: order.append((agent, len(router.calls)))) + worker.create_worker_app(router, "english", "mps", compile="on", graph_counter=lambda: 3) + assert order == [(router.agent, 0)] # before the first warmup request def test_health_reports_compile_off_by_default(): @@ -155,48 +155,67 @@ def test_health_reports_compile_off_by_default(): def test_health_flags_graphs_compiled_after_ready(monkeypatch): - monkeypatch.setattr(worker, "compile_agent", lambda agent, mode: None) + monkeypatch.setattr(worker, "compile_agent", lambda agent: None) graphs = iter([4, 4, 5]) # at readiness, first /health, second /health after a new shape compiled client = TestClient( - worker.create_worker_app(FakeRouter(), "english", "mps", compile="all", graph_counter=lambda: next(graphs)) + worker.create_worker_app(FakeRouter(), "english", "mps", compile="on", graph_counter=lambda: next(graphs)) ) first = client.get("/health").json()["compile"] - assert first == {"mode": "all", "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} + assert first == {"mode": "on", "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} assert client.get("/health").json()["compile"]["recompiled_after_ready"] is True def test_compile_failure_means_no_app(monkeypatch): - def broken(agent, mode): + def broken(agent): raise RuntimeError("inductor: unsupported op on mps") monkeypatch.setattr(worker, "compile_agent", broken) with pytest.raises(RuntimeError, match="unsupported op"): - worker.create_worker_app(FakeRouter(), "english", "mps", compile="all") + worker.create_worker_app(FakeRouter(), "english", "mps", compile="on") def test_unknown_compile_mode_is_refused(): - with pytest.raises(ValueError, match="off, all or single"): + with pytest.raises(ValueError, match="off or on"): worker.create_worker_app(FakeRouter(), "english", "mps", compile="invalid") -def test_single_row_batches_use_the_compiled_model(monkeypatch): +def test_compiled_paths_by_batch_rows(monkeypatch): import torch - class Echo(torch.nn.Module): + class Encoder(torch.nn.Module): + def forward(self, input_ids): + return "eager encoder" + + class Model(torch.nn.Module): def __init__(self): super().__init__() - self.weight = torch.nn.Parameter(torch.ones(1)) + self.encoder = Encoder() + self.head = torch.nn.Linear(2, 2) def forward(self, input_ids): - return "eager" + return ("head", self.encoder(input_ids)) + + class Stub(torch.nn.Module): + def __init__(self, kind): + super().__init__() + self.kind = kind + + def forward(self, *args): + return self.kind + + def fake_compile(module, dynamic): + assert dynamic is True + return Stub("compiled encoder" if isinstance(module, Encoder) else "whole model compiled") - monkeypatch.setattr(torch, "compile", lambda model, dynamic: lambda input_ids: "compiled") + monkeypatch.setattr(torch, "compile", fake_compile) agent = FakeAgent() - agent.model = Echo() - worker.compile_agent(agent, "single") - assert agent.model(torch.zeros(1, 7)) == "compiled" - assert agent.model(torch.zeros(3, 7)) == "eager" - assert [p.shape for p in agent.model.parameters()] == [torch.Size([1])] # one set of weights + model = Model() + agent.model = model + worker.compile_agent(agent) + assert agent.model(torch.zeros(1, 7)) == "whole model compiled" # one question + assert agent.model(torch.zeros(3, 7)) == ("head", "compiled encoder") # several: eager head, compiled encoder + assert model.encoder(torch.zeros(1, 7)) == "eager encoder" # the original model is left as it was + assert len(list(agent.model.parameters())) == len(list(model.parameters())) # one set of weights def test_every_loaded_model_is_warmed_and_described(): diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 19279d5..fae08ba 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -17,18 +17,15 @@ preloaded (LAYA_PRELOAD=0) LAYA_REQUIRE_DEVICE exit instead of serving on another 0 device than LAYA_DEVICE asked for - LAYA_WORKER_COMPILE off, all, or single: torch.compile off - (dynamic=True) the model before warmup, - for every batch or one-row batches only + LAYA_WORKER_COMPILE off or on: torch.compile (dynamic=True) off + before warmup; see compile_agent LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 fp16 weights instead of laya's fp32 upcast on MPS and CPU -The warmup also compiles every shape class it sends through the compiled model: in measured runs on an -M1 Pro the worker with `single` was ready after about 22 s instead of 8 s (`all` took over a minute). -/health counts compiled graphs at readiness and now; `recompiled_after_ready` means a request hit a -shape class the warmup did not cover. `single` exists because on MPS compiling cut one-question -latency by about 29% while compiling everything made multi-question requests slower (recipe/laya/bench). +The warmup also compiles the graphs, so the worker takes longer to become ready (20–30 s instead of +about 8 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means +a request hit a shape class the warmup did not cover. """ import logging @@ -82,7 +79,7 @@ def compiled_graphs() -> int: return int(counters["stats"]["unique_graphs"]) -COMPILE_MODES = {"0": "off", "off": "off", "": "off", "1": "all", "all": "all", "single": "single"} +COMPILE_MODES = {"": "off", "0": "off", "off": "off", "1": "on", "on": "on"} WEIGHT_MODES = {"": "fp32", "fp32": "fp32", "fp16": "fp16"} @@ -97,27 +94,34 @@ def use_fp16_weights(agent: Any) -> None: act_head.float() -def compile_agent(agent: Any, mode: str) -> None: - """`all`: every forward goes through the compiled model. `single`: batches of one row (one question) - do, everything else runs eager. Both paths share the same parameters.""" +def compile_agent(agent: Any) -> None: + """Compile the model for the batches where it pays off on MPS, sharing the same parameters. + + A batch of one row (one question) runs the whole model compiled. A batch of several rows runs only + the encoder compiled and laya's decision head eagerly: the head is two nn.TransformerEncoderLayer + with a key padding mask, which lose PyTorch's fused fast path when compiled and were about 50 ms + slower on padded multi-row batches. + """ + import copy + import torch eager = agent.model - compiled = torch.compile(eager, dynamic=True) - if mode == "all": - agent.model = compiled - return + whole = torch.compile(eager, dynamic=True) + encoder_only = copy.copy(eager) # same parameters and submodules ... + encoder_only._modules = dict(eager._modules) # ... except the encoder slot + encoder_only._modules["encoder"] = torch.compile(eager.encoder, dynamic=True) - class SingleRowCompiled(torch.nn.Module): + class Compiled(torch.nn.Module): def __init__(self): super().__init__() self.eager = eager def forward(self, input_ids, *args, **kwargs): - model = compiled if input_ids.shape[0] == 1 else self.eager + model = whole if input_ids.shape[0] == 1 else encoder_only return model(input_ids, *args, **kwargs) - agent.model = SingleRowCompiled() + agent.model = Compiled() def record_snapshot_revisions() -> dict[str, str]: @@ -182,8 +186,8 @@ def create_worker_app( Raises if compiling or warmup fails.""" from laya.serve import create_app - if compile not in ("off", "all", "single"): - raise ValueError(f"compile mode must be off, all or single, not {compile!r}") + if compile not in ("off", "on"): + raise ValueError(f"compile must be off or on, not {compile!r}") if weights not in ("fp32", "fp16"): raise ValueError(f"weights must be fp32 or fp16, not {weights!r}") names = list(router.loaded) or [model] # never load a model the worker was not asked to serve @@ -192,7 +196,7 @@ def create_worker_app( use_fp16_weights(router.load(name)) if compile != "off": for name in names: - compile_agent(router.load(name), compile) + compile_agent(router.load(name)) models = {} for name in names: warm = warmup(router, name) From b00b285ab9ecc7e94ed70ef555c632c541dc660e Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:40:27 +0800 Subject: [PATCH 07/21] Recipe: M5 results, test dependencies, first-request note Add the M5 validation from the PR discussion, where fp16 weights speed up every input even without compile. Install httpx2 for the tests: recent Starlette's TestClient needs it (httpx is only a deprecated fallback). State that the first request after ready was occasionally slow on an M5 (2 of 23 fresh starts). --- recipe/laya/apple-silicon.md | 19 +++++++++++++------ 1 file changed, 13 insertions(+), 6 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 7cbb888..230909f 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -5,8 +5,8 @@ from [`src/models/laya/`](../../src/models/laya/), puts the Rust frontend in fro the benchmark suite. The [Laya text worker](README.md) recipe covers the plain CPU setup. Validated on an M1 Pro (16 GB, 16-core GPU), macOS 26.1, Python 3.12, `laya[serve]==0.3.20`, -torch 2.14.0 and the `english` checkpoint (`convaiinnovations/laya` at `55cf4c4`). Other M-series -Macs have not been tested. +torch 2.14.0 and the `english` checkpoint (`convaiinnovations/laya` at `55cf4c4`), and by another +contributor on an M5 (10-core GPU, 32 GB, macOS 26.5.2). Other M-series Macs have not been tested. Run all commands from the repository root. @@ -17,11 +17,12 @@ or `uv python install 3.12`. ```sh python3.12 -m venv .venv -.venv/bin/python -m pip install 'laya[serve]==0.3.20' pytest +.venv/bin/python -m pip install 'laya[serve]==0.3.20' pytest httpx2 .venv/bin/python -c "import torch; print(torch.backends.mps.is_available())" ``` -The last command must print `True`. The standard macOS arm64 wheel of torch includes MPS. +The last command must print `True`. The standard macOS arm64 wheel of torch includes MPS. `pytest` and +`httpx2` are only for the tests: Starlette's `TestClient` needs `httpx2` (or, deprecated, `httpx`). ## Start the worker @@ -33,8 +34,10 @@ LAYA_REQUIRE_DEVICE=1 \ First startup downloads the checkpoint (about 850 MB). The worker loads the model, runs a warmup over short, long and multi-question requests, and only then listens on port 8000, so the first -request it accepts is already warm. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently -serving on the CPU when the model cannot be placed on MPS. +request it accepts is already warm: on an M1 Pro the first request after ready took 62–81 ms, against +0.7–1.1 s from plain laya-serve. On an M5, 21 of 23 fresh starts gave 21–36 ms and two gave 327 and +409 ms, not yet explained. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently serving on the CPU +when the model cannot be placed on MPS. Check what it is running on: @@ -64,6 +67,10 @@ request (about 57 → 35 ms in those runs), 17–20% at 198–484 tokens, 14% fo six, and cut the worker's memory from 3.5 GB to 2.8 GB. Answers stayed within 0.0031 of the fp32 worker's. The price is startup: the worker becomes ready after 20–30 s instead of about 8 s. +On an M5 the same paired comparison gave median ratios of 0.51–0.53 for one-question requests at +47–68 tokens, 0.30–0.33 at 198–484 tokens, 0.37 for three questions and 0.60 for six, most of it from +the fp16 weights, which on that GPU speed up every input even without compile. + `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. From 3dc662b77b0ce4236f65e354cb2cf462268d7920 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:11:58 +0800 Subject: [PATCH 08/21] Say that fp16 weights are for MPS and warn on CPU On CPU LAYA_WORKER_WEIGHTS=fp16 gives the same answers (max |dp| 0.0013) but a 68-token request took 334 ms instead of 138 ms. The option's description said "on MPS and CPU"; it now says it is meant for MPS, and the worker logs a warning when a model with fp16 weights is on the CPU. --- src/models/laya/README.md | 3 ++- src/models/laya/tests/test_worker.py | 11 +++++++++++ src/models/laya/worker.py | 11 +++++++++-- 3 files changed, 22 insertions(+), 3 deletions(-) diff --git a/src/models/laya/README.md b/src/models/laya/README.md index bb60ff3..0719c19 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -24,7 +24,8 @@ Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MO plus `LAYA_WORKER_COMPILE=off|on` and `LAYA_WORKER_WEIGHTS=fp32|fp16`. `on` compiles one-question requests end to end and, for several questions, only the encoder (Laya's decision head is slower compiled on MPS); `/health` reports compiled graphs at readiness and now. `fp16` keeps the checkpoint's -fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast. See the +fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast; it is meant +for MPS, on CPU fp16 is about 2.4x slower. See the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). ```sh diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index da3b276..3052e25 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -298,3 +298,14 @@ def test_fp16_weights_are_applied_to_every_loaded_model_before_warmup(monkeypatc def test_unknown_weights_mode_is_refused(): with pytest.raises(ValueError, match="fp32 or fp16"): worker.create_worker_app(FakeRouter(), "english", "mps", weights="int8") + + +def test_fp16_weights_on_cpu_are_flagged(monkeypatch, caplog): + monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: None) + with caplog.at_level("WARNING", logger="laya-worker"): + worker.create_worker_app(FakeRouter(FakeAgent(device="cpu")), "english", "cpu", weights="fp16") + assert "fp16 weights on CPU are slower" in caplog.text + caplog.clear() + with caplog.at_level("WARNING", logger="laya-worker"): + worker.create_worker_app(FakeRouter(), "english", "mps", weights="fp16") + assert "slower" not in caplog.text diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index fae08ba..576a7ae 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -21,7 +21,7 @@ before warmup; see compile_agent LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 fp16 weights instead of laya's fp32 - upcast on MPS and CPU + upcast. For MPS; on CPU fp16 is slower The warmup also compiles the graphs, so the worker takes longer to become ready (20–30 s instead of about 8 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means @@ -87,7 +87,8 @@ def compiled_graphs() -> int: def use_fp16_weights(agent: Any) -> None: """Keep the weights in fp16, the checkpoint's own precision, so the conversion is exact. laya 0.3.20 - upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features.""" + upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features. + Meant for MPS: on CPU the answers are the same but a 68-token request took 334 ms instead of 138 ms.""" agent.model.half() act_head = getattr(agent.model, "act_head", None) if act_head is not None: @@ -212,6 +213,12 @@ def create_worker_app( "models": models, } graphs_at_ready = graph_counter() if compile != "off" else None + on_cpu = [n for n, m in models.items() if m["device"].startswith("cpu")] + if weights == "fp16" and on_cpu: + log.warning( + "fp16 weights on CPU are slower than fp32 (%s); LAYA_WORKER_WEIGHTS=fp16 is meant for MPS", + ", ".join(on_cpu), + ) if info["device_mismatch"]: wrong = ", ".join(f"{n} is on {m['device']}" for n, m in models.items() if m["device_mismatch"]) message = f"asked for {info['requested_device']}, {wrong}" From 6311ae83f927477c01aec924d5e14adcd0f405eb Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 21:31:08 +0800 Subject: [PATCH 09/21] Read /health from the live model; correct the Apple Silicon recipe /health described the models as they were after warmup. laya moves a model to the CPU when a request runs out of GPU memory and keeps serving, so the device, dtypes and device_mismatch are now read on every call. Recipe, checked line by line against the measurements and a real worker: - ready time with both options is 35-39 s on the M1 Pro, not 20-30 s (that was the earlier one-question-only compile); - memory is 4.2 GB -> about 3 GB for a worker on its own (3.5 -> 2.8 GB was measured with two workers sharing the machine); - first-request numbers are attributed to the configuration they were measured on; - the uv route to Python 3.12 is `uv venv --python 3.12 --seed .venv`; - LAYA_REQUIRE_DEVICE was exercised on a real worker with MPS reported unavailable: exit code 1 with the documented message; - the first-pass report command globs *_feasibility.jsonl, and report.py skips result files of another kind instead of raising KeyError. --- recipe/laya/apple-silicon.md | 39 +++++++++++++---------- recipe/laya/bench/report.py | 7 ++++- src/models/laya/README.md | 3 +- src/models/laya/tests/test_worker.py | 13 ++++++++ src/models/laya/worker.py | 46 +++++++++++++++++----------- 5 files changed, 72 insertions(+), 36 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 230909f..acbe83d 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -12,8 +12,8 @@ Run all commands from the repository root. ## Install -Use Python 3.12. If `python3.12` is not on your `PATH`, install it with `brew install python@3.12` -or `uv python install 3.12`. +Use Python 3.12. If `python3.12` is not on your `PATH` and you have uv, replace the first command below +with `uv venv --python 3.12 --seed .venv` (`--seed` puts pip in the environment). ```sh python3.12 -m venv .venv @@ -34,10 +34,9 @@ LAYA_REQUIRE_DEVICE=1 \ First startup downloads the checkpoint (about 850 MB). The worker loads the model, runs a warmup over short, long and multi-question requests, and only then listens on port 8000, so the first -request it accepts is already warm: on an M1 Pro the first request after ready took 62–81 ms, against -0.7–1.1 s from plain laya-serve. On an M5, 21 of 23 fresh starts gave 21–36 ms and two gave 327 and -409 ms, not yet explained. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently serving on the CPU -when the model cannot be placed on MPS. +request it accepts is already warm: on an M1 Pro the first request after ready took 70–81 ms, against +0.7–1.1 s from plain laya-serve. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently serving on the +CPU when the model cannot be placed on MPS; without it the worker logs a warning and serves from the CPU. Check what it is running on: @@ -47,7 +46,9 @@ curl -s http://127.0.0.1:8000/health `device` must be `mps` and `device_mismatch` `false`. The response also names the checkpoint and revision, the weight dtype (`torch.float32`; Laya upcasts the fp16 checkpoint on MPS), the autocast -dtype Laya uses for requests with at least `mps_amp_min_rows` questions, and the warmup time. +dtype Laya uses for requests with at least `mps_amp_min_rows` questions, and the warmup time. The device +and dtypes are read on every call: if a request runs out of GPU memory, Laya moves the model to the CPU +and keeps serving, and `/health` then shows `device: cpu` and `device_mismatch: true`. ### Faster: compile and fp16 weights @@ -64,12 +65,16 @@ compiled, requests with several questions run the encoder compiled and Laya's de On the M1 Pro, with both workers running and every request sent to each back to back, the two options together lowered warm p50 against the worker without them by 37–38% for a 68-token one-question request (about 57 → 35 ms in those runs), 17–20% at 198–484 tokens, 14% for three questions and 18% for -six, and cut the worker's memory from 3.5 GB to 2.8 GB. Answers stayed within 0.0031 of the fp32 -worker's. The price is startup: the worker becomes ready after 20–30 s instead of about 8 s. +six. Answers stayed within 0.0031 of the fp32 worker's. A worker running on its own uses about 3 GB +with the options instead of 4.2 GB (2.8 GB against 3.5 GB in those paired runs, where the two workers +shared the machine). The price is startup: the worker became ready after 35–39 s instead of 8–10 s, and +its first request after that took 62–78 ms. On an M5 the same paired comparison gave median ratios of 0.51–0.53 for one-question requests at 47–68 tokens, 0.30–0.33 at 198–484 tokens, 0.37 for three questions and 0.60 for six, most of it from -the fp16 weights, which on that GPU speed up every input even without compile. +the fp16 weights, which on that GPU speed up every input even without compile. There the worker was +ready after 19 s instead of 3 s; its first request took 21–36 ms in 21 of 23 fresh starts and 327 and +409 ms in the other two, not yet explained (132–143 ms from plain laya-serve). `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. @@ -114,7 +119,7 @@ Stop the worker and frontend first; the benchmark starts its own. The scripts ar .venv/bin/python recipe/laya/bench/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve .venv/bin/python recipe/laya/bench/bench_http.py --config C4 --run feasibility \ --url http://127.0.0.1:8080 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve -.venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*.jsonl +.venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*_feasibility.jsonl ``` Runs labelled anything other than `feasibility` refuse to start on battery power or when the @@ -122,9 +127,11 @@ Runs labelled anything other than `feasibility` refuse to start on battery power ## Troubleshooting -- `device_mismatch: true`, or the worker exits with `asked for mps, english is on cpu`: MPS is not - available to this Python. Check the `torch.backends.mps.is_available()` line above; an x86_64 - Python running under Rosetta cannot use MPS. -- The worker process uses about 4 GB (Activity Monitor's Memory column, which counts MPS - allocations). On a 16 GB Mac, close other large applications before benchmarking. +- `device_mismatch: true` at startup, or the worker exits with `asked for mps, english is on cpu`: MPS + is not available to this Python. Check the `torch.backends.mps.is_available()` line above (an x86_64 + Python under Rosetta, for example, has no MPS). +- `device_mismatch: true` on a worker that started on MPS: Laya fell back to the CPU after a GPU + out-of-memory error. Free memory and restart the worker. +- The worker process uses about 4 GB, or 3 GB with fp16 weights (Activity Monitor's Memory column, + which counts MPS allocations). On a 16 GB Mac, close other large applications before benchmarking. - `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/recipe/laya/bench/report.py b/recipe/laya/bench/report.py index 23bb3d9..0d8d6f5 100644 --- a/recipe/laya/bench/report.py +++ b/recipe/laya/bench/report.py @@ -11,6 +11,7 @@ import json import math import statistics +import sys from collections import defaultdict GATE = 0.10 @@ -25,7 +26,11 @@ def read(paths): records = [] for path in paths: with open(path) as f: - records.extend(json.loads(line) for line in f if line.strip()) + rows = [json.loads(line) for line in f if line.strip()] + if any("config" not in r for r in rows): # e.g. paired.py results: summarize those with paired.py + print(f"skipping {path}: not a bench_inproc/bench_http result", file=sys.stderr) + continue + records.extend(rows) return records diff --git a/src/models/laya/README.md b/src/models/laya/README.md index 0719c19..20f54ab 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -16,7 +16,8 @@ validated on an M1 Pro). No native CUDA or Metal backend yet. returned 200 took ~230 ms against ~45 ms warm on an M1 Pro (MPS); with it, ~73 ms. - `/health` reports, for each loaded model under `models` and for `LAYA_WORKER_MODEL` at the top level, the device, weight and autocast dtypes, the checkpoint and the revision its weights were downloaded - from, and `device_mismatch` when a model is not on the device `LAYA_DEVICE` asked for. + from, and `device_mismatch` when a model is not on the device `LAYA_DEVICE` asked for. These are + read on every call, so a fallback to the CPU after startup shows up too. laya-serve reports `LAYA_DEVICE` as configured, and laya falls back to CPU with only a printed warning. `LAYA_REQUIRE_DEVICE=1` makes the worker exit instead. diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index 3052e25..52088bf 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -309,3 +309,16 @@ def test_fp16_weights_on_cpu_are_flagged(monkeypatch, caplog): with caplog.at_level("WARNING", logger="laya-worker"): worker.create_worker_app(FakeRouter(), "english", "mps", weights="fp16") assert "slower" not in caplog.text + + +def test_health_follows_a_fallback_to_cpu_after_startup(): + agent = FakeAgent(device="mps") + client = TestClient(worker.create_worker_app(FakeRouter(agent), "english", "mps", require_device=True)) + assert client.get("/health").json()["device_mismatch"] is False + agent.device = "cpu" # what laya does when a request runs out of GPU memory + agent.dtype = "torch.float32" + health = client.get("/health").json() + assert health["device"] == "cpu" + assert health["device_mismatch"] is True + assert health["models"]["english"]["device"] == "cpu" + assert health["autocast_dtype"] == "torch.float32" diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 576a7ae..40ce989 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -7,8 +7,9 @@ - It binds only after a warmup of every loaded model that covers short, long and multi-question requests (the last one crosses laya's fp16 autocast threshold on MPS), so a reachable worker is a warm one whichever model a request is routed to. -- /health reports, per loaded model, the device, weight and autocast dtypes, the checkpoint and the - revision its weights were downloaded from, and whether the device differs from the one requested. +- /health reports, per loaded model and read on every call, the device, weight and autocast dtypes, the + checkpoint and the revision its weights were downloaded from, and whether the device differs from the + one requested. laya moves a model to the CPU on a GPU out-of-memory error and keeps serving. Configuration is laya-serve's (LAYA_HOST, LAYA_PORT, LAYA_DEVICE, LAYA_MODELS, LAYA_API_KEY, ...) plus: @@ -23,8 +24,8 @@ fp16 weights instead of laya's fp32 upcast. For MPS; on CPU fp16 is slower -The warmup also compiles the graphs, so the worker takes longer to become ready (20–30 s instead of -about 8 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means +The warmup also compiles the graphs, so the worker takes longer to become ready (35–39 s instead of +8–10 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means a request hit a shape class the warmup did not cover. """ @@ -198,20 +199,29 @@ def create_worker_app( if compile != "off": for name in names: compile_agent(router.load(name)) - models = {} - for name in names: - warm = warmup(router, name) - models[name] = { - **describe(router.load(name), requested, warm["routing"], revisions), - "warmup_ms": warm["warmup_ms"], + warmed = {name: warmup(router, name) for name in names} + agents = {name: router.load(name) for name in names} + primary = model if model in agents else names[0] + + def current() -> dict[str, Any]: + """Describe the agents as they are now: laya moves a model to the CPU when a request runs out of + GPU memory, so the device at startup is not necessarily the device serving the next request.""" + models = { + name: { + **describe(agent, requested, warmed[name]["routing"], revisions), + "warmup_ms": warmed[name]["warmup_ms"], + } + for name, agent in agents.items() } - primary = model if model in models else names[0] - info = { - **models[primary], - "device_mismatch": any(m["device_mismatch"] for m in models.values()), - "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), - "models": models, - } + return { + **models[primary], + "device_mismatch": any(m["device_mismatch"] for m in models.values()), + "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), + "models": models, + } + + info = current() + models = info["models"] graphs_at_ready = graph_counter() if compile != "off" else None on_cpu = [n for n, m in models.items() if m["device"].startswith("cpu")] if weights == "fp16" and on_cpu: @@ -237,7 +247,7 @@ def health() -> dict[str, Any]: compiled.update( graphs_at_ready=graphs_at_ready, graphs_now=now, recompiled_after_ready=now > graphs_at_ready ) - return {"status": "ok", "ready": True, "loaded": router.loaded, **info, "compile": compiled} + return {"status": "ok", "ready": True, "loaded": router.loaded, **current(), "compile": compiled} return app From 9ad04e32512d6bb9e4ecde043d7ffb378ef9b20c Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:17:38 +0800 Subject: [PATCH 10/21] Run plain fp32 laya after a fallback to the CPU laya moves the model to the CPU when a request runs out of GPU memory and keeps serving. Triggered on the real GPU by lowering PyTorch's MPS memory limit, a worker with compile and fp16 weights then kept its fp16 weights and recompiled its graphs for the CPU: the request that fell back took 336 s instead of 28 s, the next one 28 s, later ones 370-550 ms. The worker's model wrapper now runs laya's own model in fp32 whenever the inputs are on the CPU, restoring the weights once. After the same fallback: 32-40 s for the request that fell back, then 140-270 ms, no graph recompiled, /health showing cpu and float32. The two options are no longer applied to a model that is on the CPU at startup. The GPU path is unchanged: a paired check gave the same ratios as before (0.62 for the 68-token request) and answers within 0.0031. --- recipe/laya/apple-silicon.md | 10 +++- src/models/laya/README.md | 5 +- src/models/laya/tests/test_worker.py | 57 ++++++++++++++---- src/models/laya/worker.py | 90 +++++++++++++++++++--------- 4 files changed, 120 insertions(+), 42 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index acbe83d..874ba9d 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -48,7 +48,9 @@ curl -s http://127.0.0.1:8000/health revision, the weight dtype (`torch.float32`; Laya upcasts the fp16 checkpoint on MPS), the autocast dtype Laya uses for requests with at least `mps_amp_min_rows` questions, and the warmup time. The device and dtypes are read on every call: if a request runs out of GPU memory, Laya moves the model to the CPU -and keeps serving, and `/health` then shows `device: cpu` and `device_mismatch: true`. +and keeps serving, and `/health` then shows `device: cpu` and `device_mismatch: true`. Triggered on the +M1 Pro by lowering PyTorch's MPS memory limit: the request that ran out of memory still returned 200 +after about 30 s, and later 68-token requests took 140–270 ms from the CPU. ### Faster: compile and fp16 weights @@ -79,6 +81,9 @@ ready after 19 s instead of 3 s; its first request took 21–36 ms in 21 of 23 f `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. +Both options apply on the GPU only. After a fallback to the CPU the worker runs Laya's fp32 model +uncompiled, like a worker started without them. + ## Start the frontend In another terminal: @@ -131,7 +136,8 @@ Runs labelled anything other than `feasibility` refuse to start on battery power is not available to this Python. Check the `torch.backends.mps.is_available()` line above (an x86_64 Python under Rosetta, for example, has no MPS). - `device_mismatch: true` on a worker that started on MPS: Laya fell back to the CPU after a GPU - out-of-memory error. Free memory and restart the worker. + out-of-memory error. It keeps answering, several times slower; free memory and restart the worker + to get back on the GPU. - The worker process uses about 4 GB, or 3 GB with fp16 weights (Activity Monitor's Memory column, which counts MPS allocations). On a 16 GB Mac, close other large applications before benchmarking. - `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/src/models/laya/README.md b/src/models/laya/README.md index 20f54ab..9ccc244 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -25,8 +25,9 @@ Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MO plus `LAYA_WORKER_COMPILE=off|on` and `LAYA_WORKER_WEIGHTS=fp32|fp16`. `on` compiles one-question requests end to end and, for several questions, only the encoder (Laya's decision head is slower compiled on MPS); `/health` reports compiled graphs at readiness and now. `fp16` keeps the checkpoint's -fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast; it is meant -for MPS, on CPU fp16 is about 2.4x slower. See the +fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast. Both apply +on the GPU only: on the CPU, including after Laya falls back to it on a GPU out-of-memory error, the +worker runs Laya's fp32 model uncompiled (fp16 is about 2.4x slower there). See the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). ```sh diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index 52088bf..92543db 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -5,6 +5,7 @@ import sys from pathlib import Path +from types import SimpleNamespace import pytest from fastapi.testclient import TestClient @@ -15,6 +16,11 @@ ANSWER = {"type": "noul", "noul": 0.9, "confidence": 0.9} +def on_gpu(rows): + """Stands in for input_ids on the GPU: the wrapper only reads its device type and number of rows.""" + return SimpleNamespace(device=SimpleNamespace(type="mps"), shape=(rows, 7)) + + class FakeAgent: def __init__(self, device="mps", dtype="torch.float16"): self.device = device @@ -212,8 +218,9 @@ def fake_compile(module, dynamic): model = Model() agent.model = model worker.compile_agent(agent) - assert agent.model(torch.zeros(1, 7)) == "whole model compiled" # one question - assert agent.model(torch.zeros(3, 7)) == ("head", "compiled encoder") # several: eager head, compiled encoder + assert agent.model(on_gpu(1)) == "whole model compiled" # one question + assert agent.model(on_gpu(3)) == ("head", "compiled encoder") # several: eager head, compiled encoder + assert agent.model(torch.zeros(1, 7)) == ("head", "eager encoder") # on the CPU: laya's model as it is assert model.encoder(torch.zeros(1, 7)) == "eager encoder" # the original model is left as it was assert len(list(agent.model.parameters())) == len(list(model.parameters())) # one set of weights @@ -282,8 +289,8 @@ def __init__(self): agent = FakeAgent() agent.model = Model() worker.use_fp16_weights(agent) - assert agent.model.encoder.weight.dtype == torch.float16 - assert agent.model.act_head.weight.dtype == torch.float32 + assert agent.model.eager.encoder.weight.dtype == torch.float16 + assert agent.model.eager.act_head.weight.dtype == torch.float32 def test_fp16_weights_are_applied_to_every_loaded_model_before_warmup(monkeypatch): @@ -300,15 +307,45 @@ def test_unknown_weights_mode_is_refused(): worker.create_worker_app(FakeRouter(), "english", "mps", weights="int8") -def test_fp16_weights_on_cpu_are_flagged(monkeypatch, caplog): - monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: None) +def test_options_are_not_applied_to_a_model_on_the_cpu(monkeypatch, caplog): + applied = [] + monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: applied.append("fp16")) + monkeypatch.setattr(worker, "compile_agent", lambda agent: applied.append("compile")) with caplog.at_level("WARNING", logger="laya-worker"): - worker.create_worker_app(FakeRouter(FakeAgent(device="cpu")), "english", "cpu", weights="fp16") - assert "fp16 weights on CPU are slower" in caplog.text + worker.create_worker_app( + FakeRouter(FakeAgent(device="cpu")), "english", "cpu", weights="fp16", compile="on", graph_counter=lambda: 0 + ) + assert applied == [] + assert "apply on the GPU only" in caplog.text caplog.clear() with caplog.at_level("WARNING", logger="laya-worker"): - worker.create_worker_app(FakeRouter(), "english", "mps", weights="fp16") - assert "slower" not in caplog.text + worker.create_worker_app(FakeRouter(), "english", "mps", weights="fp16", compile="on", graph_counter=lambda: 0) + assert applied == ["fp16", "compile"] + assert "GPU only" not in caplog.text + + +def test_after_a_fallback_to_cpu_the_model_runs_fp32_and_uncompiled(monkeypatch): + import torch + + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + self.encoder = torch.nn.Linear(4, 4) + self.act_head = torch.nn.Linear(4, 2) + + def forward(self, input_ids): + return ("eager", self.encoder.weight.dtype) + + monkeypatch.setattr(torch, "compile", lambda module, dynamic: lambda *a, **k: "compiled") + agent = FakeAgent() + agent.model = Model() + worker.use_fp16_weights(agent) + worker.compile_agent(agent) + assert agent.model(on_gpu(1)) == "compiled" + assert next(agent.model.parameters()).dtype == torch.float16 + assert agent.model(torch.zeros(1, 7)) == ("eager", torch.float32) # inputs on the CPU: laya fell back + assert {p.dtype for p in agent.model.parameters()} == {torch.float32} + assert agent.model(torch.zeros(3, 7)) == ("eager", torch.float32) def test_health_follows_a_fallback_to_cpu_after_startup(): diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 40ce989..3f575db 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -22,7 +22,10 @@ before warmup; see compile_agent LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 fp16 weights instead of laya's fp32 - upcast. For MPS; on CPU fp16 is slower + upcast + +Both options apply on the GPU only. If laya falls back to the CPU after a GPU out-of-memory error, the +worker runs laya's fp32 model uncompiled from then on, and /health shows the CPU. The warmup also compiles the graphs, so the worker takes longer to become ready (35–39 s instead of 8–10 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means @@ -86,14 +89,54 @@ def compiled_graphs() -> int: WEIGHT_MODES = {"": "fp32", "fp32": "fp32", "fp16": "fp16"} +def _served(agent: Any) -> Any: + """The module the worker puts in place of laya's model, created on first use. + + On the GPU it runs the compiled paths when there are any, otherwise laya's model. On the CPU it always + runs laya's model in fp32: laya moves the model to the CPU when a request runs out of GPU memory, and + there fp16 weights are slower (334 ms against 138 ms for a 68-token request) and the compiled graphs + would first recompile (28 s measured). So after a fallback the worker behaves like plain laya. + """ + import torch + + if getattr(agent.model, "laya_worker_wrapper", False): + return agent.model + eager = agent.model + + class Served(torch.nn.Module): + laya_worker_wrapper = True + + def __init__(self): + super().__init__() + self.eager = eager + self.fp16 = False + self.paths = None # (whole model compiled, encoder-only compiled); a tuple is not a submodule + + def forward(self, input_ids, *args, **kwargs): + if input_ids.device.type == "cpu": + if self.fp16: + self.eager.float() + self.fp16 = False + return self.eager(input_ids, *args, **kwargs) + if self.paths is None: + return self.eager(input_ids, *args, **kwargs) + whole, encoder_only = self.paths + return (whole if input_ids.shape[0] == 1 else encoder_only)(input_ids, *args, **kwargs) + + agent.model = Served() + return agent.model + + def use_fp16_weights(agent: Any) -> None: """Keep the weights in fp16, the checkpoint's own precision, so the conversion is exact. laya 0.3.20 upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features. - Meant for MPS: on CPU the answers are the same but a 68-token request took 334 ms instead of 138 ms.""" - agent.model.half() - act_head = getattr(agent.model, "act_head", None) + For the GPU only: see _served for what happens on the CPU.""" + served = _served(agent) + served.eager.half() + act_head = getattr(served.eager, "act_head", None) if act_head is not None: act_head.float() + served.fp16 = True def compile_agent(agent: Any) -> None: @@ -108,22 +151,13 @@ def compile_agent(agent: Any) -> None: import torch - eager = agent.model + served = _served(agent) + eager = served.eager whole = torch.compile(eager, dynamic=True) encoder_only = copy.copy(eager) # same parameters and submodules ... encoder_only._modules = dict(eager._modules) # ... except the encoder slot encoder_only._modules["encoder"] = torch.compile(eager.encoder, dynamic=True) - - class Compiled(torch.nn.Module): - def __init__(self): - super().__init__() - self.eager = eager - - def forward(self, input_ids, *args, **kwargs): - model = whole if input_ids.shape[0] == 1 else encoder_only - return model(input_ids, *args, **kwargs) - - agent.model = Compiled() + served.paths = (whole, encoder_only) def record_snapshot_revisions() -> dict[str, str]: @@ -193,12 +227,18 @@ def create_worker_app( if weights not in ("fp32", "fp16"): raise ValueError(f"weights must be fp32 or fp16, not {weights!r}") names = list(router.loaded) or [model] # never load a model the worker was not asked to serve - if weights == "fp16": - for name in names: - use_fp16_weights(router.load(name)) - if compile != "off": - for name in names: - compile_agent(router.load(name)) + for name in names: + agent = router.load(name) + if str(getattr(agent, "device", "")).startswith("cpu"): + if weights == "fp16" or compile != "off": + log.warning( + "%s is on the CPU: LAYA_WORKER_WEIGHTS=fp16 and LAYA_WORKER_COMPILE=on apply on the GPU only", name + ) + continue + if weights == "fp16": + use_fp16_weights(agent) + if compile != "off": + compile_agent(agent) warmed = {name: warmup(router, name) for name in names} agents = {name: router.load(name) for name in names} primary = model if model in agents else names[0] @@ -223,12 +263,6 @@ def current() -> dict[str, Any]: info = current() models = info["models"] graphs_at_ready = graph_counter() if compile != "off" else None - on_cpu = [n for n, m in models.items() if m["device"].startswith("cpu")] - if weights == "fp16" and on_cpu: - log.warning( - "fp16 weights on CPU are slower than fp32 (%s); LAYA_WORKER_WEIGHTS=fp16 is meant for MPS", - ", ".join(on_cpu), - ) if info["device_mismatch"]: wrong = ", ".join(f"{n} is on {m['device']}" for n, m in models.items() if m["device_mismatch"]) message = f"asked for {info['requested_device']}, {wrong}" From 5e260a126c0dd61aa60154a67be52ff9d08d0c0e Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Wed, 30 Sep 2026 22:34:27 +0800 Subject: [PATCH 11/21] Split the GPU optimizations out of worker.py; correct the Laya README worker.py keeps startup, warmup and /health. optimize.py holds what changes the model itself: fp16 weights, compile, the wrapper that runs laya's own fp32 model on the CPU, and the rule that the options apply on the GPU only. No behaviour change: same tests, and a paired check gave the same ratios and answers as before. README: the first-request numbers were from an early feasibility run (230 ms / 73 ms); the measured ones are 0.7-1.1 s for laya-serve and 70-81 ms for the worker. It now lists what the worker adds the same way as the PR does, says LAYA_REQUIRE_DEVICE is checked at startup, and mentions the M5 validation. --- src/models/laya/README.md | 44 ++++++------ src/models/laya/optimize.py | 99 ++++++++++++++++++++++++++ src/models/laya/tests/test_worker.py | 21 +++--- src/models/laya/worker.py | 102 +++------------------------ 4 files changed, 142 insertions(+), 124 deletions(-) create mode 100644 src/models/laya/optimize.py diff --git a/src/models/laya/README.md b/src/models/laya/README.md index 9ccc244..8710117 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -5,29 +5,33 @@ LAYA is the first planned System1-Omni model. This directory owns its complete r GPU operations and kernel implementations belong in [`backends/cuda/`](../../backends/cuda/) and [`backends/metal/`](../../backends/metal/). Setup and usage examples belong in the top-level [`recipe/`](../../../recipe/) directory. Status: the Python worker below serves LAYA through laya-serve on CPU and Apple Silicon (PyTorch MPS, -validated on an M1 Pro). No native CUDA or Metal backend yet. +validated on an M1 Pro and, by another contributor, an M5). No native CUDA or Metal backend yet. ## Worker -`worker.py` runs laya-serve (`laya[serve]==0.3.20`) with two changes: - -- It binds only after a warmup of every loaded model over short, long and multi-question requests, - so `/health` never answers for a worker that has not run a forward pass. Without it the first request after `/health` - returned 200 took ~230 ms against ~45 ms warm on an M1 Pro (MPS); with it, ~73 ms. -- `/health` reports, for each loaded model under `models` and for `LAYA_WORKER_MODEL` at the top level, - the device, weight and autocast dtypes, the checkpoint and the revision its weights were downloaded - from, and `device_mismatch` when a model is not on the device `LAYA_DEVICE` asked for. These are - read on every call, so a fallback to the CPU after startup shows up too. - laya-serve reports `LAYA_DEVICE` as configured, and laya falls back to CPU with only a printed - warning. `LAYA_REQUIRE_DEVICE=1` makes the worker exit instead. - -Configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, `LAYA_API_KEY`, ...), -plus `LAYA_WORKER_COMPILE=off|on` and `LAYA_WORKER_WEIGHTS=fp32|fp16`. `on` compiles one-question -requests end to end and, for several questions, only the encoder (Laya's decision head is slower -compiled on MPS); `/health` reports compiled graphs at readiness and now. `fp16` keeps the checkpoint's -fp16 weights (except `act_head`, which Laya feeds fp32 features) instead of the fp32 upcast. Both apply -on the GPU only: on the CPU, including after Laya falls back to it on a GPU out-of-memory error, the -worker runs Laya's fp32 model uncompiled (fp16 is about 2.4x slower there). See the +`worker.py` runs laya-serve (`laya[serve]==0.3.20`) with its request handling unchanged and adds: + +- **Warmup before readiness.** It binds only after every loaded model has run short, long and + multi-question requests, so `/health` never answers for a worker that has not run a forward pass. + On an M1 Pro (MPS) the first request after ready took 70–81 ms, against 0.7–1.1 s from laya-serve. +- **A `/health` that describes the loaded models.** For each loaded model under `models`, and for + `LAYA_WORKER_MODEL` at the top level: the device, weight and autocast dtypes, the checkpoint and the + revision its weights were downloaded from, and `device_mismatch` when a model is not on the device + `LAYA_DEVICE` asked for. laya-serve reports `LAYA_DEVICE` as configured. These are read on every call: + Laya moves a model to the CPU on a GPU out-of-memory error and keeps serving, and `/health` shows it. + `LAYA_REQUIRE_DEVICE=1` makes the worker exit at startup if a model is not on the requested device. +- **Two options that make the GPU path faster** (`optimize.py`): + - `LAYA_WORKER_COMPILE=on` compiles one-question requests end to end and, for several questions, only + the encoder (Laya's decision head is slower compiled on MPS). `/health` reports compiled graphs at + readiness and now. + - `LAYA_WORKER_WEIGHTS=fp16` keeps the checkpoint's fp16 weights instead of Laya's fp32 upcast + (except `act_head`, which Laya feeds fp32 features). + + Both apply on the GPU only. On the CPU, including after a fallback, the worker runs Laya's fp32 model + uncompiled (fp16 is about 2.4x slower there). + +Other configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, +`LAYA_API_KEY`, ...). Measurements and setup are in the [Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). ```sh diff --git a/src/models/laya/optimize.py b/src/models/laya/optimize.py new file mode 100644 index 0000000..77261d9 --- /dev/null +++ b/src/models/laya/optimize.py @@ -0,0 +1,99 @@ +"""What the Laya worker changes about the model itself to make it faster on the GPU. + +- fp16 weights: keep the checkpoint's own precision instead of laya's fp32 upcast. +- Compile: torch.compile for the batches where it pays off on MPS. + +Both go through one wrapper module that replaces `agent.model`, so that on the CPU the worker always runs +laya's own fp32 model. worker.py decides when to apply them and reports the result in /health. +""" + +from typing import Any + + +def _served(agent: Any) -> Any: + """The module the worker puts in place of laya's model, created on first use. + + On the GPU it runs the compiled paths when there are any, otherwise laya's model. On the CPU it always + runs laya's model in fp32: laya moves the model to the CPU when a request runs out of GPU memory, and + there fp16 weights are slower (334 ms against 138 ms for a 68-token request) and the compiled graphs + would first recompile (28 s measured). So after a fallback the worker behaves like plain laya. + """ + import torch + + if getattr(agent.model, "laya_worker_wrapper", False): + return agent.model + eager = agent.model + + class Served(torch.nn.Module): + laya_worker_wrapper = True + + def __init__(self): + super().__init__() + self.eager = eager + self.fp16 = False + self.paths = None # (whole model compiled, encoder-only compiled); a tuple is not a submodule + + def forward(self, input_ids, *args, **kwargs): + if input_ids.device.type == "cpu": + if self.fp16: + self.eager.float() + self.fp16 = False + return self.eager(input_ids, *args, **kwargs) + if self.paths is None: + return self.eager(input_ids, *args, **kwargs) + whole, encoder_only = self.paths + return (whole if input_ids.shape[0] == 1 else encoder_only)(input_ids, *args, **kwargs) + + agent.model = Served() + return agent.model + + +def use_fp16_weights(agent: Any) -> None: + """Keep the weights in fp16, the checkpoint's own precision, so the conversion is exact. laya 0.3.20 + upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features. + For the GPU only: see _served for what happens on the CPU.""" + served = _served(agent) + served.eager.half() + act_head = getattr(served.eager, "act_head", None) + if act_head is not None: + act_head.float() + served.fp16 = True + + +def compile_agent(agent: Any) -> None: + """Compile the model for the batches where it pays off on MPS, sharing the same parameters. + + A batch of one row (one question) runs the whole model compiled. A batch of several rows runs only + the encoder compiled and laya's decision head eagerly: the head is two nn.TransformerEncoderLayer + with a key padding mask, which lose PyTorch's fused fast path when compiled and were about 50 ms + slower on padded multi-row batches. + """ + import copy + + import torch + + served = _served(agent) + eager = served.eager + whole = torch.compile(eager, dynamic=True) + encoder_only = copy.copy(eager) # same parameters and submodules ... + encoder_only._modules = dict(eager._modules) # ... except the encoder slot + encoder_only._modules["encoder"] = torch.compile(eager.encoder, dynamic=True) + served.paths = (whole, encoder_only) + + +def apply(agent: Any, *, fp16: bool, compile: bool) -> bool: + """Apply the requested options to one agent. Does nothing and returns False for a model on the CPU.""" + if str(getattr(agent, "device", "")).startswith("cpu"): + return False + if fp16: + use_fp16_weights(agent) + if compile: + compile_agent(agent) + return True + + +def compiled_graphs() -> int: + """Graphs torch.compile has produced in this process.""" + from torch._dynamo.utils import counters + + return int(counters["stats"]["unique_graphs"]) diff --git a/src/models/laya/tests/test_worker.py b/src/models/laya/tests/test_worker.py index 92543db..40bb0fa 100644 --- a/src/models/laya/tests/test_worker.py +++ b/src/models/laya/tests/test_worker.py @@ -11,6 +11,7 @@ from fastapi.testclient import TestClient sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import optimize import worker ANSWER = {"type": "noul", "noul": 0.9, "confidence": 0.9} @@ -150,7 +151,7 @@ def test_main_exits_non_zero_when_warmup_fails(monkeypatch): def test_compile_wraps_the_model_before_warmup(monkeypatch): router = FakeRouter() order = [] - monkeypatch.setattr(worker, "compile_agent", lambda agent: order.append((agent, len(router.calls)))) + monkeypatch.setattr(optimize, "compile_agent", lambda agent: order.append((agent, len(router.calls)))) worker.create_worker_app(router, "english", "mps", compile="on", graph_counter=lambda: 3) assert order == [(router.agent, 0)] # before the first warmup request @@ -161,7 +162,7 @@ def test_health_reports_compile_off_by_default(): def test_health_flags_graphs_compiled_after_ready(monkeypatch): - monkeypatch.setattr(worker, "compile_agent", lambda agent: None) + monkeypatch.setattr(optimize, "compile_agent", lambda agent: None) graphs = iter([4, 4, 5]) # at readiness, first /health, second /health after a new shape compiled client = TestClient( worker.create_worker_app(FakeRouter(), "english", "mps", compile="on", graph_counter=lambda: next(graphs)) @@ -175,7 +176,7 @@ def test_compile_failure_means_no_app(monkeypatch): def broken(agent): raise RuntimeError("inductor: unsupported op on mps") - monkeypatch.setattr(worker, "compile_agent", broken) + monkeypatch.setattr(optimize, "compile_agent", broken) with pytest.raises(RuntimeError, match="unsupported op"): worker.create_worker_app(FakeRouter(), "english", "mps", compile="on") @@ -217,7 +218,7 @@ def fake_compile(module, dynamic): agent = FakeAgent() model = Model() agent.model = model - worker.compile_agent(agent) + optimize.compile_agent(agent) assert agent.model(on_gpu(1)) == "whole model compiled" # one question assert agent.model(on_gpu(3)) == ("head", "compiled encoder") # several: eager head, compiled encoder assert agent.model(torch.zeros(1, 7)) == ("head", "eager encoder") # on the CPU: laya's model as it is @@ -288,7 +289,7 @@ def __init__(self): agent = FakeAgent() agent.model = Model() - worker.use_fp16_weights(agent) + optimize.use_fp16_weights(agent) assert agent.model.eager.encoder.weight.dtype == torch.float16 assert agent.model.eager.act_head.weight.dtype == torch.float32 @@ -297,7 +298,7 @@ def test_fp16_weights_are_applied_to_every_loaded_model_before_warmup(monkeypatc order = [] agents = {"english": FakeAgent(), "multilingual": FakeAgent()} router = FakeRouter(agents=agents) - monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: order.append((agent, len(router.calls)))) + monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: order.append((agent, len(router.calls)))) worker.create_worker_app(router, "english", "mps", weights="fp16") assert order == [(agents["english"], 0), (agents["multilingual"], 0)] @@ -309,8 +310,8 @@ def test_unknown_weights_mode_is_refused(): def test_options_are_not_applied_to_a_model_on_the_cpu(monkeypatch, caplog): applied = [] - monkeypatch.setattr(worker, "use_fp16_weights", lambda agent: applied.append("fp16")) - monkeypatch.setattr(worker, "compile_agent", lambda agent: applied.append("compile")) + monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: applied.append("fp16")) + monkeypatch.setattr(optimize, "compile_agent", lambda agent: applied.append("compile")) with caplog.at_level("WARNING", logger="laya-worker"): worker.create_worker_app( FakeRouter(FakeAgent(device="cpu")), "english", "cpu", weights="fp16", compile="on", graph_counter=lambda: 0 @@ -339,8 +340,8 @@ def forward(self, input_ids): monkeypatch.setattr(torch, "compile", lambda module, dynamic: lambda *a, **k: "compiled") agent = FakeAgent() agent.model = Model() - worker.use_fp16_weights(agent) - worker.compile_agent(agent) + optimize.use_fp16_weights(agent) + optimize.compile_agent(agent) assert agent.model(on_gpu(1)) == "compiled" assert next(agent.model.parameters()).dtype == torch.float16 assert agent.model(torch.zeros(1, 7)) == ("eager", torch.float32) # inputs on the CPU: laya fell back diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py index 3f575db..37b5e6c 100644 --- a/src/models/laya/worker.py +++ b/src/models/laya/worker.py @@ -19,7 +19,7 @@ LAYA_REQUIRE_DEVICE exit instead of serving on another 0 device than LAYA_DEVICE asked for LAYA_WORKER_COMPILE off or on: torch.compile (dynamic=True) off - before warmup; see compile_agent + before warmup; see optimize.py LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 fp16 weights instead of laya's fp32 upcast @@ -39,6 +39,8 @@ from pathlib import Path from typing import Any +import optimize + log = logging.getLogger("laya-worker") # (words of state, questions): each shape runs twice. Short, mid-length and near-window states, then @@ -76,90 +78,10 @@ def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_ return {"warmup_ms": round((time.perf_counter() - started) * 1000, 1), "routing": routing} -def compiled_graphs() -> int: - """Graphs torch.compile has produced in this process.""" - from torch._dynamo.utils import counters - - return int(counters["stats"]["unique_graphs"]) - - COMPILE_MODES = {"": "off", "0": "off", "off": "off", "1": "on", "on": "on"} - - WEIGHT_MODES = {"": "fp32", "fp32": "fp32", "fp16": "fp16"} -def _served(agent: Any) -> Any: - """The module the worker puts in place of laya's model, created on first use. - - On the GPU it runs the compiled paths when there are any, otherwise laya's model. On the CPU it always - runs laya's model in fp32: laya moves the model to the CPU when a request runs out of GPU memory, and - there fp16 weights are slower (334 ms against 138 ms for a 68-token request) and the compiled graphs - would first recompile (28 s measured). So after a fallback the worker behaves like plain laya. - """ - import torch - - if getattr(agent.model, "laya_worker_wrapper", False): - return agent.model - eager = agent.model - - class Served(torch.nn.Module): - laya_worker_wrapper = True - - def __init__(self): - super().__init__() - self.eager = eager - self.fp16 = False - self.paths = None # (whole model compiled, encoder-only compiled); a tuple is not a submodule - - def forward(self, input_ids, *args, **kwargs): - if input_ids.device.type == "cpu": - if self.fp16: - self.eager.float() - self.fp16 = False - return self.eager(input_ids, *args, **kwargs) - if self.paths is None: - return self.eager(input_ids, *args, **kwargs) - whole, encoder_only = self.paths - return (whole if input_ids.shape[0] == 1 else encoder_only)(input_ids, *args, **kwargs) - - agent.model = Served() - return agent.model - - -def use_fp16_weights(agent: Any) -> None: - """Keep the weights in fp16, the checkpoint's own precision, so the conversion is exact. laya 0.3.20 - upcasts them to fp32 on MPS and CPU. `act_head` stays fp32 because laya feeds it `.float()` features. - For the GPU only: see _served for what happens on the CPU.""" - served = _served(agent) - served.eager.half() - act_head = getattr(served.eager, "act_head", None) - if act_head is not None: - act_head.float() - served.fp16 = True - - -def compile_agent(agent: Any) -> None: - """Compile the model for the batches where it pays off on MPS, sharing the same parameters. - - A batch of one row (one question) runs the whole model compiled. A batch of several rows runs only - the encoder compiled and laya's decision head eagerly: the head is two nn.TransformerEncoderLayer - with a key padding mask, which lose PyTorch's fused fast path when compiled and were about 50 ms - slower on padded multi-row batches. - """ - import copy - - import torch - - served = _served(agent) - eager = served.eager - whole = torch.compile(eager, dynamic=True) - encoder_only = copy.copy(eager) # same parameters and submodules ... - encoder_only._modules = dict(eager._modules) # ... except the encoder slot - encoder_only._modules["encoder"] = torch.compile(eager.encoder, dynamic=True) - served.paths = (whole, encoder_only) - - def record_snapshot_revisions() -> dict[str, str]: """Record the commit each Hugging Face checkpoint is loaded from, keyed by repo id. @@ -213,7 +135,7 @@ def create_worker_app( requested: str | None, require_device: bool = False, compile: str = "off", - graph_counter=compiled_graphs, + graph_counter=optimize.compiled_graphs, revisions: dict[str, str] | None = None, weights: str = "fp32", ): @@ -227,18 +149,10 @@ def create_worker_app( if weights not in ("fp32", "fp16"): raise ValueError(f"weights must be fp32 or fp16, not {weights!r}") names = list(router.loaded) or [model] # never load a model the worker was not asked to serve - for name in names: - agent = router.load(name) - if str(getattr(agent, "device", "")).startswith("cpu"): - if weights == "fp16" or compile != "off": - log.warning( - "%s is on the CPU: LAYA_WORKER_WEIGHTS=fp16 and LAYA_WORKER_COMPILE=on apply on the GPU only", name - ) - continue - if weights == "fp16": - use_fp16_weights(agent) - if compile != "off": - compile_agent(agent) + if weights == "fp16" or compile == "on": + for name in names: + if not optimize.apply(router.load(name), fp16=weights == "fp16", compile=compile == "on"): + log.warning("%s is on the CPU: LAYA_WORKER_WEIGHTS and LAYA_WORKER_COMPILE apply on the GPU only", name) warmed = {name: warmup(router, name) for name in names} agents = {name: router.load(name) for name in names} primary = model if model in agents else names[0] From 005a8748f24792d7bc730a44b25280706a6f8757 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 09:26:16 +0800 Subject: [PATCH 12/21] Lay the Laya worker out like the Cua-S1 worker (#19) - src/frontend/laya_mps.py: the HTTP worker, started with flags (PYTHONPATH=src python -m frontend.laya_mps --device mps --model english [--compile] [--weights fp16] [--require-device]) instead of LAYA_WORKER_* environment variables. laya-serve's own variables (LAYA_API_KEY, ...) still apply. - src/models/laya/: model-side code only. engine.py holds the warmup, the per-model description /health returns and the revision record; optimize.py is unchanged. - tests/laya/: the unit and contract tests, run with PYTHONPATH=src. - recipe/laya/requirements-mps.txt pins the validated environment. - recipe/laya/bench/: 10 scripts become 6. The answer comparison is a section of report.py, the frontend-overhead probe is paired.py's --a-url/--b-url mode, and the workload generator and token check are replaced by the committed workloads.jsonl plus its token counts in the README. The five result tables leave the repository and join the raw data as a release asset. - README: LAYA's row in the models table points at the Apple Silicon worker. No behaviour change: 26 unit and 13 contract tests pass, and a feasibility pass of every script gives the same numbers as before (68-token request 28 ms with both options, paired ratio 0.64, answers within 0.0031, frontend overhead ratio 1.00-1.01). --- README.md | 2 +- recipe/laya/apple-silicon.md | 46 +- recipe/laya/bench/README.md | 67 ++- recipe/laya/bench/bench_http.py | 16 +- recipe/laya/bench/check_workloads.py | 27 -- recipe/laya/bench/frontend_overhead.py | 80 ---- recipe/laya/bench/paired.py | 79 ++-- recipe/laya/bench/parity.py | 113 ----- recipe/laya/bench/report.py | 95 ++++ recipe/laya/bench/results/.gitignore | 10 +- .../bench/results/frontend_overhead_m1.md | 10 - recipe/laya/bench/results/measured-parity.md | 336 -------------- recipe/laya/bench/results/measured-report.md | 431 ------------------ .../bench/results/paired-all-optimizations.md | 30 -- recipe/laya/bench/results/paired-fp16.md | 30 -- recipe/laya/bench/workloads.src.py | 90 ---- recipe/laya/requirements-mps.txt | 10 + src/frontend/laya_mps.py | 139 ++++++ src/models/laya/README.md | 53 +-- src/models/laya/engine.py | 88 ++++ src/models/laya/optimize.py | 2 +- src/models/laya/worker.py | 232 ---------- .../tests => tests/laya}/test_contract.py | 29 +- .../laya/tests => tests/laya}/test_worker.py | 86 ++-- 24 files changed, 526 insertions(+), 1575 deletions(-) delete mode 100644 recipe/laya/bench/check_workloads.py delete mode 100644 recipe/laya/bench/frontend_overhead.py delete mode 100644 recipe/laya/bench/parity.py delete mode 100644 recipe/laya/bench/results/frontend_overhead_m1.md delete mode 100644 recipe/laya/bench/results/measured-parity.md delete mode 100644 recipe/laya/bench/results/measured-report.md delete mode 100644 recipe/laya/bench/results/paired-all-optimizations.md delete mode 100644 recipe/laya/bench/results/paired-fp16.md delete mode 100644 recipe/laya/bench/workloads.src.py create mode 100644 recipe/laya/requirements-mps.txt create mode 100644 src/frontend/laya_mps.py create mode 100644 src/models/laya/engine.py delete mode 100644 src/models/laya/worker.py rename {src/models/laya/tests => tests/laya}/test_contract.py (91%) rename {src/models/laya/tests => tests/laya}/test_worker.py (78%) diff --git a/README.md b/README.md index b99d91c..fe1b4fa 100644 --- a/README.md +++ b/README.md @@ -55,7 +55,7 @@ LAYA can run as an external Python worker for text requests; its in-repository m | Model | Status | | --- | --- | -| LAYA | [External worker](recipe/laya/README.md); model engine planned | +| LAYA | [External worker](recipe/laya/README.md); [Python worker on Apple Silicon (MPS) and CPU](recipe/laya/apple-silicon.md); model engine planned | | Cua-S1 4B 0.2 (`text` adapter) | [Python worker](recipe/cua_s1/text.md); [native worker](recipe/cua_s1/native.md), CUDA, run on sm_89 | CUDA and Metal coverage will be documented per model as implementations are added and validated. diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 874ba9d..0c8aed0 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -1,8 +1,9 @@ # Laya on Apple Silicon -This recipe serves Laya on the GPU of an Apple Silicon Mac (PyTorch MPS) with the Laya worker -from [`src/models/laya/`](../../src/models/laya/), puts the Rust frontend in front of it and runs -the benchmark suite. The [Laya text worker](README.md) recipe covers the plain CPU setup. +This recipe serves Laya on the GPU of an Apple Silicon Mac (PyTorch MPS) with the worker in +[`src/frontend/laya_mps.py`](../../src/frontend/laya_mps.py), puts the Rust frontend in front of it and +runs the benchmarks. The model-side code is in [`src/models/laya/`](../../src/models/laya/). The +[Laya text worker](README.md) recipe covers plain laya-serve on the CPU. Validated on an M1 Pro (16 GB, 16-core GPU), macOS 26.1, Python 3.12, `laya[serve]==0.3.20`, torch 2.14.0 and the `english` checkpoint (`convaiinnovations/laya` at `55cf4c4`), and by another @@ -17,26 +18,24 @@ with `uv venv --python 3.12 --seed .venv` (`--seed` puts pip in the environment) ```sh python3.12 -m venv .venv -.venv/bin/python -m pip install 'laya[serve]==0.3.20' pytest httpx2 +.venv/bin/python -m pip install -r recipe/laya/requirements-mps.txt .venv/bin/python -c "import torch; print(torch.backends.mps.is_available())" ``` -The last command must print `True`. The standard macOS arm64 wheel of torch includes MPS. `pytest` and -`httpx2` are only for the tests: Starlette's `TestClient` needs `httpx2` (or, deprecated, `httpx`). +The last command must print `True`. The standard macOS arm64 wheel of torch includes MPS. ## Start the worker ```sh -LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps LAYA_MODELS=english \ -LAYA_REQUIRE_DEVICE=1 \ - .venv/bin/python src/models/laya/worker.py +PYTHONPATH=src .venv/bin/python -m frontend.laya_mps --device mps --model english --require-device --port 8000 ``` First startup downloads the checkpoint (about 850 MB). The worker loads the model, runs a warmup over short, long and multi-question requests, and only then listens on port 8000, so the first request it accepts is already warm: on an M1 Pro the first request after ready took 70–81 ms, against -0.7–1.1 s from plain laya-serve. `LAYA_REQUIRE_DEVICE=1` makes it exit instead of silently serving on the -CPU when the model cannot be placed on MPS; without it the worker logs a warning and serves from the CPU. +0.7–1.1 s from plain laya-serve. `--require-device` makes it exit instead of silently serving on the CPU +when the model cannot be placed on MPS; without it the worker logs a warning and serves from the CPU. +laya-serve's environment variables still apply, e.g. `LAYA_API_KEY` for bearer authentication. Check what it is running on: @@ -55,14 +54,13 @@ after about 30 s, and later 68-token requests took 140–270 ms from the CPU. ### Faster: compile and fp16 weights ```sh -LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 LAYA_HOST=127.0.0.1 LAYA_PORT=8000 LAYA_DEVICE=mps \ -LAYA_MODELS=english LAYA_REQUIRE_DEVICE=1 \ - .venv/bin/python src/models/laya/worker.py +PYTHONPATH=src .venv/bin/python -m frontend.laya_mps --device mps --model english --require-device \ + --compile --weights fp16 --port 8000 ``` -`LAYA_WORKER_COMPILE=on` compiles the model during warmup: one-question requests run the whole model -compiled, requests with several questions run the encoder compiled and Laya's decision head as it is. -`LAYA_WORKER_WEIGHTS=fp16` keeps the checkpoint's fp16 weights instead of Laya's fp32 upcast on MPS. +`--compile` compiles the model during warmup: one-question requests run the whole model compiled, +requests with several questions run the encoder compiled and Laya's decision head as it is. +`--weights fp16` keeps the checkpoint's fp16 weights instead of Laya's fp32 upcast on MPS. On the M1 Pro, with both workers running and every request sent to each back to back, the two options together lowered warm p50 against the worker without them by 37–38% for a 68-token one-question @@ -108,9 +106,12 @@ The frontend forwards the worker's response unchanged; `compare_with_backend.py` ## Test +The tests need `pytest` and `httpx2` (Starlette's `TestClient`; `httpx` works with a deprecation warning): + ```sh -.venv/bin/python -m pytest src/models/laya/tests # unit tests, no model -LAYA_CONTRACT=1 .venv/bin/python -m pytest src/models/laya/tests # plus contract tests against a CPU worker +.venv/bin/python -m pip install pytest httpx2 +PYTHONPATH=src .venv/bin/python -m pytest tests/laya # unit tests, no model +LAYA_CONTRACT=1 PYTHONPATH=src .venv/bin/python -m pytest tests/laya # plus contract tests against a CPU worker ``` ## Benchmark @@ -119,12 +120,13 @@ Stop the worker and frontend first; the benchmark starts its own. The scripts ar [`bench/`](bench/README.md). A first pass that checks everything runs: ```sh -.venv/bin/python recipe/laya/bench/check_workloads.py .venv/bin/python recipe/laya/bench/bench_inproc.py --device mps --config C2 --run feasibility .venv/bin/python recipe/laya/bench/bench_http.py --config C3 --run feasibility --spawn .venv/bin/laya-serve .venv/bin/python recipe/laya/bench/bench_http.py --config C4 --run feasibility \ --url http://127.0.0.1:8080 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve +.venv/bin/python recipe/laya/bench/paired.py --run feasibility --a "" --b "--compile --weights fp16" .venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*_feasibility.jsonl +.venv/bin/python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_feasibility.jsonl ``` Runs labelled anything other than `feasibility` refuse to start on battery power or when the @@ -132,8 +134,8 @@ Runs labelled anything other than `feasibility` refuse to start on battery power ## Troubleshooting -- `device_mismatch: true` at startup, or the worker exits with `asked for mps, english is on cpu`: MPS - is not available to this Python. Check the `torch.backends.mps.is_available()` line above (an x86_64 +- `device_mismatch: true` at startup, or with `--require-device` the worker exits with + `asked for mps, english is on cpu`: MPS is not available to this Python. Check the `torch.backends.mps.is_available()` line above (an x86_64 Python under Rosetta, for example, has no MPS). - `device_mismatch: true` on a worker that started on MPS: Laya fell back to the CPU after a GPU out-of-memory error. It keeps answering, several times slower; free memory and restart the worker diff --git a/recipe/laya/bench/README.md b/recipe/laya/bench/README.md index 77e8a1a..026624b 100644 --- a/recipe/laya/bench/README.md +++ b/recipe/laya/bench/README.md @@ -1,71 +1,60 @@ # Laya benchmark scripts Scripts behind the numbers in the [Apple Silicon recipe](../apple-silicon.md). Each run writes raw -JSONL to `results/`; `report.py` and `parity.py` build the tables from it. +JSONL to `results/` (kept out of the repository); `report.py` builds the tables from it. | file | purpose | | --- | --- | -| `workloads.src.py` → `workloads.jsonl` | fixed inputs: W1–W6 timed, P* parity only | -| `check_workloads.py` | token count of each input with Laya's tokenizer | -| `bench_inproc.py` | Laya in-process: load, warmup, first request, warm latency, memory | -| `bench_http.py` | a `/v1/systemone` worker, optionally behind the frontend: time to ready, first request, warm latency, throughput | -| `paired.py` | two worker configurations alive at once, each request sent to both back to back; median ratio with a bootstrap interval | -| `frontend_overhead.py` | frontend cost, each request sent directly and through the frontend back to back | +| `workloads.jsonl` | the fixed inputs: W1–W6 are timed, P* are for answer comparison only. Tokens per row with Laya's tokenizer: W1 68, W2 198, W3 484, W4 68/48/47, W5 40–68, W6 47 | +| `bench_inproc.py` | Laya in-process (no HTTP): load, warmup, first request, warm latency, memory | +| `bench_http.py` | a `/v1/systemone` server, optionally started by the script and optionally behind the frontend: time to ready, first request, warm latency, throughput | +| `paired.py` | two configurations compared request by request, both alive at once: two worker flag sets, or two running servers (e.g. a worker directly and through the frontend) | | `profile_mps.py` | where a request's time goes on MPS | -| `parity.py` | answers of every run against a reference run | -| `report.py` | tables from the JSONL | -| `env.py` | versions, checkpoint, hardware and load recorded with each run | +| `report.py` | tables from the JSONL, including the run-to-run gate and the answer comparison against a reference config | +| `env.py` | shared: versions, checkpoint, hardware, power and load recorded with each run; memory footprint | ## Run -From the repository root, in the environment of the recipe: +From the repository root, in the recipe's environment (`.venv`), with the frontend built: ```sh -python recipe/laya/bench/check_workloads.py python recipe/laya/bench/bench_inproc.py --device cpu --config C1 --run m1 python recipe/laya/bench/bench_inproc.py --device mps --config C2 --run m1 python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv/bin/laya-serve python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ --frontend target/release/omni-jev --spawn .venv/bin/laya-serve -LAYA_WORKER_COMPILE=off python recipe/laya/bench/bench_http.py --config C3w --run m1 \ - --spawn .venv/bin/python src/models/laya/worker.py -LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16 python recipe/laya/bench/bench_http.py --config C3o --run m1 \ - --spawn .venv/bin/python src/models/laya/worker.py -python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl -python recipe/laya/bench/parity.py recipe/laya/bench/results/*_m[0-9].jsonl --ref C1 +python recipe/laya/bench/bench_http.py --config C3w --run m1 \ + --spawn .venv/bin/python -m frontend.laya_mps --device mps --port {port} +python recipe/laya/bench/bench_http.py --config C3o --run m1 \ + --spawn .venv/bin/python -m frontend.laya_mps --device mps --compile --weights fp16 --port {port} +python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl --ref C1 ``` Repeat with `--run m2` for a second measured run. Runs refuse to start on battery power or above a 1-minute load average of `--max-load` (default 2) unless labelled `--run feasibility`. Memory is the process's physical footprint, which on Apple Silicon includes MPS allocations. -Two worker configurations can also be compared request by request, which holds up under background -load better than separate runs: +Two configurations compared request by request, which holds up under background load better than +separate runs: ```sh -python recipe/laya/bench/paired.py --run p1 --a "LAYA_WORKER_COMPILE=off" \ - --b "LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16" +python recipe/laya/bench/paired.py --run p1 --a "" --b "--compile --weights fp16" +python recipe/laya/bench/paired.py --run f1 --a-url http://127.0.0.1:8000 --b-url http://127.0.0.1:8080 python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_p1.jsonl ``` ## Results -`results/` holds the reports from the measured runs on an M1 Pro: `measured-report.md`, -`measured-parity.md`, `frontend_overhead_m1.md`, `paired-fp16.md` and `paired-all-optimizations.md`. -`C3s` in `measured-report.md` and side B of `paired-fp16.md` ran an earlier compile mode that compiled -one-question requests only; `on` compiles them the same way and adds the encoder for several questions. The raw JSONL is published as -release assets: +The measured runs on an M1 Pro are published as assets of one release on the fork, +: -```sh -curl -LO https://github.com/cacheline999/system1-omni/releases/download/laya-mps-results-2026-09-28/laya-mps-results-2026-09-28.tar.gz -shasum -a 256 laya-mps-results-2026-09-28.tar.gz # 611ed30707ac8c98875b5aa5382360b5a7d760da166d61c626eb07ebe1ee6404 -tar xzf laya-mps-results-2026-09-28.tar.gz -C recipe/laya/bench/results -python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl -``` - -The paired fp16 runs (`paired_e4a.jsonl`, `paired_e4b.jsonl`) are in -`laya-mps-paired-fp16-2026-09-30.tar.gz` on the same release (sha256 `cc0d6f5bda6e3e0ee1f40c6966f84429e902a2f585b8e1ee33658a9be139326e`); rebuild the summary with -`paired.py --summarize`. +| asset | contents | sha256 | +| --- | --- | --- | +| `laya-mps-reports-2026-10-01.tar.gz` | the tables: baseline report and parity, frontend overhead, paired fp16, paired all optimizations | `e857da5082da4983e07a91104b20f84fb1ffd7994c56123e0a832e0d7a870cea` | +| `laya-mps-results-2026-09-28.tar.gz` | raw JSONL of the baseline runs (C1–C4, C3w, C3s) | `611ed30707ac8c98875b5aa5382360b5a7d760da166d61c626eb07ebe1ee6404` | +| `laya-mps-paired-fp16-2026-09-30.tar.gz` | raw JSONL of the paired fp16 runs | `cc0d6f5bda6e3e0ee1f40c6966f84429e902a2f585b8e1ee33658a9be139326e` | +| `laya-mps-paired-all-2026-09-30.tar.gz` | raw JSONL of the paired all-optimizations runs | `25d1b7bc9d6dff7173f9b972ebd0204f8fb4e2089fe2d27924d48ba6e589780c` | -The runs behind `paired-all-optimizations.md` (`paired_e5a.jsonl`, `paired_e5b.jsonl`) are in -`laya-mps-paired-all-2026-09-30.tar.gz` on the same release (sha256 `25d1b7bc9d6dff7173f9b972ebd0204f8fb4e2089fe2d27924d48ba6e589780c`). +Extract the raw JSONL into `results/` and run `report.py` or `paired.py --summarize` on it to rebuild +the tables. `C3s` in the baseline runs is an earlier compile mode that compiled one-question requests +only; `--compile` does the same for them and adds the encoder for several questions. diff --git a/recipe/laya/bench/bench_http.py b/recipe/laya/bench/bench_http.py index 67215b7..77e364e 100644 --- a/recipe/laya/bench/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -6,9 +6,14 @@ --url in front of it; readiness is then the frontend's /health, which proxies the worker's. Each workload runs at every --concurrency level; each client thread keeps one keep-alive connection. - python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv-laya/bin/laya-serve + python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv/bin/laya-serve python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ - --frontend target/release/omni-jev --spawn .venv-laya/bin/laya-serve + --frontend target/release/omni-jev --spawn .venv/bin/laya-serve + python recipe/laya/bench/bench_http.py --config C3o --run m1 \ + --spawn .venv/bin/python -m frontend.laya_mps --device mps --compile --weights fp16 --port {port} + +The spawned command gets LAYA_HOST/LAYA_PORT/LAYA_DEVICE/LAYA_MODELS in its environment (what laya-serve +reads), PYTHONPATH=src (for `-m frontend.laya_mps`), and `{port}` in its arguments replaced by the port. """ import argparse @@ -24,6 +29,7 @@ from urllib.parse import urlsplit HERE = Path(__file__).resolve().parent +REPO = HERE.parents[2] sys.path.insert(0, str(HERE)) from env import footprint_mb, header, noise_problems @@ -182,8 +188,10 @@ def main(): "LAYA_PRELOAD": "1", "LAYA_LOG_LEVEL": "warning", } + env["PYTHONPATH"] = str(REPO / "src") + (os.pathsep + env["PYTHONPATH"] if env.get("PYTHONPATH") else "") + command = [arg.replace("{port}", str(port)) for arg in args.spawn] spawn_log = open(Path(args.out) / f"http_{args.config}_{args.run}.worker.log", "w") # noqa: SIM115 - processes["worker"] = subprocess.Popen(args.spawn, env=env, stdout=spawn_log, stderr=subprocess.STDOUT) + processes["worker"] = subprocess.Popen(command, env=env, stdout=spawn_log, stderr=subprocess.STDOUT, cwd=REPO) if args.frontend: parts = urlsplit(args.url) env = { @@ -263,7 +271,7 @@ def emit(record): for w in order: body = body_for(w, args.model) answers, error = fetch_answers(client, body) - if error: # parity.py reports the workload as missing + if error: # the parity section of report.py reports the workload as missing emit({"type": "answers_error", "workload": w["id"], **error}) else: emit({"type": "answers", "workload": w["id"], "answers": answers}) diff --git a/recipe/laya/bench/check_workloads.py b/recipe/laya/bench/check_workloads.py deleted file mode 100644 index f36e8b8..0000000 --- a/recipe/laya/bench/check_workloads.py +++ /dev/null @@ -1,27 +0,0 @@ -"""T1 check: tokens per row of every workload, measured with Laya's own tokenizer (usage.input_tokens). - -Each question is sent alone so its row length is exact; bench workloads must land within ±10% of target. -""" - -import json -import sys -import warnings -from pathlib import Path - -warnings.filterwarnings("ignore") -import laya - -agent = laya.load("convaiinnovations/laya", device="cpu") -max_len = agent.cfg.get("max_len", 512) -failed = 0 -with open(Path(__file__).resolve().parent / "workloads.jsonl") as f: - workloads = [json.loads(line) for line in f] -for w in workloads: - rows = [agent.system_one(w["state"], {qid: q})["usage"]["input_tokens"] for qid, q in w["questions"].items()] - mean = sum(rows) / len(rows) - target = w["target_tokens_per_row"] - ok = target is None or abs(mean - target) <= 0.1 * target - ok = ok and max(rows) < max_len # a full row means Laya cut the state - failed += not ok - print(f"{'ok ' if ok else 'BAD'} {w['id']:5} rows={len(rows)} tokens/row={rows} mean={mean:.0f} target={target}") -sys.exit(1 if failed else 0) diff --git a/recipe/laya/bench/frontend_overhead.py b/recipe/laya/bench/frontend_overhead.py deleted file mode 100644 index dea614c..0000000 --- a/recipe/laya/bench/frontend_overhead.py +++ /dev/null @@ -1,80 +0,0 @@ -"""Frontend overhead, paired: each request goes to the worker directly and through the frontend, one right -after the other, alternating which goes first. Pairing cancels background load that shifts separate -runs, so the per-request difference is the frontend's cost. - -Start a worker and the frontend first (recipe/laya/apple-silicon.md), then: - - python recipe/laya/bench/frontend_overhead.py --direct http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 -""" - -import argparse -import http.client -import json -import statistics -import time -from pathlib import Path -from urllib.parse import urlsplit - -HERE = Path(__file__).resolve().parent - - -def connect(url): - parts = urlsplit(url) - return http.client.HTTPConnection(parts.hostname, parts.port or 80, timeout=120) - - -def quantile(sorted_values, p): - return sorted_values[min(len(sorted_values) - 1, int(p * len(sorted_values)))] - - -def call(conn, body): - started = time.perf_counter() - conn.request("POST", "/v1/systemone", body=body, headers={"Content-Type": "application/json"}) - response = conn.getresponse() - response.read() - if response.status != 200: - raise SystemExit(f"status {response.status}") - return (time.perf_counter() - started) * 1000 - - -def main(): - parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) - parser.add_argument("--direct", default="http://127.0.0.1:8000") - parser.add_argument("--frontend", default="http://127.0.0.1:8080") - parser.add_argument("--model", default="english") - parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) - parser.add_argument("--only", nargs="*", help="bench workload ids (default: all)") - parser.add_argument("-n", type=int, default=120, help="pairs per workload") - parser.add_argument("--discard", type=int, default=10) - args = parser.parse_args() - - with open(args.workloads) as f: - workloads = [json.loads(line) for line in f if line.strip()] - bench = [w for w in workloads if w["kind"] == "bench" and (not args.only or w["id"] in args.only)] - direct, frontend = connect(args.direct), connect(args.frontend) - - print("| workload | body bytes | direct p50 | frontend p50 | Δ p10 | Δ p50 | Δ p90 | Δ > 10 ms |") - print("|---|---|---|---|---|---|---|---|") - for w in bench: - body = json.dumps({"model": args.model, "state": w["state"], "questions": w["questions"]}).encode() - for _ in range(args.discard): - call(direct, body) - call(frontend, body) - d_ms, f_ms, delta = [], [], [] - for i in range(args.n): - if i % 2 == 0: - a, b = call(direct, body), call(frontend, body) - else: - b, a = call(frontend, body), call(direct, body) - d_ms.append(a) - f_ms.append(b) - delta.append(b - a) - delta.sort() - print( - f"| {w['id']} | {len(body)} | {statistics.median(d_ms):.1f} | {statistics.median(f_ms):.1f} " - f"| {quantile(delta, 0.1):+.1f} | {statistics.median(delta):+.1f} | {quantile(delta, 0.9):+.1f} | {sum(x > 10 for x in delta)}/{args.n} |" - ) - - -if __name__ == "__main__": - main() diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index 377fbaa..a867255 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -1,18 +1,21 @@ """Paired comparison of two worker configurations: both run at once, and every request goes to A and to B back to back, alternating which goes first, so background load that shifts both cancels out. - python recipe/laya/bench/paired.py --run p1 \ - --a "LAYA_WORKER_COMPILE=off" --b "LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16" - python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_e4*.jsonl - -Each side is src/models/laya/worker.py started with the given environment. The summary reports, per -input, the median of the per-pair ratio B/A with a 95% bootstrap interval, and B's answers against A's. + python recipe/laya/bench/paired.py --run p1 --a "" --b "--compile --weights fp16" + python recipe/laya/bench/paired.py --run f1 --a-url http://127.0.0.1:8000 --b-url http://127.0.0.1:8080 + python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_p1.jsonl + +`--a`/`--b` are extra flags for `frontend.laya_mps`, which the script starts on MPS with the english +model; `--a-url`/`--b-url` compare two servers that are already running (e.g. a worker directly and +through the Rust frontend). The summary reports, per input, the median of the per-pair ratio B/A with a +95% bootstrap interval, and B's answers against A's. """ import argparse import json import os import random +import shlex import statistics import subprocess import sys @@ -27,20 +30,24 @@ CHECKPOINT = "convaiinnovations/laya" -def spawn(env_spec, port, python, log_path): - env = { - **os.environ, - "LAYA_HOST": "127.0.0.1", - "LAYA_PORT": str(port), - "LAYA_DEVICE": "mps", - "LAYA_MODELS": "english", - "LAYA_LOG_LEVEL": "warning", - } - env.update(pair.split("=", 1) for pair in env_spec.split()) +def spawn(flags, port, python, model, log_path): + env = {**os.environ, "PYTHONPATH": str(REPO / "src")} + command = [ + python, + "-m", + "frontend.laya_mps", + "--device", + "mps", + "--model", + model, + "--port", + str(port), + "--log-level", + "warning", + *shlex.split(flags), + ] log = open(log_path, "w") # noqa: SIM115 - return subprocess.Popen( - [python, str(REPO / "src/models/laya/worker.py")], env=env, stdout=log, stderr=subprocess.STDOUT - ) + return subprocess.Popen(command, env=env, stdout=log, stderr=subprocess.STDOUT, cwd=REPO) def run(args): @@ -53,13 +60,21 @@ def run(args): parity = [w for w in workloads if w["kind"] == "parity"] out_dir = Path(args.out) out_dir.mkdir(parents=True, exist_ok=True) - sides = {"A": (args.a, args.port_a), "B": (args.b, args.port_b)} - procs = { - s: spawn(spec, port, args.python, out_dir / f"paired_{args.run}_{s}.log") for s, (spec, port) in sides.items() - } - urls = {s: f"http://127.0.0.1:{port}" for s, (_, port) in sides.items()} + if args.a_url or args.b_url: + if not (args.a_url and args.b_url) or args.a is not None or args.b is not None: + sys.exit("give both --a-url and --b-url, without --a/--b") + sides = {"A": args.a_url, "B": args.b_url} + procs = {} + urls = sides + else: + sides = {"A": (args.a or "", args.port_a), "B": (args.b or "", args.port_b)} + procs = { + s: spawn(flags, port, args.python, args.model, out_dir / f"paired_{args.run}_{s}.log") + for s, (flags, port) in sides.items() + } + urls = {s: f"http://127.0.0.1:{port}" for s, (_, port) in sides.items()} try: - health = {s: wait_ready(urls[s], {s: procs[s]}, args.ready_timeout)[1] for s in sides} + health = {s: wait_ready(urls[s], {s: procs[s]} if s in procs else {}, args.ready_timeout)[1] for s in sides} with open(out_dir / f"paired_{args.run}.jsonl", "w") as f: def emit(record): @@ -68,8 +83,8 @@ def emit(record): emit( header( CHECKPOINT, - a=args.a, - b=args.b, + a=args.a_url or f"frontend.laya_mps {args.a or ''}".strip(), + b=args.b_url or f"frontend.laya_mps {args.b or ''}".strip(), n=args.n, discard=args.discard, seed=args.seed, @@ -110,7 +125,7 @@ def emit(record): { "type": "end", "health": end_health, - "footprint_mb": {s: footprint_mb(procs[s].pid).get("footprint_mb") for s in sides}, + "footprint_mb": {s: footprint_mb(procs[s].pid).get("footprint_mb") for s in procs}, } ) finally: @@ -185,8 +200,10 @@ def main(): parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) parser.add_argument("--summarize", nargs="+", metavar="JSONL") parser.add_argument("--run") - parser.add_argument("--a", default="LAYA_WORKER_COMPILE=off", help="environment of side A") - parser.add_argument("--b", default="LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16", help="environment of side B") + parser.add_argument("--a", help="extra frontend.laya_mps flags for side A (default: none)") + parser.add_argument("--b", help="extra frontend.laya_mps flags for side B (default: --compile --weights fp16)") + parser.add_argument("--a-url", help="instead of starting workers: an already running server for side A") + parser.add_argument("--b-url", help="... and for side B") parser.add_argument("--port-a", type=int, default=8000) parser.add_argument("--port-b", type=int, default=8001) parser.add_argument("--python", default=sys.executable) @@ -202,6 +219,8 @@ def main(): if args.summarize: summarize(args.summarize) elif args.run: + if args.b is None and not args.b_url: + args.b = "--compile --weights fp16" run(args) else: parser.error("give --run or --summarize") diff --git a/recipe/laya/bench/parity.py b/recipe/laya/bench/parity.py deleted file mode 100644 index a7bb794..0000000 --- a/recipe/laya/bench/parity.py +++ /dev/null @@ -1,113 +0,0 @@ -"""Compare every run's answers with a reference run, using the tolerances declared in advance. - - python recipe/laya/bench/parity.py recipe/laya/bench/results/*.jsonl --ref C1 - -Per question: the decision must match (choice: chosen option; score: most likely level; noul: side of -0.5) and the largest absolute probability difference must stay within tolerance: 1e-3 when the request -ran in fp32, 1e-2 when it ran in fp16: fp16 weights (every request), or fp16 autocast on MPS for -rows >= the worker's amp threshold. -A flipped decision is reported with the reference margin between its top two outcomes. -Exits 1 when any question fails. -""" - -import argparse -import json -import sys - -TOLERANCE = {"fp32": 1e-3, "fp16": 1e-2} - - -def read(paths): - envs, answers = {}, {} - for path in paths: - with open(path) as f: - for line in f: - if not line.strip(): - continue - r = json.loads(line) - key = (r["config"], r["run"]) - if r["type"] == "env": - envs[key] = r - elif r["type"] == "answers": - answers.setdefault(key, {})[r["workload"]] = r["answers"] - return envs, answers - - -def outcome(answer): - """(decision, {outcome: probability}) for one question's answer.""" - kind = answer["type"] - if kind == "noul": - p = answer["noul"] - return p >= 0.5, {"yes": p} - probs = answer["probabilities"] - if kind == "choice": - return answer["choice"], probs - return max(probs, key=probs.get), probs # score: most likely level - - -def margin(probs): - if len(probs) == 1: # noul: distance from the 0.5 boundary - return abs(next(iter(probs.values())) - 0.5) - top = sorted(probs.values(), reverse=True) - return top[0] - top[1] - - -def path(env, rows): - autocast = ( - env.get("device_actual") == "mps" - and env.get("amp_dtype") == "torch.float16" - and rows >= (env.get("mps_amp_min_rows") or 10**9) - ) - return "fp16" if autocast or env.get("weights_dtype") == "torch.float16" else "fp32" - - -def main(): - parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) - parser.add_argument("files", nargs="+") - parser.add_argument("--ref", default="C1", help="reference config") - parser.add_argument("--ref-run", help="reference run label (default: the first one found)") - args = parser.parse_args() - envs, answers = read(args.files) - - refs = sorted(k for k in answers if k[0] == args.ref and (not args.ref_run or k[1] == args.ref_run)) - if not refs: - sys.exit(f"no answers for reference config {args.ref}") - ref_key = refs[0] - reference = answers[ref_key] - - print(f"Reference: {ref_key[0]} / {ref_key[1]}. Tolerances: fp32 {TOLERANCE['fp32']}, fp16 {TOLERANCE['fp16']}.\n") - print( - "| config | run | workload | question | type | path | decision ref → run | max abs Δp | ref margin | result |" - ) - print("|---|---|---|---|---|---|---|---|---|---|") - failed = total = 0 - for key in sorted(answers): - if key == ref_key: - continue - env = envs.get(key, {}) - for workload, questions in sorted(reference.items()): - got = answers[key].get(workload) - if got is None: - print(f"| {key[0]} | {key[1]} | {workload} | | | | missing | | | FAIL |") - failed += 1 - total += 1 - continue - precision = path(env, len(questions)) - for qid, ref_answer in sorted(questions.items()): - total += 1 - ref_decision, ref_probs = outcome(ref_answer) - decision, probs = outcome(got[qid]) - delta = max(abs(ref_probs.get(o, 0.0) - probs.get(o, 0.0)) for o in ref_probs.keys() | probs.keys()) - ok = decision == ref_decision and delta <= TOLERANCE[precision] - failed += not ok - flip = f"{ref_decision} → {decision}" if decision != ref_decision else f"{decision}" - print( - f"| {key[0]} | {key[1]} | {workload} | {qid} | {ref_answer['type']} | {precision} | {flip} " - f"| {delta:.4f} | {margin(ref_probs):.4f} | {'PASS' if ok else 'FAIL'} |" - ) - print(f"\n{total - failed}/{total} questions within tolerance.") - sys.exit(1 if failed else 0) - - -if __name__ == "__main__": - main() diff --git a/recipe/laya/bench/report.py b/recipe/laya/bench/report.py index 0d8d6f5..0961e12 100644 --- a/recipe/laya/bench/report.py +++ b/recipe/laya/bench/report.py @@ -2,6 +2,9 @@ python recipe/laya/bench/report.py recipe/laya/bench/results/*.jsonl +The parity section compares every run's answers with the reference config's (`--ref`, default C1): the +decision must match (choice: option; score: most likely level; noul: side of 0.5) and the largest |Δp| +must stay within 1e-3 for fp32 and 1e-2 where the run used fp16 (autocast or fp16 weights). Percentiles are nearest-rank. The run-to-run gate compares p50 across measured runs (every run whose label is not "feasibility") of the same config, workload and concurrency: (max - min) / min <= 10%. In-process results have concurrency 1. @@ -170,9 +173,100 @@ def memory(phases, ends): return table(headers, rows) +TOLERANCE = {"fp32": 1e-3, "fp16": 1e-2} + + +def read_answers(paths): + envs, answers = {}, {} + for path in paths: + with open(path) as f: + for line in f: + if not line.strip(): + continue + r = json.loads(line) + key = (r["config"], r["run"]) + if r["type"] == "env": + envs[key] = r + elif r["type"] == "answers": + answers.setdefault(key, {})[r["workload"]] = r["answers"] + return envs, answers + + +def outcome(answer): + """(decision, {outcome: probability}) for one question's answer.""" + kind = answer["type"] + if kind == "noul": + p = answer["noul"] + return p >= 0.5, {"yes": p} + probs = answer["probabilities"] + if kind == "choice": + return answer["choice"], probs + return max(probs, key=probs.get), probs # score: most likely level + + +def margin(probs): + if len(probs) == 1: # noul: distance from the 0.5 boundary + return abs(next(iter(probs.values())) - 0.5) + top = sorted(probs.values(), reverse=True) + return top[0] - top[1] + + +def path(env, rows): + autocast = ( + env.get("device_actual") == "mps" + and env.get("amp_dtype") == "torch.float16" + and rows >= (env.get("mps_amp_min_rows") or 10**9) + ) + return "fp16" if autocast or env.get("weights_dtype") == "torch.float16" else "fp32" + + +def parity(paths, ref): + """Answers of every run against the reference config, with the tolerances declared in advance.""" + envs, answers = read_answers(paths) + refs = sorted(k for k in answers if k[0] == ref) + if not refs: + return f"no answers for reference config {ref}" + ref_key = refs[0] + reference = answers[ref_key] + lines = [ + f"Reference: {ref_key[0]} / {ref_key[1]}. Tolerances: fp32 {TOLERANCE['fp32']}, fp16 {TOLERANCE['fp16']}.", + "", + "| config | run | workload | question | type | path | decision ref → run | max abs Δp | ref margin | result |", + "|---|---|---|---|---|---|---|---|---|---|", + ] + failed = total = 0 + for key in sorted(answers): + if key == ref_key: + continue + env = envs.get(key, {}) + for workload, questions in sorted(reference.items()): + got = answers[key].get(workload) + if got is None: + lines.append(f"| {key[0]} | {key[1]} | {workload} | | | | missing | | | FAIL |") + failed += 1 + total += 1 + continue + precision = path(env, len(questions)) + for qid, ref_answer in sorted(questions.items()): + total += 1 + ref_decision, ref_probs = outcome(ref_answer) + decision, probs = outcome(got[qid]) + delta = max(abs(ref_probs.get(o, 0.0) - probs.get(o, 0.0)) for o in ref_probs.keys() | probs.keys()) + ok = decision == ref_decision and delta <= TOLERANCE[precision] + failed += not ok + flip = f"{ref_decision} → {decision}" if decision != ref_decision else f"{decision}" + lines.append( + f"| {key[0]} | {key[1]} | {workload} | {qid} | {ref_answer['type']} | {precision} | {flip} " + f"| {delta:.4f} | {margin(ref_probs):.4f} | {'PASS' if ok else 'FAIL'} |" + ) + lines += ["", f"{total - failed}/{total} questions within tolerance."] + return "\n".join(lines) + + def main(): parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) parser.add_argument("files", nargs="+") + parser.add_argument("--ref", default="C1", help="reference config for the parity section (default C1)") args = parser.parse_args() records = read(args.files) @@ -188,6 +282,7 @@ def by_run(kind): if any(r["type"] == "throughput" for r in records): print("\n## Throughput\n\n" + throughput(records)) print("\n## Memory (MB)\n\n" + memory(phases, ends)) + print(f"\n## Parity against {args.ref}\n\n" + parity(args.files, args.ref)) if __name__ == "__main__": diff --git a/recipe/laya/bench/results/.gitignore b/recipe/laya/bench/results/.gitignore index a16a40b..2ef39dd 100644 --- a/recipe/laya/bench/results/.gitignore +++ b/recipe/laya/bench/results/.gitignore @@ -1,7 +1,3 @@ -# Committed: reports built from measured runs. -# Not committed: raw JSONL (published as a release asset, see ../README.md), feasibility runs, -# worker/frontend logs, runner state. -*.jsonl -*feasibility* -*.log -done_* +# Benchmark output stays out of the repository; the reports and raw data are release assets (see ../README.md). +* +!.gitignore diff --git a/recipe/laya/bench/results/frontend_overhead_m1.md b/recipe/laya/bench/results/frontend_overhead_m1.md deleted file mode 100644 index be3b463..0000000 --- a/recipe/laya/bench/results/frontend_overhead_m1.md +++ /dev/null @@ -1,10 +0,0 @@ -Paired frontend overhead, 2026-09-28T13:40Z, Apple M1 Pro, AC Power, 1-min load 4.80, omni 2a47358, laya-serve on MPS behind target/release/omni-jev - -| workload | body bytes | direct p50 | frontend p50 | Δ p10 | Δ p50 | Δ p90 | Δ > 10 ms | -|---|---|---|---|---|---|---|---| -| W1 | 416 | 45.3 | 45.6 | -0.9 | +0.3 | +1.7 | 0/120 | -| W2 | 1075 | 68.5 | 68.8 | -0.8 | +0.4 | +1.8 | 0/120 | -| W3 | 2527 | 151.5 | 151.8 | -3.7 | +0.5 | +3.5 | 0/120 | -| W4 | 657 | 71.0 | 71.3 | -0.8 | +0.3 | +1.7 | 0/120 | -| W5 | 968 | 138.9 | 139.0 | -2.8 | +0.1 | +3.6 | 1/120 | -| W6 | 197 | 39.0 | 39.2 | -0.9 | +0.2 | +1.6 | 0/120 | diff --git a/recipe/laya/bench/results/measured-parity.md b/recipe/laya/bench/results/measured-parity.md deleted file mode 100644 index 32f0b4d..0000000 --- a/recipe/laya/bench/results/measured-parity.md +++ /dev/null @@ -1,336 +0,0 @@ -Reference: C1 / m1. Tolerances: fp32 0.001, fp16 0.01. - -| config | run | workload | question | type | path | decision ref → run | max abs Δp | ref margin | result | -|---|---|---|---|---|---|---|---|---|---| -| C1 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C1 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C1 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C1 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C1 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C1 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C1 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C1 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C1 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C1 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C1 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C1 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C1 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C1 | m2 | W5 | angry | noul | fp32 | True | 0.0000 | 0.0231 | PASS | -| C1 | m2 | W5 | cancel | noul | fp32 | False | 0.0000 | 0.4696 | PASS | -| C1 | m2 | W5 | lang | choice | fp32 | en | 0.0000 | 0.1674 | PASS | -| C1 | m2 | W5 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C1 | m2 | W5 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m2 | W5 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C1 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C1 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C1 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C1 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C1 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C1 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C1 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C1 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C1 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C1 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C1 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C1 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C1 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C1 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C1 | m3 | W5 | angry | noul | fp32 | True | 0.0000 | 0.0231 | PASS | -| C1 | m3 | W5 | cancel | noul | fp32 | False | 0.0000 | 0.4696 | PASS | -| C1 | m3 | W5 | lang | choice | fp32 | en | 0.0000 | 0.1674 | PASS | -| C1 | m3 | W5 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C1 | m3 | W5 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C1 | m3 | W5 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C1 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C2 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C2 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C2 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C2 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C2 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C2 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C2 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C2 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C2 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C2 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C2 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C2 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C2 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C2 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C2 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C2 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C2 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C2 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C2 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C2 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C2 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C2 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C2 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C2 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C2 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C2 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C2 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C2 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C2 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C2 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C2 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C2 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C2 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C2 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C2 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C2 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C2 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C2 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C2 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C2 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C2 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C2 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C2 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C2 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3 | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3 | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3 | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3 | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3 | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3 | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3s | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3s | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3s | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3s | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3s | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3s | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3s | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3s | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3s | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3s | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3s | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3s | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3s | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3s | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3s | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3s | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3s | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3s | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3s | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3s | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3s | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3s | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3s | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3s | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3s | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3s | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3s | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3s | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3s | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3s | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3s | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3s | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3s | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3s | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3s | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3s | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3s | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3s | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3s | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3s | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3s | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3s | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3s | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3s | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3w | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3w | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3w | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3w | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3w | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3w | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3w | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3w | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3w | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3w | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3w | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3w | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3w | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3w | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3w | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3w | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3w | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3w | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3w | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3w | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3w | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3w | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3w | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3w | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3w | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3w | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3w | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3w | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C3w | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C3w | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C3w | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C3w | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C3w | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C3w | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C3w | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C3w | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C3w | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C3w | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C3w | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C3w | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C3w | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C3w | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C3w | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C3w | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C3w | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C3w | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m1 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C4 | m1 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m1 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C4 | m1 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m1 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C4 | m1 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m1 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C4 | m1 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m1 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C4 | m1 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m1 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C4 | m1 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C4 | m1 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m1 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m1 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C4 | m1 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C4 | m1 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C4 | m1 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C4 | m1 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C4 | m1 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C4 | m1 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C4 | m1 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m2 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C4 | m2 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m2 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C4 | m2 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m2 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C4 | m2 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m2 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C4 | m2 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m2 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C4 | m2 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m2 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C4 | m2 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C4 | m2 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m2 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m2 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C4 | m2 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C4 | m2 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C4 | m2 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C4 | m2 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C4 | m2 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C4 | m2 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C4 | m2 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m3 | P10 | route | choice | fp32 | logistics | 0.0000 | 0.1294 | PASS | -| C4 | m3 | P10L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m3 | P10L | route | choice | fp32 | returns | 0.0000 | 0.4531 | PASS | -| C4 | m3 | P10L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m3 | P2 | route | choice | fp32 | logistics | 0.0000 | 0.7298 | PASS | -| C4 | m3 | P2L | refund | noul | fp32 | True | 0.0000 | 0.3390 | PASS | -| C4 | m3 | P2L | route | choice | fp32 | logistics | 0.0000 | 0.0890 | PASS | -| C4 | m3 | P2L | urgency | score | fp32 | 2 | 0.0000 | 0.5591 | PASS | -| C4 | m3 | P5 | route | choice | fp32 | returns | 0.0000 | 0.2835 | PASS | -| C4 | m3 | W1 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m3 | W2 | route | choice | fp32 | logistics | 0.0000 | 0.1518 | PASS | -| C4 | m3 | W3 | route | choice | fp32 | logistics | 0.0000 | 0.1393 | PASS | -| C4 | m3 | W4 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | -| C4 | m3 | W4 | route | choice | fp32 | logistics | 0.0000 | 0.8265 | PASS | -| C4 | m3 | W4 | urgency | score | fp32 | 2 | 0.0000 | 0.6741 | PASS | -| C4 | m3 | W5 | angry | noul | fp16 | True | 0.0002 | 0.0231 | PASS | -| C4 | m3 | W5 | cancel | noul | fp16 | False | 0.0003 | 0.4696 | PASS | -| C4 | m3 | W5 | lang | choice | fp16 | en | 0.0005 | 0.1674 | PASS | -| C4 | m3 | W5 | refund | noul | fp16 | False | 0.0003 | 0.3833 | PASS | -| C4 | m3 | W5 | route | choice | fp16 | logistics | 0.0001 | 0.8265 | PASS | -| C4 | m3 | W5 | urgency | score | fp16 | 2 | 0.0005 | 0.6741 | PASS | -| C4 | m3 | W6 | refund | noul | fp32 | False | 0.0000 | 0.3833 | PASS | - -330/330 questions within tolerance. diff --git a/recipe/laya/bench/results/measured-report.md b/recipe/laya/bench/results/measured-report.md deleted file mode 100644 index 3889ce5..0000000 --- a/recipe/laya/bench/results/measured-report.md +++ /dev/null @@ -1,431 +0,0 @@ -## Environment - -| config | run | device | weights (autocast) | chip | os | power | load 1m | versions | ckpt | omni | -|---|---|---|---|---|---|---|---|---|---|---| -| C1 | m1 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.45 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C1 | m2 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.69 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C1 | m3 | cpu | torch.float32 (amp torch.float32, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.62 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C2 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.72 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C2 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.04 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 5.33 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.89 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3 | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 5.22 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | -| C3s | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.44 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3s | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.47 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3w | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.63 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3w | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 4.57 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C3w | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) | Apple M1 Pro | macOS 26.1 | AC Power | 5.32 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | -| C4 | m1 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.03 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C4 | m2 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 3.89 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 3062243+dirty | -| C4 | m3 | mps | torch.float32 (amp torch.float16, >= 5 rows) [assumed: laya 0.3.20 defaults] | Apple M1 Pro | macOS 26.1 | AC Power | 4.62 | laya 0.3.20 / torch 2.14.0 | 55cf4c4 | 2a47358 | - -## Phases (s) and first request per workload (ms) - -| config | run | import | load | process_to_ready | warmup | first W1 | first W2 | first W3 | first W4 | first W5 | first W6 | -|---|---|---|---|---|---|---|---|---|---|---|---| -| C1 | m1 | 1.477 | 5.511 | | 32.22 | 851.59 | 233.17 | 396.66 | 229.58 | 288.89 | 218.59 | -| C1 | m2 | 1.337 | 4.902 | | 28.264 | 676.67 | 249.27 | 509.78 | 213.0 | 298.91 | 120.46 | -| C1 | m3 | 1.425 | 5.501 | | 28.847 | 562.08 | 228.06 | 447.66 | 210.57 | 308.64 | 115.43 | -| C2 | m1 | 1.309 | 6.023 | | 11.54 | 999.09 | 77.8 | 157.55 | 86.45 | 269.89 | 39.37 | -| C2 | m2 | 0.917 | 5.562 | | 13.615 | 3241.64 | 73.52 | 161.09 | 86.16 | 326.02 | 42.86 | -| C3 | m1 | | | 7.478 | 11.876 | 1134.76 | 76.18 | 160.33 | 84.95 | 356.31 | 42.91 | -| C3 | m2 | | | 7.312 | 11.13 | 697.05 | 73.98 | 151.34 | 90.97 | 393.92 | 42.52 | -| C3 | m3 | | | 8.71 | 11.668 | 895.33 | 82.71 | 156.29 | 93.78 | 363.58 | 65.97 | -| C3s | m1 | | | 22.862 | 9.407 | 62.37 | 100.8 | 146.01 | 79.25 | 142.37 | 45.97 | -| C3s | m2 | | | 22.08 | 9.652 | 61.59 | 93.57 | 143.69 | 83.71 | 146.98 | 42.47 | -| C3w | m1 | | | 7.92 | 10.323 | 75.4 | 104.47 | 156.66 | 75.58 | 142.32 | 46.31 | -| C3w | m2 | | | 8.482 | 10.728 | 81.0 | 142.28 | 152.95 | 90.44 | 145.9 | 41.09 | -| C3w | m3 | | | 9.256 | 10.407 | 78.41 | 211.06 | 152.64 | 82.69 | 147.66 | 42.33 | -| C4 | m1 | | | 7.56 | 11.452 | 749.71 | 80.02 | 148.21 | 87.17 | 334.73 | 50.21 | -| C4 | m2 | | | 7.512 | 11.501 | 976.06 | 76.8 | 159.66 | 81.79 | 335.96 | 42.85 | -| C4 | m3 | | | 8.129 | 12.079 | 724.84 | 82.86 | 173.12 | 85.07 | 380.83 | 49.62 | - -## Warm latency (ms) - -| config | workload | conc | run | n | p50 | p95 | mean | CV | -|---|---|---|---|---|---|---|---|---| -| C1 | W1 | 1 | m1 | 300 | 136.16 | 152.83 | 139.3 | 13.3% | -| C1 | W1 | 1 | m2 | 300 | 134.18 | 174.19 | 140.56 | 13.5% | -| C1 | W1 | 1 | m3 | 300 | 143.71 | 163.08 | 147.56 | 8.8% | -| C1 | W2 | 1 | m1 | 300 | 214.92 | 256.75 | 222.03 | 12.4% | -| C1 | W2 | 1 | m2 | 300 | 211.7 | 277.15 | 222.45 | 17.3% | -| C1 | W2 | 1 | m3 | 300 | 238.15 | 364.68 | 261.25 | 36.7% | -| C1 | W3 | 1 | m1 | 300 | 423.07 | 489.32 | 429.02 | 10.0% | -| C1 | W3 | 1 | m2 | 300 | 381.17 | 443.43 | 393.26 | 11.1% | -| C1 | W3 | 1 | m3 | 300 | 427.47 | 661.61 | 469.78 | 33.3% | -| C1 | W4 | 1 | m1 | 300 | 206.81 | 230.14 | 209.84 | 7.8% | -| C1 | W4 | 1 | m2 | 300 | 197.91 | 218.13 | 201.89 | 9.6% | -| C1 | W4 | 1 | m3 | 300 | 225.78 | 273.31 | 243.74 | 42.7% | -| C1 | W5 | 1 | m1 | 300 | 399.96 | 623.75 | 432.71 | 29.0% | -| C1 | W5 | 1 | m2 | 300 | 289.94 | 319.87 | 293.91 | 5.2% | -| C1 | W5 | 1 | m3 | 300 | 316.47 | 362.75 | 323.07 | 6.8% | -| C1 | W6 | 1 | m1 | 300 | 112.68 | 120.81 | 113.63 | 5.4% | -| C1 | W6 | 1 | m2 | 300 | 108.29 | 128.27 | 111.36 | 9.5% | -| C1 | W6 | 1 | m3 | 300 | 122.7 | 163.07 | 130.51 | 24.1% | -| C2 | W1 | 1 | m1 | 300 | 45.4 | 47.88 | 45.7 | 4.6% | -| C2 | W1 | 1 | m2 | 300 | 44.15 | 46.5 | 44.51 | 4.3% | -| C2 | W2 | 1 | m1 | 300 | 69.01 | 74.36 | 69.4 | 3.2% | -| C2 | W2 | 1 | m2 | 300 | 68.26 | 73.05 | 68.72 | 3.0% | -| C2 | W3 | 1 | m1 | 300 | 150.23 | 168.83 | 152.71 | 5.5% | -| C2 | W3 | 1 | m2 | 300 | 144.64 | 155.04 | 146.47 | 5.3% | -| C2 | W4 | 1 | m1 | 300 | 71.87 | 77.93 | 73.6 | 25.9% | -| C2 | W4 | 1 | m2 | 300 | 71.14 | 78.44 | 73.13 | 23.2% | -| C2 | W5 | 1 | m1 | 300 | 144.57 | 153.01 | 145.58 | 5.4% | -| C2 | W5 | 1 | m2 | 300 | 139.31 | 146.83 | 140.19 | 3.9% | -| C2 | W6 | 1 | m1 | 300 | 37.48 | 40.59 | 37.88 | 3.8% | -| C2 | W6 | 1 | m2 | 300 | 37.53 | 41.02 | 37.67 | 5.3% | -| C3 | W1 | 1 | m1 | 300 | 46.43 | 51.36 | 47.18 | 6.7% | -| C3 | W1 | 1 | m2 | 300 | 45.16 | 47.45 | 45.4 | 2.9% | -| C3 | W1 | 1 | m3 | 300 | 46.21 | 51.04 | 47.02 | 8.1% | -| C3 | W1 | 4 | m1 | 300 | 183.78 | 193.71 | 183.85 | 5.6% | -| C3 | W1 | 4 | m2 | 300 | 179.8 | 184.82 | 179.29 | 5.6% | -| C3 | W1 | 4 | m3 | 300 | 182.29 | 195.7 | 182.58 | 6.5% | -| C3 | W2 | 1 | m1 | 300 | 68.15 | 69.76 | 68.29 | 1.4% | -| C3 | W2 | 1 | m2 | 300 | 69.29 | 73.9 | 69.89 | 3.3% | -| C3 | W2 | 1 | m3 | 300 | 72.34 | 79.75 | 73.32 | 6.4% | -| C3 | W2 | 4 | m1 | 300 | 270.14 | 274.17 | 269.09 | 5.4% | -| C3 | W2 | 4 | m2 | 300 | 277.11 | 288.8 | 276.67 | 5.6% | -| C3 | W2 | 4 | m3 | 300 | 285.48 | 446.69 | 314.42 | 25.5% | -| C3 | W3 | 1 | m1 | 300 | 147.18 | 164.15 | 149.63 | 4.6% | -| C3 | W3 | 1 | m2 | 300 | 156.25 | 180.33 | 158.78 | 7.3% | -| C3 | W3 | 1 | m3 | 300 | 155.7 | 171.08 | 158.16 | 5.0% | -| C3 | W3 | 4 | m1 | 300 | 576.78 | 646.04 | 582.0 | 6.9% | -| C3 | W3 | 4 | m2 | 300 | 604.59 | 636.61 | 603.6 | 6.6% | -| C3 | W3 | 4 | m3 | 300 | 606.93 | 696.18 | 618.08 | 7.9% | -| C3 | W4 | 1 | m1 | 300 | 76.26 | 103.39 | 82.23 | 38.3% | -| C3 | W4 | 1 | m2 | 300 | 72.09 | 76.13 | 72.49 | 3.8% | -| C3 | W4 | 1 | m3 | 300 | 71.6 | 76.35 | 72.3 | 3.4% | -| C3 | W4 | 4 | m1 | 300 | 283.54 | 289.94 | 282.96 | 5.2% | -| C3 | W4 | 4 | m2 | 300 | 290.06 | 299.55 | 288.61 | 5.7% | -| C3 | W4 | 4 | m3 | 300 | 295.12 | 341.81 | 299.56 | 8.5% | -| C3 | W5 | 1 | m1 | 300 | 138.54 | 144.85 | 139.84 | 4.6% | -| C3 | W5 | 1 | m2 | 300 | 140.16 | 145.2 | 140.5 | 2.5% | -| C3 | W5 | 1 | m3 | 300 | 141.0 | 146.11 | 142.02 | 6.4% | -| C3 | W5 | 4 | m1 | 300 | 549.68 | 556.64 | 547.68 | 5.4% | -| C3 | W5 | 4 | m2 | 300 | 563.58 | 588.67 | 564.85 | 5.9% | -| C3 | W5 | 4 | m3 | 300 | 564.72 | 605.95 | 565.53 | 5.9% | -| C3 | W6 | 1 | m1 | 300 | 41.71 | 46.22 | 42.23 | 9.1% | -| C3 | W6 | 1 | m2 | 300 | 36.99 | 42.08 | 37.79 | 5.5% | -| C3 | W6 | 1 | m3 | 300 | 38.85 | 44.38 | 39.89 | 13.9% | -| C3 | W6 | 4 | m1 | 300 | 162.26 | 181.65 | 163.49 | 7.7% | -| C3 | W6 | 4 | m2 | 300 | 147.05 | 154.68 | 146.98 | 5.9% | -| C3 | W6 | 4 | m3 | 300 | 150.12 | 168.94 | 150.83 | 7.5% | -| C3s | W1 | 1 | m1 | 300 | 32.99 | 35.83 | 33.38 | 4.8% | -| C3s | W1 | 1 | m2 | 300 | 32.55 | 33.89 | 32.68 | 2.2% | -| C3s | W1 | 4 | m1 | 300 | 129.91 | 135.95 | 130.38 | 6.2% | -| C3s | W1 | 4 | m2 | 300 | 128.91 | 132.43 | 129.01 | 6.0% | -| C3s | W2 | 1 | m1 | 300 | 63.03 | 83.07 | 66.36 | 19.3% | -| C3s | W2 | 1 | m2 | 300 | 60.73 | 64.65 | 61.12 | 2.9% | -| C3s | W2 | 4 | m1 | 300 | 259.71 | 306.33 | 266.76 | 10.1% | -| C3s | W2 | 4 | m2 | 300 | 238.37 | 248.95 | 238.59 | 5.8% | -| C3s | W3 | 1 | m1 | 300 | 134.16 | 146.03 | 135.84 | 4.2% | -| C3s | W3 | 1 | m2 | 300 | 138.18 | 150.78 | 139.85 | 4.4% | -| C3s | W3 | 4 | m1 | 300 | 544.89 | 583.88 | 547.89 | 6.6% | -| C3s | W3 | 4 | m2 | 300 | 543.35 | 564.9 | 545.68 | 6.8% | -| C3s | W4 | 1 | m1 | 300 | 72.6 | 88.29 | 75.21 | 8.7% | -| C3s | W4 | 1 | m2 | 300 | 72.83 | 77.13 | 73.2 | 3.3% | -| C3s | W4 | 4 | m1 | 300 | 295.51 | 377.02 | 308.8 | 12.5% | -| C3s | W4 | 4 | m2 | 300 | 288.58 | 299.58 | 287.73 | 5.7% | -| C3s | W5 | 1 | m1 | 300 | 138.74 | 143.5 | 139.51 | 4.5% | -| C3s | W5 | 1 | m2 | 300 | 140.03 | 144.33 | 140.38 | 2.1% | -| C3s | W5 | 4 | m1 | 300 | 555.12 | 568.42 | 553.25 | 5.5% | -| C3s | W5 | 4 | m2 | 300 | 560.2 | 571.87 | 558.3 | 5.5% | -| C3s | W6 | 1 | m1 | 300 | 26.04 | 27.02 | 26.15 | 2.0% | -| C3s | W6 | 1 | m2 | 300 | 26.43 | 30.05 | 27.01 | 9.4% | -| C3s | W6 | 4 | m1 | 300 | 103.97 | 118.23 | 105.48 | 8.7% | -| C3s | W6 | 4 | m2 | 300 | 102.84 | 106.2 | 102.83 | 5.7% | -| C3w | W1 | 1 | m1 | 300 | 46.82 | 52.47 | 47.43 | 4.8% | -| C3w | W1 | 1 | m2 | 300 | 45.76 | 48.19 | 45.99 | 2.4% | -| C3w | W1 | 1 | m3 | 300 | 45.53 | 47.85 | 45.74 | 2.7% | -| C3w | W1 | 4 | m1 | 300 | 185.89 | 220.62 | 189.55 | 8.9% | -| C3w | W1 | 4 | m2 | 300 | 181.79 | 187.96 | 181.7 | 5.7% | -| C3w | W1 | 4 | m3 | 300 | 180.96 | 184.56 | 180.41 | 5.4% | -| C3w | W2 | 1 | m1 | 300 | 68.45 | 71.76 | 68.78 | 2.2% | -| C3w | W2 | 1 | m2 | 300 | 72.4 | 79.95 | 72.95 | 5.7% | -| C3w | W2 | 1 | m3 | 300 | 68.69 | 70.34 | 68.74 | 1.3% | -| C3w | W2 | 4 | m1 | 300 | 274.76 | 310.3 | 278.37 | 7.5% | -| C3w | W2 | 4 | m2 | 300 | 283.5 | 311.64 | 284.33 | 6.9% | -| C3w | W2 | 4 | m3 | 300 | 271.99 | 275.56 | 270.72 | 5.5% | -| C3w | W3 | 1 | m1 | 300 | 144.67 | 150.49 | 145.15 | 1.9% | -| C3w | W3 | 1 | m2 | 300 | 158.77 | 185.47 | 160.35 | 7.9% | -| C3w | W3 | 1 | m3 | 300 | 146.14 | 154.61 | 147.05 | 2.8% | -| C3w | W3 | 4 | m1 | 300 | 587.08 | 621.88 | 589.07 | 5.8% | -| C3w | W3 | 4 | m2 | 300 | 630.84 | 692.64 | 630.62 | 7.2% | -| C3w | W3 | 4 | m3 | 300 | 617.72 | 658.0 | 619.19 | 6.2% | -| C3w | W4 | 1 | m1 | 300 | 70.34 | 72.85 | 70.55 | 1.7% | -| C3w | W4 | 1 | m2 | 300 | 74.18 | 79.15 | 74.51 | 3.6% | -| C3w | W4 | 1 | m3 | 300 | 70.9 | 72.72 | 71.02 | 1.6% | -| C3w | W4 | 4 | m1 | 300 | 281.56 | 297.73 | 282.03 | 6.2% | -| C3w | W4 | 4 | m2 | 300 | 292.1 | 302.51 | 290.43 | 5.8% | -| C3w | W4 | 4 | m3 | 300 | 282.83 | 287.0 | 281.54 | 5.4% | -| C3w | W5 | 1 | m1 | 300 | 138.91 | 144.43 | 139.59 | 2.9% | -| C3w | W5 | 1 | m2 | 300 | 139.93 | 143.23 | 140.23 | 2.4% | -| C3w | W5 | 1 | m3 | 300 | 138.91 | 141.94 | 139.23 | 2.1% | -| C3w | W5 | 4 | m1 | 300 | 555.48 | 573.79 | 555.16 | 5.9% | -| C3w | W5 | 4 | m2 | 300 | 559.9 | 575.41 | 557.86 | 5.5% | -| C3w | W5 | 4 | m3 | 300 | 555.77 | 562.55 | 553.06 | 5.5% | -| C3w | W6 | 1 | m1 | 300 | 40.15 | 43.83 | 40.52 | 5.2% | -| C3w | W6 | 1 | m2 | 300 | 37.95 | 42.76 | 38.56 | 5.3% | -| C3w | W6 | 1 | m3 | 300 | 38.91 | 43.52 | 39.21 | 5.1% | -| C3w | W6 | 4 | m1 | 300 | 159.07 | 174.04 | 160.42 | 7.0% | -| C3w | W6 | 4 | m2 | 300 | 148.98 | 160.82 | 149.96 | 7.0% | -| C3w | W6 | 4 | m3 | 300 | 156.96 | 173.82 | 157.85 | 7.2% | -| C4 | W1 | 1 | m1 | 300 | 51.34 | 53.4 | 51.54 | 3.9% | -| C4 | W1 | 1 | m2 | 300 | 46.31 | 51.61 | 47.32 | 13.3% | -| C4 | W1 | 1 | m3 | 300 | 47.42 | 49.16 | 47.47 | 2.4% | -| C4 | W1 | 4 | m1 | 300 | 201.31 | 204.91 | 200.77 | 5.5% | -| C4 | W1 | 4 | m2 | 300 | 182.06 | 187.56 | 181.81 | 6.2% | -| C4 | W1 | 4 | m3 | 300 | 187.8 | 192.08 | 186.87 | 5.6% | -| C4 | W2 | 1 | m1 | 300 | 87.24 | 93.59 | 88.08 | 3.8% | -| C4 | W2 | 1 | m2 | 300 | 83.78 | 90.77 | 84.15 | 5.0% | -| C4 | W2 | 1 | m3 | 300 | 71.73 | 74.42 | 71.65 | 2.2% | -| C4 | W2 | 4 | m1 | 300 | 347.52 | 354.59 | 345.44 | 5.6% | -| C4 | W2 | 4 | m2 | 300 | 293.51 | 315.78 | 292.63 | 6.6% | -| C4 | W2 | 4 | m3 | 300 | 286.32 | 313.29 | 289.29 | 8.1% | -| C4 | W3 | 1 | m1 | 300 | 153.71 | 191.37 | 158.66 | 9.3% | -| C4 | W3 | 1 | m2 | 300 | 190.42 | 203.53 | 186.58 | 7.2% | -| C4 | W3 | 1 | m3 | 300 | 168.01 | 251.87 | 184.08 | 32.3% | -| C4 | W3 | 4 | m1 | 300 | 771.69 | 962.56 | 795.76 | 11.8% | -| C4 | W3 | 4 | m2 | 300 | 759.95 | 795.4 | 749.15 | 6.8% | -| C4 | W3 | 4 | m3 | 300 | 632.37 | 697.57 | 636.88 | 7.5% | -| C4 | W4 | 1 | m1 | 300 | 90.42 | 96.35 | 90.93 | 3.3% | -| C4 | W4 | 1 | m2 | 300 | 72.31 | 76.18 | 72.74 | 3.0% | -| C4 | W4 | 1 | m3 | 300 | 74.47 | 77.65 | 74.55 | 2.6% | -| C4 | W4 | 4 | m1 | 300 | 360.37 | 373.47 | 361.38 | 8.2% | -| C4 | W4 | 4 | m2 | 300 | 290.97 | 299.16 | 289.11 | 5.8% | -| C4 | W4 | 4 | m3 | 300 | 295.8 | 301.27 | 294.11 | 5.6% | -| C4 | W5 | 1 | m1 | 300 | 138.98 | 145.95 | 139.77 | 3.0% | -| C4 | W5 | 1 | m2 | 300 | 140.01 | 145.13 | 140.46 | 2.6% | -| C4 | W5 | 1 | m3 | 300 | 150.35 | 166.48 | 151.86 | 5.5% | -| C4 | W5 | 4 | m1 | 300 | 559.72 | 572.81 | 557.28 | 5.7% | -| C4 | W5 | 4 | m2 | 300 | 562.18 | 574.27 | 559.79 | 5.7% | -| C4 | W5 | 4 | m3 | 300 | 606.57 | 622.48 | 604.12 | 5.7% | -| C4 | W6 | 1 | m1 | 300 | 41.73 | 43.16 | 41.86 | 2.2% | -| C4 | W6 | 1 | m2 | 300 | 38.31 | 42.38 | 38.86 | 4.7% | -| C4 | W6 | 1 | m3 | 300 | 39.21 | 43.03 | 39.48 | 3.8% | -| C4 | W6 | 4 | m1 | 300 | 166.55 | 169.36 | 166.38 | 5.9% | -| C4 | W6 | 4 | m2 | 300 | 147.78 | 159.22 | 148.24 | 5.7% | -| C4 | W6 | 4 | m3 | 300 | 154.91 | 169.31 | 156.36 | 6.5% | - -## Run-to-run gate (p50 spread across measured runs <= 10%) - -| config | workload | conc | runs | spread | gate | -|---|---|---|---|---|---| -| C1 | W1 | 1 | 3 | 7.1% | PASS | -| C1 | W2 | 1 | 3 | 12.5% | FAIL | -| C1 | W3 | 1 | 3 | 12.1% | FAIL | -| C1 | W4 | 1 | 3 | 14.1% | FAIL | -| C1 | W5 | 1 | 3 | 37.9% | FAIL | -| C1 | W6 | 1 | 3 | 13.3% | FAIL | -| C2 | W1 | 1 | 2 | 2.8% | PASS | -| C2 | W2 | 1 | 2 | 1.1% | PASS | -| C2 | W3 | 1 | 2 | 3.9% | PASS | -| C2 | W4 | 1 | 2 | 1.0% | PASS | -| C2 | W5 | 1 | 2 | 3.8% | PASS | -| C2 | W6 | 1 | 2 | 0.1% | PASS | -| C3 | W1 | 1 | 3 | 2.8% | PASS | -| C3 | W1 | 4 | 3 | 2.2% | PASS | -| C3 | W2 | 1 | 3 | 6.2% | PASS | -| C3 | W2 | 4 | 3 | 5.7% | PASS | -| C3 | W3 | 1 | 3 | 6.2% | PASS | -| C3 | W3 | 4 | 3 | 5.2% | PASS | -| C3 | W4 | 1 | 3 | 6.5% | PASS | -| C3 | W4 | 4 | 3 | 4.1% | PASS | -| C3 | W5 | 1 | 3 | 1.8% | PASS | -| C3 | W5 | 4 | 3 | 2.7% | PASS | -| C3 | W6 | 1 | 3 | 12.7% | FAIL | -| C3 | W6 | 4 | 3 | 10.3% | FAIL | -| C3s | W1 | 1 | 2 | 1.4% | PASS | -| C3s | W1 | 4 | 2 | 0.8% | PASS | -| C3s | W2 | 1 | 2 | 3.8% | PASS | -| C3s | W2 | 4 | 2 | 9.0% | PASS | -| C3s | W3 | 1 | 2 | 3.0% | PASS | -| C3s | W3 | 4 | 2 | 0.3% | PASS | -| C3s | W4 | 1 | 2 | 0.3% | PASS | -| C3s | W4 | 4 | 2 | 2.4% | PASS | -| C3s | W5 | 1 | 2 | 0.9% | PASS | -| C3s | W5 | 4 | 2 | 0.9% | PASS | -| C3s | W6 | 1 | 2 | 1.5% | PASS | -| C3s | W6 | 4 | 2 | 1.1% | PASS | -| C3w | W1 | 1 | 3 | 2.8% | PASS | -| C3w | W1 | 4 | 3 | 2.7% | PASS | -| C3w | W2 | 1 | 3 | 5.8% | PASS | -| C3w | W2 | 4 | 3 | 4.2% | PASS | -| C3w | W3 | 1 | 3 | 9.7% | PASS | -| C3w | W3 | 4 | 3 | 7.5% | PASS | -| C3w | W4 | 1 | 3 | 5.5% | PASS | -| C3w | W4 | 4 | 3 | 3.7% | PASS | -| C3w | W5 | 1 | 3 | 0.7% | PASS | -| C3w | W5 | 4 | 3 | 0.8% | PASS | -| C3w | W6 | 1 | 3 | 5.8% | PASS | -| C3w | W6 | 4 | 3 | 6.8% | PASS | -| C4 | W1 | 1 | 3 | 10.9% | FAIL | -| C4 | W1 | 4 | 3 | 10.6% | FAIL | -| C4 | W2 | 1 | 3 | 21.6% | FAIL | -| C4 | W2 | 4 | 3 | 21.4% | FAIL | -| C4 | W3 | 1 | 3 | 23.9% | FAIL | -| C4 | W3 | 4 | 3 | 22.0% | FAIL | -| C4 | W4 | 1 | 3 | 25.0% | FAIL | -| C4 | W4 | 4 | 3 | 23.9% | FAIL | -| C4 | W5 | 1 | 3 | 8.2% | PASS | -| C4 | W5 | 4 | 3 | 8.4% | PASS | -| C4 | W6 | 1 | 3 | 8.9% | PASS | -| C4 | W6 | 4 | 3 | 12.7% | FAIL | - -## Throughput - -| config | workload | conc | run | n | errors | elapsed s | successful req/s | -|---|---|---|---|---|---|---|---| -| C3 | W1 | 1 | m1 | 300 | 0 | 14.154 | 21.2 | -| C3 | W1 | 1 | m2 | 300 | 0 | 13.621 | 22.03 | -| C3 | W1 | 1 | m3 | 300 | 0 | 14.107 | 21.27 | -| C3 | W1 | 4 | m1 | 300 | 0 | 13.857 | 21.65 | -| C3 | W1 | 4 | m2 | 300 | 0 | 13.513 | 22.2 | -| C3 | W1 | 4 | m3 | 300 | 0 | 13.765 | 21.79 | -| C3 | W2 | 1 | m1 | 300 | 0 | 20.488 | 14.64 | -| C3 | W2 | 1 | m2 | 300 | 0 | 20.969 | 14.31 | -| C3 | W2 | 1 | m3 | 300 | 0 | 21.998 | 13.64 | -| C3 | W2 | 4 | m1 | 300 | 0 | 20.286 | 14.79 | -| C3 | W2 | 4 | m2 | 300 | 0 | 20.857 | 14.38 | -| C3 | W2 | 4 | m3 | 300 | 0 | 23.684 | 12.67 | -| C3 | W3 | 1 | m1 | 300 | 0 | 44.889 | 6.68 | -| C3 | W3 | 1 | m2 | 300 | 0 | 47.636 | 6.3 | -| C3 | W3 | 1 | m3 | 300 | 0 | 47.45 | 6.32 | -| C3 | W3 | 4 | m1 | 300 | 0 | 43.916 | 6.83 | -| C3 | W3 | 4 | m2 | 300 | 0 | 45.497 | 6.59 | -| C3 | W3 | 4 | m3 | 300 | 0 | 46.581 | 6.44 | -| C3 | W4 | 1 | m1 | 300 | 0 | 24.672 | 12.16 | -| C3 | W4 | 1 | m2 | 300 | 0 | 21.748 | 13.79 | -| C3 | W4 | 1 | m3 | 300 | 0 | 21.691 | 13.83 | -| C3 | W4 | 4 | m1 | 300 | 0 | 21.328 | 14.07 | -| C3 | W4 | 4 | m2 | 300 | 0 | 21.756 | 13.79 | -| C3 | W4 | 4 | m3 | 300 | 0 | 22.575 | 13.29 | -| C3 | W5 | 1 | m1 | 300 | 0 | 41.952 | 7.15 | -| C3 | W5 | 1 | m2 | 300 | 0 | 42.153 | 7.12 | -| C3 | W5 | 1 | m3 | 300 | 0 | 42.608 | 7.04 | -| C3 | W5 | 4 | m1 | 300 | 0 | 41.283 | 7.27 | -| C3 | W5 | 4 | m2 | 300 | 0 | 42.581 | 7.05 | -| C3 | W5 | 4 | m3 | 300 | 0 | 42.643 | 7.04 | -| C3 | W6 | 1 | m1 | 300 | 0 | 12.669 | 23.68 | -| C3 | W6 | 1 | m2 | 300 | 0 | 11.336 | 26.46 | -| C3 | W6 | 1 | m3 | 300 | 0 | 11.969 | 25.07 | -| C3 | W6 | 4 | m1 | 300 | 0 | 12.321 | 24.35 | -| C3 | W6 | 4 | m2 | 300 | 0 | 11.077 | 27.08 | -| C3 | W6 | 4 | m3 | 300 | 0 | 11.366 | 26.39 | -| C3s | W1 | 1 | m1 | 300 | 0 | 10.014 | 29.96 | -| C3s | W1 | 1 | m2 | 300 | 0 | 9.805 | 30.6 | -| C3s | W1 | 4 | m1 | 300 | 0 | 9.827 | 30.53 | -| C3s | W1 | 4 | m2 | 300 | 0 | 9.724 | 30.85 | -| C3s | W2 | 1 | m1 | 300 | 0 | 19.908 | 15.07 | -| C3s | W2 | 1 | m2 | 300 | 0 | 18.337 | 16.36 | -| C3s | W2 | 4 | m1 | 300 | 0 | 20.102 | 14.92 | -| C3s | W2 | 4 | m2 | 300 | 0 | 17.989 | 16.68 | -| C3s | W3 | 1 | m1 | 300 | 0 | 40.752 | 7.36 | -| C3s | W3 | 1 | m2 | 300 | 0 | 41.956 | 7.15 | -| C3s | W3 | 4 | m1 | 300 | 0 | 41.293 | 7.27 | -| C3s | W3 | 4 | m2 | 300 | 0 | 41.135 | 7.29 | -| C3s | W4 | 1 | m1 | 300 | 0 | 22.564 | 13.3 | -| C3s | W4 | 1 | m2 | 300 | 0 | 21.96 | 13.66 | -| C3s | W4 | 4 | m1 | 300 | 0 | 23.276 | 12.89 | -| C3s | W4 | 4 | m2 | 300 | 0 | 21.691 | 13.83 | -| C3s | W5 | 1 | m1 | 300 | 0 | 41.854 | 7.17 | -| C3s | W5 | 1 | m2 | 300 | 0 | 42.115 | 7.12 | -| C3s | W5 | 4 | m1 | 300 | 0 | 41.699 | 7.19 | -| C3s | W5 | 4 | m2 | 300 | 0 | 42.082 | 7.13 | -| C3s | W6 | 1 | m1 | 300 | 0 | 7.844 | 38.24 | -| C3s | W6 | 1 | m2 | 300 | 0 | 8.104 | 37.02 | -| C3s | W6 | 4 | m1 | 300 | 0 | 7.951 | 37.73 | -| C3s | W6 | 4 | m2 | 300 | 0 | 7.75 | 38.71 | -| C3w | W1 | 1 | m1 | 300 | 0 | 14.23 | 21.08 | -| C3w | W1 | 1 | m2 | 300 | 0 | 13.797 | 21.74 | -| C3w | W1 | 1 | m3 | 300 | 0 | 13.723 | 21.86 | -| C3w | W1 | 4 | m1 | 300 | 0 | 14.283 | 21.0 | -| C3w | W1 | 4 | m2 | 300 | 0 | 13.695 | 21.91 | -| C3w | W1 | 4 | m3 | 300 | 0 | 13.598 | 22.06 | -| C3w | W2 | 1 | m1 | 300 | 0 | 20.634 | 14.54 | -| C3w | W2 | 1 | m2 | 300 | 0 | 21.885 | 13.71 | -| C3w | W2 | 1 | m3 | 300 | 0 | 20.623 | 14.55 | -| C3w | W2 | 4 | m1 | 300 | 0 | 20.979 | 14.3 | -| C3w | W2 | 4 | m2 | 300 | 0 | 21.438 | 13.99 | -| C3w | W2 | 4 | m3 | 300 | 0 | 20.407 | 14.7 | -| C3w | W3 | 1 | m1 | 300 | 0 | 43.546 | 6.89 | -| C3w | W3 | 1 | m2 | 300 | 0 | 48.108 | 6.24 | -| C3w | W3 | 1 | m3 | 300 | 0 | 44.116 | 6.8 | -| C3w | W3 | 4 | m1 | 300 | 0 | 44.403 | 6.76 | -| C3w | W3 | 4 | m2 | 300 | 0 | 47.512 | 6.31 | -| C3w | W3 | 4 | m3 | 300 | 0 | 46.673 | 6.43 | -| C3w | W4 | 1 | m1 | 300 | 0 | 21.166 | 14.17 | -| C3w | W4 | 1 | m2 | 300 | 0 | 22.353 | 13.42 | -| C3w | W4 | 1 | m3 | 300 | 0 | 21.306 | 14.08 | -| C3w | W4 | 4 | m1 | 300 | 0 | 21.26 | 14.11 | -| C3w | W4 | 4 | m2 | 300 | 0 | 21.892 | 13.7 | -| C3w | W4 | 4 | m3 | 300 | 0 | 21.221 | 14.14 | -| C3w | W5 | 1 | m1 | 300 | 0 | 41.88 | 7.16 | -| C3w | W5 | 1 | m2 | 300 | 0 | 42.071 | 7.13 | -| C3w | W5 | 1 | m3 | 300 | 0 | 41.77 | 7.18 | -| C3w | W5 | 4 | m1 | 300 | 0 | 41.844 | 7.17 | -| C3w | W5 | 4 | m2 | 300 | 0 | 42.047 | 7.13 | -| C3w | W5 | 4 | m3 | 300 | 0 | 41.688 | 7.2 | -| C3w | W6 | 1 | m1 | 300 | 0 | 12.158 | 24.68 | -| C3w | W6 | 1 | m2 | 300 | 0 | 11.569 | 25.93 | -| C3w | W6 | 1 | m3 | 300 | 0 | 11.765 | 25.5 | -| C3w | W6 | 4 | m1 | 300 | 0 | 12.095 | 24.8 | -| C3w | W6 | 4 | m2 | 300 | 0 | 11.303 | 26.54 | -| C3w | W6 | 4 | m3 | 300 | 0 | 11.899 | 25.21 | -| C4 | W1 | 1 | m1 | 300 | 0 | 15.462 | 19.4 | -| C4 | W1 | 1 | m2 | 300 | 0 | 14.197 | 21.13 | -| C4 | W1 | 1 | m3 | 300 | 0 | 14.242 | 21.06 | -| C4 | W1 | 4 | m1 | 300 | 0 | 15.132 | 19.83 | -| C4 | W1 | 4 | m2 | 300 | 0 | 13.707 | 21.89 | -| C4 | W1 | 4 | m3 | 300 | 0 | 14.085 | 21.3 | -| C4 | W2 | 1 | m1 | 300 | 0 | 26.424 | 11.35 | -| C4 | W2 | 1 | m2 | 300 | 0 | 25.247 | 11.88 | -| C4 | W2 | 1 | m3 | 300 | 0 | 21.495 | 13.96 | -| C4 | W2 | 4 | m1 | 300 | 0 | 26.038 | 11.52 | -| C4 | W2 | 4 | m2 | 300 | 0 | 22.054 | 13.6 | -| C4 | W2 | 4 | m3 | 300 | 0 | 21.819 | 13.75 | -| C4 | W3 | 1 | m1 | 300 | 0 | 47.598 | 6.3 | -| C4 | W3 | 1 | m2 | 300 | 0 | 55.974 | 5.36 | -| C4 | W3 | 1 | m3 | 300 | 0 | 55.228 | 5.43 | -| C4 | W3 | 4 | m1 | 300 | 0 | 59.984 | 5.0 | -| C4 | W3 | 4 | m2 | 300 | 0 | 56.483 | 5.31 | -| C4 | W3 | 4 | m3 | 300 | 0 | 48.002 | 6.25 | -| C4 | W4 | 1 | m1 | 300 | 0 | 27.28 | 11.0 | -| C4 | W4 | 1 | m2 | 300 | 0 | 21.849 | 13.73 | -| C4 | W4 | 1 | m3 | 300 | 0 | 22.367 | 13.41 | -| C4 | W4 | 4 | m1 | 300 | 0 | 27.241 | 11.01 | -| C4 | W4 | 4 | m2 | 300 | 0 | 21.794 | 13.77 | -| C4 | W4 | 4 | m3 | 300 | 0 | 22.164 | 13.54 | -| C4 | W5 | 1 | m1 | 300 | 0 | 41.931 | 7.15 | -| C4 | W5 | 1 | m2 | 300 | 0 | 42.141 | 7.12 | -| C4 | W5 | 1 | m3 | 300 | 0 | 45.56 | 6.58 | -| C4 | W5 | 4 | m1 | 300 | 0 | 42.006 | 7.14 | -| C4 | W5 | 4 | m2 | 300 | 0 | 42.196 | 7.11 | -| C4 | W5 | 4 | m3 | 300 | 0 | 45.537 | 6.59 | -| C4 | W6 | 1 | m1 | 300 | 0 | 12.559 | 23.89 | -| C4 | W6 | 1 | m2 | 300 | 0 | 11.658 | 25.73 | -| C4 | W6 | 1 | m3 | 300 | 0 | 11.846 | 25.33 | -| C4 | W6 | 4 | m1 | 300 | 0 | 12.541 | 23.92 | -| C4 | W6 | 4 | m2 | 300 | 0 | 11.173 | 26.85 | -| C4 | W6 | 4 | m3 | 300 | 0 | 11.785 | 25.46 | - -## Memory (MB) - -| config | run | footprint after warmup | footprint end | footprint peak | MPS driver after warmup | MPS driver end | frontend footprint | -|---|---|---|---|---|---|---|---| -| C1 | m1 | 2030 | 2014 | 2030 | | | | -| C1 | m2 | 2025 | 2003 | 2029 | | | | -| C1 | m3 | 2022 | 2013 | 2023 | | | | -| C2 | m1 | 4169 | 3490 | 4174 | 3085 | 3102 | | -| C2 | m2 | 4170 | 3489 | 4174 | 3085 | 3102 | | -| C3 | m1 | 4184 | 3513 | 4188 | | | | -| C3 | m2 | 4182 | 3497 | 4187 | | | | -| C3 | m3 | 4181 | 3513 | 4185 | | | | -| C3s | m1 | 3927 | 3664 | 4239 | | | | -| C3s | m2 | 4086 | 3728 | 4431 | | | | -| C3w | m1 | 4198 | 3519 | 4203 | | | | -| C3w | m2 | 4193 | 3519 | 4198 | | | | -| C3w | m3 | 4199 | 3519 | 4204 | | | | -| C4 | m1 | 4184 | 3497 | 4189 | | | 3 | -| C4 | m2 | 4181 | 3513 | 4185 | | | 3 | -| C4 | m3 | 4185 | 3512 | 4189 | | | 3 | diff --git a/recipe/laya/bench/results/paired-all-optimizations.md b/recipe/laya/bench/results/paired-all-optimizations.md deleted file mode 100644 index 44cb564..0000000 --- a/recipe/laya/bench/results/paired-all-optimizations.md +++ /dev/null @@ -1,30 +0,0 @@ -Paired comparison of the worker with every optimization (B: LAYA_WORKER_COMPILE=on, LAYA_WORKER_WEIGHTS=fp16) against the worker with none (A: compile off, fp32 weights), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. - -## e5a: A = `LAYA_WORKER_COMPILE=off`, B = `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, load at start 10.84 - -| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | -|---|---|---|---|---|---| -| W1 | 300 | 55.3 | 34.7 | 0.626 | 0.620–0.633 | -| W2 | 300 | 79.6 | 63.4 | 0.798 | 0.792–0.803 | -| W3 | 300 | 184.5 | 153.2 | 0.833 | 0.826–0.842 | -| W4 | 300 | 95.3 | 82.0 | 0.859 | 0.853–0.868 | -| W5 | 300 | 181.4 | 148.7 | 0.818 | 0.808–0.825 | -| W6 | 300 | 44.8 | 27.6 | 0.616 | 0.608–0.625 | - -B vs A answers: max |Δp| 0.0031, flips [], errors [] -recompiled after ready: {'A': None, 'B': False}; footprint MB: {'A': 3501, 'B': 2849} - -## e5b: A = `LAYA_WORKER_COMPILE=off`, B = `LAYA_WORKER_COMPILE=on LAYA_WORKER_WEIGHTS=fp16`, load at start 15.12 - -| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | -|---|---|---|---|---|---| -| W1 | 300 | 59.3 | 35.9 | 0.616 | 0.609–0.628 | -| W2 | 300 | 137.0 | 108.2 | 0.795 | 0.778–0.814 | -| W3 | 300 | 199.1 | 166.7 | 0.832 | 0.825–0.838 | -| W4 | 300 | 101.2 | 87.1 | 0.863 | 0.857–0.868 | -| W5 | 300 | 187.8 | 156.2 | 0.824 | 0.818–0.834 | -| W6 | 300 | 64.7 | 36.8 | 0.584 | 0.569–0.600 | - -B vs A answers: max |Δp| 0.0031, flips [], errors [] -recompiled after ready: {'A': None, 'B': False}; footprint MB: {'A': 3533, 'B': 2783} - diff --git a/recipe/laya/bench/results/paired-fp16.md b/recipe/laya/bench/results/paired-fp16.md deleted file mode 100644 index f748c29..0000000 --- a/recipe/laya/bench/results/paired-fp16.md +++ /dev/null @@ -1,30 +0,0 @@ -Paired comparison of fp32 (A) and fp16 (B) weights with the one-question path compiled (then `LAYA_WORKER_COMPILE=single`, now the one-question part of `on`), both alive at once; built with `paired.py --summarize`. M1 Pro, AC power, checkpoint 55cf4c4. - -## e4a: A = `LAYA_WORKER_COMPILE=single`, B = `LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16`, load at start 31.32 - -| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | -|---|---|---|---|---|---| -| W1 | 300 | 33.2 | 28.5 | 0.857 | 0.855–0.861 | -| W2 | 300 | 62.8 | 56.9 | 0.910 | 0.907–0.913 | -| W3 | 300 | 152.8 | 137.0 | 0.900 | 0.897–0.905 | -| W4 | 300 | 71.6 | 74.6 | 1.042 | 1.040–1.045 | -| W5 | 300 | 164.0 | 145.9 | 0.894 | 0.891–0.899 | -| W6 | 300 | 26.5 | 22.7 | 0.855 | 0.853–0.859 | - -B vs A answers: max |Δp| 0.0031, flips [], errors [] -recompiled after ready: {'A': False, 'B': False}; footprint MB: {'A': 3647, 'B': 2717} - -## e4b: A = `LAYA_WORKER_COMPILE=single`, B = `LAYA_WORKER_COMPILE=single LAYA_WORKER_WEIGHTS=fp16`, load at start 18.75 - -| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval | -|---|---|---|---|---|---| -| W1 | 300 | 34.9 | 29.7 | 0.855 | 0.851–0.861 | -| W2 | 300 | 68.3 | 61.8 | 0.910 | 0.908–0.913 | -| W3 | 300 | 158.8 | 143.3 | 0.902 | 0.897–0.908 | -| W4 | 300 | 90.0 | 93.6 | 1.040 | 1.032–1.047 | -| W5 | 300 | 161.7 | 143.2 | 0.891 | 0.889–0.896 | -| W6 | 300 | 32.7 | 28.0 | 0.850 | 0.839–0.861 | - -B vs A answers: max |Δp| 0.0031, flips [], errors [] -recompiled after ready: {'A': False, 'B': False}; footprint MB: {'A': 3646, 'B': 2721} - diff --git a/recipe/laya/bench/workloads.src.py b/recipe/laya/bench/workloads.src.py deleted file mode 100644 index 92f5981..0000000 --- a/recipe/laya/bench/workloads.src.py +++ /dev/null @@ -1,90 +0,0 @@ -"""Writes workloads.jsonl. Edit here, not the JSONL: the text is fixed so runs stay comparable.""" - -import json -from pathlib import Path - -ROUTE = { - "type": "choice", - "instructions": "Route the ticket to the queue that owns it.", - "criteria": { - "logistics": "Shipping, delivery and tracking", - "payment": "Charges, invoices and refunds", - "returns": "Returns and exchanges", - "account": "Login, password and profile", - "human": "Anything else", - }, -} -URGENCY = { - "type": "score", - "instructions": "How urgent is the ticket?", - "criteria": ["Not urgent", "Needs attention soon", "Needs attention immediately"], -} -REFUND = {"type": "noul", "instructions": "Does the customer ask for a refund?"} -ANGRY = {"type": "noul", "instructions": "Is the customer angry?"} -CANCEL = {"type": "noul", "instructions": "Does the customer want to cancel the order?"} -LANG = { - "type": "choice", - "instructions": "Which language is the ticket written in?", - "criteria": {"en": "English", "de": "German", "fr": "French"}, -} - -SHORT = "My package never arrived and tracking has not updated in ten days." -LONG = ( - "Hello, I ordered a pair of running shoes three weeks ago and paid for express delivery. " - "The confirmation email said the parcel would arrive within two business days, but the tracking " - "page has shown 'label created' ever since. I contacted the courier and they told me they never " - "received the parcel from your warehouse. In the meantime I was charged twice on my credit card, " - "once for the original amount and once for a slightly different amount that I do not recognise. " - "I need the shoes for a race next weekend, so please either ship them today with a tracking number " - "that actually works or cancel the order and refund both charges. I have been a customer for years " - "and this is the first time something like this has happened." -) -NEAR = " ".join([LONG] * 3) - - -def options(n): - names = [ - "logistics", - "payment", - "returns", - "account", - "human", - "billing", - "technical", - "sales", - "legal", - "security", - ] - return { - "type": "choice", - "instructions": ROUTE["instructions"], - "criteria": {k: f"The {k} team" for k in names[:n]}, - } - - -W = [ - # id, kind, state, questions, target tokens per row, measured with check_workloads.py - ("W1", "bench", SHORT, {"route": ROUTE}, 68), - ("W2", "bench", LONG, {"route": ROUTE}, 200), - ("W3", "bench", NEAR, {"route": ROUTE}, 480), - ("W4", "bench", SHORT, {"route": ROUTE, "urgency": URGENCY, "refund": REFUND}, 54), - ( - "W5", - "bench", - SHORT, - {"route": ROUTE, "urgency": URGENCY, "refund": REFUND, "angry": ANGRY, "cancel": CANCEL, "lang": LANG}, - 49, - ), - ("W6", "bench", SHORT, {"refund": REFUND}, 47), - ("P2", "parity", SHORT, {"route": options(2)}, None), - ("P5", "parity", SHORT, {"route": options(5)}, None), - ("P10", "parity", SHORT, {"route": options(10)}, None), - ("P2L", "parity", LONG, {"route": options(2), "urgency": URGENCY, "refund": REFUND}, None), - ("P10L", "parity", LONG, {"route": options(10), "urgency": URGENCY, "refund": REFUND}, None), -] - -with open(Path(__file__).resolve().parent / "workloads.jsonl", "w") as f: - f.writelines( - json.dumps({"id": wid, "kind": kind, "target_tokens_per_row": target, "state": state, "questions": qs}) + "\n" - for wid, kind, state, qs, target in W - ) diff --git a/recipe/laya/requirements-mps.txt b/recipe/laya/requirements-mps.txt new file mode 100644 index 0000000..dca8e7a --- /dev/null +++ b/recipe/laya/requirements-mps.txt @@ -0,0 +1,10 @@ +# The environment the Apple Silicon recipe was validated in (M1 Pro and M5, Python 3.12). +# laya[serve] pulls in torch, transformers, fastapi and uvicorn; the pins below are the versions it resolved to. +laya[serve]==0.3.20 +torch==2.14.0 +transformers==5.17.0 +tokenizers==0.23.2 +safetensors==0.8.0 +huggingface-hub==1.33.0 +fastapi==0.141.1 +uvicorn==0.54.0 diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py new file mode 100644 index 0000000..827b6d0 --- /dev/null +++ b/src/frontend/laya_mps.py @@ -0,0 +1,139 @@ +"""HTTP worker for Laya on Apple Silicon (PyTorch MPS) and CPU: `GET /health` and `POST /v1/systemone`. + +PYTHONPATH=src python -m frontend.laya_mps --device mps --model english [--compile] [--weights fp16] + +It is laya-serve (`laya[serve]==0.3.20`) with its request handling unchanged and three changes: + +- It binds only after every loaded model has run a warmup over short, long and multi-question requests, + so a reachable worker is a warm one. laya-serve answers /health before any forward pass. +- /health describes the loaded models as they are now: device, weight and autocast dtypes, checkpoint + and the revision the weights were loaded from, and `device_mismatch` when a model is not on the + requested device. laya-serve reports the configured device, and laya moves a model to the CPU on a + GPU out-of-memory error and keeps serving. `--require-device` exits at startup instead of serving + from another device. +- `--compile` and `--weights fp16` make the GPU path faster (models/laya/optimize.py). Both apply on + the GPU only; on the CPU, including after a fallback, the worker runs laya's fp32 model uncompiled. + +laya-serve's own environment variables still apply, notably LAYA_API_KEY for bearer authentication. +""" + +from __future__ import annotations + +import argparse +import logging +import sys +from typing import Any + +from models.laya import engine, optimize + +log = logging.getLogger("laya-worker") + + +def build_app( + router: Any, + model: str, + requested: str | None, + *, + require_device: bool = False, + compile: bool = False, + fp16: bool = False, + graph_counter=optimize.compiled_graphs, + revisions: dict[str, str] | None = None, +): + """Apply the options, warm up every loaded model, then return laya's app with /health replaced. + `model` is the one summarised at the top of /health, and the one loaded if nothing is preloaded. + Raises if an option or the warmup fails, so the caller never binds a worker that cannot answer.""" + from laya.serve import create_app + + names = list(router.loaded) or [model] # never load a model the worker was not asked to serve + if fp16 or compile: + for name in names: + if not optimize.apply(router.load(name), fp16=fp16, compile=compile): + log.warning("%s is on the CPU: --compile and --weights fp16 apply on the GPU only", name) + warmed = {name: engine.warmup(router, name) for name in names} + agents = {name: router.load(name) for name in names} + primary = model if model in agents else names[0] + + def current() -> dict[str, Any]: + """The agents as they are now, not as they were at startup (see the module docstring).""" + models = { + name: { + **engine.describe(agent, requested, warmed[name]["routing"], revisions), + "warmup_ms": warmed[name]["warmup_ms"], + } + for name, agent in agents.items() + } + return { + **models[primary], + "device_mismatch": any(m["device_mismatch"] for m in models.values()), + "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), + "models": models, + } + + info = current() + graphs_at_ready = graph_counter() if compile else None + if info["device_mismatch"]: + wrong = ", ".join(f"{n} is on {m['device']}" for n, m in info["models"].items() if m["device_mismatch"]) + message = f"asked for {info['requested_device']}, {wrong}" + if require_device: + raise RuntimeError(message) + log.warning(message) + + app = create_app(router) + app.router.routes[:] = [r for r in app.router.routes if getattr(r, "path", None) != "/health"] + + @app.get("/health") + def health() -> dict[str, Any]: + compiled: dict[str, Any] = {"enabled": compile} + if compile: + now = graph_counter() + compiled.update( + graphs_at_ready=graphs_at_ready, graphs_now=now, recompiled_after_ready=now > graphs_at_ready + ) + return {"status": "ok", "ready": True, "loaded": router.loaded, **current(), "compile": compiled} + + return app + + +def make_router(device: str | None, model: str) -> Any: + """laya's Router with one checkpoint preloaded.""" + from laya.router import Router + + router = Router(device=device) + router.preload([model]) + return router + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--device", default=None, help="torch device for laya: mps, cpu (default: laya's choice)") + parser.add_argument("--model", default="english", help="laya checkpoint to serve: english, multilingual, ...") + parser.add_argument("--compile", action="store_true", help="torch.compile the GPU path during warmup") + parser.add_argument("--weights", default="fp32", choices=["fp32", "fp16"], help="weight precision on the GPU") + parser.add_argument("--require-device", action="store_true", help="exit if a model is not on --device") + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--port", type=int, default=8000) + parser.add_argument("--log-level", default="info") + args = parser.parse_args() + + import uvicorn + + logging.basicConfig(level=logging.INFO, format="%(name)s: %(message)s") + revisions = engine.record_snapshot_revisions() + try: + app = build_app( + make_router(args.device, args.model), + args.model, + args.device, + require_device=args.require_device, + compile=args.compile, + fp16=args.weights == "fp16", + revisions=revisions, + ) + except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever + sys.exit(f"laya-worker: not starting: {exc}") + uvicorn.run(app, host=args.host, port=args.port, log_level=args.log_level) + + +if __name__ == "__main__": + main() diff --git a/src/models/laya/README.md b/src/models/laya/README.md index 8710117..6170aba 100644 --- a/src/models/laya/README.md +++ b/src/models/laya/README.md @@ -7,35 +7,30 @@ GPU operations and kernel implementations belong in [`backends/cuda/`](../../bac Status: the Python worker below serves LAYA through laya-serve on CPU and Apple Silicon (PyTorch MPS, validated on an M1 Pro and, by another contributor, an M5). No native CUDA or Metal backend yet. -## Worker - -`worker.py` runs laya-serve (`laya[serve]==0.3.20`) with its request handling unchanged and adds: - -- **Warmup before readiness.** It binds only after every loaded model has run short, long and - multi-question requests, so `/health` never answers for a worker that has not run a forward pass. - On an M1 Pro (MPS) the first request after ready took 70–81 ms, against 0.7–1.1 s from laya-serve. -- **A `/health` that describes the loaded models.** For each loaded model under `models`, and for - `LAYA_WORKER_MODEL` at the top level: the device, weight and autocast dtypes, the checkpoint and the - revision its weights were downloaded from, and `device_mismatch` when a model is not on the device - `LAYA_DEVICE` asked for. laya-serve reports `LAYA_DEVICE` as configured. These are read on every call: - Laya moves a model to the CPU on a GPU out-of-memory error and keeps serving, and `/health` shows it. - `LAYA_REQUIRE_DEVICE=1` makes the worker exit at startup if a model is not on the requested device. -- **Two options that make the GPU path faster** (`optimize.py`): - - `LAYA_WORKER_COMPILE=on` compiles one-question requests end to end and, for several questions, only - the encoder (Laya's decision head is slower compiled on MPS). `/health` reports compiled graphs at - readiness and now. - - `LAYA_WORKER_WEIGHTS=fp16` keeps the checkpoint's fp16 weights instead of Laya's fp32 upcast - (except `act_head`, which Laya feeds fp32 features). - - Both apply on the GPU only. On the CPU, including after a fallback, the worker runs Laya's fp32 model - uncompiled (fp16 is about 2.4x slower there). - -Other configuration is laya-serve's (`LAYA_HOST`, `LAYA_PORT`, `LAYA_DEVICE`, `LAYA_MODELS`, -`LAYA_API_KEY`, ...). Measurements and setup are in the -[Apple Silicon recipe](../../../recipe/laya/apple-silicon.md). +## Files + +- [`src/frontend/laya_mps.py`](../../frontend/laya_mps.py): the HTTP worker. laya-serve (`laya[serve]==0.3.20`) + with its request handling unchanged, started as `PYTHONPATH=src python -m frontend.laya_mps --device mps`. +- `engine.py`: what the worker runs before readiness (a warmup of every loaded model over short, long and + multi-question requests) and what `/health` reports about a loaded model, read on every call: device, + weight and autocast dtypes, checkpoint and the revision the weights were loaded from, `device_mismatch`. +- `optimize.py`: the two GPU options. `--compile` compiles one-question requests end to end and, for several + questions, only the encoder (Laya's decision head is slower compiled on MPS). `--weights fp16` keeps the + checkpoint's fp16 weights instead of Laya's fp32 upcast (`act_head` stays fp32). Both apply on the GPU + only: on the CPU, including after Laya falls back to it on a GPU out-of-memory error, the worker runs + Laya's fp32 model uncompiled. +- Tests: [`tests/laya/`](../../../tests/laya/). The unit tests use a fake router; `LAYA_CONTRACT=1` adds + contract tests against a real worker on the CPU. + +What the worker changes against laya-serve, with measurements, is in the +[Apple Silicon recipe](../../../recipe/laya/apple-silicon.md): it binds only after the warmup (laya-serve +answers `/health` before any forward pass, so its first request took 0.7–1.1 s against 70–81 ms), and +`/health` tells the device the model is actually on (laya-serve reports the configured one; Laya falls +back to the CPU with only a printed warning). `--require-device` exits at startup if a model is not on +the requested device. ```sh -LAYA_DEVICE=mps LAYA_MODELS=english python src/models/laya/worker.py -python -m pytest src/models/laya/tests # unit tests, no model -LAYA_CONTRACT=1 python -m pytest src/models/laya/tests # plus contract tests on CPU, loads the checkpoint +PYTHONPATH=src python -m frontend.laya_mps --device mps --model english +PYTHONPATH=src python -m pytest tests/laya # unit tests, no model +LAYA_CONTRACT=1 PYTHONPATH=src python -m pytest tests/laya # plus contract tests on CPU, loads the checkpoint ``` diff --git a/src/models/laya/engine.py b/src/models/laya/engine.py new file mode 100644 index 0000000..92af219 --- /dev/null +++ b/src/models/laya/engine.py @@ -0,0 +1,88 @@ +"""Laya as a served model: what to run before readiness and what to report about a loaded model. + +laya's `Router` and `Agent` do the loading and the forward passes. This module adds the warmup every +loaded model runs before the worker binds, the description of a loaded model that `/health` returns, and +the record of which checkpoint revision laya actually loaded. The HTTP worker is `frontend/laya_mps.py`; +the GPU optimizations are `optimize.py` next to this file. +""" + +import time +from pathlib import Path +from typing import Any + +# (words of state, questions): each shape runs twice. Short, mid-length and near-window states, then +# 3 and 6 questions (6 is at or above laya's MPS fp16 autocast threshold of 5 rows). +_CHOICE = { + "type": "choice", + "instructions": "Which team should handle this?", + "criteria": {"billing": "Charges and refunds", "technical": "Software problems", "other": "Anything else"}, +} +_SCORE = {"type": "score", "instructions": "How urgent is it?", "criteria": ["Low", "Medium", "High"]} +_NOUL = {"type": "noul", "instructions": "Does the customer ask for a refund?"} +WARMUP_SHAPES = [ + (10, {"q": _CHOICE}), + (150, {"q": _CHOICE}), + (400, {"q": _CHOICE}), + (10, {"a": _CHOICE, "b": _SCORE, "c": _NOUL}), + (10, {f"q{i}": q for i, q in enumerate([_CHOICE, _SCORE, _NOUL, _CHOICE, _SCORE, _NOUL])}), +] +WARMUP_REPEATS = 2 + + +def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_REPEATS) -> dict[str, Any]: + """Run every shape `repeats` times. Any failure propagates: a worker that cannot answer must not bind.""" + started = time.perf_counter() + routing = None + for words, questions in shapes: + state = " ".join(["refund"] * words) + for _ in range(repeats): + result = router.predict(state, questions, model=model) + routing = result.get("routing") or routing + return {"warmup_ms": round((time.perf_counter() - started) * 1000, 1), "routing": routing} + + +def record_snapshot_revisions() -> dict[str, str]: + """Record the commit each Hugging Face checkpoint is loaded from, keyed by repo id. + + laya calls huggingface_hub.snapshot_download while loading and keeps only the repo id; the + returned path (.../snapshots//...) is the only place the loaded revision appears. Call this + before the router loads anything. A checkpoint loaded from a local path records nothing. + """ + import huggingface_hub + + revisions: dict[str, str] = {} + original = huggingface_hub.snapshot_download + + def recording(repo_id, *args, **kwargs): + path = original(repo_id, *args, **kwargs) + parts = Path(path).parts + if "snapshots" in parts[:-1]: + revisions[repo_id] = parts[parts.index("snapshots") + 1] + return path + + huggingface_hub.snapshot_download = recording + return revisions + + +def describe( + agent: Any, requested: str | None, routing: dict[str, Any] | None, revisions: dict[str, str] | None = None +) -> dict[str, Any]: + """What /health reports about one loaded agent, read from the agent as it is now.""" + device = str(getattr(agent, "device", "unknown")) + model = getattr(agent, "model", None) + try: + weights = str(next(model.parameters()).dtype) if model is not None else None + except (AttributeError, StopIteration, TypeError): + weights = None + repo = (routing or {}).get("repo") + requested_type = requested.split(":")[0] if requested else None + return { + "device": device, + "requested_device": requested or "auto", + "device_mismatch": bool(requested_type) and device.split(":")[0] != requested_type, + "weights_dtype": weights, + "autocast_dtype": str(getattr(agent, "dtype", None)), + "mps_amp_min_rows": getattr(agent, "mps_amp_min_rows", None), + "checkpoint": repo, + "revision": (revisions or {}).get(repo), + } diff --git a/src/models/laya/optimize.py b/src/models/laya/optimize.py index 77261d9..5c66948 100644 --- a/src/models/laya/optimize.py +++ b/src/models/laya/optimize.py @@ -4,7 +4,7 @@ - Compile: torch.compile for the batches where it pays off on MPS. Both go through one wrapper module that replaces `agent.model`, so that on the CPU the worker always runs -laya's own fp32 model. worker.py decides when to apply them and reports the result in /health. +laya's own fp32 model. frontend/laya_mps.py decides when to apply them and reports the result in /health. """ from typing import Any diff --git a/src/models/laya/worker.py b/src/models/laya/worker.py deleted file mode 100644 index 37b5e6c..0000000 --- a/src/models/laya/worker.py +++ /dev/null @@ -1,232 +0,0 @@ -"""Laya worker: laya-serve with warmup before readiness and the actual device in /health. - -laya-serve (laya 0.3.20) answers /health as soon as it binds, before any forward pass, and reports -LAYA_DEVICE as configured rather than where the model ended up. This worker reuses laya's app and -request handling unchanged and fixes both: - -- It binds only after a warmup of every loaded model that covers short, long and multi-question - requests (the last one crosses laya's fp16 autocast threshold on MPS), so a reachable worker is a - warm one whichever model a request is routed to. -- /health reports, per loaded model and read on every call, the device, weight and autocast dtypes, the - checkpoint and the revision its weights were downloaded from, and whether the device differs from the - one requested. laya moves a model to the CPU on a GPU out-of-memory error and keeps serving. - -Configuration is laya-serve's (LAYA_HOST, LAYA_PORT, LAYA_DEVICE, LAYA_MODELS, LAYA_API_KEY, ...) plus: - - LAYA_WORKER_MODEL model summarised at the top of /health; english - also the one loaded when nothing is - preloaded (LAYA_PRELOAD=0) - LAYA_REQUIRE_DEVICE exit instead of serving on another 0 - device than LAYA_DEVICE asked for - LAYA_WORKER_COMPILE off or on: torch.compile (dynamic=True) off - before warmup; see optimize.py - LAYA_WORKER_WEIGHTS fp32 or fp16: keep the checkpoint's fp32 - fp16 weights instead of laya's fp32 - upcast - -Both options apply on the GPU only. If laya falls back to the CPU after a GPU out-of-memory error, the -worker runs laya's fp32 model uncompiled from then on, and /health shows the CPU. - -The warmup also compiles the graphs, so the worker takes longer to become ready (35–39 s instead of -8–10 s on an M1 Pro). /health counts compiled graphs at readiness and now; `recompiled_after_ready` means -a request hit a shape class the warmup did not cover. -""" - -import logging -import os -import sys -import time -from pathlib import Path -from typing import Any - -import optimize - -log = logging.getLogger("laya-worker") - -# (words of state, questions): each shape runs twice. Short, mid-length and near-window states, then -# 3 and 6 questions (6 is at or above laya's MPS fp16 autocast threshold of 5 rows). -_CHOICE = { - "type": "choice", - "instructions": "Which team should handle this?", - "criteria": {"billing": "Charges and refunds", "technical": "Software problems", "other": "Anything else"}, -} -_SCORE = {"type": "score", "instructions": "How urgent is it?", "criteria": ["Low", "Medium", "High"]} -_NOUL = {"type": "noul", "instructions": "Does the customer ask for a refund?"} -WARMUP_SHAPES = [ - (10, {"q": _CHOICE}), - (150, {"q": _CHOICE}), - (400, {"q": _CHOICE}), - (10, {"a": _CHOICE, "b": _SCORE, "c": _NOUL}), - (10, {f"q{i}": q for i, q in enumerate([_CHOICE, _SCORE, _NOUL, _CHOICE, _SCORE, _NOUL])}), -] -WARMUP_REPEATS = 2 - - -def _env_bool(name: str) -> bool: - return os.environ.get(name, "").strip().lower() in ("1", "true", "yes", "on") - - -def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_REPEATS) -> dict[str, Any]: - """Run every shape `repeats` times. Any failure propagates: a worker that cannot answer must not bind.""" - started = time.perf_counter() - routing = None - for words, questions in shapes: - state = " ".join(["refund"] * words) - for _ in range(repeats): - result = router.predict(state, questions, model=model) - routing = result.get("routing") or routing - return {"warmup_ms": round((time.perf_counter() - started) * 1000, 1), "routing": routing} - - -COMPILE_MODES = {"": "off", "0": "off", "off": "off", "1": "on", "on": "on"} -WEIGHT_MODES = {"": "fp32", "fp32": "fp32", "fp16": "fp16"} - - -def record_snapshot_revisions() -> dict[str, str]: - """Record the commit each Hugging Face checkpoint is loaded from, keyed by repo id. - - laya calls huggingface_hub.snapshot_download while loading and keeps only the repo id; the - returned path (.../snapshots//...) is the only place the loaded revision appears. Call this - before the router loads anything. A checkpoint loaded from a local path records nothing. - """ - import huggingface_hub - - revisions: dict[str, str] = {} - original = huggingface_hub.snapshot_download - - def recording(repo_id, *args, **kwargs): - path = original(repo_id, *args, **kwargs) - parts = Path(path).parts - if "snapshots" in parts[:-1]: - revisions[repo_id] = parts[parts.index("snapshots") + 1] - return path - - huggingface_hub.snapshot_download = recording - return revisions - - -def describe( - agent: Any, requested: str | None, routing: dict[str, Any] | None, revisions: dict[str, str] | None = None -) -> dict[str, Any]: - """What /health reports about one loaded agent.""" - device = str(getattr(agent, "device", "unknown")) - model = getattr(agent, "model", None) - try: - weights = str(next(model.parameters()).dtype) if model is not None else None - except (AttributeError, StopIteration, TypeError): - weights = None - repo = (routing or {}).get("repo") - requested_type = requested.split(":")[0] if requested else None - return { - "device": device, - "requested_device": requested or "auto", - "device_mismatch": bool(requested_type) and device.split(":")[0] != requested_type, - "weights_dtype": weights, - "autocast_dtype": str(getattr(agent, "dtype", None)), - "mps_amp_min_rows": getattr(agent, "mps_amp_min_rows", None), - "checkpoint": repo, - "revision": (revisions or {}).get(repo), - } - - -def create_worker_app( - router: Any, - model: str, - requested: str | None, - require_device: bool = False, - compile: str = "off", - graph_counter=optimize.compiled_graphs, - revisions: dict[str, str] | None = None, - weights: str = "fp32", -): - """Optionally compile, warm up every loaded model, then return laya's app with /health replaced. - `model` is the one summarised at the top of /health, and the one loaded if nothing is preloaded. - Raises if compiling or warmup fails.""" - from laya.serve import create_app - - if compile not in ("off", "on"): - raise ValueError(f"compile must be off or on, not {compile!r}") - if weights not in ("fp32", "fp16"): - raise ValueError(f"weights must be fp32 or fp16, not {weights!r}") - names = list(router.loaded) or [model] # never load a model the worker was not asked to serve - if weights == "fp16" or compile == "on": - for name in names: - if not optimize.apply(router.load(name), fp16=weights == "fp16", compile=compile == "on"): - log.warning("%s is on the CPU: LAYA_WORKER_WEIGHTS and LAYA_WORKER_COMPILE apply on the GPU only", name) - warmed = {name: warmup(router, name) for name in names} - agents = {name: router.load(name) for name in names} - primary = model if model in agents else names[0] - - def current() -> dict[str, Any]: - """Describe the agents as they are now: laya moves a model to the CPU when a request runs out of - GPU memory, so the device at startup is not necessarily the device serving the next request.""" - models = { - name: { - **describe(agent, requested, warmed[name]["routing"], revisions), - "warmup_ms": warmed[name]["warmup_ms"], - } - for name, agent in agents.items() - } - return { - **models[primary], - "device_mismatch": any(m["device_mismatch"] for m in models.values()), - "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), - "models": models, - } - - info = current() - models = info["models"] - graphs_at_ready = graph_counter() if compile != "off" else None - if info["device_mismatch"]: - wrong = ", ".join(f"{n} is on {m['device']}" for n, m in models.items() if m["device_mismatch"]) - message = f"asked for {info['requested_device']}, {wrong}" - if require_device: - raise RuntimeError(message) - log.warning(message) - - app = create_app(router) - app.router.routes[:] = [r for r in app.router.routes if getattr(r, "path", None) != "/health"] - - @app.get("/health") - def health() -> dict[str, Any]: - compiled = {"mode": compile} - if compile != "off": - now = graph_counter() - compiled.update( - graphs_at_ready=graphs_at_ready, graphs_now=now, recompiled_after_ready=now > graphs_at_ready - ) - return {"status": "ok", "ready": True, "loaded": router.loaded, **current(), "compile": compiled} - - return app - - -def main() -> None: - import uvicorn - from laya.serve import _resolve_port, build_router - - logging.basicConfig(level=logging.INFO, format="%(name)s: %(message)s") - model = os.environ.get("LAYA_WORKER_MODEL", "english") - requested = os.environ.get("LAYA_DEVICE") or None - revisions = record_snapshot_revisions() - try: - app = create_worker_app( - build_router(), - model, - requested, - require_device=_env_bool("LAYA_REQUIRE_DEVICE"), - compile=COMPILE_MODES.get(os.environ.get("LAYA_WORKER_COMPILE", "").strip().lower(), "invalid"), - revisions=revisions, - weights=WEIGHT_MODES.get(os.environ.get("LAYA_WORKER_WEIGHTS", "").strip().lower(), "invalid"), - ) - except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever - sys.exit(f"laya-worker: not starting: {exc}") - uvicorn.run( - app, - host=os.environ.get("LAYA_HOST", "127.0.0.1"), - port=_resolve_port(), - log_level=os.environ.get("LAYA_LOG_LEVEL", "info"), - ) - - -if __name__ == "__main__": - main() diff --git a/src/models/laya/tests/test_contract.py b/tests/laya/test_contract.py similarity index 91% rename from src/models/laya/tests/test_contract.py rename to tests/laya/test_contract.py index cd01468..212219d 100644 --- a/src/models/laya/tests/test_contract.py +++ b/tests/laya/test_contract.py @@ -1,7 +1,7 @@ """Contract tests against a real worker process on CPU. They load the Laya checkpoint, so they only run with LAYA_CONTRACT=1: - LAYA_CONTRACT=1 python -m pytest src/models/laya/tests/test_contract.py + LAYA_CONTRACT=1 PYTHONPATH=src python -m pytest tests/laya/test_contract.py """ import http.client @@ -18,7 +18,7 @@ pytestmark = pytest.mark.skipif(os.environ.get("LAYA_CONTRACT") != "1", reason="set LAYA_CONTRACT=1") -WORKER = Path(__file__).resolve().parents[1] / "worker.py" +SRC = Path(__file__).resolve().parents[2] / "src" TOKEN = "contract-test-token" STATE = "I was charged twice for my order. Please refund the duplicate today." CHOICE = { @@ -62,16 +62,21 @@ def decide(port, questions, **kwargs): @pytest.fixture(scope="module") def worker(): port = free_port() - env = { - **os.environ, - "LAYA_HOST": "127.0.0.1", - "LAYA_PORT": str(port), - "LAYA_DEVICE": "cpu", - "LAYA_MODELS": "english", - "LAYA_API_KEY": TOKEN, - "LAYA_LOG_LEVEL": "warning", - } - process = subprocess.Popen([sys.executable, str(WORKER)], env=env) + env = {**os.environ, "PYTHONPATH": str(SRC), "LAYA_API_KEY": TOKEN} + command = [ + sys.executable, + "-m", + "frontend.laya_mps", + "--device", + "cpu", + "--model", + "english", + "--port", + str(port), + "--log-level", + "warning", + ] + process = subprocess.Popen(command, env=env) deadline = time.monotonic() + 600 while time.monotonic() < deadline: assert process.poll() is None, f"worker exited with {process.returncode}" diff --git a/src/models/laya/tests/test_worker.py b/tests/laya/test_worker.py similarity index 78% rename from src/models/laya/tests/test_worker.py rename to tests/laya/test_worker.py index 40bb0fa..712e1af 100644 --- a/src/models/laya/tests/test_worker.py +++ b/tests/laya/test_worker.py @@ -4,15 +4,13 @@ """ import sys -from pathlib import Path from types import SimpleNamespace import pytest from fastapi.testclient import TestClient -sys.path.insert(0, str(Path(__file__).resolve().parents[1])) -import optimize -import worker +from frontend import laya_mps as worker +from models.laya import engine, optimize ANSWER = {"type": "noul", "noul": 0.9, "confidence": 0.9} @@ -61,30 +59,30 @@ def predict(self, state, questions, model=None): def test_warmup_covers_short_long_and_fp16_multi_question_shapes(): router = FakeRouter() - worker.warmup(router, "english") + engine.warmup(router, "english") words = {len(state.split()) for state, _, _ in router.calls} rows = {len(questions) for _, questions, _ in router.calls} assert min(words) <= 20 and max(words) >= 400 assert max(rows) >= router.agent.mps_amp_min_rows # crosses laya's fp16 autocast threshold on MPS assert {q["type"] for _, questions, _ in router.calls for q in questions.values()} == {"choice", "score", "noul"} - assert len(router.calls) == len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + assert len(router.calls) == len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS assert all(model == "english" for _, _, model in router.calls) def test_warmup_runs_before_the_app_exists(): router = FakeRouter() - worker.create_worker_app(router, "english", "mps") - assert len(router.calls) == len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + worker.build_app(router, "english", "mps") + assert len(router.calls) == len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS def test_warmup_failure_raises_and_no_app_is_built(): with pytest.raises(RuntimeError, match="out of memory"): - worker.create_worker_app(FakeRouter(fail_on_call=3), "english", "mps") + worker.build_app(FakeRouter(fail_on_call=3), "english", "mps") def test_health_reports_the_agent_device_not_the_requested_one(): router = FakeRouter(FakeAgent(device="cpu", dtype="torch.float32")) - health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() + health = TestClient(worker.build_app(router, "english", "mps")).get("/health").json() assert health["device"] == "cpu" assert health["requested_device"] == "mps" assert health["device_mismatch"] is True @@ -92,7 +90,7 @@ def test_health_reports_the_agent_device_not_the_requested_one(): def test_health_on_the_requested_device(): - health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() + health = TestClient(worker.build_app(FakeRouter(), "english", "mps")).get("/health").json() assert health["device"] == "mps" assert health["device_mismatch"] is False assert health["autocast_dtype"] == "torch.float16" @@ -102,15 +100,12 @@ def test_health_on_the_requested_device(): def test_device_index_is_not_a_mismatch(): router = FakeRouter(FakeAgent(device="cuda:0")) - assert ( - TestClient(worker.create_worker_app(router, "english", "cuda")).get("/health").json()["device_mismatch"] - is False - ) + assert TestClient(worker.build_app(router, "english", "cuda")).get("/health").json()["device_mismatch"] is False def test_auto_device_is_never_a_mismatch(): router = FakeRouter(FakeAgent(device="cpu")) - health = TestClient(worker.create_worker_app(router, "english", None)).get("/health").json() + health = TestClient(worker.build_app(router, "english", None)).get("/health").json() assert health["requested_device"] == "auto" assert health["device_mismatch"] is False @@ -118,17 +113,17 @@ def test_auto_device_is_never_a_mismatch(): def test_require_device_refuses_to_serve_on_another_device(): router = FakeRouter(FakeAgent(device="cpu")) with pytest.raises(RuntimeError, match="asked for mps, english is on cpu"): - worker.create_worker_app(router, "english", "mps", require_device=True) + worker.build_app(router, "english", "mps", require_device=True) def test_only_one_health_route_remains(): - app = worker.create_worker_app(FakeRouter(), "english", "mps") + app = worker.build_app(FakeRouter(), "english", "mps") assert [r.path for r in app.router.routes if getattr(r, "path", None) == "/health"] == ["/health"] def test_decisions_still_go_through_laya_serve(): router = FakeRouter() - client = TestClient(worker.create_worker_app(router, "english", "mps")) + client = TestClient(worker.build_app(router, "english", "mps")) before = len(router.calls) response = client.post( "/v1/systemone", @@ -140,9 +135,8 @@ def test_decisions_still_go_through_laya_serve(): def test_main_exits_non_zero_when_warmup_fails(monkeypatch): - import laya.serve - - monkeypatch.setattr(laya.serve, "build_router", lambda: FakeRouter(fail_on_call=1)) + monkeypatch.setattr(worker, "make_router", lambda device, model: FakeRouter(fail_on_call=1)) + monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps"]) monkeypatch.setattr("uvicorn.run", lambda *a, **k: pytest.fail("must not bind")) with pytest.raises(SystemExit, match="not starting"): worker.main() @@ -152,23 +146,23 @@ def test_compile_wraps_the_model_before_warmup(monkeypatch): router = FakeRouter() order = [] monkeypatch.setattr(optimize, "compile_agent", lambda agent: order.append((agent, len(router.calls)))) - worker.create_worker_app(router, "english", "mps", compile="on", graph_counter=lambda: 3) + worker.build_app(router, "english", "mps", compile=True, graph_counter=lambda: 3) assert order == [(router.agent, 0)] # before the first warmup request def test_health_reports_compile_off_by_default(): - health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() - assert health["compile"] == {"mode": "off"} + health = TestClient(worker.build_app(FakeRouter(), "english", "mps")).get("/health").json() + assert health["compile"] == {"enabled": False} def test_health_flags_graphs_compiled_after_ready(monkeypatch): monkeypatch.setattr(optimize, "compile_agent", lambda agent: None) graphs = iter([4, 4, 5]) # at readiness, first /health, second /health after a new shape compiled client = TestClient( - worker.create_worker_app(FakeRouter(), "english", "mps", compile="on", graph_counter=lambda: next(graphs)) + worker.build_app(FakeRouter(), "english", "mps", compile=True, graph_counter=lambda: next(graphs)) ) first = client.get("/health").json()["compile"] - assert first == {"mode": "on", "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} + assert first == {"enabled": True, "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} assert client.get("/health").json()["compile"]["recompiled_after_ready"] is True @@ -178,12 +172,7 @@ def broken(agent): monkeypatch.setattr(optimize, "compile_agent", broken) with pytest.raises(RuntimeError, match="unsupported op"): - worker.create_worker_app(FakeRouter(), "english", "mps", compile="on") - - -def test_unknown_compile_mode_is_refused(): - with pytest.raises(ValueError, match="off or on"): - worker.create_worker_app(FakeRouter(), "english", "mps", compile="invalid") + worker.build_app(FakeRouter(), "english", "mps", compile=True) def test_compiled_paths_by_batch_rows(monkeypatch): @@ -229,19 +218,19 @@ def fake_compile(module, dynamic): def test_every_loaded_model_is_warmed_and_described(): agents = {"english": FakeAgent(), "multilingual": FakeAgent(device="cpu", dtype="torch.float32")} router = FakeRouter(agents=agents) - health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() - per_model = len(worker.WARMUP_SHAPES) * worker.WARMUP_REPEATS + health = TestClient(worker.build_app(router, "english", "mps")).get("/health").json() + per_model = len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS assert [m for _, _, m in router.calls].count("multilingual") == per_model assert [m for _, _, m in router.calls].count("english") == per_model assert set(health["models"]) == {"english", "multilingual"} - assert health["device"] == "mps" # the top level summarises LAYA_WORKER_MODEL + assert health["device"] == "mps" # the top level summarises --model assert health["models"]["multilingual"]["device"] == "cpu" assert health["device_mismatch"] is True # one model off the requested device is enough def test_a_model_that_is_not_preloaded_is_not_loaded_for_warmup(): router = FakeRouter(agents={"multilingual": FakeAgent()}) - health = TestClient(worker.create_worker_app(router, "english", "mps")).get("/health").json() + health = TestClient(worker.build_app(router, "english", "mps")).get("/health").json() assert "english" not in router.loads assert {m for _, _, m in router.calls} == {"multilingual"} assert set(health["models"]) == {"multilingual"} @@ -249,15 +238,15 @@ def test_a_model_that_is_not_preloaded_is_not_loaded_for_warmup(): def test_nothing_preloaded_warms_the_worker_model(): router = FakeRouter(agents={}) - worker.create_worker_app(router, "english", "mps") + worker.build_app(router, "english", "mps") assert {m for _, _, m in router.calls} == {"english"} def test_revision_comes_from_the_loaded_snapshot_not_a_guess(): revisions = {"convaiinnovations/laya": "55cf4c4"} - health = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps", revisions=revisions)).get("/health") + health = TestClient(worker.build_app(FakeRouter(), "english", "mps", revisions=revisions)).get("/health") assert health.json()["revision"] == "55cf4c4" - unknown = TestClient(worker.create_worker_app(FakeRouter(), "english", "mps")).get("/health").json() + unknown = TestClient(worker.build_app(FakeRouter(), "english", "mps")).get("/health").json() assert unknown["revision"] is None @@ -269,7 +258,7 @@ def test_record_snapshot_revisions_reads_the_downloaded_path(monkeypatch): "/local/checkpoint": "/local/checkpoint", } monkeypatch.setattr(huggingface_hub, "snapshot_download", lambda repo_id, **kwargs: paths[repo_id]) - revisions = worker.record_snapshot_revisions() + revisions = engine.record_snapshot_revisions() assert ( huggingface_hub.snapshot_download("convaiinnovations/laya", allow_patterns=["*"]) == paths["convaiinnovations/laya"] @@ -299,28 +288,23 @@ def test_fp16_weights_are_applied_to_every_loaded_model_before_warmup(monkeypatc agents = {"english": FakeAgent(), "multilingual": FakeAgent()} router = FakeRouter(agents=agents) monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: order.append((agent, len(router.calls)))) - worker.create_worker_app(router, "english", "mps", weights="fp16") + worker.build_app(router, "english", "mps", fp16=True) assert order == [(agents["english"], 0), (agents["multilingual"], 0)] -def test_unknown_weights_mode_is_refused(): - with pytest.raises(ValueError, match="fp32 or fp16"): - worker.create_worker_app(FakeRouter(), "english", "mps", weights="int8") - - def test_options_are_not_applied_to_a_model_on_the_cpu(monkeypatch, caplog): applied = [] monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: applied.append("fp16")) monkeypatch.setattr(optimize, "compile_agent", lambda agent: applied.append("compile")) with caplog.at_level("WARNING", logger="laya-worker"): - worker.create_worker_app( - FakeRouter(FakeAgent(device="cpu")), "english", "cpu", weights="fp16", compile="on", graph_counter=lambda: 0 + worker.build_app( + FakeRouter(FakeAgent(device="cpu")), "english", "cpu", fp16=True, compile=True, graph_counter=lambda: 0 ) assert applied == [] assert "apply on the GPU only" in caplog.text caplog.clear() with caplog.at_level("WARNING", logger="laya-worker"): - worker.create_worker_app(FakeRouter(), "english", "mps", weights="fp16", compile="on", graph_counter=lambda: 0) + worker.build_app(FakeRouter(), "english", "mps", fp16=True, compile=True, graph_counter=lambda: 0) assert applied == ["fp16", "compile"] assert "GPU only" not in caplog.text @@ -351,7 +335,7 @@ def forward(self, input_ids): def test_health_follows_a_fallback_to_cpu_after_startup(): agent = FakeAgent(device="mps") - client = TestClient(worker.create_worker_app(FakeRouter(agent), "english", "mps", require_device=True)) + client = TestClient(worker.build_app(FakeRouter(agent), "english", "mps", require_device=True)) assert client.get("/health").json()["device_mismatch"] is False agent.device = "cpu" # what laya does when a request runs out of GPU memory agent.dtype = "torch.float32" From 9e8ddb157365bcb10cf8d267792759861954daa8 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:37:10 +0800 Subject: [PATCH 13/21] Laya worker: keep measurements out of comments, report compile.active, honour --log-level - Docstrings in optimize.py and profile_mps.py no longer carry one machine's numbers or a private spec id. - /health gains compile.active, false once no model runs the compiled path (after a CPU fallback). - --log-level applies to the worker's own log as well as uvicorn's. - Startup failures print their traceback before exiting. - Warn when laya's autocast row threshold is above what the warmup covers. - compile_agent is a no-op on an already compiled agent; the wrapper is one class, found by isinstance. --- recipe/laya/apple-silicon.md | 1 + recipe/laya/bench/profile_mps.py | 2 +- src/frontend/laya_mps.py | 26 ++++++++++++++-- src/models/laya/engine.py | 5 ++-- src/models/laya/optimize.py | 51 ++++++++++++++++++++------------ tests/laya/test_worker.py | 46 +++++++++++++++++++++++++++- 6 files changed, 105 insertions(+), 26 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 0c8aed0..09071d5 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -78,6 +78,7 @@ ready after 19 s instead of 3 s; its first request took 21–36 ms in 21 of 23 f `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. +`active` is `false` once no model runs the compiled path any more, i.e. after a fallback to the CPU. Both options apply on the GPU only. After a fallback to the CPU the worker runs Laya's fp32 model uncompiled, like a worker started without them. diff --git a/recipe/laya/bench/profile_mps.py b/recipe/laya/bench/profile_mps.py index 4a9298d..187d931 100644 --- a/recipe/laya/bench/profile_mps.py +++ b/recipe/laya/bench/profile_mps.py @@ -1,4 +1,4 @@ -"""Where a Laya request's time goes on MPS (spec R2). Wraps laya's stages on the loaded instance; laya +"""Where a Laya request's time goes on MPS. Wraps laya's stages on the loaded instance; laya itself is not modified. Three measurements: diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py index 827b6d0..7d49d99 100644 --- a/src/frontend/laya_mps.py +++ b/src/frontend/laya_mps.py @@ -70,6 +70,17 @@ def current() -> dict[str, Any]: "models": models, } + for name, agent in agents.items(): + autocast_rows = getattr(agent, "mps_amp_min_rows", None) + if str(agent.device).startswith("mps") and autocast_rows and autocast_rows > engine.WARMUP_MAX_ROWS: + log.warning( + "%s: laya autocasts from %d questions but the warmup stops at %d; the first request that " + "large is not warm", + name, + autocast_rows, + engine.WARMUP_MAX_ROWS, + ) + info = current() graphs_at_ready = graph_counter() if compile else None if info["device_mismatch"]: @@ -88,7 +99,10 @@ def health() -> dict[str, Any]: if compile: now = graph_counter() compiled.update( - graphs_at_ready=graphs_at_ready, graphs_now=now, recompiled_after_ready=now > graphs_at_ready + active=any(optimize.compile_active(agent) for agent in agents.values()), + graphs_at_ready=graphs_at_ready, + graphs_now=now, + recompiled_after_ready=now > graphs_at_ready, ) return {"status": "ok", "ready": True, "loaded": router.loaded, **current(), "compile": compiled} @@ -113,12 +127,17 @@ def main() -> None: parser.add_argument("--require-device", action="store_true", help="exit if a model is not on --device") parser.add_argument("--host", default="127.0.0.1") parser.add_argument("--port", type=int, default=8000) - parser.add_argument("--log-level", default="info") + parser.add_argument( + "--log-level", + default="info", + choices=["critical", "error", "warning", "info", "debug"], + help="for the worker's own log and uvicorn's", + ) args = parser.parse_args() import uvicorn - logging.basicConfig(level=logging.INFO, format="%(name)s: %(message)s") + logging.basicConfig(level=args.log_level.upper(), format="%(name)s: %(message)s") revisions = engine.record_snapshot_revisions() try: app = build_app( @@ -131,6 +150,7 @@ def main() -> None: revisions=revisions, ) except Exception as exc: # noqa: BLE001 -- any failure before binding means not ready, ever + log.exception("startup failed") sys.exit(f"laya-worker: not starting: {exc}") uvicorn.run(app, host=args.host, port=args.port, log_level=args.log_level) diff --git a/src/models/laya/engine.py b/src/models/laya/engine.py index 92af219..694aa16 100644 --- a/src/models/laya/engine.py +++ b/src/models/laya/engine.py @@ -10,8 +10,8 @@ from pathlib import Path from typing import Any -# (words of state, questions): each shape runs twice. Short, mid-length and near-window states, then -# 3 and 6 questions (6 is at or above laya's MPS fp16 autocast threshold of 5 rows). +# (words of state, questions). Short, mid-length and near-window states, then several questions; the last +# shape has enough rows to reach laya's fp16 autocast on MPS (`agent.mps_amp_min_rows`). _CHOICE = { "type": "choice", "instructions": "Which team should handle this?", @@ -27,6 +27,7 @@ (10, {f"q{i}": q for i, q in enumerate([_CHOICE, _SCORE, _NOUL, _CHOICE, _SCORE, _NOUL])}), ] WARMUP_REPEATS = 2 +WARMUP_MAX_ROWS = max(len(questions) for _, questions in WARMUP_SHAPES) def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_REPEATS) -> dict[str, Any]: diff --git a/src/models/laya/optimize.py b/src/models/laya/optimize.py index 5c66948..6283be1 100644 --- a/src/models/laya/optimize.py +++ b/src/models/laya/optimize.py @@ -7,27 +7,17 @@ laya's own fp32 model. frontend/laya_mps.py decides when to apply them and reports the result in /health. """ +import functools from typing import Any -def _served(agent: Any) -> Any: - """The module the worker puts in place of laya's model, created on first use. - - On the GPU it runs the compiled paths when there are any, otherwise laya's model. On the CPU it always - runs laya's model in fp32: laya moves the model to the CPU when a request runs out of GPU memory, and - there fp16 weights are slower (334 ms against 138 ms for a 68-token request) and the compiled graphs - would first recompile (28 s measured). So after a fallback the worker behaves like plain laya. - """ +@functools.cache +def _served_class() -> type: + """The module class the worker puts in place of laya's model (built on first use: torch is imported late).""" import torch - if getattr(agent.model, "laya_worker_wrapper", False): - return agent.model - eager = agent.model - class Served(torch.nn.Module): - laya_worker_wrapper = True - - def __init__(self): + def __init__(self, eager): super().__init__() self.eager = eager self.fp16 = False @@ -44,7 +34,19 @@ def forward(self, input_ids, *args, **kwargs): whole, encoder_only = self.paths return (whole if input_ids.shape[0] == 1 else encoder_only)(input_ids, *args, **kwargs) - agent.model = Served() + return Served + + +def _served(agent: Any) -> Any: + """Wrap `agent.model` once and return the wrapper. + + On the GPU it runs the compiled paths when there are any, otherwise laya's model. On the CPU it always + runs laya's model in fp32: laya moves the model to the CPU when a request runs out of GPU memory, and + there fp16 weights are slower than fp32 and the compiled graphs would have to recompile first. So after + a fallback the worker behaves like plain laya. + """ + if not isinstance(agent.model, _served_class()): + agent.model = _served_class()(agent.model) return agent.model @@ -65,14 +67,16 @@ def compile_agent(agent: Any) -> None: A batch of one row (one question) runs the whole model compiled. A batch of several rows runs only the encoder compiled and laya's decision head eagerly: the head is two nn.TransformerEncoderLayer - with a key padding mask, which lose PyTorch's fused fast path when compiled and were about 50 ms - slower on padded multi-row batches. + with a key padding mask, which lose PyTorch's fused fast path when compiled and get slower on + padded multi-row batches. Does nothing for an agent that is already compiled. """ import copy import torch served = _served(agent) + if served.paths is not None: + return eager = served.eager whole = torch.compile(eager, dynamic=True) encoder_only = copy.copy(eager) # same parameters and submodules ... @@ -81,9 +85,13 @@ def compile_agent(agent: Any) -> None: served.paths = (whole, encoder_only) +def _on_cpu(agent: Any) -> bool: + return str(getattr(agent, "device", "")).startswith("cpu") + + def apply(agent: Any, *, fp16: bool, compile: bool) -> bool: """Apply the requested options to one agent. Does nothing and returns False for a model on the CPU.""" - if str(getattr(agent, "device", "")).startswith("cpu"): + if _on_cpu(agent): return False if fp16: use_fp16_weights(agent) @@ -92,6 +100,11 @@ def apply(agent: Any, *, fp16: bool, compile: bool) -> bool: return True +def compile_active(agent: Any) -> bool: + """Whether requests to this agent run the compiled paths now. False on the CPU, also after a fallback.""" + return getattr(getattr(agent, "model", None), "paths", None) is not None and not _on_cpu(agent) + + def compiled_graphs() -> int: """Graphs torch.compile has produced in this process.""" from torch._dynamo.utils import counters diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 712e1af..48ed395 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -162,7 +162,13 @@ def test_health_flags_graphs_compiled_after_ready(monkeypatch): worker.build_app(FakeRouter(), "english", "mps", compile=True, graph_counter=lambda: next(graphs)) ) first = client.get("/health").json()["compile"] - assert first == {"enabled": True, "graphs_at_ready": 4, "graphs_now": 4, "recompiled_after_ready": False} + assert first == { + "enabled": True, + "active": False, # compile_agent is stubbed out here + "graphs_at_ready": 4, + "graphs_now": 4, + "recompiled_after_ready": False, + } assert client.get("/health").json()["compile"]["recompiled_after_ready"] is True @@ -344,3 +350,41 @@ def test_health_follows_a_fallback_to_cpu_after_startup(): assert health["device_mismatch"] is True assert health["models"]["english"]["device"] == "cpu" assert health["autocast_dtype"] == "torch.float32" + + +def test_health_compile_active_follows_a_fallback_to_cpu(monkeypatch): + import torch + + compiles = [] + monkeypatch.setattr(torch, "compile", lambda module, dynamic: compiles.append(module) or module) + agent = FakeAgent() + agent.model = torch.nn.Sequential() + agent.model.encoder = torch.nn.Identity() + client = TestClient(worker.build_app(FakeRouter(agent), "english", "mps", compile=True, graph_counter=lambda: 2)) + assert client.get("/health").json()["compile"]["active"] is True + optimize.compile_agent(agent) # a second name for the same agent must not compile again + assert len(compiles) == 2 + agent.device = "cpu" + assert client.get("/health").json()["compile"]["active"] is False + + +def test_warning_when_the_warmup_does_not_reach_layas_autocast_rows(caplog): + agent = FakeAgent() + agent.mps_amp_min_rows = engine.WARMUP_MAX_ROWS + 1 + with caplog.at_level("WARNING", logger="laya-worker"): + worker.build_app(FakeRouter(agent), "english", "mps") + assert "the warmup stops at" in caplog.text + caplog.clear() + with caplog.at_level("WARNING", logger="laya-worker"): + worker.build_app(FakeRouter(), "english", "mps") + assert "the warmup stops at" not in caplog.text + + +def test_log_level_applies_to_the_workers_own_log(monkeypatch): + seen = {} + monkeypatch.setattr(worker, "make_router", lambda device, model: FakeRouter()) + monkeypatch.setattr(worker.logging, "basicConfig", lambda **kw: seen.update(kw)) + monkeypatch.setattr("uvicorn.run", lambda app, **kw: seen.update(uvicorn=kw["log_level"])) + monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps", "--log-level", "warning"]) + worker.main() + assert (seen["level"], seen["uvicorn"]) == ("WARNING", "warning") From 3a7850751df2a3fc62b1e677c09b3971bc37a8cc Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:58:06 +0800 Subject: [PATCH 14/21] Bench scripts: mark the imports that follow the sys.path setup (E402) --- recipe/laya/bench/bench_http.py | 2 +- recipe/laya/bench/bench_inproc.py | 6 +++--- recipe/laya/bench/paired.py | 4 ++-- recipe/laya/bench/profile_mps.py | 8 ++++---- 4 files changed, 10 insertions(+), 10 deletions(-) diff --git a/recipe/laya/bench/bench_http.py b/recipe/laya/bench/bench_http.py index 77e364e..c38bc4f 100644 --- a/recipe/laya/bench/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -31,7 +31,7 @@ HERE = Path(__file__).resolve().parent REPO = HERE.parents[2] sys.path.insert(0, str(HERE)) -from env import footprint_mb, header, noise_problems +from env import footprint_mb, header, noise_problems # noqa: E402 CHECKPOINT = "convaiinnovations/laya" # what laya-serve's "english" model resolves to (laya/router.py) diff --git a/recipe/laya/bench/bench_inproc.py b/recipe/laya/bench/bench_inproc.py index 5eb8e09..46cd764 100644 --- a/recipe/laya/bench/bench_inproc.py +++ b/recipe/laya/bench/bench_inproc.py @@ -16,14 +16,14 @@ T_START = time.perf_counter() warnings.filterwarnings("ignore") -import laya -import torch +import laya # noqa: E402 +import torch # noqa: E402 T_IMPORT = time.perf_counter() - T_START HERE = Path(__file__).resolve().parent sys.path.insert(0, str(HERE)) -from env import footprint_mb, header, noise_problems +from env import footprint_mb, header, noise_problems # noqa: E402 def load_workloads(path): diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index a867255..4c358aa 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -24,8 +24,8 @@ HERE = Path(__file__).resolve().parent REPO = HERE.parents[2] sys.path.insert(0, str(HERE)) -from bench_http import Client, body_for, fetch_answers, wait_ready -from env import footprint_mb, header, noise_problems +from bench_http import Client, body_for, fetch_answers, wait_ready # noqa: E402 +from env import footprint_mb, header, noise_problems # noqa: E402 CHECKPOINT = "convaiinnovations/laya" diff --git a/recipe/laya/bench/profile_mps.py b/recipe/laya/bench/profile_mps.py index 187d931..ea7512c 100644 --- a/recipe/laya/bench/profile_mps.py +++ b/recipe/laya/bench/profile_mps.py @@ -25,13 +25,13 @@ from pathlib import Path warnings.filterwarnings("ignore") -import laya -import laya.agent -import torch +import laya # noqa: E402 +import laya.agent # noqa: E402 +import torch # noqa: E402 HERE = Path(__file__).resolve().parent sys.path.insert(0, str(HERE)) -from env import header, noise_problems +from env import header, noise_problems # noqa: E402 STAGES = ["encode", "collate", "dispatch", "gpu_wait", "copy_back", "decode", "other"] From 584680dcb087b543344f7cf7cc9fb616d56a66aa Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Fri, 2 Oct 2026 09:13:20 +0800 Subject: [PATCH 15/21] Laya worker: prepare checkpoints loaded while serving; fix report parity on paired files and bundled revisions - A checkpoint laya loads for a later request gets the same options, warmup and device check through the Router's on_load hook; an evicted one drops out of /health; one that cannot be prepared is unloaded. /health names a checkpoint under `preparing` while that runs. - report.py: the parity section uses the records read() already filtered, so paired files no longer crash it. - /health finds the revision of a bundled checkpoint under the download repository. - Comments: drop a private spec id, headings repeated as comments, and a docstring restating its function. --- recipe/laya/apple-silicon.md | 6 ++ recipe/laya/bench/bench_http.py | 4 +- recipe/laya/bench/profile_mps.py | 3 - recipe/laya/bench/report.py | 25 +++---- src/frontend/laya_mps.py | 117 ++++++++++++++++++++----------- src/models/laya/engine.py | 5 +- tests/laya/test_worker.py | 93 ++++++++++++++++++++++++ 7 files changed, 192 insertions(+), 61 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 09071d5..087f487 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -37,6 +37,12 @@ request it accepts is already warm: on an M1 Pro the first request after ready t when the model cannot be placed on MPS; without it the worker logs a warning and serves from the CPU. laya-serve's environment variables still apply, e.g. `LAYA_API_KEY` for bearer authentication. +Laya loads another checkpoint when a request names it (`"model": "multilingual"`) or its routing picks it +(a non-English state). The worker prepares that checkpoint the same way inside that first request, so +that request takes seconds (other requests wait behind it; `/health` names it under `preparing` +meanwhile), and `/health` lists it from then on. With `--require-device`, a checkpoint +that does not land on the requested device is unloaded again and the request fails with 500. + Check what it is running on: ```sh diff --git a/recipe/laya/bench/bench_http.py b/recipe/laya/bench/bench_http.py index c38bc4f..055964f 100644 --- a/recipe/laya/bench/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -221,7 +221,7 @@ def memory(): def emit(record): f.write(json.dumps({**common, **record}) + "\n") - # /health's device is what the worker reports; laya-serve 0.3.20 echoes LAYA_DEVICE (issue #3, G2). + # /health's device is what the worker reports; laya-serve 0.3.20 echoes LAYA_DEVICE. emit( header( CHECKPOINT, @@ -299,7 +299,7 @@ def emit(record): "n": len(results), "errors": len(results) - ok, "elapsed_s": round(elapsed, 3), - "rps": round(ok / elapsed, 2), # successful requests only + "rps": round(ok / elapsed, 2), } ) diff --git a/recipe/laya/bench/profile_mps.py b/recipe/laya/bench/profile_mps.py index ea7512c..5bf0bdd 100644 --- a/recipe/laya/bench/profile_mps.py +++ b/recipe/laya/bench/profile_mps.py @@ -162,7 +162,6 @@ def emit(record): ) ) - # 1. Length sweep. print("## Length sweep (1 choice question, 5 options)\n") print("| words | tokens | p50 ms |\n|---|---|---|") points = [] @@ -191,7 +190,6 @@ def emit(record): emit({"type": "fit", "a_ms": round(a, 3), "b_ms_per_token": round(b, 5), "r2": round(r2, 4)}) print(f"\nwall ≈ {a:.1f} ms + {b:.3f} ms/token × tokens (R² {r2:.3f})") - # 2. Stage split. print("\n## Stages (median ms per request)\n") columns = ["wall", *STAGES, "gpu_exec"] print("| workload | tokens | rows | " + " | ".join(columns) + " |\n|" + "---|" * (len(columns) + 3)) @@ -219,7 +217,6 @@ def emit(record): cells = " | ".join(f"{medians[c]:.1f}" if c in medians else "" for c in columns) print(f"| {wid} | {tokens} | {len(w['questions'])} | {cells} |") - # 3. Operators on the host. print("\n## Host operators for W1 (torch.profiler, CPU)\n") w = workloads["W1"] with torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CPU]) as prof: diff --git a/recipe/laya/bench/report.py b/recipe/laya/bench/report.py index 0961e12..54274f1 100644 --- a/recipe/laya/bench/report.py +++ b/recipe/laya/bench/report.py @@ -176,19 +176,14 @@ def memory(phases, ends): TOLERANCE = {"fp32": 1e-3, "fp16": 1e-2} -def read_answers(paths): +def read_answers(records): envs, answers = {}, {} - for path in paths: - with open(path) as f: - for line in f: - if not line.strip(): - continue - r = json.loads(line) - key = (r["config"], r["run"]) - if r["type"] == "env": - envs[key] = r - elif r["type"] == "answers": - answers.setdefault(key, {})[r["workload"]] = r["answers"] + for r in records: + key = (r["config"], r["run"]) + if r["type"] == "env": + envs[key] = r + elif r["type"] == "answers": + answers.setdefault(key, {})[r["workload"]] = r["answers"] return envs, answers @@ -220,9 +215,9 @@ def path(env, rows): return "fp16" if autocast or env.get("weights_dtype") == "torch.float16" else "fp32" -def parity(paths, ref): +def parity(records, ref): """Answers of every run against the reference config, with the tolerances declared in advance.""" - envs, answers = read_answers(paths) + envs, answers = read_answers(records) refs = sorted(k for k in answers if k[0] == ref) if not refs: return f"no answers for reference config {ref}" @@ -282,7 +277,7 @@ def by_run(kind): if any(r["type"] == "throughput" for r in records): print("\n## Throughput\n\n" + throughput(records)) print("\n## Memory (MB)\n\n" + memory(phases, ends)) - print(f"\n## Parity against {args.ref}\n\n" + parity(args.files, args.ref)) + print(f"\n## Parity against {args.ref}\n\n" + parity(records, args.ref)) if __name__ == "__main__": diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py index 7d49d99..e1dd802 100644 --- a/src/frontend/laya_mps.py +++ b/src/frontend/laya_mps.py @@ -5,7 +5,9 @@ It is laya-serve (`laya[serve]==0.3.20`) with its request handling unchanged and three changes: - It binds only after every loaded model has run a warmup over short, long and multi-question requests, - so a reachable worker is a warm one. laya-serve answers /health before any forward pass. + so a reachable worker is a warm one. laya-serve answers /health before any forward pass. A checkpoint + laya loads later (a request that names another model, or routes to it) gets the same options and + warmup inside that first request. - /health describes the loaded models as they are now: device, weight and autocast dtypes, checkpoint and the revision the weights were loaded from, and `device_mismatch` when a model is not on the requested device. laya-serve reports the configured device, and laya moves a model to the CPU on a @@ -40,37 +42,25 @@ def build_app( graph_counter=optimize.compiled_graphs, revisions: dict[str, str] | None = None, ): - """Apply the options, warm up every loaded model, then return laya's app with /health replaced. - `model` is the one summarised at the top of /health, and the one loaded if nothing is preloaded. - Raises if an option or the warmup fails, so the caller never binds a worker that cannot answer.""" + """Prepare every loaded model (options, warmup, device check), then return laya's app with /health + replaced. A checkpoint laya loads later, for a request that names or routes to it, is prepared the same + way before it answers. `model` is the one summarised at the top of /health, and the one loaded if + nothing is preloaded. Raises if preparing fails, so the caller never binds a worker that cannot answer.""" from laya.serve import create_app - names = list(router.loaded) or [model] # never load a model the worker was not asked to serve - if fp16 or compile: - for name in names: - if not optimize.apply(router.load(name), fp16=fp16, compile=compile): - log.warning("%s is on the CPU: --compile and --weights fp16 apply on the GPU only", name) - warmed = {name: engine.warmup(router, name) for name in names} - agents = {name: router.load(name) for name in names} - primary = model if model in agents else names[0] + resident: dict[str, tuple[Any, dict[str, Any]]] = {} # prepared checkpoints: name -> (agent, warmup result) + preparing: set[str] = set() + graphs_at_ready = None - def current() -> dict[str, Any]: - """The agents as they are now, not as they were at startup (see the module docstring).""" - models = { - name: { - **engine.describe(agent, requested, warmed[name]["routing"], revisions), - "warmup_ms": warmed[name]["warmup_ms"], - } - for name, agent in agents.items() - } - return { - **models[primary], - "device_mismatch": any(m["device_mismatch"] for m in models.values()), - "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), - "models": models, - } + def apply_options(name: str, agent: Any) -> None: + if (fp16 or compile) and not optimize.apply(agent, fp16=fp16, compile=compile): + log.warning("%s is on the CPU: --compile and --weights fp16 apply on the GPU only", name) - for name, agent in agents.items(): + def make_ready(name: str, agent: Any) -> str | None: + """Warm the checkpoint up and start describing it. Returns where it is if not on the requested device.""" + nonlocal graphs_at_ready + warmed = engine.warmup(router, name) + resident[name] = (agent, warmed) autocast_rows = getattr(agent, "mps_amp_min_rows", None) if str(agent.device).startswith("mps") and autocast_rows and autocast_rows > engine.WARMUP_MAX_ROWS: log.warning( @@ -80,15 +70,56 @@ def current() -> dict[str, Any]: autocast_rows, engine.WARMUP_MAX_ROWS, ) + if compile: + graphs_at_ready = graph_counter() + described = engine.describe(agent, requested, warmed["routing"], revisions) + return f"{name} is on {described['device']}" if described["device_mismatch"] else None + + def check_device(misplaced: list[str]) -> None: + if misplaced: + message = f"asked for {requested}, {', '.join(misplaced)}" + if require_device: + raise RuntimeError(message) + log.warning(message) + + class Lifecycle: + """laya Router hooks: a checkpoint loaded while serving is prepared like the ones loaded at startup, + or not kept at all; an evicted one is no longer described.""" + + def on_load(self, ctx: Any) -> None: + preparing.add(ctx.model) + try: + apply_options(ctx.model, ctx.agent) + check_device(list(filter(None, [make_ready(ctx.model, ctx.agent)]))) + except Exception: + log.exception("%s could not be prepared and is unloaded", ctx.model) + router.unload(ctx.model) + raise + finally: + preparing.discard(ctx.model) + + def on_evict(self, ctx: Any) -> None: + resident.pop(ctx.model, None) + + names = list(router.loaded) or [model] # never load a model the worker was not asked to serve + startup = {name: router.load(name) for name in names} + for name, agent in startup.items(): + apply_options(name, agent) + check_device(list(filter(None, [make_ready(name, agent) for name, agent in startup.items()]))) + router.add_hook(Lifecycle()) - info = current() - graphs_at_ready = graph_counter() if compile else None - if info["device_mismatch"]: - wrong = ", ".join(f"{n} is on {m['device']}" for n, m in info["models"].items() if m["device_mismatch"]) - message = f"asked for {info['requested_device']}, {wrong}" - if require_device: - raise RuntimeError(message) - log.warning(message) + def current() -> dict[str, Any]: + """The agents as they are now, not as they were at startup (see the module docstring).""" + models = { + name: {**engine.describe(agent, requested, warmed["routing"], revisions), "warmup_ms": warmed["warmup_ms"]} + for name, (agent, warmed) in list(resident.items()) + } + return { + **models.get(model, next(iter(models.values()), {})), + "device_mismatch": any(m["device_mismatch"] for m in models.values()), + "warmup_ms": round(sum(m["warmup_ms"] for m in models.values()), 1), + "models": models, + } app = create_app(router) app.router.routes[:] = [r for r in app.router.routes if getattr(r, "path", None) != "/health"] @@ -99,18 +130,24 @@ def health() -> dict[str, Any]: if compile: now = graph_counter() compiled.update( - active=any(optimize.compile_active(agent) for agent in agents.values()), + active=any(optimize.compile_active(agent) for agent, _ in list(resident.values())), graphs_at_ready=graphs_at_ready, graphs_now=now, - recompiled_after_ready=now > graphs_at_ready, + recompiled_after_ready=now > graphs_at_ready and not preparing, ) - return {"status": "ok", "ready": True, "loaded": router.loaded, **current(), "compile": compiled} + return { + "status": "ok", + "ready": True, + "loaded": router.loaded, + "preparing": sorted(preparing), + **current(), + "compile": compiled, + } return app def make_router(device: str | None, model: str) -> Any: - """laya's Router with one checkpoint preloaded.""" from laya.router import Router router = Router(device=device) diff --git a/src/models/laya/engine.py b/src/models/laya/engine.py index 694aa16..0be5d1b 100644 --- a/src/models/laya/engine.py +++ b/src/models/laya/engine.py @@ -76,6 +76,9 @@ def describe( except (AttributeError, StopIteration, TypeError): weights = None repo = (routing or {}).get("repo") + # laya names a bundled checkpoint "//"; the download is recorded under "/". + revisions = revisions or {} + revision = revisions.get(repo) or revisions.get("/".join(str(repo).split("/")[:2])) requested_type = requested.split(":")[0] if requested else None return { "device": device, @@ -85,5 +88,5 @@ def describe( "autocast_dtype": str(getattr(agent, "dtype", None)), "mps_amp_min_rows": getattr(agent, "mps_amp_min_rows", None), "checkpoint": repo, - "revision": (revisions or {}).get(repo), + "revision": revision, } diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 48ed395..68f1192 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -35,6 +35,7 @@ def __init__(self, agent=None, fail_on_call=None, agents=None): self.agents = agents if agents is not None else {"english": self.agent} self.calls = [] self.loads = [] + self.hooks = [] self.fail_on_call = fail_on_call @property @@ -45,6 +46,20 @@ def load(self, name): self.loads.append(name) return self.agents.setdefault(name, self.agent) + def add_hook(self, hook): + self.hooks.append(hook) + + def unload(self, name): + self.agents.pop(name, None) + for hook in self.hooks: + hook.on_evict(SimpleNamespace(model=name)) + + def load_while_serving(self, name, agent): + """What laya's Router.load does for a checkpoint that is not resident yet.""" + self.agents[name] = agent + for hook in self.hooks: + hook.on_load(SimpleNamespace(model=name, agent=agent)) + def predict(self, state, questions, model=None): self.calls.append((state, questions, model)) if self.fail_on_call is not None and len(self.calls) == self.fail_on_call: @@ -388,3 +403,81 @@ def test_log_level_applies_to_the_workers_own_log(monkeypatch): monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps", "--log-level", "warning"]) worker.main() assert (seen["level"], seen["uvicorn"]) == ("WARNING", "warning") + + +def test_a_checkpoint_loaded_while_serving_is_prepared_and_described(monkeypatch): + applied = [] + monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: applied.append(agent)) + router = FakeRouter() + client = TestClient(worker.build_app(router, "english", "mps", fp16=True)) + before = len(router.calls) + late = FakeAgent(device="cpu", dtype="torch.float32") + router.load_while_serving("multilingual", late) + assert applied == [router.agent] # the late one is on the CPU, where the options do not apply + assert [m for _, _, m in router.calls[before:]] == ["multilingual"] * ( + len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS + ) + health = client.get("/health").json() + assert set(health["models"]) == {"english", "multilingual"} + assert health["models"]["multilingual"]["device"] == "cpu" + assert health["device_mismatch"] is True + assert health["preparing"] == [] + router.unload("multilingual") # eviction + health = client.get("/health").json() + assert set(health["models"]) == {"english"} + assert health["device_mismatch"] is False + + +def test_a_late_checkpoint_that_cannot_be_prepared_is_not_kept(): + router = FakeRouter() + client = TestClient(worker.build_app(router, "english", "mps", require_device=True)) + with pytest.raises(RuntimeError, match="multilingual is on cpu"): + router.load_while_serving("multilingual", FakeAgent(device="cpu")) + assert router.loaded == ["english"] + assert set(client.get("/health").json()["models"]) == {"english"} + router.fail_on_call = len(router.calls) + 1 + with pytest.raises(RuntimeError, match="out of memory"): # warmup fails + router.load_while_serving("multilingual", FakeAgent()) + assert router.loaded == ["english"] + + +def test_compile_baseline_moves_when_a_late_checkpoint_compiles(monkeypatch): + monkeypatch.setattr(optimize, "compile_agent", lambda agent: None) + graphs = iter([3, 7, 7]) # startup, after the late checkpoint compiled, /health + router = FakeRouter() + client = TestClient(worker.build_app(router, "english", "mps", compile=True, graph_counter=lambda: next(graphs))) + router.load_while_serving("multilingual", FakeAgent()) + compiled = client.get("/health").json()["compile"] + assert (compiled["graphs_at_ready"], compiled["recompiled_after_ready"]) == (7, False) + + +def test_revision_of_a_bundled_checkpoint_is_found_under_the_download_repo(): + routing = {"repo": "convaiinnovations/laya/multilingual"} + described = engine.describe(FakeAgent(), "mps", routing, {"convaiinnovations/laya": "abc123"}) + assert (described["checkpoint"], described["revision"]) == ("convaiinnovations/laya/multilingual", "abc123") + assert ( + engine.describe(FakeAgent(), "mps", {"repo": "/models/laya"}, {"convaiinnovations/laya": "abc123"})["revision"] + is None + ) + + +def test_startup_error_names_every_checkpoint_off_the_requested_device(): + agents = {"english": FakeAgent(device="cpu"), "multilingual": FakeAgent(device="cpu")} + with pytest.raises(RuntimeError, match="asked for mps, english is on cpu, multilingual is on cpu"): + worker.build_app(FakeRouter(agents=agents), "english", "mps", require_device=True) + + +def test_health_names_a_checkpoint_while_it_is_being_prepared(): + router = FakeRouter() + client = TestClient(worker.build_app(router, "english", "mps")) + seen = [] + predict = router.predict + + def predict_and_look(state, questions, model=None): + seen.append(client.get("/health").json()["preparing"]) + return predict(state, questions, model=model) + + router.predict = predict_and_look + router.load_while_serving("multilingual", FakeAgent()) + assert seen[0] == ["multilingual"] + assert client.get("/health").json()["preparing"] == [] From 87c920f0f1846a32563e2a34063e3da5c494ab0c Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Fri, 2 Oct 2026 10:22:21 +0800 Subject: [PATCH 16/21] Laya worker: restore checkpoints evicted for a failed late load; say which laya-serve variables apply - A late checkpoint that cannot be prepared is unloaded and the checkpoints laya evicted for it are loaded again. - Only the variables laya-serve's app reads apply (LAYA_API_KEY); the worker warns about the launcher's ones. - A second app built on the same router replaces the first app's hooks. - paired.py: timed requests are not retried and failed pairs are counted; the summary shows each side's device at the end and flags a side that left its device or compiled path. - bench: wait_ready keeps polling on a malformed response; cleanup terminates every process before waiting. --- recipe/laya/apple-silicon.md | 7 ++- recipe/laya/bench/bench_http.py | 8 +++- recipe/laya/bench/paired.py | 32 +++++++++++--- src/frontend/laya_mps.py | 78 ++++++++++++++++++++++++--------- tests/laya/test_worker.py | 57 ++++++++++++++++++++++++ 5 files changed, 153 insertions(+), 29 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 087f487..670c933 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -35,13 +35,16 @@ over short, long and multi-question requests, and only then listens on port 8000 request it accepts is already warm: on an M1 Pro the first request after ready took 70–81 ms, against 0.7–1.1 s from plain laya-serve. `--require-device` makes it exit instead of silently serving on the CPU when the model cannot be placed on MPS; without it the worker logs a warning and serves from the CPU. -laya-serve's environment variables still apply, e.g. `LAYA_API_KEY` for bearer authentication. +Of laya-serve's environment variables, `LAYA_API_KEY` (bearer authentication) still applies. Those its +launcher reads do not: device, model, host, port and log level are the flags above, and `LAYA_THREADS` and +`LAYA_AUTO_TASK` are not read. The worker warns at startup if any of them is set. Laya loads another checkpoint when a request names it (`"model": "multilingual"`) or its routing picks it (a non-English state). The worker prepares that checkpoint the same way inside that first request, so that request takes seconds (other requests wait behind it; `/health` names it under `preparing` meanwhile), and `/health` lists it from then on. With `--require-device`, a checkpoint -that does not land on the requested device is unloaded again and the request fails with 500. +that does not land on the requested device is unloaded again and the request fails with 500. If Laya +evicted another checkpoint to make room for it (it keeps two by default), the worker loads that one again. Check what it is running on: diff --git a/recipe/laya/bench/bench_http.py b/recipe/laya/bench/bench_http.py index 055964f..7f07dcc 100644 --- a/recipe/laya/bench/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -81,7 +81,7 @@ def wait_ready(url, processes, timeout_s): _, status, body = client.request("GET", "/health") if status == 200: return time.perf_counter() - started, json.loads(body) - except OSError: + except (OSError, http.client.HTTPException): pass finally: client.close() @@ -315,7 +315,11 @@ def emit(record): finally: for proc in processes.values(): proc.terminate() - proc.wait(timeout=30) + for proc in processes.values(): + try: + proc.wait(timeout=30) + except subprocess.TimeoutExpired: + proc.kill() print(out) diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index 4c358aa..e0ac27b 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -12,6 +12,7 @@ """ import argparse +import http.client import json import os import random @@ -106,7 +107,10 @@ def emit(record): first, second = ("A", "B") if i % 2 == 0 else ("B", "A") ms = {} for s in (first, second): - t, status, _ = clients[s].request("POST", "/v1/systemone", body, retry=True) + try: + t, status, _ = clients[s].request("POST", "/v1/systemone", body) + except (OSError, http.client.HTTPException): + t, status = None, 0 ms[s] = t if status == 200 else None if i >= args.discard: emit( @@ -131,7 +135,11 @@ def emit(record): finally: for p in procs.values(): p.terminate() - p.wait(timeout=30) + for p in procs.values(): + try: + p.wait(timeout=30) + except subprocess.TimeoutExpired: + p.kill() print(out_dir / f"paired_{args.run}.jsonl") @@ -165,10 +173,12 @@ def summarize(paths): env = next(r for r in records if r["type"] == "env") print(f"## {env['run']}: A = `{env['a']}`, B = `{env['b']}`, load at start {env['loadavg_1m']}\n") print("| input | pairs | A p50 ms | B p50 ms | median B/A | 95% interval |\n|---|---|---|---|---|---|") - pairs = {} + pairs, failed = {}, 0 for r in records: if r["type"] == "pair" and r["a_ms"] and r["b_ms"]: pairs.setdefault(r["workload"], []).append(r) + elif r["type"] == "pair": + failed += 1 for wid in sorted(pairs): ps = pairs[wid] med, lo, hi = median_interval([p["b_ms"] / p["a_ms"] for p in ps]) @@ -192,8 +202,20 @@ def summarize(paths): flips.append((wid, q, round(margin(a), 4))) end = next(r for r in records if r["type"] == "end") compile_state = {s: h.get("compile", {}).get("recompiled_after_ready") for s, h in end["health"].items()} - print(f"\nB vs A answers: max |Δp| {worst:.4f}, flips {flips}, errors {errors}") - print(f"recompiled after ready: {compile_state}; footprint MB: {end['footprint_mb']}\n") + devices = {s: h.get("device") for s, h in end["health"].items()} + off_gpu = [ + s + for s, h in end["health"].items() + if h.get("device_mismatch") + or (h.get("compile", {}).get("enabled") and not h["compile"].get("active", True)) + ] + print(f"\nB vs A answers: max |Δp| {worst:.4f}, flips {flips}, errors {errors}; failed pairs: {failed}") + print(f"recompiled after ready: {compile_state}; device at end: {devices}; footprint MB: {end['footprint_mb']}") + if off_gpu: + print( + f"**Side {', '.join(off_gpu)} left its device or compiled path during the run; the ratios above mix both.**" + ) + print() def main(): diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py index e1dd802..a05eba8 100644 --- a/src/frontend/laya_mps.py +++ b/src/frontend/laya_mps.py @@ -16,13 +16,17 @@ - `--compile` and `--weights fp16` make the GPU path faster (models/laya/optimize.py). Both apply on the GPU only; on the CPU, including after a fallback, the worker runs laya's fp32 model uncompiled. -laya-serve's own environment variables still apply, notably LAYA_API_KEY for bearer authentication. +Of laya-serve's environment variables, the ones its app reads still apply, notably LAYA_API_KEY for bearer +authentication. The ones its launcher reads do not, because the flags above replace it: LAYA_DEVICE, +LAYA_MODELS, LAYA_PRELOAD, LAYA_HOST, LAYA_PORT, LAYA_LOG_LEVEL, LAYA_THREADS and LAYA_AUTO_TASK. The +worker warns at startup about any of those that are set. """ from __future__ import annotations import argparse import logging +import os import sys from typing import Any @@ -30,6 +34,26 @@ log = logging.getLogger("laya-worker") +# Read by laya-serve's launcher (laya.serve.build_router and main), which this worker does not run. +UNREAD_LAYA_SERVE_VARIABLES = ( + "LAYA_DEVICE", + "LAYA_MODELS", + "LAYA_PRELOAD", + "LAYA_HOST", + "LAYA_PORT", + "LAYA_LOG_LEVEL", + "LAYA_THREADS", + "LAYA_AUTO_TASK", +) + + +class Lifecycle: + """laya Router hooks that hand checkpoint loads and evictions to one worker app.""" + + def __init__(self, on_load, on_evict): + self.on_load = on_load + self.on_evict = on_evict + def build_app( router: Any, @@ -82,31 +106,42 @@ def check_device(misplaced: list[str]) -> None: raise RuntimeError(message) log.warning(message) - class Lifecycle: - """laya Router hooks: a checkpoint loaded while serving is prepared like the ones loaded at startup, - or not kept at all; an evicted one is no longer described.""" - - def on_load(self, ctx: Any) -> None: - preparing.add(ctx.model) - try: - apply_options(ctx.model, ctx.agent) - check_device(list(filter(None, [make_ready(ctx.model, ctx.agent)]))) - except Exception: - log.exception("%s could not be prepared and is unloaded", ctx.model) - router.unload(ctx.model) - raise - finally: - preparing.discard(ctx.model) - - def on_evict(self, ctx: Any) -> None: - resident.pop(ctx.model, None) + evicted: list[str] = [] # what laya dropped to make room for the checkpoint it is loading + + def on_evict(ctx: Any) -> None: + resident.pop(ctx.model, None) + evicted.append(ctx.model) + + def on_load(ctx: Any) -> None: + """A checkpoint loaded while serving is prepared like the ones loaded at startup, or not kept at all; + in that case the checkpoints laya evicted for it are loaded again.""" + made_room = [name for name in evicted if name != ctx.model] + evicted.clear() + preparing.add(ctx.model) + try: + apply_options(ctx.model, ctx.agent) + check_device(list(filter(None, [make_ready(ctx.model, ctx.agent)]))) + except Exception: + log.exception("%s could not be prepared and is unloaded", ctx.model) + router.unload(ctx.model) + evicted.clear() + for name in made_room: + try: + router.load(name) + except Exception: # noqa: BLE001 -- the request fails for the first reason either way + log.exception("%s was evicted for %s and could not be loaded again", name, ctx.model) + raise + finally: + preparing.discard(ctx.model) names = list(router.loaded) or [model] # never load a model the worker was not asked to serve startup = {name: router.load(name) for name in names} for name, agent in startup.items(): apply_options(name, agent) check_device(list(filter(None, [make_ready(name, agent) for name, agent in startup.items()]))) - router.add_hook(Lifecycle()) + for hook in [h for h in getattr(router, "hooks", ()) if isinstance(h, Lifecycle)]: + router.remove_hook(hook) # an app built earlier on this router + router.add_hook(Lifecycle(on_load, on_evict)) def current() -> dict[str, Any]: """The agents as they are now, not as they were at startup (see the module docstring).""" @@ -175,6 +210,9 @@ def main() -> None: import uvicorn logging.basicConfig(level=args.log_level.upper(), format="%(name)s: %(message)s") + unread = [name for name in UNREAD_LAYA_SERVE_VARIABLES if os.environ.get(name)] + if unread: + log.warning("%s: read by laya-serve's launcher, not by this worker; use the flags", ", ".join(unread)) revisions = engine.record_snapshot_revisions() try: app = build_app( diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 68f1192..651b101 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -49,6 +49,9 @@ def load(self, name): def add_hook(self, hook): self.hooks.append(hook) + def remove_hook(self, hook): + self.hooks.remove(hook) + def unload(self, name): self.agents.pop(name, None) for hook in self.hooks: @@ -481,3 +484,57 @@ def predict_and_look(state, questions, model=None): router.load_while_serving("multilingual", FakeAgent()) assert seen[0] == ["multilingual"] assert client.get("/health").json()["preparing"] == [] + + +class StubAgent: + """Stands in for laya.agent.Agent under laya's real Router: no weights, fixed answers.""" + + devices: dict = {} # subfolder -> the device that checkpoint lands on + + def __init__(self, repo, device=None, token=None, subfolder=None): + self.device = self.devices.get(subfolder, device) + self.dtype = "torch.float16" + self.mps_amp_min_rows = 5 + + def system_one(self, state, questions, lang=None, **_): + return {"model": "stub", "answers": {qid: ANSWER for qid in questions}, "usage": {}} + + +def test_a_failed_late_load_gives_back_the_checkpoint_laya_evicted_for_it(monkeypatch): + import laya.agent + from laya.router import Router + + monkeypatch.setattr(laya.agent, "Agent", StubAgent) + monkeypatch.setattr(StubAgent, "devices", {"typed-decisions": "cpu"}) + router = Router(device="mps") + router.preload(["english", "multilingual"]) # laya keeps two checkpoints by default + client = TestClient(worker.build_app(router, "english", "mps", require_device=True)) + with pytest.raises(RuntimeError, match="typed-decisions is on cpu"): + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model="typed-decisions") + assert sorted(router.loaded) == ["english", "multilingual"] + health = client.get("/health").json() + assert set(health["models"]) == {"english", "multilingual"} + assert health["preparing"] == [] + + +def test_building_a_second_app_on_a_router_replaces_the_first_apps_hooks(): + router = FakeRouter() + worker.build_app(router, "english", "mps") + worker.build_app(router, "english", "mps") + assert len(router.hooks) == 1 + before = len(router.calls) + router.load_while_serving("multilingual", FakeAgent()) + assert len(router.calls) - before == len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS + + +def test_main_warns_about_laya_serve_variables_it_does_not_read(monkeypatch, caplog): + monkeypatch.setattr(worker, "make_router", lambda device, model: FakeRouter()) + monkeypatch.setattr(worker.logging, "basicConfig", lambda **kw: None) + monkeypatch.setattr("uvicorn.run", lambda app, **kw: None) + monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps"]) + monkeypatch.setenv("LAYA_THREADS", "4") + monkeypatch.setenv("LAYA_API_KEY", "k") + with caplog.at_level("WARNING", logger="laya-worker"): + worker.main() + assert "LAYA_THREADS" in caplog.text + assert "LAYA_API_KEY" not in caplog.text From 145088415ea566a884b4512c55e0918ee4f7ebdb Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Fri, 2 Oct 2026 11:48:28 +0800 Subject: [PATCH 17/21] Laya recipe: say what a late checkpoint load costs through the frontend; bench spawn fixes - Recipe: the first request for a late checkpoint took about 70 s with --compile on the M1 Pro, past the frontend's 60 s backend timeout (504, then fine on retry); how to avoid it; memory with two checkpoints. - /health: graphs compiled by a late checkpoint that is then unloaded are not reported as recompiles. - bench_http.py, paired.py: a process that fails to start no longer leaves the other one running. - bench_http.py: {device} and {model} in the spawn command, so --device/--model reach frontend.laya_mps. - First-pass report compares against its own C2 run. --- recipe/laya/apple-silicon.md | 20 ++++++--- recipe/laya/bench/README.md | 4 +- recipe/laya/bench/bench_http.py | 76 ++++++++++++++++++--------------- recipe/laya/bench/paired.py | 8 ++-- src/frontend/laya_mps.py | 3 ++ tests/laya/test_worker.py | 18 ++++++++ 6 files changed, 81 insertions(+), 48 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 670c933..3848cc9 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -40,11 +40,17 @@ launcher reads do not: device, model, host, port and log level are the flags abo `LAYA_AUTO_TASK` are not read. The worker warns at startup if any of them is set. Laya loads another checkpoint when a request names it (`"model": "multilingual"`) or its routing picks it -(a non-English state). The worker prepares that checkpoint the same way inside that first request, so -that request takes seconds (other requests wait behind it; `/health` names it under `preparing` -meanwhile), and `/health` lists it from then on. With `--require-device`, a checkpoint -that does not land on the requested device is unloaded again and the request fails with 500. If Laya -evicted another checkpoint to make room for it (it keeps two by default), the worker loads that one again. +(a non-English state). The worker prepares that checkpoint the same way inside that first request; other +requests wait behind it, `/health` names it under `preparing` meanwhile and lists it afterwards. On the +M1 Pro that first request took about 5–10 s without the options and about 70 s with `--compile` (plus the +download the first time, 680 MB for `multilingual`). The frontend gives a backend 60 s, so with `--compile` +it answered that request with 504 while the worker finished preparing; the same request sent again then +took 35 ms. To avoid that, send one request for each further checkpoint straight to the worker after +startup. Each resident checkpoint needs its own memory (see Troubleshooting). + +With `--require-device`, a checkpoint that does not land on the requested device is unloaded again and +the request fails with 500. If Laya evicted another checkpoint to make room for it (it keeps two by +default), the worker loads that one again. Check what it is running on: @@ -135,7 +141,7 @@ Stop the worker and frontend first; the benchmark starts its own. The scripts ar .venv/bin/python recipe/laya/bench/bench_http.py --config C4 --run feasibility \ --url http://127.0.0.1:8080 --frontend target/release/omni-jev --spawn .venv/bin/laya-serve .venv/bin/python recipe/laya/bench/paired.py --run feasibility --a "" --b "--compile --weights fp16" -.venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*_feasibility.jsonl +.venv/bin/python recipe/laya/bench/report.py recipe/laya/bench/results/*_feasibility.jsonl --ref C2 .venv/bin/python recipe/laya/bench/paired.py --summarize recipe/laya/bench/results/paired_feasibility.jsonl ``` @@ -151,5 +157,5 @@ Runs labelled anything other than `feasibility` refuse to start on battery power out-of-memory error. It keeps answering, several times slower; free memory and restart the worker to get back on the GPU. - The worker process uses about 4 GB, or 3 GB with fp16 weights (Activity Monitor's Memory column, - which counts MPS allocations). On a 16 GB Mac, close other large applications before benchmarking. + which counts MPS allocations), with one checkpoint loaded; a second one Laya loads later adds its own. On a 16 GB Mac, close other large applications before benchmarking. - `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/recipe/laya/bench/README.md b/recipe/laya/bench/README.md index 026624b..4f8d818 100644 --- a/recipe/laya/bench/README.md +++ b/recipe/laya/bench/README.md @@ -24,9 +24,9 @@ python recipe/laya/bench/bench_http.py --config C3 --run m1 --spawn .venv/bin/la python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ --frontend target/release/omni-jev --spawn .venv/bin/laya-serve python recipe/laya/bench/bench_http.py --config C3w --run m1 \ - --spawn .venv/bin/python -m frontend.laya_mps --device mps --port {port} + --spawn .venv/bin/python -m frontend.laya_mps --device {device} --model {model} --port {port} python recipe/laya/bench/bench_http.py --config C3o --run m1 \ - --spawn .venv/bin/python -m frontend.laya_mps --device mps --compile --weights fp16 --port {port} + --spawn .venv/bin/python -m frontend.laya_mps --device {device} --model {model} --compile --weights fp16 --port {port} python recipe/laya/bench/report.py recipe/laya/bench/results/*_m[0-9].jsonl --ref C1 ``` diff --git a/recipe/laya/bench/bench_http.py b/recipe/laya/bench/bench_http.py index 7f07dcc..9626157 100644 --- a/recipe/laya/bench/bench_http.py +++ b/recipe/laya/bench/bench_http.py @@ -10,10 +10,13 @@ python recipe/laya/bench/bench_http.py --config C4 --run m1 --url http://127.0.0.1:8080 \ --frontend target/release/omni-jev --spawn .venv/bin/laya-serve python recipe/laya/bench/bench_http.py --config C3o --run m1 \ - --spawn .venv/bin/python -m frontend.laya_mps --device mps --compile --weights fp16 --port {port} + --spawn .venv/bin/python -m frontend.laya_mps --device {device} --model {model} \ + --compile --weights fp16 --port {port} The spawned command gets LAYA_HOST/LAYA_PORT/LAYA_DEVICE/LAYA_MODELS in its environment (what laya-serve -reads), PYTHONPATH=src (for `-m frontend.laya_mps`), and `{port}` in its arguments replaced by the port. +reads) and PYTHONPATH=src (for `-m frontend.laya_mps`). `{port}`, `{device}` and `{model}` in its arguments +are replaced by the port and by --device and --model, which is how `frontend.laya_mps` gets them: it takes +flags and does not read those variables. """ import argparse @@ -147,7 +150,7 @@ def main(): parser.add_argument("--frontend", help="Rust frontend binary to start on --url in front of the spawned worker") parser.add_argument("--backend-port", type=int, default=8000, help="worker port when --frontend is used") parser.add_argument("--spawn", nargs=argparse.REMAINDER, help="start this worker command, then benchmark it") - parser.add_argument("--device", default="mps", help="LAYA_DEVICE for a spawned worker") + parser.add_argument("--device", default="mps", help="device for a spawned worker: LAYA_DEVICE and {device}") parser.add_argument("--ready-timeout", type=float, default=600) parser.add_argument("--workloads", default=str(HERE / "workloads.jsonl")) parser.add_argument("--only", nargs="*", help="bench workload ids to run (default: all)") @@ -177,44 +180,47 @@ def main(): Path(args.out).mkdir(parents=True, exist_ok=True) processes = {} - if args.spawn: - port = args.backend_port if args.frontend else urlsplit(args.url).port - env = { - **os.environ, - "LAYA_HOST": "127.0.0.1", - "LAYA_PORT": str(port), - "LAYA_DEVICE": args.device, - "LAYA_MODELS": args.model, - "LAYA_PRELOAD": "1", - "LAYA_LOG_LEVEL": "warning", - } - env["PYTHONPATH"] = str(REPO / "src") + (os.pathsep + env["PYTHONPATH"] if env.get("PYTHONPATH") else "") - command = [arg.replace("{port}", str(port)) for arg in args.spawn] - spawn_log = open(Path(args.out) / f"http_{args.config}_{args.run}.worker.log", "w") # noqa: SIM115 - processes["worker"] = subprocess.Popen(command, env=env, stdout=spawn_log, stderr=subprocess.STDOUT, cwd=REPO) - if args.frontend: - parts = urlsplit(args.url) - env = { - **os.environ, - "OMNI_JEV_BIND": f"{parts.hostname}:{parts.port}", - "OMNI_JEV_BACKEND_URL": f"http://127.0.0.1:{args.backend_port}", - } - frontend_log = open(Path(args.out) / f"http_{args.config}_{args.run}.frontend.log", "w") # noqa: SIM115 - processes["frontend"] = subprocess.Popen( - [args.frontend], env=env, stdout=frontend_log, stderr=subprocess.STDOUT - ) - process = processes.get("worker") - frontend = processes.get("frontend") def memory(): - mem = footprint_mb(process.pid) if process else {} - if frontend: - mem["frontend_footprint_mb"] = footprint_mb(frontend.pid).get("footprint_mb") + mem = footprint_mb(processes["worker"].pid) if "worker" in processes else {} + if "frontend" in processes: + mem["frontend_footprint_mb"] = footprint_mb(processes["frontend"].pid).get("footprint_mb") return mem out = Path(args.out) / f"http_{args.config}_{args.run}.jsonl" common = {"config": args.config, "run": args.run} try: + if args.spawn: + port = args.backend_port if args.frontend else urlsplit(args.url).port + env = { + **os.environ, + "LAYA_HOST": "127.0.0.1", + "LAYA_PORT": str(port), + "LAYA_DEVICE": args.device, + "LAYA_MODELS": args.model, + "LAYA_PRELOAD": "1", + "LAYA_LOG_LEVEL": "warning", + } + env["PYTHONPATH"] = str(REPO / "src") + (os.pathsep + env["PYTHONPATH"] if env.get("PYTHONPATH") else "") + placeholders = {"{port}": str(port), "{device}": args.device, "{model}": args.model} + command = list(args.spawn) + for placeholder, value in placeholders.items(): + command = [arg.replace(placeholder, value) for arg in command] + spawn_log = open(Path(args.out) / f"http_{args.config}_{args.run}.worker.log", "w") # noqa: SIM115 + processes["worker"] = subprocess.Popen( + command, env=env, stdout=spawn_log, stderr=subprocess.STDOUT, cwd=REPO + ) + if args.frontend: + parts = urlsplit(args.url) + env = { + **os.environ, + "OMNI_JEV_BIND": f"{parts.hostname}:{parts.port}", + "OMNI_JEV_BACKEND_URL": f"http://127.0.0.1:{args.backend_port}", + } + frontend_log = open(Path(args.out) / f"http_{args.config}_{args.run}.frontend.log", "w") # noqa: SIM115 + processes["frontend"] = subprocess.Popen( + [args.frontend], env=env, stdout=frontend_log, stderr=subprocess.STDOUT + ) ready_s, health = wait_ready(args.url, processes, args.ready_timeout) with open(out, "w") as f: @@ -257,7 +263,7 @@ def emit(record): emit( { "type": "phase", - "process_to_ready_s": round(ready_s, 3) if process else None, + "process_to_ready_s": round(ready_s, 3) if "worker" in processes else None, "warmup_s": round(warmup_s, 3), "first_ms": {k: round(v, 2) for k, v in first_ms.items()}, "routing": routing, diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index e0ac27b..bdf4d1f 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -69,12 +69,12 @@ def run(args): urls = sides else: sides = {"A": (args.a or "", args.port_a), "B": (args.b or "", args.port_b)} - procs = { - s: spawn(flags, port, args.python, args.model, out_dir / f"paired_{args.run}_{s}.log") - for s, (flags, port) in sides.items() - } + procs = {} urls = {s: f"http://127.0.0.1:{port}" for s, (_, port) in sides.items()} try: + if not (args.a_url or args.b_url): + for s, (flags, port) in sides.items(): + procs[s] = spawn(flags, port, args.python, args.model, out_dir / f"paired_{args.run}_{s}.log") health = {s: wait_ready(urls[s], {s: procs[s]} if s in procs else {}, args.ready_timeout)[1] for s in sides} with open(out_dir / f"paired_{args.run}.jsonl", "w") as f: diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py index a05eba8..1b07a56 100644 --- a/src/frontend/laya_mps.py +++ b/src/frontend/laya_mps.py @@ -115,6 +115,7 @@ def on_evict(ctx: Any) -> None: def on_load(ctx: Any) -> None: """A checkpoint loaded while serving is prepared like the ones loaded at startup, or not kept at all; in that case the checkpoints laya evicted for it are loaded again.""" + nonlocal graphs_at_ready made_room = [name for name in evicted if name != ctx.model] evicted.clear() preparing.add(ctx.model) @@ -130,6 +131,8 @@ def on_load(ctx: Any) -> None: router.load(name) except Exception: # noqa: BLE001 -- the request fails for the first reason either way log.exception("%s was evicted for %s and could not be loaded again", name, ctx.model) + if compile: + graphs_at_ready = graph_counter() # graphs the unloaded checkpoint compiled are not recompiles raise finally: preparing.discard(ctx.model) diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 651b101..4b0b1ac 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -538,3 +538,21 @@ def test_main_warns_about_laya_serve_variables_it_does_not_read(monkeypatch, cap worker.main() assert "LAYA_THREADS" in caplog.text assert "LAYA_API_KEY" not in caplog.text + + +def test_graphs_compiled_by_a_late_checkpoint_that_fails_are_not_reported_as_recompiles(monkeypatch): + monkeypatch.setattr(optimize, "compile_agent", lambda agent: None) + router = FakeRouter() + graphs = {"n": 0} + predict = router.predict + + def predict_and_compile(state, questions, model=None): + graphs["n"] += 1 + return predict(state, questions, model=model) + + router.predict = predict_and_compile + client = TestClient(worker.build_app(router, "english", "mps", compile=True, graph_counter=lambda: graphs["n"])) + router.fail_on_call = len(router.calls) + 3 + with pytest.raises(RuntimeError, match="out of memory"): + router.load_while_serving("multilingual", FakeAgent()) + assert client.get("/health").json()["compile"]["recompiled_after_ready"] is False From 543acef57ddde36eba84e50c05c954d23e1353d7 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:22:53 +0800 Subject: [PATCH 18/21] Laya worker: drop comments that repeat the code or the test name --- src/models/laya/engine.py | 2 +- src/models/laya/optimize.py | 3 +-- tests/laya/test_worker.py | 20 ++++++++++---------- 3 files changed, 12 insertions(+), 13 deletions(-) diff --git a/src/models/laya/engine.py b/src/models/laya/engine.py index 0be5d1b..4bf008b 100644 --- a/src/models/laya/engine.py +++ b/src/models/laya/engine.py @@ -31,7 +31,7 @@ def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_REPEATS) -> dict[str, Any]: - """Run every shape `repeats` times. Any failure propagates: a worker that cannot answer must not bind.""" + """Any failure propagates: a worker that cannot answer must not bind.""" started = time.perf_counter() routing = None for words, questions in shapes: diff --git a/src/models/laya/optimize.py b/src/models/laya/optimize.py index 6283be1..2b6d369 100644 --- a/src/models/laya/optimize.py +++ b/src/models/laya/optimize.py @@ -90,7 +90,7 @@ def _on_cpu(agent: Any) -> bool: def apply(agent: Any, *, fp16: bool, compile: bool) -> bool: - """Apply the requested options to one agent. Does nothing and returns False for a model on the CPU.""" + """Does nothing and returns False for a model on the CPU.""" if _on_cpu(agent): return False if fp16: @@ -106,7 +106,6 @@ def compile_active(agent: Any) -> bool: def compiled_graphs() -> int: - """Graphs torch.compile has produced in this process.""" from torch._dynamo.utils import counters return int(counters["stats"]["unique_graphs"]) diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 4b0b1ac..7b9e6e2 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -81,7 +81,7 @@ def test_warmup_covers_short_long_and_fp16_multi_question_shapes(): words = {len(state.split()) for state, _, _ in router.calls} rows = {len(questions) for _, questions, _ in router.calls} assert min(words) <= 20 and max(words) >= 400 - assert max(rows) >= router.agent.mps_amp_min_rows # crosses laya's fp16 autocast threshold on MPS + assert max(rows) >= router.agent.mps_amp_min_rows assert {q["type"] for _, questions, _ in router.calls for q in questions.values()} == {"choice", "score", "noul"} assert len(router.calls) == len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS assert all(model == "english" for _, _, model in router.calls) @@ -165,7 +165,7 @@ def test_compile_wraps_the_model_before_warmup(monkeypatch): order = [] monkeypatch.setattr(optimize, "compile_agent", lambda agent: order.append((agent, len(router.calls)))) worker.build_app(router, "english", "mps", compile=True, graph_counter=lambda: 3) - assert order == [(router.agent, 0)] # before the first warmup request + assert order == [(router.agent, 0)] def test_health_reports_compile_off_by_default(): @@ -232,9 +232,9 @@ def fake_compile(module, dynamic): model = Model() agent.model = model optimize.compile_agent(agent) - assert agent.model(on_gpu(1)) == "whole model compiled" # one question - assert agent.model(on_gpu(3)) == ("head", "compiled encoder") # several: eager head, compiled encoder - assert agent.model(torch.zeros(1, 7)) == ("head", "eager encoder") # on the CPU: laya's model as it is + assert agent.model(on_gpu(1)) == "whole model compiled" + assert agent.model(on_gpu(3)) == ("head", "compiled encoder") + assert agent.model(torch.zeros(1, 7)) == ("head", "eager encoder") assert model.encoder(torch.zeros(1, 7)) == "eager encoder" # the original model is left as it was assert len(list(agent.model.parameters())) == len(list(model.parameters())) # one set of weights @@ -249,7 +249,7 @@ def test_every_loaded_model_is_warmed_and_described(): assert set(health["models"]) == {"english", "multilingual"} assert health["device"] == "mps" # the top level summarises --model assert health["models"]["multilingual"]["device"] == "cpu" - assert health["device_mismatch"] is True # one model off the requested device is enough + assert health["device_mismatch"] is True def test_a_model_that_is_not_preloaded_is_not_loaded_for_warmup(): @@ -352,7 +352,7 @@ def forward(self, input_ids): optimize.compile_agent(agent) assert agent.model(on_gpu(1)) == "compiled" assert next(agent.model.parameters()).dtype == torch.float16 - assert agent.model(torch.zeros(1, 7)) == ("eager", torch.float32) # inputs on the CPU: laya fell back + assert agent.model(torch.zeros(1, 7)) == ("eager", torch.float32) assert {p.dtype for p in agent.model.parameters()} == {torch.float32} assert agent.model(torch.zeros(3, 7)) == ("eager", torch.float32) @@ -361,7 +361,7 @@ def test_health_follows_a_fallback_to_cpu_after_startup(): agent = FakeAgent(device="mps") client = TestClient(worker.build_app(FakeRouter(agent), "english", "mps", require_device=True)) assert client.get("/health").json()["device_mismatch"] is False - agent.device = "cpu" # what laya does when a request runs out of GPU memory + agent.device = "cpu" agent.dtype = "torch.float32" health = client.get("/health").json() assert health["device"] == "cpu" @@ -425,7 +425,7 @@ def test_a_checkpoint_loaded_while_serving_is_prepared_and_described(monkeypatch assert health["models"]["multilingual"]["device"] == "cpu" assert health["device_mismatch"] is True assert health["preparing"] == [] - router.unload("multilingual") # eviction + router.unload("multilingual") health = client.get("/health").json() assert set(health["models"]) == {"english"} assert health["device_mismatch"] is False @@ -439,7 +439,7 @@ def test_a_late_checkpoint_that_cannot_be_prepared_is_not_kept(): assert router.loaded == ["english"] assert set(client.get("/health").json()["models"]) == {"english"} router.fail_on_call = len(router.calls) + 1 - with pytest.raises(RuntimeError, match="out of memory"): # warmup fails + with pytest.raises(RuntimeError, match="out of memory"): router.load_while_serving("multilingual", FakeAgent()) assert router.loaded == ["english"] From f2b7c6ad12d643ce3e981e6f3cbe94437471429c Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Fri, 2 Oct 2026 19:00:36 +0800 Subject: [PATCH 19/21] Laya: third review fixes, requirement-derived tests on CPU and GPU, and what the warm numbers leave out Worker and benchmark scripts - /health keeps each checkpoint's revision from its own load: revisions are recorded per checkpoint (repo or repo/subfolder) and read once when the checkpoint has been prepared. - report.py: the parity section lists every run with an env record; a run whose answer probes all failed, or an answer missing a question, shows as failures instead of disappearing or raising. - paired.py --summarize: a run without an end record, a side without answers or a question missing on one side is reported instead of raising. Tests - Contract tests: answers match Laya run directly (choice, score, noul, combined, six questions), the port opens only after the warmup, the checkpoint stays loaded; LAYA_CONTRACT_DEVICE=mps and LAYA_CONTRACT_FLAGS run the same contract on the GPU, without and with the options. - /health is checked against Laya's real Router after random sequences of loads, failed loads and evictions. - Unit tests from the issue's requirements, and for behaviours a mutation run showed were executed but not asserted (late checkpoints get the GPU options before their warmup, logging, defaults, /health values). Recipe - What back-to-back latencies leave out, measured on the M1 Pro: a request after 2-5 s of idle takes 105-115 ms against 25 ms; a new input length costs about 15 ms once; with both options the footprint grows about 5 MB per input length seen, to 5.3 GB after all of them. - Checkpoint download: 97 s for 846 MB. Contract tests on the GPU. --- recipe/laya/apple-silicon.md | 37 +++++- recipe/laya/bench/paired.py | 16 ++- recipe/laya/bench/report.py | 20 ++- src/frontend/laya_mps.py | 8 +- src/models/laya/engine.py | 29 ++++- tests/laya/test_bench.py | 87 +++++++++++++ tests/laya/test_contract.py | 90 +++++++++++-- tests/laya/test_worker.py | 242 +++++++++++++++++++++++++++++++++-- 8 files changed, 482 insertions(+), 47 deletions(-) create mode 100644 tests/laya/test_bench.py diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index 3848cc9..be6d824 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -30,7 +30,8 @@ The last command must print `True`. The standard macOS arm64 wheel of torch incl PYTHONPATH=src .venv/bin/python -m frontend.laya_mps --device mps --model english --require-device --port 8000 ``` -First startup downloads the checkpoint (about 850 MB). The worker loads the model, runs a warmup +First startup downloads the checkpoint (846 MB; 97 s into an empty cache at 8.7 MB/s when measured). The +worker loads the model, runs a warmup over short, long and multi-question requests, and only then listens on port 8000, so the first request it accepts is already warm: on an M1 Pro the first request after ready took 70–81 ms, against 0.7–1.1 s from plain laya-serve. `--require-device` makes it exit instead of silently serving on the CPU @@ -82,7 +83,7 @@ together lowered warm p50 against the worker without them by 37–38% for a 68-t request (about 57 → 35 ms in those runs), 17–20% at 198–484 tokens, 14% for three questions and 18% for six. Answers stayed within 0.0031 of the fp32 worker's. A worker running on its own uses about 3 GB with the options instead of 4.2 GB (2.8 GB against 3.5 GB in those paired runs, where the two workers -shared the machine). The price is startup: the worker became ready after 35–39 s instead of 8–10 s, and +shared the machine), measured on the six benchmark inputs; see below for how it grows. The price is startup: the worker became ready after 35–39 s instead of 8–10 s, and its first request after that took 62–78 ms. On an M5 the same paired comparison gave median ratios of 0.51–0.53 for one-question requests at @@ -91,6 +92,24 @@ the fp16 weights, which on that GPU speed up every input even without compile. T ready after 19 s instead of 3 s; its first request took 21–36 ms in 21 of 23 fresh starts and 327 and 409 ms in the other two, not yet explained (132–143 ms from plain laya-serve). +### What the warm numbers leave out + +The latencies above are for requests sent back to back. Measured on the M1 Pro: + +- **Idle gaps.** A request that follows a pause is slower, with or without the options, because the GPU + has slowed down in the meantime. For a short one-question request (25 ms back to back with the options, + 40 ms without) it took about 50 ms after 0.2–1 s of idle and 105–115 ms after 2–5 s (60–68 ms and + 114–127 ms without the options). This is also why the first request after ready costs more than a warm + one. An agent that asks once every few seconds sees these numbers, not the back-to-back ones. A + heartbeat of one forward pass every 0.5 s held it at about 45 ms in a probe, for 8% GPU load; the worker + does not do this. +- **New input lengths.** The first request of a length the worker has not seen costs about 15 ms more + once with the options (6 ms without). It is not a recompile (`recompiled_after_ready` stays `false`). +- **Memory grows with the lengths seen.** With both options, what PyTorch caches per input length adds + about 5 MB each: the 3 GB below became 3.3 GB after 100 new lengths and 5.3 GB after all 477, more than + the 4.0 GB of a worker without the options, whose footprint did not grow. `torch.mps.empty_cache()` + releases it, and those lengths then pay their first-request cost again. + `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. `active` is `false` once no model runs the compiled path any more, i.e. after a fallback to the CPU. @@ -130,6 +149,16 @@ PYTHONPATH=src .venv/bin/python -m pytest tests/laya # unit t LAYA_CONTRACT=1 PYTHONPATH=src .venv/bin/python -m pytest tests/laya # plus contract tests against a CPU worker ``` +The contract tests start a real worker and check readiness, the three decision types, error responses, and +that its answers match Laya run directly in fp32 on the CPU. On an Apple Silicon Mac, run them against the +GPU as well, without and with the options: + +```sh +LAYA_CONTRACT=1 LAYA_CONTRACT_DEVICE=mps PYTHONPATH=src .venv/bin/python -m pytest tests/laya/test_contract.py +LAYA_CONTRACT=1 LAYA_CONTRACT_DEVICE=mps LAYA_CONTRACT_FLAGS="--compile --weights fp16" \ + PYTHONPATH=src .venv/bin/python -m pytest tests/laya/test_contract.py +``` + ## Benchmark Stop the worker and frontend first; the benchmark starts its own. The scripts are listed in @@ -156,6 +185,6 @@ Runs labelled anything other than `feasibility` refuse to start on battery power - `device_mismatch: true` on a worker that started on MPS: Laya fell back to the CPU after a GPU out-of-memory error. It keeps answering, several times slower; free memory and restart the worker to get back on the GPU. -- The worker process uses about 4 GB, or 3 GB with fp16 weights (Activity Monitor's Memory column, - which counts MPS allocations), with one checkpoint loaded; a second one Laya loads later adds its own. On a 16 GB Mac, close other large applications before benchmarking. +- The worker process uses about 4 GB, or 3 GB with fp16 weights growing towards 5 GB as it sees more input + lengths (Activity Monitor's Memory column, which counts MPS allocations), with one checkpoint loaded; a second one Laya loads later adds its own. On a 16 GB Mac, close other large applications before benchmarking. - `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index bdf4d1f..2f4c580 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -192,15 +192,22 @@ def summarize(paths): answers.setdefault(r["workload"], {})[r["side"]] = r worst, flips, errors = 0.0, [], [] for wid, sides in answers.items(): - if sides["A"]["error"] or sides["B"]["error"]: + if any(sides.get(s, {"error": "missing"}).get("error") for s in "AB"): errors.append(wid) continue for q, a in sides["A"]["answers"].items(): - b = sides["B"]["answers"][q] - worst = max(worst, max(abs(flat(a)[k] - flat(b)[k]) for k in flat(a))) + b = sides["B"]["answers"].get(q) + if b is None: + errors.append(f"{wid}/{q}") + continue + worst = max(worst, max(abs(flat(a)[k] - flat(b).get(k, 0.0)) for k in flat(a))) if decision(a) != decision(b): flips.append((wid, q, round(margin(a), 4))) - end = next(r for r in records if r["type"] == "end") + print(f"\nB vs A answers: max |Δp| {worst:.4f}, flips {flips}, errors {errors}; failed pairs: {failed}") + end = next((r for r in records if r["type"] == "end"), None) + if end is None: + print("**The run did not finish: no end record, so no device, recompile or memory check.**\n") + continue compile_state = {s: h.get("compile", {}).get("recompiled_after_ready") for s, h in end["health"].items()} devices = {s: h.get("device") for s, h in end["health"].items()} off_gpu = [ @@ -209,7 +216,6 @@ def summarize(paths): if h.get("device_mismatch") or (h.get("compile", {}).get("enabled") and not h["compile"].get("active", True)) ] - print(f"\nB vs A answers: max |Δp| {worst:.4f}, flips {flips}, errors {errors}; failed pairs: {failed}") print(f"recompiled after ready: {compile_state}; device at end: {devices}; footprint MB: {end['footprint_mb']}") if off_gpu: print( diff --git a/recipe/laya/bench/report.py b/recipe/laya/bench/report.py index 54274f1..7505afc 100644 --- a/recipe/laya/bench/report.py +++ b/recipe/laya/bench/report.py @@ -177,14 +177,17 @@ def memory(phases, ends): def read_answers(records): - envs, answers = {}, {} + """Per (config, run): the env record, the answers per workload, and the probes that failed.""" + envs, answers, errors = {}, {}, {} for r in records: key = (r["config"], r["run"]) if r["type"] == "env": envs[key] = r elif r["type"] == "answers": answers.setdefault(key, {})[r["workload"]] = r["answers"] - return envs, answers + elif r["type"] == "answers_error": + errors.setdefault(key, {})[r["workload"]] = f"status {r.get('status')}: {r.get('detail')}" + return envs, answers, errors def outcome(answer): @@ -217,7 +220,7 @@ def path(env, rows): def parity(records, ref): """Answers of every run against the reference config, with the tolerances declared in advance.""" - envs, answers = read_answers(records) + envs, answers, errors = read_answers(records) refs = sorted(k for k in answers if k[0] == ref) if not refs: return f"no answers for reference config {ref}" @@ -230,20 +233,25 @@ def parity(records, ref): "|---|---|---|---|---|---|---|---|---|---|", ] failed = total = 0 - for key in sorted(answers): + for key in sorted(envs.keys() | answers.keys() | errors.keys()): # a run with no answers at all still counts if key == ref_key: continue env = envs.get(key, {}) for workload, questions in sorted(reference.items()): - got = answers[key].get(workload) + got = answers.get(key, {}).get(workload) if got is None: - lines.append(f"| {key[0]} | {key[1]} | {workload} | | | | missing | | | FAIL |") + why = errors.get(key, {}).get(workload, "missing") + lines.append(f"| {key[0]} | {key[1]} | {workload} | | | | {why} | | | FAIL |") failed += 1 total += 1 continue precision = path(env, len(questions)) for qid, ref_answer in sorted(questions.items()): total += 1 + if qid not in got: + lines.append(f"| {key[0]} | {key[1]} | {workload} | {qid} | | | missing | | | FAIL |") + failed += 1 + continue ref_decision, ref_probs = outcome(ref_answer) decision, probs = outcome(got[qid]) delta = max(abs(ref_probs.get(o, 0.0) - probs.get(o, 0.0)) for o in ref_probs.keys() | probs.keys()) diff --git a/src/frontend/laya_mps.py b/src/frontend/laya_mps.py index 1b07a56..c29eda4 100644 --- a/src/frontend/laya_mps.py +++ b/src/frontend/laya_mps.py @@ -84,6 +84,7 @@ def make_ready(name: str, agent: Any) -> str | None: """Warm the checkpoint up and start describing it. Returns where it is if not on the requested device.""" nonlocal graphs_at_ready warmed = engine.warmup(router, name) + warmed["revision"] = engine.loaded_revision(revisions, warmed["routing"]) # fixed here: see engine resident[name] = (agent, warmed) autocast_rows = getattr(agent, "mps_amp_min_rows", None) if str(agent.device).startswith("mps") and autocast_rows and autocast_rows > engine.WARMUP_MAX_ROWS: @@ -96,7 +97,7 @@ def make_ready(name: str, agent: Any) -> str | None: ) if compile: graphs_at_ready = graph_counter() - described = engine.describe(agent, requested, warmed["routing"], revisions) + described = engine.describe(agent, requested, warmed["routing"], warmed["revision"]) return f"{name} is on {described['device']}" if described["device_mismatch"] else None def check_device(misplaced: list[str]) -> None: @@ -149,7 +150,10 @@ def on_load(ctx: Any) -> None: def current() -> dict[str, Any]: """The agents as they are now, not as they were at startup (see the module docstring).""" models = { - name: {**engine.describe(agent, requested, warmed["routing"], revisions), "warmup_ms": warmed["warmup_ms"]} + name: { + **engine.describe(agent, requested, warmed["routing"], warmed["revision"]), + "warmup_ms": warmed["warmup_ms"], + } for name, (agent, warmed) in list(resident.items()) } return { diff --git a/src/models/laya/engine.py b/src/models/laya/engine.py index 4bf008b..2ec8cd5 100644 --- a/src/models/laya/engine.py +++ b/src/models/laya/engine.py @@ -42,12 +42,24 @@ def warmup(router: Any, model: str, shapes=WARMUP_SHAPES, repeats: int = WARMUP_ return {"warmup_ms": round((time.perf_counter() - started) * 1000, 1), "routing": routing} +def _checkpoint_name(repo_id: str, allow_patterns: Any) -> str: + """laya's name for what one download fetched: "", or "/" for a bundled checkpoint. + laya restricts each download to one checkpoint's files, which all sit under its subfolder if it has one.""" + patterns = [allow_patterns] if isinstance(allow_patterns, str) else list(allow_patterns or [""]) + folders = {pattern.split("/")[0] if "/" in pattern else "" for pattern in patterns} + subfolder = folders.pop() if len(folders) == 1 else "" + return f"{repo_id}/{subfolder}" if subfolder else repo_id + + def record_snapshot_revisions() -> dict[str, str]: - """Record the commit each Hugging Face checkpoint is loaded from, keyed by repo id. + """Record the commit each Hugging Face checkpoint was last downloaded at, keyed by laya's name for it + (`routing["repo"]`). laya calls huggingface_hub.snapshot_download while loading and keeps only the repo id; the returned path (.../snapshots//...) is the only place the loaded revision appears. Call this - before the router loads anything. A checkpoint loaded from a local path records nothing. + before the router loads anything. A checkpoint loaded from a local path records nothing. An entry + changes when the same checkpoint is downloaded again, so read it with `loaded_revision` right after a + checkpoint has loaded and keep that value. """ import huggingface_hub @@ -58,15 +70,21 @@ def recording(repo_id, *args, **kwargs): path = original(repo_id, *args, **kwargs) parts = Path(path).parts if "snapshots" in parts[:-1]: - revisions[repo_id] = parts[parts.index("snapshots") + 1] + name = _checkpoint_name(repo_id, kwargs.get("allow_patterns")) + revisions[name] = parts[parts.index("snapshots") + 1] return path huggingface_hub.snapshot_download = recording return revisions +def loaded_revision(revisions: dict[str, str] | None, routing: dict[str, Any] | None) -> str | None: + """The commit of the checkpoint that has just loaded, if it was downloaded.""" + return (revisions or {}).get((routing or {}).get("repo")) + + def describe( - agent: Any, requested: str | None, routing: dict[str, Any] | None, revisions: dict[str, str] | None = None + agent: Any, requested: str | None, routing: dict[str, Any] | None, revision: str | None = None ) -> dict[str, Any]: """What /health reports about one loaded agent, read from the agent as it is now.""" device = str(getattr(agent, "device", "unknown")) @@ -76,9 +94,6 @@ def describe( except (AttributeError, StopIteration, TypeError): weights = None repo = (routing or {}).get("repo") - # laya names a bundled checkpoint "//"; the download is recorded under "/". - revisions = revisions or {} - revision = revisions.get(repo) or revisions.get("/".join(str(repo).split("/")[:2])) requested_type = requested.split(":")[0] if requested else None return { "device": device, diff --git a/tests/laya/test_bench.py b/tests/laya/test_bench.py new file mode 100644 index 0000000..b448aec --- /dev/null +++ b/tests/laya/test_bench.py @@ -0,0 +1,87 @@ +"""Checks on the benchmark inputs, the run header and the documented commands in recipe/laya. No model.""" + +import json +import subprocess +import sys +from pathlib import Path + +REPO = Path(__file__).resolve().parents[2] +BENCH = REPO / "recipe/laya/bench" +sys.path.insert(0, str(BENCH)) + +import env as bench_env # noqa: E402 + + +def rows(path): + return [json.loads(line) for line in Path(path).read_text().splitlines()] + + +def test_header_records_what_makes_two_runs_comparable(monkeypatch): + record = bench_env.header("some/repo", extra=1) + assert record["type"] == "env" and record["extra"] == 1 and record["checkpoint"] == "some/repo" + head = subprocess.run(["git", "-C", str(REPO), "rev-parse", "HEAD"], capture_output=True, text=True).stdout.strip() + assert record["omni_sha"] == head and isinstance(record["omni_dirty"], bool) + import laya # noqa: F401 + from importlib.metadata import version + + assert (record["laya"], record["torch"], record["transformers"]) == tuple( + version(package) for package in ("laya", "torch", "transformers") + ) + assert record["argv"] == sys.argv and record["python"] == ".".join(map(str, sys.version_info[:3])) + assert record["power"] and record["loadavg_1m"] >= 0 and record["chip"] and record["mem_gb"] > 0 + assert record["utc"].endswith("+00:00") + + +def test_the_fixed_inputs_span_question_types_lengths_and_option_counts(): + workloads = rows(BENCH / "workloads.jsonl") + assert len({w["id"] for w in workloads}) == len(workloads) + questions = [q for w in workloads for q in w["questions"].values()] + assert {q["type"] for q in questions} == {"choice", "score", "noul"} + assert {len(q["criteria"]) for q in questions if q["type"] == "choice"} >= {2, 5, 10} + bench = [w for w in workloads if w["kind"] == "bench"] + words = sorted(len(w["state"].split()) for w in bench) + assert ( + words[0] < 20 and 100 < max(w for w in words if w < 200) and words[-1] > 300 + ) # short, medium, near the window + assert {len(w["questions"]) for w in bench} >= {1, 3, 6} # below and above laya's autocast threshold of 5 rows + parity = [w for w in workloads if w["kind"] == "parity"] + assert {q["type"] for w in parity for q in w["questions"].values()} == {"choice", "score", "noul"} + + +DOCUMENTS = ["recipe/laya/apple-silicon.md", "recipe/laya/bench/README.md", "src/models/laya/README.md"] +SCRIPTS = { + "frontend.laya_mps": "src/frontend/laya_mps.py", + **{f"recipe/laya/bench/{name}.py": f"recipe/laya/bench/{name}.py" + for name in ("bench_http", "bench_inproc", "paired", "profile_mps", "report")}, +} # fmt: skip + + +def documented_commands(): + import re + + for document in DOCUMENTS: + text = (REPO / document).read_text() + for block in re.findall(r"```sh\n(.*?)```", text, flags=re.S): + for command in block.replace("\\\n", " ").splitlines(): + for name, source in SCRIPTS.items(): + if name in command: + yield document, command, name, source + + +def test_documented_commands_use_flags_and_files_that_exist(): + import re + + commands = list(documented_commands()) + assert len(commands) >= 15 + for document, command, name, source in commands: + assert (REPO / source).exists(), f"{document}: {source}" + options = set(re.findall(r'add_argument\(\s*"(--?[a-z][a-z-]*)"', (REPO / source).read_text())) + ours = command.split("--spawn")[0] if name.startswith("recipe") else command.split(name, 1)[1] + if name == "recipe/laya/bench/paired.py": + ours = re.sub(r'"[^"]*"', "", ours) # --a/--b carry flags of the worker, checked below + for worker_flags in re.findall(r'--[ab] "([^"]*)"', command): + assert set(re.findall(r"--[a-z-]+", worker_flags)) <= set( + re.findall(r'add_argument\(\s*"(--[a-z-]+)"', (REPO / SCRIPTS["frontend.laya_mps"]).read_text()) + ), f"{document}: {command}" + used = set(re.findall(r"(? 0 # nothing listened while the model loaded and warmed up + assert startup["first_health"]["ready"] is True and startup["first_health"]["warmup_ms"] > 0 + + +def test_the_checkpoint_stays_loaded_across_requests(worker): + port, _ = worker + for _ in range(5): + assert decide(port, {"q": NOUL})[0] == 200 + health = json.loads(call(port, "GET", "/health")[1]) + assert health["loaded"] == ["english"] and health["warmup_ms"] == startup["first_health"]["warmup_ms"] -def test_first_request_after_ready_is_warm(worker): +def test_first_request_after_ready_is_not_a_cold_start(worker): port, (status, _, first_ms) = worker assert status == 200 - warm = [decide(port, {"q": CHOICE})[2] for _ in range(20)] - assert first_ms <= 2 * statistics.median(warm), f"first {first_ms:.0f} ms, warm p50 {statistics.median(warm):.0f}" + warm = statistics.median(decide(port, {"q": CHOICE})[2] for _ in range(20)) + # Without the warmup the first request costs several hundred ms more than a warm one. A few tens of ms + # remain on MPS: the GPU has been idle since the warmup, and any request after a pause pays that. + assert first_ms <= warm + 100, f"first {first_ms:.0f} ms, warm p50 {warm:.0f} ms" @pytest.mark.parametrize( @@ -142,6 +183,37 @@ def test_decisions(worker, questions, kinds): assert answer["choice"] in CHOICE["criteria"] +SIX = {f"q{i}": q for i, q in enumerate([CHOICE, SCORE, NOUL, CHOICE, SCORE, NOUL])} + + +@pytest.mark.parametrize( + "questions", + [{"q": CHOICE}, {"q": SCORE}, {"q": NOUL}, {"a": CHOICE, "b": SCORE, "c": NOUL}, SIX], + ids=["choice", "score", "noul", "combined", "six-questions"], +) +def test_answers_match_laya_itself(worker, reference, questions): + port, _ = worker + status, body, _ = decide(port, questions) + assert status == 200 + served = json.loads(body) + expected = reference.system_one(STATE, questions) + assert set(expected) <= set(served) # the worker adds `routing`, it drops nothing + assert served["usage"] == expected["usage"] + reduced = "fp16" in FLAGS or (DEVICE == "mps" and len(questions) >= startup["first_health"]["mps_amp_min_rows"]) + tolerance = 1e-2 if reduced else 1e-3 + for qid, want in expected["answers"].items(): + got = served["answers"][qid] + assert set(got) == set(want) + if want["type"] == "noul": + assert got["noul"] == pytest.approx(want["noul"], abs=tolerance) + continue + assert set(got["probabilities"]) == set(want["probabilities"]) + for option, p in want["probabilities"].items(): + assert got["probabilities"][option] == pytest.approx(p, abs=tolerance) + if want["type"] == "choice": + assert got["choice"] == want["choice"] == max(got["probabilities"], key=got["probabilities"].get) + + def test_same_request_same_answer(worker): port, _ = worker first, second = (json.loads(decide(port, {"a": CHOICE, "b": NOUL})[1])["answers"] for _ in range(2)) diff --git a/tests/laya/test_worker.py b/tests/laya/test_worker.py index 7b9e6e2..b8f81f7 100644 --- a/tests/laya/test_worker.py +++ b/tests/laya/test_worker.py @@ -4,6 +4,7 @@ """ import sys +from pathlib import Path from types import SimpleNamespace import pytest @@ -152,12 +153,13 @@ def test_decisions_still_go_through_laya_serve(): assert len(router.calls) == before + 1 -def test_main_exits_non_zero_when_warmup_fails(monkeypatch): +def test_main_exits_non_zero_when_warmup_fails(monkeypatch, caplog): monkeypatch.setattr(worker, "make_router", lambda device, model: FakeRouter(fail_on_call=1)) monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps"]) monkeypatch.setattr("uvicorn.run", lambda *a, **k: pytest.fail("must not bind")) - with pytest.raises(SystemExit, match="not starting"): + with caplog.at_level("ERROR", logger="laya-worker"), pytest.raises(SystemExit, match="not starting"): worker.main() + assert "startup failed" in caplog.text and "Traceback" in caplog.text and "out of memory" in caplog.text def test_compile_wraps_the_model_before_warmup(monkeypatch): @@ -289,6 +291,11 @@ def test_record_snapshot_revisions_reads_the_downloaded_path(monkeypatch): ) huggingface_hub.snapshot_download("/local/checkpoint") assert revisions == {"convaiinnovations/laya": "55cf4c4abc"} + huggingface_hub.snapshot_download( + "convaiinnovations/laya", allow_patterns=["multilingual/model.safetensors", "multilingual/tokenizer/*"] + ) + huggingface_hub.snapshot_download("convaiinnovations/laya", allow_patterns=["model.safetensors", "tokenizer/*"]) + assert set(revisions) == {"convaiinnovations/laya", "convaiinnovations/laya/multilingual"} def test_fp16_weights_keep_act_head_in_fp32(): @@ -402,10 +409,11 @@ def test_log_level_applies_to_the_workers_own_log(monkeypatch): seen = {} monkeypatch.setattr(worker, "make_router", lambda device, model: FakeRouter()) monkeypatch.setattr(worker.logging, "basicConfig", lambda **kw: seen.update(kw)) - monkeypatch.setattr("uvicorn.run", lambda app, **kw: seen.update(uvicorn=kw["log_level"])) + monkeypatch.setattr("uvicorn.run", lambda app, **kw: seen.update(uvicorn=kw)) monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps", "--log-level", "warning"]) worker.main() - assert (seen["level"], seen["uvicorn"]) == ("WARNING", "warning") + assert (seen["level"], seen["uvicorn"]["log_level"]) == ("WARNING", "warning") + assert (seen["uvicorn"]["host"], seen["uvicorn"]["port"]) == ("127.0.0.1", 8000) # local only by default def test_a_checkpoint_loaded_while_serving_is_prepared_and_described(monkeypatch): @@ -454,16 +462,6 @@ def test_compile_baseline_moves_when_a_late_checkpoint_compiles(monkeypatch): assert (compiled["graphs_at_ready"], compiled["recompiled_after_ready"]) == (7, False) -def test_revision_of_a_bundled_checkpoint_is_found_under_the_download_repo(): - routing = {"repo": "convaiinnovations/laya/multilingual"} - described = engine.describe(FakeAgent(), "mps", routing, {"convaiinnovations/laya": "abc123"}) - assert (described["checkpoint"], described["revision"]) == ("convaiinnovations/laya/multilingual", "abc123") - assert ( - engine.describe(FakeAgent(), "mps", {"repo": "/models/laya"}, {"convaiinnovations/laya": "abc123"})["revision"] - is None - ) - - def test_startup_error_names_every_checkpoint_off_the_requested_device(): agents = {"english": FakeAgent(device="cpu"), "multilingual": FakeAgent(device="cpu")} with pytest.raises(RuntimeError, match="asked for mps, english is on cpu, multilingual is on cpu"): @@ -556,3 +554,219 @@ def predict_and_compile(state, questions, model=None): with pytest.raises(RuntimeError, match="out of memory"): router.load_while_serving("multilingual", FakeAgent()) assert client.get("/health").json()["compile"]["recompiled_after_ready"] is False + + +class Repository: + """What the downloading stub agents see: the repository's current commit, and where checkpoints land.""" + + commit = 0 + device: dict = {} # subfolder -> device + broken: set = set() # subfolders whose forward raises + + +class DownloadingAgent: + """laya.agent.Agent without weights. It downloads the way laya does, so the worker's recording sees it.""" + + def __init__(self, repo, device=None, token=None, subfolder=None): + import huggingface_hub + + prefix = f"{subfolder}/" if subfolder else "" + path = huggingface_hub.snapshot_download( + repo, token=token, allow_patterns=[prefix + "model.safetensors", prefix + "tokenizer/*"] + ) + self.loaded_commit = Path(path).name + self.subfolder = subfolder + self.device = Repository.device.get(subfolder, device) + self.dtype = "torch.float16" + self.mps_amp_min_rows = 5 + + def system_one(self, state, questions, lang=None, **_): + if self.subfolder in Repository.broken: + raise RuntimeError("MPS backend out of memory") + return {"model": "stub", "answers": {qid: ANSWER for qid in questions}, "usage": {}} + + +@pytest.fixture +def laya_router(monkeypatch): + """laya's real Router over downloading stub agents, and the revisions the worker records for them.""" + import huggingface_hub + import laya.agent + from laya.router import Router + + monkeypatch.setattr(Repository, "commit", 0) + monkeypatch.setattr(Repository, "device", {}) + monkeypatch.setattr(Repository, "broken", set()) + monkeypatch.setattr( + huggingface_hub, "snapshot_download", lambda repo, **kw: f"/hf/snapshots/commit-{Repository.commit}" + ) + monkeypatch.setattr(laya.agent, "Agent", DownloadingAgent) + return Router(device="mps"), engine.record_snapshot_revisions() + + +def assert_health_matches(client, router): + health = client.get("/health").json() + agents = dict(router._agents) + assert set(health["models"]) == set(router.loaded) == set(agents) + assert health["preparing"] == [] + for name, agent in agents.items(): + assert health["models"][name]["device"] == str(agent.device) + assert health["models"][name]["revision"] == agent.loaded_commit + assert health["device_mismatch"] == any(str(agent.device) != "mps" for agent in agents.values()) + + +def test_checkpoints_of_one_repository_keep_the_revision_of_their_own_load(laya_router): + router, revisions = laya_router + router.load("english") + Repository.commit = 1 # the repository moves on between the two loads + router.load("multilingual") + client = TestClient(worker.build_app(router, "english", "mps", revisions=revisions)) + assert_health_matches(client, router) + Repository.commit = 2 + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model="typed-decisions") + assert_health_matches(client, router) + + +@pytest.mark.parametrize("require_device", [False, True]) +@pytest.mark.parametrize("seed", range(8)) +def test_health_matches_the_router_after_any_sequence_of_loads(laya_router, seed, require_device): + import random + + router, revisions = laya_router + names = ["english", "multilingual", "typed-decisions"] + subfolder = {"english": None, "multilingual": "multilingual", "typed-decisions": "typed-decisions"} + rng = random.Random(seed) + router.preload(rng.sample(names, rng.choice([1, 2]))) + client = TestClient( + worker.build_app(router, router.loaded[0], "mps", require_device=require_device, revisions=revisions) + ) + assert_health_matches(client, router) + for _ in range(12): + Repository.commit += rng.random() < 0.3 + Repository.device = {subfolder[n]: "cpu" for n in names if rng.random() < 0.25} + Repository.broken = {subfolder[n] for n in names if rng.random() < 0.15} + try: + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model=rng.choice(names)) + except RuntimeError: + pass # a failed late load or forward: the request fails, /health must still match the router + Repository.broken = set() + assert_health_matches(client, router) + + +def test_health_answers_with_no_resident_checkpoint(laya_router): + router, revisions = laya_router + router.preload(["english"]) + client = TestClient(worker.build_app(router, "english", "mps", revisions=revisions)) + router.unload() + assert client.get("/health").json()["models"] == {} + + +def test_a_checkpoint_loaded_while_serving_gets_the_gpu_options_before_its_warmup(monkeypatch): + applied = [] + router = FakeRouter() + monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: applied.append(("fp16", len(router.calls)))) + monkeypatch.setattr(optimize, "compile_agent", lambda agent: applied.append(("compile", len(router.calls)))) + worker.build_app(router, "english", "mps", fp16=True, compile=True, graph_counter=lambda: 0) + warmed_at_startup = len(router.calls) + router.load_while_serving("multilingual", FakeAgent()) + assert applied[2:] == [("fp16", warmed_at_startup), ("compile", warmed_at_startup)] + + +def test_a_checkpoint_off_the_requested_device_is_logged_when_it_is_allowed(caplog): + with caplog.at_level("WARNING", logger="laya-worker"): + worker.build_app(FakeRouter(FakeAgent(device="cpu")), "english", "mps") + assert "asked for mps, english is on cpu" in caplog.text + + +def test_no_warning_when_the_warmup_just_reaches_layas_autocast_rows(caplog): + agent = FakeAgent() + agent.mps_amp_min_rows = engine.WARMUP_MAX_ROWS + with caplog.at_level("WARNING", logger="laya-worker"): + worker.build_app(FakeRouter(agent), "english", "mps") + assert "the warmup stops at" not in caplog.text + + +def test_failed_late_loads_are_logged_with_their_cause(laya_router, caplog): + router, revisions = laya_router + router.preload(["english", "multilingual"]) + worker.build_app(router, "english", "mps", require_device=True, revisions=revisions) + Repository.device = {"typed-decisions": "cpu", None: "cpu"} # english cannot come back either + with caplog.at_level("ERROR", logger="laya-worker"), pytest.raises(RuntimeError): + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model="typed-decisions") + assert "typed-decisions could not be prepared and is unloaded" in caplog.text + assert "english was evicted for typed-decisions and could not be loaded again" in caplog.text + assert "asked for mps, typed-decisions is on cpu" in caplog.text # the traceback of the cause + assert router.loaded == ["multilingual"] + + +def test_a_failed_late_load_reloads_only_what_was_evicted_for_it(laya_router, monkeypatch): + router, revisions = laya_router + router.max_loaded = 1 + router.load("english") + worker.build_app(router, "english", "mps", require_device=True, revisions=revisions) + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model="multilingual") # evicts english + built = [] + original = DownloadingAgent.__init__ + monkeypatch.setattr( + DownloadingAgent, + "__init__", + lambda self, repo, **kw: built.append(kw.get("subfolder")) or original(self, repo, **kw), + ) + Repository.device = {"typed-decisions": "cpu"} + with pytest.raises(RuntimeError): + router.predict("refund me", {"r": {"type": "noul", "instructions": "?"}}, model="typed-decisions") + assert built == ["typed-decisions", "multilingual"] and router.loaded == ["multilingual"] + + +def test_describe_reads_the_weight_dtype_from_the_model_as_it_is(): + import torch + + agent = FakeAgent() + assert engine.describe(agent, "mps", None)["weights_dtype"] is None # no model to read + agent.model = torch.nn.Linear(2, 2) + assert engine.describe(agent, "mps", None)["weights_dtype"] == "torch.float32" + agent.model.half() + assert engine.describe(agent, "mps", None)["weights_dtype"] == "torch.float16" + + +def test_warmup_time_is_reported_in_milliseconds(monkeypatch): + clock = iter([10.0, 10.25]) + monkeypatch.setattr(engine.time, "perf_counter", lambda: next(clock)) + assert engine.warmup(FakeRouter(), "english")["warmup_ms"] == 250.0 + + +def test_apply_says_whether_the_options_were_applied(monkeypatch): + monkeypatch.setattr(optimize, "use_fp16_weights", lambda agent: None) + assert optimize.apply(FakeAgent(device="cpu"), fp16=True, compile=False) is False + assert optimize.apply(FakeAgent(device="mps"), fp16=True, compile=False) is True + + +def test_the_checkpoint_is_built_once_and_serves_every_request(laya_router, monkeypatch): + router, revisions = laya_router + router.preload(["english"]) + built = [] + original = DownloadingAgent.__init__ + monkeypatch.setattr( + DownloadingAgent, "__init__", lambda self, repo, **kw: built.append(repo) or original(self, repo, **kw) + ) + client = TestClient(worker.build_app(router, "english", "mps", revisions=revisions)) + body = {"model": "english", "state": "refund me", "questions": {"r": {"type": "noul", "instructions": "?"}}} + for _ in range(3): + response = client.post("/v1/systemone", json=body) + assert response.status_code == 200 and response.json()["answers"]["r"] == ANSWER + assert built == [] and router.loaded == ["english"] + + +def test_main_binds_only_after_every_warmup_request(monkeypatch): + router = FakeRouter() + bound_after = [] + monkeypatch.setattr(worker, "make_router", lambda device, model: router) + monkeypatch.setattr(worker.logging, "basicConfig", lambda **kw: None) + monkeypatch.setattr("uvicorn.run", lambda app, **kw: bound_after.append(len(router.calls))) + monkeypatch.setattr(sys, "argv", ["laya_mps", "--device", "mps"]) + worker.main() + assert bound_after == [len(engine.WARMUP_SHAPES) * engine.WARMUP_REPEATS] + + +def test_the_warmup_asks_every_question_type(): + kinds = {q["type"] for _, questions in engine.WARMUP_SHAPES for q in questions.values()} + assert kinds == {"choice", "score", "noul"} From 668f3315a24bdf7890da26bf02e32a02745f4c63 Mon Sep 17 00:00:00 2001 From: cacheline999 <326908201+cacheline999@users.noreply.github.com> Date: Sat, 3 Oct 2026 00:51:31 +0800 Subject: [PATCH 20/21] Laya: keep profile runs out of the parity section; header test off macOS; memory growth is from compile - report.py: the parity section lists benchmark runs (they write a phase record before any answers) and runs with answers or failed probes; a profile_mps.py run, which never answers, no longer shows as failures. - The run-header test asserts the macOS machine probes only on macOS, and checks that they degrade to None elsewhere. - Recipe: the per-length memory growth comes from --compile, not from fp16 weights. --- recipe/laya/apple-silicon.md | 14 ++++++------ recipe/laya/bench/report.py | 13 ++++++----- tests/laya/test_bench.py | 42 +++++++++++++++++++++++++++++++++++- 3 files changed, 57 insertions(+), 12 deletions(-) diff --git a/recipe/laya/apple-silicon.md b/recipe/laya/apple-silicon.md index be6d824..9fb097f 100644 --- a/recipe/laya/apple-silicon.md +++ b/recipe/laya/apple-silicon.md @@ -105,10 +105,11 @@ The latencies above are for requests sent back to back. Measured on the M1 Pro: does not do this. - **New input lengths.** The first request of a length the worker has not seen costs about 15 ms more once with the options (6 ms without). It is not a recompile (`recompiled_after_ready` stays `false`). -- **Memory grows with the lengths seen.** With both options, what PyTorch caches per input length adds - about 5 MB each: the 3 GB below became 3.3 GB after 100 new lengths and 5.3 GB after all 477, more than - the 4.0 GB of a worker without the options, whose footprint did not grow. `torch.mps.empty_cache()` - releases it, and those lengths then pay their first-request cost again. +- **Memory grows with the lengths seen.** With `--compile`, PyTorch keeps host memory for every input + length the compiled model has run, about 5 MB each (fp16 weights alone add little): the 3 GB above became + 3.3 GB after 100 new lengths and 5.3 GB after all 477, more than the 4.0 GB of a worker without the + options. `torch.mps.empty_cache()` releases most of it, and those lengths then pay their first-request + cost again. `/health` reports under `compile` how many graphs existed when the worker became ready and how many exist now; `recompiled_after_ready: true` means a request shape was not covered by the warmup. @@ -185,6 +186,7 @@ Runs labelled anything other than `feasibility` refuse to start on battery power - `device_mismatch: true` on a worker that started on MPS: Laya fell back to the CPU after a GPU out-of-memory error. It keeps answering, several times slower; free memory and restart the worker to get back on the GPU. -- The worker process uses about 4 GB, or 3 GB with fp16 weights growing towards 5 GB as it sees more input - lengths (Activity Monitor's Memory column, which counts MPS allocations), with one checkpoint loaded; a second one Laya loads later adds its own. On a 16 GB Mac, close other large applications before benchmarking. +- The worker process uses about 4 GB, or 3 GB with fp16 weights (Activity Monitor's Memory column, which + counts MPS allocations), with one checkpoint loaded; with `--compile` it grows towards 5 GB as it sees + more input lengths (see "What the warm numbers leave out"); a second one Laya loads later adds its own. On a 16 GB Mac, close other large applications before benchmarking. - `Address already in use`: another worker or frontend still holds port 8000 or 8080. diff --git a/recipe/laya/bench/report.py b/recipe/laya/bench/report.py index 7505afc..97a4b42 100644 --- a/recipe/laya/bench/report.py +++ b/recipe/laya/bench/report.py @@ -177,17 +177,20 @@ def memory(phases, ends): def read_answers(records): - """Per (config, run): the env record, the answers per workload, and the probes that failed.""" - envs, answers, errors = {}, {}, {} + """Per (config, run): the env record, the answers per workload, the probes that failed, and which runs are + benchmark runs (bench_inproc/bench_http write a phase record before any answers; profile_mps never answers).""" + envs, answers, errors, benchmarks = {}, {}, {}, set() for r in records: key = (r["config"], r["run"]) if r["type"] == "env": envs[key] = r + elif r["type"] == "phase": + benchmarks.add(key) elif r["type"] == "answers": answers.setdefault(key, {})[r["workload"]] = r["answers"] elif r["type"] == "answers_error": errors.setdefault(key, {})[r["workload"]] = f"status {r.get('status')}: {r.get('detail')}" - return envs, answers, errors + return envs, answers, errors, benchmarks def outcome(answer): @@ -220,7 +223,7 @@ def path(env, rows): def parity(records, ref): """Answers of every run against the reference config, with the tolerances declared in advance.""" - envs, answers, errors = read_answers(records) + envs, answers, errors, benchmarks = read_answers(records) refs = sorted(k for k in answers if k[0] == ref) if not refs: return f"no answers for reference config {ref}" @@ -233,7 +236,7 @@ def parity(records, ref): "|---|---|---|---|---|---|---|---|---|---|", ] failed = total = 0 - for key in sorted(envs.keys() | answers.keys() | errors.keys()): # a run with no answers at all still counts + for key in sorted(benchmarks | answers.keys() | errors.keys()): # a benchmark run without answers still counts if key == ref_key: continue env = envs.get(key, {}) diff --git a/tests/laya/test_bench.py b/tests/laya/test_bench.py index b448aec..6dfff4e 100644 --- a/tests/laya/test_bench.py +++ b/tests/laya/test_bench.py @@ -28,7 +28,9 @@ def test_header_records_what_makes_two_runs_comparable(monkeypatch): version(package) for package in ("laya", "torch", "transformers") ) assert record["argv"] == sys.argv and record["python"] == ".".join(map(str, sys.version_info[:3])) - assert record["power"] and record["loadavg_1m"] >= 0 and record["chip"] and record["mem_gb"] > 0 + assert record["loadavg_1m"] >= 0 + if sys.platform == "darwin": # the machine probes use macOS tools; elsewhere they are None (test below) + assert record["power"] and record["chip"] and record["mem_gb"] > 0 assert record["utc"].endswith("+00:00") @@ -85,3 +87,41 @@ def test_documented_commands_use_flags_and_files_that_exist(): ), f"{document}: {command}" used = set(re.findall(r"(? Date: Sat, 3 Oct 2026 01:08:49 +0800 Subject: [PATCH 21/21] Laya bench and tests: two-sided answer check in paired summaries; readiness test checks the warmup - paired.py --summarize compares the questions of both sides: a question missing on either side, or answered with a different type, is listed under errors. - Contract test: drop the check that the port refused connections before ready, which plain laya-serve also passes while it loads; the test keeps the part that fails for it, a finished warmup in the first /health. --- recipe/laya/bench/paired.py | 7 ++++--- tests/laya/test_bench.py | 28 ++++++++++++++++++++++++++++ tests/laya/test_contract.py | 7 +++---- 3 files changed, 35 insertions(+), 7 deletions(-) diff --git a/recipe/laya/bench/paired.py b/recipe/laya/bench/paired.py index 2f4c580..ecc077b 100644 --- a/recipe/laya/bench/paired.py +++ b/recipe/laya/bench/paired.py @@ -195,9 +195,10 @@ def summarize(paths): if any(sides.get(s, {"error": "missing"}).get("error") for s in "AB"): errors.append(wid) continue - for q, a in sides["A"]["answers"].items(): - b = sides["B"]["answers"].get(q) - if b is None: + a_answers, b_answers = sides["A"]["answers"], sides["B"]["answers"] + for q in sorted(a_answers.keys() | b_answers.keys()): + a, b = a_answers.get(q), b_answers.get(q) + if a is None or b is None or a.get("type") != b.get("type"): errors.append(f"{wid}/{q}") continue worst = max(worst, max(abs(flat(a)[k] - flat(b).get(k, 0.0)) for k in flat(a))) diff --git a/tests/laya/test_bench.py b/tests/laya/test_bench.py index 6dfff4e..f0e9374 100644 --- a/tests/laya/test_bench.py +++ b/tests/laya/test_bench.py @@ -125,3 +125,31 @@ def test_parity_lists_benchmark_runs_but_not_runs_that_never_answer(): assert "| C3 | x | W1 | | | | missing | | | FAIL |" in out assert "| C4 | x | W1 | | | | status 500: x | | | FAIL |" in out assert out.endswith("0/2 questions within tolerance.") + + +# ---------------------------------------------------------------------------------------------- paired.py +import paired # noqa: E402 + +CHOICE_ANSWER = {"type": "choice", "choice": "a", "probabilities": {"a": 0.7, "b": 0.3}} + + +def summary_of(tmp_path, capsys, a, b): + records = [{"type": "env", "run": "x", "a": "A", "b": "B", "loadavg_1m": 1.0}] + records += [{"type": "answers", "side": side, "workload": "W1", "answers": ans, "error": None} + for side, ans in (("A", a), ("B", b))] # fmt: skip + records.append({"type": "end", "health": {"A": {"device": "mps"}, "B": {"device": "mps"}}, "footprint_mb": {}}) + path = tmp_path / "paired_x.jsonl" + path.write_text("\n".join(json.dumps(r) for r in records)) + paired.summarize([str(path)]) + return capsys.readouterr().out + + +def test_paired_summary_reports_a_question_missing_on_either_side(tmp_path, capsys): + both = {"q": CHOICE_ANSWER, "r": {"type": "noul", "noul": 0.9}} + assert "errors ['W1/r']" in summary_of(tmp_path, capsys, both, {"q": CHOICE_ANSWER}) + assert "errors ['W1/r']" in summary_of(tmp_path, capsys, {"q": CHOICE_ANSWER}, both) + + +def test_paired_summary_reports_answers_of_different_shape(tmp_path, capsys): + out = summary_of(tmp_path, capsys, {"q": CHOICE_ANSWER}, {"q": {"type": "noul", "noul": 0.9}}) + assert "errors ['W1/q']" in out diff --git a/tests/laya/test_contract.py b/tests/laya/test_contract.py index 56462e0..17d779f 100644 --- a/tests/laya/test_contract.py +++ b/tests/laya/test_contract.py @@ -88,7 +88,6 @@ def worker(): ] process = subprocess.Popen(command, env=env) deadline = time.monotonic() + 600 - startup["refused"] = 0 while time.monotonic() < deadline: assert process.poll() is None, f"worker exited with {process.returncode}" try: @@ -97,7 +96,6 @@ def worker(): startup["first_health"] = json.loads(body) break except OSError: - startup["refused"] += 1 time.sleep(0.1) else: pytest.fail("worker not ready in 600 s") @@ -133,8 +131,9 @@ def test_health_reports_the_loaded_model(worker): assert len(health["revision"]) == 40 -def test_the_port_opens_only_after_the_warmup(worker): - assert startup["refused"] > 0 # nothing listened while the model loaded and warmed up +def test_the_first_health_answer_comes_after_the_warmup(worker): + # laya-serve also refuses connections while it loads, then answers /health before any forward pass; + # the worker's first answer must already report a finished warmup. assert startup["first_health"]["ready"] is True and startup["first_health"]["warmup_ms"] > 0