Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .github/workflows/anyio.yml
Original file line number Diff line number Diff line change
Expand Up @@ -69,9 +69,9 @@ jobs:
cache-suffix: anyio-${{ matrix.os_name }}-py${{ matrix.python_version }}

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Cache Cargo registry
uses: Swatinem/rust-cache@v2
Expand Down
40 changes: 32 additions & 8 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -48,9 +48,9 @@ jobs:
python-version: "3.13"

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Cache Cargo registry
uses: Swatinem/rust-cache@v2
Expand Down Expand Up @@ -80,9 +80,9 @@ jobs:
python-version: "3.13"

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"
components: clippy

- name: Cache Cargo registry
Expand All @@ -91,6 +91,30 @@ jobs:
- name: Run Clippy
run: cargo clippy --all-targets --all-features --locked -- -D warnings

allocator-miri:
name: allocator miri (strict provenance)
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- name: Check out repository
uses: actions/checkout@v6
with:
persist-credentials: false

- name: Set up Rust
uses: dtolnay/rust-toolchain@master
with:
toolchain: "nightly-2026-09-25"
components: miri,rust-src

- name: Cache Cargo registry
uses: Swatinem/rust-cache@v2

- name: Run allocator tests under Miri
env:
MIRIFLAGS: -Zmiri-strict-provenance
run: cargo miri test --manifest-path tools/vibeio-check/Cargo.toml --features scheduler-batch-cache batch_

type-check:
name: pyright
runs-on: ubuntu-latest
Expand Down Expand Up @@ -165,9 +189,9 @@ jobs:
cache-suffix: anyio

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Cache Cargo registry
uses: Swatinem/rust-cache@v2
Expand Down Expand Up @@ -283,9 +307,9 @@ jobs:
cache-suffix: ${{ matrix.os_name }}-py${{ matrix.python_version }}

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Cache Cargo registry
uses: Swatinem/rust-cache@v2
Expand Down
4 changes: 2 additions & 2 deletions .github/workflows/vibeio-native.yml
Original file line number Diff line number Diff line change
Expand Up @@ -47,9 +47,9 @@ jobs:
persist-credentials: false

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"
components: clippy, rustfmt

- name: Cache runtime harness build
Expand Down
8 changes: 4 additions & 4 deletions .github/workflows/wheels.yml
Original file line number Diff line number Diff line change
Expand Up @@ -64,9 +64,9 @@ jobs:
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Install LLVM tools for PGO
if: github.event_name == 'workflow_dispatch' && inputs.pgo
Expand Down Expand Up @@ -114,9 +114,9 @@ jobs:
uses: astral-sh/setup-uv@08807647e7069bb48b6ef5acd8ec9567f424441b

- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
uses: dtolnay/rust-toolchain@master
with:
toolchain: "1.98.1"
toolchain: "nightly-2026-09-25"

- name: Cache Cargo registry and target
uses: Swatinem/rust-cache@v2
Expand Down
2 changes: 2 additions & 0 deletions Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,8 @@ windows-sys = { version = "0.61", features = ["Wdk_Foundation", "Wdk_Storage_Fil

[features]
default = []
# Opt-in: faster scheduler turns, but measured TCP workloads can regress.
scheduler-batch-cache = []
blocking-default = ["dep:rusty_pool"]
# Embedded-runtime modules are opt-in; production wheels retain the default
# networking/timer build unless a feature is explicitly requested.
Expand Down
32 changes: 32 additions & 0 deletions benches/compare_event_loops.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@

import argparse
import asyncio
import contextvars
import ctypes
import gc
import importlib
Expand Down Expand Up @@ -373,6 +374,37 @@ async def tiny_task() -> None:
)


async def bench_task_options(
loop_name: str, iterations: int, batch_size: int
) -> ChildResult:
"""Exercise named, explicit-context Task construction (Python 3.11+)."""

async def tiny_task() -> None:
await asyncio.sleep(0)

loop = asyncio.get_running_loop()
context = contextvars.copy_context()
start = time.perf_counter()
remaining = iterations
while remaining > 0:
current_batch = min(batch_size, remaining)
tasks = [
loop.create_task(tiny_task(), name="benchmark-task", context=context)
for _ in range(current_batch)
]
await asyncio.gather(*tasks)
remaining -= current_batch
return ChildResult(
loop_name,
"task_options",
time.perf_counter() - start,
iterations,
baseline_rss_bytes=0,
peak_rss_bytes=0,
peak_rss_delta_bytes=0,
)


async def maybe_wait_closed(writer: asyncio.StreamWriter) -> None:
wait_closed = getattr(writer, "wait_closed", None)
if wait_closed is None:
Expand Down
146 changes: 146 additions & 0 deletions docs/allocator-batch-results.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,146 @@
# Scheduler batch allocation reuse

Measured on 2026-09-25. This change uses the stabilized allocator API available
in the pinned `nightly-2026-09-25` compiler, without an `allocator_ext` feature
gate. It does **not** claim compatibility with an older stable-channel compiler.

The cache is **opt-in** through `scheduler-batch-cache`, not enabled in default
builds or wheels. Its scheduler-turn improvement comes with a confirmed TCP
slowdown, so the general-purpose event loop retains its original allocation
and drain path.

## What changes

`Runtime::block_on` previously allocated and freed its ready-task vector on
every entry. With the feature enabled, the runtime owns a single-slot, exact-layout
`System` allocation cache, used through `Allocator` and `Vec::with_capacity_in`.
Successive calls reuse that storage; batches within a call keep the vector's
capacity. Tests verify one system allocation across repeated runtime turns,
including root-future unwinding and reuse afterward.

The usual retained allocation is 2 KiB on 64-bit systems, with a hard 16 KiB
ceiling per runtime. Runtime destruction frees it. This reduces allocator
traffic, not the live size of tasks or a guaranteed amount of process RSS.
No mutex, allocator `Arc` clone, size-class search, or public configuration is
introduced. The cache sits outside `RuntimeInner` to preserve the hot shared
scheduler state's layout.

Allocator-aware `Vec::drain` is not in the stabilized subset. A private full
drain uses stable vector primitives, retains the allocation, preserves FIFO
order, and drops unconsumed tasks on unwind. Strict-provenance Miri tests cover
partial consumption, destructor panics, zero-sized elements, forgotten drains,
alignment, growth, zeroing, disjoint simultaneous allocations, and runtime
panic cleanup. Kani's older compiler retains an ordinary-vector fallback;
its proofs do not verify the custom allocator.

The previous shared size-class allocator remains rejected. Early versions of
this change also regressed busy turns when replacing the vector each batch or
using optional task slots. Neither implementation is retained.

## Native scheduler measurement

Host: Intel Core i9-9900K, Linux x86-64; Rust 1.100.0-nightly
(`f7575a9da`, LLVM 23.1.1). Identical `runtime_turns.rs` source and lockfile were
built with the same compiler and release profile, against `d4f56f1` and the
new implementation with `--features scheduler-batch-cache`. No other builds
or tests ran during timing. CPU affinity
was unrestricted; these are local results, not a cross-platform guarantee.

Each comparison used 24 randomized, balanced baseline/candidate process pairs,
10,000 warm-up turns per process, and paired bootstrap 95% confidence intervals.
Negative percentages mean less elapsed time.

| Workload | Turns per sample | Baseline median | Candidate median | Change (95% interval) |
| --- | ---: | ---: | ---: | ---: |
| Idle scheduler entry/exit | 7,000,000 | 0.76260 s | 0.64408 s | -15.41% [-15.87%, -14.88%] |
| 64 continuously ready tasks | 500,000 | 1.12497 s | 1.12506 s | +0.30% [-0.48%, +1.14%] |

The idle-turn gain passed the declared 1% timing threshold; the busy result is
inconclusive and within the 3% regression budget at this confidence level.
These measurements isolate scheduler turns, not Python or network throughput.

Reproduce with `scripts/compare_runtime_turns.py` as described in
[Development](development.md). Use `--iterations 7000000 --tasks 0 --seed 999`
for idle turns and `--iterations 500000 --tasks 64 --seed 1000` for busy turns.
Both use `--blocks 24` and separate output directories.

Local raw plans, binary hashes, paired samples, and reports are in
`target/allocator-batch-optin-idle` and `target/allocator-batch-optin-active`.
These ignored directories are local experiment artifacts, not checked-in data.

## Python application workloads

The six-workload holdout used CPython 3.14.7, 24 paired process blocks,
seed 997, and `tcp_streams` as the predeclared primary. Compiler versions,
non-PGO flags, benchmark source, and artifact hashes were checked by the
hot-path harness. Each interval below uses 99.583% confidence (Bonferroni
adjustment across six timing and six RSS metrics).

| Workload | Time change (interval) | Peak RSS change (interval) |
| --- | ---: | ---: |
| Callbacks | -0.14% [-0.70%, +0.36%] | +0.00% [-0.01%, +0.02%] |
| Tasks | -0.64% [-1.50%, +0.14%] | -0.19% [-0.41%, +0.01%] |
| TCP streams | +2.10% [+0.64%, +3.56%] | +0.00% [-0.23%, +0.23%] |
| HTTP keep-alive | -0.30% [-6.67%, +3.78%] | +0.01% [-0.28%, +0.32%] |
| Mixed streams | -0.71% [-2.98%, +1.33%] | +0.03% [-0.29%, +0.38%] |
| Bulk transfer | -0.43% [-2.01%, +0.82%] | -0.00% [-0.02%, +0.01%] |

The application-wide gate is **inconclusive**, not passed: the primary did
not improve, TCP showed a small slowdown, and the TCP/HTTP upper timing bounds
exceed the 3% budget. Peak RSS stayed well within budget, but there is no
established application-wide memory reduction or throughput improvement.

An independent 40-pair TCP confirmation (seed 998) found **+2.75%** elapsed time,
with a 97.5% interval of **[+1.79%, +3.85%]**. RSS changed +0.04%
[-0.14%, +0.24%]. This confirmed the tradeoff and is why the cache is opt-in.
Raw confirmation data is in `target/allocator-batch-final-tcp-confirmation`.

The exact baseline is `d4f56f1`; the local cache-enabled measurement snapshot is
`a87092ec214b6daa9288818109f58cb6034846d3`. It predates the final feature guard:
these application results evaluate the cache integration, not the final default
build (which uses the original path). Artifact directories
are `target/allocator-batch-baseline-final` and `target/allocator-batch-final`;
the complete plan, samples, manifests and report are under
`target/allocator-batch-final-holdout`.

The following command records the cache-enabled snapshot comparison. To repeat
the experiment on later revisions, use a cache-enabled candidate build; an
ordinary default build no longer enables the cache. Use fresh output directories.

```bash
.venv/bin/python scripts/hotpath_lab.py compare \
--baseline target/allocator-batch-baseline-final \
--candidate target/allocator-batch-final \
--primary tcp_streams --suite holdout --blocks 24 --seed 997 \
--out target/allocator-batch-final-holdout
```

## Correctness checks

- Root Rust suites: 307 passed with default features; 407 with all features.
- Embedded runtime all-feature suite: 283 passed; 16 doctests passed.
- Strict-provenance Miri: 10 allocator/drain/runtime regression tests passed.
- Kani: all 31 `merge_` harnesses passed, using the ordinary-vector fallback.
- Clippy: both crates, all targets/features, warnings denied.
- Python: 192 passed and 2 skipped on **each** isolated extension; tooling,
stress, and slow-network markers were excluded (73 deselected).
- Final default development install: the same 192 Python tests passed, with
2 skipped and 73 deselected.
- Repository tooling: 67 passed. The new comparison runner also passes Ruff.

Relevant commands:

```bash
python3 scripts/run_rust_tests.py --all-features
cargo test --manifest-path tools/vibeio-check/Cargo.toml --all-features --locked
MIRIFLAGS=-Zmiri-strict-provenance cargo miri test \
--manifest-path tools/vibeio-check/Cargo.toml --features scheduler-batch-cache batch_
cargo kani --harness merge_ -j 2 --output-format terse
PYTEST_ADDOPTS="-m 'not tooling and not stress and not slow_network'" \
.venv/bin/python scripts/hotpath_lab.py test \
--artifact target/allocator-batch-final --python-child
```

The Python command was also run against `target/allocator-batch-baseline-final`.
These checks do not replace cross-platform CI or the excluded stress/network
suites.
Loading
Loading