From 0122103274b1c1e2e2175269608ae652da3fbb55 Mon Sep 17 00:00:00 2001 From: prabod Date: Wed, 12 Aug 2026 17:17:10 +1000 Subject: [PATCH] sync: header and binding mirrors for 0.2.2 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The half of the 0.2.2 sync that had to wait for the engine release. `embed_norm_eps` widens BaseRTModelConfig from 1540 to 1704 bytes. The rust-sys ABI job links the latest engine RELEASE from this repo and compares it against the in-tree mirror, so syncing these before v0.2.2 was published failed by construction — 1540 from the binary against 1704 from source. v0.2.2 is now released here, so the two agree. Verified the way that job does it: downloaded basert-engine-macos-arm64-0.2.2.tar.gz from this repo's release, pointed BASERT_LIB_DIR at it, and ran the sys crate — 15 passed, including model_config_size_matches_library, the assertion that previously failed. The mirrors are one ABI struct across C, Swift, Rust, Python and Node; the model config is returned BY VALUE, so a mirror even four bytes short lets the engine write past the caller's allocation. --- bindings/node/src/index.ts | 12 +++++ bindings/python/baseRT/__init__.py | 14 +++++- bindings/python/tests/test_baseRT.py | 4 +- bindings/rust/baseRT-sys/src/lib.rs | 19 +++++++- .../swift/Sources/CBaseRT/include/types.h | 44 +++++++++++++++++++ include/baseRT/types.h | 44 +++++++++++++++++++ 6 files changed, 133 insertions(+), 4 deletions(-) diff --git a/bindings/node/src/index.ts b/bindings/node/src/index.ts index b5ae1aa..d4763d5 100644 --- a/bindings/node/src/index.ts +++ b/bindings/node/src/index.ts @@ -174,6 +174,18 @@ const ModelConfigC = koffi.struct("BaseRTModelConfig", { rope_high_freq_factor: "float", rope_orig_max_pos: "uint32", rope_scaling_type: "uint32", + // Muse Glimmer (0 / empty = not applicable) + qk_scale_factor: "float", + output_multiplier: "float", + post_norm_eps: "float", + nope_layers: koffi.array("uint8", 64), + embed_norm_eps: "float", + vision_window_layers: koffi.array("uint8", 64), + vision_window_size: "uint32", + vision_pos_embed_h: "uint32", + vision_pos_embed_w: "uint32", + vision_adapter_dim: "uint32", + video_token_id: "uint32", }); const SamplingConfigC = koffi.struct("BaseRTSamplingConfig", { diff --git a/bindings/python/baseRT/__init__.py b/bindings/python/baseRT/__init__.py index aaa0441..ecb8a8c 100644 --- a/bindings/python/baseRT/__init__.py +++ b/bindings/python/baseRT/__init__.py @@ -21,7 +21,7 @@ from pathlib import Path from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple, Union -__version__ = "0.2.1" +__version__ = "0.2.2" # --------------------------------------------------------------------------- # Library loading @@ -195,6 +195,18 @@ class BaseRTModelConfig(ctypes.Structure): ("rope_high_freq_factor", ctypes.c_float), ("rope_orig_max_pos", ctypes.c_uint32), ("rope_scaling_type", ctypes.c_uint32), + # Muse Glimmer (0 / empty = not applicable) + ("qk_scale_factor", ctypes.c_float), + ("output_multiplier", ctypes.c_float), + ("post_norm_eps", ctypes.c_float), + ("nope_layers", ctypes.c_uint8 * 64), + ("embed_norm_eps", ctypes.c_float), + ("vision_window_layers", ctypes.c_uint8 * 64), + ("vision_window_size", ctypes.c_uint32), + ("vision_pos_embed_h", ctypes.c_uint32), + ("vision_pos_embed_w", ctypes.c_uint32), + ("vision_adapter_dim", ctypes.c_uint32), + ("video_token_id", ctypes.c_uint32), ] diff --git a/bindings/python/tests/test_baseRT.py b/bindings/python/tests/test_baseRT.py index e47a81e..55f40a8 100644 --- a/bindings/python/tests/test_baseRT.py +++ b/bindings/python/tests/test_baseRT.py @@ -80,7 +80,7 @@ def test_field_count(self): # Keep this in sync with include/baseRT/types.h. The runtime ABI-size # assertion below catches layout drift; this count catches accidental # omission of same-sized fields from the ctypes mirror. - assert len(BaseRTModelConfig._fields_) == 95 + assert len(BaseRTModelConfig._fields_) == 106 def test_architecture_field_is_char_array(self): # architecture should be a fixed 32-byte char array @@ -567,7 +567,7 @@ def test_model_config_size(self): # Exact sizeof(BaseRTModelConfig) from include/baseRT/types.h; the # library cross-check happens at import via baseRT_model_config_sizeof. # Must match the Rust mirror test (bindings/rust/baseRT-sys). - assert ctypes.sizeof(BaseRTModelConfig) == 1540 + assert ctypes.sizeof(BaseRTModelConfig) == 1704 def test_sampling_config_size(self): size = ctypes.sizeof(BaseRTSamplingConfig) diff --git a/bindings/rust/baseRT-sys/src/lib.rs b/bindings/rust/baseRT-sys/src/lib.rs index 0b20e97..7cccdbc 100644 --- a/bindings/rust/baseRT-sys/src/lib.rs +++ b/bindings/rust/baseRT-sys/src/lib.rs @@ -130,6 +130,18 @@ pub struct BaseRTModelConfig { pub rope_high_freq_factor: c_float, pub rope_orig_max_pos: u32, pub rope_scaling_type: u32, + // Muse Glimmer (0 / empty = not applicable) + pub qk_scale_factor: c_float, + pub output_multiplier: c_float, + pub post_norm_eps: c_float, + pub nope_layers: [u8; 64], + pub embed_norm_eps: c_float, + pub vision_window_layers: [u8; 64], + pub vision_window_size: u32, + pub vision_pos_embed_h: u32, + pub vision_pos_embed_w: u32, + pub vision_adapter_dim: u32, + pub video_token_id: u32, } /// Transcription result statistics. @@ -453,7 +465,7 @@ mod tests { "config struct unexpectedly small ({})", mem::size_of::() ); - assert_eq!(mem::size_of::(), 1540); + assert_eq!(mem::size_of::(), 1704); } #[test] @@ -547,6 +559,11 @@ mod tests { assert_eq!(&base.rope_scaling_factor as *const _ as usize - base_ptr, 1520); assert_eq!(&base.rope_orig_max_pos as *const _ as usize - base_ptr, 1532); assert_eq!(&base.rope_scaling_type as *const _ as usize - base_ptr, 1536); + assert_eq!(&base.qk_scale_factor as *const _ as usize - base_ptr, 1540); + assert_eq!(&base.nope_layers as *const _ as usize - base_ptr, 1552); + assert_eq!(&base.embed_norm_eps as *const _ as usize - base_ptr, 1616); + assert_eq!(&base.vision_window_layers as *const _ as usize - base_ptr, 1620); + assert_eq!(&base.video_token_id as *const _ as usize - base_ptr, 1700); } #[test] diff --git a/bindings/swift/Sources/CBaseRT/include/types.h b/bindings/swift/Sources/CBaseRT/include/types.h index d441e21..e607967 100644 --- a/bindings/swift/Sources/CBaseRT/include/types.h +++ b/bindings/swift/Sources/CBaseRT/include/types.h @@ -190,6 +190,50 @@ typedef struct { // 1 = llama3 (piecewise), 2 = linear (uniform divisor), 3 = other / // unsupported (skipped with a warning rather than mis-applied). uint32_t rope_scaling_type; + + // ── Muse Glimmer (0 / empty = not applicable) ──────────────────── + // Dense SWA/global decoder with a perception (ViT) tower. Shaped like + // Gemma 3 (four zero-centered per-layer norms, final logit softcap) with + // four twists the fields below carry. + // + // Multiplier applied to Q AFTER a scaleless (weightless) per-head + // RMSNorm, ON TOP OF the usual 1/sqrt(head_dim) — unlike + // `attention_scale`, which REPLACES that term. 0 = no QK-norm/scale. + // Because the norm is weightless there are no q_norm/k_norm tensors. + float qk_scale_factor; + // Scale applied to the logits BEFORE the final tanh softcap + // (`logit_softcap`). 0 = no multiplier. + float output_multiplier; + // Epsilon for the post-attention / post-FFN norms, which differs from + // `norm_eps` here (1e-8 vs 1e-5). 0 = reuse `norm_eps` everywhere. + float post_norm_eps; + // Per-layer NoPE mask: bit i = 1 means layer i applies NO rotary at all + // (the reference passes `position_embeddings=None` there). On Muse + // Glimmer these are exactly the global/full-attention layers, so the + // mask is the complement of `swa_layers` — but it is stored + // independently because the two are separate config keys upstream + // (`layer_types` vs `layer_rope_theta`) and need not agree. + // Up to 512 layers (64 bytes). All zero = every layer gets RoPE. + uint8_t nope_layers[64]; + // Epsilon for a weightless RMSNorm the runtime applies to the token + // embedding right after lookup. 0 = already folded into the embedding + // rows at convert time (the HF path does this exactly), or no such norm. + // Only the GGUF path sets it: k-quant passthrough keeps the rows packed, + // so folding would mean dequantizing them. + float embed_norm_eps; + + // Perception-tower extras (vision_arch == 2). The tower is a LayerNorm + // ViT with SEPARATE biased q/k/v projections, alternating window/full + // attention, a bias-free Linear patch embed over flattened pixels, and a + // bilinearly-resampled learned position grid. + // Per-layer window-attention mask: bit i = 1 means layer i attends only + // within its window; 0 = full attention across all patches. + uint8_t vision_window_layers[64]; + uint32_t vision_window_size; // window side in PIXELS (pos_emb_height * patch_size, e.g. 448) + uint32_t vision_pos_embed_h; // learned position grid height (e.g. 32) + uint32_t vision_pos_embed_w; // learned position grid width (e.g. 32) + uint32_t vision_adapter_dim; // projector hidden width (`projector_hidden_size`, e.g. 4096) + uint32_t video_token_id; // text-side placeholder token for video features (0 = none) } BaseRTModelConfig; /// Transcription result statistics. diff --git a/include/baseRT/types.h b/include/baseRT/types.h index d441e21..e607967 100644 --- a/include/baseRT/types.h +++ b/include/baseRT/types.h @@ -190,6 +190,50 @@ typedef struct { // 1 = llama3 (piecewise), 2 = linear (uniform divisor), 3 = other / // unsupported (skipped with a warning rather than mis-applied). uint32_t rope_scaling_type; + + // ── Muse Glimmer (0 / empty = not applicable) ──────────────────── + // Dense SWA/global decoder with a perception (ViT) tower. Shaped like + // Gemma 3 (four zero-centered per-layer norms, final logit softcap) with + // four twists the fields below carry. + // + // Multiplier applied to Q AFTER a scaleless (weightless) per-head + // RMSNorm, ON TOP OF the usual 1/sqrt(head_dim) — unlike + // `attention_scale`, which REPLACES that term. 0 = no QK-norm/scale. + // Because the norm is weightless there are no q_norm/k_norm tensors. + float qk_scale_factor; + // Scale applied to the logits BEFORE the final tanh softcap + // (`logit_softcap`). 0 = no multiplier. + float output_multiplier; + // Epsilon for the post-attention / post-FFN norms, which differs from + // `norm_eps` here (1e-8 vs 1e-5). 0 = reuse `norm_eps` everywhere. + float post_norm_eps; + // Per-layer NoPE mask: bit i = 1 means layer i applies NO rotary at all + // (the reference passes `position_embeddings=None` there). On Muse + // Glimmer these are exactly the global/full-attention layers, so the + // mask is the complement of `swa_layers` — but it is stored + // independently because the two are separate config keys upstream + // (`layer_types` vs `layer_rope_theta`) and need not agree. + // Up to 512 layers (64 bytes). All zero = every layer gets RoPE. + uint8_t nope_layers[64]; + // Epsilon for a weightless RMSNorm the runtime applies to the token + // embedding right after lookup. 0 = already folded into the embedding + // rows at convert time (the HF path does this exactly), or no such norm. + // Only the GGUF path sets it: k-quant passthrough keeps the rows packed, + // so folding would mean dequantizing them. + float embed_norm_eps; + + // Perception-tower extras (vision_arch == 2). The tower is a LayerNorm + // ViT with SEPARATE biased q/k/v projections, alternating window/full + // attention, a bias-free Linear patch embed over flattened pixels, and a + // bilinearly-resampled learned position grid. + // Per-layer window-attention mask: bit i = 1 means layer i attends only + // within its window; 0 = full attention across all patches. + uint8_t vision_window_layers[64]; + uint32_t vision_window_size; // window side in PIXELS (pos_emb_height * patch_size, e.g. 448) + uint32_t vision_pos_embed_h; // learned position grid height (e.g. 32) + uint32_t vision_pos_embed_w; // learned position grid width (e.g. 32) + uint32_t vision_adapter_dim; // projector hidden width (`projector_hidden_size`, e.g. 4096) + uint32_t video_token_id; // text-side placeholder token for video features (0 = none) } BaseRTModelConfig; /// Transcription result statistics.