From 591097e46c7dcbe2f4d7c8b4dc1a81ac55324ade Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Fri, 2 Oct 2026 12:02:44 +0800 Subject: [PATCH 1/2] cua_s1: preprocess RGB images for native vision on CPU --- Cargo.lock | 1 + recipe/cua_s1/native.md | 3 + recipe/cua_s1/native_image_preprocess.md | 78 +++++ src/models/cua_s1/native/Cargo.toml | 7 + .../cua_s1/native/THIRD_PARTY_NOTICES.md | 24 ++ .../native/examples/preprocess_image.rs | 75 +++++ src/models/cua_s1/native/licenses/APACHE-2.0 | 203 ++++++++++++ .../cua_s1/native/licenses/PILLOW-LICENSE | 30 ++ .../cua_s1/native/licenses/PYTORCH-LICENSE | 84 +++++ .../cua_s1/native/src/image_preprocess.rs | 246 ++++++++++++++ src/models/cua_s1/native/src/lib.rs | 1 + .../fixtures/image_preprocess/README.md | 30 ++ .../fixtures/image_preprocess/generate.py | 119 +++++++ .../fixtures/image_preprocess/manifest.json | 311 ++++++++++++++++++ .../image_preprocess/preprocessor_config.json | 21 ++ tests/cua_s1/image_preprocess.rs | 223 +++++++++++++ 16 files changed, 1456 insertions(+) create mode 100644 recipe/cua_s1/native_image_preprocess.md create mode 100644 src/models/cua_s1/native/THIRD_PARTY_NOTICES.md create mode 100644 src/models/cua_s1/native/examples/preprocess_image.rs create mode 100644 src/models/cua_s1/native/licenses/APACHE-2.0 create mode 100644 src/models/cua_s1/native/licenses/PILLOW-LICENSE create mode 100644 src/models/cua_s1/native/licenses/PYTORCH-LICENSE create mode 100644 src/models/cua_s1/native/src/image_preprocess.rs create mode 100644 tests/cua_s1/fixtures/image_preprocess/README.md create mode 100644 tests/cua_s1/fixtures/image_preprocess/generate.py create mode 100644 tests/cua_s1/fixtures/image_preprocess/manifest.json create mode 100644 tests/cua_s1/fixtures/image_preprocess/preprocessor_config.json create mode 100644 tests/cua_s1/image_preprocess.rs diff --git a/Cargo.lock b/Cargo.lock index 4b9adda..fad6575 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -954,6 +954,7 @@ dependencies = [ "safetensors 0.8.0", "serde", "serde_json", + "sha2", "tokenizers", "tokio", ] diff --git a/recipe/cua_s1/native.md b/recipe/cua_s1/native.md index 018feff..8b1d839 100644 --- a/recipe/cua_s1/native.md +++ b/recipe/cua_s1/native.md @@ -41,3 +41,6 @@ cargo test -p omni-cua-s1-native CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ cargo test --release -p omni-cua-s1-native --test kernels -- --ignored ``` + +For the separate decoded-RGB8 CPU preprocessing API and example, see +[native image preprocessing](native_image_preprocess.md). diff --git a/recipe/cua_s1/native_image_preprocess.md b/recipe/cua_s1/native_image_preprocess.md new file mode 100644 index 0000000..eb63043 --- /dev/null +++ b/recipe/cua_s1/native_image_preprocess.md @@ -0,0 +1,78 @@ +# Native CPU image preprocessing + +The native crate exposes `image_preprocess::preprocess_rgb8(width, height, rgb)` +for **already decoded, interleaved RGB8** data. It prepares the image tensor for +the fixed Qwen3.5-4B / Cua-S1 4B processor. It does not decode PNG/JPEG, fetch +URLs, handle HTTP requests, run the vision encoder, or use a GPU. The existing +native text worker remains separate. + +The input must contain exactly `width * height * 3` bytes, in row-major RGB +order. Both dimensions must be nonzero and at most 2048; the area must be at +most 1,048,576 pixels and the aspect ratio at most 200. The library checks +geometry, lengths, and allocation arithmetic before creating image buffers. +These are input limits; smart resize can produce a side longer than 2048 for +very narrow inputs. + +The fixed processor uses a factor of 32, minimum area 65,536 and maximum area +16,777,216. Smart resize follows Python ties-to-even rounding and floating-point +square-root scaling with floor/ceil. Resampling matches the CPU torchvision +uint8 bicubic antialias path: Keys cubic coefficient `a = -0.5`, float64 weights, +per-axis int16 fixed-point coefficients, horizontal then vertical passes, and +rounding/clamping to uint8 after each pass. Unchanged axes bypass resampling. +The maximum-area downscale branch is retained for parity with the processor, +although the smaller input cap makes it unreachable through this API. + +`ProcessedImage` contains: + +- `pixel_values: Vec`, contiguous `[patches, 1536]` values normalized as + `(pixel - 127.5) / 127.5` using float32 operations. +- `image_grid_thw: [usize; 3]`, equal to `[1, resized_height / 16, resized_width / 16]`. +- `resized_width` and `resized_height`. +- `image_tokens()`, the patch count divided by four for the 2×2 spatial merge. + +Packing order is `block_y, block_x, merge_y (2), merge_x (2), channel (3), +temporal repeat (2), patch_y (16), patch_x (16)`. Each single image is repeated +across the two temporal positions. There is no video input support. + +## Run without a GPU + +From the repository root, provide a raw RGB8 file and its dimensions: + +```sh +cargo run --locked -p omni-cua-s1-native --example preprocess_image -- \ + 256 256 image.rgb pixel_values.f32 +``` + +The example prints the tensor shape, grid, resized dimensions and image token +count. The optional fourth argument writes every output float as little-endian +float32, with no header. It validates argument count, decimal dimensions and +exact file length, and bounds the input read. No model weights, Python, CUDA +library or image decoder are needed for this command. + +## Reference and validation + +Reference hashes are produced by the actual Hugging Face `AutoImageProcessor` +on CPU, with Python 3.12 and these exact package pins: Transformers 5.17.0, +PyTorch 2.14.0, torchvision 0.29.0, NumPy 2.5.3 and Pillow 11.3.0. The processor +configuration is the Qwen3.5-4B file at revision +`851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a` (see the fixture manifest for its URL +and SHA-256). Fixtures and regeneration instructions live in +[`tests/cua_s1/fixtures/image_preprocess/`](../../tests/cua_s1/fixtures/image_preprocess/). + +```sh +cargo test --locked -p omni-cua-s1-native --test image_preprocess +cargo test --release --locked -p omni-cua-s1-native --test image_preprocess +``` + +The tests compare input hashes, shape/grid metadata and SHA-256 of **every +little-endian float32 output byte** for 14 deterministic images. Cases include +tiny images, noise, ramps, checkerboards, ties-to-even dimensions, unchanged +axes, one- and two-axis resizes, extreme aspect ratios and the input area cap. +Additional checks cover all RGB byte values, channel/temporal/patch order, +constant images, inclusive limits and malformed geometry/buffers. This +validates the pinned CPU preprocessing behavior; it does not establish CUDA +preprocessing, image decoding, vision inference, or end-to-end model parity. + +The implementation adapts upstream algorithms; retained attributions and +license texts are in +[`THIRD_PARTY_NOTICES.md`](../../src/models/cua_s1/native/THIRD_PARTY_NOTICES.md). diff --git a/src/models/cua_s1/native/Cargo.toml b/src/models/cua_s1/native/Cargo.toml index 2e8caf6..0861b42 100644 --- a/src/models/cua_s1/native/Cargo.toml +++ b/src/models/cua_s1/native/Cargo.toml @@ -22,3 +22,10 @@ serde_json = { version = "1.0.149", features = ["float_roundtrip", "preserve_ord # the onig regex backend, as in the Python tokenizers wheel tokenizers = { version = "=0.22.2", default-features = false, features = ["onig"] } tokio = { version = "1.49.0", features = ["macros", "net", "rt-multi-thread", "sync"] } + +[[test]] +name = "image_preprocess" +path = "../../../../tests/cua_s1/image_preprocess.rs" + +[dev-dependencies] +sha2 = "0.10" diff --git a/src/models/cua_s1/native/THIRD_PARTY_NOTICES.md b/src/models/cua_s1/native/THIRD_PARTY_NOTICES.md new file mode 100644 index 0000000..54b788e --- /dev/null +++ b/src/models/cua_s1/native/THIRD_PARTY_NOTICES.md @@ -0,0 +1,24 @@ +# Image preprocessing attributions + +`src/image_preprocess.rs` is a Rust adaptation of the following algorithms. +It is modified for decoded interleaved RGB8 input, fixed Cua-S1 4B settings, +bounded dimensions, standard-library buffers, and a standalone CPU API. + +- PyTorch 2.14.0, `aten/src/ATen/native/cpu/UpSampleKernel.cpp` + (`_compute_indices_min_size_weights_aa`, `_compute_index_ranges_int16_weights`, + and the separable uint8 horizontal/vertical loops), plus the cubic polynomial + helpers in `aten/src/ATen/native/UpSample.h`. See [PyTorch license](licenses/PYTORCH-LICENSE) + for the retained copyright notices, redistribution conditions, and disclaimer. +- PyTorch's bicubic filter credits Pillow's `src/libImaging/Resample.c`. + The retained PIL/Pillow notice is in [Pillow license](licenses/PILLOW-LICENSE). +- Transformers 5.17.0, + `src/transformers/models/qwen2_vl/image_processing_qwen2_vl.py` (smart resize + and patch ordering) and `src/transformers/image_processing_backends.py` + (fused normalization). Copyright 2024 The Qwen team, Alibaba Group and the + HuggingFace Inc. team. All rights reserved. The backend file is + Copyright 2025 The HuggingFace Inc. team. Licensed under the + [Apache License, Version 2.0](licenses/APACHE-2.0). + +No upstream runtime or image decoder is linked by this module. The above +notices and license texts must accompany redistributed adaptations as required +by their respective licenses. diff --git a/src/models/cua_s1/native/examples/preprocess_image.rs b/src/models/cua_s1/native/examples/preprocess_image.rs new file mode 100644 index 0000000..310c661 --- /dev/null +++ b/src/models/cua_s1/native/examples/preprocess_image.rs @@ -0,0 +1,75 @@ +//! CPU-only RGB8 preprocessing; run from the repository root with: +//! cargo run -p omni-cua-s1-native --example preprocess_image -- 256 256 image.rgb + +use std::{ + env, + fs::File, + io::{BufWriter, Read, Write}, +}; + +use anyhow::{Context, Result, ensure}; +use omni_cua_s1_native::image_preprocess::preprocess_rgb8; + +fn main() -> Result<()> { + let args: Vec<_> = env::args_os().skip(1).collect(); + ensure!( + args.len() == 3 || args.len() == 4, + "usage: preprocess_image WIDTH HEIGHT RAW_RGB_PATH [OUTPUT_F32_PATH]" + ); + let dimension = |index: usize| -> Result { + let text = args[index] + .to_str() + .context("dimensions must be UTF-8 decimal integers")?; + ensure!( + !text.is_empty() && text.bytes().all(|byte| byte.is_ascii_digit()), + "dimensions must be unsigned decimal integers" + ); + text.parse().context("dimension is too large") + }; + let width = dimension(0)?; + let height = dimension(1)?; + // Bound the file read before allocating. The library validates the complete + // contract too; these checks keep malformed CLI inputs cheap to reject. + ensure!( + width > 0 && height > 0 && width <= 2048 && height <= 2048, + "dimensions must be in 1..=2048" + ); + let area = width.checked_mul(height).context("image area overflow")?; + ensure!( + area <= 1_048_576, + "image area must not exceed 1048576 pixels" + ); + ensure!( + width.max(height) <= width.min(height) * 200, + "image aspect ratio must not exceed 200" + ); + let expected = area.checked_mul(3).context("RGB length overflow")?; + let mut rgb = Vec::with_capacity(expected + 1); + File::open(&args[2]) + .context("opening RGB input")? + .take((expected + 1) as u64) + .read_to_end(&mut rgb) + .context("reading RGB input")?; + ensure!( + rgb.len() == expected, + "RGB input must contain exactly {expected} bytes" + ); + let image = preprocess_rgb8(width, height, &rgb)?; + println!( + "pixel_values shape: [{}, 1536]", + image.pixel_values.len() / 1536 + ); + println!("image_grid_thw: {:?}", image.image_grid_thw); + println!("resized: {}x{}", image.resized_width, image.resized_height); + println!("image_tokens: {}", image.image_tokens()); + if let Some(path) = args.get(3) { + let mut output = BufWriter::new(File::create(path).context("creating float32 output")?); + for value in image.pixel_values { + output + .write_all(&value.to_le_bytes()) + .context("writing float32 output")?; + } + output.flush().context("flushing float32 output")?; + } + Ok(()) +} diff --git a/src/models/cua_s1/native/licenses/APACHE-2.0 b/src/models/cua_s1/native/licenses/APACHE-2.0 new file mode 100644 index 0000000..68b7d66 --- /dev/null +++ b/src/models/cua_s1/native/licenses/APACHE-2.0 @@ -0,0 +1,203 @@ +Copyright 2018- The Hugging Face team. All rights reserved. + + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "[]" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright [yyyy] [name of copyright owner] + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/src/models/cua_s1/native/licenses/PILLOW-LICENSE b/src/models/cua_s1/native/licenses/PILLOW-LICENSE new file mode 100644 index 0000000..10dd42d --- /dev/null +++ b/src/models/cua_s1/native/licenses/PILLOW-LICENSE @@ -0,0 +1,30 @@ +The Python Imaging Library (PIL) is + + Copyright © 1997-2011 by Secret Labs AB + Copyright © 1995-2011 by Fredrik Lundh and contributors + +Pillow is the friendly PIL fork. It is + + Copyright © 2010 by Jeffrey A. Clark and contributors + +Like PIL, Pillow is licensed under the open source MIT-CMU License: + +By obtaining, using, and/or copying this software and/or its associated +documentation, you agree that you have read, understood, and will comply +with the following terms and conditions: + +Permission to use, copy, modify and distribute this software and its +documentation for any purpose and without fee is hereby granted, +provided that the above copyright notice appears in all copies, and that +both that copyright notice and this permission notice appear in supporting +documentation, and that the name of Secret Labs AB or the author not be +used in advertising or publicity pertaining to distribution of the software +without specific, written prior permission. + +SECRET LABS AB AND THE AUTHOR DISCLAIMS ALL WARRANTIES WITH REGARD TO THIS +SOFTWARE, INCLUDING ALL IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS. +IN NO EVENT SHALL SECRET LABS AB OR THE AUTHOR BE LIABLE FOR ANY SPECIAL, +INDIRECT OR CONSEQUENTIAL DAMAGES OR ANY DAMAGES WHATSOEVER RESULTING FROM +LOSS OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE +OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE OR +PERFORMANCE OF THIS SOFTWARE. diff --git a/src/models/cua_s1/native/licenses/PYTORCH-LICENSE b/src/models/cua_s1/native/licenses/PYTORCH-LICENSE new file mode 100644 index 0000000..c23172f --- /dev/null +++ b/src/models/cua_s1/native/licenses/PYTORCH-LICENSE @@ -0,0 +1,84 @@ +From PyTorch: + +Copyright (c) 2016- Facebook, Inc (Adam Paszke) +Copyright (c) 2014- Facebook, Inc (Soumith Chintala) +Copyright (c) 2011-2014 Idiap Research Institute (Ronan Collobert) +Copyright (c) 2012-2014 Deepmind Technologies (Koray Kavukcuoglu) +Copyright (c) 2011-2012 NEC Laboratories America (Koray Kavukcuoglu) +Copyright (c) 2011-2013 NYU (Clement Farabet) +Copyright (c) 2006-2010 NEC Laboratories America (Ronan Collobert, Leon Bottou, Iain Melvin, Jason Weston) +Copyright (c) 2006 Idiap Research Institute (Samy Bengio) +Copyright (c) 2001-2004 Idiap Research Institute (Ronan Collobert, Samy Bengio, Johnny Mariethoz) + +From Caffe2: + +Copyright (c) 2016-present, Facebook Inc. All rights reserved. + +All contributions by Facebook: +Copyright (c) 2016 Facebook Inc. + +All contributions by Google: +Copyright (c) 2015 Google Inc. +All rights reserved. + +All contributions by Yangqing Jia: +Copyright (c) 2015 Yangqing Jia +All rights reserved. + +All contributions by Kakao Brain: +Copyright 2019-2020 Kakao Brain + +All contributions by Cruise LLC: +Copyright (c) 2022 Cruise LLC. +All rights reserved. + +All contributions by Tri Dao: +Copyright (c) 2024 Tri Dao. +All rights reserved. + +All contributions by Arm: +Copyright (c) 2021, 2023-2025 Arm Limited and/or its affiliates + +All contributions from Caffe: +Copyright(c) 2013, 2014, 2015, the respective contributors +All rights reserved. + +All other contributions: +Copyright(c) 2015, 2016 the respective contributors +All rights reserved. + +Caffe2 uses a copyright model similar to Caffe: each contributor holds +copyright over their contributions to Caffe2. The project versioning records +all such contribution and copyright details. If a contributor wants to further +mark their specific copyright on a particular contribution, they should +indicate their copyright solely in the commit message of the change when it is +committed. + +All rights reserved. + +Redistribution and use in source and binary forms, with or without +modification, are permitted provided that the following conditions are met: + +1. Redistributions of source code must retain the above copyright + notice, this list of conditions and the following disclaimer. + +2. Redistributions in binary form must reproduce the above copyright + notice, this list of conditions and the following disclaimer in the + documentation and/or other materials provided with the distribution. + +3. Neither the names of Facebook, Deepmind Technologies, NYU, NEC Laboratories America + and IDIAP Research Institute nor the names of its contributors may be + used to endorse or promote products derived from this software without + specific prior written permission. + +THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" +AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE +IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE +ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE +LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR +CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF +SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS +INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN +CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) +ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +POSSIBILITY OF SUCH DAMAGE. diff --git a/src/models/cua_s1/native/src/image_preprocess.rs b/src/models/cua_s1/native/src/image_preprocess.rs new file mode 100644 index 0000000..9009fb2 --- /dev/null +++ b/src/models/cua_s1/native/src/image_preprocess.rs @@ -0,0 +1,246 @@ +//! CPU preprocessing for the fixed Qwen3.5-4B / Cua-S1 image processor. +//! +//! Input is already decoded, interleaved RGB8. No image codecs, GPU, model +//! weights, or HTTP handling are involved. The output is row-major +//! `[patches, 1536]`, ready for a separate vision encoder. +//! +//! The resize algorithm is a Rust adaptation of PyTorch's CPU uint8 bicubic +//! antialias implementation (which credits Pillow). Smart resize and packing +//! follow Transformers' Qwen2VLImageProcessor. See `../THIRD_PARTY_NOTICES.md`. + +use std::borrow::Cow; + +use anyhow::{Result, ensure}; + +const PATCH_SIZE: usize = 16; +const MERGE_SIZE: usize = 2; +const FACTOR: usize = PATCH_SIZE * MERGE_SIZE; +const PATCH_VALUES: usize = 3 * 2 * PATCH_SIZE * PATCH_SIZE; +const MIN_PIXELS: usize = 65_536; +const MAX_PIXELS: usize = 16_777_216; + +/// Normalized image patches and the spatial metadata used by the vision model. +#[derive(Debug)] +pub struct ProcessedImage { + /// Contiguous row-major `[patches, 1536]` float32 values. + pub pixel_values: Vec, + /// `[1, resized_height / 16, resized_width / 16]`. + pub image_grid_thw: [usize; 3], + pub resized_width: usize, + pub resized_height: usize, +} + +impl ProcessedImage { + /// Number of image tokens after the model's 2-by-2 spatial merge. + pub fn image_tokens(&self) -> usize { + self.pixel_values.len() / PATCH_VALUES / (MERGE_SIZE * MERGE_SIZE) + } +} + +/// Preprocess a decoded RGB8 image using the fixed 4B processor settings. +/// +/// Rejects empty dimensions, sides over 2048, area over 1,048,576 pixels, +/// aspect ratios over 200, and buffers whose length is not `width * height * 3`. +/// Geometry and buffer arithmetic are checked before allocating image buffers. +pub fn preprocess_rgb8(width: usize, height: usize, rgb: &[u8]) -> Result { + ensure!(width > 0 && height > 0, "image dimensions must be nonzero"); + ensure!( + width <= 2048 && height <= 2048, + "image sides must not exceed 2048" + ); + let area = width + .checked_mul(height) + .ok_or_else(|| anyhow::anyhow!("image area overflow"))?; + ensure!( + area <= 1_048_576, + "image area must not exceed 1048576 pixels" + ); + ensure!( + width.max(height) <= width.min(height) * 200, + "image aspect ratio must not exceed 200" + ); + let input_len = area + .checked_mul(3) + .ok_or_else(|| anyhow::anyhow!("RGB buffer length overflow"))?; + ensure!( + rgb.len() == input_len, + "RGB buffer length must be {input_len}, got {}", + rgb.len() + ); + + let (resized_width, resized_height) = smart_resize(width, height); + let resized_area = resized_width + .checked_mul(resized_height) + .ok_or_else(|| anyhow::anyhow!("resized area overflow"))?; + let resized_len = resized_area + .checked_mul(3) + .ok_or_else(|| anyhow::anyhow!("resized buffer length overflow"))?; + let horizontal_len = resized_width + .checked_mul(height) + .and_then(|area| area.checked_mul(3)) + .ok_or_else(|| anyhow::anyhow!("horizontal buffer length overflow"))?; + let output_len = resized_area + .checked_mul(6) + .ok_or_else(|| anyhow::anyhow!("patch buffer length overflow"))?; + output_len + .checked_mul(std::mem::size_of::()) + .ok_or_else(|| anyhow::anyhow!("patch buffer byte length overflow"))?; + + let mut resized = Cow::Borrowed(rgb); + if resized_width != width { + let axis = AxisWeights::new(width, resized_width); + let mut horizontal = vec![0; horizontal_len]; + for y in 0..height { + for (x, kernel) in axis.kernels.iter().enumerate() { + for channel in 0..3 { + horizontal[(y * resized_width + x) * 3 + channel] = + axis.apply(kernel, |source_x| rgb[(y * width + source_x) * 3 + channel]); + } + } + } + resized = Cow::Owned(horizontal); + } + if resized_height != height { + let axis = AxisWeights::new(height, resized_height); + let mut vertical = vec![0; resized_len]; + for (y, kernel) in axis.kernels.iter().enumerate() { + for x in 0..resized_width { + for channel in 0..3 { + vertical[(y * resized_width + x) * 3 + channel] = axis + .apply(kernel, |source_y| { + resized[(source_y * resized_width + x) * 3 + channel] + }); + } + } + } + resized = Cow::Owned(vertical); + } + + let mut pixel_values = Vec::with_capacity(output_len); + for block_y in 0..resized_height / FACTOR { + for block_x in 0..resized_width / FACTOR { + for merge_y in 0..MERGE_SIZE { + for merge_x in 0..MERGE_SIZE { + for channel in 0..3 { + for _temporal in 0..2 { + for patch_y in 0..PATCH_SIZE { + for patch_x in 0..PATCH_SIZE { + let y = block_y * FACTOR + merge_y * PATCH_SIZE + patch_y; + let x = block_x * FACTOR + merge_x * PATCH_SIZE + patch_x; + let pixel = resized[(y * resized_width + x) * 3 + channel]; + // Match the fused float32 torchvision normalization, + // including its operation order (no reciprocal multiply). + pixel_values.push((f32::from(pixel) - 127.5) / 127.5); + } + } + } + } + } + } + } + } + Ok(ProcessedImage { + pixel_values, + image_grid_thw: [1, resized_height / PATCH_SIZE, resized_width / PATCH_SIZE], + resized_width, + resized_height, + }) +} + +fn smart_resize(width: usize, height: usize) -> (usize, usize) { + // Python round uses ties-to-even; Rust's ordinary round does not. + let mut w = (width as f64 / FACTOR as f64).round_ties_even() as usize * FACTOR; + let mut h = (height as f64 / FACTOR as f64).round_ties_even() as usize * FACTOR; + if w * h > MAX_PIXELS { + let beta = ((width * height) as f64 / MAX_PIXELS as f64).sqrt(); + w = ((width as f64 / beta / FACTOR as f64).floor() as usize * FACTOR).max(FACTOR); + h = ((height as f64 / beta / FACTOR as f64).floor() as usize * FACTOR).max(FACTOR); + } else if w * h < MIN_PIXELS { + let beta = (MIN_PIXELS as f64 / (width * height) as f64).sqrt(); + w = (width as f64 * beta / FACTOR as f64).ceil() as usize * FACTOR; + h = (height as f64 * beta / FACTOR as f64).ceil() as usize * FACTOR; + } + (w, h) +} + +struct Kernel { + start: usize, + weights: Vec, +} + +struct AxisWeights { + kernels: Vec, + precision: u32, +} + +impl AxisWeights { + fn new(input: usize, output: usize) -> Self { + let scale = input as f64 / output as f64; + let support = 2.0 * scale.max(1.0); + let invscale = if scale >= 1.0 { 1.0 / scale } else { 1.0 }; + let max_size = support.ceil() as usize * 2 + 1; + let mut maximum = 0.0_f64; + let mut floating = Vec::with_capacity(output); + for index in 0..output { + let center = scale * (index as f64 + 0.5); + // C++ conversion truncates toward zero before clamping the bounds. + let start = ((center - support + 0.5) as isize).max(0) as usize; + let end = ((center + support + 0.5) as usize).min(input); + let count = end.saturating_sub(start).min(max_size); + let mut weights: Vec = (0..count) + .map(|j| cubic((j as f64 + start as f64 - center + 0.5) * invscale)) + .collect(); + let total: f64 = weights.iter().sum(); + if total != 0.0 { + for weight in &mut weights { + *weight /= total; + maximum = maximum.max(*weight); + } + } + floating.push((start, weights)); + } + // One precision for the whole axis, as in PyTorch's int16 path. + let mut precision = 0; + while precision < 22 { + if (0.5 + maximum * f64::from(1 << (precision + 1))) as i32 >= (1 << 15) { + break; + } + precision += 1; + } + let multiplier = f64::from(1 << precision); + let kernels = floating + .into_iter() + .map(|(start, weights)| Kernel { + start, + weights: weights + .into_iter() + .map(|weight| { + let value = weight * multiplier; + (value + if value < 0.0 { -0.5 } else { 0.5 }) as i16 + }) + .collect(), + }) + .collect(); + Self { kernels, precision } + } + + fn apply(&self, kernel: &Kernel, pixel: impl Fn(usize) -> u8) -> u8 { + let mut accumulator = 1_i32 << (self.precision - 1); + for (offset, &weight) in kernel.weights.iter().enumerate() { + accumulator += i32::from(pixel(kernel.start + offset)) * i32::from(weight); + } + (accumulator >> self.precision).clamp(0, 255) as u8 + } +} + +fn cubic(x: f64) -> f64 { + let x = x.abs(); + const A: f64 = -0.5; + if x < 1.0 { + ((A + 2.0) * x - (A + 3.0)) * x * x + 1.0 + } else if x < 2.0 { + ((A * x - 5.0 * A) * x + 8.0 * A) * x - 4.0 * A + } else { + 0.0 + } +} diff --git a/src/models/cua_s1/native/src/lib.rs b/src/models/cua_s1/native/src/lib.rs index 0d3aa5f..ff12872 100644 --- a/src/models/cua_s1/native/src/lib.rs +++ b/src/models/cua_s1/native/src/lib.rs @@ -5,5 +5,6 @@ pub mod contract; pub mod cuda; pub mod engine; +pub mod image_preprocess; pub mod json; pub mod model; diff --git a/tests/cua_s1/fixtures/image_preprocess/README.md b/tests/cua_s1/fixtures/image_preprocess/README.md new file mode 100644 index 0000000..1eff2a5 --- /dev/null +++ b/tests/cua_s1/fixtures/image_preprocess/README.md @@ -0,0 +1,30 @@ +# Native RGB preprocessing reference fixtures + +`preprocessor_config.json` is from pinned Qwen3.5-4B revision +`851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a`: +[upstream configuration](https://huggingface.co/Qwen/Qwen3.5-4B/resolve/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a/preprocessor_config.json). + +`manifest.json` contains complete FP32 output byte hashes generated by the actual +Transformers `AutoImageProcessor` with this configuration on CPU. It records +package versions, platform, configuration fingerprint, deterministic RGB input +hashes, resized dimensions, grid, patch shape and image token count. No weights +or generated output tensors are committed. Rust tests duplicate only the input +generator and compare every output byte through SHA-256. + +The input generator covers constant RGB, spatial/channel ramps, checkerboards +and seeded xorshift32 noise. Cases include no resize, upsampling, downsampling, +Python ties-to-even dimensions, portrait/wide inputs, the 200:1 aspect boundary +and the request pixel cap. These establish CPU preprocessing parity on the +recorded reference environment; they are not an encoder or GPU accuracy test. + +Regenerate with Python 3.12 and the pinned packages (no GPU or model downloads): + +```sh +python -m pip install torch==2.14.0 torchvision==0.29.0 transformers==5.17.0 \ + numpy==2.5.3 Pillow==11.3.0 +python tests/cua_s1/fixtures/image_preprocess/generate.py +``` + +The generator rejects mismatched package versions. Keep new generated hashes +reviewable against the recorded source; do not update expected values merely +to make a native mismatch disappear. diff --git a/tests/cua_s1/fixtures/image_preprocess/generate.py b/tests/cua_s1/fixtures/image_preprocess/generate.py new file mode 100644 index 0000000..70efc1b --- /dev/null +++ b/tests/cua_s1/fixtures/image_preprocess/generate.py @@ -0,0 +1,119 @@ +"""Regenerate CPU parity hashes using the actual pinned Hugging Face processor. + +Run in a Python 3.12 environment with torch==2.14.0, torchvision==0.29.0, +transformers==5.17.0, numpy==2.5.3 and Pillow==11.3.0. No model weights or GPU. +""" + +import hashlib +import importlib.metadata +import json +import platform +from pathlib import Path + +import numpy as np +import torch +from PIL import Image +from transformers import AutoImageProcessor + +PACKAGES = { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "numpy": "2.5.3", + "Pillow": "11.3.0", +} +CASES = [ + ("single_pixel", 1, 1, "constant", 1), + ("tiny_noise", 17, 19, "noise", 17), + ("aligned_noise", 256, 256, "noise", 123), + ("odd_noise", 319, 241, "noise", 42), + ("tie_down", 272, 256, "noise", 999), + ("tie_up", 304, 256, "noise", 101), + ("both_down", 271, 271, "checker", 1), + ("one_axis_down", 256, 257, "ramp", 1), + ("wide", 640, 320, "ramp", 1), + ("portrait", 319, 641, "noise", 23), + ("aspect_limit", 200, 1, "noise", 47), + ("tall_aspect_limit", 7, 1400, "checker", 1), + ("input_pixel_limit", 2048, 512, "noise", 55), + ("large_round_up", 1600, 600, "noise", 77), +] + + +def pixels(width, height, pattern, seed): + """Only input generation is duplicated in Rust; outputs come from HF.""" + if pattern == "noise": + out = bytearray(width * height * 3) + state = seed + for i in range(len(out)): + state ^= (state << 13) & 0xFFFFFFFF + state ^= state >> 17 + state ^= (state << 5) & 0xFFFFFFFF + out[i] = state & 255 + return bytes(out) + return bytes( + ( + [0, 128, 255][c] + if pattern == "constant" + else (255 if (x + y + c) % 2 else 0) + if pattern == "checker" + else (x * 13 + y * 7 + c * 83) % 256 + ) + for y in range(height) + for x in range(width) + for c in range(3) + ) + + +def evaluate(processor, name, width, height, pattern, seed): + raw = pixels(width, height, pattern, seed) + array = np.frombuffer(raw, dtype=np.uint8).reshape(height, width, 3) + result = processor(images=Image.fromarray(array), return_tensors="pt", device="cpu") + tensor = result["pixel_values"].contiguous() + grid = result["image_grid_thw"][0].tolist() + assert tensor.dtype == torch.float32 and tensor.device.type == "cpu" + return { + "name": name, + "width": width, + "height": height, + "pattern": pattern, + "seed": seed, + "input_sha256": hashlib.sha256(raw).hexdigest(), + "grid": grid, + "shape": list(tensor.shape), + "image_tokens": grid[1] * grid[2] // 4, + "resized_width": grid[2] * 16, + "resized_height": grid[1] * 16, + "output_sha256": hashlib.sha256( + tensor.numpy().astype(" f32 { + (f32::from(pixel) - 127.5) / 127.5 +} + +#[test] +fn tiny_rgb_is_upscaled_and_temporally_repeated() { + let image = preprocess_rgb8(1, 1, &[0, 127, 255]).unwrap(); + assert_eq!((image.resized_width, image.resized_height), (256, 256)); + assert_eq!(image.image_grid_thw, [1, 16, 16]); + assert_eq!(image.image_tokens(), 64); + assert_eq!(image.pixel_values.len(), 256 * 1536); + for patch in image.pixel_values.as_chunks::<1536>().0 { + for (channel, expected) in [0, 127, 255].into_iter().enumerate() { + assert!( + patch[channel * 512..(channel + 1) * 512] + .iter() + .all(|&value| value.to_bits() == normalized(expected).to_bits()) + ); + } + } +} + +#[test] +fn identity_resize_preserves_every_byte_and_patch_merge_order() { + let width = 256; + let height = 256; + let rgb: Vec = (0..height) + .flat_map(|y| (0..width).flat_map(move |x| [x as u8, y as u8, (x ^ y) as u8])) + .collect(); + let image = preprocess_rgb8(width, height, &rgb).unwrap(); + for block_y in 0..8 { + for block_x in 0..8 { + for merge_y in 0..2 { + for merge_x in 0..2 { + let patch = ((block_y * 8 + block_x) * 2 + merge_y) * 2 + merge_x; + for channel in 0..3 { + for temporal in 0..2 { + for py in 0..16 { + for px in 0..16 { + let x = block_x * 32 + merge_x * 16 + px; + let y = block_y * 32 + merge_y * 16 + py; + let index = patch * 1536 + + channel * 512 + + temporal * 256 + + py * 16 + + px; + assert_eq!( + image.pixel_values[index].to_bits(), + normalized(rgb[(y * width + x) * 3 + channel]).to_bits() + ); + } + } + } + } + } + } + } + } +} + +#[test] +fn smart_resize_uses_python_ties_even_rounding() { + for (side, expected) in [(272, 256), (304, 320)] { + let image = preprocess_rgb8(side, side, &vec![128; side * side * 3]).unwrap(); + assert_eq!( + (image.resized_width, image.resized_height), + (expected, expected) + ); + assert!( + image + .pixel_values + .iter() + .all(|&value| value == normalized(128)) + ); + } +} + +#[test] +fn rejects_invalid_geometry_before_buffer_length_validation() { + for (width, height) in [(0, 1), (1, 0), (0, 0)] { + assert_eq!( + preprocess_rgb8(width, height, &[]).unwrap_err().to_string(), + "image dimensions must be nonzero", + "{width}x{height}" + ); + } + for (width, height) in [(usize::MAX, 1), (1, usize::MAX), (usize::MAX, usize::MAX)] { + assert_eq!( + preprocess_rgb8(width, height, &[]).unwrap_err().to_string(), + "image sides must not exceed 2048", + "{width}x{height}" + ); + } + for (width, height, expected_error) in [ + (2049, 32, "image sides must not exceed 2048"), + (32, 2049, "image sides must not exceed 2048"), + (1025, 1024, "image area must not exceed 1048576 pixels"), + (201, 1, "image aspect ratio must not exceed 200"), + (1, 201, "image aspect ratio must not exceed 200"), + ] { + // A valid byte length ensures the geometry check itself rejects this + // image, rather than accidentally passing due to a truncated buffer. + let rgb = vec![0; width * height * 3]; + assert_eq!( + preprocess_rgb8(width, height, &rgb) + .unwrap_err() + .to_string(), + expected_error, + "{width}x{height}" + ); + } +} + +#[test] +fn rejects_incorrect_rgb_buffer_lengths() { + for rgb in [&[][..], &[1, 2][..], &[1, 2, 3, 4][..]] { + assert_eq!( + preprocess_rgb8(1, 1, rgb).unwrap_err().to_string(), + format!("RGB buffer length must be 3, got {}", rgb.len()) + ); + } +} + +#[test] +fn accepts_input_limits_inclusively() { + for (width, height) in [(200, 1), (1, 200), (2048, 512), (512, 2048)] { + let image = preprocess_rgb8(width, height, &vec![255; width * height * 3]).unwrap(); + assert_eq!(image.resized_width % 32, 0); + assert_eq!(image.resized_height % 32, 0); + assert!(image.pixel_values.iter().all(|&value| value == 1.0)); + } +} + +#[test] +fn matches_pinned_processor_full_output_hashes() { + use sha2::{Digest, Sha256}; + let manifest: serde_json::Value = + serde_json::from_str(include_str!("fixtures/image_preprocess/manifest.json")).unwrap(); + assert_eq!( + format!( + "{:x}", + Sha256::digest(include_bytes!( + "fixtures/image_preprocess/preprocessor_config.json" + )) + ), + manifest["config_sha256"].as_str().unwrap(), + "processor configuration must match the reference manifest" + ); + for case in manifest["cases"].as_array().unwrap() { + let name = case["name"].as_str().unwrap(); + let width = case["width"].as_u64().unwrap() as usize; + let height = case["height"].as_u64().unwrap() as usize; + let pattern = case["pattern"].as_str().unwrap(); + let mut state = case["seed"].as_u64().unwrap() as u32; + let mut rgb = Vec::with_capacity(width * height * 3); + for y in 0..height { + for x in 0..width { + for channel in 0..3 { + rgb.push(match pattern { + "noise" => { + state ^= state << 13; + state ^= state >> 17; + state ^= state << 5; + state as u8 + } + "constant" => [0, 128, 255][channel], + "checker" => { + if (x + y + channel) % 2 == 0 { + 0 + } else { + 255 + } + } + "ramp" => ((x * 13 + y * 7 + channel * 83) % 256) as u8, + _ => panic!("unknown fixture pattern {pattern}"), + }); + } + } + } + assert_eq!( + format!("{:x}", Sha256::digest(&rgb)), + case["input_sha256"].as_str().unwrap(), + "{name} input" + ); + let image = preprocess_rgb8(width, height, &rgb).unwrap(); + assert_eq!( + serde_json::json!(image.image_grid_thw), + case["grid"], + "{name} grid" + ); + assert_eq!( + serde_json::json!([image.pixel_values.len() / 1536, 1536]), + case["shape"], + "{name} shape" + ); + assert_eq!( + image.image_tokens(), + case["image_tokens"].as_u64().unwrap() as usize, + "{name} tokens" + ); + assert_eq!( + image.resized_width, + case["resized_width"].as_u64().unwrap() as usize, + "{name} width" + ); + assert_eq!( + image.resized_height, + case["resized_height"].as_u64().unwrap() as usize, + "{name} height" + ); + let mut hash = Sha256::new(); + for value in image.pixel_values { + hash.update(value.to_le_bytes()); + } + assert_eq!( + format!("{:x}", hash.finalize()), + case["output_sha256"].as_str().unwrap(), + "{name} output" + ); + } +} From 9ee058ff1c767993c66e01358a20c163f47b947b Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Fri, 2 Oct 2026 13:19:12 +0800 Subject: [PATCH 2/2] chore(cua_s1): limit PR diff to core implementation --- Cargo.lock | 1 - recipe/cua_s1/native.md | 3 - recipe/cua_s1/native_image_preprocess.md | 78 ----- src/models/cua_s1/native/Cargo.toml | 7 - .../native/examples/preprocess_image.rs | 75 ----- .../fixtures/image_preprocess/README.md | 30 -- .../fixtures/image_preprocess/generate.py | 119 ------- .../fixtures/image_preprocess/manifest.json | 311 ------------------ .../image_preprocess/preprocessor_config.json | 21 -- tests/cua_s1/image_preprocess.rs | 223 ------------- 10 files changed, 868 deletions(-) delete mode 100644 recipe/cua_s1/native_image_preprocess.md delete mode 100644 src/models/cua_s1/native/examples/preprocess_image.rs delete mode 100644 tests/cua_s1/fixtures/image_preprocess/README.md delete mode 100644 tests/cua_s1/fixtures/image_preprocess/generate.py delete mode 100644 tests/cua_s1/fixtures/image_preprocess/manifest.json delete mode 100644 tests/cua_s1/fixtures/image_preprocess/preprocessor_config.json delete mode 100644 tests/cua_s1/image_preprocess.rs diff --git a/Cargo.lock b/Cargo.lock index fad6575..4b9adda 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -954,7 +954,6 @@ dependencies = [ "safetensors 0.8.0", "serde", "serde_json", - "sha2", "tokenizers", "tokio", ] diff --git a/recipe/cua_s1/native.md b/recipe/cua_s1/native.md index 8b1d839..018feff 100644 --- a/recipe/cua_s1/native.md +++ b/recipe/cua_s1/native.md @@ -41,6 +41,3 @@ cargo test -p omni-cua-s1-native CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ cargo test --release -p omni-cua-s1-native --test kernels -- --ignored ``` - -For the separate decoded-RGB8 CPU preprocessing API and example, see -[native image preprocessing](native_image_preprocess.md). diff --git a/recipe/cua_s1/native_image_preprocess.md b/recipe/cua_s1/native_image_preprocess.md deleted file mode 100644 index eb63043..0000000 --- a/recipe/cua_s1/native_image_preprocess.md +++ /dev/null @@ -1,78 +0,0 @@ -# Native CPU image preprocessing - -The native crate exposes `image_preprocess::preprocess_rgb8(width, height, rgb)` -for **already decoded, interleaved RGB8** data. It prepares the image tensor for -the fixed Qwen3.5-4B / Cua-S1 4B processor. It does not decode PNG/JPEG, fetch -URLs, handle HTTP requests, run the vision encoder, or use a GPU. The existing -native text worker remains separate. - -The input must contain exactly `width * height * 3` bytes, in row-major RGB -order. Both dimensions must be nonzero and at most 2048; the area must be at -most 1,048,576 pixels and the aspect ratio at most 200. The library checks -geometry, lengths, and allocation arithmetic before creating image buffers. -These are input limits; smart resize can produce a side longer than 2048 for -very narrow inputs. - -The fixed processor uses a factor of 32, minimum area 65,536 and maximum area -16,777,216. Smart resize follows Python ties-to-even rounding and floating-point -square-root scaling with floor/ceil. Resampling matches the CPU torchvision -uint8 bicubic antialias path: Keys cubic coefficient `a = -0.5`, float64 weights, -per-axis int16 fixed-point coefficients, horizontal then vertical passes, and -rounding/clamping to uint8 after each pass. Unchanged axes bypass resampling. -The maximum-area downscale branch is retained for parity with the processor, -although the smaller input cap makes it unreachable through this API. - -`ProcessedImage` contains: - -- `pixel_values: Vec`, contiguous `[patches, 1536]` values normalized as - `(pixel - 127.5) / 127.5` using float32 operations. -- `image_grid_thw: [usize; 3]`, equal to `[1, resized_height / 16, resized_width / 16]`. -- `resized_width` and `resized_height`. -- `image_tokens()`, the patch count divided by four for the 2×2 spatial merge. - -Packing order is `block_y, block_x, merge_y (2), merge_x (2), channel (3), -temporal repeat (2), patch_y (16), patch_x (16)`. Each single image is repeated -across the two temporal positions. There is no video input support. - -## Run without a GPU - -From the repository root, provide a raw RGB8 file and its dimensions: - -```sh -cargo run --locked -p omni-cua-s1-native --example preprocess_image -- \ - 256 256 image.rgb pixel_values.f32 -``` - -The example prints the tensor shape, grid, resized dimensions and image token -count. The optional fourth argument writes every output float as little-endian -float32, with no header. It validates argument count, decimal dimensions and -exact file length, and bounds the input read. No model weights, Python, CUDA -library or image decoder are needed for this command. - -## Reference and validation - -Reference hashes are produced by the actual Hugging Face `AutoImageProcessor` -on CPU, with Python 3.12 and these exact package pins: Transformers 5.17.0, -PyTorch 2.14.0, torchvision 0.29.0, NumPy 2.5.3 and Pillow 11.3.0. The processor -configuration is the Qwen3.5-4B file at revision -`851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a` (see the fixture manifest for its URL -and SHA-256). Fixtures and regeneration instructions live in -[`tests/cua_s1/fixtures/image_preprocess/`](../../tests/cua_s1/fixtures/image_preprocess/). - -```sh -cargo test --locked -p omni-cua-s1-native --test image_preprocess -cargo test --release --locked -p omni-cua-s1-native --test image_preprocess -``` - -The tests compare input hashes, shape/grid metadata and SHA-256 of **every -little-endian float32 output byte** for 14 deterministic images. Cases include -tiny images, noise, ramps, checkerboards, ties-to-even dimensions, unchanged -axes, one- and two-axis resizes, extreme aspect ratios and the input area cap. -Additional checks cover all RGB byte values, channel/temporal/patch order, -constant images, inclusive limits and malformed geometry/buffers. This -validates the pinned CPU preprocessing behavior; it does not establish CUDA -preprocessing, image decoding, vision inference, or end-to-end model parity. - -The implementation adapts upstream algorithms; retained attributions and -license texts are in -[`THIRD_PARTY_NOTICES.md`](../../src/models/cua_s1/native/THIRD_PARTY_NOTICES.md). diff --git a/src/models/cua_s1/native/Cargo.toml b/src/models/cua_s1/native/Cargo.toml index 0861b42..2e8caf6 100644 --- a/src/models/cua_s1/native/Cargo.toml +++ b/src/models/cua_s1/native/Cargo.toml @@ -22,10 +22,3 @@ serde_json = { version = "1.0.149", features = ["float_roundtrip", "preserve_ord # the onig regex backend, as in the Python tokenizers wheel tokenizers = { version = "=0.22.2", default-features = false, features = ["onig"] } tokio = { version = "1.49.0", features = ["macros", "net", "rt-multi-thread", "sync"] } - -[[test]] -name = "image_preprocess" -path = "../../../../tests/cua_s1/image_preprocess.rs" - -[dev-dependencies] -sha2 = "0.10" diff --git a/src/models/cua_s1/native/examples/preprocess_image.rs b/src/models/cua_s1/native/examples/preprocess_image.rs deleted file mode 100644 index 310c661..0000000 --- a/src/models/cua_s1/native/examples/preprocess_image.rs +++ /dev/null @@ -1,75 +0,0 @@ -//! CPU-only RGB8 preprocessing; run from the repository root with: -//! cargo run -p omni-cua-s1-native --example preprocess_image -- 256 256 image.rgb - -use std::{ - env, - fs::File, - io::{BufWriter, Read, Write}, -}; - -use anyhow::{Context, Result, ensure}; -use omni_cua_s1_native::image_preprocess::preprocess_rgb8; - -fn main() -> Result<()> { - let args: Vec<_> = env::args_os().skip(1).collect(); - ensure!( - args.len() == 3 || args.len() == 4, - "usage: preprocess_image WIDTH HEIGHT RAW_RGB_PATH [OUTPUT_F32_PATH]" - ); - let dimension = |index: usize| -> Result { - let text = args[index] - .to_str() - .context("dimensions must be UTF-8 decimal integers")?; - ensure!( - !text.is_empty() && text.bytes().all(|byte| byte.is_ascii_digit()), - "dimensions must be unsigned decimal integers" - ); - text.parse().context("dimension is too large") - }; - let width = dimension(0)?; - let height = dimension(1)?; - // Bound the file read before allocating. The library validates the complete - // contract too; these checks keep malformed CLI inputs cheap to reject. - ensure!( - width > 0 && height > 0 && width <= 2048 && height <= 2048, - "dimensions must be in 1..=2048" - ); - let area = width.checked_mul(height).context("image area overflow")?; - ensure!( - area <= 1_048_576, - "image area must not exceed 1048576 pixels" - ); - ensure!( - width.max(height) <= width.min(height) * 200, - "image aspect ratio must not exceed 200" - ); - let expected = area.checked_mul(3).context("RGB length overflow")?; - let mut rgb = Vec::with_capacity(expected + 1); - File::open(&args[2]) - .context("opening RGB input")? - .take((expected + 1) as u64) - .read_to_end(&mut rgb) - .context("reading RGB input")?; - ensure!( - rgb.len() == expected, - "RGB input must contain exactly {expected} bytes" - ); - let image = preprocess_rgb8(width, height, &rgb)?; - println!( - "pixel_values shape: [{}, 1536]", - image.pixel_values.len() / 1536 - ); - println!("image_grid_thw: {:?}", image.image_grid_thw); - println!("resized: {}x{}", image.resized_width, image.resized_height); - println!("image_tokens: {}", image.image_tokens()); - if let Some(path) = args.get(3) { - let mut output = BufWriter::new(File::create(path).context("creating float32 output")?); - for value in image.pixel_values { - output - .write_all(&value.to_le_bytes()) - .context("writing float32 output")?; - } - output.flush().context("flushing float32 output")?; - } - Ok(()) -} diff --git a/tests/cua_s1/fixtures/image_preprocess/README.md b/tests/cua_s1/fixtures/image_preprocess/README.md deleted file mode 100644 index 1eff2a5..0000000 --- a/tests/cua_s1/fixtures/image_preprocess/README.md +++ /dev/null @@ -1,30 +0,0 @@ -# Native RGB preprocessing reference fixtures - -`preprocessor_config.json` is from pinned Qwen3.5-4B revision -`851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a`: -[upstream configuration](https://huggingface.co/Qwen/Qwen3.5-4B/resolve/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a/preprocessor_config.json). - -`manifest.json` contains complete FP32 output byte hashes generated by the actual -Transformers `AutoImageProcessor` with this configuration on CPU. It records -package versions, platform, configuration fingerprint, deterministic RGB input -hashes, resized dimensions, grid, patch shape and image token count. No weights -or generated output tensors are committed. Rust tests duplicate only the input -generator and compare every output byte through SHA-256. - -The input generator covers constant RGB, spatial/channel ramps, checkerboards -and seeded xorshift32 noise. Cases include no resize, upsampling, downsampling, -Python ties-to-even dimensions, portrait/wide inputs, the 200:1 aspect boundary -and the request pixel cap. These establish CPU preprocessing parity on the -recorded reference environment; they are not an encoder or GPU accuracy test. - -Regenerate with Python 3.12 and the pinned packages (no GPU or model downloads): - -```sh -python -m pip install torch==2.14.0 torchvision==0.29.0 transformers==5.17.0 \ - numpy==2.5.3 Pillow==11.3.0 -python tests/cua_s1/fixtures/image_preprocess/generate.py -``` - -The generator rejects mismatched package versions. Keep new generated hashes -reviewable against the recorded source; do not update expected values merely -to make a native mismatch disappear. diff --git a/tests/cua_s1/fixtures/image_preprocess/generate.py b/tests/cua_s1/fixtures/image_preprocess/generate.py deleted file mode 100644 index 70efc1b..0000000 --- a/tests/cua_s1/fixtures/image_preprocess/generate.py +++ /dev/null @@ -1,119 +0,0 @@ -"""Regenerate CPU parity hashes using the actual pinned Hugging Face processor. - -Run in a Python 3.12 environment with torch==2.14.0, torchvision==0.29.0, -transformers==5.17.0, numpy==2.5.3 and Pillow==11.3.0. No model weights or GPU. -""" - -import hashlib -import importlib.metadata -import json -import platform -from pathlib import Path - -import numpy as np -import torch -from PIL import Image -from transformers import AutoImageProcessor - -PACKAGES = { - "torch": "2.14.0", - "torchvision": "0.29.0", - "transformers": "5.17.0", - "numpy": "2.5.3", - "Pillow": "11.3.0", -} -CASES = [ - ("single_pixel", 1, 1, "constant", 1), - ("tiny_noise", 17, 19, "noise", 17), - ("aligned_noise", 256, 256, "noise", 123), - ("odd_noise", 319, 241, "noise", 42), - ("tie_down", 272, 256, "noise", 999), - ("tie_up", 304, 256, "noise", 101), - ("both_down", 271, 271, "checker", 1), - ("one_axis_down", 256, 257, "ramp", 1), - ("wide", 640, 320, "ramp", 1), - ("portrait", 319, 641, "noise", 23), - ("aspect_limit", 200, 1, "noise", 47), - ("tall_aspect_limit", 7, 1400, "checker", 1), - ("input_pixel_limit", 2048, 512, "noise", 55), - ("large_round_up", 1600, 600, "noise", 77), -] - - -def pixels(width, height, pattern, seed): - """Only input generation is duplicated in Rust; outputs come from HF.""" - if pattern == "noise": - out = bytearray(width * height * 3) - state = seed - for i in range(len(out)): - state ^= (state << 13) & 0xFFFFFFFF - state ^= state >> 17 - state ^= (state << 5) & 0xFFFFFFFF - out[i] = state & 255 - return bytes(out) - return bytes( - ( - [0, 128, 255][c] - if pattern == "constant" - else (255 if (x + y + c) % 2 else 0) - if pattern == "checker" - else (x * 13 + y * 7 + c * 83) % 256 - ) - for y in range(height) - for x in range(width) - for c in range(3) - ) - - -def evaluate(processor, name, width, height, pattern, seed): - raw = pixels(width, height, pattern, seed) - array = np.frombuffer(raw, dtype=np.uint8).reshape(height, width, 3) - result = processor(images=Image.fromarray(array), return_tensors="pt", device="cpu") - tensor = result["pixel_values"].contiguous() - grid = result["image_grid_thw"][0].tolist() - assert tensor.dtype == torch.float32 and tensor.device.type == "cpu" - return { - "name": name, - "width": width, - "height": height, - "pattern": pattern, - "seed": seed, - "input_sha256": hashlib.sha256(raw).hexdigest(), - "grid": grid, - "shape": list(tensor.shape), - "image_tokens": grid[1] * grid[2] // 4, - "resized_width": grid[2] * 16, - "resized_height": grid[1] * 16, - "output_sha256": hashlib.sha256( - tensor.numpy().astype(" f32 { - (f32::from(pixel) - 127.5) / 127.5 -} - -#[test] -fn tiny_rgb_is_upscaled_and_temporally_repeated() { - let image = preprocess_rgb8(1, 1, &[0, 127, 255]).unwrap(); - assert_eq!((image.resized_width, image.resized_height), (256, 256)); - assert_eq!(image.image_grid_thw, [1, 16, 16]); - assert_eq!(image.image_tokens(), 64); - assert_eq!(image.pixel_values.len(), 256 * 1536); - for patch in image.pixel_values.as_chunks::<1536>().0 { - for (channel, expected) in [0, 127, 255].into_iter().enumerate() { - assert!( - patch[channel * 512..(channel + 1) * 512] - .iter() - .all(|&value| value.to_bits() == normalized(expected).to_bits()) - ); - } - } -} - -#[test] -fn identity_resize_preserves_every_byte_and_patch_merge_order() { - let width = 256; - let height = 256; - let rgb: Vec = (0..height) - .flat_map(|y| (0..width).flat_map(move |x| [x as u8, y as u8, (x ^ y) as u8])) - .collect(); - let image = preprocess_rgb8(width, height, &rgb).unwrap(); - for block_y in 0..8 { - for block_x in 0..8 { - for merge_y in 0..2 { - for merge_x in 0..2 { - let patch = ((block_y * 8 + block_x) * 2 + merge_y) * 2 + merge_x; - for channel in 0..3 { - for temporal in 0..2 { - for py in 0..16 { - for px in 0..16 { - let x = block_x * 32 + merge_x * 16 + px; - let y = block_y * 32 + merge_y * 16 + py; - let index = patch * 1536 - + channel * 512 - + temporal * 256 - + py * 16 - + px; - assert_eq!( - image.pixel_values[index].to_bits(), - normalized(rgb[(y * width + x) * 3 + channel]).to_bits() - ); - } - } - } - } - } - } - } - } -} - -#[test] -fn smart_resize_uses_python_ties_even_rounding() { - for (side, expected) in [(272, 256), (304, 320)] { - let image = preprocess_rgb8(side, side, &vec![128; side * side * 3]).unwrap(); - assert_eq!( - (image.resized_width, image.resized_height), - (expected, expected) - ); - assert!( - image - .pixel_values - .iter() - .all(|&value| value == normalized(128)) - ); - } -} - -#[test] -fn rejects_invalid_geometry_before_buffer_length_validation() { - for (width, height) in [(0, 1), (1, 0), (0, 0)] { - assert_eq!( - preprocess_rgb8(width, height, &[]).unwrap_err().to_string(), - "image dimensions must be nonzero", - "{width}x{height}" - ); - } - for (width, height) in [(usize::MAX, 1), (1, usize::MAX), (usize::MAX, usize::MAX)] { - assert_eq!( - preprocess_rgb8(width, height, &[]).unwrap_err().to_string(), - "image sides must not exceed 2048", - "{width}x{height}" - ); - } - for (width, height, expected_error) in [ - (2049, 32, "image sides must not exceed 2048"), - (32, 2049, "image sides must not exceed 2048"), - (1025, 1024, "image area must not exceed 1048576 pixels"), - (201, 1, "image aspect ratio must not exceed 200"), - (1, 201, "image aspect ratio must not exceed 200"), - ] { - // A valid byte length ensures the geometry check itself rejects this - // image, rather than accidentally passing due to a truncated buffer. - let rgb = vec![0; width * height * 3]; - assert_eq!( - preprocess_rgb8(width, height, &rgb) - .unwrap_err() - .to_string(), - expected_error, - "{width}x{height}" - ); - } -} - -#[test] -fn rejects_incorrect_rgb_buffer_lengths() { - for rgb in [&[][..], &[1, 2][..], &[1, 2, 3, 4][..]] { - assert_eq!( - preprocess_rgb8(1, 1, rgb).unwrap_err().to_string(), - format!("RGB buffer length must be 3, got {}", rgb.len()) - ); - } -} - -#[test] -fn accepts_input_limits_inclusively() { - for (width, height) in [(200, 1), (1, 200), (2048, 512), (512, 2048)] { - let image = preprocess_rgb8(width, height, &vec![255; width * height * 3]).unwrap(); - assert_eq!(image.resized_width % 32, 0); - assert_eq!(image.resized_height % 32, 0); - assert!(image.pixel_values.iter().all(|&value| value == 1.0)); - } -} - -#[test] -fn matches_pinned_processor_full_output_hashes() { - use sha2::{Digest, Sha256}; - let manifest: serde_json::Value = - serde_json::from_str(include_str!("fixtures/image_preprocess/manifest.json")).unwrap(); - assert_eq!( - format!( - "{:x}", - Sha256::digest(include_bytes!( - "fixtures/image_preprocess/preprocessor_config.json" - )) - ), - manifest["config_sha256"].as_str().unwrap(), - "processor configuration must match the reference manifest" - ); - for case in manifest["cases"].as_array().unwrap() { - let name = case["name"].as_str().unwrap(); - let width = case["width"].as_u64().unwrap() as usize; - let height = case["height"].as_u64().unwrap() as usize; - let pattern = case["pattern"].as_str().unwrap(); - let mut state = case["seed"].as_u64().unwrap() as u32; - let mut rgb = Vec::with_capacity(width * height * 3); - for y in 0..height { - for x in 0..width { - for channel in 0..3 { - rgb.push(match pattern { - "noise" => { - state ^= state << 13; - state ^= state >> 17; - state ^= state << 5; - state as u8 - } - "constant" => [0, 128, 255][channel], - "checker" => { - if (x + y + channel) % 2 == 0 { - 0 - } else { - 255 - } - } - "ramp" => ((x * 13 + y * 7 + channel * 83) % 256) as u8, - _ => panic!("unknown fixture pattern {pattern}"), - }); - } - } - } - assert_eq!( - format!("{:x}", Sha256::digest(&rgb)), - case["input_sha256"].as_str().unwrap(), - "{name} input" - ); - let image = preprocess_rgb8(width, height, &rgb).unwrap(); - assert_eq!( - serde_json::json!(image.image_grid_thw), - case["grid"], - "{name} grid" - ); - assert_eq!( - serde_json::json!([image.pixel_values.len() / 1536, 1536]), - case["shape"], - "{name} shape" - ); - assert_eq!( - image.image_tokens(), - case["image_tokens"].as_u64().unwrap() as usize, - "{name} tokens" - ); - assert_eq!( - image.resized_width, - case["resized_width"].as_u64().unwrap() as usize, - "{name} width" - ); - assert_eq!( - image.resized_height, - case["resized_height"].as_u64().unwrap() as usize, - "{name} height" - ); - let mut hash = Sha256::new(); - for value in image.pixel_values { - hash.update(value.to_le_bytes()); - } - assert_eq!( - format!("{:x}", hash.finalize()), - case["output_sha256"].as_str().unwrap(), - "{name} output" - ); - } -}