diff --git a/.github/workflows/copilot-setup-steps.yml b/.github/workflows/copilot-setup-steps.yml index 2dd40ce3..2e930544 100644 --- a/.github/workflows/copilot-setup-steps.yml +++ b/.github/workflows/copilot-setup-steps.yml @@ -20,7 +20,7 @@ jobs: submodules: recursive - name: Install compiler and CMake - run: sudo apt-get install -y cmake clang + run: sudo apt-get install -y cmake gcc-14 g++-14 - name: Install torch (CPU) and the build backend run: | @@ -37,6 +37,6 @@ jobs: - name: Build and install bartorch env: - CC: clang - CXX: clang++ + CC: gcc-14 + CXX: g++-14 run: pip install -e . --no-build-isolation diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index 9c5b1197..89d987ca 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -107,9 +107,10 @@ jobs: python-version: "3.12" cache: pip - # libbartorch is a clang build: BART's nested functions become Blocks. - - name: Install the compiler, OpenMP and CMake - run: sudo apt-get install -y cmake clang libomp-dev + # GCC 14, as upstream BART builds on Linux: its heap trampolines are + # what BART's nested functions need. + - name: Install the compiler and CMake + run: sudo apt-get install -y cmake gcc-14 g++-14 - name: Install torch (CPU) and the build backend run: | @@ -119,8 +120,8 @@ jobs: - name: Build and install bartorch env: - CC: clang - CXX: clang++ + CC: gcc-14 + CXX: g++-14 run: pip install -e . --no-build-isolation - name: Install the documentation and example requirements diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index 8297ef88..cf59360f 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -182,19 +182,25 @@ jobs: -w wheelhouse/ --plat manylinux_2_28_x86_64 \ --exclude libcudart.so.12 --exclude libcufft.so.11 \ --exclude libcublas.so.12 --exclude libcublasLt.so.12 - - name: Check that it loads without a card + # The wheel carries no interpreter, so it is loaded from each Python the + # package supports, as the CPU wheels are tested from each. + - name: Check that it loads without a card, on every Python run: | - /opt/python/cp312-cp312/bin/python -m pip install wheelhouse/*.whl \ - torch --index-url https://download.pytorch.org/whl/cpu \ - --extra-index-url https://pypi.org/simple - /opt/python/cp312-cp312/bin/python -c " + for py in cp310-cp310 cp311-cp311 cp312-cp312 cp313-cp313 cp314-cp314; do + python=/opt/python/$py/bin/python + $python -m pip install wheelhouse/*.whl \ + torch --index-url https://download.pytorch.org/whl/cpu \ + --extra-index-url https://pypi.org/simple + $python -c " + import sys import bartorch from bartorch import _cuda info = bartorch.build_info() - print(info) + print(sys.version.split()[0], info) assert 'cuda=ON' in info, info print('devices:', _cuda.device_count()) " + done - uses: actions/upload-artifact@v4 with: name: cuda-wheel diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index bbef760a..337f4ad8 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -10,7 +10,10 @@ env: FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true" jobs: - # libbartorch is a clang build: BART's nested functions become Blocks. + # The toolchains are upstream BART's: GCC 14 on Linux, whose heap + # trampolines carry BART's nested functions, and Apple's clang on macOS, + # where they become Blocks. Each platform runs the oldest and the newest + # Python the package supports; the wheel workflow covers every version. tests: strategy: fail-fast: false @@ -18,19 +21,24 @@ jobs: include: - os: ubuntu-latest python-version: "3.10" - cc: clang + cc: gcc-14 + cxx: g++-14 + # MKL as the BLAS, LAPACK and FFT source, in one of the two jobs. + extras: "[mkl]" - os: ubuntu-latest - python-version: "3.12" - cc: clang + python-version: "3.14" + cc: gcc-14 + cxx: g++-14 # One job measures coverage for Codecov; the rest only test. coverage: true - - os: ubuntu-latest - python-version: "3.12" - cc: gcc-14 - extras: "[mkl]" - os: macos-latest - python-version: "3.12" + python-version: "3.10" + cc: clang + cxx: clang++ + - os: macos-latest + python-version: "3.14" cc: clang + cxx: clang++ runs-on: ${{ matrix.os }} # Codecov takes the upload on GitHub's OIDC token, so no secret is needed. permissions: @@ -49,9 +57,9 @@ jobs: python-version: ${{ matrix.python-version }} cache: pip - - name: Install compiler, OpenMP and CMake (Linux) + - name: Install compiler and CMake (Linux) if: runner.os == 'Linux' - run: sudo apt-get install -y cmake ${{ matrix.cc }} libomp-dev + run: sudo apt-get install -y cmake ${{ matrix.cc }} ${{ matrix.cxx }} - name: Install CMake (macOS) if: runner.os == 'macOS' @@ -79,7 +87,7 @@ jobs: - name: Build and install bartorch env: CC: ${{ matrix.cc }} - CXX: ${{ matrix.cc == 'gcc-14' && 'g++-14' || 'clang++' }} + CXX: ${{ matrix.cxx }} run: pip install -e ".${{ matrix.extras }}" --no-build-isolation -v # Both wheels carry their own copy of LLVM's OpenMP runtime, and the @@ -154,8 +162,8 @@ jobs: python-version: "3.12" cache: pip - - name: Install the CUDA toolkit, clang and OpenMP - run: sudo apt-get update -qq && sudo apt-get install -y --no-install-recommends cmake clang libomp-dev nvidia-cuda-toolkit + - name: Install the CUDA toolkit and GCC + run: sudo apt-get update -qq && sudo apt-get install -y --no-install-recommends cmake gcc-14 g++-14 g++-12 nvidia-cuda-toolkit - name: Install torch (CPU) and the build backend run: | @@ -170,10 +178,13 @@ jobs: - name: Install torchvision from the same index as torch run: pip install torchvision --index-url https://download.pytorch.org/whl/cpu + # BART is GCC 14's, for its heap trampolines; the host side of the + # kernels is GCC 12's, the newest this distribution's nvcc accepts. - name: Build with BART's CUDA kernels env: - CC: clang - CXX: clang++ + CC: gcc-14 + CXX: g++-14 + CUDAHOSTCXX: g++-12 run: pip install -e . --no-build-isolation --config-settings=cmake.define.BARTORCH_CUDA=ON - name: The CUDA build must report its device code and run on the host diff --git a/AGENTS.md b/AGENTS.md index 4c3e4a3e..beaa3985 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -614,12 +614,16 @@ the stack needs an executable stack, which glibc 2.41 refuses to `dlopen`; so GCC 14's `-ftrampoline-impl=heap` is required and older GCC is rejected at configure time. BART's own `NOEXEC_STACK` workaround does not help here: it parses a trampoline layout GCC emits only for non-PIC executables, not for a -shared library. Both compilers are built and tested in CI. +shared library. CI builds with upstream BART's toolchains: GCC 14 on Linux +and Apple's clang on macOS. clang on Linux, with the vendored Blocks runtime, +still builds and is not tested in CI. -Windows is not a platform here. BART does not build on it, and nothing in -this repository carries a path toward one: no `.dll` among the names the -loader tries, no `__declspec(dllexport)`, no `win32` branch picking a -different library. WSL2 is a Linux install and is the answer. +Windows is not a platform here. BART builds there with MinGW, but Windows is +LLP64: `long` is 32 bits, and BART holds every dimension, size and byte +stride in a `long`, so no array could exceed 2 GiB. No macro fixes that +without editing BART: `#define long long long` breaks every `long long` BART +writes, a typedef breaks every `unsigned long`, and its `sscanf("%lu")` calls +would still write four bytes. WSL2 is a Linux install and is the answer. The compiler's own runtime is linked statically on Linux, because otherwise the toolchain's floor becomes the target system's: a GCC 14 build asks @@ -1109,8 +1113,8 @@ of it is written. What comes back is finite stack memory. `_call`'s `_PRECONDITIONS` refuses those combinations before the command runs, which is where any further "BART does not define this" case belongs. -Windows is not on this list because it is not a target: BART does not build -there, and WSL2 is a Linux install like any other. +Windows is not on this list because it is not a target: BART's 32-bit `long` +there caps every array at 2 GiB, and WSL2 is a Linux install like any other. A tool that takes device memory as it stands. BART guards the host reads that would break -- `estimate_im_dims` copies to the host when it is handed one -- diff --git a/docs/guides/developer/prerequisites.md b/docs/guides/developer/prerequisites.md index 8980dd2f..243b8efa 100644 --- a/docs/guides/developer/prerequisites.md +++ b/docs/guides/developer/prerequisites.md @@ -15,5 +15,6 @@ suite is run from `src/` without installing, install it separately skip, because the substitution declining is an error. `pip install mkl deepinv` enables the tests that need MKL and the DeepInverse adapter. -Windows is not a target: BART does not build on it, and WSL2 is used as a Linux +Windows is not a target: BART stores array sizes and strides in `long`, which +is 32 bits there and limits every array to 2 GiB. WSL2 is used as a Linux environment. diff --git a/docs/guides/developer/pull-requests.md b/docs/guides/developer/pull-requests.md index 40edf072..5d00ed05 100644 --- a/docs/guides/developer/pull-requests.md +++ b/docs/guides/developer/pull-requests.md @@ -11,7 +11,8 @@ template: | Validation | The commands run and their results; for a numerical change, the reference and the tolerance; hardware or optional dependencies that were not available | | Public API and documentation | Changes to public behaviour and the documentation updated for them | -CI runs the test matrix (Linux with clang and GCC 14, macOS, and a CUDA build +CI runs the test matrix (Linux with GCC 14 and macOS with clang, each on the +oldest and newest supported Python, and a CUDA build without a device), the lint and spelling checks, and two documentation builds: the reference without the compiled library, and the executed gallery, which is published from `main` and from release tags. A pull request is merged when CI passes and a diff --git a/docs/guides/user/prerequisites.md b/docs/guides/user/prerequisites.md index c86ef41d..80cc6f80 100644 --- a/docs/guides/user/prerequisites.md +++ b/docs/guides/user/prerequisites.md @@ -18,7 +18,7 @@ | macOS 11 or later on Apple silicon | Wheel | CPU only; see {ref}`macos-openmp` | | Linux aarch64 | Source distribution | BART is compiled on installation; FINUFFT publishes no wheel for this platform and is built from source as well | | macOS on Intel | Source distribution | BART is compiled on installation; FINUFFT releases after 2.4.0 have no wheel for this platform and are built from source as well | -| Windows | Not supported | BART does not build on Windows; WSL2 provides a Linux environment | +| Windows | Not supported | BART stores array sizes and strides in `long`, which is 32 bits on Windows, limiting every array to 2 GiB; WSL2 provides a Linux environment | A source installation needs the toolchain listed under {ref}`source-builds`. Apple MPS devices are not supported; the device paths are CPU and CUDA. diff --git a/pyproject.toml b/pyproject.toml index 62c1eef3..b8860b3f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -80,7 +80,7 @@ deepinv = [ # MKL covers every BLAS and LAPACK routine BART calls and is about twice as # fast as the alternatives on the work that leans on LAPACK. There is no MKL # wheel for macOS, where SciPy's OpenBLAS and Accelerate serve instead, and -# Windows is not a target: BART does not build there. +# Windows is not a target: BART's `long` is 32 bits there. mkl = [ "mkl>=2024; sys_platform == 'linux' and platform_machine == 'x86_64'", ]