From dddfb9e83d62fe602847a30ccb2f97032d3072de Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 07:47:54 +0000 Subject: [PATCH 01/12] executorch: add riscv64 build (cp312/cp313/cp314) Mirrors upstream's own default pybind wheel build (setup.py + CMake --preset pybind): XNNPACK CPU delegate, portable/optimized/quantized kernels, CoreML's portable util+inmemoryfs+pybind pieces (no Apple delegate on Linux), OpenVINO backend (dlopen-based, no build-time SDK). QNN, CUDA and Vulkan stay off, same as upstream's own CI on riscv64/ARM hosts (QNN auto-skips under GITHUB_ACTIONS; CUDA/Vulkan are opt-in and neither is requested). torch, pytorch-tokenizers and coremltools (all hard runtime deps) are already on pypi.riseproject.dev for riscv64 cp312/cp313/cp314; torchao>=0.18.0 resolves to its py3-none-any wheel there (no riscv64 tag exists on PyPI, but its own x86_64 wheel doesn't match this platform anyway, so pip falls back to the pure-Python one). cp310/cp311 are dropped: torch and pytorch-tokenizers ship neither for riscv64. The smoke test replaces upstream's torchvision-based MobileNetV3 example (torchvision has no riscv64 wheel) with a small hand-built model, keeping upstream's own registered-backend assertions and exercising both the portable (non-delegated) and XNNPACK-delegated export+execute paths for real. A small packaging patch widens license-files so the wheel also carries the licences of the third-party libraries statically linked into its extensions (XNNPACK, cpuinfo, pthreadpool, FP16, FXdiv, flatbuffers, flatcc, Eigen, pocketfft, nlohmann/json, pybind11) -- upstream's own wheel ships only its own LICENSE. --- .github/workflows/build-executorch.yml | 210 ++++++++++++++++++ docs/packages/executorch.yaml | 5 + ...de-vendored-third-party-licences-in-.patch | 36 +++ 3 files changed, 251 insertions(+) create mode 100644 .github/workflows/build-executorch.yml create mode 100644 docs/packages/executorch.yaml create mode 100644 patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml new file mode 100644 index 00000000000..264b4fb61b8 --- /dev/null +++ b/.github/workflows/build-executorch.yml @@ -0,0 +1,210 @@ +# SPDX-FileCopyrightText: 2026 The RISE Project +# SPDX-License-Identifier: MIT +--- +# This workflow is based on: +# https://github.com/pytorch/executorch/blob/v1.4.1/.github/workflows/build-wheels-linux.yml +name: Build executorch wheels (riscv64) + +on: + workflow_dispatch: + inputs: + version: + description: 'Version glob to (re)build; empty builds every version of docs/packages/executorch.yaml not released yet' + required: false + default: '' + pull_request: + branches: [main] + paths: + - '.github/workflows/build-executorch.yml' + - 'docs/packages/executorch.yaml' + - 'patches/executorch/**' + push: + branches: [main] + paths: + - '.github/workflows/build-executorch.yml' + - 'docs/packages/executorch.yaml' + - 'patches/executorch/**' + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + +permissions: + contents: read + +env: + MANYLINUX_RISCV64_IMAGE: quay.io/pypa/manylinux_2_39_riscv64 + +jobs: + setup: + uses: $/.github/workflows/_setup.yml + with: + package: executorch + version: ${{ inputs.version }} + + build_wheels: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + # cp310/cp311 are dropped: torch and pytorch-tokenizers (both hard runtime + # deps) have no riscv64 wheel for either on pypi.riseproject.dev. cp314t + # is dropped too: upstream itself ships no free-threaded wheel. + python: ["cp312", "cp313", "cp314"] + name: Build executorch ${{ matrix.version }} ${{ matrix.python }}-manylinux_riscv64 + runs-on: ubuntu-24.04-riscv + timeout-minutes: 720 + + env: + EXECUTORCH_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout executorch v${{ env.EXECUTORCH_VERSION }} + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: pytorch/executorch + ref: v${{ env.EXECUTORCH_VERSION }} + submodules: recursive + persist-credentials: false + + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + path: python-wheels + persist-credentials: false + + - name: Apply patches + run: git apply python-wheels/patches/executorch/${{ env.EXECUTORCH_VERSION }}/*.patch + + # Every library below is statically linked into the compiled extensions, and + # setuptools' license-files (widened by the patch above) only reaches the + # project root. + - name: Stage the vendored licences + run: | + set -euo pipefail + cp backends/xnnpack/third-party/FP16/LICENSE LICENSE.FP16 + cp backends/xnnpack/third-party/FXdiv/LICENSE LICENSE.FXdiv + cp backends/xnnpack/third-party/XNNPACK/LICENSE LICENSE.XNNPACK + cp backends/xnnpack/third-party/cpuinfo/LICENSE LICENSE.cpuinfo + cp backends/xnnpack/third-party/pthreadpool/LICENSE LICENSE.pthreadpool + cp third-party/flatbuffers/LICENSE LICENSE.flatbuffers + cp third-party/flatcc/LICENSE LICENSE.flatcc + cp kernels/optimized/third-party/eigen/LICENSE LICENSE.eigen + cp third-party/pocketfft/LICENSE.md LICENSE.pocketfft + cp third-party/json/LICENSE.MIT LICENSE.json + cp third-party/pybind11/LICENSE LICENSE.pybind11 + + # Upstream's own smoke test (.ci/scripts/wheel/test_linux.py) exercises the + # examples/ tree, which needs torchvision for its MobileNetV3 model; + # torchvision has no riscv64 wheel. This test keeps upstream's own + # assertions (registered backend names) and replaces the model with a small + # hand-built one, run through both the portable (non-delegated) path and a + # real XNNPACK-delegated export, so the compiled extension, the portable + # kernels and the XNNPACK backend are all genuinely exercised end to end. + - name: Write the riscv64 smoke test + run: | + cat > riscv64_smoke_test.py <<'EOF' + import platform + + import torch + from torch.export import export + + from executorch.backends.xnnpack.partition.xnnpack_partitioner import ( + XnnpackPartitioner, + ) + from executorch.exir import to_edge, to_edge_transform_and_lower + from executorch.extension.pybindings.portable_lib import ( + _get_registered_backend_names, + _load_for_executorch_from_buffer, + ) + + registered = _get_registered_backend_names() + assert "XnnpackBackend" in registered, registered + assert "OpenvinoBackend" in registered, registered + if platform.machine() in ("x86_64", "amd64"): + assert "QnnBackend" in registered, registered + + + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + self.linear = torch.nn.Linear(4, 4) + + def forward(self, x): + return self.linear(x).relu() + + + model = Model().eval() + example_inputs = (torch.randn(2, 4),) + expected = model(*example_inputs) + + portable_prog = to_edge(export(model, example_inputs, strict=True)).to_executorch() + portable_module = _load_for_executorch_from_buffer(portable_prog.buffer) + portable_out = portable_module.forward(example_inputs)[0] + assert torch.allclose(portable_out, expected, atol=1e-4), (portable_out, expected) + print("portable (non-delegated) execution OK") + + xnnpack_prog = to_edge_transform_and_lower( + export(model, example_inputs, strict=True), + partitioner=[XnnpackPartitioner()], + ).to_executorch() + xnnpack_module = _load_for_executorch_from_buffer(xnnpack_prog.buffer) + xnnpack_out = xnnpack_module.forward(example_inputs)[0] + assert torch.allclose(xnnpack_out, expected, atol=1e-4), (xnnpack_out, expected) + print("XNNPACK-delegated execution OK") + EOF + + - uses: pypa/cibuildwheel@1828c10ab37f080699c7b81cea34097c684a7074 # v4.2.0 + with: + only: ${{ matrix.python }}-manylinux_riscv64 + env: + CIBW_MANYLINUX_RISCV64_IMAGE: ${{ env.MANYLINUX_RISCV64_IMAGE }} + CIBW_ENVIRONMENT: >- + CMAKE_BUILD_PARALLEL_LEVEL=8 + PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + CIBW_TEST_SOURCES: riscv64_smoke_test.py + CIBW_TEST_COMMAND: python riscv64_smoke_test.py + + - name: Check the extensions and the vendored licences made it into the wheel + run: | + python3 - wheelhouse/*.whl <<'EOF' + import sys, zipfile + expected = {"LICENSE", "LICENSE.FP16", "LICENSE.FXdiv", "LICENSE.XNNPACK", + "LICENSE.cpuinfo", "LICENSE.pthreadpool", "LICENSE.flatbuffers", + "LICENSE.flatcc", "LICENSE.eigen", "LICENSE.pocketfft", + "LICENSE.json", "LICENSE.pybind11"} + for whl in sys.argv[1:]: + names = zipfile.ZipFile(whl).namelist() + assert [n for n in names + if n.startswith("executorch/extension/pybindings/_portable_lib.") + and n.endswith(".so")], whl + licenses = {n.split(".dist-info/licenses/", 1)[1] for n in names + if ".dist-info/licenses/" in n and not n.endswith("/")} + assert licenses == expected, f"{whl}: {sorted(licenses)}" + print(whl, "ok") + EOF + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: executorch-${{ env.EXECUTORCH_VERSION }}-${{ matrix.python }}-manylinux_riscv64 + path: wheelhouse/*.whl + if-no-files-found: error + + publish: + name: Publish executorch ${{ matrix.version }} + needs: [setup, build_wheels] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + permissions: + contents: write + pull-requests: write + uses: $/.github/workflows/_publish-wheel.yml + secrets: + app-private-key: ${{ secrets.RISEPROJECT_APP_PRIVATE_KEY }} + with: + artifact-pattern: executorch-${{ matrix.version }}-*-manylinux_riscv64 diff --git a/docs/packages/executorch.yaml b/docs/packages/executorch.yaml new file mode 100644 index 00000000000..ae60b1cf3ff --- /dev/null +++ b/docs/packages/executorch.yaml @@ -0,0 +1,5 @@ +package-name: executorch +source-code: https://github.com/pytorch/executorch +license: BSD-3-Clause AND MIT AND BSD-2-Clause AND Apache-2.0 AND MPL-2.0 +versions: +- version: 1.4.1 diff --git a/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch new file mode 100644 index 00000000000..7fa69c2403b --- /dev/null +++ b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch @@ -0,0 +1,36 @@ +From 0000000000000000000000000000000000000001 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Thu, 10 Jul 2026 00:00:00 +0000 +Subject: [PATCH] packaging: Include vendored third-party licences in the + wheel + +The compiled extensions in this wheel statically link FP16, FXdiv, +XNNPACK, cpuinfo, pthreadpool, flatbuffers, flatcc, Eigen, pocketfft, +nlohmann/json and pybind11, but `license-files` only reaches the +project's own LICENSE. RISE distributes these wheels and must also +carry the licence of every statically linked dependency, so widen the +glob to pick up the `LICENSE.` files the riscv64 build stages at +the project root before `pip wheel` runs. + +Upstream-Status: Inappropriate [only relevant to a distributor that +ships vendored static licences alongside the wheel; upstream's own +PyPI wheel carries only the project's own LICENSE] + +Signed-off-by: RISE Project CI +--- + pyproject.toml | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/pyproject.toml b/pyproject.toml +index 0000000000..0000000000 100644 +--- a/pyproject.toml ++++ b/pyproject.toml +@@ -100,7 +100,7 @@ + # package, and package_data to restrict the set up non-python files we + # include. See also setuptools/discovery.py for custom finders. + [tool.setuptools] +-license-files = ["LICENSE"] ++license-files = ["LICENSE", "LICENSE.*"] + + [tool.setuptools.package-dir] + # Tell setuptools to follow the symlink: src/executorch/* -> * for all first level From 161c3ab77cc6cbe100d9780149e4eb8750e1e7d2 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 09:03:22 +0000 Subject: [PATCH 02/12] executorch: fix check_patches and cmake-from-source CI failures - Fold the packaging patch's Upstream-Status: Inappropriate reason onto one line: check_patch.py's regex has no re.DOTALL, so it only reads the tag's first physical line, and a bracket closed two lines down never reaches the format check. - Add CIBW_BEFORE_ALL_LINUX: dnf -y install openssl-devel. No riscv64 cmake wheel is published yet for the pinned version, so pip builds it from source; its own bootstrap's Utilities/cmcurl needs OpenSSL's dev headers, which the manylinux_2_39_riscv64 image doesn't ship by default. --- .github/workflows/build-executorch.yml | 4 ++++ ...-packaging-Include-vendored-third-party-licences-in-.patch | 4 +--- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index 264b4fb61b8..6aa95bad958 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -161,6 +161,10 @@ jobs: only: ${{ matrix.python }}-manylinux_riscv64 env: CIBW_MANYLINUX_RISCV64_IMAGE: ${{ env.MANYLINUX_RISCV64_IMAGE }} + # No riscv64 cmake wheel exists, so pip builds it from source as a + # build dependency; its own CMake bootstrap needs OpenSSL's dev + # headers to find libcrypto. + CIBW_BEFORE_ALL_LINUX: dnf -y install openssl-devel CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ diff --git a/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch index 7fa69c2403b..7e65c08e10e 100644 --- a/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch +++ b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch @@ -12,9 +12,7 @@ carry the licence of every statically linked dependency, so widen the glob to pick up the `LICENSE.` files the riscv64 build stages at the project root before `pip wheel` runs. -Upstream-Status: Inappropriate [only relevant to a distributor that -ships vendored static licences alongside the wheel; upstream's own -PyPI wheel carries only the project's own LICENSE] +Upstream-Status: Inappropriate [only relevant to a distributor bundling vendored third-party licences; reproduces identically on any architecture] Signed-off-by: RISE Project CI --- From 53206a23d4dc3141a6c16dbbc6573f46c68757ee Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 12:59:52 +0000 Subject: [PATCH 03/12] executorch: patch CMake source-dir-name check for cibuildwheel cibuildwheel always mounts the source into its manylinux container at the fixed path /project, which trips CMakeLists.txt's hard requirement that the checkout be named exactly `executorch` (upstream tracks lifting this at pytorch/executorch#6475). Add a second patch that provides the same `` include path through a configure-time symlink instead, so the build no longer depends on the checkout directory's name. Validated by applying the patch to a real v1.4.1 checkout and by exercising the isolated CMake snippet under both the original and patched forms: unpatched fails with the reported FATAL_ERROR when the directory isn't named `executorch`, and patched configures cleanly with the symlink resolving to the real source tree. --- ...rce-directory-name-check-for-cibuild.patch | 65 +++++++++++++++++++ 1 file changed, 65 insertions(+) create mode 100644 patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch diff --git a/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch b/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch new file mode 100644 index 00000000000..41e36734ac7 --- /dev/null +++ b/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch @@ -0,0 +1,65 @@ +From 0000000000000000000000000000000000000002 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Thu, 10 Jul 2026 00:00:00 +0000 +Subject: [PATCH] cmake: stop requiring the checkout dir be named `executorch` + +CMakeLists.txt hard-fails configure unless the checkout's directory is +named exactly `executorch`, because `_common_include_directories` adds +the checkout's *parent* directory so in-tree code can say +`#include ` (upstream tracks lifting this +restriction at https://github.com/pytorch/executorch/issues/6475). + +cibuildwheel always mounts/copies the source into its manylinux +container at the fixed path `/project`, so this check always trips +there, regardless of what directory this repo is checked out to on the +runner. Provide the same `` include path through a +configure-time symlink instead of requiring a literally named parent +directory, so the build works unmodified under cibuildwheel. + +Upstream-Status: Inappropriate [only relevant to cibuildwheel, which always mounts the source at a fixed /project path inside its build container; see pytorch/executorch#6475] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 20 +++++++++----------- + 1 file changed, 9 insertions(+), 11 deletions(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index be0cec2..4668d0a 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -447,22 +447,20 @@ if(NOT DEFINED CMAKE_POSITION_INDEPENDENT_CODE + list(APPEND _common_compile_options $<$>:-fPIC>) + endif() + +-# Let files say "include ". +-# TODO(#6475): This requires/assumes that the repo lives in a directory named +-# exactly `executorch`. Check the assumption first. Remove this check once we +-# stop relying on the assumption. +-cmake_path(GET CMAKE_CURRENT_SOURCE_DIR FILENAME _repo_dir_name) +-if(NOT "${_repo_dir_name}" STREQUAL "executorch") +- message( +- FATAL_ERROR +- "The ExecuTorch repo must be cloned into a directory named exactly " +- "`executorch`; found `${_repo_dir_name}`. See " +- "https://github.com/pytorch/executorch/issues/6475 for progress on a " +- "fix for this restriction." ++# Let files say "include ", regardless of what ++# the checkout directory is actually named (see ++# https://github.com/pytorch/executorch/issues/6475). Provide that include ++# path unconditionally through a configure-time symlink, rather than ++# requiring the checkout itself be named `executorch`. ++set(_executorch_include_shim "${CMAKE_BINARY_DIR}/_executorch_include") ++file(MAKE_DIRECTORY "${_executorch_include_shim}") ++if(NOT EXISTS "${_executorch_include_shim}/executorch") ++ file(CREATE_LINK "${CMAKE_CURRENT_SOURCE_DIR}" ++ "${_executorch_include_shim}/executorch" SYMBOLIC + ) + endif() + set(_common_include_directories +- $ ++ $ + $ + $ + $ +-- +2.34.1 From 0d4e71f16f244a668fbfaa3a39ce4350925e50d3 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 16:45:22 +0000 Subject: [PATCH 04/12] executorch: install torch at build time, disable PEP 517 isolation CMake's find_package_torch_headers/get_torch_base_path resolves torch by running `import torch` in the same interpreter running the build backend. torch is a runtime dependency (setup.py), not declared in [build-system].requires, so pip's isolated build venv never has it and CMake configure fails with AttributeError: 'NoneType' object has no attribute 'submodule_search_locations'. Preinstall build-system.requires plus torch in CIBW_BEFORE_BUILD and disable isolation via CIBW_BUILD_FRONTEND, mirroring the same pattern already used in build-torchaudio.yml and build-torchcodec.yml. --- .github/workflows/build-executorch.yml | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index 6aa95bad958..0cc69bb20ff 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -165,6 +165,18 @@ jobs: # build dependency; its own CMake bootstrap needs OpenSSL's dev # headers to find libcrypto. CIBW_BEFORE_ALL_LINUX: dnf -y install openssl-devel + # get_torch_base_path (tools/cmake/Utils.cmake) runs `import torch` in + # the same interpreter as the build backend to add it to + # CMAKE_PREFIX_PATH, but torch is only a runtime dependency + # (setup.py), not a build-system.requires entry, so PEP 517 isolation + # hides it from that interpreter. Preinstall build-system.requires + # plus torch and turn isolation off (mirrors build-torchaudio.yml / + # build-torchcodec.yml). + CIBW_BUILD_FRONTEND: "pip; args: --no-build-isolation" + CIBW_BEFORE_BUILD: >- + pip install "cmake>=3.24,<4.0.0" "packaging>=24.2" pyyaml + "setuptools>=77.0.3" wheel zstd certifi && + pip install --only-binary=:all: "torch>=2.13.0a0" CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ From 993ab938dd06fae2b857949f5253bd2140d6a88f Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 19:00:30 +0000 Subject: [PATCH 05/12] executorch: build pybind extensions as MODULE, not SHARED The riscv64 build now fails during CMake configure, not link: CMake Error at .../FindPython/Support.cmake:4243 (message): Python_ADD_LIBRARY: dependent target 'Python::Python' is not defined. Did you miss to request COMPONENT 'Development.Embed'? CMake Error at third-party/pybind11/tools/pybind11NewTools.cmake:276 (target_link_libraries): Cannot specify link libraries for target "executorchcoreml" which is not built by this project. Every pybind11_add_module( SHARED ...) call in upstream's CMakeLists (portable_lib, data_loader, selective_build, _training_lib, _llm_runner, executorchcoreml) makes CMake's Python_add_library() require the Python::Python target (COMPONENT Development.Embed) to exist at configure time. quay.io/pypa/manylinux_2_39_riscv64, like every vanilla pypa manylinux image, ships only a static libpythonX.Y.a and never satisfies Development.Embed, so find_package(Python) never defines Python::Python and configure aborts on the first such call it reaches (coreml's, by subdirectory order) -- the second CMake error above is just fallout from the first one leaving that target half-defined. These are all ordinary dlopen()'d Python extension modules -- upstream's own strip_python_lib() helper immediately strips the resulting Python::Python/pybind11::embed link dependency right back out after creating them, precisely because they must resolve Python symbols from the running interpreter rather than linking a build-time libpython. Dropping SHARED (pybind11's own default, MODULE) only requires Development.Module, which manylinux images do provide, and produces the same importable extension without ever needing Development.Embed. Patch under patches/executorch/1.4.1/ per the porting skill, marked "To upstream" since this is a real portability bug on any manylinux-like environment lacking an embeddable libpython, not riscv64-specific. --- ...bind-extensions-as-MODULE-not-SHARED.patch | 118 ++++++++++++++++++ 1 file changed, 118 insertions(+) create mode 100644 patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch diff --git a/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch new file mode 100644 index 00000000000..74a47453b26 --- /dev/null +++ b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch @@ -0,0 +1,118 @@ +From 0000000000000000000000000000000000000003 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: build pybind extensions as MODULE, not SHARED + +pybind11_add_module( SHARED ...) makes CMake's Python_add_library() +require the Python::Python target (COMPONENT Development.Embed) to exist +at configure time, even though strip_python_lib() immediately removes the +resulting Python::Python/pybind11::embed link dependency again -- these +are ordinary dlopen()'d extension modules that resolve Python symbols +from the running interpreter, never anything that embeds Python. + +Vanilla manylinux images (unlike the custom manylinux-builder images +PyTorch's own CI uses) ship only a static libpythonX.Y.a and never +satisfy Development.Embed, so find_package(Python) never defines +Python::Python and every pybind11_add_module(... SHARED ...) call fails +configure outright: + + CMake Error: Python_ADD_LIBRARY: dependent target 'Python::Python' is + not defined. Did you miss to request COMPONENT 'Development.Embed'? + +Dropping SHARED (the pybind11 default, MODULE) only requires +Development.Module, which these images do provide, and produces the +exact same importable extension: strip_python_lib()'s job of avoiding a +runtime libpython link was already redundant with MODULE all along. + +Upstream-Status: To upstream [not yet submitted; a real portability bug on +any vanilla manylinux image lacking an embeddable libpython +(Development.Embed), independent of riscv64 -- pytorch/executorch's own +Linux CI apparently runs under a custom manylinux-builder image that does +provide one, so upstream has likely never hit this] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 4 ++-- + backends/apple/coreml/CMakeLists.txt | 2 +- + codegen/tools/CMakeLists.txt | 2 +- + extension/llm/runner/CMakeLists.txt | 2 +- + extension/training/CMakeLists.txt | 2 +- + 5 files changed, 6 insertions(+), 6 deletions(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index be0cec2..a02db93 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -1157,7 +1157,7 @@ if(EXECUTORCH_BUILD_PYBIND) + target_link_libraries(util PRIVATE torch c10 executorch extension_tensor) + + # pybind portable_lib +- pybind11_add_module(portable_lib SHARED extension/pybindings/pybindings.cpp) ++ pybind11_add_module(portable_lib extension/pybindings/pybindings.cpp) + strip_python_lib(portable_lib) + # The actual output file needs a leading underscore so it can coexist with + # portable_lib.py in the same python package. PyTorch requires C++20, so +@@ -1190,7 +1190,7 @@ if(EXECUTORCH_BUILD_PYBIND) + # pybind data_loader module - provides PyDataLoader type for external + # pybinding extensions to create custom data loaders + pybind11_add_module( +- data_loader SHARED extension/pybindings/pybindings_data_loader.cpp ++ data_loader extension/pybindings/pybindings_data_loader.cpp + ) + strip_python_lib(data_loader) + target_include_directories(data_loader PRIVATE ${_common_include_directories}) +diff --git a/backends/apple/coreml/CMakeLists.txt b/backends/apple/coreml/CMakeLists.txt +index 5391a35..03612c5 100644 +--- a/backends/apple/coreml/CMakeLists.txt ++++ b/backends/apple/coreml/CMakeLists.txt +@@ -172,7 +172,7 @@ install( + + if(EXECUTORCH_BUILD_PYBIND) + pybind11_add_module( +- executorchcoreml SHARED runtime/inmemoryfs/inmemory_filesystem_py.cpp ++ executorchcoreml runtime/inmemoryfs/inmemory_filesystem_py.cpp + runtime/inmemoryfs/inmemory_filesystem_utils.cpp + ) + strip_python_lib(executorchcoreml) +diff --git a/codegen/tools/CMakeLists.txt b/codegen/tools/CMakeLists.txt +index b829e83..efdfdf3 100644 +--- a/codegen/tools/CMakeLists.txt ++++ b/codegen/tools/CMakeLists.txt +@@ -8,7 +8,7 @@ + # Check if pybind11 is available + + # Create the selective_build pybind11 module +-pybind11_add_module(selective_build SHARED selective_build.cpp) ++pybind11_add_module(selective_build selective_build.cpp) + strip_python_lib(selective_build) + + # Set the output name to match the module name +diff --git a/extension/llm/runner/CMakeLists.txt b/extension/llm/runner/CMakeLists.txt +index 3e9f811..8a5e808 100644 +--- a/extension/llm/runner/CMakeLists.txt ++++ b/extension/llm/runner/CMakeLists.txt +@@ -110,7 +110,7 @@ endif() + if(EXECUTORCH_BUILD_PYBIND) + # Create the Python extension module for LLM runners + pybind11_add_module( +- _llm_runner SHARED ${CMAKE_CURRENT_SOURCE_DIR}/pybindings.cpp ++ _llm_runner ${CMAKE_CURRENT_SOURCE_DIR}/pybindings.cpp + ) + strip_python_lib(_llm_runner) + +diff --git a/extension/training/CMakeLists.txt b/extension/training/CMakeLists.txt +index aaaeb8d..3591af0 100644 +--- a/extension/training/CMakeLists.txt ++++ b/extension/training/CMakeLists.txt +@@ -63,7 +63,7 @@ if(EXECUTORCH_BUILD_PYBIND) + endif() + + pybind11_add_module( +- _training_lib SHARED ++ _training_lib + ${CMAKE_CURRENT_SOURCE_DIR}/pybindings/_training_lib.cpp + ) + strip_python_lib(_training_lib) +-- +2.34.1 + From 08ab1ec1bd4e0a0c5c39b1d5b337230f82a3a94a Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 19:07:55 +0000 Subject: [PATCH 06/12] executorch: reflow 0003 patch's Upstream-Status comment onto one line The bracketed comment on the To upstream line spanned 5 lines, so check_patch.py's single-line regex only captured the first line's text (an unclosed '[') and rejected the format. Reflow it onto a single line per this repo's Upstream-Status convention, keeping the same substance: not yet submitted, a real portability bug on any vanilla manylinux image lacking Development.Embed, independent of riscv64, and likely unseen upstream because pytorch/executorch's own CI runs a custom manylinux-builder image that does provide an embeddable libpython. --- ...cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch index 74a47453b26..88b12e65428 100644 --- a/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch +++ b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch @@ -24,11 +24,7 @@ Development.Module, which these images do provide, and produces the exact same importable extension: strip_python_lib()'s job of avoiding a runtime libpython link was already redundant with MODULE all along. -Upstream-Status: To upstream [not yet submitted; a real portability bug on -any vanilla manylinux image lacking an embeddable libpython -(Development.Embed), independent of riscv64 -- pytorch/executorch's own -Linux CI apparently runs under a custom manylinux-builder image that does -provide one, so upstream has likely never hit this] +Upstream-Status: To upstream [not yet submitted; a real portability bug on any vanilla manylinux image lacking an embeddable libpython (Development.Embed), independent of riscv64 -- pytorch/executorch's own CI runs a custom manylinux-builder image that does provide one, so upstream has likely never hit this] Signed-off-by: RISE Project CI --- From f30db86eb2fb3b0bd2d32043a510088234265c25 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 20:49:30 +0000 Subject: [PATCH 07/12] executorch: keep portable_lib SHARED for llm/runner linkage Configure now fails past everything 0003 fixed, on a problem specific to one of the six pybind targets it converted to MODULE: CMake Error at extension/llm/runner/CMakeLists.txt:122 (target_link_libraries): Target "portable_lib" of type MODULE_LIBRARY may not be linked into another target. One may link only to INTERFACE, OBJECT, STATIC or SHARED libraries, or to executables with the ENABLE_EXPORTS property set. extension/llm/runner/CMakeLists.txt does target_link_libraries(_llm_runner PRIVATE ... portable_lib ...) -- portable_lib is both a Python extension module and a library another CMake target links against, unlike the other five 0003 touched. CMake refuses to target_link_libraries() a MODULE_LIBRARY into anything else no matter how it was created; there is no pybind11_add_module() argument that opts out of this, so portable_lib has to go back to SHARED and the Development.Embed gap has to be closed a different way. FindPython's __Python_ADD_LIBRARY() (Modules/FindPython/Support.cmake) requires a Python::Python target to already exist before it will add_library(... SHARED ...), then links the new target against it -- but nothing requires that target to carry a real embeddable libpython. Defining Python::Python ourselves as a link-free INTERFACE IMPORTED target when Development.Embed didn't find one satisfies the existence check; strip_python_lib() (already applied to portable_lib) then strips it back out of portable_lib's real link line exactly as it already does for the other five, so no embeddable libpython is ever actually linked. This reproduces upstream's own working configuration, since SHARED is what pybind11_add_module(portable_lib ...) already used before 0003, on their manylinux-builder images that do have Development.Embed. Patch under patches/executorch/1.4.1/ per the porting skill, marked "To upstream" since this is the same class of portability bug as 0003, not riscv64-specific. --- ...able_lib-SHARED-for-llm-runner-linka.patch | 78 +++++++++++++++++++ 1 file changed, 78 insertions(+) create mode 100644 patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch diff --git a/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch b/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch new file mode 100644 index 00000000000..7b5b79ced54 --- /dev/null +++ b/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch @@ -0,0 +1,78 @@ +From 0000000000000000000000000000000000000004 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: keep portable_lib SHARED for llm/runner linkage + +0003 converted all six pybind11_add_module( SHARED ...) call sites to +plain pybind11_add_module( ...) (MODULE), since vanilla manylinux +images never satisfy Python_add_library(SHARED)'s Python::Python +(Development.Embed) requirement. portable_lib is not like the other five, +though: extension/llm/runner/CMakeLists.txt does +target_link_libraries(_llm_runner PRIVATE ... portable_lib ...), and CMake +refuses to link a MODULE_LIBRARY into another target at all: + + CMake Error: Target "portable_lib" of type MODULE_LIBRARY may not be + linked into another target. One may link only to INTERFACE, OBJECT, + STATIC or SHARED libraries, or to executables with the ENABLE_EXPORTS + property set. + +This is a CMake generator restriction on the target type itself (checked in +every CMake version's link-dependency analysis), not something any +pybind11_add_module() argument can opt out of -- so portable_lib has to stay +SHARED, and the Development.Embed gap has to be closed some other way. + +FindPython's __Python_ADD_LIBRARY() (Modules/FindPython/Support.cmake) +requires the Python::Python target to already exist before it will even +add_library(... SHARED ...), and then links the new target against it: + + if (NOT type STREQUAL "MODULE" AND NOT TARGET ${prefix}::Python) + message (SEND_ERROR "...dependent target '${prefix}::Python' is not + defined. Did you miss to request COMPONENT 'Development.Embed'?") + ... + target_link_libraries (${name} PRIVATE ${prefix}::Python) + +Nothing requires that target to carry a real embeddable libpython, though: +defining Python::Python ourselves as a link-free INTERFACE IMPORTED target +when Development.Embed didn't find one satisfies the existence check, and +strip_python_lib() (already applied to portable_lib right below) then +removes Python::Python/pybind11::embed from portable_lib's real link line +exactly as it already does for the other five, now-MODULE targets -- so no +embeddable libpython is ever actually linked, and portable_lib comes out +identical to what it built as on upstream's own manylinux-builder CI, which +does have Development.Embed and never exercises this fallback. + +Upstream-Status: To upstream [not yet submitted; a real portability bug on any vanilla manylinux image lacking an embeddable libpython (Development.Embed), independent of riscv64, same as 0003] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 15 ++++++++++++++- + 1 file changed, 14 insertions(+), 1 deletion(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index 72ee20a..52da31b 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -1155,7 +1155,20 @@ if(EXECUTORCH_BUILD_PYBIND) + target_link_libraries(util PRIVATE torch c10 executorch extension_tensor) + + # pybind portable_lib +- pybind11_add_module(portable_lib extension/pybindings/pybindings.cpp) ++ # Unlike the other five pybind11_add_module() call sites, portable_lib is ++ # itself target_link_libraries()'d into extension/llm/runner's _llm_runner, ++ # and CMake refuses to link a MODULE_LIBRARY into another target -- so this ++ # one has to stay SHARED. Python_add_library(SHARED) unconditionally ++ # requires a Python::Python target to already exist (Development.Embed), ++ # which vanilla manylinux images never provide; define it as a link-free ++ # stand-in when missing so the SHARED add_library() call is allowed to ++ # proceed. strip_python_lib() below removes it (and pybind11::embed) from ++ # portable_lib's real link line either way, so no embeddable libpython is ++ # ever actually required. ++ if(NOT TARGET Python::Python) ++ add_library(Python::Python INTERFACE IMPORTED) ++ endif() ++ pybind11_add_module(portable_lib SHARED extension/pybindings/pybindings.cpp) + strip_python_lib(portable_lib) + # The actual output file needs a leading underscore so it can coexist with + # portable_lib.py in the same python package. PyTorch requires C++20, so +-- +2.34.1 From 9f0b7b330ea1d509481c0102d8e978ab5d88421d Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 20:49:46 +0000 Subject: [PATCH 08/12] executorch: drop FFHT SIMD arch fatal error, portable fallback exists The second independent configure failure past 0003 is unrelated to pybind/MODULE-vs-SHARED entirely: CMake Error at extension/llm/custom_ops/CMakeLists.txt:61 (message): Unsupported CMAKE_SYSTEM_PROCESSOR riscv64. (If 32-bit x86, try using fht_avx.c and send a PR if it works!) extension/llm/custom_ops/CMakeLists.txt picks a vendored FFHT (Fast Hadamard Transform) SIMD kernel source file based on CMAKE_SYSTEM_PROCESSOR -- fht_neon.c for aarch64/arm64/armv7, fht_avx.c for x86_64/AMD64 -- and hard-errors for anything else. But nothing on riscv64 needs an FFHT SIMD kernel at all. extension/llm/custom_ops/spinquant/fast_hadamard_transform.h's fast_hadamard_transform_ffht_impl() -- the only caller of fht_float()/ fht_double() anywhere outside the vendored FFHT sources themselves -- branches on compiler-defined architecture macros, not CMAKE_SYSTEM_PROCESSOR: #if defined(__aarch64__) || defined(__x86_64__) fht_float(vec, log2_vec_size); #else fast_hadamard_transform_simple_impl(vec, log2_vec_size); #endif so on riscv64 the portable, scalar fast_hadamard_transform_simple_impl() already runs instead, and fht_float()/fht_double() are never referenced. The CMakeLists.txt FATAL_ERROR was refusing to configure over a kernel source file that would never have been needed. Replace it with a STATUS message and add no FFHT source for these architectures -- the vendored fht_avx.c/fht_neon.c/fht_sse.c/dumb_fht.c files use a different, narrower API regardless (dumb_fht.c is void dumb_fht(float*, int), not fht_float()'s int fht_float(float*, int), with no fht_double()/_oop equivalents at all), so none of them was ever a drop-in match here. Patch under patches/executorch/1.4.1/ per the porting skill, marked "To upstream" since the FATAL_ERROR blocks every architecture outside x86_64/aarch64/armv7 even though a portable fallback needing none of them already exists, not riscv64-specific. --- ...-simd-arch-fatal-error-portable-fall.patch | 73 +++++++++++++++++++ 1 file changed, 73 insertions(+) create mode 100644 patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch diff --git a/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch b/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch new file mode 100644 index 00000000000..5ec70df461b --- /dev/null +++ b/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch @@ -0,0 +1,73 @@ +From 0000000000000000000000000000000000000005 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: drop FFHT SIMD arch fatal error, portable fallback exists + +extension/llm/custom_ops/CMakeLists.txt picks a vendored FFHT (Fast +Hadamard Transform) SIMD kernel source file to compile based on +CMAKE_SYSTEM_PROCESSOR -- fht_neon.c for aarch64/arm64/armv7, fht_avx.c for +x86_64/AMD64 -- and hard-errors for every other architecture, riscv64 +included: + + CMake Error: Unsupported CMAKE_SYSTEM_PROCESSOR riscv64. (If 32-bit x86, + try using fht_avx.c and send a PR if it works!) + +But nothing on such an architecture actually needs an FFHT SIMD kernel. +extension/llm/custom_ops/spinquant/fast_hadamard_transform.h's +fast_hadamard_transform_ffht_impl() -- the only caller of fht_float()/ +fht_double() outside the vendored FFHT sources themselves -- already +branches on the compiler-defined architecture macros, not +CMAKE_SYSTEM_PROCESSOR: + + #if defined(__aarch64__) || defined(__x86_64__) + fht_float(vec, log2_vec_size); + normalize_after_fht(vec, log2_vec_size); + #else + fast_hadamard_transform_simple_impl(vec, log2_vec_size); + #endif + +so on any architecture other than __aarch64__/__x86_64__ -- riscv64 +included -- the portable, scalar fast_hadamard_transform_simple_impl() is +already used at compile time, and the FFHT SIMD kernel's fht_float()/ +fht_double() symbols are never referenced, let alone linked. The +CMakeLists.txt FATAL_ERROR was refusing to configure over a source file +that would never have been needed. Replace it with a STATUS message and +add no FFHT source at all for these architectures; the vendored fht_avx.c/ +fht_neon.c/fht_sse.c/dumb_fht.c files use a different, narrower API +(void dumb_fht(float*, int) vs. fht_float()'s int fht_float(float*, int), +and no fht_double()/_oop equivalents) and were never a drop-in match +regardless of architecture. + +Upstream-Status: To upstream [not yet submitted; the FATAL_ERROR blocks every architecture outside x86_64/aarch64/armv7 even though fast_hadamard_transform.h already has a portable fallback that needs none of them, independent of riscv64] + +Signed-off-by: RISE Project CI +--- + extension/llm/custom_ops/CMakeLists.txt | 12 +++++++++--- + 1 file changed, 9 insertions(+), 3 deletions(-) + +diff --git a/extension/llm/custom_ops/CMakeLists.txt b/extension/llm/custom_ops/CMakeLists.txt +index fa847cb..d9f6e4f 100644 +--- a/extension/llm/custom_ops/CMakeLists.txt ++++ b/extension/llm/custom_ops/CMakeLists.txt +@@ -58,10 +58,16 @@ elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64)") + "extension/llm/custom_ops/spinquant/third-party/FFHT/fht_avx.c" + ) + else() ++ # No FFHT SIMD kernel exists for this architecture, but none is required: ++ # fast_hadamard_transform_ffht_impl() (fast_hadamard_transform.h) only ++ # calls fht_float() when __aarch64__ or __x86_64__ is defined, and already ++ # falls back at compile time to the portable ++ # fast_hadamard_transform_simple_impl() everywhere else, so nothing here ++ # links against fht_float/fht_double. + message( +- FATAL_ERROR +- "Unsupported CMAKE_SYSTEM_PROCESSOR ${CMAKE_SYSTEM_PROCESSOR}. (If \ +-32-bit x86, try using fht_avx.c and send a PR if it works!)" ++ STATUS ++ "No FFHT SIMD kernel for CMAKE_SYSTEM_PROCESSOR ${CMAKE_SYSTEM_PROCESSOR}; " ++ "using the portable fast_hadamard_transform_simple_impl fallback." + ) + endif() + +-- +2.34.1 From c86a1cfc4671691fe9cca79af7002564baebaad6 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 23:17:56 +0000 Subject: [PATCH 09/12] executorch: pin torch to 2.13.0, matching torch_pin.py's own ABI freeze The cp314 leg's build step got past CMake configure (0003/0004/0005) and failed at actual compilation instead, in two clusters: no matching constructor for c10::complex::complex(c10::complex), and "expected ',' or '...' before C10_LIFETIMEBOUND" across ArrayRef.h/ TensorAccessor.h/Tensor.h. torch_pin.py pins TORCH_VERSION="2.13.0", the release executorch's vendored runtime/core/portable_type/c10 header snapshot is frozen against. CIBW_BEFORE_BUILD copied setup.py's own unbounded "torch>=2.13.0a0" floor, so once pypi.riseproject.dev started serving torch 2.14.0 for every interpreter, pip resolved that instead of 2.13.0 (confirmed from the job log: "Successfully installed ... torch-2.14.0+cpu"). The real 2.14.0 headers and the vendored 2.13.0-era copy then collide in the same translation unit: 2.14.0's c10::complex gained constructors the vendored, 2.13.0-era class doesn't have, and C10_LIFETIMEBOUND (added between 2.13.0 and 2.14.0, confirmed by diffing torch/headeronly/macros/ Macros.h between the two tags) resolves to the vendored, pre-macro copy and is left undefined. torch 2.13.0 has a riscv64 wheel on our registry for every interpreter in this matrix, so this is a pin problem, not an unsupported-interpreter one (gotcha 615) - pin the build-time install to exactly what torch_pin.py names instead of dropping cp314, and pin CIBW_TEST_REQUIRES the same way so the test install doesn't resolve a newer torch than the extension was actually compiled against. --- .github/workflows/build-executorch.yml | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index 0cc69bb20ff..d3793c9354a 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -173,13 +173,23 @@ jobs: # plus torch and turn isolation off (mirrors build-torchaudio.yml / # build-torchcodec.yml). CIBW_BUILD_FRONTEND: "pip; args: --no-build-isolation" + # torch_pin.py pins TORCH_VERSION="2.13.0": this CMake build compiles + # against the installed torch's real headers alongside executorch's own + # vendored, 2.13.0-frozen copy of them (runtime/core/portable_type/c10), + # so an unbounded floor (setup.py's own unpinned "torch>=2.13.0a0") lets + # pip resolve a newer registry release whose headers conflict with the + # vendored ones in the same translation unit (gotcha 615). CIBW_BEFORE_BUILD: >- pip install "cmake>=3.24,<4.0.0" "packaging>=24.2" pyyaml "setuptools>=77.0.3" wheel zstd certifi && - pip install --only-binary=:all: "torch>=2.13.0a0" + pip install --only-binary=:all: "torch==2.13.0" CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + # Match the build-time pin (gotcha 615): the wheel's own metadata still + # says "torch>=2.13.0a0", so without this the test install would resolve + # a newer torch than the extension was actually compiled against. + CIBW_TEST_REQUIRES: torch==2.13.0 CIBW_TEST_SOURCES: riscv64_smoke_test.py CIBW_TEST_COMMAND: python riscv64_smoke_test.py From 10db2acb53f0a553e093d83067cf729851194490 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Tue, 29 Sep 2026 03:49:35 +0000 Subject: [PATCH 10/12] executorch: exclude torch's own libs from auditwheel repair Compiling now succeeds (torch pinned to 2.13.0), but the default CIBW_REPAIR_WHEEL_COMMAND then fails: auditwheel can't locate libc10.so because it isn't on any path it searches. libtorch/libc10/libgomp etc. are provided by the torch wheel at runtime (torch is imported before executorch and loads them RTLD_GLOBAL), so exclude them the same way build-torchaudio.yml and build-vllm.yml do (gotcha 17). --- .github/workflows/build-executorch.yml | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index d3793c9354a..44eaae35662 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -186,6 +186,17 @@ jobs: CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + # The torch wheel already carries these and loads them RTLD_GLOBAL (gotcha 17). + CIBW_REPAIR_WHEEL_COMMAND: >- + auditwheel repair -w {dest_dir} {wheel} + --exclude libtorch.so + --exclude libtorch_cpu.so + --exclude libtorch_python.so + --exclude libtorch_global_deps.so + --exclude libc10.so + --exclude libgomp.so.1 + --exclude libgfortran.so.5 + --exclude libopenblas.so.0 # Match the build-time pin (gotcha 615): the wheel's own metadata still # says "torch>=2.13.0a0", so without this the test install would resolve # a newer torch than the extension was actually compiled against. From d7e44f09bc34fbd77f72be0094f0291da152a26a Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Wed, 30 Sep 2026 09:03:07 +0000 Subject: [PATCH 11/12] executorch: TEMP add gdb backtrace capture to diagnose cpuinfo abort Restrict the matrix to cp312 only, keep debug symbols (CFLAGS/CXXFLAGS=-g and CMAKE_ARGS=-DCMAKE_BUILD_TYPE=RelWithDebInfo, matching build-tesserocr.yml's precedent), install gdb, and run the smoke test under gdb to capture a native backtrace of the deterministic cpuinfo_get_uarch abort. To be reset once the backtrace is read. --- .github/workflows/build-executorch.yml | 32 ++++++++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index 44eaae35662..d6c6936cf84 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -52,7 +52,9 @@ jobs: # cp310/cp311 are dropped: torch and pytorch-tokenizers (both hard runtime # deps) have no riscv64 wheel for either on pypi.riseproject.dev. cp314t # is dropped too: upstream itself ships no free-threaded wheel. - python: ["cp312", "cp313", "cp314"] + # TEMP: restricted to cp312 only while diagnosing the cpuinfo abort + # under gdb (gotcha 115) - restore ["cp312", "cp313", "cp314"] once done. + python: ["cp312"] name: Build executorch ${{ matrix.version }} ${{ matrix.python }}-manylinux_riscv64 runs-on: ubuntu-24.04-riscv timeout-minutes: 720 @@ -183,9 +185,17 @@ jobs: pip install "cmake>=3.24,<4.0.0" "packaging>=24.2" pyyaml "setuptools>=77.0.3" wheel zstd certifi && pip install --only-binary=:all: "torch==2.13.0" + # TEMP diagnostic (gotcha 115): keep debug symbols so gdb's backtrace + # resolves symbols instead of bare addresses. setup.py's get_build_type() + # only offers Debug/Release via $DEBUG, but appends $CMAKE_ARGS after its + # own -DCMAKE_BUILD_TYPE flag, so CMAKE_ARGS wins (RelWithDebInfo matches + # this repo's own precedent in build-tesserocr.yml, and keeps optimizations + # on so the crash still reproduces the same way as the real release build). CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + CFLAGS=-g CXXFLAGS=-g + CMAKE_ARGS=-DCMAKE_BUILD_TYPE=RelWithDebInfo # The torch wheel already carries these and loads them RTLD_GLOBAL (gotcha 17). CIBW_REPAIR_WHEEL_COMMAND: >- auditwheel repair -w {dest_dir} {wheel} @@ -202,7 +212,25 @@ jobs: # a newer torch than the extension was actually compiled against. CIBW_TEST_REQUIRES: torch==2.13.0 CIBW_TEST_SOURCES: riscv64_smoke_test.py - CIBW_TEST_COMMAND: python riscv64_smoke_test.py + # TEMP diagnostic (gotcha 115): gdb works on the real riscv64 runner + # (not under QEMU). The abort is 100% deterministic across all 3 + # python legs already observed, so a single gdb run is enough - no + # retry loop needed. Fail the step either way so the log is easy to + # spot: once with the backtrace if the abort reproduces, once with a + # clear "did not reproduce" message otherwise. + CIBW_BEFORE_TEST_LINUX: dnf -y install gdb + CIBW_TEST_COMMAND: >- + gdb -batch -ex run -ex "thread apply all bt 25" --args python riscv64_smoke_test.py + > /tmp/gdb-backtrace.log 2>&1; + gdb_status=$?; + cat /tmp/gdb-backtrace.log; + echo "gdb exit status: $gdb_status"; + if grep -qE "SIGABRT|Aborted" /tmp/gdb-backtrace.log; + then echo "=== crash reproduced under gdb (backtrace above) ==="; + exit 1; + else echo "=== SIGABRT/Aborted NOT found - crash did not reproduce under gdb ==="; + exit 1; + fi - name: Check the extensions and the vendored licences made it into the wheel run: | From 55fe9ace178cd513d99fe56f1fac75a8a5339ca6 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Fri, 2 Oct 2026 00:46:10 +0000 Subject: [PATCH 12/12] executorch: drop the gdb diagnostic, run the smoke test with faulthandler The gdb run (cp312 only, RelWithDebInfo) never produced a backtrace: the job sat in the cibuildwheel step for 2h40m and then lost its runner, so no log was ever uploaded. Its test command also exits 1 on both branches, so it could never pass. Restore the real cp312/cp313/cp314 release build. The failure it was chasing is unchanged and deterministic on every leg: the wheel builds and repairs fine, then the smoke test aborts with Error in cpuinfo: failed to parse file /sys/devices/system/cpu/cpu0/topology/core_id: "-1" is not an unsigned number Fatal error in cpuinfo: cpuinfo_get_uarch called before cpuinfo is initialized This runner's kernel reports core_id -1 (gotcha 14), cpuinfo's riscv64 Linux init gives up, and something in the wheel calls cpuinfo_get_uarch() without checking that cpuinfo_initialize() succeeded. Run the smoke test under -X faulthandler so the SIGABRT at least dumps the Python frame that triggers it, which pins down the native caller to patch. --- .github/workflows/build-executorch.yml | 32 ++------------------------ 1 file changed, 2 insertions(+), 30 deletions(-) diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml index d6c6936cf84..92e66d681e5 100644 --- a/.github/workflows/build-executorch.yml +++ b/.github/workflows/build-executorch.yml @@ -52,9 +52,7 @@ jobs: # cp310/cp311 are dropped: torch and pytorch-tokenizers (both hard runtime # deps) have no riscv64 wheel for either on pypi.riseproject.dev. cp314t # is dropped too: upstream itself ships no free-threaded wheel. - # TEMP: restricted to cp312 only while diagnosing the cpuinfo abort - # under gdb (gotcha 115) - restore ["cp312", "cp313", "cp314"] once done. - python: ["cp312"] + python: ["cp312", "cp313", "cp314"] name: Build executorch ${{ matrix.version }} ${{ matrix.python }}-manylinux_riscv64 runs-on: ubuntu-24.04-riscv timeout-minutes: 720 @@ -185,17 +183,9 @@ jobs: pip install "cmake>=3.24,<4.0.0" "packaging>=24.2" pyyaml "setuptools>=77.0.3" wheel zstd certifi && pip install --only-binary=:all: "torch==2.13.0" - # TEMP diagnostic (gotcha 115): keep debug symbols so gdb's backtrace - # resolves symbols instead of bare addresses. setup.py's get_build_type() - # only offers Debug/Release via $DEBUG, but appends $CMAKE_ARGS after its - # own -DCMAKE_BUILD_TYPE flag, so CMAKE_ARGS wins (RelWithDebInfo matches - # this repo's own precedent in build-tesserocr.yml, and keeps optimizations - # on so the crash still reproduces the same way as the real release build). CIBW_ENVIRONMENT: >- CMAKE_BUILD_PARALLEL_LEVEL=8 PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ - CFLAGS=-g CXXFLAGS=-g - CMAKE_ARGS=-DCMAKE_BUILD_TYPE=RelWithDebInfo # The torch wheel already carries these and loads them RTLD_GLOBAL (gotcha 17). CIBW_REPAIR_WHEEL_COMMAND: >- auditwheel repair -w {dest_dir} {wheel} @@ -212,25 +202,7 @@ jobs: # a newer torch than the extension was actually compiled against. CIBW_TEST_REQUIRES: torch==2.13.0 CIBW_TEST_SOURCES: riscv64_smoke_test.py - # TEMP diagnostic (gotcha 115): gdb works on the real riscv64 runner - # (not under QEMU). The abort is 100% deterministic across all 3 - # python legs already observed, so a single gdb run is enough - no - # retry loop needed. Fail the step either way so the log is easy to - # spot: once with the backtrace if the abort reproduces, once with a - # clear "did not reproduce" message otherwise. - CIBW_BEFORE_TEST_LINUX: dnf -y install gdb - CIBW_TEST_COMMAND: >- - gdb -batch -ex run -ex "thread apply all bt 25" --args python riscv64_smoke_test.py - > /tmp/gdb-backtrace.log 2>&1; - gdb_status=$?; - cat /tmp/gdb-backtrace.log; - echo "gdb exit status: $gdb_status"; - if grep -qE "SIGABRT|Aborted" /tmp/gdb-backtrace.log; - then echo "=== crash reproduced under gdb (backtrace above) ==="; - exit 1; - else echo "=== SIGABRT/Aborted NOT found - crash did not reproduce under gdb ==="; - exit 1; - fi + CIBW_TEST_COMMAND: python -X faulthandler riscv64_smoke_test.py - name: Check the extensions and the vendored licences made it into the wheel run: |