diff --git a/.github/workflows/build-executorch.yml b/.github/workflows/build-executorch.yml new file mode 100644 index 00000000000..2741ac79404 --- /dev/null +++ b/.github/workflows/build-executorch.yml @@ -0,0 +1,250 @@ +# SPDX-FileCopyrightText: 2026 The RISE Project +# SPDX-License-Identifier: MIT +--- +# This workflow is based on: +# https://github.com/pytorch/executorch/blob/v1.4.1/.github/workflows/build-wheels-linux.yml +name: Build executorch wheels (riscv64) + +on: + workflow_dispatch: + inputs: + version: + description: 'Version glob to (re)build; empty builds every version of docs/packages/executorch.yaml not released yet' + required: false + default: '' + pull_request: + branches: [main] + paths: + - '.github/workflows/build-executorch.yml' + - 'docs/packages/executorch.yaml' + - 'patches/executorch/**' + push: + branches: [main] + paths: + - '.github/workflows/build-executorch.yml' + - 'docs/packages/executorch.yaml' + - 'patches/executorch/**' + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + +permissions: + contents: read + +env: + MANYLINUX_RISCV64_IMAGE: quay.io/pypa/manylinux_2_39_riscv64 + +jobs: + setup: + uses: $/.github/workflows/_setup.yml + with: + package: executorch + version: ${{ inputs.version }} + + build_wheels: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + # cp310/cp311 are dropped: torch and pytorch-tokenizers (both hard runtime + # deps) have no riscv64 wheel for either on pypi.riseproject.dev. cp314t + # is dropped too: upstream itself ships no free-threaded wheel. + python: ["cp312", "cp313", "cp314"] + name: Build executorch ${{ matrix.version }} ${{ matrix.python }}-manylinux_riscv64 + runs-on: ubuntu-24.04-riscv + timeout-minutes: 720 + + env: + EXECUTORCH_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout executorch v${{ env.EXECUTORCH_VERSION }} + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: pytorch/executorch + ref: v${{ env.EXECUTORCH_VERSION }} + submodules: recursive + persist-credentials: false + + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + path: python-wheels + persist-credentials: false + + - name: Apply patches + run: git apply python-wheels/patches/executorch/${{ env.EXECUTORCH_VERSION }}/*.patch + + # Every library below is statically linked into the compiled extensions, and + # setuptools' license-files (widened by the patch above) only reaches the + # project root. + - name: Stage the vendored licences + run: | + set -euo pipefail + cp backends/xnnpack/third-party/FP16/LICENSE LICENSE.FP16 + cp backends/xnnpack/third-party/FXdiv/LICENSE LICENSE.FXdiv + cp backends/xnnpack/third-party/XNNPACK/LICENSE LICENSE.XNNPACK + cp backends/xnnpack/third-party/cpuinfo/LICENSE LICENSE.cpuinfo + cp backends/xnnpack/third-party/pthreadpool/LICENSE LICENSE.pthreadpool + cp third-party/flatbuffers/LICENSE LICENSE.flatbuffers + cp third-party/flatcc/LICENSE LICENSE.flatcc + cp kernels/optimized/third-party/eigen/LICENSE LICENSE.eigen + cp third-party/pocketfft/LICENSE.md LICENSE.pocketfft + cp third-party/json/LICENSE.MIT LICENSE.json + cp third-party/pybind11/LICENSE LICENSE.pybind11 + + # Upstream's own smoke test (.ci/scripts/wheel/test_linux.py) exercises the + # examples/ tree, which needs torchvision for its MobileNetV3 model; + # torchvision has no riscv64 wheel. This test keeps upstream's own + # assertions (registered backend names) and replaces the model with a small + # hand-built one, run through both the portable (non-delegated) path and a + # real XNNPACK-delegated export, so the compiled extension, the portable + # kernels and the XNNPACK backend are all genuinely exercised end to end. + - name: Write the riscv64 smoke test + run: | + cat > riscv64_smoke_test.py <<'EOF' + import platform + + import torch + from torch.export import export + + from executorch.backends.xnnpack.partition.xnnpack_partitioner import ( + XnnpackPartitioner, + ) + from executorch.exir import to_edge, to_edge_transform_and_lower + from executorch.extension.pybindings.portable_lib import ( + _get_registered_backend_names, + _load_for_executorch_from_buffer, + ) + + registered = _get_registered_backend_names() + assert "XnnpackBackend" in registered, registered + assert "OpenvinoBackend" in registered, registered + if platform.machine() in ("x86_64", "amd64"): + assert "QnnBackend" in registered, registered + + + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + self.linear = torch.nn.Linear(4, 4) + + def forward(self, x): + return self.linear(x).relu() + + + model = Model().eval() + example_inputs = (torch.randn(2, 4),) + expected = model(*example_inputs) + + portable_prog = to_edge(export(model, example_inputs, strict=True)).to_executorch() + portable_module = _load_for_executorch_from_buffer(portable_prog.buffer) + portable_out = portable_module.forward(example_inputs)[0] + assert torch.allclose(portable_out, expected, atol=1e-4), (portable_out, expected) + print("portable (non-delegated) execution OK") + + xnnpack_prog = to_edge_transform_and_lower( + export(model, example_inputs, strict=True), + partitioner=[XnnpackPartitioner()], + ).to_executorch() + xnnpack_module = _load_for_executorch_from_buffer(xnnpack_prog.buffer) + xnnpack_out = xnnpack_module.forward(example_inputs)[0] + assert torch.allclose(xnnpack_out, expected, atol=1e-4), (xnnpack_out, expected) + print("XNNPACK-delegated execution OK") + EOF + + - uses: pypa/cibuildwheel@1828c10ab37f080699c7b81cea34097c684a7074 # v4.2.0 + with: + only: ${{ matrix.python }}-manylinux_riscv64 + env: + CIBW_MANYLINUX_RISCV64_IMAGE: ${{ env.MANYLINUX_RISCV64_IMAGE }} + # No riscv64 cmake wheel exists, so pip builds it from source as a + # build dependency; its own CMake bootstrap needs OpenSSL's dev + # headers to find libcrypto. + CIBW_BEFORE_ALL_LINUX: dnf -y install openssl-devel + # get_torch_base_path (tools/cmake/Utils.cmake) runs `import torch` in + # the same interpreter as the build backend to add it to + # CMAKE_PREFIX_PATH, but torch is only a runtime dependency + # (setup.py), not a build-system.requires entry, so PEP 517 isolation + # hides it from that interpreter. Preinstall build-system.requires + # plus torch and turn isolation off (mirrors build-torchaudio.yml / + # build-torchcodec.yml). + CIBW_BUILD_FRONTEND: "pip; args: --no-build-isolation" + # torch_pin.py pins TORCH_VERSION="2.13.0": this CMake build compiles + # against the installed torch's real headers alongside executorch's own + # vendored, 2.13.0-frozen copy of them (runtime/core/portable_type/c10), + # so an unbounded floor (setup.py's own unpinned "torch>=2.13.0a0") lets + # pip resolve a newer registry release whose headers conflict with the + # vendored ones in the same translation unit (gotcha 615). + CIBW_BEFORE_BUILD: >- + pip install "cmake>=3.24,<4.0.0" "packaging>=24.2" pyyaml + "setuptools>=77.0.3" wheel zstd certifi && + pip install --only-binary=:all: "torch==2.13.0" + # Without BUILD_VERSION, setup.py appends + to version.txt + # (mirrors build-torchaudio.yml / build-torchcodec.yml). + CIBW_ENVIRONMENT: >- + BUILD_VERSION=${{ env.EXECUTORCH_VERSION }} + CMAKE_BUILD_PARALLEL_LEVEL=8 + PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + # The torch wheel already carries these and loads them RTLD_GLOBAL (gotcha 17). + CIBW_REPAIR_WHEEL_COMMAND: >- + auditwheel repair -w {dest_dir} {wheel} + --exclude libtorch.so + --exclude libtorch_cpu.so + --exclude libtorch_python.so + --exclude libtorch_global_deps.so + --exclude libc10.so + --exclude libgomp.so.1 + --exclude libgfortran.so.5 + --exclude libopenblas.so.0 + # Match the build-time pin (gotcha 615): the wheel's own metadata still + # says "torch>=2.13.0a0", so without this the test install would resolve + # a newer torch than the extension was actually compiled against. + CIBW_TEST_REQUIRES: torch==2.13.0 + CIBW_TEST_SOURCES: riscv64_smoke_test.py + CIBW_TEST_COMMAND: python -X faulthandler riscv64_smoke_test.py + + - name: Check the extensions and the vendored licences made it into the wheel + run: | + python3 - wheelhouse/*.whl <<'EOF' + import sys, zipfile + expected = {"LICENSE", "LICENSE.FP16", "LICENSE.FXdiv", "LICENSE.XNNPACK", + "LICENSE.cpuinfo", "LICENSE.pthreadpool", "LICENSE.flatbuffers", + "LICENSE.flatcc", "LICENSE.eigen", "LICENSE.pocketfft", + "LICENSE.json", "LICENSE.pybind11"} + for whl in sys.argv[1:]: + names = zipfile.ZipFile(whl).namelist() + assert [n for n in names + if n.startswith("executorch/extension/pybindings/_portable_lib.") + and n.endswith(".so")], whl + licenses = {n.split(".dist-info/licenses/", 1)[1] for n in names + if ".dist-info/licenses/" in n and not n.endswith("/")} + assert licenses == expected, f"{whl}: {sorted(licenses)}" + print(whl, "ok") + EOF + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: executorch-${{ env.EXECUTORCH_VERSION }}-${{ matrix.python }}-manylinux_riscv64 + path: wheelhouse/*.whl + if-no-files-found: error + + publish: + name: Publish executorch ${{ matrix.version }} + needs: [setup, build_wheels] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + permissions: + contents: write + pull-requests: write + uses: $/.github/workflows/_publish-wheel.yml + secrets: + app-private-key: ${{ secrets.RISEPROJECT_APP_PRIVATE_KEY }} + with: + artifact-pattern: executorch-${{ matrix.version }}-*-manylinux_riscv64 diff --git a/docs/packages/executorch.yaml b/docs/packages/executorch.yaml new file mode 100644 index 00000000000..ae60b1cf3ff --- /dev/null +++ b/docs/packages/executorch.yaml @@ -0,0 +1,5 @@ +package-name: executorch +source-code: https://github.com/pytorch/executorch +license: BSD-3-Clause AND MIT AND BSD-2-Clause AND Apache-2.0 AND MPL-2.0 +versions: +- version: 1.4.1 diff --git a/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch new file mode 100644 index 00000000000..7e65c08e10e --- /dev/null +++ b/patches/executorch/1.4.1/0001-packaging-Include-vendored-third-party-licences-in-.patch @@ -0,0 +1,34 @@ +From 0000000000000000000000000000000000000001 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Thu, 10 Jul 2026 00:00:00 +0000 +Subject: [PATCH] packaging: Include vendored third-party licences in the + wheel + +The compiled extensions in this wheel statically link FP16, FXdiv, +XNNPACK, cpuinfo, pthreadpool, flatbuffers, flatcc, Eigen, pocketfft, +nlohmann/json and pybind11, but `license-files` only reaches the +project's own LICENSE. RISE distributes these wheels and must also +carry the licence of every statically linked dependency, so widen the +glob to pick up the `LICENSE.` files the riscv64 build stages at +the project root before `pip wheel` runs. + +Upstream-Status: Inappropriate [only relevant to a distributor bundling vendored third-party licences; reproduces identically on any architecture] + +Signed-off-by: RISE Project CI +--- + pyproject.toml | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) + +diff --git a/pyproject.toml b/pyproject.toml +index 0000000000..0000000000 100644 +--- a/pyproject.toml ++++ b/pyproject.toml +@@ -100,7 +100,7 @@ + # package, and package_data to restrict the set up non-python files we + # include. See also setuptools/discovery.py for custom finders. + [tool.setuptools] +-license-files = ["LICENSE"] ++license-files = ["LICENSE", "LICENSE.*"] + + [tool.setuptools.package-dir] + # Tell setuptools to follow the symlink: src/executorch/* -> * for all first level diff --git a/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch b/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch new file mode 100644 index 00000000000..41e36734ac7 --- /dev/null +++ b/patches/executorch/1.4.1/0002-cmake-relax-source-directory-name-check-for-cibuild.patch @@ -0,0 +1,65 @@ +From 0000000000000000000000000000000000000002 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Thu, 10 Jul 2026 00:00:00 +0000 +Subject: [PATCH] cmake: stop requiring the checkout dir be named `executorch` + +CMakeLists.txt hard-fails configure unless the checkout's directory is +named exactly `executorch`, because `_common_include_directories` adds +the checkout's *parent* directory so in-tree code can say +`#include ` (upstream tracks lifting this +restriction at https://github.com/pytorch/executorch/issues/6475). + +cibuildwheel always mounts/copies the source into its manylinux +container at the fixed path `/project`, so this check always trips +there, regardless of what directory this repo is checked out to on the +runner. Provide the same `` include path through a +configure-time symlink instead of requiring a literally named parent +directory, so the build works unmodified under cibuildwheel. + +Upstream-Status: Inappropriate [only relevant to cibuildwheel, which always mounts the source at a fixed /project path inside its build container; see pytorch/executorch#6475] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 20 +++++++++----------- + 1 file changed, 9 insertions(+), 11 deletions(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index be0cec2..4668d0a 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -447,22 +447,20 @@ if(NOT DEFINED CMAKE_POSITION_INDEPENDENT_CODE + list(APPEND _common_compile_options $<$>:-fPIC>) + endif() + +-# Let files say "include ". +-# TODO(#6475): This requires/assumes that the repo lives in a directory named +-# exactly `executorch`. Check the assumption first. Remove this check once we +-# stop relying on the assumption. +-cmake_path(GET CMAKE_CURRENT_SOURCE_DIR FILENAME _repo_dir_name) +-if(NOT "${_repo_dir_name}" STREQUAL "executorch") +- message( +- FATAL_ERROR +- "The ExecuTorch repo must be cloned into a directory named exactly " +- "`executorch`; found `${_repo_dir_name}`. See " +- "https://github.com/pytorch/executorch/issues/6475 for progress on a " +- "fix for this restriction." ++# Let files say "include ", regardless of what ++# the checkout directory is actually named (see ++# https://github.com/pytorch/executorch/issues/6475). Provide that include ++# path unconditionally through a configure-time symlink, rather than ++# requiring the checkout itself be named `executorch`. ++set(_executorch_include_shim "${CMAKE_BINARY_DIR}/_executorch_include") ++file(MAKE_DIRECTORY "${_executorch_include_shim}") ++if(NOT EXISTS "${_executorch_include_shim}/executorch") ++ file(CREATE_LINK "${CMAKE_CURRENT_SOURCE_DIR}" ++ "${_executorch_include_shim}/executorch" SYMBOLIC + ) + endif() + set(_common_include_directories +- $ ++ $ + $ + $ + $ +-- +2.34.1 diff --git a/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch new file mode 100644 index 00000000000..88b12e65428 --- /dev/null +++ b/patches/executorch/1.4.1/0003-cmake-build-pybind-extensions-as-MODULE-not-SHARED.patch @@ -0,0 +1,114 @@ +From 0000000000000000000000000000000000000003 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: build pybind extensions as MODULE, not SHARED + +pybind11_add_module( SHARED ...) makes CMake's Python_add_library() +require the Python::Python target (COMPONENT Development.Embed) to exist +at configure time, even though strip_python_lib() immediately removes the +resulting Python::Python/pybind11::embed link dependency again -- these +are ordinary dlopen()'d extension modules that resolve Python symbols +from the running interpreter, never anything that embeds Python. + +Vanilla manylinux images (unlike the custom manylinux-builder images +PyTorch's own CI uses) ship only a static libpythonX.Y.a and never +satisfy Development.Embed, so find_package(Python) never defines +Python::Python and every pybind11_add_module(... SHARED ...) call fails +configure outright: + + CMake Error: Python_ADD_LIBRARY: dependent target 'Python::Python' is + not defined. Did you miss to request COMPONENT 'Development.Embed'? + +Dropping SHARED (the pybind11 default, MODULE) only requires +Development.Module, which these images do provide, and produces the +exact same importable extension: strip_python_lib()'s job of avoiding a +runtime libpython link was already redundant with MODULE all along. + +Upstream-Status: To upstream [not yet submitted; a real portability bug on any vanilla manylinux image lacking an embeddable libpython (Development.Embed), independent of riscv64 -- pytorch/executorch's own CI runs a custom manylinux-builder image that does provide one, so upstream has likely never hit this] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 4 ++-- + backends/apple/coreml/CMakeLists.txt | 2 +- + codegen/tools/CMakeLists.txt | 2 +- + extension/llm/runner/CMakeLists.txt | 2 +- + extension/training/CMakeLists.txt | 2 +- + 5 files changed, 6 insertions(+), 6 deletions(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index be0cec2..a02db93 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -1157,7 +1157,7 @@ if(EXECUTORCH_BUILD_PYBIND) + target_link_libraries(util PRIVATE torch c10 executorch extension_tensor) + + # pybind portable_lib +- pybind11_add_module(portable_lib SHARED extension/pybindings/pybindings.cpp) ++ pybind11_add_module(portable_lib extension/pybindings/pybindings.cpp) + strip_python_lib(portable_lib) + # The actual output file needs a leading underscore so it can coexist with + # portable_lib.py in the same python package. PyTorch requires C++20, so +@@ -1190,7 +1190,7 @@ if(EXECUTORCH_BUILD_PYBIND) + # pybind data_loader module - provides PyDataLoader type for external + # pybinding extensions to create custom data loaders + pybind11_add_module( +- data_loader SHARED extension/pybindings/pybindings_data_loader.cpp ++ data_loader extension/pybindings/pybindings_data_loader.cpp + ) + strip_python_lib(data_loader) + target_include_directories(data_loader PRIVATE ${_common_include_directories}) +diff --git a/backends/apple/coreml/CMakeLists.txt b/backends/apple/coreml/CMakeLists.txt +index 5391a35..03612c5 100644 +--- a/backends/apple/coreml/CMakeLists.txt ++++ b/backends/apple/coreml/CMakeLists.txt +@@ -172,7 +172,7 @@ install( + + if(EXECUTORCH_BUILD_PYBIND) + pybind11_add_module( +- executorchcoreml SHARED runtime/inmemoryfs/inmemory_filesystem_py.cpp ++ executorchcoreml runtime/inmemoryfs/inmemory_filesystem_py.cpp + runtime/inmemoryfs/inmemory_filesystem_utils.cpp + ) + strip_python_lib(executorchcoreml) +diff --git a/codegen/tools/CMakeLists.txt b/codegen/tools/CMakeLists.txt +index b829e83..efdfdf3 100644 +--- a/codegen/tools/CMakeLists.txt ++++ b/codegen/tools/CMakeLists.txt +@@ -8,7 +8,7 @@ + # Check if pybind11 is available + + # Create the selective_build pybind11 module +-pybind11_add_module(selective_build SHARED selective_build.cpp) ++pybind11_add_module(selective_build selective_build.cpp) + strip_python_lib(selective_build) + + # Set the output name to match the module name +diff --git a/extension/llm/runner/CMakeLists.txt b/extension/llm/runner/CMakeLists.txt +index 3e9f811..8a5e808 100644 +--- a/extension/llm/runner/CMakeLists.txt ++++ b/extension/llm/runner/CMakeLists.txt +@@ -110,7 +110,7 @@ endif() + if(EXECUTORCH_BUILD_PYBIND) + # Create the Python extension module for LLM runners + pybind11_add_module( +- _llm_runner SHARED ${CMAKE_CURRENT_SOURCE_DIR}/pybindings.cpp ++ _llm_runner ${CMAKE_CURRENT_SOURCE_DIR}/pybindings.cpp + ) + strip_python_lib(_llm_runner) + +diff --git a/extension/training/CMakeLists.txt b/extension/training/CMakeLists.txt +index aaaeb8d..3591af0 100644 +--- a/extension/training/CMakeLists.txt ++++ b/extension/training/CMakeLists.txt +@@ -63,7 +63,7 @@ if(EXECUTORCH_BUILD_PYBIND) + endif() + + pybind11_add_module( +- _training_lib SHARED ++ _training_lib + ${CMAKE_CURRENT_SOURCE_DIR}/pybindings/_training_lib.cpp + ) + strip_python_lib(_training_lib) +-- +2.34.1 + diff --git a/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch b/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch new file mode 100644 index 00000000000..7b5b79ced54 --- /dev/null +++ b/patches/executorch/1.4.1/0004-cmake-keep-portable_lib-SHARED-for-llm-runner-linka.patch @@ -0,0 +1,78 @@ +From 0000000000000000000000000000000000000004 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: keep portable_lib SHARED for llm/runner linkage + +0003 converted all six pybind11_add_module( SHARED ...) call sites to +plain pybind11_add_module( ...) (MODULE), since vanilla manylinux +images never satisfy Python_add_library(SHARED)'s Python::Python +(Development.Embed) requirement. portable_lib is not like the other five, +though: extension/llm/runner/CMakeLists.txt does +target_link_libraries(_llm_runner PRIVATE ... portable_lib ...), and CMake +refuses to link a MODULE_LIBRARY into another target at all: + + CMake Error: Target "portable_lib" of type MODULE_LIBRARY may not be + linked into another target. One may link only to INTERFACE, OBJECT, + STATIC or SHARED libraries, or to executables with the ENABLE_EXPORTS + property set. + +This is a CMake generator restriction on the target type itself (checked in +every CMake version's link-dependency analysis), not something any +pybind11_add_module() argument can opt out of -- so portable_lib has to stay +SHARED, and the Development.Embed gap has to be closed some other way. + +FindPython's __Python_ADD_LIBRARY() (Modules/FindPython/Support.cmake) +requires the Python::Python target to already exist before it will even +add_library(... SHARED ...), and then links the new target against it: + + if (NOT type STREQUAL "MODULE" AND NOT TARGET ${prefix}::Python) + message (SEND_ERROR "...dependent target '${prefix}::Python' is not + defined. Did you miss to request COMPONENT 'Development.Embed'?") + ... + target_link_libraries (${name} PRIVATE ${prefix}::Python) + +Nothing requires that target to carry a real embeddable libpython, though: +defining Python::Python ourselves as a link-free INTERFACE IMPORTED target +when Development.Embed didn't find one satisfies the existence check, and +strip_python_lib() (already applied to portable_lib right below) then +removes Python::Python/pybind11::embed from portable_lib's real link line +exactly as it already does for the other five, now-MODULE targets -- so no +embeddable libpython is ever actually linked, and portable_lib comes out +identical to what it built as on upstream's own manylinux-builder CI, which +does have Development.Embed and never exercises this fallback. + +Upstream-Status: To upstream [not yet submitted; a real portability bug on any vanilla manylinux image lacking an embeddable libpython (Development.Embed), independent of riscv64, same as 0003] + +Signed-off-by: RISE Project CI +--- + CMakeLists.txt | 15 ++++++++++++++- + 1 file changed, 14 insertions(+), 1 deletion(-) + +diff --git a/CMakeLists.txt b/CMakeLists.txt +index 72ee20a..52da31b 100644 +--- a/CMakeLists.txt ++++ b/CMakeLists.txt +@@ -1155,7 +1155,20 @@ if(EXECUTORCH_BUILD_PYBIND) + target_link_libraries(util PRIVATE torch c10 executorch extension_tensor) + + # pybind portable_lib +- pybind11_add_module(portable_lib extension/pybindings/pybindings.cpp) ++ # Unlike the other five pybind11_add_module() call sites, portable_lib is ++ # itself target_link_libraries()'d into extension/llm/runner's _llm_runner, ++ # and CMake refuses to link a MODULE_LIBRARY into another target -- so this ++ # one has to stay SHARED. Python_add_library(SHARED) unconditionally ++ # requires a Python::Python target to already exist (Development.Embed), ++ # which vanilla manylinux images never provide; define it as a link-free ++ # stand-in when missing so the SHARED add_library() call is allowed to ++ # proceed. strip_python_lib() below removes it (and pybind11::embed) from ++ # portable_lib's real link line either way, so no embeddable libpython is ++ # ever actually required. ++ if(NOT TARGET Python::Python) ++ add_library(Python::Python INTERFACE IMPORTED) ++ endif() ++ pybind11_add_module(portable_lib SHARED extension/pybindings/pybindings.cpp) + strip_python_lib(portable_lib) + # The actual output file needs a leading underscore so it can coexist with + # portable_lib.py in the same python package. PyTorch requires C++20, so +-- +2.34.1 diff --git a/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch b/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch new file mode 100644 index 00000000000..5ec70df461b --- /dev/null +++ b/patches/executorch/1.4.1/0005-cmake-drop-FFHT-simd-arch-fatal-error-portable-fall.patch @@ -0,0 +1,73 @@ +From 0000000000000000000000000000000000000005 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] cmake: drop FFHT SIMD arch fatal error, portable fallback exists + +extension/llm/custom_ops/CMakeLists.txt picks a vendored FFHT (Fast +Hadamard Transform) SIMD kernel source file to compile based on +CMAKE_SYSTEM_PROCESSOR -- fht_neon.c for aarch64/arm64/armv7, fht_avx.c for +x86_64/AMD64 -- and hard-errors for every other architecture, riscv64 +included: + + CMake Error: Unsupported CMAKE_SYSTEM_PROCESSOR riscv64. (If 32-bit x86, + try using fht_avx.c and send a PR if it works!) + +But nothing on such an architecture actually needs an FFHT SIMD kernel. +extension/llm/custom_ops/spinquant/fast_hadamard_transform.h's +fast_hadamard_transform_ffht_impl() -- the only caller of fht_float()/ +fht_double() outside the vendored FFHT sources themselves -- already +branches on the compiler-defined architecture macros, not +CMAKE_SYSTEM_PROCESSOR: + + #if defined(__aarch64__) || defined(__x86_64__) + fht_float(vec, log2_vec_size); + normalize_after_fht(vec, log2_vec_size); + #else + fast_hadamard_transform_simple_impl(vec, log2_vec_size); + #endif + +so on any architecture other than __aarch64__/__x86_64__ -- riscv64 +included -- the portable, scalar fast_hadamard_transform_simple_impl() is +already used at compile time, and the FFHT SIMD kernel's fht_float()/ +fht_double() symbols are never referenced, let alone linked. The +CMakeLists.txt FATAL_ERROR was refusing to configure over a source file +that would never have been needed. Replace it with a STATUS message and +add no FFHT source at all for these architectures; the vendored fht_avx.c/ +fht_neon.c/fht_sse.c/dumb_fht.c files use a different, narrower API +(void dumb_fht(float*, int) vs. fht_float()'s int fht_float(float*, int), +and no fht_double()/_oop equivalents) and were never a drop-in match +regardless of architecture. + +Upstream-Status: To upstream [not yet submitted; the FATAL_ERROR blocks every architecture outside x86_64/aarch64/armv7 even though fast_hadamard_transform.h already has a portable fallback that needs none of them, independent of riscv64] + +Signed-off-by: RISE Project CI +--- + extension/llm/custom_ops/CMakeLists.txt | 12 +++++++++--- + 1 file changed, 9 insertions(+), 3 deletions(-) + +diff --git a/extension/llm/custom_ops/CMakeLists.txt b/extension/llm/custom_ops/CMakeLists.txt +index fa847cb..d9f6e4f 100644 +--- a/extension/llm/custom_ops/CMakeLists.txt ++++ b/extension/llm/custom_ops/CMakeLists.txt +@@ -58,10 +58,16 @@ elseif(CMAKE_SYSTEM_PROCESSOR MATCHES "^(x86_64|AMD64)") + "extension/llm/custom_ops/spinquant/third-party/FFHT/fht_avx.c" + ) + else() ++ # No FFHT SIMD kernel exists for this architecture, but none is required: ++ # fast_hadamard_transform_ffht_impl() (fast_hadamard_transform.h) only ++ # calls fht_float() when __aarch64__ or __x86_64__ is defined, and already ++ # falls back at compile time to the portable ++ # fast_hadamard_transform_simple_impl() everywhere else, so nothing here ++ # links against fht_float/fht_double. + message( +- FATAL_ERROR +- "Unsupported CMAKE_SYSTEM_PROCESSOR ${CMAKE_SYSTEM_PROCESSOR}. (If \ +-32-bit x86, try using fht_avx.c and send a PR if it works!)" ++ STATUS ++ "No FFHT SIMD kernel for CMAKE_SYSTEM_PROCESSOR ${CMAKE_SYSTEM_PROCESSOR}; " ++ "using the portable fast_hadamard_transform_simple_impl fallback." + ) + endif() + +-- +2.34.1 diff --git a/patches/executorch/1.4.1/0006-xnnpack-only-query-cpuinfo-uarchs-once-cpuinfo-is-in.patch b/patches/executorch/1.4.1/0006-xnnpack-only-query-cpuinfo-uarchs-once-cpuinfo-is-in.patch new file mode 100644 index 00000000000..8632bcb6d9a --- /dev/null +++ b/patches/executorch/1.4.1/0006-xnnpack-only-query-cpuinfo-uarchs-once-cpuinfo-is-in.patch @@ -0,0 +1,73 @@ +From 0000000000000000000000000000000000000006 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Fri, 9 Oct 2026 00:00:00 +0000 +Subject: [PATCH 6/7] xnnpack: only query cpuinfo uarchs once cpuinfo is + initialized + +init_hardware_config() checks cpuinfo_initialize() before reading the L1/L2 +data cache sizes, but the loop that fills hardware_config.uarch[] right +after it sits outside that check and calls cpuinfo_get_uarch() +unconditionally. When cpuinfo cannot initialize, cpuinfo_get_uarch() is +fatal: + + Fatal error in cpuinfo: cpuinfo_get_uarch called before cpuinfo is initialized + +and the process aborts. XnnpackBackend's static registration calls +xnn_initialize() while _portable_lib is being loaded, so this kills +`import executorch.extension.pybindings.portable_lib` outright. + +xnn_init_hardware_config() deliberately skips its cpuinfo_initialize() +gate on XNN_ARCH_RISCV and XNN_ARCH_PPC64 (both detect their vector +extensions through getauxval() instead), so on those architectures +init_hardware_config() is reached with cpuinfo uninitialized whenever +cpuinfo cannot parse the host's topology - e.g. a riscv64 kernel that +reports /sys/devices/system/cpu/cpuN/topology/core_id as -1. Only query +the uarchs when cpuinfo_initialize() succeeds and otherwise record +xnn_uarch_unknown, which is what the !XNN_ENABLE_CPUINFO branch already +does. + +Upstream-Status: To upstream [not yet submitted to google/XNNPACK; unchanged on XNNPACK master, affects every architecture whose xnn_init_hardware_config() skips the cpuinfo_initialize() gate (riscv64, ppc64) on hosts where cpuinfo cannot initialize] + +Signed-off-by: RISE Project CI +--- + .../XNNPACK/src/configs/hardware-config.c | 22 ++++++++++++------- + 1 file changed, 14 insertions(+), 8 deletions(-) + +diff --git a/backends/xnnpack/third-party/XNNPACK/src/configs/hardware-config.c b/backends/xnnpack/third-party/XNNPACK/src/configs/hardware-config.c +index da518a1..620c202 100644 +--- a/backends/xnnpack/third-party/XNNPACK/src/configs/hardware-config.c ++++ b/backends/xnnpack/third-party/XNNPACK/src/configs/hardware-config.c +@@ -448,16 +448,22 @@ static void init_hardware_config(void) { + } + } + ++ if (cpuinfo_initialize()) { + #if XNN_MAX_UARCH_TYPES > 1 +- // Print what we think we know about the microarchs. +- xnn_log_info("cpuinfo_get_uarchs_count: %u.", cpuinfo_get_uarchs_count()); +- for (int i = 0; i < cpuinfo_get_uarchs_count(); i++) { +- xnn_log_info("cpu_get_uarch(%i): 0x%x", i, cpuinfo_get_uarch(i)->uarch); +- } ++ // Print what we think we know about the microarchs. ++ xnn_log_info("cpuinfo_get_uarchs_count: %u.", cpuinfo_get_uarchs_count()); ++ for (int i = 0; i < cpuinfo_get_uarchs_count(); i++) { ++ xnn_log_info("cpu_get_uarch(%i): 0x%x", i, cpuinfo_get_uarch(i)->uarch); ++ } + #endif // XNN_MAX_UARCH_TYPES > 1 +- for (size_t i = 0; i < XNN_MAX_UARCH_TYPES; ++i) { +- const struct cpuinfo_uarch_info* uarch = cpuinfo_get_uarch(i); +- hardware_config.uarch[i] = uarch ? cpuinfo_to_xnn_uarch(uarch->uarch) : xnn_uarch_unknown; ++ for (size_t i = 0; i < XNN_MAX_UARCH_TYPES; ++i) { ++ const struct cpuinfo_uarch_info* uarch = cpuinfo_get_uarch(i); ++ hardware_config.uarch[i] = uarch ? cpuinfo_to_xnn_uarch(uarch->uarch) : xnn_uarch_unknown; ++ } ++ } else { ++ for (size_t i = 0; i < XNN_MAX_UARCH_TYPES; ++i) { ++ hardware_config.uarch[i] = xnn_uarch_unknown; ++ } + } + #else + xnn_log_warning("Unable to determine L1/L2 data cache properties."); +-- +2.43.0 + diff --git a/patches/executorch/1.4.1/0007-threadpool-fall-back-to-the-online-CPU-count-when-cp.patch b/patches/executorch/1.4.1/0007-threadpool-fall-back-to-the-online-CPU-count-when-cp.patch new file mode 100644 index 00000000000..d74a1e05fb5 --- /dev/null +++ b/patches/executorch/1.4.1/0007-threadpool-fall-back-to-the-online-CPU-count-when-cp.patch @@ -0,0 +1,63 @@ +From 0000000000000000000000000000000000000007 Mon Sep 17 00:00:00 2001 +From: RISE Project CI +Date: Fri, 9 Oct 2026 00:00:00 +0000 +Subject: [PATCH 7/7] threadpool: fall back to the online CPU count when + cpuinfo fails + +get_threadpool() returns nullptr when cpuinfo_initialize() fails, but +none of its callers handle that: get_pthreadpool() aborts with "Failed to +acquire an instance of ThreadPool!" (reached from XNNCompiler.cpp when +creating every XNNPACK runtime), and parallel_for(), +_threadpool_get_thread_count() and the LLM sdpa op dereference the +result directly. So on any host where cpuinfo cannot initialize - e.g. a +riscv64 kernel that reports /sys/devices/system/cpu/cpuN/topology/core_id +as -1 - running an XNNPACK-delegated or parallelised program crashes. + +cpuinfo is only used here to choose a thread count: pthreadpool itself is +built with PTHREADPOOL_USE_CPUINFO=0 and starts its threads without it. +Size the pool from std::thread::hardware_concurrency() (capped at the +same tsan limit of 63) when cpuinfo is unavailable, instead of leaving +every caller without a threadpool. + +Upstream-Status: To upstream [not yet submitted; unchanged on pytorch/executorch main, affects any host where cpuinfo_initialize() fails, independent of riscv64] + +Signed-off-by: RISE Project CI +--- + extension/threadpool/threadpool.cpp | 13 ++++++++----- + 1 file changed, 8 insertions(+), 5 deletions(-) + +diff --git a/extension/threadpool/threadpool.cpp b/extension/threadpool/threadpool.cpp +index 2845b4a..a3a8c60 100644 +--- a/extension/threadpool/threadpool.cpp ++++ b/extension/threadpool/threadpool.cpp +@@ -11,6 +11,7 @@ + + #include + #include ++#include + + #include + #include +@@ -127,12 +128,14 @@ void ThreadPool::run( + ThreadPool* get_threadpool() { + executorch::runtime::runtime_init(); + +- if (!cpuinfo_initialize()) { +- ET_LOG(Error, "cpuinfo initialization failed"); +- return nullptr; // NOLINT(facebook-hte-NullableReturn) +- } +- + static const int num_threads = ([]() { ++ if (!cpuinfo_initialize()) { ++ // pthreadpool does not need cpuinfo, so size the pool from the online ++ // CPU count instead of leaving every caller without a threadpool. ++ ET_LOG(Info, "cpuinfo initialization failed, using the online CPU count"); ++ const uint32_t online = std::thread::hardware_concurrency(); ++ return std::min(std::max(online, 1u), 63u); ++ } + #if defined(EXECUTORCH_THREADPOOL_USE_ALL_LOGICAL_CORES) + // Use threads=cores. + auto result = cpuinfo_get_processors_count(); +-- +2.43.0 +