diff --git a/.github/workflows/build-implicit.yml b/.github/workflows/build-implicit.yml new file mode 100644 index 00000000000..fb49159917b --- /dev/null +++ b/.github/workflows/build-implicit.yml @@ -0,0 +1,174 @@ +# SPDX-FileCopyrightText: 2026 The RISE Project +# SPDX-License-Identifier: MIT +--- +# This workflow is based on: https://github.com/benfred/implicit/blob/0.7.3/.github/workflows/build.yml +name: Build implicit wheels (riscv64) + +on: + workflow_dispatch: + inputs: + version: + description: 'Version glob to (re)build; empty builds every version of docs/packages/implicit.yaml not released yet' + required: false + default: '' + pull_request: + branches: [main] + paths: + - '.github/workflows/build-implicit.yml' + - 'docs/packages/implicit.yaml' + push: + branches: [main] + paths: + - '.github/workflows/build-implicit.yml' + - 'docs/packages/implicit.yaml' + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + +permissions: + contents: read # to fetch code (actions/checkout) + +env: + MANYLINUX_RISCV64_IMAGE: quay.io/pypa/manylinux_2_39_riscv64 + +jobs: + setup: + uses: $/.github/workflows/_setup.yml + with: + package: implicit + version: ${{ inputs.version }} + + build_wheels: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + name: Build implicit ${{ matrix.version }} ${{ matrix.python }}-manylinux_riscv64 + runs-on: ubuntu-24.04-riscv + timeout-minutes: 90 + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + # Matches upstream's own cp39-cp314 matrix, floored to our cp312. cp314t + # is skipped here too: upstream's own pyproject.toml cibuildwheel `skip` + # list excludes it ("*t-*") and its build.yml matrix never builds it. + python: ["cp312", "cp313", "cp314"] + + env: + IMPLICIT_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout implicit v${{ env.IMPLICIT_VERSION }} + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: benfred/implicit + ref: v${{ env.IMPLICIT_VERSION }} + persist-credentials: false + + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + path: python-wheels + persist-credentials: false + + # libgomp's dynamic and guided schedules segfault on the riscv64 runners + # (riseproject-dev/python-wheels#617, same defect as patches/lightgbm/4.7.0/0002-*); + # static is unaffected. + - name: Patch implicit source + run: git apply python-wheels/patches/implicit/${{ env.IMPLICIT_VERSION }}/00*.patch + + - uses: pypa/cibuildwheel@1828c10ab37f080699c7b81cea34097c684a7074 # v4.2.0 + with: + output-dir: wheelhouse/ + only: ${{ matrix.python }}-manylinux_riscv64 + env: + CIBW_MANYLINUX_RISCV64_IMAGE: ${{ env.MANYLINUX_RISCV64_IMAGE }} + CIBW_ENVIRONMENT: >- + PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + PIP_ONLY_BINARY=numpy,scipy,cython + # Upstream's pyproject.toml sets repair-wheel-command="" globally, only + # to avoid auditwheel vendoring CUDA libraries on manylinux_x86_64 (see + # ci/install_cuda_13.sh, gated to that arch alone). No CUDA toolkit is + # present on riscv64 - CMakeLists.txt's find_package(CUDAToolkit) simply + # skips the GPU extension - so run the normal auditwheel repair here to + # properly vendor libgomp (CMake links OpenMP automatically when found). + CIBW_REPAIR_WHEEL_COMMAND: auditwheel repair -w {dest_dir} {wheel} + CIBW_TEST_REQUIRES: pytest + CIBW_TEST_ENVIRONMENT: >- + PIP_EXTRA_INDEX_URL=https://pypi.riseproject.dev/simple/ + PIP_ONLY_BINARY=numpy,scipy + CIBW_TEST_COMMAND: >- + python -c "import implicit.gpu; assert not implicit.gpu.HAS_CUDA" && + python -m pytest {project}/tests + + - name: Check wheel contents + run: | + python3 - wheelhouse/*.whl <<'EOF' + import sys, zipfile + for path in sys.argv[1:]: + names = zipfile.ZipFile(path).namelist() + assert any(n.endswith(".so") for n in names), (path, names) + licences = {n.rsplit("/", 1)[-1] for n in names + if ".dist-info/licenses/" in n} - {""} + assert licences == {"LICENSE"}, (path, licences) + print(path, "ok") + EOF + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: implicit-${{ env.IMPLICIT_VERSION }}-${{ matrix.python }}-manylinux_riscv64 + path: wheelhouse/*.whl + if-no-files-found: error + + gpl_sources: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + name: Collect GPL sources + runs-on: ubuntu-24.04-riscv + + env: + IMPLICIT_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + # CMake links OpenMP whenever find_package(OpenMP) succeeds (implicit's own + # CMakeLists.txt has no way to disable it), so auditwheel repair above + # vendors the image's libgomp into implicit.libs/. + - uses: ./actions/collect-gpl-sources + with: + image: ${{ env.MANYLINUX_RISCV64_IMAGE }} + packages: gcc + output: gpl-sources.tar + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: implicit-${{ env.IMPLICIT_VERSION }}-gpl-sources + path: gpl-sources.tar + if-no-files-found: error + + publish: + name: Publish implicit ${{ matrix.version }} + needs: [setup, build_wheels, gpl_sources] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + permissions: + contents: write + pull-requests: write + uses: $/.github/workflows/_publish-wheel.yml + secrets: + app-private-key: ${{ secrets.RISEPROJECT_APP_PRIVATE_KEY }} + with: + artifact-pattern: implicit-${{ matrix.version }}-*-manylinux_riscv64 + gpl-sources-artifact: implicit-${{ matrix.version }}-gpl-sources + gpl-sources-description: gcc diff --git a/docs/packages/implicit.yaml b/docs/packages/implicit.yaml new file mode 100644 index 00000000000..09fea32b45e --- /dev/null +++ b/docs/packages/implicit.yaml @@ -0,0 +1,5 @@ +package-name: implicit +source-code: https://github.com/benfred/implicit +license: MIT +versions: +- version: 0.7.3 diff --git a/patches/implicit/0.7.3/0001-use-schedule-static-for-the-dynamic-and-guided-Open.patch b/patches/implicit/0.7.3/0001-use-schedule-static-for-the-dynamic-and-guided-Open.patch new file mode 100644 index 00000000000..74beb80ba7d --- /dev/null +++ b/patches/implicit/0.7.3/0001-use-schedule-static-for-the-dynamic-and-guided-Open.patch @@ -0,0 +1,107 @@ +From a47c062213910e0d38a4c00bec5f63be23b56f21 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 07:55:34 +0000 +Subject: [PATCH] use schedule(static) for the dynamic and guided OpenMP + loops + +Upstream-Status: Inappropriate [works around a libgomp defect on the riscv64 runners, riseproject-dev/python-wheels#617] + +Same libgomp defect as patches/lightgbm/4.7.0/0002-*: the dynamic and +guided work-share schedules fault inside gomp_iter_dynamic_next()/ +gomp_iter_guided_next() on the riscv64 machines these wheels are built +and tested on, while schedule(static) on the same loops is clean. + +cp314-manylinux_riscv64 segfaulted (exit 139) on run 36389716255, +inside tests/recommender_base_test.py::test_fit_non_csr_matrix, at +implicit/cpu/als.py:164 (AlternatingLeastSquares.fit -> the CG solver). +The crash is in the ALS OpenMP core, not the test: that test is simply +the first one whose model.fit() reaches implicit.cpu._als's +prange(..., schedule='dynamic', chunksize=8) loops, so any fit() call +is equally exposed. The backtrace faults inside +implicit.libs/libgomp-*.so.1.0.0, called from implicit/cpu/_als.so, +called back into libgomp - the same shape as the lightgbm case (fault +frame in libgomp itself, sibling frame in the project's compiled +parallel region). + +implicit already uses schedule='static' for the BPR and LMF solvers +(implicit/cpu/bpr.pyx, implicit/cpu/lmf.pyx); only ALS +(implicit/cpu/_als.pyx, 4 loops), the nearest-neighbours all-pairs KNN +(implicit/_nearest_neighbours.pyx, 1 loop) and top-k (implicit/cpu/topk.pyx, +1 guided loop) ask for dynamic/guided. Moving those 6 loops to static +costs load balancing on the (mostly similar-cost) per-user/per-item +iterations and buys a wheel that does not crash. Revert once the +runners' toolchain is fixed. +--- + implicit/_nearest_neighbours.pyx | 2 +- + implicit/cpu/_als.pyx | 8 ++++---- + implicit/cpu/topk.pyx | 2 +- + 3 files changed, 6 insertions(+), 6 deletions(-) + +diff --git a/implicit/_nearest_neighbours.pyx b/implicit/_nearest_neighbours.pyx +index 85f32f0..67b11ae 100644 +--- a/implicit/_nearest_neighbours.pyx ++++ b/implicit/_nearest_neighbours.pyx +@@ -142,7 +142,7 @@ def all_pairs_knn(users, unsigned int K=100, int num_threads=0, show_progress=Tr + topk = new TopK[int, double](K) + + try: +- for i in prange(item_count, schedule='dynamic', chunksize=8): ++ for i in prange(item_count, schedule='static'): + for index1 in range(item_indptr[i], item_indptr[i+1]): + u = item_indices[index1] + w1 = item_data[index1] +diff --git a/implicit/cpu/_als.pyx b/implicit/cpu/_als.pyx +index 630e3f1..0601e02 100644 +--- a/implicit/cpu/_als.pyx ++++ b/implicit/cpu/_als.pyx +@@ -93,7 +93,7 @@ def _least_squares(YtY, integral[:] indptr, integral[:] indices, float[:] data, + A = malloc(sizeof(floating) * factors * factors) + b = malloc(sizeof(floating) * factors) + try: +- for u in prange(users, schedule='dynamic', chunksize=8): ++ for u in prange(users, schedule='static'): + # if we have no items for this user, skip and set to zero + if indptr[u] == indptr[u+1]: + memset(&X[u, 0], 0, sizeof(floating) * factors) +@@ -174,7 +174,7 @@ def _least_squares_cg(integral[:] indptr, integral[:] indices, float[:] data, + p = malloc(sizeof(floating) * N) + r = malloc(sizeof(floating) * N) + try: +- for u in prange(users, schedule='dynamic', chunksize=8): ++ for u in prange(users, schedule='static'): + # start from previous iteration + x = &X[u, 0] + +@@ -274,7 +274,7 @@ def _calculate_loss(Cui, integral[:] indptr, integral[:] indices, float[:] data, + with nogil, parallel(num_threads=num_threads): + r = malloc(sizeof(floating) * N) + try: +- for u in prange(users, schedule='dynamic', chunksize=8): ++ for u in prange(users, schedule='static'): + # calculates (A.dot(Xu) - 2 * b).dot(Xu), without calculating A + temp = 1.0 + symv(b"U", &N, &temp, &YtY[0, 0], &N, &X[u, 0], &one, &zero, r, &one) +@@ -298,7 +298,7 @@ def _calculate_loss(Cui, integral[:] indptr, integral[:] indices, float[:] data, + loss += dot(&N, r, &one, &X[u, 0], &one) + user_norm += dot(&N, &X[u, 0], &one, &X[u, 0], &one) + +- for i in prange(items, schedule='dynamic', chunksize=8): ++ for i in prange(items, schedule='static'): + item_norm += dot(&N, &Y[i, 0], &one, &Y[i, 0], &one) + + finally: +diff --git a/implicit/cpu/topk.pyx b/implicit/cpu/topk.pyx +index 05dca11..35911ad 100644 +--- a/implicit/cpu/topk.pyx ++++ b/implicit/cpu/topk.pyx +@@ -34,7 +34,7 @@ def topk(items, query, int k, item_norms=None, filter_query_items=None, filter_i + + cdef int startidx, endidx, batch + +- for batch in prange(batches, schedule="guided", num_threads=num_threads, nogil=True): ++ for batch in prange(batches, schedule="static", num_threads=num_threads, nogil=True): + startidx = batch * batch_size + endidx = min(startidx + batch_size, query_rows) + with gil: +-- +2.43.0