From 39eb3777fd2f74896c86f7b772bd6d2091891c94 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 07:39:02 +0000 Subject: [PATCH 01/13] ray-haproxy: add riscv64 wheel build Ports HAProxy 2.8.25 for riscv64: ray-haproxy bundles a prebuilt HAProxy binary (statically vendoring OpenSSL and Lua it builds from source, plus PCRE and libxcrypt discovered via ldd) for Ray Serve. The build mirrors ray-project/ray-haproxy's own release.yml, run against manylinux_2_39_riscv64 instead of manylinux2014 (no riscv64 image exists for that profile). Patches add a riscv64 arch branch and switch to PCRE2, since manylinux_2_39_riscv64 (Rocky 10) dropped the legacy PCRE1 package. --- .github/workflows/build-ray-haproxy.yml | 245 ++++++++++++++++++ docs/packages/ray-haproxy.yaml | 5 + ...ist-add-riscv64-and-build-against-PC.patch | 69 +++++ ...vendoring-recognise-riscv64-binaries.patch | 28 ++ ...PARTY_LICENSES-note-PCRE2-on-riscv64.patch | 28 ++ 5 files changed, 375 insertions(+) create mode 100644 .github/workflows/build-ray-haproxy.yml create mode 100644 docs/packages/ray-haproxy.yaml create mode 100644 patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch create mode 100644 patches/ray-haproxy/2.8.25/0002-verify-vendoring-recognise-riscv64-binaries.patch create mode 100644 patches/ray-haproxy/2.8.25/0003-THIRD_PARTY_LICENSES-note-PCRE2-on-riscv64.patch diff --git a/.github/workflows/build-ray-haproxy.yml b/.github/workflows/build-ray-haproxy.yml new file mode 100644 index 00000000000..311d4a08e45 --- /dev/null +++ b/.github/workflows/build-ray-haproxy.yml @@ -0,0 +1,245 @@ +# SPDX-FileCopyrightText: 2026 The RISE Project +# SPDX-License-Identifier: MIT +--- +# This workflow is based on: +# https://github.com/ray-project/ray-haproxy/blob/v2.8.25/.github/workflows/release.yml +name: Build ray-haproxy wheels (riscv64) + +on: + workflow_dispatch: + inputs: + version: + description: 'Version glob to (re)build; empty builds every version of docs/packages/ray-haproxy.yaml not released yet' + required: false + default: '' + pull_request: + branches: [main] + paths: + - '.github/workflows/build-ray-haproxy.yml' + - 'docs/packages/ray-haproxy.yaml' + - 'patches/ray-haproxy/**' + push: + branches: [main] + paths: + - '.github/workflows/build-ray-haproxy.yml' + - 'docs/packages/ray-haproxy.yaml' + - 'patches/ray-haproxy/**' + +concurrency: + group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + cancel-in-progress: true + +permissions: + contents: read # to fetch code (actions/checkout) + +env: + MANYLINUX_RISCV64_IMAGE: quay.io/pypa/manylinux_2_39_riscv64 + +jobs: + setup: + uses: $/.github/workflows/_setup.yml + with: + package: ray-haproxy + version: ${{ inputs.version }} + + build_wheel: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + name: Build ray-haproxy ${{ matrix.version }} py3-none-manylinux_riscv64 + runs-on: ubuntu-24.04-riscv + timeout-minutes: 60 + + env: + RAY_HAPROXY_VERSION: ${{ matrix.version }} + HAPROXY_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout ray-haproxy v${{ matrix.version }} + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: ray-project/ray-haproxy + ref: v${{ env.RAY_HAPROXY_VERSION }} + persist-credentials: false + + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + path: python-wheels + persist-credentials: false + + - name: Apply patches + run: git apply -v python-wheels/patches/ray-haproxy/${{ env.RAY_HAPROXY_VERSION }}/*.patch + + # Upstream's own build step, run against manylinux_2_39_riscv64 instead of + # manylinux2014 (which has no riscv64 image), then packaged the same way + # its release.yml does outside the container. + - name: Build HAProxy and package the wheel + run: | + docker run --rm \ + -v "$(pwd)":/workspace \ + --workdir /workspace \ + -e HAPROXY_VERSION \ + -e OUTPUT_DIR=/workspace/dist \ + "${{ env.MANYLINUX_RISCV64_IMAGE }}" \ + bash -c ' + set -euxo pipefail + ./ci/build/build-haproxy-dist.sh + mkdir -p ray_haproxy/bin/lib + tar -xzf dist/haproxy-linux-riscv64.tar.gz -C ray_haproxy/bin/ + /opt/python/cp312-cp312/bin/python -m pip install -q wheel setuptools + /opt/python/cp312-cp312/bin/python setup.py bdist_wheel --plat-name manylinux_2_39_riscv64 + ' + + - name: Verify the wheel ships the riscv64 binary + run: | + python3 - dist/*.whl <<'EOF' + import sys, zipfile + with zipfile.ZipFile(sys.argv[1]) as zf: + names = zf.namelist() + print("\n".join(names)) + binary = next(n for n in names if n.endswith("ray_haproxy/bin/haproxy")) + header = zf.read(binary)[:20] + assert header[:4] == b"\x7fELF", header + assert header[18] == 0xF3, header # e_machine == EM_RISCV + libs = [n for n in names if "ray_haproxy/bin/lib/" in n and not n.endswith("lib/")] + assert libs, "no vendored shared libraries in the wheel" + licenses = sorted(n.rsplit("/", 1)[-1] for n in names if ".data/data/" in n) + assert licenses == ["LICENSE", "THIRD_PARTY_LICENSES"], licenses + EOF + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: ray-haproxy-${{ env.RAY_HAPROXY_VERSION }}-py3-none-manylinux_riscv64 + path: dist/*.whl + if-no-files-found: error + + test_wheel: + name: Test ray-haproxy ${{ matrix.version }} on Python ${{ matrix.python-version }} + needs: [setup, build_wheel] + if: needs.setup.outputs.versions != '[]' + runs-on: ubuntu-24.04-riscv + timeout-minutes: 30 + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + # Upstream tests 3.9-3.12; trimmed to the interpreters this registry targets. + python-version: ['3.12', '3.13', '3.14'] + + env: + RAY_HAPROXY_VERSION: ${{ matrix.version }} + + steps: + # Checked out beside the workspace root (not into it) so the package's + # own ray_haproxy/ source directory can't shadow the installed wheel + # when the smoke test below imports ray_haproxy (gotcha 187). + - name: Checkout ray-haproxy v${{ matrix.version }} (tests) + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + repository: ray-project/ray-haproxy + ref: v${{ env.RAY_HAPROXY_VERSION }} + path: ray-haproxy-src + persist-credentials: false + + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + path: python-wheels + persist-credentials: false + + - name: Apply patches + run: git -C ray-haproxy-src apply -v ../python-wheels/patches/ray-haproxy/${{ env.RAY_HAPROXY_VERSION }}/*.patch + + - name: Download wheel + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: ray-haproxy-${{ env.RAY_HAPROXY_VERSION }}-py3-none-manylinux_riscv64 + path: wheelhouse + + - name: Install Python + uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1 + with: + python-version: ${{ matrix.python-version }} + activate-environment: true + enable-cache: false + + - name: Install the wheel + run: uv pip install --reinstall --no-index --find-links wheelhouse ray-haproxy + + - name: Run upstream's smoke test + run: | + python -c " + from ray_haproxy import get_haproxy_binary + import subprocess, sys + binary = get_haproxy_binary() + print(f'Python {sys.version}') + print(f'Binary: {binary}') + result = subprocess.run([binary, '-v'], capture_output=True, text=True) + print(result.stdout or result.stderr) + assert result.returncode == 0, f'haproxy -v failed: {result.returncode}' + print('OK') + " + + # Runs upstream's own vendoring checker (RPATH, ldd resolution, ELF + # sanity) directly on real riscv64 hardware, in place of upstream's + # `verify` job, which spins up six distro containers, several of + # which (amazonlinux, rockylinux:9) have no riscv64 image to run. + - name: Run upstream's vendoring verification + run: | + BINARY="$(python -c 'from ray_haproxy import get_haproxy_binary; print(get_haproxy_binary())')" + chmod +x ray-haproxy-src/ci/verify-vendoring.sh + ray-haproxy-src/ci/verify-vendoring.sh "$BINARY" "$(dirname "$BINARY")/lib" + + gpl_sources: + needs: [setup] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + name: Collect GPL sources for ray-haproxy ${{ matrix.version }} + runs-on: ubuntu-24.04-riscv + + env: + RAY_HAPROXY_VERSION: ${{ matrix.version }} + + steps: + - name: Checkout python-wheels + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - uses: ./actions/collect-gpl-sources + with: + image: ${{ env.MANYLINUX_RISCV64_IMAGE }} + packages: gcc + output: gpl-sources.tar + + - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: ray-haproxy-${{ env.RAY_HAPROXY_VERSION }}-gpl-sources + path: gpl-sources.tar + if-no-files-found: error + + publish: + name: Publish ray-haproxy ${{ matrix.version }} + needs: [setup, build_wheel, test_wheel, gpl_sources] + if: needs.setup.outputs.versions != '[]' + strategy: + fail-fast: false + matrix: + version: ${{ fromJSON(needs.setup.outputs.versions) }} + permissions: + contents: write + pull-requests: write + uses: $/.github/workflows/_publish-wheel.yml + secrets: + app-private-key: ${{ secrets.RISEPROJECT_APP_PRIVATE_KEY }} + with: + artifact-pattern: ray-haproxy-${{ matrix.version }}-py3-none-manylinux_riscv64 + gpl-sources-artifact: ray-haproxy-${{ matrix.version }}-gpl-sources + gpl-sources-description: gcc diff --git a/docs/packages/ray-haproxy.yaml b/docs/packages/ray-haproxy.yaml new file mode 100644 index 00000000000..a9ae875bac0 --- /dev/null +++ b/docs/packages/ray-haproxy.yaml @@ -0,0 +1,5 @@ +package-name: ray-haproxy +source-code: https://github.com/ray-project/ray-haproxy +license: GNU General Public License v2 (GPLv2) +versions: +- version: 2.8.25 diff --git a/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch new file mode 100644 index 00000000000..7c83198b1b5 --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch @@ -0,0 +1,69 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: add riscv64 and build against PCRE2 there + +build-haproxy-dist.sh's arch normalisation only knows x86_64/aarch64/arm64, +so it exits with "Unsupported architecture: riscv64" before doing anything: + + Unsupported architecture: riscv64 + +Add a riscv64 case. Its dependency install and HAProxy `make` flags also +assume PCRE1 (`pcre-devel`, `USE_PCRE=1`), which manylinux2014 (CentOS 7) +carries; manylinux_2_39_riscv64 (Rocky 10) dropped the legacy PCRE1 package +and only ships `pcre2-devel`. Branch on the riscv64 arch label to install +`pcre2-devel` and build with `USE_PCRE2=1 USE_PCRE2_JIT=1` instead, leaving +the x86_64/aarch64 codepath untouched. + +Upstream-Status: Inappropriate [manylinux2014 (x86_64/aarch64) still has pcre-devel; this is a difference between manylinux images, not something upstream's existing targets need] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -62,6 +62,7 @@ + x86_64) ARCH_LABEL="x86_64" ;; + aarch64) ARCH_LABEL="arm64" ;; + arm64) ARCH_LABEL="arm64" ;; # macOS (future) ++ riscv64) ARCH_LABEL="riscv64" ;; + *) echo "Unsupported architecture: $ARCH"; exit 1 ;; + esac + +@@ -91,7 +92,14 @@ + # lua-devel — for HAProxy USE_LUA=1 (Lua 5.1 on CentOS 7) + # --------------------------------------------------------------------------- + echo "==> Installing build dependencies" +-yum install -y perl-IPC-Cmd pcre-devel zlib-devel readline-devel 2>/dev/null ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ # manylinux_2_39_riscv64 (Rocky 10) dropped the legacy PCRE1 package; ++ # only pcre2-devel is available there. HAProxy's USE_PCRE2 build option ++ # is the equivalent for that library. ++ yum install -y perl-IPC-Cmd pcre2-devel zlib-devel readline-devel 2>/dev/null ++else ++ yum install -y perl-IPC-Cmd pcre-devel zlib-devel readline-devel 2>/dev/null ++fi + + # --------------------------------------------------------------------------- + # 1. Build OpenSSL from source +@@ -159,13 +167,19 @@ + tar -xzf "$BUILD_DIR/haproxy.tar.gz" -C "$BUILD_DIR" --strip-components=1 + + echo "==> Compiling HAProxy" ++# manylinux_2_39_riscv64 only has pcre2-devel (see the riscv64 branch above), ++# so build against PCRE2 there instead of PCRE1. ++PCRE_MAKE_VARS=(USE_PCRE=1) ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ PCRE_MAKE_VARS=(USE_PCRE2=1 USE_PCRE2_JIT=1) ++fi + make -C "$BUILD_DIR" \ + TARGET=linux-glibc \ + USE_OPENSSL=1 \ + SSL_INC="$DEPS_DIR/include" \ + SSL_LIB="$OPENSSL_LIB_DIR" \ + USE_ZLIB=1 \ +- USE_PCRE=1 \ ++ "${PCRE_MAKE_VARS[@]}" \ + USE_LUA=1 \ + LUA_INC="$DEPS_DIR/include" \ + LUA_LIB="$DEPS_DIR/lib" \ diff --git a/patches/ray-haproxy/2.8.25/0002-verify-vendoring-recognise-riscv64-binaries.patch b/patches/ray-haproxy/2.8.25/0002-verify-vendoring-recognise-riscv64-binaries.patch new file mode 100644 index 00000000000..7a17d711c4e --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0002-verify-vendoring-recognise-riscv64-binaries.patch @@ -0,0 +1,28 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci: verify-vendoring: recognise riscv64 binaries + +The ELF sanity check only matches `file`'s output for x86-64 and aarch64, so +it fails a riscv64 binary that is otherwise fine: + + FAIL: Unexpected binary type: ELF 64-bit LSB pie executable, UCB RISC-V, ... + +Add a branch for `file`'s "RISC-V" architecture string. + +Upstream-Status: Inappropriate [only needed once a riscv64 leg exists to call this script; upstream's own x86_64/aarch64 legs never hit this path] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/verify-vendoring.sh b/ci/verify-vendoring.sh +--- a/ci/verify-vendoring.sh ++++ b/ci/verify-vendoring.sh +@@ -128,6 +128,8 @@ + pass "Binary is ELF 64-bit x86-64" + elif echo "$FILE_TYPE" | grep -q "ELF 64-bit.*aarch64"; then + pass "Binary is ELF 64-bit aarch64" ++elif echo "$FILE_TYPE" | grep -q "ELF 64-bit.*RISC-V"; then ++ pass "Binary is ELF 64-bit RISC-V" + else + fail "Unexpected binary type: $FILE_TYPE" + fi diff --git a/patches/ray-haproxy/2.8.25/0003-THIRD_PARTY_LICENSES-note-PCRE2-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0003-THIRD_PARTY_LICENSES-note-PCRE2-on-riscv64.patch new file mode 100644 index 00000000000..14edc099aa5 --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0003-THIRD_PARTY_LICENSES-note-PCRE2-on-riscv64.patch @@ -0,0 +1,28 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] THIRD_PARTY_LICENSES: note PCRE2 on riscv64 + +The riscv64 build links PCRE2 (previous patch), not the classic PCRE1 this +file's "PCRE" entry names and links to (sourceforge.net/projects/pcre, +pcre.org). Both share the same BSD-style licence text already quoted here, +but point the Source line at PCRE2's own repo too so the notice matches what +the riscv64 wheel actually vendors. + +Upstream-Status: Inappropriate [only the riscv64 build in this fork links PCRE2; upstream's own x86_64/aarch64 wheels still vendor PCRE1] + +Signed-off-by: Ludovic Henry +--- +diff --git a/THIRD_PARTY_LICENSES b/THIRD_PARTY_LICENSES +--- a/THIRD_PARTY_LICENSES ++++ b/THIRD_PARTY_LICENSES +@@ -33,7 +33,9 @@ + PCRE (Perl Compatible Regular Expressions) + ------------------------------------------- + License: BSD License ++Note: the riscv64 wheel vendors PCRE2 instead; same licence. + Source: https://sourceforge.net/projects/pcre/ ++ https://github.com/PCRE2Project/pcre2 (PCRE2, riscv64) + https://www.pcre.org/ + + Redistribution and use in source and binary forms, with or without From 44611dbe48ed344fe5b1e064f5af4eea09c68ee4 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 08:33:33 +0000 Subject: [PATCH 02/13] ray-haproxy: install perl-FindBin on riscv64 CI failed with OpenSSL's Configure script unable to find FindBin.pm: Can't locate FindBin.pm in @INC (you may need to install the FindBin module) ... at .../openssl-3.0.15/Configure line 15. Rocky 10 (manylinux_2_39_riscv64) splits FindBin.pm out of core Perl into perl-FindBin; manylinux2014's older Perl carries it without a separate package, so upstream never needed it. Install it alongside perl-IPC-Cmd in the riscv64 branch. --- ...haproxy-dist-add-riscv64-and-build-against-PC.patch | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch index 7c83198b1b5..5726ae29d69 100644 --- a/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch +++ b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch @@ -15,6 +15,14 @@ and only ships `pcre2-devel`. Branch on the riscv64 arch label to install `pcre2-devel` and build with `USE_PCRE2=1 USE_PCRE2_JIT=1` instead, leaving the x86_64/aarch64 codepath untouched. +Rocky 10 also splits `FindBin.pm` out of its base Perl into `perl-FindBin`, +which OpenSSL's `Configure` needs and manylinux2014's Perl carries without +a separate package: + + Can't locate FindBin.pm in @INC (you may need to install the FindBin module) ... at .../openssl-3.0.15/Configure line 15. + +Install it alongside `perl-IPC-Cmd` on riscv64. + Upstream-Status: Inappropriate [manylinux2014 (x86_64/aarch64) still has pcre-devel; this is a difference between manylinux images, not something upstream's existing targets need] Signed-off-by: Ludovic Henry @@ -39,7 +47,7 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + # manylinux_2_39_riscv64 (Rocky 10) dropped the legacy PCRE1 package; + # only pcre2-devel is available there. HAProxy's USE_PCRE2 build option + # is the equivalent for that library. -+ yum install -y perl-IPC-Cmd pcre2-devel zlib-devel readline-devel 2>/dev/null ++ yum install -y perl-IPC-Cmd perl-FindBin pcre2-devel zlib-devel readline-devel 2>/dev/null +else + yum install -y perl-IPC-Cmd pcre-devel zlib-devel readline-devel 2>/dev/null +fi From 57b87fe15acfdd7974c38547ee700dcc25b53068 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 12:35:50 +0000 Subject: [PATCH 03/13] ray-haproxy: install perl-lib on riscv64 CI got past the FindBin fix (44611dbe48) and now fails on the next `use` line in OpenSSL's Configure script: Can't locate lib.pm in @INC (you may need to install the lib module) ... at .../openssl-3.0.15/Configure line 16. Rocky 10 (manylinux_2_39_riscv64) splits the `lib` pragma out of core Perl into perl-lib the same way it split FindBin.pm into perl-FindBin; manylinux2014's older Perl carries it without a separate package. Install it alongside perl-IPC-Cmd/perl-FindBin in the riscv64 branch. --- ...haproxy-dist-add-riscv64-and-build-against-PC.patch | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch index 5726ae29d69..b6d2832c7b9 100644 --- a/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch +++ b/patches/ray-haproxy/2.8.25/0001-build-haproxy-dist-add-riscv64-and-build-against-PC.patch @@ -21,7 +21,13 @@ a separate package: Can't locate FindBin.pm in @INC (you may need to install the FindBin module) ... at .../openssl-3.0.15/Configure line 15. -Install it alongside `perl-IPC-Cmd` on riscv64. +Rocky 10 splits the `lib` pragma out the same way, into `perl-lib`, and +`Configure` needs that too (it's `use`d two lines later than `FindBin`, so +this only surfaced once the FindBin install let Configure get further): + + Can't locate lib.pm in @INC (you may need to install the lib module) ... at .../openssl-3.0.15/Configure line 16. + +Install both alongside `perl-IPC-Cmd` on riscv64. Upstream-Status: Inappropriate [manylinux2014 (x86_64/aarch64) still has pcre-devel; this is a difference between manylinux images, not something upstream's existing targets need] @@ -47,7 +53,7 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + # manylinux_2_39_riscv64 (Rocky 10) dropped the legacy PCRE1 package; + # only pcre2-devel is available there. HAProxy's USE_PCRE2 build option + # is the equivalent for that library. -+ yum install -y perl-IPC-Cmd perl-FindBin pcre2-devel zlib-devel readline-devel 2>/dev/null ++ yum install -y perl-IPC-Cmd perl-FindBin perl-lib pcre2-devel zlib-devel readline-devel 2>/dev/null +else + yum install -y perl-IPC-Cmd pcre-devel zlib-devel readline-devel 2>/dev/null +fi From 509d6d13175097e646c0042bc041aba3231378b2 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 17:23:51 +0000 Subject: [PATCH 04/13] ray-haproxy: skip ldd's self-exec check on riscv64 build-haproxy-dist.sh's own post-patchelf sanity echo re-runs ldd on the haproxy binary and fails with "not a dynamic executable" on manylinux_2_39_riscv64, even though patchelf's own --print-rpath reads the same file back cleanly one line above and the plain execve further down (haproxy -v) runs it directly. Replace ldd's self-exec check with a static readelf NEEDED-vs-vendored check on riscv64, leaving x86_64/aarch64 untouched. --- ...-skip-ldd-self-exec-check-on-riscv64.patch | 67 +++++++++++++++++++ 1 file changed, 67 insertions(+) create mode 100644 patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch diff --git a/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch new file mode 100644 index 00000000000..e67a46e2c3a --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch @@ -0,0 +1,67 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: skip ldd's self-exec check on riscv64 + +After vendoring and `patchelf --set-rpath`, the script's own sanity echo +re-runs `ldd` on the just-rewritten binary. On manylinux_2_39_riscv64 that +call fails outright: + + Verifying all deps resolve: + not a dynamic executable + +`ldd` resolves a binary's dependencies by re-executing it with +`LD_TRACE_LOADED_OBJECTS=1` set (the dynamic linker intercepts and prints +the deps instead of running the program). That self-exec trick is what +reports "not a dynamic executable" here, even though the ELF itself is +fine: `patchelf --print-rpath` reads the very same file back cleanly one +line above, and vendoring already collected every non-allowlisted `NEEDED` +entry into `$STAGE_DIR/lib` via a plain `ldd` call earlier in this same +script, before `strip`/`patchelf` touched the binary — dependency +resolution was never in question, only this later ldd self-exec on +manylinux_2_39_riscv64's still-young riscv64 glibc port. The plain execve a +few lines further down (`"$STAGE_DIR/haproxy" -v`) runs the identical +binary directly and is unaffected. + +Replace the self-exec check with the equivalent static one on riscv64: +read the binary's `NEEDED` entries with `readelf -d` and confirm each is +either manylinux-allowlisted or vendored, without asking `ldd` to +re-execute the binary at all. + +Upstream-Status: Inappropriate [ldd's self-exec check is fine on x86_64/aarch64; this works around manylinux_2_39_riscv64's own riscv64 glibc/ldd, not something upstream's existing targets hit] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -261,7 +261,28 @@ find "$STAGE_DIR/lib" -name '*.so*' -exec patchelf --set-rpath '$ORIGIN' {} \; + + echo " RPATH: $(patchelf --print-rpath "$STAGE_DIR/haproxy")" + echo " Verifying all deps resolve:" +-LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ # ldd resolves deps by re-executing the binary with ++ # LD_TRACE_LOADED_OBJECTS=1; on manylinux_2_39_riscv64 that self-exec ++ # trick reports "not a dynamic executable" against a binary patchelf has ++ # just rewritten, even though the ELF itself is fine (patchelf's own ++ # --print-rpath above reads it back cleanly, and the plain execve a few ++ # lines down, "$STAGE_DIR/haproxy" -v, runs it directly). Check NEEDED ++ # entries statically instead of asking ldd to self-exec the binary. ++ MISSING="" ++ for needed in $(readelf -d "$STAGE_DIR/haproxy" | grep NEEDED | sed -E 's/.*\[(.*)\]/\1/'); do ++ echo "$needed" | grep -qE "$MANYLINUX_ALLOWLIST" && continue ++ [ -f "$STAGE_DIR/lib/$needed" ] && continue ++ MISSING="$MISSING $needed" ++ done ++ if [ -n "$MISSING" ]; then ++ echo "ERROR: dependencies not allowlisted or vendored:$MISSING" ++ exit 1 ++ fi ++ echo " All NEEDED entries are allowlisted or vendored" ++else ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" ++fi + + # --------------------------------------------------------------------------- + # 5. Sanity checks From 13ec914fb2f6ccb630e0b04abbe97a14eb57855a Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 20:00:44 +0000 Subject: [PATCH 05/13] ray-haproxy: vendor our own OpenSSL on riscv64 The staged haproxy binary segfaulted with no output right when the build script exec'd it, immediately after vendoring reported success. The vendored libssl.so.3/libcrypto.so.3 turned out to be Rocky 10's system OpenSSL (rpm reports 3.5.5), not the OpenSSL 3.0.15 this same script builds from source and links HAProxy against a few lines earlier: ldd's default resolution (no LD_LIBRARY_PATH/RPATH set yet at that point) finds Rocky 10's own libssl.so.3 on manylinux_2_39_riscv64, something manylinux2014 (x86_64/aarch64) never ships, so this only bites riscv64. Point ldd at our own build before vendoring runs so it collects the matching OpenSSL instead of Rocky 10's unrelated one. --- ...st-vendor-our-own-openssl-on-riscv64.patch | 67 +++++++++++++++++++ 1 file changed, 67 insertions(+) create mode 100644 patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch diff --git a/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch new file mode 100644 index 00000000000..ef33473a7b7 --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch @@ -0,0 +1,67 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: vendor our own OpenSSL on riscv64 + +Once the ldd self-exec check (0004) was worked around, the build got one +step further and then died with no output at all right where the staged +binary is first executed: + + Vendoring: libssl.so.3 (/lib64/lp64d/libssl.so.3) + Vendoring: libcrypto.so.3 (/lib64/lp64d/libcrypto.so.3) + ... + All NEEDED entries are allowlisted or vendored + ./ci/build/build-haproxy-dist.sh: line 310: 18549 Segmentation fault LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v + +`/lib64/lp64d/libssl.so.3` is not a build path -- it is Rocky 10's own +system OpenSSL (`rpm -q openssl-libs` on the manylinux_2_39_riscv64 base +reports 3.5.5), not the OpenSSL 3.0.15 this same script builds from source +a few lines above and points HAProxy's SSL_INC/SSL_LIB at. `collect_deps` +resolves each NEEDED entry with a plain `ldd`, which at that point in the +script has no LD_LIBRARY_PATH pointing at the just-built OpenSSL and no +RPATH on the binary yet (patchelf runs later), so it falls through to the +dynamic linker's default search path -- and manylinux_2_39_riscv64 (Rocky +10) already has a libssl.so.3/libcrypto.so.3 there. manylinux2014 (CentOS +7, x86_64/aarch64) never hits this: it has no libssl.so.3 at all, ldd +correctly reports "not found", and the "vendor OpenSSL .so files we built +from source" fallback a few lines down fills in the right copy. On +riscv64 that fallback never fires, because collect_deps has already +placed a file under the same soname (Rocky 10's, not ours) and the +fallback's `[ -f "$STAGE_DIR/lib/$soname" ] && continue` guard skips it. + +The wheel ends up shipping a HAProxy linked against OpenSSL 3.0.15 headers +but running against Rocky 10's unrelated 3.5.5 build -- five minor +releases apart, with its own (RHEL10-target) hwcap-gated codepaths -- and +it now segfaults before HAProxy's own code ever prints anything, which +points at process-startup library-constructor code in that mismatched +libcrypto rather than anything in HAProxy's own `-v` handling. + +Point ldd at the OpenSSL we just built before collect_deps runs, on +riscv64 only, so it vendors that copy instead of falling through to +Rocky 10's. + +Upstream-Status: Inappropriate [only manylinux_2_39_riscv64 (Rocky 10) already ships a colliding libssl.so.3/libcrypto.so.3 on the default loader path; manylinux2014 (x86_64/aarch64) doesn't carry that soname so upstream's existing targets never hit this] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -235,6 +235,17 @@ collect_deps() { + done + } + ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ # manylinux_2_39_riscv64 (Rocky 10) already ships libssl.so.3/libcrypto.so.3 ++ # on the default loader search path; manylinux2014 (CentOS 7) doesn't carry ++ # that soname at all. Without this, ldd's default resolution below (run ++ # with no LD_LIBRARY_PATH yet, since patchelf hasn't set one on the binary) ++ # finds Rocky 10's own OpenSSL build instead of the one just built from ++ # source above, and collect_deps vendors *that* -- a different OpenSSL ++ # build than the one HAProxy was actually configured, compiled and linked ++ # against a few lines up. ++ export LD_LIBRARY_PATH="$OPENSSL_LIB_DIR" ++fi + collect_deps "$STAGE_DIR/haproxy" + + # Also vendor the OpenSSL .so files we built from source — ldd sees them by From 288a1198ef01bdc18d993b93befb85bba01163bb Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 21:09:04 +0000 Subject: [PATCH 06/13] ray-haproxy: TEMP capture strace/core backtrace on riscv64 haproxy -v crash Two inference-only fixes (ldd self-exec skip, vendor our own OpenSSL) have not resolved the segfault at the end of build-haproxy-dist.sh on riscv64. Add a temporary, riscv64-only diagnostic that runs the crashing invocation under strace -f, and attempts a gdb backtrace against any core dump produced, to get a real crash signal from the actual CI runner instead of guessing further. This commit is diagnostic-only and is expected to be superseded by the real fix once the backtrace is in hand. --- ...race-core-backtrace-on-riscv64-crash.patch | 79 +++++++++++++++++++ 1 file changed, 79 insertions(+) create mode 100644 patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch diff --git a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch new file mode 100644 index 00000000000..6b3f25b7b2f --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch @@ -0,0 +1,79 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Mon, 28 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: TEMP capture strace/core backtrace on riscv64 + haproxy -v crash + +Two fixes so far (ldd's self-exec quirk, then vendoring our own OpenSSL +instead of Rocky 10's system one) have not resolved the `haproxy -v` +segfault at the end of the build: + + Verifying all deps resolve: + All NEEDED entries are allowlisted or vendored + ./ci/build/build-haproxy-dist.sh: line 321: 18518 Segmentation fault LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v + +Both fixes were made by inference from log text alone, without access to +riscv64 hardware to actually run the binary and inspect the crash. This +adds a TEMPORARY, riscv64-only diagnostic around that exact invocation: +`ulimit -c unlimited`, run it under `strace -f` (printing the tail of the +trace on failure), and if a core file turns up, run `gdb -batch -ex bt` +(plus registers and the faulting instructions) against it. `gdb`/`strace` +are installed on demand since neither is guaranteed present in +manylinux_2_39_riscv64. + +This patch exists to get a real crash signal from an actual CI run on the +riscv64 runner; it is not a fix and is meant to be removed (or replaced by +the real fix) once that signal is in hand. + +Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -318,7 +318,43 @@ else + echo " Stack canary: NO — consider -fstack-protector-strong" + fi + +-LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ # TEMPORARY diagnostic (not a real fix): capture a backtrace/strace from ++ # this exact invocation on the real riscv64 runner, since this segfaults ++ # with zero output and no prior local riscv64 hardware access has been ++ # available to reproduce it directly. ++ dnf install -y gdb strace 2>/dev/null || true ++ ulimit -c unlimited ++ echo " core_pattern: $(cat /proc/sys/kernel/core_pattern 2>/dev/null || echo unknown)" ++ set +e ++ strace -f -o /tmp/haproxy-v-strace.log \ ++ env LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v ++ STATUS=$? ++ set -e ++ echo " haproxy -v exit status: $STATUS" ++ if [ "$STATUS" -ne 0 ]; then ++ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); last strace lines:" ++ tail -n 80 /tmp/haproxy-v-strace.log 2>/dev/null || echo " (no strace log)" ++ echo "==> DIAGNOSTIC: searching for a core dump" ++ CORE="" ++ for d in . /workspace /tmp "$STAGE_DIR"; do ++ for f in "$d"/core "$d"/core.*; do ++ [ -f "$f" ] && CORE="$f" && break 2 ++ done ++ done ++ if [ -n "$CORE" ]; then ++ echo " found core: $CORE" ++ gdb -batch -ex 'set pagination off' -ex bt -ex 'info registers' \ ++ -ex 'x/20i $pc-40' -ex 'info sharedlibrary' \ ++ "$STAGE_DIR/haproxy" "$CORE" 2>&1 ++ else ++ echo " no core dump found" ++ fi ++ exit "$STATUS" ++ fi ++else ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v ++fi + + # --------------------------------------------------------------------------- + # 6. Package From 058f6396246d66c6bf0145fdc73ea06732d2a99e Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 22:56:13 +0000 Subject: [PATCH 07/13] ray-haproxy: fix TEMP riscv64 diagnostic so it survives locked-down ulimit -c The diagnostic added in 288a1198 called `ulimit -c unlimited` unconditionally before running the crashing `haproxy -v` under strace. On the real riscv64 runner the container's core-dump hard limit is locked to 0, so that call itself failed: ./ci/build/build-haproxy-dist.sh: line 327: ulimit: core file size: cannot modify limit: Operation not permitted and under this script's `set -euxo pipefail`, that nonzero exit aborted the whole script before the strace-wrapped `haproxy -v`, the core-file search, or the gdb backtrace ever ran. No diagnostic signal was captured. Make `ulimit -c unlimited` non-fatal by running it as an `if` condition (which `set -e` does not treat as abort-worthy), logging the hard limit and the outcome instead of relying on it to succeed. Since the hard limit being locked to 0 may be a permanent constraint of this runner/container, not just this run, don't depend on core dumps at all: `strace -f`'s own output (last syscalls, mapped files, and the signal delivery itself) is now the primary signal, and its full log is dumped to the job log whenever the invocation exits nonzero. The core-file search and gdb backtrace stay as a bonus path if a core does happen to appear, but nothing depends on it anymore. Still a TEMPORARY riscv64-only diagnostic, not a fix. Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] --- ...race-core-backtrace-on-riscv64-crash.patch | 54 +++++++++++++++---- 1 file changed, 44 insertions(+), 10 deletions(-) diff --git a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch index 6b3f25b7b2f..eab314edff4 100644 --- a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch +++ b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch @@ -15,12 +15,33 @@ segfault at the end of the build: Both fixes were made by inference from log text alone, without access to riscv64 hardware to actually run the binary and inspect the crash. This adds a TEMPORARY, riscv64-only diagnostic around that exact invocation: -`ulimit -c unlimited`, run it under `strace -f` (printing the tail of the -trace on failure), and if a core file turns up, run `gdb -batch -ex bt` -(plus registers and the faulting instructions) against it. `gdb`/`strace` -are installed on demand since neither is guaranteed present in +run it under `strace -f` (dumping the full trace to the job log on +failure), and if a core file turns up, also run `gdb -batch -ex bt` (plus +registers and the faulting instructions) against it. `gdb`/`strace` are +installed on demand since neither is guaranteed present in manylinux_2_39_riscv64. +The first attempt at this diagnostic never got that far: it called +`ulimit -c unlimited` unconditionally, and under this script's +`set -euxo pipefail`, the runner's core-dump hard limit being locked to 0 +by the outer container runtime made that call itself fail and abort the +whole script before the crashing command, the core-file search, or the +gdb backtrace ever ran: + + ./ci/build/build-haproxy-dist.sh: line 327: ulimit: core file size: cannot modify limit: Operation not permitted + +This revision makes the `ulimit -c unlimited` call non-fatal (it now runs +as an `if` condition, which `set -e` does not treat as abort-worthy) and +logs the hard limit and outcome instead of relying on it. Since core +dumps may simply be unavailable in this runner/container setup — not just +this one run — `strace -f`'s own output (last syscalls, mapped files, and +the signal delivery itself) is now the primary diagnostic signal: the +full trace is written to a file and dumped to the job log whenever the +invocation exits nonzero, independent of ulimit or core files. The +core-file search and gdb backtrace remain as a bonus path if a core does +happen to appear (e.g. via `/proc/sys/kernel/core_pattern`), but nothing +here depends on it. + This patch exists to get a real crash signal from an actual CI run on the riscv64 runner; it is not a fix and is meant to be removed (or replaced by the real fix) once that signal is in hand. @@ -32,7 +53,7 @@ Signed-off-by: Ludovic Henry diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh --- a/ci/build/build-haproxy-dist.sh +++ b/ci/build/build-haproxy-dist.sh -@@ -318,7 +318,43 @@ else +@@ -318,7 +318,56 @@ else echo " Stack canary: NO — consider -fstack-protector-strong" fi @@ -43,7 +64,15 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + # with zero output and no prior local riscv64 hardware access has been + # available to reproduce it directly. + dnf install -y gdb strace 2>/dev/null || true -+ ulimit -c unlimited ++ echo " core-dump hard limit: $(ulimit -Hc 2>/dev/null || echo unknown)" ++ # The container runtime may lock the core-dump hard limit to 0, in which ++ # case raising the soft limit fails with "Operation not permitted". Do ++ # not let that abort the script under `set -e` (it is inside an `if` ++ # condition, so a nonzero status here does not trigger errexit): strace ++ # below is the primary diagnostic signal and does not need core dumps. ++ if ! ulimit -c unlimited 2>&1; then ++ echo " ulimit -c unlimited failed: core dumps are unavailable in this container; strace is the primary signal below" ++ fi + echo " core_pattern: $(cat /proc/sys/kernel/core_pattern 2>/dev/null || echo unknown)" + set +e + strace -f -o /tmp/haproxy-v-strace.log \ @@ -52,9 +81,14 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + set -e + echo " haproxy -v exit status: $STATUS" + if [ "$STATUS" -ne 0 ]; then -+ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); last strace lines:" -+ tail -n 80 /tmp/haproxy-v-strace.log 2>/dev/null || echo " (no strace log)" -+ echo "==> DIAGNOSTIC: searching for a core dump" ++ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); full strace log:" ++ if [ -s /tmp/haproxy-v-strace.log ]; then ++ wc -l /tmp/haproxy-v-strace.log ++ cat /tmp/haproxy-v-strace.log ++ else ++ echo " (no strace log, or empty — strace itself may have failed to run)" ++ fi ++ echo "==> DIAGNOSTIC: searching for a core dump (bonus signal only; not required)" + CORE="" + for d in . /workspace /tmp "$STAGE_DIR"; do + for f in "$d"/core "$d"/core.*; do @@ -67,7 +101,7 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + -ex 'x/20i $pc-40' -ex 'info sharedlibrary' \ + "$STAGE_DIR/haproxy" "$CORE" 2>&1 + else -+ echo " no core dump found" ++ echo " no core dump found (expected if core dumps are unavailable here)" + fi + exit "$STATUS" + fi From 39a3a653a26110be8df8a5e5f87d117593f9325e Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Mon, 28 Sep 2026 23:30:20 +0000 Subject: [PATCH 08/13] ray-haproxy: fix TEMP riscv64 diagnostic misreporting strace-missing as haproxy's exit code The diagnostic added in 288a1198/058f6396 wrapped the crashing haproxy -v invocation in `strace -f -o ... env ...` unconditionally and took STATUS from that wrapped command's exit code. The fixed diagnostic just ran for the first time and reported real signal: haproxy -v exit status: 127 ./ci/build/build-haproxy-dist.sh: line 338: strace: command not found Exit 127 is bash's own "command not found" status, not haproxy's. `dnf install -y gdb strace 2>/dev/null || true` silently left no working `strace` on PATH, so the shell failed to exec `strace` itself before haproxy -v ever ran (wrapped or otherwise). This diagnostic run produced zero information about the original segfault. Restructure so a missing diagnostic tool can't be mistaken for the thing being diagnosed: run the plain, unwrapped invocation first and on its own so STATUS is always its real exit code, then gather diagnostics afterwards and independently of STATUS -- LD_DEBUG=all (glibc's own dynamic-linker trace, built into every glibc so it can't hit this same "tool missing" failure mode), a guarded `ldd`, and `strace` only when `command -v strace` actually finds it on PATH. Still a TEMPORARY riscv64-only diagnostic, not a fix. --- ...race-core-backtrace-on-riscv64-crash.patch | 134 +++++++++++------- 1 file changed, 84 insertions(+), 50 deletions(-) diff --git a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch index eab314edff4..3795b7ae7ea 100644 --- a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch +++ b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch @@ -14,37 +14,38 @@ segfault at the end of the build: Both fixes were made by inference from log text alone, without access to riscv64 hardware to actually run the binary and inspect the crash. This -adds a TEMPORARY, riscv64-only diagnostic around that exact invocation: -run it under `strace -f` (dumping the full trace to the job log on -failure), and if a core file turns up, also run `gdb -batch -ex bt` (plus -registers and the faulting instructions) against it. `gdb`/`strace` are -installed on demand since neither is guaranteed present in -manylinux_2_39_riscv64. +adds a TEMPORARY, riscv64-only diagnostic around that exact invocation. -The first attempt at this diagnostic never got that far: it called -`ulimit -c unlimited` unconditionally, and under this script's -`set -euxo pipefail`, the runner's core-dump hard limit being locked to 0 -by the outer container runtime made that call itself fail and abort the -whole script before the crashing command, the core-file search, or the -gdb backtrace ever ran: +The first revision of this diagnostic (which wrapped the invocation in +`strace -f -o ... env ...` unconditionally, deciding STATUS from that +wrapped command's own exit code) got real CI signal for the first time, +but it wasn't signal about HAProxy at all: - ./ci/build/build-haproxy-dist.sh: line 327: ulimit: core file size: cannot modify limit: Operation not permitted + haproxy -v exit status: 127 + ==> DIAGNOSTIC: haproxy -v failed (status 127); full strace log: + ./ci/build/build-haproxy-dist.sh: line 338: strace: command not found -This revision makes the `ulimit -c unlimited` call non-fatal (it now runs -as an `if` condition, which `set -e` does not treat as abort-worthy) and -logs the hard limit and outcome instead of relying on it. Since core -dumps may simply be unavailable in this runner/container setup — not just -this one run — `strace -f`'s own output (last syscalls, mapped files, and -the signal delivery itself) is now the primary diagnostic signal: the -full trace is written to a file and dumped to the job log whenever the -invocation exits nonzero, independent of ulimit or core files. The -core-file search and gdb backtrace remain as a bonus path if a core does -happen to appear (e.g. via `/proc/sys/kernel/core_pattern`), but nothing -here depends on it. +Exit 127 is bash's own "command not found" status. `dnf install -y gdb +strace 2>/dev/null || true` had silently not left a working `strace` on +PATH (redirected stderr and `|| true` hid whatever dnf actually did), so +the following `strace -f -o ... env LD_LIBRARY_PATH=... "$STAGE_DIR/haproxy" +-v` line failed at the shell's attempt to exec `strace` itself -- HAProxy +was never invoked, wrapped or otherwise, and nothing here says anything +about the original segfault yet. -This patch exists to get a real crash signal from an actual CI run on the -riscv64 runner; it is not a fix and is meant to be removed (or replaced by -the real fix) once that signal is in hand. +Restructure so a missing diagnostic tool can't be mistaken for the thing +being diagnosed: + + - Run the plain, unwrapped invocation first and on its own; STATUS is + always its real exit code, never a wrapper's. + - Only if that fails, gather diagnostics afterwards, independently of + STATUS: LD_DEBUG=all (glibc's own dynamic-linker trace, built into + every glibc, so unlike strace it needs no package install and can't + hit this same "tool not on PATH" failure mode), a `set +e`-guarded + `ldd`, and `strace` only when `command -v strace` actually finds it + on PATH after the install attempt. + +Still a TEMPORARY riscv64-only diagnostic, not a fix. Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] @@ -53,41 +54,74 @@ Signed-off-by: Ludovic Henry diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh --- a/ci/build/build-haproxy-dist.sh +++ b/ci/build/build-haproxy-dist.sh -@@ -318,7 +318,56 @@ else +@@ -318,7 +318,89 @@ else echo " Stack canary: NO — consider -fstack-protector-strong" fi -LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v +if [ "$ARCH_LABEL" = "riscv64" ]; then -+ # TEMPORARY diagnostic (not a real fix): capture a backtrace/strace from -+ # this exact invocation on the real riscv64 runner, since this segfaults -+ # with zero output and no prior local riscv64 hardware access has been -+ # available to reproduce it directly. -+ dnf install -y gdb strace 2>/dev/null || true -+ echo " core-dump hard limit: $(ulimit -Hc 2>/dev/null || echo unknown)" -+ # The container runtime may lock the core-dump hard limit to 0, in which -+ # case raising the soft limit fails with "Operation not permitted". Do -+ # not let that abort the script under `set -e` (it is inside an `if` -+ # condition, so a nonzero status here does not trigger errexit): strace -+ # below is the primary diagnostic signal and does not need core dumps. -+ if ! ulimit -c unlimited 2>&1; then -+ echo " ulimit -c unlimited failed: core dumps are unavailable in this container; strace is the primary signal below" -+ fi -+ echo " core_pattern: $(cat /proc/sys/kernel/core_pattern 2>/dev/null || echo unknown)" ++ # TEMPORARY diagnostic (not a real fix): capture real signal from this ++ # exact invocation on the riscv64 runner. STATUS always reflects the ++ # plain, unwrapped invocation below, run first and on its own -- any ++ # diagnostic tooling (strace, gdb) is gathered afterwards, independently, ++ # and never decides STATUS, so a diagnostic tool that fails to run can't ++ # be mistaken for the thing being diagnosed. + set +e -+ strace -f -o /tmp/haproxy-v-strace.log \ -+ env LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v + STATUS=$? + set -e + echo " haproxy -v exit status: $STATUS" + if [ "$STATUS" -ne 0 ]; then -+ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); full strace log:" -+ if [ -s /tmp/haproxy-v-strace.log ]; then -+ wc -l /tmp/haproxy-v-strace.log -+ cat /tmp/haproxy-v-strace.log ++ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); gathering diagnostics" ++ dnf install -y gdb strace 2>/dev/null || true ++ echo " core-dump hard limit: $(ulimit -Hc 2>/dev/null || echo unknown)" ++ # The container runtime may lock the core-dump hard limit to 0, in ++ # which case raising the soft limit fails with "Operation not ++ # permitted". Do not let that abort the script under `set -e` (it is ++ # inside an `if` condition, so a nonzero status here does not trigger ++ # errexit). ++ if ! ulimit -c unlimited 2>&1; then ++ echo " ulimit -c unlimited failed: core dumps are unavailable in this container" ++ fi ++ echo " core_pattern: $(cat /proc/sys/kernel/core_pattern 2>/dev/null || echo unknown)" ++ ++ echo "==> DIAGNOSTIC: LD_DEBUG=all trace (glibc's own dynamic-linker trace;" ++ echo " built into every glibc, unlike strace, so it needs no package install):" ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all "$STAGE_DIR/haproxy" -v \ ++ > /tmp/haproxy-v-lddebug.log 2>&1 ++ echo " (LD_DEBUG re-run exit status: $?, informational only)" ++ if [ -s /tmp/haproxy-v-lddebug.log ]; then ++ wc -l /tmp/haproxy-v-lddebug.log ++ cat /tmp/haproxy-v-lddebug.log ++ else ++ echo " (empty LD_DEBUG log)" ++ fi ++ ++ echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" ++ echo " executable\" on this riscv64 ld.so per the self-exec quirk above):" ++ set +e ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" ++ echo " ldd exit: $?" ++ set -e ++ ++ if command -v strace >/dev/null 2>&1; then ++ echo "==> DIAGNOSTIC: strace -f trace:" ++ set +e ++ strace -f -o /tmp/haproxy-v-strace.log \ ++ env LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v ++ set -e ++ if [ -s /tmp/haproxy-v-strace.log ]; then ++ wc -l /tmp/haproxy-v-strace.log ++ cat /tmp/haproxy-v-strace.log ++ else ++ echo " (empty strace log)" ++ fi + else -+ echo " (no strace log, or empty — strace itself may have failed to run)" ++ echo "==> DIAGNOSTIC: strace not available after install attempt (dnf install" ++ echo " may have failed, e.g. no network/repo access in this container);" ++ echo " skipping strace, LD_DEBUG=all above is the primary signal" + fi ++ + echo "==> DIAGNOSTIC: searching for a core dump (bonus signal only; not required)" + CORE="" + for d in . /workspace /tmp "$STAGE_DIR"; do From 9f282d781aae01ca51744fe3aa04a080a9d8af2b Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Tue, 29 Sep 2026 02:41:07 +0000 Subject: [PATCH 09/13] ray-haproxy: fix TEMP riscv64 diagnostic swallowing its own LD_DEBUG crash haproxy -v really does segfault on riscv64 (status 139), plain and under LD_DEBUG=all alike -- confirmed by the last CI run. But the LD_DEBUG=all re-run was a bare command under `set -e`, unlike the plain invocation above it which is guarded with `set +e` / capture-status / `set -e`. Its own segfault therefore aborted the script immediately, before the following echo/cat of /tmp/haproxy-v-lddebug.log could run -- the trace glibc's dynamic linker had already written to that file (unbuffered, as it goes) never reached the job log. Guard the LD_DEBUG=all re-run the same way the plain invocation is guarded, and print the log unconditionally afterwards regardless of the re-run's own exit status, tailed to the last 500 lines since the trace can be large and the crash-adjacent lines at the end are what matter. --- ...race-core-backtrace-on-riscv64-crash.patch | 31 +++++++++++++++++-- 1 file changed, 28 insertions(+), 3 deletions(-) diff --git a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch index 3795b7ae7ea..934f7334e5b 100644 --- a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch +++ b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch @@ -45,6 +45,25 @@ being diagnosed: `ldd`, and `strace` only when `command -v strace` actually finds it on PATH after the install attempt. +That revision ran on riscv64 CI for the first time and got the real +crash: `haproxy -v` genuinely segfaults, plain and under `LD_DEBUG=all` +alike. But the `LD_DEBUG=all` re-run was never wrapped in the same +`set +e` / capture-status / `set -e` pattern the plain invocation above +it uses -- it was a bare command under `set -e`, so its own segfault +(exit 139) aborted the script right there, before the following +`echo`/`cat` could run: + + ==> DIAGNOSTIC: LD_DEBUG=all trace (...) + ##[error]Process completed with exit code 139. + +The trace glibc's dynamic linker had already written to +`/tmp/haproxy-v-lddebug.log` (unbuffered, as it goes, not just on clean +exit) never made it into the job log. Guard that re-run the same way the +plain invocation is guarded, and print the log's tail unconditionally +afterwards regardless of the re-run's own exit status -- capped to the +last 500 lines since a full `LD_DEBUG=all` trace can be large and the +crash-adjacent lines at the end are what matter. + Still a TEMPORARY riscv64-only diagnostic, not a fix. Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] @@ -54,7 +73,7 @@ Signed-off-by: Ludovic Henry diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh --- a/ci/build/build-haproxy-dist.sh +++ b/ci/build/build-haproxy-dist.sh -@@ -318,7 +318,89 @@ else +@@ -318,7 +318,95 @@ else echo " Stack canary: NO — consider -fstack-protector-strong" fi @@ -87,12 +106,18 @@ diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh + + echo "==> DIAGNOSTIC: LD_DEBUG=all trace (glibc's own dynamic-linker trace;" + echo " built into every glibc, unlike strace, so it needs no package install):" ++ set +e + LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all "$STAGE_DIR/haproxy" -v \ + > /tmp/haproxy-v-lddebug.log 2>&1 -+ echo " (LD_DEBUG re-run exit status: $?, informational only)" ++ LDDEBUG_STATUS=$? ++ set -e ++ echo " LD_DEBUG re-run exit status: $LDDEBUG_STATUS (informational only)" + if [ -s /tmp/haproxy-v-lddebug.log ]; then + wc -l /tmp/haproxy-v-lddebug.log -+ cat /tmp/haproxy-v-lddebug.log ++ echo " last 500 lines (ld.so writes this trace to fd 2 unbuffered as it" ++ echo " runs, so the crash-adjacent lines survive even though the process" ++ echo " never reached a clean exit):" ++ tail -n 500 /tmp/haproxy-v-lddebug.log + else + echo " (empty LD_DEBUG log)" + fi From 55757461dd13ac8d4b46d17bc93b9c5acba1276d Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Tue, 29 Sep 2026 04:05:16 +0000 Subject: [PATCH 10/13] ray-haproxy: TEMP probe LD_DEBUG_OUTPUT on riscv64 haproxy -v crash The previous diagnostic round finally captured a clean, correctly-guarded LD_DEBUG=all re-run, and it came back completely empty. glibc writes that trace via direct, unbuffered write(2) calls rather than through stdio, so the existing whole-process stdout/stderr redirect isn't expected to be swallowing it. Add a second, independent re-run using LD_DEBUG_OUTPUT, which glibc opens and writes to directly by path, to tell a buffering artifact in this harness apart from a crash too early for ld.so's own debug-print code to run at all. Also confirmed from the build flags: no CPU= or ARCH= is passed to HAProxy's make, HAProxy's own Makefile defaults for TARGET=linux-glibc add no -march/-mcpu on an unset CPU/ARCH, and a leaked -march=native would fail to compile outright on this toolchain (see build-pyscipopt.yml) rather than merely segfault at runtime -- no evidence of an ISA mismatch from the build flags themselves. Still a TEMPORARY riscv64-only diagnostic, not a fix. --- ...EMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch | 103 ++++++++++++++++++ 1 file changed, 103 insertions(+) create mode 100644 patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch diff --git a/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch new file mode 100644 index 00000000000..d5f125696a9 --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch @@ -0,0 +1,103 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Tue, 29 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: TEMP probe LD_DEBUG_OUTPUT on riscv64 haproxy -v + crash + +The previous diagnostic round finally captured a clean, correctly-guarded +LD_DEBUG=all re-run, and it came back completely empty: + + ==> DIAGNOSTIC: LD_DEBUG=all trace (...) + LD_DEBUG re-run exit status: 139 (informational only) + (empty LD_DEBUG log) + +Two things about this are worth separating out before reading anything +else into it: + +- glibc's dynamic linker writes the LD_DEBUG trace with direct, unbuffered + write(2) calls to the target fd, not through C stdio -- so redirecting + the whole process's stdout/stderr to a regular file (`> file 2>&1`, as + the existing re-run does) is not expected to introduce the kind of + userspace buffering that would swallow this specific trace even if the + process crashes moments later. +- The two prior fixes to this same diagnostic (exit-127-from-missing- + strace, then the unguarded LD_DEBUG re-run aborting the script under + `set -e` before its own log could be printed) both turned out to be bugs + in the diagnostic harness itself, not in HAProxy -- so it is worth + double-checking this empty log isn't a third instance of the same + pattern before treating it as real signal about the crash. + +Add a second, independent re-run using LD_DEBUG_OUTPUT, which glibc opens +and writes to directly by path rather than inheriting fd 1/2 from the +shell at all. If that log also comes back empty, the two independent +sinks agreeing rules out this specific harness's redirection as the +explanation, and points instead at the crash happening before ld.so's own +code starts running (e.g. in the kernel's ELF-loading/entry trampoline, +ahead of the interpreter's first debug-print call). + +Separately: the `readelf -d`-based NEEDED check earlier in this script +(riscv64's replacement for ldd's self-exec check, which reports "not a +dynamic executable" against this exact binary on manylinux_2_39_riscv64 +per the commit that added it) has been confirmed passing again on this +build, and the same platform ldd has, in this round, additionally been +seen emitting a symbol-version warning about an unrelated system library +(`libcurl.so.4` wanting a newer OPENSSL_3.2.0 than our vendored 3.0.x +libssl provides) while walking dependencies it wasn't asked to vendor -- +further evidence this image's riscv64 ldd is doing something unreliable +of its own accord, not that the haproxy binary it's being pointed at is +malformed. + +No compiler flag in this script, in HAProxy's own Makefile defaults for +TARGET=linux-glibc, or in the riscv64 branches added by earlier patches +here selects a CPU/ISA baseline (no CPU= or ARCH= is passed to HAProxy's +`make`, and HAProxy's default CPU_CFLAGS/ARCH_FLAGS for an unset CPU/ARCH +add no -march/-mcpu at all). A leaked `-march=native` would also fail to +compile outright on this toolchain rather than produce a binary that +merely segfaults at runtime, per gcc's behaviour on this same +manylinux_2_39_riscv64 image (see build-pyscipopt.yml, which strips +`-march=native` from a vendored dependency's build for exactly this +reason). Still no evidence of an ISA mismatch from the build flags; not +changing anything CPU/ISA-related in this round. + +Still a TEMPORARY riscv64-only diagnostic, not a fix. + +Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -359,7 +359,31 @@ if [ "$ARCH_LABEL" = "riscv64" ]; then + echo " never reached a clean exit):" + tail -n 500 /tmp/haproxy-v-lddebug.log + else +- echo " (empty LD_DEBUG log)" ++ echo " (empty LD_DEBUG log; re-running with LD_DEBUG_OUTPUT below, which" ++ echo " glibc writes through a different path than the shell's own" ++ echo " redirect of the whole process's stdout/stderr)" ++ fi ++ ++ echo "==> DIAGNOSTIC: LD_DEBUG=all + LD_DEBUG_OUTPUT trace (glibc opens this" ++ echo " file itself and writes to it directly, independently of whatever" ++ echo " the shell did to fd 1/2 above -- an empty log here too would mean" ++ echo " the crash happens before ld.so's own code starts running at all," ++ echo " e.g. in the kernel's ELF-loading/entry trampoline):" ++ rm -f /tmp/haproxy-v-lddebug-output.* ++ set +e ++ LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all \ ++ LD_DEBUG_OUTPUT=/tmp/haproxy-v-lddebug-output \ ++ "$STAGE_DIR/haproxy" -v > /dev/null 2>&1 ++ LDDEBUG_OUTPUT_STATUS=$? ++ set -e ++ echo " LD_DEBUG_OUTPUT re-run exit status: $LDDEBUG_OUTPUT_STATUS (informational only)" ++ LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)" ++ if [ -n "$LDDEBUG_OUTPUT_FILE" ] && [ -s "$LDDEBUG_OUTPUT_FILE" ]; then ++ echo " $LDDEBUG_OUTPUT_FILE:" ++ wc -l "$LDDEBUG_OUTPUT_FILE" ++ tail -n 500 "$LDDEBUG_OUTPUT_FILE" ++ else ++ echo " (empty or missing LD_DEBUG_OUTPUT file too)" + fi + + echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" From a1d8e3726c6cb7b18aa71e142624d575e0d526a6 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Tue, 29 Sep 2026 04:40:16 +0000 Subject: [PATCH 11/13] ray-haproxy: TEMP print and un-break LD_DEBUG_OUTPUT diagnostic The LD_DEBUG_OUTPUT re-run finally ran cleanly last round but the log still cut off right after its "exit status: 139 (informational only)" line, with the job then failing at exit code 2 -- two bugs in the diagnostic harness itself, not new signal about the crash: - glibc appends the PID to LD_DEBUG_OUTPUT's path (e.g. "/tmp/haproxy-v-lddebug-output."), so locating the file needs a glob. - The glob was done via `ls ... | head -n1`, and this script inherits `set -euxo pipefail` from its `bash -c` wrapper (bash propagates SHELLOPTS to child scripts). With no file matching, `ls` exits 2 and, under pipefail, that becomes the pipeline's own status even though `head` succeeds -- which errexit then treats as a failed assignment and kills the script before the following `if` ever runs. Replace it with a bare glob `for` loop (no pipeline to fail under pipefail) that prints every matching file's content, and says so explicitly when no file matches at all, distinct from a matched file that's merely empty. Still a TEMPORARY riscv64-only diagnostic, not a fix. --- ...P-print-and-un-break-LD_DEBUG_OUTPUT.patch | 105 ++++++++++++++++++ 1 file changed, 105 insertions(+) create mode 100644 patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch diff --git a/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch b/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch new file mode 100644 index 00000000000..9780c2534c8 --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch @@ -0,0 +1,105 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Tue, 29 Sep 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: TEMP print and un-break LD_DEBUG_OUTPUT diagnostic + +The previous round finally got the LD_DEBUG_OUTPUT re-run itself running +cleanly (no step-killing crash, a correctly captured "exit status: 139 +(informational only)"), but the CI log still stops dead right after that +line: + + LD_DEBUG_OUTPUT re-run exit status: 139 (informational only) + ##[error]Process completed with exit code 2. + +Nothing after it ever reads or prints the file glibc actually wrote. Two +separate bugs, both in the diagnostic harness itself: + +- glibc's dynamic linker does not write LD_DEBUG_OUTPUT= to that + exact path -- it appends the process's PID, producing + "." (e.g. "/tmp/haproxy-v-lddebug-output.18524"), and the + PID isn't known from the shell's side ahead of the crash. The line that + was meant to locate that file -- + `LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)"` + -- has to glob for it, which it does, but through a pipeline. +- This script inherits `set -euxo pipefail` from the `bash -c` wrapper + that invokes it (via bash's automatic SHELLOPTS propagation to child + bash processes), so pipefail is live here even though this script never + sets it itself. With no file matching the glob, the unquoted pattern is + passed to `ls` literally, `ls` exits 2 ("No such file or directory", + redirected to /dev/null), and with pipefail that pipeline's *overall* + status is `ls`'s 2 even though the trailing `head -n1` exits 0. That + nonzero status lands on a plain assignment, which errexit does not + forgive, so the script dies right there with exit 2 -- before the `if` + a few lines down ever runs to print anything. This is the same shape of + self-inflicted harness bug as the previous two rounds (missing strace + reported as haproxy's own exit code; an unguarded re-run aborting the + script under `set -e` before its own log could be printed), not new + information about the crash. + +Replace the `ls | head` pipeline with a bare glob `for` loop: with no +match, bash leaves the pattern unexpanded and `[ -f "$f" ]` just reports +false, so there is no pipeline left to fail under pipefail. Print every +matching file's content (there should only ever be one, but loop instead +of assuming), and when nothing matches, say so explicitly and distinctly +from a matched-but-empty file, since "no file at all" is itself the +signal that the crash happens before ld.so's own code -- including its +LD_DEBUG_OUTPUT file open -- ever runs. + +Still a TEMPORARY riscv64-only diagnostic, not a fix. + +Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -369,21 +369,39 @@ if [ "$ARCH_LABEL" = "riscv64" ]; then + echo " the shell did to fd 1/2 above -- an empty log here too would mean" + echo " the crash happens before ld.so's own code starts running at all," + echo " e.g. in the kernel's ELF-loading/entry trampoline):" +- rm -f /tmp/haproxy-v-lddebug-output.* ++ LDDEBUG_OUTPUT_PATH=/tmp/haproxy-v-lddebug-output ++ rm -f "${LDDEBUG_OUTPUT_PATH}".* + set +e + LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all \ +- LD_DEBUG_OUTPUT=/tmp/haproxy-v-lddebug-output \ ++ LD_DEBUG_OUTPUT="$LDDEBUG_OUTPUT_PATH" \ + "$STAGE_DIR/haproxy" -v > /dev/null 2>&1 + LDDEBUG_OUTPUT_STATUS=$? + set -e + echo " LD_DEBUG_OUTPUT re-run exit status: $LDDEBUG_OUTPUT_STATUS (informational only)" +- LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)" +- if [ -n "$LDDEBUG_OUTPUT_FILE" ] && [ -s "$LDDEBUG_OUTPUT_FILE" ]; then +- echo " $LDDEBUG_OUTPUT_FILE:" +- wc -l "$LDDEBUG_OUTPUT_FILE" +- tail -n 500 "$LDDEBUG_OUTPUT_FILE" +- else +- echo " (empty or missing LD_DEBUG_OUTPUT file too)" ++ # glibc's dynamic linker appends the process's PID to LD_DEBUG_OUTPUT ++ # rather than writing the exact path given (so the file is e.g. ++ # "${LDDEBUG_OUTPUT_PATH}.18524", not "${LDDEBUG_OUTPUT_PATH}" itself), ++ # and the PID isn't known from the shell's side ahead of the crash. ++ # Glob for it instead of piping `ls` through `head`: with this ++ # script's inherited `pipefail`, `ls nomatch 2>/dev/null | head -n1` ++ # would itself exit non-zero (ls's own status, even though head ++ # succeeds) and abort the script under `set -e` before ever printing ++ # anything -- which is exactly what silently ate this section last ++ # run. A bare `for` glob has no such pipeline to fail: with no match ++ # bash leaves the pattern unexpanded and `[ -f "$f" ]` just reports ++ # false. ++ LDDEBUG_OUTPUT_FOUND=0 ++ for f in "${LDDEBUG_OUTPUT_PATH}".*; do ++ if [ -f "$f" ]; then ++ LDDEBUG_OUTPUT_FOUND=1 ++ echo " LD_DEBUG_OUTPUT file: $f" ++ wc -l "$f" ++ tail -n 500 "$f" ++ fi ++ done ++ if [ "$LDDEBUG_OUTPUT_FOUND" -eq 0 ]; then ++ echo " no LD_DEBUG_OUTPUT file was created at all (not even an empty" ++ echo " one) -- distinct from a file existing but being empty" + fi + + echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" From 87dbc5a83f0d90f18928bb44eb3ea1042b0632b8 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Thu, 1 Oct 2026 17:52:57 +0000 Subject: [PATCH 12/13] ray-haproxy: link haproxy above mmap_min_addr on riscv64 The `haproxy -v` segfault was never about OpenSSL. HAProxy links a non-PIE executable, riscv64's default link base is 0x10000 (= the runners' vm.mmap_min_addr), and `patchelf --set-rpath` prepends a program-header segment below that base, so execve kills the binary before ld.so runs (gotcha 595). That explains every earlier symptom: the empty LD_DEBUG trace, the never-created LD_DEBUG_OUTPUT file, and ldd's "not a dynamic executable" on the patchelf'd binary. - New 0004: pass -Wl,-Ttext-segment=0x200000 via HAProxy's ADDLIB hook on riscv64. - Drop the old 0004 (skip ldd's post-patchelf check): ldd failed because of the same exec crash, so upstream's plain check is restored. - Rework 0005 to scope LD_LIBRARY_PATH to the collect_deps ldd walk. Exported script-wide it reached binutils' readelf, which links libcurl via libdebuginfod and failed to start against OpenSSL 3.0.15 ("OPENSSL_3.2.0 not found ... libcurl.so.4"), silently disabling the hardening checks. haproxy itself never links libcurl. - Drop the TEMP diagnostic patches 0006-0008. - Collect libxcrypt's source in gpl_sources: libcrypt.so.2 (LGPL-2.1) is vendored from the image. --- .github/workflows/build-ray-haproxy.yml | 4 +- ...-link-above-mmap_min_addr-on-riscv64.patch | 57 ++++++ ...-skip-ldd-self-exec-check-on-riscv64.patch | 67 ------- ...st-vendor-our-own-openssl-on-riscv64.patch | 77 ++++---- ...race-core-backtrace-on-riscv64-crash.patch | 172 ------------------ ...EMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch | 103 ----------- ...P-print-and-un-break-LD_DEBUG_OUTPUT.patch | 105 ----------- 7 files changed, 91 insertions(+), 494 deletions(-) create mode 100644 patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-link-above-mmap_min_addr-on-riscv64.patch delete mode 100644 patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch delete mode 100644 patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch delete mode 100644 patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch delete mode 100644 patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch diff --git a/.github/workflows/build-ray-haproxy.yml b/.github/workflows/build-ray-haproxy.yml index 311d4a08e45..2f8f1c837b6 100644 --- a/.github/workflows/build-ray-haproxy.yml +++ b/.github/workflows/build-ray-haproxy.yml @@ -216,7 +216,7 @@ jobs: - uses: ./actions/collect-gpl-sources with: image: ${{ env.MANYLINUX_RISCV64_IMAGE }} - packages: gcc + packages: gcc libxcrypt output: gpl-sources.tar - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 @@ -242,4 +242,4 @@ jobs: with: artifact-pattern: ray-haproxy-${{ matrix.version }}-py3-none-manylinux_riscv64 gpl-sources-artifact: ray-haproxy-${{ matrix.version }}-gpl-sources - gpl-sources-description: gcc + gpl-sources-description: gcc and the copyleft libraries bundled in the wheel diff --git a/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-link-above-mmap_min_addr-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-link-above-mmap_min_addr-on-riscv64.patch new file mode 100644 index 00000000000..f7f72b8f3df --- /dev/null +++ b/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-link-above-mmap_min_addr-on-riscv64.patch @@ -0,0 +1,57 @@ +From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 +From: Ludovic Henry +Date: Thu, 1 Oct 2026 00:00:00 +0000 +Subject: [PATCH] ci/build: link haproxy above mmap_min_addr on riscv64 + +The staged binary is killed with SIGSEGV the first time the script runs +it, with no output from HAProxy or from the dynamic linker: + + Segmentation fault LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v + +HAProxy's Makefile links a non-PIE executable, and GNU ld's riscv64 +default linker script places its first PT_LOAD at 0x10000 (x86_64 and +aarch64 use 0x400000). `patchelf --set-rpath '$ORIGIN/lib'` has to grow +the program headers of an ET_EXEC it cannot relocate, so it maps them in +a new segment just below the old base, under 0x10000 -- which is the +kernel's vm.mmap_min_addr on the riscv64 runners. execve then fails past +its point of no return and the process dies before ld.so's first +instruction: LD_DEBUG=all writes nothing and LD_DEBUG_OUTPUT is never +even created, and ldd (which asks ld.so to load the binary) reports "not +a dynamic executable" for the patchelf'd file. The vendored libraries +play no part in it: the crash is identical whether Rocky 10's OpenSSL +3.5.5 or the OpenSSL 3.0.15 built here is vendored. + +Pass `-Wl,-Ttext-segment=0x200000` through HAProxy's ADDLIB link hook on +riscv64, so patchelf's extra segment still lands above mmap_min_addr, as +it already does on x86_64/aarch64 with their higher default base. + +Upstream-Status: Inappropriate [riscv64-only: x86_64/aarch64 link at 0x400000, so patchelf's prepended segment never drops below mmap_min_addr there] + +Signed-off-by: Ludovic Henry +--- +diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh +--- a/ci/build/build-haproxy-dist.sh ++++ b/ci/build/build-haproxy-dist.sh +@@ -173,6 +173,14 @@ PCRE_MAKE_VARS=(USE_PCRE=1) + if [ "$ARCH_LABEL" = "riscv64" ]; then + PCRE_MAKE_VARS=(USE_PCRE2=1 USE_PCRE2_JIT=1) + fi ++# riscv64's default non-PIE link base is 0x10000, which is also the kernel's ++# vm.mmap_min_addr: the program header segment `patchelf --set-rpath` prepends ++# below it further down can't be mapped, and execve kills the binary with ++# SIGSEGV before ld.so runs. Link it higher, like x86_64/aarch64 already are. ++LINK_MAKE_VARS=() ++if [ "$ARCH_LABEL" = "riscv64" ]; then ++ LINK_MAKE_VARS=(ADDLIB=-Wl,-Ttext-segment=0x200000) ++fi + make -C "$BUILD_DIR" \ + TARGET=linux-glibc \ + USE_OPENSSL=1 \ +@@ -184,6 +192,7 @@ make -C "$BUILD_DIR" \ + LUA_INC="$DEPS_DIR/include" \ + LUA_LIB="$DEPS_DIR/lib" \ + USE_PROMEX=1 \ ++ "${LINK_MAKE_VARS[@]}" \ + -j"$(nproc)" + + # --------------------------------------------------------------------------- diff --git a/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch deleted file mode 100644 index e67a46e2c3a..00000000000 --- a/patches/ray-haproxy/2.8.25/0004-build-haproxy-dist-skip-ldd-self-exec-check-on-riscv64.patch +++ /dev/null @@ -1,67 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Ludovic Henry -Date: Mon, 28 Sep 2026 00:00:00 +0000 -Subject: [PATCH] ci/build: skip ldd's self-exec check on riscv64 - -After vendoring and `patchelf --set-rpath`, the script's own sanity echo -re-runs `ldd` on the just-rewritten binary. On manylinux_2_39_riscv64 that -call fails outright: - - Verifying all deps resolve: - not a dynamic executable - -`ldd` resolves a binary's dependencies by re-executing it with -`LD_TRACE_LOADED_OBJECTS=1` set (the dynamic linker intercepts and prints -the deps instead of running the program). That self-exec trick is what -reports "not a dynamic executable" here, even though the ELF itself is -fine: `patchelf --print-rpath` reads the very same file back cleanly one -line above, and vendoring already collected every non-allowlisted `NEEDED` -entry into `$STAGE_DIR/lib` via a plain `ldd` call earlier in this same -script, before `strip`/`patchelf` touched the binary — dependency -resolution was never in question, only this later ldd self-exec on -manylinux_2_39_riscv64's still-young riscv64 glibc port. The plain execve a -few lines further down (`"$STAGE_DIR/haproxy" -v`) runs the identical -binary directly and is unaffected. - -Replace the self-exec check with the equivalent static one on riscv64: -read the binary's `NEEDED` entries with `readelf -d` and confirm each is -either manylinux-allowlisted or vendored, without asking `ldd` to -re-execute the binary at all. - -Upstream-Status: Inappropriate [ldd's self-exec check is fine on x86_64/aarch64; this works around manylinux_2_39_riscv64's own riscv64 glibc/ldd, not something upstream's existing targets hit] - -Signed-off-by: Ludovic Henry ---- -diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh ---- a/ci/build/build-haproxy-dist.sh -+++ b/ci/build/build-haproxy-dist.sh -@@ -261,7 +261,28 @@ find "$STAGE_DIR/lib" -name '*.so*' -exec patchelf --set-rpath '$ORIGIN' {} \; - - echo " RPATH: $(patchelf --print-rpath "$STAGE_DIR/haproxy")" - echo " Verifying all deps resolve:" --LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" -+if [ "$ARCH_LABEL" = "riscv64" ]; then -+ # ldd resolves deps by re-executing the binary with -+ # LD_TRACE_LOADED_OBJECTS=1; on manylinux_2_39_riscv64 that self-exec -+ # trick reports "not a dynamic executable" against a binary patchelf has -+ # just rewritten, even though the ELF itself is fine (patchelf's own -+ # --print-rpath above reads it back cleanly, and the plain execve a few -+ # lines down, "$STAGE_DIR/haproxy" -v, runs it directly). Check NEEDED -+ # entries statically instead of asking ldd to self-exec the binary. -+ MISSING="" -+ for needed in $(readelf -d "$STAGE_DIR/haproxy" | grep NEEDED | sed -E 's/.*\[(.*)\]/\1/'); do -+ echo "$needed" | grep -qE "$MANYLINUX_ALLOWLIST" && continue -+ [ -f "$STAGE_DIR/lib/$needed" ] && continue -+ MISSING="$MISSING $needed" -+ done -+ if [ -n "$MISSING" ]; then -+ echo "ERROR: dependencies not allowlisted or vendored:$MISSING" -+ exit 1 -+ fi -+ echo " All NEEDED entries are allowlisted or vendored" -+else -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" -+fi - - # --------------------------------------------------------------------------- - # 5. Sanity checks diff --git a/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch index ef33473a7b7..97ae66eec84 100644 --- a/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch +++ b/patches/ray-haproxy/2.8.25/0005-build-haproxy-dist-vendor-our-own-openssl-on-riscv64.patch @@ -1,67 +1,54 @@ From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 From: Ludovic Henry -Date: Mon, 28 Sep 2026 00:00:00 +0000 +Date: Thu, 1 Oct 2026 00:00:00 +0000 Subject: [PATCH] ci/build: vendor our own OpenSSL on riscv64 -Once the ldd self-exec check (0004) was worked around, the build got one -step further and then died with no output at all right where the staged -binary is first executed: +collect_deps resolves the binary's NEEDED entries with a plain ldd, run +before patchelf gives it an RPATH. On manylinux_2_39_riscv64 that finds +Rocky 10's own OpenSSL (3.5.5) on the default loader path and vendors it: Vendoring: libssl.so.3 (/lib64/lp64d/libssl.so.3) Vendoring: libcrypto.so.3 (/lib64/lp64d/libcrypto.so.3) - ... - All NEEDED entries are allowlisted or vendored - ./ci/build/build-haproxy-dist.sh: line 310: 18549 Segmentation fault LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v -`/lib64/lp64d/libssl.so.3` is not a build path -- it is Rocky 10's own -system OpenSSL (`rpm -q openssl-libs` on the manylinux_2_39_riscv64 base -reports 3.5.5), not the OpenSSL 3.0.15 this same script builds from source -a few lines above and points HAProxy's SSL_INC/SSL_LIB at. `collect_deps` -resolves each NEEDED entry with a plain `ldd`, which at that point in the -script has no LD_LIBRARY_PATH pointing at the just-built OpenSSL and no -RPATH on the binary yet (patchelf runs later), so it falls through to the -dynamic linker's default search path -- and manylinux_2_39_riscv64 (Rocky -10) already has a libssl.so.3/libcrypto.so.3 there. manylinux2014 (CentOS -7, x86_64/aarch64) never hits this: it has no libssl.so.3 at all, ldd -correctly reports "not found", and the "vendor OpenSSL .so files we built -from source" fallback a few lines down fills in the right copy. On -riscv64 that fallback never fires, because collect_deps has already -placed a file under the same soname (Rocky 10's, not ours) and the -fallback's `[ -f "$STAGE_DIR/lib/$soname" ] && continue` guard skips it. +manylinux2014 has no libssl.so.3, so on x86_64/aarch64 ldd reports it +"not found" and the "vendor the OpenSSL .so files we built from source" +fallback that follows ships the 3.0.15 HAProxy was compiled and linked +against. On riscv64 that fallback skips both sonames because Rocky 10's +copies are already in lib/. -The wheel ends up shipping a HAProxy linked against OpenSSL 3.0.15 headers -but running against Rocky 10's unrelated 3.5.5 build -- five minor -releases apart, with its own (RHEL10-target) hwcap-gated codepaths -- and -it now segfaults before HAProxy's own code ever prints anything, which -points at process-startup library-constructor code in that mismatched -libcrypto rather than anything in HAProxy's own `-v` handling. +Point that ldd walk at the OpenSSL built here, on riscv64 only. Keep it +scoped to the collect_deps call: exported for the rest of the script, +it also reaches binutils, whose readelf links libcurl through +libdebuginfod, and Rocky 10's libcurl needs a newer OpenSSL than 3.0.15 +exports, so every later readelf fails to start: -Point ldd at the OpenSSL we just built before collect_deps runs, on -riscv64 only, so it vendors that copy instead of falling through to -Rocky 10's. + readelf: .../lib/libssl.so.3: version `OPENSSL_3.2.0' not found (required by /lib64/lp64d/libcurl.so.4) -Upstream-Status: Inappropriate [only manylinux_2_39_riscv64 (Rocky 10) already ships a colliding libssl.so.3/libcrypto.so.3 on the default loader path; manylinux2014 (x86_64/aarch64) doesn't carry that soname so upstream's existing targets never hit this] +which silently turned the script's hardening checks into "NO". + +Upstream-Status: Inappropriate [manylinux2014 (x86_64/aarch64) has no libssl.so.3 on the default loader path, so upstream's targets already vendor the OpenSSL they build] Signed-off-by: Ludovic Henry --- diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh --- a/ci/build/build-haproxy-dist.sh +++ b/ci/build/build-haproxy-dist.sh -@@ -235,6 +235,17 @@ collect_deps() { +@@ -244,7 +244,17 @@ collect_deps() { done } - + +-collect_deps "$STAGE_DIR/haproxy" +if [ "$ARCH_LABEL" = "riscv64" ]; then -+ # manylinux_2_39_riscv64 (Rocky 10) already ships libssl.so.3/libcrypto.so.3 -+ # on the default loader search path; manylinux2014 (CentOS 7) doesn't carry -+ # that soname at all. Without this, ldd's default resolution below (run -+ # with no LD_LIBRARY_PATH yet, since patchelf hasn't set one on the binary) -+ # finds Rocky 10's own OpenSSL build instead of the one just built from -+ # source above, and collect_deps vendors *that* -- a different OpenSSL -+ # build than the one HAProxy was actually configured, compiled and linked -+ # against a few lines up. -+ export LD_LIBRARY_PATH="$OPENSSL_LIB_DIR" ++ # manylinux_2_39_riscv64 (Rocky 10) ships its own libssl.so.3/libcrypto.so.3 ++ # on the default loader path, which manylinux2014 doesn't, so a plain ldd ++ # would vendor Rocky's OpenSSL instead of the one built above. Scope this to ++ # the ldd walk: exported script-wide it also reaches binutils, whose ++ # readelf links libcurl (via libdebuginfod) and then fails to start against ++ # this older OpenSSL. ++ LD_LIBRARY_PATH="$OPENSSL_LIB_DIR" collect_deps "$STAGE_DIR/haproxy" ++else ++ collect_deps "$STAGE_DIR/haproxy" +fi - collect_deps "$STAGE_DIR/haproxy" - + # Also vendor the OpenSSL .so files we built from source — ldd sees them by + # their build path, but make sure their sonames are in lib/. diff --git a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch b/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch deleted file mode 100644 index 934f7334e5b..00000000000 --- a/patches/ray-haproxy/2.8.25/0006-build-haproxy-dist-TEMP-capture-strace-core-backtrace-on-riscv64-crash.patch +++ /dev/null @@ -1,172 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Ludovic Henry -Date: Mon, 28 Sep 2026 00:00:00 +0000 -Subject: [PATCH] ci/build: TEMP capture strace/core backtrace on riscv64 - haproxy -v crash - -Two fixes so far (ldd's self-exec quirk, then vendoring our own OpenSSL -instead of Rocky 10's system one) have not resolved the `haproxy -v` -segfault at the end of the build: - - Verifying all deps resolve: - All NEEDED entries are allowlisted or vendored - ./ci/build/build-haproxy-dist.sh: line 321: 18518 Segmentation fault LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v - -Both fixes were made by inference from log text alone, without access to -riscv64 hardware to actually run the binary and inspect the crash. This -adds a TEMPORARY, riscv64-only diagnostic around that exact invocation. - -The first revision of this diagnostic (which wrapped the invocation in -`strace -f -o ... env ...` unconditionally, deciding STATUS from that -wrapped command's own exit code) got real CI signal for the first time, -but it wasn't signal about HAProxy at all: - - haproxy -v exit status: 127 - ==> DIAGNOSTIC: haproxy -v failed (status 127); full strace log: - ./ci/build/build-haproxy-dist.sh: line 338: strace: command not found - -Exit 127 is bash's own "command not found" status. `dnf install -y gdb -strace 2>/dev/null || true` had silently not left a working `strace` on -PATH (redirected stderr and `|| true` hid whatever dnf actually did), so -the following `strace -f -o ... env LD_LIBRARY_PATH=... "$STAGE_DIR/haproxy" --v` line failed at the shell's attempt to exec `strace` itself -- HAProxy -was never invoked, wrapped or otherwise, and nothing here says anything -about the original segfault yet. - -Restructure so a missing diagnostic tool can't be mistaken for the thing -being diagnosed: - - - Run the plain, unwrapped invocation first and on its own; STATUS is - always its real exit code, never a wrapper's. - - Only if that fails, gather diagnostics afterwards, independently of - STATUS: LD_DEBUG=all (glibc's own dynamic-linker trace, built into - every glibc, so unlike strace it needs no package install and can't - hit this same "tool not on PATH" failure mode), a `set +e`-guarded - `ldd`, and `strace` only when `command -v strace` actually finds it - on PATH after the install attempt. - -That revision ran on riscv64 CI for the first time and got the real -crash: `haproxy -v` genuinely segfaults, plain and under `LD_DEBUG=all` -alike. But the `LD_DEBUG=all` re-run was never wrapped in the same -`set +e` / capture-status / `set -e` pattern the plain invocation above -it uses -- it was a bare command under `set -e`, so its own segfault -(exit 139) aborted the script right there, before the following -`echo`/`cat` could run: - - ==> DIAGNOSTIC: LD_DEBUG=all trace (...) - ##[error]Process completed with exit code 139. - -The trace glibc's dynamic linker had already written to -`/tmp/haproxy-v-lddebug.log` (unbuffered, as it goes, not just on clean -exit) never made it into the job log. Guard that re-run the same way the -plain invocation is guarded, and print the log's tail unconditionally -afterwards regardless of the re-run's own exit status -- capped to the -last 500 lines since a full `LD_DEBUG=all` trace can be large and the -crash-adjacent lines at the end are what matter. - -Still a TEMPORARY riscv64-only diagnostic, not a fix. - -Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] - -Signed-off-by: Ludovic Henry ---- -diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh ---- a/ci/build/build-haproxy-dist.sh -+++ b/ci/build/build-haproxy-dist.sh -@@ -318,7 +318,95 @@ else - echo " Stack canary: NO — consider -fstack-protector-strong" - fi - --LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v -+if [ "$ARCH_LABEL" = "riscv64" ]; then -+ # TEMPORARY diagnostic (not a real fix): capture real signal from this -+ # exact invocation on the riscv64 runner. STATUS always reflects the -+ # plain, unwrapped invocation below, run first and on its own -- any -+ # diagnostic tooling (strace, gdb) is gathered afterwards, independently, -+ # and never decides STATUS, so a diagnostic tool that fails to run can't -+ # be mistaken for the thing being diagnosed. -+ set +e -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v -+ STATUS=$? -+ set -e -+ echo " haproxy -v exit status: $STATUS" -+ if [ "$STATUS" -ne 0 ]; then -+ echo "==> DIAGNOSTIC: haproxy -v failed (status $STATUS); gathering diagnostics" -+ dnf install -y gdb strace 2>/dev/null || true -+ echo " core-dump hard limit: $(ulimit -Hc 2>/dev/null || echo unknown)" -+ # The container runtime may lock the core-dump hard limit to 0, in -+ # which case raising the soft limit fails with "Operation not -+ # permitted". Do not let that abort the script under `set -e` (it is -+ # inside an `if` condition, so a nonzero status here does not trigger -+ # errexit). -+ if ! ulimit -c unlimited 2>&1; then -+ echo " ulimit -c unlimited failed: core dumps are unavailable in this container" -+ fi -+ echo " core_pattern: $(cat /proc/sys/kernel/core_pattern 2>/dev/null || echo unknown)" -+ -+ echo "==> DIAGNOSTIC: LD_DEBUG=all trace (glibc's own dynamic-linker trace;" -+ echo " built into every glibc, unlike strace, so it needs no package install):" -+ set +e -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all "$STAGE_DIR/haproxy" -v \ -+ > /tmp/haproxy-v-lddebug.log 2>&1 -+ LDDEBUG_STATUS=$? -+ set -e -+ echo " LD_DEBUG re-run exit status: $LDDEBUG_STATUS (informational only)" -+ if [ -s /tmp/haproxy-v-lddebug.log ]; then -+ wc -l /tmp/haproxy-v-lddebug.log -+ echo " last 500 lines (ld.so writes this trace to fd 2 unbuffered as it" -+ echo " runs, so the crash-adjacent lines survive even though the process" -+ echo " never reached a clean exit):" -+ tail -n 500 /tmp/haproxy-v-lddebug.log -+ else -+ echo " (empty LD_DEBUG log)" -+ fi -+ -+ echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" -+ echo " executable\" on this riscv64 ld.so per the self-exec quirk above):" -+ set +e -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" ldd "$STAGE_DIR/haproxy" -+ echo " ldd exit: $?" -+ set -e -+ -+ if command -v strace >/dev/null 2>&1; then -+ echo "==> DIAGNOSTIC: strace -f trace:" -+ set +e -+ strace -f -o /tmp/haproxy-v-strace.log \ -+ env LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v -+ set -e -+ if [ -s /tmp/haproxy-v-strace.log ]; then -+ wc -l /tmp/haproxy-v-strace.log -+ cat /tmp/haproxy-v-strace.log -+ else -+ echo " (empty strace log)" -+ fi -+ else -+ echo "==> DIAGNOSTIC: strace not available after install attempt (dnf install" -+ echo " may have failed, e.g. no network/repo access in this container);" -+ echo " skipping strace, LD_DEBUG=all above is the primary signal" -+ fi -+ -+ echo "==> DIAGNOSTIC: searching for a core dump (bonus signal only; not required)" -+ CORE="" -+ for d in . /workspace /tmp "$STAGE_DIR"; do -+ for f in "$d"/core "$d"/core.*; do -+ [ -f "$f" ] && CORE="$f" && break 2 -+ done -+ done -+ if [ -n "$CORE" ]; then -+ echo " found core: $CORE" -+ gdb -batch -ex 'set pagination off' -ex bt -ex 'info registers' \ -+ -ex 'x/20i $pc-40' -ex 'info sharedlibrary' \ -+ "$STAGE_DIR/haproxy" "$CORE" 2>&1 -+ else -+ echo " no core dump found (expected if core dumps are unavailable here)" -+ fi -+ exit "$STATUS" -+ fi -+else -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" "$STAGE_DIR/haproxy" -v -+fi - - # --------------------------------------------------------------------------- - # 6. Package diff --git a/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch b/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch deleted file mode 100644 index d5f125696a9..00000000000 --- a/patches/ray-haproxy/2.8.25/0007-build-haproxy-dist-TEMP-probe-LD_DEBUG_OUTPUT-on-riscv64.patch +++ /dev/null @@ -1,103 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Ludovic Henry -Date: Tue, 29 Sep 2026 00:00:00 +0000 -Subject: [PATCH] ci/build: TEMP probe LD_DEBUG_OUTPUT on riscv64 haproxy -v - crash - -The previous diagnostic round finally captured a clean, correctly-guarded -LD_DEBUG=all re-run, and it came back completely empty: - - ==> DIAGNOSTIC: LD_DEBUG=all trace (...) - LD_DEBUG re-run exit status: 139 (informational only) - (empty LD_DEBUG log) - -Two things about this are worth separating out before reading anything -else into it: - -- glibc's dynamic linker writes the LD_DEBUG trace with direct, unbuffered - write(2) calls to the target fd, not through C stdio -- so redirecting - the whole process's stdout/stderr to a regular file (`> file 2>&1`, as - the existing re-run does) is not expected to introduce the kind of - userspace buffering that would swallow this specific trace even if the - process crashes moments later. -- The two prior fixes to this same diagnostic (exit-127-from-missing- - strace, then the unguarded LD_DEBUG re-run aborting the script under - `set -e` before its own log could be printed) both turned out to be bugs - in the diagnostic harness itself, not in HAProxy -- so it is worth - double-checking this empty log isn't a third instance of the same - pattern before treating it as real signal about the crash. - -Add a second, independent re-run using LD_DEBUG_OUTPUT, which glibc opens -and writes to directly by path rather than inheriting fd 1/2 from the -shell at all. If that log also comes back empty, the two independent -sinks agreeing rules out this specific harness's redirection as the -explanation, and points instead at the crash happening before ld.so's own -code starts running (e.g. in the kernel's ELF-loading/entry trampoline, -ahead of the interpreter's first debug-print call). - -Separately: the `readelf -d`-based NEEDED check earlier in this script -(riscv64's replacement for ldd's self-exec check, which reports "not a -dynamic executable" against this exact binary on manylinux_2_39_riscv64 -per the commit that added it) has been confirmed passing again on this -build, and the same platform ldd has, in this round, additionally been -seen emitting a symbol-version warning about an unrelated system library -(`libcurl.so.4` wanting a newer OPENSSL_3.2.0 than our vendored 3.0.x -libssl provides) while walking dependencies it wasn't asked to vendor -- -further evidence this image's riscv64 ldd is doing something unreliable -of its own accord, not that the haproxy binary it's being pointed at is -malformed. - -No compiler flag in this script, in HAProxy's own Makefile defaults for -TARGET=linux-glibc, or in the riscv64 branches added by earlier patches -here selects a CPU/ISA baseline (no CPU= or ARCH= is passed to HAProxy's -`make`, and HAProxy's default CPU_CFLAGS/ARCH_FLAGS for an unset CPU/ARCH -add no -march/-mcpu at all). A leaked `-march=native` would also fail to -compile outright on this toolchain rather than produce a binary that -merely segfaults at runtime, per gcc's behaviour on this same -manylinux_2_39_riscv64 image (see build-pyscipopt.yml, which strips -`-march=native` from a vendored dependency's build for exactly this -reason). Still no evidence of an ISA mismatch from the build flags; not -changing anything CPU/ISA-related in this round. - -Still a TEMPORARY riscv64-only diagnostic, not a fix. - -Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] - -Signed-off-by: Ludovic Henry ---- -diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh ---- a/ci/build/build-haproxy-dist.sh -+++ b/ci/build/build-haproxy-dist.sh -@@ -359,7 +359,31 @@ if [ "$ARCH_LABEL" = "riscv64" ]; then - echo " never reached a clean exit):" - tail -n 500 /tmp/haproxy-v-lddebug.log - else -- echo " (empty LD_DEBUG log)" -+ echo " (empty LD_DEBUG log; re-running with LD_DEBUG_OUTPUT below, which" -+ echo " glibc writes through a different path than the shell's own" -+ echo " redirect of the whole process's stdout/stderr)" -+ fi -+ -+ echo "==> DIAGNOSTIC: LD_DEBUG=all + LD_DEBUG_OUTPUT trace (glibc opens this" -+ echo " file itself and writes to it directly, independently of whatever" -+ echo " the shell did to fd 1/2 above -- an empty log here too would mean" -+ echo " the crash happens before ld.so's own code starts running at all," -+ echo " e.g. in the kernel's ELF-loading/entry trampoline):" -+ rm -f /tmp/haproxy-v-lddebug-output.* -+ set +e -+ LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all \ -+ LD_DEBUG_OUTPUT=/tmp/haproxy-v-lddebug-output \ -+ "$STAGE_DIR/haproxy" -v > /dev/null 2>&1 -+ LDDEBUG_OUTPUT_STATUS=$? -+ set -e -+ echo " LD_DEBUG_OUTPUT re-run exit status: $LDDEBUG_OUTPUT_STATUS (informational only)" -+ LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)" -+ if [ -n "$LDDEBUG_OUTPUT_FILE" ] && [ -s "$LDDEBUG_OUTPUT_FILE" ]; then -+ echo " $LDDEBUG_OUTPUT_FILE:" -+ wc -l "$LDDEBUG_OUTPUT_FILE" -+ tail -n 500 "$LDDEBUG_OUTPUT_FILE" -+ else -+ echo " (empty or missing LD_DEBUG_OUTPUT file too)" - fi - - echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" diff --git a/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch b/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch deleted file mode 100644 index 9780c2534c8..00000000000 --- a/patches/ray-haproxy/2.8.25/0008-build-haproxy-dist-TEMP-print-and-un-break-LD_DEBUG_OUTPUT.patch +++ /dev/null @@ -1,105 +0,0 @@ -From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001 -From: Ludovic Henry -Date: Tue, 29 Sep 2026 00:00:00 +0000 -Subject: [PATCH] ci/build: TEMP print and un-break LD_DEBUG_OUTPUT diagnostic - -The previous round finally got the LD_DEBUG_OUTPUT re-run itself running -cleanly (no step-killing crash, a correctly captured "exit status: 139 -(informational only)"), but the CI log still stops dead right after that -line: - - LD_DEBUG_OUTPUT re-run exit status: 139 (informational only) - ##[error]Process completed with exit code 2. - -Nothing after it ever reads or prints the file glibc actually wrote. Two -separate bugs, both in the diagnostic harness itself: - -- glibc's dynamic linker does not write LD_DEBUG_OUTPUT= to that - exact path -- it appends the process's PID, producing - "." (e.g. "/tmp/haproxy-v-lddebug-output.18524"), and the - PID isn't known from the shell's side ahead of the crash. The line that - was meant to locate that file -- - `LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)"` - -- has to glob for it, which it does, but through a pipeline. -- This script inherits `set -euxo pipefail` from the `bash -c` wrapper - that invokes it (via bash's automatic SHELLOPTS propagation to child - bash processes), so pipefail is live here even though this script never - sets it itself. With no file matching the glob, the unquoted pattern is - passed to `ls` literally, `ls` exits 2 ("No such file or directory", - redirected to /dev/null), and with pipefail that pipeline's *overall* - status is `ls`'s 2 even though the trailing `head -n1` exits 0. That - nonzero status lands on a plain assignment, which errexit does not - forgive, so the script dies right there with exit 2 -- before the `if` - a few lines down ever runs to print anything. This is the same shape of - self-inflicted harness bug as the previous two rounds (missing strace - reported as haproxy's own exit code; an unguarded re-run aborting the - script under `set -e` before its own log could be printed), not new - information about the crash. - -Replace the `ls | head` pipeline with a bare glob `for` loop: with no -match, bash leaves the pattern unexpanded and `[ -f "$f" ]` just reports -false, so there is no pipeline left to fail under pipefail. Print every -matching file's content (there should only ever be one, but loop instead -of assuming), and when nothing matches, say so explicitly and distinctly -from a matched-but-empty file, since "no file at all" is itself the -signal that the crash happens before ld.so's own code -- including its -LD_DEBUG_OUTPUT file open -- ever runs. - -Still a TEMPORARY riscv64-only diagnostic, not a fix. - -Upstream-Status: Inappropriate [temporary riscv64 CI diagnostic, not meant to be carried or upstreamed] - -Signed-off-by: Ludovic Henry ---- -diff --git a/ci/build/build-haproxy-dist.sh b/ci/build/build-haproxy-dist.sh ---- a/ci/build/build-haproxy-dist.sh -+++ b/ci/build/build-haproxy-dist.sh -@@ -369,21 +369,39 @@ if [ "$ARCH_LABEL" = "riscv64" ]; then - echo " the shell did to fd 1/2 above -- an empty log here too would mean" - echo " the crash happens before ld.so's own code starts running at all," - echo " e.g. in the kernel's ELF-loading/entry trampoline):" -- rm -f /tmp/haproxy-v-lddebug-output.* -+ LDDEBUG_OUTPUT_PATH=/tmp/haproxy-v-lddebug-output -+ rm -f "${LDDEBUG_OUTPUT_PATH}".* - set +e - LD_LIBRARY_PATH="$STAGE_DIR/lib" LD_DEBUG=all \ -- LD_DEBUG_OUTPUT=/tmp/haproxy-v-lddebug-output \ -+ LD_DEBUG_OUTPUT="$LDDEBUG_OUTPUT_PATH" \ - "$STAGE_DIR/haproxy" -v > /dev/null 2>&1 - LDDEBUG_OUTPUT_STATUS=$? - set -e - echo " LD_DEBUG_OUTPUT re-run exit status: $LDDEBUG_OUTPUT_STATUS (informational only)" -- LDDEBUG_OUTPUT_FILE="$(ls /tmp/haproxy-v-lddebug-output.* 2>/dev/null | head -n1)" -- if [ -n "$LDDEBUG_OUTPUT_FILE" ] && [ -s "$LDDEBUG_OUTPUT_FILE" ]; then -- echo " $LDDEBUG_OUTPUT_FILE:" -- wc -l "$LDDEBUG_OUTPUT_FILE" -- tail -n 500 "$LDDEBUG_OUTPUT_FILE" -- else -- echo " (empty or missing LD_DEBUG_OUTPUT file too)" -+ # glibc's dynamic linker appends the process's PID to LD_DEBUG_OUTPUT -+ # rather than writing the exact path given (so the file is e.g. -+ # "${LDDEBUG_OUTPUT_PATH}.18524", not "${LDDEBUG_OUTPUT_PATH}" itself), -+ # and the PID isn't known from the shell's side ahead of the crash. -+ # Glob for it instead of piping `ls` through `head`: with this -+ # script's inherited `pipefail`, `ls nomatch 2>/dev/null | head -n1` -+ # would itself exit non-zero (ls's own status, even though head -+ # succeeds) and abort the script under `set -e` before ever printing -+ # anything -- which is exactly what silently ate this section last -+ # run. A bare `for` glob has no such pipeline to fail: with no match -+ # bash leaves the pattern unexpanded and `[ -f "$f" ]` just reports -+ # false. -+ LDDEBUG_OUTPUT_FOUND=0 -+ for f in "${LDDEBUG_OUTPUT_PATH}".*; do -+ if [ -f "$f" ]; then -+ LDDEBUG_OUTPUT_FOUND=1 -+ echo " LD_DEBUG_OUTPUT file: $f" -+ wc -l "$f" -+ tail -n 500 "$f" -+ fi -+ done -+ if [ "$LDDEBUG_OUTPUT_FOUND" -eq 0 ]; then -+ echo " no LD_DEBUG_OUTPUT file was created at all (not even an empty" -+ echo " one) -- distinct from a file existing but being empty" - fi - - echo "==> DIAGNOSTIC: ldd (informational only; may report \"not a dynamic" From 414c62d811b6dac05ac49f926fef5e0aa421bfa3 Mon Sep 17 00:00:00 2001 From: Ludovic Henry Date: Thu, 1 Oct 2026 19:05:25 +0000 Subject: [PATCH 13/13] ray-haproxy: fix the test job's patch glob The test jobs never ran before (the build always failed first), and now fail at "Apply patches": error: can't open patch '../python-wheels/patches/ray-haproxy/2.8.25/*.patch': No such file or directory The shell expands the glob from the workspace root, not from git -C's directory, so the relative ../python-wheels pattern matches nothing and reaches git literally. Anchor it at $GITHUB_WORKSPACE, as build-sglang-router.yml does. --- .github/workflows/build-ray-haproxy.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/build-ray-haproxy.yml b/.github/workflows/build-ray-haproxy.yml index 2f8f1c837b6..4808bf58c06 100644 --- a/.github/workflows/build-ray-haproxy.yml +++ b/.github/workflows/build-ray-haproxy.yml @@ -152,7 +152,7 @@ jobs: persist-credentials: false - name: Apply patches - run: git -C ray-haproxy-src apply -v ../python-wheels/patches/ray-haproxy/${{ env.RAY_HAPROXY_VERSION }}/*.patch + run: git -C ray-haproxy-src apply -v "$GITHUB_WORKSPACE"/python-wheels/patches/ray-haproxy/${{ env.RAY_HAPROXY_VERSION }}/*.patch - name: Download wheel uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1