Skip to content

Commit 73ecf15

Browse files
warp-lang: Add version 1.17.0 (#2375)
* warp-lang: Add version 1.17.0 Signed-off-by: riseproject-dev[bot] <330740410+riseproject-dev[bot]@users.noreply.github.com> * warp-lang: Carry the 1.16.0 patches forward to 1.17.0 0003-0005 apply unchanged. Two need refreshing against 1.17.0: - 0001: upstream moved machine_architecture() out of build_dll.py and setup.py into the new warp/_src/build_architecture.py, as an alias table plus an Architecture literal. riscv64 is now added there instead of to the two copies; the setup.py platform table, help strings and build_llvm.py RISCV backend selection are the same changes as before, rebased onto setup.py's new windows-aarch64 entries. - 0002: context only, the neighbouring aarch64 block is now `#if defined(__aarch64__) || defined(_M_ARM64)`. --------- Signed-off-by: riseproject-dev[bot] <330740410+riseproject-dev[bot]@users.noreply.github.com> Co-authored-by: riseproject-dev[bot] <330740410+riseproject-dev[bot]@users.noreply.github.com> Co-authored-by: Ludovic Henry <git@ludovic.dev>
1 parent a8ae8b5 commit 73ecf15

6 files changed

Lines changed: 459 additions & 0 deletions

‎docs/packages/warp-lang.yaml‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -21,3 +21,4 @@ versions:
2121
gpl-sources:
2222
filename: gpl-sources.tar
2323
description: gcc
24+
- version: 1.17.0
Lines changed: 132 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,132 @@
1+
From 8b64e2e85baff0f1e96beb7b3ad4d25ac8cbd82e Mon Sep 17 00:00:00 2001
2+
From: Ludovic Henry <git@ludovic.dev>
3+
Date: Sun, 20 Sep 2026 12:00:00 +0000
4+
Subject: [PATCH 1/4] Add riscv64 to the build and packaging platform tables
5+
6+
Upstream-Status: To upstream [no riscv64 CI runner or prebuilt Clang/LLVM SDK upstream yet]
7+
8+
machine_architecture() in warp/_src/build_architecture.py (shared by
9+
build_lib.py, build_llvm.py, build_dll.py and setup.py) raises
10+
"Unrecognized machine architecture 'riscv64'", which aborts build_lib.py before
11+
anything is compiled.
12+
13+
setup.py's `platforms` table is what WarpBDistWheel.get_tag() reads to stamp the
14+
wheel's platform tag, so riscv64 needs an entry there as well. manylinux_2_39 is
15+
the flavour of the image these wheels are built in
16+
(quay.io/pypa/manylinux_2_39_riscv64, Rocky Linux 10, glibc 2.39).
17+
18+
build_llvm.py picks LLVM_TARGETS_TO_BUILD from the same canonical architecture
19+
string and otherwise falls back to X86, which would build an LLVM unable to emit
20+
code for the machine it runs on; RISCV is the matching backend name.
21+
22+
Nothing else in the tree is architecture-specific: warp/native carries no SIMD
23+
intrinsics and the Linux compile flags are a generic -O3 -fPIC --std=c++17.
24+
---
25+
build_llvm.py | 4 +++-
26+
setup.py | 11 ++++++-----
27+
warp/_src/build_architecture.py | 5 +++--
28+
3 files changed, 12 insertions(+), 8 deletions(-)
29+
30+
diff --git a/build_llvm.py b/build_llvm.py
31+
index 46cc828..7ac9c9b 100644
32+
--- a/build_llvm.py
33+
+++ b/build_llvm.py
34+
@@ -166,7 +166,7 @@ def build_llvm_clang_from_source_for_arch(args, arch: str, llvm_source: str) ->
35+
36+
Args:
37+
args: Command line arguments
38+
- arch: Architecture to build for ("aarch64" or "x86_64")
39+
+ arch: Architecture to build for ("aarch64", "riscv64" or "x86_64")
40+
llvm_source: Path to the LLVM source code
41+
"""
42+
43+
@@ -224,6 +224,8 @@ def build_llvm_clang_from_source_for_arch(args, arch: str, llvm_source: str) ->
44+
45+
if arch == "aarch64":
46+
target_backend = "AArch64"
47+
+ elif arch == "riscv64":
48+
+ target_backend = "RISCV"
49+
else:
50+
target_backend = "X86"
51+
52+
diff --git a/setup.py b/setup.py
53+
index 327bae0..998d9da 100644
54+
--- a/setup.py
55+
+++ b/setup.py
56+
@@ -29,14 +29,14 @@ parser.add_argument(
57+
"-P",
58+
type=str,
59+
default="",
60+
- help="Wheel platform: windows-x86_64|windows-aarch64|linux-x86_64|linux-aarch64|macos-aarch64",
61+
+ help="Wheel platform: windows-x86_64|windows-aarch64|linux-x86_64|linux-aarch64|linux-riscv64|macos-aarch64",
62+
)
63+
parser.add_argument(
64+
"--manylinux",
65+
"-M",
66+
type=str,
67+
default="manylinux_2_28",
68+
- help="Manylinux flavor for Linux wheels: manylinux_2_28|manylinux_2_34",
69+
+ help="Manylinux flavor for Linux wheels: manylinux_2_28|manylinux_2_34|manylinux_2_39",
70+
)
71+
args = parser.parse_known_args()[0]
72+
73+
@@ -73,6 +73,7 @@ platforms = [
74+
Platform("windows", "aarch64", "Windows ARM64", ".dll", "win_arm64"),
75+
Platform("linux", "x86_64", "Linux x86-64", ".so", "manylinux_2_28_x86_64"),
76+
Platform("linux", "aarch64", "Linux AArch64", ".so", "manylinux_2_34_aarch64"),
77+
+ Platform("linux", "riscv64", "Linux RISC-V 64", ".so", "manylinux_2_39_riscv64"),
78+
Platform("macos", "aarch64", "macOS ARM64", ".dylib", "macosx_11_0_arm64"),
79+
]
80+
81+
@@ -129,7 +130,7 @@ if args.command == "bdist_wheel":
82+
if len(detected_platforms) > 1:
83+
print("Libraries for multiple platforms were detected.")
84+
print("Run `python -m build --wheel -C--build-option=-P<platform>` to select a specific one.")
85+
- print("Available platforms: windows-x86_64, windows-aarch64, linux-x86_64, linux-aarch64, macos-aarch64")
86+
+ print("Available platforms: windows-x86_64, windows-aarch64, linux-x86_64, linux-aarch64, linux-riscv64, macos-aarch64")
87+
# Select the libraries corresponding with the this machine's platform
88+
for p in platforms:
89+
if p.os == machine_os() and p.arch == machine_architecture():
90+
@@ -151,8 +152,8 @@ class WarpBDistWheel(bdist_wheel):
91+
# setuptools.Command can validate the command line options.
92+
user_options: ClassVar[list[tuple[str, str, str]]] = [
93+
*bdist_wheel.user_options,
94+
- ("platform=", "P", "Wheel platform: windows-x86_64|windows-aarch64|linux-x86_64|linux-aarch64|macos-aarch64"),
95+
- ("manylinux=", "M", "Manylinux flavor for Linux wheels: manylinux_2_28|manylinux_2_34"),
96+
+ ("platform=", "P", "Wheel platform: windows-x86_64|windows-aarch64|linux-x86_64|linux-aarch64|linux-riscv64|macos-aarch64"),
97+
+ ("manylinux=", "M", "Manylinux flavor for Linux wheels: manylinux_2_28|manylinux_2_34|manylinux_2_39"),
98+
]
99+
100+
def initialize_options(self):
101+
diff --git a/warp/_src/build_architecture.py b/warp/_src/build_architecture.py
102+
index 9df7173..68775d4 100644
103+
--- a/warp/_src/build_architecture.py
104+
+++ b/warp/_src/build_architecture.py
105+
@@ -3,7 +3,7 @@
106+
107+
"""Detect and normalize architectures used by Warp builds.
108+
109+
-This module maps platform-specific x86-64 and ARM64 names to the
110+
+This module maps platform-specific x86-64, ARM64 and RV64 names to the
111+
canonical identifiers shared by Warp's build, packaging, and runtime code.
112+
"""
113+
114+
@@ -12,13 +12,14 @@ from __future__ import annotations
115+
import platform
116+
from typing import Literal
117+
118+
-Architecture = Literal["x86_64", "aarch64"]
119+
+Architecture = Literal["x86_64", "aarch64", "riscv64"]
120+
121+
_ARCHITECTURE_ALIASES: dict[str, Architecture] = {
122+
"amd64": "x86_64",
123+
"x86_64": "x86_64",
124+
"arm64": "aarch64",
125+
"aarch64": "aarch64",
126+
+ "riscv64": "riscv64",
127+
}
128+
129+
130+
--
131+
2.43.0
132+
Lines changed: 63 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,63 @@
1+
From bb045ca51b3f98290f3194192d5a72fde6af1667 Mon Sep 17 00:00:00 2001
2+
From: Ludovic Henry <git@ludovic.dev>
3+
Date: Sun, 20 Sep 2026 12:05:00 +0000
4+
Subject: [PATCH 2/4] clang: name the rv64gc/lp64d target ABI for JIT-compiled
5+
kernels
6+
7+
Upstream-Status: To upstream [no riscv64 runner upstream to regression-test it on]
8+
9+
warp-clang.so drives Clang through CompilerInvocation, i.e. at the cc1 level,
10+
where none of the driver's per-target defaults are applied. On RV64 that leaves
11+
clang::RISCVTargetInfo with a bare rv64i ISA (initFeatureMap sees no
12+
-target-feature flags) and, because the invocation names no ABI, an ABIFLen of
13+
zero -- the soft-float lp64 ABI.
14+
15+
warp.so, crt.cpp and warp-clang.so itself are compiled by the host gcc for
16+
rv64gc/lp64d, so every call a JIT-compiled kernel makes into them that passes or
17+
returns a float or a double -- _wp_isfinite(double) and the other crt.cpp shims
18+
-- would read the value from the wrong register class. The base ISA is wrong for
19+
the same reason: without M/A/F/D the frontend emits integer-multiply and
20+
soft-float libcalls for operations the hardware performs natively.
21+
22+
Named explicitly rather than derived from the triple, because
23+
LLVM_DEFAULT_TARGET_TRIPLE is whatever LLVM was configured with (build_llvm.py
24+
sets riscv64-pc-linux) and cc1 infers neither a RISC-V ISA nor an ABI from the
25+
triple's environment field.
26+
27+
Mirrors the x86_64 (+f16c) and aarch64 (+reserve-x28) blocks beside it.
28+
---
29+
warp/native/clang/clang.cpp | 18 ++++++++++++++++++
30+
1 file changed, 18 insertions(+)
31+
32+
diff --git a/warp/native/clang/clang.cpp b/warp/native/clang/clang.cpp
33+
index a9fc417..d2ca34f 100644
34+
--- a/warp/native/clang/clang.cpp
35+
+++ b/warp/native/clang/clang.cpp
36+
@@ -256,6 +256,24 @@ static std::unique_ptr<clang::CompilerInstance> create_compiler(
37+
args.push_back("+f16c");
38+
#endif
39+
40+
+#if defined(__riscv) && __riscv_xlen == 64
41+
+ // cc1 defaults RV64 to a bare rv64i ISA and, with no ABI named, to soft-float lp64.
42+
+ // warp.so and crt.cpp are compiled rv64gc/lp64d, so without this every kernel call
43+
+ // into them (_wp_isfinite(double), ...) would pass floats in the wrong registers.
44+
+ args.push_back("-target-abi");
45+
+ args.push_back("lp64d");
46+
+ args.push_back("-target-feature");
47+
+ args.push_back("+m");
48+
+ args.push_back("-target-feature");
49+
+ args.push_back("+a");
50+
+ args.push_back("-target-feature");
51+
+ args.push_back("+f");
52+
+ args.push_back("-target-feature");
53+
+ args.push_back("+d");
54+
+ args.push_back("-target-feature");
55+
+ args.push_back("+c");
56+
+#endif
57+
+
58+
#if defined(__aarch64__) || defined(_M_ARM64)
59+
if (tiles_in_stack_memory) {
60+
// Static memory support is broken on AArch64 CPUs. As a workaround we reserve some stack memory on kernel
61+
--
62+
2.43.0
63+
Lines changed: 167 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,167 @@
1+
From 7bd5afd2c143e7db533bafc04c17ef4555c45a4b Mon Sep 17 00:00:00 2001
2+
From: Ludovic Henry <git@ludovic.dev>
3+
Date: Mon, 21 Sep 2026 10:08:00 +0000
4+
Subject: [PATCH 3/4] Convert fp16 in software on RISC-V targets without Zfh
5+
6+
Upstream-Status: To upstream [no riscv64 runner upstream to regression-test it on]
7+
8+
float_to_half()/half_to_float() convert through Clang's native _Float16 in
9+
kernel code. RV64GC has no half-precision hardware, so Clang lowers both
10+
conversions to the compiler-rt libcalls __extendhfsf2/__truncsfhf2, and
11+
wp_load_obj() (warp/native/clang/clang.cpp) resolves a JIT-compiled module's
12+
externals from a curated CRT table that carries no compiler builtins. Every
13+
module holding an fp16 kernel therefore failed to materialize:
14+
15+
JIT session error: Symbols not found: [ __extendhfsf2, __truncsfhf2 ]
16+
Failed to lookup symbol: Failed to materialize symbols: ...
17+
RuntimeError: Failed to find forward kernel '...' for device 'cpu'
18+
19+
and took the module's non-fp16 kernels down with it -- 649 of the 1552 errors
20+
in a full riscv64 run of warp.tests, spread over test_codegen, test_print,
21+
test_spatial, test_transform, test_fabricarray and 40 more modules.
22+
23+
x86_64 sidesteps this by adding +f16c to the cc1 invocation, which lowers both
24+
conversions to vcvtph2ps/vcvtps2ph; aarch64 gets fcvt from armv8-a. RISC-V has
25+
no such baseline extension -- Zfh is optional, is not part of rv64gc, and
26+
asking for it would emit fcvt.h.s on hardware that need not implement it -- so
27+
convert in software instead.
28+
29+
The two routines are bit-for-bit what _Float16 yields where the hardware does
30+
have half support: round to nearest with ties to even, signalling NaNs quieted,
31+
NaN payloads preserved. Verified exhaustively against (_Float16) on x86_64
32+
(+f16c) over all 2^32 float bit patterns and all 2^16 half bit patterns, zero
33+
differences, so a riscv64 kernel rounds exactly as an x86_64 or aarch64 one
34+
does rather than picking up the ties-away rounding of the Giesen routine
35+
warp.cpp uses for the host-side conversions.
36+
37+
Guarded on !__riscv_zfh so a Zfh-enabled target keeps the native path.
38+
---
39+
warp/native/builtin.h | 105 ++++++++++++++++++++++++++++++++++++++++++
40+
1 file changed, 105 insertions(+)
41+
42+
diff --git a/warp/native/builtin.h b/warp/native/builtin.h
43+
index f0c23b5..b1ad4b7 100644
44+
--- a/warp/native/builtin.h
45+
+++ b/warp/native/builtin.h
46+
@@ -367,6 +367,109 @@ CUDA_CALLABLE inline float bfloat16_to_float(wp_bfloat16 x) { return wp_bfloat16
47+
48+
#elif defined(__clang__)
49+
50+
+#if defined(__riscv) && !defined(__riscv_zfh)
51+
+
52+
+// RISC-V without Zfh has no half-precision hardware, so Clang lowers every _Float16 conversion
53+
+// to the compiler-rt libcalls __extendhfsf2/__truncsfhf2. A JIT-compiled module resolves its
54+
+// externals from the curated CRT table in wp_load_obj() (warp/native/clang/clang.cpp), which
55+
+// carries no compiler builtins, so a module holding any fp16 kernel fails to materialize with
56+
+// "JIT session error: Symbols not found: [ __extendhfsf2, __truncsfhf2 ]".
57+
+//
58+
+// Convert in software instead. Both routines are bit-for-bit what _Float16 yields where the
59+
+// hardware does have half support -- round to nearest with ties to even, signalling NaNs
60+
+// quieted, payloads preserved -- verified exhaustively over all 2^32 floats and all 2^16
61+
+// halves, so kernels round exactly as they do on x86_64 (+f16c) and aarch64.
62+
+CUDA_CALLABLE inline half float_to_half(float x)
63+
+{
64+
+ unsigned int bits;
65+
+ memcpy(&bits, &x, sizeof(bits));
66+
+
67+
+ const unsigned int sign = (bits >> 16) & 0x8000u;
68+
+ const unsigned int magnitude = bits & 0x7fffffffu;
69+
+
70+
+ unsigned int u;
71+
+ if (magnitude >= 0x7f800000u) // Inf or NaN
72+
+ {
73+
+ const unsigned int mantissa = magnitude & 0x007fffffu;
74+
+ u = sign | 0x7c00u | (mantissa ? ((mantissa >> 13) | 0x0200u) : 0u); // NaN quieted, Inf kept
75+
+ }
76+
+ else if (magnitude >= 0x47800000u) // 65536 and above overflows the half range
77+
+ {
78+
+ u = sign | 0x7c00u;
79+
+ }
80+
+ else if (magnitude >= 0x38800000u) // 2^-14 and above is a normal half
81+
+ {
82+
+ u = sign | (((magnitude >> 23) - 127 + 15) << 10) | ((magnitude >> 13) & 0x03ffu);
83+
+ const unsigned int rest = magnitude & 0x1fffu; // the bits that do not fit
84+
+ if (rest > 0x1000u || (rest == 0x1000u && (u & 1u)))
85+
+ u += 1; // round to nearest, ties to even
86+
+ }
87+
+ else
88+
+ {
89+
+ const unsigned int exponent = magnitude >> 23;
90+
+ const unsigned int shift = 126 - exponent; // a subnormal half's ulp is 2^-24
91+
+ if (exponent == 0 || shift > 24)
92+
+ {
93+
+ u = sign; // underflows to zero
94+
+ }
95+
+ else
96+
+ {
97+
+ const unsigned int mantissa = (magnitude & 0x007fffffu) | 0x00800000u; // implicit bit
98+
+ const unsigned int rest = mantissa & ((1u << shift) - 1u);
99+
+ const unsigned int tie = 1u << (shift - 1);
100+
+ u = sign | (mantissa >> shift);
101+
+ if (rest > tie || (rest == tie && (u & 1u)))
102+
+ u += 1; // round to nearest, ties to even
103+
+ }
104+
+ }
105+
+
106+
+ half h;
107+
+ h.u = static_cast<unsigned short>(u);
108+
+ return h;
109+
+}
110+
+
111+
+CUDA_CALLABLE inline float half_to_float(half h)
112+
+{
113+
+ const unsigned int sign = (static_cast<unsigned int>(h.u) & 0x8000u) << 16;
114+
+ const unsigned int exponent = (static_cast<unsigned int>(h.u) >> 10) & 0x001fu;
115+
+ const unsigned int mantissa = static_cast<unsigned int>(h.u) & 0x03ffu;
116+
+
117+
+ unsigned int bits;
118+
+ if (exponent == 0x1fu) // Inf or NaN
119+
+ {
120+
+ bits = sign | 0x7f800000u | (mantissa ? ((mantissa << 13) | 0x00400000u) : 0u);
121+
+ }
122+
+ else if (exponent == 0u) // zero or subnormal
123+
+ {
124+
+ if (mantissa == 0u)
125+
+ {
126+
+ bits = sign;
127+
+ }
128+
+ else
129+
+ {
130+
+ // Renormalize: shift the mantissa up until its leading one leaves the field.
131+
+ unsigned int m = mantissa;
132+
+ unsigned int e = 127 - 15 + 1;
133+
+ while ((m & 0x0400u) == 0u)
134+
+ {
135+
+ m <<= 1;
136+
+ e -= 1;
137+
+ }
138+
+ bits = sign | (e << 23) | ((m & 0x03ffu) << 13);
139+
+ }
140+
+ }
141+
+ else
142+
+ {
143+
+ bits = sign | ((exponent + 127 - 15) << 23) | (mantissa << 13);
144+
+ }
145+
+
146+
+ float val;
147+
+ memcpy(&val, &bits, sizeof(val));
148+
+ return val;
149+
+}
150+
+
151+
+#else
152+
+
153+
// _Float16 is Clang's native half-precision floating-point type
154+
CUDA_CALLABLE inline half float_to_half(float x)
155+
{
156+
@@ -381,6 +484,8 @@ CUDA_CALLABLE inline float half_to_float(half h)
157+
return static_cast<float>(f16);
158+
}
159+
160+
+#endif // __riscv && !__riscv_zfh
161+
+
162+
#ifndef WP_NO_BFLOAT16
163+
CUDA_CALLABLE inline wp_bfloat16 float_to_bfloat16(float x)
164+
{
165+
--
166+
2.43.0
167+

0 commit comments

Comments
 (0)