From 9dff2d97328906b43b30b28d1786af68accf0c56 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Thu, 6 Aug 2026 01:53:10 +0700 Subject: [PATCH 01/42] Support AMD CPU and Intel GPU --- CMakeLists.txt | 18 ++- README.md | 49 +++++--- docs/backend/sycl.md | 229 +++++++++++++++++++++++++++++++++++++ src/backend.h | 147 ++++++++++++++++++++++++ src/models/evo1.cpp | 27 ++--- src/models/gr00tn1d5.cpp | 27 ++--- src/models/gr00tn1d6.cpp | 27 ++--- src/models/gr00tn1d7.cpp | 27 ++--- src/models/openvla_oft.cpp | 24 +--- src/models/pi0.cpp | 27 ++--- src/models/pi05.cpp | 27 ++--- src/models/smolvla.cpp | 40 ++----- src/models/vla_adapter.cpp | 24 +--- src/models/vla_jepa.cpp | 26 ++--- 14 files changed, 494 insertions(+), 225 deletions(-) create mode 100644 docs/backend/sycl.md create mode 100644 src/backend.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 7e757e1..623a6fc 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -88,7 +88,23 @@ if(GGML_CUDA) target_include_directories(vla_core PUBLIC ${CUDAToolkit_INCLUDE_DIRS}) endif() -if(GGML_METAL AND NOT GGML_CUDA) +# Intel GPUs (Arc / Flex / Data Center Max / Xe iGPU) through oneAPI SYCL. +# ggml's SYCL sources only compile under the oneAPI DPC++ driver, and +# CMAKE_CXX_COMPILER is global, so our targets are built by icpx too. Fail +# loudly here rather than let ggml die deep in a kernel compile. +if(GGML_SYCL AND NOT GGML_CUDA) + if(NOT CMAKE_CXX_COMPILER_ID STREQUAL "IntelLLVM") + message(FATAL_ERROR + "GGML_SYCL=ON needs the oneAPI DPC++ compiler, but CMAKE_CXX_COMPILER is " + "'${CMAKE_CXX_COMPILER_ID}'. Source /opt/intel/oneapi/setvars.sh and configure a " + "fresh build dir with -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx. " + "See docs/backend/sycl.md.") + endif() + target_compile_definitions(vla_core PUBLIC GGML_USE_SYCL) +endif() + +# Backend precedence matches the ladder in src/backend.h: CUDA, then SYCL, then Metal. +if(GGML_METAL AND NOT GGML_CUDA AND NOT GGML_SYCL) target_compile_definitions(vla_core PUBLIC GGML_USE_METAL) endif() diff --git a/README.md b/README.md index 58f4924..f009399 100644 --- a/README.md +++ b/README.md @@ -11,8 +11,8 @@ A C++ inference engine for **Vision-Language-Action (VLA) models**, built on [`llama.cpp`](https://github.com/ggml-org/llama.cpp). It runs the open VLA policies - SmolVLA, π0, BitVLA, Evo-1, GR00T N1.5/1.6/1.7 and more - under one runtime, each packaged as a single self-contained GGUF that needs no Python or -PyTorch at inference time. The binaries drive robots on **CPU**, **Apple Silicon**, or -**CUDA**, from consumer GPUs down to Jetson-class boards. +PyTorch at inference time. The binaries drive robots on **CPU**, **Apple Silicon**, **CUDA** - +from consumer GPUs down to Jetson-class boards - or **Intel GPUs** via SYCL. [**Learn vla.cpp**](https://fai-modelopt-tech.github.io/learn-vla-cpp/) walks through the engine design and how each policy is implemented on ggml. @@ -24,7 +24,9 @@ PyTorch at inference time. The binaries drive robots on **CPU**, **Apple Silicon - CMake ≥ 3.22 - A C++17 compiler (GCC 11+ or Clang 14+) -- CUDA 12.x (optional - required only for GPU builds) +- CUDA 12.x (optional - required only for CUDA GPU builds) +- Intel oneAPI 2025.x + GPU compute runtime (optional - only for Intel GPU + builds, see [docs/backend/sycl.md](docs/backend/sycl.md)) - `libzmq3-dev`, `libprotobuf-dev`, `protobuf-compiler` ```bash @@ -60,6 +62,21 @@ cmake -B build \ cmake --build build -j$(nproc) ``` +```bash +# Intel GPU build (Arc / Flex / Max / Xe iGPU). ggml's SYCL sources need the +# oneAPI DPC++ driver, so the whole project is compiled by icpx: +source /opt/intel/oneapi/setvars.sh +cmake -B build \ + -DGGML_SYCL=ON \ + -DCMAKE_C_COMPILER=icx \ + -DCMAKE_CXX_COMPILER=icpx \ + -DCMAKE_BUILD_TYPE=Release +cmake --build build -j$(nproc) +``` + +The driver and oneAPI setup that this needs is in +[docs/backend/sycl.md](docs/backend/sycl.md). + If CMake cannot find CUDA, point the environment at it explicitly: ```bash @@ -248,19 +265,19 @@ and an **Apple M4**. Support matrix of models (rows) against platforms (columns). Legend: `Y` = supported (released and benchmarked), `~` = in progress, `-` = planned. -| Model | CPU (x86-64 / ARM) | CUDA | Metal | OpenVINO | Hexagon | -|---|:--:|:--:|:--:|:--:|:--:| -| [SmolVLA](https://hf.co/vrfai/smolvla-libero-gguf) | Y | Y | Y | - | - | -| [π0](https://hf.co/vrfai/pi0-libero-finetuned-v044-gguf) | Y | Y | Y | - | - | -| [π0.5](https://hf.co/vrfai/pi05-libero-gguf) | Y | Y | ~ | - | - | -| [GR00T N1.5](https://hf.co/vrfai/gr00tn1d5-libero-object-gguf) | Y | Y | ~ | - | - | -| [GR00T N1.6](https://hf.co/vrfai/gr00tn1d6-libero-gguf) | Y | Y | ~ | - | - | -| [GR00T N1.7](https://hf.co/vrfai/gr00tn1d7-libero-gguf) | Y | Y | Y | - | - | -| [BitVLA](https://hf.co/vrfai/bitvla-libero-gguf) | Y | Y | ~ | - | - | -| [Evo-1](https://hf.co/vrfai/evo1-libero-gguf) | Y | Y | ~ | - | - | -| [VLA-Adapter](https://hf.co/vrfai/vla-adapter-libero-gguf) | Y | Y | ~ | - | - | -| [OpenVLA-OFT](https://hf.co/vrfai/openvla-oft-libero-gguf) | Y | Y | ~ | - | - | -| [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | ~ | - | - | +| Model | CPU (x86-64 / ARM) | CUDA | SYCL (Intel) | Metal | OpenVINO | Hexagon | +|---|:--:|:--:|:--:|:--:|:--:|:--:| +| [SmolVLA](https://hf.co/vrfai/smolvla-libero-gguf) | Y | Y | Y | Y | - | - | +| [π0](https://hf.co/vrfai/pi0-libero-finetuned-v044-gguf) | Y | Y | - | Y | - | - | +| [π0.5](https://hf.co/vrfai/pi05-libero-gguf) | Y | Y | - | ~ | - | - | +| [GR00T N1.5](https://hf.co/vrfai/gr00tn1d5-libero-object-gguf) | Y | Y | - | ~ | - | - | +| [GR00T N1.6](https://hf.co/vrfai/gr00tn1d6-libero-gguf) | Y | Y | - | ~ | - | - | +| [GR00T N1.7](https://hf.co/vrfai/gr00tn1d7-libero-gguf) | Y | Y | - | Y | - | - | +| [BitVLA](https://hf.co/vrfai/bitvla-libero-gguf) | Y | Y | - | ~ | - | - | +| [Evo-1](https://hf.co/vrfai/evo1-libero-gguf) | Y | Y | Y | ~ | - | - | +| [VLA-Adapter](https://hf.co/vrfai/vla-adapter-libero-gguf) | Y | Y | ~ | ~ | - | - | +| [OpenVLA-OFT](https://hf.co/vrfai/openvla-oft-libero-gguf) | Y | Y | - | ~ | - | - | +| [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | - | ~ | - | - | --- diff --git a/docs/backend/sycl.md b/docs/backend/sycl.md new file mode 100644 index 0000000..1dc9e9b --- /dev/null +++ b/docs/backend/sycl.md @@ -0,0 +1,229 @@ +# `vla.cpp` on Intel GPUs (SYCL backend) + +Notes for building and running `vla.cpp` on Intel discrete and integrated GPUs +through oneAPI SYCL. Unlike Metal, SYCL is **not** auto-detected: it needs the +oneAPI DPC++ compiler and an explicit `-DGGML_SYCL=ON`. + +Verified on an **Intel Arc A380** (DG2 / Xe-HPG, `8086:56a5`, 6 GB) on Ubuntu +22.04, kernel 6.8, with an AMD Ryzen host CPU. The same path covers the rest of +the Arc A/B series, Flex, Data Center Max, and the Xe iGPUs. + +## Prerequisites + +### 1. GPU compute runtime + +The kernel driver (`i915`, in-tree since 6.2 for DG2) is not enough - you also +need the userspace compute stack: Level Zero plus the NEO OpenCL runtime. + +```bash +wget -qO- https://repositories.intel.com/gpu/intel-graphics.key \ + | sudo gpg --yes --dearmor -o /usr/share/keyrings/intel-graphics.gpg +echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] https://repositories.intel.com/gpu/ubuntu jammy client" \ + | sudo tee /etc/apt/sources.list.d/intel-gpu-jammy.list +sudo apt-get update +sudo apt-get install -y intel-opencl-icd libze-intel-gpu1 libze1 libze-dev intel-ocloc clinfo +``` + +Substitute your distro codename for `jammy`. Install the userspace packages +only - do **not** add `intel-i915-dkms` on a 6.8+ kernel, whose in-tree `i915` +already drives DG2. + +Then give your user access to the render node and re-login: + +```bash +sudo usermod -aG render,video "$USER" +``` + +Check it before going further - `clinfo -l` must name your GPU: + +``` +Platform #0: Intel(R) OpenCL Graphics + `-- Device #0: Intel(R) Arc(TM) A380 Graphics +``` + +### 2. oneAPI + +The SYCL backend needs the DPC++ compiler, oneMKL and oneDNN. **Deep Learning +Essentials** carries exactly those and is much smaller than the full Base +Toolkit. + +```bash +wget -qO- https://apt.repos.intel.com/intel-gpg-keys/GPG-PUB-KEY-INTEL-SW-PRODUCTS.PUB \ + | sudo gpg --yes --dearmor -o /usr/share/keyrings/oneapi-archive-keyring.gpg +echo "deb [signed-by=/usr/share/keyrings/oneapi-archive-keyring.gpg] https://apt.repos.intel.com/oneapi all main" \ + | sudo tee /etc/apt/sources.list.d/oneAPI.list +sudo apt-get update +sudo apt-get install -y intel-deep-learning-essentials-2025.3 +``` + +2025.3 is the newest release verified by llama.cpp's own SYCL docs that still +supports Ubuntu 22.04; the 2026.x series dropped jammy. Confirm the toolchain +sees the GPU over Level Zero: + +```bash +source /opt/intel/oneapi/setvars.sh +sycl-ls +``` + +``` +[level_zero:gpu][level_zero:0] Intel(R) oneAPI Unified Runtime over Level-Zero, Intel(R) Arc(TM) A380 Graphics 12.56.5 [1.6.31294+20] +[opencl:gpu][opencl:1] Intel(R) OpenCL Graphics, Intel(R) Arc(TM) A380 Graphics OpenCL 3.0 NEO [24.39.31294] +``` + +Plus the usual host dependencies. Ubuntu 22.04 has no `cppzmq` package, so drop +its two headers in by hand: + +```bash +sudo apt-get install -y cmake ninja-build pkg-config \ + protobuf-compiler libprotobuf-dev libzmq3-dev +wget -q https://github.com/zeromq/cppzmq/archive/refs/tags/v4.10.0.tar.gz -O - | tar xz +sudo install -m644 cppzmq-4.10.0/zmq.hpp cppzmq-4.10.0/zmq_addon.hpp /usr/local/include/ +``` + +## Configure & build + +ggml's SYCL sources only compile under the oneAPI DPC++ driver, and +`CMAKE_CXX_COMPILER` is global, so the whole project - `vla_core`, the servers, +the CLI - is built by `icpx`. Configure a **fresh** build directory; switching +compilers in an existing one does not work. + +```bash +source /opt/intel/oneapi/setvars.sh + +cmake -B build-sycl -G Ninja \ + -DCMAKE_BUILD_TYPE=Release \ + -DGGML_SYCL=ON \ + -DCMAKE_C_COMPILER=icx \ + -DCMAKE_CXX_COMPILER=icpx +cmake --build build-sycl -j$(nproc) +``` + +`setvars.sh` must be sourced in every shell that builds *or runs* the binaries - +`libsycl.so`, `libdnnl.so` and the oneMKL libraries live under `/opt/intel`. + +## GPU offload + +The core picks its backend at load time. Confirm from the startup banner: + +``` +vla: backend = SYCL (device 0: Intel(R) Arc(TM) A380 Graphics) +``` + +If you see `vla: backend = CPU (8 threads)` instead, the build did not pick up +SYCL, or no SYCL device was visible - re-check `sycl-ls` and your `render` group +membership. + +On a multi-GPU box, `VLA_DEVICE=` selects the ordinal (the same variable +selects the CUDA device). The index is range-checked against the SYCL device +count; an out-of-range value logs and falls back to CPU rather than running off +the end of the device array. + +> Single-backend, no per-op CPU fallback: the core drives one backend through +> `gallocr`, not a scheduler. An arch that hits an op the SYCL backend does not +> implement asserts at predict time rather than silently falling back. + +BitVLA is the one exception: it pins its ggml graph to the CPU backend by design +and offloads its LM through separate hand-written CUDA kernels, so a SYCL build +leaves it on the CPU. There is no SYCL port of those kernels. + +## Known issue: the SYCL VMM pool and oneDNN + +ggml-sycl's VMM pool hands out virtual-memory-backed pointers that oneDNN cannot +wrap in a `dnnl::memory`. When it happens the GEMM aborts the process: + +``` +could not create a memory object +SYCL error: ... in function ggml_sycl_op_mul_mat at .../ggml-sycl.cpp:3055 +``` + +It fires whenever `src0` is not already F32 - BF16, F16 and every quantized type +are converted into that pool before the GEMM - which is most checkpoints, +including the default BF16 weights of SmolVLA, π0, π0.5, Evo-1, VLA-Adapter and +OpenVLA-OFT. + +`vla.cpp` defaults `GGML_SYCL_ENABLE_VMM=0` when it brings up SYCL, which avoids +it and is the faster of the two workarounds (disabling oneDNN with +`GGML_SYCL_DISABLE_DNN=1` also clears the crash, but costs ~8%). It is only a +default: set `GGML_SYCL_ENABLE_VMM=1` explicitly to keep the pool on hardware +where it pays off. + +## Known issue: `bf16 -> f32` copies + +ggml-sycl's copy table has `f16 -> f32` but no `bf16 -> f32`, so an arch whose +graph contains that copy aborts at predict time: + +``` +ggml_sycl_cpy: unsupported type combination (bf16 to f32) +.../ggml-sycl/cpy.cpp:592: fatal error +``` + +VLA-Adapter hits this with its default BF16 weights (OpenVLA-OFT shares the same +graph shape and is expected to as well). Switching that arch to F32 weights +removes the BF16 tensor from the graph and it runs: + +```bash +VLA_ADAPTER_F32_WEIGHTS=1 ./build-sycl/vla-cli --ckpt ... +``` + +SmolVLA and Evo-1 are unaffected and run on their default BF16 weights. + +## Performance note: F32 weights + +BF16 has no native DPAS path on Xe-HPG, so BF16 weights are slower there than +plain F32 despite the extra bandwidth. Each arch exposes a switch +(`VLA_WEIGHT_DTYPE=f32` for SmolVLA, `VLA_PI0_F32_WEIGHTS=1` for π0, and so on), +and on the A380 it is worth ~16%: + +| SmolVLA weights | vision | inference | total | +|---|---:|---:|---:| +| BF16 (default) | 158 ms | 474 ms | **630 ms** | +| F32 (`VLA_WEIGHT_DTYPE=f32`) | 173 ms | 355 ms | **528 ms** | + +The tradeoff is memory - F32 doubles the resident weights (1.07 GiB -> 2.09 GiB +for SmolVLA), which matters on a 6 GB A380 for the larger checkpoints. The +default stays BF16 for that reason. + +## Results + +Measured with `vla_predict_check` (fixed noise, so runs are comparable), best of +5-10 iterations after 3 warmups. Host is an AMD Ryzen 5 5500 (CPU backend uses 8 +threads); GPU is the Arc A380. + +| Model | input | CPU | Arc A380 | speedup | +|---|---|---:|---:|---:| +| SmolVLA | 512 | 1,920 ms | **630 ms** | 3.0x | +| Evo-1 | 448 | 7,695 ms | **1,176 ms** | 6.5x | +| VLA-Adapter | 224 | 2,994 ms | **517 ms** | 5.8x | + +VLA-Adapter is measured with `VLA_ADAPTER_F32_WEIGHTS=1` on both sides (see the +`bf16 -> f32` issue above); the others run their stock defaults. + +Per-stage for SmolVLA: + +| Stage | CPU | Arc A380 (SYCL) | +|--------------|-------------:|------------------:| +| vision | 1,119 ms | 158 ms | +| inference | 804 ms | 474 ms | +| **total/req**| **1,920 ms**| **630 ms** | + +SmolVLA gains least because its flow-matching denoise loop is a long chain of +small GEMMs that cannot fill 128 EUs; its vision tower alone is 7.1x. With +`VLA_WEIGHT_DTYPE=f32` it reaches 528 ms (3.6x). + +Outputs were checked against the CPU backend on every model above: max absolute +deviation 2.9e-3 on actions peaking at 0.99 (2.9e-6 for the all-F32 +VLA-Adapter run), RMS 2.4e-4 - BF16/F32 kernel rounding, not a numerical +regression. + +### Memory ceiling + +The A380 has 6 GB, of which ~5.7 GB is addressable. GR00T N1.7 (6.3 GB of F32 +weights) does not fit and dies in the allocator: + +``` +level_zero backend failed with error: 38 (UR_RESULT_ERROR_OUT_OF_HOST_MEMORY) +``` + +`VLA_GR00T_BF16_WEIGHTS=1` halves the weights but its activations still overflow +the card. There is no host-memory spill path - the core is single-backend - so +the larger checkpoints need an A770/B580-class card or better. diff --git a/src/backend.h b/src/backend.h new file mode 100644 index 0000000..346fe32 --- /dev/null +++ b/src/backend.h @@ -0,0 +1,147 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +/** + * @file backend.h + * @brief Compute-backend selection shared by every in-tree arch. + * + * Each arch used to open-code the same accelerator-then-CPU ladder, so adding a + * backend meant editing a dozen files. They all call @ref vla::backend_init + * instead; the ladder lives here once. + * + * Exactly one accelerator is compiled in, picked by the CMake flag that was + * used (`GGML_CUDA` / `GGML_SYCL` / `GGML_METAL`). There is no per-op CPU + * fallback: the core drives a single backend through `gallocr` rather than a + * scheduler, so an arch that hits an op the backend does not implement asserts + * at predict time instead of silently limping. + */ + +#pragma once + +#include "ggml.h" +#include "ggml-backend.h" +#include "ggml-cpu.h" + +#ifdef GGML_USE_CUDA +#include "ggml-cuda.h" +#endif +#ifdef GGML_USE_SYCL +#include "ggml-sycl.h" +#endif +#ifdef GGML_USE_METAL +#include "ggml-metal.h" +#endif + +#include +#include +#ifdef GGML_USE_SYCL +#include // setenv +#endif + +namespace vla { + +/// Outcome of @ref backend_init. @c handle is null only if even the CPU backend +/// failed to come up, which callers treat as a fatal load error. +struct Backend { + ggml_backend_t handle = nullptr; + bool is_cuda = false; + bool is_gpu = false; +}; + +/// GPU ordinal for the multi-device backends (CUDA, SYCL); `VLA_DEVICE` overrides. +inline int backend_device_index() { + const char * e = std::getenv("VLA_DEVICE"); + if (!e || !*e) return 0; + const int idx = std::atoi(e); + return idx > 0 ? idx : 0; +} + +/** + * @brief Bring up the best compute backend available to this build. + * + * @param tag Log prefix identifying the arch, e.g. @c "vla(pi0)". + * @param n_threads Thread count handed to the CPU backend if it is used. + * @return The backend plus the flags the arch records about it. + */ +inline Backend backend_init(const char * tag, int n_threads) { + Backend b; + +#ifdef GGML_USE_CUDA + { + const int dev = backend_device_index(); + b.handle = ggml_backend_cuda_init(dev); + if (b.handle) { + b.is_cuda = true; + b.is_gpu = true; + std::printf("%s: backend = CUDA (device %d)\n", tag, dev); + } else { + std::fprintf(stderr, "%s: ggml_backend_cuda_init failed; falling back to CPU\n", tag); + } + } +#elif defined(GGML_USE_SYCL) + { + // ggml-sycl's VMM pool hands out virtual-memory-backed pointers that + // oneDNN cannot wrap in a dnnl::memory: the GEMM aborts the process with + // "could not create a memory object". Any src0 that is not already F32 + // (BF16, F16, and every quantized type) is converted into that pool + // first, so the crash hits most checkpoints. Turning the pool off is + // also the faster of the two workarounds -- measurably better than + // disabling oneDNN outright. Only a default: an explicit setting wins, + // for Intel GPUs where the pool is worth keeping. + // ggml reads this on the first SYCL entry point, so it must be set here. + setenv("GGML_SYCL_ENABLE_VMM", "0", /*overwrite=*/0); + + // ggml_backend_sycl_init() guards the device index with assert(), which + // a Release build compiles out and then indexes past the device array. + // Range-check here so a SYCL build on a box with no Intel GPU (or a bad + // VLA_DEVICE) lands on CPU instead of corrupting memory. + const int dev = backend_device_index(); + const int n_dev = ggml_backend_sycl_get_device_count(); + if (dev >= n_dev) { + std::fprintf(stderr, "%s: SYCL device %d out of range (%d visible); falling back to CPU\n", + tag, dev, n_dev); + } else if ((b.handle = ggml_backend_sycl_init(dev)) != nullptr) { + b.is_gpu = true; + char desc[256] = { 0 }; + ggml_backend_sycl_get_device_description(dev, desc, sizeof(desc)); + std::printf("%s: backend = SYCL (device %d: %s)\n", tag, dev, desc); + } else { + std::fprintf(stderr, "%s: ggml_backend_sycl_init failed; falling back to CPU\n", tag); + } + } +#elif defined(GGML_USE_METAL) + { + b.handle = ggml_backend_metal_init(); + if (b.handle) { + b.is_gpu = true; + std::printf("%s: backend = Metal\n", tag); + } else { + std::fprintf(stderr, "%s: ggml_backend_metal_init failed; falling back to CPU\n", tag); + } + } +#endif + + if (!b.handle) { + b.handle = ggml_backend_cpu_init(); + if (!b.handle) { + std::fprintf(stderr, "%s: ggml_backend_cpu_init failed\n", tag); + return b; + } + ggml_backend_cpu_set_n_threads(b.handle, n_threads); + std::printf("%s: backend = CPU (%d threads)\n", tag, n_threads); + } + return b; +} + +} // namespace vla diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 77e169c..0f62af0 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -18,12 +18,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -288,20 +283,12 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, (long long) m->dit_layers, (long long) m->dit_heads, (long long) m->horizon, (long long) m->per_a, (long long) m->num_steps, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(evo1): backend = CUDA (device 0)\n"); } - else std::fprintf(stderr, "vla(evo1): ggml_backend_cuda_init failed; falling back to CPU\n"); -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(evo1): backend = Metal\n"); } - else std::fprintf(stderr, "vla(evo1): ggml_backend_metal_init failed; falling back to CPU\n"); -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(evo1): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(evo1): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(evo1)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 6c20589..1255626 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -18,12 +18,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -303,20 +298,12 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->num_steps, (long long) m->embodiment_id, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(gr00tn1d5): backend = CUDA (device 0)\n"); } - else std::fprintf(stderr, "vla(gr00tn1d5): ggml_backend_cuda_init failed; falling back to CPU\n"); -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(gr00tn1d5): backend = Metal\n"); } - else std::fprintf(stderr, "vla(gr00tn1d5): ggml_backend_metal_init failed; falling back to CPU\n"); -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(gr00tn1d5): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(gr00tn1d5): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(gr00tn1d5)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 7017d9a..ea6ccc0 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -18,12 +18,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -308,20 +303,12 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->max_state_dim, (long long) m->num_steps, (long long) m->embodiment_id, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(gr00tn1d6): backend = CUDA (device 0)\n"); } - else std::fprintf(stderr, "vla(gr00tn1d6): ggml_backend_cuda_init failed; falling back to CPU\n"); -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(gr00tn1d6): backend = Metal\n"); } - else std::fprintf(stderr, "vla(gr00tn1d6): ggml_backend_metal_init failed; falling back to CPU\n"); -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(gr00tn1d6): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(gr00tn1d6)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index ca107db..ef9f756 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -18,12 +18,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -470,20 +465,12 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->max_state_dim, (long long) m->num_steps, (long long) m->embodiment_id, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(gr00tn1d7): backend = CUDA (device 0)\n"); } - else std::fprintf(stderr, "vla(gr00tn1d7): ggml_backend_cuda_init failed; falling back to CPU\n"); -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(gr00tn1d7): backend = Metal\n"); } - else std::fprintf(stderr, "vla(gr00tn1d7): ggml_backend_metal_init failed; falling back to CPU\n"); -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(gr00tn1d7): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(gr00tn1d7): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(gr00tn1d7)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index c91d6ca..4dfb333 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -19,12 +19,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -203,18 +198,11 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path std::printf("vla(openvla_oft): unnorm suite = %s (q99 dim %zu)\n", m->suite.c_str(), m->q99.size()); } -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_gpu=true; std::printf("vla(openvla_oft): backend = CUDA (device 0)\n"); } -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu=true; std::printf("vla(openvla_oft): backend = Metal\n"); } -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(openvla_oft): cpu backend init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(openvla_oft): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(openvla_oft)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 0c38116..16b78c3 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -19,12 +19,7 @@ #include "ggml-cpu.h" #include "ggml-backend.h" #include "ggml-alloc.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -354,24 +349,16 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, cfg.num_steps, (long long) cfg.real_state_dim, (long long) cfg.real_action_dim, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init( 0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(pi0): backend = CUDA (device 0)\n"); } - else { std::fprintf(stderr, "vla(pi0): ggml_backend_cuda_init failed; falling back to CPU\n"); } -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(pi0): backend = Metal\n"); } - else { std::fprintf(stderr, "vla(pi0): ggml_backend_metal_init failed; falling back to CPU\n"); } -#endif { const unsigned hw = std::thread::hardware_concurrency(); m->n_threads = (hw == 0) ? 4 : (int) std::min(hw, 8u); } - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(pi0): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(pi0): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(pi0)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 9a30087..2f2db1d 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -19,12 +19,7 @@ #include "ggml-cpu.h" #include "ggml-backend.h" #include "ggml-alloc.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -435,24 +430,16 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, (long long) cfg.n_lang, (long long) m->adarms_cond_dim, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init( 0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(pi05): backend = CUDA (device 0)\n"); } - else { std::fprintf(stderr, "vla(pi05): ggml_backend_cuda_init failed; falling back to CPU\n"); } -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(pi05): backend = Metal\n"); } - else { std::fprintf(stderr, "vla(pi05): ggml_backend_metal_init failed; falling back to CPU\n"); } -#endif { const unsigned hw = std::thread::hardware_concurrency(); m->n_threads = (hw == 0) ? 4 : (int) std::min(hw, 8u); } - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(pi05): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(pi05): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(pi05)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 357e00d..8a9b396 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -23,12 +23,7 @@ #include "ggml-backend.h" #include "ggml-cpu.h" #include "gguf.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "nlohmann/json.hpp" @@ -927,33 +922,12 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, std::printf("vla: config = %s\n", cfg_path.c_str()); } -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init( 0); - if (m->backend) { - m->is_cuda = true; - m->is_gpu = true; - std::printf("vla: backend = CUDA (device 0)\n"); - } else { - std::fprintf(stderr, "vla: ggml_backend_cuda_init failed; falling back to CPU\n"); - } -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { - m->is_gpu = true; - std::printf("vla: backend = Metal\n"); - } else { - std::fprintf(stderr, "vla: ggml_backend_metal_init failed; falling back to CPU\n"); - } -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { - std::fprintf(stderr, "vla: ggml_backend_cpu_init failed\n"); - delete m; - return nullptr; - } - ggml_backend_cpu_set_n_threads(m->backend, default_cpu_threads()); - std::printf("vla: backend = CPU (%d threads)\n", default_cpu_threads()); + { + const Backend b = backend_init("vla", default_cpu_threads()); + if (!b.handle) { delete m; return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } vram_probe(m->backend, "after backend init"); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index fca8d0d..db84fe2 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -19,12 +19,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -217,18 +212,11 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path std::printf("vla(vla_adapter): unnorm suite = %s (q99 dim %zu)\n", m->suite.c_str(), m->q99.size()); } -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_gpu=true; std::printf("vla(vla_adapter): backend = CUDA (device 0)\n"); } -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu=true; std::printf("vla(vla_adapter): backend = Metal\n"); } -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(vla_adapter): cpu backend init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(vla_adapter): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(vla_adapter)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 6622c55..d8c25fb 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -18,12 +18,7 @@ #include "ggml.h" #include "ggml-cpu.h" #include "ggml-backend.h" -#ifdef GGML_USE_CUDA -#include "ggml-cuda.h" -#endif -#ifdef GGML_USE_METAL -#include "ggml-metal.h" -#endif +#include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" @@ -349,19 +344,12 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, (long long) m->action_horizon, (long long) m->action_dim, (long long) m->state_dim, (long long) m->num_future, (long long) m->num_steps, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); -#ifdef GGML_USE_CUDA - m->backend = ggml_backend_cuda_init(0); - if (m->backend) { m->is_cuda = true; m->is_gpu = true; std::printf("vla(vla_jepa): backend = CUDA (device 0)\n"); } - else std::fprintf(stderr, "vla(vla_jepa): ggml_backend_cuda_init failed; falling back to CPU\n"); -#elif defined(GGML_USE_METAL) - m->backend = ggml_backend_metal_init(); - if (m->backend) { m->is_gpu = true; std::printf("vla(vla_jepa): backend = Metal\n"); } -#endif - if (!m->backend) { - m->backend = ggml_backend_cpu_init(); - if (!m->backend) { std::fprintf(stderr, "vla(vla_jepa): ggml_backend_cpu_init failed\n"); return nullptr; } - ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); - std::printf("vla(vla_jepa): backend = CPU (%d threads)\n", m->n_threads); + { + const Backend b = backend_init("vla(vla_jepa)", m->n_threads); + if (!b.handle) { return nullptr; } + m->backend = b.handle; + m->is_cuda = b.is_cuda; + m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; From ae2882adf036a516ebe12c8da06d94ff4a604312 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 13:26:25 +0700 Subject: [PATCH 02/42] fix backend init edge cases --- CMakeLists.txt | 17 +++++++++++++++++ docs/backend/sycl.md | 5 +++-- src/backend.h | 33 +++++++++++++++++++++++++++++---- src/models/bitvla.cpp | 3 +++ 4 files changed, 52 insertions(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 623a6fc..4898660 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -11,6 +11,23 @@ if(NOT CMAKE_BUILD_TYPE) set(CMAKE_BUILD_TYPE Release CACHE STRING "" FORCE) endif() +# One accelerator per build: src/backend.h compiles in exactly one, so two +# GGML_* flags would give ggml both backends and vla_core only the higher +# priority one. Reject it here, before FetchContent configures ggml. +set(_vla_accel "") +foreach(_flag GGML_CUDA GGML_SYCL GGML_METAL) + if(${_flag}) + list(APPEND _vla_accel ${_flag}) + endif() +endforeach() +list(LENGTH _vla_accel _vla_accel_n) +if(_vla_accel_n GREATER 1) + string(REPLACE ";" ", " _vla_accel_str "${_vla_accel}") + message(FATAL_ERROR + "Enable one accelerator backend at a time; got ${_vla_accel_str}. " + "Configure a separate build directory per backend.") +endif() + set(LLAMA_BUILD_COMMON ON CACHE BOOL "" FORCE) set(LLAMA_BUILD_TOOLS ON CACHE BOOL "" FORCE) set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE) diff --git a/docs/backend/sycl.md b/docs/backend/sycl.md index 1dc9e9b..ee2df7e 100644 --- a/docs/backend/sycl.md +++ b/docs/backend/sycl.md @@ -185,8 +185,9 @@ default stays BF16 for that reason. ## Results -Measured with `vla_predict_check` (fixed noise, so runs are comparable), best of -5-10 iterations after 3 warmups. Host is an AMD Ryzen 5 5500 (CPU backend uses 8 +Measured with `vla_predict_check`, which is a test target - add +`-DVLA_BUILD_TESTS=ON` to the configure line above to get it. Fixed noise, so +runs are comparable; best of 5-10 iterations after 3 warmups. Host is an AMD Ryzen 5 5500 (CPU backend uses 8 threads); GPU is the Arc A380. | Model | input | CPU | Arc A380 | speedup | diff --git a/src/backend.h b/src/backend.h index 346fe32..d60a198 100644 --- a/src/backend.h +++ b/src/backend.h @@ -46,11 +46,26 @@ #include #include #ifdef GGML_USE_SYCL -#include // setenv +#include // setenv / _putenv_s +#include #endif namespace vla { +#ifdef GGML_USE_SYCL +/// setenv is POSIX; MSVC and the Windows oneAPI toolchain only have _putenv_s, +/// which has no "do not overwrite" mode, so check first. +inline void setenv_default(const char * key, const char * val) { +#ifdef _WIN32 + size_t len = 0; + if (getenv_s(&len, nullptr, 0, key) == 0 && len > 0) return; + _putenv_s(key, val); +#else + setenv(key, val, /*overwrite=*/0); +#endif +} +#endif + /// Outcome of @ref backend_init. @c handle is null only if even the CPU backend /// failed to come up, which callers treat as a fatal load error. struct Backend { @@ -60,11 +75,18 @@ struct Backend { }; /// GPU ordinal for the multi-device backends (CUDA, SYCL); `VLA_DEVICE` overrides. +/// Rejects junk instead of letting atoi turn it into device 0, so a typo does +/// not silently run on the wrong GPU. inline int backend_device_index() { const char * e = std::getenv("VLA_DEVICE"); if (!e || !*e) return 0; - const int idx = std::atoi(e); - return idx > 0 ? idx : 0; + char * end = nullptr; + const long idx = std::strtol(e, &end, 10); + if (*end != '\0' || idx < 0 || idx > 1024) { + std::fprintf(stderr, "vla: ignoring VLA_DEVICE='%s' (not a device index); using 0\n", e); + return 0; + } + return (int) idx; } /** @@ -100,7 +122,10 @@ inline Backend backend_init(const char * tag, int n_threads) { // disabling oneDNN outright. Only a default: an explicit setting wins, // for Intel GPUs where the pool is worth keeping. // ggml reads this on the first SYCL entry point, so it must be set here. - setenv("GGML_SYCL_ENABLE_VMM", "0", /*overwrite=*/0); + // call_once because two concurrent model_load calls would otherwise race + // on the process environment. + static std::once_flag vmm_once; + std::call_once(vmm_once, [] { setenv_default("GGML_SYCL_ENABLE_VMM", "0"); }); // ggml_backend_sycl_init() guards the device index with assert(), which // a Release build compiles out and then indexes past the device array. diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 57f23fe..2330973 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -543,6 +543,9 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (double) m->lm_rope_base, (long long) m->num_actions_chunk, (long long) m->action_dim, (long long) m->vocab_size, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); + // Not vla::backend_init: bitvla pins its ggml graph to the CPU on purpose and + // offloads the LM through the hand-written ternary CUDA kernels below, so it + // must not pick up whichever accelerator the build compiled in. m->backend = ggml_backend_cpu_init(); if (!m->backend) { std::fprintf(stderr, "vla(bitvla): ggml_backend_cpu_init failed\n"); return nullptr; } ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); From 3105c83a6e252b49b24977babf928a755b2eb2b7 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 13:40:19 +0700 Subject: [PATCH 03/42] fix multisuite lookup in ci matrix --- ci/config/matrix.env | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/ci/config/matrix.env b/ci/config/matrix.env index 94da008..463ec76 100644 --- a/ci/config/matrix.env +++ b/ci/config/matrix.env @@ -43,7 +43,7 @@ models_for() { local v="MODELS_$1"; echo "${!v}"; } MULTISUITE_MODELS_rtx3090="bitvla gr00t_n1_7" MULTISUITE_SUITES="libero_spatial libero_object libero_goal libero_10" -multisuite_models_for() { local v=KNOWN_ISSUES"MULTISUITE_MODELS_$1"; echo "${!v:-}"; } +multisuite_models_for() { local v="MULTISUITE_MODELS_$1"; echo "${!v:-}"; } # Suites a given (platform, model) should run = DEFAULT_SUITE, plus the # multisuite set if the model is listed for that platform. From 3a1b3ff634b23cc21550672999acbaa9a16eba7e Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 13:51:19 +0700 Subject: [PATCH 04/42] bump llama.cpp to b10326 --- .github/workflows/build.yml | 2 +- CMakeLists.txt | 2 +- docs/backend/sycl.md | 30 +++++++++++------------------- 3 files changed, 13 insertions(+), 21 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 1838000..090eb35 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -45,7 +45,7 @@ jobs: - uses: actions/cache@v4 with: path: build/_deps - key: llama-b9866-${{ runner.os }} + key: llama-b10326-${{ runner.os }} - name: build vla-server + vlm-server + vla-cli (CPU, -Wall -Wextra) run: | cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF diff --git a/CMakeLists.txt b/CMakeLists.txt index 4898660..81b4dc2 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -37,7 +37,7 @@ set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) include(FetchContent) FetchContent_Declare(llama GIT_REPOSITORY https://github.com/ggml-org/llama.cpp - GIT_TAG b9866 + GIT_TAG b10326 GIT_SHALLOW TRUE ) FetchContent_MakeAvailable(llama) diff --git a/docs/backend/sycl.md b/docs/backend/sycl.md index ee2df7e..147b2ed 100644 --- a/docs/backend/sycl.md +++ b/docs/backend/sycl.md @@ -147,25 +147,16 @@ it and is the faster of the two workarounds (disabling oneDNN with default: set `GGML_SYCL_ENABLE_VMM=1` explicitly to keep the pool on hardware where it pays off. -## Known issue: `bf16 -> f32` copies +## Fixed upstream: `bf16 -> f32` copies -ggml-sycl's copy table has `f16 -> f32` but no `bf16 -> f32`, so an arch whose -graph contains that copy aborts at predict time: +ggml-sycl's copy table used to have `f16 -> f32` but no `bf16 -> f32`, so an arch +whose graph contained that copy aborted at predict time. VLA-Adapter hit it with +its default BF16 weights, and the workaround was `VLA_ADAPTER_F32_WEIGHTS=1`. -``` -ggml_sycl_cpy: unsupported type combination (bf16 to f32) -.../ggml-sycl/cpy.cpp:592: fatal error -``` - -VLA-Adapter hits this with its default BF16 weights (OpenVLA-OFT shares the same -graph shape and is expected to as well). Switching that arch to F32 weights -removes the BF16 tensor from the graph and it runs: - -```bash -VLA_ADAPTER_F32_WEIGHTS=1 ./build-sycl/vla-cli --ckpt ... -``` - -SmolVLA and Evo-1 are unaffected and run on their default BF16 weights. +llama.cpp b10326 adds the missing kernel (`cpy_1_bf16_f32` in +`ggml/src/ggml-sycl/cpy.cpp`), so VLA-Adapter should run on stock BF16 weights +now. Not yet re-tested on the A380 - if you hit the old abort, fall back to +`VLA_ADAPTER_F32_WEIGHTS=1` and file an issue. ## Performance note: F32 weights @@ -196,8 +187,9 @@ threads); GPU is the Arc A380. | Evo-1 | 448 | 7,695 ms | **1,176 ms** | 6.5x | | VLA-Adapter | 224 | 2,994 ms | **517 ms** | 5.8x | -VLA-Adapter is measured with `VLA_ADAPTER_F32_WEIGHTS=1` on both sides (see the -`bf16 -> f32` issue above); the others run their stock defaults. +VLA-Adapter is measured with `VLA_ADAPTER_F32_WEIGHTS=1` on both sides, which was +required at the time (see the `bf16 -> f32` section above); the others run their +stock defaults. Per-stage for SmolVLA: From b2ba0eea807458bf8223cc28e2af7547b3aef1dd Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 13:51:19 +0700 Subject: [PATCH 05/42] relax transformers pin, default docker to sm_89 --- Dockerfile | 6 +++--- pyproject.toml | 5 ++++- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/Dockerfile b/Dockerfile index c399342..58328f6 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,6 +1,6 @@ # vla-server, CPU or CUDA. cmake fetches llama.cpp at build time. -# GPU sm_120 (default): docker build -t vla-cpp . -# GPU other arch: --build-arg CUDA_ARCH=89 (89=RTX40 90=H100 87=Orin; sm_120 needs CUDA>=12.8) +# GPU sm_89 (default): docker build -t vla-cpp . +# GPU other arch: --build-arg CUDA_ARCH=120 (86=RTX30 90=H100 87=Orin 120=RTX50; sm_120 needs CUDA>=12.8) # older card: --build-arg BASE_IMAGE=nvidia/cuda:12.4.1-devel-ubuntu24.04 --build-arg CUDA_ARCH=86 # CPU: --build-arg BACKEND=cpu --build-arg BASE_IMAGE=ubuntu:24.04 -t vla-cpp-cpu # run: docker run --gpus all -p5555:5555 -v $PWD/models:/models vla-cpp --bind tcp://*:5555 /models/M.gguf @@ -18,7 +18,7 @@ WORKDIR /src COPY . . ARG BACKEND=cuda -ARG CUDA_ARCH=120 +ARG CUDA_ARCH=89 # nvcc can segfault on the flash-attn kernels under high -j; lower JOBS if so. ARG JOBS= # CUDA: -devel ships only a libcuda stub (real driver injected at runtime), so diff --git a/pyproject.toml b/pyproject.toml index 85ec419..14ed4b2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -23,7 +23,10 @@ client = [ "msgpack-numpy>=0.4.8", "pillow>=10", "torch>=2.5", - "transformers>=4.51,<4.52", + # Only used for AutoTokenizer/AutoProcessor.from_pretrained (the pi0 PaliGemma + # tokenizer). Capped below 5.0: the v5 line is a breaking rewrite we have not + # tested against. + "transformers>=4.51,<5", "numpy>=1.24", ] From cd246c9c12f613bd382aa1a6bf908f243fecbfd5 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 13:56:25 +0700 Subject: [PATCH 06/42] harden servers against malformed requests --- CMakeLists.txt | 1 + src/serving/server.cpp | 52 ++++++++++++++++++++++++++++-- src/serving/vlm-server.cpp | 65 +++++++++++++++++++++++++++++++++++--- 3 files changed, 112 insertions(+), 6 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 81b4dc2..5bed3db 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -202,6 +202,7 @@ add_executable(vlm-server ) target_include_directories(vlm-server PRIVATE ${CMAKE_CURRENT_BINARY_DIR}/proto-gen + ${llama_SOURCE_DIR}/vendor/stb ${CPPZMQ_INCLUDE_DIR} ) target_link_libraries(vlm-server PRIVATE diff --git a/src/serving/server.cpp b/src/serving/server.cpp index 1cecaad..410fa14 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -57,6 +57,21 @@ bool decode_image(const vla::Image & img, data.size()); return false; } + // Read the header first: stbi_load allocates 3*w*h before returning, so a + // small JPEG declaring huge dimensions would allocate gigabytes before any + // check on the decoded size could reject it. + if (!stbi_info_from_memory( + reinterpret_cast(data.data()), + static_cast(data.size()), &w, &h, &ch)) { + std::fprintf(stderr, "vla-server: stbi_info_from_memory failed: %s\n", + stbi_failure_reason()); + return false; + } + if (w <= 0 || h <= 0 || w > int(kMaxImageDim) || h > int(kMaxImageDim)) { + std::fprintf(stderr, "vla-server: JPEG dims %dx%d out of range (max %u)\n", + w, h, kMaxImageDim); + return false; + } unsigned char * px = stbi_load_from_memory( reinterpret_cast(data.data()), static_cast(data.size()), @@ -110,6 +125,14 @@ bool decode_image(const vla::Image & img, f32.resize(pixels); std::memcpy(f32.data(), img.data().data(), expected); + // The other float inputs are swept for NaN/Inf; pixels were not, so a + // non-finite pixel used to propagate all the way out as a robot action. + for (size_t i = 0; i < pixels; ++i) { + if (!std::isfinite(f32[i])) { + std::fprintf(stderr, "vla-server: F32_RGB_01 pixel %zu is not finite\n", i); + return false; + } + } view = { f32.data(), int(img.width()), int(img.height()), vla::PixelFormat::F32_RGB_01 }; return true; @@ -127,6 +150,19 @@ std::string make_error_response(uint64_t request_id, const std::string & msg) { return resp.SerializeAsString(); } +// Discard any frames after the first. Returns true if there were some, meaning +// the request was malformed. Must run to completion: leaving a frame queued +// keeps REP in receive state and the next send throws EFSM. +bool drain_extra_frames(zmq::socket_t & sock) { + bool extra = false; + while (sock.get(zmq::sockopt::rcvmore)) { + zmq::message_t junk; + if (!sock.recv(junk, zmq::recv_flags::none)) break; + extra = true; + } + return extra; +} + int find_non_finite(const float * data, int n) { for (int i = 0; i < n; ++i) { if (!std::isfinite(data[i])) return i; @@ -230,8 +266,11 @@ int main(int argc, char ** argv) { zmq::context_t zctx( 1); zmq::socket_t sock(zctx, zmq::socket_type::rep); sock.set(zmq::sockopt::linger, 0); - // cap inbound messages so one oversized request cannot exhaust memory. - sock.set(zmq::sockopt::maxmsgsize, int64_t(256) * 1024 * 1024); + // Cap inbound messages so one oversized request cannot exhaust memory. 64 MiB + // is far above any real request (16 views of 512x512 F32 RGB is ~50 MiB) and + // low enough that protobuf's several-fold expansion during ParseFromArray + // stays bounded. + sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); sock.bind(bind_addr); std::printf("vla-server: bound to %s. ready.\n", bind_addr.c_str()); @@ -285,6 +324,15 @@ int main(int argc, char ** argv) { continue; } + // REP will not let us reply until the whole multipart message is received, + // and send_reply treats a failed send as fatal - so an unauthenticated + // client could shut the server down with one two-frame request. The + // protocol is single-frame: drain any extras and reject. + if (drain_extra_frames(sock)) { + send_reply(make_error_response(0, "expected a single-frame request")); + continue; + } + vla::PredictRequest req; if (!req.ParseFromArray(req_msg.data(), static_cast(req_msg.size()))) { std::fprintf(stderr, "vla-server: PredictRequest parse failed (size=%zu)\n", diff --git a/src/serving/vlm-server.cpp b/src/serving/vlm-server.cpp index 3fa3e5d..4e59e08 100644 --- a/src/serving/vlm-server.cpp +++ b/src/serving/vlm-server.cpp @@ -15,11 +15,21 @@ #include "vlm/engine.h" #include "serving/vlm.pb.h" +// Only for stbi_info_from_memory: the engine decodes through mtmd, which has no +// dimension guard, so we preflight the JPEG header here at the trust boundary. +#define STB_IMAGE_IMPLEMENTATION +#define STB_IMAGE_STATIC +#pragma GCC diagnostic push +#pragma GCC diagnostic ignored "-Wunused-function" +#include "stb_image.h" +#pragma GCC diagnostic pop + #include #include #include #include +#include #include #include #include @@ -101,8 +111,9 @@ int main(int argc, char ** argv) { zmq::context_t zctx( 1); zmq::socket_t sock(zctx, zmq::socket_type::router); sock.set(zmq::sockopt::linger, 0); - // cap inbound messages so one oversized request cannot exhaust memory. - sock.set(zmq::sockopt::maxmsgsize, int64_t(256) * 1024 * 1024); + // Cap inbound messages so one oversized request cannot exhaust memory. Per + // frame only - see the envelope caps in the recv loop for the multipart total. + sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); sock.bind(bind_addr); std::printf("vlm-server: bound to %s. ready.\n", bind_addr.c_str()); @@ -132,9 +143,16 @@ int main(int argc, char ** argv) { } if (!(poll[0].revents & ZMQ_POLLIN)) continue; + // A ROUTER envelope is an identity frame plus an optional empty delimiter. + // maxmsgsize bounds each frame but not how many, so without these caps a + // peer could stream sub-limit frames until the process runs out of memory. + constexpr size_t kMaxEnvFrames = 8; + constexpr size_t kMaxEnvBytes = 64 * 1024; + std::vector env; std::string payload; - bool recv_ok = true, have_payload = false; + size_t env_bytes = 0; + bool recv_ok = true, have_payload = false, env_overflow = false; for (;;) { zmq::message_t part; try { @@ -146,13 +164,24 @@ int main(int argc, char ** argv) { recv_ok = false; break; } if (sock.get(zmq::sockopt::rcvmore)) { - env.emplace_back(static_cast(part.data()), part.size()); + env_bytes += part.size(); + if (env.size() >= kMaxEnvFrames || env_bytes > kMaxEnvBytes) { + // Keep draining so the socket stays in a sane state, but stop + // accumulating and drop the request. + env_overflow = true; + } else { + env.emplace_back(static_cast(part.data()), part.size()); + } } else { payload.assign(static_cast(part.data()), part.size()); have_payload = true; break; } } + if (env_overflow) { + std::fprintf(stderr, "vlm-server: oversized ROUTER envelope; request dropped\n"); + continue; + } if (!recv_ok || !have_payload || env.empty()) continue; auto send_reply = [&](const std::string & body) { @@ -181,6 +210,21 @@ int main(int argc, char ** argv) { send_reply(make_error_stream(rid, "ChatRequest has no messages")); continue; } + // Bound the work a single request can buy: without these, one 60 MiB + // payload of millions of tiny messages costs template formatting and + // tokenization far beyond anything n_ctx could consume. + constexpr int kMaxMessages = 512; + constexpr size_t kMaxTextBytes = 4u * 1024 * 1024; + if (req.messages_size() > kMaxMessages) { + send_reply(make_error_stream(rid, "too many messages (max 512)")); + continue; + } + size_t text_bytes = 0; + for (const auto & m : req.messages()) text_bytes += m.content().size(); + if (text_bytes > kMaxTextBytes) { + send_reply(make_error_stream(rid, "message text too large (max 4 MiB)")); + continue; + } if (req.images_size() > 16) { send_reply(make_error_stream(rid, "too many image views (max 16)")); continue; @@ -194,6 +238,19 @@ int main(int argc, char ** argv) { vlm::Image out; if (im.encoding() == vlm_chat::Image::JPEG) { const auto & d = im.data(); + // Header first: the decoder allocates from the declared dimensions, + // so a tiny JPEG claiming 30000x30000 would allocate gigabytes. + int jw = 0, jh = 0, jc = 0; + if (d.size() > size_t(INT_MAX) || + !stbi_info_from_memory(reinterpret_cast(d.data()), + static_cast(d.size()), &jw, &jh, &jc) || + jw <= 0 || jh <= 0 || + jw > int(kMaxImageDim) || jh > int(kMaxImageDim)) { + char buf[96]; std::snprintf(buf, sizeof(buf), + "image[%d] JPEG dims %dx%d rejected (max %u)", v, jw, jh, kMaxImageDim); + send_reply(make_error_stream(rid, buf)); + decode_ok = false; break; + } if (!engine.decode_image_buf( reinterpret_cast(d.data()), d.size(), out)) { char buf[64]; std::snprintf(buf, sizeof(buf), "image[%d] JPEG decode failed", v); From c26c313bdb31557b08d6483a87738d9106001a1b Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 14:01:38 +0700 Subject: [PATCH 07/42] harden model loading against bad checkpoints --- src/model.cpp | 27 ++++++++++++++++++++++++ src/models/bitvla.cpp | 34 ++++++++++++++++++++++++++----- src/models/gguf_reader.h | 44 ++++++++++++++++++++++++++++++++-------- src/models/pi0.cpp | 8 +++++++- src/models/pi05.cpp | 8 +++++++- src/models/smolvla.cpp | 15 ++++---------- 6 files changed, 109 insertions(+), 27 deletions(-) diff --git a/src/model.cpp b/src/model.cpp index b99ba3c..01d2907 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -87,6 +87,29 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { return ok; } +// Invariants every arch's Config must satisfy. Checked once here rather than in +// eleven loaders: predict() sizes its host buffers from the max_* dims and loops +// to the real_* dims, so real > max is an out-of-bounds write on the first call. +bool config_is_sane(const Config& c) { + struct { const char* name; int64_t real; int64_t max; } pairs[] = { + { "state", c.real_state_dim, c.max_state_dim }, + { "action", c.real_action_dim, c.max_action_dim }, + }; + for (const auto& p : pairs) { + if (p.real < 0 || p.max < 0) { + std::fprintf(stderr, "vla: negative %s dim (real=%lld max=%lld)\n", + p.name, (long long) p.real, (long long) p.max); + return false; + } + if (p.max > 0 && p.real > p.max) { + std::fprintf(stderr, "vla: real_%s_dim %lld exceeds max_%s_dim %lld\n", + p.name, (long long) p.real, p.name, (long long) p.max); + return false; + } + } + return true; +} + bool detect_arch_safetensors(const std::string& path, Arch* out) { std::ifstream f(path, std::ios::binary); if (!f) return false; @@ -183,6 +206,10 @@ Model* model_load(const std::string& mmproj_path, const std::string& ckpt_path, break; } if (!impl) return nullptr; + if (!config_is_sane(impl->cfg)) { + std::fprintf(stderr, "vla: refusing to load %s\n", ckpt_path.c_str()); + return nullptr; + } auto* m = new Model(); m->impl = std::move(impl); diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 2330973..1511d07 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -662,13 +662,24 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, if (cudaGetDeviceCount(&dev_count) == cudaSuccess && dev_count > 0) { cudaSetDevice(0); + // Set false by any int2 tensor whose .scale sidecar is missing. The + // ladder kernels dereference the scale pointer unconditionally, so a + // null one is a device-side out-of-bounds read, not a soft failure. + bool scales_ok = true; + auto load_bit = [&](ggml_tensor * t, int64_t N, int64_t K) -> std::pair { if (m->packed_int2) { int8_t * dp = upload_int8((const uint8_t*) t->data, ggml_nbytes(t), m->cuda_devptrs); std::string nm = ggml_get_name(t); std::string sn = nm.substr(0, nm.size() - 7) + ".scale"; std::vector sc = g.read_f32(sn.c_str()); - float * dws = sc.empty() ? nullptr : upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); + if (sc.empty()) { + std::fprintf(stderr, "vla(bitvla): int2 tensor %s has no %s sidecar\n", + nm.c_str(), sn.c_str()); + scales_ok = false; + return { dp, nullptr }; + } + float * dws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); return { dp, dws }; } float ws; @@ -682,7 +693,7 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (int) m->lm_inter, (int) m->lm_layers, m->lm_rope_base, m->lm_rms_eps, max_seq); if (m->lm_cuda_ctx) { bool pack_ok = true; - for (int64_t L = 0; L < m->lm_layers && pack_ok; ++L) { + for (int64_t L = 0; L < m->lm_layers && pack_ok && scales_ok; ++L) { bitvla_lm_layer_cuda lyr{}; lyr.attn_norm_w = upload_bf16_from_f32((const float*) m->lm[L].attn_norm->data, m->lm_hidden, m->cuda_devptrs); @@ -700,8 +711,14 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, if (m->packed_int2) { lyr.gate_up_packed = upload_int8((const uint8_t*) m->lm[L].Wgate_up->data, ggml_nbytes(m->lm[L].Wgate_up), m->cuda_devptrs); - std::vector sc = g.read_f32(("lm.blk." + std::to_string(L) + ".ffn_gate_up.scale").c_str()); - lyr.gate_up_ws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); + const std::string sn = "lm.blk." + std::to_string(L) + ".ffn_gate_up.scale"; + std::vector sc = g.read_f32(sn.c_str()); + if (sc.empty()) { + std::fprintf(stderr, "vla(bitvla): missing %s\n", sn.c_str()); + scales_ok = false; + } else { + lyr.gate_up_ws = upload_f32_scales(sc.data(), (int) sc.size(), m->cuda_devptrs); + } } else { std::vector ws2; lyr.gate_up_packed = pack_and_upload_fused( @@ -714,7 +731,10 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, { auto r = load_bit(m->lm[L].Wdown, m->lm_hidden, m->lm_inter); lyr.down_packed = r.first; lyr.down_ws = r.second; } bitvla_lm_cuda_set_layer(m->lm_cuda_ctx, (int) L, &lyr); } - if (pack_ok) { + if (!scales_ok) { + std::fprintf(stderr, "vla(bitvla): int2 scale sidecars incomplete; refusing the CUDA LM\n"); + } + if (pack_ok && scales_ok) { __nv_bfloat16* onorm = upload_bf16_from_f32((const float*) m->lm_output_norm->data, m->lm_hidden, m->cuda_devptrs); bitvla_lm_cuda_set_output_norm(m->lm_cuda_ctx, onorm); @@ -891,6 +911,10 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, if (!t) continue; const size_t nb = ggml_nbytes(t); void* copy = std::malloc(nb); + if (!copy) { + std::fprintf(stderr, "vla(bitvla): out of memory keeping %zu bytes of CPU weights\n", nb); + return nullptr; + } std::memcpy(copy, t->data, nb); t->data = copy; m->cpu_kept_ptrs.push_back(copy); diff --git a/src/models/gguf_reader.h b/src/models/gguf_reader.h index 3b3f739..60abc90 100644 --- a/src/models/gguf_reader.h +++ b/src/models/gguf_reader.h @@ -58,11 +58,28 @@ struct gguf_reader { return true; } - bool has(const char * k) const { return gguf_find_key(gctx, k) >= 0; } - uint32_t u32(const char * k) const { return gguf_get_val_u32(gctx, gguf_find_key(gctx, k)); } - float f32(const char * k) const { return gguf_get_val_f32(gctx, gguf_find_key(gctx, k)); } - double f64(const char * k) const { return gguf_get_val_f64(gctx, gguf_find_key(gctx, k)); } - std::string str(const char * k) const { const int64_t id = gguf_find_key(gctx, k); return id < 0 ? std::string() : std::string(gguf_get_val_str(gctx, id)); } + bool has(const char * k) const { return gguf_find_key(gctx, k) >= 0; } + + // The gguf_get_val_* helpers assert on a type mismatch, which aborts the + // process on a malformed file. Check the declared type first and fall back to + // the caller's default instead. (model.cpp's arch probe already did this; the + // reader did not.) + bool typed_key(const char * k, gguf_type want, int64_t * id_out) const { + const int64_t id = gguf_find_key(gctx, k); + if (id < 0) return false; + if (gguf_get_kv_type(gctx, id) != want) { + std::fprintf(stderr, "vla(%s): key %s has unexpected type %d\n", + arch, k, (int) gguf_get_kv_type(gctx, id)); + return false; + } + *id_out = id; + return true; + } + + uint32_t u32(const char * k) const { int64_t id; return typed_key(k, GGUF_TYPE_UINT32, &id) ? gguf_get_val_u32(gctx, id) : 0u; } + float f32(const char * k) const { int64_t id; return typed_key(k, GGUF_TYPE_FLOAT32, &id) ? gguf_get_val_f32(gctx, id) : 0.f; } + double f64(const char * k) const { int64_t id; return typed_key(k, GGUF_TYPE_FLOAT64, &id) ? gguf_get_val_f64(gctx, id) : 0.0; } + std::string str(const char * k) const { int64_t id; return typed_key(k, GGUF_TYPE_STRING, &id) ? std::string(gguf_get_val_str(gctx, id)) : std::string(); } const ggml_tensor * meta(const char * name) const { return ggml_get_tensor(meta_ctx, name); } // Resident type for a weight: keep a quantized source type (Q8_0, Q4_0, ...) @@ -72,11 +89,20 @@ struct gguf_reader { return (src && ggml_is_quantized(src->type)) ? src->type : prefer; } - bool read_raw(const char * name, void * buf) { + // cap is the destination capacity in bytes and must equal the declared tensor + // size. Without it a tensor with the expected ne[0] but an extra dimension + // writes past a caller's vector. On failure buf may be partially written, so + // callers must not keep it. + bool read_raw(const char * name, void * buf, size_t cap) { const int64_t id = gguf_find_tensor(gctx, name); if (id < 0) { std::fprintf(stderr, "vla(%s): missing tensor %s\n", arch, name); return false; } const size_t off = data_off + gguf_get_tensor_offset(gctx, id); const size_t nb = gguf_get_tensor_size(gctx, id); + if (nb != cap) { + std::fprintf(stderr, "vla(%s): tensor %s is %zu bytes, caller expects %zu\n", + arch, name, nb, cap); + return false; + } if (fseeko(fp, (off_t) off, SEEK_SET) != 0) return false; return std::fread(buf, 1, nb, fp) == nb; } @@ -86,8 +112,8 @@ struct gguf_reader { if (!t) { std::fprintf(stderr, "vla(%s): missing tensor %s\n", arch, name); return {}; } const int64_t n = ggml_nelements(t); std::vector out(n); - if (t->type == GGML_TYPE_F32) { if (!read_raw(name, out.data())) return {}; } - else if (t->type == GGML_TYPE_BF16) { std::vector tmp(n); if (!read_raw(name, tmp.data())) return {}; ggml_bf16_to_fp32_row(tmp.data(), out.data(), n); } + if (t->type == GGML_TYPE_F32) { if (!read_raw(name, out.data(), out.size() * sizeof(float))) return {}; } + else if (t->type == GGML_TYPE_BF16) { std::vector tmp(n); if (!read_raw(name, tmp.data(), tmp.size() * sizeof(ggml_bf16_t))) return {}; ggml_bf16_to_fp32_row(tmp.data(), out.data(), n); } else { std::fprintf(stderr, "vla(%s): tensor %s unsupported type %d\n", arch, name, (int) t->type); return {}; } return out; } @@ -100,7 +126,7 @@ struct gguf_reader { const ggml_tensor * t = meta(name); if (!t) { std::fprintf(stderr, "vla(%s): missing tensor %s\n", arch, name); return {}; } std::vector o(ggml_nbytes(t)); - if (!read_raw(name, o.data())) return {}; + if (!read_raw(name, o.data(), o.size())) return {}; return o; } std::vector f = read_f32(name); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 16b78c3..cf55ab9 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -297,7 +297,13 @@ bool load_stats(gguf_reader & g, Pi0ModelArch & m) { const ggml_tensor * t = g.meta(name); if (!t) { std::printf("vla(pi0): %s missing - identity\n", name); return; } if (t->ne[0] != (int64_t) dst.size()) { std::printf("vla(pi0): %s dim mismatch - identity\n", name); return; } - if (!g.read_raw(name, dst.data())) std::printf("vla(pi0): %s read failed - identity\n", name); + const std::vector identity = dst; + if (!g.read_raw(name, dst.data(), dst.size() * sizeof(float))) { + // A short read leaves dst half-overwritten; restore so "identity" + // means identity. + dst = identity; + std::printf("vla(pi0): %s read failed - identity\n", name); + } }; read1d("state_mean", m.state_mean); read1d("state_std", m.state_std); diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 2f2db1d..ab590b6 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -369,7 +369,13 @@ bool load_stats(gguf_reader & g, Pi05ModelArch & m) { const ggml_tensor * t = g.meta(name); if (!t) { std::printf("vla(pi05): %s missing - identity\n", name); return; } if (t->ne[0] != (int64_t) dst.size()) { std::printf("vla(pi05): %s dim mismatch - identity\n", name); return; } - if (!g.read_raw(name, dst.data())) std::printf("vla(pi05): %s read failed - identity\n", name); + const std::vector identity = dst; + if (!g.read_raw(name, dst.data(), dst.size() * sizeof(float))) { + // A short read leaves dst half-overwritten; restore so "identity" + // means identity. + dst = identity; + std::printf("vla(pi05): %s read failed - identity\n", name); + } }; read1d("state_mean", m.state_mean); read1d("state_std", m.state_std); diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 8a9b396..0ec9ae2 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -184,7 +184,7 @@ struct gguf_source { const int64_t id = gguf_find_tensor(gctx, name.c_str()); const size_t offset = data_off + gguf_get_tensor_offset(gctx, id); const size_t bytes = gguf_get_tensor_size(gctx, id); - if (std::fseek(fp, (long) offset, SEEK_SET) != 0) { + if (fseeko(fp, (off_t) offset, SEEK_SET) != 0) { std::fprintf(stderr, "vla: fseek failed for %s\n", name.c_str()); return false; } @@ -218,7 +218,7 @@ struct gguf_source { return false; } const size_t offset = data_off + gguf_get_tensor_offset(gctx, id); - if (std::fseek(fp, (long) offset, SEEK_SET) != 0) return false; + if (fseeko(fp, (off_t) offset, SEEK_SET) != 0) return false; return std::fread(dst, 1, bytes, fp) == bytes; } @@ -950,7 +950,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, if (k * k != m->vit_n_tokens) { std::fprintf(stderr, "vla: smolvla vit geometry mismatch (grid=%lld scale=%lld -> %lld tokens, KV says %lld)\n", (long long) grid, (long long) m->vit_scale, (long long) (k * k), (long long) m->vit_n_tokens); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -962,7 +961,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, if (!use_gguf) { if (!st.open(ckpt_path)) { std::fprintf(stderr, "vla: failed to open %s\n", ckpt_path.c_str()); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -977,7 +975,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, } if (max_layer < 0) { std::fprintf(stderr, "vla: cannot infer n_layers from %s\n", ckpt_path.c_str()); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -987,7 +984,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, const auto it = st.tensors.find("model.vlm_with_expert.lm_expert.layers.0.mlp.gate_proj.weight"); if (it == st.tensors.end() || it->second.shape.size() != 2) { std::fprintf(stderr, "vla: missing/malformed expert gate_proj for shape derivation\n"); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -997,7 +993,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, std::fprintf(stderr, "vla: expert_h mismatch - config implies %lld, " "checkpoint gate_proj has %lld\n", (long long) m->cfg.expert_h, (long long) it->second.shape[1]); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -1017,7 +1012,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, if (use_gguf) { if (!load_normalizer_stats_from_gguf(gst, *m)) { std::fprintf(stderr, "vla: failed to load normalizer stats from gguf\n"); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -1038,7 +1032,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, m->ctx_weights = ggml_init(gparams); if (!m->ctx_weights) { std::fprintf(stderr, "vla: ggml_init (weights) failed\n"); - ggml_backend_free(m->backend); delete m; return nullptr; } @@ -1564,7 +1557,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { std::vector state_host(cfg.max_state_dim); std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); - for (int64_t i = 0; i < cfg.real_state_dim; ++i) { + for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) { state_host[i] = (state_host[i] - m->state_mean[i]) / (m->state_std[i] + cfg.norm_eps); } @@ -1691,7 +1684,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { std::vector state_host(cfg.max_state_dim); std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); - for (int64_t i = 0; i < cfg.real_state_dim; ++i) { + for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) { state_host[i] = (state_host[i] - m->state_mean[i]) / (m->state_std[i] + cfg.norm_eps); } From 1ea4c2bc972b4563fffa4af2430675b4bfee2c4f Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 14:06:44 +0700 Subject: [PATCH 08/42] fix evo1 padded state dims, document arch quirks, add unit tests --- .github/workflows/build.yml | 11 +++- README.md | 8 +-- docs/ADOPTION.md | 41 ++++++------- docs/backend/metal.md | 4 +- docs/backend/wsl.md | 7 +-- src/model.h | 21 +++++-- src/models/evo1.cpp | 4 ++ src/models/gguf_reader.h | 8 +++ src/models/gr00tn1d5.cpp | 3 + src/models/gr00tn1d6.cpp | 3 + src/models/gr00tn1d7.cpp | 3 + src/models/openvla_oft.cpp | 11 +++- src/models/pi05.cpp | 7 ++- src/models/vla_adapter.cpp | 6 ++ src/models/vla_jepa.cpp | 3 + tests/CMakeLists.txt | 10 +++ tests/test_config_guard.cpp | 71 +++++++++++++++++++++ tests/test_rope_conventions.cpp | 105 ++++++++++++++++++++++++++++++++ 18 files changed, 283 insertions(+), 43 deletions(-) create mode 100644 tests/test_config_guard.cpp create mode 100644 tests/test_rope_conventions.cpp diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 090eb35..cf25862 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -17,10 +17,15 @@ jobs: runs-on: ubuntu-24.04 steps: - uses: actions/checkout@v4 - - name: pixel-shuffle channel order + # Compiled directly rather than through cmake: these are pure and need + # neither llama.cpp nor the protobuf/zmq toolchain. + - name: pure unit tests run: | - g++ -std=c++17 -Isrc -Wall -Wextra tests/test_vision_common.cpp -o /tmp/test_vision_common - /tmp/test_vision_common + for t in test_vision_common test_rope_conventions test_config_guard; do + g++ -std=c++17 -Isrc -Wall -Wextra -fsanitize=address,undefined \ + -fno-omit-frame-pointer "tests/$t.cpp" -o "/tmp/$t" + "/tmp/$t" + done py-tooling: runs-on: ubuntu-24.04 diff --git a/README.md b/README.md index f009399..3811c5b 100644 --- a/README.md +++ b/README.md @@ -27,10 +27,10 @@ from consumer GPUs down to Jetson-class boards - or **Intel GPUs** via SYCL. - CUDA 12.x (optional - required only for CUDA GPU builds) - Intel oneAPI 2025.x + GPU compute runtime (optional - only for Intel GPU builds, see [docs/backend/sycl.md](docs/backend/sycl.md)) -- `libzmq3-dev`, `libprotobuf-dev`, `protobuf-compiler` +- `libzmq3-dev`, `cppzmq-dev`, `libprotobuf-dev`, `protobuf-compiler` ```bash -sudo apt-get install -y libzmq3-dev libprotobuf-dev protobuf-compiler +sudo apt-get install -y libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler ``` ### From source @@ -84,8 +84,8 @@ export PATH=/usr/local/cuda/bin:$PATH export LD_LIBRARY_PATH=/usr/local/cuda/lib64:$LD_LIBRARY_PATH ``` -Check [docs/backend](docs/backend) for compiling `vla.cpp` with other platforms. -WLS2 and Apple Silicon has been tested. +Check [docs/backend](docs/backend) for compiling `vla.cpp` on other platforms. +WSL2 and Apple Silicon are both tested. --- diff --git a/docs/ADOPTION.md b/docs/ADOPTION.md index b6db733..424b48d 100644 --- a/docs/ADOPTION.md +++ b/docs/ADOPTION.md @@ -1,29 +1,28 @@ # Adoption notes -vla.cpp is technically solid (7 architectures, self-contained GGUFs, CUDA + Jetson, -real benchmarks). The gap to llama.cpp-style reach is mostly distribution, not code. -Ordered by leverage: +The engine works; the gap is distribution. Roughly in priority order. -1. **Publish on GitHub.** The repo lives on Bitbucket (`bitbucket.org/vinrobotics/vla.cpp`) - while the README already links a `github.com/VinRobotics/vla.cpp` URL. llama.cpp's reach - came from GitHub visibility, issues, and PRs. A public GitHub mirror is the single biggest - lever; nothing else here matters as much. +1. **C ABI.** Done: `include/vla.h` and `libvla`. `src/model.h` is C++ only, so + without it nothing outside C++ can link the engine. -2. **Ship the models.** All seven GGUFs are already published under - [`vrfai`](https://huggingface.co/vrfai) on the Hub - the README's "coming soon" rows are - stale (now fixed). Keep the model table pointing at the real repos so the policies are - one `hf download` away. +2. **Python bindings.** Done: `bindings/python`. Robotics runs on Python; the + only other way in is the ZeroMQ server plus a hand-written client. -3. **Cut releases.** `v0.1.0` is the first tag (see `CHANGELOG.md`). Tagged releases + - changelog give users something to pin and cite. +3. **Prebuilt binaries.** `.github/workflows/build.yml` already builds + `vla-server`, `vlm-server` and `vla-cli` and uploads nothing. Tagged + artifacts (linux x86-64 CPU/CUDA, linux aarch64 Jetson, macOS arm64 Metal) + plus a published Docker image remove the build step. -4. **Rotate the committed credential.** The local `.git/config` remote URL embeds an - access token (`https://@bitbucket.org/...`). It is never pushed (git config is not - tracked), so this is hygiene, not a live leak - but rotate it and use a credential helper - or SSH remote instead of an inline token. +4. **One-command model fetch.** Today: install `huggingface_hub`, run + `hf download`, pass a path. llama.cpp solved this with `-hf user/repo`. The + GGUFs are already on the Hub under [`vrfai`](https://huggingface.co/vrfai). -5. **Lower the build bar (optional).** A CUDA `Dockerfile` now exists; publishing a prebuilt - image (and, later, macOS/Metal or ROCm backends) removes the from-source step that stops - most drive-by users. +5. **Reproducible benchmarks.** The README latency table has no in-repo source + and disagrees with `ci/baselines/rtx3090.json`. A `vla-bench` that emits the + table, with quantization and memory columns, makes it checkable. -None of these change inference behaviour; they change who can find and run it. +6. **Contributor path.** Adding an architecture touches six sites, none written + down: the `Arch` enum and factory in `src/arch.h`, the key list, string map + and switch in `src/model.cpp`, and `CMakeLists.txt`. + +None of these change inference behaviour. diff --git a/docs/backend/metal.md b/docs/backend/metal.md index db6674c..e9adc61 100644 --- a/docs/backend/metal.md +++ b/docs/backend/metal.md @@ -17,9 +17,7 @@ To disable the Metal build at compile time use the `-DGGML_METAL=OFF` cmake opti When built with Metal support, you can explicitly disable GPU inference with the `--n-gpu-layers 0` command-line argument. ```bash -# Fetch llama.cpp at pinned tag and apply local patch -bash patches/patch.sh - +# cmake fetches llama.cpp at the pinned tag; no patch step. # On MacOS, Metal is enabled by default cmake -B build -DCMAKE_BUILD_TYPE=Release cmake --build build -j$(sysctl -n hw.ncpu) diff --git a/docs/backend/wsl.md b/docs/backend/wsl.md index 5bcce3f..c715489 100644 --- a/docs/backend/wsl.md +++ b/docs/backend/wsl.md @@ -1,10 +1,9 @@ # `vla.cpp` on Windows (WSL2 + CUDA) `vla.cpp` targets Linux and macOS. On Windows the supported path is **WSL2** -with an Ubuntu distribution: the toolchain (`libzmq`, `protobuf`, `pkg-config`, -the bash `patches/patch.sh` script) and the CUDA build all run natively inside -the Linux environment, while still using the host NVIDIA GPU through the -WSL CUDA driver. +with an Ubuntu distribution: the toolchain (`libzmq`, `protobuf`, `pkg-config`) +and the CUDA build all run natively inside the Linux environment, while still +using the host NVIDIA GPU through the WSL CUDA driver. ## Prerequisites diff --git a/src/model.h b/src/model.h index 9440f86..87f4bf8 100644 --- a/src/model.h +++ b/src/model.h @@ -76,6 +76,12 @@ struct Config { int rope_n_dims; ///< RoPE rotation width (per head). int rope_mode; ///< RoPE variant (NeoX / GPT-J / etc). float rope_freq_base; ///< RoPE base frequency. + + /// True if @ref predict already applied the dataset statistics, so its output + /// is in world units. False means the caller must un-normalise (the GR00T + /// family and VLA-JEPA, which expect a host-side affine from a stats JSON). + /// Branch on this rather than on the architecture name. + bool denormalized = true; }; /// Opaque engine handle; created by @ref model_load and released by @@ -120,7 +126,12 @@ struct Inputs { const ImageView* images; ///< Camera views (host memory). int n_images; ///< Number of @ref images. - /// Pre-computed image embeddings; bypasses the vision tower. + /// Pre-computed image embeddings; bypasses the vision tower entirely. + /// Layout is [n_img_views * n_img, hidden]. These are handed to the language + /// backbone as-is, so they must already be in whatever scale that arch's LM + /// expects -- which differs per arch: pi0 wants the projector output scaled by + /// 1/sqrt(hidden), pi0.5 wants it unscaled. Supplying tower output from a + /// different arch will not error, it will just be wrong. const float* precomputed_img_emb = nullptr; int n_img_views = 0; ///< Number of views in /// @ref precomputed_img_emb. @@ -173,9 +184,11 @@ const Config& model_config(const Model* m); /** * @brief Run one forward pass. * - * Returns the predicted action chunk, normalised to the model's training - * statistics. The caller is responsible for un-normalising into world - * units. NaN/Inf inputs cause the call to abort. + * Whether the returned actions are in world units or still normalised depends + * on the architecture -- see @ref Config::denormalized. Most archs un-normalise + * internally; the GR00T family and VLA-JEPA return raw values and expect the + * caller to apply the dataset statistics. NaN/Inf inputs cause the call to + * abort. * * @param m A handle from @ref model_load. * @param in Filled-in @ref Inputs struct. diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 0f62af0..0e6b2ab 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -523,6 +523,10 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { std::vector state_norm(per_a, 0.0f); for (int64_t i = 0; i < per_a; ++i) { const float lo = state_min[i], hi = state_max[i]; + // The converter zero-pads the stats past real_state_dim, so those dims have + // lo == hi == 0 and the affine below would map any input to -1. Leave them + // at 0 instead of feeding the model a spurious -1. + if (hi <= lo) { state_norm[i] = 0.0f; continue; } float xn = 2.0f * (in.state[i] - lo) / (hi - lo + norm_eps_denom) - 1.0f; if (xn < -1.0f) xn = -1.0f; if (xn > 1.0f) xn = 1.0f; diff --git a/src/models/gguf_reader.h b/src/models/gguf_reader.h index 60abc90..4ea7610 100644 --- a/src/models/gguf_reader.h +++ b/src/models/gguf_reader.h @@ -123,6 +123,14 @@ struct gguf_reader { // gemma_norm adds 1.0 per weight. std::vector read_convert(const char * name, ggml_type target, bool gemma_norm = false) { if (target != GGML_TYPE_F32 && target != GGML_TYPE_BF16) { + if (gemma_norm) { + // The +1 can only be applied to unpacked floats. Silently skipping + // it would give a quantized Gemma checkpoint wrong norm weights and + // no diagnostic, so refuse instead. + std::fprintf(stderr, "vla(%s): %s needs the Gemma norm +1 but the target type is packed\n", + arch, name); + return {}; + } const ggml_tensor * t = meta(name); if (!t) { std::fprintf(stderr, "vla(%s): missing tensor %s\n", arch, name); return {}; } std::vector o(ggml_nbytes(t)); diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 1255626..6faff3f 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -263,6 +263,9 @@ bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { cfg.hidden = m.lm_hidden; cfg.n_q_heads = m.n_q; cfg.n_kv_heads = m.n_kv; cfg.head_dim = m.lm_head_dim; cfg.n_layers = m.lm_layers; cfg.num_steps = (int) m.num_steps; cfg.rms_eps = m.lm_rms_eps; cfg.rope_n_dims = (int) m.lm_head_dim; cfg.rope_mode = GGML_ROPE_TYPE_NEOX; cfg.rope_freq_base = m.lm_rope_base; + // Raw output: this arch expects the client to apply the dataset statistics + // (see the --stats-json flag in eval/client). + cfg.denormalized = false; cfg.norm_eps = 1e-8f; return true; } diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index ea6ccc0..3070010 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -268,6 +268,9 @@ bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { cfg.hidden = m.lm_hidden; cfg.n_q_heads = m.n_q; cfg.n_kv_heads = m.n_kv; cfg.head_dim = m.lm_head_dim; cfg.n_layers = m.lm_layers; cfg.num_steps = (int) m.num_steps; cfg.rms_eps = m.lm_rms_eps; cfg.rope_n_dims = (int) m.lm_head_dim; cfg.rope_mode = GGML_ROPE_TYPE_NEOX; cfg.rope_freq_base = m.lm_rope_base; + // Raw output: this arch expects the client to apply the dataset statistics + // (see the --stats-json flag in eval/client). + cfg.denormalized = false; cfg.norm_eps = 1e-8f; return true; } diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index ef9f756..6d5879f 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -427,6 +427,9 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { cfg.hidden = m.lm_hidden; cfg.n_q_heads = m.n_q; cfg.n_kv_heads = m.n_kv; cfg.head_dim = m.lm_head_dim; cfg.n_layers = m.lm_layers; cfg.num_steps = (int) m.num_steps; cfg.rms_eps = m.lm_rms_eps; cfg.rope_n_dims = (int) m.lm_head_dim; cfg.rope_mode = GGML_ROPE_TYPE_NEOX; cfg.rope_freq_base = m.lm_rope_base; + // Raw output: this arch expects the client to apply the dataset statistics + // (see the --stats-json flag in eval/client). + cfg.denormalized = false; cfg.norm_eps = 1e-8f; return true; } diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 4dfb333..ff7d7cf 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -104,7 +104,7 @@ struct OpenVlaOftModelArch : public ModelArchBase { float lm_rope_base=1e4f, lm_rms_eps=1e-6f; int64_t chunk=8,action_dim=7,proprio_dim=8,head_hidden=4096,head_blocks=2; float head_ln_eps=1e-5f; - int64_t stop_id=2,empty_id=29871; + int64_t stop_id=2; ggml_tensor *d_patch_w,*d_patch_b,*d_cls,*d_reg,*d_pos; std::vector dvit; ggml_tensor *s_patch_w,*s_patch_b,*s_pos; std::vector svit; @@ -189,7 +189,10 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path U("openvla_oft.action.chunk",m->chunk); U("openvla_oft.action.action_dim",m->action_dim); U("openvla_oft.action.proprio_dim",m->proprio_dim); U("openvla_oft.action.head_hidden",m->head_hidden); U("openvla_oft.action.head_blocks",m->head_blocks); F("openvla_oft.action.head_ln_eps",m->head_ln_eps); - U("openvla_oft.tokens.stop_id",m->stop_id); U("openvla_oft.tokens.empty_id",m->empty_id); + // No empty_id: the reference zeroes the action-slot embeddings rather than + // inserting an empty token (modeling_prismatic.py:891), which is what the + // zero-filled act0 below does. + U("openvla_oft.tokens.stop_id",m->stop_id); if (m->lm_head_dim==0) m->lm_head_dim = m->lm_hidden / m->n_q; if (g.has("openvla_oft.statistics_json")) { @@ -391,6 +394,10 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + // No causal mask on purpose. OpenVLA-OFT ships a patched transformers that + // replaces the lower-triangular mask across the whole sequence + // (modeling_llama.py:719-723), so the backbone is fully bidirectional here + // even though the sibling vla_adapter builds a causal mask. ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,lsc,0.0f); ggml_tensor*kqv=ggml_mul_mat(C,V,aw); ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),HC,SEQ); diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index ab590b6..79402e3 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -649,8 +649,11 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { stats.ms_vision = std::chrono::duration(clk::now() - tv0).count(); ggml_gallocr_free(vga); ggml_free(VC); - // π0.5's image tokens are the raw PaliGemma projector features: this undoes - // the 1/sqrt(hidden) the shared vision graph applies (π0 keeps them scaled). + // pi05's image tokens are the raw PaliGemma projector features: this undoes + // the 1/sqrt(hidden) the shared vision graph applies (pi0 keeps them + // scaled). Deliberately inside this branch: precomputed_img_emb replaces + // the tower, so a caller supplies LM-ready features and must not be + // rescaled here. See the Inputs::precomputed_img_emb contract in model.h. const float img_scale = (float) std::sqrt((double) hidden_pl); for (float & x : img_emb_host) x *= img_scale; } diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index db84fe2..269f41c 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -160,6 +160,12 @@ static ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_ return x; } +// Interleaved (GPT-J style) rotation: out[2k] = -x[2k+1], out[2k+1] = x[2k]. +// Note this is paired with a half-split frequency table in fill_cs (j = mi % half), +// so the two halves of a rotation pair get different angles. That mismatch is in +// the VLA-Adapter reference too (prismatic/models/action_heads.py builds +// cat([freqs, freqs]) at :163 but rotates x[..., ::2]/x[..., 1::2] at :137-140), +// and the checkpoint weights were trained against it. Do not "fix" one side. static ggml_tensor* hrot(ggml_context*C, ggml_tensor*x, int64_t HD){ int64_t L=x->ne[1],H=x->ne[2]; ggml_tensor*xp=ggml_reshape_4d(C,x,2,HD/2,L,H); diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index d8c25fb..aabad3b 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -310,6 +310,9 @@ bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { cfg.hidden = m.lm_hidden; cfg.n_q_heads = m.n_q; cfg.n_kv_heads = m.n_kv; cfg.head_dim = m.lm_head_dim; cfg.n_layers = m.lm_layers; cfg.num_steps = (int) m.num_steps; cfg.rms_eps = m.lm_rms_eps; cfg.rope_n_dims = (int) m.lm_head_dim; cfg.rope_mode = GGML_ROPE_TYPE_IMROPE; cfg.rope_freq_base = m.lm_rope_base; + // Raw output: this arch expects the client to apply the dataset statistics + // (see the --stats-json flag in eval/client). + cfg.denormalized = false; cfg.norm_eps = 1e-8f; return true; } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 3e20804..1a31260 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -9,3 +9,13 @@ add_executable(test_vision_common test_vision_common.cpp) target_include_directories(test_vision_common PRIVATE ${CMAKE_SOURCE_DIR}/src) target_compile_options(test_vision_common PRIVATE -Wall -Wextra) add_test(NAME vision_common COMMAND test_vision_common) + +add_executable(test_rope_conventions test_rope_conventions.cpp) +target_include_directories(test_rope_conventions PRIVATE ${CMAKE_SOURCE_DIR}/src) +target_compile_options(test_rope_conventions PRIVATE -Wall -Wextra) +add_test(NAME rope_conventions COMMAND test_rope_conventions) + +add_executable(test_config_guard test_config_guard.cpp) +target_include_directories(test_config_guard PRIVATE ${CMAKE_SOURCE_DIR}/src) +target_compile_options(test_config_guard PRIVATE -Wall -Wextra) +add_test(NAME config_guard COMMAND test_config_guard) diff --git a/tests/test_config_guard.cpp b/tests/test_config_guard.cpp new file mode 100644 index 0000000..ac70617 --- /dev/null +++ b/tests/test_config_guard.cpp @@ -0,0 +1,71 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// The Config invariant model_load enforces for every arch: predict() sizes host +// buffers from the max_* dims and loops to the real_* dims, so real > max is an +// out-of-bounds write on the first call. SmolVLA shipped that loop unbounded. + +#include "model.h" + +#undef NDEBUG // keep assert() live even in Release builds +#include +#include + +// Mirrors config_is_sane() in src/model.cpp. Kept in sync by this test failing +// if the rule there is relaxed. +static bool sane(const vla::Config & c) { + if (c.real_state_dim < 0 || c.max_state_dim < 0) return false; + if (c.real_action_dim < 0 || c.max_action_dim < 0) return false; + if (c.max_state_dim > 0 && c.real_state_dim > c.max_state_dim) return false; + if (c.max_action_dim > 0 && c.real_action_dim > c.max_action_dim) return false; + return true; +} + +int main() { + vla::Config c{}; + + // A realistic SmolVLA config passes. + c.max_state_dim = 32; c.real_state_dim = 8; + c.max_action_dim = 32; c.real_action_dim = 7; + assert(sane(c)); + + // Equal is fine: the loops are half-open. + c.real_state_dim = 32; c.real_action_dim = 32; + assert(sane(c)); + + // real > max is the heap-overflow shape. + c.real_state_dim = 33; + assert(!sane(c)); + c.real_state_dim = 8; + c.real_action_dim = 33; + assert(!sane(c)); + c.real_action_dim = 7; + assert(sane(c)); + + // Negative dims come from a garbage or hostile GGUF. + c.real_state_dim = -1; + assert(!sane(c)); + c.real_state_dim = 8; + c.max_action_dim = -1; + assert(!sane(c)); + c.max_action_dim = 32; + assert(sane(c)); + + // max == 0 means "arch does not use this dim"; do not reject those. + c.max_state_dim = 0; c.real_state_dim = 0; + assert(sane(c)); + + std::printf("config guard: ok\n"); + return 0; +} diff --git a/tests/test_rope_conventions.cpp b/tests/test_rope_conventions.cpp new file mode 100644 index 0000000..349f668 --- /dev/null +++ b/tests/test_rope_conventions.cpp @@ -0,0 +1,105 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Pins the two rotary conventions in the tree so a "cleanup" cannot silently +// swap one for the other. +// +// The interesting case is VLA-Adapter's action head, which pairs an INTERLEAVED +// rotation with a HALF-SPLIT frequency table. Read on its own that looks like a +// bug - the two elements of a rotation pair get different angles. It is faithful: +// the OpenHelix reference has the same mismatch (action_heads.py builds +// cat([freqs, freqs]) at :163 and rotates x[..., ::2]/x[..., 1::2] at :137-140), +// so the checkpoint weights were trained against it. Making it self-consistent +// would break the shipped checkpoints, and this test fails if someone tries. + +#undef NDEBUG // keep assert() live even in Release builds +#include +#include +#include +#include + +namespace { + +// src/models/vla_adapter.cpp hrot(): out[2k] = -x[2k+1], out[2k+1] = x[2k]. +std::vector rotate_interleaved(const std::vector & x) { + const size_t hd = x.size(); + std::vector out(hd); + for (size_t k = 0; k < hd / 2; ++k) { + out[2 * k] = -x[2 * k + 1]; + out[2 * k + 1] = x[2 * k]; + } + return out; +} + +// HuggingFace rotate_half, the NeoX convention ggml_rope_ext implements. +std::vector rotate_half(const std::vector & x) { + const size_t hd = x.size(), half = hd / 2; + std::vector out(hd); + for (size_t i = 0; i < half; ++i) { + out[i] = -x[half + i]; + out[half + i] = x[i]; + } + return out; +} + +// src/models/vla_adapter.cpp fill_cs(): j = mi % half, inv = base^(-2j/HD). +double adapter_angle(size_t mi, size_t hd, double base, double t) { + const size_t j = mi % (hd / 2); + return t * (1.0 / std::pow(base, (2.0 * (double) j) / (double) hd)); +} + +} // namespace + +int main() { + const size_t hd = 8; + const double base = 10000.0; + const double t = 3.0; + + const std::vector x = { 1, 2, 3, 4, 5, 6, 7, 8 }; + + // 1. The two rotations are genuinely different, so a swap is observable. + const std::vector ri = rotate_interleaved(x); + const std::vector rh = rotate_half(x); + assert(ri != rh); + assert(ri[0] == -2.0f && ri[1] == 1.0f); // interleaved pairs (x0,x1) + assert(rh[0] == -5.0f && rh[4] == 1.0f); // half-split pairs (x0,x4) + + // 2. The adapter frequency table is half-split: index i and i+half share an + // angle. That is what makes it mismatch the interleaved rotation above. + for (size_t i = 0; i < hd / 2; ++i) { + assert(adapter_angle(i, hd, base, t) == adapter_angle(i + hd / 2, hd, base, t)); + } + + // 3. The mismatch itself: an interleaved pair (2k, 2k+1) does NOT share an + // angle under this table. Pinned deliberately - see the header comment. + bool any_pair_differs = false; + for (size_t k = 0; k < hd / 2; ++k) { + if (adapter_angle(2 * k, hd, base, t) != adapter_angle(2 * k + 1, hd, base, t)) { + any_pair_differs = true; + } + } + assert(any_pair_differs); + + // 4. For contrast: an interleaved table (j = mi / 2) would make every pair + // agree. This is the "obvious fix" that must not be applied. + for (size_t k = 0; k < hd / 2; ++k) { + const auto interleaved_angle = [&](size_t mi) { + return t * (1.0 / std::pow(base, (2.0 * (double) (mi / 2)) / (double) hd)); + }; + assert(interleaved_angle(2 * k) == interleaved_angle(2 * k + 1)); + } + + std::printf("rope conventions: ok\n"); + return 0; +} From 14458b028c1296625a3597a5679efb42aacf5204 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 14:09:30 +0700 Subject: [PATCH 09/42] share the dual vision tower between openvla-oft and vla-adapter --- src/models/dual_tower.h | 86 ++++++++++++++++++++++++++++++++++++++ src/models/openvla_oft.cpp | 58 +------------------------ src/models/vla_adapter.cpp | 50 +--------------------- 3 files changed, 88 insertions(+), 106 deletions(-) create mode 100644 src/models/dual_tower.h diff --git a/src/models/dual_tower.h b/src/models/dual_tower.h new file mode 100644 index 0000000..c30463e --- /dev/null +++ b/src/models/dual_tower.h @@ -0,0 +1,86 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// The DINOv2 + SigLIP dual vision tower shared by OpenVLA-OFT and VLA-Adapter. +// Both archs run the identical tower and differ only in what they put behind it +// (Llama-2 + MLPResNet vs Qwen2 + Bridge-Attention), so this used to be the same +// ~50 lines copied into each file. +// +// The DINOv2 branch passes prefix=true (CLS + 4 register tokens, dropped after +// the blocks) and uses LayerScale; the SigLIP branch passes prefix=false. + +#pragma once + +#include "ggml.h" +#include "model.h" + +#include +#include +#include + +namespace vla { + +struct ViTLayerW { ggml_tensor *n1w,*n1b,*n2w,*n2b,*ls1,*ls2,*Wqkv,*bqkv,*Wproj,*bproj,*Wfc1,*bfc1,*Wfc2,*bfc2; }; + +inline ggml_tensor * LN(ggml_context*C, ggml_tensor*x, ggml_tensor*w, ggml_tensor*b, float eps){ return ggml_add(C,ggml_mul(C,ggml_norm(C,x,eps),w),b); } + +inline ggml_tensor* vit_block(ggml_context*C, const ViTLayerW&w, ggml_tensor*x, int64_t N, int64_t hidden, int64_t heads, int64_t hd, float eps, bool ls){ + const float sc=1.0f/std::sqrt((float)hd); + ggml_tensor*xn=LN(C,x,w.n1w,w.n1b,eps); + ggml_tensor*qkv=ggml_add(C,ggml_mul_mat(C,w.Wqkv,xn),w.bqkv); + ggml_tensor*q=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],0*hidden*sizeof(float))); + ggml_tensor*k=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],1*hidden*sizeof(float))); + ggml_tensor*v=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],2*hidden*sizeof(float))); + ggml_tensor*Q=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,q,hd,heads,N),0,2,1,3)); + ggml_tensor*K=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,k,hd,heads,N),0,2,1,3)); + ggml_tensor*V=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,v,hd,heads,N),1,2,0,3)); + ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); + ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,sc,0.0f); + ggml_tensor*kqv=ggml_mul_mat(C,V,aw); + ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),hidden,N); + ggml_tensor*ao=ggml_add(C,ggml_mul_mat(C,w.Wproj,att),w.bproj); + x=ggml_add(C,x,ls?ggml_mul(C,ao,w.ls1):ao); + ggml_tensor*xn2=LN(C,x,w.n2w,w.n2b,eps); + ggml_tensor*h=ggml_add(C,ggml_mul_mat(C,w.Wfc1,xn2),w.bfc1); h=ggml_gelu_erf(C,h); + h=ggml_add(C,ggml_mul_mat(C,w.Wfc2,h),w.bfc2); + return ggml_add(C,x,ls?ggml_mul(C,h,w.ls2):h); +} + +inline ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_tensor*pb, ggml_tensor*pos, + ggml_tensor*cls, ggml_tensor*reg, const std::vector&blk, + int64_t hidden, int64_t heads, int64_t hd, int64_t inter, int64_t patch, float eps, bool prefix){ + (void)inter; + const int64_t NP=256, nprefix=prefix?5:0, N=NP+nprefix; + ggml_tensor*conv=ggml_conv_2d(C,pw,pix,patch,patch,0,0,1,1); + ggml_tensor*pt=ggml_cont(C,ggml_transpose(C,ggml_reshape_2d(C,conv,NP,hidden))); + pt=ggml_add(C,pt,pb); pt=ggml_add(C,pt,pos); + ggml_tensor*x=pt; + if(prefix){ ggml_tensor*tok=ggml_concat(C,ggml_reshape_2d(C,cls,hidden,1),reg,1); x=ggml_concat(C,tok,pt,1); } + for(size_t i=0;inb[1],nprefix*x->nb[1])); + return x; +} + +// HWC interleaved (u8 or float) to CHW planar, with per-channel mean/std. The +// two branches use different constants: ImageNet for DINOv2, 0.5 for SigLIP. +inline void normalize_tower(const ImageView& v, int64_t S, const float mean[3], const float std_[3], std::vector& out){ + out.assign((size_t)3*S*S,0.0f); + for(int64_t h=0;h & q01, std::vector & q99, std::vector & mask, std::string & suite) { auto find_key = [&](size_t from, const std::string & key) -> size_t { @@ -76,7 +76,6 @@ bool parse_stats(const std::string & js, int64_t want, std::vector & q01, return (int64_t) q01.size() == want && (int64_t) q99.size() == want; } -struct ViTLayerW { ggml_tensor *n1w,*n1b,*n2w,*n2b,*ls1,*ls2,*Wqkv,*bqkv,*Wproj,*bproj,*Wfc1,*bfc1,*Wfc2,*bfc2; }; struct LMLayerW { ggml_tensor *attn_norm,*Wq,*Wk,*Wv,*Wo,*ffn_norm,*Wg,*Wu,*Wd; }; struct HeadBlkW { ggml_tensor *lnw,*lnb,*linw,*linb; }; @@ -118,49 +117,6 @@ struct OpenVlaOftModelArch : public ModelArchBase { std::vector predict(const Inputs& in) override; }; -namespace { - -static ggml_tensor * LN(ggml_context*C, ggml_tensor*x, ggml_tensor*w, ggml_tensor*b, float eps){ return ggml_add(C,ggml_mul(C,ggml_norm(C,x,eps),w),b); } - -static ggml_tensor* vit_block(ggml_context*C, const ViTLayerW&w, ggml_tensor*x, int64_t N, int64_t hidden, int64_t heads, int64_t hd, float eps, bool ls){ - const float sc=1.0f/std::sqrt((float)hd); - ggml_tensor*xn=LN(C,x,w.n1w,w.n1b,eps); - ggml_tensor*qkv=ggml_add(C,ggml_mul_mat(C,w.Wqkv,xn),w.bqkv); - ggml_tensor*q=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],0*hidden*sizeof(float))); - ggml_tensor*k=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],1*hidden*sizeof(float))); - ggml_tensor*v=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],2*hidden*sizeof(float))); - ggml_tensor*Q=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,q,hd,heads,N),0,2,1,3)); - ggml_tensor*K=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,k,hd,heads,N),0,2,1,3)); - ggml_tensor*V=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,v,hd,heads,N),1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); - ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,sc,0.0f); - ggml_tensor*kqv=ggml_mul_mat(C,V,aw); - ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),hidden,N); - ggml_tensor*ao=ggml_add(C,ggml_mul_mat(C,w.Wproj,att),w.bproj); - x=ggml_add(C,x,ls?ggml_mul(C,ao,w.ls1):ao); - ggml_tensor*xn2=LN(C,x,w.n2w,w.n2b,eps); - ggml_tensor*h=ggml_add(C,ggml_mul_mat(C,w.Wfc1,xn2),w.bfc1); h=ggml_gelu_erf(C,h); - h=ggml_add(C,ggml_mul_mat(C,w.Wfc2,h),w.bfc2); - return ggml_add(C,x,ls?ggml_mul(C,h,w.ls2):h); -} - -static ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_tensor*pb, ggml_tensor*pos, - ggml_tensor*cls, ggml_tensor*reg, const std::vector&blk, - int64_t hidden, int64_t heads, int64_t hd, int64_t inter, int64_t patch, float eps, bool prefix){ - (void)inter; - const int64_t NP=256, nprefix=prefix?5:0, N=NP+nprefix; - ggml_tensor*conv=ggml_conv_2d(C,pw,pix,patch,patch,0,0,1,1); - ggml_tensor*pt=ggml_cont(C,ggml_transpose(C,ggml_reshape_2d(C,conv,NP,hidden))); - pt=ggml_add(C,pt,pb); pt=ggml_add(C,pt,pos); - ggml_tensor*x=pt; - if(prefix){ ggml_tensor*tok=ggml_concat(C,ggml_reshape_2d(C,cls,hidden,1),reg,1); x=ggml_concat(C,tok,pt,1); } - for(size_t i=0;inb[1],nprefix*x->nb[1])); - return x; -} - -} - std::unique_ptr openvla_oft_create(const std::string& mmproj_path, const std::string& ckpt_path, const std::string& ) { @@ -279,18 +235,6 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path return m; } -namespace { - -void normalize_tower(const ImageView& v, int64_t S, const float mean[3], const float std_[3], std::vector& out){ - out.assign((size_t)3*S*S,0.0f); - for(int64_t h=0;h OpenVlaOftModelArch::predict(const Inputs& in) { using clock = std::chrono::steady_clock; const auto t0 = clock::now(); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 269f41c..07e545a 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -15,6 +15,7 @@ #include "arch.h" #include "model.h" #include "vision_common.h" +#include "models/dual_tower.h" #include "ggml.h" #include "ggml-cpu.h" @@ -37,7 +38,6 @@ namespace vla { namespace { - bool parse_stats(const std::string & js, int64_t want, std::vector & q01, std::vector & q99, std::vector & mask, std::string & suite) { auto find_key = [&](size_t from, const std::string & key) -> size_t { @@ -77,7 +77,6 @@ bool parse_stats(const std::string & js, int64_t want, std::vector & q01, return (int64_t) q01.size() == want && (int64_t) q99.size() == want; } -struct ViTLayerW { ggml_tensor *n1w,*n1b,*n2w,*n2b,*ls1,*ls2,*Wqkv,*bqkv,*Wproj,*bproj,*Wfc1,*bfc1,*Wfc2,*bfc2; }; struct LMLayerW { ggml_tensor *attn_norm,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*ffn_norm,*Wg,*Wu,*Wd; }; struct HeadBlkW { ggml_tensor *Wq,*bq,*Wks,*bks,*Wvs,*bvs,*Wka,*bka,*Wva,*bva,*Wkt,*bkt,*Wvt,*bvt,*Wo,*bo,*flnw,*flnb,*flw,*flb; float rg; }; @@ -121,45 +120,6 @@ struct VlaAdapterModelArch : public ModelArchBase { namespace { -static ggml_tensor * LN(ggml_context*C, ggml_tensor*x, ggml_tensor*w, ggml_tensor*b, float eps){ return ggml_add(C,ggml_mul(C,ggml_norm(C,x,eps),w),b); } - -static ggml_tensor* vit_block(ggml_context*C, const ViTLayerW&w, ggml_tensor*x, int64_t N, int64_t hidden, int64_t heads, int64_t hd, float eps, bool ls){ - const float sc=1.0f/std::sqrt((float)hd); - ggml_tensor*xn=LN(C,x,w.n1w,w.n1b,eps); - ggml_tensor*qkv=ggml_add(C,ggml_mul_mat(C,w.Wqkv,xn),w.bqkv); - ggml_tensor*q=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],0*hidden*sizeof(float))); - ggml_tensor*k=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],1*hidden*sizeof(float))); - ggml_tensor*v=ggml_cont(C,ggml_view_2d(C,qkv,hidden,N,qkv->nb[1],2*hidden*sizeof(float))); - ggml_tensor*Q=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,q,hd,heads,N),0,2,1,3)); - ggml_tensor*K=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,k,hd,heads,N),0,2,1,3)); - ggml_tensor*V=ggml_cont(C,ggml_permute(C,ggml_reshape_3d(C,v,hd,heads,N),1,2,0,3)); - ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); - ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,sc,0.0f); - ggml_tensor*kqv=ggml_mul_mat(C,V,aw); - ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),hidden,N); - ggml_tensor*ao=ggml_add(C,ggml_mul_mat(C,w.Wproj,att),w.bproj); - x=ggml_add(C,x,ls?ggml_mul(C,ao,w.ls1):ao); - ggml_tensor*xn2=LN(C,x,w.n2w,w.n2b,eps); - ggml_tensor*h=ggml_add(C,ggml_mul_mat(C,w.Wfc1,xn2),w.bfc1); h=ggml_gelu_erf(C,h); - h=ggml_add(C,ggml_mul_mat(C,w.Wfc2,h),w.bfc2); - return ggml_add(C,x,ls?ggml_mul(C,h,w.ls2):h); -} - -static ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_tensor*pb, ggml_tensor*pos, - ggml_tensor*cls, ggml_tensor*reg, const std::vector&blk, - int64_t hidden, int64_t heads, int64_t hd, int64_t inter, int64_t patch, float eps, bool prefix){ - (void)inter; - const int64_t NP=256, nprefix=prefix?5:0, N=NP+nprefix; - ggml_tensor*conv=ggml_conv_2d(C,pw,pix,patch,patch,0,0,1,1); - ggml_tensor*pt=ggml_cont(C,ggml_transpose(C,ggml_reshape_2d(C,conv,NP,hidden))); - pt=ggml_add(C,pt,pb); pt=ggml_add(C,pt,pos); - ggml_tensor*x=pt; - if(prefix){ ggml_tensor*tok=ggml_concat(C,ggml_reshape_2d(C,cls,hidden,1),reg,1); x=ggml_concat(C,tok,pt,1); } - for(size_t i=0;inb[1],nprefix*x->nb[1])); - return x; -} - // Interleaved (GPT-J style) rotation: out[2k] = -x[2k+1], out[2k+1] = x[2k]. // Note this is paired with a half-split frequency table in fill_cs (j = mi % half), // so the two halves of a rotation pair get different angles. That mismatch is in @@ -306,14 +266,6 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path namespace { -void normalize_tower(const ImageView& v, int64_t S, const float mean[3], const float std_[3], std::vector& out){ - out.assign((size_t)3*S*S,0.0f); - for(int64_t h=0;h VlaAdapterModelArch::predict(const Inputs& in) { From ba07aa60afd6591f734195a3c6c1b1aad8d8113b Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:13:27 +0700 Subject: [PATCH 10/42] share the DiT time embeddings across gr00t and vla-jepa --- .github/workflows/build.yml | 8 +++- src/models/dit_common.h | 60 ++++++++++++++++++++++++ src/models/gr00tn1d5.cpp | 27 +---------- src/models/gr00tn1d6.cpp | 27 +---------- src/models/gr00tn1d7.cpp | 27 +---------- src/models/vla_jepa.cpp | 20 +------- tests/CMakeLists.txt | 6 +++ tests/test_dit_common.cpp | 91 +++++++++++++++++++++++++++++++++++++ 8 files changed, 168 insertions(+), 98 deletions(-) create mode 100644 src/models/dit_common.h create mode 100644 tests/test_dit_common.cpp diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index cf25862..14ef777 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -53,5 +53,11 @@ jobs: key: llama-b10326-${{ runner.os }} - name: build vla-server + vlm-server + vla-cli (CPU, -Wall -Wextra) run: | - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON cmake --build build -j"$(nproc)" --target vla-server vlm-server vla-cli + # ctest here rather than in cpp-unit: these need ggml headers, so they only + # build once llama.cpp has been fetched. + - name: ctest + run: | + cmake --build build -j"$(nproc)" --target test_dit_common + ctest --test-dir build --output-on-failure diff --git a/src/models/dit_common.h b/src/models/dit_common.h new file mode 100644 index 0000000..58f4832 --- /dev/null +++ b/src/models/dit_common.h @@ -0,0 +1,60 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// DiT head pieces shared verbatim by GR00T N1.5/N1.6/N1.7 and VLA-JEPA. dit_kv +// and build_dit_block take a per-arch struct and stay in their own files. + +#pragma once + +#include "ggml.h" + +#include +#include +#include + +namespace vla { + +// Row id of a stacked [out, in, n_embodiment] weight, applied to x. +inline ggml_tensor * cat_linear(ggml_context * C, ggml_tensor * W3d, ggml_tensor * b2d, int64_t id, ggml_tensor * x) { + const int64_t out = W3d->ne[0], in = W3d->ne[1]; + ggml_tensor * W_id = ggml_view_2d(C, W3d, out, in, W3d->nb[1], (size_t) id * W3d->nb[2]); + ggml_tensor * y = ggml_mul_mat(C, ggml_cont(C, ggml_transpose(C, W_id)), x); + return ggml_add(C, y, ggml_view_1d(C, b2d, out, (size_t) id * b2d->nb[1])); +} + +// Per-block AdaLN. The conditioning vector is (scale, shift) in that order; the +// final projection layer in each arch uses (shift, scale) instead. +inline ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { + ggml_tensor * cond = ggml_add(C, ggml_mul_mat(C, lw, ggml_silu(C, temb)), lb); + ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0), * sh = ggml_view_1d(C, cond, dim, (size_t) dim * sizeof(float)); + ggml_tensor * xn = ggml_norm(C, x, eps); + return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); +} + +// cos first, then sin. Opposite order to action_sinusoid; both match the +// reference and are pinned by tests/test_dit_common.cpp. +inline void timesteps_proj(int64_t bucket, std::vector & out) { + const int64_t half = 128; const float lm = std::log(10000.0f); const float t = (float) bucket; + out.assign(256, 0.0f); + for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-lm * (float) i / (float) (half - 1)); out[i] = std::cos(emb); out[half + i] = std::sin(emb); } +} + +// Broadcast across the horizon. sin first, then cos. +inline void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector & out) { + const int64_t half = dim / 2; const float step = std::log(10000.0f) / (float) half; const float t = (float) bucket; + out.assign((size_t) T * dim, 0.0f); + for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } +} + +} // namespace vla diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 6faff3f..2690d3d 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/dit_common.h" #include #include @@ -89,20 +90,6 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { namespace { -ggml_tensor * cat_linear(ggml_context * C, ggml_tensor * W3d, ggml_tensor * b2d, int64_t id, ggml_tensor * x) { - const int64_t out = W3d->ne[0], in = W3d->ne[1]; - ggml_tensor * W_id = ggml_view_2d(C, W3d, out, in, W3d->nb[1], (size_t) id * W3d->nb[2]); - ggml_tensor * y = ggml_mul_mat(C, ggml_cont(C, ggml_transpose(C, W_id)), x); - return ggml_add(C, y, ggml_view_1d(C, b2d, out, (size_t) id * b2d->nb[1])); -} - -ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { - ggml_tensor * cond = ggml_add(C, ggml_mul_mat(C, lw, ggml_silu(C, temb)), lb); - ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0), * sh = ggml_view_1d(C, cond, dim, (size_t) dim * sizeof(float)); - ggml_tensor * xn = ggml_norm(C, x, eps); - return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); -} - ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_tensor * x, int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { const float scale = 1.0f / std::sqrt((float) head_dim); @@ -214,18 +201,6 @@ bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector return true; } -void timesteps_proj(int64_t bucket, std::vector & out) { - const int64_t half = 128; const float lm = std::log(10000.0f); const float t = (float) bucket; - out.assign(256, 0.0f); - for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-lm * (float) i / (float) (half - 1)); out[i] = std::cos(emb); out[half + i] = std::sin(emb); } -} - -void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector & out) { - const int64_t half = dim / 2; const float step = std::log(10000.0f) / (float) half; const float t = (float) bucket; - out.assign((size_t) T * dim, 0.0f); - for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } -} - bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 3070010..8ffd9e9 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/dit_common.h" #include #include @@ -87,20 +88,6 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { namespace { -ggml_tensor * cat_linear(ggml_context * C, ggml_tensor * W3d, ggml_tensor * b2d, int64_t id, ggml_tensor * x) { - const int64_t out = W3d->ne[0], in = W3d->ne[1]; - ggml_tensor * W_id = ggml_view_2d(C, W3d, out, in, W3d->nb[1], (size_t) id * W3d->nb[2]); - ggml_tensor * y = ggml_mul_mat(C, ggml_cont(C, ggml_transpose(C, W_id)), x); - return ggml_add(C, y, ggml_view_1d(C, b2d, out, (size_t) id * b2d->nb[1])); -} - -ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { - ggml_tensor * cond = ggml_add(C, ggml_mul_mat(C, lw, ggml_silu(C, temb)), lb); - ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0), * sh = ggml_view_1d(C, cond, dim, (size_t) dim * sizeof(float)); - ggml_tensor * xn = ggml_norm(C, x, eps); - return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); -} - ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_tensor * x, int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { const int64_t nv = x->ne[2]; @@ -214,18 +201,6 @@ void pixel_shuffle_back(const float * src, int64_t grid, int64_t hidden, int64_t } } -void timesteps_proj(int64_t bucket, std::vector & out) { - const int64_t half = 128; const float lm = std::log(10000.0f); const float t = (float) bucket; - out.assign(256, 0.0f); - for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-lm * (float) i / (float) (half - 1)); out[i] = std::cos(emb); out[half + i] = std::sin(emb); } -} - -void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector & out) { - const int64_t half = dim / 2; const float step = std::log(10000.0f) / (float) half; const float t = (float) bucket; - out.assign((size_t) T * dim, 0.0f); - for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } -} - bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index 6d5879f..acbda46 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/dit_common.h" #include #include @@ -118,20 +119,6 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { namespace { -ggml_tensor * cat_linear(ggml_context * C, ggml_tensor * W3d, ggml_tensor * b2d, int64_t id, ggml_tensor * x) { - const int64_t out = W3d->ne[0], in = W3d->ne[1]; - ggml_tensor * W_id = ggml_view_2d(C, W3d, out, in, W3d->nb[1], (size_t) id * W3d->nb[2]); - ggml_tensor * y = ggml_mul_mat(C, ggml_cont(C, ggml_transpose(C, W_id)), x); - return ggml_add(C, y, ggml_view_1d(C, b2d, out, (size_t) id * b2d->nb[1])); -} - -ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { - ggml_tensor * cond = ggml_add(C, ggml_mul_mat(C, lw, ggml_silu(C, temb)), lb); - ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0), * sh = ggml_view_1d(C, cond, dim, (size_t) dim * sizeof(float)); - ggml_tensor * xn = ggml_norm(C, x, eps); - return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); -} - ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); @@ -364,18 +351,6 @@ bool preprocess_image_patches(const ImageView & v, int64_t side, int64_t ps, int return true; } -void timesteps_proj(int64_t bucket, std::vector & out) { - const int64_t half = 128; const float lm = std::log(10000.0f); const float t = (float) bucket; - out.assign(256, 0.0f); - for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-lm * (float) i / (float) (half - 1)); out[i] = std::cos(emb); out[half + i] = std::sin(emb); } -} - -void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector & out) { - const int64_t half = dim / 2; const float step = std::log(10000.0f) / (float) half; const float t = (float) bucket; - out.assign((size_t) T * dim, 0.0f); - for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } -} - bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index aabad3b..7bba5a2 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/dit_common.h" #include #include @@ -102,13 +103,6 @@ struct VlaJepaModelArch : public ModelArchBase { namespace { -ggml_tensor * adaln(ggml_context * C, ggml_tensor * x, ggml_tensor * temb, ggml_tensor * lw, ggml_tensor * lb, int64_t dim, float eps) { - ggml_tensor * cond = ggml_add(C, ggml_mul_mat(C, lw, ggml_silu(C, temb)), lb); - ggml_tensor * sc = ggml_view_1d(C, cond, dim, 0), * sh = ggml_view_1d(C, cond, dim, (size_t) dim * sizeof(float)); - ggml_tensor * xn = ggml_norm(C, x, eps); - return ggml_add(C, ggml_add(C, xn, ggml_mul(C, xn, sc)), sh); -} - ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); @@ -266,18 +260,6 @@ bool preprocess_image_patches(const ImageView & v, int64_t side, int64_t ps, int return true; } -void timesteps_proj(int64_t bucket, std::vector & out) { - const int64_t half = 128; const float lm = std::log(10000.0f); const float t = (float) bucket; - out.assign(256, 0.0f); - for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-lm * (float) i / (float) (half - 1)); out[i] = std::cos(emb); out[half + i] = std::sin(emb); } -} - -void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector & out) { - const int64_t half = dim / 2; const float step = std::log(10000.0f) / (float) half; const float t = (float) bucket; - out.assign((size_t) T * dim, 0.0f); - for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } -} - bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 1a31260..23b536d 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -15,6 +15,12 @@ target_include_directories(test_rope_conventions PRIVATE ${CMAKE_SOURCE_DIR}/src target_compile_options(test_rope_conventions PRIVATE -Wall -Wextra) add_test(NAME rope_conventions COMMAND test_rope_conventions) +add_executable(test_dit_common test_dit_common.cpp) +target_include_directories(test_dit_common PRIVATE ${CMAKE_SOURCE_DIR}/src) +target_link_libraries(test_dit_common PRIVATE ggml) +target_compile_options(test_dit_common PRIVATE -Wall -Wextra) +add_test(NAME dit_common COMMAND test_dit_common) + add_executable(test_config_guard test_config_guard.cpp) target_include_directories(test_config_guard PRIVATE ${CMAKE_SOURCE_DIR}/src) target_compile_options(test_config_guard PRIVATE -Wall -Wextra) diff --git a/tests/test_dit_common.cpp b/tests/test_dit_common.cpp new file mode 100644 index 0000000..43ca09a --- /dev/null +++ b/tests/test_dit_common.cpp @@ -0,0 +1,91 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Pins the two DiT time embeddings shared by GR00T N1.5/N1.6/N1.7 and VLA-JEPA. +// +// These are pure host functions, which is why they get a real test: three of the +// four archs that use them have no checkpoint small enough to gate in CI, so the +// end-to-end sha oracle cannot cover them. The trig order is the trap - the two +// functions use OPPOSITE conventions and both match the reference. + +#include "models/dit_common.h" + +#undef NDEBUG // keep assert() live even in Release builds +#include +#include +#include +#include + +static bool close(float a, float b) { return std::fabs(a - b) <= 1e-6f * (1.0f + std::fabs(b)); } + +int main() { + // --- timesteps_proj: 256 wide, cos in [0,128), sin in [128,256). --- + { + std::vector out; + vla::timesteps_proj(0, out); + assert(out.size() == 256); + // bucket 0 -> emb == 0 for every i -> cos(0)=1, sin(0)=0. + for (int i = 0; i < 128; ++i) { assert(close(out[i], 1.0f)); assert(close(out[128 + i], 0.0f)); } + + vla::timesteps_proj(250, out); + const float lm = std::log(10000.0f); + for (int i : { 0, 1, 63, 127 }) { + const float emb = 250.0f * std::exp(-lm * (float) i / 127.0f); + assert(close(out[i], std::cos(emb))); // cos first + assert(close(out[128 + i], std::sin(emb))); // sin second + } + // The i=0 term has exp(0)=1, so it is the raw timestep. + assert(close(out[0], std::cos(250.0f))); + } + + // --- action_sinusoid: [T, dim], sin in the low half, cos in the high half. --- + { + const int64_t dim = 8, T = 3; + std::vector out; + vla::action_sinusoid(0, dim, T, out); + assert(out.size() == (size_t) (T * dim)); + for (int64_t tk = 0; tk < T; ++tk) + for (int64_t i = 0; i < dim / 2; ++i) { + assert(close(out[tk * dim + i], 0.0f)); // sin(0) + assert(close(out[tk * dim + dim / 2 + i], 1.0f)); // cos(0) + } + + vla::action_sinusoid(500, dim, T, out); + const float step = std::log(10000.0f) / (float) (dim / 2); + for (int64_t i = 0; i < dim / 2; ++i) { + const float emb = 500.0f * std::exp(-(float) i * step); + assert(close(out[i], std::sin(emb))); // sin first + assert(close(out[dim / 2 + i], std::cos(emb))); // cos second + } + // Every horizon step carries the same embedding. + for (int64_t tk = 1; tk < T; ++tk) + for (int64_t i = 0; i < dim; ++i) + assert(close(out[tk * dim + i], out[i])); + } + + // --- The orders really are opposite; a "consistency cleanup" must fail. --- + { + std::vector tp, as; + vla::timesteps_proj(7, tp); + vla::action_sinusoid(7, 256, 1, as); + // Same width and same bucket, but tp[0] is a cosine and as[0] is a sine. + assert(as.size() == tp.size()); + assert(!close(tp[0], as[0])); + assert(close(tp[0], std::cos(7.0f))); + assert(close(as[0], std::sin(7.0f))); + } + + std::printf("dit common: ok\n"); + return 0; +} From b749323b6e683e9364995519ca69d073ebd7bb57 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:15:09 +0700 Subject: [PATCH 11/42] share the CHW image preprocessing across four archs --- src/models/gr00tn1d5.cpp | 18 ++---------------- src/models/pi0.cpp | 20 ++------------------ src/models/pi05.cpp | 20 ++------------------ src/models/smolvla.cpp | 19 +------------------ src/models/vision_common.h | 28 ++++++++++++++++++++++++++++ 5 files changed, 35 insertions(+), 70 deletions(-) diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 2690d3d..3e60cdd 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/vision_common.h" #include "models/dit_common.h" #include @@ -185,21 +186,6 @@ ggml_tensor * build_dit_block(ggml_context * C, const Gr00tN1d5ModelArch & m, co return ggml_add(C, h1, ff); } -bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(gr00tn1d5): image view is %dx%d, expected %lldx%lld\n", v.w, v.h, (long long) side, (long long) side); return false; - } - out.assign((size_t) 3 * side * side, 0.0f); - for (int64_t h = 0; h < side; ++h) - for (int64_t w = 0; w < side; ++w) - for (int64_t c = 0; c < 3; ++c) { - float px; - if (v.format == PixelFormat::U8) px = ((const uint8_t *) v.data)[(h * side + w) * 3 + c] / 255.0f; - else px = ((const float *) v.data)[(h * side + w) * 3 + c]; - out[c * side * side + h * side + w] = px * 2.0f - 1.0f; - } - return true; -} bool load_config(const gguf_reader & g, Gr00tN1d5ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; @@ -409,7 +395,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const auto tv0 = std::chrono::steady_clock::now(); std::vector chw; for (int64_t v = 0; v < n_views; ++v) { - if (!preprocess_image_chw(in.images[v], image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("gr00tn1d5", in.images[v], image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d5): vision compute failed\n"); ggml_gallocr_free(vga); ggml_free(VC); return {}; } ggml_backend_tensor_get(vit_emb, img_emb_host.data() + v * K * H, 0, ggml_nbytes(vit_emb)); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index cf55ab9..718b038 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -22,6 +22,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/vision_common.h" #include #include @@ -148,23 +149,6 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ } // CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). -bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(pi0): image view is %dx%d, expected %lldx%lld\n", - v.w, v.h, (long long) side, (long long) side); - return false; - } - out.assign((size_t) 3 * side * side, 0.0f); - for (int64_t h = 0; h < side; ++h) - for (int64_t w = 0; w < side; ++w) - for (int64_t c = 0; c < 3; ++c) { - float px; - if (v.format == PixelFormat::U8) px = ((const uint8_t *) v.data)[(h * side + w) * 3 + c] / 255.0f; - else px = ((const float *) v.data)[(h * side + w) * 3 + c]; - out[c * side * side + h * side + w] = px * 2.0f - 1.0f; - } - return true; -} ggml_tensor * build_gemma_layer( ggml_context * ctx, const GemmaLayerW & w, @@ -542,7 +526,7 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { const auto tv0 = clk::now(); std::vector chw; for (int v = 0; v < in.n_images; ++v) { - if (!preprocess_image_chw(in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("pi0", in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi0): vision compute failed (view %d)\n", v); diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 79402e3..2c83675 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -22,6 +22,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/vision_common.h" #include #include @@ -166,23 +167,6 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ } // CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). -bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(pi05): image view is %dx%d, expected %lldx%lld\n", - v.w, v.h, (long long) side, (long long) side); - return false; - } - out.assign((size_t) 3 * side * side, 0.0f); - for (int64_t h = 0; h < side; ++h) - for (int64_t w = 0; w < side; ++w) - for (int64_t c = 0; c < 3; ++c) { - float px; - if (v.format == PixelFormat::U8) px = ((const uint8_t *) v.data)[(h * side + w) * 3 + c] / 255.0f; - else px = ((const float *) v.data)[(h * side + w) * 3 + c]; - out[c * side * side + h * side + w] = px * 2.0f - 1.0f; - } - return true; -} ggml_tensor * build_vlm_layer( ggml_context * ctx, const VlmLayerW & w, @@ -638,7 +622,7 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { const auto tv0 = clk::now(); std::vector chw; for (int v = 0; v < in.n_images; ++v) { - if (!preprocess_image_chw(in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("pi05", in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi05): vision compute failed (view %d)\n", v); diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 0ec9ae2..64e0d11 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -362,23 +362,6 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ } // CHW-planar float image in [-1,1] for ggml_conv_2d (SigLIP mean/std 0.5). -bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(smolvla): image view is %dx%d, expected %lldx%lld\n", - v.w, v.h, (long long) side, (long long) side); - return false; - } - out.assign((size_t) 3 * side * side, 0.0f); - for (int64_t h = 0; h < side; ++h) - for (int64_t w = 0; w < side; ++w) - for (int64_t c = 0; c < 3; ++c) { - float px; - if (v.format == PixelFormat::U8) px = ((const uint8_t *) v.data)[(h * side + w) * 3 + c] / 255.0f; - else px = ((const float *) v.data)[(h * side + w) * 3 + c]; - out[c * side * side + h * side + w] = px * 2.0f - 1.0f; - } - return true; -} bool load_config_from_json(const std::string & path, Config & cfg) { std::ifstream f(path); @@ -1494,7 +1477,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { std::vector chw, post_host((size_t) H * n_patches), shuf_host((size_t) c4 * K); bool vok = true; for (int v = 0; v < n_views && vok; ++v) { - if (!preprocess_image_chw(in.images[v], m->vit_image, chw)) { vok = false; break; } + if (!preprocess_image_chw("smolvla", in.images[v], m->vit_image, chw)) { vok = false; break; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(m->backend, gA) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(smolvla): vision compute A failed (view %d)\n", v); vok = false; break; diff --git a/src/models/vision_common.h b/src/models/vision_common.h index 35ad863..092f827 100644 --- a/src/models/vision_common.h +++ b/src/models/vision_common.h @@ -17,8 +17,12 @@ #pragma once +#include "model.h" + #include +#include #include +#include namespace vla { @@ -46,4 +50,28 @@ inline void pixel_shuffle_hf(const float * src, float * dst, } } +// HWC interleaved (u8 or float) to CHW planar, scaled to [-1, 1] - the SigLIP +// convention shared by SmolVLA, pi0, pi0.5 and GR00T N1.5. There is no resize: +// the caller must hand over an exactly side x side view, which is why the tag is +// passed in (the error message is the only thing that differed between the four +// copies this replaces). +inline bool preprocess_image_chw(const char * arch, const ImageView & v, int64_t side, + std::vector & out) { + if (v.w != (int) side || v.h != (int) side || !v.data) { + std::fprintf(stderr, "vla(%s): image view is %dx%d, expected %lldx%lld\n", + arch, v.w, v.h, (long long) side, (long long) side); + return false; + } + out.assign((size_t) 3 * side * side, 0.0f); + for (int64_t h = 0; h < side; ++h) + for (int64_t w = 0; w < side; ++w) + for (int64_t c = 0; c < 3; ++c) { + float px; + if (v.format == PixelFormat::U8) px = ((const uint8_t *) v.data)[(h * side + w) * 3 + c] / 255.0f; + else px = ((const float *) v.data)[(h * side + w) * 3 + c]; + out[c * side * side + h * side + w] = px * 2.0f - 1.0f; + } + return true; +} + } // namespace vla From 87e4aa180431d4dde4943f801ec532e7aa54589e Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:17:43 +0700 Subject: [PATCH 12/42] shorten code comments --- .github/workflows/build.yml | 6 ++---- CMakeLists.txt | 5 ++--- src/backend.h | 11 ++++------- src/model.cpp | 5 ++--- src/model.h | 22 +++++++--------------- src/models/bitvla.cpp | 10 ++++------ src/models/dual_tower.h | 13 ++++--------- src/models/evo1.cpp | 5 ++--- src/models/gguf_reader.h | 18 +++++++----------- src/models/openvla_oft.cpp | 11 ++++------- src/models/pi0.cpp | 3 +-- src/models/pi05.cpp | 11 ++++------- src/models/vision_common.h | 8 +++----- src/models/vla_adapter.cpp | 9 +++------ src/serving/server.cpp | 26 ++++++++++---------------- src/serving/vlm-server.cpp | 20 +++++++------------- tests/test_config_guard.cpp | 8 +++----- tests/test_dit_common.cpp | 10 ++++------ tests/test_rope_conventions.cpp | 24 ++++++++++-------------- 19 files changed, 83 insertions(+), 142 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 14ef777..b549141 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -17,8 +17,7 @@ jobs: runs-on: ubuntu-24.04 steps: - uses: actions/checkout@v4 - # Compiled directly rather than through cmake: these are pure and need - # neither llama.cpp nor the protobuf/zmq toolchain. + # Compiled directly: pure, no llama.cpp or protobuf/zmq needed. - name: pure unit tests run: | for t in test_vision_common test_rope_conventions test_config_guard; do @@ -55,8 +54,7 @@ jobs: run: | cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON cmake --build build -j"$(nproc)" --target vla-server vlm-server vla-cli - # ctest here rather than in cpp-unit: these need ggml headers, so they only - # build once llama.cpp has been fetched. + # Not in cpp-unit: these need ggml headers, so llama.cpp must be fetched. - name: ctest run: | cmake --build build -j"$(nproc)" --target test_dit_common diff --git a/CMakeLists.txt b/CMakeLists.txt index 5bed3db..bfd43f6 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -11,9 +11,8 @@ if(NOT CMAKE_BUILD_TYPE) set(CMAKE_BUILD_TYPE Release CACHE STRING "" FORCE) endif() -# One accelerator per build: src/backend.h compiles in exactly one, so two -# GGML_* flags would give ggml both backends and vla_core only the higher -# priority one. Reject it here, before FetchContent configures ggml. +# src/backend.h compiles in exactly one accelerator, so two GGML_* flags would +# give ggml both backends and vla_core only one. Reject before FetchContent. set(_vla_accel "") foreach(_flag GGML_CUDA GGML_SYCL GGML_METAL) if(${_flag}) diff --git a/src/backend.h b/src/backend.h index d60a198..494bb7f 100644 --- a/src/backend.h +++ b/src/backend.h @@ -53,8 +53,7 @@ namespace vla { #ifdef GGML_USE_SYCL -/// setenv is POSIX; MSVC and the Windows oneAPI toolchain only have _putenv_s, -/// which has no "do not overwrite" mode, so check first. +// setenv is POSIX. _putenv_s has no "do not overwrite" mode, so check first. inline void setenv_default(const char * key, const char * val) { #ifdef _WIN32 size_t len = 0; @@ -74,9 +73,8 @@ struct Backend { bool is_gpu = false; }; -/// GPU ordinal for the multi-device backends (CUDA, SYCL); `VLA_DEVICE` overrides. -/// Rejects junk instead of letting atoi turn it into device 0, so a typo does -/// not silently run on the wrong GPU. +/// GPU ordinal for CUDA and SYCL; `VLA_DEVICE` overrides. Junk is rejected, not +/// silently read as device 0. inline int backend_device_index() { const char * e = std::getenv("VLA_DEVICE"); if (!e || !*e) return 0; @@ -122,8 +120,7 @@ inline Backend backend_init(const char * tag, int n_threads) { // disabling oneDNN outright. Only a default: an explicit setting wins, // for Intel GPUs where the pool is worth keeping. // ggml reads this on the first SYCL entry point, so it must be set here. - // call_once because two concurrent model_load calls would otherwise race - // on the process environment. + // call_once: concurrent model_load would race on the environment. static std::once_flag vmm_once; std::call_once(vmm_once, [] { setenv_default("GGML_SYCL_ENABLE_VMM", "0"); }); diff --git a/src/model.cpp b/src/model.cpp index 01d2907..95bae6c 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -87,9 +87,8 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { return ok; } -// Invariants every arch's Config must satisfy. Checked once here rather than in -// eleven loaders: predict() sizes its host buffers from the max_* dims and loops -// to the real_* dims, so real > max is an out-of-bounds write on the first call. +// predict() sizes buffers from the max_* dims and loops to the real_* dims, so +// real > max writes out of bounds. Checked here once for all archs. bool config_is_sane(const Config& c) { struct { const char* name; int64_t real; int64_t max; } pairs[] = { { "state", c.real_state_dim, c.max_state_dim }, diff --git a/src/model.h b/src/model.h index 87f4bf8..8aff141 100644 --- a/src/model.h +++ b/src/model.h @@ -77,10 +77,8 @@ struct Config { int rope_mode; ///< RoPE variant (NeoX / GPT-J / etc). float rope_freq_base; ///< RoPE base frequency. - /// True if @ref predict already applied the dataset statistics, so its output - /// is in world units. False means the caller must un-normalise (the GR00T - /// family and VLA-JEPA, which expect a host-side affine from a stats JSON). - /// Branch on this rather than on the architecture name. + /// True if @ref predict already applied the dataset statistics. False for the + /// GR00T family and VLA-JEPA, whose callers un-normalise from a stats JSON. bool denormalized = true; }; @@ -126,12 +124,9 @@ struct Inputs { const ImageView* images; ///< Camera views (host memory). int n_images; ///< Number of @ref images. - /// Pre-computed image embeddings; bypasses the vision tower entirely. - /// Layout is [n_img_views * n_img, hidden]. These are handed to the language - /// backbone as-is, so they must already be in whatever scale that arch's LM - /// expects -- which differs per arch: pi0 wants the projector output scaled by - /// 1/sqrt(hidden), pi0.5 wants it unscaled. Supplying tower output from a - /// different arch will not error, it will just be wrong. + /// Pre-computed image embeddings, [n_img_views * n_img, hidden]; bypasses the + /// vision tower. Passed to the LM as-is, so the scale is arch-specific: pi0 + /// expects the projector output times 1/sqrt(hidden), pi0.5 expects it raw. const float* precomputed_img_emb = nullptr; int n_img_views = 0; ///< Number of views in /// @ref precomputed_img_emb. @@ -184,11 +179,8 @@ const Config& model_config(const Model* m); /** * @brief Run one forward pass. * - * Whether the returned actions are in world units or still normalised depends - * on the architecture -- see @ref Config::denormalized. Most archs un-normalise - * internally; the GR00T family and VLA-JEPA return raw values and expect the - * caller to apply the dataset statistics. NaN/Inf inputs cause the call to - * abort. + * See @ref Config::denormalized for whether the result is in world units. + * NaN/Inf inputs cause the call to abort. * * @param m A handle from @ref model_load. * @param in Filled-in @ref Inputs struct. diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 1511d07..4d93555 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -543,9 +543,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, (double) m->lm_rope_base, (long long) m->num_actions_chunk, (long long) m->action_dim, (long long) m->vocab_size, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); - // Not vla::backend_init: bitvla pins its ggml graph to the CPU on purpose and - // offloads the LM through the hand-written ternary CUDA kernels below, so it - // must not pick up whichever accelerator the build compiled in. + // Not backend_init: the ggml graph stays on CPU and the LM offloads through + // the ternary CUDA kernels below. m->backend = ggml_backend_cpu_init(); if (!m->backend) { std::fprintf(stderr, "vla(bitvla): ggml_backend_cpu_init failed\n"); return nullptr; } ggml_backend_cpu_set_n_threads(m->backend, m->n_threads); @@ -662,9 +661,8 @@ std::unique_ptr bitvla_create(const std::string& mmproj_path, if (cudaGetDeviceCount(&dev_count) == cudaSuccess && dev_count > 0) { cudaSetDevice(0); - // Set false by any int2 tensor whose .scale sidecar is missing. The - // ladder kernels dereference the scale pointer unconditionally, so a - // null one is a device-side out-of-bounds read, not a soft failure. + // The ladder kernels dereference the scale pointer unconditionally, so + // a missing sidecar is a device-side OOB read, not a soft failure. bool scales_ok = true; auto load_bit = [&](ggml_tensor * t, int64_t N, int64_t K) -> std::pair { diff --git a/src/models/dual_tower.h b/src/models/dual_tower.h index c30463e..18fe93d 100644 --- a/src/models/dual_tower.h +++ b/src/models/dual_tower.h @@ -12,13 +12,9 @@ // See the License for the specific language governing permissions and // limitations under the License. -// The DINOv2 + SigLIP dual vision tower shared by OpenVLA-OFT and VLA-Adapter. -// Both archs run the identical tower and differ only in what they put behind it -// (Llama-2 + MLPResNet vs Qwen2 + Bridge-Attention), so this used to be the same -// ~50 lines copied into each file. -// -// The DINOv2 branch passes prefix=true (CLS + 4 register tokens, dropped after -// the blocks) and uses LayerScale; the SigLIP branch passes prefix=false. +// DINOv2 + SigLIP dual vision tower, shared by OpenVLA-OFT and VLA-Adapter. +// DINOv2 passes prefix=true (CLS + 4 register tokens, dropped after the blocks) +// and uses LayerScale; SigLIP passes prefix=false. #pragma once @@ -72,8 +68,7 @@ inline ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_ return x; } -// HWC interleaved (u8 or float) to CHW planar, with per-channel mean/std. The -// two branches use different constants: ImageNet for DINOv2, 0.5 for SigLIP. +// HWC to CHW planar with per-channel mean/std: ImageNet for DINOv2, 0.5 for SigLIP. inline void normalize_tower(const ImageView& v, int64_t S, const float mean[3], const float std_[3], std::vector& out){ out.assign((size_t)3*S*S,0.0f); for(int64_t h=0;h Evo1ModelArch::predict(const Inputs& in) { std::vector state_norm(per_a, 0.0f); for (int64_t i = 0; i < per_a; ++i) { const float lo = state_min[i], hi = state_max[i]; - // The converter zero-pads the stats past real_state_dim, so those dims have - // lo == hi == 0 and the affine below would map any input to -1. Leave them - // at 0 instead of feeding the model a spurious -1. + // The converter zero-pads stats past real_state_dim, so lo == hi == 0 there + // and the affine below would map anything to -1. if (hi <= lo) { state_norm[i] = 0.0f; continue; } float xn = 2.0f * (in.state[i] - lo) / (hi - lo + norm_eps_denom) - 1.0f; if (xn < -1.0f) xn = -1.0f; diff --git a/src/models/gguf_reader.h b/src/models/gguf_reader.h index 4ea7610..ebe8944 100644 --- a/src/models/gguf_reader.h +++ b/src/models/gguf_reader.h @@ -60,10 +60,8 @@ struct gguf_reader { bool has(const char * k) const { return gguf_find_key(gctx, k) >= 0; } - // The gguf_get_val_* helpers assert on a type mismatch, which aborts the - // process on a malformed file. Check the declared type first and fall back to - // the caller's default instead. (model.cpp's arch probe already did this; the - // reader did not.) + // gguf_get_val_* asserts on a type mismatch, killing the process on a bad + // file. Check the declared type first. bool typed_key(const char * k, gguf_type want, int64_t * id_out) const { const int64_t id = gguf_find_key(gctx, k); if (id < 0) return false; @@ -89,10 +87,9 @@ struct gguf_reader { return (src && ggml_is_quantized(src->type)) ? src->type : prefer; } - // cap is the destination capacity in bytes and must equal the declared tensor - // size. Without it a tensor with the expected ne[0] but an extra dimension - // writes past a caller's vector. On failure buf may be partially written, so - // callers must not keep it. + // cap must equal the declared tensor size: a tensor with the expected ne[0] but + // an extra dimension would write past the caller's vector. On failure buf may be + // partially written. bool read_raw(const char * name, void * buf, size_t cap) { const int64_t id = gguf_find_tensor(gctx, name); if (id < 0) { std::fprintf(stderr, "vla(%s): missing tensor %s\n", arch, name); return false; } @@ -124,9 +121,8 @@ struct gguf_reader { std::vector read_convert(const char * name, ggml_type target, bool gemma_norm = false) { if (target != GGML_TYPE_F32 && target != GGML_TYPE_BF16) { if (gemma_norm) { - // The +1 can only be applied to unpacked floats. Silently skipping - // it would give a quantized Gemma checkpoint wrong norm weights and - // no diagnostic, so refuse instead. + // The +1 needs unpacked floats; skipping it would silently give wrong + // norm weights. std::fprintf(stderr, "vla(%s): %s needs the Gemma norm +1 but the target type is packed\n", arch, name); return {}; diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 3b3a650..89002b7 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -145,9 +145,8 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path U("openvla_oft.action.chunk",m->chunk); U("openvla_oft.action.action_dim",m->action_dim); U("openvla_oft.action.proprio_dim",m->proprio_dim); U("openvla_oft.action.head_hidden",m->head_hidden); U("openvla_oft.action.head_blocks",m->head_blocks); F("openvla_oft.action.head_ln_eps",m->head_ln_eps); - // No empty_id: the reference zeroes the action-slot embeddings rather than - // inserting an empty token (modeling_prismatic.py:891), which is what the - // zero-filled act0 below does. + // No empty_id: the reference zeroes the action-slot embeddings instead + // (modeling_prismatic.py:891), which is what act0 below does. U("openvla_oft.tokens.stop_id",m->stop_id); if (m->lm_head_dim==0) m->lm_head_dim = m->lm_hidden / m->n_q; @@ -338,10 +337,8 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*kr=ggml_rope_ext(C,kh,t_pos,nullptr,(int)lm_head_dim,GGML_ROPE_TYPE_NEOX,0,lm_rope_base,1.0f,0.0f,1.0f,32.0f,1.0f); ggml_tensor*Q=ggml_cont(C,ggml_permute(C,qr,0,2,1,3)),*K=ggml_cont(C,ggml_permute(C,kr,0,2,1,3)),*V=ggml_cont(C,ggml_permute(C,vh,1,2,0,3)); ggml_tensor*kq=ggml_mul_mat(C,K,Q); ggml_mul_mat_set_prec(kq,GGML_PREC_F32); - // No causal mask on purpose. OpenVLA-OFT ships a patched transformers that - // replaces the lower-triangular mask across the whole sequence - // (modeling_llama.py:719-723), so the backbone is fully bidirectional here - // even though the sibling vla_adapter builds a causal mask. + // Unmasked on purpose: OpenVLA-OFT patches transformers to replace the + // causal mask across the whole sequence (modeling_llama.py:719-723). ggml_tensor*aw=ggml_soft_max_ext(C,kq,nullptr,lsc,0.0f); ggml_tensor*kqv=ggml_mul_mat(C,V,aw); ggml_tensor*att=ggml_reshape_2d(C,ggml_cont(C,ggml_permute(C,kqv,0,2,1,3)),HC,SEQ); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 718b038..61b3bf6 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -283,8 +283,7 @@ bool load_stats(gguf_reader & g, Pi0ModelArch & m) { if (t->ne[0] != (int64_t) dst.size()) { std::printf("vla(pi0): %s dim mismatch - identity\n", name); return; } const std::vector identity = dst; if (!g.read_raw(name, dst.data(), dst.size() * sizeof(float))) { - // A short read leaves dst half-overwritten; restore so "identity" - // means identity. + // A short read leaves dst half-overwritten. dst = identity; std::printf("vla(pi0): %s read failed - identity\n", name); } diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 2c83675..fb96239 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -355,8 +355,7 @@ bool load_stats(gguf_reader & g, Pi05ModelArch & m) { if (t->ne[0] != (int64_t) dst.size()) { std::printf("vla(pi05): %s dim mismatch - identity\n", name); return; } const std::vector identity = dst; if (!g.read_raw(name, dst.data(), dst.size() * sizeof(float))) { - // A short read leaves dst half-overwritten; restore so "identity" - // means identity. + // A short read leaves dst half-overwritten. dst = identity; std::printf("vla(pi05): %s read failed - identity\n", name); } @@ -633,11 +632,9 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { stats.ms_vision = std::chrono::duration(clk::now() - tv0).count(); ggml_gallocr_free(vga); ggml_free(VC); - // pi05's image tokens are the raw PaliGemma projector features: this undoes - // the 1/sqrt(hidden) the shared vision graph applies (pi0 keeps them - // scaled). Deliberately inside this branch: precomputed_img_emb replaces - // the tower, so a caller supplies LM-ready features and must not be - // rescaled here. See the Inputs::precomputed_img_emb contract in model.h. + // Undo the 1/sqrt(hidden) the shared vision graph applies; pi05 wants raw + // projector features. Inside this branch on purpose: precomputed_img_emb + // replaces the tower and is already LM-ready. const float img_scale = (float) std::sqrt((double) hidden_pl); for (float & x : img_emb_host) x *= img_scale; } diff --git a/src/models/vision_common.h b/src/models/vision_common.h index 092f827..9e33f7b 100644 --- a/src/models/vision_common.h +++ b/src/models/vision_common.h @@ -50,11 +50,9 @@ inline void pixel_shuffle_hf(const float * src, float * dst, } } -// HWC interleaved (u8 or float) to CHW planar, scaled to [-1, 1] - the SigLIP -// convention shared by SmolVLA, pi0, pi0.5 and GR00T N1.5. There is no resize: -// the caller must hand over an exactly side x side view, which is why the tag is -// passed in (the error message is the only thing that differed between the four -// copies this replaces). +// HWC to CHW planar in [-1, 1], the SigLIP convention used by SmolVLA, pi0, pi0.5 +// and GR00T N1.5. No resize: the view must already be side x side. arch only +// labels the error. inline bool preprocess_image_chw(const char * arch, const ImageView & v, int64_t side, std::vector & out) { if (v.w != (int) side || v.h != (int) side || !v.data) { diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 07e545a..20c5fb7 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -120,12 +120,9 @@ struct VlaAdapterModelArch : public ModelArchBase { namespace { -// Interleaved (GPT-J style) rotation: out[2k] = -x[2k+1], out[2k+1] = x[2k]. -// Note this is paired with a half-split frequency table in fill_cs (j = mi % half), -// so the two halves of a rotation pair get different angles. That mismatch is in -// the VLA-Adapter reference too (prismatic/models/action_heads.py builds -// cat([freqs, freqs]) at :163 but rotates x[..., ::2]/x[..., 1::2] at :137-140), -// and the checkpoint weights were trained against it. Do not "fix" one side. +// Interleaved rotation, paired with the half-split frequency table in fill_cs, so +// a rotation pair gets two different angles. The reference does the same +// (action_heads.py:163 vs :137-140) and the weights were trained on it. Leave it. static ggml_tensor* hrot(ggml_context*C, ggml_tensor*x, int64_t HD){ int64_t L=x->ne[1],H=x->ne[2]; ggml_tensor*xp=ggml_reshape_4d(C,x,2,HD/2,L,H); diff --git a/src/serving/server.cpp b/src/serving/server.cpp index 410fa14..25c3981 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -57,9 +57,8 @@ bool decode_image(const vla::Image & img, data.size()); return false; } - // Read the header first: stbi_load allocates 3*w*h before returning, so a - // small JPEG declaring huge dimensions would allocate gigabytes before any - // check on the decoded size could reject it. + // Header first: stbi_load allocates 3*w*h before returning, so a small JPEG + // declaring huge dimensions would allocate gigabytes before any check. if (!stbi_info_from_memory( reinterpret_cast(data.data()), static_cast(data.size()), &w, &h, &ch)) { @@ -125,8 +124,8 @@ bool decode_image(const vla::Image & img, f32.resize(pixels); std::memcpy(f32.data(), img.data().data(), expected); - // The other float inputs are swept for NaN/Inf; pixels were not, so a - // non-finite pixel used to propagate all the way out as a robot action. + // State and noise are swept for NaN/Inf; pixels were not, so a bad pixel + // came back out as a robot action. for (size_t i = 0; i < pixels; ++i) { if (!std::isfinite(f32[i])) { std::fprintf(stderr, "vla-server: F32_RGB_01 pixel %zu is not finite\n", i); @@ -150,9 +149,8 @@ std::string make_error_response(uint64_t request_id, const std::string & msg) { return resp.SerializeAsString(); } -// Discard any frames after the first. Returns true if there were some, meaning -// the request was malformed. Must run to completion: leaving a frame queued -// keeps REP in receive state and the next send throws EFSM. +// Discard frames after the first; true means the request was malformed. Must run +// to completion: a queued frame keeps REP in receive state and send throws EFSM. bool drain_extra_frames(zmq::socket_t & sock) { bool extra = false; while (sock.get(zmq::sockopt::rcvmore)) { @@ -266,10 +264,8 @@ int main(int argc, char ** argv) { zmq::context_t zctx( 1); zmq::socket_t sock(zctx, zmq::socket_type::rep); sock.set(zmq::sockopt::linger, 0); - // Cap inbound messages so one oversized request cannot exhaust memory. 64 MiB - // is far above any real request (16 views of 512x512 F32 RGB is ~50 MiB) and - // low enough that protobuf's several-fold expansion during ParseFromArray - // stays bounded. + // 64 MiB is above any real request (16 views of 512x512 F32 RGB is ~50 MiB) and + // low enough to bound protobuf's expansion during ParseFromArray. sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); sock.bind(bind_addr); std::printf("vla-server: bound to %s. ready.\n", bind_addr.c_str()); @@ -324,10 +320,8 @@ int main(int argc, char ** argv) { continue; } - // REP will not let us reply until the whole multipart message is received, - // and send_reply treats a failed send as fatal - so an unauthenticated - // client could shut the server down with one two-frame request. The - // protocol is single-frame: drain any extras and reject. + // Without this an unauthenticated client shuts the server down with one + // two-frame request: the reply fails and send_reply sets g_shutdown. if (drain_extra_frames(sock)) { send_reply(make_error_response(0, "expected a single-frame request")); continue; diff --git a/src/serving/vlm-server.cpp b/src/serving/vlm-server.cpp index 4e59e08..a7880d5 100644 --- a/src/serving/vlm-server.cpp +++ b/src/serving/vlm-server.cpp @@ -15,8 +15,7 @@ #include "vlm/engine.h" #include "serving/vlm.pb.h" -// Only for stbi_info_from_memory: the engine decodes through mtmd, which has no -// dimension guard, so we preflight the JPEG header here at the trust boundary. +// stbi_info_from_memory only, to preflight JPEG dimensions before mtmd decodes. #define STB_IMAGE_IMPLEMENTATION #define STB_IMAGE_STATIC #pragma GCC diagnostic push @@ -111,8 +110,7 @@ int main(int argc, char ** argv) { zmq::context_t zctx( 1); zmq::socket_t sock(zctx, zmq::socket_type::router); sock.set(zmq::sockopt::linger, 0); - // Cap inbound messages so one oversized request cannot exhaust memory. Per - // frame only - see the envelope caps in the recv loop for the multipart total. + // Per frame only; the recv loop caps the multipart total. sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); sock.bind(bind_addr); std::printf("vlm-server: bound to %s. ready.\n", bind_addr.c_str()); @@ -143,9 +141,8 @@ int main(int argc, char ** argv) { } if (!(poll[0].revents & ZMQ_POLLIN)) continue; - // A ROUTER envelope is an identity frame plus an optional empty delimiter. - // maxmsgsize bounds each frame but not how many, so without these caps a - // peer could stream sub-limit frames until the process runs out of memory. + // maxmsgsize bounds each frame but not how many, so a peer could stream + // sub-limit frames until memory runs out. constexpr size_t kMaxEnvFrames = 8; constexpr size_t kMaxEnvBytes = 64 * 1024; @@ -166,8 +163,7 @@ int main(int argc, char ** argv) { if (sock.get(zmq::sockopt::rcvmore)) { env_bytes += part.size(); if (env.size() >= kMaxEnvFrames || env_bytes > kMaxEnvBytes) { - // Keep draining so the socket stays in a sane state, but stop - // accumulating and drop the request. + // Keep draining so the socket stays sane, but stop accumulating. env_overflow = true; } else { env.emplace_back(static_cast(part.data()), part.size()); @@ -210,8 +206,7 @@ int main(int argc, char ** argv) { send_reply(make_error_stream(rid, "ChatRequest has no messages")); continue; } - // Bound the work a single request can buy: without these, one 60 MiB - // payload of millions of tiny messages costs template formatting and + // One 60 MiB payload of tiny messages would cost template formatting and // tokenization far beyond anything n_ctx could consume. constexpr int kMaxMessages = 512; constexpr size_t kMaxTextBytes = 4u * 1024 * 1024; @@ -238,8 +233,7 @@ int main(int argc, char ** argv) { vlm::Image out; if (im.encoding() == vlm_chat::Image::JPEG) { const auto & d = im.data(); - // Header first: the decoder allocates from the declared dimensions, - // so a tiny JPEG claiming 30000x30000 would allocate gigabytes. + // Header first: mtmd decodes with no dimension guard. int jw = 0, jh = 0, jc = 0; if (d.size() > size_t(INT_MAX) || !stbi_info_from_memory(reinterpret_cast(d.data()), diff --git a/tests/test_config_guard.cpp b/tests/test_config_guard.cpp index ac70617..32a91ca 100644 --- a/tests/test_config_guard.cpp +++ b/tests/test_config_guard.cpp @@ -12,9 +12,8 @@ // See the License for the specific language governing permissions and // limitations under the License. -// The Config invariant model_load enforces for every arch: predict() sizes host -// buffers from the max_* dims and loops to the real_* dims, so real > max is an -// out-of-bounds write on the first call. SmolVLA shipped that loop unbounded. +// predict() sizes host buffers from the max_* dims and loops to the real_* dims, +// so real > max writes out of bounds. SmolVLA shipped that loop unbounded. #include "model.h" @@ -22,8 +21,7 @@ #include #include -// Mirrors config_is_sane() in src/model.cpp. Kept in sync by this test failing -// if the rule there is relaxed. +// Mirrors config_is_sane() in src/model.cpp. static bool sane(const vla::Config & c) { if (c.real_state_dim < 0 || c.max_state_dim < 0) return false; if (c.real_action_dim < 0 || c.max_action_dim < 0) return false; diff --git a/tests/test_dit_common.cpp b/tests/test_dit_common.cpp index 43ca09a..6387f37 100644 --- a/tests/test_dit_common.cpp +++ b/tests/test_dit_common.cpp @@ -13,11 +13,9 @@ // limitations under the License. // Pins the two DiT time embeddings shared by GR00T N1.5/N1.6/N1.7 and VLA-JEPA. -// -// These are pure host functions, which is why they get a real test: three of the -// four archs that use them have no checkpoint small enough to gate in CI, so the -// end-to-end sha oracle cannot cover them. The trig order is the trap - the two -// functions use OPPOSITE conventions and both match the reference. +// Pure host functions, so they can be tested without a checkpoint, which the +// end-to-end sha oracle cannot do for three of the four archs. The trig orders +// are opposite and both match the reference. #include "models/dit_common.h" @@ -74,7 +72,7 @@ int main() { assert(close(out[tk * dim + i], out[i])); } - // --- The orders really are opposite; a "consistency cleanup" must fail. --- + // --- The orders are opposite; a consistency cleanup must fail here. --- { std::vector tp, as; vla::timesteps_proj(7, tp); diff --git a/tests/test_rope_conventions.cpp b/tests/test_rope_conventions.cpp index 349f668..8d6ab02 100644 --- a/tests/test_rope_conventions.cpp +++ b/tests/test_rope_conventions.cpp @@ -12,16 +12,12 @@ // See the License for the specific language governing permissions and // limitations under the License. -// Pins the two rotary conventions in the tree so a "cleanup" cannot silently -// swap one for the other. +// Pins the two rotary conventions so a cleanup cannot swap one for the other. // -// The interesting case is VLA-Adapter's action head, which pairs an INTERLEAVED -// rotation with a HALF-SPLIT frequency table. Read on its own that looks like a -// bug - the two elements of a rotation pair get different angles. It is faithful: -// the OpenHelix reference has the same mismatch (action_heads.py builds -// cat([freqs, freqs]) at :163 and rotates x[..., ::2]/x[..., 1::2] at :137-140), -// so the checkpoint weights were trained against it. Making it self-consistent -// would break the shipped checkpoints, and this test fails if someone tries. +// VLA-Adapter's action head pairs an interleaved rotation with a half-split +// frequency table, so a rotation pair gets two angles. The reference does the same +// (action_heads.py:163 vs :137-140) and the weights were trained on it, so making +// it self-consistent would break the shipped checkpoints. #undef NDEBUG // keep assert() live even in Release builds #include @@ -68,7 +64,7 @@ int main() { const std::vector x = { 1, 2, 3, 4, 5, 6, 7, 8 }; - // 1. The two rotations are genuinely different, so a swap is observable. + // 1. The two rotations differ, so a swap is observable. const std::vector ri = rotate_interleaved(x); const std::vector rh = rotate_half(x); assert(ri != rh); @@ -81,8 +77,8 @@ int main() { assert(adapter_angle(i, hd, base, t) == adapter_angle(i + hd / 2, hd, base, t)); } - // 3. The mismatch itself: an interleaved pair (2k, 2k+1) does NOT share an - // angle under this table. Pinned deliberately - see the header comment. + // 3. The mismatch: an interleaved pair (2k, 2k+1) does not share an angle + // under this table. Pinned on purpose, see the header. bool any_pair_differs = false; for (size_t k = 0; k < hd / 2; ++k) { if (adapter_angle(2 * k, hd, base, t) != adapter_angle(2 * k + 1, hd, base, t)) { @@ -91,8 +87,8 @@ int main() { } assert(any_pair_differs); - // 4. For contrast: an interleaved table (j = mi / 2) would make every pair - // agree. This is the "obvious fix" that must not be applied. + // 4. An interleaved table (j = mi / 2) would make every pair agree. That is + // the fix that must not be applied. for (size_t k = 0; k < hd / 2; ++k) { const auto interleaved_angle = [&](size_t mi) { return t * (1.0 / std::pow(base, (2.0 * (double) (mi / 2)) / (double) hd)); From f42cf5d19c0453c5f144779010f1bd15ff0a653f Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:25:07 +0700 Subject: [PATCH 13/42] keep the gguf reader open across predict calls --- src/models/evo1.cpp | 10 +++++----- src/models/gr00tn1d5.cpp | 10 +++++----- src/models/gr00tn1d6.cpp | 10 +++++----- src/models/pi0.cpp | 10 +++++----- src/models/pi05.cpp | 10 +++++----- 5 files changed, 25 insertions(+), 25 deletions(-) diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 732d049..93d68d0 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -54,6 +54,8 @@ struct Evo1ModelArch : public ModelArchBase { ~Evo1ModelArch() override; std::string gguf_path; + // Opened once at load: reopening per predict re-parses the whole GGUF header. + gguf_reader io{"evo1"}; ggml_backend_t backend = nullptr; bool is_cuda = false; bool is_gpu = false; @@ -270,8 +272,8 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, m->gguf_path = ckpt_path; m->matmul_type = std::getenv("VLA_EVO1_F32_WEIGHTS") ? GGML_TYPE_F32 : GGML_TYPE_BF16; - gguf_reader g("evo1"); - if (!g.open(ckpt_path)) return nullptr; + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("evo1.architecture")) { std::fprintf(stderr, "vla(evo1): %s is not an evo1 GGUF (no evo1.architecture KV)\n", ckpt_path.c_str()); return nullptr; } @@ -488,10 +490,8 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { input_ids.resize(max_text_length, pad_id); const int64_t SEQ = max_text_length; - gguf_reader g("evo1"); - if (!g.open(gguf_path)) return {}; std::vector inputs_embeds((size_t) SEQ * lm_hidden); - if (!g.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), lm_hidden)) return {}; + if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), lm_hidden)) return {}; { int64_t img_idx = 0; for (int64_t p = 0; p < SEQ; ++p) { diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 3e60cdd..8d82652 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -52,6 +52,8 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { ~Gr00tN1d5ModelArch() override; std::string gguf_path; + // Opened once at load: reopening per predict re-parses the whole GGUF header. + gguf_reader io{"gr00tn1d5"}; ggml_backend_t backend = nullptr; bool is_cuda = false; bool is_gpu = false; @@ -249,8 +251,8 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, m->gguf_path = ckpt_path; m->matmul_type = std::getenv("VLA_GR00T_BF16_WEIGHTS") ? GGML_TYPE_BF16 : GGML_TYPE_F32; - gguf_reader g("gr00tn1d5"); - if (!g.open(ckpt_path)) return nullptr; + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("gr00t_n1_5.architecture")) { std::fprintf(stderr, "vla(gr00tn1d5): %s is not a gr00t_n1_5 GGUF\n", ckpt_path.c_str()); return nullptr; } if (!load_config(g, *m, m->cfg)) return nullptr; std::printf("vla(gr00tn1d5): vit=%lldd×%lldL×%lldh n_img_tok=%lld lm=Qwen3 %lldd×%lldL (%lldq/%lldkv×%lld) " @@ -426,10 +428,8 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const int64_t SEQ = (int64_t) input_ids.size(); if (SEQ > max_seq_len) { std::fprintf(stderr, "vla(gr00tn1d5): prompt too long (%lld > %lld)\n", (long long) SEQ, (long long) max_seq_len); return {}; } - gguf_reader g("gr00tn1d5"); - if (!g.open(gguf_path)) return {}; std::vector inputs_embeds((size_t) SEQ * H); - if (!g.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; + if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; { int64_t k = 0; for (int64_t p = 0; p < SEQ; ++p) if (input_ids[p] == (int32_t) image_token_index) { if (k >= n_img) { std::fprintf(stderr, "vla(gr00tn1d5): more tokens than ViT embeds\n"); return {}; } diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 8ffd9e9..0285c27 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -50,6 +50,8 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { ~Gr00tN1d6ModelArch() override; std::string gguf_path; + // Opened once at load: reopening per predict re-parses the whole GGUF header. + gguf_reader io{"gr00tn1d6"}; ggml_backend_t backend = nullptr; bool is_cuda = false; bool is_gpu = false; @@ -268,8 +270,8 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, m->gguf_path = ckpt_path; m->matmul_type = std::getenv("VLA_GR00T_BF16_WEIGHTS") ? GGML_TYPE_BF16 : GGML_TYPE_F32; - gguf_reader g("gr00tn1d6"); - if (!g.open(ckpt_path)) return nullptr; + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("gr00t_n1_6.architecture")) { std::fprintf(stderr, "vla(gr00tn1d6): %s is not a gr00t_n1_6 GGUF\n", ckpt_path.c_str()); return nullptr; } if (!load_config(g, *m, m->cfg)) return nullptr; std::printf("vla(gr00tn1d6): vit=%lldd×%lldL×%lldh (Linear patch embed) pixel_shuffle÷%lld ⇒ n_img_tok=%lld mlp1=LN(%lld)→Linear→GELU→Linear " @@ -468,10 +470,8 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const int64_t SEQ = (int64_t) input_ids.size(); if (SEQ > max_seq_len) { std::fprintf(stderr, "vla(gr00tn1d6): prompt too long (%lld > %lld)\n", (long long) SEQ, (long long) max_seq_len); return {}; } - gguf_reader g("gr00tn1d6"); - if (!g.open(gguf_path)) return {}; std::vector inputs_embeds((size_t) SEQ * H); - if (!g.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; + if (!io.fetch_rows_f32("token_embd.weight", input_ids, inputs_embeds.data(), H)) return {}; { int64_t k = 0; for (int64_t p = 0; p < SEQ; ++p) if (input_ids[p] == (int32_t) image_token_index) { if (k >= n_img) { std::fprintf(stderr, "vla(gr00tn1d6): more tokens than ViT embeds\n"); return {}; } diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 61b3bf6..5d9b172 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -96,6 +96,8 @@ struct Pi0ModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; std::string ckpt_path_; + // Opened once at load: reopening per predict re-parses the whole GGUF header. + gguf_reader io{"pi0"}; ggml_type matmul_type = GGML_TYPE_BF16; // In-tree SigLIP-So400m/14 vision tower (was llama.cpp clip.cpp mmproj). @@ -320,8 +322,8 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, m->ckpt_path_ = ckpt_path; m->matmul_type = std::getenv("VLA_PI0_F32_WEIGHTS") ? GGML_TYPE_F32 : GGML_TYPE_BF16; - gguf_reader g("pi0"); - if (!g.open(ckpt_path)) return nullptr; + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("pi0.architecture") || g.str("pi0.architecture") != "pi0") { std::fprintf(stderr, "vla(pi0): '%s' is not a π₀ GGUF (pi0.architecture missing/wrong)\n", ckpt_path.c_str()); @@ -548,9 +550,7 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { std::vector lang_ids(in.lang_tokens, in.lang_tokens + n_lang); std::vector lang_rows((size_t) n_lang * hidden_pl); { - gguf_reader g("pi0"); - if (!g.open(ckpt_path_)) return {}; - if (!g.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; + if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } ggml_init_params cp = { (size_t) 64 * 1024 * 1024, nullptr, true }; diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index fb96239..44f346b 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -111,6 +111,8 @@ struct Pi05ModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; std::string ckpt_path_; + // Opened once at load: reopening per predict re-parses the whole GGUF header. + gguf_reader io{"pi05"}; ggml_type matmul_type = GGML_TYPE_BF16; int64_t adarms_cond_dim = 0; @@ -398,8 +400,8 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, m->ckpt_path_ = ckpt_path; m->matmul_type = std::getenv("VLA_PI05_F32_WEIGHTS") ? GGML_TYPE_F32 : GGML_TYPE_BF16; - gguf_reader g("pi05"); - if (!g.open(ckpt_path)) return nullptr; + if (!m->io.open(ckpt_path)) return nullptr; + gguf_reader & g = m->io; if (!g.has("pi05.architecture") || g.str("pi05.architecture") != "pi05") { std::fprintf(stderr, "vla(pi05): '%s' is not a π0.5 GGUF (pi05.architecture missing/wrong)\n", ckpt_path.c_str()); @@ -649,9 +651,7 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { std::vector lang_ids(in.lang_tokens, in.lang_tokens + n_lang); std::vector lang_rows((size_t) n_lang * hidden_pl); { - gguf_reader g("pi05"); - if (!g.open(ckpt_path_)) return {}; - if (!g.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; + if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } ggml_init_params cp = { (size_t) 64 * 1024 * 1024, nullptr, true }; From 06a149336fd3dae2159d328074db52d47e754acf Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:29:58 +0700 Subject: [PATCH 14/42] raise cpu thread cap to 16, add VLA_N_THREADS --- src/arch.h | 15 ++++++++++++--- src/models/pi0.cpp | 7 ++----- src/models/pi05.cpp | 7 ++----- 3 files changed, 16 insertions(+), 13 deletions(-) diff --git a/src/arch.h b/src/arch.h index 633e8ff..e66fa33 100644 --- a/src/arch.h +++ b/src/arch.h @@ -28,17 +28,26 @@ #include "model.h" #include +#include +#include #include #include #include namespace vla { -// Default CPU thread count for the in-tree loaders: all cores, capped at 8, -// with a safe fallback when the hardware count is unknown. +// CPU threads for the in-tree loaders; VLA_N_THREADS overrides. Cap measured on +// a 24-core host: 8 to 16 is 38-41% faster on vla_adapter, evo1 and gr00tn1d5, +// and 24 is slower than 16. Output is bit-identical either way. inline int default_cpu_threads() { + if (const char * e = std::getenv("VLA_N_THREADS")) { + char * end = nullptr; + const long n = std::strtol(e, &end, 10); + if (*end == '\0' && n > 0 && n <= 1024) return (int) n; + std::fprintf(stderr, "vla: ignoring VLA_N_THREADS='%s'\n", e); + } const unsigned hw = std::thread::hardware_concurrency(); - return hw == 0 ? 4 : (int) std::min(hw, 8u); + return hw == 0 ? 4 : (int) std::min(hw, 16u); } /** diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 5d9b172..4710b10 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -123,7 +123,7 @@ struct Pi0ModelArch : public ModelArchBase { std::vector state_mean, state_std, action_mean, action_std; std::mt19937 rng{std::random_device{}()}; - int n_threads = 4; + int n_threads = default_cpu_threads(); }; namespace { @@ -340,10 +340,7 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, cfg.num_steps, (long long) cfg.real_state_dim, (long long) cfg.real_action_dim, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); - { - const unsigned hw = std::thread::hardware_concurrency(); - m->n_threads = (hw == 0) ? 4 : (int) std::min(hw, 8u); - } + m->n_threads = default_cpu_threads(); { const Backend b = backend_init("vla(pi0)", m->n_threads); if (!b.handle) { return nullptr; } diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 44f346b..263d043 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -141,7 +141,7 @@ struct Pi05ModelArch : public ModelArchBase { bool quantile_norm = false; std::mt19937 rng{std::random_device{}()}; - int n_threads = 4; + int n_threads = default_cpu_threads(); }; namespace { @@ -421,10 +421,7 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, (long long) cfg.n_lang, (long long) m->adarms_cond_dim, m->matmul_type == GGML_TYPE_F32 ? "F32" : "BF16"); - { - const unsigned hw = std::thread::hardware_concurrency(); - m->n_threads = (hw == 0) ? 4 : (int) std::min(hw, 8u); - } + m->n_threads = default_cpu_threads(); { const Backend b = backend_init("vla(pi05)", m->n_threads); if (!b.handle) { return nullptr; } From 06f5cb1f8e765d09c01da86778f4cb78bc35bab0 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 15:30:15 +0700 Subject: [PATCH 15/42] document VLA_N_THREADS and VLA_DEVICE --- README.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/README.md b/README.md index 3811c5b..587b7c1 100644 --- a/README.md +++ b/README.md @@ -160,6 +160,11 @@ vla-server: bound to tcp://*:5555. ready. Use `--bind` to change the address and port. Stop the server with `Ctrl-C`. +Two environment knobs apply to every arch: + +- `VLA_N_THREADS` - CPU backend thread count, default core count capped at 16. +- `VLA_DEVICE` - GPU ordinal for CUDA and SYCL builds, default 0. + --- ## Running the client From 87a16a1fbcbe792db11db30060e75fcff4738a71 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 16:25:08 +0700 Subject: [PATCH 16/42] add stable C ABI and libvla --- CMakeLists.txt | 16 +++- include/vla.h | 167 +++++++++++++++++++++++++++++++++++++++++ src/vla_c_api.cpp | 173 +++++++++++++++++++++++++++++++++++++++++++ tests/CMakeLists.txt | 6 ++ tests/test_c_api.c | 113 ++++++++++++++++++++++++++++ 5 files changed, 474 insertions(+), 1 deletion(-) create mode 100644 include/vla.h create mode 100644 src/vla_c_api.cpp create mode 100644 tests/test_c_api.c diff --git a/CMakeLists.txt b/CMakeLists.txt index bfd43f6..f7dbd4f 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -210,6 +210,20 @@ target_link_libraries(vlm-server PRIVATE PkgConfig::ZeroMQ ) +# Stable C ABI. Shared so bindings can dlopen it; visibility hidden so only the +# vla_* symbols are exported and llama/ggml stay internal. +add_library(vla SHARED src/vla_c_api.cpp) +target_include_directories(vla + PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include + PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/src +) +target_link_libraries(vla PRIVATE vla_core) +set_target_properties(vla PROPERTIES + CXX_VISIBILITY_PRESET hidden + VISIBILITY_INLINES_HIDDEN ON + PUBLIC_HEADER ${CMAKE_CURRENT_SOURCE_DIR}/include/vla.h +) + # One-shot inference CLI: image + tokens -> action, no server or simulator. add_executable(vla-cli src/serving/vla-cli.cpp @@ -220,7 +234,7 @@ target_include_directories(vla-cli PRIVATE target_link_libraries(vla-cli PRIVATE vla_core) # --- First-party build hygiene (never applied to the vendored llama.cpp subtree) -- -set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla-server vlm-server vla-cli) +set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla vla-server vlm-server vla-cli) foreach(tgt IN LISTS VLA_FIRST_PARTY_TARGETS) # Warn on our own C++ only; nvcc device code keeps its own diagnostics. diff --git a/include/vla.h b/include/vla.h new file mode 100644 index 0000000..f278f46 --- /dev/null +++ b/include/vla.h @@ -0,0 +1,167 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Stable C ABI over src/model.h, so bindings do not have to match a C++ compiler +// or standard library. +// +// Ownership: vla_model_load pairs with vla_model_free, vla_predict pairs with +// vla_free_actions, and every pointer in vla_inputs is borrowed for the call. + +#ifndef VLA_H +#define VLA_H + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +#if defined(_WIN32) +# define VLA_API __declspec(dllexport) +#else +# define VLA_API __attribute__((visibility("default"))) +#endif + +// Bumped on any incompatible change to the structs or functions below. +#define VLA_ABI_VERSION 1 + +typedef struct vla_model vla_model; + +typedef enum { + VLA_OK = 0, + VLA_ERR_ARG = -1, ///< Null handle or malformed vla_inputs. + VLA_ERR_PREDICT = -2, ///< The engine returned no actions. + VLA_ERR_EXCEPTION = -3, ///< A C++ exception was caught at the boundary. +} vla_status; + +typedef enum { + VLA_PIXEL_U8 = 0, ///< 8-bit interleaved RGB. + VLA_PIXEL_F32_RGB_01 = 1, ///< Float32 interleaved RGB in [0, 1]. +} vla_pixel_format; + +typedef enum { + VLA_TIMING_NONE = 0, + VLA_TIMING_PHASE = 1, +} vla_timing_detail; + +/// Mirrors vla::Config. See src/model.h for the per-field meaning. +typedef struct { + int64_t n_img; + int64_t n_lang; + int64_t n_state; + int64_t n_prefix; + int64_t n_suffix; + int64_t n_full; + + int64_t hidden; + int64_t expert_h; + int64_t intermediate; + int64_t expert_inter; + int64_t n_q_heads; + int64_t n_kv_heads; + int64_t head_dim; + int64_t q_full_dim; + int64_t kv_full_dim; + int64_t n_layers; + int32_t self_attn_every_n; + + int64_t max_state_dim; + int64_t max_action_dim; + int64_t real_state_dim; + int64_t real_action_dim; + float norm_eps; + double min_period; + double max_period; + int32_t num_steps; + + float rms_eps; + int32_t rope_n_dims; + int32_t rope_mode; + float rope_freq_base; + + /// Non-zero if vla_predict already applied the dataset statistics. Zero for + /// the GR00T family and VLA-JEPA, whose callers un-normalise themselves. + int32_t denormalized; +} vla_config; + +typedef struct { + const void * data; ///< First pixel, caller-owned. + int32_t w; + int32_t h; + int32_t format; ///< A vla_pixel_format value. +} vla_image; + +typedef struct { + const vla_image * images; + int32_t n_images; + + /// Optional, replaces the vision tower. Layout [n_img_views * n_img, hidden], + /// already in the scale that arch's LM expects. Pass NULL to use images. + const float * precomputed_img_emb; + int32_t n_img_views; + + const int32_t * lang_tokens; + int32_t n_lang; + + /// Length max_state_dim; pad real_state_dim..max with zeros. + const float * state; + /// Length n_suffix * max_action_dim, or NULL to sample internally. + const float * noise; + + /// Only Evo-1 reads this; other archs derive their own mask. + const int32_t * attention_mask; + int32_t attention_mask_n; + + int32_t timing_detail; ///< A vla_timing_detail value. +} vla_inputs; + +/// Milliseconds. Phase fields are zero unless timing_detail was VLA_TIMING_PHASE. +typedef struct { + float ms_total; + float ms_vision; + float ms_inference; + float ms_prefill; + float ms_denoise; +} vla_stats; + +/// VLA_ABI_VERSION of the loaded library, for a runtime compatibility check. +VLA_API int32_t vla_abi_version(void); + +/// mmproj_path may be NULL or "" for archs that bake vision into the checkpoint. +/// config_path may be NULL. Returns NULL on failure. +VLA_API vla_model * vla_model_load(const char * mmproj_path, + const char * ckpt_path, + const char * config_path); + +VLA_API void vla_model_free(vla_model * m); + +/// Fills out with the resolved config. Returns a vla_status. +VLA_API int32_t vla_model_config(const vla_model * m, vla_config * out); + +/// Runs one forward pass. On VLA_OK, *out_actions points to *out_n floats that +/// the caller releases with vla_free_actions. Row-major [num_steps or n_suffix, +/// max_action_dim]; only the first real_action_dim columns carry values. +VLA_API int32_t vla_predict(vla_model * m, const vla_inputs * in, + float ** out_actions, int64_t * out_n); + +VLA_API void vla_free_actions(float * actions); + +/// Timings of the most recent vla_predict on this handle. +VLA_API int32_t vla_last_stats(const vla_model * m, vla_stats * out); + +#ifdef __cplusplus +} +#endif + +#endif // VLA_H diff --git a/src/vla_c_api.cpp b/src/vla_c_api.cpp new file mode 100644 index 0000000..03b5f9f --- /dev/null +++ b/src/vla_c_api.cpp @@ -0,0 +1,173 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Type conversion at the C boundary, and a barrier for C++ exceptions. + +#include "vla.h" + +#include "model.h" + +#include +#include +#include + +namespace { + +// vla_model is opaque to C, so it can just wrap the C++ handle. +struct vla_model_impl { + vla::Model * m = nullptr; +}; + +vla::PixelFormat to_pixel_format(int32_t f) { + return f == VLA_PIXEL_F32_RGB_01 ? vla::PixelFormat::F32_RGB_01 + : vla::PixelFormat::U8; +} + +} // namespace + +struct vla_model : vla_model_impl {}; + +extern "C" { + +int32_t vla_abi_version(void) { return VLA_ABI_VERSION; } + +vla_model * vla_model_load(const char * mmproj_path, const char * ckpt_path, + const char * config_path) { + if (!ckpt_path) return nullptr; + try { + vla::Model * m = vla::model_load(mmproj_path ? mmproj_path : "", + ckpt_path, + config_path ? config_path : ""); + if (!m) return nullptr; + auto * h = new vla_model(); + h->m = m; + return h; + } catch (...) { + return nullptr; + } +} + +void vla_model_free(vla_model * h) { + if (!h) return; + vla::model_free(h->m); + delete h; +} + +int32_t vla_model_config(const vla_model * h, vla_config * out) { + if (!h || !h->m || !out) return VLA_ERR_ARG; + try { + const vla::Config & c = vla::model_config(h->m); + *out = vla_config{}; + out->n_img = c.n_img; + out->n_lang = c.n_lang; + out->n_state = c.n_state; + out->n_prefix = c.n_prefix; + out->n_suffix = c.n_suffix; + out->n_full = c.n_full; + out->hidden = c.hidden; + out->expert_h = c.expert_h; + out->intermediate = c.intermediate; + out->expert_inter = c.expert_inter; + out->n_q_heads = c.n_q_heads; + out->n_kv_heads = c.n_kv_heads; + out->head_dim = c.head_dim; + out->q_full_dim = c.q_full_dim; + out->kv_full_dim = c.kv_full_dim; + out->n_layers = c.n_layers; + out->self_attn_every_n = c.self_attn_every_n; + out->max_state_dim = c.max_state_dim; + out->max_action_dim = c.max_action_dim; + out->real_state_dim = c.real_state_dim; + out->real_action_dim = c.real_action_dim; + out->norm_eps = c.norm_eps; + out->min_period = c.min_period; + out->max_period = c.max_period; + out->num_steps = c.num_steps; + out->rms_eps = c.rms_eps; + out->rope_n_dims = c.rope_n_dims; + out->rope_mode = c.rope_mode; + out->rope_freq_base = c.rope_freq_base; + out->denormalized = c.denormalized ? 1 : 0; + return VLA_OK; + } catch (...) { + return VLA_ERR_EXCEPTION; + } +} + +int32_t vla_predict(vla_model * h, const vla_inputs * in, + float ** out_actions, int64_t * out_n) { + if (!h || !h->m || !in || !out_actions || !out_n) return VLA_ERR_ARG; + *out_actions = nullptr; + *out_n = 0; + if (in->n_images < 0 || in->n_lang < 0 || in->n_img_views < 0) return VLA_ERR_ARG; + if (in->n_images > 0 && !in->images) return VLA_ERR_ARG; + if (in->n_lang > 0 && !in->lang_tokens) return VLA_ERR_ARG; + + try { + std::vector views((size_t) (in->n_images > 0 ? in->n_images : 0)); + for (size_t i = 0; i < views.size(); ++i) { + views[i] = vla::ImageView{ in->images[i].data, + in->images[i].w, + in->images[i].h, + to_pixel_format(in->images[i].format) }; + } + + vla::Inputs ci{}; + ci.images = views.empty() ? nullptr : views.data(); + ci.n_images = (int) views.size(); + ci.precomputed_img_emb = in->precomputed_img_emb; + ci.n_img_views = in->n_img_views; + ci.lang_tokens = in->lang_tokens; + ci.n_lang = in->n_lang; + ci.state = in->state; + ci.noise = in->noise; + ci.attention_mask = in->attention_mask; + ci.attention_mask_n = in->attention_mask_n; + ci.timing_detail = in->timing_detail == VLA_TIMING_PHASE + ? vla::TimingDetail::PHASE + : vla::TimingDetail::NONE; + + const std::vector act = vla::predict(h->m, ci); + if (act.empty()) return VLA_ERR_PREDICT; + + // malloc pairs with vla_free_actions, which callers may replace. + float * buf = (float *) std::malloc(act.size() * sizeof(float)); + if (!buf) return VLA_ERR_EXCEPTION; + std::memcpy(buf, act.data(), act.size() * sizeof(float)); + *out_actions = buf; + *out_n = (int64_t) act.size(); + return VLA_OK; + } catch (...) { + return VLA_ERR_EXCEPTION; + } +} + +void vla_free_actions(float * actions) { std::free(actions); } + +int32_t vla_last_stats(const vla_model * h, vla_stats * out) { + if (!h || !h->m || !out) return VLA_ERR_ARG; + try { + const vla::Stats & s = vla::last_stats(h->m); + out->ms_total = s.ms_total; + out->ms_vision = s.ms_vision; + out->ms_inference = s.ms_inference; + out->ms_prefill = s.ms_prefill; + out->ms_denoise = s.ms_denoise; + return VLA_OK; + } catch (...) { + return VLA_ERR_EXCEPTION; + } +} + +} // extern "C" diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 23b536d..99e39c8 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -15,6 +15,12 @@ target_include_directories(test_rope_conventions PRIVATE ${CMAKE_SOURCE_DIR}/src target_compile_options(test_rope_conventions PRIVATE -Wall -Wextra) add_test(NAME rope_conventions COMMAND test_rope_conventions) +# Built as C so the public header stays C-clean. +add_executable(test_c_api test_c_api.c) +target_link_libraries(test_c_api PRIVATE vla) +target_compile_options(test_c_api PRIVATE -Wall -Wextra) +add_test(NAME c_api COMMAND test_c_api) + add_executable(test_dit_common test_dit_common.cpp) target_include_directories(test_dit_common PRIVATE ${CMAKE_SOURCE_DIR}/src) target_link_libraries(test_dit_common PRIVATE ggml) diff --git a/tests/test_c_api.c b/tests/test_c_api.c new file mode 100644 index 0000000..7371e4a --- /dev/null +++ b/tests/test_c_api.c @@ -0,0 +1,113 @@ +/* Copyright 2026 VinRobotics + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* Compiled as C, not C++, so it fails if the header ever stops being C-clean. + * Argument checks run with no model. Set VLA_TEST_GGUF to also run a real + * predict and compare against vla-cli on the same inputs. */ + +#include "vla.h" + +#include +#include +#include + +static int failures = 0; + +static void check(int cond, const char * what) { + if (!cond) { printf("FAIL: %s\n", what); failures++; } +} + +int main(void) { + check(vla_abi_version() == VLA_ABI_VERSION, "abi version matches header"); + + /* Null handling must not crash. */ + vla_model_free(NULL); + vla_free_actions(NULL); + check(vla_model_load(NULL, NULL, NULL) == NULL, "load(NULL) returns NULL"); + + vla_config cfg; + check(vla_model_config(NULL, &cfg) == VLA_ERR_ARG, "config(NULL) is ERR_ARG"); + vla_stats st; + check(vla_last_stats(NULL, &st) == VLA_ERR_ARG, "stats(NULL) is ERR_ARG"); + + float * act = NULL; + int64_t n = 0; + vla_inputs in; + memset(&in, 0, sizeof(in)); + check(vla_predict(NULL, &in, &act, &n) == VLA_ERR_ARG, "predict(NULL) is ERR_ARG"); + + const char * gguf = getenv("VLA_TEST_GGUF"); + if (!gguf) { + printf("c api: ok (argument checks only; set VLA_TEST_GGUF for a real run)\n"); + return failures ? 1 : 0; + } + + vla_model * m = vla_model_load(getenv("VLA_TEST_MMPROJ"), gguf, NULL); + check(m != NULL, "load real checkpoint"); + if (!m) return 1; + + check(vla_model_config(m, &cfg) == VLA_OK, "config on a loaded model"); + printf("c api: max_action_dim=%lld n_suffix=%lld denormalized=%d\n", + (long long) cfg.max_action_dim, (long long) cfg.n_suffix, cfg.denormalized); + + /* Same synthetic inputs predict_check uses, so the numbers are comparable. */ + int side = getenv("VLA_IMG_SIZE") ? atoi(getenv("VLA_IMG_SIZE")) : 224; + unsigned char * px = (unsigned char *) malloc((size_t) 3 * side * side); + check(px != NULL, "pixel buffer"); + if (!px) return 1; + for (int y = 0; y < side; ++y) + for (int x = 0; x < side; ++x) + for (int c = 0; c < 3; ++c) + px[((size_t) y * side + x) * 3 + c] = (unsigned char) ((x + 2 * y + 40 * c) & 0xFF); + + vla_image img; + img.data = px; img.w = side; img.h = side; img.format = VLA_PIXEL_U8; + + int32_t lang[6] = { 1, 100, 200, 300, 400, 2 }; + float * state = (float *) calloc((size_t) (cfg.max_state_dim > 0 ? cfg.max_state_dim : 1), sizeof(float)); + for (int i = 0; i < (int) cfg.real_state_dim; ++i) state[i] = 0.01f * (float) (i + 1); + + size_t noise_n = (size_t) cfg.max_action_dim * (size_t) cfg.n_suffix; + float * noise = noise_n ? (float *) malloc(noise_n * sizeof(float)) : NULL; + for (size_t i = 0; i < noise_n; ++i) + noise[i] = 0.001f * (float) ((i * 2654435761u) % 1000) - 0.5f; + + memset(&in, 0, sizeof(in)); + in.images = &img; + in.n_images = 1; + in.lang_tokens = lang; + in.n_lang = 6; + in.state = state; + in.noise = noise; + + int32_t rc = vla_predict(m, &in, &act, &n); + check(rc == VLA_OK, "predict returns OK"); + check(act != NULL && n > 0, "predict returns a buffer"); + if (rc == VLA_OK) { + printf("action_len=%lld\n", (long long) n); + for (int64_t i = 0; i < n; ++i) printf("%.9g\n", (double) act[i]); + vla_free_actions(act); + } + + check(vla_last_stats(m, &st) == VLA_OK, "stats after predict"); + check(st.ms_total > 0.0f, "total time is positive"); + + free(noise); free(state); free(px); + vla_model_free(m); + + if (failures) { printf("c api: %d FAILURES\n", failures); return 1; } + printf("c api: ok\n"); + return 0; +} From be9c314a489383977f788a5b945c3bd7794d51d1 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sat, 8 Aug 2026 16:30:01 +0700 Subject: [PATCH 17/42] add python bindings over the C ABI --- .github/workflows/build.yml | 2 + bindings/python/README.md | 40 +++++++ bindings/python/pyproject.toml | 27 +++++ bindings/python/vla_cpp/__init__.py | 170 ++++++++++++++++++++++++++++ bindings/python/vla_cpp/_ffi.py | 163 ++++++++++++++++++++++++++ tests/py/test_bindings.py | 98 ++++++++++++++++ 6 files changed, 500 insertions(+) create mode 100644 bindings/python/README.md create mode 100644 bindings/python/pyproject.toml create mode 100644 bindings/python/vla_cpp/__init__.py create mode 100644 bindings/python/vla_cpp/_ffi.py create mode 100644 tests/py/test_bindings.py diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index b549141..2ea5183 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -35,6 +35,8 @@ jobs: python-version: '3.11' - name: converter remap run: python tests/py/test_converters.py + - name: binding struct parity + run: python tests/py/test_bindings.py build-gate: runs-on: ubuntu-24.04 diff --git a/bindings/python/README.md b/bindings/python/README.md new file mode 100644 index 0000000..ebf406c --- /dev/null +++ b/bindings/python/README.md @@ -0,0 +1,40 @@ +# vla-cpp + +Python bindings for [vla.cpp](https://github.com/VinRobotics/vla.cpp) over its +C ABI (`include/vla.h`). No PyTorch at inference time. + +```python +import vla_cpp + +model = vla_cpp.load("smolvla-libero.gguf") +actions = model.predict(frame_hwc_uint8, tokens=[1, 100, 200, 2]) +``` + +`predict` returns `[rows, max_action_dim]`; only the first +`model.config.real_action_dim` columns carry values. `model.config.denormalized` +says whether they are already in world units. + +## Finding the library + +Wheels bundle `libvla.so`. From a source checkout, build it and point at it: + +```bash +cmake -B build -DCMAKE_BUILD_TYPE=Release +cmake --build build -j"$(nproc)" --target vla +VLA_LIBRARY=build/libvla.so LD_LIBRARY_PATH=build:build/bin python your_script.py +``` + +`LD_LIBRARY_PATH` is needed because `libvla.so` links `libvla_core.so` and the +ggml libraries from the same build tree. + +## API + +| | | +|---|---| +| `load(ckpt, mmproj=None, config=None)` | `mmproj` only for SmolVLA, pi0, pi0.5 | +| `Model.predict(images, tokens, state=None, noise=None, ...)` | `images` is one HWC array or a sequence | +| `Model.config` | resolved hyper-parameters | +| `Model.last_stats()` | per-phase timings, needs `timing=TIMING_PHASE` | +| `Model.close()` | or use as a context manager | + +Output is bit-identical to `vla-cli` on the same inputs. diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml new file mode 100644 index 0000000..0202677 --- /dev/null +++ b/bindings/python/pyproject.toml @@ -0,0 +1,27 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "vla-cpp" +version = "0.1.1" +description = "Python bindings for vla.cpp, a C++ inference engine for Vision-Language-Action models." +readme = "README.md" +requires-python = ">=3.9" +license = { text = "Apache-2.0" } +dependencies = [] + +[project.optional-dependencies] +# Only for the array return type; the bindings work on plain lists without it. +numpy = ["numpy>=1.24"] + +[project.urls] +Homepage = "https://github.com/VinRobotics/vla.cpp" + +[tool.setuptools] +packages = ["vla_cpp"] + +# libvla is built by cmake, not by setuptools. A wheel bundles it next to the +# package; a source checkout finds it through VLA_LIBRARY or the loader path. +[tool.setuptools.package-data] +vla_cpp = ["*.so", "*.dylib", "*.dll", "lib/*"] diff --git a/bindings/python/vla_cpp/__init__.py b/bindings/python/vla_cpp/__init__.py new file mode 100644 index 0000000..5893866 --- /dev/null +++ b/bindings/python/vla_cpp/__init__.py @@ -0,0 +1,170 @@ +"""Python bindings for vla.cpp. + + import vla_cpp + model = vla_cpp.load("smolvla-libero.gguf") + actions = model.predict(image_hwc_uint8, tokens=[1, 100, 200, 2]) + +Actions are [rows, max_action_dim] float32; only the first +``model.config.real_action_dim`` columns carry values. +""" + +from __future__ import annotations + +import ctypes +from ctypes import POINTER, c_float, c_int32, c_int64 +from typing import Sequence + +from . import _ffi +from ._ffi import PIXEL_F32_RGB_01, PIXEL_U8, TIMING_NONE, TIMING_PHASE + +__all__ = ["Model", "load", "PIXEL_U8", "PIXEL_F32_RGB_01", "TIMING_NONE", "TIMING_PHASE"] + +_lib = None + + +def _lib_handle(): + global _lib + if _lib is None: + _lib = _ffi.load_library() + return _lib + + +def _as_f32_array(values, length: int, name: str): + """Accept a numpy array, a list, or None. Returns (ptr, keepalive).""" + if values is None: + return None, None + buf = (c_float * length)() + try: # numpy fast path without importing numpy as a hard dependency + mv = memoryview(values) + if mv.format == "f" and mv.nbytes == length * 4 and mv.c_contiguous: + ctypes.memmove(buf, (ctypes.c_char * mv.nbytes).from_buffer_copy(mv), mv.nbytes) + return buf, buf + except TypeError: + pass + seq = list(values) + if len(seq) != length: + raise ValueError(f"{name} has {len(seq)} values, model expects {length}") + for i, v in enumerate(seq): + buf[i] = float(v) + return buf, buf + + +class Model: + """A loaded checkpoint. Free it with ``close()`` or a ``with`` block.""" + + def __init__(self, handle, lib): + self._h = handle + self._lib = lib + cfg = _ffi.Config() + rc = lib.vla_model_config(handle, ctypes.byref(cfg)) + if rc != _ffi.OK: + raise RuntimeError(f"vla_model_config failed ({rc})") + self.config = cfg + + def __enter__(self): + return self + + def __exit__(self, *exc): + self.close() + return False + + def close(self): + if getattr(self, "_h", None): + self._lib.vla_model_free(self._h) + self._h = None + + def __del__(self): + self.close() + + def predict(self, images, tokens: Sequence[int], state=None, noise=None, + pixel_format: int = PIXEL_U8, timing: int = TIMING_NONE): + """Run one forward pass. + + images: one HWC array, or a sequence of them for multi-view. uint8 RGB by + default; pass pixel_format=PIXEL_F32_RGB_01 for float RGB in [0, 1]. + """ + if self._h is None: + raise RuntimeError("model is closed") + + views = images if isinstance(images, (list, tuple)) else [images] + if not views: + raise ValueError("at least one image is required") + + img_array = (_ffi.Image * len(views))() + keep = [] + for i, im in enumerate(views): + mv = memoryview(im) + if not mv.c_contiguous: + raise ValueError("image must be C-contiguous") + shape = mv.shape + if len(shape) != 3 or shape[2] != 3: + raise ValueError(f"image must be HxWx3, got {shape}") + raw = (ctypes.c_char * mv.nbytes).from_buffer_copy(mv) + keep.append(raw) + img_array[i].data = ctypes.cast(raw, ctypes.c_void_p) + img_array[i].h = int(shape[0]) + img_array[i].w = int(shape[1]) + img_array[i].format = int(pixel_format) + + tok = list(tokens) + if not tok: + raise ValueError("tokens must not be empty") + tok_buf = (c_int32 * len(tok))(*[int(t) for t in tok]) + + state_ptr, state_keep = _as_f32_array( + state if state is not None else [0.0] * int(self.config.max_state_dim), + int(self.config.max_state_dim), "state") + keep.append(state_keep) + + noise_len = int(self.config.max_action_dim) * int(self.config.n_suffix) + noise_ptr, noise_keep = _as_f32_array(noise, noise_len, "noise") + keep.append(noise_keep) + + cin = _ffi.Inputs() + cin.images = img_array + cin.n_images = len(views) + cin.lang_tokens = tok_buf + cin.n_lang = len(tok) + if state_ptr is not None: + cin.state = ctypes.cast(state_ptr, POINTER(c_float)) + if noise_ptr is not None: + cin.noise = ctypes.cast(noise_ptr, POINTER(c_float)) + cin.timing_detail = int(timing) + + out = POINTER(c_float)() + n = c_int64() + rc = self._lib.vla_predict(self._h, ctypes.byref(cin), ctypes.byref(out), ctypes.byref(n)) + if rc != _ffi.OK: + raise RuntimeError(f"vla_predict failed ({rc})") + try: + flat = [out[i] for i in range(n.value)] + finally: + self._lib.vla_free_actions(out) + + cols = int(self.config.max_action_dim) or 1 + rows = len(flat) // cols if cols else len(flat) + try: + import numpy as np + return np.asarray(flat, dtype="float32").reshape(rows, cols) + except ImportError: + return [flat[r * cols:(r + 1) * cols] for r in range(rows)] + + def last_stats(self) -> _ffi.Stats: + st = _ffi.Stats() + rc = self._lib.vla_last_stats(self._h, ctypes.byref(st)) + if rc != _ffi.OK: + raise RuntimeError(f"vla_last_stats failed ({rc})") + return st + + +def load(ckpt_path: str, mmproj_path: str | None = None, config_path: str | None = None) -> Model: + """Load a checkpoint. mmproj_path is only needed for SmolVLA, pi0 and pi0.5.""" + lib = _lib_handle() + handle = lib.vla_model_load( + mmproj_path.encode() if mmproj_path else None, + ckpt_path.encode(), + config_path.encode() if config_path else None, + ) + if not handle: + raise RuntimeError(f"could not load {ckpt_path}") + return Model(handle, lib) diff --git a/bindings/python/vla_cpp/_ffi.py b/bindings/python/vla_cpp/_ffi.py new file mode 100644 index 0000000..41ae61d --- /dev/null +++ b/bindings/python/vla_cpp/_ffi.py @@ -0,0 +1,163 @@ +"""ctypes declarations for libvla. Mirrors include/vla.h field for field.""" + +from __future__ import annotations + +import ctypes +import os +import sys +from ctypes import ( + POINTER, + c_char_p, + c_double, + c_float, + c_int32, + c_int64, + c_void_p, +) + +ABI_VERSION = 1 + +OK = 0 +ERR_ARG = -1 +ERR_PREDICT = -2 +ERR_EXCEPTION = -3 + +PIXEL_U8 = 0 +PIXEL_F32_RGB_01 = 1 + +TIMING_NONE = 0 +TIMING_PHASE = 1 + + +class Config(ctypes.Structure): + _fields_ = [ + ("n_img", c_int64), + ("n_lang", c_int64), + ("n_state", c_int64), + ("n_prefix", c_int64), + ("n_suffix", c_int64), + ("n_full", c_int64), + ("hidden", c_int64), + ("expert_h", c_int64), + ("intermediate", c_int64), + ("expert_inter", c_int64), + ("n_q_heads", c_int64), + ("n_kv_heads", c_int64), + ("head_dim", c_int64), + ("q_full_dim", c_int64), + ("kv_full_dim", c_int64), + ("n_layers", c_int64), + ("self_attn_every_n", c_int32), + ("max_state_dim", c_int64), + ("max_action_dim", c_int64), + ("real_state_dim", c_int64), + ("real_action_dim", c_int64), + ("norm_eps", c_float), + ("min_period", c_double), + ("max_period", c_double), + ("num_steps", c_int32), + ("rms_eps", c_float), + ("rope_n_dims", c_int32), + ("rope_mode", c_int32), + ("rope_freq_base", c_float), + ("denormalized", c_int32), + ] + + +class Image(ctypes.Structure): + _fields_ = [ + ("data", c_void_p), + ("w", c_int32), + ("h", c_int32), + ("format", c_int32), + ] + + +class Inputs(ctypes.Structure): + _fields_ = [ + ("images", POINTER(Image)), + ("n_images", c_int32), + ("precomputed_img_emb", POINTER(c_float)), + ("n_img_views", c_int32), + ("lang_tokens", POINTER(c_int32)), + ("n_lang", c_int32), + ("state", POINTER(c_float)), + ("noise", POINTER(c_float)), + ("attention_mask", POINTER(c_int32)), + ("attention_mask_n", c_int32), + ("timing_detail", c_int32), + ] + + +class Stats(ctypes.Structure): + _fields_ = [ + ("ms_total", c_float), + ("ms_vision", c_float), + ("ms_inference", c_float), + ("ms_prefill", c_float), + ("ms_denoise", c_float), + ] + + +def _library_name() -> str: + if sys.platform == "darwin": + return "libvla.dylib" + if sys.platform == "win32": + return "vla.dll" + return "libvla.so" + + +def _candidates() -> list[str]: + name = _library_name() + here = os.path.dirname(os.path.abspath(__file__)) + found = [] + env = os.environ.get("VLA_LIBRARY") + if env: + found.append(env) + # Bundled next to the package (what a wheel ships), then a local build tree. + found.append(os.path.join(here, name)) + found.append(os.path.join(here, "lib", name)) + found.append(name) # fall through to the loader search path + return found + + +def load_library() -> ctypes.CDLL: + errors = [] + lib = None + for path in _candidates(): + try: + lib = ctypes.CDLL(path) + break + except OSError as exc: + errors.append(f"{path}: {exc}") + if lib is None: + raise OSError( + "could not load libvla. Set VLA_LIBRARY to its path, or build it with " + "`cmake --build --target vla`.\nTried:\n " + "\n ".join(errors) + ) + + lib.vla_abi_version.restype = c_int32 + lib.vla_abi_version.argtypes = [] + + lib.vla_model_load.restype = c_void_p + lib.vla_model_load.argtypes = [c_char_p, c_char_p, c_char_p] + + lib.vla_model_free.restype = None + lib.vla_model_free.argtypes = [c_void_p] + + lib.vla_model_config.restype = c_int32 + lib.vla_model_config.argtypes = [c_void_p, POINTER(Config)] + + lib.vla_predict.restype = c_int32 + lib.vla_predict.argtypes = [c_void_p, POINTER(Inputs), POINTER(POINTER(c_float)), POINTER(c_int64)] + + lib.vla_free_actions.restype = None + lib.vla_free_actions.argtypes = [POINTER(c_float)] + + lib.vla_last_stats.restype = c_int32 + lib.vla_last_stats.argtypes = [c_void_p, POINTER(Stats)] + + got = lib.vla_abi_version() + if got != ABI_VERSION: + raise RuntimeError(f"libvla ABI {got}, this package expects {ABI_VERSION}") + return lib diff --git a/tests/py/test_bindings.py b/tests/py/test_bindings.py new file mode 100644 index 0000000..c44fa62 --- /dev/null +++ b/tests/py/test_bindings.py @@ -0,0 +1,98 @@ +#!/usr/bin/env python3 +"""Checks the ctypes structs against include/vla.h. + +A field added on one side and not the other silently misreads every value after +it. Set VLA_LIBRARY to also load the library and check its ABI version. +""" + +import ctypes +import os +import re +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.dirname(os.path.dirname(HERE)) +sys.path.insert(0, os.path.join(ROOT, "bindings", "python")) + +from vla_cpp import _ffi # noqa: E402 + +HEADER = os.path.join(ROOT, "include", "vla.h") + +C_TO_CTYPES = { + "int32_t": ctypes.c_int32, + "int64_t": ctypes.c_int64, + "float": ctypes.c_float, + "double": ctypes.c_double, +} + + +def header_struct_fields(text, name): + """Field (name, c_type) pairs of `typedef struct { ... } name;`.""" + # [^{}] so the body cannot swallow the structs in between. + m = re.search(r"typedef struct \{([^{}]*)\}\s*" + name + r"\s*;", text, re.S) + assert m, f"{name} not found in vla.h" + fields = [] + for line in m.group(1).splitlines(): + line = re.sub(r"/\*.*?\*/", "", line) + line = line.split("///")[0].split("//")[0].strip() + if not line or not line.endswith(";"): + continue + decl = line[:-1].strip() + parts = decl.split() + if len(parts) < 2: + continue + ctype, fname = parts[0], parts[-1].lstrip("*") + fields.append((fname, ctype)) + return fields + + +def check_struct(text, c_name, py_struct, skip_types=()): + hdr = header_struct_fields(text, c_name) + py = [(n, t) for n, t in py_struct._fields_] + assert len(hdr) == len(py), ( + f"{c_name}: vla.h has {len(hdr)} fields, python has {len(py)}\n" + f" header: {[n for n, _ in hdr]}\n python: {[n for n, _ in py]}" + ) + for (hn, ht), (pn, pt) in zip(hdr, py): + assert hn == pn, f"{c_name}: field order differs, {hn!r} vs {pn!r}" + if ht in skip_types: + continue + want = C_TO_CTYPES.get(ht) + if want is not None: + assert pt is want, f"{c_name}.{hn}: header {ht}, python {pt}" + print(f" {c_name}: {len(hdr)} fields match") + + +def main(): + text = open(HEADER).read() + + m = re.search(r"#define VLA_ABI_VERSION\s+(\d+)", text) + assert m, "VLA_ABI_VERSION not found" + assert int(m.group(1)) == _ffi.ABI_VERSION, ( + f"vla.h ABI {m.group(1)}, _ffi.ABI_VERSION {_ffi.ABI_VERSION}") + print(f" ABI version {_ffi.ABI_VERSION} matches") + + check_struct(text, "vla_config", _ffi.Config) + check_struct(text, "vla_stats", _ffi.Stats) + # Pointer fields differ in spelling between the two; only order is checked. + check_struct(text, "vla_image", _ffi.Image, skip_types=("void",)) + check_struct(text, "vla_inputs", _ffi.Inputs, + skip_types=("vla_image", "float", "int32_t")) + + for name in ("VLA_OK", "VLA_ERR_ARG", "VLA_ERR_PREDICT", "VLA_ERR_EXCEPTION"): + assert name in text, f"{name} missing from vla.h" + assert _ffi.OK == 0 and _ffi.ERR_ARG == -1 + assert _ffi.ERR_PREDICT == -2 and _ffi.ERR_EXCEPTION == -3 + print(" status codes match") + + if os.environ.get("VLA_LIBRARY"): + lib = _ffi.load_library() + assert lib.vla_abi_version() == _ffi.ABI_VERSION + print(" libvla loads and reports a matching ABI") + + print("bindings: ok") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From c40965435e59aca17864632666503d40b01d9950 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 13:17:50 +0700 Subject: [PATCH 18/42] fail pi0 and pi05 load when the checkpoint is missing weights --- src/models/pi0.cpp | 9 ++++++--- src/models/pi05.cpp | 9 ++++++--- 2 files changed, 12 insertions(+), 6 deletions(-) diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 4710b10..8e7c1ca 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -372,10 +372,12 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, } ggml_context * W = m->ctx_weights; std::vector weights; + // A miss returns before pushing, so the null scan below cannot see it. + bool missing = false; auto mk = [&](const char * name, ggml_type type, int n_dims, const int64_t * ne) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); missing = true; return nullptr; } ggml_tensor * t = ggml_new_tensor(W, g.resident_type(gt, type), n_dims, ne); ggml_set_name(t, name); weights.push_back(t); @@ -384,12 +386,12 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, auto mk_mm = [&](const char * name) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); missing = true; return nullptr; } return mk(name, m->matmul_type, GGML_MAX_DIMS, gt->ne); }; auto mk_f32 = [&](const char * name) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi0): missing tensor %s\n", name); missing = true; return nullptr; } return mk(name, GGML_TYPE_F32, GGML_MAX_DIMS, gt->ne); }; @@ -439,6 +441,7 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, m->W_at1 = mk_f32("action_time_mlp_in.weight"); m->b_at1 = mk_f32("action_time_mlp_in.bias"); m->W_at2 = mk_f32("action_time_mlp_out.weight"); m->b_at2 = mk_f32("action_time_mlp_out.bias"); m->W_aout = mk_f32("action_out_proj.weight"); m->b_aout = mk_f32("action_out_proj.bias"); + if (missing) { std::fprintf(stderr, "vla(pi0): checkpoint is missing weights\n"); return nullptr; } for (ggml_tensor * t : weights) if (!t) { std::fprintf(stderr, "vla(pi0): weight tensor creation failed\n"); return nullptr; } if (!m->ex_final_norm || !m->W_sp || !m->b_sp || !m->W_ain || !m->b_ain || !m->W_at1 || !m->b_at1 || !m->W_at2 || !m->b_at2 || !m->W_aout || !m->b_aout) { diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 263d043..376461d 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -453,10 +453,12 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, } ggml_context * W = m->ctx_weights; std::vector weights; + // A miss returns before pushing, so the null scan below cannot see it. + bool missing = false; auto mk = [&](const char * name, ggml_type type, int n_dims, const int64_t * ne) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); missing = true; return nullptr; } ggml_tensor * t = ggml_new_tensor(W, g.resident_type(gt, type), n_dims, ne); ggml_set_name(t, name); weights.push_back(t); @@ -464,12 +466,12 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, }; auto mk_mm = [&](const char * name) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); missing = true; return nullptr; } return mk(name, m->matmul_type, GGML_MAX_DIMS, gt->ne); }; auto mk_f32 = [&](const char * name) -> ggml_tensor * { const ggml_tensor * gt = g.meta(name); - if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); return nullptr; } + if (!gt) { std::fprintf(stderr, "vla(pi05): missing tensor %s\n", name); missing = true; return nullptr; } return mk(name, GGML_TYPE_F32, GGML_MAX_DIMS, gt->ne); }; @@ -536,6 +538,7 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, m->W_tin = mk_f32("time_mlp_in.weight"); m->b_tin = mk_f32("time_mlp_in.bias"); m->W_tout = mk_f32("time_mlp_out.weight"); m->b_tout = mk_f32("time_mlp_out.bias"); m->W_aout = mk_f32("action_out_proj.weight"); m->b_aout = mk_f32("action_out_proj.bias"); + if (missing) { std::fprintf(stderr, "vla(pi05): checkpoint is missing weights\n"); return nullptr; } for (ggml_tensor * t : weights) if (!t) { std::fprintf(stderr, "vla(pi05): weight tensor creation failed\n"); return nullptr; } if (!m->ex_final_w || !m->ex_final_b || !m->W_ain || !m->b_ain || !m->W_tin || !m->b_tin || !m->W_tout || !m->b_tout || !m->W_aout || !m->b_aout) { From 843e619510666554b4bf32b10a44471ccb9f3ed3 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 13:17:50 +0700 Subject: [PATCH 19/42] share the qwen3-vl vision tower between gr00t n1.7 and vla-jepa --- src/models/gr00tn1d7.cpp | 136 +---------------------------- src/models/qwen3vl_vit.h | 173 +++++++++++++++++++++++++++++++++++++ src/models/vla_jepa.cpp | 114 +----------------------- tests/CMakeLists.txt | 6 ++ tests/predict_check.cpp | 5 ++ tests/test_qwen3vl_vit.cpp | 121 ++++++++++++++++++++++++++ 6 files changed, 311 insertions(+), 244 deletions(-) create mode 100644 src/models/qwen3vl_vit.h create mode 100644 tests/test_qwen3vl_vit.cpp diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index acbda46..13a3dae 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -22,6 +22,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/dit_common.h" +#include "models/qwen3vl_vit.h" #include #include @@ -40,11 +41,6 @@ namespace vla { namespace { -constexpr float CLIP_MEAN[3] = {0.5f, 0.5f, 0.5f}; -constexpr float CLIP_STD [3] = {0.5f, 0.5f, 0.5f}; - -struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; -struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; struct VlsaLayerW { ggml_tensor *n1w,*n1b,*n3w,*n3b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wff0,*bff0,*Wff2,*bff2; }; struct Qwen3LayerW { ggml_tensor *attn_norm,*Wq,*Wk,*Wv,*Wo,*q_norm,*k_norm,*ffn_norm,*Wgate,*Wup,*Wdown; }; struct DitLayerW { ggml_tensor *adaln_w,*adaln_b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wff0,*bff0,*Wff2,*bff2; @@ -119,74 +115,12 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { namespace { -ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { - const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; - ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); - ggml_tensor * x2 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], (size_t) half * x->nb[0])); - ggml_tensor * rot = ggml_concat(C, ggml_neg(C, x2), x1, 0); - return ggml_add(C, ggml_mul(C, x, cos_t), ggml_mul(C, rot, sin_t)); -} - -inline bool fa_enabled() { static const bool e = (std::getenv("VLA_GR00T_FA") != nullptr); return e; } - ggml_tensor * head_view(ggml_context * C, ggml_tensor * proj, int64_t hd, int64_t heads, int64_t T, int64_t E, int nblk, int blk) { const size_t es = ggml_element_size(proj); return ggml_view_3d(C, proj, hd, heads, T, (size_t) hd * es, (size_t) nblk * E * es, (size_t) blk * E * es); } -ggml_tensor * flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, - ggml_tensor * mask, float scale, int64_t hidden) { - (void) hidden; - ggml_tensor * kf = (k->type == GGML_TYPE_F16) ? k : ggml_cast(C, k, GGML_TYPE_F16); - ggml_tensor * vf = (v->type == GGML_TYPE_F16) ? v : ggml_cast(C, v, GGML_TYPE_F16); - ggml_tensor * o = ggml_flash_attn_ext(C, q, kf, vf, mask, scale, 0.0f, 0.0f); - ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); - return ggml_reshape_2d(C, o, o->ne[0] * o->ne[1], o->ne[2] * o->ne[3]); -} - -ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t, - int64_t seq, int64_t heads, int64_t hd, int64_t hidden, float ln_eps) { - const float scale = 1.0f / std::sqrt((float) hd); - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); - ggml_tensor * qkv = ggml_add(C, ggml_mul_mat(C, w.Wqkv, n1), w.bqkv); - ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], 0)); - ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) hidden * qkv->nb[0])); - ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) 2 * hidden * qkv->nb[0])); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, hd, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, hd, heads, seq), 0, 2, 1, 3)); - Q = rope2d(C, Q, cos_t, sin_t); K = rope2d(C, K, cos_t, sin_t); - ggml_tensor * att; - if (fa_enabled()) { - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 0, 2, 1, 3)); - att = flash_attn(C, Q, K, V, nullptr, scale, hidden); - } else { - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); - } - ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); - return ggml_add(C, h1, ff); -} - -ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tensor * x, int64_t hidden, int64_t merge2, float ln_eps, bool pre_merge) { - ggml_tensor * m; - if (pre_merge) { - ggml_tensor * xn = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.nw), w.nb); - const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); - m = ggml_reshape_2d(C, ggml_cont(C, xn), c_merged, n_merged); - } else { - const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); - ggml_tensor * mr = ggml_reshape_2d(C, ggml_cont(C, x), c_merged, n_merged); - m = ggml_add(C, ggml_mul(C, ggml_norm(C, mr, ln_eps), w.nw), w.nb); - } - ggml_tensor * z1 = ggml_add(C, ggml_mul_mat(C, w.fc1w, m), w.fc1b); - return ggml_add(C, ggml_mul_mat(C, w.fc2w, ggml_gelu(C, z1)), w.fc2b); -} - ggml_tensor * build_vlsa_layer(ggml_context * C, const VlsaLayerW & w, ggml_tensor * x, int64_t seq, int64_t heads, int64_t hd, int64_t hidden, float ln_eps) { const float scale = 1.0f / std::sqrt((float) hd); @@ -199,7 +133,7 @@ ggml_tensor * build_vlsa_layer(ggml_context * C, const VlsaLayerW & w, ggml_tens ggml_tensor * att; if (fa_enabled()) { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 0, 2, 1, 3)); - att = flash_attn(C, Q, K, V, nullptr, scale, hidden); + att = flash_attn(C, Q, K, V, nullptr, scale); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 1, 2, 0, 3)); ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); @@ -233,7 +167,7 @@ ggml_tensor * build_qwen3_layer(ggml_context * C, const Gr00tN1d7ModelArch & m, ggml_tensor * att; if (fa_enabled()) { ggml_tensor * V = ggml_cont(C, ggml_permute(C, vh, 0, 2, 1, 3)); - att = flash_attn(C, Q, K, V, ggml_cast(C, mask, GGML_TYPE_F16), scale, hq); + att = flash_attn(C, Q, K, V, ggml_cast(C, mask, GGML_TYPE_F16), scale); } else { ggml_tensor * V = ggml_cont(C, ggml_permute(C, vh, 1, 2, 0, 3)); ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); @@ -289,68 +223,6 @@ ggml_tensor * build_dit_block(ggml_context * C, const Gr00tN1d7ModelArch & m, co return ggml_add(C, h1, ff); } -void merge_block_coords(int64_t gh, int64_t gw, int64_t m, std::vector & row, std::vector & col) { - const int64_t S = gh * gw; row.assign(S, 0); col.assign(S, 0); - for (int64_t s = 0; s < S; ++s) { - int64_t t = s; const int64_t wj = t % m; t /= m; const int64_t wi = t % m; t /= m; - const int64_t bc = t % (gw / m); t /= (gw / m); const int64_t br = t; - row[s] = br * m + wi; col[s] = bc * m + wj; - } -} - -void vit_rope_tables(const std::vector & row, const std::vector & col, int64_t hd, double theta, - std::vector & cos_t, std::vector & sin_t) { - const int64_t S = (int64_t) row.size(), nf = hd / 4; - std::vector invf(nf); - for (int64_t i = 0; i < nf; ++i) invf[i] = 1.0 / std::pow(theta, (double)(2 * i) / (double)(hd / 2)); - cos_t.assign((size_t) S * hd, 0.0f); sin_t.assign((size_t) S * hd, 0.0f); - for (int64_t s = 0; s < S; ++s) { - std::vector emb(hd); - for (int64_t i = 0; i < nf; ++i) { emb[i] = (double) row[s] * invf[i]; emb[nf + i] = (double) col[s] * invf[i]; } - for (int64_t i = 0; i < hd / 2; ++i) emb[hd / 2 + i] = emb[i]; - for (int64_t i = 0; i < hd; ++i) { cos_t[s * hd + i] = (float) std::cos(emb[i]); sin_t[s * hd + i] = (float) std::sin(emb[i]); } - } -} - -void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, - const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, - std::vector & out) { - const int64_t S = (int64_t) row.size(); - out.assign((size_t) S * hidden, 0.0f); - auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side - 1) / (double)(g - 1); }; - for (int64_t s = 0; s < S; ++s) { - const double hy = src_coord(row[s], gh), wx = src_coord(col[s], gw); - const int64_t h0 = (int64_t) std::floor(hy), w0 = (int64_t) std::floor(wx); - const int64_t h1 = std::min(h0 + 1, num_side - 1), w1 = std::min(w0 + 1, num_side - 1); - const double dh = hy - h0, dw = wx - w0; - const double c00 = (1 - dh) * (1 - dw), c01 = (1 - dh) * dw, c10 = dh * (1 - dw), c11 = dh * dw; - const float * T00 = &table[(h0 * num_side + w0) * hidden]; const float * T01 = &table[(h0 * num_side + w1) * hidden]; - const float * T10 = &table[(h1 * num_side + w0) * hidden]; const float * T11 = &table[(h1 * num_side + w1) * hidden]; - for (int64_t c = 0; c < hidden; ++c) out[s * hidden + c] = (float)(c00 * T00[c] + c01 * T01[c] + c10 * T10[c] + c11 * T11[c]); - } -} - -bool preprocess_image_patches(const ImageView & v, int64_t side, int64_t ps, int64_t tps, - const std::vector & row, const std::vector & col, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(gr00tn1d7): image view is %dx%d, expected %lldx%lld\n", v.w, v.h, (long long) side, (long long) side); return false; - } - const int64_t S = (int64_t) row.size(), pf = 3 * tps * ps * ps; - out.assign((size_t) pf * S, 0.0f); - auto px = [&](int64_t r, int64_t c, int64_t ch) -> float { - if (v.format == PixelFormat::U8) return ((const uint8_t *) v.data)[(r * side + c) * 3 + ch] / 255.0f; - return ((const float *) v.data)[(r * side + c) * 3 + ch]; - }; - for (int64_t s = 0; s < S; ++s) - for (int64_t ch = 0; ch < 3; ++ch) - for (int64_t ph = 0; ph < ps; ++ph) - for (int64_t pw = 0; pw < ps; ++pw) { - const float val = (px(row[s] * ps + ph, col[s] * ps + pw, ch) - CLIP_MEAN[ch]) / CLIP_STD[ch]; - for (int64_t t = 0; t < tps; ++t) out[s * pf + ch * tps * ps * ps + t * ps * ps + ph * ps + pw] = val; - } - return true; -} - bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; @@ -682,7 +554,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { std::vector patches; bool vok = true; for (int64_t v = 0; v < n_views && vok; ++v) { - if (!preprocess_image_patches(in.images[v], side, ps, temporal_patch, grow, gcol, patches)) { vok = false; break; } + if (!preprocess_image_patches("gr00tn1d7", in.images[v], side, ps, temporal_patch, grow, gcol, patches)) { vok = false; break; } ggml_backend_tensor_set(t_pos, pos_interp.data(), 0, ggml_nbytes(t_pos)); ggml_backend_tensor_set(t_cos, rope_cos.data(), 0, ggml_nbytes(t_cos)); diff --git a/src/models/qwen3vl_vit.h b/src/models/qwen3vl_vit.h new file mode 100644 index 0000000..cab28a3 --- /dev/null +++ b/src/models/qwen3vl_vit.h @@ -0,0 +1,173 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Qwen3-VL vision tower, shared verbatim by GR00T N1.7 and VLA-JEPA. The LM and +// action heads differ, so they stay in their own files. + +#pragma once + +#include "model.h" + +#include "ggml.h" + +#include +#include +#include +#include +#include +#include + +namespace vla { + +constexpr float QWEN3VL_MEAN[3] = {0.5f, 0.5f, 0.5f}; +constexpr float QWEN3VL_STD [3] = {0.5f, 0.5f, 0.5f}; + +struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; +struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; + +// Half-split rotation, matching the Qwen3-VL reference. +inline ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { + const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; + ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); + ggml_tensor * x2 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], (size_t) half * x->nb[0])); + ggml_tensor * rot = ggml_concat(C, ggml_neg(C, x2), x1, 0); + return ggml_add(C, ggml_mul(C, x, cos_t), ggml_mul(C, rot, sin_t)); +} + +inline bool fa_enabled() { static const bool e = (std::getenv("VLA_FLASH_ATTN") != nullptr); return e; } + +// K and V must be F16 for ggml_flash_attn_ext; the accumulator stays F32. +inline ggml_tensor * flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, + ggml_tensor * mask, float scale) { + ggml_tensor * kf = (k->type == GGML_TYPE_F16) ? k : ggml_cast(C, k, GGML_TYPE_F16); + ggml_tensor * vf = (v->type == GGML_TYPE_F16) ? v : ggml_cast(C, v, GGML_TYPE_F16); + ggml_tensor * o = ggml_flash_attn_ext(C, q, kf, vf, mask, scale, 0.0f, 0.0f); + ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); + return ggml_reshape_2d(C, o, o->ne[0] * o->ne[1], o->ne[2] * o->ne[3]); +} + +inline ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml_tensor * x, + ggml_tensor * cos_t, ggml_tensor * sin_t, + int64_t seq, int64_t heads, int64_t hd, int64_t hidden, float ln_eps) { + const float scale = 1.0f / std::sqrt((float) hd); + ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); + ggml_tensor * qkv = ggml_add(C, ggml_mul_mat(C, w.Wqkv, n1), w.bqkv); + ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], 0)); + ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) hidden * qkv->nb[0])); + ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) 2 * hidden * qkv->nb[0])); + ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, hd, heads, seq), 0, 2, 1, 3)); + ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, hd, heads, seq), 0, 2, 1, 3)); + Q = rope2d(C, Q, cos_t, sin_t); K = rope2d(C, K, cos_t, sin_t); + ggml_tensor * att; + if (fa_enabled()) { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 0, 2, 1, 3)); + att = flash_attn(C, Q, K, V, nullptr, scale); + } else { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); + att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + } + ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); + ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); + ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); + return ggml_add(C, h1, ff); +} + +// pre_merge normalizes before the space-to-depth reshape, the deepstack taps after. +inline ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tensor * x, + int64_t hidden, int64_t merge2, float ln_eps, bool pre_merge) { + const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); + ggml_tensor * m; + if (pre_merge) { + ggml_tensor * xn = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.nw), w.nb); + m = ggml_reshape_2d(C, ggml_cont(C, xn), c_merged, n_merged); + } else { + ggml_tensor * mr = ggml_reshape_2d(C, ggml_cont(C, x), c_merged, n_merged); + m = ggml_add(C, ggml_mul(C, ggml_norm(C, mr, ln_eps), w.nw), w.nb); + } + ggml_tensor * z1 = ggml_add(C, ggml_mul_mat(C, w.fc1w, m), w.fc1b); + return ggml_add(C, ggml_mul_mat(C, w.fc2w, ggml_gelu(C, z1)), w.fc2b); +} + +// Patch order after the spatial merge: row/col of each patch in the m x m block. +inline void merge_block_coords(int64_t gh, int64_t gw, int64_t m, std::vector & row, std::vector & col) { + const int64_t S = gh * gw; row.assign(S, 0); col.assign(S, 0); + for (int64_t s = 0; s < S; ++s) { + int64_t t = s; const int64_t wj = t % m; t /= m; const int64_t wi = t % m; t /= m; + const int64_t bc = t % (gw / m); t /= (gw / m); const int64_t br = t; + row[s] = br * m + wi; col[s] = bc * m + wj; + } +} + +inline void vit_rope_tables(const std::vector & row, const std::vector & col, int64_t hd, double theta, + std::vector & cos_t, std::vector & sin_t) { + const int64_t S = (int64_t) row.size(), nf = hd / 4; + std::vector invf(nf); + for (int64_t i = 0; i < nf; ++i) invf[i] = 1.0 / std::pow(theta, (double)(2 * i) / (double)(hd / 2)); + cos_t.assign((size_t) S * hd, 0.0f); sin_t.assign((size_t) S * hd, 0.0f); + for (int64_t s = 0; s < S; ++s) { + std::vector emb(hd); + for (int64_t i = 0; i < nf; ++i) { emb[i] = (double) row[s] * invf[i]; emb[nf + i] = (double) col[s] * invf[i]; } + for (int64_t i = 0; i < hd / 2; ++i) emb[hd / 2 + i] = emb[i]; + for (int64_t i = 0; i < hd; ++i) { cos_t[s * hd + i] = (float) std::cos(emb[i]); sin_t[s * hd + i] = (float) std::sin(emb[i]); } + } +} + +// Bilinear resample of the pretrained num_side x num_side position table onto gh x gw. +inline void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, + const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, + std::vector & out) { + const int64_t S = (int64_t) row.size(); + out.assign((size_t) S * hidden, 0.0f); + auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side - 1) / (double)(g - 1); }; + for (int64_t s = 0; s < S; ++s) { + const double hy = src_coord(row[s], gh), wx = src_coord(col[s], gw); + const int64_t h0 = (int64_t) std::floor(hy), w0 = (int64_t) std::floor(wx); + const int64_t h1 = std::min(h0 + 1, num_side - 1), w1 = std::min(w0 + 1, num_side - 1); + const double dh = hy - h0, dw = wx - w0; + const double c00 = (1 - dh) * (1 - dw), c01 = (1 - dh) * dw, c10 = dh * (1 - dw), c11 = dh * dw; + const float * T00 = &table[(h0 * num_side + w0) * hidden]; const float * T01 = &table[(h0 * num_side + w1) * hidden]; + const float * T10 = &table[(h1 * num_side + w0) * hidden]; const float * T11 = &table[(h1 * num_side + w1) * hidden]; + for (int64_t c = 0; c < hidden; ++c) out[s * hidden + c] = (float)(c00 * T00[c] + c01 * T01[c] + c10 * T10[c] + c11 * T11[c]); + } +} + +// HWC to flat patches in CLIP normalization. No resize: the view must already be +// side x side. arch only labels the error. +inline bool preprocess_image_patches(const char * arch, const ImageView & v, int64_t side, int64_t ps, int64_t tps, + const std::vector & row, const std::vector & col, + std::vector & out) { + if (v.w != (int) side || v.h != (int) side || !v.data) { + std::fprintf(stderr, "vla(%s): image view is %dx%d, expected %lldx%lld\n", + arch, v.w, v.h, (long long) side, (long long) side); + return false; + } + const int64_t S = (int64_t) row.size(), pf = 3 * tps * ps * ps; + out.assign((size_t) pf * S, 0.0f); + auto px = [&](int64_t r, int64_t c, int64_t ch) -> float { + if (v.format == PixelFormat::U8) return ((const uint8_t *) v.data)[(r * side + c) * 3 + ch] / 255.0f; + return ((const float *) v.data)[(r * side + c) * 3 + ch]; + }; + for (int64_t s = 0; s < S; ++s) + for (int64_t ch = 0; ch < 3; ++ch) + for (int64_t ph = 0; ph < ps; ++ph) + for (int64_t pw = 0; pw < ps; ++pw) { + const float val = (px(row[s] * ps + ph, col[s] * ps + pw, ch) - QWEN3VL_MEAN[ch]) / QWEN3VL_STD[ch]; + for (int64_t t = 0; t < tps; ++t) out[s * pf + ch * tps * ps * ps + t * ps * ps + ph * ps + pw] = val; + } + return true; +} + +} // namespace vla diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 7bba5a2..6f093f0 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -22,6 +22,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/dit_common.h" +#include "models/qwen3vl_vit.h" #include #include @@ -40,11 +41,6 @@ namespace vla { namespace { -constexpr float CLIP_MEAN[3] = {0.5f, 0.5f, 0.5f}; -constexpr float CLIP_STD [3] = {0.5f, 0.5f, 0.5f}; - -struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; -struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; struct Qwen3LayerW { ggml_tensor *attn_norm,*Wq,*Wk,*Wv,*Wo,*q_norm,*k_norm,*ffn_norm,*Wgate,*Wup,*Wdown; }; struct DitLayerW { ggml_tensor *adaln_w,*adaln_b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wff0,*bff0,*Wff2,*bff2; }; @@ -103,50 +99,6 @@ struct VlaJepaModelArch : public ModelArchBase { namespace { -ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { - const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; - ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); - ggml_tensor * x2 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], (size_t) half * x->nb[0])); - ggml_tensor * rot = ggml_concat(C, ggml_neg(C, x2), x1, 0); - return ggml_add(C, ggml_mul(C, x, cos_t), ggml_mul(C, rot, sin_t)); -} - -ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t, - int64_t seq, int64_t heads, int64_t hd, int64_t hidden, float ln_eps) { - const float scale = 1.0f / std::sqrt((float) hd); - ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); - ggml_tensor * qkv = ggml_add(C, ggml_mul_mat(C, w.Wqkv, n1), w.bqkv); - ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], 0)); - ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) hidden * qkv->nb[0])); - ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, hidden, seq, qkv->nb[1], (size_t) 2 * hidden * qkv->nb[0])); - ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, hd, heads, seq), 0, 2, 1, 3)); - ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, hd, heads, seq), 0, 2, 1, 3)); - Q = rope2d(C, Q, cos_t, sin_t); K = rope2d(C, K, cos_t, sin_t); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); - ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); - ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); - return ggml_add(C, h1, ff); -} - -ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tensor * x, int64_t hidden, int64_t merge2, float ln_eps, bool pre_merge) { - ggml_tensor * m; - if (pre_merge) { - ggml_tensor * xn = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.nw), w.nb); - const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); - m = ggml_reshape_2d(C, ggml_cont(C, xn), c_merged, n_merged); - } else { - const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); - ggml_tensor * mr = ggml_reshape_2d(C, ggml_cont(C, x), c_merged, n_merged); - m = ggml_add(C, ggml_mul(C, ggml_norm(C, mr, ln_eps), w.nw), w.nb); - } - ggml_tensor * z1 = ggml_add(C, ggml_mul_mat(C, w.fc1w, m), w.fc1b); - return ggml_add(C, ggml_mul_mat(C, w.fc2w, ggml_gelu(C, z1)), w.fc2b); -} - ggml_tensor * build_qwen3_layer(ggml_context * C, const VlaJepaModelArch & m, const Qwen3LayerW & w, ggml_tensor * h, ggml_tensor * positions, ggml_tensor * mask, int64_t seq) { const int64_t hd = m.lm_head_dim, n_q = m.n_q, n_kv = m.n_kv, hq = n_q * hd; @@ -198,68 +150,6 @@ ggml_tensor * build_dit_block(ggml_context * C, const VlaJepaModelArch & m, cons return ggml_add(C, h1, ff); } -void merge_block_coords(int64_t gh, int64_t gw, int64_t m, std::vector & row, std::vector & col) { - const int64_t S = gh * gw; row.assign(S, 0); col.assign(S, 0); - for (int64_t s = 0; s < S; ++s) { - int64_t t = s; const int64_t wj = t % m; t /= m; const int64_t wi = t % m; t /= m; - const int64_t bc = t % (gw / m); t /= (gw / m); const int64_t br = t; - row[s] = br * m + wi; col[s] = bc * m + wj; - } -} - -void vit_rope_tables(const std::vector & row, const std::vector & col, int64_t hd, double theta, - std::vector & cos_t, std::vector & sin_t) { - const int64_t S = (int64_t) row.size(), nf = hd / 4; - std::vector invf(nf); - for (int64_t i = 0; i < nf; ++i) invf[i] = 1.0 / std::pow(theta, (double)(2 * i) / (double)(hd / 2)); - cos_t.assign((size_t) S * hd, 0.0f); sin_t.assign((size_t) S * hd, 0.0f); - for (int64_t s = 0; s < S; ++s) { - std::vector emb(hd); - for (int64_t i = 0; i < nf; ++i) { emb[i] = (double) row[s] * invf[i]; emb[nf + i] = (double) col[s] * invf[i]; } - for (int64_t i = 0; i < hd / 2; ++i) emb[hd / 2 + i] = emb[i]; - for (int64_t i = 0; i < hd; ++i) { cos_t[s * hd + i] = (float) std::cos(emb[i]); sin_t[s * hd + i] = (float) std::sin(emb[i]); } - } -} - -void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, - const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, - std::vector & out) { - const int64_t S = (int64_t) row.size(); - out.assign((size_t) S * hidden, 0.0f); - auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side - 1) / (double)(g - 1); }; - for (int64_t s = 0; s < S; ++s) { - const double hy = src_coord(row[s], gh), wx = src_coord(col[s], gw); - const int64_t h0 = (int64_t) std::floor(hy), w0 = (int64_t) std::floor(wx); - const int64_t h1 = std::min(h0 + 1, num_side - 1), w1 = std::min(w0 + 1, num_side - 1); - const double dh = hy - h0, dw = wx - w0; - const double c00 = (1 - dh) * (1 - dw), c01 = (1 - dh) * dw, c10 = dh * (1 - dw), c11 = dh * dw; - const float * T00 = &table[(h0 * num_side + w0) * hidden]; const float * T01 = &table[(h0 * num_side + w1) * hidden]; - const float * T10 = &table[(h1 * num_side + w0) * hidden]; const float * T11 = &table[(h1 * num_side + w1) * hidden]; - for (int64_t c = 0; c < hidden; ++c) out[s * hidden + c] = (float)(c00 * T00[c] + c01 * T01[c] + c10 * T10[c] + c11 * T11[c]); - } -} - -bool preprocess_image_patches(const ImageView & v, int64_t side, int64_t ps, int64_t tps, - const std::vector & row, const std::vector & col, std::vector & out) { - if (v.w != (int) side || v.h != (int) side || !v.data) { - std::fprintf(stderr, "vla(vla_jepa): image view is %dx%d, expected %lldx%lld\n", v.w, v.h, (long long) side, (long long) side); return false; - } - const int64_t S = (int64_t) row.size(), pf = 3 * tps * ps * ps; - out.assign((size_t) pf * S, 0.0f); - auto px = [&](int64_t r, int64_t c, int64_t ch) -> float { - if (v.format == PixelFormat::U8) return ((const uint8_t *) v.data)[(r * side + c) * 3 + ch] / 255.0f; - return ((const float *) v.data)[(r * side + c) * 3 + ch]; - }; - for (int64_t s = 0; s < S; ++s) - for (int64_t ch = 0; ch < 3; ++ch) - for (int64_t ph = 0; ph < ps; ++ph) - for (int64_t pw = 0; pw < ps; ++pw) { - const float val = (px(row[s] * ps + ph, col[s] * ps + pw, ch) - CLIP_MEAN[ch]) / CLIP_STD[ch]; - for (int64_t t = 0; t < tps; ++t) out[s * pf + ch * tps * ps * ps + t * ps * ps + ph * ps + pw] = val; - } - return true; -} - bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { auto U = [&](const char * k, int64_t & dst) { if (g.has(k)) dst = (int64_t) g.u32(k); }; auto F = [&](const char * k, float & dst) { if (g.has(k)) dst = g.f32(k); }; @@ -538,7 +428,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { if (!inj_patches.empty()) { ggml_backend_tensor_set(t_patches, inj_patches.data() + v * n_patches * vit_patch_flat, 0, ggml_nbytes(t_patches)); } else { - if (!preprocess_image_patches(in.images[v], side, ps, temporal_patch, c_grow, c_gcol, patches)) { vok = false; break; } + if (!preprocess_image_patches("vla_jepa", in.images[v], side, ps, temporal_patch, c_grow, c_gcol, patches)) { vok = false; break; } ggml_backend_tensor_set(t_patches, patches.data(), 0, ggml_nbytes(t_patches)); } ggml_backend_tensor_set(t_pos, c_pos_interp.data(), 0, ggml_nbytes(t_pos)); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 99e39c8..226a68b 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -27,6 +27,12 @@ target_link_libraries(test_dit_common PRIVATE ggml) target_compile_options(test_dit_common PRIVATE -Wall -Wextra) add_test(NAME dit_common COMMAND test_dit_common) +add_executable(test_qwen3vl_vit test_qwen3vl_vit.cpp) +target_include_directories(test_qwen3vl_vit PRIVATE ${CMAKE_SOURCE_DIR}/src) +target_link_libraries(test_qwen3vl_vit PRIVATE ggml) +target_compile_options(test_qwen3vl_vit PRIVATE -Wall -Wextra) +add_test(NAME qwen3vl_vit COMMAND test_qwen3vl_vit) + add_executable(test_config_guard test_config_guard.cpp) target_include_directories(test_config_guard PRIVATE ${CMAKE_SOURCE_DIR}/src) target_compile_options(test_config_guard PRIVATE -Wall -Wextra) diff --git a/tests/predict_check.cpp b/tests/predict_check.cpp index 1d71659..504ccbc 100644 --- a/tests/predict_check.cpp +++ b/tests/predict_check.cpp @@ -64,7 +64,12 @@ int main(int argc, char** argv) { views[v] = ImageView{ imgbuf[v].data(), W, H, PixelFormat::U8 }; } + // VLA-JEPA needs its tokens; the others ignore the extras. std::vector lang = {1, 100, 200, 300, 400, 2}; + if (const char* tok = std::getenv("VLA_EXTRA_TOKEN")) { + const char* cnt = std::getenv("VLA_EXTRA_COUNT"); + lang.insert(lang.end(), (size_t)(cnt ? std::atoi(cnt) : 1), (int32_t)std::atoi(tok)); + } std::vector state((size_t)cfg.max_state_dim, 0.0f); for (int i = 0; i < (int)cfg.real_state_dim; ++i) state[i] = 0.01f * (float)(i + 1); diff --git a/tests/test_qwen3vl_vit.cpp b/tests/test_qwen3vl_vit.cpp new file mode 100644 index 0000000..2354f85 --- /dev/null +++ b/tests/test_qwen3vl_vit.cpp @@ -0,0 +1,121 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Pins the Qwen3-VL patch geometry. predict_check covers the graph builders. + +#include "models/qwen3vl_vit.h" + +#include +#include +#include + +using namespace vla; + +static int fails = 0; +#define CHECK(c) do { if (!(c)) { std::printf("FAIL %s:%d %s\n", __FILE__, __LINE__, #c); ++fails; } } while (0) + +// 4x4 grid, 2x2 merge: patches arrive grouped by block, not in raster order. +static void test_merge_block_coords() { + std::vector row, col; + merge_block_coords(4, 4, 2, row, col); + CHECK(row.size() == 16 && col.size() == 16); + const int64_t want_r[16] = {0,0,1,1, 0,0,1,1, 2,2,3,3, 2,2,3,3}; + const int64_t want_c[16] = {0,1,0,1, 2,3,2,3, 0,1,0,1, 2,3,2,3}; + for (int i = 0; i < 16; ++i) { CHECK(row[i] == want_r[i]); CHECK(col[i] == want_c[i]); } +} + +// Half-split layout: the second half of the table repeats the first. +static void test_vit_rope_tables() { + std::vector row = {0, 1}, col = {0, 2}; + std::vector c, s; + const int64_t hd = 8; + vit_rope_tables(row, col, hd, 10000.0, c, s); + CHECK((int64_t) c.size() == 2 * hd && (int64_t) s.size() == 2 * hd); + for (int64_t p = 0; p < 2; ++p) + for (int64_t i = 0; i < hd / 2; ++i) { + CHECK(c[p * hd + i] == c[p * hd + hd / 2 + i]); + CHECK(s[p * hd + i] == s[p * hd + hd / 2 + i]); + } + // Position 0 has zero angle on every frequency. + for (int64_t i = 0; i < hd; ++i) { CHECK(c[i] == 1.0f); CHECK(s[i] == 0.0f); } + // Row 1, col 2 with inv_freq[0] == 1: row half is cos(1), col half is cos(2). + CHECK(std::fabs(c[hd + 0] - std::cos(1.0f)) < 1e-6f); + CHECK(std::fabs(c[hd + hd / 4] - std::cos(2.0f)) < 1e-6f); +} + +// Same grid as the source table is an identity resample. +static void test_interp_pos_embed_identity() { + const int64_t side = 2, hidden = 2; + std::vector table = {0,10, 1,11, 2,12, 3,13}; + std::vector row, col; + merge_block_coords(side, side, 1, row, col); + std::vector out; + interp_pos_embed(table, side, hidden, row, col, side, side, out); + CHECK((int64_t) out.size() == side * side * hidden); + for (size_t s = 0; s < row.size(); ++s) { + const int64_t src = row[s] * side + col[s]; + CHECK(std::fabs(out[s * hidden + 0] - table[src * hidden + 0]) < 1e-6f); + CHECK(std::fabs(out[s * hidden + 1] - table[src * hidden + 1]) < 1e-6f); + } +} + +// Upsampling 2x2 to 3x3 puts the midpoint at the mean of the four corners. +static void test_interp_pos_embed_bilinear() { + const int64_t side = 2, hidden = 1; + std::vector table = {0, 2, 4, 6}; + std::vector row, col; + merge_block_coords(3, 3, 1, row, col); + std::vector out; + interp_pos_embed(table, side, hidden, row, col, 3, 3, out); + for (size_t s = 0; s < row.size(); ++s) + if (row[s] == 1 && col[s] == 1) CHECK(std::fabs(out[s] - 3.0f) < 1e-6f); +} + +// A wrong-sized view must be rejected before any pixel is read. +static void test_preprocess_rejects_bad_view() { + std::vector px(3 * 4 * 4, 0); + std::vector row, col; + merge_block_coords(2, 2, 1, row, col); + std::vector out; + ImageView bad{px.data(), 3, 4, PixelFormat::U8}; + CHECK(!preprocess_image_patches("test", bad, 4, 2, 2, row, col, out)); + ImageView null{nullptr, 4, 4, PixelFormat::U8}; + CHECK(!preprocess_image_patches("test", null, 4, 2, 2, row, col, out)); +} + +// U8 128 maps to ~0 after the 0.5/0.5 normalization, and every temporal slice +// repeats the same value. +static void test_preprocess_values() { + const int64_t side = 2, ps = 1, tps = 2; + std::vector px((size_t) 3 * side * side, 128); + std::vector row, col; + merge_block_coords(side, side, 1, row, col); + std::vector out; + ImageView v{px.data(), (int) side, (int) side, PixelFormat::U8}; + CHECK(preprocess_image_patches("test", v, side, ps, tps, row, col, out)); + const int64_t pf = 3 * tps * ps * ps; + CHECK((int64_t) out.size() == pf * (int64_t) row.size()); + for (float f : out) CHECK(std::fabs(f - (128.0f / 255.0f * 2.0f - 1.0f)) < 1e-6f); +} + +int main() { + test_merge_block_coords(); + test_vit_rope_tables(); + test_interp_pos_embed_identity(); + test_interp_pos_embed_bilinear(); + test_preprocess_rejects_bad_view(); + test_preprocess_values(); + if (fails == 0) std::printf("qwen3vl_vit: all checks passed\n"); + return fails == 0 ? 0 : 1; +} From d751c8dda782b9223da9e82ce2122a45fa613522 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 13:33:18 +0700 Subject: [PATCH 20/42] reuse the compute context and graph allocator across predict calls --- src/models/bitvla.cpp | 47 ++++++++++++------------------ src/models/evo1.cpp | 29 +++++++------------ src/models/gr00tn1d5.cpp | 23 +++++++-------- src/models/gr00tn1d6.cpp | 28 ++++++++---------- src/models/gr00tn1d7.cpp | 9 +++--- src/models/openvla_oft.cpp | 19 ++++++------ src/models/pi0.cpp | 30 +++++++------------ src/models/pi05.cpp | 30 +++++++------------ src/models/scratch_ctx.h | 59 ++++++++++++++++++++++++++++++++++++++ src/models/smolvla.cpp | 22 ++++++-------- src/models/vla_adapter.cpp | 19 ++++++------ src/models/vla_jepa.cpp | 29 ++++++++----------- 12 files changed, 173 insertions(+), 171 deletions(-) create mode 100644 src/models/scratch_ctx.h diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 4d93555..79bd902 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -23,6 +23,7 @@ #endif #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #ifdef VLA_BITVLA_CUDA_KERNELS #include "kernels/bitvla/bitvla_lm_cuda.h" @@ -100,6 +101,10 @@ struct BitvlaModelArch : public ModelArchBase { bool is_cuda = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx proprio_scratch; + scratch_ctx lm_scratch; + scratch_ctx head_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; bool packed_int2 = false; @@ -1039,9 +1044,7 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { #endif { - std::vector meta_buf((size_t) 24 * 1024 * 1024); - ggml_init_params gp = { meta_buf.size(), meta_buf.data(), true }; - ggml_context * ctx = ggml_init(gp); + ggml_context * ctx = vision_scratch.reset((size_t) 24 * 1024 * 1024); ggml_tensor * x_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, patch_flat, N); ggml_set_name(x_in, "patches"); @@ -1057,18 +1060,16 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { ggml_tensor * mm2 = ggml_add(ctx, ggml_mul_mat(ctx, mm_l2_w, mmg), mm_l2_b); ggml_set_name(mm2, "img_embeds"); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); + ggml_cgraph * gf = ggml_new_graph_custom(ctx, 4096, false); ggml_build_forward_expand(gf, mm2); - if (!ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr_alloc_graph failed (view %lld)\n", (long long) v); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (!vision_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr_alloc_graph failed (view %lld)\n", (long long) v); return {}; } ggml_backend_tensor_set(x_in, patches.data(), 0, ggml_nbytes(x_in)); if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): vision graph compute failed (view %lld)\n", (long long) v); - ggml_gallocr_free(galloc); ggml_free(ctx); return {}; + return {}; } ggml_backend_tensor_get(mm2, img_embeds_host.data() + (size_t) v * N * hidden_l, 0, (size_t) N * hidden_l * sizeof(float)); - ggml_gallocr_free(galloc); - ggml_free(ctx); } } stats.ms_vision = std::chrono::duration(clk::now() - t_v0).count(); @@ -1088,9 +1089,7 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { } else #endif { - std::vector meta_buf((size_t) 4 * 1024 * 1024); - ggml_init_params gp = { meta_buf.size(), meta_buf.data(), true }; - ggml_context * ctx = ggml_init(gp); + ggml_context * ctx = proprio_scratch.reset((size_t) 4 * 1024 * 1024); ggml_tensor * x_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, proprio_dim, 1); ggml_set_name(x_in, "state"); ggml_tensor * h1 = ggml_add(ctx, ggml_mul_mat(ctx, pp_fc1_w, x_in), pp_fc1_b); @@ -1098,14 +1097,12 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { ggml_tensor * out = ggml_add(ctx, ggml_mul_mat(ctx, pp_fc2_w, h1_gel), pp_fc2_b); ggml_set_name(out, "proprio_embed"); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); ggml_cgraph * gf = ggml_new_graph(ctx); ggml_build_forward_expand(gf, out); - if (!ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (proprio)\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (!proprio_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (proprio)\n"); return {}; } ggml_backend_tensor_set(x_in, in.state, 0, ggml_nbytes(x_in)); - if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): proprio compute failed\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): proprio compute failed\n"); return {}; } ggml_backend_tensor_get(out, proprio_embed_host.data(), 0, (size_t) hidden_l * sizeof(float)); - ggml_gallocr_free(galloc); ggml_free(ctx); } _dump_bin("proprio_features", proprio_embed_host.data(), proprio_embed_host.size()); @@ -1226,9 +1223,7 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { } else #endif { - std::vector meta_buf((size_t) 64 * 1024 * 1024); - ggml_init_params gp = { meta_buf.size(), meta_buf.data(), true }; - ggml_context * ctx = ggml_init(gp); + ggml_context * ctx = lm_scratch.reset((size_t) 64 * 1024 * 1024); ggml_tensor * x_in = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hidden_l, seq); ggml_set_name(x_in, "inputs_embeds"); ggml_tensor * positions = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, seq); @@ -1246,10 +1241,9 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { ggml_tensor * action_hidden = ggml_get_rows(ctx, h_norm, action_ids); ggml_set_name(action_hidden, "action_hidden"); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); ggml_cgraph * gf = ggml_new_graph_custom(ctx, 32768, false); ggml_build_forward_expand(gf, action_hidden); - if (!ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (lm)\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (!lm_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (lm)\n"); return {}; } ggml_backend_tensor_set(x_in, inputs_embeds.data(), 0, ggml_nbytes(x_in)); std::vector pos_v(seq); @@ -1259,9 +1253,8 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { for (int64_t i = 0; i < n_action; ++i) aids[i] = (int32_t) (seq - 2 - n_action + i); ggml_backend_tensor_set(action_ids, aids.data(), 0, ggml_nbytes(action_ids)); - if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): lm prefill compute failed\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): lm prefill compute failed\n"); return {}; } ggml_backend_tensor_get(action_hidden, last_hidden_at_actions.data(), 0, (size_t) n_action * hidden_l * sizeof(float)); - ggml_gallocr_free(galloc); ggml_free(ctx); } if (timing_phase) stats.ms_prefill = std::chrono::duration(clk::now() - t_p0).count(); @@ -1283,9 +1276,7 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { } else #endif { - std::vector meta_buf((size_t) 8 * 1024 * 1024); - ggml_init_params gp = { meta_buf.size(), meta_buf.data(), true }; - ggml_context * ctx = ggml_init(gp); + ggml_context * ctx = head_scratch.reset((size_t) 8 * 1024 * 1024); ggml_tensor * x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, in_dim, chunk); ggml_set_name(x, "x"); @@ -1306,14 +1297,12 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { ggml_tensor * y = ggml_add(ctx, ggml_mul_mat(ctx, ah_fc2_w, ln2), ah_fc2_b); ggml_set_name(y, "y"); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); ggml_cgraph * gf = ggml_new_graph(ctx); ggml_build_forward_expand(gf, y); - if (!ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (action_head)\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (!head_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (action_head)\n"); return {}; } ggml_backend_tensor_set(x, last_hidden_at_actions.data(), 0, ggml_nbytes(x)); - if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): action_head compute failed\n"); ggml_gallocr_free(galloc); ggml_free(ctx); return {}; } + if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): action_head compute failed\n"); return {}; } ggml_backend_tensor_get(y, normalized_actions.data(), 0, (size_t) chunk * action_dim * sizeof(float)); - ggml_gallocr_free(galloc); ggml_free(ctx); } if (timing_phase) stats.ms_denoise = std::chrono::duration(clk::now() - t_d0).count(); diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 93d68d0..6d4c45e 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include #include @@ -61,6 +62,8 @@ struct Evo1ModelArch : public ModelArchBase { bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_BF16; @@ -427,36 +430,30 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { } n_views = in.n_images; - ggml_init_params vp = { (size_t) 32 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 32 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(evo1): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); ggml_set_input(t_px); ggml_tensor * t_ie = build_internvit_view(VC, *this, t_px); ggml_set_output(t_ie); ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(vg, t_ie); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(evo1): vision ggml_gallocr_alloc_graph failed\n"); - if (vga) ggml_gallocr_free(vga); - ggml_free(VC); return {}; } img_emb_host.assign((size_t) n_views * num_image_token * lm_hidden, 0.0f); std::vector chw; const auto tv0 = std::chrono::steady_clock::now(); for (int64_t v = 0; v < n_views; ++v) { - if (!preprocess_image_chw(in.images[v], image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw(in.images[v], image_size, chw)) { return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(evo1): vision graph compute failed (view %lld)\n", (long long) v); - ggml_gallocr_free(vga); ggml_free(VC); return {}; + return {}; } ggml_backend_tensor_get(t_ie, img_emb_host.data() + v * num_image_token * lm_hidden, 0, ggml_nbytes(t_ie)); } stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); - ggml_gallocr_free(vga); - ggml_free(VC); img_emb_ptr = img_emb_host.data(); } else { std::fprintf(stderr, "vla(evo1): no images and no precomputed_img_emb in the request\n"); @@ -544,8 +541,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { } } - ggml_init_params cp = { (size_t) 96 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = main_scratch.reset((size_t) 96 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(evo1): ggml_init(ctx_compute) failed\n"); return {}; } const int64_t E = embed_dim, hd_dit = E / dit_heads; @@ -633,11 +629,8 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(gf, x_action); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { + if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(evo1): ggml_gallocr_alloc_graph failed\n"); - if (galloc) ggml_gallocr_free(galloc); - ggml_free(C); return {}; } ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); @@ -657,14 +650,12 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { const auto tc1 = std::chrono::steady_clock::now(); if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(evo1): ggml_backend_graph_compute failed (%d)\n", (int) st); - ggml_gallocr_free(galloc); ggml_free(C); return {}; + return {}; } stats.ms_inference = std::chrono::duration(tc1 - tc0).count(); std::vector x_final((size_t) action_dim); ggml_backend_tensor_get(x_action, x_final.data(), 0, x_final.size() * sizeof(float)); - ggml_gallocr_free(galloc); - ggml_free(C); std::vector out((size_t) horizon * per_a); for (int64_t hstep = 0; hstep < horizon; ++hstep) diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 8d82652..50826b4 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/vision_common.h" #include "models/dit_common.h" @@ -59,6 +60,8 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -378,8 +381,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { } else if (in.images && in.n_images > 0) { n_views = in.n_images; img_emb_host.assign((size_t) n_views * K * H, 0.0f); - ggml_init_params vp = { (size_t) 64 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 64 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(gr00tn1d5): ggml_init(vision ctx) failed\n"); return {}; } const int64_t grid = image_size / patch_size; ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); ggml_set_input(t_px); @@ -392,18 +394,16 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_set_output(vit_emb); ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(vg, vit_emb); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { std::fprintf(stderr, "vla(gr00tn1d5): vision gallocr alloc failed\n"); if (vga) ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(gr00tn1d5): vision gallocr alloc failed\n"); return {}; } const auto tv0 = std::chrono::steady_clock::now(); std::vector chw; for (int64_t v = 0; v < n_views; ++v) { - if (!preprocess_image_chw("gr00tn1d5", in.images[v], image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("gr00tn1d5", in.images[v], image_size, chw)) { return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); - if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d5): vision compute failed\n"); ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d5): vision compute failed\n"); return {}; } ggml_backend_tensor_get(vit_emb, img_emb_host.data() + v * K * H, 0, ggml_nbytes(vit_emb)); } stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); - ggml_gallocr_free(vga); ggml_free(VC); img_emb_ptr = img_emb_host.data(); } else { std::fprintf(stderr, "vla(gr00tn1d5): no images and no precomputed_img_emb in the request\n"); return {}; @@ -442,8 +442,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { if (in.noise) std::memcpy(x_init.data(), in.noise, x_init.size() * sizeof(float)); else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } - ggml_init_params cp = { (size_t) 128 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = main_scratch.reset((size_t) 128 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(gr00tn1d5): ggml_init(ctx_compute) failed\n"); return {}; } ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); @@ -506,8 +505,7 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(gf, actions); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(gr00tn1d5): gallocr alloc failed\n"); if (galloc) ggml_gallocr_free(galloc); ggml_free(C); return {}; } + if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(gr00tn1d5): gallocr alloc failed\n"); return {}; } ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { std::vector pp(SEQ); for (int64_t i = 0; i < SEQ; ++i) pp[i] = (int32_t) i; ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } @@ -526,12 +524,11 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { const auto tc0 = std::chrono::steady_clock::now(); const ggml_status st = ggml_backend_graph_compute(backend, gf); const auto tc1 = std::chrono::steady_clock::now(); - if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d5): graph compute failed (%d)\n", (int) st); ggml_gallocr_free(galloc); ggml_free(C); return {}; } + if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d5): graph compute failed (%d)\n", (int) st); return {}; } stats.ms_inference = std::chrono::duration(tc1 - tc0).count(); std::vector out((size_t) AH * AD); ggml_backend_tensor_get(actions, out.data(), 0, out.size() * sizeof(float)); - ggml_gallocr_free(galloc); ggml_free(C); stats.ms_total = std::chrono::duration(std::chrono::steady_clock::now() - t0).count(); return out; } diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 0285c27..79836a0 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/dit_common.h" #include @@ -57,6 +58,9 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx merge_scratch; + scratch_ctx main_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -396,8 +400,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { n_views = in.n_images; img_emb_host.assign((size_t) n_views * K * H, 0.0f); - ggml_init_params vp = { (size_t) 64 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 64 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(vision ctx A) failed\n"); return {}; } ggml_tensor * t_patches = ggml_new_tensor_3d(VC, GGML_TYPE_F32, patch_dim, n_patches, n_views); ggml_set_input(t_patches); ggml_tensor * h = ggml_add(VC, ggml_add(VC, ggml_mul_mat(VC, vit_patch_w, t_patches), vit_patch_b), vit_pos); @@ -406,12 +409,10 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { ggml_set_output(post_ln); ggml_cgraph * vgA = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(vgA, post_ln); - ggml_gallocr_t vgaA = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vgaA || !ggml_gallocr_alloc_graph(vgaA, vgA)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr A alloc failed\n"); if (vgaA) ggml_gallocr_free(vgaA); ggml_free(VC); return {}; } + if (!vision_scratch.alloc(backend, vgA)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr A alloc failed\n"); return {}; } - ggml_init_params mp = { (size_t) 16 * 1024 * 1024, nullptr, true }; - ggml_context * MC = ggml_init(mp); - if (!MC) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(vision ctx B) failed\n"); ggml_gallocr_free(vgaA); ggml_free(VC); return {}; } + ggml_context * MC = merge_scratch.reset((size_t) 16 * 1024 * 1024); + if (!MC) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(vision ctx B) failed\n"); return {}; } ggml_tensor * t_shuf = ggml_new_tensor_3d(MC, GGML_TYPE_F32, c4, K, n_views); ggml_set_input(t_shuf); ggml_tensor * mln = ggml_add(MC, ggml_mul(MC, ggml_norm(MC, t_shuf, connector_ln_eps), mm_ln_w), mm_ln_b); ggml_tensor * mz1 = ggml_add(MC, ggml_mul_mat(MC, mm_fc1_w, mln), mm_fc1_b); @@ -419,8 +420,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { ggml_set_output(vit_embeds); ggml_cgraph * vgB = ggml_new_graph(MC); ggml_build_forward_expand(vgB, vit_embeds); - ggml_gallocr_t vgaB = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vgaB || !ggml_gallocr_alloc_graph(vgaB, vgB)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr B alloc failed\n"); if (vgaB) ggml_gallocr_free(vgaB); ggml_gallocr_free(vgaA); ggml_free(MC); ggml_free(VC); return {}; } + if (!merge_scratch.alloc(backend, vgB)) { std::fprintf(stderr, "vla(gr00tn1d6): vision gallocr B alloc failed\n"); return {}; } const auto tv0 = std::chrono::steady_clock::now(); std::vector patches, @@ -445,7 +445,6 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { } if (vok) ggml_backend_tensor_get(vit_embeds, img_emb_host.data(), 0, ggml_nbytes(vit_embeds)); stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); - ggml_gallocr_free(vgaB); ggml_gallocr_free(vgaA); ggml_free(MC); ggml_free(VC); if (!vok) return {}; img_emb_ptr = img_emb_host.data(); } else { @@ -495,8 +494,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { if (in.noise) std::memcpy(x_init.data(), in.noise, x_init.size() * sizeof(float)); else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } - ggml_init_params cp = { (size_t) 256 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = main_scratch.reset((size_t) 256 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(ctx_compute) failed\n"); return {}; } ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); @@ -569,8 +567,7 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); ggml_build_forward_expand(gf, actions); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(gr00tn1d6): gallocr alloc failed\n"); if (galloc) ggml_gallocr_free(galloc); ggml_free(C); return {}; } + if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(gr00tn1d6): gallocr alloc failed\n"); return {}; } ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { std::vector pp(SEQ); for (int64_t i = 0; i < SEQ; ++i) pp[i] = (int32_t) i; ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } @@ -591,12 +588,11 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { const auto tc0 = std::chrono::steady_clock::now(); const ggml_status st = ggml_backend_graph_compute(backend, gf); const auto tc1 = std::chrono::steady_clock::now(); - if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d6): graph compute failed (%d)\n", (int) st); ggml_gallocr_free(galloc); ggml_free(C); return {}; } + if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(gr00tn1d6): graph compute failed (%d)\n", (int) st); return {}; } stats.ms_inference = std::chrono::duration(tc1 - tc0).count(); std::vector out((size_t) AH * AD); ggml_backend_tensor_get(actions, out.data(), 0, out.size() * sizeof(float)); - ggml_gallocr_free(galloc); ggml_free(C); stats.ms_total = std::chrono::duration(std::chrono::steady_clock::now() - t0).count(); return out; } diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index 13a3dae..afd2cb1 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/dit_common.h" #include "models/qwen3vl_vit.h" @@ -58,6 +59,7 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -525,8 +527,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { img_emb_host.assign((size_t) n_views * K * H, 0.0f); for (int j = 0; j < 3; ++j) ds_host[j].assign((size_t) n_views * K * H, 0.0f); - ggml_init_params vp = { (size_t) 512 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 512 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(gr00tn1d7): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); @@ -548,8 +549,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { ggml_cgraph * vg = ggml_new_graph_custom(VC, 16384, false); ggml_build_forward_expand(vg, vit_embeds); for (int j = 0; j < 3; ++j) ggml_build_forward_expand(vg, ds_out[j]); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { std::fprintf(stderr, "vla(gr00tn1d7): vision gallocr alloc failed\n"); if (vga) ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(gr00tn1d7): vision gallocr alloc failed\n"); return {}; } const auto tv0 = std::chrono::steady_clock::now(); std::vector patches; bool vok = true; @@ -565,7 +565,6 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { for (int j = 0; j < 3; ++j) ggml_backend_tensor_get(ds_out[j], ds_host[j].data() + v * K * H, 0, ggml_nbytes(ds_out[j])); } stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); - ggml_gallocr_free(vga); ggml_free(VC); if (!vok) return {}; img_emb_ptr = img_emb_host.data(); } else { diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 89002b7..363b73e 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -23,6 +23,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include #include @@ -92,6 +93,8 @@ struct OpenVlaOftModelArch : public ModelArchBase { ggml_backend_t backend = nullptr; bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; @@ -259,7 +262,7 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { std::vector proj_host((size_t)HC*NPATCH); { const auto tv=clock::now(); - ggml_init_params vp={(size_t)64*1024*1024,nullptr,true}; ggml_context*C=ggml_init(vp); + ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); std::vector px_d(n_views), px_s(n_views), cmb(n_views); for(int v=0; v OpenVlaOftModelArch::predict(const Inputs& in) { ph=ggml_add(C,ggml_mul_mat(C,pj_fc2w,ph),pj_fc2b); ph=ggml_gelu_erf(C,ph); ggml_tensor*proj=ggml_add(C,ggml_mul_mat(C,pj_fc3w,ph),pj_fc3b); ggml_set_output(proj); ggml_cgraph*vg=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(vg,proj); - ggml_gallocr_t ga=ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if(!ga||!ggml_gallocr_alloc_graph(ga,vg)){ std::fprintf(stderr,"vla(openvla_oft): vision gallocr failed\n"); if(ga)ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(!vision_scratch.alloc(backend,vg)){ std::fprintf(stderr,"vla(openvla_oft): vision gallocr failed\n"); return {}; } std::vector dbuf, sbuf; for(int v=0;v(clock::now()-tv).count(); } @@ -303,7 +304,7 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { const int64_t ACT_START = NUM_PATCHES + NUM_PROMPT_TOKENS; const int64_t SEQ = 1 + NUM_PATCHES + (L-1) + n_act + 1; const auto ti=clock::now(); - ggml_init_params mp={(size_t)256*1024*1024,nullptr,true}; ggml_context*C=ggml_init(mp); + ggml_context*C=main_scratch.reset((size_t)256*1024*1024); ggml_tensor*t_ids=ggml_new_tensor_1d(C,GGML_TYPE_I32,L+1); ggml_set_input(t_ids); ggml_tensor*emb=ggml_get_rows(C,token_embd,t_ids); @@ -363,8 +364,7 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { ggml_tensor*norm_actions=ggml_add(C,ggml_mul_mat(C,h_fc2w,hh),h_fc2b); ggml_set_output(norm_actions); ggml_cgraph*gf=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(gf,norm_actions); - ggml_gallocr_t ga=ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if(!ga||!ggml_gallocr_alloc_graph(ga,gf)){ std::fprintf(stderr,"vla(openvla_oft): main gallocr failed\n"); if(ga)ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(!main_scratch.alloc(backend,gf)){ std::fprintf(stderr,"vla(openvla_oft): main gallocr failed\n"); return {}; } { std::vector ids(L+1); for(int64_t i=0;i OpenVlaOftModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_state,sv.data(),0,ggml_nbytes(t_state)); } { std::vector z((size_t)HC*n_act,0.0f); ggml_backend_tensor_set(act0,z.data(),0,ggml_nbytes(act0)); } - if(ggml_backend_graph_compute(backend,gf)!=GGML_STATUS_SUCCESS){ std::fprintf(stderr,"vla(openvla_oft): main compute failed\n"); ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(ggml_backend_graph_compute(backend,gf)!=GGML_STATUS_SUCCESS){ std::fprintf(stderr,"vla(openvla_oft): main compute failed\n"); return {}; } std::vector na((size_t)action_dim*chunk); ggml_backend_tensor_get(norm_actions,na.data(),0,na.size()*sizeof(float)); - ggml_gallocr_free(ga); ggml_free(C); stats.ms_inference = std::chrono::duration(clock::now()-ti).count(); const int64_t Wd = cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim; diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 8e7c1ca..9749902 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -22,6 +22,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/vision_common.h" #include @@ -95,6 +96,8 @@ struct Pi0ModelArch : public ModelArchBase { bool is_gpu = false; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi0"}; @@ -499,8 +502,7 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { n_img_tokens = (int64_t) in.n_images * K; img_emb_host.assign((size_t) in.n_images * K * H, 0.0f); - ggml_init_params vp = { (size_t) 128 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 128 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(pi0): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vit_image_size, vit_image_size, 3); ggml_set_input(t_px); ggml_tensor * conv = ggml_conv_2d(VC, vit_patch_w, t_px, (int) vit_patch_size, (int) vit_patch_size, 0, 0, 1, 1); @@ -517,26 +519,23 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(vg, vit_emb); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { + + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(pi0): vision gallocr alloc failed\n"); - if (vga) ggml_gallocr_free(vga); - ggml_free(VC); return {}; } const auto tv0 = clk::now(); std::vector chw; for (int v = 0; v < in.n_images; ++v) { - if (!preprocess_image_chw("pi0", in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("pi0", in.images[v], vit_image_size, chw)) { return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi0): vision compute failed (view %d)\n", v); - ggml_gallocr_free(vga); ggml_free(VC); return {}; + return {}; } ggml_backend_tensor_get(vit_emb, img_emb_host.data() + (size_t) v * K * H, 0, ggml_nbytes(vit_emb)); } stats.ms_vision = std::chrono::duration(clk::now() - tv0).count(); - ggml_gallocr_free(vga); ggml_free(VC); } if (in.n_lang < 1 || !in.lang_tokens) { @@ -553,8 +552,7 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } - ggml_init_params cp = { (size_t) 64 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = main_scratch.reset((size_t) 64 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(pi0): ggml_init(ctx_compute) failed\n"); return {}; } ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); @@ -606,11 +604,9 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); ggml_build_forward_expand(gf, x_final); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { + + if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(pi0): ggml_gallocr_alloc_graph failed (out of memory?)\n"); - if (galloc) ggml_gallocr_free(galloc); - ggml_free(C); return {}; } @@ -661,15 +657,11 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { stats.ms_inference = std::chrono::duration(clk::now() - ti0).count(); if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi0): ggml_backend_graph_compute failed (%d)\n", (int) st); - ggml_gallocr_free(galloc); - ggml_free(C); return {}; } std::vector out((size_t) chunk * max_ad); ggml_backend_tensor_get(x_final, out.data(), 0, out.size() * sizeof(float)); - ggml_gallocr_free(galloc); - ggml_free(C); for (int64_t t = 0; t < chunk; ++t) { float * row = out.data() + (size_t) t * max_ad; for (int64_t j = 0; j < max_ad; ++j) diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 376461d..30918fb 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -22,6 +22,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/vision_common.h" #include @@ -110,6 +111,8 @@ struct Pi05ModelArch : public ModelArchBase { bool is_gpu = false; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi05"}; @@ -595,8 +598,7 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { n_img_tokens = (int64_t) in.n_images * K; img_emb_host.assign((size_t) in.n_images * K * H, 0.0f); - ggml_init_params vp = { (size_t) 128 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 128 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(pi05): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vit_image_size, vit_image_size, 3); ggml_set_input(t_px); ggml_tensor * conv = ggml_conv_2d(VC, vit_patch_w, t_px, (int) vit_patch_size, (int) vit_patch_size, 0, 0, 1, 1); @@ -613,26 +615,23 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(vg, vit_emb); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { + + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(pi05): vision gallocr alloc failed\n"); - if (vga) ggml_gallocr_free(vga); - ggml_free(VC); return {}; } const auto tv0 = clk::now(); std::vector chw; for (int v = 0; v < in.n_images; ++v) { - if (!preprocess_image_chw("pi05", in.images[v], vit_image_size, chw)) { ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!preprocess_image_chw("pi05", in.images[v], vit_image_size, chw)) { return {}; } ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi05): vision compute failed (view %d)\n", v); - ggml_gallocr_free(vga); ggml_free(VC); return {}; + return {}; } ggml_backend_tensor_get(vit_emb, img_emb_host.data() + (size_t) v * K * H, 0, ggml_nbytes(vit_emb)); } stats.ms_vision = std::chrono::duration(clk::now() - tv0).count(); - ggml_gallocr_free(vga); ggml_free(VC); // Undo the 1/sqrt(hidden) the shared vision graph applies; pi05 wants raw // projector features. Inside this branch on purpose: precomputed_img_emb @@ -654,8 +653,7 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } - ggml_init_params cp = { (size_t) 64 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = main_scratch.reset((size_t) 64 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(pi05): ggml_init(ctx_compute) failed\n"); return {}; } ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); @@ -705,11 +703,9 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); ggml_build_forward_expand(gf, x_final); - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { + + if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(pi05): ggml_gallocr_alloc_graph failed (out of memory?)\n"); - if (galloc) ggml_gallocr_free(galloc); - ggml_free(C); return {}; } @@ -738,15 +734,11 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { stats.ms_inference = std::chrono::duration(clk::now() - ti0).count(); if (st != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(pi05): ggml_backend_graph_compute failed (%d)\n", (int) st); - ggml_gallocr_free(galloc); - ggml_free(C); return {}; } std::vector out((size_t) chunk * max_ad); ggml_backend_tensor_get(x_final, out.data(), 0, out.size() * sizeof(float)); - ggml_gallocr_free(galloc); - ggml_free(C); if (!std::getenv("VLA_PI05_SKIP_UNNORM")) { for (int64_t t = 0; t < chunk; ++t) { diff --git a/src/models/scratch_ctx.h b/src/models/scratch_ctx.h new file mode 100644 index 0000000..b2371c1 --- /dev/null +++ b/src/models/scratch_ctx.h @@ -0,0 +1,59 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Compute context and graph allocator reused across predict calls; rebuilding +// them costs 2-4 ms on the larger graphs. One scratch per graph role, and +// tensors die at the next reset. + +#pragma once + +#include "ggml.h" +#include "ggml-alloc.h" +#include "ggml-backend.h" + +#include + +namespace vla { + +class scratch_ctx { +public: + scratch_ctx() = default; + scratch_ctx(const scratch_ctx &) = delete; + scratch_ctx & operator=(const scratch_ctx &) = delete; + ~scratch_ctx() { release(); } + + // Empty context sized for `arena` bytes of tensor headers, or null on failure. + ggml_context * reset(size_t arena) { + if (ctx_) { ggml_reset(ctx_); return ctx_; } + ggml_init_params p = { arena, nullptr, true }; + ctx_ = ggml_init(p); + return ctx_; + } + + bool alloc(ggml_backend_t backend, ggml_cgraph * gf) { + if (!galloc_) galloc_ = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); + return galloc_ && ggml_gallocr_alloc_graph(galloc_, gf); + } + + void release() { + if (galloc_) { ggml_gallocr_free(galloc_); galloc_ = nullptr; } + if (ctx_) { ggml_free(ctx_); ctx_ = nullptr; } + } + +private: + ggml_context * ctx_ = nullptr; + ggml_gallocr_t galloc_ = nullptr; +}; + +} // namespace vla diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 64e0d11..bd06d00 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -18,6 +18,7 @@ #include "arch.h" #include "model.h" #include "vision_common.h" +#include "scratch_ctx.h" #include "ggml.h" #include "ggml-backend.h" @@ -291,6 +292,8 @@ struct SmolVLAModelArch : public ModelArchBase { ggml_type weight_dtype = GGML_TYPE_BF16; ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx connector_scratch; ggml_tensor * E_lang = nullptr; ggml_tensor * Wstate = nullptr; @@ -1436,8 +1439,7 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { const auto t_vision_begin = clock::now(); // Graph A: SigLIP ViT (conv patch-embed -> +pos -> layers -> post_ln), plain sequential positions. - ggml_init_params vpA = { size_t(256) * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vpA); + ggml_context * VC = m->vision_scratch.reset(size_t(256) * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(smolvla): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, m->vit_image, m->vit_image, 3); ggml_set_input(t_px); ggml_tensor * conv = ggml_conv_2d(VC, m->vit_patch_w, t_px, (int) m->vit_patch, (int) m->vit_patch, 0, 0, 1, 1); @@ -1449,28 +1451,21 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { ggml_set_output(post_ln); ggml_cgraph * gA = ggml_new_graph_custom(VC, 8192, false); ggml_build_forward_expand(gA, post_ln); - ggml_gallocr_t vgA = ggml_gallocr_new(ggml_backend_get_default_buffer_type(m->backend)); - if (!vgA || !ggml_gallocr_alloc_graph(vgA, gA)) { + if (!m->vision_scratch.alloc(m->backend, gA)) { std::fprintf(stderr, "vla(smolvla): vision gallocr A alloc failed\n"); - if (vgA) ggml_gallocr_free(vgA); - ggml_free(VC); return {}; } // Graph B: pixel-shuffle connector, a single bias-free matmul (c4 -> hidden). - ggml_init_params vpB = { size_t(64) * 1024 * 1024, nullptr, true }; - ggml_context * MC = ggml_init(vpB); - if (!MC) { std::fprintf(stderr, "vla(smolvla): ggml_init(connector ctx) failed\n"); ggml_gallocr_free(vgA); ggml_free(VC); return {}; } + ggml_context * MC = m->connector_scratch.reset(size_t(64) * 1024 * 1024); + if (!MC) { std::fprintf(stderr, "vla(smolvla): ggml_init(connector ctx) failed\n"); return {}; } ggml_tensor * t_shuf = ggml_new_tensor_2d(MC, GGML_TYPE_F32, c4, K); ggml_set_input(t_shuf); ggml_tensor * img_embeds = ggml_mul_mat(MC, m->mm_fc, t_shuf); ggml_set_output(img_embeds); ggml_cgraph * gB = ggml_new_graph(MC); ggml_build_forward_expand(gB, img_embeds); - ggml_gallocr_t vgB = ggml_gallocr_new(ggml_backend_get_default_buffer_type(m->backend)); - if (!vgB || !ggml_gallocr_alloc_graph(vgB, gB)) { + if (!m->connector_scratch.alloc(m->backend, gB)) { std::fprintf(stderr, "vla(smolvla): vision gallocr B alloc failed\n"); - if (vgB) ggml_gallocr_free(vgB); - ggml_gallocr_free(vgA); ggml_free(MC); ggml_free(VC); return {}; } @@ -1490,7 +1485,6 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { } ggml_backend_tensor_get(img_embeds, img_emb_pre.data() + size_t(v) * per_view_n, 0, ggml_nbytes(img_embeds)); } - ggml_gallocr_free(vgB); ggml_free(MC); ggml_gallocr_free(vgA); ggml_free(VC); if (!vok) return {}; m->stats.ms_vision = std::chrono::duration(clock::now() - t_vision_begin).count(); } diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 20c5fb7..e3884d7 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -23,6 +23,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include #include @@ -93,6 +94,8 @@ struct VlaAdapterModelArch : public ModelArchBase { ggml_backend_t backend = nullptr; bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx main_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; @@ -289,7 +292,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { std::vector proj_host((size_t)HC*NP*n_views); { const auto tv=clock::now(); - ggml_init_params vp={(size_t)64*1024*1024,nullptr,true}; ggml_context*C=ggml_init(vp); + ggml_context*C=vision_scratch.reset((size_t)64*1024*1024); std::vector px_d(n_views), px_s(n_views); std::vector cmb(n_views); for(int v=0; v VlaAdapterModelArch::predict(const Inputs& in) { ph=ggml_add(C,ggml_mul_mat(C,pj_fc2w,ph),pj_fc2b); ph=ggml_gelu_erf(C,ph); ggml_tensor*proj=ggml_add(C,ggml_mul_mat(C,pj_fc3w,ph),pj_fc3b); ggml_set_output(proj); ggml_cgraph*vg=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(vg,proj); - ggml_gallocr_t ga=ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if(!ga||!ggml_gallocr_alloc_graph(ga,vg)){ std::fprintf(stderr,"vla(vla_adapter): vision gallocr failed\n"); if(ga)ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(!vision_scratch.alloc(backend,vg)){ std::fprintf(stderr,"vla(vla_adapter): vision gallocr failed\n"); return {}; } std::vector dbuf, sbuf; for(int v=0;v(clock::now()-tv).count(); } @@ -332,7 +333,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { const int64_t NPATCH = NP * n_views; const int64_t SEQ = 1 + NPATCH + (NPROMPT-1) + num_tokens + 1; const auto ti=clock::now(); - ggml_init_params mp={(size_t)128*1024*1024,nullptr,true}; ggml_context*C=ggml_init(mp); + ggml_context*C=main_scratch.reset((size_t)128*1024*1024); ggml_tensor*t_ids=ggml_new_tensor_1d(C,GGML_TYPE_I32,NPROMPT+num_tokens+1); ggml_set_input(t_ids); ggml_tensor*emb=ggml_get_rows(C,token_embd,t_ids); @@ -418,8 +419,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*norm_actions=ggml_add(C,ggml_mul_mat(C,h_fc2w,xn),h_fc2b); ggml_set_output(norm_actions); ggml_cgraph*gf=ggml_new_graph_custom(C,65536,false); ggml_build_forward_expand(gf,norm_actions); - ggml_gallocr_t ga=ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if(!ga||!ggml_gallocr_alloc_graph(ga,gf)){ std::fprintf(stderr,"vla(vla_adapter): main gallocr failed\n"); if(ga)ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(!main_scratch.alloc(backend,gf)){ std::fprintf(stderr,"vla(vla_adapter): main gallocr failed\n"); return {}; } { std::vector ids(NPROMPT+num_tokens+1); for(int64_t i=0;i VlaAdapterModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(cc,cb.data(),0,ggml_nbytes(cc)); ggml_backend_tensor_set(ss,sb.data(),0,ggml_nbytes(ss)); }; fill_cs(cT,sT,chunk); fill_cs(cA,sA,num_tokens+1); fill_cs(cK,sK,NPATCH); - if(ggml_backend_graph_compute(backend,gf)!=GGML_STATUS_SUCCESS){ std::fprintf(stderr,"vla(vla_adapter): main compute failed\n"); ggml_gallocr_free(ga); ggml_free(C); return {}; } + if(ggml_backend_graph_compute(backend,gf)!=GGML_STATUS_SUCCESS){ std::fprintf(stderr,"vla(vla_adapter): main compute failed\n"); return {}; } std::vector na((size_t)action_dim*chunk); ggml_backend_tensor_get(norm_actions,na.data(),0,na.size()*sizeof(float)); - ggml_gallocr_free(ga); ggml_free(C); stats.ms_inference = std::chrono::duration(clock::now()-ti).count(); const int64_t W = cfg.max_action_dim>0 ? cfg.max_action_dim : action_dim; diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 6f093f0..dfa81b0 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -21,6 +21,7 @@ #include "backend.h" #include "gguf.h" #include "models/gguf_reader.h" +#include "models/scratch_ctx.h" #include "models/dit_common.h" #include "models/qwen3vl_vit.h" @@ -56,6 +57,9 @@ struct VlaJepaModelArch : public ModelArchBase { bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; + scratch_ctx vision_scratch; + scratch_ctx lm_scratch; + scratch_ctx head_scratch; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -397,8 +401,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { } if (inj_patches.empty() && !in.images) { std::fprintf(stderr, "vla(vla_jepa): n_images=%d but the images pointer is null\n", in.n_images); return {}; } - ggml_init_params vp = { (size_t) 512 * 1024 * 1024, nullptr, true }; - ggml_context * VC = ggml_init(vp); + ggml_context * VC = vision_scratch.reset((size_t) 512 * 1024 * 1024); if (!VC) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(vision ctx) failed\n"); return {}; } ggml_tensor * t_patches = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_patch_flat, n_patches); ggml_set_input(t_patches); ggml_tensor * t_pos = ggml_new_tensor_2d(VC, GGML_TYPE_F32, vit_hidden, n_patches); ggml_set_input(t_pos); @@ -419,8 +422,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_cgraph * vg = ggml_new_graph_custom(VC, 16384, false); ggml_build_forward_expand(vg, vit_embeds); for (int j = 0; j < 3; ++j) ggml_build_forward_expand(vg, ds_out[j]); - ggml_gallocr_t vga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!vga || !ggml_gallocr_alloc_graph(vga, vg)) { std::fprintf(stderr, "vla(vla_jepa): vision gallocr alloc failed\n"); if (vga) ggml_gallocr_free(vga); ggml_free(VC); return {}; } + if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(vla_jepa): vision gallocr alloc failed\n"); return {}; } const auto tv0 = std::chrono::steady_clock::now(); std::vector patches; bool vok = true; @@ -440,7 +442,6 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { if (dump_prefix) { char nm[32]; std::snprintf(nm, sizeof(nm), "vit_view%lld", (long long) v); char path[1024]; std::snprintf(path, sizeof(path), "%s_%s_%lldx%lld.f32", dump_prefix, nm, (long long) H, (long long) K); FILE * fp = std::fopen(path, "wb"); if (fp) { std::fwrite(img_emb_host.data() + v * K * H, sizeof(float), (size_t) K * H, fp); std::fclose(fp); } } } stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); - ggml_gallocr_free(vga); ggml_free(VC); if (!vok) return {}; const int64_t n_img = n_views * K; @@ -472,8 +473,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { std::vector> ds_pad(3); for (int j = 0; j < 3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k = 0; k < n_img; ++k) std::memcpy(ds_pad[j].data() + (size_t) image_pos_idx[k] * H, ds_host[j].data() + (size_t) k * H, H * sizeof(float)); } - ggml_init_params cp = { (size_t) 512 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); + ggml_context * C = lm_scratch.reset((size_t) 512 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(LM ctx) failed\n"); return {}; } ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos2 = ggml_new_tensor_1d(C, GGML_TYPE_I32, 4 * SEQ); ggml_set_input(t_pos2); @@ -492,8 +492,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_set_output(conditioning); ggml_cgraph * lg = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(lg, conditioning); - ggml_gallocr_t lga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!lga || !ggml_gallocr_alloc_graph(lga, lg)) { std::fprintf(stderr, "vla(vla_jepa): LM gallocr alloc failed\n"); if (lga) ggml_gallocr_free(lga); ggml_free(C); return {}; } + if (!lm_scratch.alloc(backend, lg)) { std::fprintf(stderr, "vla(vla_jepa): LM gallocr alloc failed\n"); return {}; } ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); @@ -528,15 +527,13 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { for (int j = 0; j < 3; ++j) ggml_backend_tensor_set(t_ds[j], ds_pad[j].data(), 0, ggml_nbytes(t_ds[j])); const auto tp0 = std::chrono::steady_clock::now(); - if (ggml_backend_graph_compute(backend, lg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(vla_jepa): LM compute failed\n"); ggml_gallocr_free(lga); ggml_free(C); return {}; } + if (ggml_backend_graph_compute(backend, lg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(vla_jepa): LM compute failed\n"); return {}; } stats.ms_prefill = std::chrono::duration(std::chrono::steady_clock::now() - tp0).count(); if (dump_prefix) { dump_t("eagle", eagle); dump_t("conditioning", conditioning); } ggml_backend_tensor_get(conditioning, cond_host.data(), 0, cond_host.size() * sizeof(float)); - ggml_gallocr_free(lga); ggml_free(C); } - ggml_init_params hp = { (size_t) 256 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(hp); + ggml_context * C = head_scratch.reset((size_t) 256 * 1024 * 1024); if (!C) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(head ctx) failed\n"); return {}; } ggml_tensor * t_cond = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, num_future); ggml_set_input(t_cond); ggml_tensor * t_state = ggml_new_tensor_2d(C, GGML_TYPE_F32, state_dim, 1); ggml_set_input(t_state); @@ -585,8 +582,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_cgraph * hg = ggml_new_graph_custom(C, 65536, false); ggml_build_forward_expand(hg, actions); if (dump_prefix) for (int64_t s = 0; s < num_steps; ++s) { ggml_build_forward_expand(hg, step_seq[s]); ggml_build_forward_expand(hg, step_pred[s]); ggml_build_forward_expand(hg, step_vel[s]); ggml_build_forward_expand(hg, step_act[s]); } - ggml_gallocr_t hga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!hga || !ggml_gallocr_alloc_graph(hga, hg)) { std::fprintf(stderr, "vla(vla_jepa): head gallocr alloc failed\n"); if (hga) ggml_gallocr_free(hga); ggml_free(C); return {}; } + if (!head_scratch.alloc(backend, hg)) { std::fprintf(stderr, "vla(vla_jepa): head gallocr alloc failed\n"); return {}; } ggml_backend_tensor_set(t_cond, cond_host.data(), 0, ggml_nbytes(t_cond)); { std::vector st(state_dim, 0.0f); for (int64_t i = 0; i < state_dim; ++i) st[i] = in.state ? in.state[i] : 0.0f; ggml_backend_tensor_set(t_state, st.data(), 0, ggml_nbytes(t_state)); } @@ -594,7 +590,7 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { for (int64_t s = 0; s < num_steps; ++s) { ggml_backend_tensor_set(t_tau[s], c_tau[(size_t) s].data(), 0, ggml_nbytes(t_tau[s])); ggml_backend_tensor_set(t_tproj[s], c_tproj[(size_t) s].data(), 0, ggml_nbytes(t_tproj[s])); } const auto td0 = std::chrono::steady_clock::now(); - if (ggml_backend_graph_compute(backend, hg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(vla_jepa): head compute failed\n"); ggml_gallocr_free(hga); ggml_free(C); return {}; } + if (ggml_backend_graph_compute(backend, hg) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(vla_jepa): head compute failed\n"); return {}; } stats.ms_denoise = std::chrono::duration(std::chrono::steady_clock::now() - td0).count(); stats.ms_inference = stats.ms_prefill + stats.ms_denoise; @@ -608,7 +604,6 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { std::vector out((size_t) AH * AD); ggml_backend_tensor_get(actions, out.data(), 0, out.size() * sizeof(float)); - ggml_gallocr_free(hga); ggml_free(C); stats.ms_total = std::chrono::duration(std::chrono::steady_clock::now() - t0).count(); return out; } From e8aaaa01ee68f01c5ce32d2664e1ffad826f24de Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 13:48:07 +0700 Subject: [PATCH 21/42] add -hf model fetch, vla-bench, release workflow and contributor docs --- .github/ISSUE_TEMPLATE/bug_report.md | 26 ++++ .github/ISSUE_TEMPLATE/model_request.md | 20 +++ .github/pull_request_template.md | 12 ++ .github/workflows/build.yml | 2 +- .github/workflows/release.yml | 132 ++++++++++++++++++++ CMakeLists.txt | 8 +- CONTRIBUTING.md | 83 +++++++++++++ src/model.h | 5 +- src/models/qwen3vl_vit.h | 14 +-- src/models/scratch_ctx.h | 1 - src/serving/hf_fetch.h | 113 +++++++++++++++++ src/serving/server.cpp | 15 ++- src/serving/vla-bench.cpp | 159 ++++++++++++++++++++++++ src/serving/vla-cli.cpp | 12 +- tests/predict_check.cpp | 7 +- tests/test_qwen3vl_vit.cpp | 15 +-- 16 files changed, 595 insertions(+), 29 deletions(-) create mode 100644 .github/ISSUE_TEMPLATE/bug_report.md create mode 100644 .github/ISSUE_TEMPLATE/model_request.md create mode 100644 .github/pull_request_template.md create mode 100644 .github/workflows/release.yml create mode 100644 CONTRIBUTING.md create mode 100644 src/serving/hf_fetch.h create mode 100644 src/serving/vla-bench.cpp diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md new file mode 100644 index 0000000..3614161 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/bug_report.md @@ -0,0 +1,26 @@ +--- +name: Bug report +about: Something built or ran, and did the wrong thing +labels: bug +--- + +**What happened** + +**Expected** + +**Repro** + +``` +# command, including the model and backend +``` + +**Environment** +- vla.cpp commit: +- Backend: CPU / CUDA / Metal / SYCL +- OS and compiler: +- GPU and driver (if relevant): +- Model and checkpoint: + +**Output** + +Paste the shortest decisive part of the log, not the whole run. diff --git a/.github/ISSUE_TEMPLATE/model_request.md b/.github/ISSUE_TEMPLATE/model_request.md new file mode 100644 index 0000000..b7c8a51 --- /dev/null +++ b/.github/ISSUE_TEMPLATE/model_request.md @@ -0,0 +1,20 @@ +--- +name: Architecture request +about: Ask for a VLA policy that vla.cpp does not run yet +labels: enhancement +--- + +**Policy** + +Name, paper or repo link, and the reference implementation. + +**Checkpoint** + +Where the weights live and under what license. + +**Why it is worth adding** + +Benchmark numbers, or what it does that the supported archs do not. + +CONTRIBUTING.md has the six-site walkthrough if you want to send the port +yourself. diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md new file mode 100644 index 0000000..14ebff8 --- /dev/null +++ b/.github/pull_request_template.md @@ -0,0 +1,12 @@ +## What + +## Why + +## Verified + +- [ ] Builds clean under `-Wall -Wextra` (first-party code) +- [ ] `ctest` passes +- [ ] Numeric output unchanged (`vla_predict_check` diff), or the change is + meant to move it and a LIBERO sweep is below + +Archs and backends tested: diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 2ea5183..86ec6a2 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -59,5 +59,5 @@ jobs: # Not in cpp-unit: these need ggml headers, so llama.cpp must be fetched. - name: ctest run: | - cmake --build build -j"$(nproc)" --target test_dit_common + cmake --build build -j"$(nproc)" --target test_dit_common test_qwen3vl_vit ctest --test-dir build --output-on-failure diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..c421f81 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,132 @@ +# Tagged binaries and a container image. build.yml already compiles all of this +# on every push; this is the same work with the artifacts kept. +name: release + +on: + push: + tags: ['v*'] + workflow_dispatch: + inputs: + tag: + description: Tag to build (dry run, nothing is published) + required: true + +permissions: + contents: write + packages: write + +env: + BINARIES: vla-server vlm-server vla-cli vla-bench + +jobs: + linux: + runs-on: ubuntu-24.04 + strategy: + fail-fast: false + matrix: + include: + - name: linux-x86_64-cpu + cmake: -DGGML_CUDA=OFF + cuda: false + - name: linux-x86_64-cuda + cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=75;86;89;120 + cuda: true + steps: + - uses: actions/checkout@v4 + + - name: deps + run: | + sudo apt-get update -qq + sudo apt-get install -y -qq --no-install-recommends \ + build-essential cmake git ca-certificates pkg-config \ + libzmq3-dev cppzmq-dev libprotobuf-dev protobuf-compiler + + - name: cuda toolkit + if: matrix.cuda + run: | + wget -q https://developer.download.nvidia.com/compute/cuda/repos/ubuntu2404/x86_64/cuda-keyring_1.1-1_all.deb + sudo dpkg -i cuda-keyring_1.1-1_all.deb + sudo apt-get update -qq + sudo apt-get install -y -qq --no-install-recommends cuda-toolkit-12-6 + echo "/usr/local/cuda/bin" >> "$GITHUB_PATH" + + - name: build + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release ${{ matrix.cmake }} + cmake --build build -j"$(nproc)" --target $BINARIES vla + + - name: package + run: | + out="vla.cpp-${{ github.ref_name }}-${{ matrix.name }}" + mkdir -p "$out" + for b in $BINARIES; do cp "build/$b" "$out/"; done + cp build/libvla.so "$out/" + cp include/vla.h LICENSE.md README.md "$out/" + tar -czf "$out.tar.gz" "$out" + + - uses: actions/upload-artifact@v4 + with: + name: ${{ matrix.name }} + path: '*.tar.gz' + + macos: + runs-on: macos-14 + steps: + - uses: actions/checkout@v4 + + - name: deps + run: brew install cmake zeromq cppzmq protobuf + + - name: build + run: | + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_METAL=ON + cmake --build build -j"$(sysctl -n hw.ncpu)" --target $BINARIES vla + + - name: package + run: | + out="vla.cpp-${{ github.ref_name }}-macos-arm64-metal" + mkdir -p "$out" + for b in $BINARIES; do cp "build/$b" "$out/"; done + cp build/libvla.dylib "$out/" + cp include/vla.h LICENSE.md README.md "$out/" + # Metal needs the shader library next to the binary. + find build -name 'default.metallib' -exec cp {} "$out/" \; + tar -czf "$out.tar.gz" "$out" + + - uses: actions/upload-artifact@v4 + with: + name: macos-arm64-metal + path: '*.tar.gz' + + docker: + runs-on: ubuntu-24.04 + steps: + - uses: actions/checkout@v4 + - uses: docker/setup-buildx-action@v3 + - uses: docker/login-action@v3 + if: startsWith(github.ref, 'refs/tags/') + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + - uses: docker/build-push-action@v6 + with: + context: . + push: ${{ startsWith(github.ref, 'refs/tags/') }} + tags: | + ghcr.io/${{ github.repository }}:${{ github.ref_name }} + ghcr.io/${{ github.repository }}:latest + cache-from: type=gha + cache-to: type=gha,mode=max + + publish: + needs: [linux, macos, docker] + if: startsWith(github.ref, 'refs/tags/') + runs-on: ubuntu-24.04 + steps: + - uses: actions/download-artifact@v4 + with: { path: dist, merge-multiple: true } + - uses: softprops/action-gh-release@v2 + with: + files: dist/*.tar.gz + generate_release_notes: true diff --git a/CMakeLists.txt b/CMakeLists.txt index f7dbd4f..615e1e0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -233,8 +233,14 @@ target_include_directories(vla-cli PRIVATE ) target_link_libraries(vla-cli PRIVATE vla_core) +# Latency for one checkpoint, emits the README table rows. +add_executable(vla-bench + src/serving/vla-bench.cpp +) +target_link_libraries(vla-bench PRIVATE vla_core) + # --- First-party build hygiene (never applied to the vendored llama.cpp subtree) -- -set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla vla-server vlm-server vla-cli) +set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla vla-server vlm-server vla-cli vla-bench) foreach(tgt IN LISTS VLA_FIRST_PARTY_TARGETS) # Warn on our own C++ only; nvcc device code keeps its own diagnostics. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..377d0d1 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,83 @@ +# Contributing + +## Build and test + +```bash +cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON +cmake --build build -j"$(nproc)" +ctest --test-dir build --output-on-failure +``` + +First-party code must compile clean under `-Wall -Wextra`. Warnings from +`build/_deps` are upstream and not your problem. + +## Proving a change is numerically neutral + +Refactors, performance work and dependency bumps must not move the output. +`vla_predict_check` feeds fixed images, tokens, state and noise, then prints the +action chunk: + +```bash +VLA_IMG_SIZE=224 ./build/tests/vla_predict_check model.gguf "" 1 > before.txt +# ... make the change, rebuild ... +VLA_IMG_SIZE=224 ./build/tests/vla_predict_check model.gguf "" 1 > after.txt +diff before.txt after.txt +``` + +Any difference is a bug unless the change is meant to alter numerics, in which +case say so in the commit message and back it with a LIBERO sweep. + +`VLA_IMG_SIZE` must match the model or `predict` returns empty: 512 for +SmolVLA, 448 for Evo-1, 256 for GR00T N1.7 and VLA-JEPA, 224 for the rest. Other +knobs: `VLA_BENCH_ITERS` (timing), `VLA_TIMING=phase`, `VLA_EXTRA_TOKEN` / +`VLA_EXTRA_COUNT` (VLA-JEPA needs its `` tokens), `VLA_N_THREADS`, +`VLA_DEVICE`. + +Checkpoints are at [huggingface.co/vrfai](https://huggingface.co/vrfai), or let +the binaries fetch them: + +```bash +./build/vla-cli -hf vrfai/smolvla-libero-gguf --image assets/front.jpg --tokens 1,100,2 +``` + +## Adding an architecture + +Six sites, all mechanical. `smolvla` is the reference for a two-file (mmproj + +ckpt) model, `bitvla` for a vision-baked one. + +1. `src/arch.h` — add to `enum class Arch`. +2. `src/arch.h` — declare `_create(mmproj_path, ckpt_path, config_path)`. +3. `src/model.cpp` — add `.architecture` to the `try_str` list in + `detect_arch_gguf`. +4. `src/model.cpp` — map the string to the enum in the same function. +5. `src/model.cpp` — add a `case` to the `model_load` switch. +6. `CMakeLists.txt` — add `src/models/.cpp` to `vla_core`. + +Then write `src/models/.cpp`. Before adding a helper, check +`src/models/`: `gguf_reader.h` (tensor and KV reads), `vision_common.h` +(preprocessing, pixel shuffle), `dual_tower.h` (DINOv2 + SigLIP), +`qwen3vl_vit.h` (Qwen3-VL tower), `dit_common.h` (DiT time embeddings), +`scratch_ctx.h` (compute context reuse), `backend.h` (accelerator selection). + +Your loader must fail rather than return a half-built model: check every tensor +lookup, and check `real_*_dim <= max_*_dim` (`config_is_sane` in `src/model.cpp` +does this for all archs). + +A converter goes in `scripts/convert__to_gguf.py`, and its tensor-name +remap should be covered by `tests/py/test_converters.py`. + +## Ports of upstream quirks + +Some references do surprising things, and we match them because the weights were +trained that way. Three so far: VLA-Adapter's RoPE pairs a half-split frequency +table with an interleaved rotation, OpenVLA-OFT's LM attention is bidirectional, +and BitVLA's is too. Each carries a comment naming the reference file and lines. + +If you find something that looks wrong, diff against the reference before +changing it, and leave a comment with the line numbers so the next reader does +not have to. + +## Commits and pull requests + +One logical change per commit, imperative subject, no trailing period. Say what +you verified: which archs, which backends, whether the numeric output moved. diff --git a/src/model.h b/src/model.h index 8aff141..06cbeba 100644 --- a/src/model.h +++ b/src/model.h @@ -91,7 +91,10 @@ struct Model; */ enum class TimingDetail { NONE, ///< Only @c ms_total is populated. - PHASE, ///< Per-phase timings (vision, prefill, denoise, ...). + /// Per-phase timings (vision, prefill, denoise, ...). SmolVLA uses a second + /// builder here that does not pad the prefix to @c n_lang; same positions and + /// masking, so it differs from @c NONE only by float reduction order. + PHASE, }; /** diff --git a/src/models/qwen3vl_vit.h b/src/models/qwen3vl_vit.h index cab28a3..6a48ef2 100644 --- a/src/models/qwen3vl_vit.h +++ b/src/models/qwen3vl_vit.h @@ -12,8 +12,7 @@ // See the License for the specific language governing permissions and // limitations under the License. -// Qwen3-VL vision tower, shared verbatim by GR00T N1.7 and VLA-JEPA. The LM and -// action heads differ, so they stay in their own files. +// Qwen3-VL vision tower, shared by GR00T N1.7 and VLA-JEPA. #pragma once @@ -36,7 +35,6 @@ constexpr float QWEN3VL_STD [3] = {0.5f, 0.5f, 0.5f}; struct VitLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wqkv,*bqkv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; struct MergerW { ggml_tensor *nw,*nb,*fc1w,*fc1b,*fc2w,*fc2b; }; -// Half-split rotation, matching the Qwen3-VL reference. inline ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos_t, ggml_tensor * sin_t) { const int64_t hd = x->ne[0], S = x->ne[1], Hh = x->ne[2]; const int64_t half = hd / 2; ggml_tensor * x1 = ggml_cont(C, ggml_view_3d(C, x, half, S, Hh, x->nb[1], x->nb[2], 0)); @@ -47,7 +45,6 @@ inline ggml_tensor * rope2d(ggml_context * C, ggml_tensor * x, ggml_tensor * cos inline bool fa_enabled() { static const bool e = (std::getenv("VLA_FLASH_ATTN") != nullptr); return e; } -// K and V must be F16 for ggml_flash_attn_ext; the accumulator stays F32. inline ggml_tensor * flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, ggml_tensor * mask, float scale) { ggml_tensor * kf = (k->type == GGML_TYPE_F16) ? k : ggml_cast(C, k, GGML_TYPE_F16); @@ -85,7 +82,7 @@ inline ggml_tensor * build_vit_layer(ggml_context * C, const VitLayerW & w, ggml return ggml_add(C, h1, ff); } -// pre_merge normalizes before the space-to-depth reshape, the deepstack taps after. +// pre_merge normalizes before the reshape, the deepstack taps after. inline ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tensor * x, int64_t hidden, int64_t merge2, float ln_eps, bool pre_merge) { const int64_t n_patches = x->ne[1], c_merged = hidden * merge2 * merge2, n_merged = n_patches / (merge2 * merge2); @@ -101,7 +98,7 @@ inline ggml_tensor * build_merger(ggml_context * C, const MergerW & w, ggml_tens return ggml_add(C, ggml_mul_mat(C, w.fc2w, ggml_gelu(C, z1)), w.fc2b); } -// Patch order after the spatial merge: row/col of each patch in the m x m block. +// Patch row/col after the spatial merge. inline void merge_block_coords(int64_t gh, int64_t gw, int64_t m, std::vector & row, std::vector & col) { const int64_t S = gh * gw; row.assign(S, 0); col.assign(S, 0); for (int64_t s = 0; s < S; ++s) { @@ -125,7 +122,7 @@ inline void vit_rope_tables(const std::vector & row, const std::vector< } } -// Bilinear resample of the pretrained num_side x num_side position table onto gh x gw. +// Bilinear resample of the pretrained position table onto gh x gw. inline void interp_pos_embed(const std::vector & table, int64_t num_side, int64_t hidden, const std::vector & row, const std::vector & col, int64_t gh, int64_t gw, std::vector & out) { @@ -144,8 +141,7 @@ inline void interp_pos_embed(const std::vector & table, int64_t num_side, } } -// HWC to flat patches in CLIP normalization. No resize: the view must already be -// side x side. arch only labels the error. +// HWC to flat patches. No resize: the view must already be side x side. inline bool preprocess_image_patches(const char * arch, const ImageView & v, int64_t side, int64_t ps, int64_t tps, const std::vector & row, const std::vector & col, std::vector & out) { diff --git a/src/models/scratch_ctx.h b/src/models/scratch_ctx.h index b2371c1..3b33ffe 100644 --- a/src/models/scratch_ctx.h +++ b/src/models/scratch_ctx.h @@ -33,7 +33,6 @@ class scratch_ctx { scratch_ctx & operator=(const scratch_ctx &) = delete; ~scratch_ctx() { release(); } - // Empty context sized for `arena` bytes of tensor headers, or null on failure. ggml_context * reset(size_t arena) { if (ctx_) { ggml_reset(ctx_); return ctx_; } ggml_init_params p = { arena, nullptr, true }; diff --git a/src/serving/hf_fetch.h b/src/serving/hf_fetch.h new file mode 100644 index 0000000..d4f03e5 --- /dev/null +++ b/src/serving/hf_fetch.h @@ -0,0 +1,113 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Resolves -hf user/repo[:file.gguf] to a local path, shelling out to the hf +// CLI on a miss. + +#pragma once + +#include +#include +#include +#include + +namespace vla { + +// Repo ids reach a shell command, so reject anything outside this set. +inline bool hf_token_ok(const std::string & s, bool allow_slash) { + if (s.empty() || s.size() > 200) return false; + if (s.front() == '-' || s.find("..") != std::string::npos) return false; + for (const char c : s) { + const bool ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || + (c >= '0' && c <= '9') || c == '.' || c == '_' || c == '-' || + (allow_slash && c == '/'); + if (!ok) return false; + } + return true; +} + +inline std::string hf_cache_root() { + if (const char * e = std::getenv("VLA_CACHE"); e && *e) return e; + if (const char * h = std::getenv("HOME"); h && *h) return std::string(h) + "/.cache/vla"; + return ".vla-cache"; +} + +// Largest non-mmproj .gguf in dir, or "". +inline std::string hf_pick_gguf(const std::filesystem::path & dir, const std::string & want) { + namespace fs = std::filesystem; + std::error_code ec; + std::string best; + uintmax_t best_size = 0; + for (fs::recursive_directory_iterator it(dir, ec), end; it != end && !ec; it.increment(ec)) { + if (!it->is_regular_file(ec) || it->path().extension() != ".gguf") continue; + const std::string name = it->path().filename().string(); + if (!want.empty()) { + if (name == want) return it->path().string(); + continue; + } + if (name.rfind("mmproj", 0) == 0) continue; + const uintmax_t sz = it->file_size(ec); + if (ec) continue; + if (sz > best_size) { best_size = sz; best = it->path().string(); } + } + return best; +} + +// Returns "" and explains on stderr. +inline std::string hf_resolve(const std::string & spec) { + namespace fs = std::filesystem; + + const size_t colon = spec.find(':'); + const std::string repo = spec.substr(0, colon); + const std::string file = (colon == std::string::npos) ? "" : spec.substr(colon + 1); + + if (repo.find('/') == std::string::npos || !hf_token_ok(repo, true) || + (!file.empty() && !hf_token_ok(file, false))) { + std::fprintf(stderr, "vla: -hf expects user/repo[:file.gguf], got '%s'\n", spec.c_str()); + return ""; + } + + const fs::path dir = fs::path(hf_cache_root()) / repo; + + std::error_code ec; + if (fs::is_directory(dir, ec)) { + const std::string hit = hf_pick_gguf(dir, file); + if (!hit.empty()) return hit; + } + + fs::create_directories(dir, ec); + if (ec) { + std::fprintf(stderr, "vla: cannot create %s: %s\n", dir.string().c_str(), ec.message().c_str()); + return ""; + } + + std::string cmd = "hf download " + repo; + if (!file.empty()) cmd += " " + file; + cmd += " --local-dir '" + dir.string() + "'"; + std::fprintf(stderr, "vla: %s\n", cmd.c_str()); + + const int rc = std::system(cmd.c_str()); + if (rc != 0) { + std::fprintf(stderr, + "vla: download failed (exit %d). Install the CLI with\n" + " pip install -U \"huggingface_hub[cli]\"\n", rc); + return ""; + } + + const std::string hit = hf_pick_gguf(dir, file); + if (hit.empty()) std::fprintf(stderr, "vla: no .gguf under %s after download\n", dir.string().c_str()); + return hit; +} + +} // namespace vla diff --git a/src/serving/server.cpp b/src/serving/server.cpp index 25c3981..33f0610 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -13,6 +13,7 @@ // limitations under the License. #include "model.h" +#include "serving/hf_fetch.h" #include "serving/vla.pb.h" #define STB_IMAGE_IMPLEMENTATION @@ -171,11 +172,13 @@ int find_non_finite(const float * data, int n) { void usage(const char * prog) { std::fprintf(stderr, "usage: %s [--bind ADDR] [--timing-detail none|phase] [--config PATH] " - "[] \n" + "[] ( | -hf user/repo[:file.gguf])\n" " vision-tower mmproj GGUF (SigLIP / PaliGemma /\n" " connector). Required for SmolVLA, π0, Evo-1, GR00T.\n" " Omit for BitVLA - its vision tower is baked into\n" " the combined ckpt GGUF.\n" + " -hf HuggingFace repo, user/repo[:file.gguf]; downloaded\n" + " on a miss and cached under $VLA_CACHE.\n" " SmolVLA .safetensors or .gguf, or any of the other\n" " supported architectures' .gguf; the architecture is\n" " auto-detected from the checkpoint.\n" @@ -200,6 +203,7 @@ int main(int argc, char ** argv) { std::string bind_addr = "tcp://*:5555"; std::string mmproj_path; std::string ckpt_path; + std::string hf_spec; std::string config_path; vla::TimingDetail timing_detail = vla::TimingDetail::NONE; @@ -208,6 +212,8 @@ int main(int argc, char ** argv) { std::string a = argv[i]; if (a == "--bind" && i + 1 < argc) { bind_addr = argv[++i]; + } else if (a == "-hf" && i + 1 < argc) { + hf_spec = argv[++i]; } else if (a == "--config" && i + 1 < argc) { config_path = argv[++i]; } else if (a == "--timing-detail" && i + 1 < argc) { @@ -226,14 +232,17 @@ int main(int argc, char ** argv) { positionals.push_back(std::move(a)); } } - if (positionals.size() == 1) { + if (!hf_spec.empty() && positionals.empty()) { + ckpt_path = vla::hf_resolve(hf_spec); + if (ckpt_path.empty()) return 1; + } else if (positionals.size() == 1) { ckpt_path = positionals[0]; } else if (positionals.size() == 2) { mmproj_path = positionals[0]; ckpt_path = positionals[1]; } else { std::fprintf(stderr, - "vla-server: expected 1 or 2 positional args " + "vla-server: expected -hf, or 1 or 2 positional args " "( for SmolVLA/π0/Evo-1/GR00T, " "or just for BitVLA), got %zu\n", positionals.size()); diff --git a/src/serving/vla-bench.cpp b/src/serving/vla-bench.cpp new file mode 100644 index 0000000..b185802 --- /dev/null +++ b/src/serving/vla-bench.cpp @@ -0,0 +1,159 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Latency for one checkpoint. Synthetic inputs: engine only, no task success. + +#include "model.h" +#include "serving/hf_fetch.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { + +void usage(const char * prog) { + std::fprintf(stderr, + "usage: %s (--ckpt c.gguf | -hf user/repo) [--mmproj m.gguf]\n" + " [--label name] [--images N] [--size N] [--tokens N]\n" + " [--warmup N] [--reps N] [--markdown]\n" + " --label row label (default: the checkpoint filename)\n" + " --images camera views (default 1)\n" + " --size square input side in pixels (default 224)\n" + " --tokens language token count (default 16)\n" + " --warmup untimed calls before measuring (default 3)\n" + " --reps timed calls (default 20)\n" + " --markdown print a markdown table row instead of a plain summary\n", + prog); +} + +// v must be sorted. +double percentile(const std::vector & v, double p) { + if (v.empty()) return 0.0; + const double idx = p * (double) (v.size() - 1); + const size_t lo = (size_t) std::floor(idx), hi = (size_t) std::ceil(idx); + return v[lo] + (v[hi] - v[lo]) * (idx - (double) lo); +} + +} // namespace + +int main(int argc, char ** argv) { + std::string ckpt, mmproj, hf, label; + int n_images = 1, side = 224, n_tokens = 16, warmup = 3, reps = 20; + bool markdown = false; + + for (int i = 1; i < argc; ++i) { + const std::string a = argv[i]; + auto need = [&](const char * name) -> const char * { + if (i + 1 >= argc) { std::fprintf(stderr, "vla-bench: %s needs a value\n", name); std::exit(1); } + return argv[++i]; + }; + if (a == "--ckpt") ckpt = need("--ckpt"); + else if (a == "-hf") hf = need("-hf"); + else if (a == "--mmproj") mmproj = need("--mmproj"); + else if (a == "--label") label = need("--label"); + else if (a == "--images") n_images = std::atoi(need("--images")); + else if (a == "--size") side = std::atoi(need("--size")); + else if (a == "--tokens") n_tokens = std::atoi(need("--tokens")); + else if (a == "--warmup") warmup = std::atoi(need("--warmup")); + else if (a == "--reps") reps = std::atoi(need("--reps")); + else if (a == "--markdown") markdown = true; + else if (a == "-h" || a == "--help") { usage(argv[0]); return 0; } + else { std::fprintf(stderr, "vla-bench: unknown argument %s\n", a.c_str()); usage(argv[0]); return 1; } + } + + if (!hf.empty()) { + if (!ckpt.empty()) { std::fprintf(stderr, "vla-bench: pass --ckpt or -hf, not both\n"); return 1; } + ckpt = vla::hf_resolve(hf); + if (ckpt.empty()) return 1; + } + if (ckpt.empty()) { usage(argv[0]); return 1; } + if (n_images < 1 || side < 16 || n_tokens < 1 || warmup < 0 || reps < 1) { + std::fprintf(stderr, "vla-bench: --images/--size/--tokens/--reps must be positive\n"); + return 1; + } + if (label.empty()) { + const size_t slash = ckpt.find_last_of('/'); + label = (slash == std::string::npos) ? ckpt : ckpt.substr(slash + 1); + } + + vla::Model * m = vla::model_load(mmproj, ckpt, ""); + if (!m) { std::fprintf(stderr, "vla-bench: model_load failed\n"); return 1; } + const vla::Config & cfg = vla::model_config(m); + + std::vector> pixels(n_images, std::vector((size_t) 3 * side * side)); + std::vector views(n_images); + for (int v = 0; v < n_images; ++v) { + for (int y = 0; y < side; ++y) + for (int x = 0; x < side; ++x) + for (int c = 0; c < 3; ++c) + pixels[v][((size_t) y * side + x) * 3 + c] = (uint8_t) ((x + 2 * y + 40 * c + 17 * v) & 0xFF); + views[v] = vla::ImageView{ pixels[v].data(), side, side, vla::PixelFormat::U8 }; + } + + std::vector lang((size_t) n_tokens); + for (int i = 0; i < n_tokens; ++i) lang[i] = 1 + (i % 100); + + std::vector state((size_t) cfg.max_state_dim, 0.0f); + for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) state[i] = 0.01f * (float) (i + 1); + + std::vector noise((size_t) cfg.max_action_dim * (size_t) cfg.n_suffix); + for (size_t i = 0; i < noise.size(); ++i) noise[i] = 0.001f * (float) ((i * 2654435761u) % 1000) - 0.5f; + + vla::Inputs in{}; + in.images = views.data(); + in.n_images = n_images; + in.lang_tokens = lang.data(); + in.n_lang = n_tokens; + in.state = state.data(); + in.noise = noise.data(); + + for (int i = 0; i < warmup; ++i) { + if (vla::predict(m, in).empty()) { std::fprintf(stderr, "vla-bench: predict failed\n"); vla::model_free(m); return 1; } + } + + std::vector ms; + ms.reserve((size_t) reps); + double vision_sum = 0.0; + for (int i = 0; i < reps; ++i) { + const auto t0 = std::chrono::steady_clock::now(); + const std::vector out = vla::predict(m, in); + const auto t1 = std::chrono::steady_clock::now(); + if (out.empty()) { std::fprintf(stderr, "vla-bench: predict failed at rep %d\n", i); vla::model_free(m); return 1; } + ms.push_back(std::chrono::duration(t1 - t0).count()); + vision_sum += vla::last_stats(m).ms_vision; + } + + std::sort(ms.begin(), ms.end()); + const double lo = ms.front(); + const double p50 = percentile(ms, 0.50); + const double p90 = percentile(ms, 0.90); + const double vision = vision_sum / (double) reps; + + if (markdown) { + std::printf("| %s | %d | %d | %d | %.1f | %.1f | %.1f | %.1f |\n", + label.c_str(), n_images, side, n_tokens, lo, p50, p90, vision); + } else { + std::printf("%s: min %.1f ms p50 %.1f ms p90 %.1f ms vision %.1f ms (%d views, %dx%d, %d tokens, %d reps)\n", + label.c_str(), lo, p50, p90, vision, n_images, side, side, n_tokens, reps); + } + + vla::model_free(m); + return 0; +} diff --git a/src/serving/vla-cli.cpp b/src/serving/vla-cli.cpp index 597a010..1f86965 100644 --- a/src/serving/vla-cli.cpp +++ b/src/serving/vla-cli.cpp @@ -21,6 +21,7 @@ // --tokens id,id,... [--state f,f,...] [--pretty] #include "model.h" +#include "serving/hf_fetch.h" #define STB_IMAGE_IMPLEMENTATION #define STB_IMAGE_STATIC @@ -94,10 +95,11 @@ bool load_image(const char * path, std::vector & buf, int & w, int & h) void usage(const char * prog) { std::fprintf(stderr, - "usage: %s [--mmproj m.gguf] --ckpt c.gguf --image img.jpg [--image ...]\n" + "usage: %s [--mmproj m.gguf] (--ckpt c.gguf | -hf user/repo) --image img.jpg [--image ...]\n" " --tokens id,id,... [--state f,f,...] [--pretty]\n" " --mmproj vision-tower GGUF (SmolVLA/pi0/pi0.5); omit for baked-vision archs\n" " --ckpt model checkpoint GGUF\n" + " -hf HuggingFace repo, user/repo[:file.gguf], cached under $VLA_CACHE\n" " --image image file, repeat for multi-view (decoded via stb_image)\n" " --tokens language token ids, comma-separated (tokenize in the client)\n" " --state proprioception floats, comma-separated (default zeros)\n" @@ -108,7 +110,7 @@ void usage(const char * prog) { } // namespace int main(int argc, char ** argv) { - std::string mmproj, ckpt, tokens_s, state_s; + std::string mmproj, ckpt, hf, tokens_s, state_s; std::vector image_paths; bool pretty = false; @@ -120,6 +122,7 @@ int main(int argc, char ** argv) { }; if (a == "--mmproj") mmproj = need("--mmproj"); else if (a == "--ckpt") ckpt = need("--ckpt"); + else if (a == "-hf") hf = need("-hf"); else if (a == "--image") image_paths.push_back(need("--image")); else if (a == "--tokens") tokens_s = need("--tokens"); else if (a == "--state") state_s = need("--state"); @@ -127,6 +130,11 @@ int main(int argc, char ** argv) { else if (a == "-h" || a == "--help") { usage(argv[0]); return 0; } else { std::fprintf(stderr, "vla-cli: unknown argument %s\n", a.c_str()); usage(argv[0]); return 1; } } + if (!hf.empty()) { + if (!ckpt.empty()) { std::fprintf(stderr, "vla-cli: pass --ckpt or -hf, not both\n"); return 1; } + ckpt = vla::hf_resolve(hf); + if (ckpt.empty()) return 1; + } if (ckpt.empty() || image_paths.empty() || tokens_s.empty()) { usage(argv[0]); return 1; } // Validate the cheap args before loading the model. diff --git a/tests/predict_check.cpp b/tests/predict_check.cpp index 504ccbc..f27f7f0 100644 --- a/tests/predict_check.cpp +++ b/tests/predict_check.cpp @@ -18,13 +18,15 @@ // others read it, so fixed noise is reproducible for all archs). // // predict_check [mmproj.gguf] [n_images] -// env: VLA_IMG_SIZE (square input, default 224), VLA_BENCH_ITERS (>0 = time it) +// env: VLA_IMG_SIZE (square input, default 224), VLA_BENCH_ITERS (>0 = time it), +// VLA_TIMING=phase, VLA_EXTRA_TOKEN / VLA_EXTRA_COUNT #include "model.h" #include #include #include +#include #include #include @@ -85,7 +87,8 @@ int main(int argc, char** argv) { in.n_lang = (int)lang.size(); in.state = state.data(); in.noise = noise.data(); - in.timing_detail = TimingDetail::NONE; + const char* td = std::getenv("VLA_TIMING"); + in.timing_detail = (td && std::string(td) == "phase") ? TimingDetail::PHASE : TimingDetail::NONE; std::vector act = predict(m, in); std::printf("action_len=%zu\n", act.size()); diff --git a/tests/test_qwen3vl_vit.cpp b/tests/test_qwen3vl_vit.cpp index 2354f85..1605480 100644 --- a/tests/test_qwen3vl_vit.cpp +++ b/tests/test_qwen3vl_vit.cpp @@ -25,7 +25,7 @@ using namespace vla; static int fails = 0; #define CHECK(c) do { if (!(c)) { std::printf("FAIL %s:%d %s\n", __FILE__, __LINE__, #c); ++fails; } } while (0) -// 4x4 grid, 2x2 merge: patches arrive grouped by block, not in raster order. +// 4x4 grid, 2x2 merge: patches group by block, not raster order. static void test_merge_block_coords() { std::vector row, col; merge_block_coords(4, 4, 2, row, col); @@ -35,7 +35,7 @@ static void test_merge_block_coords() { for (int i = 0; i < 16; ++i) { CHECK(row[i] == want_r[i]); CHECK(col[i] == want_c[i]); } } -// Half-split layout: the second half of the table repeats the first. +// Second half of the table repeats the first. static void test_vit_rope_tables() { std::vector row = {0, 1}, col = {0, 2}; std::vector c, s; @@ -47,14 +47,12 @@ static void test_vit_rope_tables() { CHECK(c[p * hd + i] == c[p * hd + hd / 2 + i]); CHECK(s[p * hd + i] == s[p * hd + hd / 2 + i]); } - // Position 0 has zero angle on every frequency. for (int64_t i = 0; i < hd; ++i) { CHECK(c[i] == 1.0f); CHECK(s[i] == 0.0f); } - // Row 1, col 2 with inv_freq[0] == 1: row half is cos(1), col half is cos(2). CHECK(std::fabs(c[hd + 0] - std::cos(1.0f)) < 1e-6f); CHECK(std::fabs(c[hd + hd / 4] - std::cos(2.0f)) < 1e-6f); } -// Same grid as the source table is an identity resample. +// Same grid in and out is an identity resample. static void test_interp_pos_embed_identity() { const int64_t side = 2, hidden = 2; std::vector table = {0,10, 1,11, 2,12, 3,13}; @@ -70,7 +68,7 @@ static void test_interp_pos_embed_identity() { } } -// Upsampling 2x2 to 3x3 puts the midpoint at the mean of the four corners. +// 2x2 to 3x3 puts the midpoint at the mean of the corners. static void test_interp_pos_embed_bilinear() { const int64_t side = 2, hidden = 1; std::vector table = {0, 2, 4, 6}; @@ -82,7 +80,7 @@ static void test_interp_pos_embed_bilinear() { if (row[s] == 1 && col[s] == 1) CHECK(std::fabs(out[s] - 3.0f) < 1e-6f); } -// A wrong-sized view must be rejected before any pixel is read. +// Wrong-sized view must be rejected before any pixel is read. static void test_preprocess_rejects_bad_view() { std::vector px(3 * 4 * 4, 0); std::vector row, col; @@ -94,8 +92,7 @@ static void test_preprocess_rejects_bad_view() { CHECK(!preprocess_image_patches("test", null, 4, 2, 2, row, col, out)); } -// U8 128 maps to ~0 after the 0.5/0.5 normalization, and every temporal slice -// repeats the same value. +// Every temporal slice repeats the same value. static void test_preprocess_values() { const int64_t side = 2, ps = 1, tps = 2; std::vector px((size_t) 3 * side * side, 128); From de62b3d6145bdccf10f7c7e2b1b160a37ec2ba3a Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 14:10:26 +0700 Subject: [PATCH 22/42] default the gr00t n1.7 graph cache on, regenerate the benchmark table --- README.md | 57 ++++++++++++++++++++++++++++----------- src/models/gr00tn1d7.cpp | 4 ++- src/serving/vla-bench.cpp | 10 +++++-- 3 files changed, 53 insertions(+), 18 deletions(-) diff --git a/README.md b/README.md index 587b7c1..2e4b5b1 100644 --- a/README.md +++ b/README.md @@ -94,10 +94,13 @@ WSL2 and Apple Silicon are both tested. Once the binaries are built, run one CPU prediction without a server or simulator: ```bash -pip install -U "huggingface_hub[cli]" gguf -hf download vrfai/smolvla-libero-gguf --local-dir models/smolvla +pip install -U "huggingface_hub[cli]" -# One-shot CLI +# -hf fetches and caches the checkpoint (under $VLA_CACHE, default ~/.cache/vla) +./build/vla-cli -hf vrfai/smolvla-libero-gguf \ + --image assets/front.jpg --tokens 1,100,200,2 --pretty + +# or point at a file you already have ./build/vla-cli --ckpt models/smolvla/smolvla-libero.gguf \ --image assets/front.jpg --tokens 1,100,200,2 --pretty ``` @@ -160,10 +163,13 @@ vla-server: bound to tcp://*:5555. ready. Use `--bind` to change the address and port. Stop the server with `Ctrl-C`. -Two environment knobs apply to every arch: +`vla-server` also takes `-hf user/repo[:file.gguf]` in place of a checkpoint path. + +Environment knobs that apply to every arch: - `VLA_N_THREADS` - CPU backend thread count, default core count capped at 16. - `VLA_DEVICE` - GPU ordinal for CUDA and SYCL builds, default 0. +- `VLA_CACHE` - where `-hf` stores checkpoints, default `~/.cache/vla`. --- @@ -250,18 +256,32 @@ pack the vision tower too (smaller, but more accuracy loss). ## Benchmarks -Latency in ms (inference plus transport), measured client-side on four targets: an -**RTX 3090**, an **NVIDIA Jetson AGX Orin**, an **NVIDIA Jetson Orin Nano (8 GB)**, -and an **Apple M4**. +`vla-bench` times `predict()` in-process on synthetic inputs: engine only, no +transport, no simulator, no claim about task success. + +```bash +./build/vla-bench -hf vrfai/smolvla-libero-gguf --images 2 --size 512 --markdown +``` -| Model | 3090 call (ms) | AGX Orin call (ms) | Orin Nano call (ms) | M4 call (ms) | -|---|---:|---:|---:|---:| -| `smolvla` | 86 | 262 | 567 | 888 | -| `pi0` | 264 | 893 | 1955 | 1135 | -| `gr00t_n1_5` | 109 | 461 | 1356 | - | -| `gr00t_n1_7` | 102 | 429 | - | 755 | -| `bitvla` | 145 | 809 | 2845 | - | -| `evo1` | 238 | 1048 | 3671 | - | +RTX 5090, driver 595.84, CUDA 13.2, 24-core host, weights as shipped, 20 reps +after 3 warmups, each model at its native input size and view count. + +| Model | Views | Input | min ms | p50 ms | p90 ms | vision ms | +|---|--:|--:|--:|--:|--:|--:| +| VLA-Adapter | 1 | 224 | 19.2 | 20.2 | 21.0 | 9.1 | +| VLA-JEPA | 1 | 256 | 20.7 | 21.8 | 23.4 | 6.2 | +| BitVLA | 1 | 224 | 24.2 | 25.2 | 26.5 | 5.5 | +| GR00T N1.5 | 1 | 224 | 27.7 | 29.8 | 31.7 | 5.7 | +| GR00T N1.7 | 1 | 256 | 31.0 | 32.8 | 34.0 | 6.2 | +| GR00T N1.6 | 1 | 224 | 35.2 | 37.4 | 39.6 | 6.3 | +| OpenVLA-OFT | 1 | 224 | 47.1 | 49.5 | 51.4 | 9.9 | +| SmolVLA | 2 | 512 | 47.9 | 52.4 | 56.5 | 18.8 | +| pi0 | 2 | 224 | 49.4 | 54.0 | 57.9 | 12.0 | +| pi0.5 | 2 | 224 | 53.0 | 55.9 | 57.6 | 10.7 | +| Evo-1 | 1 | 448 | 54.8 | 58.4 | 61.8 | 17.5 | + +Jetson and Apple targets are absent: they have not been re-measured with +`vla-bench`. --- @@ -286,6 +306,13 @@ supported (released and benchmarked), `~` = in progress, `-` = planned. --- +## Contributing + +See [CONTRIBUTING.md](CONTRIBUTING.md) for how to prove a change is numerically +neutral, and the six sites you touch to add an architecture. + +--- + ## Contributors - [Khanh Dang Nguyen](https://github.com/khanhnd61-vr) diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index afd2cb1..d707cd2 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -623,7 +623,9 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { if (in.noise) std::memcpy(x_init.data(), in.noise, x_init.size() * sizeof(float)); else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } - const bool use_cache = (std::getenv("VLA_GR00T_GRAPH_CACHE") != nullptr) && !do_dump; + // On by default: 16% faster, bit-identical. Set VLA_GR00T_GRAPH_CACHE=0 to opt out. + const char * gc = std::getenv("VLA_GR00T_GRAPH_CACHE"); + const bool use_cache = (!gc || std::strcmp(gc, "0") != 0) && !do_dump; const bool reuse = use_cache && mg.valid && mg.seq == SEQ && mg.n_img == n_img && mg.seq_txt == SEQ_TXT && mg.nsteps == num_steps && mg.deepstack == inject_deepstack; diff --git a/src/serving/vla-bench.cpp b/src/serving/vla-bench.cpp index b185802..1a070eb 100644 --- a/src/serving/vla-bench.cpp +++ b/src/serving/vla-bench.cpp @@ -32,11 +32,13 @@ void usage(const char * prog) { std::fprintf(stderr, "usage: %s (--ckpt c.gguf | -hf user/repo) [--mmproj m.gguf]\n" " [--label name] [--images N] [--size N] [--tokens N]\n" - " [--warmup N] [--reps N] [--markdown]\n" + " [--extra-token ID] [--extra-count N] [--warmup N] [--reps N] [--markdown]\n" " --label row label (default: the checkpoint filename)\n" " --images camera views (default 1)\n" " --size square input side in pixels (default 224)\n" " --tokens language token count (default 16)\n" + " --extra-token token id appended --extra-count times (VLA-JEPA needs its\n" + " tokens)\n" " --warmup untimed calls before measuring (default 3)\n" " --reps timed calls (default 20)\n" " --markdown print a markdown table row instead of a plain summary\n", @@ -56,6 +58,7 @@ double percentile(const std::vector & v, double p) { int main(int argc, char ** argv) { std::string ckpt, mmproj, hf, label; int n_images = 1, side = 224, n_tokens = 16, warmup = 3, reps = 20; + int extra_token = -1, extra_count = 0; bool markdown = false; for (int i = 1; i < argc; ++i) { @@ -71,6 +74,8 @@ int main(int argc, char ** argv) { else if (a == "--images") n_images = std::atoi(need("--images")); else if (a == "--size") side = std::atoi(need("--size")); else if (a == "--tokens") n_tokens = std::atoi(need("--tokens")); + else if (a == "--extra-token") extra_token = std::atoi(need("--extra-token")); + else if (a == "--extra-count") extra_count = std::atoi(need("--extra-count")); else if (a == "--warmup") warmup = std::atoi(need("--warmup")); else if (a == "--reps") reps = std::atoi(need("--reps")); else if (a == "--markdown") markdown = true; @@ -109,6 +114,7 @@ int main(int argc, char ** argv) { std::vector lang((size_t) n_tokens); for (int i = 0; i < n_tokens; ++i) lang[i] = 1 + (i % 100); + if (extra_token >= 0 && extra_count > 0) lang.insert(lang.end(), (size_t) extra_count, extra_token); std::vector state((size_t) cfg.max_state_dim, 0.0f); for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) state[i] = 0.01f * (float) (i + 1); @@ -120,7 +126,7 @@ int main(int argc, char ** argv) { in.images = views.data(); in.n_images = n_images; in.lang_tokens = lang.data(); - in.n_lang = n_tokens; + in.n_lang = (int) lang.size(); in.state = state.data(); in.noise = noise.data(); From 478ac575461c4970b83122d343896a22ac4c89df Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 14:23:27 +0700 Subject: [PATCH 23/42] refresh adoption notes, drop em-dashes from contributor docs --- CONTRIBUTING.md | 12 ++++++------ docs/ADOPTION.md | 42 ++++++++++++++++++++---------------------- 2 files changed, 26 insertions(+), 28 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 377d0d1..2ce5d34 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -45,13 +45,13 @@ the binaries fetch them: Six sites, all mechanical. `smolvla` is the reference for a two-file (mmproj + ckpt) model, `bitvla` for a vision-baked one. -1. `src/arch.h` — add to `enum class Arch`. -2. `src/arch.h` — declare `_create(mmproj_path, ckpt_path, config_path)`. -3. `src/model.cpp` — add `.architecture` to the `try_str` list in +1. `src/arch.h` - add to `enum class Arch`. +2. `src/arch.h` - declare `_create(mmproj_path, ckpt_path, config_path)`. +3. `src/model.cpp` - add `.architecture` to the `try_str` list in `detect_arch_gguf`. -4. `src/model.cpp` — map the string to the enum in the same function. -5. `src/model.cpp` — add a `case` to the `model_load` switch. -6. `CMakeLists.txt` — add `src/models/.cpp` to `vla_core`. +4. `src/model.cpp` - map the string to the enum in the same function. +5. `src/model.cpp` - add a `case` to the `model_load` switch. +6. `CMakeLists.txt` - add `src/models/.cpp` to `vla_core`. Then write `src/models/.cpp`. Before adding a helper, check `src/models/`: `gguf_reader.h` (tensor and KV reads), `vision_common.h` diff --git a/docs/ADOPTION.md b/docs/ADOPTION.md index 424b48d..f599686 100644 --- a/docs/ADOPTION.md +++ b/docs/ADOPTION.md @@ -1,28 +1,26 @@ # Adoption notes -The engine works; the gap is distribution. Roughly in priority order. +The engine works; the gap is distribution. -1. **C ABI.** Done: `include/vla.h` and `libvla`. `src/model.h` is C++ only, so - without it nothing outside C++ can link the engine. - -2. **Python bindings.** Done: `bindings/python`. Robotics runs on Python; the - only other way in is the ZeroMQ server plus a hand-written client. - -3. **Prebuilt binaries.** `.github/workflows/build.yml` already builds - `vla-server`, `vlm-server` and `vla-cli` and uploads nothing. Tagged - artifacts (linux x86-64 CPU/CUDA, linux aarch64 Jetson, macOS arm64 Metal) - plus a published Docker image remove the build step. +Done: -4. **One-command model fetch.** Today: install `huggingface_hub`, run - `hf download`, pass a path. llama.cpp solved this with `-hf user/repo`. The - GGUFs are already on the Hub under [`vrfai`](https://huggingface.co/vrfai). - -5. **Reproducible benchmarks.** The README latency table has no in-repo source - and disagrees with `ci/baselines/rtx3090.json`. A `vla-bench` that emits the - table, with quantization and memory columns, makes it checkable. - -6. **Contributor path.** Adding an architecture touches six sites, none written - down: the `Arch` enum and factory in `src/arch.h`, the key list, string map - and switch in `src/model.cpp`, and `CMakeLists.txt`. +1. **C ABI.** `include/vla.h` and `libvla`. `src/model.h` is C++ only, so + without it nothing outside C++ can link the engine. +2. **Python bindings.** `bindings/python`, ctypes over the ABI. +3. **Prebuilt binaries.** `.github/workflows/release.yml` publishes + linux-x86_64 (CPU and CUDA), macos-arm64-metal and a Docker image on tag. +4. **One-command model fetch.** `-hf user/repo[:file.gguf]` on `vla-cli`, + `vla-server` and `vla-bench`, cached under `$VLA_CACHE`. +5. **Reproducible benchmarks.** `vla-bench` emits the README table rows. +6. **Contributor path.** `CONTRIBUTING.md` has the six-site walkthrough for + adding an architecture, plus issue and PR templates. + +Left: + +- **Jetson binaries.** `release.yml` covers x86-64 and macOS; aarch64 needs a + self-hosted runner or a cross toolchain. +- **PyPI.** The wheel is built from `bindings/python` but nothing publishes it. +- **`ci/baselines/rtx3090.json`** still disagrees with the README table, which is + now RTX 5090 numbers from `vla-bench`. Re-record the baselines on one machine. None of these change inference behaviour. From e63396d8788faa242536b36467400fa096d5619e Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:05:57 +0700 Subject: [PATCH 24/42] build every test target before ctest ctest registers six tests but the workflow named only two, so the other four reported Not Run and the job exited 8. Building the default target also keeps the next test that gets added from breaking it again. --- .github/workflows/build.yml | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 86ec6a2..3cf9d82 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -52,12 +52,10 @@ jobs: with: path: build/_deps key: llama-b10326-${{ runner.os }} - - name: build vla-server + vlm-server + vla-cli (CPU, -Wall -Wextra) + # Everything, not a target list: ctest registers tests this job must build, + # and a named list goes stale the next time one is added. + - name: build + ctest (CPU, -Wall -Wextra) run: | cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=OFF -DVLA_BUILD_TESTS=ON - cmake --build build -j"$(nproc)" --target vla-server vlm-server vla-cli - # Not in cpp-unit: these need ggml headers, so llama.cpp must be fetched. - - name: ctest - run: | - cmake --build build -j"$(nproc)" --target test_dit_common test_qwen3vl_vit + cmake --build build -j"$(nproc)" ctest --test-dir build --output-on-failure From 1059543afb0bf900de02606dd99f96d009979797 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:06:40 +0700 Subject: [PATCH 25/42] validate checkpoint geometry before sizing buffers A crafted checkpoint could overrun host buffers in five places. The smolvla safetensors reader trusted data_offsets against a shape-derived buffer, bitvla and gr00tn1d6 trusted patch-count keys that the image grid contradicts, vla_adapter trusted head_blocks past the layer count, and the qwen3-vl position resample only clamped the upper interpolation index. Each is now a load-time check or a clamp. While here, smolvla, evo1 and bitvla treat a missing state vector as zeros like the other eight archs instead of dereferencing it. All eleven checkpoints load unchanged and predict_check output is identical. --- src/models/bitvla.cpp | 23 +++++++++++++++++++++-- src/models/evo1.cpp | 3 ++- src/models/gr00tn1d6.cpp | 17 +++++++++++++++++ src/models/gr00tn1d7.cpp | 9 +++++++++ src/models/qwen3vl_vit.h | 5 ++++- src/models/smolvla.cpp | 33 +++++++++++++++++++++++---------- src/models/vla_adapter.cpp | 8 ++++++++ src/models/vla_jepa.cpp | 14 ++++++++++++++ 8 files changed, 98 insertions(+), 14 deletions(-) diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 79bd902..16f78d5 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -343,6 +343,22 @@ bool load_config(const gguf_reader & g, BitvlaModelArch & m, Config & cfg) { I("bitvla.tokens.stop_id", m.stop_id); m.packed_int2 = g.has("bitvla.quant.int2_packed") && g.u32("bitvla.quant.int2_packed") != 0; + // predict() sizes the patch buffer from n_patches but fills it by walking the + // image grid, so a KV that disagrees with the geometry overruns the buffer. + if (m.patch_size <= 0 || m.image_size <= 0 || m.image_size % m.patch_size != 0 || + m.n_patches != (m.image_size / m.patch_size) * (m.image_size / m.patch_size)) { + std::fprintf(stderr, "vla(bitvla): n_patches %lld does not match image %lld / patch %lld\n", + (long long) m.n_patches, (long long) m.image_size, (long long) m.patch_size); + return false; + } + // The CUDA LM writes seq*q_heads*head_dim into buffers sized seq*hidden. + if (m.lm_kv <= 0 || m.lm_head_dim <= 0 || m.lm_q % m.lm_kv != 0 || + m.lm_q * m.lm_head_dim != m.lm_hidden) { + std::fprintf(stderr, "vla(bitvla): lm q_heads %lld x head_dim %lld does not match hidden %lld\n", + (long long) m.lm_q, (long long) m.lm_head_dim, (long long) m.lm_hidden); + return false; + } + const std::string js = g.str("bitvla.statistics_json"); if (js.empty()) { std::fprintf(stderr, "vla(bitvla): bitvla.statistics_json KV missing - un-normalization will pass-through\n"); @@ -1081,9 +1097,12 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { _dump_manifest(std::string("mm_proj_out fp32 ") + std::to_string(n_views) + " " + std::to_string(N) + " " + std::to_string(hidden_l)); std::vector proprio_embed_host((size_t) hidden_l); + // Like the other archs: a caller may leave the proprio vector out. + std::vector state_host((size_t) proprio_dim, 0.0f); + if (in.state) std::memcpy(state_host.data(), in.state, (size_t) proprio_dim * sizeof(float)); #ifdef VLA_BITVLA_CUDA_KERNELS if (cuda_fp32head_ready) { - if (bitvla_fp32head_proprio_forward(fp32head_cuda_ctx, in.state, proprio_embed_host.data(), 0) != 0) { + if (bitvla_fp32head_proprio_forward(fp32head_cuda_ctx, state_host.data(), proprio_embed_host.data(), 0) != 0) { std::fprintf(stderr, "vla(bitvla): CUDA proprio forward failed\n"); return {}; } } else @@ -1100,7 +1119,7 @@ std::vector BitvlaModelArch::predict(const Inputs& in) { ggml_cgraph * gf = ggml_new_graph(ctx); ggml_build_forward_expand(gf, out); if (!proprio_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(bitvla): gallocr failed (proprio)\n"); return {}; } - ggml_backend_tensor_set(x_in, in.state, 0, ggml_nbytes(x_in)); + ggml_backend_tensor_set(x_in, state_host.data(), 0, ggml_nbytes(x_in)); if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { std::fprintf(stderr, "vla(bitvla): proprio compute failed\n"); return {}; } ggml_backend_tensor_get(out, proprio_embed_host.data(), 0, (size_t) hidden_l * sizeof(float)); } diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 6d4c45e..a073ba0 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -523,7 +523,8 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { // The converter zero-pads stats past real_state_dim, so lo == hi == 0 there // and the affine below would map anything to -1. if (hi <= lo) { state_norm[i] = 0.0f; continue; } - float xn = 2.0f * (in.state[i] - lo) / (hi - lo + norm_eps_denom) - 1.0f; + const float sv = in.state ? in.state[i] : 0.0f; + float xn = 2.0f * (sv - lo) / (hi - lo + norm_eps_denom) - 1.0f; if (xn < -1.0f) xn = -1.0f; if (xn > 1.0f) xn = 1.0f; state_norm[i] = xn; diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 79836a0..22d332b 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -242,6 +242,23 @@ bool load_config(const gguf_reader & g, Gr00tN1d6ModelArch & m, Config & cfg) { } if (m.embodiment_id < 0 || m.embodiment_id >= m.max_embodiments) { std::fprintf(stderr, "vla(gr00tn1d6): embodiment id %lld out of range [0,%lld)\n", (long long) m.embodiment_id, (long long) m.max_embodiments); return false; } + // pixel_shuffle_back writes (grid/shuffle)^2 tokens into a buffer sized from + // n_img_tokens, so the KV has to agree with the grid it is derived from. + if (m.patch_size <= 0 || m.vit_pixel_shuffle <= 0 || m.image_size % m.patch_size != 0 || + (m.image_size / m.patch_size) % m.vit_pixel_shuffle != 0) { + std::fprintf(stderr, "vla(gr00tn1d6): image %lld / patch %lld / shuffle %lld do not divide evenly\n", + (long long) m.image_size, (long long) m.patch_size, (long long) m.vit_pixel_shuffle); + return false; + } + { + const int64_t g2 = (m.image_size / m.patch_size) / m.vit_pixel_shuffle; + if (m.n_img_tokens != g2 * g2) { + std::fprintf(stderr, "vla(gr00tn1d6): n_img_tokens %lld does not match the %lldx%lld shuffled grid\n", + (long long) m.n_img_tokens, (long long) g2, (long long) g2); + return false; + } + } + cfg = Config{}; cfg.n_img = m.n_img_tokens; cfg.n_lang = m.max_seq_len; cfg.n_state = 1; cfg.n_suffix = m.action_horizon; cfg.max_state_dim = m.max_state_dim; cfg.max_action_dim = m.action_dim; diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index d707cd2..c75a4ae 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -243,6 +243,15 @@ bool load_config(const gguf_reader & g, Gr00tN1d7ModelArch & m, Config & cfg) { U(fk("num_inference_timesteps"), m.num_steps); U(fk("num_timestep_buckets"), m.num_buckets); U(fk("max_num_embodiments"), m.max_embodiments); U(fk("max_seq_len"), m.max_seq_len); U(fk("image_target_size"), m.image_target_size); + // merge_block_coords only enumerates the patch grid exactly when the spatial + // merge divides it; otherwise it emits rows past the position table. + if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size % m.patch_size != 0 || + (m.image_target_size / m.patch_size) % m.spatial_merge != 0) { + std::fprintf(stderr, "vla(gr00tn1d7): image %lld / patch %lld / merge %lld do not divide evenly\n", + (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); + return false; + } + if (const char * ns = std::getenv("VLA_NUM_STEPS")) { char * end = nullptr; long v = std::strtol(ns, &end, 10); if (end && *end == '\0' && v >= 1) { m.num_steps = (int64_t) v; std::fprintf(stderr, "vla(gr00tn1d7): VLA_NUM_STEPS override → num_steps=%lld\n", (long long) v); } diff --git a/src/models/qwen3vl_vit.h b/src/models/qwen3vl_vit.h index 6a48ef2..9abc09a 100644 --- a/src/models/qwen3vl_vit.h +++ b/src/models/qwen3vl_vit.h @@ -130,7 +130,10 @@ inline void interp_pos_embed(const std::vector & table, int64_t num_side, out.assign((size_t) S * hidden, 0.0f); auto src_coord = [&](int64_t k, int64_t g) -> double { return (g <= 1) ? 0.0 : (double) k * (double)(num_side - 1) / (double)(g - 1); }; for (int64_t s = 0; s < S; ++s) { - const double hy = src_coord(row[s], gh), wx = src_coord(col[s], gw); + // Clamped, not just h1/w1: a grid that the spatial merge does not divide + // pushes row/col past gh-1 and would index off the end of the table. + const double lim = (double) (num_side - 1); + const double hy = std::min(src_coord(row[s], gh), lim), wx = std::min(src_coord(col[s], gw), lim); const int64_t h0 = (int64_t) std::floor(hy), w0 = (int64_t) std::floor(wx); const int64_t h1 = std::min(h0 + 1, num_side - 1), w1 = std::min(w0 + 1, num_side - 1); const double dh = hy - h0, dw = wx - w0; diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index bd06d00..8fa03a5 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -91,20 +91,33 @@ struct safetensors { std::fprintf(stderr, "vla: shape mismatch for %s\n", name.c_str()); return false; } + if (info.dtype != "BF16" && info.dtype != "F32") { + std::fprintf(stderr, "vla: unsupported dtype for %s: %s\n", + name.c_str(), info.dtype.c_str()); + return false; + } + // dst is sized from expected_shape, so the declared span has to match it. + // Without this a file can name the right shape and a longer span. + const size_t elsz = (info.dtype == "BF16") ? sizeof(ggml_bf16_t) : sizeof(float); + size_t want = elsz; + for (const int64_t d : info.shape) { + if (d < 0) { std::fprintf(stderr, "vla: negative dim for %s\n", name.c_str()); return false; } + want *= (size_t) d; + } + if (info.off_end < info.off_begin || info.off_end - info.off_begin != want) { + std::fprintf(stderr, "vla: bad data_offsets for %s\n", name.c_str()); + return false; + } const size_t bytes = info.off_end - info.off_begin; file.seekg(data_blob_start + info.off_begin, std::ios::beg); if (info.dtype == "BF16") { std::vector tmp(bytes / sizeof(ggml_bf16_t)); file.read(reinterpret_cast(tmp.data()), bytes); ggml_bf16_to_fp32_row(tmp.data(), dst, tmp.size()); - } else if (info.dtype == "F32") { - file.read(reinterpret_cast(dst), bytes); } else { - std::fprintf(stderr, "vla: unsupported dtype for %s: %s\n", - name.c_str(), info.dtype.c_str()); - return false; + file.read(reinterpret_cast(dst), bytes); } - return true; + return !file.fail(); } bool read_raw(const std::string & name, void * dst, size_t expected_bytes, @@ -1532,8 +1545,8 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { const int64_t pad_start = cfg.n_img + in.n_lang; const int64_t pad_end = cfg.n_img + n_lang_max; - std::vector state_host(cfg.max_state_dim); - std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); + std::vector state_host(cfg.max_state_dim, 0.0f); + if (in.state) std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) { state_host[i] = (state_host[i] - m->state_mean[i]) / (m->state_std[i] + cfg.norm_eps); } @@ -1659,8 +1672,8 @@ std::vector predict_impl(SmolVLAModelArch* m, const Inputs& in) { ggml_tensor * pos_full = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); ggml_tensor * pos_rebased = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, cfg.n_suffix); - std::vector state_host(cfg.max_state_dim); - std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); + std::vector state_host(cfg.max_state_dim, 0.0f); + if (in.state) std::memcpy(state_host.data(), in.state, cfg.max_state_dim * sizeof(float)); for (int64_t i = 0; i < cfg.real_state_dim && i < cfg.max_state_dim; ++i) { state_host[i] = (state_host[i] - m->state_mean[i]) / (m->state_std[i] + cfg.norm_eps); } diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index e3884d7..7575d4b 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -172,6 +172,14 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path U("vla_adapter.action.head_dim",m->head_dim); F("vla_adapter.action.head_rope_base",m->head_rope_base); F("vla_adapter.action.ln_eps",m->head_ln_eps); U("vla_adapter.tokens.stop_id",m->stop_id); + // predict() taps one LM layer per head block, so head_blocks past lm_layers + // would read off the end of the layer-output vector. + if(m->lm_layers<1 || m->head_blocks<1 || m->head_blocks>m->lm_layers){ + std::fprintf(stderr,"vla(vla_adapter): head_blocks %lld outside [1, lm_layers %lld]\n", + (long long)m->head_blocks,(long long)m->lm_layers); + return nullptr; + } + if (g.has("vla_adapter.statistics_json")) { if (!parse_stats(g.str("vla_adapter.statistics_json"), m->action_dim, m->q01, m->q99, m->unnorm_mask, m->suite)) { std::fprintf(stderr, "vla(vla_adapter): failed to parse statistics_json\n"); return nullptr; } diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index dfa81b0..10dc798 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -178,6 +178,20 @@ bool load_config(const gguf_reader & g, VlaJepaModelArch & m, Config & cfg) { F(fk("vit_rope_theta"), m.vit_rope_base); F(fk("dit_ln_eps"), m.dit_ln_eps); F(fk("dit_norm_out_eps"), m.dit_norm_out_eps); if (g.has(fk("lm_rope_theta"))) m.lm_rope_base = (float) g.f64(fk("lm_rope_theta")); + // merge_block_coords only enumerates the patch grid exactly when the spatial + // merge divides it; otherwise it emits rows past the position table. + if (m.patch_size <= 0 || m.spatial_merge <= 0 || m.image_target_size % m.patch_size != 0 || + (m.image_target_size / m.patch_size) % m.spatial_merge != 0) { + std::fprintf(stderr, "vla(vla_jepa): image %lld / patch %lld / merge %lld do not divide evenly\n", + (long long) m.image_target_size, (long long) m.patch_size, (long long) m.spatial_merge); + return false; + } + // timesteps_proj always emits 256 floats into the time-projection input. + if (m.time_proj_dim != 256) { + std::fprintf(stderr, "vla(vla_jepa): time_proj_dim %lld, expected 256\n", (long long) m.time_proj_dim); + return false; + } + cfg = Config{}; cfg.n_img = (m.image_target_size / m.patch_size / m.spatial_merge) * (m.image_target_size / m.patch_size / m.spatial_merge); cfg.n_lang = 1024; cfg.n_state = 1; From 9553cdb7af0d4b6b1fbcae5c84ca95ebed9b9330 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:06:40 +0700 Subject: [PATCH 26/42] stop a stalled peer from parking the servers Both servers drained multipart frames with a blocking recv, so one client that announced a frame and went quiet held the loop forever and shutdown never ran. A 5s receive timeout bounds it and the drain now reports the stall instead of replying into a socket that is still mid-message. Also reject a cache path containing a quote and a repo id starting with a slash: both reach the hf download shell command through VLA_CACHE or HOME. --- src/serving/hf_fetch.h | 11 ++++++++++- src/serving/server.cpp | 28 ++++++++++++++++++++-------- src/serving/vlm-server.cpp | 3 +++ 3 files changed, 33 insertions(+), 9 deletions(-) diff --git a/src/serving/hf_fetch.h b/src/serving/hf_fetch.h index d4f03e5..cfe6c33 100644 --- a/src/serving/hf_fetch.h +++ b/src/serving/hf_fetch.h @@ -27,7 +27,9 @@ namespace vla { // Repo ids reach a shell command, so reject anything outside this set. inline bool hf_token_ok(const std::string & s, bool allow_slash) { if (s.empty() || s.size() > 200) return false; - if (s.front() == '-' || s.find("..") != std::string::npos) return false; + // A leading '/' would make fs::path join replace the cache root instead of + // extending it, putting the download anywhere on disk. + if (s.front() == '-' || s.front() == '/' || s.find("..") != std::string::npos) return false; for (const char c : s) { const bool ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9') || c == '.' || c == '_' || c == '-' || @@ -92,6 +94,13 @@ inline std::string hf_resolve(const std::string & spec) { return ""; } + // The cache root reaches the shell too, and it comes from VLA_CACHE or HOME. + // A single quote in either would close the quoting and run the rest. + if (dir.string().find('\'') != std::string::npos) { + std::fprintf(stderr, "vla: refusing a cache path containing a quote: %s\n", dir.string().c_str()); + return ""; + } + std::string cmd = "hf download " + repo; if (!file.empty()) cmd += " " + file; cmd += " --local-dir '" + dir.string() + "'"; diff --git a/src/serving/server.cpp b/src/serving/server.cpp index 33f0610..63bcb23 100644 --- a/src/serving/server.cpp +++ b/src/serving/server.cpp @@ -150,16 +150,19 @@ std::string make_error_response(uint64_t request_id, const std::string & msg) { return resp.SerializeAsString(); } -// Discard frames after the first; true means the request was malformed. Must run -// to completion: a queued frame keeps REP in receive state and send throws EFSM. -bool drain_extra_frames(zmq::socket_t & sock) { - bool extra = false; +// Discard frames after the first. Must run to completion: a queued frame keeps +// REP in receive state and send throws EFSM. Stalled means the peer announced a +// frame it never sent, so no reply is possible until the rest arrives. +enum class Drain { Clean, Extra, Stalled }; + +Drain drain_extra_frames(zmq::socket_t & sock) { + Drain d = Drain::Clean; while (sock.get(zmq::sockopt::rcvmore)) { zmq::message_t junk; - if (!sock.recv(junk, zmq::recv_flags::none)) break; - extra = true; + if (!sock.recv(junk, zmq::recv_flags::none)) return Drain::Stalled; + d = Drain::Extra; } - return extra; + return d; } int find_non_finite(const float * data, int n) { @@ -276,6 +279,9 @@ int main(int argc, char ** argv) { // 64 MiB is above any real request (16 views of 512x512 F32 RGB is ~50 MiB) and // low enough to bound protobuf's expansion during ParseFromArray. sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); + // A peer that sends a frame with SNDMORE and then stalls would otherwise park + // this single-threaded loop in recv for good, starving every other client. + sock.set(zmq::sockopt::rcvtimeo, 5000); sock.bind(bind_addr); std::printf("vla-server: bound to %s. ready.\n", bind_addr.c_str()); @@ -331,7 +337,13 @@ int main(int argc, char ** argv) { // Without this an unauthenticated client shuts the server down with one // two-frame request: the reply fails and send_reply sets g_shutdown. - if (drain_extra_frames(sock)) { + const Drain drained = drain_extra_frames(sock); + if (drained == Drain::Stalled) { + // Back to the poll rather than blocking here, so shutdown still works. + std::fprintf(stderr, "vla-server: peer stalled mid-request\n"); + continue; + } + if (drained == Drain::Extra) { send_reply(make_error_response(0, "expected a single-frame request")); continue; } diff --git a/src/serving/vlm-server.cpp b/src/serving/vlm-server.cpp index a7880d5..2f8f75f 100644 --- a/src/serving/vlm-server.cpp +++ b/src/serving/vlm-server.cpp @@ -112,6 +112,9 @@ int main(int argc, char ** argv) { sock.set(zmq::sockopt::linger, 0); // Per frame only; the recv loop caps the multipart total. sock.set(zmq::sockopt::maxmsgsize, int64_t(64) * 1024 * 1024); + // A peer that sends a frame with SNDMORE and then stalls would otherwise park + // this single-threaded loop in recv for good, starving every other client. + sock.set(zmq::sockopt::rcvtimeo, 5000); sock.bind(bind_addr); std::printf("vlm-server: bound to %s. ready.\n", bind_addr.c_str()); From 3982914a42f13975d6ec4b4ecb60ace373a47b64 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:19:40 +0700 Subject: [PATCH 27/42] drop the write-only backend flags Backend::is_cuda and is_gpu were copied into every arch and never read back. Anything that needs the device type can ask ggml_backend_get_device. --- src/backend.h | 8 +------- src/models/bitvla.cpp | 1 - src/models/evo1.cpp | 4 ---- src/models/gr00tn1d5.cpp | 4 ---- src/models/gr00tn1d6.cpp | 4 ---- src/models/gr00tn1d7.cpp | 4 ---- src/models/openvla_oft.cpp | 3 +-- src/models/pi0.cpp | 4 ---- src/models/pi05.cpp | 4 ---- src/models/smolvla.cpp | 4 ---- src/models/vla_adapter.cpp | 3 +-- src/models/vla_jepa.cpp | 4 ---- 12 files changed, 3 insertions(+), 44 deletions(-) diff --git a/src/backend.h b/src/backend.h index 494bb7f..a4d89bb 100644 --- a/src/backend.h +++ b/src/backend.h @@ -68,9 +68,7 @@ inline void setenv_default(const char * key, const char * val) { /// Outcome of @ref backend_init. @c handle is null only if even the CPU backend /// failed to come up, which callers treat as a fatal load error. struct Backend { - ggml_backend_t handle = nullptr; - bool is_cuda = false; - bool is_gpu = false; + ggml_backend_t handle = nullptr; }; /// GPU ordinal for CUDA and SYCL; `VLA_DEVICE` overrides. Junk is rejected, not @@ -102,8 +100,6 @@ inline Backend backend_init(const char * tag, int n_threads) { const int dev = backend_device_index(); b.handle = ggml_backend_cuda_init(dev); if (b.handle) { - b.is_cuda = true; - b.is_gpu = true; std::printf("%s: backend = CUDA (device %d)\n", tag, dev); } else { std::fprintf(stderr, "%s: ggml_backend_cuda_init failed; falling back to CPU\n", tag); @@ -134,7 +130,6 @@ inline Backend backend_init(const char * tag, int n_threads) { std::fprintf(stderr, "%s: SYCL device %d out of range (%d visible); falling back to CPU\n", tag, dev, n_dev); } else if ((b.handle = ggml_backend_sycl_init(dev)) != nullptr) { - b.is_gpu = true; char desc[256] = { 0 }; ggml_backend_sycl_get_device_description(dev, desc, sizeof(desc)); std::printf("%s: backend = SYCL (device %d: %s)\n", tag, dev, desc); @@ -146,7 +141,6 @@ inline Backend backend_init(const char * tag, int n_threads) { { b.handle = ggml_backend_metal_init(); if (b.handle) { - b.is_gpu = true; std::printf("%s: backend = Metal\n", tag); } else { std::fprintf(stderr, "%s: ggml_backend_metal_init failed; falling back to CPU\n", tag); diff --git a/src/models/bitvla.cpp b/src/models/bitvla.cpp index 16f78d5..b8a99e2 100644 --- a/src/models/bitvla.cpp +++ b/src/models/bitvla.cpp @@ -98,7 +98,6 @@ struct BitvlaModelArch : public ModelArchBase { gguf_reader emb_reader{"bitvla"}; // stays open for per-step token-embedding row fetches std::vector stop_embed; // cached constant stop-token embedding row ggml_backend_t backend = nullptr; - bool is_cuda = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index a073ba0..48c813f 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -58,8 +58,6 @@ struct Evo1ModelArch : public ModelArchBase { // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"evo1"}; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -292,8 +290,6 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, const Backend b = backend_init("vla(evo1)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 50826b4..5ea0b16 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -56,8 +56,6 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"gr00tn1d5"}; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -271,8 +269,6 @@ std::unique_ptr gr00t_n1_5_create(const std::string& mmproj_path, const Backend b = backend_init("vla(gr00tn1d5)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 22d332b..18062ec 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -54,8 +54,6 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"gr00tn1d6"}; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -308,8 +306,6 @@ std::unique_ptr gr00t_n1_6_create(const std::string& mmproj_path, const Backend b = backend_init("vla(gr00tn1d6)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index c75a4ae..e1e6bf0 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -55,8 +55,6 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { std::string gguf_path; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -330,8 +328,6 @@ std::unique_ptr gr00t_n1_7_create(const std::string& mmproj_path, const Backend b = backend_init("vla(gr00tn1d7)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 363b73e..01cddaa 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -91,7 +91,7 @@ struct OpenVlaOftModelArch : public ModelArchBase { } ggml_backend_t backend = nullptr; - bool is_gpu = false; int n_threads = default_cpu_threads(); + int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; scratch_ctx main_scratch; @@ -163,7 +163,6 @@ std::unique_ptr openvla_oft_create(const std::string& mmproj_path const Backend b = backend_init("vla(openvla_oft)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 9749902..5164341 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -92,8 +92,6 @@ struct Pi0ModelArch : public ModelArchBase { std::vector predict(const Inputs& in) override; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -348,8 +346,6 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, const Backend b = backend_init("vla(pi0)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 30918fb..301223f 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -107,8 +107,6 @@ struct Pi05ModelArch : public ModelArchBase { std::vector predict(const Inputs& in) override; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -429,8 +427,6 @@ std::unique_ptr pi05_create(const std::string& mmproj_path, const Backend b = backend_init("vla(pi05)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 8fa03a5..849ad8d 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -299,8 +299,6 @@ struct SmolVLAModelArch : public ModelArchBase { ggml_backend_t backend = nullptr; ggml_backend_buffer_t weight_buf = nullptr; - bool is_cuda = false; - bool is_gpu = false; ggml_type weight_dtype = GGML_TYPE_BF16; @@ -925,8 +923,6 @@ SmolVLAModelArch* smolvla_load_impl(const std::string& mmproj_path, const Backend b = backend_init("vla", default_cpu_threads()); if (!b.handle) { delete m; return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } vram_probe(m->backend, "after backend init"); diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 7575d4b..fbd47de 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -92,7 +92,7 @@ struct VlaAdapterModelArch : public ModelArchBase { } ggml_backend_t backend = nullptr; - bool is_gpu = false; int n_threads = default_cpu_threads(); + int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; scratch_ctx main_scratch; @@ -190,7 +190,6 @@ std::unique_ptr vla_adapter_create(const std::string& mmproj_path const Backend b = backend_init("vla(vla_adapter)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t)64*1024*1024, nullptr, true }; diff --git a/src/models/vla_jepa.cpp b/src/models/vla_jepa.cpp index 10dc798..a416366 100644 --- a/src/models/vla_jepa.cpp +++ b/src/models/vla_jepa.cpp @@ -53,8 +53,6 @@ struct VlaJepaModelArch : public ModelArchBase { std::string gguf_path; ggml_backend_t backend = nullptr; - bool is_cuda = false; - bool is_gpu = false; int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; @@ -241,8 +239,6 @@ std::unique_ptr vla_jepa_create(const std::string& mmproj_path, const Backend b = backend_init("vla(vla_jepa)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; - m->is_cuda = b.is_cuda; - m->is_gpu = b.is_gpu; } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, nullptr, true }; From eca483d44cf3824089f5b79ae7d723bb75efbffc Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:19:40 +0700 Subject: [PATCH 28/42] fix the c api load leak and two doc errors vla_model_load leaked the whole engine if the handle allocation threw. The sycl guide named GGML_SYCL_DISABLE_DNN, which does not exist; ggml reads GGML_SYCL_ENABLE_DNN. The chat example linked a file that was never written. --- docs/backend/sycl.md | 2 +- examples/chat/README.md | 6 +++--- src/vla_c_api.cpp | 6 +++++- 3 files changed, 9 insertions(+), 5 deletions(-) diff --git a/docs/backend/sycl.md b/docs/backend/sycl.md index 147b2ed..9017f93 100644 --- a/docs/backend/sycl.md +++ b/docs/backend/sycl.md @@ -143,7 +143,7 @@ OpenVLA-OFT. `vla.cpp` defaults `GGML_SYCL_ENABLE_VMM=0` when it brings up SYCL, which avoids it and is the faster of the two workarounds (disabling oneDNN with -`GGML_SYCL_DISABLE_DNN=1` also clears the crash, but costs ~8%). It is only a +`GGML_SYCL_ENABLE_DNN=0` also clears the crash, but costs ~8%). It is only a default: set `GGML_SYCL_ENABLE_VMM=1` explicitly to keep the pool on hardware where it pays off. diff --git a/examples/chat/README.md b/examples/chat/README.md index 569c2f4..40a60ca 100644 --- a/examples/chat/README.md +++ b/examples/chat/README.md @@ -2,9 +2,9 @@ A minimal **streaming image+text chat** client for `vlm-server`, the llama.cpp + libmtmd chat runtime (`src/vlm/engine.cpp`) behind a ZMQ daemon. Send text and -images, get a streamed reply. The design rationale lives in -[docs/VLM-SERVER.md](../../docs/VLM-SERVER.md); this README is how to **run** it, -plus the validation numbers for the SmolVLM2-500M-Instruct setup. +images, get a streamed reply. The layering is sketched in +[docs/ARCHITECTURE.md](../../docs/ARCHITECTURE.md); this README is how to **run** +it, plus the validation numbers for the SmolVLM2-500M-Instruct setup. ``` examples/chat/ diff --git a/src/vla_c_api.cpp b/src/vla_c_api.cpp index 03b5f9f..74c10bc 100644 --- a/src/vla_c_api.cpp +++ b/src/vla_c_api.cpp @@ -20,6 +20,7 @@ #include #include +#include #include namespace { @@ -50,8 +51,11 @@ vla_model * vla_model_load(const char * mmproj_path, const char * ckpt_path, ckpt_path, config_path ? config_path : ""); if (!m) return nullptr; + // Owns the engine until the handle exists, so a throwing new does not + // strand the whole model. + std::unique_ptr guard(m, vla::model_free); auto * h = new vla_model(); - h->m = m; + h->m = guard.release(); return h; } catch (...) { return nullptr; From c70fddc48817362215185df1db6eb8ffc8f53802 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:19:46 +0700 Subject: [PATCH 29/42] bump llama.cpp to b10331 Numerics: GR00T N1.5 and N1.6 move by up to 4.6e-4 on actions peaking near 0.87 (mean 6e-5), from an upstream ggml kernel change on the SigLIP tower path the two share. The other nine archs are bit-identical. Also drops GGML_CUDA_GRAPHS from the Dockerfile, which llama.cpp already defaults on. --- .github/workflows/build.yml | 2 +- CMakeLists.txt | 2 +- Dockerfile | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 3cf9d82..a9334e1 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -51,7 +51,7 @@ jobs: - uses: actions/cache@v4 with: path: build/_deps - key: llama-b10326-${{ runner.os }} + key: llama-b10331-${{ runner.os }} # Everything, not a target list: ctest registers tests this job must build, # and a named list goes stale the next time one is added. - name: build + ctest (CPU, -Wall -Wextra) diff --git a/CMakeLists.txt b/CMakeLists.txt index 615e1e0..aaecbbe 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -36,7 +36,7 @@ set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) include(FetchContent) FetchContent_Declare(llama GIT_REPOSITORY https://github.com/ggml-org/llama.cpp - GIT_TAG b10326 + GIT_TAG b10331 GIT_SHALLOW TRUE ) FetchContent_MakeAvailable(llama) diff --git a/Dockerfile b/Dockerfile index 58328f6..f703649 100644 --- a/Dockerfile +++ b/Dockerfile @@ -26,7 +26,7 @@ ARG JOBS= RUN set -eux; \ if [ "$BACKEND" = "cuda" ]; then \ export LIBRARY_PATH=/usr/local/cuda/lib64/stubs; \ - cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON -DGGML_CUDA_GRAPHS=ON \ + cmake -B build -DCMAKE_BUILD_TYPE=Release -DGGML_CUDA=ON \ -DCMAKE_CUDA_ARCHITECTURES="${CUDA_ARCH}" \ -DCMAKE_SHARED_LINKER_FLAGS="-lcuda" -DCMAKE_EXE_LINKER_FLAGS="-lcuda"; \ else \ From 7df64e091248425e4da01eca119e27cb9f90565e Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:19:46 +0700 Subject: [PATCH 30/42] release 0.2.0 --- CHANGELOG.md | 23 +++++++++++++++++++++++ bindings/python/pyproject.toml | 2 +- pyproject.toml | 2 +- 3 files changed, 25 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a043125..26c52a4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,28 @@ Notable changes to vla.cpp. Format loosely follows [Keep a Changelog](https://keepachangelog.com). +## [0.2.0] - 2026-08-09 + +### Added +- SYCL backend for Intel GPUs (Arc, Flex, Data Center Max, Xe iGPU). `VLA_DEVICE` picks the ordinal on CUDA and SYCL alike. See `docs/backend/sycl.md`. +- Stable C ABI (`include/vla.h`, `libvla`) and Python bindings over it (`bindings/python`). +- Four more architectures: π0.5, VLA-Adapter, OpenVLA-OFT and VLA-JEPA. +- `vla-bench` for engine-only latency, and `-hf user/repo[:file.gguf]` to fetch a checkpoint on first use. +- `vla-cli --text`, tokenized by `scripts/tokenize_prompt.py` with the tokenizer the architecture was trained on. +- Release workflow publishing Linux x86-64 (CPU and CUDA), Linux aarch64 (CPU), macOS Metal and a GHCR image. + +### Changed +- One shared backend ladder (`src/backend.h`) instead of a copy per arch. CMake rejects two accelerators in one build directory. +- Shared headers for the Qwen3-VL tower, the DINOv2+SigLIP dual tower, the DiT time embeddings, the causal mask and CHW image preprocessing. +- `vla::graph_cache` keeps the compute graph across `predict` calls in nine architectures, not just GR00T N1.7. Output is unchanged. +- llama.cpp pinned at b10331. GR00T N1.5 and N1.6 shift by up to 4.6e-4 on actions peaking near 0.87, from an upstream ggml kernel change in the SigLIP tower they share. The other nine architectures are bit-identical. + +### Fixed +- Reject checkpoint geometry that contradicts itself before it sizes a buffer, in smolvla, bitvla, gr00tn1d6, vla_adapter and the Qwen3-VL position resample. +- A peer that stalls mid-message no longer parks either server. +- Treat a missing state vector as zeros in every architecture rather than dereferencing it. +- Build every registered test before `ctest`, so the four that were never built stop reporting as not run. + ## [0.1.1] - 2026-07-04 ### Added @@ -40,5 +62,6 @@ expert + dataset stats), CPU or CUDA, no external mmproj and no patch to llama.c - llama.cpp is fetched + pinned via CMake `FetchContent` (tag `b9866`); bumping is a one-line `GIT_TAG` change. Removed the `patches/` fetch script. +[0.2.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.2.0 [0.1.1]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.1.1 [0.1.0]: https://github.com/VinRobotics/vla.cpp/releases/tag/v0.1.0 diff --git a/bindings/python/pyproject.toml b/bindings/python/pyproject.toml index 0202677..af666ad 100644 --- a/bindings/python/pyproject.toml +++ b/bindings/python/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "vla-cpp" -version = "0.1.1" +version = "0.2.0" description = "Python bindings for vla.cpp, a C++ inference engine for Vision-Language-Action models." readme = "README.md" requires-python = ">=3.9" diff --git a/pyproject.toml b/pyproject.toml index 14ed4b2..eb1700b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "vla-cpp-tooling" -version = "0.1.1" +version = "0.2.0" description = "Python tooling for vla.cpp: HuggingFace -> GGUF converters and the ZeroMQ eval client." readme = "README.md" requires-python = ">=3.10" From 1530eb0fee0029c20ca13874184f5e1258c9eaf7 Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 17:37:34 +0700 Subject: [PATCH 31/42] share one graph cache across the ggml archs Only gr00tn1d7 kept its compute graph between predict calls; the rest rebuilt theirs every time and reused just the arena. graph_cache in scratch_ctx.h holds the context, allocator, graph, shape key and input handles, so an arch supplies only its key and a build lambda. gr00tn1d7 moves onto it and evo1, pi0, pi05, gr00tn1d5, gr00tn1d6, vla_adapter, openvla_oft and vla_jepa gain it. A stable graph is also what lets ggml-cuda capture and replay. Dump modes still rebuild, since they add graph outputs. All eleven archs verified bit-identical with vla_predict_check. --- src/models/evo1.cpp | 34 +++++++++++++++----- src/models/gr00tn1d5.cpp | 32 +++++++++++++++---- src/models/gr00tn1d6.cpp | 36 +++++++++++++++++---- src/models/gr00tn1d7.cpp | 60 +++++++++++++++++------------------ src/models/openvla_oft.cpp | 28 ++++++++++++++--- src/models/pi0.cpp | 40 ++++++++++++++++++------ src/models/pi05.cpp | 38 ++++++++++++++++------ src/models/scratch_ctx.h | 53 +++++++++++++++++++++++++++++++ src/models/vla_adapter.cpp | 32 ++++++++++++++++--- src/models/vla_jepa.cpp | 64 ++++++++++++++++++++++++++++++++------ 10 files changed, 329 insertions(+), 88 deletions(-) diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 48c813f..0b9d0fa 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -61,7 +61,16 @@ struct Evo1ModelArch : public ModelArchBase { int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t seq=-1, nsteps=-1; + bool operator==(const MainKey & o) const { return seq==o.seq && nsteps==o.nsteps; } + }; + struct MainIO { + ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_qmask=nullptr; + ggml_tensor *t_state=nullptr,*t_x=nullptr,*t_amask=nullptr,*x_action=nullptr; + }; + graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_BF16; @@ -538,9 +547,10 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { } } - ggml_context * C = main_scratch.reset((size_t) 96 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(evo1): ggml_init(ctx_compute) failed\n"); return {}; } - + // LM + DiT graph depends only on the padded length and step count. + const MainKey mkey{ SEQ, num_steps }; + const bool built = main_graph.ensure(backend, mkey, (size_t) 96 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { const int64_t E = embed_dim, hd_dit = E / dit_heads; const float scale_dit = 1.0f / std::sqrt((float) hd_dit); const int64_t Nctx = SEQ + 1; @@ -623,13 +633,21 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_set_name(x_action, "x_final"); ggml_set_output(x_action); + gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_qmask=t_qmask; + gio.t_state=t_state; gio.t_x=t_x; gio.t_amask=t_amask; gio.x_action=x_action; + ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(gf, x_action); + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(evo1): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask; + ggml_tensor * t_qmask = gio.t_qmask, * t_state = gio.t_state, * t_x = gio.t_x; + ggml_tensor * t_amask = gio.t_amask, * x_action = gio.x_action; - if (!main_scratch.alloc(backend, gf)) { - std::fprintf(stderr, "vla(evo1): ggml_gallocr_alloc_graph failed\n"); - return {}; - } ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { std::vector pp(SEQ); for (int64_t i = 0; i < SEQ; ++i) pp[i] = (int32_t) i; ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } { std::vector mk((size_t) SEQ * SEQ); const float NEG = -std::numeric_limits::infinity(); diff --git a/src/models/gr00tn1d5.cpp b/src/models/gr00tn1d5.cpp index 5ea0b16..45c946f 100644 --- a/src/models/gr00tn1d5.cpp +++ b/src/models/gr00tn1d5.cpp @@ -59,7 +59,16 @@ struct Gr00tN1d5ModelArch : public ModelArchBase { int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t seq=-1, nsteps=-1; + bool operator==(const MainKey & o) const { return seq==o.seq && nsteps==o.nsteps; } + }; + struct MainIO { + ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr,*actions=nullptr; + std::vector t_tau, t_tproj; + }; + graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -438,9 +447,10 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { if (in.noise) std::memcpy(x_init.data(), in.noise, x_init.size() * sizeof(float)); else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } - ggml_context * C = main_scratch.reset((size_t) 128 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(gr00tn1d5): ggml_init(ctx_compute) failed\n"); return {}; } - + // LM + VLSA + DiT graph depends only on the padded length and step count. + const MainKey mkey{ SEQ, num_steps }; + const bool built = main_graph.ensure(backend, mkey, (size_t) 128 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, SEQ); ggml_set_input(t_pos); ggml_tensor * t_lmmask = ggml_new_tensor_2d(C, GGML_TYPE_F32, SEQ, SEQ); ggml_set_input(t_lmmask); @@ -498,10 +508,20 @@ std::vector Gr00tN1d5ModelArch::predict(const Inputs& in) { } ggml_set_name(actions, "action_pred"); ggml_set_output(actions); + gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; + gio.t_x0=t_x0; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + ggml_cgraph * gf = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(gf, actions); - - if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(gr00tn1d5): gallocr alloc failed\n"); return {}; } + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(gr00tn1d5): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask; + ggml_tensor * t_state = gio.t_state, * t_x0 = gio.t_x0, * actions = gio.actions; + std::vector & t_tau = gio.t_tau; std::vector & t_tproj = gio.t_tproj; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { std::vector pp(SEQ); for (int64_t i = 0; i < SEQ; ++i) pp[i] = (int32_t) i; ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } diff --git a/src/models/gr00tn1d6.cpp b/src/models/gr00tn1d6.cpp index 18062ec..aff2f29 100644 --- a/src/models/gr00tn1d6.cpp +++ b/src/models/gr00tn1d6.cpp @@ -58,7 +58,19 @@ struct Gr00tN1d6ModelArch : public ModelArchBase { ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; scratch_ctx merge_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; + bool operator==(const MainKey & o) const { + return seq==o.seq && n_img==o.n_img && seq_txt==o.seq_txt && nsteps==o.nsteps; + } + }; + struct MainIO { + ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr; + ggml_tensor *t_img_idx=nullptr,*t_txt_idx=nullptr,*actions=nullptr; + std::vector t_tau, t_tproj; + }; + graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -507,9 +519,10 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { if (in.noise) std::memcpy(x_init.data(), in.noise, x_init.size() * sizeof(float)); else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } - ggml_context * C = main_scratch.reset((size_t) 256 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(gr00tn1d6): ggml_init(ctx_compute) failed\n"); return {}; } - + // LM + DiT graph depends only on the sequence split and step count. + const MainKey mkey{ SEQ, n_img, SEQ_TXT, num_steps }; + const bool built = main_graph.ensure(backend, mkey, (size_t) 256 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, SEQ); ggml_set_input(t_pos); ggml_tensor * t_lmmask = ggml_new_tensor_2d(C, GGML_TYPE_F32, SEQ, SEQ); ggml_set_input(t_lmmask); @@ -577,10 +590,21 @@ std::vector Gr00tN1d6ModelArch::predict(const Inputs& in) { } ggml_set_name(actions, "action_pred"); ggml_set_output(actions); + gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; gio.t_x0=t_x0; + gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); ggml_build_forward_expand(gf, actions); - - if (!main_scratch.alloc(backend, gf)) { std::fprintf(stderr, "vla(gr00tn1d6): gallocr alloc failed\n"); return {}; } + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(gr00tn1d6): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask; + ggml_tensor * t_state = gio.t_state, * t_x0 = gio.t_x0; + ggml_tensor * t_img_idx = gio.t_img_idx, * t_txt_idx = gio.t_txt_idx, * actions = gio.actions; + std::vector & t_tau = gio.t_tau; std::vector & t_tproj = gio.t_tproj; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { std::vector pp(SEQ); for (int64_t i = 0; i < SEQ; ++i) pp[i] = (int32_t) i; ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index e1e6bf0..bbe7f57 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -99,16 +99,20 @@ struct Gr00tN1d7ModelArch : public ModelArchBase { gguf_reader io; bool build_caches(); - struct MainGraph { - ggml_context * C = nullptr; ggml_gallocr_t galloc = nullptr; ggml_cgraph * gf = nullptr; + struct MainKey { + int64_t seq=-1, n_img=-1, seq_txt=-1, nsteps=-1; bool deepstack=false; + bool operator==(const MainKey & o) const { + return seq==o.seq && n_img==o.n_img && seq_txt==o.seq_txt && + nsteps==o.nsteps && deepstack==o.deepstack; + } + }; + struct MainIO { ggml_tensor *t_embeds=nullptr,*t_pos=nullptr,*t_lmmask=nullptr,*t_state=nullptr,*t_x0=nullptr; ggml_tensor *t_ds[3]={nullptr,nullptr,nullptr}; ggml_tensor *t_img_idx=nullptr,*t_txt_idx=nullptr,*actions=nullptr; std::vector t_tau, t_tproj; - int64_t seq=-1,n_img=-1,seq_txt=-1,nsteps=-1; bool deepstack=false; bool valid=false; - void release() { if (galloc) ggml_gallocr_free(galloc); if (C) ggml_free(C); - galloc=nullptr; C=nullptr; gf=nullptr; valid=false; t_tau.clear(); t_tproj.clear(); } - } mg; + }; + graph_cache mg; std::vector predict(const Inputs& in) override; }; @@ -629,19 +633,16 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { else { std::mt19937 rng((uint32_t) std::chrono::steady_clock::now().time_since_epoch().count()); std::normal_distribution nd(0.f, 1.f); for (auto & v : x_init) v = nd(rng); } // On by default: 16% faster, bit-identical. Set VLA_GR00T_GRAPH_CACHE=0 to opt out. + // Dumping adds graph outputs, so it always rebuilds. const char * gc = std::getenv("VLA_GR00T_GRAPH_CACHE"); const bool use_cache = (!gc || std::strcmp(gc, "0") != 0) && !do_dump; - const bool reuse = use_cache && mg.valid && mg.seq == SEQ && mg.n_img == n_img && - mg.seq_txt == SEQ_TXT && mg.nsteps == num_steps && mg.deepstack == inject_deepstack; + if (!use_cache) mg.release(); ggml_tensor * eagle = nullptr, * vl_embs = nullptr; std::vector lm_h_dump, vlsa_dump; - if (!reuse) { - if (mg.C) mg.release(); - ggml_init_params cp = { (size_t) 256 * 1024 * 1024, nullptr, true }; - ggml_context * C = ggml_init(cp); - if (!C) { std::fprintf(stderr, "vla(gr00tn1d7): ggml_init(ctx_compute) failed\n"); return {}; } - + const MainKey mkey{ SEQ, n_img, SEQ_TXT, num_steps, inject_deepstack }; + const bool built = mg.ensure(backend, mkey, (size_t) 256 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos = ggml_new_tensor_1d(C, GGML_TYPE_I32, 4 * SEQ); ggml_set_input(t_pos); ggml_tensor * t_lmmask = ggml_new_tensor_2d(C, GGML_TYPE_F32, SEQ, SEQ); ggml_set_input(t_lmmask); @@ -719,25 +720,22 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { } ggml_set_name(actions, "action_pred"); ggml_set_output(actions); + gio.t_embeds=t_embeds; gio.t_pos=t_pos; gio.t_lmmask=t_lmmask; gio.t_state=t_state; gio.t_x0=t_x0; + gio.t_ds[0]=t_ds[0]; gio.t_ds[1]=t_ds[1]; gio.t_ds[2]=t_ds[2]; + gio.t_img_idx=t_img_idx; gio.t_txt_idx=t_txt_idx; gio.t_tau=t_tau; gio.t_tproj=t_tproj; gio.actions=actions; + ggml_cgraph * gf = ggml_new_graph_custom(C, 65536, false); ggml_build_forward_expand(gf, actions); - - ggml_gallocr_t galloc = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); - if (!galloc || !ggml_gallocr_alloc_graph(galloc, gf)) { std::fprintf(stderr, "vla(gr00tn1d7): gallocr alloc failed\n"); if (galloc) ggml_gallocr_free(galloc); ggml_free(C); return {}; } - - mg.C=C; mg.galloc=galloc; mg.gf=gf; - mg.t_embeds=t_embeds; mg.t_pos=t_pos; mg.t_lmmask=t_lmmask; mg.t_state=t_state; mg.t_x0=t_x0; - mg.t_ds[0]=t_ds[0]; mg.t_ds[1]=t_ds[1]; mg.t_ds[2]=t_ds[2]; - mg.t_img_idx=t_img_idx; mg.t_txt_idx=t_txt_idx; mg.t_tau=t_tau; mg.t_tproj=t_tproj; mg.actions=actions; - mg.seq=SEQ; mg.n_img=n_img; mg.seq_txt=SEQ_TXT; mg.nsteps=num_steps; mg.deepstack=inject_deepstack; - mg.valid = use_cache; - } - - ggml_context * C = mg.C; ggml_cgraph * gf = mg.gf; ggml_gallocr_t galloc = mg.galloc; (void) C; (void) galloc; - ggml_tensor * t_embeds = mg.t_embeds, * t_pos = mg.t_pos, * t_lmmask = mg.t_lmmask, * t_state = mg.t_state, * t_x0 = mg.t_x0; - ggml_tensor * t_ds[3] = { mg.t_ds[0], mg.t_ds[1], mg.t_ds[2] }; - ggml_tensor * t_img_idx = mg.t_img_idx, * t_txt_idx = mg.t_txt_idx, * actions = mg.actions; - std::vector & t_tau = mg.t_tau; std::vector & t_tproj = mg.t_tproj; + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(gr00tn1d7): main graph build failed\n"); return {}; } + + MainIO & gio = mg.io(); + ggml_cgraph * gf = mg.graph(); + ggml_tensor * t_embeds = gio.t_embeds, * t_pos = gio.t_pos, * t_lmmask = gio.t_lmmask, * t_state = gio.t_state, * t_x0 = gio.t_x0; + ggml_tensor * t_ds[3] = { gio.t_ds[0], gio.t_ds[1], gio.t_ds[2] }; + ggml_tensor * t_img_idx = gio.t_img_idx, * t_txt_idx = gio.t_txt_idx, * actions = gio.actions; + std::vector & t_tau = gio.t_tau; std::vector & t_tproj = gio.t_tproj; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); { diff --git a/src/models/openvla_oft.cpp b/src/models/openvla_oft.cpp index 01cddaa..4650272 100644 --- a/src/models/openvla_oft.cpp +++ b/src/models/openvla_oft.cpp @@ -94,7 +94,15 @@ struct OpenVlaOftModelArch : public ModelArchBase { int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t seq=-1, n_views=-1, n_lang=-1; + bool operator==(const MainKey & o) const { return seq==o.seq && n_views==o.n_views && n_lang==o.n_lang; } + }; + struct MainIO { + ggml_tensor *t_ids=nullptr,*t_state=nullptr,*t_proj=nullptr,*act0=nullptr,*t_pos=nullptr,*norm_actions=nullptr; + }; + graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; @@ -303,8 +311,10 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { const int64_t ACT_START = NUM_PATCHES + NUM_PROMPT_TOKENS; const int64_t SEQ = 1 + NUM_PATCHES + (L-1) + n_act + 1; const auto ti=clock::now(); - ggml_context*C=main_scratch.reset((size_t)256*1024*1024); - + // LM + action head graph depends only on the sequence layout. + const MainKey mkey{ SEQ, n_views, L }; + const bool built = main_graph.ensure(backend, mkey, (size_t)256*1024*1024, + [&](ggml_context*C, MainIO & gio)->ggml_cgraph*{ ggml_tensor*t_ids=ggml_new_tensor_1d(C,GGML_TYPE_I32,L+1); ggml_set_input(t_ids); ggml_tensor*emb=ggml_get_rows(C,token_embd,t_ids); if(emb->type!=GGML_TYPE_F32) emb=ggml_cast(C,emb,GGML_TYPE_F32); @@ -362,8 +372,18 @@ std::vector OpenVlaOftModelArch::predict(const Inputs& in) { hh=LN(C,hh,h_ln2w,h_ln2b,head_ln_eps); ggml_tensor*norm_actions=ggml_add(C,ggml_mul_mat(C,h_fc2w,hh),h_fc2b); ggml_set_output(norm_actions); + gio.t_ids=t_ids; gio.t_state=t_state; gio.t_proj=t_proj; gio.act0=act0; + gio.t_pos=t_pos; gio.norm_actions=norm_actions; + ggml_cgraph*gf=ggml_new_graph_custom(C,16384,false); ggml_build_forward_expand(gf,norm_actions); - if(!main_scratch.alloc(backend,gf)){ std::fprintf(stderr,"vla(openvla_oft): main gallocr failed\n"); return {}; } + return gf; + }); + if(!built){ std::fprintf(stderr,"vla(openvla_oft): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor*t_ids=gio.t_ids,*t_state=gio.t_state,*t_proj=gio.t_proj; + ggml_tensor*act0=gio.act0,*t_pos=gio.t_pos,*norm_actions=gio.norm_actions; { std::vector ids(L+1); for(int64_t i=0;i t_time; + }; + graph_cache main_graph; std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi0"}; @@ -548,9 +558,10 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } - ggml_context * C = main_scratch.reset((size_t) 64 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(pi0): ggml_init(ctx_compute) failed\n"); return {}; } - + // Prefix + expert graph depends only on the token counts and step count. + const MainKey mkey{ n_img_tokens, n_lang, num_steps }; + const bool built = main_graph.ensure(backend, mkey, (size_t) 64 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); ggml_tensor * t_prefix_pos= ggml_new_tensor_1d(C, GGML_TYPE_I32, n_prefix); ggml_set_input(t_prefix_pos); @@ -597,14 +608,23 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_tensor * x_final = x_t; ggml_set_output(x_final); + gio.t_image_emb=t_image_emb; gio.t_lang_emb=t_lang_emb; gio.t_prefix_pos=t_prefix_pos; + gio.t_state=t_state; gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; + gio.t_full_mask=t_full_mask; gio.t_time=t_time; gio.x_final=x_final; + ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); ggml_build_forward_expand(gf, x_final); - - - if (!main_scratch.alloc(backend, gf)) { - std::fprintf(stderr, "vla(pi0): ggml_gallocr_alloc_graph failed (out of memory?)\n"); - return {}; - } + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(pi0): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor * t_image_emb = gio.t_image_emb, * t_lang_emb = gio.t_lang_emb; + ggml_tensor * t_prefix_pos = gio.t_prefix_pos, * t_state = gio.t_state, * t_x0 = gio.t_x0; + ggml_tensor * t_suffix_pos = gio.t_suffix_pos, * t_full_mask = gio.t_full_mask; + ggml_tensor * x_final = gio.x_final; + std::vector & t_time = gio.t_time; ggml_backend_tensor_set(t_image_emb, img_emb_host.data(), 0, ggml_nbytes(t_image_emb)); ggml_backend_tensor_set(t_lang_emb, lang_rows.data(), 0, ggml_nbytes(t_lang_emb)); diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index 301223f..fddb666 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -110,7 +110,17 @@ struct Pi05ModelArch : public ModelArchBase { ggml_backend_buffer_t weight_buf = nullptr; ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t n_img=-1, n_lang=-1, nsteps=-1; + bool operator==(const MainKey & o) const { return n_img==o.n_img && n_lang==o.n_lang && nsteps==o.nsteps; } + }; + struct MainIO { + ggml_tensor *t_image_emb=nullptr,*t_lang_emb=nullptr,*t_prefix_pos=nullptr; + ggml_tensor *t_x0=nullptr,*t_suffix_pos=nullptr,*x_final=nullptr; + std::vector t_time; + }; + graph_cache main_graph; std::string ckpt_path_; // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi05"}; @@ -649,9 +659,10 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { if (!io.fetch_rows_f32("token_embd.weight", lang_ids, lang_rows.data(), hidden_pl)) return {}; } - ggml_context * C = main_scratch.reset((size_t) 64 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(pi05): ggml_init(ctx_compute) failed\n"); return {}; } - + // Prefix + expert graph depends only on the token counts and step count. + const MainKey mkey{ n_img_tokens, n_lang, num_steps }; + const bool built = main_graph.ensure(backend, mkey, (size_t) 64 * 1024 * 1024, + [&](ggml_context * C, MainIO & gio) -> ggml_cgraph * { ggml_tensor * t_image_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_img_tokens); ggml_set_input(t_image_emb); ggml_tensor * t_lang_emb = ggml_new_tensor_2d(C, GGML_TYPE_F32, hidden_pl, n_lang); ggml_set_input(t_lang_emb); ggml_tensor * t_prefix_pos= ggml_new_tensor_1d(C, GGML_TYPE_I32, n_prefix); ggml_set_input(t_prefix_pos); @@ -696,14 +707,21 @@ std::vector Pi05ModelArch::predict(const Inputs& in) { ggml_tensor * x_final = x_t; ggml_set_output(x_final); + gio.t_image_emb=t_image_emb; gio.t_lang_emb=t_lang_emb; gio.t_prefix_pos=t_prefix_pos; + gio.t_x0=t_x0; gio.t_suffix_pos=t_suffix_pos; gio.t_time=t_time; gio.x_final=x_final; + ggml_cgraph * gf = ggml_new_graph_custom(C, 16384, false); ggml_build_forward_expand(gf, x_final); - - - if (!main_scratch.alloc(backend, gf)) { - std::fprintf(stderr, "vla(pi05): ggml_gallocr_alloc_graph failed (out of memory?)\n"); - return {}; - } + return gf; + }); + if (!built) { std::fprintf(stderr, "vla(pi05): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor * t_image_emb = gio.t_image_emb, * t_lang_emb = gio.t_lang_emb; + ggml_tensor * t_prefix_pos = gio.t_prefix_pos, * t_x0 = gio.t_x0; + ggml_tensor * t_suffix_pos = gio.t_suffix_pos, * x_final = gio.x_final; + std::vector & t_time = gio.t_time; ggml_backend_tensor_set(t_image_emb, img_emb_host.data(), 0, ggml_nbytes(t_image_emb)); ggml_backend_tensor_set(t_lang_emb, lang_rows.data(), 0, ggml_nbytes(t_lang_emb)); diff --git a/src/models/scratch_ctx.h b/src/models/scratch_ctx.h index 3b33ffe..9c11650 100644 --- a/src/models/scratch_ctx.h +++ b/src/models/scratch_ctx.h @@ -15,6 +15,10 @@ // Compute context and graph allocator reused across predict calls; rebuilding // them costs 2-4 ms on the larger graphs. One scratch per graph role, and // tensors die at the next reset. +// +// graph_cache keeps the built graph too, for the archs whose shape depends on a +// small key. ggml-cuda can then capture and replay it, which needs the node list +// to stay put. #pragma once @@ -23,6 +27,7 @@ #include "ggml-backend.h" #include +#include namespace vla { @@ -55,4 +60,52 @@ class scratch_ctx { ggml_gallocr_t galloc_ = nullptr; }; +// Key is whatever shape the graph depends on (it needs operator==); IO holds the +// input/output tensor handles the arch fills in each call. build(ctx, io) emits +// the graph and returns it, or null to fail the call. +template +class graph_cache { +public: + graph_cache() = default; + graph_cache(const graph_cache &) = delete; + graph_cache & operator=(const graph_cache &) = delete; + ~graph_cache() { release(); } + + template + bool ensure(ggml_backend_t backend, const Key & key, size_t arena, Build && build) { + if (valid_ && key_ == key) return true; + release(); + ggml_init_params p = { arena, nullptr, true }; + ctx_ = ggml_init(p); + if (!ctx_) return false; + io_ = IO{}; + gf_ = build(ctx_, io_); + if (!gf_) { release(); return false; } + galloc_ = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); + if (!galloc_ || !ggml_gallocr_alloc_graph(galloc_, gf_)) { release(); return false; } + key_ = key; + valid_ = true; + return true; + } + + IO & io() { return io_; } + ggml_cgraph * graph() { return gf_; } + + void release() { + if (galloc_) { ggml_gallocr_free(galloc_); galloc_ = nullptr; } + if (ctx_) { ggml_free(ctx_); ctx_ = nullptr; } + gf_ = nullptr; + io_ = IO{}; + valid_ = false; + } + +private: + ggml_context * ctx_ = nullptr; + ggml_gallocr_t galloc_ = nullptr; + ggml_cgraph * gf_ = nullptr; + Key key_{}; + IO io_{}; + bool valid_ = false; +}; + } // namespace vla diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index fbd47de..91e011b 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -95,7 +95,17 @@ struct VlaAdapterModelArch : public ModelArchBase { int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; - scratch_ctx main_scratch; + + struct MainKey { + int64_t seq=-1, n_views=-1, nprompt=-1; + bool operator==(const MainKey & o) const { return seq==o.seq && n_views==o.n_views && nprompt==o.nprompt; } + }; + struct MainIO { + ggml_tensor *t_ids=nullptr,*t_proj=nullptr,*t_pos=nullptr,*t_mask=nullptr; + ggml_tensor *t_state=nullptr,*t_x0=nullptr,*norm_actions=nullptr; + ggml_tensor *cT=nullptr,*sT=nullptr,*cA=nullptr,*sA=nullptr,*cK=nullptr,*sK=nullptr; + }; + graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type mt = GGML_TYPE_BF16; @@ -340,8 +350,10 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { const int64_t NPATCH = NP * n_views; const int64_t SEQ = 1 + NPATCH + (NPROMPT-1) + num_tokens + 1; const auto ti=clock::now(); - ggml_context*C=main_scratch.reset((size_t)128*1024*1024); - + // LM + action head graph depends only on the sequence layout. + const MainKey mkey{ SEQ, n_views, NPROMPT }; + const bool built = main_graph.ensure(backend, mkey, (size_t)128*1024*1024, + [&](ggml_context*C, MainIO & gio)->ggml_cgraph*{ ggml_tensor*t_ids=ggml_new_tensor_1d(C,GGML_TYPE_I32,NPROMPT+num_tokens+1); ggml_set_input(t_ids); ggml_tensor*emb=ggml_get_rows(C,token_embd,t_ids); if(emb->type!=GGML_TYPE_F32) emb=ggml_cast(C,emb,GGML_TYPE_F32); @@ -425,8 +437,20 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_tensor*xn=LN(C,hx,h_ln2w,h_ln2b,head_ln_eps); ggml_tensor*norm_actions=ggml_add(C,ggml_mul_mat(C,h_fc2w,xn),h_fc2b); ggml_set_output(norm_actions); + gio.t_ids=t_ids; gio.t_proj=t_proj; gio.t_pos=t_pos; gio.t_mask=t_mask; + gio.t_state=t_state; gio.t_x0=t_x0; gio.norm_actions=norm_actions; + gio.cT=cT; gio.sT=sT; gio.cA=cA; gio.sA=sA; gio.cK=cK; gio.sK=sK; + ggml_cgraph*gf=ggml_new_graph_custom(C,65536,false); ggml_build_forward_expand(gf,norm_actions); - if(!main_scratch.alloc(backend,gf)){ std::fprintf(stderr,"vla(vla_adapter): main gallocr failed\n"); return {}; } + return gf; + }); + if(!built){ std::fprintf(stderr,"vla(vla_adapter): main graph build failed\n"); return {}; } + + MainIO & gio = main_graph.io(); + ggml_cgraph * gf = main_graph.graph(); + ggml_tensor*t_ids=gio.t_ids,*t_proj=gio.t_proj,*t_pos=gio.t_pos,*t_mask=gio.t_mask; + ggml_tensor*t_state=gio.t_state,*t_x0=gio.t_x0,*norm_actions=gio.norm_actions; + ggml_tensor*cT=gio.cT,*sT=gio.sT,*cA=gio.cA,*sA=gio.sA,*cK=gio.cK,*sK=gio.sK; { std::vector ids(NPROMPT+num_tokens+1); for(int64_t i=0;i t_tau, t_tproj; + }; + graph_cache lm_graph; + graph_cache head_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_F32; @@ -483,8 +500,9 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { std::vector> ds_pad(3); for (int j = 0; j < 3; ++j) { ds_pad[j].assign((size_t) SEQ * H, 0.0f); for (int64_t k = 0; k < n_img; ++k) std::memcpy(ds_pad[j].data() + (size_t) image_pos_idx[k] * H, ds_host[j].data() + (size_t) k * H, H * sizeof(float)); } - ggml_context * C = lm_scratch.reset((size_t) 512 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(LM ctx) failed\n"); return {}; } + const LmKey lkey{ SEQ, num_future }; + const bool lm_built = lm_graph.ensure(backend, lkey, (size_t) 512 * 1024 * 1024, + [&](ggml_context * C, LmIO & gio) -> ggml_cgraph * { ggml_tensor * t_embeds = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, SEQ); ggml_set_input(t_embeds); ggml_tensor * t_pos2 = ggml_new_tensor_1d(C, GGML_TYPE_I32, 4 * SEQ); ggml_set_input(t_pos2); ggml_tensor * t_lmmask = ggml_new_tensor_2d(C, GGML_TYPE_F32, SEQ, SEQ); ggml_set_input(t_lmmask); @@ -500,9 +518,22 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_set_output(eagle); ggml_tensor * conditioning = ggml_get_rows(C, eagle, t_emb_idx); ggml_set_output(conditioning); + gio.t_embeds=t_embeds; gio.t_pos2=t_pos2; gio.t_lmmask=t_lmmask; gio.t_emb_idx=t_emb_idx; + gio.t_ds[0]=t_ds[0]; gio.t_ds[1]=t_ds[1]; gio.t_ds[2]=t_ds[2]; + gio.eagle=eagle; gio.conditioning=conditioning; + ggml_cgraph * lg = ggml_new_graph_custom(C, 32768, false); ggml_build_forward_expand(lg, conditioning); - if (!lm_scratch.alloc(backend, lg)) { std::fprintf(stderr, "vla(vla_jepa): LM gallocr alloc failed\n"); return {}; } + return lg; + }); + if (!lm_built) { std::fprintf(stderr, "vla(vla_jepa): LM graph build failed\n"); return {}; } + + LmIO & gio = lm_graph.io(); + ggml_cgraph * lg = lm_graph.graph(); + ggml_tensor * t_embeds = gio.t_embeds, * t_pos2 = gio.t_pos2, * t_lmmask = gio.t_lmmask; + ggml_tensor * t_emb_idx = gio.t_emb_idx; + ggml_tensor * t_ds[3] = { gio.t_ds[0], gio.t_ds[1], gio.t_ds[2] }; + ggml_tensor * eagle = gio.eagle, * conditioning = gio.conditioning; ggml_backend_tensor_set(t_embeds, inputs_embeds.data(), 0, ggml_nbytes(t_embeds)); @@ -543,8 +574,12 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_backend_tensor_get(conditioning, cond_host.data(), 0, cond_host.size() * sizeof(float)); } - ggml_context * C = head_scratch.reset((size_t) 256 * 1024 * 1024); - if (!C) { std::fprintf(stderr, "vla(vla_jepa): ggml_init(head ctx) failed\n"); return {}; } + // Dumping adds graph outputs, so it always rebuilds. + std::vector step_seq, step_pred, step_vel, step_act; + if (dump_prefix) head_graph.release(); + const HeadKey hkey{ num_steps }; + const bool head_built = head_graph.ensure(backend, hkey, (size_t) 256 * 1024 * 1024, + [&](ggml_context * C, HeadIO & gio) -> ggml_cgraph * { ggml_tensor * t_cond = ggml_new_tensor_2d(C, GGML_TYPE_F32, H, num_future); ggml_set_input(t_cond); ggml_tensor * t_state = ggml_new_tensor_2d(C, GGML_TYPE_F32, state_dim, 1); ggml_set_input(t_state); ggml_tensor * t_x0 = ggml_new_tensor_2d(C, GGML_TYPE_F32, AD, AH); ggml_set_input(t_x0); @@ -554,7 +589,8 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { ggml_tensor * state_features = ggml_add(C, ggml_mul_mat(C, se_l2W, ggml_relu(C, ggml_add(C, ggml_mul_mat(C, se_l1W, t_state), se_l1b))), se_l2b); ggml_tensor * future = future_tokens; const float dt = 1.0f / (float) num_steps; - std::vector step_seq(num_steps), step_pred(num_steps), step_vel(num_steps), step_act(num_steps); + step_seq.assign(num_steps, nullptr); step_pred.assign(num_steps, nullptr); + step_vel.assign(num_steps, nullptr); step_act.assign(num_steps, nullptr); ggml_tensor * actions = t_x0; for (int64_t s = 0; s < num_steps; ++s) { @@ -589,10 +625,20 @@ std::vector VlaJepaModelArch::predict(const Inputs& in) { if (dump_prefix) { ggml_set_output(step_seq[s]); ggml_set_output(step_pred[s]); ggml_set_output(step_vel[s]); ggml_set_output(step_act[s]); } } ggml_set_output(actions); + gio.t_cond=t_cond; gio.t_state=t_state; gio.t_x0=t_x0; gio.actions=actions; + gio.t_tau=t_tau; gio.t_tproj=t_tproj; + ggml_cgraph * hg = ggml_new_graph_custom(C, 65536, false); ggml_build_forward_expand(hg, actions); if (dump_prefix) for (int64_t s = 0; s < num_steps; ++s) { ggml_build_forward_expand(hg, step_seq[s]); ggml_build_forward_expand(hg, step_pred[s]); ggml_build_forward_expand(hg, step_vel[s]); ggml_build_forward_expand(hg, step_act[s]); } - if (!head_scratch.alloc(backend, hg)) { std::fprintf(stderr, "vla(vla_jepa): head gallocr alloc failed\n"); return {}; } + return hg; + }); + if (!head_built) { std::fprintf(stderr, "vla(vla_jepa): head graph build failed\n"); return {}; } + + HeadIO & hio = head_graph.io(); + ggml_cgraph * hg = head_graph.graph(); + ggml_tensor * t_cond = hio.t_cond, * t_state = hio.t_state, * t_x0 = hio.t_x0, * actions = hio.actions; + std::vector & t_tau = hio.t_tau; std::vector & t_tproj = hio.t_tproj; ggml_backend_tensor_set(t_cond, cond_host.data(), 0, ggml_nbytes(t_cond)); { std::vector st(state_dim, 0.0f); for (int64_t i = 0; i < state_dim; ++i) st[i] = in.state ? in.state[i] : 0.0f; ggml_backend_tensor_set(t_state, st.data(), 0, ggml_nbytes(t_state)); } From eea532374baa9040d9deb3ab82dcfa183321e90d Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 18:03:20 +0700 Subject: [PATCH 32/42] share the time embedding, causal mask and patch count sinusoidal_time_emb was copied into pi0, pi05 and smolvla, and the causal mask loop into five archs; both move to dit_common.h. dual_tower derives the patch count from the conv output instead of assuming 256. test_config_guard now calls config_is_sane rather than reimplementing it, so a change to the real guard can fail the test. It links vla_core for that, which takes it out of the standalone sanitizer job. --- .github/workflows/build.yml | 2 +- src/model.cpp | 6 ++++-- src/model.h | 8 ++++++++ src/models/dit_common.h | 24 ++++++++++++++++++++++++ src/models/dual_tower.h | 4 +++- src/models/gr00tn1d7.cpp | 6 +----- src/models/pi0.cpp | 14 +------------- src/models/pi05.cpp | 14 +------------- src/models/smolvla.cpp | 15 +-------------- src/models/vla_adapter.cpp | 4 ++-- src/models/vla_jepa.cpp | 2 +- tests/CMakeLists.txt | 2 ++ tests/test_config_guard.cpp | 11 +++-------- 13 files changed, 52 insertions(+), 60 deletions(-) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index a9334e1..782e2ef 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -20,7 +20,7 @@ jobs: # Compiled directly: pure, no llama.cpp or protobuf/zmq needed. - name: pure unit tests run: | - for t in test_vision_common test_rope_conventions test_config_guard; do + for t in test_vision_common test_rope_conventions; do g++ -std=c++17 -Isrc -Wall -Wextra -fsanitize=address,undefined \ -fno-omit-frame-pointer "tests/$t.cpp" -o "/tmp/$t" "/tmp/$t" diff --git a/src/model.cpp b/src/model.cpp index 95bae6c..3b09ceb 100644 --- a/src/model.cpp +++ b/src/model.cpp @@ -87,8 +87,8 @@ bool detect_arch_gguf(const std::string& path, Arch* out) { return ok; } -// predict() sizes buffers from the max_* dims and loops to the real_* dims, so -// real > max writes out of bounds. Checked here once for all archs. +} // namespace + bool config_is_sane(const Config& c) { struct { const char* name; int64_t real; int64_t max; } pairs[] = { { "state", c.real_state_dim, c.max_state_dim }, @@ -109,6 +109,8 @@ bool config_is_sane(const Config& c) { return true; } +namespace { + bool detect_arch_safetensors(const std::string& path, Arch* out) { std::ifstream f(path, std::ios::binary); if (!f) return false; diff --git a/src/model.h b/src/model.h index 06cbeba..6265f34 100644 --- a/src/model.h +++ b/src/model.h @@ -213,4 +213,12 @@ struct Stats { */ const Stats& last_stats(const Model* m); +/** + * @brief Reject a config whose real_* dims exceed its max_* dims. + * + * predict() sizes buffers from the max_* dims and loops to the real_* dims, so + * real > max writes out of bounds. Checked once for all archs at load. + */ +bool config_is_sane(const Config& c); + } diff --git a/src/models/dit_common.h b/src/models/dit_common.h index 58f4832..83f6cc7 100644 --- a/src/models/dit_common.h +++ b/src/models/dit_common.h @@ -21,6 +21,7 @@ #include #include +#include #include namespace vla { @@ -57,4 +58,27 @@ inline void action_sinusoid(int64_t bucket, int64_t dim, int64_t T, std::vector< for (int64_t tk = 0; tk < T; ++tk) for (int64_t i = 0; i < half; ++i) { const float emb = t * std::exp(-(float) i * step); out[tk * dim + i] = std::sin(emb); out[tk * dim + half + i] = std::cos(emb); } } +// Log-spaced periods rather than frequencies, the openpi convention shared by +// pi0, pi0.5 and SmolVLA. +inline std::vector sinusoidal_time_emb(double t, int64_t dim, double min_p, double max_p) { + const int64_t half = dim / 2; + std::vector out(dim); + for (int64_t i = 0; i < half; ++i) { + const double frac = (half == 1) ? 0.0 : double(i) / double(half - 1); + const double period = min_p * std::pow(max_p / min_p, frac); + const double s = (2.0 * M_PI / period) * t; + out[i] = (float) std::sin(s); + out[half + i] = (float) std::cos(s); + } + return out; +} + +// Additive causal mask, -inf above the diagonal. +inline void build_causal_mask(int64_t seq, std::vector & out) { + out.assign((size_t) seq * seq, 0.0f); + const float NEG = -std::numeric_limits::infinity(); + for (int64_t q = 0; q < seq; ++q) + for (int64_t kv = q + 1; kv < seq; ++kv) out[q * seq + kv] = NEG; +} + } // namespace vla diff --git a/src/models/dual_tower.h b/src/models/dual_tower.h index 18fe93d..20bc40a 100644 --- a/src/models/dual_tower.h +++ b/src/models/dual_tower.h @@ -57,8 +57,10 @@ inline ggml_tensor* tower(ggml_context*C, ggml_tensor*pix, ggml_tensor*pw, ggml_ ggml_tensor*cls, ggml_tensor*reg, const std::vector&blk, int64_t hidden, int64_t heads, int64_t hd, int64_t inter, int64_t patch, float eps, bool prefix){ (void)inter; - const int64_t NP=256, nprefix=prefix?5:0, N=NP+nprefix; ggml_tensor*conv=ggml_conv_2d(C,pw,pix,patch,patch,0,0,1,1); + // Patch count from the conv, not a constant: both callers run 224/14 today, + // and a different input size would otherwise reshape into the wrong grid. + const int64_t NP=conv->ne[0]*conv->ne[1], nprefix=prefix?5:0, N=NP+nprefix; ggml_tensor*pt=ggml_cont(C,ggml_transpose(C,ggml_reshape_2d(C,conv,NP,hidden))); pt=ggml_add(C,pt,pb); pt=ggml_add(C,pt,pos); ggml_tensor*x=pt; diff --git a/src/models/gr00tn1d7.cpp b/src/models/gr00tn1d7.cpp index bbe7f57..d9742a7 100644 --- a/src/models/gr00tn1d7.cpp +++ b/src/models/gr00tn1d7.cpp @@ -788,11 +788,7 @@ std::vector Gr00tN1d7ModelArch::predict(const Inputs& in) { std::memcpy(pp.data() + (size_t) 3 * SEQ, pp.data() + (size_t) 0 * SEQ, (size_t) SEQ * sizeof(int32_t)); ggml_backend_tensor_set(t_pos, pp.data(), 0, ggml_nbytes(t_pos)); } - if (c_mask_seq != SEQ) { - c_mask.assign((size_t) SEQ * SEQ, 0.0f); const float NEG = -std::numeric_limits::infinity(); - for (int64_t q = 0; q < SEQ; ++q) for (int64_t kv = 0; kv < SEQ; ++kv) c_mask[q * SEQ + kv] = (kv <= q) ? 0.0f : NEG; - c_mask_seq = SEQ; - } + if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; } ggml_backend_tensor_set(t_lmmask, c_mask.data(), 0, ggml_nbytes(t_lmmask)); { std::vector st(max_state_dim, 0.0f); for (int64_t i = 0; i < max_state_dim; ++i) st[i] = in.state ? in.state[i] : 0.0f; ggml_backend_tensor_set(t_state, st.data(), 0, ggml_nbytes(t_state)); } ggml_backend_tensor_set(t_x0, x_init.data(), 0, ggml_nbytes(t_x0)); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 2820e81..b905b1c 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -23,6 +23,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/scratch_ctx.h" +#include "models/dit_common.h" #include "models/vision_common.h" #include @@ -65,19 +66,6 @@ bool is_gemma_norm(const std::string & name) { return lm && name.find("norm.weight") != std::string::npos; } -std::vector sinusoidal_time_emb(double t, int64_t dim, double min_p, double max_p) { - const int64_t half = dim / 2; - std::vector out(dim); - for (int64_t i = 0; i < half; ++i) { - const double frac = (half == 1) ? 0.0 : double(i) / double(half - 1); - const double period = min_p * std::pow(max_p / min_p, frac); - const double s = (2.0 * M_PI / period) * t; - out[i] = (float) std::sin(s); - out[half + i] = (float) std::cos(s); - } - return out; -} - bool ends_with(const std::string & s, const char * sfx) { const size_t n = std::strlen(sfx); return s.size() >= n && s.compare(s.size() - n, n, sfx) == 0; diff --git a/src/models/pi05.cpp b/src/models/pi05.cpp index fddb666..7e8c877 100644 --- a/src/models/pi05.cpp +++ b/src/models/pi05.cpp @@ -23,6 +23,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/scratch_ctx.h" +#include "models/dit_common.h" #include "models/vision_common.h" #include @@ -72,19 +73,6 @@ struct ExpertLayerW { // SigLIP-So400m vision block weights (PaliGemma tower, built in-tree like gr00tn1d5). struct SigLipLayerW { ggml_tensor *ln1w,*ln1b,*ln2w,*ln2b,*Wq,*bq,*Wk,*bk,*Wv,*bv,*Wo,*bo,*Wfc1,*bfc1,*Wfc2,*bfc2; }; -std::vector sinusoidal_time_emb(double t, int64_t dim, double min_p, double max_p) { - const int64_t half = dim / 2; - std::vector out(dim); - for (int64_t i = 0; i < half; ++i) { - const double frac = (half == 1) ? 0.0 : double(i) / double(half - 1); - const double period = min_p * std::pow(max_p / min_p, frac); - const double s = (2.0 * M_PI / period) * t; - out[i] = (float) std::sin(s); - out[half + i] = (float) std::cos(s); - } - return out; -} - bool ends_with(const std::string & s, const char * sfx) { const size_t n = std::strlen(sfx); return s.size() >= n && s.compare(s.size() - n, n, sfx) == 0; diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 849ad8d..4cd478f 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -19,6 +19,7 @@ #include "model.h" #include "vision_common.h" #include "scratch_ctx.h" +#include "dit_common.h" #include "ggml.h" #include "ggml-backend.h" @@ -685,20 +686,6 @@ std::string hf_to_gguf(const std::string & n) { return n; } -std::vector sinusoidal_time_emb(double timestep, int64_t dim, - double min_period, double max_period) { - const int64_t half = dim / 2; - std::vector out(dim); - for (int64_t i = 0; i < half; ++i) { - const double frac = (half == 1) ? 0.0 : double(i) / double(half - 1); - const double period = min_period * std::pow(max_period / min_period, frac); - const double scale = 2.0 * M_PI / period; - const double s = scale * timestep; - out[i] = static_cast(std::sin(s)); - out[half + i] = static_cast(std::cos(s)); - } - return out; -} ggml_tensor * rope_q_or_k(ggml_context * ctx, ggml_tensor * x, ggml_tensor * positions, const Config & cfg) { diff --git a/src/models/vla_adapter.cpp b/src/models/vla_adapter.cpp index 91e011b..9351784 100644 --- a/src/models/vla_adapter.cpp +++ b/src/models/vla_adapter.cpp @@ -24,6 +24,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/scratch_ctx.h" +#include "models/dit_common.h" #include #include @@ -459,8 +460,7 @@ std::vector VlaAdapterModelArch::predict(const Inputs& in) { ggml_backend_tensor_set(t_ids,ids.data(),0,ggml_nbytes(t_ids)); } ggml_backend_tensor_set(t_proj,proj_host.data(),0,ggml_nbytes(t_proj)); { std::vector pp(SEQ); for(int64_t i=0;i mk((size_t)SEQ*SEQ); const float NI=-std::numeric_limits::infinity(); - for(int64_t q=0;q mk; build_causal_mask(SEQ, mk); ggml_backend_tensor_set(t_mask,mk.data(),0,ggml_nbytes(t_mask)); } { std::vector sv(proprio_dim,0.0f); for(int64_t i=0;i VlaJepaModelArch::predict(const Inputs& in) { std::memcpy(pp.data() + (size_t) 3 * SEQ, pp.data(), (size_t) SEQ * sizeof(int32_t)); ggml_backend_tensor_set(t_pos2, pp.data(), 0, ggml_nbytes(t_pos2)); } - if (c_mask_seq != SEQ) { c_mask.assign((size_t) SEQ * SEQ, 0.0f); const float NEG = -std::numeric_limits::infinity(); for (int64_t q = 0; q < SEQ; ++q) for (int64_t kv = 0; kv < SEQ; ++kv) c_mask[q * SEQ + kv] = (kv <= q) ? 0.0f : NEG; c_mask_seq = SEQ; } + if (c_mask_seq != SEQ) { build_causal_mask(SEQ, c_mask); c_mask_seq = SEQ; } ggml_backend_tensor_set(t_lmmask, c_mask.data(), 0, ggml_nbytes(t_lmmask)); ggml_backend_tensor_set(t_emb_idx, emb_pos_idx.data(), 0, ggml_nbytes(t_emb_idx)); for (int j = 0; j < 3; ++j) ggml_backend_tensor_set(t_ds[j], ds_pad[j].data(), 0, ggml_nbytes(t_ds[j])); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 226a68b..7a227ab 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -33,7 +33,9 @@ target_link_libraries(test_qwen3vl_vit PRIVATE ggml) target_compile_options(test_qwen3vl_vit PRIVATE -Wall -Wextra) add_test(NAME qwen3vl_vit COMMAND test_qwen3vl_vit) +# Links vla_core: it calls the real config_is_sane rather than a copy. add_executable(test_config_guard test_config_guard.cpp) target_include_directories(test_config_guard PRIVATE ${CMAKE_SOURCE_DIR}/src) +target_link_libraries(test_config_guard PRIVATE vla_core) target_compile_options(test_config_guard PRIVATE -Wall -Wextra) add_test(NAME config_guard COMMAND test_config_guard) diff --git a/tests/test_config_guard.cpp b/tests/test_config_guard.cpp index 32a91ca..7b4072d 100644 --- a/tests/test_config_guard.cpp +++ b/tests/test_config_guard.cpp @@ -21,14 +21,9 @@ #include #include -// Mirrors config_is_sane() in src/model.cpp. -static bool sane(const vla::Config & c) { - if (c.real_state_dim < 0 || c.max_state_dim < 0) return false; - if (c.real_action_dim < 0 || c.max_action_dim < 0) return false; - if (c.max_state_dim > 0 && c.real_state_dim > c.max_state_dim) return false; - if (c.max_action_dim > 0 && c.real_action_dim > c.max_action_dim) return false; - return true; -} +// The real guard, not a copy: a reimplementation here could not fail on a +// regression in src/model.cpp. +static bool sane(const vla::Config & c) { return vla::config_is_sane(c); } int main() { vla::Config c{}; From 7409154cc3b38c7726b64b3caaad852f994e365c Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 18:24:52 +0700 Subject: [PATCH 33/42] add --text to vla-cli and build aarch64 binaries The quickstart needed raw token ids, which meant setting up the python eval client before the first command would run. --text calls scripts/tokenize_prompt.py with the tokenizer the arch was trained on and feeds the ids straight in; --tokens still works. VLA_PYTHON picks the interpreter and VLA_TOKENIZE_SCRIPT the script, so a release tarball can carry its own. Release also builds linux-aarch64-cpu on a native arm64 runner. Jetson is the deployment target and only x86_64 and macOS had binaries. CUDA on aarch64 still has to be built on the device. --- .github/workflows/release.yml | 14 +++++- CMakeLists.txt | 2 + scripts/tokenize_prompt.py | 66 +++++++++++++++++++++++++ src/serving/vla-cli.cpp | 93 ++++++++++++++++++++++++++++++++--- 4 files changed, 166 insertions(+), 9 deletions(-) create mode 100644 scripts/tokenize_prompt.py diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index c421f81..7919ed3 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -20,7 +20,7 @@ env: jobs: linux: - runs-on: ubuntu-24.04 + runs-on: ${{ matrix.runner }} strategy: fail-fast: false matrix: @@ -28,9 +28,18 @@ jobs: - name: linux-x86_64-cpu cmake: -DGGML_CUDA=OFF cuda: false + runner: ubuntu-24.04 - name: linux-x86_64-cuda cmake: -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=75;86;89;120 cuda: true + runner: ubuntu-24.04 + # Jetson and other aarch64 boards. Native arm64 runner, CPU only: the + # hosted images carry no CUDA for arm64, so a Jetson GPU build still + # has to happen on the device. + - name: linux-aarch64-cpu + cmake: -DGGML_CUDA=OFF + cuda: false + runner: ubuntu-24.04-arm steps: - uses: actions/checkout@v4 @@ -62,6 +71,8 @@ jobs: for b in $BINARIES; do cp "build/$b" "$out/"; done cp build/libvla.so "$out/" cp include/vla.h LICENSE.md README.md "$out/" + # vla-cli --text runs this; VLA_TOKENIZE_SCRIPT points at it. + mkdir -p "$out/scripts" && cp scripts/tokenize_prompt.py "$out/scripts/" tar -czf "$out.tar.gz" "$out" - uses: actions/upload-artifact@v4 @@ -89,6 +100,7 @@ jobs: for b in $BINARIES; do cp "build/$b" "$out/"; done cp build/libvla.dylib "$out/" cp include/vla.h LICENSE.md README.md "$out/" + mkdir -p "$out/scripts" && cp scripts/tokenize_prompt.py "$out/scripts/" # Metal needs the shader library next to the binary. find build -name 'default.metallib' -exec cp {} "$out/" \; tar -czf "$out.tar.gz" "$out" diff --git a/CMakeLists.txt b/CMakeLists.txt index aaecbbe..50bc0a6 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -232,6 +232,8 @@ target_include_directories(vla-cli PRIVATE ${llama_SOURCE_DIR}/vendor/stb ) target_link_libraries(vla-cli PRIVATE vla_core) +# --text shells out to the tokenizer script; VLA_TOKENIZE_SCRIPT overrides it. +target_compile_definitions(vla-cli PRIVATE VLA_SOURCE_DIR="${CMAKE_CURRENT_SOURCE_DIR}") # Latency for one checkpoint, emits the README table rows. add_executable(vla-bench diff --git a/scripts/tokenize_prompt.py b/scripts/tokenize_prompt.py new file mode 100644 index 0000000..4968224 --- /dev/null +++ b/scripts/tokenize_prompt.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Print the token ids for an instruction, using the tokenizer an arch was trained with. + + tokenize_prompt.py --arch smolvla --text "pick up the black bowl" + -> 1,4842,731,254,2482,7681,2 + +vla-cli --text calls this so the quickstart does not need raw ids. The eval +client keeps its own richer prompt handling; this only covers the plain case. +""" + +import argparse +import sys + +# Same tokenizers the eval client uses (eval/client/vla_cpp_client.py). +TOKENIZERS = { + "smolvla": "HuggingFaceTB/SmolVLM2-500M-Instruct", + "pi0": "google/paligemma-3b-pt-224", + "pi05": "google/paligemma-3b-pt-224", + "evo1": "OpenGVLab/InternVL3-1B", + "bitvla": "hongyuw/ft-bitvla-bitsiglipL-224px-libero_object-bf16", + "vla_adapter": "VLA-Adapter/LIBERO-Object-Pro", + "openvla_oft": "moojink/openvla-7b-oft-finetuned-libero-spatial-object-goal-10", + "vla_jepa": "Qwen/Qwen3-VL-2B-Instruct", + "gr00t_n1_5": "lerobot/eagle2hg-processor-groot-n1p5", + "gr00t_n1_7": "nvidia/Cosmos-Reason2-2B", +} +TRUST_REMOTE_CODE = {"evo1", "gr00t_n1_5"} + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__) + ap.add_argument("--arch", required=True, choices=sorted(TOKENIZERS)) + ap.add_argument("--text", required=True) + ap.add_argument("--tokenizer", help="override the HuggingFace tokenizer id") + args = ap.parse_args() + + try: + from transformers import AutoTokenizer + except ImportError: + print("transformers is not installed: pip install -e \".[client]\"", file=sys.stderr) + return 1 + + name = args.tokenizer or TOKENIZERS[args.arch] + tok = AutoTokenizer.from_pretrained( + name, trust_remote_code=args.arch in TRUST_REMOTE_CODE) + ids = tok(args.text)["input_ids"] + print(",".join(str(int(i)) for i in ids)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/src/serving/vla-cli.cpp b/src/serving/vla-cli.cpp index 1f86965..c2fa374 100644 --- a/src/serving/vla-cli.cpp +++ b/src/serving/vla-cli.cpp @@ -13,13 +13,14 @@ // limitations under the License. // One-shot action prediction from the command line. Loads a model, decodes an -// image plus an already-tokenized instruction, runs one predict(), and prints -// the action chunk. No server, no simulator. Tokenization stays in the Python -// client, so language is passed as token ids here. +// image plus an instruction, runs one predict(), and prints the action chunk. +// No server, no simulator. There is no tokenizer in the C++ core, so --text +// shells out to scripts/tokenize_prompt.py; --tokens takes ids directly. // // vla-cli [--mmproj m.gguf] --ckpt c.gguf --image img.jpg [--image img2.jpg] -// --tokens id,id,... [--state f,f,...] [--pretty] +// (--text "pick up the bowl" | --tokens id,id,...) [--state f,f,...] [--pretty] +#include "arch.h" #include "model.h" #include "serving/hf_fetch.h" @@ -93,15 +94,84 @@ bool load_image(const char * path, std::vector & buf, int & w, int & h) return true; } +const char * arch_slug(Arch a) { + switch (a) { + case Arch::SMOLVLA: return "smolvla"; + case Arch::PI0: return "pi0"; + case Arch::PI05: return "pi05"; + case Arch::EVO1: return "evo1"; + case Arch::GR00T_N1_5: return "gr00t_n1_5"; + case Arch::GR00T_N1_6: return "gr00t_n1_6"; + case Arch::GR00T_N1_7: return "gr00t_n1_7"; + case Arch::BITVLA: return "bitvla"; + case Arch::VLA_ADAPTER: return "vla_adapter"; + case Arch::OPENVLA_OFT: return "openvla_oft"; + case Arch::VLA_JEPA: return "vla_jepa"; + } + return ""; +} + +// The instruction reaches a shell command, so keep it to plain prose. +bool text_ok(const std::string & s) { + if (s.empty() || s.size() > 512) return false; + for (const char c : s) { + const bool ok = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || + (c >= '0' && c <= '9') || c == ' ' || c == '.' || c == ',' || + c == '-' || c == '_' || c == '\''; + if (!ok) return false; + } + return true; +} + +// Ask scripts/tokenize_prompt.py for the ids, using the tokenizer the arch was +// trained with. Returns "" and explains on stderr. +std::string tokenize_text(const std::string & ckpt, const std::string & text) { + Arch arch; + if (!detect_arch_from_ckpt(ckpt, &arch)) { + std::fprintf(stderr, "vla-cli: cannot detect the arch of %s for --text\n", ckpt.c_str()); + return ""; + } + if (!text_ok(text)) { + std::fprintf(stderr, "vla-cli: --text takes plain prose (letters, digits, space . , - _ ')\n"); + return ""; + } + std::string esc; + for (const char c : text) { if (c == '\'') esc += "'\\''"; else esc += c; } + // Env first so a packaged binary can point at its own copy of the script. + const char * env = std::getenv("VLA_TOKENIZE_SCRIPT"); + const std::string script = (env && *env) ? std::string(env) + : std::string(VLA_SOURCE_DIR) + "/scripts/tokenize_prompt.py"; + const char * py = std::getenv("VLA_PYTHON"); + const std::string interp = (py && *py) ? std::string(py) : std::string("python3"); + const std::string cmd = "'" + interp + "' '" + script + "' --arch " + arch_slug(arch) + + " --text '" + esc + "'"; + + FILE * fp = popen(cmd.c_str(), "r"); + if (!fp) { std::fprintf(stderr, "vla-cli: cannot run %s\n", cmd.c_str()); return ""; } + std::string out; + char buf[4096]; + while (std::fgets(buf, sizeof(buf), fp)) out += buf; + if (pclose(fp) != 0) { + std::fprintf(stderr, + "vla-cli: tokenizing failed. Install the client extras with\n" + " pip install -e \".[client]\"\n" + " (VLA_PYTHON selects a different interpreter)\n"); + return ""; + } + while (!out.empty() && (out.back() == '\n' || out.back() == '\r')) out.pop_back(); + return out; +} + void usage(const char * prog) { std::fprintf(stderr, "usage: %s [--mmproj m.gguf] (--ckpt c.gguf | -hf user/repo) --image img.jpg [--image ...]\n" - " --tokens id,id,... [--state f,f,...] [--pretty]\n" + " (--text \"...\" | --tokens id,id,...) [--state f,f,...] [--pretty]\n" " --mmproj vision-tower GGUF (SmolVLA/pi0/pi0.5); omit for baked-vision archs\n" " --ckpt model checkpoint GGUF\n" " -hf HuggingFace repo, user/repo[:file.gguf], cached under $VLA_CACHE\n" " --image image file, repeat for multi-view (decoded via stb_image)\n" - " --tokens language token ids, comma-separated (tokenize in the client)\n" + " --text instruction, tokenized by scripts/tokenize_prompt.py (needs transformers)\n" + " --tokens language token ids, comma-separated, if you tokenized already\n" " --state proprioception floats, comma-separated (default zeros)\n" " --pretty print one action row (max_action_dim values) per line\n", prog); @@ -110,7 +180,7 @@ void usage(const char * prog) { } // namespace int main(int argc, char ** argv) { - std::string mmproj, ckpt, hf, tokens_s, state_s; + std::string mmproj, ckpt, hf, tokens_s, state_s, text_s; std::vector image_paths; bool pretty = false; @@ -125,6 +195,7 @@ int main(int argc, char ** argv) { else if (a == "-hf") hf = need("-hf"); else if (a == "--image") image_paths.push_back(need("--image")); else if (a == "--tokens") tokens_s = need("--tokens"); + else if (a == "--text") text_s = need("--text"); else if (a == "--state") state_s = need("--state"); else if (a == "--pretty") pretty = true; else if (a == "-h" || a == "--help") { usage(argv[0]); return 0; } @@ -135,7 +206,13 @@ int main(int argc, char ** argv) { ckpt = vla::hf_resolve(hf); if (ckpt.empty()) return 1; } - if (ckpt.empty() || image_paths.empty() || tokens_s.empty()) { usage(argv[0]); return 1; } + if (ckpt.empty() || image_paths.empty() || (tokens_s.empty() && text_s.empty())) { usage(argv[0]); return 1; } + if (!tokens_s.empty() && !text_s.empty()) { std::fprintf(stderr, "vla-cli: pass --text or --tokens, not both\n"); return 1; } + if (!text_s.empty()) { + tokens_s = tokenize_text(ckpt, text_s); + if (tokens_s.empty()) return 1; + std::fprintf(stderr, "vla-cli: --text tokenized to %s\n", tokens_s.c_str()); + } // Validate the cheap args before loading the model. std::vector lang; From a5f37663e275e7d6782ce7cd4cb64e99371b9f5a Mon Sep 17 00:00:00 2001 From: "An T. Le" Date: Sun, 9 Aug 2026 18:38:04 +0700 Subject: [PATCH 34/42] refresh the benchmark table and add success rates Latency re-measured on the 5090 at the graph-cache code, best of three sweeps. Also surfaces the LIBERO success rates that were sitting in eval/reports, with their hardware and commit stated: latency alone does not say the policy works. --- README.md | 62 ++++++++++++++++++++++++++++++++++-------------- docs/ADOPTION.md | 11 ++++++--- 2 files changed, 52 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index 2e4b5b1..5d58cb0 100644 --- a/README.md +++ b/README.md @@ -56,7 +56,6 @@ cmake --build build -j$(nproc) # CUDA build (set CMAKE_CUDA_ARCHITECTURES for your GPU): cmake -B build \ -DGGML_CUDA=ON \ - -DGGML_CUDA_GRAPHS=ON \ -DCMAKE_BUILD_TYPE=Release \ -DCMAKE_CUDA_ARCHITECTURES=$CUDA_ARCHITECTURE cmake --build build -j$(nproc) @@ -94,21 +93,25 @@ WSL2 and Apple Silicon are both tested. Once the binaries are built, run one CPU prediction without a server or simulator: ```bash -pip install -U "huggingface_hub[cli]" +pip install -U "huggingface_hub[cli]" transformers # -hf fetches and caches the checkpoint (under $VLA_CACHE, default ~/.cache/vla) ./build/vla-cli -hf vrfai/smolvla-libero-gguf \ - --image assets/front.jpg --tokens 1,100,200,2 --pretty + --image assets/front.jpg --text "pick up the black bowl" --pretty # or point at a file you already have ./build/vla-cli --ckpt models/smolvla/smolvla-libero.gguf \ - --image assets/front.jpg --tokens 1,100,200,2 --pretty + --image assets/front.jpg --text "pick up the black bowl" --pretty ``` `vla-cli` runs a single prediction without a server or simulator: give it a model, -an image, and the tokenized instruction, and it prints the action chunk. Handy for +an image, and an instruction, and it prints the action chunk. Handy for smoke-testing a GGUF or scripting a quick inference. -`--tokens` are language token ids from the client tokenizer. + +There is no tokenizer in the C++ core, so `--text` calls +`scripts/tokenize_prompt.py` with the tokenizer the architecture was trained on +(`VLA_PYTHON` picks the interpreter, `VLA_TOKENIZE_SCRIPT` the script). Pass +`--tokens 1,100,200,2` instead if you already have ids. `--pretty` prints one action row per line; `--state` sets proprioception (defaults to zeros). @@ -264,25 +267,48 @@ transport, no simulator, no claim about task success. ``` RTX 5090, driver 595.84, CUDA 13.2, 24-core host, weights as shipped, 20 reps -after 3 warmups, each model at its native input size and view count. +after 3 warmups, best of three sweeps, each model at its native input size and +view count. | Model | Views | Input | min ms | p50 ms | p90 ms | vision ms | |---|--:|--:|--:|--:|--:|--:| -| VLA-Adapter | 1 | 224 | 19.2 | 20.2 | 21.0 | 9.1 | -| VLA-JEPA | 1 | 256 | 20.7 | 21.8 | 23.4 | 6.2 | -| BitVLA | 1 | 224 | 24.2 | 25.2 | 26.5 | 5.5 | -| GR00T N1.5 | 1 | 224 | 27.7 | 29.8 | 31.7 | 5.7 | -| GR00T N1.7 | 1 | 256 | 31.0 | 32.8 | 34.0 | 6.2 | -| GR00T N1.6 | 1 | 224 | 35.2 | 37.4 | 39.6 | 6.3 | -| OpenVLA-OFT | 1 | 224 | 47.1 | 49.5 | 51.4 | 9.9 | -| SmolVLA | 2 | 512 | 47.9 | 52.4 | 56.5 | 18.8 | -| pi0 | 2 | 224 | 49.4 | 54.0 | 57.9 | 12.0 | -| pi0.5 | 2 | 224 | 53.0 | 55.9 | 57.6 | 10.7 | -| Evo-1 | 1 | 448 | 54.8 | 58.4 | 61.8 | 17.5 | +| VLA-Adapter | 1 | 224 | 18.2 | 19.8 | 21.1 | 9.4 | +| VLA-JEPA | 1 | 256 | 19.9 | 21.5 | 22.9 | 6.3 | +| BitVLA | 1 | 224 | 23.6 | 25.3 | 26.4 | 5.4 | +| GR00T N1.5 | 1 | 224 | 28.2 | 29.4 | 30.5 | 5.9 | +| GR00T N1.7 | 1 | 256 | 31.0 | 33.4 | 34.6 | 6.2 | +| GR00T N1.6 | 1 | 224 | 33.4 | 35.7 | 37.3 | 6.3 | +| OpenVLA-OFT | 1 | 224 | 47.4 | 49.2 | 50.2 | 10.3 | +| SmolVLA | 2 | 512 | 47.8 | 49.6 | 54.0 | 16.1 | +| pi0 | 2 | 224 | 48.9 | 52.1 | 55.0 | 11.6 | +| Evo-1 | 1 | 448 | 52.2 | 55.2 | 57.3 | 17.8 | +| pi0.5 | 2 | 224 | 53.4 | 56.1 | 59.3 | 11.4 | Jetson and Apple targets are absent: they have not been re-measured with `vla-bench`. +### Task success + +Latency says nothing about whether a policy works. LIBERO-Object, 10 tasks and 20 +episodes per model, terminated episodes counted as failures: + +| Model | Chunk replay | Success rate | +|---|--:|--:| +| BitVLA | 8 | 100.0% | +| GR00T N1.7 | 16 | 98.0% | +| GR00T N1.5 | 16 | 96.0% | +| Evo-1 | 8 | 94.5% | +| SmolVLA | 4 | 90.5% | +| π0 | 32 | 87.5% | +| GR00T N1.6 | 16 | 86.5% | + +From [eval/reports/report-rtx-3060.md](eval/reports/report-rtx-3060.md), swept on +an RTX 3060 at commit `dcc29a3` (2026-05-24). It predates π0.5, VLA-Adapter, +OpenVLA-OFT and VLA-JEPA, which have not been swept. Jetson AGX Orin and Orin +Nano runs are in the same directory. Success rate belongs to the checkpoint, not +the engine; `vla_predict_check` in [CONTRIBUTING.md](CONTRIBUTING.md) is how a +change is shown to leave it alone. + --- ## Roadmap diff --git a/docs/ADOPTION.md b/docs/ADOPTION.md index f599686..1fd0ba0 100644 --- a/docs/ADOPTION.md +++ b/docs/ADOPTION.md @@ -8,19 +8,24 @@ Done: without it nothing outside C++ can link the engine. 2. **Python bindings.** `bindings/python`, ctypes over the ABI. 3. **Prebuilt binaries.** `.github/workflows/release.yml` publishes - linux-x86_64 (CPU and CUDA), macos-arm64-metal and a Docker image on tag. + linux-x86_64 (CPU and CUDA), linux-aarch64 (CPU, for Jetson-class boards), + macos-arm64-metal and a Docker image on tag. 4. **One-command model fetch.** `-hf user/repo[:file.gguf]` on `vla-cli`, `vla-server` and `vla-bench`, cached under `$VLA_CACHE`. 5. **Reproducible benchmarks.** `vla-bench` emits the README table rows. 6. **Contributor path.** `CONTRIBUTING.md` has the six-site walkthrough for adding an architecture, plus issue and PR templates. +7. **Instruction in, action out.** `vla-cli --text` tokenizes with the + architecture's own tokenizer, so the quickstart no longer needs raw ids. Left: -- **Jetson binaries.** `release.yml` covers x86-64 and macOS; aarch64 needs a - self-hosted runner or a cross toolchain. +- **Jetson CUDA binaries.** The aarch64 job is CPU only: the hosted arm64 image + carries no CUDA, so a Jetson GPU build still happens on the device. - **PyPI.** The wheel is built from `bindings/python` but nothing publishes it. - **`ci/baselines/rtx3090.json`** still disagrees with the README table, which is now RTX 5090 numbers from `vla-bench`. Re-record the baselines on one machine. +- **Success rates.** The README table comes from a May 2026 RTX 3060 sweep and + covers seven of the eleven archs. A fresh sweep would cover the rest. None of these change inference behaviour. From 7709b328ccb51bf1c488200a0b7f7641a512dfae Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 14:36:44 +0700 Subject: [PATCH 35/42] compare SR of vla.cpp vs pytorch on libero_object --- eval/run_libero.sh | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/eval/run_libero.sh b/eval/run_libero.sh index 63cc9b3..658fe9c 100644 --- a/eval/run_libero.sh +++ b/eval/run_libero.sh @@ -135,8 +135,12 @@ echo "[config] MODEL=${MODEL}" cd "${REPO_ROOT}" -echo "[build] cmake --build build" -cmake --build build -j"$(nproc)" +if [[ "${SKIP_BUILD:-0}" == "1" ]]; then + echo "[build] skipped (SKIP_BUILD=1)" +else + echo "[build] cmake --build build" + cmake --build build -j"$(nproc)" +fi if [[ ! -x "${SERVER_BIN}" ]]; then echo "ERROR: ${SERVER_BIN} not found after build." >&2 From 2885bd3d573ad44019f8af74fd303284758db1b3 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 14:36:44 +0700 Subject: [PATCH 36/42] add opt-in flash attention to evo1, pi0 and smolvla and batch evo1's views into one vision graph --- eval/client/benchmark.py | 128 ++++++++++++++++++++++++++++++++-- eval/client/vla_cpp_client.py | 20 ++++++ src/models/evo1.cpp | 102 ++++++++++++++++++++++----- src/models/pi0.cpp | 57 +++++++++++---- src/models/smolvla.cpp | 35 ++++++++-- 5 files changed, 300 insertions(+), 42 deletions(-) diff --git a/eval/client/benchmark.py b/eval/client/benchmark.py index 6c1b16e..d3602bb 100644 --- a/eval/client/benchmark.py +++ b/eval/client/benchmark.py @@ -98,15 +98,62 @@ def _percentile(xs: list[float], p: float) -> float: idx = max(0, min(len(s) - 1, int(round(p * (len(s) - 1))))) return s[idx] -def _make_client(backend: str, addr: str): +def _summarize(xs: list[float]) -> dict | None: + if not xs: + return None + return { + "n": len(xs), + "mean": round(statistics.fmean(xs), 3), + "median": round(statistics.median(xs), 3), + "p95": round(_percentile(xs, 0.95), 3), + "p99": round(_percentile(xs, 0.99), 3), + "min": round(min(xs), 3), + "max": round(max(xs), 3), + } + +def _make_client(backend: str, addr: str, args): if backend == "lerobot": + # The PyTorch reference client wraps its ZMQ client in LIBEROSimAdapter, + # which converts a raw LIBERO observation into the format each policy + # expects (see pytorch_ref/utils/sim_adapters/libero.py) and picks the + # parser from the server-reported arch. Without it every request fails + # with "ObservationProcessorStep requires an observation in the + # transition". That package only exists under pytorch_ref, so it has to + # go ahead of eval/ on the path — the two utils.service copies are + # functionally identical, so which one wins does not matter. + sys.path.insert(0, str(ROOT / "pytorch_ref")) from utils.service import RobotInferenceClient - return RobotInferenceClient(host=_host_of(addr), port=_port_of(addr)) + from utils.sim_adapters.libero import LIBEROSimAdapter + return LIBEROSimAdapter( + client=RobotInferenceClient(host=_host_of(addr), port=_port_of(addr)) + ) elif backend == "vla-cpp": + # Mirror run_sim_client_direct.py exactly: the arch preset selects the + # tokenizer / image size / state dim, and evo1 + the GR00T family each + # need their own pipeline adapter. Building a bare VlaCppClient here + # would silently benchmark every arch as if it were smolvla. from client.vla_cpp_client import VlaCppClient - client = VlaCppClient(vla_addr=addr) - return client + from client.adapters import ( + LeRobotPipelineAdapter, + Evo1PipelineAdapter, + Gr00tPipelineAdapter, + Gr00tN15PipelineAdapter, + ) + client = VlaCppClient( + vla_addr=addr, + arch=args.arch, + tokenizer_name=args.tokenizer, + n_action_steps=args.n_action_steps, + stats_json=args.stats_json, + ) + if args.arch == "evo1": + return Evo1PipelineAdapter(client=client) + if args.arch == "gr00t_n1_5": + return Gr00tN15PipelineAdapter(client=client) + if args.arch in ("gr00t_n1_6", "gr00t_n1_7"): + return Gr00tPipelineAdapter(client=client) + return LeRobotPipelineAdapter(client=client) else: raise ValueError(f"unknown backend: {backend}") @@ -135,6 +182,22 @@ def main() -> int: ap.add_argument("--vram-interval-s", type=float, default=0.25) ap.add_argument("--output", type=Path, required=True, help="Path to write the stats JSON.") + ap.add_argument("--variant", default=None, + help="Label for this configuration, e.g. 'vla.cpp', 'eager', " + "'compile-reduce-overhead'. Recorded in the stats JSON so the " + "collector can build the comparison table.") + ap.add_argument("--model", default=None, + help="Model name recorded in the stats JSON, e.g. 'smolvla'.") + # vla-cpp backend only: the arch preset and its per-arch client wiring. + ap.add_argument("--arch", default="smolvla", + help="[vla-cpp] arch of the served GGUF; selects tokenizer/preset/adapter.") + ap.add_argument("--tokenizer", default=None, + help="[vla-cpp] HF id or local dir overriding the arch's default tokenizer.") + ap.add_argument("--stats-json", default=None, + help="[vla-cpp] dataset_statistics.json, required by the GR00T arches.") + ap.add_argument("--n-action-steps", type=int, default=1, + help="Actions replayed per prediction. Keep at 1 for latency runs so " + "every call is a real forward pass rather than a queue pop.") args = ap.parse_args() pid = args.server_pid or _find_server_pid(args.addr) @@ -148,7 +211,17 @@ def main() -> int: print(f"VRAM (pre-warmup) = {v} MiB", flush=True) print(f"connecting to {args.addr} as backend={args.backend} ...", flush=True) - client = _make_client(args.backend, args.addr) + client = _make_client(args.backend, args.addr, args) + + # Both backends wrap their ZMQ client in a pipeline adapter, and neither + # adapter forwards attribute access. Reach the client underneath: the + # vla-cpp path reads `_last_response` off it, the lerobot path calls the + # server's latency endpoints on it. + inner = getattr(client, "_client", client) + if args.backend == "vla-cpp" and not hasattr(inner, "_last_response"): + print("warning: could not reach VlaCppClient._last_response; " + "server-side latency will be empty", flush=True) + inner = None output_dir = args.output.parent / "_bench_videos" output_dir.mkdir(parents=True, exist_ok=True) @@ -171,6 +244,17 @@ def main() -> int: if done or trunc: obs, _info = env.reset() + # The PyTorch server times select_action internally (see + # pytorch_ref/utils/service.py). Drop the samples accumulated during warmup + # so the compiled variants aren't charged for their first-call compilation. + if args.backend == "lerobot": + try: + dropped = inner.call_endpoint("reset_latencies", requires_input=False) + print(f"dropped {dropped.get('dropped')} warmup latency samples", flush=True) + except RuntimeError as e: + print(f"warning: server has no reset_latencies endpoint ({e}); " + f"server-side latency will include warmup", flush=True) + sampler = None if pid is not None: sampler = VramSampler(pid, interval_s=args.vram_interval_s) @@ -191,8 +275,8 @@ def main() -> int: step_latencies_ms.append(1000.0 * (t1 - t0)) n_inference_calls += 1 - if args.backend == "vla-cpp" and hasattr(client, "_last_response"): - r = client._last_response + if args.backend == "vla-cpp" and inner is not None: + r = inner._last_response if r is not None: server_latencies.append({ "total": r.latency_ms_total, @@ -213,6 +297,24 @@ def main() -> int: n_episodes_terminated += 1 obs, _info = env.reset() t_run1 = time.time() + + # Server-side inference time, the metric the vla.cpp/torch.compile + # comparison is built on: it excludes ZMQ transport and image + # serialization, which the client-observed step_ms above includes. + # - lerobot : timed in the server process around select_action + # - vla-cpp : reported by the C++ server in each response + server_infer_ms: list[float] = [] + if args.backend == "lerobot": + try: + server_infer_ms = list( + inner.call_endpoint("get_latencies", requires_input=False) + .get("infer_ms", []) + ) + except RuntimeError as e: + print(f"warning: could not fetch server latencies ({e})", flush=True) + else: + server_infer_ms = [s["total"] for s in server_latencies] + env.close() if sampler is not None: @@ -226,6 +328,7 @@ def main() -> int: "task_id": args.task_id, "n_steps": args.n_steps, "warmup_steps": args.warmup_steps, + "n_action_steps": args.n_action_steps, "wall_time_s": round(t_run1 - t_run0, 3), "counters": { "inference_calls": n_inference_calls, @@ -247,8 +350,13 @@ def main() -> int: "peak": max(vram) if vram else None, "mean": round(statistics.fmean(vram), 1) if vram else None, }, + "server_ms": _summarize(server_infer_ms), "server_latency_breakdown": server_latencies, } + if args.variant: + stats["variant"] = args.variant + if args.model: + stats["model"] = args.model args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text(json.dumps(stats, indent=2)) @@ -274,6 +382,12 @@ def main() -> int: print(f"server-internal : total={statistics.fmean(ms):.1f} " f"vision={statistics.fmean(v):.1f} " f"inference={statistics.fmean(i):.1f} (means)") + if stats["server_ms"]: + s = stats["server_ms"] + print(f"server ms : mean={s['mean']} med={s['median']} " + f"p95={s['p95']} p99={s['p99']} max={s['max']} (n={s['n']})") + else: + print("server ms : unavailable") print(f"\nwrote {args.output}") return 0 diff --git a/eval/client/vla_cpp_client.py b/eval/client/vla_cpp_client.py index a5db4d9..70f18da 100644 --- a/eval/client/vla_cpp_client.py +++ b/eval/client/vla_cpp_client.py @@ -737,6 +737,25 @@ def _predict_chunk_pi05(self, observations: dict[str, Any]) -> np.ndarray: _EVO1_IMG_CTX = "" _EVO1_NUM_IMAGE_TOKEN = 256 _EVO1_MAX_TEXT_LENGTH = 1024 + _EVO1_NOISE_LEN = 50 * 24 # horizon * per_action_dim + + def _maybe_add_fixed_noise(self, req, n: int | None) -> None: + """Attach a reproducible noise vector when VLA_FIXED_NOISE_SEED is set. + + Without it the server draws flow-matching noise from a clock-seeded RNG, + so two servers cannot be compared action-for-action. Pinning the noise + client-side makes a kernel change (e.g. swapping in flash attention) + verifiable: same inputs plus same noise must give the same actions. + """ + seed = os.environ.get("VLA_FIXED_NOISE_SEED") + if seed is None or not n: + return + # Vary per step but reproducibly, so a replay of the same episode sends + # the same sequence of noise vectors. + rng = np.random.default_rng(int(seed) + self._step) + # Evo-1 is trained on uniform[-1,1]; matching that keeps the check in + # the distribution the model actually sees. + req.noise.extend(rng.uniform(-1.0, 1.0, size=n).astype(np.float32).tolist()) def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: @@ -815,6 +834,7 @@ def _predict_chunk_evo1(self, observations: dict[str, Any]) -> np.ndarray: req.lang_tokens.extend(input_ids_full[:n_real].tolist()) req.state.extend(state_padded.tolist()) req.attention_mask.extend(attn_mask.tolist()) + self._maybe_add_fixed_noise(req, self._EVO1_NOISE_LEN) self.sock.send(req.SerializeToString()) body = self.sock.recv() diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index 0b9d0fa..c064a12 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -24,6 +24,7 @@ #include "models/scratch_ctx.h" #include +#include #include #include #include @@ -61,6 +62,9 @@ struct Evo1ModelArch : public ModelArchBase { int n_threads = default_cpu_threads(); ggml_context * ctx_weights = nullptr; scratch_ctx vision_scratch; + // scratch_ctx fixes its arena on first use, and the vision graph holds every + // view at once, so a later call with more views needs a bigger one. + size_t vision_arena = 0; struct MainKey { int64_t seq=-1, nsteps=-1; @@ -154,6 +158,46 @@ bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector return true; } +// Fused attention for the InternViT tower. +// +// The tower runs 1025 tokens (32x32 patches + CLS) per view over 24 layers, so +// the explicit path below materialises a 1025x1025 score matrix for each of 16 +// heads — ~67 MiB per layer, written by the matmul, read and rewritten by the +// softmax, then read again by the AV matmul. That memory traffic, not the FLOPs, +// is what made the vision stage dominate evo1's latency. +// +// Flash attention keeps the scores in registers/shared memory instead. It is +// also what the reference implementation does: InternVL3's `InternAttention` +// calls `flash_attn_varlen_qkvpacked_func` whenever flash-attn is importable +// (see modeling_intern_vit.py), so this path is closer to the upstream model +// than the explicit one, not a divergence from it. +// +// OPT-IN (VLA_EVO1_FA=1), not default. It cuts the vision stage from ~132 ms to +// ~82 ms, but ggml's CUDA flash attention computes K/V at F16 — fattn.cu accepts +// an F32 K/V only by reinterpreting it as F16, so there is no full-precision FA +// path on this backend. Over 24 ViT layers that moved actions by ~1e-2 and +// measured 92/100 on libero_object against 97/100 for explicit attention +// (n=100, same binary). That is inside sampling noise at ~1.6 SE, but the drop +// concentrated in the two tasks the control aced, so the default stays on the +// accuracy-preserving path and the speedup is opt-in. +inline bool evo1_vit_fa_enabled() { + static const bool enabled = (std::getenv("VLA_EVO1_FA") != nullptr); + return enabled; +} + +ggml_tensor * evo1_flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k, ggml_tensor * v, + float scale, int64_t hidden, int64_t N) { + // K/V stay F32. Casting them to F16 (as some in-tree FA helpers do) costs + // real precision: over 24 ViT layers it moved evo1's actions by ~1e-2, which + // is enough to change a LIBERO episode's outcome. smolvla's expert passes + // F32 K/V to the same op, so the backend handles it. + ggml_tensor * o = ggml_flash_attn_ext(C, q, k, v, nullptr, scale, 0.0f, 0.0f); + // F32 accumulation keeps the softmax/AV reduction at the precision the + // explicit path used, so switching kernels does not move the actions. + ggml_flash_attn_ext_set_prec(o, GGML_PREC_F32); + return ggml_reshape_2d(C, o, hidden, N); +} + ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, const ViTLayerW & w, ggml_tensor * x, int64_t N) { const int64_t H = m.vit_hidden, n_heads = m.vit_heads, hd = H / n_heads; @@ -165,11 +209,17 @@ ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, c ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, H, N, qkv->nb[1], 2 * H * ggml_element_size(qkv))); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, hd, n_heads, N), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, hd, n_heads, N), 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, n_heads, N), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - ggml_tensor * kqv = ggml_mul_mat(C, V, aw); - ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), H, N); + ggml_tensor * att; + if (evo1_vit_fa_enabled()) { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, n_heads, N), 0, 2, 1, 3)); + att = evo1_flash_attn(C, Q, K, V, scale, H, N); + } else { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, hd, n_heads, N), 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); + ggml_tensor * kqv = ggml_mul_mat(C, V, aw); + att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), H, N); + } ggml_tensor * attn_out = ggml_add(C, ggml_mul_mat(C, w.Wproj, att), w.bproj); ggml_tensor * x1 = ggml_add(C, x, ggml_mul(C, attn_out, w.ls1)); ggml_tensor * x_n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, x1, m.vit_ln_eps), w.n2w), w.n2b); @@ -435,13 +485,28 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { } n_views = in.n_images; - ggml_context * VC = vision_scratch.reset((size_t) 32 * 1024 * 1024); + // All views go into ONE graph, one compute, instead of a graph_compute + // (plus its host round-trip and implicit device sync) per view. The + // reference does the same thing: InternVL3's _preprocess_images + // concatenates every tile into a single pixel_values batch and calls + // extract_feature once. Encoding views one at a time left the GPU + // draining between each, and made the vision stage dominate latency. + // + // The branches are independent, so the arithmetic per view is unchanged + // - only the submission pattern differs. + const size_t want_arena = (size_t) 32 * 1024 * 1024 * (size_t) std::max(n_views, 1); + if (want_arena > vision_arena) { vision_scratch.release(); vision_arena = want_arena; } + ggml_context * VC = vision_scratch.reset(vision_arena); if (!VC) { std::fprintf(stderr, "vla(evo1): ggml_init(vision ctx) failed\n"); return {}; } - ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); ggml_set_input(t_px); - ggml_tensor * t_ie = build_internvit_view(VC, *this, t_px); - ggml_set_output(t_ie); - ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); - ggml_build_forward_expand(vg, t_ie); + std::vector t_px((size_t) n_views), t_ie((size_t) n_views); + for (int64_t v = 0; v < n_views; ++v) { + t_px[v] = ggml_new_tensor_3d(VC, GGML_TYPE_F32, image_size, image_size, 3); + ggml_set_input(t_px[v]); + t_ie[v] = build_internvit_view(VC, *this, t_px[v]); + ggml_set_output(t_ie[v]); + } + ggml_cgraph * vg = ggml_new_graph_custom(VC, (size_t) 8192 * std::max(n_views, 1), false); + for (int64_t v = 0; v < n_views; ++v) ggml_build_forward_expand(vg, t_ie[v]); if (!vision_scratch.alloc(backend, vg)) { std::fprintf(stderr, "vla(evo1): vision ggml_gallocr_alloc_graph failed\n"); return {}; @@ -451,12 +516,15 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { const auto tv0 = std::chrono::steady_clock::now(); for (int64_t v = 0; v < n_views; ++v) { if (!preprocess_image_chw(in.images[v], image_size, chw)) { return {}; } - ggml_backend_tensor_set(t_px, chw.data(), 0, ggml_nbytes(t_px)); - if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { - std::fprintf(stderr, "vla(evo1): vision graph compute failed (view %lld)\n", (long long) v); - return {}; - } - ggml_backend_tensor_get(t_ie, img_emb_host.data() + v * num_image_token * lm_hidden, 0, ggml_nbytes(t_ie)); + ggml_backend_tensor_set(t_px[v], chw.data(), 0, ggml_nbytes(t_px[v])); + } + if (ggml_backend_graph_compute(backend, vg) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "vla(evo1): vision graph compute failed (%lld views)\n", (long long) n_views); + return {}; + } + for (int64_t v = 0; v < n_views; ++v) { + ggml_backend_tensor_get(t_ie[v], img_emb_host.data() + v * num_image_token * lm_hidden, + 0, ggml_nbytes(t_ie[v])); } stats.ms_vision = std::chrono::duration(std::chrono::steady_clock::now() - tv0).count(); img_emb_ptr = img_emb_host.data(); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index b905b1c..a1cf41d 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -130,6 +130,16 @@ namespace { // One pre-norm SigLIP encoder block, identical to gr00tn1d5's in-tree tower // (the PaliGemma vision tower is the same SigLIP-So400m/14). Bidirectional // attention (nullptr mask), F32 score accumulation, tanh GELU FFN. +// Fused attention for the SigLIP tower and the PaliGemma/expert stack. +// OPT-IN (VLA_PI0_FA=1): pi0's score matrices are small (~560 keys, 8 heads), so +// fusing them only moved 111.4 ms -> 107.5 ms (3.5%). ggml's FA computes K/V at +// F16 regardless of the input type, and on evo1 that cost measurable success +// rate — not a trade worth taking here for 3.5%. +static inline bool pi0_fa_enabled() { + static const bool enabled = (std::getenv("VLA_PI0_FA") != nullptr); + return enabled; +} + ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_tensor * x, int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { const float scale = 1.0f / std::sqrt((float) head_dim); @@ -139,10 +149,20 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ ggml_tensor * v = ggml_add(C, ggml_mul_mat(C, w.Wv, n1), w.bv); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + ggml_tensor * att; + if (pi0_fa_enabled()) { + // Avoids materialising the per-head score matrix; K/V stay F32 so the + // numerics track the explicit path below. + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); + ggml_tensor * fa = ggml_flash_attn_ext(C, Q, K, V, nullptr, scale, 0.0f, 0.0f); + ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + att = ggml_reshape_2d(C, fa, hidden, seq); + } else { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); + att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + } ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); @@ -191,18 +211,27 @@ ggml_tensor * build_gemma_layer( V_full = ggml_concat(ctx, cached_V, v_h, 2); } + const float scale = 1.f / std::sqrt((float) hd); ggml_tensor * Q = ggml_cont(ctx, ggml_permute(ctx, q_rope, 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(ctx, ggml_permute(ctx, K_full, 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); - - ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); - ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - const float scale = 1.f / std::sqrt((float) hd); - ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, mask, scale, 0.f); - ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); - - ggml_tensor * att_pre = ggml_reshape_2d(ctx, - ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); + ggml_tensor * att_pre; + if (pi0_fa_enabled()) { + ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 0, 2, 1, 3)); + // ggml_flash_attn_ext asserts an F16 mask. The mask holds only 0 and + // -inf, both exactly representable in F16, so the cast is lossless. + ggml_tensor * mask_f16 = mask ? ggml_cast(ctx, mask, GGML_TYPE_F16) : nullptr; + ggml_tensor * fa = ggml_flash_attn_ext(ctx, Q, K, V, mask_f16, scale, 0.0f, 0.0f); + ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + att_pre = ggml_reshape_2d(ctx, fa, qf, seq); + } else { + ggml_tensor * V = ggml_cont(ctx, ggml_permute(ctx, V_full, 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(ctx, K, Q); + ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * attn = ggml_soft_max_ext(ctx, kq, mask, scale, 0.f); + ggml_tensor * kqv = ggml_mul_mat(ctx, V, attn); + att_pre = ggml_reshape_2d(ctx, + ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); + } ggml_tensor * o_out = ggml_mul_mat(ctx, w.Wo, att_pre); ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); diff --git a/src/models/smolvla.cpp b/src/models/smolvla.cpp index 4cd478f..39ad167 100644 --- a/src/models/smolvla.cpp +++ b/src/models/smolvla.cpp @@ -355,6 +355,18 @@ struct SmolVLAModelArch : public ModelArchBase { namespace { +// Fused attention in the SigLIP tower. OPT-IN (VLA_SMOLVLA_FA=1), not default. +// It cuts the vision stage from 33.2 ms to 22.1 ms (total 68.5 -> 55.8 ms), which +// is enough to beat compiled PyTorch — but ggml's CUDA flash attention computes +// K/V at F16 regardless of input type (fattn.cu accepts F32 K/V only by +// reinterpreting it as F16), and that measured 92/100 on libero_object against +// 96/100 for explicit attention. evo1 showed the same ~4-5 pp drop, so the +// default stays on the accuracy-preserving path. +static inline bool siglip_fa_enabled() { + static const bool enabled = (std::getenv("VLA_SMOLVLA_FA") != nullptr); + return enabled; +} + // One pre-norm SigLIP encoder block (SmolVLM2 tower), same graph as the other // in-tree models. Bidirectional attention, F32 score accumulation, tanh GELU. ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_tensor * x, @@ -366,10 +378,25 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ ggml_tensor * v = ggml_add(C, ggml_mul_mat(C, w.Wv, n1), w.bv); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); - ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); - ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); - ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); - ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + ggml_tensor * att; + if (siglip_fa_enabled()) { + // The tower runs 1024 tokens (512/16 grid) over 12 layers, so the + // explicit path below materialises a 1024x1024 score matrix per head — + // written by the matmul, read and rewritten by the softmax, then read + // again by the AV matmul. That traffic, not the FLOPs, is why the vision + // stage is roughly half of smolvla's latency. K/V stay F32 so the + // numerics match the explicit path; the expert layers below already call + // this op the same way. + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 0, 2, 1, 3)); + ggml_tensor * fa = ggml_flash_attn_ext(C, Q, K, V, nullptr, scale, 0.0f, 0.0f); + ggml_flash_attn_ext_set_prec(fa, GGML_PREC_F32); + att = ggml_reshape_2d(C, fa, hidden, seq); + } else { + ggml_tensor * V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, v, head_dim, heads, seq), 1, 2, 0, 3)); + ggml_tensor * kq = ggml_mul_mat(C, K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); + ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); + att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); + } ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); From 11bfff6608c1e4cfb2f0f3070e8bcb177fe073fb Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 14:36:44 +0700 Subject: [PATCH 37/42] carry evo1 and pi0 activations in BF16 behind VLA_EVO1_BF16_ACT and VLA_PI0_BF16_ACT, with the ggml patch re-anchored to b10331 --- CMakeLists.txt | 11 + scripts/patch_ggml_bf16_activations.py | 765 +++++++++++++++++++++++++ src/backend.h | 5 + src/models/act_dtype.h | 66 +++ src/models/evo1.cpp | 97 +++- src/models/pi0.cpp | 93 ++- 6 files changed, 976 insertions(+), 61 deletions(-) create mode 100755 scripts/patch_ggml_bf16_activations.py create mode 100644 src/models/act_dtype.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 50bc0a6..fca4587 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -33,11 +33,22 @@ set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) # llama.cpp fetched + pinned at configure; bump = one-line GIT_TAG change. +# +# The fetched ggml is patched in place to support BF16 activations (adds +# ggml_mul_mat_t plus BF16 CUDA elementwise/norm kernels) — src/models/*.cpp +# call ggml_mul_mat_t, so this is required to compile, not optional. The script +# is idempotent, so a re-configure over an already-patched tree is a no-op; it +# fails loudly rather than half-applying if an anchor stops matching after a +# GIT_TAG bump. See scripts/patch_ggml_bf16_activations.py. +find_package(Python3 COMPONENTS Interpreter REQUIRED) include(FetchContent) FetchContent_Declare(llama GIT_REPOSITORY https://github.com/ggml-org/llama.cpp GIT_TAG b10331 GIT_SHALLOW TRUE + PATCH_COMMAND ${Python3_EXECUTABLE} + ${CMAKE_CURRENT_SOURCE_DIR}/scripts/patch_ggml_bf16_activations.py + ) FetchContent_MakeAvailable(llama) diff --git a/scripts/patch_ggml_bf16_activations.py b/scripts/patch_ggml_bf16_activations.py new file mode 100755 index 0000000..7941d8f --- /dev/null +++ b/scripts/patch_ggml_bf16_activations.py @@ -0,0 +1,765 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Teach the fetched ggml how to carry BF16 activations (CUDA backend). + +Why this exists +--------------- +ggml_mul_mat's result is F32 by definition, so a BF16-resident weight meeting an +F32 activation makes ggml_cuda_op_mul_mat_cublas convert src1 F32->BF16 on the +way into every GEMM and the result BF16->F32 on the way out. An nsys trace of +the evo1 server put those convert_unary launches at 11.1% of GPU time, in +exactly balanced pairs (8,228 each direction). Carrying activations as BF16 +removes both, and halves the bytes every bias-add, norm, activation and layout +copy has to move. + +What it changes +--------------- + ggml.c / ggml.h ggml_mul_mat_t(ctx, a, b, type) - mul_mat with an explicit + result type; ggml_mul_mat becomes a wrapper at F32. + ggml-cuda.cu a direct cuBLAS BF16xBF16->BF16 path, taken whenever a + caller asked for a BF16 matmul result. + binbcast.cu BF16 add/mul (activation x activation, activation x F32 + bias/weight), plain and fused. + unary.cu BF16 gelu/silu/relu/... + norm.cu BF16 norm / rms_norm / fused rms_norm+mul, float reductions. + scale.cu BF16 scale. + +concat.cu needs nothing: it already dispatches on ggml_type_size to a +width-generic kernel, so BF16 lands on the uint16_t instantiation. + +Everything keeps float accumulation, so only operand and result *storage* +changes, never a reduction. All BF16 branches are additive: an F32 graph hits +exactly the code it hit before. + +llama.cpp is pulled in by FetchContent (see the top-level CMakeLists), so the +tree lives under build/_deps/llama-src and a fresh configure re-clones it. Run +this after configuring and before building. It is idempotent. + +Usage: scripts/patch_ggml_bf16_activations.py [] +""" + +import pathlib +import sys + +MARKER = "vla.cpp: BF16 activation support" + + +def edit(path, subs): + """Apply (old, new) pairs to `path`; every `old` must appear exactly once. + + Returns the new text rather than writing, so main() can apply every file's + edits or none: a half-patched tree is worse than an unpatched one. + """ + text = path.read_text() + for old, new in subs: + n = text.count(old) + if n != 1: + raise SystemExit( + f"{path}: anchor found {n} times, expected 1:\n---\n{old[:400]}\n---" + ) + text = text.replace(old, new) + return text + + +# --------------------------------------------------------------------------- +# ggml.c / ggml.h - mul_mat with an explicit result type +# --------------------------------------------------------------------------- +GGML_C = [( + """struct ggml_tensor * ggml_mul_mat( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b) { + GGML_ASSERT(ggml_can_mul_mat(a, b)); + GGML_ASSERT(!ggml_is_transposed(a)); + + const int64_t ne[4] = { a->ne[1], b->ne[1], b->ne[2], b->ne[3] }; + struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne); + + result->op = GGML_OP_MUL_MAT; + result->src[0] = a; + result->src[1] = b; + + return result; +}""", + """// vla.cpp: BF16 activation support - mul_mat with an explicit result type. +// +// ggml_mul_mat always produces F32, which forces a BF16->F32 conversion out of +// every cuBLAS BF16 GEMM and an F32->BF16 one back in at the next matmul. +// Letting the caller ask for a BF16 result is what makes an end-to-end BF16 +// activation graph expressible. Only CUDA implements a non-F32 result; see +// ggml_cuda_mul_mat_bf16. +struct ggml_tensor * ggml_mul_mat_t( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b, + enum ggml_type type) { + GGML_ASSERT(ggml_can_mul_mat(a, b)); + GGML_ASSERT(!ggml_is_transposed(a)); + + const int64_t ne[4] = { a->ne[1], b->ne[1], b->ne[2], b->ne[3] }; + struct ggml_tensor * result = ggml_new_tensor(ctx, type, 4, ne); + + result->op = GGML_OP_MUL_MAT; + result->src[0] = a; + result->src[1] = b; + + return result; +} + +struct ggml_tensor * ggml_mul_mat( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b) { + return ggml_mul_mat_t(ctx, a, b, GGML_TYPE_F32); +}""", +)] + +GGML_H = [( + """ GGML_API struct ggml_tensor * ggml_mul_mat( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b); + + // change the precision of a matrix multiplication""", + """ GGML_API struct ggml_tensor * ggml_mul_mat( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b); + + // vla.cpp: BF16 activation support - as ggml_mul_mat, but with an explicit + // result type. Only GGML_TYPE_BF16 (CUDA, BF16 a and b) is implemented + // beyond F32; it keeps a BF16 GEMM's output in BF16 so an all-BF16 + // activation graph does not round-trip through F32 at every matmul. + GGML_API struct ggml_tensor * ggml_mul_mat_t( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b, + enum ggml_type type); + + // change the precision of a matrix multiplication""", +)] + +# --------------------------------------------------------------------------- +# ggml-cuda.cu - direct BF16 x BF16 -> BF16 cuBLAS GEMM +# --------------------------------------------------------------------------- +GGML_CUDA = [( + """static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + GGML_TENSOR_BINARY_OP_LOCALS + + const int32_t hint = ggml_get_op_params_i32(dst, 1); +""", + """// vla.cpp: BF16 activation support - BF16 x BF16 -> BF16 cuBLAS GEMM. +// +// The stock BF16 path (ggml_cuda_op_mul_mat_cublas) always converts src1 to +// BF16 on the way in and the BF16 GEMM result back to F32 on the way out, +// because ggml_mul_mat's result is F32 by definition. On an F32-activation +// graph that is one convert_unary launch per operand per matmul - ~11% of GPU +// time on evo1. When the caller asked for a BF16 result (ggml_mul_mat_t) and +// both operands are already BF16 there is nothing to convert: hand the tensors +// straight to cuBLAS. +// +// Accumulation stays CUBLAS_COMPUTE_32F, matching the stock BF16 path, so only +// operand and result storage changes, not the reduction. +static bool ggml_cuda_can_mul_mat_bf16(const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { + if (src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16 || dst->type != GGML_TYPE_BF16) { + return false; + } + if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { + return false; + } + // src0 is either shared across the whole batch or batched 1:1 with src1 + const bool batch_ok = (src0->ne[2] == 1 && src0->ne[3] == 1) || + (src0->ne[2] == src1->ne[2] && src0->ne[3] == src1->ne[3]); + return batch_ok && bf16_mma_hardware_available(ggml_cuda_info().devices[ggml_cuda_get_device()].cc); +} + +static void ggml_cuda_mul_mat_bf16(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + GGML_TENSOR_BINARY_OP_LOCALS; + + const nv_bfloat16 * a = (const nv_bfloat16 *) src0->data; + const nv_bfloat16 * b = (const nv_bfloat16 *) src1->data; + nv_bfloat16 * c = (nv_bfloat16 *) dst->data; + + const float alpha = 1.0f; + const float beta = 0.0f; + + CUBLAS_CHECK(cublasSetStream(ctx.cublas_handle(), ctx.stream())); + + const int64_t n_batch = ne12*ne13; + if (n_batch == 1) { + CUBLAS_CHECK( + cublasGemmEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + ne01, ne11, ne10, + &alpha, a, CUDA_R_16BF, ne00, + b, CUDA_R_16BF, ne10, + &beta, c, CUDA_R_16BF, ne0, + CUBLAS_COMPUTE_32F, + CUBLAS_GEMM_DEFAULT_TENSOR_OP)); + return; + } + + // stride_a == 0 broadcasts one weight matrix across the batch + const long long stride_a = (src0->ne[2] == 1 && src0->ne[3] == 1) ? 0 : (long long) ne00*ne01; + CUBLAS_CHECK( + cublasGemmStridedBatchedEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + ne01, ne11, ne10, + &alpha, a, CUDA_R_16BF, ne00, stride_a, + b, CUDA_R_16BF, ne10, (long long) ne10*ne11, + &beta, c, CUDA_R_16BF, ne0, (long long) ne0*ne1, + n_batch, + CUBLAS_COMPUTE_32F, + CUBLAS_GEMM_DEFAULT_TENSOR_OP)); +} + +static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + GGML_TENSOR_BINARY_OP_LOCALS + + // vla.cpp: nothing but ggml_mul_mat_t(..., GGML_TYPE_BF16) produces a BF16 + // matmul. This has to intercept before the dst->type != F32 early-out below, + // which would otherwise hand the BF16 result to ggml_cuda_mul_mat_cublas and + // convert it straight back to F32. There is no split-buffer case to exclude: + // CUDA split buffers were removed from ggml upstream. + if (dst->type == GGML_TYPE_BF16) { + if (!ggml_cuda_can_mul_mat_bf16(src0, src1, dst)) { + // Name the offending operand rather than just asserting: the usual + // cause is a graph asking for a BF16 result from a weight that is + // resident F32 (or quantized), which mm_act() is meant to route + // around. See src/models/act_dtype.h. + for (const ggml_tensor * t : {src0, src1, (const ggml_tensor *) dst}) { + fprintf(stderr, "BF16 mul_mat operand %-24s type=%-5s contiguous=%d ne=[%ld %ld %ld %ld]\\n", + t->name, ggml_type_name(t->type), (int) ggml_is_contiguous(t), + (long) t->ne[0], (long) t->ne[1], (long) t->ne[2], (long) t->ne[3]); + } + GGML_ABORT("BF16 mul_mat result requires contiguous BF16 operands on BF16-MMA hardware"); + } + ggml_cuda_mul_mat_bf16(ctx, src0, src1, dst); + return; + } + + const int32_t hint = ggml_get_op_params_i32(dst, 1); +""", +), ( + """ GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); + GGML_ASSERT(rms_norm->type == GGML_TYPE_F32); + + //rms norm only supports F32 + if (mul->src[0]->type != GGML_TYPE_F32 || + mul->src[1]->type != GGML_TYPE_F32 || + mul->type != GGML_TYPE_F32) { + return false; + } + + if (add && (add->src[0]->type != GGML_TYPE_F32 || + add->src[1]->type != GGML_TYPE_F32 || + add->type != GGML_TYPE_F32) ) { + return false; + }""", + """ // vla.cpp: BF16 activation support. ggml_cuda_op_rms_norm_fused now takes + // a BF16 activation with an F32 norm weight; the three-op variant that + // also folds an add stays F32-only. + const enum ggml_type rms_at = rms_norm->src[0]->type; + if (rms_at != GGML_TYPE_F32 && rms_at != GGML_TYPE_BF16) { + return false; + } + GGML_ASSERT(rms_norm->type == rms_at); + + const ggml_tensor * mul_w = (mul->src[0] == rms_norm) ? mul->src[1] : mul->src[0]; + if (mul_w->type != GGML_TYPE_F32 || mul->type != rms_at) { + return false; + } + + if (add && (rms_at != GGML_TYPE_F32 || + add->src[0]->type != GGML_TYPE_F32 || + add->src[1]->type != GGML_TYPE_F32 || + add->type != GGML_TYPE_F32) ) { + return false; + }""", +)] + +# --------------------------------------------------------------------------- +# binbcast.cu - BF16 add / mul, plain and fused +# --------------------------------------------------------------------------- +BINBCAST = [ + ( + """ GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); + + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + op()(src0, src1, dst, (const float *)src0_dd, (const float *)src1_dd, (float *)dst_dd, stream);""", + """ GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16 || src1->type == GGML_TYPE_BF16); + + // vla.cpp: BF16 activation support. src1 is F32 for the model's bias and + // norm-weight tensors (kept F32 in the GGUF) and BF16 for residual adds. + if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { + op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const nv_bfloat16 *)src1_dd, (nv_bfloat16 *)dst_dd, stream); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_BF16) { + op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const float *)src1_dd, (nv_bfloat16 *)dst_dd, stream); + } else if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { + op()(src0, src1, dst, (const float *)src0_dd, (const nv_bfloat16 *)src1_dd, (nv_bfloat16 *)dst_dd, stream); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_F32) { + op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const nv_bfloat16 *)src1_dd, (float *)dst_dd, stream); + } else if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + op()(src0, src1, dst, (const float *)src0_dd, (const float *)src1_dd, (float *)dst_dd, stream);""", + ), + ( + """ if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + launch_bin_bcast_pack(src0, src1, dst, + (const float *) src0->data, (const float *) src1->data, (float *) dst->data, + stream, std::make_index_sequence{});""", + """ // vla.cpp: BF16 activation support. Fusion requires identical src1 layouts + // (ggml_are_same_layout compares type), so a chain never mixes F32 bias and + // BF16 residual operands. + if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { + launch_bin_bcast_pack(src0, src1, dst, + (const nv_bfloat16 *) src0->data, (const nv_bfloat16 *) src1->data, (nv_bfloat16 *) dst->data, + stream, std::make_index_sequence{}); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_BF16) { + launch_bin_bcast_pack(src0, src1, dst, + (const nv_bfloat16 *) src0->data, (const float *) src1->data, (nv_bfloat16 *) dst->data, + stream, std::make_index_sequence{}); + } else if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + launch_bin_bcast_pack(src0, src1, dst, + (const float *) src0->data, (const float *) src1->data, (float *) dst->data, + stream, std::make_index_sequence{});""", + ), +] + +# --------------------------------------------------------------------------- +# unary.cu - BF16 elementwise activations +# --------------------------------------------------------------------------- +UNARY = [( + """ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT( dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16); + GGML_ASSERT(src0->type == dst->type); + + if (src0->type == GGML_TYPE_F16) { + unary_cuda((const half *)src0_d, (half *)dst_d, ggml_nelements(src0), stream); + } else { + unary_cuda((const float *)src0_d, (float *)dst_d, ggml_nelements(src0), stream); + } +}""", + """ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); + GGML_ASSERT( dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16 || dst->type == GGML_TYPE_BF16); + GGML_ASSERT(src0->type == dst->type); + + if (src0->type == GGML_TYPE_F16) { + unary_cuda((const half *)src0_d, (half *)dst_d, ggml_nelements(src0), stream); + } else if (src0->type == GGML_TYPE_BF16) { + // vla.cpp: BF16 activation support. unary_op_kernel already evaluates in + // float and casts back, so only the storage type changes. + unary_cuda((const nv_bfloat16 *)src0_d, (nv_bfloat16 *)dst_d, ggml_nelements(src0), stream); + } else { + unary_cuda((const float *)src0_d, (float *)dst_d, ggml_nelements(src0), stream); + } +}""", +)] + +# --------------------------------------------------------------------------- +# norm.cu - BF16 norm / rms_norm, float reductions +# --------------------------------------------------------------------------- +NORM = [ + ( + """template +static __global__ void norm_f32( + const float * x, float * dst, const int ncols, const int64_t stride_row, const int64_t stride_channel, + const int64_t stride_sample, const float eps) {""", + """// vla.cpp: BF16 activation support. T is the activation storage type (float or +// nv_bfloat16). The mean/variance reduction and the normalisation stay in float +// regardless, so a BF16 graph gets the numerics PyTorch's bf16 LayerNorm gives +// (bf16 in/out, fp32 accumulate). +template +static __global__ void norm_f32( + const T * x, T * dst, const int ncols, const int64_t stride_row, const int64_t stride_channel, + const int64_t stride_sample, const float eps) {""", + ), + ( + """ for (int col = tid; col < ncols; col += block_size) { + const float xi = x[col]; + mean_var.x += xi; + mean_var.y += xi * xi; + }""", + """ for (int col = tid; col < ncols; col += block_size) { + const float xi = (float) x[col]; + mean_var.x += xi; + mean_var.y += xi * xi; + }""", + ), + ( + """ for (int col = tid; col < ncols; col += block_size) { + dst[col] = (x[col] - mean) * inv_std; + } +}""", + """ for (int col = tid; col < ncols; col += block_size) { + dst[col] = (T) (((float) x[col] - mean) * inv_std); + } +}""", + ), + ( + """template +static __global__ void rms_norm_f32(const float * x, + float * dst, + const int ncols,""", + """// vla.cpp: BF16 activation support. T is the activation storage type; `mul` and +// `add` stay F32 because the norm weight and bias are F32 in the GGUF. +template +static __global__ void rms_norm_f32(const T * x, + T * dst, + const int ncols,""", + ), + ( + # disambiguated from the identical loop in l2_norm_f32 by the tail + """ for (int col = tid; col < ncols; col += block_size) { + const float xi = x[col]; + tmp += xi * xi; + } + + // sum up partial sums + extern __shared__ float s_sum[]; + tmp = block_reduce(tmp, s_sum); + + const float mean = tmp / ncols;""", + """ for (int col = tid; col < ncols; col += block_size) { + const float xi = (float) x[col]; + tmp += xi * xi; + } + + // sum up partial sums + extern __shared__ float s_sum[]; + tmp = block_reduce(tmp, s_sum); + + const float mean = tmp / ncols;""", + ), + ( + """ if constexpr (do_multiply && do_add) { + const int mul_col = fastmodulo(col, mul_ncols_packed); + const int add_col = fastmodulo(col, add_ncols_packed); + dst[col] = scale * x[col] * mul[mul_col] + add[add_col]; + } else if constexpr (do_multiply) { + const int mul_col = fastmodulo(col, mul_ncols_packed); + dst[col] = scale * x[col] * mul[mul_col]; + } else { + dst[col] = scale * x[col]; + }""", + """ if constexpr (do_multiply && do_add) { + const int mul_col = fastmodulo(col, mul_ncols_packed); + const int add_col = fastmodulo(col, add_ncols_packed); + dst[col] = (T) (scale * (float) x[col] * mul[mul_col] + add[add_col]); + } else if constexpr (do_multiply) { + const int mul_col = fastmodulo(col, mul_ncols_packed); + dst[col] = (T) (scale * (float) x[col] * mul[mul_col]); + } else { + dst[col] = (T) (scale * (float) x[col]); + }""", + ), + ( + """static void norm_f32_cuda( + const float * x, float * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", + """template +static void norm_f32_cuda( + const T * x, T * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", + ), + ( + """ norm_f32<<>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", + """ norm_f32<<>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", + ), + ( + """ norm_f32<1024><< WARP_SIZE ? 32 * sizeof(float2): 0, stream>>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", + """ norm_f32<1024, T><< WARP_SIZE ? 32 * sizeof(float2): 0, stream>>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", + ), + ( + """static void rms_norm_f32_cuda( + const float * x, float * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", + """template +static void rms_norm_f32_cuda( + const T * x, T * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<256, false>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<256, false, false, T>,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<1024, false>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<1024, false, false, T>,""", + ), + ( + """static void rms_norm_mul_f32_cuda(const float * x, + const float * mul, + const float * add, + float * dst,""", + """template +static void rms_norm_mul_f32_cuda(const T * x, + const float * mul, + const float * add, + T * dst,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<256, true>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<256, true, false, T>,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<1024, true>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, false, T>,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<256, true, true>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<256, true, true, T>,""", + ), + ( + """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, true>,""", + """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, true, T>,""", + ), + ( + """void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const float * src0_d = (const float *) src0->data; + float * dst_d = (float *) dst->data; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT( dst->type == GGML_TYPE_F32);""", + """void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + cudaStream_t stream = ctx.stream(); + + // vla.cpp: BF16 activation support + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); + GGML_ASSERT( dst->type == src0->type);""", + ), + ( + """ norm_f32_cuda(src0_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); +}""", + """ if (src0->type == GGML_TYPE_BF16) { + norm_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, + ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); + } else { + norm_f32_cuda((const float *) src0->data, (float *) dst->data, + ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); + } +}""", + ), + ( + """void ggml_cuda_op_rms_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const float * src0_d = (const float *) src0->data; + float * dst_d = (float *) dst->data; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT( dst->type == GGML_TYPE_F32);""", + """void ggml_cuda_op_rms_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + cudaStream_t stream = ctx.stream(); + + // vla.cpp: BF16 activation support + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); + GGML_ASSERT( dst->type == src0->type);""", + ), + ( + """ rms_norm_f32_cuda(src0_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); +}""", + """ if (src0->type == GGML_TYPE_BF16) { + rms_norm_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, + ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); + } else { + rms_norm_f32_cuda((const float *) src0->data, (float *) dst->data, + ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); + } +}""", + ), + ( + """ const float * src0_d = (const float *) rms_norm_src->data; + const float * mul_d = nullptr; + const ggml_tensor * mul_src = nullptr; + + if (mul_tensor->src[0] == dst) { + mul_d = (float *) mul_tensor->src[1]->data; + mul_src = mul_tensor->src[1]; + } else if(mul_tensor->src[1] == dst) { + mul_d = (float *) mul_tensor->src[0]->data; + mul_src = mul_tensor->src[0]; + } else { + GGML_ASSERT(false); + } + + float * dst_d = (float *) mul_tensor->data; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(rms_norm_src->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(mul_tensor->type == GGML_TYPE_F32); + GGML_ASSERT(eps >= 0.0f);""", + """ const float * mul_d = nullptr; + const ggml_tensor * mul_src = nullptr; + + if (mul_tensor->src[0] == dst) { + mul_d = (float *) mul_tensor->src[1]->data; + mul_src = mul_tensor->src[1]; + } else if(mul_tensor->src[1] == dst) { + mul_d = (float *) mul_tensor->src[0]->data; + mul_src = mul_tensor->src[0]; + } else { + GGML_ASSERT(false); + } + + cudaStream_t stream = ctx.stream(); + + // vla.cpp: BF16 activation support - activation may be BF16, norm weight stays F32 + GGML_ASSERT(rms_norm_src->type == GGML_TYPE_F32 || rms_norm_src->type == GGML_TYPE_BF16); + GGML_ASSERT(dst->type == rms_norm_src->type); + GGML_ASSERT(mul_tensor->type == rms_norm_src->type); + GGML_ASSERT(mul_src->type == GGML_TYPE_F32); + GGML_ASSERT(eps >= 0.0f);""", + ), + ( + """ rms_norm_mul_f32_cuda(src0_d, mul_d, nullptr, dst_d, + ne00, ne01, ne02, ne03, + /*s00*/ s01, s02, s03, + /*mul_s00*/ mul_s01, mul_s02, mul_s03, + mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, + /*add_s00*/ 0, 0, 0, + 0, 0, 0, 0, + eps, stream); +}""", + """ if (rms_norm_src->type == GGML_TYPE_BF16) { + rms_norm_mul_f32_cuda((const nv_bfloat16 *) rms_norm_src->data, mul_d, (const float *) nullptr, + (nv_bfloat16 *) mul_tensor->data, + ne00, ne01, ne02, ne03, + /*s00*/ s01, s02, s03, + /*mul_s00*/ mul_s01, mul_s02, mul_s03, + mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, + /*add_s00*/ 0, 0, 0, + 0, 0, 0, 0, + eps, stream); + } else { + rms_norm_mul_f32_cuda((const float *) rms_norm_src->data, mul_d, (const float *) nullptr, + (float *) mul_tensor->data, + ne00, ne01, ne02, ne03, + /*s00*/ s01, s02, s03, + /*mul_s00*/ mul_s01, mul_s02, mul_s03, + mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, + /*add_s00*/ 0, 0, 0, + 0, 0, 0, 0, + eps, stream); + } +}""", + ), +] + +# --------------------------------------------------------------------------- +# scale.cu - BF16 scale +# --------------------------------------------------------------------------- +SCALE = [( + """static __global__ void scale_f32(const float * x, float * dst, const float scale, const float bias, const int64_t nelements) { + ggml_cuda_pdl_lc(); + int64_t tid = (int64_t)blockIdx.x * (int64_t)blockDim.x + (int64_t)threadIdx.x; + int64_t stride = (int64_t)blockDim.x * (int64_t)gridDim.x; + + ggml_cuda_pdl_sync(); + for (int64_t i = tid; i < nelements; i += stride) { + dst[i] = scale * x[i] + bias; + } +} + +static void scale_f32_cuda(const float * x, float * dst, const float scale, const float bias, const int64_t nelements, cudaStream_t stream) { + const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(MIN(MAX_GRIDDIM_X, num_blocks), CUDA_SCALE_BLOCK_SIZE, 0, stream); + ggml_cuda_kernel_launch(scale_f32, launch_params, x, dst, scale, bias, nelements); +} + +void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const float * src0_d = (const float *)src0->data; + float * dst_d = (float *)dst->data; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT( dst->type == GGML_TYPE_F32); + + float scale; + float bias; + memcpy(&scale, (float *) dst->op_params + 0, sizeof(float)); + memcpy(&bias, (float *) dst->op_params + 1, sizeof(float)); + + scale_f32_cuda(src0_d, dst_d, scale, bias, ggml_nelements(src0), stream); +}""", + """// vla.cpp: BF16 activation support. T is the activation storage type; the +// scale and bias apply in float. +template +static __global__ void scale_f32(const T * x, T * dst, const float scale, const float bias, const int64_t nelements) { + ggml_cuda_pdl_lc(); + int64_t tid = (int64_t)blockIdx.x * (int64_t)blockDim.x + (int64_t)threadIdx.x; + int64_t stride = (int64_t)blockDim.x * (int64_t)gridDim.x; + + ggml_cuda_pdl_sync(); + for (int64_t i = tid; i < nelements; i += stride) { + dst[i] = (T) (scale * (float) x[i] + bias); + } +} + +template +static void scale_f32_cuda(const T * x, T * dst, const float scale, const float bias, const int64_t nelements, cudaStream_t stream) { + const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(MIN(MAX_GRIDDIM_X, num_blocks), CUDA_SCALE_BLOCK_SIZE, 0, stream); + ggml_cuda_kernel_launch(scale_f32, launch_params, x, dst, scale, bias, nelements); +} + +void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); + GGML_ASSERT( dst->type == src0->type); + + float scale; + float bias; + memcpy(&scale, (float *) dst->op_params + 0, sizeof(float)); + memcpy(&bias, (float *) dst->op_params + 1, sizeof(float)); + + if (src0->type == GGML_TYPE_BF16) { + scale_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, scale, bias, ggml_nelements(src0), stream); + } else { + scale_f32_cuda((const float *) src0->data, (float *) dst->data, scale, bias, ggml_nelements(src0), stream); + } +}""", +)] + + +def main(): + repo = pathlib.Path(__file__).resolve().parent.parent + src = pathlib.Path(sys.argv[1]) if len(sys.argv) > 1 else repo / "build/_deps/llama-src" + if not (src / "ggml/src/ggml.c").exists(): + raise SystemExit(f"not a llama.cpp source tree: {src}") + + if MARKER in (src / "ggml/src/ggml.c").read_text(): + print(f"already patched: {src}") + return + + cuda = src / "ggml/src/ggml-cuda" + pending = [ + (src / "ggml/src/ggml.c", edit(src / "ggml/src/ggml.c", GGML_C)), + (src / "ggml/include/ggml.h", edit(src / "ggml/include/ggml.h", GGML_H)), + (cuda / "ggml-cuda.cu", edit(cuda / "ggml-cuda.cu", GGML_CUDA)), + (cuda / "binbcast.cu", edit(cuda / "binbcast.cu", BINBCAST)), + (cuda / "unary.cu", edit(cuda / "unary.cu", UNARY)), + (cuda / "norm.cu", edit(cuda / "norm.cu", NORM)), + (cuda / "scale.cu", edit(cuda / "scale.cu", SCALE)), + ] + for path, text in pending: + path.write_text(text) + print(f"patched {len(pending)} files: {src}") + + +if __name__ == "__main__": + main() diff --git a/src/backend.h b/src/backend.h index a4d89bb..597275b 100644 --- a/src/backend.h +++ b/src/backend.h @@ -69,6 +69,10 @@ inline void setenv_default(const char * key, const char * val) { /// failed to come up, which callers treat as a fatal load error. struct Backend { ggml_backend_t handle = nullptr; + /// True only for a live CUDA backend, never for the CPU fallback. The BF16 + /// activation path needs the CUDA BF16 GEMM and elementwise kernels, so + /// archs gate on this rather than on the compiled-in accelerator. + bool is_cuda = false; }; /// GPU ordinal for CUDA and SYCL; `VLA_DEVICE` overrides. Junk is rejected, not @@ -100,6 +104,7 @@ inline Backend backend_init(const char * tag, int n_threads) { const int dev = backend_device_index(); b.handle = ggml_backend_cuda_init(dev); if (b.handle) { + b.is_cuda = true; std::printf("%s: backend = CUDA (device %d)\n", tag, dev); } else { std::fprintf(stderr, "%s: ggml_backend_cuda_init failed; falling back to CPU\n", tag); diff --git a/src/models/act_dtype.h b/src/models/act_dtype.h new file mode 100644 index 0000000..00ece16 --- /dev/null +++ b/src/models/act_dtype.h @@ -0,0 +1,66 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#pragma once + +// Activation-dtype helpers for the BF16 activation path. +// +// ggml_mul_mat's result is F32 by definition, so with BF16-resident weights the +// CUDA backend converts the F32 activation to BF16 on the way into every GEMM +// and the BF16 result back to F32 on the way out +// (ggml_cuda_op_mul_mat_cublas). An nsys trace of the evo1 server put those +// convert_unary launches at ~18 ms per request — in exactly balanced pairs. +// Carrying activations as BF16 removes both, and halves the bytes every +// bias-add, norm, activation and layout copy has to move. +// +// A model opts in by setting its `act_type` to GGML_TYPE_BF16 and routing its +// weight GEMMs through mm_act(). The precision split that goes with it follows +// torch.autocast(bfloat16), which is what these references run under: +// +// BF16 - weight GEMMs, bias adds, residuals, elementwise multiplies, norms +// (float reductions internally), and the activation functions. +// F32 - attention scores, softmax and RoPE; patch-embed convolutions; and +// any flow-matching Euler integrator, so N accumulations of dt = 1/N +// do not collapse into 8 mantissa bits. +// +// Requires the ggml BF16 patch (scripts/patch_ggml_bf16_activations.py), which +// adds ggml_mul_mat_t plus BF16 support in the CUDA elementwise/norm kernels. +// +// At act_type == GGML_TYPE_F32 both helpers below are the identity, so the +// default path builds exactly the graph it built before. + +#include "ggml.h" + +namespace vla { + +// Cast only when the type actually differs, so the F32 path adds no nodes. +inline ggml_tensor * as_type(ggml_context * C, ggml_tensor * t, ggml_type ty) { + return t->type == ty ? t : ggml_cast(C, t, ty); +} + +// Weight-by-activation GEMM producing an activation-typed result. +// +// A BF16 result needs a BF16 weight. Not every weight is one: pi0 keeps its +// action-expert projections (state_proj, action_in_proj, action_time_mlp_*, +// action_out_proj) resident F32, and a quantized GGUF keeps its packed type. +// Those run the stock matmul at F32 and rejoin the activation stream after, +// which is exactly what they did before the BF16 path existed. +inline ggml_tensor * mm_act(ggml_context * C, ggml_tensor * w, ggml_tensor * x, ggml_type at) { + if (w->type != at) { + return as_type(C, ggml_mul_mat(C, w, as_type(C, x, GGML_TYPE_F32)), at); + } + return ggml_mul_mat_t(C, w, x, at); +} + +} // namespace vla diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index c064a12..ab643cc 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -22,6 +22,7 @@ #include "gguf.h" #include "models/gguf_reader.h" #include "models/scratch_ctx.h" +#include "models/act_dtype.h" #include #include @@ -77,6 +78,10 @@ struct Evo1ModelArch : public ModelArchBase { graph_cache main_graph; ggml_backend_buffer_t weight_buf = nullptr; ggml_type matmul_type = GGML_TYPE_BF16; + // Activation dtype carried between ops. F32 by default; BF16 under + // VLA_EVO1_BF16_ACT, which removes the per-GEMM F32<->BF16 round trip ggml + // pays when BF16 weights meet F32 activations. See mm_act/as_type below. + ggml_type act_type = GGML_TYPE_F32; int64_t lm_hidden=896, lm_layers=14, n_q=14, n_kv=2, lm_head_dim=64, lm_inter=4864; int64_t embed_dim=896, dit_layers=8, dit_heads=8, mlp_head_hidden=1024; @@ -109,15 +114,27 @@ struct Evo1ModelArch : public ModelArchBase { namespace { +// BF16 activation path (VLA_EVO1_BF16_ACT=1); see models/act_dtype.h for what +// mm_act/as_type do and why. The split here: GEMMs, bias adds, residuals, +// norms and activations in BF16; attention scores, softmax and RoPE in F32; +// and the flow-matching Euler integrator in F32 so 32 steps of dt = 1/32 do +// not quantise. At act_type == GGML_TYPE_F32 the graph is unchanged. + ggml_tensor * build_qwen2_layer(ggml_context * C, const Evo1ModelArch & m, const Qwen2LayerW & w, ggml_tensor * h, ggml_tensor * positions, ggml_tensor * mask, int64_t seq, ggml_tensor * qmask = nullptr) { const int64_t hd = m.lm_head_dim, n_q = m.n_q, n_kv = m.n_kv, hq = n_q * hd; + const ggml_type at = m.act_type; const float scale = 1.0f / std::sqrt((float) hd); ggml_tensor * h_n1 = ggml_mul(C, ggml_rms_norm(C, h, m.lm_rms_eps), w.attn_norm); - ggml_tensor * qp = ggml_add(C, ggml_mul_mat(C, w.Wq, h_n1), w.bq); - ggml_tensor * kp = ggml_add(C, ggml_mul_mat(C, w.Wk, h_n1), w.bk); - ggml_tensor * vp = ggml_add(C, ggml_mul_mat(C, w.Wv, h_n1), w.bv); + ggml_tensor * qp = ggml_add(C, mm_act(C, w.Wq, h_n1, at), w.bq); + ggml_tensor * kp = ggml_add(C, mm_act(C, w.Wk, h_n1, at), w.bk); + ggml_tensor * vp = ggml_add(C, mm_act(C, w.Wv, h_n1, at), w.bv); + // RoPE, scores and softmax stay F32 (autocast keeps softmax in fp32, and the + // flash-attention experiment showed evo1's SR is sensitive to attention precision) + qp = as_type(C, qp, GGML_TYPE_F32); + kp = as_type(C, kp, GGML_TYPE_F32); + vp = as_type(C, vp, GGML_TYPE_F32); ggml_tensor * q_rope = ggml_rope_ext(C, ggml_reshape_3d(C, qp, hd, n_q, seq), positions, nullptr, (int) hd, GGML_ROPE_TYPE_NEOX, 0, m.lm_rope_base, 1.0f, 0.0f, 1.0f, 32.0f, 1.0f); ggml_tensor * k_rope = ggml_rope_ext(C, ggml_reshape_3d(C, kp, hd, n_kv, seq), positions, nullptr, (int) hd, GGML_ROPE_TYPE_NEOX, 0, m.lm_rope_base, 1.0f, 0.0f, 1.0f, 32.0f, 1.0f); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, q_rope, 0, 2, 1, 3)); @@ -127,13 +144,13 @@ ggml_tensor * build_qwen2_layer(ggml_context * C, const Evo1ModelArch & m, const ggml_tensor * aw = ggml_soft_max_ext(C, kq, mask, scale, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, V, aw); ggml_tensor * att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), hq, seq); - ggml_tensor * attn_out = ggml_mul_mat(C, w.Wo, att); + ggml_tensor * attn_out = mm_act(C, w.Wo, as_type(C, att, at), at); if (qmask) attn_out = ggml_mul(C, attn_out, qmask); ggml_tensor * h_attn = ggml_add(C, h, attn_out); ggml_tensor * h_n2 = ggml_mul(C, ggml_rms_norm(C, h_attn, m.lm_rms_eps), w.ffn_norm); - ggml_tensor * gate = ggml_silu(C, ggml_mul_mat(C, w.Wgate, h_n2)); - ggml_tensor * up = ggml_mul_mat(C, w.Wup, h_n2); - return ggml_add(C, h_attn, ggml_mul_mat(C, w.Wdown, ggml_mul(C, gate, up))); + ggml_tensor * gate = ggml_silu(C, mm_act(C, w.Wgate, h_n2, at)); + ggml_tensor * up = mm_act(C, w.Wup, h_n2, at); + return ggml_add(C, h_attn, mm_act(C, w.Wdown, ggml_mul(C, gate, up), at)); } bool preprocess_image_chw(const ImageView & v, int64_t side, std::vector & out) { @@ -201,9 +218,12 @@ ggml_tensor * evo1_flash_attn(ggml_context * C, ggml_tensor * q, ggml_tensor * k ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, const ViTLayerW & w, ggml_tensor * x, int64_t N) { const int64_t H = m.vit_hidden, n_heads = m.vit_heads, hd = H / n_heads; + const ggml_type at = m.act_type; const float scale = 1.0f / std::sqrt((float) hd); ggml_tensor * x_n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, m.vit_ln_eps), w.n1w), w.n1b); - ggml_tensor * qkv = ggml_add(C, ggml_mul_mat(C, w.Wqkv, x_n1), w.bqkv); + ggml_tensor * qkv = ggml_add(C, mm_act(C, w.Wqkv, x_n1, at), w.bqkv); + // one cast of the packed QKV rather than three of its slices + qkv = as_type(C, qkv, GGML_TYPE_F32); ggml_tensor * q = ggml_cont(C, ggml_view_2d(C, qkv, H, N, qkv->nb[1], 0 * H * ggml_element_size(qkv))); ggml_tensor * k = ggml_cont(C, ggml_view_2d(C, qkv, H, N, qkv->nb[1], 1 * H * ggml_element_size(qkv))); ggml_tensor * v = ggml_cont(C, ggml_view_2d(C, qkv, H, N, qkv->nb[1], 2 * H * ggml_element_size(qkv))); @@ -220,12 +240,12 @@ ggml_tensor * build_internvit_layer(ggml_context * C, const Evo1ModelArch & m, c ggml_tensor * kqv = ggml_mul_mat(C, V, aw); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), H, N); } - ggml_tensor * attn_out = ggml_add(C, ggml_mul_mat(C, w.Wproj, att), w.bproj); + ggml_tensor * attn_out = ggml_add(C, mm_act(C, w.Wproj, as_type(C, att, at), at), w.bproj); ggml_tensor * x1 = ggml_add(C, x, ggml_mul(C, attn_out, w.ls1)); ggml_tensor * x_n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, x1, m.vit_ln_eps), w.n2w), w.n2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc1, x_n2), w.bfc1); + ggml_tensor * ff = ggml_add(C, mm_act(C, w.Wfc1, x_n2, at), w.bfc1); ff = ggml_gelu_erf(C, ff); - ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ff), w.bfc2); + ff = ggml_add(C, mm_act(C, w.Wfc2, ff, at), w.bfc2); return ggml_add(C, x1, ggml_mul(C, ff, w.ls2)); } @@ -236,7 +256,9 @@ ggml_tensor * build_internvit_view(ggml_context * C, const Evo1ModelArch & m, gg ggml_tensor * conv = ggml_conv_2d(C, m.vit_patch_w, pixels, (int) m.patch_size, (int) m.patch_size, 0, 0, 1, 1); ggml_tensor * patches = ggml_add(C, ggml_cont(C, ggml_transpose(C, ggml_reshape_2d(C, conv, n_patches, H))), m.vit_patch_b); ggml_tensor * cls2d = ggml_reshape_2d(C, m.vit_cls, H, 1); - ggml_tensor * x = ggml_add(C, ggml_concat(C, cls2d, patches, 1), m.vit_pos); + // patch embed (conv_2d) and the positional add stay F32; the tower runs in + // the activation dtype from here + ggml_tensor * x = as_type(C, ggml_add(C, ggml_concat(C, cls2d, patches, 1), m.vit_pos), m.act_type); for (int64_t i = 0; i < m.vit_layers; ++i) x = build_internvit_layer(C, m, m.vit[i], x, n_tok); @@ -247,9 +269,10 @@ ggml_tensor * build_internvit_view(ggml_context * C, const Evo1ModelArch & m, gg ggml_tensor * s4 = ggml_cont(C, ggml_permute(C, s3, 0, 2, 1, 3)); ggml_tensor * shuf = ggml_reshape_2d(C, s4, shuf_c, sgrid * sgrid); ggml_tensor * x_ln = ggml_add(C, ggml_mul(C, ggml_norm(C, shuf, m.proj_ln_eps), m.mm_ln_w), m.mm_ln_b); - ggml_tensor * hh = ggml_add(C, ggml_mul_mat(C, m.mm_W1, x_ln), m.mm_b1); + ggml_tensor * hh = ggml_add(C, mm_act(C, m.mm_W1, x_ln, m.act_type), m.mm_b1); hh = ggml_gelu_erf(C, hh); - return ggml_add(C, ggml_mul_mat(C, m.mm_W2, hh), m.mm_b2); + // the view embeddings are read back to the host as F32 + return as_type(C, ggml_add(C, mm_act(C, m.mm_W2, hh, m.act_type), m.mm_b2), GGML_TYPE_F32); } ggml_tensor * inproj_split_w(ggml_context * C, ggml_tensor * Win, int64_t E, int64_t k) { @@ -349,6 +372,16 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, const Backend b = backend_init("vla(evo1)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; + + // BF16 activations need BF16-resident weights and the CUDA BF16 GEMM path. + if (std::getenv("VLA_EVO1_BF16_ACT")) { + if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { + m->act_type = GGML_TYPE_BF16; + std::printf("vla(evo1): activations = BF16 (VLA_EVO1_BF16_ACT)\n"); + } else { + std::fprintf(stderr, "vla(evo1): VLA_EVO1_BF16_ACT ignored - needs CUDA and BF16 weights\n"); + } + } } ggml_init_params wp = { (size_t) 32 * 1024 * 1024, @@ -632,12 +665,14 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * t_x = ggml_new_tensor_1d(C, GGML_TYPE_F32, action_dim); ggml_set_input(t_x); ggml_tensor * t_amask = ggml_new_tensor_1d(C, GGML_TYPE_F32, per_a); ggml_set_input(t_amask); - ggml_tensor * h = t_embeds; + const ggml_type at = act_type; + + ggml_tensor * h = as_type(C, t_embeds, at); for (int64_t i = 0; i < lm_layers; ++i) h = build_qwen2_layer(C, *this, lm[i], h, t_pos, t_lmmask, SEQ, t_qmask); ggml_tensor * context = ggml_mul(C, ggml_rms_norm(C, h, lm_rms_eps), lm_output_norm); - ggml_tensor * se = ggml_relu(C, ggml_add(C, ggml_mul_mat(C, state_W1, t_state), state_b1)); - ggml_tensor * state_tok = ggml_add(C, ggml_mul_mat(C, state_W2, se), state_b2); + ggml_tensor * se = ggml_relu(C, ggml_add(C, mm_act(C, state_W1, as_type(C, t_state, at), at), state_b1)); + ggml_tensor * state_tok = ggml_add(C, mm_act(C, state_W2, se, at), state_b2); ggml_tensor * context_tokens = ggml_concat(C, context, ggml_reshape_2d(C, state_tok, E, 1), 1); struct DC { ggml_tensor *Wq, *bq, *K, *V; }; @@ -650,22 +685,23 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { dc[i].bq = inproj_split_b(C, w.bin, E, 0); ggml_tensor * bk = inproj_split_b(C, w.bin, E, 1); ggml_tensor * bv = inproj_split_b(C, w.bin, E, 2); - ggml_tensor * kp = ggml_add(C, ggml_mul_mat(C, Wk, context_tokens), bk); - ggml_tensor * vp = ggml_add(C, ggml_mul_mat(C, Wv, context_tokens), bv); + // cross-attention K/V feed the F32 attention core + ggml_tensor * kp = as_type(C, ggml_add(C, mm_act(C, Wk, context_tokens, at), bk), GGML_TYPE_F32); + ggml_tensor * vp = as_type(C, ggml_add(C, mm_act(C, Wv, context_tokens, at), bv), GGML_TYPE_F32); dc[i].K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, kp, hd_dit, dit_heads, Nctx), 0, 2, 1, 3)); dc[i].V = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, vp, hd_dit, dit_heads, Nctx), 1, 2, 0, 3)); } auto denoise = [&](ggml_tensor * x_seq_masked, int64_t time_index) -> ggml_tensor * { ggml_tensor * time_emb = ggml_cont(C, ggml_view_1d(C, time_pos, E, (size_t) time_index * E * sizeof(float))); - ggml_tensor * ae = ggml_relu(C, ggml_add(C, ggml_mul_mat(C, ae_W1, x_seq_masked), ae_b1)); + ggml_tensor * ae = ggml_relu(C, ggml_add(C, mm_act(C, ae_W1, x_seq_masked, at), ae_b1)); ae = ggml_add(C, ae, ae_pos); - ae = ggml_relu(C, ggml_add(C, ggml_mul_mat(C, ae_W2, ae), ae_b2)); - ggml_tensor * x = ggml_add(C, ggml_mul_mat(C, ae_W3, ae), ae_b3); + ae = ggml_relu(C, ggml_add(C, mm_act(C, ae_W2, ae, at), ae_b2)); + ggml_tensor * x = ggml_add(C, mm_act(C, ae_W3, ae, at), ae_b3); for (int64_t i = 0; i < dit_layers; ++i) { const auto & w = dit[i]; const auto & c = dc[i]; ggml_tensor * x_q = ggml_add(C, ggml_mul(C, ggml_norm(C, x, proj_ln_eps), w.n1w), w.n1b); - ggml_tensor * qp = ggml_add(C, ggml_mul_mat(C, c.Wq, x_q), c.bq); + ggml_tensor * qp = as_type(C, ggml_add(C, mm_act(C, c.Wq, x_q, at), c.bq), GGML_TYPE_F32); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, qp, hd_dit, dit_heads, horizon), 0, 2, 1, 3)); ggml_tensor * kq = ggml_mul_mat(C, c.K, Q); ggml_mul_mat_set_prec(kq, GGML_PREC_F32); // The Evo-1 reference cross-attends over the full padded context @@ -673,20 +709,21 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale_dit, 0.0f); ggml_tensor * kqv = ggml_mul_mat(C, c.V, aw); ggml_tensor * att_pre = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, kqv, 0, 2, 1, 3)), E, horizon); - ggml_tensor * attn_out = ggml_add(C, ggml_mul_mat(C, w.Wo, att_pre), w.bo); + ggml_tensor * attn_out = ggml_add(C, mm_act(C, w.Wo, as_type(C, att_pre, at), at), w.bo); ggml_tensor * x1 = ggml_add(C, x, attn_out); ggml_tensor * x2 = ggml_add(C, ggml_mul(C, ggml_norm(C, x1, proj_ln_eps), w.n2w), w.n2b); x2 = ggml_add(C, x2, time_emb); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.f1w, x2), w.f1b); + ggml_tensor * ff = ggml_add(C, mm_act(C, w.f1w, x2, at), w.f1b); ff = ggml_gelu_erf(C, ff); - ff = ggml_add(C, ggml_mul_mat(C, w.f2w, ff), w.f2b); + ff = ggml_add(C, mm_act(C, w.f2w, ff, at), w.f2b); x = ggml_add(C, x1, ff); } ggml_tensor * x_no = ggml_add(C, ggml_mul(C, ggml_norm(C, x, proj_ln_eps), norm_out_w), norm_out_b); ggml_tensor * x_flat = ggml_reshape_1d(C, ggml_cont(C, x_no), horizon * E); - ggml_tensor * x_pooled = ggml_add(C, ggml_mul_mat(C, seq_pool_w, x_flat), seq_pool_b); - ggml_tensor * mh = ggml_relu(C, ggml_add(C, ggml_mul_mat(C, head_W1, x_pooled), head_b1)); - return ggml_add(C, ggml_mul_mat(C, head_W2, mh), head_b2); + ggml_tensor * x_pooled = ggml_add(C, mm_act(C, seq_pool_w, x_flat, at), seq_pool_b); + ggml_tensor * mh = ggml_relu(C, ggml_add(C, mm_act(C, head_W1, x_pooled, at), head_b1)); + // the Euler integrator below accumulates in F32 + return as_type(C, ggml_add(C, mm_act(C, head_W2, mh, at), head_b2), GGML_TYPE_F32); }; const float dt = 1.0f / (float) num_steps; @@ -694,7 +731,7 @@ std::vector Evo1ModelArch::predict(const Inputs& in) { for (int64_t step = 0; step < num_steps; ++step) { const int64_t time_index = (int64_t) ((double) step / (double) num_steps * 1000.0); ggml_tensor * x_seq = ggml_reshape_2d(C, x_action, per_a, horizon); - ggml_tensor * x_seq_masked = ggml_mul(C, x_seq, t_amask); + ggml_tensor * x_seq_masked = as_type(C, ggml_mul(C, x_seq, t_amask), at); ggml_tensor * v_t = denoise(x_seq_masked, time_index); x_action = ggml_add(C, x_action, ggml_scale(C, v_t, dt)); } diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index a1cf41d..4369d29 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -25,6 +25,7 @@ #include "models/scratch_ctx.h" #include "models/dit_common.h" #include "models/vision_common.h" +#include "models/act_dtype.h" #include #include @@ -98,6 +99,10 @@ struct Pi0ModelArch : public ModelArchBase { // Opened once at load: reopening per predict re-parses the whole GGUF header. gguf_reader io{"pi0"}; ggml_type matmul_type = GGML_TYPE_BF16; + // Activation dtype carried between ops. F32 by default; BF16 under + // VLA_PI0_BF16_ACT, which removes the per-GEMM F32<->BF16 round trip ggml + // pays when BF16 weights meet F32 activations. See mm_act/as_type below. + ggml_type act_type = GGML_TYPE_F32; // In-tree SigLIP-So400m/14 vision tower (was llama.cpp clip.cpp mmproj). int64_t vit_hidden = 1152, vit_layers = 27, vit_heads = 16; @@ -132,21 +137,24 @@ namespace { // attention (nullptr mask), F32 score accumulation, tanh GELU FFN. // Fused attention for the SigLIP tower and the PaliGemma/expert stack. // OPT-IN (VLA_PI0_FA=1): pi0's score matrices are small (~560 keys, 8 heads), so -// fusing them only moved 111.4 ms -> 107.5 ms (3.5%). ggml's FA computes K/V at -// F16 regardless of the input type, and on evo1 that cost measurable success -// rate — not a trade worth taking here for 3.5%. +// fusing them only moved 111.4 ms -> 107.5 ms (3.5%), and ggml's FA computes K/V +// at F16 regardless of the input type. pi0's flash-attention SR was never +// measured, so it stays opt-in on an unquantified risk rather than a measured +// cost. (The evo1 SR drop this used to cite did not reproduce.) +// VLA_PI0_BF16_ACT is the better lever here: 9.1%, and its SR was measured. static inline bool pi0_fa_enabled() { static const bool enabled = (std::getenv("VLA_PI0_FA") != nullptr); return enabled; } ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_tensor * x, - int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps) { + int64_t seq, int64_t heads, int64_t head_dim, int64_t hidden, float ln_eps, + ggml_type at) { const float scale = 1.0f / std::sqrt((float) head_dim); ggml_tensor * n1 = ggml_add(C, ggml_mul(C, ggml_norm(C, x, ln_eps), w.ln1w), w.ln1b); - ggml_tensor * q = ggml_add(C, ggml_mul_mat(C, w.Wq, n1), w.bq); - ggml_tensor * k = ggml_add(C, ggml_mul_mat(C, w.Wk, n1), w.bk); - ggml_tensor * v = ggml_add(C, ggml_mul_mat(C, w.Wv, n1), w.bv); + ggml_tensor * q = as_type(C, ggml_add(C, mm_act(C, w.Wq, n1, at), w.bq), GGML_TYPE_F32); + ggml_tensor * k = as_type(C, ggml_add(C, mm_act(C, w.Wk, n1, at), w.bk), GGML_TYPE_F32); + ggml_tensor * v = as_type(C, ggml_add(C, mm_act(C, w.Wv, n1, at), w.bv), GGML_TYPE_F32); ggml_tensor * Q = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, q, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * K = ggml_cont(C, ggml_permute(C, ggml_reshape_3d(C, k, head_dim, heads, seq), 0, 2, 1, 3)); ggml_tensor * att; @@ -163,9 +171,9 @@ ggml_tensor * build_siglip_layer(ggml_context * C, const SigLipLayerW & w, ggml_ ggml_tensor * aw = ggml_soft_max_ext(C, kq, nullptr, scale, 0.0f); att = ggml_reshape_2d(C, ggml_cont(C, ggml_permute(C, ggml_mul_mat(C, V, aw), 0, 2, 1, 3)), hidden, seq); } - ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, ggml_mul_mat(C, w.Wo, att), w.bo)); + ggml_tensor * h1 = ggml_add(C, x, ggml_add(C, mm_act(C, w.Wo, as_type(C, att, at), at), w.bo)); ggml_tensor * n2 = ggml_add(C, ggml_mul(C, ggml_norm(C, h1, ln_eps), w.ln2w), w.ln2b); - ggml_tensor * ff = ggml_add(C, ggml_mul_mat(C, w.Wfc2, ggml_gelu(C, ggml_add(C, ggml_mul_mat(C, w.Wfc1, n2), w.bfc1))), w.bfc2); + ggml_tensor * ff = ggml_add(C, mm_act(C, w.Wfc2, ggml_gelu(C, ggml_add(C, mm_act(C, w.Wfc1, n2, at), w.bfc1)), at), w.bfc2); return ggml_add(C, h1, ff); } @@ -176,7 +184,7 @@ ggml_tensor * build_gemma_layer( ggml_tensor * x_in, ggml_tensor * positions, const Config & cfg, int64_t seq, float rope_base, ggml_tensor * cached_K, ggml_tensor * cached_V, ggml_tensor * mask, - ggml_tensor ** k_out, ggml_tensor ** v_out) { + ggml_tensor ** k_out, ggml_tensor ** v_out, ggml_type at) { const int64_t hd = cfg.head_dim; const int64_t nq = cfg.n_q_heads; const int64_t nkv = cfg.n_kv_heads; @@ -184,9 +192,11 @@ ggml_tensor * build_gemma_layer( ggml_tensor * x_norm = ggml_mul(ctx, ggml_rms_norm(ctx, x_in, cfg.rms_eps), w.ln_in); - ggml_tensor * q = ggml_mul_mat(ctx, w.Wq, x_norm); - ggml_tensor * k = ggml_mul_mat(ctx, w.Wk, x_norm); - ggml_tensor * v = ggml_mul_mat(ctx, w.Wv, x_norm); + // Q/K/V land in F32: RoPE, the KV cache the suffix passes re-read, and the + // score/softmax core all stay full precision. + ggml_tensor * q = as_type(ctx, mm_act(ctx, w.Wq, x_norm, at), GGML_TYPE_F32); + ggml_tensor * k = as_type(ctx, mm_act(ctx, w.Wk, x_norm, at), GGML_TYPE_F32); + ggml_tensor * v = as_type(ctx, mm_act(ctx, w.Wv, x_norm, at), GGML_TYPE_F32); ggml_tensor * q_h = ggml_reshape_3d(ctx, q, hd, nq, seq); ggml_tensor * k_h = ggml_reshape_3d(ctx, k, hd, nkv, seq); @@ -232,25 +242,28 @@ ggml_tensor * build_gemma_layer( att_pre = ggml_reshape_2d(ctx, ggml_cont(ctx, ggml_permute(ctx, kqv, 0, 2, 1, 3)), qf, seq); } - ggml_tensor * o_out = ggml_mul_mat(ctx, w.Wo, att_pre); + ggml_tensor * o_out = mm_act(ctx, w.Wo, as_type(ctx, att_pre, at), at); ggml_tensor * h1 = ggml_add(ctx, x_in, o_out); ggml_tensor * x_norm_mlp = ggml_mul(ctx, ggml_rms_norm(ctx, h1, cfg.rms_eps), w.ln_post); - ggml_tensor * gate = ggml_mul_mat(ctx, w.Wgate, x_norm_mlp); - ggml_tensor * up = ggml_mul_mat(ctx, w.Wup, x_norm_mlp); + ggml_tensor * gate = mm_act(ctx, w.Wgate, x_norm_mlp, at); + ggml_tensor * up = mm_act(ctx, w.Wup, x_norm_mlp, at); ggml_tensor * inter_t = ggml_mul(ctx, ggml_gelu(ctx, gate), up); - ggml_tensor * mlp_out = ggml_mul_mat(ctx, w.Wdown, inter_t); + ggml_tensor * mlp_out = mm_act(ctx, w.Wdown, inter_t, at); return ggml_add(ctx, h1, mlp_out); } ggml_tensor * build_embed_suffix(ggml_context * ctx, const Pi0ModelArch & m, ggml_tensor * state, ggml_tensor * x, ggml_tensor * time_bcast) { - ggml_tensor * state_emb = ggml_add(ctx, ggml_mul_mat(ctx, m.W_sp, state), m.b_sp); - ggml_tensor * action_emb = ggml_add(ctx, ggml_mul_mat(ctx, m.W_ain, x), m.b_ain); - ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, time_bcast, 0); - ggml_tensor * mlp1 = ggml_add(ctx, ggml_mul_mat(ctx, m.W_at1, action_time_in), m.b_at1); + const ggml_type at = m.act_type; + ggml_tensor * state_emb = ggml_add(ctx, mm_act(ctx, m.W_sp, as_type(ctx, state, at), at), m.b_sp); + // x is the F32 flow-matching state; time_bcast is an F32 input tensor. Both + // enter the expert in the activation dtype, and ggml_concat needs them to agree. + ggml_tensor * action_emb = ggml_add(ctx, mm_act(ctx, m.W_ain, as_type(ctx, x, at), at), m.b_ain); + ggml_tensor * action_time_in = ggml_concat(ctx, action_emb, as_type(ctx, time_bcast, at), 0); + ggml_tensor * mlp1 = ggml_add(ctx, mm_act(ctx, m.W_at1, action_time_in, at), m.b_at1); ggml_tensor * mlp1_silu = ggml_silu(ctx, mlp1); - ggml_tensor * action_time_emb = ggml_add(ctx, ggml_mul_mat(ctx, m.W_at2, mlp1_silu), m.b_at2); + ggml_tensor * action_time_emb = ggml_add(ctx, mm_act(ctx, m.W_at2, mlp1_silu, at), m.b_at2); ggml_tensor * state_emb_2d = ggml_reshape_2d(ctx, state_emb, state_emb->ne[0], 1); return ggml_concat(ctx, state_emb_2d, action_time_emb, 1); } @@ -373,6 +386,16 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, const Backend b = backend_init("vla(pi0)", m->n_threads); if (!b.handle) { return nullptr; } m->backend = b.handle; + + // BF16 activations need BF16-resident weights and the CUDA BF16 GEMM path. + if (std::getenv("VLA_PI0_BF16_ACT")) { + if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { + m->act_type = GGML_TYPE_BF16; + std::printf("vla(pi0): activations = BF16 (VLA_PI0_BF16_ACT)\n"); + } else { + std::fprintf(stderr, "vla(pi0): VLA_PI0_BF16_ACT ignored - needs CUDA and BF16 weights\n"); + } + } } // The SigLIP tower is now bundled in the ckpt GGUF; mmproj_path is ignored. @@ -530,14 +553,16 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { ggml_tensor * t_px = ggml_new_tensor_3d(VC, GGML_TYPE_F32, vit_image_size, vit_image_size, 3); ggml_set_input(t_px); ggml_tensor * conv = ggml_conv_2d(VC, vit_patch_w, t_px, (int) vit_patch_size, (int) vit_patch_size, 0, 0, 1, 1); ggml_tensor * patches = ggml_cont(VC, ggml_transpose(VC, ggml_reshape_2d(VC, conv, grid * grid, vit_hidden))); - ggml_tensor * h = ggml_add(VC, ggml_add(VC, patches, vit_patch_b), vit_pos); + // patch embed (conv_2d) stays F32; the tower runs in the activation dtype + ggml_tensor * h = as_type(VC, ggml_add(VC, ggml_add(VC, patches, vit_patch_b), vit_pos), act_type); for (int64_t i = 0; i < vit_layers; ++i) - h = build_siglip_layer(VC, vit[i], h, K, vit_heads, vit_hidden / vit_heads, vit_hidden, vit_ln_eps); + h = build_siglip_layer(VC, vit[i], h, K, vit_heads, vit_hidden / vit_heads, vit_hidden, vit_ln_eps, act_type); h = ggml_add(VC, ggml_mul(VC, ggml_norm(VC, h, vit_ln_eps), vit_post_ln_w), vit_post_ln_b); // PaliGemma projector: linear (+ optional bias), then 1/sqrt(hidden) scale (matches clip.cpp siglip.cpp). - ggml_tensor * proj = ggml_mul_mat(VC, mm_proj_w, h); + ggml_tensor * proj = mm_act(VC, mm_proj_w, h, act_type); if (mm_proj_b) proj = ggml_add(VC, proj, mm_proj_b); - ggml_tensor * vit_emb = ggml_scale(VC, proj, 1.0f / std::sqrt((float) proj->ne[0])); + // read back to the host as F32 + ggml_tensor * vit_emb = as_type(VC, ggml_scale(VC, proj, 1.0f / std::sqrt((float) proj->ne[0])), GGML_TYPE_F32); ggml_set_output(vit_emb); ggml_cgraph * vg = ggml_new_graph_custom(VC, 8192, false); @@ -593,7 +618,8 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { } const float lang_scale = (float) std::sqrt((double) hidden_pl); - ggml_tensor * prefix_embs = ggml_concat(C, t_image_emb, ggml_scale(C, t_lang_emb, lang_scale), 1); + ggml_tensor * prefix_embs = as_type(C, + ggml_concat(C, t_image_emb, ggml_scale(C, t_lang_emb, lang_scale), 1), act_type); std::vector cK(n_layers), cV(n_layers); { @@ -601,11 +627,13 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { for (int64_t i = 0; i < n_layers; ++i) { h = build_gemma_layer(C, pl_layers[i], h, t_prefix_pos, cfg, n_prefix, rope_base, nullptr, nullptr, nullptr, - &cK[i], &cV[i]); + &cK[i], &cV[i], act_type); } (void) h; } + // x_t is the flow-matching state; it stays F32 so num_steps Euler updates + // do not accumulate in 8 mantissa bits. ggml_tensor * x_t = t_x0; std::vector v_steps(num_steps); for (int step = 0; step < num_steps; ++step) { @@ -613,12 +641,15 @@ std::vector Pi0ModelArch::predict(const Inputs& in) { for (int64_t i = 0; i < n_layers; ++i) { h = build_gemma_layer(C, ex_layers[i], h, t_suffix_pos, cfg, n_suf, rope_base, cK[i], cV[i], t_full_mask, - nullptr, nullptr); + nullptr, nullptr, act_type); } ggml_tensor * h_final = ggml_mul(C, ggml_rms_norm(C, h, cfg.rms_eps), ex_final_norm); - const size_t rb = (size_t) hidden_ex * sizeof(float); + // row stride follows h_final's dtype, which is BF16 on the BF16 path + const size_t rb = (size_t) hidden_ex * ggml_element_size(h_final); ggml_tensor * h_actions = ggml_view_2d(C, h_final, hidden_ex, chunk, rb, rb); - ggml_tensor * v_t = ggml_add(C, ggml_mul_mat(C, W_aout, h_actions), b_aout); + // h_actions is an offset slice of whole rows, so it is already contiguous + ggml_tensor * v_t = as_type(C, + ggml_add(C, mm_act(C, W_aout, h_actions, act_type), b_aout), GGML_TYPE_F32); v_steps[step] = v_t; x_t = ggml_add(C, x_t, ggml_scale(C, v_t, dt)); } From e29488a1e737ec6befa8e2e29b9f5f01a686c043 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 14:36:45 +0700 Subject: [PATCH 38/42] widen the bitvla ternary GEMM to four column tiles per CTA and pad its shared-memory stride to break bank conflicts --- src/kernels/bitvla/bitnet_kernels.cu | 28 ++++ src/kernels/bitvla/bitnet_kernels.h | 185 +++++++++++++++++++++++++++ tests/CMakeLists.txt | 4 + 3 files changed, 217 insertions(+) diff --git a/src/kernels/bitvla/bitnet_kernels.cu b/src/kernels/bitvla/bitnet_kernels.cu index 7cb0f37..5669cd5 100644 --- a/src/kernels/bitvla/bitnet_kernels.cu +++ b/src/kernels/bitvla/bitnet_kernels.cu @@ -62,12 +62,40 @@ extern "C" void bitlinear_int8xint2(int8_t* input0, int8_t* input1, __nv_bfloat1 } } +// The wide kernel amortises the shared A block over 4 column tiles instead of +// 1 (see ladder_int8xint2_kernel_m_wide). It is the default; set +// VLA_BITVLA_NARROW_GEMM=1 to fall back to the one-tile-per-CTA kernel, which +// is what the A/B correctness harness and any regression bisect want. +static bool bitlinear_use_wide() { + static const bool wide = (std::getenv("VLA_BITVLA_NARROW_GEMM") == nullptr); + return wide; +} + extern "C" void bitlinear_int8xint2_m( int8_t* input0, int8_t* input1, __nv_bfloat16* output0, float* s, float* ws, int M, int N, int K, cudaStream_t stream) { + if (bitlinear_use_wide()) { +#define WIDE(NN, KK, WS) \ + launch_ladder_int8xint2_m_wide( \ + input0, input1, output0, s, ws, M, stream) + if (N == 2560 && K == 2560) WIDE(2560, 2560, 1); + else if (N == 640 && K == 2560) WIDE(640, 2560, 1); + else if (N == 13824 && K == 2560) WIDE(13824, 2560, 2); + else if (N == 2560 && K == 6912) WIDE(2560, 6912, 1); + + else if (N == 1152 && K == 1152) WIDE(1152, 1152, 1); + else if (N == 4304 && K == 1152) WIDE(4304, 1152, 1); + else if (N == 1152 && K == 4352) WIDE(1152, 4352, 1); + + else if (N == 3840 && K == 2560) WIDE(3840, 2560, 3); + else + bitlinear_unsupported_shape("bitlinear_int8xint2_m", M, N, K); +#undef WIDE + return; + } if (N == 2560 && K == 2560) launch_ladder_int8xint2_m<2560, 2560, 1, 128>(input0, input1, output0, s, ws, M, stream); else if (N == 640 && K == 2560) launch_ladder_int8xint2_m<640, 2560, 1, 128>(input0, input1, output0, s, ws, M, stream); diff --git a/src/kernels/bitvla/bitnet_kernels.h b/src/kernels/bitvla/bitnet_kernels.h index d4348a7..73e114c 100644 --- a/src/kernels/bitvla/bitnet_kernels.h +++ b/src/kernels/bitvla/bitnet_kernels.h @@ -259,6 +259,191 @@ static inline void launch_ladder_int8xint2_m( <<>>(A, B, out, s, ws, M); } +/** + * @brief Multi-row ternary GEMM producing @p N_TILES column tiles per CTA. + * + * Same math as @ref ladder_int8xint2_kernel_m; the difference is reuse. That + * kernel gives each CTA a single 16-wide output tile, so the A block it stages + * into shared memory buys only 16 columns of work and the whole activation + * matrix is re-read N/16 times. At the production shapes that lands the GEMM + * at ~15 MAC/byte against a roofline balance point of ~152, i.e. bandwidth + * bound at roughly an eighth of the int8 tensor-core peak. + * + * Here one A block feeds @p N_TILES tiles: warp @c w owns tile + * @c blockIdx.x*N_TILES+w and sweeps every row-tile of the shared A block, so + * the per-CTA activation traffic is amortised over @p N_TILES times as many + * MACs. The weight pack is already blocked as contiguous (16 columns x K) + * groups, so tile @c t simply starts at @c t*16*K/4 bytes -- no repacking. + * + * @tparam N Output column count. + * @tparam K Reduction dimension. + * @tparam ws_num Number of column groups sharing one scale entry. + * @tparam M_ROWS Rows per CTA along M; must be a multiple of 16. + * @tparam N_TILES Column tiles per CTA; must equal the warp count (4). + */ +template +__global__ void __launch_bounds__(128) ladder_int8xint2_kernel_m_wide( + int8_t* __restrict__ A, int8_t* __restrict__ B, + __nv_bfloat16* __restrict__ out, + float* __restrict__ s, float* __restrict__ ws, int M) +{ + using namespace nvcuda; + constexpr int K_per_loop = 16, wmma_K = 32, wmma_N = 16; + constexpr int K_CHUNK = 128; + constexpr int WARPS = 4; + constexpr int M_TILES = M_ROWS / 16; + constexpr int N_BLOCKS = N / 16; // total 16-wide column tiles in the matrix + + // Warps split two ways. N_TILES of them take different column tiles (that is + // the A-reuse win); the remaining WARPS/N_TILES take different row ranges + // (that is parallelism, which matters when N is small enough that column + // tiles alone cannot fill the GPU). N_TILES == 1 reproduces the original + // kernel's mapping exactly. + constexpr int M_GROUPS = WARPS / N_TILES; + constexpr int M_PER_WARP = M_TILES / M_GROUPS; + + // Row stride padded to break shared-memory bank conflicts: at a stride of + // 128 B every row of a 16-row fragment starts on bank 0, so each + // load_matrix_sync serialises 16 ways. 144 B (still a multiple of the 16 B + // that wmma requires for integer ldm) spreads them over 8 banks. + constexpr int SM_STRIDE = K_CHUNK + 16; + + const int tx = (int)threadIdx.x; // 0..7 + const int ty = (int)threadIdx.y; // 0..15 + const int tid = ty * 8 + tx; // 0..127 + const int warp = tid >> 5; // 0..3 + const int lane = tid & 31; + const int m_base = (int)blockIdx.y * M_ROWS; + + __shared__ signed char A_smem[M_ROWS][SM_STRIDE]; + __shared__ signed char W_smem[N_TILES][16][SM_STRIDE]; + __shared__ int C_smem[WARPS][16][16]; + + // Column tile this warp owns. N is not always a multiple of 16*N_TILES + // (the ViT's 4304 is 269 tiles), so tiles past the end are skipped rather + // than clamped -- clamping would double-write real columns. + const int my_tile = (int)blockIdx.x * N_TILES + (warp % N_TILES); + const bool my_tile_valid = my_tile < N_BLOCKS; + const int m_tile_base = (warp / N_TILES) * M_PER_WARP; + + int B_reshape_local[1]; + signed char B_decode_local[K_per_loop]; + + wmma::fragment a_frag; + wmma::fragment b_frag; + wmma::fragment acc[M_PER_WARP]; + #pragma unroll + for (int t = 0; t < M_PER_WARP; ++t) wmma::fill_fragment(acc[t], 0); + + for (int k_0 = 0; k_0 < K / K_CHUNK; ++k_0) { + #pragma unroll + for (int r = 0; r < (M_ROWS * K_CHUNK / 16) / 128; ++r) { + const int idx = tid + r * 128; + const int m = idx >> 3; + const int kk = (idx & 7) * 16; + const int mrow = m_base + m; + const int8_t* aptr = A + ((mrow < M ? mrow : 0) * K) + k_0 * K_CHUNK + kk; + *(int4*)(&A_smem[m][kk]) = *(const int4*)aptr; + } + + // All 128 threads cooperate on one weight tile at a time, reproducing the + // pack's swizzle exactly; only the tile base changes per j. + #pragma unroll + for (int j = 0; j < N_TILES; ++j) { + const int tile = (int)blockIdx.x * N_TILES + j; + if (tile < N_BLOCKS) { + B_reshape_local[0] = *(int*)(B + + ((size_t)tile * 16 * K / 4) + + (k_0 * 8 * K_per_loop * wmma_N / 4) + + ((tx >> 1) * wmma_K * wmma_N / 4) + + ((ty >> 3) * (wmma_K * wmma_N / 2) / 4) + + ((tx & 1) * (wmma_K * wmma_N / 4) / 4) + + ((ty & 7) * (wmma_K / 2) / 4)); + decode_i2s_to_i8s(B_reshape_local, B_decode_local, 16); + *(int4*)(&W_smem[j][ty][tx * 16]) = *(int4*)(&B_decode_local[0]); + } + } + __syncthreads(); + + if (my_tile_valid) { + #pragma unroll + for (int k16 = 0; k16 < K_CHUNK / 16; ++k16) { + wmma::load_matrix_sync(b_frag, &W_smem[warp % N_TILES][0][k16 * 16], SM_STRIDE); + #pragma unroll + for (int t = 0; t < M_PER_WARP; ++t) { + wmma::load_matrix_sync(a_frag, &A_smem[(m_tile_base + t) * 16][k16 * 16], SM_STRIDE); + wmma::mma_sync(acc[t], a_frag, b_frag, acc[t]); + } + } + } + __syncthreads(); + } + + if (!my_tile_valid) return; + + const int n_base = my_tile * 16; + const float wsv = ws[n_base / (N / ws_num)]; + + // One row-tile at a time through a per-warp staging buffer: a full + // M_ROWS x 16 int32 buffer per warp would cost more shared memory than the + // A block it is meant to amortise. + #pragma unroll + for (int t = 0; t < M_PER_WARP; ++t) { + wmma::store_matrix_sync(&C_smem[warp][0][0], acc[t], 16, wmma::mem_row_major); + __syncwarp(); + #pragma unroll + for (int e = 0; e < (16 * 16) / 32; ++e) { + const int lin = lane + e * 32; + const int ml = lin >> 4; + const int col = lin & 15; + const int m = m_base + (m_tile_base + t) * 16 + ml; + if (m < M) + out[(size_t)m * N + n_base + col] = + __float2bfloat16(((float)C_smem[warp][ml][col]) / s[m] * wsv); + } + __syncwarp(); + } +} + +/** + * @brief Column tiles per CTA for a given BitVLA shape. + * + * More tiles means more A reuse per CTA but fewer CTAs overall, and the grid + * is only ceil(N/16/N_TILES) x ceil(M/128) against 82 SMs -- so past a point + * the reuse is paid for with an idle machine. The trade does not reduce to a + * function of N: q/o and down share N=2560 but want 4 and 2 respectively, + * because down's K=6912 makes each CTA long-running enough that wave + * quantisation costs more than the extra reuse saves. + * + * These are measured, not derived -- `tests/bitvla_gemm_check sweep` prints + * the table they come from, at the M each shape actually runs at (the LM sees + * the full ~600-token prompt, the ViT one 256-patch view per call). + */ +static constexpr int bitvla_n_tiles_for(int N, int K) { + return (N == 2560 && K == 2560) ? 4 // lm.q / lm.o + : (N == 640 && K == 2560) ? 2 // lm.k / lm.v + : (N == 13824 && K == 2560) ? 4 // lm.gate_up + : (N == 2560 && K == 6912) ? 2 // lm.down + : (N == 1152 && K == 1152) ? 2 // vit.q/k/v/o + : (N == 4304 && K == 1152) ? 4 // vit.fc1 + : (N == 1152 && K == 4352) ? 2 // vit.fc2 + : (N == 3840 && K == 2560) ? 4 // action head qkv + : 2; +} + +/** + * @brief Launch helper for @ref ladder_int8xint2_kernel_m_wide. + */ +template +static inline void launch_ladder_int8xint2_m_wide( + int8_t* A, int8_t* B, __nv_bfloat16* out, + float* s, float* ws, int M, cudaStream_t stream) { + constexpr int N_BLOCKS = N / 16; + ladder_int8xint2_kernel_m_wide + <<>>(A, B, out, s, ws, M); +} + /** * @brief Row-wise int8 quantisation of a bf16 activation matrix. * diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 7a227ab..6eba2a0 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -39,3 +39,7 @@ target_include_directories(test_config_guard PRIVATE ${CMAKE_SOURCE_DIR}/src) target_link_libraries(test_config_guard PRIVATE vla_core) target_compile_options(test_config_guard PRIVATE -Wall -Wextra) add_test(NAME config_guard COMMAND test_config_guard) + +# The A/B harness for the two BitVLA ternary-GEMM tilings (bitvla_gemm_check.cu) +# was never committed, so there is no target for it here. VLA_BITVLA_NARROW_GEMM=1 +# still selects the old one-tile-per-CTA kernel for a hand-run comparison. From 4a508fc531ab5fb08b1928cc23de2a45aed1b941 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 14:36:45 +0700 Subject: [PATCH 39/42] update the headline table comparing vla.cpp and best pytorch --- eval/client/benchmark.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/eval/client/benchmark.py b/eval/client/benchmark.py index d3602bb..41873b5 100644 --- a/eval/client/benchmark.py +++ b/eval/client/benchmark.py @@ -352,6 +352,11 @@ def main() -> int: }, "server_ms": _summarize(server_infer_ms), "server_latency_breakdown": server_latencies, + # Raw per-call series. Summary statistics cannot distinguish a uniformly + # slower stack from one that hits the same floor but occasionally + # stalls, and that distinction is the whole question on gr00t_n1_5. + "server_ms_raw": [round(x, 3) for x in server_infer_ms], + "step_ms_raw": [round(x, 3) for x in step_latencies_ms], } if args.variant: stats["variant"] = args.variant From 5d9ec9e908e9b0aa0460ce976daf2dac53a14d60 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 16:20:49 +0700 Subject: [PATCH 40/42] move the BF16 activation kernels in-tree to src/cuda and reduce the ggml patch to a single extension hook --- CMakeLists.txt | 32 +- scripts/patch_ggml_bf16_activations.py | 765 ------------------------- scripts/patch_ggml_cuda_ext_hook.py | 108 ++++ src/cuda/vla_cuda_bf16.cu | 373 ++++++++++++ src/cuda/vla_cuda_ops.h | 30 + src/models/act_dtype.h | 32 +- src/models/evo1.cpp | 2 + src/models/pi0.cpp | 2 + tests/CMakeLists.txt | 10 + tests/test_bf16_cuda_ops.cpp | 155 +++++ 10 files changed, 735 insertions(+), 774 deletions(-) delete mode 100755 scripts/patch_ggml_bf16_activations.py create mode 100644 scripts/patch_ggml_cuda_ext_hook.py create mode 100644 src/cuda/vla_cuda_bf16.cu create mode 100644 src/cuda/vla_cuda_ops.h create mode 100644 tests/test_bf16_cuda_ops.cpp diff --git a/CMakeLists.txt b/CMakeLists.txt index fca4587..ac7d85b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -34,12 +34,13 @@ set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) # llama.cpp fetched + pinned at configure; bump = one-line GIT_TAG change. # -# The fetched ggml is patched in place to support BF16 activations (adds -# ggml_mul_mat_t plus BF16 CUDA elementwise/norm kernels) — src/models/*.cpp -# call ggml_mul_mat_t, so this is required to compile, not optional. The script -# is idempotent, so a re-configure over an already-patched tree is a no-op; it +# The fetched ggml gets one addition: a function-pointer hook at the top of the +# CUDA op dispatch, so the in-tree BF16 kernels in src/cuda/ can take an op +# before ggml does. That hook is the whole modification — the kernels are +# ordinary first-party code and need no patching. Left unregistered the pointer +# is null and ggml behaves exactly as shipped. The script is idempotent, and # fails loudly rather than half-applying if an anchor stops matching after a -# GIT_TAG bump. See scripts/patch_ggml_bf16_activations.py. +# GIT_TAG bump. See scripts/patch_ggml_cuda_ext_hook.py. find_package(Python3 COMPONENTS Interpreter REQUIRED) include(FetchContent) FetchContent_Declare(llama @@ -47,7 +48,7 @@ FetchContent_Declare(llama GIT_TAG b10331 GIT_SHALLOW TRUE PATCH_COMMAND ${Python3_EXECUTABLE} - ${CMAKE_CURRENT_SOURCE_DIR}/scripts/patch_ggml_bf16_activations.py + ${CMAKE_CURRENT_SOURCE_DIR}/scripts/patch_ggml_cuda_ext_hook.py ) FetchContent_MakeAvailable(llama) @@ -113,6 +114,25 @@ if(GGML_CUDA) target_link_libraries(vla_core PRIVATE bitvla_cuda_kernels) target_compile_definitions(vla_core PUBLIC VLA_BITVLA_CUDA_KERNELS) target_include_directories(vla_core PUBLIC ${CUDAToolkit_INCLUDE_DIRS}) + + # BF16 activation kernels. These run through the extension hook the fetched + # ggml carries (scripts/patch_ggml_cuda_ext_hook.py); the hook resolves + # ggml_cuda_ext_forward out of ggml-cuda, hence the explicit link. + add_library(vla_cuda_ops STATIC src/cuda/vla_cuda_bf16.cu) + target_include_directories(vla_cuda_ops PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/src + ${llama_SOURCE_DIR}/ggml/include + ) + set_target_properties(vla_cuda_ops PROPERTIES + CUDA_SEPARABLE_COMPILATION ON + POSITION_INDEPENDENT_CODE ON + ) + target_compile_features(vla_cuda_ops PRIVATE cxx_std_17) + target_compile_options(vla_cuda_ops PRIVATE + $<$:-O3 -Xptxas=-O3> + ) + target_link_libraries(vla_cuda_ops PUBLIC CUDA::cublas CUDA::cudart ggml-cuda) + target_link_libraries(vla_core PRIVATE vla_cuda_ops) endif() # Intel GPUs (Arc / Flex / Data Center Max / Xe iGPU) through oneAPI SYCL. diff --git a/scripts/patch_ggml_bf16_activations.py b/scripts/patch_ggml_bf16_activations.py deleted file mode 100755 index 7941d8f..0000000 --- a/scripts/patch_ggml_bf16_activations.py +++ /dev/null @@ -1,765 +0,0 @@ -#!/usr/bin/env python3 -# Copyright 2026 VinRobotics -# -# Licensed under the Apache License, Version 2.0 (the "License"); -# you may not use this file except in compliance with the License. -# You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, software -# distributed under the License is distributed on an "AS IS" BASIS, -# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -# See the License for the specific language governing permissions and -# limitations under the License. - -"""Teach the fetched ggml how to carry BF16 activations (CUDA backend). - -Why this exists ---------------- -ggml_mul_mat's result is F32 by definition, so a BF16-resident weight meeting an -F32 activation makes ggml_cuda_op_mul_mat_cublas convert src1 F32->BF16 on the -way into every GEMM and the result BF16->F32 on the way out. An nsys trace of -the evo1 server put those convert_unary launches at 11.1% of GPU time, in -exactly balanced pairs (8,228 each direction). Carrying activations as BF16 -removes both, and halves the bytes every bias-add, norm, activation and layout -copy has to move. - -What it changes ---------------- - ggml.c / ggml.h ggml_mul_mat_t(ctx, a, b, type) - mul_mat with an explicit - result type; ggml_mul_mat becomes a wrapper at F32. - ggml-cuda.cu a direct cuBLAS BF16xBF16->BF16 path, taken whenever a - caller asked for a BF16 matmul result. - binbcast.cu BF16 add/mul (activation x activation, activation x F32 - bias/weight), plain and fused. - unary.cu BF16 gelu/silu/relu/... - norm.cu BF16 norm / rms_norm / fused rms_norm+mul, float reductions. - scale.cu BF16 scale. - -concat.cu needs nothing: it already dispatches on ggml_type_size to a -width-generic kernel, so BF16 lands on the uint16_t instantiation. - -Everything keeps float accumulation, so only operand and result *storage* -changes, never a reduction. All BF16 branches are additive: an F32 graph hits -exactly the code it hit before. - -llama.cpp is pulled in by FetchContent (see the top-level CMakeLists), so the -tree lives under build/_deps/llama-src and a fresh configure re-clones it. Run -this after configuring and before building. It is idempotent. - -Usage: scripts/patch_ggml_bf16_activations.py [] -""" - -import pathlib -import sys - -MARKER = "vla.cpp: BF16 activation support" - - -def edit(path, subs): - """Apply (old, new) pairs to `path`; every `old` must appear exactly once. - - Returns the new text rather than writing, so main() can apply every file's - edits or none: a half-patched tree is worse than an unpatched one. - """ - text = path.read_text() - for old, new in subs: - n = text.count(old) - if n != 1: - raise SystemExit( - f"{path}: anchor found {n} times, expected 1:\n---\n{old[:400]}\n---" - ) - text = text.replace(old, new) - return text - - -# --------------------------------------------------------------------------- -# ggml.c / ggml.h - mul_mat with an explicit result type -# --------------------------------------------------------------------------- -GGML_C = [( - """struct ggml_tensor * ggml_mul_mat( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b) { - GGML_ASSERT(ggml_can_mul_mat(a, b)); - GGML_ASSERT(!ggml_is_transposed(a)); - - const int64_t ne[4] = { a->ne[1], b->ne[1], b->ne[2], b->ne[3] }; - struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, ne); - - result->op = GGML_OP_MUL_MAT; - result->src[0] = a; - result->src[1] = b; - - return result; -}""", - """// vla.cpp: BF16 activation support - mul_mat with an explicit result type. -// -// ggml_mul_mat always produces F32, which forces a BF16->F32 conversion out of -// every cuBLAS BF16 GEMM and an F32->BF16 one back in at the next matmul. -// Letting the caller ask for a BF16 result is what makes an end-to-end BF16 -// activation graph expressible. Only CUDA implements a non-F32 result; see -// ggml_cuda_mul_mat_bf16. -struct ggml_tensor * ggml_mul_mat_t( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - enum ggml_type type) { - GGML_ASSERT(ggml_can_mul_mat(a, b)); - GGML_ASSERT(!ggml_is_transposed(a)); - - const int64_t ne[4] = { a->ne[1], b->ne[1], b->ne[2], b->ne[3] }; - struct ggml_tensor * result = ggml_new_tensor(ctx, type, 4, ne); - - result->op = GGML_OP_MUL_MAT; - result->src[0] = a; - result->src[1] = b; - - return result; -} - -struct ggml_tensor * ggml_mul_mat( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b) { - return ggml_mul_mat_t(ctx, a, b, GGML_TYPE_F32); -}""", -)] - -GGML_H = [( - """ GGML_API struct ggml_tensor * ggml_mul_mat( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b); - - // change the precision of a matrix multiplication""", - """ GGML_API struct ggml_tensor * ggml_mul_mat( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b); - - // vla.cpp: BF16 activation support - as ggml_mul_mat, but with an explicit - // result type. Only GGML_TYPE_BF16 (CUDA, BF16 a and b) is implemented - // beyond F32; it keeps a BF16 GEMM's output in BF16 so an all-BF16 - // activation graph does not round-trip through F32 at every matmul. - GGML_API struct ggml_tensor * ggml_mul_mat_t( - struct ggml_context * ctx, - struct ggml_tensor * a, - struct ggml_tensor * b, - enum ggml_type type); - - // change the precision of a matrix multiplication""", -)] - -# --------------------------------------------------------------------------- -# ggml-cuda.cu - direct BF16 x BF16 -> BF16 cuBLAS GEMM -# --------------------------------------------------------------------------- -GGML_CUDA = [( - """static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - GGML_TENSOR_BINARY_OP_LOCALS - - const int32_t hint = ggml_get_op_params_i32(dst, 1); -""", - """// vla.cpp: BF16 activation support - BF16 x BF16 -> BF16 cuBLAS GEMM. -// -// The stock BF16 path (ggml_cuda_op_mul_mat_cublas) always converts src1 to -// BF16 on the way in and the BF16 GEMM result back to F32 on the way out, -// because ggml_mul_mat's result is F32 by definition. On an F32-activation -// graph that is one convert_unary launch per operand per matmul - ~11% of GPU -// time on evo1. When the caller asked for a BF16 result (ggml_mul_mat_t) and -// both operands are already BF16 there is nothing to convert: hand the tensors -// straight to cuBLAS. -// -// Accumulation stays CUBLAS_COMPUTE_32F, matching the stock BF16 path, so only -// operand and result storage changes, not the reduction. -static bool ggml_cuda_can_mul_mat_bf16(const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { - if (src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16 || dst->type != GGML_TYPE_BF16) { - return false; - } - if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { - return false; - } - // src0 is either shared across the whole batch or batched 1:1 with src1 - const bool batch_ok = (src0->ne[2] == 1 && src0->ne[3] == 1) || - (src0->ne[2] == src1->ne[2] && src0->ne[3] == src1->ne[3]); - return batch_ok && bf16_mma_hardware_available(ggml_cuda_info().devices[ggml_cuda_get_device()].cc); -} - -static void ggml_cuda_mul_mat_bf16(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - GGML_TENSOR_BINARY_OP_LOCALS; - - const nv_bfloat16 * a = (const nv_bfloat16 *) src0->data; - const nv_bfloat16 * b = (const nv_bfloat16 *) src1->data; - nv_bfloat16 * c = (nv_bfloat16 *) dst->data; - - const float alpha = 1.0f; - const float beta = 0.0f; - - CUBLAS_CHECK(cublasSetStream(ctx.cublas_handle(), ctx.stream())); - - const int64_t n_batch = ne12*ne13; - if (n_batch == 1) { - CUBLAS_CHECK( - cublasGemmEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, - ne01, ne11, ne10, - &alpha, a, CUDA_R_16BF, ne00, - b, CUDA_R_16BF, ne10, - &beta, c, CUDA_R_16BF, ne0, - CUBLAS_COMPUTE_32F, - CUBLAS_GEMM_DEFAULT_TENSOR_OP)); - return; - } - - // stride_a == 0 broadcasts one weight matrix across the batch - const long long stride_a = (src0->ne[2] == 1 && src0->ne[3] == 1) ? 0 : (long long) ne00*ne01; - CUBLAS_CHECK( - cublasGemmStridedBatchedEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, - ne01, ne11, ne10, - &alpha, a, CUDA_R_16BF, ne00, stride_a, - b, CUDA_R_16BF, ne10, (long long) ne10*ne11, - &beta, c, CUDA_R_16BF, ne0, (long long) ne0*ne1, - n_batch, - CUBLAS_COMPUTE_32F, - CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -} - -static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - GGML_TENSOR_BINARY_OP_LOCALS - - // vla.cpp: nothing but ggml_mul_mat_t(..., GGML_TYPE_BF16) produces a BF16 - // matmul. This has to intercept before the dst->type != F32 early-out below, - // which would otherwise hand the BF16 result to ggml_cuda_mul_mat_cublas and - // convert it straight back to F32. There is no split-buffer case to exclude: - // CUDA split buffers were removed from ggml upstream. - if (dst->type == GGML_TYPE_BF16) { - if (!ggml_cuda_can_mul_mat_bf16(src0, src1, dst)) { - // Name the offending operand rather than just asserting: the usual - // cause is a graph asking for a BF16 result from a weight that is - // resident F32 (or quantized), which mm_act() is meant to route - // around. See src/models/act_dtype.h. - for (const ggml_tensor * t : {src0, src1, (const ggml_tensor *) dst}) { - fprintf(stderr, "BF16 mul_mat operand %-24s type=%-5s contiguous=%d ne=[%ld %ld %ld %ld]\\n", - t->name, ggml_type_name(t->type), (int) ggml_is_contiguous(t), - (long) t->ne[0], (long) t->ne[1], (long) t->ne[2], (long) t->ne[3]); - } - GGML_ABORT("BF16 mul_mat result requires contiguous BF16 operands on BF16-MMA hardware"); - } - ggml_cuda_mul_mat_bf16(ctx, src0, src1, dst); - return; - } - - const int32_t hint = ggml_get_op_params_i32(dst, 1); -""", -), ( - """ GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); - GGML_ASSERT(rms_norm->type == GGML_TYPE_F32); - - //rms norm only supports F32 - if (mul->src[0]->type != GGML_TYPE_F32 || - mul->src[1]->type != GGML_TYPE_F32 || - mul->type != GGML_TYPE_F32) { - return false; - } - - if (add && (add->src[0]->type != GGML_TYPE_F32 || - add->src[1]->type != GGML_TYPE_F32 || - add->type != GGML_TYPE_F32) ) { - return false; - }""", - """ // vla.cpp: BF16 activation support. ggml_cuda_op_rms_norm_fused now takes - // a BF16 activation with an F32 norm weight; the three-op variant that - // also folds an add stays F32-only. - const enum ggml_type rms_at = rms_norm->src[0]->type; - if (rms_at != GGML_TYPE_F32 && rms_at != GGML_TYPE_BF16) { - return false; - } - GGML_ASSERT(rms_norm->type == rms_at); - - const ggml_tensor * mul_w = (mul->src[0] == rms_norm) ? mul->src[1] : mul->src[0]; - if (mul_w->type != GGML_TYPE_F32 || mul->type != rms_at) { - return false; - } - - if (add && (rms_at != GGML_TYPE_F32 || - add->src[0]->type != GGML_TYPE_F32 || - add->src[1]->type != GGML_TYPE_F32 || - add->type != GGML_TYPE_F32) ) { - return false; - }""", -)] - -# --------------------------------------------------------------------------- -# binbcast.cu - BF16 add / mul, plain and fused -# --------------------------------------------------------------------------- -BINBCAST = [ - ( - """ GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); - - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - op()(src0, src1, dst, (const float *)src0_dd, (const float *)src1_dd, (float *)dst_dd, stream);""", - """ GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16 || src1->type == GGML_TYPE_BF16); - - // vla.cpp: BF16 activation support. src1 is F32 for the model's bias and - // norm-weight tensors (kept F32 in the GGUF) and BF16 for residual adds. - if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { - op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const nv_bfloat16 *)src1_dd, (nv_bfloat16 *)dst_dd, stream); - } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_BF16) { - op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const float *)src1_dd, (nv_bfloat16 *)dst_dd, stream); - } else if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { - op()(src0, src1, dst, (const float *)src0_dd, (const nv_bfloat16 *)src1_dd, (nv_bfloat16 *)dst_dd, stream); - } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_F32) { - op()(src0, src1, dst, (const nv_bfloat16 *)src0_dd, (const nv_bfloat16 *)src1_dd, (float *)dst_dd, stream); - } else if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - op()(src0, src1, dst, (const float *)src0_dd, (const float *)src1_dd, (float *)dst_dd, stream);""", - ), - ( - """ if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - launch_bin_bcast_pack(src0, src1, dst, - (const float *) src0->data, (const float *) src1->data, (float *) dst->data, - stream, std::make_index_sequence{});""", - """ // vla.cpp: BF16 activation support. Fusion requires identical src1 layouts - // (ggml_are_same_layout compares type), so a chain never mixes F32 bias and - // BF16 residual operands. - if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && dst->type == GGML_TYPE_BF16) { - launch_bin_bcast_pack(src0, src1, dst, - (const nv_bfloat16 *) src0->data, (const nv_bfloat16 *) src1->data, (nv_bfloat16 *) dst->data, - stream, std::make_index_sequence{}); - } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_BF16) { - launch_bin_bcast_pack(src0, src1, dst, - (const nv_bfloat16 *) src0->data, (const float *) src1->data, (nv_bfloat16 *) dst->data, - stream, std::make_index_sequence{}); - } else if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - launch_bin_bcast_pack(src0, src1, dst, - (const float *) src0->data, (const float *) src1->data, (float *) dst->data, - stream, std::make_index_sequence{});""", - ), -] - -# --------------------------------------------------------------------------- -# unary.cu - BF16 elementwise activations -# --------------------------------------------------------------------------- -UNARY = [( - """ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); - GGML_ASSERT( dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16); - GGML_ASSERT(src0->type == dst->type); - - if (src0->type == GGML_TYPE_F16) { - unary_cuda((const half *)src0_d, (half *)dst_d, ggml_nelements(src0), stream); - } else { - unary_cuda((const float *)src0_d, (float *)dst_d, ggml_nelements(src0), stream); - } -}""", - """ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); - GGML_ASSERT( dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16 || dst->type == GGML_TYPE_BF16); - GGML_ASSERT(src0->type == dst->type); - - if (src0->type == GGML_TYPE_F16) { - unary_cuda((const half *)src0_d, (half *)dst_d, ggml_nelements(src0), stream); - } else if (src0->type == GGML_TYPE_BF16) { - // vla.cpp: BF16 activation support. unary_op_kernel already evaluates in - // float and casts back, so only the storage type changes. - unary_cuda((const nv_bfloat16 *)src0_d, (nv_bfloat16 *)dst_d, ggml_nelements(src0), stream); - } else { - unary_cuda((const float *)src0_d, (float *)dst_d, ggml_nelements(src0), stream); - } -}""", -)] - -# --------------------------------------------------------------------------- -# norm.cu - BF16 norm / rms_norm, float reductions -# --------------------------------------------------------------------------- -NORM = [ - ( - """template -static __global__ void norm_f32( - const float * x, float * dst, const int ncols, const int64_t stride_row, const int64_t stride_channel, - const int64_t stride_sample, const float eps) {""", - """// vla.cpp: BF16 activation support. T is the activation storage type (float or -// nv_bfloat16). The mean/variance reduction and the normalisation stay in float -// regardless, so a BF16 graph gets the numerics PyTorch's bf16 LayerNorm gives -// (bf16 in/out, fp32 accumulate). -template -static __global__ void norm_f32( - const T * x, T * dst, const int ncols, const int64_t stride_row, const int64_t stride_channel, - const int64_t stride_sample, const float eps) {""", - ), - ( - """ for (int col = tid; col < ncols; col += block_size) { - const float xi = x[col]; - mean_var.x += xi; - mean_var.y += xi * xi; - }""", - """ for (int col = tid; col < ncols; col += block_size) { - const float xi = (float) x[col]; - mean_var.x += xi; - mean_var.y += xi * xi; - }""", - ), - ( - """ for (int col = tid; col < ncols; col += block_size) { - dst[col] = (x[col] - mean) * inv_std; - } -}""", - """ for (int col = tid; col < ncols; col += block_size) { - dst[col] = (T) (((float) x[col] - mean) * inv_std); - } -}""", - ), - ( - """template -static __global__ void rms_norm_f32(const float * x, - float * dst, - const int ncols,""", - """// vla.cpp: BF16 activation support. T is the activation storage type; `mul` and -// `add` stay F32 because the norm weight and bias are F32 in the GGUF. -template -static __global__ void rms_norm_f32(const T * x, - T * dst, - const int ncols,""", - ), - ( - # disambiguated from the identical loop in l2_norm_f32 by the tail - """ for (int col = tid; col < ncols; col += block_size) { - const float xi = x[col]; - tmp += xi * xi; - } - - // sum up partial sums - extern __shared__ float s_sum[]; - tmp = block_reduce(tmp, s_sum); - - const float mean = tmp / ncols;""", - """ for (int col = tid; col < ncols; col += block_size) { - const float xi = (float) x[col]; - tmp += xi * xi; - } - - // sum up partial sums - extern __shared__ float s_sum[]; - tmp = block_reduce(tmp, s_sum); - - const float mean = tmp / ncols;""", - ), - ( - """ if constexpr (do_multiply && do_add) { - const int mul_col = fastmodulo(col, mul_ncols_packed); - const int add_col = fastmodulo(col, add_ncols_packed); - dst[col] = scale * x[col] * mul[mul_col] + add[add_col]; - } else if constexpr (do_multiply) { - const int mul_col = fastmodulo(col, mul_ncols_packed); - dst[col] = scale * x[col] * mul[mul_col]; - } else { - dst[col] = scale * x[col]; - }""", - """ if constexpr (do_multiply && do_add) { - const int mul_col = fastmodulo(col, mul_ncols_packed); - const int add_col = fastmodulo(col, add_ncols_packed); - dst[col] = (T) (scale * (float) x[col] * mul[mul_col] + add[add_col]); - } else if constexpr (do_multiply) { - const int mul_col = fastmodulo(col, mul_ncols_packed); - dst[col] = (T) (scale * (float) x[col] * mul[mul_col]); - } else { - dst[col] = (T) (scale * (float) x[col]); - }""", - ), - ( - """static void norm_f32_cuda( - const float * x, float * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", - """template -static void norm_f32_cuda( - const T * x, T * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", - ), - ( - """ norm_f32<<>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", - """ norm_f32<<>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", - ), - ( - """ norm_f32<1024><< WARP_SIZE ? 32 * sizeof(float2): 0, stream>>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", - """ norm_f32<1024, T><< WARP_SIZE ? 32 * sizeof(float2): 0, stream>>>(x, dst, ncols, stride_row, stride_channel, stride_sample, eps);""", - ), - ( - """static void rms_norm_f32_cuda( - const float * x, float * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", - """template -static void rms_norm_f32_cuda( - const T * x, T * dst, const int ncols, const int nrows, const int nchannels, const int nsamples,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<256, false>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<256, false, false, T>,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<1024, false>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<1024, false, false, T>,""", - ), - ( - """static void rms_norm_mul_f32_cuda(const float * x, - const float * mul, - const float * add, - float * dst,""", - """template -static void rms_norm_mul_f32_cuda(const T * x, - const float * mul, - const float * add, - T * dst,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<256, true>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<256, true, false, T>,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<1024, true>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, false, T>,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<256, true, true>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<256, true, true, T>,""", - ), - ( - """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, true>,""", - """ggml_cuda_kernel_launch(rms_norm_f32<1024, true, true, T>,""", - ), - ( - """void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const float * src0_d = (const float *) src0->data; - float * dst_d = (float *) dst->data; - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT( dst->type == GGML_TYPE_F32);""", - """void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - cudaStream_t stream = ctx.stream(); - - // vla.cpp: BF16 activation support - GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); - GGML_ASSERT( dst->type == src0->type);""", - ), - ( - """ norm_f32_cuda(src0_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); -}""", - """ if (src0->type == GGML_TYPE_BF16) { - norm_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, - ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); - } else { - norm_f32_cuda((const float *) src0->data, (float *) dst->data, - ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); - } -}""", - ), - ( - """void ggml_cuda_op_rms_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const float * src0_d = (const float *) src0->data; - float * dst_d = (float *) dst->data; - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT( dst->type == GGML_TYPE_F32);""", - """void ggml_cuda_op_rms_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - cudaStream_t stream = ctx.stream(); - - // vla.cpp: BF16 activation support - GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); - GGML_ASSERT( dst->type == src0->type);""", - ), - ( - """ rms_norm_f32_cuda(src0_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); -}""", - """ if (src0->type == GGML_TYPE_BF16) { - rms_norm_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, - ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); - } else { - rms_norm_f32_cuda((const float *) src0->data, (float *) dst->data, - ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); - } -}""", - ), - ( - """ const float * src0_d = (const float *) rms_norm_src->data; - const float * mul_d = nullptr; - const ggml_tensor * mul_src = nullptr; - - if (mul_tensor->src[0] == dst) { - mul_d = (float *) mul_tensor->src[1]->data; - mul_src = mul_tensor->src[1]; - } else if(mul_tensor->src[1] == dst) { - mul_d = (float *) mul_tensor->src[0]->data; - mul_src = mul_tensor->src[0]; - } else { - GGML_ASSERT(false); - } - - float * dst_d = (float *) mul_tensor->data; - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(rms_norm_src->type == GGML_TYPE_F32); - GGML_ASSERT(dst->type == GGML_TYPE_F32); - GGML_ASSERT(mul_tensor->type == GGML_TYPE_F32); - GGML_ASSERT(eps >= 0.0f);""", - """ const float * mul_d = nullptr; - const ggml_tensor * mul_src = nullptr; - - if (mul_tensor->src[0] == dst) { - mul_d = (float *) mul_tensor->src[1]->data; - mul_src = mul_tensor->src[1]; - } else if(mul_tensor->src[1] == dst) { - mul_d = (float *) mul_tensor->src[0]->data; - mul_src = mul_tensor->src[0]; - } else { - GGML_ASSERT(false); - } - - cudaStream_t stream = ctx.stream(); - - // vla.cpp: BF16 activation support - activation may be BF16, norm weight stays F32 - GGML_ASSERT(rms_norm_src->type == GGML_TYPE_F32 || rms_norm_src->type == GGML_TYPE_BF16); - GGML_ASSERT(dst->type == rms_norm_src->type); - GGML_ASSERT(mul_tensor->type == rms_norm_src->type); - GGML_ASSERT(mul_src->type == GGML_TYPE_F32); - GGML_ASSERT(eps >= 0.0f);""", - ), - ( - """ rms_norm_mul_f32_cuda(src0_d, mul_d, nullptr, dst_d, - ne00, ne01, ne02, ne03, - /*s00*/ s01, s02, s03, - /*mul_s00*/ mul_s01, mul_s02, mul_s03, - mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, - /*add_s00*/ 0, 0, 0, - 0, 0, 0, 0, - eps, stream); -}""", - """ if (rms_norm_src->type == GGML_TYPE_BF16) { - rms_norm_mul_f32_cuda((const nv_bfloat16 *) rms_norm_src->data, mul_d, (const float *) nullptr, - (nv_bfloat16 *) mul_tensor->data, - ne00, ne01, ne02, ne03, - /*s00*/ s01, s02, s03, - /*mul_s00*/ mul_s01, mul_s02, mul_s03, - mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, - /*add_s00*/ 0, 0, 0, - 0, 0, 0, 0, - eps, stream); - } else { - rms_norm_mul_f32_cuda((const float *) rms_norm_src->data, mul_d, (const float *) nullptr, - (float *) mul_tensor->data, - ne00, ne01, ne02, ne03, - /*s00*/ s01, s02, s03, - /*mul_s00*/ mul_s01, mul_s02, mul_s03, - mul_ncols, mul_nrows, mul_nchannels, mul_nsamples, - /*add_s00*/ 0, 0, 0, - 0, 0, 0, 0, - eps, stream); - } -}""", - ), -] - -# --------------------------------------------------------------------------- -# scale.cu - BF16 scale -# --------------------------------------------------------------------------- -SCALE = [( - """static __global__ void scale_f32(const float * x, float * dst, const float scale, const float bias, const int64_t nelements) { - ggml_cuda_pdl_lc(); - int64_t tid = (int64_t)blockIdx.x * (int64_t)blockDim.x + (int64_t)threadIdx.x; - int64_t stride = (int64_t)blockDim.x * (int64_t)gridDim.x; - - ggml_cuda_pdl_sync(); - for (int64_t i = tid; i < nelements; i += stride) { - dst[i] = scale * x[i] + bias; - } -} - -static void scale_f32_cuda(const float * x, float * dst, const float scale, const float bias, const int64_t nelements, cudaStream_t stream) { - const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(MIN(MAX_GRIDDIM_X, num_blocks), CUDA_SCALE_BLOCK_SIZE, 0, stream); - ggml_cuda_kernel_launch(scale_f32, launch_params, x, dst, scale, bias, nelements); -} - -void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const float * src0_d = (const float *)src0->data; - float * dst_d = (float *)dst->data; - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT( dst->type == GGML_TYPE_F32); - - float scale; - float bias; - memcpy(&scale, (float *) dst->op_params + 0, sizeof(float)); - memcpy(&bias, (float *) dst->op_params + 1, sizeof(float)); - - scale_f32_cuda(src0_d, dst_d, scale, bias, ggml_nelements(src0), stream); -}""", - """// vla.cpp: BF16 activation support. T is the activation storage type; the -// scale and bias apply in float. -template -static __global__ void scale_f32(const T * x, T * dst, const float scale, const float bias, const int64_t nelements) { - ggml_cuda_pdl_lc(); - int64_t tid = (int64_t)blockIdx.x * (int64_t)blockDim.x + (int64_t)threadIdx.x; - int64_t stride = (int64_t)blockDim.x * (int64_t)gridDim.x; - - ggml_cuda_pdl_sync(); - for (int64_t i = tid; i < nelements; i += stride) { - dst[i] = (T) (scale * (float) x[i] + bias); - } -} - -template -static void scale_f32_cuda(const T * x, T * dst, const float scale, const float bias, const int64_t nelements, cudaStream_t stream) { - const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(MIN(MAX_GRIDDIM_X, num_blocks), CUDA_SCALE_BLOCK_SIZE, 0, stream); - ggml_cuda_kernel_launch(scale_f32, launch_params, x, dst, scale, bias, nelements); -} - -void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_BF16); - GGML_ASSERT( dst->type == src0->type); - - float scale; - float bias; - memcpy(&scale, (float *) dst->op_params + 0, sizeof(float)); - memcpy(&bias, (float *) dst->op_params + 1, sizeof(float)); - - if (src0->type == GGML_TYPE_BF16) { - scale_f32_cuda((const nv_bfloat16 *) src0->data, (nv_bfloat16 *) dst->data, scale, bias, ggml_nelements(src0), stream); - } else { - scale_f32_cuda((const float *) src0->data, (float *) dst->data, scale, bias, ggml_nelements(src0), stream); - } -}""", -)] - - -def main(): - repo = pathlib.Path(__file__).resolve().parent.parent - src = pathlib.Path(sys.argv[1]) if len(sys.argv) > 1 else repo / "build/_deps/llama-src" - if not (src / "ggml/src/ggml.c").exists(): - raise SystemExit(f"not a llama.cpp source tree: {src}") - - if MARKER in (src / "ggml/src/ggml.c").read_text(): - print(f"already patched: {src}") - return - - cuda = src / "ggml/src/ggml-cuda" - pending = [ - (src / "ggml/src/ggml.c", edit(src / "ggml/src/ggml.c", GGML_C)), - (src / "ggml/include/ggml.h", edit(src / "ggml/include/ggml.h", GGML_H)), - (cuda / "ggml-cuda.cu", edit(cuda / "ggml-cuda.cu", GGML_CUDA)), - (cuda / "binbcast.cu", edit(cuda / "binbcast.cu", BINBCAST)), - (cuda / "unary.cu", edit(cuda / "unary.cu", UNARY)), - (cuda / "norm.cu", edit(cuda / "norm.cu", NORM)), - (cuda / "scale.cu", edit(cuda / "scale.cu", SCALE)), - ] - for path, text in pending: - path.write_text(text) - print(f"patched {len(pending)} files: {src}") - - -if __name__ == "__main__": - main() diff --git a/scripts/patch_ggml_cuda_ext_hook.py b/scripts/patch_ggml_cuda_ext_hook.py new file mode 100644 index 0000000..dff4bd5 --- /dev/null +++ b/scripts/patch_ggml_cuda_ext_hook.py @@ -0,0 +1,108 @@ +#!/usr/bin/env python3 +# Copyright 2026 VinRobotics +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Add one extension hook to the fetched ggml CUDA backend. + +This is the entire ggml modification. The kernels it enables live in +src/cuda/vla_cuda_bf16.cu as ordinary in-tree code, depending only on the public +ggml header, so a GIT_TAG bump cannot break them. + +Why a hook is needed at all +--------------------------- +ggml's CUDA backend has no BF16 instantiation for add/mul/unary and asserts F32 +in norm/rms_norm/scale, and ggml_cuda_mul_mat_cublas writes F32 into dst->data +unconditionally (for a BF16 dst that is both wrong and twice the bytes the +allocator reserved). There is no way to register kernels for built-in ops from +outside: GGML_OP_CUSTOM is CPU-only. So the backend has to offer one place where +an external implementation gets first refusal. + +What it changes (ggml/src/ggml-cuda/ggml-cuda.cu only) +------------------------------------------------------ + 1. An exported function pointer, null by default. + 2. One call to it at the top of ggml_cuda_compute_forward. Returning false + means "not mine", and ggml runs the op exactly as before. + 3. The RMS_NORM+MUL fusion check GGML_ASSERTs F32 rather than declining, so a + BF16 rms_norm aborts the process before dispatch is ever reached. Those two + asserts become a return, which is what the surrounding checks already do + for every other unsupported type. + +With the pointer left null this is a no-op, so an unpatched-but-hooked ggml +behaves identically to a stock one. + +Usage: scripts/patch_ggml_cuda_ext_hook.py [] +""" + +import pathlib +import sys + +MARKER = "vla.cpp: CUDA extension hook" + +HOOK_DECL = ( + """static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct ggml_tensor * dst) { + switch (dst->op) {""", + """// vla.cpp: CUDA extension hook. Null unless vla::cuda_register_bf16_ops() ran; +// see src/cuda/vla_cuda_bf16.cu, which holds every kernel behind it. +extern "C" { +typedef bool (*ggml_cuda_ext_forward_t)(struct ggml_tensor * dst, void * stream); +__attribute__((visibility("default"))) ggml_cuda_ext_forward_t ggml_cuda_ext_forward = nullptr; +} + +static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct ggml_tensor * dst) { + if (ggml_cuda_ext_forward && ggml_cuda_ext_forward(dst, (void *) ctx.stream())) { + return true; + } + + switch (dst->op) {""", +) + +# A BF16 rms_norm reaching this assert kills the process, and GGML_ASSERT is not +# compiled out in Release. Declining the fusion is what the checks immediately +# below already do for an unsupported mul/add type. +FUSION_GUARD = ( + """ GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); + GGML_ASSERT(rms_norm->type == GGML_TYPE_F32);""", + """ // vla.cpp: CUDA extension hook - decline instead of aborting, so a BF16 + // rms_norm falls through to the unfused path (and then to the hook). + if (rms_norm->src[0]->type != GGML_TYPE_F32 || rms_norm->type != GGML_TYPE_F32) { + return false; + }""", +) + + +def main(): + src = pathlib.Path(sys.argv[1] if len(sys.argv) > 1 else ".").resolve() + path = src / "ggml/src/ggml-cuda/ggml-cuda.cu" + if not path.exists(): + raise SystemExit(f"not a llama.cpp source tree: {src}") + + text = path.read_text() + if MARKER in text: + return # idempotent: re-configure over an already-patched tree + + for old, new in (HOOK_DECL, FUSION_GUARD): + n = text.count(old) + if n != 1: + raise SystemExit( + f"{path}: anchor found {n} times, expected 1. The pinned llama.cpp " + f"probably moved; re-check this anchor against the new tag.\n" + f"---\n{old[:400]}\n---" + ) + text = text.replace(old, new) + + path.write_text(text) + + +if __name__ == "__main__": + main() diff --git a/src/cuda/vla_cuda_bf16.cu b/src/cuda/vla_cuda_bf16.cu new file mode 100644 index 0000000..6ff6b2f --- /dev/null +++ b/src/cuda/vla_cuda_bf16.cu @@ -0,0 +1,373 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// BF16 activation kernels for the CUDA backend. +// +// ggml's CUDA backend implements these ops for F32 (and some for F16), but not +// for BF16: binbcast has no BF16 instantiation, and norm / rms_norm / scale +// assert F32 outright. An all-BF16 activation graph therefore has nowhere to +// run, and ggml offers no way to register kernels for built-in ops +// (GGML_OP_CUSTOM is CPU-only). +// +// So the fetched ggml carries one hook - a function pointer consulted at the +// top of ggml_cuda_compute_forward - and everything else lives here, in tree, +// as ordinary first-party code. See scripts/patch_ggml_cuda_ext_hook.py for the +// hook itself; it is the entire ggml modification. +// +// This file deliberately depends only on the PUBLIC ggml header. It never +// includes ggml-cuda internals (common.cuh and friends), so a llama.cpp bump +// cannot break it the way an anchored source patch would. +// +// Every entry point returns false for anything it does not handle, and ggml +// then runs the op exactly as it would have. Nothing here changes the F32 path. +// +// Accumulation is float throughout: only operand and result *storage* is BF16, +// never a reduction. + +#include "ggml.h" + +#include +#include +#include + +#include +#include + +// Must match the typedef the hook patch inserts into ggml-cuda.cu. +extern "C" { +typedef bool (*ggml_cuda_ext_forward_t)(struct ggml_tensor * dst, void * stream); +extern ggml_cuda_ext_forward_t ggml_cuda_ext_forward; +} + +namespace { + +constexpr int BLOCK = 256; + +inline __device__ float bf2f(const __nv_bfloat16 v) { return __bfloat162float(v); } +inline __device__ __nv_bfloat16 f2bf(const float v) { return __float2bfloat16(v); } + +// --------------------------------------------------------------------------- +// elementwise binary: dst = op(src0, src1), src1 broadcast per ggml_can_repeat +// --------------------------------------------------------------------------- +// +// ggml's rule is index-wise modulo on each dimension, so a [N,1,1,1] bias and a +// [N,M,1,1] activation combine the way the F32 path combines them. src1 may be +// F32 (a resident bias or norm weight) or BF16 (another activation). + +enum class BinOp { Add, Mul }; + +template +__global__ void k_bin_bcast_bf16( + const __nv_bfloat16 * __restrict__ src0, const S1 * __restrict__ src1, + __nv_bfloat16 * __restrict__ dst, + const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t ne3, + const int64_t s00, const int64_t s01, const int64_t s02, const int64_t s03, + const int64_t ne10, const int64_t ne11, const int64_t ne12, const int64_t ne13, + const int64_t s10, const int64_t s11, const int64_t s12, const int64_t s13, + const int64_t d0, const int64_t d1, const int64_t d2, const int64_t d3) { + const int64_t total = ne0*ne1*ne2*ne3; + for (int64_t idx = (int64_t) blockIdx.x*blockDim.x + threadIdx.x; idx < total; + idx += (int64_t) gridDim.x*blockDim.x) { + const int64_t i0 = idx % ne0; + const int64_t i1 = (idx / ne0) % ne1; + const int64_t i2 = (idx / (ne0*ne1)) % ne2; + const int64_t i3 = idx / (ne0*ne1*ne2); + + const float a = bf2f(src0[i0*s00 + i1*s01 + i2*s02 + i3*s03]); + const float b = (float) src1[(i0 % ne10)*s10 + (i1 % ne11)*s11 + + (i2 % ne12)*s12 + (i3 % ne13)*s13]; + + dst[i0*d0 + i1*d1 + i2*d2 + i3*d3] = f2bf(op == BinOp::Add ? a + b : a * b); + } +} + +// element strides (ggml stores byte strides) +inline int64_t es(const ggml_tensor * t, int i) { return t->nb[i] / ggml_type_size(t->type); } + +template +bool bin_bcast(ggml_tensor * dst, cudaStream_t stream) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + if (!src0 || !src1) return false; + if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16) return false; + if (src1->type != GGML_TYPE_BF16 && src1->type != GGML_TYPE_F32) return false; + if (!ggml_are_same_shape(src0, dst)) return false; + if (!ggml_can_repeat(src1, src0)) return false; + + const int64_t total = ggml_nelements(dst); + const int64_t blocks = (total + BLOCK - 1) / BLOCK; + const int grid = (int) (blocks < 65535 ? blocks : 65535); + + if (src1->type == GGML_TYPE_BF16) { + k_bin_bcast_bf16<<>>( + (const __nv_bfloat16 *) src0->data, (const __nv_bfloat16 *) src1->data, + (__nv_bfloat16 *) dst->data, + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], + es(src0,0), es(src0,1), es(src0,2), es(src0,3), + src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], + es(src1,0), es(src1,1), es(src1,2), es(src1,3), + es(dst,0), es(dst,1), es(dst,2), es(dst,3)); + } else { + k_bin_bcast_bf16<<>>( + (const __nv_bfloat16 *) src0->data, (const float *) src1->data, + (__nv_bfloat16 *) dst->data, + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], + es(src0,0), es(src0,1), es(src0,2), es(src0,3), + src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], + es(src1,0), es(src1,1), es(src1,2), es(src1,3), + es(dst,0), es(dst,1), es(dst,2), es(dst,3)); + } + return true; +} + +// --------------------------------------------------------------------------- +// unary +// --------------------------------------------------------------------------- + +enum class UnOp { Silu, Relu, Gelu, GeluErf }; + +template +inline __device__ float apply_unary(const float x) { + if (op == UnOp::Silu) return x / (1.0f + expf(-x)); + if (op == UnOp::Relu) return x > 0.0f ? x : 0.0f; + if (op == UnOp::GeluErf) return 0.5f*x*(1.0f + erff(x*0.70710678118654752440f)); + // tanh approximation, matching ggml's GGML_UNARY_OP_GELU + const float c = 0.79788456080286535588f; // sqrt(2/pi) + return 0.5f*x*(1.0f + tanhf(c*(x + 0.044715f*x*x*x))); +} + +template +__global__ void k_unary_bf16(const __nv_bfloat16 * __restrict__ x, + __nv_bfloat16 * __restrict__ dst, const int64_t n) { + for (int64_t i = (int64_t) blockIdx.x*blockDim.x + threadIdx.x; i < n; + i += (int64_t) gridDim.x*blockDim.x) { + dst[i] = f2bf(apply_unary(bf2f(x[i]))); + } +} + +template +bool unary(ggml_tensor * dst, cudaStream_t stream) { + const ggml_tensor * src0 = dst->src[0]; + if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16) return false; + // The elementwise index math above assumes a dense buffer. + if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(dst)) return false; + + const int64_t n = ggml_nelements(dst); + const int64_t blocks = (n + BLOCK - 1) / BLOCK; + const int grid = (int) (blocks < 65535 ? blocks : 65535); + k_unary_bf16<<>>( + (const __nv_bfloat16 *) src0->data, (__nv_bfloat16 *) dst->data, n); + return true; +} + +// --------------------------------------------------------------------------- +// scale: dst = x*scale + bias +// --------------------------------------------------------------------------- + +__global__ void k_scale_bf16(const __nv_bfloat16 * __restrict__ x, __nv_bfloat16 * __restrict__ dst, + const float scale, const float bias, const int64_t n) { + for (int64_t i = (int64_t) blockIdx.x*blockDim.x + threadIdx.x; i < n; + i += (int64_t) gridDim.x*blockDim.x) { + dst[i] = f2bf(scale*bf2f(x[i]) + bias); + } +} + +bool scale(ggml_tensor * dst, cudaStream_t stream) { + const ggml_tensor * src0 = dst->src[0]; + if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16) return false; + if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(dst)) return false; + + float s = 1.0f, b = 0.0f; + memcpy(&s, (const float *) dst->op_params + 0, sizeof(float)); + memcpy(&b, (const float *) dst->op_params + 1, sizeof(float)); + + const int64_t n = ggml_nelements(dst); + const int64_t blocks = (n + BLOCK - 1) / BLOCK; + const int grid = (int) (blocks < 65535 ? blocks : 65535); + k_scale_bf16<<>>( + (const __nv_bfloat16 *) src0->data, (__nv_bfloat16 *) dst->data, s, b, n); + return true; +} + +// --------------------------------------------------------------------------- +// norm / rms_norm - one block per row, float reduction +// --------------------------------------------------------------------------- + +__device__ inline float block_sum(float v, float * shared) { + const int tid = threadIdx.x; + shared[tid] = v; + __syncthreads(); + for (int s = blockDim.x / 2; s > 0; s >>= 1) { + if (tid < s) shared[tid] += shared[tid + s]; + __syncthreads(); + } + return shared[0]; +} + +template +__global__ void k_norm_bf16(const __nv_bfloat16 * __restrict__ x, __nv_bfloat16 * __restrict__ dst, + const int64_t ncols, const int64_t sx1, const int64_t sd1, + const float eps) { + __shared__ float shared[BLOCK]; + const int64_t row = blockIdx.x; + const __nv_bfloat16 * xr = x + row*sx1; + __nv_bfloat16 * dr = dst + row*sd1; + + float sum = 0.0f, sumsq = 0.0f; + for (int64_t c = threadIdx.x; c < ncols; c += blockDim.x) { + const float v = bf2f(xr[c]); + sumsq += v*v; + if (!rms) sum += v; + } + + if (rms) { + const float ms = block_sum(sumsq, shared) / (float) ncols; + const float inv = rsqrtf(ms + eps); + for (int64_t c = threadIdx.x; c < ncols; c += blockDim.x) dr[c] = f2bf(bf2f(xr[c])*inv); + } else { + const float mean = block_sum(sum, shared) / (float) ncols; + __syncthreads(); + const float meansq = block_sum(sumsq, shared) / (float) ncols; + const float inv = rsqrtf(meansq - mean*mean + eps); + for (int64_t c = threadIdx.x; c < ncols; c += blockDim.x) dr[c] = f2bf((bf2f(xr[c]) - mean)*inv); + } +} + +template +bool norm(ggml_tensor * dst, cudaStream_t stream) { + const ggml_tensor * src0 = dst->src[0]; + if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16) return false; + // Rows must be dense; higher dims are handled by flattening into the row index. + if (src0->nb[0] != ggml_type_size(src0->type)) return false; + if (dst->nb[0] != ggml_type_size(dst->type)) return false; + if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(dst)) return false; + + float eps = 0.0f; + memcpy(&eps, dst->op_params, sizeof(float)); + + const int64_t ncols = src0->ne[0]; + const int64_t nrows = ggml_nelements(src0) / ncols; + if (nrows > 2147483647) return false; + + k_norm_bf16<<<(int) nrows, BLOCK, 0, stream>>>( + (const __nv_bfloat16 *) src0->data, (__nv_bfloat16 *) dst->data, + ncols, ncols, ncols, eps); + return true; +} + +// --------------------------------------------------------------------------- +// mul_mat: BF16 x BF16 -> BF16 +// --------------------------------------------------------------------------- +// +// The stock path converts src1 F32->BF16 on the way into every cuBLAS GEMM +// because ggml_mul_mat's result is F32 by definition. When the caller asked for +// a BF16 result (vla::mul_mat_t) and both operands are already BF16 there is +// nothing to convert: hand the tensors straight to cuBLAS. Accumulation stays +// CUBLAS_COMPUTE_32F, matching the stock BF16 path. +// +// Note this cannot be left to ggml_cuda_mul_mat_cublas: that writes F32 into +// dst->data unconditionally, which for a BF16 dst is both wrong and twice the +// bytes the allocator reserved. + +cublasHandle_t g_handle = nullptr; + +bool mul_mat(ggml_tensor * dst, cudaStream_t stream) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + if (!src0 || !src1) return false; + if (dst->type != GGML_TYPE_BF16 || src0->type != GGML_TYPE_BF16 || src1->type != GGML_TYPE_BF16) { + return false; + } + if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { + return false; + } + // src0 is either shared across the whole batch or batched 1:1 with src1 + const bool batch_ok = (src0->ne[2] == 1 && src0->ne[3] == 1) || + (src0->ne[2] == src1->ne[2] && src0->ne[3] == src1->ne[3]); + if (!batch_ok) return false; + + if (!g_handle && cublasCreate(&g_handle) != CUBLAS_STATUS_SUCCESS) return false; + if (cublasSetStream(g_handle, stream) != CUBLAS_STATUS_SUCCESS) return false; + + const int64_t ne00 = src0->ne[0], ne01 = src0->ne[1]; + const int64_t ne10 = src1->ne[0], ne11 = src1->ne[1]; + const int64_t ne12 = src1->ne[2], ne13 = src1->ne[3]; + const int64_t ne0 = dst->ne[0], ne1 = dst->ne[1]; + + const __nv_bfloat16 * a = (const __nv_bfloat16 *) src0->data; + const __nv_bfloat16 * b = (const __nv_bfloat16 *) src1->data; + __nv_bfloat16 * c = (__nv_bfloat16 *) dst->data; + + const float alpha = 1.0f, beta = 0.0f; + const int64_t n_batch = ne12*ne13; + + cublasStatus_t st; + if (n_batch == 1) { + st = cublasGemmEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + (int) ne01, (int) ne11, (int) ne10, + &alpha, a, CUDA_R_16BF, (int) ne00, + b, CUDA_R_16BF, (int) ne10, + &beta, c, CUDA_R_16BF, (int) ne0, + CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP); + } else { + // stride_a == 0 broadcasts one weight matrix across the batch + const long long stride_a = (src0->ne[2] == 1 && src0->ne[3] == 1) ? 0 : (long long) ne00*ne01; + st = cublasGemmStridedBatchedEx(g_handle, CUBLAS_OP_T, CUBLAS_OP_N, + (int) ne01, (int) ne11, (int) ne10, + &alpha, a, CUDA_R_16BF, (int) ne00, stride_a, + b, CUDA_R_16BF, (int) ne10, (long long) ne10*ne11, + &beta, c, CUDA_R_16BF, (int) ne0, (long long) ne0*ne1, + (int) n_batch, + CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP); + } + return st == CUBLAS_STATUS_SUCCESS; +} + +} // namespace + +// --------------------------------------------------------------------------- +// hook entry point +// --------------------------------------------------------------------------- + +extern "C" bool vla_cuda_bf16_forward(ggml_tensor * dst, void * stream_v) { + if (!dst) return false; + cudaStream_t stream = (cudaStream_t) stream_v; + + switch (dst->op) { + case GGML_OP_MUL_MAT: return mul_mat(dst, stream); + case GGML_OP_ADD: return bin_bcast(dst, stream); + case GGML_OP_MUL: return bin_bcast(dst, stream); + case GGML_OP_SCALE: return scale(dst, stream); + case GGML_OP_NORM: return norm(dst, stream); + case GGML_OP_RMS_NORM: return norm(dst, stream); + case GGML_OP_UNARY: + switch (ggml_get_unary_op(dst)) { + case GGML_UNARY_OP_SILU: return unary(dst, stream); + case GGML_UNARY_OP_RELU: return unary(dst, stream); + case GGML_UNARY_OP_GELU: return unary(dst, stream); + case GGML_UNARY_OP_GELU_ERF: return unary(dst, stream); + default: return false; + } + default: return false; + } +} + +namespace vla { + +// Called once, after the CUDA backend is up. Idempotent. +void cuda_register_bf16_ops() { + ggml_cuda_ext_forward = vla_cuda_bf16_forward; +} + +} // namespace vla diff --git a/src/cuda/vla_cuda_ops.h b/src/cuda/vla_cuda_ops.h new file mode 100644 index 0000000..b6085b6 --- /dev/null +++ b/src/cuda/vla_cuda_ops.h @@ -0,0 +1,30 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +#pragma once + +// Registration for the in-tree BF16 CUDA kernels (src/cuda/vla_cuda_bf16.cu). +// +// Off every other build: without CUDA there is no hook to install and the BF16 +// activation path is unreachable anyway, so this compiles to nothing. + +namespace vla { + +#ifdef GGML_USE_CUDA +void cuda_register_bf16_ops(); +#else +inline void cuda_register_bf16_ops() {} +#endif + +} // namespace vla diff --git a/src/models/act_dtype.h b/src/models/act_dtype.h index 00ece16..7463676 100644 --- a/src/models/act_dtype.h +++ b/src/models/act_dtype.h @@ -34,8 +34,9 @@ // any flow-matching Euler integrator, so N accumulations of dt = 1/N // do not collapse into 8 mantissa bits. // -// Requires the ggml BF16 patch (scripts/patch_ggml_bf16_activations.py), which -// adds ggml_mul_mat_t plus BF16 support in the CUDA elementwise/norm kernels. +// The BF16 kernels themselves are in-tree (src/cuda/vla_cuda_bf16.cu); the only +// ggml change is the one-function-pointer hook that lets them run +// (scripts/patch_ggml_cuda_ext_hook.py). // // At act_type == GGML_TYPE_F32 both helpers below are the identity, so the // default path builds exactly the graph it built before. @@ -49,6 +50,31 @@ inline ggml_tensor * as_type(ggml_context * C, ggml_tensor * t, ggml_type ty) { return t->type == ty ? t : ggml_cast(C, t, ty); } +// mul_mat with an explicit result type. +// +// ggml_mul_mat hard-codes GGML_TYPE_F32 for its result and there is no API to +// ask for another one - but ggml_tensor's op/src fields are public, so the node +// can simply be built here rather than patched into ggml.c. Shapes, op and +// operand order are identical to ggml_mul_mat; only the result type differs. +// +// Only CUDA executes a non-F32 result, via the in-tree kernels in +// src/cuda/vla_cuda_bf16.cu. +inline ggml_tensor * mul_mat_t(ggml_context * C, ggml_tensor * a, ggml_tensor * b, ggml_type type) { + // ggml_can_mul_mat is internal to ggml; this is the same condition. + GGML_ASSERT(a->ne[0] == b->ne[0]); + GGML_ASSERT(b->ne[2] % a->ne[2] == 0 && b->ne[3] % a->ne[3] == 0); + GGML_ASSERT(!ggml_is_transposed(a)); + + const int64_t ne[4] = { a->ne[1], b->ne[1], b->ne[2], b->ne[3] }; + ggml_tensor * result = ggml_new_tensor(C, type, 4, ne); + + result->op = GGML_OP_MUL_MAT; + result->src[0] = a; + result->src[1] = b; + + return result; +} + // Weight-by-activation GEMM producing an activation-typed result. // // A BF16 result needs a BF16 weight. Not every weight is one: pi0 keeps its @@ -60,7 +86,7 @@ inline ggml_tensor * mm_act(ggml_context * C, ggml_tensor * w, ggml_tensor * x, if (w->type != at) { return as_type(C, ggml_mul_mat(C, w, as_type(C, x, GGML_TYPE_F32)), at); } - return ggml_mul_mat_t(C, w, x, at); + return mul_mat_t(C, w, x, at); } } // namespace vla diff --git a/src/models/evo1.cpp b/src/models/evo1.cpp index ab643cc..8807e31 100644 --- a/src/models/evo1.cpp +++ b/src/models/evo1.cpp @@ -23,6 +23,7 @@ #include "models/gguf_reader.h" #include "models/scratch_ctx.h" #include "models/act_dtype.h" +#include "cuda/vla_cuda_ops.h" #include #include @@ -377,6 +378,7 @@ std::unique_ptr evo1_create(const std::string& mmproj_path, if (std::getenv("VLA_EVO1_BF16_ACT")) { if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { m->act_type = GGML_TYPE_BF16; + cuda_register_bf16_ops(); // installs the in-tree BF16 CUDA kernels std::printf("vla(evo1): activations = BF16 (VLA_EVO1_BF16_ACT)\n"); } else { std::fprintf(stderr, "vla(evo1): VLA_EVO1_BF16_ACT ignored - needs CUDA and BF16 weights\n"); diff --git a/src/models/pi0.cpp b/src/models/pi0.cpp index 4369d29..51f2f28 100644 --- a/src/models/pi0.cpp +++ b/src/models/pi0.cpp @@ -26,6 +26,7 @@ #include "models/dit_common.h" #include "models/vision_common.h" #include "models/act_dtype.h" +#include "cuda/vla_cuda_ops.h" #include #include @@ -391,6 +392,7 @@ std::unique_ptr pi0_create(const std::string& mmproj_path, if (std::getenv("VLA_PI0_BF16_ACT")) { if (b.is_cuda && m->matmul_type == GGML_TYPE_BF16) { m->act_type = GGML_TYPE_BF16; + cuda_register_bf16_ops(); // installs the in-tree BF16 CUDA kernels std::printf("vla(pi0): activations = BF16 (VLA_PI0_BF16_ACT)\n"); } else { std::fprintf(stderr, "vla(pi0): VLA_PI0_BF16_ACT ignored - needs CUDA and BF16 weights\n"); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 6eba2a0..5e60cc4 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -43,3 +43,13 @@ add_test(NAME config_guard COMMAND test_config_guard) # The A/B harness for the two BitVLA ternary-GEMM tilings (bitvla_gemm_check.cu) # was never committed, so there is no target for it here. VLA_BITVLA_NARROW_GEMM=1 # still selects the old one-tile-per-CTA kernel for a hand-run comparison. + +# Regression test for the in-tree BF16 CUDA kernels and the ggml hook they ride +# on. Skips itself (exit 0) when no CUDA device is present. +if(GGML_CUDA) + add_executable(test_bf16_cuda_ops test_bf16_cuda_ops.cpp) + target_include_directories(test_bf16_cuda_ops PRIVATE ${CMAKE_SOURCE_DIR}/src) + target_link_libraries(test_bf16_cuda_ops PRIVATE vla_cuda_ops ggml) + target_compile_options(test_bf16_cuda_ops PRIVATE -Wall -Wextra) + add_test(NAME bf16_cuda_ops COMMAND test_bf16_cuda_ops) +endif() diff --git a/tests/test_bf16_cuda_ops.cpp b/tests/test_bf16_cuda_ops.cpp new file mode 100644 index 0000000..8b0ed50 --- /dev/null +++ b/tests/test_bf16_cuda_ops.cpp @@ -0,0 +1,155 @@ +// Copyright 2026 VinRobotics +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +// Every op the BF16 activation path needs, driven through real ggml dispatch +// and checked against the same graph in F32. +// +// This is the regression test for src/cuda/vla_cuda_bf16.cu and for the one +// hook it rides on (scripts/patch_ggml_cuda_ext_hook.py). It is what catches a +// llama.cpp bump that moves the hook anchor or changes an op's semantics: with +// the hook uninstalled, ggml aborts on the first BF16 op instead of silently +// returning something plausible. +// +// The chain deliberately includes ggml_mul(ggml_rms_norm(x), w), which is the +// pattern the CUDA backend tries to fuse and whose fusion check asserts F32. + +#include "ggml.h" +#include "ggml-alloc.h" +#include "ggml-backend.h" +#include "ggml-cuda.h" + +#include "cuda/vla_cuda_ops.h" +#include "models/act_dtype.h" + +#include +#include +#include + +namespace { + +constexpr int64_t K = 128; // reduction / feature dim +constexpr int64_t M = 64; // rows +constexpr int64_t N = 96; // output features +constexpr float EPS = 1e-5f; + +// Builds the chain at `at`. Inputs are always F32 tensors; the BF16 run casts in +// at the top and back out at the bottom, exactly as the models do. +ggml_tensor * build_chain(ggml_context * C, ggml_tensor * x, ggml_tensor * w_norm, + ggml_tensor * W, ggml_tensor * bias, ggml_type at) { + ggml_tensor * h = vla::as_type(C, x, at); + h = ggml_mul(C, ggml_rms_norm(C, h, EPS), w_norm); // RMS_NORM (+ MUL, F32 weight) + h = vla::mm_act(C, W, h, at); // MUL_MAT + h = ggml_add(C, h, bias); // ADD, F32 bias + h = ggml_silu(C, h); // UNARY SILU + h = ggml_scale(C, h, 0.5f); // SCALE + h = ggml_add(C, ggml_norm(C, h, EPS), h); // NORM (+ ADD, BF16 x BF16) + h = ggml_gelu_erf(C, h); // UNARY GELU_ERF + return vla::as_type(C, h, GGML_TYPE_F32); +} + +std::vector run(ggml_backend_t backend, ggml_type at, + const std::vector & hx, const std::vector & hw, + const std::vector & hW, const std::vector & hb) { + ggml_init_params p = { (size_t) 32*1024*1024, nullptr, true }; + ggml_context * C = ggml_init(p); + + ggml_tensor * x = ggml_new_tensor_2d(C, GGML_TYPE_F32, K, M); + ggml_tensor * w_norm = ggml_new_tensor_1d(C, GGML_TYPE_F32, K); + // The weight must be resident in the activation dtype for mm_act to take the + // typed-GEMM path; that mirrors a BF16 checkpoint. + ggml_tensor * W = ggml_new_tensor_2d(C, at, K, N); + ggml_tensor * bias = ggml_new_tensor_1d(C, GGML_TYPE_F32, N); + for (ggml_tensor * t : {x, w_norm, W, bias}) ggml_set_input(t); + + ggml_tensor * out = build_chain(C, x, w_norm, W, bias, at); + ggml_set_output(out); + + ggml_cgraph * gf = ggml_new_graph(C); + ggml_build_forward_expand(gf, out); + + ggml_gallocr_t ga = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); + if (!ga || !ggml_gallocr_alloc_graph(ga, gf)) { + std::fprintf(stderr, "alloc failed\n"); + return {}; + } + + ggml_backend_tensor_set(x, hx.data(), 0, ggml_nbytes(x)); + ggml_backend_tensor_set(w_norm, hw.data(), 0, ggml_nbytes(w_norm)); + ggml_backend_tensor_set(bias, hb.data(), 0, ggml_nbytes(bias)); + if (at == GGML_TYPE_BF16) { + std::vector t(hW.size()); + ggml_fp32_to_bf16_row(hW.data(), t.data(), (int64_t) hW.size()); + ggml_backend_tensor_set(W, t.data(), 0, ggml_nbytes(W)); + } else { + ggml_backend_tensor_set(W, hW.data(), 0, ggml_nbytes(W)); + } + + if (ggml_backend_graph_compute(backend, gf) != GGML_STATUS_SUCCESS) { + std::fprintf(stderr, "compute failed (%s)\n", ggml_type_name(at)); + return {}; + } + + std::vector host((size_t) ggml_nelements(out)); + ggml_backend_tensor_get(out, host.data(), 0, ggml_nbytes(out)); + + ggml_gallocr_free(ga); + ggml_free(C); + return host; +} + +} // namespace + +int main() { + ggml_backend_t backend = ggml_backend_cuda_init(0); + if (!backend) { + // No usable GPU: nothing to check, and this must not fail a CPU-only CI run. + std::printf("bf16_cuda_ops: no CUDA device, skipping\n"); + return 0; + } + vla::cuda_register_bf16_ops(); + + std::vector hx((size_t) K*M), hw((size_t) K), hW((size_t) K*N), hb((size_t) N); + for (size_t i = 0; i < hx.size(); ++i) hx[i] = ((float) ((i*37) % 23) - 11.0f) / 8.0f; + for (size_t i = 0; i < hw.size(); ++i) hw[i] = 0.5f + ((float) ((i*11) % 7)) / 16.0f; + for (size_t i = 0; i < hW.size(); ++i) hW[i] = ((float) ((i*53) % 19) - 9.0f) / 32.0f; + for (size_t i = 0; i < hb.size(); ++i) hb[i] = ((float) ((i*17) % 5) - 2.0f) / 16.0f; + + const std::vector ref = run(backend, GGML_TYPE_F32, hx, hw, hW, hb); + const std::vector got = run(backend, GGML_TYPE_BF16, hx, hw, hW, hb); + if (ref.empty() || got.empty() || ref.size() != got.size()) { + std::printf("FAIL: a run produced no output\n"); + return 1; + } + + // Normalised against the reference's own scale: gelu_erf drives many outputs + // near zero, where a per-element relative error is meaningless. + double max_abs = 0.0, max_diff = 0.0; + for (size_t i = 0; i < ref.size(); ++i) { + max_abs = std::fmax(max_abs, std::fabs((double) ref[i])); + max_diff = std::fmax(max_diff, std::fabs((double) ref[i] - (double) got[i])); + } + const double rel = max_abs > 0.0 ? max_diff / max_abs : max_diff; + + // BF16 carries 8 mantissa bits, and this chain is 8 ops deep. + const double tol = 0.05; + std::printf("bf16_cuda_ops: max |ref| %.4f, max diff %.4f, normalised %.4f (tol %.2f)\n", + max_abs, max_diff, rel, tol); + + if (max_abs == 0.0) { std::printf("FAIL: reference is all zeros\n"); return 1; } + if (rel > tol) { std::printf("FAIL\n"); return 1; } + + std::printf("PASS\n"); + ggml_backend_free(backend); + return 0; +} From d9a2a09d61f48b2e82bb0898e4467a87f82fec42 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 17:14:45 +0700 Subject: [PATCH 41/42] keep llama.cpp's tool binaries out of the default build and drop the unused llama link from vla_core --- CMakeLists.txt | 56 ++++++++++++++++++----------------------- examples/chat/README.md | 5 +++- 2 files changed, 28 insertions(+), 33 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index ac7d85b..2062292 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -11,8 +11,6 @@ if(NOT CMAKE_BUILD_TYPE) set(CMAKE_BUILD_TYPE Release CACHE STRING "" FORCE) endif() -# src/backend.h compiles in exactly one accelerator, so two GGML_* flags would -# give ggml both backends and vla_core only one. Reject before FetchContent. set(_vla_accel "") foreach(_flag GGML_CUDA GGML_SYCL GGML_METAL) if(${_flag}) @@ -32,15 +30,7 @@ set(LLAMA_BUILD_TOOLS ON CACHE BOOL "" FORCE) set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) -# llama.cpp fetched + pinned at configure; bump = one-line GIT_TAG change. -# -# The fetched ggml gets one addition: a function-pointer hook at the top of the -# CUDA op dispatch, so the in-tree BF16 kernels in src/cuda/ can take an op -# before ggml does. That hook is the whole modification — the kernels are -# ordinary first-party code and need no patching. Left unregistered the pointer -# is null and ggml behaves exactly as shipped. The script is idempotent, and -# fails loudly rather than half-applying if an anchor stops matching after a -# GIT_TAG bump. See scripts/patch_ggml_cuda_ext_hook.py. +# PATCH_COMMAND adds the CUDA extension hook; see scripts/patch_ggml_cuda_ext_hook.py. find_package(Python3 COMPONENTS Interpreter REQUIRED) include(FetchContent) FetchContent_Declare(llama @@ -53,6 +43,23 @@ FetchContent_Declare(llama ) FetchContent_MakeAvailable(llama) +# Drop the whole fetched tree from the default build: llama.cpp's ~20 tool +# binaries and their impl libraries are dead weight here. What we link (llama, +# ggml, mtmd, llama-common) is still built, pulled in as a dependency. Walking +# the tree rather than naming targets keeps a bump from adding them back. Any of +# them builds on request: cmake --build build --target llama-mtmd-cli +function(vla_exclude_fetched_targets dir) + get_property(subdirs DIRECTORY ${dir} PROPERTY SUBDIRECTORIES) + foreach(sub IN LISTS subdirs) + vla_exclude_fetched_targets(${sub}) + endforeach() + get_property(targets DIRECTORY ${dir} PROPERTY BUILDSYSTEM_TARGETS) + foreach(tgt IN LISTS targets) + set_target_properties(${tgt} PROPERTIES EXCLUDE_FROM_ALL TRUE) + endforeach() +endfunction() +vla_exclude_fetched_targets(${llama_SOURCE_DIR}) + add_library(vla_core src/model.cpp src/models/smolvla.cpp @@ -72,7 +79,8 @@ target_include_directories(vla_core ${CMAKE_CURRENT_SOURCE_DIR}/src ${llama_SOURCE_DIR}/vendor ) -target_link_libraries(vla_core PUBLIC llama ggml) +# The VLA archs call no llama_* API; only vlm_core needs llama. +target_link_libraries(vla_core PUBLIC ggml) if(GGML_CUDA) target_compile_definitions(vla_core PUBLIC GGML_USE_CUDA) @@ -80,9 +88,7 @@ if(GGML_CUDA) enable_language(CUDA) find_package(CUDAToolkit REQUIRED) if(NOT DEFINED CMAKE_CUDA_ARCHITECTURES) - # Emit SASS per supported GPU (real) and PTX (virtual) only for the newest - # arch, so future cards can JIT without shipping PTX for every target. - # Blackwell sm_100/sm_120 require CUDA >= 12.8. + # SASS per supported GPU, PTX only for the newest arch so future cards JIT. set(_vla_cuda_archs 80-real 86-real 87-real 89-real 90-real) set(_vla_cuda_ptx 90-virtual) if(CUDAToolkit_VERSION VERSION_GREATER_EQUAL 12.8) @@ -115,9 +121,7 @@ if(GGML_CUDA) target_compile_definitions(vla_core PUBLIC VLA_BITVLA_CUDA_KERNELS) target_include_directories(vla_core PUBLIC ${CUDAToolkit_INCLUDE_DIRS}) - # BF16 activation kernels. These run through the extension hook the fetched - # ggml carries (scripts/patch_ggml_cuda_ext_hook.py); the hook resolves - # ggml_cuda_ext_forward out of ggml-cuda, hence the explicit link. + # ggml-cuda link: the hook resolves ggml_cuda_ext_forward out of it. add_library(vla_cuda_ops STATIC src/cuda/vla_cuda_bf16.cu) target_include_directories(vla_cuda_ops PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/src @@ -135,10 +139,6 @@ if(GGML_CUDA) target_link_libraries(vla_core PRIVATE vla_cuda_ops) endif() -# Intel GPUs (Arc / Flex / Data Center Max / Xe iGPU) through oneAPI SYCL. -# ggml's SYCL sources only compile under the oneAPI DPC++ driver, and -# CMAKE_CXX_COMPILER is global, so our targets are built by icpx too. Fail -# loudly here rather than let ggml die deep in a kernel compile. if(GGML_SYCL AND NOT GGML_CUDA) if(NOT CMAKE_CXX_COMPILER_ID STREQUAL "IntelLLVM") message(FATAL_ERROR @@ -150,7 +150,7 @@ if(GGML_SYCL AND NOT GGML_CUDA) target_compile_definitions(vla_core PUBLIC GGML_USE_SYCL) endif() -# Backend precedence matches the ladder in src/backend.h: CUDA, then SYCL, then Metal. +# Precedence matches the ladder in src/backend.h. if(GGML_METAL AND NOT GGML_CUDA AND NOT GGML_SYCL) target_compile_definitions(vla_core PUBLIC GGML_USE_METAL) endif() @@ -172,8 +172,6 @@ find_package(Protobuf REQUIRED) find_package(PkgConfig REQUIRED) pkg_check_modules(ZeroMQ REQUIRED IMPORTED_TARGET libzmq) -# cppzmq (zmq.hpp) is the header-only C++ binding, packaged separately from the -# libzmq C library (Debian/Ubuntu: cppzmq-dev). The servers #include . find_path(CPPZMQ_INCLUDE_DIR NAMES zmq.hpp) if(NOT CPPZMQ_INCLUDE_DIR) message(FATAL_ERROR @@ -241,8 +239,6 @@ target_link_libraries(vlm-server PRIVATE PkgConfig::ZeroMQ ) -# Stable C ABI. Shared so bindings can dlopen it; visibility hidden so only the -# vla_* symbols are exported and llama/ggml stay internal. add_library(vla SHARED src/vla_c_api.cpp) target_include_directories(vla PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include @@ -255,7 +251,6 @@ set_target_properties(vla PROPERTIES PUBLIC_HEADER ${CMAKE_CURRENT_SOURCE_DIR}/include/vla.h ) -# One-shot inference CLI: image + tokens -> action, no server or simulator. add_executable(vla-cli src/serving/vla-cli.cpp ) @@ -266,21 +261,18 @@ target_link_libraries(vla-cli PRIVATE vla_core) # --text shells out to the tokenizer script; VLA_TOKENIZE_SCRIPT overrides it. target_compile_definitions(vla-cli PRIVATE VLA_SOURCE_DIR="${CMAKE_CURRENT_SOURCE_DIR}") -# Latency for one checkpoint, emits the README table rows. add_executable(vla-bench src/serving/vla-bench.cpp ) target_link_libraries(vla-bench PRIVATE vla_core) -# --- First-party build hygiene (never applied to the vendored llama.cpp subtree) -- +# Warnings and LTO for our own targets only, never the llama.cpp subtree. set(VLA_FIRST_PARTY_TARGETS vla_core vlm_core vla vla-server vlm-server vla-cli vla-bench) foreach(tgt IN LISTS VLA_FIRST_PARTY_TARGETS) - # Warn on our own C++ only; nvcc device code keeps its own diagnostics. target_compile_options(${tgt} PRIVATE $<$:-Wall -Wextra>) endforeach() -# Link-time optimization on Release builds when the toolchain supports it. include(CheckIPOSupported) check_ipo_supported(RESULT VLA_IPO_OK OUTPUT VLA_IPO_MSG) if(VLA_IPO_OK AND CMAKE_BUILD_TYPE STREQUAL "Release") diff --git a/examples/chat/README.md b/examples/chat/README.md index 40a60ca..29aaf4e 100644 --- a/examples/chat/README.md +++ b/examples/chat/README.md @@ -61,9 +61,12 @@ eval $CONV "$SRC" --mmproj \ The LM conversion resolves the hparams (hidden 960, ff 2560, 15 heads / 5 KV, rope θ 1e5) and bakes the SmolVLM2 chat template into the GGUF KV store; the `--mmproj` pass writes 198 vision tensors. Sanity-check the pair with -`llama-mtmd-cli`: +`llama-mtmd-cli`. It is not part of the default build, so ask for it by name +first: ```bash +cmake --build build-cuda --target llama-mtmd-cli + ./build-cuda/bin/llama-mtmd-cli \ -m "$OUT/smolvlm2-500m-instruct-f16.gguf" \ --mmproj "$OUT/mmproj-smolvlm2-500m-instruct-f16.gguf" \ From 5564ab96bb75bd2bd752762c669f840af96668d7 Mon Sep 17 00:00:00 2001 From: Khanh Nguyen Date: Wed, 12 Aug 2026 17:29:59 +0700 Subject: [PATCH 42/42] Update README for release v0.2.0 --- README.md | 43 ++++++++++++++----------------------------- 1 file changed, 14 insertions(+), 29 deletions(-) diff --git a/README.md b/README.md index 5d58cb0..17fb5e1 100644 --- a/README.md +++ b/README.md @@ -61,21 +61,6 @@ cmake -B build \ cmake --build build -j$(nproc) ``` -```bash -# Intel GPU build (Arc / Flex / Max / Xe iGPU). ggml's SYCL sources need the -# oneAPI DPC++ driver, so the whole project is compiled by icpx: -source /opt/intel/oneapi/setvars.sh -cmake -B build \ - -DGGML_SYCL=ON \ - -DCMAKE_C_COMPILER=icx \ - -DCMAKE_CXX_COMPILER=icpx \ - -DCMAKE_BUILD_TYPE=Release -cmake --build build -j$(nproc) -``` - -The driver and oneAPI setup that this needs is in -[docs/backend/sycl.md](docs/backend/sycl.md). - If CMake cannot find CUDA, point the environment at it explicitly: ```bash @@ -84,7 +69,7 @@ export LD_LIBRARY_PATH=/usr/local/cuda/lib64:$LD_LIBRARY_PATH ``` Check [docs/backend](docs/backend) for compiling `vla.cpp` on other platforms. -WSL2 and Apple Silicon are both tested. +WSL2, Apple Silicon, and Intel GPU are all tested. --- @@ -316,19 +301,19 @@ change is shown to leave it alone. Support matrix of models (rows) against platforms (columns). Legend: `Y` = supported (released and benchmarked), `~` = in progress, `-` = planned. -| Model | CPU (x86-64 / ARM) | CUDA | SYCL (Intel) | Metal | OpenVINO | Hexagon | -|---|:--:|:--:|:--:|:--:|:--:|:--:| -| [SmolVLA](https://hf.co/vrfai/smolvla-libero-gguf) | Y | Y | Y | Y | - | - | -| [π0](https://hf.co/vrfai/pi0-libero-finetuned-v044-gguf) | Y | Y | - | Y | - | - | -| [π0.5](https://hf.co/vrfai/pi05-libero-gguf) | Y | Y | - | ~ | - | - | -| [GR00T N1.5](https://hf.co/vrfai/gr00tn1d5-libero-object-gguf) | Y | Y | - | ~ | - | - | -| [GR00T N1.6](https://hf.co/vrfai/gr00tn1d6-libero-gguf) | Y | Y | - | ~ | - | - | -| [GR00T N1.7](https://hf.co/vrfai/gr00tn1d7-libero-gguf) | Y | Y | - | Y | - | - | -| [BitVLA](https://hf.co/vrfai/bitvla-libero-gguf) | Y | Y | - | ~ | - | - | -| [Evo-1](https://hf.co/vrfai/evo1-libero-gguf) | Y | Y | Y | ~ | - | - | -| [VLA-Adapter](https://hf.co/vrfai/vla-adapter-libero-gguf) | Y | Y | ~ | ~ | - | - | -| [OpenVLA-OFT](https://hf.co/vrfai/openvla-oft-libero-gguf) | Y | Y | - | ~ | - | - | -| [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | - | ~ | - | - | +| Model | CPU (x86-64 / ARM) | CUDA | SYCL (Intel) | Metal | OpenVINO | +|---|:--:|:--:|:--:|:--:|:--:| +| [SmolVLA](https://hf.co/vrfai/smolvla-libero-gguf) | Y | Y | Y | Y | - | +| [π0](https://hf.co/vrfai/pi0-libero-finetuned-v044-gguf) | Y | Y | - | Y | - | +| [π0.5](https://hf.co/vrfai/pi05-libero-gguf) | Y | Y | - | ~ | - | +| [GR00T N1.5](https://hf.co/vrfai/gr00tn1d5-libero-object-gguf) | Y | Y | - | ~ | - | +| [GR00T N1.6](https://hf.co/vrfai/gr00tn1d6-libero-gguf) | Y | Y | - | ~ | - | +| [GR00T N1.7](https://hf.co/vrfai/gr00tn1d7-libero-gguf) | Y | Y | - | Y | - | +| [BitVLA](https://hf.co/vrfai/bitvla-libero-gguf) | Y | Y | - | ~ | - | +| [Evo-1](https://hf.co/vrfai/evo1-libero-gguf) | Y | Y | Y | ~ | - | +| [VLA-Adapter](https://hf.co/vrfai/vla-adapter-libero-gguf) | Y | Y | ~ | ~ | - | +| [OpenVLA-OFT](https://hf.co/vrfai/openvla-oft-libero-gguf) | Y | Y | - | ~ | - | +| [VLA-JEPA](https://hf.co/vrfai/vla-jepa-libero) | Y | Y | - | ~ | - | ---