diff --git a/.github/workflows/windows-vulkan.yml b/.github/workflows/windows-vulkan.yml new file mode 100644 index 0000000..a5f346d --- /dev/null +++ b/.github/workflows/windows-vulkan.yml @@ -0,0 +1,86 @@ +# Smoke-builds the whisper.cpp engine crate with the Vulkan GPU backend on Windows. +# +# Unlike the sherpa-onnx DirectML path (which compiles onnxruntime from source for hours and has no +# upstream prebuilt), whisper.cpp + Vulkan builds in minutes and covers any Vulkan GPU — AMD, Intel, +# or NVIDIA — with ggml's built-in CPU fallback. This validates that the from-source Vulkan build +# links cleanly so we can wire it into the Windows app. Runs on demand. +name: Windows Vulkan whisper smoke + +on: + workflow_dispatch: {} + # Validate on pushes to the feature branch (a brand-new workflow can't be dispatched from a + # non-default branch). Drop this trigger once the path is proven and the workflow lands on main. + push: + branches: [feat/windows-vulkan-whisper] + +env: + CARGO_TERM_COLOR: always + # Keep the build tree shallow: ggml-vulkan's vulkan-shaders-gen sub-build nests deeply, and from the + # default target dir its .pdb path runs past Windows' 260-char MAX_PATH ("cannot open program + # database"). A short target root keeps every nested path well under the limit. + CARGO_TARGET_DIR: 'D:\b' + +jobs: + build: + name: whisper.cpp + Vulkan (windows) + runs-on: windows-latest + timeout-minutes: 60 + steps: + - uses: actions/checkout@v4 + with: + submodules: recursive # the vendored whisper.cpp is a submodule + - uses: dtolnay/rust-toolchain@stable + - uses: actions/setup-node@v4 + with: + node-version: 20 + # Put cl.exe + the MSVC include/lib env on PATH so the Ninja generator (and ggml-vulkan's + # vulkan-shaders-gen sub-build) can find the compiler. Ninja itself is preinstalled. + - name: Set up the MSVC build environment + uses: ilammy/msvc-dev-cmd@v1 + with: + arch: x64 + - name: Install the Vulkan SDK + uses: humbletim/install-vulkan-sdk@v1.2 + with: + version: 1.4.309.0 + cache: true + - name: Install the frontend deps + working-directory: app + run: npm ci + # `tauri build` runs the frontend build (beforeBuildCommand) + the cargo release build + the NSIS + # bundler in one step, producing the real Windows GPU installer: default features (sherpa on CPU) + # plus whisper-vulkan (whisper on the GPU, CPU fallback). This also proves the exact release + # recipe — that tauri's bundler honours CARGO_TARGET_DIR=D:\b — before wiring it into release.yml. + - name: Build the Windows GPU installer (whisper.cpp Vulkan) + working-directory: app + shell: pwsh + run: | + Write-Host "VULKAN_SDK = $env:VULKAN_SDK" + npm run tauri build -- --features whisper-vulkan + # Show where the NSIS bundler dropped the installer so the release.yml recipe can target the same + # path (confirms CARGO_TARGET_DIR is honoured by the bundler, not just the compiler). + - name: Locate the installer (diagnostic) + if: always() + shell: pwsh + run: | + Get-ChildItem -Recurse "D:\b","app\src-tauri\target" -Include *-setup.exe -ErrorAction SilentlyContinue | + Select-Object FullName, Length | Format-Table -Auto + # Upload the installer so it can be downloaded and tested on real Windows hardware (GPU + non-GPU). + - name: Upload the Windows GPU installer + if: always() + uses: actions/upload-artifact@v4 + with: + name: wisp-windows-gpu-installer + if-no-files-found: error + # tauri honours CARGO_TARGET_DIR, so the NSIS bundle lands under D:\b (confirmed by the + # locate step). A single absolute path keeps upload-artifact's root unambiguous. + path: D:\b\release\bundle\nsis\*-setup.exe + # Always list what the CMake build actually produced, so a link-path/lib-name mismatch in + # build.rs is visible from the log even when the build fails at the link step. + - name: List the produced whisper/ggml libs (diagnostic) + if: always() + shell: pwsh + run: | + Get-ChildItem -Recurse "$env:CARGO_TARGET_DIR" -Include *.lib,*.dll -ErrorAction SilentlyContinue | + Where-Object { $_.Name -match '(?i)whisper|ggml|vulkan' } | + Select-Object FullName, Length | Sort-Object FullName | Format-Table -Auto diff --git a/app/src-tauri/Cargo.toml b/app/src-tauri/Cargo.toml index 32dea9b..395f850 100644 --- a/app/src-tauri/Cargo.toml +++ b/app/src-tauri/Cargo.toml @@ -39,6 +39,9 @@ sherpa-cpu = ["wisp-engine-sherpa/download-binaries"] # weekly smoke workflow (cargo build --no-default-features --features sherpa-gpu) — a ~2h compile, # too slow to gate every PR. Mutually exclusive with sherpa-cpu (sherpa-rs-sys enforces this). sherpa-gpu = ["wisp-engine-sherpa/directml"] +# whisper.cpp with the Vulkan GPU backend on Windows (generic GPU — AMD/Intel/NVIDIA — with a CPU +# fallback). Off by default; the Windows GPU release turns it on. No effect on macOS (always Metal). +whisper-vulkan = ["wisp-engine-whisper-cpp/vulkan", "wisp-models/whisper-vulkan"] [build-dependencies] tauri-build = { version = "2", features = [] } @@ -82,6 +85,9 @@ wisp-applespeech = { path = "../../crates/wisp-applespeech" } # ScreenCaptureKit, so Windows gets the same one-click "capture the meeting" Live experience. [target.'cfg(target_os = "windows")'.dependencies] wisp-loopback = { path = "../../crates/wisp-loopback" } +# whisper.cpp ASR engine — built with the Vulkan GPU backend only under the `whisper-vulkan` feature +# (generic GPU + CPU fallback); an empty shell in the default CPU build. +wisp-engine-whisper-cpp = { path = "../../crates/wisp-engine-whisper-cpp" } # Total physical memory (GlobalMemoryStatusEx) to recommend a machine-appropriate model. Same # `windows` major as wisp-loopback, so no extra version in the tree. windows = { version = "0.61", features = ["Win32_System_SystemInformation"] } diff --git a/app/src-tauri/src/lib.rs b/app/src-tauri/src/lib.rs index 266dcb4..21d68e7 100644 --- a/app/src-tauri/src/lib.rs +++ b/app/src-tauri/src/lib.rs @@ -475,9 +475,9 @@ fn coerce_param(raw: &serde_json::Value, kind: &ParamKind) -> Option } } -/// Builds the GPU (Metal) whisper.cpp engine from a downloaded GGUF model. macOS only; elsewhere -/// this family isn't offered, so the stub just reports it. -#[cfg(target_os = "macos")] +/// Builds the GPU whisper.cpp engine from a downloaded GGUF model — Metal on macOS, Vulkan on Windows +/// (under the `whisper-vulkan` feature). Where the engine isn't built, the stub below reports it. +#[cfg(any(target_os = "macos", all(target_os = "windows", feature = "whisper-vulkan")))] fn build_whisper_cpp_engine( descriptor: &ModelDescriptor, dir: &Path, @@ -493,14 +493,14 @@ fn build_whisper_cpp_engine( Ok(Box::new(engine)) } -#[cfg(not(target_os = "macos"))] +#[cfg(not(any(target_os = "macos", all(target_os = "windows", feature = "whisper-vulkan"))))] fn build_whisper_cpp_engine( _descriptor: &ModelDescriptor, _dir: &Path, _language: &str, ) -> WispResult> { Err(WispError::Engine( - "the whisper.cpp GPU engine is only available on macOS".to_owned(), + "the whisper.cpp GPU engine is only available on macOS, and on Windows GPU builds".to_owned(), )) } diff --git a/crates/wisp-engine-whisper-cpp/Cargo.toml b/crates/wisp-engine-whisper-cpp/Cargo.toml index a4be81b..1926d6a 100644 --- a/crates/wisp-engine-whisper-cpp/Cargo.toml +++ b/crates/wisp-engine-whisper-cpp/Cargo.toml @@ -10,6 +10,12 @@ wisp-core = { path = "../wisp-core" } # Physical-core detection for the decode thread count (pure Rust; no native lib linked). num_cpus = "1" +[features] +# Build whisper.cpp with the Vulkan backend on Windows (generic GPU — AMD/Intel/NVIDIA — with a CPU +# fallback). Off by default so the standard Windows build stays a no-op shell and needs no Vulkan SDK. +# No effect on macOS, which always builds the Metal + Core ML backend. +vulkan = [] + [build-dependencies] cmake = "0.1" bindgen = "0.70" diff --git a/crates/wisp-engine-whisper-cpp/build.rs b/crates/wisp-engine-whisper-cpp/build.rs index 670a6ac..1f3c38b 100644 --- a/crates/wisp-engine-whisper-cpp/build.rs +++ b/crates/wisp-engine-whisper-cpp/build.rs @@ -1,11 +1,19 @@ -//! Builds the vendored whisper.cpp (with the Metal backend) and generates Rust bindings for its -//! C API. macOS-only for now; on other targets this is a no-op so the crate is an empty shell. +//! Builds the vendored whisper.cpp and generates Rust bindings for its C API. +//! +//! - **macOS**: Metal + Core ML backend (always). +//! - **Windows**: Vulkan backend — generic GPU across AMD/Intel/NVIDIA, with ggml's built-in CPU +//! fallback — but only when the `vulkan` feature is on, so the default Windows build stays a no-op +//! shell and needs no Vulkan SDK. +//! - **Other targets**: a no-op, so the crate is an empty shell. use std::env; -use std::path::PathBuf; +use std::path::{Path, PathBuf}; fn main() { - if env::var("CARGO_CFG_TARGET_OS").as_deref() != Ok("macos") { + let target_os = env::var("CARGO_CFG_TARGET_OS").unwrap_or_default(); + let windows_vulkan = target_os == "windows" && env::var("CARGO_FEATURE_VULKAN").is_ok(); + + if target_os != "macos" && !windows_vulkan { return; } @@ -16,9 +24,19 @@ fn main() { "whisper.cpp submodule missing — run `git submodule update --init --recursive`" ); - // Build whisper.cpp + ggml as static libs with the Metal backend, embedding the Metal shader - // library into the binary so nothing extra has to ship at runtime. - let dst = cmake::Config::new(&src) + if target_os == "macos" { + build_macos(&src); + } else { + build_windows_vulkan(&src); + } + + generate_bindings(&src); +} + +/// Builds whisper.cpp + ggml as static libs with the Metal backend, embedding the Metal shader +/// library into the binary so nothing extra has to ship at runtime, plus the Core ML encoder path. +fn build_macos(src: &Path) { + let dst = cmake::Config::new(src) .profile("Release") // ggml uses std::filesystem (introduced in macOS 10.15). Pin a modern deployment target so // the C++ build doesn't inherit a lower one from the embedding app's build environment @@ -79,7 +97,72 @@ fn main() { ] { println!("cargo:rustc-link-lib=framework={framework}"); } +} + +/// Builds whisper.cpp + ggml as static libs with the Vulkan backend (generic GPU; ggml falls back to +/// CPU when no Vulkan device is present). The Vulkan headers + loader import lib come from the Vulkan +/// SDK at build time (CI installs it; `VULKAN_SDK` points at it); at runtime `vulkan-1.dll` ships with +/// the GPU driver, so nothing extra has to be bundled. +fn build_windows_vulkan(src: &Path) { + let mut config = cmake::Config::new(src); + config + // Ninja, not the Visual Studio generator: ggml-vulkan builds its `vulkan-shaders-gen` host + // tool via ExternalProject, and the VS generator fails to initialize the compiler in that + // sub-build ("No CMAKE_C_COMPILER could be found"). Ninja + the MSVC dev environment (set up + // in CI) lets every sub-build find cl.exe. Requires Ninja + the MSVC env on PATH. + .generator("Ninja") + .profile("Release") + .define("BUILD_SHARED_LIBS", "OFF") + .define("WHISPER_BUILD_EXAMPLES", "OFF") + .define("WHISPER_BUILD_TESTS", "OFF") + .define("WHISPER_BUILD_SERVER", "OFF") + .define("GGML_VULKAN", "ON") + .define("GGML_OPENMP", "OFF"); + + // Help CMake find the Vulkan SDK. ggml-vulkan only appends the SDK to CMAKE_PREFIX_PATH when + // VULKAN_SDK is set in the *cmake* environment — which cmake-rs does not forward by default — so + // pass it through explicitly. With that plus an explicit prefix path and the loader paths, + // find_package locates Vulkan, SPIRV-Headers, and SPIRV-Tools without relying on auto-detection. + if let Ok(sdk) = env::var("VULKAN_SDK") { + config.env("VULKAN_SDK", &sdk); + // The SDK drops its package configs flat in Lib/cmake (e.g. SPIRV-HeadersConfig.cmake) rather + // than in per-package subdirs, so add that dir itself to the prefix path for find_package. + config.define("CMAKE_PREFIX_PATH", format!("{sdk};{sdk}/Lib/cmake")); + config.define("Vulkan_INCLUDE_DIR", format!("{sdk}/Include")); + config.define("Vulkan_LIBRARY", format!("{sdk}/Lib/vulkan-1.lib")); + } + + let dst = config.build(); + + // MSVC is multi-config, so the libs land under per-config (`Release`) subdirs of the build tree + // and/or the install prefix — search the likely spots so linking is robust to either layout. + let build = dst.join("build"); + for dir in [ + dst.join("lib"), + build.join("src"), + build.join("src/Release"), + build.join("ggml/src"), + build.join("ggml/src/Release"), + build.join("ggml/src/ggml-vulkan"), + build.join("ggml/src/ggml-vulkan/Release"), + ] { + println!("cargo:rustc-link-search=native={}", dir.display()); + } + + for lib in ["whisper", "ggml", "ggml-cpu", "ggml-vulkan", "ggml-base"] { + println!("cargo:rustc-link-lib=static={lib}"); + } + + // The Vulkan loader import library, from the Vulkan SDK. + if let Ok(sdk) = env::var("VULKAN_SDK") { + println!("cargo:rustc-link-search=native={}\\Lib", sdk); + } + println!("cargo:rustc-link-lib=vulkan-1"); +} +/// Generates the Rust FFI bindings for whisper.cpp's C API. Platform-independent — it only parses the +/// public headers, so the same bindings serve every backend. +fn generate_bindings(src: &Path) { let whisper_h = src.join("include/whisper.h"); let bindings = bindgen::Builder::default() .header(whisper_h.to_string_lossy()) diff --git a/crates/wisp-engine-whisper-cpp/src/lib.rs b/crates/wisp-engine-whisper-cpp/src/lib.rs index 45205c3..9e8f00e 100644 --- a/crates/wisp-engine-whisper-cpp/src/lib.rs +++ b/crates/wisp-engine-whisper-cpp/src/lib.rs @@ -1,10 +1,12 @@ -//! whisper.cpp ASR engine with Metal (GPU) acceleration. +//! whisper.cpp ASR engine with GPU acceleration — Metal on macOS, Vulkan on Windows. //! //! Wraps the vendored whisper.cpp behind [`wisp_core::AsrEngine`] so it drops into the pipeline -//! like the sherpa engines — but runs on the Apple GPU via Metal instead of CPU-only ONNX, which -//! makes large models (e.g. large-v3-turbo) usable in real time. macOS-only for now. +//! like the sherpa engines — but runs on the GPU (Apple Metal, or Vulkan across AMD/Intel/NVIDIA) +//! instead of CPU-only ONNX, which makes large models (e.g. large-v3-turbo) usable in real time. -#![cfg(target_os = "macos")] +// macOS always builds the Metal + Core ML backend; Windows builds the Vulkan backend only under the +// `vulkan` feature (see build.rs). Elsewhere this crate is an empty shell. +#![cfg(any(target_os = "macos", all(target_os = "windows", feature = "vulkan")))] mod sys { #![allow( diff --git a/crates/wisp-models/Cargo.toml b/crates/wisp-models/Cargo.toml index 4e0858e..46d4c2f 100644 --- a/crates/wisp-models/Cargo.toml +++ b/crates/wisp-models/Cargo.toml @@ -11,6 +11,10 @@ repository.workspace = true default = ["http"] # Provides HttpDownloader (network model downloads via ureq). http = ["dep:ureq"] +# Set when the app is built with the whisper.cpp Vulkan GPU backend, so the catalog offers Whisper +# models on Windows (they run on the GPU, or on CPU via the backend's fallback). Off, Whisper stays +# macOS/Metal-only. The app's `whisper-vulkan` feature turns this on. +whisper-vulkan = [] [dependencies] wisp-core = { path = "../wisp-core" } diff --git a/crates/wisp-models/src/machine.rs b/crates/wisp-models/src/machine.rs index 424fdf9..a41bbd7 100644 --- a/crates/wisp-models/src/machine.rs +++ b/crates/wisp-models/src/machine.rs @@ -169,7 +169,14 @@ fn resolve(ideal: &str, catalog: &[ModelDescriptor]) -> ModelId { /// scattered `cfg` checks. pub fn family_runnable(family: ModelFamily, accelerator: Accelerator) -> bool { match family { - ModelFamily::WhisperCpp | ModelFamily::AppleSpeech => accelerator == Accelerator::Metal, + // whisper.cpp runs on the Apple GPU (Metal), or — in a Windows GPU build — on Vulkan with a + // built-in CPU fallback, so it's offered on Windows too (no GPU required to run, just to + // accelerate). Apple on-device speech stays macOS/Metal-only. + ModelFamily::WhisperCpp => { + accelerator == Accelerator::Metal + || (cfg!(target_os = "windows") && cfg!(feature = "whisper-vulkan")) + } + ModelFamily::AppleSpeech => accelerator == Accelerator::Metal, _ => true, } }