diff --git a/.forgejo/workflows/release.yml b/.forgejo/workflows/release.yml index 1875af9..e53b875 100644 --- a/.forgejo/workflows/release.yml +++ b/.forgejo/workflows/release.yml @@ -16,10 +16,15 @@ jobs: "https://git.breadway.dev/${GITHUB_REPOSITORY}.git" src - name: build - run: cd src && cargo build --release --locked + # --features full (breadmill's npu+rocm+cuda combined) ships one binary + # that can use any backend at runtime via --npu/--rocm/--cuda or + # `backend` in config.toml. All three are ort load-dynamic (dlopen) + # EPs, so this doesn't require the NPU/ROCm/CUDA toolkits to be + # present on the build host — see breadmill/Cargo.toml. + run: cd src && cargo build --release --locked --workspace --features full - name: test - run: cd src && cargo test --release --locked --workspace + run: cd src && cargo test --release --locked --workspace --features full - name: prepare artifacts run: | diff --git a/Cargo.lock b/Cargo.lock index f751ae5..2e7fa2b 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -133,7 +133,7 @@ dependencies = [ [[package]] name = "breadmill" -version = "0.2.0" +version = "0.2.1" dependencies = [ "breadsearch-shared", "hex", diff --git a/README.md b/README.md index 700cd82..446787d 100644 --- a/README.md +++ b/README.md @@ -24,6 +24,7 @@ Optional features: | `npu` | AMD XDNA NPU via VitisAI ONNX Runtime EP (requires Ryzen AI SDK) | | `rocm` | AMD iGPU via the MIGraphX ONNX Runtime EP (ROCm-backed) | | `cuda` | NVIDIA GPU via the CUDA ONNX Runtime EP | +| `full` | All three of the above in one binary | ``` # NPU build @@ -34,6 +35,9 @@ cargo build --release -p breadmill --features rocm # CUDA (NVIDIA GPU) build cargo build --release -p breadmill --features cuda + +# All backends in one binary (what the release build ships) +cargo build --release -p breadmill --features full ``` `rocm`/`cuda`/`npu` all use `ort`'s `load-dynamic` mode: at runtime, breadmill @@ -43,6 +47,15 @@ build actually has the matching execution provider compiled in — breadmill logs a clear `Successfully registered` / `not enabled in this build` line for this at startup (see [GPU backend notes](#gpu-backend-notes) below). +Because all three are dlopen-based, `full` doesn't require the NPU/ROCm/CUDA +toolkits to be installed at build time — only at run time, and only for +whichever single backend you actually select via `--npu`/`--rocm`/`--cuda` +or `backend` in config.toml. The **released binaries are built with +`full`**: same binary works CPU-only out of the box, and picks up NPU/ROCm/CUDA +acceleration on a machine that has the matching ONNX Runtime available, +without needing a different download. An explicit `--npu`/`--rocm`/`--cuda` +flag always overrides `backend` in config.toml, not the other way around. + ## Setup **1. Fetch the embedding model** (~550 MB, downloaded once from Hugging Face): diff --git a/breadmill/Cargo.toml b/breadmill/Cargo.toml index f9cc209..5e3822b 100644 --- a/breadmill/Cargo.toml +++ b/breadmill/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "breadmill" -version = "0.2.0" +version = "0.2.1" edition = "2021" license = "MIT" @@ -14,6 +14,13 @@ npu = ["ort/vitis", "ort/load-dynamic"] # libonnxruntime_providers_rocm.so, which most distros don't package. rocm = ["ort/migraphx", "ort/load-dynamic"] cuda = ["ort/cuda", "ort/load-dynamic"] +# All backends in one binary. Safe to combine: every backend here uses +# ort's load-dynamic (dlopen) mode, so none of this links against an actual +# NPU/ROCm/CUDA toolkit at build time — which ONNX Runtime actually gets +# loaded (and thus which EPs are really available) is decided at runtime by +# ORT_DYLIB_PATH / the dynamic linker, per the --npu/--rocm/--cuda flag or +# `backend` config value in use for that run. +full = ["npu", "rocm", "cuda"] [[bin]] name = "breadmill" diff --git a/breadmill/src/main.rs b/breadmill/src/main.rs index b5bc99c..8d2e9bd 100644 --- a/breadmill/src/main.rs +++ b/breadmill/src/main.rs @@ -105,17 +105,34 @@ fn run_daemon(force_reindex: bool, use_npu: bool, use_rocm: bool, use_cuda: bool std::fs::create_dir_all(&state_dir).map_err(|e| e.to_string())?; std::fs::create_dir_all(&cache_dir).map_err(|e| e.to_string())?; - let backend = if use_npu || config.model.backend == "npu" { - eprintln!("breadmill: NPU backend selected"); - Backend::Npu { cache_dir: cache_dir.clone() } - } else if use_rocm || config.model.backend == "rocm" { - eprintln!("breadmill: ROCm backend selected"); - Backend::Rocm - } else if use_cuda || config.model.backend == "cuda" { - eprintln!("breadmill: CUDA backend selected"); - Backend::Cuda + // CLI flags always override config — otherwise an explicit --cuda/--npu on + // the command line would silently lose to an unrelated `backend = "..."` + // already sitting in config.toml, since that's whatever earlier branch a + // fixed if/else-if priority order happened to check first. + let backend_name = if use_npu { + "npu" + } else if use_rocm { + "rocm" + } else if use_cuda { + "cuda" } else { - Backend::Cpu + config.model.backend.as_str() + }; + + let backend = match backend_name { + "npu" => { + eprintln!("breadmill: NPU backend selected"); + Backend::Npu { cache_dir: cache_dir.clone() } + } + "rocm" => { + eprintln!("breadmill: ROCm backend selected"); + Backend::Rocm + } + "cuda" => { + eprintln!("breadmill: CUDA backend selected"); + Backend::Cuda + } + _ => Backend::Cpu, }; let store = Store::open(&state_dir, dim)?;