From 0ca65c7dcaf0bfb61f9e32a8da5381b0c4e41b0f Mon Sep 17 00:00:00 2001 From: James Dumay Date: Fri, 27 Mar 2026 21:36:19 +1100 Subject: [PATCH] Add ROCm-aware Linux build and device detection --- CONTRIBUTING.md | 14 +-- Justfile | 54 +++++++--- README.md | 2 +- RELEASE.md | 2 +- mesh-llm/src/election.rs | 6 +- mesh-llm/src/hardware.rs | 97 +++++++++++++----- mesh-llm/src/launch.rs | 64 +++++++----- mesh-llm/src/main.rs | 2 +- scripts/build-linux.sh | 177 +++++++++++++++++++++++++++------ scripts/detect-llama-device.sh | 44 ++++++++ scripts/detect-rocm-arch.sh | 77 ++++++++++++++ 11 files changed, 430 insertions(+), 109 deletions(-) create mode 100755 scripts/detect-llama-device.sh create mode 100755 scripts/detect-rocm-arch.sh diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f9b6d6958..e0e00d017 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -11,9 +11,9 @@ This file covers local build and development workflows for this repository. **macOS**: Apple Silicon. Metal is used automatically. -**Linux**: x86_64 with an NVIDIA GPU. Requires the CUDA toolkit (`nvcc` in your `PATH`). On Arch Linux, CUDA is typically at `/opt/cuda`; on Ubuntu/Debian it's at `/usr/local/cuda`. Auto-detection finds the right SM architecture for your GPU. +**Linux NVIDIA**: x86_64 with an NVIDIA GPU. Requires the CUDA toolkit (`nvcc` in your `PATH`). On Arch Linux, CUDA is typically at `/opt/cuda`; on Ubuntu/Debian it's at `/usr/local/cuda`. Auto-detection finds the right SM architecture for your GPU. -**Linux AMD**: ROCm/HIP is supported when ROCm is installed (typically under `/opt/rocm`). +**Linux AMD**: ROCm/HIP is supported when ROCm is installed. Typical installs expose `hipcc`, `hipconfig`, and `rocm-smi` under `/opt/rocm/bin`. ## Build from source @@ -23,7 +23,7 @@ Build everything (llama.cpp fork, mesh binary, and UI production build): just build ``` -On Linux, make sure `nvcc` is in your `PATH` first: +On Linux, `just build` auto-detects CUDA vs ROCm. For NVIDIA, make sure `nvcc` is in your `PATH` first: ```bash # Arch Linux @@ -33,22 +33,22 @@ PATH=/opt/cuda/bin:$PATH just build PATH=/usr/local/cuda/bin:$PATH just build ``` -The build script auto-detects your GPU's CUDA architecture. To override: +For NVIDIA builds, the script auto-detects your GPU's CUDA architecture. To override: ```bash just build cuda_arch=90 # e.g. H100 ``` -For AMD ROCm builds: +For AMD ROCm builds, you can force the backend explicitly: ```bash -scripts/build-linux-amd.sh +just build backend=rocm ``` To override the AMD GPU target list: ```bash -scripts/build-linux-amd.sh "gfx90a;gfx942;gfx1100" +just build backend=rocm rocm_arch="gfx90a;gfx942;gfx1100" ``` Create a portable bundle: diff --git a/Justfile b/Justfile index 64e6508ab..396cf63d4 100644 --- a/Justfile +++ b/Justfile @@ -7,14 +7,16 @@ ui_dir := mesh_dir / "ui" models_dir := env("HOME") / ".models" model := models_dir / "GLM-4.7-Flash-Q4_K_M.gguf" -# Build for the current platform (macOS→Metal, Linux→CUDA with auto-detected arch) +# Build for the current platform (macOS→Metal, Linux→CUDA/ROCm auto-detected) [macos] build: build-mac -# Pass cuda_arch to override auto-detection (e.g. just build cuda_arch=90) +# Linux overrides: +# just build backend=cuda cuda_arch='120;86' +# just build backend=rocm rocm_arch='gfx942;gfx90a' [linux] -build cuda_arch="": - @scripts/build-linux.sh "{{ cuda_arch }}" +build backend="" cuda_arch="" rocm_arch="": + @scripts/build-linux.sh --backend "{{ backend }}" --cuda-arch "{{ cuda_arch }}" --rocm-arch "{{ rocm_arch }}" # Build on macOS Apple Silicon (Metal + RPC) build-mac: @@ -47,11 +49,9 @@ build-mac: echo "Mesh binary: target/release/mesh-llm" fi -# Build on Linux with CUDA — delegates to scripts/build-linux.sh - -# cuda_arch overrides auto-detection (see scripts/detect-cuda-arch.sh for supported GPUs) -build-linux cuda_arch="": - @scripts/build-linux.sh "{{ cuda_arch }}" +# Build on Linux with CUDA or ROCm — delegates to scripts/build-linux.sh +build-linux backend="" cuda_arch="" rocm_arch="": + @scripts/build-linux.sh --backend "{{ backend }}" --cuda-arch "{{ cuda_arch }}" --rocm-arch "{{ rocm_arch }}" # Build release artifacts for the current platform. @@ -61,7 +61,24 @@ release-build: # Build a Linux CUDA release artifact with an explicit architecture list. release-build-cuda cuda_arch="75;80;86;89;90;120": - @scripts/build-linux.sh "{{ cuda_arch }}" + @scripts/build-linux.sh --backend cuda --cuda-arch "{{ cuda_arch }}" + +# Build a Linux AMD ROCm release artifact with an explicit architecture list. +release-build-amd amd_arch="gfx90a;gfx942;gfx1100;gfx1101;gfx1102;gfx1200;gfx1201": + @scripts/build-linux-amd.sh "{{ amd_arch }}" + +# Build a Linux AMD ROCm release artifact inside Docker. +release-rocm-docker amd_arch="" image="rocm/dev-ubuntu-24.04:7.0-complete" platform="": + #!/usr/bin/env bash + set -euo pipefail + ARGS=(--build-only --image "{{ image }}") + if [ -n "{{ amd_arch }}" ]; then + ARGS+=(--rocm-arch "{{ amd_arch }}") + fi + if [ -n "{{ platform }}" ]; then + ARGS+=(--platform "{{ platform }}") + fi + exec scripts/run-rocm-docker-build.sh "${ARGS[@]}" # Bump release version consistently across source and Cargo manifests. release-version version: @@ -83,8 +100,14 @@ download-model: # ── Raw TCP (no mesh) ────────────────────────────────────────── # Start rpc-server (worker) with local GGUF loading -worker host="0.0.0.0" port="50052" device="MTL0" gguf=model: - {{ build_dir }}/bin/rpc-server --host {{ host }} --port {{ port }} -d {{ device }} --gguf {{ gguf }} +worker host="0.0.0.0" port="50052" device="" gguf=model: + #!/usr/bin/env bash + set -euo pipefail + DEVICE="{{ device }}" + if [ -z "$DEVICE" ]; then + DEVICE="$(scripts/detect-llama-device.sh)" + fi + exec {{ build_dir }}/bin/rpc-server --host {{ host }} --port {{ port }} -d "$DEVICE" --gguf {{ gguf }} # Start llama-server (orchestrator) pointing at an RPC worker serve rpc="127.0.0.1:50052" port="8080" gguf=model: @@ -98,8 +121,9 @@ serve rpc="127.0.0.1:50052" port="8080" gguf=model: local: build download-model #!/usr/bin/env bash set -euo pipefail + DEVICE="$(scripts/detect-llama-device.sh)" echo "Starting rpc-server (worker)..." - {{ build_dir }}/bin/rpc-server --host 127.0.0.1 --port 50052 -d MTL0 --gguf {{ model }} & + {{ build_dir }}/bin/rpc-server --host 127.0.0.1 --port 50052 -d "$DEVICE" --gguf {{ model }} & WORKER_PID=$! sleep 3 echo "Starting llama-server (orchestrator)..." @@ -174,6 +198,10 @@ release-bundle version output="dist": release-bundle-cuda version output="dist": MESH_RELEASE_FLAVOR=cuda scripts/package-release.sh "{{ version }}" "{{ output }}" +# Create Linux AMD ROCm release archive(s). +release-bundle-amd version output="dist": + MESH_RELEASE_FLAVOR=rocm scripts/package-release.sh "{{ version }}" "{{ output }}" + # Run the UI with Vite HMR and proxy /api to mesh-llm (default: http://127.0.0.1:3131) ui-dev api="http://127.0.0.1:3131" port="5173": #!/usr/bin/env bash diff --git a/README.md b/README.md index 47998cd90..8e3381575 100644 --- a/README.md +++ b/README.md @@ -36,7 +36,7 @@ cd mesh-llm just build ``` -Requires: `just`, `cmake`, Rust toolchain, Node.js + npm. NVIDIA GPU builds need `nvcc` (CUDA toolkit). AMD GPU builds need ROCm/HIP. CPU-only and Jetson/Tegra also work. See [CONTRIBUTING.md](CONTRIBUTING.md) for details. +Requires: `just`, `cmake`, Rust toolchain, Node.js + npm. NVIDIA GPU builds need `nvcc` (CUDA toolkit). AMD GPU builds need ROCm/HIP. CPU-only and Jetson/Tegra also work. For source builds, `just build` auto-detects CUDA vs ROCm on Linux, or you can force `backend=rocm`. See [CONTRIBUTING.md](CONTRIBUTING.md) for details. ## Run Once installed, you can run: diff --git a/RELEASE.md b/RELEASE.md index dba52885b..d107673ec 100644 --- a/RELEASE.md +++ b/RELEASE.md @@ -16,7 +16,7 @@ just build ``` -This clones/updates the llama.cpp fork if needed, builds with `-DGGML_METAL=ON -DGGML_RPC=ON -DBUILD_SHARED_LIBS=OFF -DLLAMA_OPENSSL=OFF`, and builds the Rust mesh-llm binary. +On macOS, this clones/updates the llama.cpp fork if needed, builds with `-DGGML_METAL=ON -DGGML_RPC=ON -DBUILD_SHARED_LIBS=OFF -DLLAMA_OPENSSL=OFF`, and builds the Rust mesh-llm binary. Linux release workflows build CPU, CUDA, and ROCm variants separately. ### 2. Verify no homebrew dependencies diff --git a/mesh-llm/src/election.rs b/mesh-llm/src/election.rs index 90eaa3777..e17125976 100644 --- a/mesh-llm/src/election.rs +++ b/mesh-llm/src/election.rs @@ -1020,7 +1020,7 @@ async fn start_llama( } // Build --rpc list: only remote workers. - // The host's own GPU is used directly via Metal — no need to route + // The host's own GPU is used directly on the local backend — no need to route // through the local rpc-server (which would add unnecessary TCP round trips). let all_ports = tunnel_mgr.peer_ports_map().await; let mut rpc_ports: Vec = Vec::new(); @@ -1031,7 +1031,7 @@ async fn start_llama( } // Calculate tensor split from VRAM. - // Device order: RPC workers first (matching --rpc order), then Metal (host) last. + // Device order: RPC workers first (matching --rpc order), then the local host device last. let my_vram_f = my_vram as f64; let mut all_vrams: Vec = Vec::new(); for id in &worker_ids { @@ -1043,7 +1043,7 @@ async fn start_llama( }); } } - all_vrams.push(my_vram_f); // Metal is last device + all_vrams.push(my_vram_f); // Host device is last let total: f64 = all_vrams.iter().sum(); let split = if total > 0.0 && !rpc_ports.is_empty() { let s: Vec = all_vrams diff --git a/mesh-llm/src/hardware.rs b/mesh-llm/src/hardware.rs index 67e337760..e926e173e 100644 --- a/mesh-llm/src/hardware.rs +++ b/mesh-llm/src/hardware.rs @@ -53,18 +53,32 @@ pub fn parse_macos_cpu_brand(output: &str) -> Option { } } -/// Parse `rocm-smi --showproductname` output → GPU name from "Card series:" line. +/// Parse `rocm-smi --showproductname` output → GPU names from "Card series:" lines. #[cfg(any(target_os = "linux", test))] -pub fn parse_rocm_gpu_name(output: &str) -> Option { +pub fn parse_rocm_gpu_names(output: &str) -> Vec { + let mut names = Vec::new(); for line in output.lines() { if let Some(pos) = line.find("Card series:") { let val = line[pos + "Card series:".len()..].trim(); if !val.is_empty() { - return Some(val.to_string()); + names.push(val.to_string()); } } } - None + names +} + +/// Parse `rocm-smi --showmeminfo vram --csv` output → per-GPU VRAM bytes. +#[cfg(any(target_os = "linux", test))] +pub fn parse_rocm_gpu_vrams(output: &str) -> Vec { + output + .lines() + .skip(1) + .filter_map(|line| { + let total = line.split(',').nth(1)?; + total.trim().parse::().ok() + }) + .collect() } /// Summarize GPU names: empty→None, 1→name, N identical→"N× name", N mixed→"a, b". @@ -239,7 +253,7 @@ impl Collector for DefaultCollector { survey.vram_bytes = vram + (ram_offload as f64 * 0.75) as u64; } else { // Try AMD ROCm (mesh.rs:295-316) - let rocm_vram: Option = (|| { + let rocm_vram: Option> = (|| { let out = std::process::Command::new("rocm-smi") .args(["--showmeminfo", "vram", "--csv"]) .output() @@ -248,18 +262,13 @@ impl Collector for DefaultCollector { return None; } let s = String::from_utf8(out.stdout).ok()?; - for line in s.lines().skip(1) { - if let Some(total) = line.split(',').nth(1) { - if let Ok(bytes) = total.trim().parse::() { - return Some(bytes); - } - } - } - None + let vrams = parse_rocm_gpu_vrams(&s); + if vrams.is_empty() { None } else { Some(vrams) } })(); - if let Some(vram) = rocm_vram { - survey.gpu_vram = vec![vram]; + if let Some(per_gpu) = rocm_vram { + let vram: u64 = per_gpu.iter().sum(); + survey.gpu_vram = per_gpu; let ram_offload = system_ram.saturating_sub(vram); survey.vram_bytes = vram + (ram_offload as f64 * 0.75) as u64; } else if system_ram > 0 { @@ -302,15 +311,12 @@ impl Collector for DefaultCollector { if let Some(out) = out { if out.status.success() { if let Ok(s) = String::from_utf8(out.stdout) { + let names = parse_rocm_gpu_names(&s); if metrics.contains(&Metric::GpuName) { - survey.gpu_name = parse_rocm_gpu_name(&s); + survey.gpu_name = summarize_gpu_name(&names); } if metrics.contains(&Metric::GpuCount) { - let count = s - .lines() - .filter(|l| l.trim_start().starts_with("GPU[")) - .count(); - survey.gpu_count = u8::try_from(count).unwrap_or(u8::MAX); + survey.gpu_count = u8::try_from(names.len()).unwrap_or(u8::MAX); } } } @@ -455,18 +461,61 @@ mod tests { } #[test] - fn test_parse_rocm_gpu_name() { + fn test_parse_rocm_gpu_names_single() { let fixture = "\ ======================= ROCm System Management Interface ======================= ================================= Product Info ================================= GPU[0]\t\t: Card series:\t\t\tNavi31 [Radeon RX 7900 XTX] ================================================================================"; assert_eq!( - parse_rocm_gpu_name(fixture), - Some("Navi31 [Radeon RX 7900 XTX]".to_string()) + parse_rocm_gpu_names(fixture), + vec!["Navi31 [Radeon RX 7900 XTX]".to_string()] ); } + #[test] + fn test_parse_rocm_gpu_names_multi() { + let fixture = "\ +======================= ROCm System Management Interface ======================= +================================= Product Info ================================= +GPU[0]\t\t: Card series:\t\t\tAMD Instinct MI300X +GPU[1]\t\t: Card series:\t\t\tAMD Instinct MI300X +================================================================================"; + assert_eq!( + parse_rocm_gpu_names(fixture), + vec![ + "AMD Instinct MI300X".to_string(), + "AMD Instinct MI300X".to_string() + ] + ); + } + + #[test] + fn test_parse_rocm_gpu_vrams_single() { + let fixture = "\ +device,VRAM Total Memory (B),VRAM Total Used Memory (B) +card0,25753026560,416378880"; + assert_eq!(parse_rocm_gpu_vrams(fixture), vec![25753026560]); + } + + #[test] + fn test_parse_rocm_gpu_vrams_multi() { + let fixture = "\ +device,VRAM Total Memory (B),VRAM Total Used Memory (B) +card0,25753026560,416378880 +card1,25753026560,512000000"; + assert_eq!(parse_rocm_gpu_vrams(fixture), vec![25753026560, 25753026560]); + } + + #[test] + fn test_parse_rocm_gpu_vrams_ignores_invalid_rows() { + let fixture = "\ +device,VRAM Total Memory (B),VRAM Total Used Memory (B) +card0,25753026560,416378880 +card1,not-a-number,512000000"; + assert_eq!(parse_rocm_gpu_vrams(fixture), vec![25753026560]); + } + #[test] fn test_summarize_gpu_name_single() { assert_eq!( diff --git a/mesh-llm/src/launch.rs b/mesh-llm/src/launch.rs index de321c5b7..82c441c12 100644 --- a/mesh-llm/src/launch.rs +++ b/mesh-llm/src/launch.rs @@ -16,6 +16,16 @@ fn temp_log_path(name: &str) -> PathBuf { std::env::temp_dir().join(name) } +fn command_has_output(command: &str, args: &[&str]) -> bool { + let Ok(output) = std::process::Command::new(command).args(args).output() else { + return false; + }; + output.status.success() + && String::from_utf8_lossy(&output.stdout) + .lines() + .any(|line| !line.trim().is_empty()) +} + /// Start a local rpc-server and return the port it's listening on. /// Picks an available port automatically. /// If `gguf_path` is provided, passes `--gguf` so the server loads weights from the local file. @@ -286,21 +296,30 @@ pub async fn start_llama_server( args.push("--tensor-split".to_string()); args.push(ts.to_string()); } + let local_device = detect_device(); if let Some(draft_path) = draft { if draft_path.exists() { - args.push("-md".to_string()); - args.push(draft_path.to_string_lossy().to_string()); - args.push("-ngld".to_string()); - args.push("99".to_string()); - args.push("--device-draft".to_string()); - args.push("MTL0".to_string()); - args.push("--draft-max".to_string()); - args.push(draft_max.to_string()); - tracing::info!( - "Speculative decoding: draft={}, draft-max={}", - draft_path.display(), - draft_max - ); + if local_device != "CPU" { + args.push("-md".to_string()); + args.push(draft_path.to_string_lossy().to_string()); + args.push("-ngld".to_string()); + args.push("99".to_string()); + args.push("--device-draft".to_string()); + args.push(local_device.clone()); + args.push("--draft-max".to_string()); + args.push(draft_max.to_string()); + tracing::info!( + "Speculative decoding: draft={}, draft-max={}, device={}", + draft_path.display(), + draft_max, + local_device + ); + } else { + tracing::warn!( + "Draft model present at {} but no GPU backend detected, skipping speculative decoding", + draft_path.display() + ); + } } else { tracing::warn!( "Draft model not found at {}, skipping speculative decoding", @@ -383,19 +402,8 @@ fn detect_device() -> String { } // Linux: check for NVIDIA CUDA - if let Ok(output) = std::process::Command::new("nvidia-smi") - .args(["--query-gpu=name", "--format=csv,noheader"]) - .output() - { - if output.status.success() { - let gpu_count = String::from_utf8_lossy(&output.stdout) - .lines() - .filter(|l| !l.trim().is_empty()) - .count(); - if gpu_count > 0 { - return "CUDA0".to_string(); - } - } + if command_has_output("nvidia-smi", &["--query-gpu=name", "--format=csv,noheader"]) { + return "CUDA0".to_string(); } // Linux: check for NVIDIA Tegra/Jetson (tegrastats — Jetson AGX/NX devices support CUDA) @@ -412,7 +420,9 @@ fn detect_device() -> String { } // Linux: check for AMD ROCm/HIP - if std::path::Path::new("/opt/rocm").exists() { + if command_has_output("rocm-smi", &["--showproductname"]) + || command_has_output("rocminfo", &[]) + { return "HIP0".to_string(); } diff --git a/mesh-llm/src/main.rs b/mesh-llm/src/main.rs index fea10d714..338d22ef3 100644 --- a/mesh-llm/src/main.rs +++ b/mesh-llm/src/main.rs @@ -122,7 +122,7 @@ struct Cli { #[arg(long, hide = true)] bin_dir: Option, - /// Device for rpc-server (e.g. MTL0, CPU). + /// Device for rpc-server (e.g. MTL0, CUDA0, HIP0, Vulkan0, CPU). #[arg(long, hide = true)] device: Option, diff --git a/scripts/build-linux.sh b/scripts/build-linux.sh index 5281d1050..29ec91c7c 100755 --- a/scripts/build-linux.sh +++ b/scripts/build-linux.sh @@ -1,10 +1,13 @@ #!/usr/bin/env bash -# build-linux.sh — build llama.cpp (CUDA) + mesh-llm on Linux +# build-linux.sh — build llama.cpp + mesh-llm on Linux # -# Usage: scripts/build-linux.sh [--clean] [cuda_arch] -# --clean Wipe the build dir before cmake (required on arch change). -# cuda_arch SM integer for CMAKE_CUDA_ARCHITECTURES (e.g. 87, 90, 120). -# If omitted, scripts/detect-cuda-arch.sh is invoked to detect it. +# Usage: +# scripts/build-linux.sh [--clean] [--backend cuda|rocm] [--cuda-arch SM_LIST] [--rocm-arch GFX_LIST] +# +# Examples: +# scripts/build-linux.sh +# scripts/build-linux.sh --backend cuda --cuda-arch '120;86' +# scripts/build-linux.sh --backend rocm --rocm-arch 'gfx942;gfx90a' # # Must be run from the repository root. @@ -19,39 +22,133 @@ MESH_DIR="$REPO_ROOT/mesh-llm" UI_DIR="$MESH_DIR/ui" CLEAN=0 +BACKEND="" CUDA_ARCH="" +ROCM_ARCH="" -for ARG in "$@"; do - case "$ARG" in - --clean) CLEAN=1 ;; - *) [[ -z "$CUDA_ARCH" ]] && CUDA_ARCH="$ARG" ;; +while [[ $# -gt 0 ]]; do + case "$1" in + --clean) + CLEAN=1 + shift + ;; + --backend) + BACKEND="${2:-}" + shift 2 + ;; + --cuda-arch) + CUDA_ARCH="${2:-}" + shift 2 + ;; + --rocm-arch) + ROCM_ARCH="${2:-}" + shift 2 + ;; + *) + # Backward compatibility: treat a bare arg as cuda_arch. + [[ -z "$CUDA_ARCH" ]] && CUDA_ARCH="$1" + shift + ;; esac done -if [[ -z "$CUDA_ARCH" ]]; then - echo "No cuda_arch specified — running auto-detection..." - CUDA_ARCH="$("$SCRIPT_DIR/detect-cuda-arch.sh")" - echo "Using SM ${CUDA_ARCH}" -fi +detect_backend() { + if command -v nvidia-smi &>/dev/null; then + echo cuda + return 0 + fi + if command -v tegrastats &>/dev/null; then + echo cuda + return 0 + fi + if command -v nvcc &>/dev/null; then + echo cuda + return 0 + fi + if command -v rocm-smi &>/dev/null; then + echo rocm + return 0 + fi + if command -v rocminfo &>/dev/null; then + echo rocm + return 0 + fi + if command -v hipcc &>/dev/null; then + echo rocm + return 0 + fi + if [[ -x /opt/rocm/bin/hipcc ]]; then + echo rocm + return 0 + fi + echo cuda +} -# Locate nvcc — check PATH first, then common install locations -if ! command -v nvcc &>/dev/null; then +locate_nvcc() { + if command -v nvcc &>/dev/null; then + return 0 + fi for CANDIDATE in /usr/local/cuda/bin /opt/cuda/bin /usr/cuda/bin; do if [[ -x "$CANDIDATE/nvcc" ]]; then export PATH="$CANDIDATE:$PATH" - break + return 0 fi done + return 1 +} + +locate_hip_toolchain() { + if command -v hipcc &>/dev/null; then + return 0 + fi + for CANDIDATE in /opt/rocm/bin /usr/lib/rocm/bin /usr/local/rocm/bin; do + if [[ -x "$CANDIDATE/hipcc" ]]; then + export PATH="$CANDIDATE:$PATH" + return 0 + fi + done + return 1 +} + +if [[ -z "$BACKEND" ]]; then + BACKEND="$(detect_backend)" fi -if ! command -v nvcc &>/dev/null; then - echo "Error: nvcc not found. Install the CUDA toolkit and ensure nvcc is in your PATH." >&2 - echo " Arch Linux: sudo pacman -S cuda" >&2 - echo " Ubuntu/Debian: sudo apt install nvidia-cuda-toolkit" >&2 - exit 1 -fi - -echo "Using nvcc: $(command -v nvcc) ($(nvcc --version | grep release | awk '{print $5}' | tr -d ','))" +case "$BACKEND" in + cuda) + locate_nvcc || { + echo "Error: nvcc not found. Install the CUDA toolkit and ensure nvcc is in your PATH." >&2 + echo " Arch Linux: sudo pacman -S cuda" >&2 + echo " Ubuntu/Debian: sudo apt install nvidia-cuda-toolkit" >&2 + exit 1 + } + if [[ -z "$CUDA_ARCH" ]]; then + echo "No cuda_arch specified — running auto-detection..." + CUDA_ARCH="$("$SCRIPT_DIR/detect-cuda-arch.sh")" + echo "Using SM ${CUDA_ARCH}" + fi + echo "Building Linux backend: CUDA" + echo "Using nvcc: $(command -v nvcc) ($(nvcc --version | grep release | awk '{print $5}' | tr -d ','))" + ;; + rocm) + locate_hip_toolchain || { + echo "Error: hipcc not found. Install ROCm and ensure hipcc is in your PATH." >&2 + echo " Typical location: /opt/rocm/bin/hipcc" >&2 + exit 1 + } + if [[ -z "$ROCM_ARCH" ]]; then + echo "No rocm_arch specified — running auto-detection..." + ROCM_ARCH="$("$SCRIPT_DIR/detect-rocm-arch.sh")" + echo "Using AMDGPU_TARGETS ${ROCM_ARCH}" + fi + echo "Building Linux backend: ROCm/HIP" + echo "Using hipcc: $(command -v hipcc)" + ;; + *) + echo "Error: unsupported backend '$BACKEND' (expected 'cuda' or 'rocm')." >&2 + exit 1 + ;; +esac if [[ ! -d "$LLAMA_DIR" ]]; then echo "Cloning michaelneale/llama.cpp (rebase-upstream-master)..." @@ -74,13 +171,29 @@ if [[ "$CLEAN" -eq 1 && -d "$BUILD_DIR" ]]; then rm -rf "$BUILD_DIR" fi -cmake -B "$BUILD_DIR" -S "$LLAMA_DIR" \ - -DGGML_CUDA=ON \ - -DGGML_METAL=OFF \ - -DGGML_RPC=ON \ - -DBUILD_SHARED_LIBS=OFF \ - -DLLAMA_OPENSSL=OFF \ - -DCMAKE_CUDA_ARCHITECTURES="$CUDA_ARCH" +if [[ "$BACKEND" == "cuda" ]]; then + cmake -B "$BUILD_DIR" -S "$LLAMA_DIR" \ + -DGGML_CUDA=ON \ + -DGGML_HIP=OFF \ + -DGGML_METAL=OFF \ + -DGGML_RPC=ON \ + -DBUILD_SHARED_LIBS=OFF \ + -DLLAMA_OPENSSL=OFF \ + -DCMAKE_CUDA_ARCHITECTURES="$CUDA_ARCH" +else + if command -v hipconfig &>/dev/null; then + export HIPCXX="$(hipconfig -l)/clang" + export HIP_PATH="$(hipconfig -R)" + fi + cmake -B "$BUILD_DIR" -S "$LLAMA_DIR" \ + -DGGML_CUDA=OFF \ + -DGGML_HIP=ON \ + -DGGML_METAL=OFF \ + -DGGML_RPC=ON \ + -DBUILD_SHARED_LIBS=OFF \ + -DLLAMA_OPENSSL=OFF \ + -DAMDGPU_TARGETS="$ROCM_ARCH" +fi cmake --build "$BUILD_DIR" --config Release -j"$(nproc)" echo "llama.cpp build complete: $BUILD_DIR/bin/" diff --git a/scripts/detect-llama-device.sh b/scripts/detect-llama-device.sh new file mode 100755 index 000000000..1e78c570e --- /dev/null +++ b/scripts/detect-llama-device.sh @@ -0,0 +1,44 @@ +#!/usr/bin/env bash +# detect-llama-device.sh — pick the best llama.cpp device string for this host + +set -euo pipefail + +if [[ "$(uname -s)" == "Darwin" ]]; then + echo MTL0 + exit 0 +fi + +if command -v nvidia-smi &>/dev/null; then + if nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | grep -q '[^[:space:]]'; then + echo CUDA0 + exit 0 + fi +fi + +if command -v tegrastats &>/dev/null; then + echo CUDA0 + exit 0 +fi + +if command -v rocm-smi &>/dev/null; then + if rocm-smi --showproductname 2>/dev/null | grep -q '^GPU\['; then + echo HIP0 + exit 0 + fi +fi + +if command -v rocminfo &>/dev/null; then + if rocminfo 2>/dev/null | grep -q 'gfx'; then + echo HIP0 + exit 0 + fi +fi + +if command -v vulkaninfo &>/dev/null; then + if vulkaninfo --summary >/dev/null 2>&1; then + echo Vulkan0 + exit 0 + fi +fi + +echo CPU diff --git a/scripts/detect-rocm-arch.sh b/scripts/detect-rocm-arch.sh new file mode 100755 index 000000000..6ed0b8436 --- /dev/null +++ b/scripts/detect-rocm-arch.sh @@ -0,0 +1,77 @@ +#!/usr/bin/env bash +# detect-rocm-arch.sh — detect AMDGPU_TARGETS values for ROCm builds +# +# Outputs a semicolon-separated list of gfx targets, e.g. "gfx942;gfx90a". + +set -euo pipefail + +die() { + echo "ERROR: $*" >&2 + exit 1 +} + +ARCHES=() + +add_arch() { + local arch="$1" + [[ "$arch" =~ ^gfx[0-9a-z]+$ ]] || return 0 + for existing in "${ARCHES[@]:-}"; do + [[ "$existing" == "$arch" ]] && return 0 + done + ARCHES+=("$arch") +} + +if command -v amdgpu-arch &>/dev/null; then + while IFS= read -r arch; do + arch="${arch//[[:space:]]/}" + [[ -n "$arch" ]] && add_arch "$arch" + done < <(amdgpu-arch 2>/dev/null || true) +fi + +if [[ ${#ARCHES[@]} -eq 0 ]] && command -v rocminfo &>/dev/null; then + while IFS= read -r arch; do + arch="${arch//[[:space:]]/}" + [[ -n "$arch" ]] && add_arch "$arch" + done < <(rocminfo 2>/dev/null | grep -oE 'gfx[0-9a-z]+' || true) +fi + +if [[ ${#ARCHES[@]} -eq 0 ]] && command -v rocm-smi &>/dev/null; then + while IFS= read -r series; do + case "$series" in + *MI300X*|*MI300*) + add_arch gfx942 + ;; + *MI250*|*MI210*|*MI200*) + add_arch gfx90a + ;; + *MI100*) + add_arch gfx908 + ;; + *RX\ 7900*|*Navi31*) + add_arch gfx1100 + ;; + *RX\ 7800*|*RX\ 7700*|*Navi32*) + add_arch gfx1101 + ;; + *RX\ 7600*|*Navi33*) + add_arch gfx1102 + ;; + esac + done < <(rocm-smi --showproductname 2>/dev/null | sed -n 's/.*Card series:[[:space:]]*//p' || true) +fi + +if [[ ${#ARCHES[@]} -eq 0 ]]; then + die "Could not detect ROCm architecture automatically. +Pass the arch explicitly: + just build backend=rocm rocm_arch=gfx942 + +Common values: + gfx942 MI300X / MI300 + gfx90a MI250 / MI210 + gfx908 MI100 + gfx1100 Radeon RX 7900 / Navi31 + gfx1101 Radeon RX 7800 / RX 7700 / Navi32 + gfx1102 Radeon RX 7600 / Navi33" +fi + +(IFS=';'; echo "${ARCHES[*]}")