From cffeceb334192e8ba68909dbd7a4cbcfa2414352 Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 08:54:32 +0200 Subject: [PATCH 01/10] detect Intel Arc Battlemage GPUs and fix the SYCL stack path detect_gpu() only matched Alchemist/DG2 device IDs (0x56xx/0x569x) and additionally gated on an lspci marketing string ("Intel ... Arc"), so Battlemage cards (0xe2xx, e.g. B580/B570/Arc Pro B60; lspci reports them as "Battlemage G31 [Intel Graphics]") were silently missed and the installer fell back to CPU-only mode. Verified on real hardware: a dual Battlemage G31 host (2 x 0xe223, 24 GB each) was detected as "No GPU detected -> CPU-only tier T1". Changes: - installers/lib/detection.sh: detect Arc purely from sysfs vendor/device IDs (lspci only names the card now), add the Battlemage 0xe2xx family, and count/sum VRAM across all matching cards instead of returning on the first one (single-card behavior unchanged). - scripts/detect-hardware.sh: add Intel Arc detection to the diagnostics tool (previously Jetson/NVIDIA/AMD/Apple only), including lmem VRAM, xe/i915 driver state, multi-GPU count, and ARC/ARC_LITE tier mapping. - config/gpu-database.json: add Intel known_gpus (B60/B580/B570/BMG-G31, A770/A750/A580/A380), intel heuristic classes (>=10 GB -> ARC, 4-10 GB -> ARC_LITE), an intel bandwidth table, and an sycl default. - scripts/classify-hardware.sh: map intel/sycl backends to docker-compose.intel.yml and add sycl to the bandwidth default table. - resolve-compose-stack.sh + compose-select.sh: prefer the pinned prebuilt image overlay (docker-compose.intel.yml, same image phase 08 pulls) over docker-compose.arc.yml (oneAPI 2025.0 source build; its toolchain predates Battlemage support). arc.yml remains as fallback. - tests/test-intel-arc-detection.sh: mock-sysfs tests covering dual Battlemage, Alchemist, missing lspci/product_name/lmem, and rejection of integrated iGPUs. Registered in tests/ci-suite.txt. --- ods/config/gpu-database.json | 180 ++++++++++++++++++++++++++ ods/installers/lib/compose-select.sh | 13 +- ods/installers/lib/detection.sh | 68 ++++++---- ods/scripts/classify-hardware.sh | 4 +- ods/scripts/detect-hardware.sh | 61 +++++++++ ods/scripts/resolve-compose-stack.sh | 11 +- ods/tests/ci-suite.txt | 1 + ods/tests/test-intel-arc-detection.sh | 90 +++++++++++++ 8 files changed, 391 insertions(+), 37 deletions(-) create mode 100755 ods/tests/test-intel-arc-detection.sh diff --git a/ods/config/gpu-database.json b/ods/config/gpu-database.json index c2bb97bd8c..f0882e5c4e 100644 --- a/ods/config/gpu-database.json +++ b/ods/config/gpu-database.json @@ -287,6 +287,166 @@ "backend": "nvidia", "tier": "NV_ULTRA" } + }, + { + "id": "arc_pro_b60", + "match": { + "device_ids": ["0xe223"], + "name_patterns": ["Arc Pro B60", "Battlemage G31", "BMG-G31"] + }, + "specs": { + "label": "Intel Arc Pro B60 (Battlemage G31)", + "vendor": "intel", + "architecture": "xe2", + "memory_type": "discrete", + "memory_mb": 24576, + "memory_source": "vram", + "bandwidth_gbps": 456 + }, + "recommended": { + "backend": "intel", + "tier": "ARC" + } + }, + { + "id": "arc_b580", + "match": { + "device_ids": ["0xe20b"], + "name_patterns": ["Arc B580", "B580"] + }, + "specs": { + "label": "Intel Arc B580 (Battlemage)", + "vendor": "intel", + "architecture": "xe2", + "memory_type": "discrete", + "memory_mb": 12288, + "memory_source": "vram", + "bandwidth_gbps": 456 + }, + "recommended": { + "backend": "intel", + "tier": "ARC" + } + }, + { + "id": "arc_b570", + "match": { + "device_ids": ["0xe20c"], + "name_patterns": ["Arc B570", "B570"] + }, + "specs": { + "label": "Intel Arc B570 (Battlemage)", + "vendor": "intel", + "architecture": "xe2", + "memory_type": "discrete", + "memory_mb": 10240, + "memory_source": "vram", + "bandwidth_gbps": 380 + }, + "recommended": { + "backend": "intel", + "tier": "ARC" + } + }, + { + "id": "battlemage_g31_generic", + "match": { + "device_ids": ["0xe220", "0xe221", "0xe222", "0xe224"], + "name_patterns": ["Battlemage"] + }, + "specs": { + "label": "Intel Battlemage G31", + "vendor": "intel", + "architecture": "xe2", + "memory_type": "discrete", + "memory_mb": 24576, + "memory_source": "vram", + "bandwidth_gbps": 456 + }, + "recommended": { + "backend": "intel", + "tier": "ARC" + } + }, + { + "id": "arc_a770", + "match": { + "device_ids": ["0x56a0"], + "name_patterns": ["Arc A770", "A770"] + }, + "specs": { + "label": "Intel Arc A770 (Alchemist)", + "vendor": "intel", + "architecture": "xe", + "memory_type": "discrete", + "memory_mb": 16384, + "memory_source": "vram", + "bandwidth_gbps": 560 + }, + "recommended": { + "backend": "intel", + "tier": "ARC" + } + }, + { + "id": "arc_a750", + "match": { + "device_ids": ["0x56a1"], + "name_patterns": ["Arc A750", "A750"] + }, + "specs": { + "label": "Intel Arc A750 (Alchemist)", + "vendor": "intel", + "architecture": "xe", + "memory_type": "discrete", + "memory_mb": 8192, + "memory_source": "vram", + "bandwidth_gbps": 512 + }, + "recommended": { + "backend": "intel", + "tier": "ARC_LITE" + } + }, + { + "id": "arc_a580", + "match": { + "device_ids": ["0x56a2"], + "name_patterns": ["Arc A580", "A580"] + }, + "specs": { + "label": "Intel Arc A580 (Alchemist)", + "vendor": "intel", + "architecture": "xe", + "memory_type": "discrete", + "memory_mb": 8192, + "memory_source": "vram", + "bandwidth_gbps": 512 + }, + "recommended": { + "backend": "intel", + "tier": "ARC_LITE" + } + }, + { + "id": "arc_a380", + "match": { + "device_ids": ["0x56a5"], + "name_patterns": ["Arc A380", "A380"] + }, + "specs": { + "label": "Intel Arc A380 (Alchemist)", + "vendor": "intel", + "architecture": "xe", + "memory_type": "discrete", + "memory_mb": 6144, + "memory_source": "vram", + "bandwidth_gbps": 186 + }, + "recommended": { + "backend": "intel", + "tier": "ARC_LITE" + } } ], "known_gpu_bandwidth": { @@ -384,6 +544,15 @@ "M1 Max": 400, "M1 Pro": 200, "M1": 68 + }, + "intel": { + "Arc Pro B60": 456, + "B580": 456, + "B570": 380, + "A770": 560, + "A750": 512, + "A580": 512, + "A380": 186 } }, "heuristic_classes": [ @@ -487,6 +656,16 @@ "match": { "vendor": "apple", "memory_type": "unified", "min_ram_mb": 0 }, "recommended": { "backend": "apple", "tier": "T1" } }, + { + "id": "intel_arc", + "match": { "vendor": "intel", "memory_type": "discrete", "min_vram_mb": 10240 }, + "recommended": { "backend": "intel", "tier": "ARC" } + }, + { + "id": "intel_arc_lite", + "match": { "vendor": "intel", "memory_type": "discrete", "min_vram_mb": 4096 }, + "recommended": { "backend": "intel", "tier": "ARC_LITE" } + }, { "id": "cpu_only", "match": { "vendor": "none", "memory_type": "none", "min_ram_mb": 0 }, @@ -497,6 +676,7 @@ "bandwidth_gbps": { "cuda": 220, "rocm": 180, + "sycl": 300, "metal": 160, "cpu_x86": 70, "cpu_arm": 50 diff --git a/ods/installers/lib/compose-select.sh b/ods/installers/lib/compose-select.sh index 6ea10e3570..157a329acd 100755 --- a/ods/installers/lib/compose-select.sh +++ b/ods/installers/lib/compose-select.sh @@ -62,14 +62,15 @@ resolve_compose_config() { COMPOSE_FILE="docker-compose.amd.yml" fi elif [[ "$TIER" == "ARC" || "$TIER" == "ARC_LITE" || "$GPU_BACKEND" == "intel" || "$GPU_BACKEND" == "sycl" ]]; then - # Prefer docker-compose.arc.yml (oneAPI build-from-source) when present; - # fall back to docker-compose.intel.yml (pre-built image) if arc.yml is absent. - if [[ -f "$SCRIPT_DIR/docker-compose.base.yml" && -f "$SCRIPT_DIR/docker-compose.arc.yml" ]]; then - COMPOSE_FLAGS="-f docker-compose.base.yml -f docker-compose.arc.yml" - COMPOSE_FILE="docker-compose.arc.yml" - elif [[ -f "$SCRIPT_DIR/docker-compose.base.yml" && -f "$SCRIPT_DIR/docker-compose.intel.yml" ]]; then + # Prefer docker-compose.intel.yml (pre-built pinned llama.cpp image, + # the same image phase 08 pulls) over docker-compose.arc.yml + # (oneAPI source build, ~10-20 min and too old for Battlemage). + if [[ -f "$SCRIPT_DIR/docker-compose.base.yml" && -f "$SCRIPT_DIR/docker-compose.intel.yml" ]]; then COMPOSE_FLAGS="-f docker-compose.base.yml -f docker-compose.intel.yml" COMPOSE_FILE="docker-compose.intel.yml" + elif [[ -f "$SCRIPT_DIR/docker-compose.base.yml" && -f "$SCRIPT_DIR/docker-compose.arc.yml" ]]; then + COMPOSE_FLAGS="-f docker-compose.base.yml -f docker-compose.arc.yml" + COMPOSE_FILE="docker-compose.arc.yml" fi else if [[ -f "$SCRIPT_DIR/docker-compose.base.yml" && -f "$SCRIPT_DIR/docker-compose.nvidia.yml" ]]; then diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index bcbcd44542..fb9f529593 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -448,34 +448,50 @@ detect_gpu() { fi fi - # Try Intel Arc via lspci + sysfs - if lspci 2>/dev/null | grep -qi 'VGA.*Intel.*Arc'; then - for card_dir in "$_drm_sys"/card*/device; do - [[ -d "$card_dir" ]] || continue - local vendor device - vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue - device=$(cat "$card_dir/device" 2>/dev/null) || continue - # Intel vendor ID: 0x8086, Arc device IDs: 0x56a0-0x56c1 (Alchemist), 0x5690-0x569f (DG2) - if [[ "$vendor" == "0x8086" ]] && [[ "$device" =~ ^0x(56[a-c][0-9a-f]|569[0-9a-f])$ ]]; then - GPU_BACKEND="intel" - GPU_MEMORY_TYPE="discrete" - GPU_DEVICE_ID="$device" - GPU_COUNT=1 - # Try to get VRAM size from sysfs (lmem_total_bytes on Arc) - local vram_bytes - vram_bytes=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram_bytes=0 - GPU_VRAM=$(( vram_bytes / 1048576 )) # in MB - # Try marketing name from sysfs or lspci - if [[ -f "$card_dir/product_name" ]]; then - GPU_NAME=$(cat "$card_dir/product_name" 2>/dev/null) || GPU_NAME="Intel Arc" - else - GPU_NAME=$(lspci | grep -i 'VGA.*Intel.*Arc' | sed 's/.*: //' | head -1) - [[ -z "$GPU_NAME" ]] && GPU_NAME="Intel Arc ($GPU_DEVICE_ID)" - fi - log "GPU: $GPU_NAME (${GPU_VRAM}MB VRAM, Intel Arc)" - return 0 + # Try Intel Arc via sysfs. Detection must not gate on an lspci marketing + # string: Battlemage enumerates as e.g. "Battlemage G31 [Intel Graphics]" + # (no "Arc" in the name), and lspci may not be installed at all. + local -a _intel_dirs=() + for card_dir in "$_drm_sys"/card*/device; do + [[ -d "$card_dir" ]] || continue + local vendor device + vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue + device=$(cat "$card_dir/device" 2>/dev/null) || continue + # Intel vendor ID: 0x8086. Discrete Arc device families: + # Alchemist / DG2: 0x56a0-0x56bf, 0x5690-0x569f + # Battlemage (BMG, B570/B580/Arc Pro B-series): 0xe2xx + if [[ "$vendor" == "0x8086" ]] && [[ "$device" =~ ^0x(56[a-c][0-9a-f]|569[0-9a-f]|e2[0-9a-f]{2})$ ]]; then + _intel_dirs+=("$card_dir") + fi + done + + if [[ ${#_intel_dirs[@]} -gt 0 ]]; then + GPU_BACKEND="intel" + GPU_MEMORY_TYPE="discrete" + GPU_COUNT=${#_intel_dirs[@]} + GPU_VRAM=0 + GPU_NAME="" + for card_dir in "${_intel_dirs[@]}"; do + device=$(cat "$card_dir/device" 2>/dev/null) || device="" + [[ -z "${GPU_DEVICE_ID:-}" ]] && GPU_DEVICE_ID="$device" + # Per-GPU local memory size (lmem_total_bytes on discrete Arc) + local vram_bytes + vram_bytes=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram_bytes=0 + GPU_VRAM=$(( GPU_VRAM + vram_bytes / 1048576 )) # in MB + # Marketing name from sysfs; only the first card names the set. + if [[ -z "$GPU_NAME" && -f "$card_dir/product_name" ]]; then + GPU_NAME=$(cat "$card_dir/product_name" 2>/dev/null) || GPU_NAME="" fi done + if [[ -z "$GPU_NAME" ]] && command -v lspci >/dev/null 2>&1; then + GPU_NAME=$(lspci 2>/dev/null | grep -iE 'VGA|3D|Display' | grep -i 'intel' | sed 's/.*: //' | head -1) + fi + [[ -z "$GPU_NAME" ]] && GPU_NAME="Intel Arc (${GPU_DEVICE_ID:-unknown})" + if [[ $GPU_COUNT -gt 1 ]]; then + GPU_NAME="${GPU_NAME} × ${GPU_COUNT}" + fi + log "GPU: $GPU_NAME (${GPU_VRAM}MB VRAM, Intel Arc)" + return 0 fi # Try AMD GPUs (discrete RDNA + APU) via sysfs. An integrated GPU next to diff --git a/ods/scripts/classify-hardware.sh b/ods/scripts/classify-hardware.sh index 91d74df07d..3d0777aa0f 100755 --- a/ods/scripts/classify-hardware.sh +++ b/ods/scripts/classify-hardware.sh @@ -78,6 +78,8 @@ OVERLAY_MAP = { "amd": ["docker-compose.base.yml", "docker-compose.amd.yml"], "nvidia": ["docker-compose.base.yml", "docker-compose.nvidia.yml"], "apple": ["docker-compose.base.yml", "docker-compose.apple.yml"], + "intel": ["docker-compose.base.yml", "docker-compose.intel.yml"], + "sycl": ["docker-compose.base.yml", "docker-compose.intel.yml"], "cpu": ["docker-compose.base.yml", "docker-compose.cpu.yml"], } @@ -171,7 +173,7 @@ if bandwidth == 0 and gpu_name: if bandwidth == 0: # Fall back to default bandwidth - backend_key_map = {"nvidia": "cuda", "amd": "rocm", "apple": "metal"} + backend_key_map = {"nvidia": "cuda", "amd": "rocm", "apple": "metal", "intel": "sycl", "sycl": "sycl"} bk = backend_key_map.get(gpu_vendor, "cpu_x86") bandwidth = db.get("defaults", {}).get("bandwidth_gbps", {}).get(bk, 0) diff --git a/ods/scripts/detect-hardware.sh b/ods/scripts/detect-hardware.sh index 6e4d293209..d849bff250 100755 --- a/ods/scripts/detect-hardware.sh +++ b/ods/scripts/detect-hardware.sh @@ -290,6 +290,40 @@ detect_amd_sysfs() { return 1 } +# Detect Intel discrete Arc GPUs via sysfs +# Output: gpu_name|vram_bytes_total|count|driver_loaded|device_id +detect_intel_sysfs() { + local count=0 total_vram=0 gpu_name="" first_dev="" driver_loaded="false" + for card_dir in /sys/class/drm/card*/device; do + [[ -d "$card_dir" ]] || continue + local vendor device + vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue + device=$(cat "$card_dir/device" 2>/dev/null) || continue + # 0x8086 = Intel. Discrete Arc families only (integrated Xe iGPUs share + # system RAM and are not inference targets): + # Alchemist / DG2: 0x56a0-0x56bf, 0x5690-0x569f + # Battlemage (BMG, B570/B580/Arc Pro B-series): 0xe2xx + [[ "$vendor" == "0x8086" ]] || continue + [[ "$device" =~ ^0x(56[a-c][0-9a-f]|569[0-9a-f]|e2[0-9a-f]{2})$ ]] || continue + count=$(( count + 1 )) + [[ -z "$first_dev" ]] && first_dev="$device" + local vram + vram=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram=0 + total_vram=$(( total_vram + vram )) + if [[ -z "$gpu_name" && -f "$card_dir/product_name" ]]; then + gpu_name=$(cat "$card_dir/product_name" 2>/dev/null) || gpu_name="" + fi + done + (( count > 0 )) || return 1 + # xe drives Battlemage+; i915 drives Alchemist/DG2 + if lsmod 2>/dev/null | grep -qE '^(xe|i915) '; then + driver_loaded="true" + fi + [[ -z "$gpu_name" ]] && gpu_name="Intel Arc ($first_dev)" + (( count > 1 )) && gpu_name="${gpu_name} × ${count}" + echo "${gpu_name}|${total_vram}|${count}|${driver_loaded}|${first_dev}" +} + # Count AMD GPUs via sysfs count_amd_gpus() { local count=0 @@ -496,6 +530,8 @@ tier_description() { AP_ULTRA) echo "Apple Ultra (96GB+): high-end local profile via CPU inference in Docker" ;; AP_PRO) echo "Apple Pro (36GB+): balanced local profile via CPU inference in Docker" ;; AP_BASE) echo "Apple Base (<36GB): compact local profile via CPU inference in Docker" ;; + ARC) echo "Intel Arc (10GB+): SYCL-accelerated local profile (B570/B580/A770/Arc Pro)" ;; + ARC_LITE) echo "Intel Arc Lite (<10GB): SYCL-accelerated compact profile (A380/A750)" ;; esac } @@ -526,6 +562,8 @@ tier_model() { AP_ULTRA) echo "gemma-4-31b-it-Q4_K_M.gguf" ;; AP_PRO) echo "gemma-4-e4b-it-Q4_K_M.gguf" ;; AP_BASE) echo "gemma-4-e2b-it-Q4_K_M.gguf" ;; + ARC) echo "gemma-4-e4b-it" ;; + ARC_LITE) echo "gemma-4-e2b-it" ;; esac return fi @@ -542,6 +580,8 @@ tier_model() { AP_ULTRA) echo "qwen3-coder-next-Q4_K_M.gguf" ;; AP_PRO) echo "qwen3.5-9b-Q4_K_M.gguf" ;; AP_BASE) echo "qwen3.5-2b-Q4_K_M.gguf" ;; + ARC) echo "qwen3.5-9b" ;; + ARC_LITE) echo "qwen3.5-4b" ;; esac } @@ -682,6 +722,23 @@ main() { fi fi + # Try Intel Arc (discrete) if no NVIDIA/AMD GPU matched + if [[ -z "$gpu_name" ]]; then + local intel_out="" + if intel_out=$(detect_intel_sysfs 2>/dev/null); then + local _intel_vram _intel_driver + IFS='|' read -r gpu_name _intel_vram gpu_count _intel_driver device_id <<< "$intel_out" + gpu_vram_mb=$(( $(as_int "$_intel_vram") / 1048576 )) + gpu_type="intel" + gpu_architecture="arc" + memory_type="discrete" + driver_loaded="$_intel_driver" + if command -v vulkaninfo &>/dev/null; then + vulkaninfo --summary 2>/dev/null | grep -qi intel && vulkan_available="true" || true + fi + fi + fi + # Try Apple Silicon if macOS if [[ -z "$gpu_name" && "$os" == "macos" ]]; then local apple_out @@ -711,6 +768,10 @@ main() { local unified_gb unified_gb=$((gpu_vram_mb / 1024)) tier=$(get_apple_tier "$unified_gb") + elif [[ "$gpu_type" == "intel" ]]; then + # Discrete Arc: ARC for ≥10GB cards (B570/B580/A770, Arc Pro B-series), + # ARC_LITE for smaller cards (A380/A750). Mirrors tier-map.sh. + if (( gpu_vram_mb / 1024 >= 10 )); then tier="ARC"; else tier="ARC_LITE"; fi else tier=$(get_tier "$gpu_vram_mb") fi diff --git a/ods/scripts/resolve-compose-stack.sh b/ods/scripts/resolve-compose-stack.sh index c13c8301e3..037f9f7133 100755 --- a/ods/scripts/resolve-compose-stack.sh +++ b/ods/scripts/resolve-compose-stack.sh @@ -215,12 +215,15 @@ elif gpu_backend == "amd": resolved = ["docker-compose.base.yml", "docker-compose.amd.yml"] primary = "docker-compose.amd.yml" elif gpu_backend in ("intel", "sycl") or tier in ("ARC", "ARC_LITE"): - if existing(["docker-compose.base.yml", "docker-compose.arc.yml"]): - resolved = ["docker-compose.base.yml", "docker-compose.arc.yml"] - primary = "docker-compose.arc.yml" - elif existing(["docker-compose.base.yml", "docker-compose.intel.yml"]): + # Prefer the pre-built pinned image overlay (matches the image pulled in + # phase 08). docker-compose.arc.yml is the opt-in oneAPI source build — + # ~10-20 min build and its 2025.0 toolchain predates Battlemage support. + if existing(["docker-compose.base.yml", "docker-compose.intel.yml"]): resolved = ["docker-compose.base.yml", "docker-compose.intel.yml"] primary = "docker-compose.intel.yml" + elif existing(["docker-compose.base.yml", "docker-compose.arc.yml"]): + resolved = ["docker-compose.base.yml", "docker-compose.arc.yml"] + primary = "docker-compose.arc.yml" elif existing(["docker-compose.base.yml"]): resolved = ["docker-compose.base.yml"] primary = "docker-compose.base.yml" diff --git a/ods/tests/ci-suite.txt b/ods/tests/ci-suite.txt index b008cd4169..fc5e54cd6f 100644 --- a/ods/tests/ci-suite.txt +++ b/ods/tests/ci-suite.txt @@ -77,6 +77,7 @@ tests/test-install-menu-hermes-flag.sh tests/test-installed-feature-defaults.sh tests/test-installed-root-group-repair.sh tests/test-installer-log-safety.sh +tests/test-intel-arc-detection.sh tests/test-jetson-detection.sh tests/test-lean-linux-defaults.sh tests/test-lemonade-retirement-guard.py diff --git a/ods/tests/test-intel-arc-detection.sh b/ods/tests/test-intel-arc-detection.sh new file mode 100755 index 0000000000..98759bd5eb --- /dev/null +++ b/ods/tests/test-intel-arc-detection.sh @@ -0,0 +1,90 @@ +#!/usr/bin/env bash +# Intel discrete Arc detection must not depend on lspci marketing strings — +# Battlemage enumerates as "Battlemage G31 [Intel Graphics]" (no "Arc" in the +# name) and lspci may not be installed. Device IDs are the contract: +# Alchemist / DG2: 0x56a0-0x56bf, 0x5690-0x569f +# Battlemage (BMG): 0xe2xx (B570/B580, Arc Pro B-series, BMG-G31) +# Integrated Xe iGPUs (Meteor Lake 0x7dxx, Raptor Lake 0xa7xx, ...) share +# system RAM and must NOT be picked up as inference GPUs. +# +# Fixtures are mock /sys/class/drm trees (ODS_DRM_SYS), matching +# test-amd-igpu-dgpu-selection.sh. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +tmp="$(mktemp -d)" +trap 'rm -rf "$tmp"' EXIT + +fail() { printf '[FAIL] %s\n' "$*" >&2; exit 1; } +pass() { printf '[PASS] %s\n' "$*"; } + +# make_intel_card DRM_ROOT CARD DEVICE_ID LMEM_MB [NAME] +make_intel_card() { + local dev="$1/$2/device" + mkdir -p "$dev" + printf '0x8086\n' > "$dev/vendor" + printf '%s\n' "$3" > "$dev/device" + [[ "$4" -gt 0 ]] && printf '%s\n' "$(( $4 * 1048576 ))" > "$dev/lmem_total_bytes" + [[ -z "${5:-}" ]] || printf '%s\n' "$5" > "$dev/product_name" +} + +detect() ( + export ODS_DRM_SYS="$1" + log() { :; }; warn() { :; }; ai() { :; }; ai_ok() { :; }; ai_warn() { :; }; ai_bad() { :; } + lspci() { return 1; } + nvidia-smi() { return 1; } + SCRIPT_DIR="$ROOT" + # shellcheck source=../installers/lib/detection.sh + . "$ROOT/installers/lib/detection.sh" + detect_gpu >/dev/null + printf '%s|%s|%s|%s|%s|%s\n' "$GPU_BACKEND" "$GPU_COUNT" "$GPU_NAME" "$GPU_VRAM" "$GPU_MEMORY_TYPE" "$GPU_DEVICE_ID" +) + +# Dual Battlemage G31 (0xe223, 24 GB each): the lspci name contains no "Arc". +bmg_duo="$tmp/bmg-duo/drm" +make_intel_card "$bmg_duo" card0 0xe223 24576 "Arc Pro B60" +make_intel_card "$bmg_duo" card1 0xe223 24576 "Arc Pro B60" + +got="$(detect "$bmg_duo")" +[[ "$got" == "intel|2|Arc Pro B60 × 2|49152|discrete|0xe223" ]] \ + || fail "dual Battlemage must detect as 2x Intel Arc with summed VRAM, got: $got" +pass "Battlemage 0xe223 detected as Intel Arc, multi-GPU VRAM summed" + +# Single Alchemist A770 (0x56a0, 16 GB). +alc="$tmp/alchemist/drm" +make_intel_card "$alc" card0 0x56a0 16384 "Intel Arc A770" + +got="$(detect "$alc")" +[[ "$got" == "intel|1|Intel Arc A770|16384|discrete|0x56a0" ]] \ + || fail "Alchemist A770 must still detect, got: $got" +pass "Alchemist 0x56a0 unchanged" + +# Battlemage without product_name and without lspci: ID-only naming. +bmg_noname="$tmp/bmg-noname/drm" +make_intel_card "$bmg_noname" card0 0xe20b 12288 + +got="$(detect "$bmg_noname")" +[[ "$got" == "intel|1|Intel Arc (0xe20b)|12288|discrete|0xe20b" ]] \ + || fail "unnamed Battlemage must fall back to a device-ID name, got: $got" +pass "no lspci / no product_name still detects" + +# Battlemage on a kernel that does not expose lmem_total_bytes: still detects. +bmg_novram="$tmp/bmg-novram/drm" +make_intel_card "$bmg_novram" card0 0xe20b 0 + +got="$(detect "$bmg_novram")" +[[ "$got" == intel\|1\|*0\|discrete\|0xe20b ]] \ + || fail "missing lmem_total_bytes must not hide the GPU, got: $got" +pass "missing lmem_total_bytes still detects the card" + +# Integrated Intel GPUs must never become the inference backend. +igpu="$tmp/igpu/drm" +make_intel_card "$igpu" card0 0x7d55 0 # Meteor Lake Arc iGPU +make_intel_card "$igpu" card1 0xa7a0 0 # Raptor Lake UHD iGPU + +got="$(detect "$igpu")" +[[ "$got" == cpu\|* ]] \ + || fail "integrated Xe iGPUs must not be detected as Arc, got: $got" +pass "integrated iGPUs ignored" + +echo "All Intel Arc detection tests passed." From eee8f41ac5bc555b433430921987a81500a9d83a Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:24:15 +0200 Subject: [PATCH 02/10] xe driver (Battlemage): read VRAM from the PCI BAR aperture when lmem_total_bytes is absent MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit i915 exposes lmem_total_bytes in sysfs, but the xe driver — which binds Battlemage by default on current kernels — has no per-device VRAM file (verified on kernel 7.0, dual 0xe223). Fall back to the largest PCI BAR aperture from the device's resource file, which is the local-memory window (32 GiB BAR on a 24 GB Arc Pro B60). A card with neither source still detects with VRAM=0. --- ods/installers/lib/detection.sh | 15 ++++++++++++++- ods/scripts/detect-hardware.sh | 13 +++++++++++++ ods/tests/test-intel-arc-detection.sh | 16 +++++++++++++++- 3 files changed, 42 insertions(+), 2 deletions(-) diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index fb9f529593..25fd12c6d1 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -474,9 +474,22 @@ detect_gpu() { for card_dir in "${_intel_dirs[@]}"; do device=$(cat "$card_dir/device" 2>/dev/null) || device="" [[ -z "${GPU_DEVICE_ID:-}" ]] && GPU_DEVICE_ID="$device" - # Per-GPU local memory size (lmem_total_bytes on discrete Arc) + # Per-GPU local memory size. i915 exposes lmem_total_bytes; the + # newer xe driver (Battlemage's default) does not — use the + # largest PCI BAR aperture, which is the local-memory window. local vram_bytes vram_bytes=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram_bytes=0 + if [[ ! "$vram_bytes" =~ ^[0-9]+$ ]] || (( vram_bytes == 0 )); then + local _bar_start _bar_end _bar_max=0 + if [[ -f "$card_dir/resource" ]]; then + while read -r _bar_start _bar_end _; do + [[ "$_bar_start" == 0x* && "$_bar_end" == 0x* && "$_bar_end" != "0x0000000000000000" ]] || continue + (( _bar_end > _bar_start )) || continue + (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true + done < "$card_dir/resource" + fi + vram_bytes=$_bar_max + fi GPU_VRAM=$(( GPU_VRAM + vram_bytes / 1048576 )) # in MB # Marketing name from sysfs; only the first card names the set. if [[ -z "$GPU_NAME" && -f "$card_dir/product_name" ]]; then diff --git a/ods/scripts/detect-hardware.sh b/ods/scripts/detect-hardware.sh index d849bff250..91baa3f708 100755 --- a/ods/scripts/detect-hardware.sh +++ b/ods/scripts/detect-hardware.sh @@ -309,6 +309,19 @@ detect_intel_sysfs() { [[ -z "$first_dev" ]] && first_dev="$device" local vram vram=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram=0 + if [[ ! "$vram" =~ ^[0-9]+$ ]] || (( vram == 0 )); then + # xe (Battlemage's driver) does not expose lmem_total_bytes; the + # largest PCI BAR aperture is the local-memory window instead. + local _bar_start _bar_end _bar_max=0 + if [[ -f "$card_dir/resource" ]]; then + while read -r _bar_start _bar_end _; do + [[ "$_bar_start" == 0x* && "$_bar_end" == 0x* && "$_bar_end" != "0x0000000000000000" ]] || continue + (( _bar_end > _bar_start )) || continue + (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true + done < "$card_dir/resource" + fi + vram=$_bar_max + fi total_vram=$(( total_vram + vram )) if [[ -z "$gpu_name" && -f "$card_dir/product_name" ]]; then gpu_name=$(cat "$card_dir/product_name" 2>/dev/null) || gpu_name="" diff --git a/ods/tests/test-intel-arc-detection.sh b/ods/tests/test-intel-arc-detection.sh index 98759bd5eb..40379ff159 100755 --- a/ods/tests/test-intel-arc-detection.sh +++ b/ods/tests/test-intel-arc-detection.sh @@ -68,7 +68,21 @@ got="$(detect "$bmg_noname")" || fail "unnamed Battlemage must fall back to a device-ID name, got: $got" pass "no lspci / no product_name still detects" -# Battlemage on a kernel that does not expose lmem_total_bytes: still detects. +# Battlemage on the xe driver: no lmem_total_bytes, VRAM comes from the +# largest PCI BAR aperture (32 GiB aperture on the real BMG-G31 below). +bmg_xe="$tmp/bmg-xe/drm" +make_intel_card "$bmg_xe" card0 0xe223 0 +dev="$bmg_xe/card0/device" +printf '0x0000002800000000 0x0000002800ffffff 0x000000000014220c\n' > "$dev/resource" +printf '0x0000001800000000 0x0000001fffffffff 0x000000000014220c\n' >> "$dev/resource" +printf '0x00000000fc200000 0x00000000fc3fffff 0x0000000000046200\n' >> "$dev/resource" + +got="$(detect "$bmg_xe")" +[[ "$got" == "intel|1|Intel Arc (0xe223)|32768|discrete|0xe223" ]] \ + || fail "xe card without lmem must read VRAM from the largest BAR, got: $got" +pass "xe fallback reads VRAM from the PCI BAR aperture" + +# Battlemage with no VRAM evidence at all: still detects. bmg_novram="$tmp/bmg-novram/drm" make_intel_card "$bmg_novram" card0 0xe20b 0 From aaed4fd355e4b58e0a82731fd0398c271e8c76d2 Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:25:19 +0200 Subject: [PATCH 03/10] detect-hardware: read module list from /proc/modules, not lsmod lsmod lives in sbin, which user-level PATHs often lack (hit on the Battlemage test host), leaving driver_loaded=false while xe is bound. --- ods/scripts/detect-hardware.sh | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/ods/scripts/detect-hardware.sh b/ods/scripts/detect-hardware.sh index 91baa3f708..8f66086f0e 100755 --- a/ods/scripts/detect-hardware.sh +++ b/ods/scripts/detect-hardware.sh @@ -328,8 +328,9 @@ detect_intel_sysfs() { fi done (( count > 0 )) || return 1 - # xe drives Battlemage+; i915 drives Alchemist/DG2 - if lsmod 2>/dev/null | grep -qE '^(xe|i915) '; then + # xe drives Battlemage+; i915 drives Alchemist/DG2. /proc/modules is + # PATH-independent (lsmod lives in sbin, often missing from user PATHs). + if grep -qE '^(xe|i915) ' /proc/modules 2>/dev/null; then driver_loaded="true" fi [[ -z "$gpu_name" ]] && gpu_name="Intel Arc ($first_dev)" From 801ad91c4fb73cefb0366f29e23310def37e71fd Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:29:43 +0200 Subject: [PATCH 04/10] add Intel backend contract (pinned server-intel image) config/backends/ had amd/apple/cpu/nvidia but no intel.json, so the capability path logged 'Could not load backend contract for intel' and fell back to defaults. Uses the same pinned ghcr.io/ggml-org/llama.cpp server-intel image that docker-compose.intel.yml and phase 08 pull. --- ods/config/backends/intel.json | 14 ++++++++++++++ 1 file changed, 14 insertions(+) create mode 100644 ods/config/backends/intel.json diff --git a/ods/config/backends/intel.json b/ods/config/backends/intel.json new file mode 100644 index 0000000000..2b686df6d7 --- /dev/null +++ b/ods/config/backends/intel.json @@ -0,0 +1,14 @@ +{ + "id": "intel", + "llm_engine": "llama-server", + "service_name": "llama-server", + "public_api_port": 8080, + "public_health_url": "http://127.0.0.1:8080/health", + "provider_name": "local-llama", + "provider_url": "http://llama-server:8080/v1", + "runtime": { + "llama_server": { + "linux_image": "ghcr.io/ggml-org/llama.cpp:server-intel-b9014@sha256:9c7bbaad3663523a3deb8927d3cfbf58d33f00a7634c69843e9eeeda01568c1b" + } + } +} From accceafcd848f58d89ec6f5439962951d00e9cb5 Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:31:08 +0200 Subject: [PATCH 05/10] 02-detection: don't abort the phase when lspci can't name the Arc card MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Arc validation grep chain ran inside a command substitution without a guard, so on Battlemage (lspci name 'Battlemage G31 [Intel Graphics]' — no 'Arc'/'B###' token) the pipeline failed and set -e killed phase 02 with no error output. Guard with || true and match the Battlemage name. Also adds config/backends/intel.json, previously missing: the capability path logged 'Could not load backend contract for intel'. --- ods/installers/phases/02-detection.sh | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/ods/installers/phases/02-detection.sh b/ods/installers/phases/02-detection.sh index bd5b013ce5..7ec74e0dc3 100755 --- a/ods/installers/phases/02-detection.sh +++ b/ods/installers/phases/02-detection.sh @@ -299,11 +299,14 @@ if [[ $GPU_COUNT -gt 0 && "$GPU_BACKEND" == "intel" ]]; then # detect_gpu() already confirmed it via sysfs; this adds a human-readable log line. _arc_pci_name="" if command -v lspci &>/dev/null; then + # Battlemage's lspci name is "Battlemage G31 [Intel Graphics]" — no + # "Arc" string — and a no-match grep would otherwise abort the phase + # under set -e via the command substitution. _arc_pci_name=$(lspci 2>/dev/null \ | grep -i 'VGA\|Display\|3D' \ - | grep -i 'Intel.*Arc\|Arc.*Intel\|Intel.*A[0-9][0-9][0-9]\|Intel.*B[0-9][0-9][0-9]' \ + | grep -iE 'Intel.*Arc|Arc.*Intel|Intel.*A[0-9][0-9][0-9]|Intel.*B[0-9][0-9][0-9]|Battlemage' \ | head -1 \ - | sed 's/.*: //') + | sed 's/.*: //') || true if [[ -n "$_arc_pci_name" ]]; then ai_ok "lspci: $_arc_pci_name" else @@ -312,7 +315,7 @@ if [[ $GPU_COUNT -gt 0 && "$GPU_BACKEND" == "intel" ]]; then | grep -i 'VGA\|Display\|3D' \ | grep -i 'Intel' \ | head -1 \ - | sed 's/.*: //') + | sed 's/.*: //') || true [[ -n "$_arc_pci_name" ]] && ai_ok "lspci: $_arc_pci_name (Intel GPU)" \ || ai_warn "lspci: Intel Arc sysfs entry found but lspci VGA entry not visible — IOMMU or PCIe bridge may obscure it" fi From 35a8b715747215c7cc7581a60a66d30136352efa Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:38:37 +0200 Subject: [PATCH 06/10] emit Intel multi-GPU topology so phase 03 GPU assignment works Phase 02 only built GPU_TOPOLOGY_JSON for NVIDIA and AMD; on a multi-Arc host (verified: 2x BMG-G31) phase 03's assign_gpus.py got an empty topology and aborted with 'ERROR: no GPUs found in topology'. Adds detect_intel_topo() (same gpus[]/links[] schema as the AMD path; links stay empty since SYCL has no P2P rank and llama.cpp picks devices by ONEAPI_DEVICE_SELECTOR index) and calls it when GPU_BACKEND=intel and count>1. Shared the per-card VRAM reader (lmem_total_bytes, else largest PCI BAR) and the Arc device-ID predicate as lib functions so the topology scan and detect_gpu can't drift apart. --- ods/installers/lib/detection.sh | 86 +++++++++++++++++++++------ ods/installers/phases/02-detection.sh | 8 +++ ods/tests/test-intel-arc-detection.sh | 12 ++++ 3 files changed, 87 insertions(+), 19 deletions(-) diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index 25fd12c6d1..062ed8de0b 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -324,6 +324,68 @@ select_cpu_fallback_tier() { fi } +# Per-card local-memory size in bytes for a discrete Intel GPU. +# i915 exposes lmem_total_bytes; the newer xe driver (Battlemage's default) +# does not — use the largest PCI BAR aperture, the local-memory window. +# Echoes 0 when neither source exists. +intel_card_vram_bytes() { + local card_dir="$1" bytes + bytes=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || bytes=0 + if [[ "$bytes" =~ ^[0-9]+$ ]] && (( bytes > 0 )); then + echo "$bytes" + return 0 + fi + local _bar_start _bar_end _bar_max=0 + if [[ -f "$card_dir/resource" ]]; then + while read -r _bar_start _bar_end _; do + [[ "$_bar_start" == 0x* && "$_bar_end" == 0x* && "$_bar_end" != "0x0000000000000000" ]] || continue + (( _bar_end > _bar_start )) || continue + (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true + done < "$card_dir/resource" + fi + echo "$_bar_max" +} + +# Discrete Intel Arc device-ID check. Alchemist/DG2: 0x56xx/0x569x; +# Battlemage (BMG, B570/B580/Arc Pro B-series): 0xe2xx. +intel_is_arc_device() { + [[ "$1" =~ ^0x(56[a-c][0-9a-f]|569[0-9a-f]|e2[0-9a-f]{2})$ ]] +} + +# Emit a minimal topology JSON for multi-GPU Intel Arc systems so phase 03's +# assign_gpus.py can place services. SYCL has no P2P fabric ranking — links +# are empty; llama.cpp picks devices by ONEAPI_DEVICE_SELECTOR index anyway. +detect_intel_topo() { + local _drm_sys="${ODS_DRM_SYS:-/sys/class/drm}" + local gpus_tsv="" idx=0 + local card_dir + for card_dir in "$_drm_sys"/card*/device; do + [[ -d "$card_dir" ]] || continue + local vendor device + vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue + device=$(cat "$card_dir/device" 2>/dev/null) || continue + [[ "$vendor" == "0x8086" ]] && intel_is_arc_device "$device" || continue + local name vram_bytes vram_gb uuid + vram_bytes=$(intel_card_vram_bytes "$card_dir") + vram_gb=$(LC_ALL=C awk -v bytes="$vram_bytes" 'BEGIN { printf "%.1f", bytes / 1073741824 }') + uuid=$(readlink -f "$card_dir" 2>/dev/null | grep -oP '[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]' | tail -1) || uuid="" + [[ -z "$uuid" ]] && uuid="card${idx}" + name=$(cat "$card_dir/product_name" 2>/dev/null) || name="" + [[ -z "$name" ]] && name="Intel Arc ($device)" + gpus_tsv+="${idx} ${name} ${vram_gb} ${uuid}"$'\n' + idx=$((idx + 1)) + done + (( idx > 0 )) || { echo "{}"; return 1; } + jq -n --argjson gpus "$(printf '%s' "$gpus_tsv" | jq -Rn '[inputs | split("\t") | { + index: (.[0] | tonumber), + name: .[1], + memory_gb: (.[2] | tonumber), + uuid: .[3], + memory_type: "discrete" + }]')" \ + '{gpu_count: ($gpus | length), vendor: "intel", gpus: $gpus, links: []}' +} + detect_gpu() { GPU_BACKEND="cpu" # default to CPU-only fallback GPU_MEMORY_TYPE="none" @@ -457,10 +519,10 @@ detect_gpu() { local vendor device vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue device=$(cat "$card_dir/device" 2>/dev/null) || continue - # Intel vendor ID: 0x8086. Discrete Arc device families: - # Alchemist / DG2: 0x56a0-0x56bf, 0x5690-0x569f - # Battlemage (BMG, B570/B580/Arc Pro B-series): 0xe2xx - if [[ "$vendor" == "0x8086" ]] && [[ "$device" =~ ^0x(56[a-c][0-9a-f]|569[0-9a-f]|e2[0-9a-f]{2})$ ]]; then + # Intel vendor ID: 0x8086; discrete Arc families only — see + # intel_is_arc_device(). Integrated Xe iGPUs share system RAM and are + # not inference targets. + if [[ "$vendor" == "0x8086" ]] && intel_is_arc_device "$device"; then _intel_dirs+=("$card_dir") fi done @@ -474,22 +536,8 @@ detect_gpu() { for card_dir in "${_intel_dirs[@]}"; do device=$(cat "$card_dir/device" 2>/dev/null) || device="" [[ -z "${GPU_DEVICE_ID:-}" ]] && GPU_DEVICE_ID="$device" - # Per-GPU local memory size. i915 exposes lmem_total_bytes; the - # newer xe driver (Battlemage's default) does not — use the - # largest PCI BAR aperture, which is the local-memory window. local vram_bytes - vram_bytes=$(cat "$card_dir/lmem_total_bytes" 2>/dev/null) || vram_bytes=0 - if [[ ! "$vram_bytes" =~ ^[0-9]+$ ]] || (( vram_bytes == 0 )); then - local _bar_start _bar_end _bar_max=0 - if [[ -f "$card_dir/resource" ]]; then - while read -r _bar_start _bar_end _; do - [[ "$_bar_start" == 0x* && "$_bar_end" == 0x* && "$_bar_end" != "0x0000000000000000" ]] || continue - (( _bar_end > _bar_start )) || continue - (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true - done < "$card_dir/resource" - fi - vram_bytes=$_bar_max - fi + vram_bytes=$(intel_card_vram_bytes "$card_dir") GPU_VRAM=$(( GPU_VRAM + vram_bytes / 1048576 )) # in MB # Marketing name from sysfs; only the first card names the set. if [[ -z "$GPU_NAME" && -f "$card_dir/product_name" ]]; then diff --git a/ods/installers/phases/02-detection.sh b/ods/installers/phases/02-detection.sh index 7ec74e0dc3..dc571ec6e9 100755 --- a/ods/installers/phases/02-detection.sh +++ b/ods/installers/phases/02-detection.sh @@ -382,6 +382,14 @@ if [[ $GPU_COUNT -gt 0 && "$GPU_BACKEND" == "intel" ]]; then _arc_vram_gb=$((GPU_VRAM / 1024)) ai_ok "Intel Arc detected: $GPU_NAME (${_arc_vram_gb} GB VRAM, device ${GPU_DEVICE_ID:-unknown})" log "Intel Arc backend: GPU_BACKEND=intel, VRAM=${GPU_VRAM}MB, Level Zero=${_level_zero_ok}" + + # Multi-GPU: emit a minimal topology so phase 03's assignment step has a + # GPU list to work with. SYCL has no P2P fabric ranking — links stay empty. + if [[ $GPU_COUNT -gt 1 ]] && declare -F detect_intel_topo >/dev/null 2>&1; then + GPU_TOPOLOGY_JSON=$(detect_intel_topo 2>>"$LOG_FILE") || GPU_TOPOLOGY_JSON="{}" + GPU_TOTAL_VRAM=$GPU_VRAM + log "Intel topology: $(echo "$GPU_TOPOLOGY_JSON" | jq -r '.gpu_count // 0') GPU(s)" + fi fi # ----------------------------------------------------------------------------- diff --git a/ods/tests/test-intel-arc-detection.sh b/ods/tests/test-intel-arc-detection.sh index 40379ff159..b2d9e11dbe 100755 --- a/ods/tests/test-intel-arc-detection.sh +++ b/ods/tests/test-intel-arc-detection.sh @@ -101,4 +101,16 @@ got="$(detect "$igpu")" || fail "integrated Xe iGPUs must not be detected as Arc, got: $got" pass "integrated iGPUs ignored" +# Multi-GPU topology JSON for phase 03's assign_gpus.py (mock sysfs). +topo="$(export ODS_DRM_SYS="$bmg_duo" + log() { :; }; warn() { :; }; ai() { :; }; ai_ok() { :; }; ai_warn() { :; }; ai_bad() { :; } + SCRIPT_DIR="$ROOT" + . "$ROOT/installers/lib/detection.sh" + detect_intel_topo)" +got="$(jq -r '[.vendor, .gpu_count, (.links | length)] | join("|")' <<<"$topo")" +[[ "$got" == "intel|2|0" ]] || fail "intel topology must list both GPUs, got: $got" +got="$(jq -r '[.gpus[].memory_gb] | join(",")' <<<"$topo")" +[[ "$got" == "24,24" ]] || fail "topology VRAM per card, got: $got" +pass "intel multi-GPU topology feeds assign_gpus.py" + echo "All Intel Arc detection tests passed." From 3eecdb89d523c1be649a7a6d5262230a342ebc2c Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 10:43:06 +0200 Subject: [PATCH 07/10] 02-detection: place Intel topology after GPU_TOPOLOGY_JSON init The first version assigned GPU_TOPOLOGY_JSON before the variable's '{}' initialization later in the phase, so the topology was discarded and phase 03 still saw gpu_count=0. --- ods/installers/phases/02-detection.sh | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/ods/installers/phases/02-detection.sh b/ods/installers/phases/02-detection.sh index dc571ec6e9..832c9c478a 100755 --- a/ods/installers/phases/02-detection.sh +++ b/ods/installers/phases/02-detection.sh @@ -382,14 +382,6 @@ if [[ $GPU_COUNT -gt 0 && "$GPU_BACKEND" == "intel" ]]; then _arc_vram_gb=$((GPU_VRAM / 1024)) ai_ok "Intel Arc detected: $GPU_NAME (${_arc_vram_gb} GB VRAM, device ${GPU_DEVICE_ID:-unknown})" log "Intel Arc backend: GPU_BACKEND=intel, VRAM=${GPU_VRAM}MB, Level Zero=${_level_zero_ok}" - - # Multi-GPU: emit a minimal topology so phase 03's assignment step has a - # GPU list to work with. SYCL has no P2P fabric ranking — links stay empty. - if [[ $GPU_COUNT -gt 1 ]] && declare -F detect_intel_topo >/dev/null 2>&1; then - GPU_TOPOLOGY_JSON=$(detect_intel_topo 2>>"$LOG_FILE") || GPU_TOPOLOGY_JSON="{}" - GPU_TOTAL_VRAM=$GPU_VRAM - log "Intel topology: $(echo "$GPU_TOPOLOGY_JSON" | jq -r '.gpu_count // 0') GPU(s)" - fi fi # ----------------------------------------------------------------------------- @@ -398,6 +390,16 @@ fi GPU_TOPOLOGY_JSON="{}" GPU_HAS_NVLINK="false" GPU_TOTAL_VRAM=0 + +# Intel Arc multi-GPU: emit a minimal topology so phase 03's assignment step +# has a GPU list to work with. SYCL has no P2P fabric ranking — links stay +# empty; llama.cpp picks devices by ONEAPI_DEVICE_SELECTOR index anyway. +if [[ $GPU_COUNT -gt 1 && "$GPU_BACKEND" == "intel" ]] && declare -F detect_intel_topo >/dev/null 2>&1; then + GPU_TOPOLOGY_JSON=$(detect_intel_topo 2>>"$LOG_FILE") || GPU_TOPOLOGY_JSON="{}" + GPU_TOTAL_VRAM=$GPU_VRAM + log "Intel topology: $(echo "$GPU_TOPOLOGY_JSON" | jq -r '.gpu_count // 0') GPU(s)" +fi + if [[ $GPU_COUNT -gt 1 && "$GPU_BACKEND" == "nvidia" ]]; then ai "Detecting multi-GPU topology..." if [[ -f "$SCRIPT_DIR/installers/lib/nvidia-topo.sh" ]]; then From 738595719b62d9b0ca3e47475f978b3b689bf6de Mon Sep 17 00:00:00 2001 From: SergiioB Date: Wed, 7 Oct 2026 11:03:45 +0200 Subject: [PATCH 08/10] preflight + config renderer: recognize intel/sycl backend and ARC tier minimums - render-runtime-configs.py rejected --gpu-backend sycl (argparse choices lacked intel/sycl), aborting phase 06 on every Arc host. - preflight-engine had no ARC/ARC_LITE entries in the tier rank, RAM, or disk maps, so an Arc host inherited the tier-2 50GB disk floor even though the SYCL model is ~3-6GB. ARC -> 16GB RAM / 30GB disk, ARC_LITE -> 8GB / 20GB. - Preflight treated intel/sycl as 'unknown backend' warning; now a pass. Also scale the xe BAR-aperture VRAM fallback by 3/4: the aperture is the next power of two above real VRAM (32 GiB BAR on a 24 GB Arc Pro B60), so the raw value over-reports and inflates model-fit checks. --- ods/installers/lib/detection.sh | 7 ++++++- ods/scripts/detect-hardware.sh | 6 +++++- ods/scripts/preflight-engine.sh | 15 +++++++++++++++ ods/scripts/render-runtime-configs.py | 3 ++- ods/tests/test-intel-arc-detection.sh | 4 ++-- 5 files changed, 30 insertions(+), 5 deletions(-) diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index 062ed8de0b..262e1ea183 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -343,7 +343,12 @@ intel_card_vram_bytes() { (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true done < "$card_dir/resource" fi - echo "$_bar_max" + # The BAR aperture is sized to the next power of two above real VRAM + # (32 GiB BAR on a 24 GB Arc Pro B60, 16 GiB on a 12 GB B580). Scaling by + # 3/4 lands on the exact VRAM for every shipping BMG part and stays + # conservative — over-reporting VRAM makes model selection promise more + # than the card can hold. + echo $(( _bar_max * 3 / 4 )) } # Discrete Intel Arc device-ID check. Alchemist/DG2: 0x56xx/0x569x; diff --git a/ods/scripts/detect-hardware.sh b/ods/scripts/detect-hardware.sh index 8f66086f0e..56dfb40407 100755 --- a/ods/scripts/detect-hardware.sh +++ b/ods/scripts/detect-hardware.sh @@ -312,6 +312,10 @@ detect_intel_sysfs() { if [[ ! "$vram" =~ ^[0-9]+$ ]] || (( vram == 0 )); then # xe (Battlemage's driver) does not expose lmem_total_bytes; the # largest PCI BAR aperture is the local-memory window instead. + # The BAR is sized to the next power of two above real VRAM + # (32 GiB on a 24 GB Arc Pro B60, 16 GiB on a 12 GB B580), so + # scale by 3/4 — over-reporting VRAM makes model selection + # promise more than the card can hold. local _bar_start _bar_end _bar_max=0 if [[ -f "$card_dir/resource" ]]; then while read -r _bar_start _bar_end _; do @@ -320,7 +324,7 @@ detect_intel_sysfs() { (( _bar_end - _bar_start + 1 > _bar_max )) && _bar_max=$(( _bar_end - _bar_start + 1 )) || true done < "$card_dir/resource" fi - vram=$_bar_max + vram=$(( _bar_max * 3 / 4 )) fi total_vram=$(( total_vram + vram )) if [[ -z "$gpu_name" && -f "$card_dir/product_name" ]]; then diff --git a/ods/scripts/preflight-engine.sh b/ods/scripts/preflight-engine.sh index 76cbaa1fd0..a39f4d17b1 100755 --- a/ods/scripts/preflight-engine.sh +++ b/ods/scripts/preflight-engine.sh @@ -151,6 +151,8 @@ tier_rank_map = { "T2": 2, "T3": 3, "T4": 4, + "ARC": 2, + "ARC_LITE": 1, "SH_COMPACT": 3, "SH_LARGE": 4, } @@ -168,6 +170,8 @@ min_ram_map = { "T3": 48, "4": 64, "T4": 64, + "ARC": 16, + "ARC_LITE": 8, "SH_COMPACT": 64, "SH_LARGE": 96, } @@ -186,6 +190,10 @@ min_disk_map = { "T3": 80, "4": 150, "T4": 150, + # ARC tiers install a ~3-6GB SYCL model on top of the core images; the + # generic tier-2 minimum (50GB) oversizes them. + "ARC": 30, + "ARC_LITE": 20, "SH_COMPACT": 80, "SH_LARGE": 120, } @@ -373,6 +381,13 @@ elif gpu_backend == "nvidia": f"NVIDIA backend selected ({gpu_name}, {gpu_vram_mb}MB VRAM).", "", ) +elif gpu_backend in {"intel", "sycl"}: + add_check( + "gpu-backend", + "pass", + f"Intel Arc backend selected ({gpu_name}, {gpu_vram_mb}MB VRAM).", + "", + ) elif gpu_backend == "apple": add_check( "gpu-backend", diff --git a/ods/scripts/render-runtime-configs.py b/ods/scripts/render-runtime-configs.py index 36d0f65a54..69e7c27a49 100644 --- a/ods/scripts/render-runtime-configs.py +++ b/ods/scripts/render-runtime-configs.py @@ -732,7 +732,8 @@ def parse_args(argv: list[str]) -> argparse.Namespace: # from before round F still passes them on every render (R13). parser.add_argument("--lemonade-model-id", default="", help=argparse.SUPPRESS) parser.add_argument("--lemonade-api-base", default="", help=argparse.SUPPRESS) - parser.add_argument("--gpu-backend", choices=["amd", "apple", "cpu", "nvidia"], default="nvidia") + # intel = detection value, sycl = tier-map value; both mean llama.cpp SYCL + parser.add_argument("--gpu-backend", choices=["amd", "apple", "cpu", "intel", "nvidia", "sycl"], default="nvidia") parser.add_argument( "--ods-mode", choices=["local", "cloud", "hybrid", *LEGACY_ODS_MODES], default="local", ) diff --git a/ods/tests/test-intel-arc-detection.sh b/ods/tests/test-intel-arc-detection.sh index b2d9e11dbe..efa0e93f67 100755 --- a/ods/tests/test-intel-arc-detection.sh +++ b/ods/tests/test-intel-arc-detection.sh @@ -78,8 +78,8 @@ printf '0x0000001800000000 0x0000001fffffffff 0x000000000014220c\n' >> "$dev/res printf '0x00000000fc200000 0x00000000fc3fffff 0x0000000000046200\n' >> "$dev/resource" got="$(detect "$bmg_xe")" -[[ "$got" == "intel|1|Intel Arc (0xe223)|32768|discrete|0xe223" ]] \ - || fail "xe card without lmem must read VRAM from the largest BAR, got: $got" +[[ "$got" == "intel|1|Intel Arc (0xe223)|24576|discrete|0xe223" ]] \ + || fail "xe card without lmem must read VRAM from the largest BAR (scaled), got: $got" pass "xe fallback reads VRAM from the PCI BAR aperture" # Battlemage with no VRAM evidence at all: still detects. From 7c92215ba5a7da9a3969a856ff9c49049a416b91 Mon Sep 17 00:00:00 2001 From: SergiioB Date: Thu, 8 Oct 2026 19:05:41 +0200 Subject: [PATCH 09/10] detection: emit integer memory_gb in intel topology when VRAM is whole The awk "%.1f" format always emitted a decimal (24.0); assign_gpus.py consumers and the detection test expect integral GB values as integers. Print %d when the GiB value is whole, %.1f otherwise. --- ods/installers/lib/detection.sh | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index 262e1ea183..fc7d434957 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -372,7 +372,10 @@ detect_intel_topo() { [[ "$vendor" == "0x8086" ]] && intel_is_arc_device "$device" || continue local name vram_bytes vram_gb uuid vram_bytes=$(intel_card_vram_bytes "$card_dir") - vram_gb=$(LC_ALL=C awk -v bytes="$vram_bytes" 'BEGIN { printf "%.1f", bytes / 1073741824 }') + vram_gb=$(LC_ALL=C awk -v bytes="$vram_bytes" 'BEGIN { + gb = bytes / 1073741824 + if (gb == int(gb)) printf "%d", gb; else printf "%.1f", gb + }') uuid=$(readlink -f "$card_dir" 2>/dev/null | grep -oP '[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-9]' | tail -1) || uuid="" [[ -z "$uuid" ]] && uuid="card${idx}" name=$(cat "$card_dir/product_name" 2>/dev/null) || name="" From d5e35d88835799c9ef2e36ed82f920019b565c8f Mon Sep 17 00:00:00 2001 From: Mike Bradley Date: Thu, 8 Oct 2026 14:06:50 -0700 Subject: [PATCH 10/10] fix(intel): bound B570 memory and tolerate optional PCI names --- ods/installers/lib/detection.sh | 16 +++++++++----- ods/scripts/detect-hardware.sh | 10 ++++++--- ods/tests/test-intel-arc-detection.sh | 30 +++++++++++++++++++++++++++ 3 files changed, 48 insertions(+), 8 deletions(-) diff --git a/ods/installers/lib/detection.sh b/ods/installers/lib/detection.sh index fc7d434957..6c186e14b0 100755 --- a/ods/installers/lib/detection.sh +++ b/ods/installers/lib/detection.sh @@ -345,10 +345,16 @@ intel_card_vram_bytes() { fi # The BAR aperture is sized to the next power of two above real VRAM # (32 GiB BAR on a 24 GB Arc Pro B60, 16 GiB on a 12 GB B580). Scaling by - # 3/4 lands on the exact VRAM for every shipping BMG part and stays - # conservative — over-reporting VRAM makes model selection promise more - # than the card can hold. - echo $(( _bar_max * 3 / 4 )) + # 3/4 matches those cards, but B570 has only 10 GiB behind the same + # 16 GiB aperture as B580. Cap that known SKU so model placement cannot + # spend the extra 2 GiB of address space as physical memory. + bytes=$(( _bar_max * 3 / 4 )) + local device + device=$(cat "$card_dir/device" 2>/dev/null) || device="" + if [[ "$device" == "0xe20c" ]] && (( bytes > 10737418240 )); then + bytes=10737418240 + fi + echo "$bytes" } # Discrete Intel Arc device-ID check. Alchemist/DG2: 0x56xx/0x569x; @@ -553,7 +559,7 @@ detect_gpu() { fi done if [[ -z "$GPU_NAME" ]] && command -v lspci >/dev/null 2>&1; then - GPU_NAME=$(lspci 2>/dev/null | grep -iE 'VGA|3D|Display' | grep -i 'intel' | sed 's/.*: //' | head -1) + GPU_NAME=$(lspci 2>/dev/null | grep -iE 'VGA|3D|Display' | grep -i 'intel' | sed 's/.*: //' | head -1) || true fi [[ -z "$GPU_NAME" ]] && GPU_NAME="Intel Arc (${GPU_DEVICE_ID:-unknown})" if [[ $GPU_COUNT -gt 1 ]]; then diff --git a/ods/scripts/detect-hardware.sh b/ods/scripts/detect-hardware.sh index 56dfb40407..55c695e911 100755 --- a/ods/scripts/detect-hardware.sh +++ b/ods/scripts/detect-hardware.sh @@ -294,7 +294,8 @@ detect_amd_sysfs() { # Output: gpu_name|vram_bytes_total|count|driver_loaded|device_id detect_intel_sysfs() { local count=0 total_vram=0 gpu_name="" first_dev="" driver_loaded="false" - for card_dir in /sys/class/drm/card*/device; do + local drm_root="${1:-/sys/class/drm}" + for card_dir in "$drm_root"/card*/device; do [[ -d "$card_dir" ]] || continue local vendor device vendor=$(cat "$card_dir/vendor" 2>/dev/null) || continue @@ -314,8 +315,8 @@ detect_intel_sysfs() { # largest PCI BAR aperture is the local-memory window instead. # The BAR is sized to the next power of two above real VRAM # (32 GiB on a 24 GB Arc Pro B60, 16 GiB on a 12 GB B580), so - # scale by 3/4 — over-reporting VRAM makes model selection - # promise more than the card can hold. + # scale by 3/4, capped below for the 10 GiB B570, which shares + # B580's 16 GiB aperture without its 12 GiB physical memory. local _bar_start _bar_end _bar_max=0 if [[ -f "$card_dir/resource" ]]; then while read -r _bar_start _bar_end _; do @@ -325,6 +326,9 @@ detect_intel_sysfs() { done < "$card_dir/resource" fi vram=$(( _bar_max * 3 / 4 )) + if [[ "$device" == "0xe20c" ]] && (( vram > 10737418240 )); then + vram=10737418240 + fi fi total_vram=$(( total_vram + vram )) if [[ -z "$gpu_name" && -f "$card_dir/product_name" ]]; then diff --git a/ods/tests/test-intel-arc-detection.sh b/ods/tests/test-intel-arc-detection.sh index efa0e93f67..70888aac45 100755 --- a/ods/tests/test-intel-arc-detection.sh +++ b/ods/tests/test-intel-arc-detection.sh @@ -82,6 +82,36 @@ got="$(detect "$bmg_xe")" || fail "xe card without lmem must read VRAM from the largest BAR (scaled), got: $got" pass "xe fallback reads VRAM from the PCI BAR aperture" +# B570 has 10 GiB, although its 16 GiB BAR is the same size as B580's. +# Address-space padding must never become available model memory. +b570="$tmp/b570/drm" +make_intel_card "$b570" card0 0xe20c 0 "Intel Arc B570" +printf '0x0000001000000000 0x00000013ffffffff 0x000000000014220c\n' > "$b570/card0/device/resource" +got="$(detect "$b570")" +[[ "$got" == "intel|1|Intel Arc B570|10240|discrete|0xe20c" ]] \ + || fail "B570 BAR must not inflate 10 GiB VRAM, got: $got" +portable="$(source "$ROOT/scripts/detect-hardware.sh"; detect_intel_sysfs "$b570")" +IFS='|' read -r _name bytes count _driver _device <<< "$portable" +[[ "$bytes" == 10737418240 && "$count" == 1 ]] \ + || fail "portable detector must report the same B570 capacity, got: $portable" +pass "both detectors cap B570 aperture padding at physical VRAM" + +# The installer calls detect_gpu directly with errexit. A substitution around +# the function masks that behavior, so exercise it in a fresh strict shell. +rm "$b570/card0/device/product_name" +for lspci_status in 0 1; do + got="$(ODS_DRM_SYS="$b570" SCRIPT_DIR="$ROOT" LSPCI_STATUS="$lspci_status" \ + bash -c 'set -euo pipefail + log(){ :; }; warn(){ :; } + nvidia-smi(){ return 1; }; lspci(){ return "$LSPCI_STATUS"; } + source "$SCRIPT_DIR/installers/lib/detection.sh" + detect_gpu >/dev/null + printf "%s|%s" "$GPU_BACKEND" "$GPU_NAME"')" \ + || fail "optional lspci name lookup aborted strict-shell detection" + [[ "$got" == "intel|Intel Arc (0xe20c)" ]] || fail "ID fallback lost: $got" +done +pass "strict installer survives failed or empty lspci naming" + # Battlemage with no VRAM evidence at all: still detects. bmg_novram="$tmp/bmg-novram/drm" make_intel_card "$bmg_novram" card0 0xe20b 0