58 lines
1.8 KiB
YAML
58 lines
1.8 KiB
YAML
# ODS — Intel Arc GPU Overlay (SYCL backend)
|
|
# Requires: Intel Arc GPU with oneAPI Level Zero runtime on host
|
|
# Use with: docker compose -f docker-compose.base.yml -f docker-compose.intel.yml up -d
|
|
#
|
|
# Supported hardware:
|
|
# ARC tier — Arc A770 16 GB (and future Arc B-series ≥12 GB)
|
|
# ARC_LITE tier — Arc A750 8 GB, A380 6 GB
|
|
#
|
|
# Host prerequisites:
|
|
# apt install intel-opencl-icd intel-level-zero-gpu level-zero
|
|
# usermod -aG video,render $USER (then re-login)
|
|
# # Verify: clinfo | grep -i "intel arc"
|
|
|
|
services:
|
|
llama-server:
|
|
image: ${LLAMA_SERVER_IMAGE:-ghcr.io/ggml-org/llama.cpp:server-intel-b8248}
|
|
devices:
|
|
- /dev/dri:/dev/dri
|
|
group_add:
|
|
- "${VIDEO_GID:-44}"
|
|
- "${RENDER_GID:-992}"
|
|
environment:
|
|
# Level Zero selects the Intel Arc GPU; persistent cache avoids
|
|
# recompiling SYCL kernels on every container start.
|
|
- ONEAPI_DEVICE_SELECTOR=level_zero:gpu
|
|
- SYCL_CACHE_PERSISTENT=1
|
|
# Enable Intel GPU System Management Interface for telemetry
|
|
- ZES_ENABLE_SYSMAN=1
|
|
command:
|
|
- --model
|
|
- /models/${GGUF_FILE:-Qwen3.5-9B-Q4_K_M.gguf}
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- "8080"
|
|
- --n-gpu-layers
|
|
- "${N_GPU_LAYERS:-99}"
|
|
- --ctx-size
|
|
- "${CTX_SIZE:-32768}"
|
|
- --metrics
|
|
deploy:
|
|
resources:
|
|
limits:
|
|
cpus: '${LLAMA_CPU_LIMIT:-16.0}'
|
|
memory: ${LLAMA_SERVER_MEMORY_LIMIT:-24G}
|
|
reservations:
|
|
cpus: '${LLAMA_CPU_RESERVATION:-2.0}'
|
|
memory: 4G
|
|
|
|
dashboard-api:
|
|
environment:
|
|
# Hard-code sycl backend so the dashboard uses Intel sysfs GPU detection
|
|
# regardless of .env state.
|
|
- GPU_BACKEND=sycl
|
|
volumes:
|
|
- /sys/class/drm:/sys/class/drm:ro
|
|
- /sys/class/hwmon:/sys/class/hwmon:ro
|