diff --git a/docker/docker-xpu/Dockerfile b/docker/docker-xpu/Dockerfile new file mode 100644 index 000000000..c5010f3dd --- /dev/null +++ b/docker/docker-xpu/Dockerfile @@ -0,0 +1,88 @@ +# https://hub.docker.com/r/intel/deep-learning-essentials/tags +# Unlike the CUDA/ROCm bases this image ships no PyTorch, and its version must match the +# host Intel GPU driver — see README.md. + +ARG BASE_IMAGE=intel/deep-learning-essentials:2026.1.0-devel-ubuntu24.04 +FROM ${BASE_IMAGE} + +# Installation arguments +ARG PIP_INDEX=https://pypi.org/simple +ARG PYTORCH_INDEX=https://download.pytorch.org/whl/xpu +# Must stay in the same compute-runtime series as the base image — see README.md ("ocloc"). +ARG OCLOC_VERSION=26.18.38308.1 + +# Define environments +ENV DEBIAN_FRONTEND=noninteractive +ENV PIP_ROOT_USER_ACTION=ignore +# The base image's Python is distro-managed (PEP 668); nothing to protect in a container. +ENV PIP_BREAK_SYSTEM_PACKAGES=1 +# expandable_segments trips a Level-Zero bug on XPU — see README.md ("Multi-GPU and OPTIM_TORCH"). +ENV OPTIM_TORCH=0 + +# Use Bash instead of default /bin/sh +SHELL ["/bin/bash", "-c"] + +# Set the working directory +WORKDIR /app + +# Install pip, Python dev headers and a C/C++ toolchain — see README.md ("Python.h: No such file or directory"). +RUN PY_MM="$(python3 -c 'import sys; print(f"{sys.version_info.major}.{sys.version_info.minor}")')" && \ + apt-get update -y && \ + apt-get install -y --no-install-recommends \ + "python${PY_MM}-dev" python3-pip build-essential && \ + apt-get clean && rm -rf /var/lib/apt/lists/* + +# Install ocloc — pinned .deb rather than apt, see README.md ("ocloc and torch.compile"). +RUN wget -q "https://github.com/intel/compute-runtime/releases/download/${OCLOC_VERSION}/intel-ocloc_${OCLOC_VERSION}-0_amd64.deb" \ + -O /tmp/intel-ocloc.deb && \ + dpkg -i /tmp/intel-ocloc.deb && \ + rm -f /tmp/intel-ocloc.deb + +# Change pip source — pip is deliberately not upgraded, see README.md ("How pip is installed"). +RUN pip config set global.index-url "${PIP_INDEX}" && \ + pip config set global.extra-index-url "${PIP_INDEX}" && \ + pip install --no-cache-dir packaging wheel setuptools editables "hatchling>=1.18.0" + +# Copy the application into the image +COPY . /app + +# Install PyTorch for XPU — all three wheels pinned as a set, see README.md ("Pinned torch stack") +RUN pip install --no-cache-dir -r requirements/xpu.txt --index-url "${PYTORCH_INDEX}" + +# Install LLaMA Factory +# metrics.txt is explicit because this project defines no extras, so `.[metrics]` would no-op. +RUN pip install --no-cache-dir -e . --no-build-isolation && \ + pip install --no-cache-dir -r requirements/metrics.txt + +# Optional accelerators — see README.md ("Optional accelerators"). +# py-cpuinfo must land first: deepspeed's own setup.py imports deepspeed/ops/adam/cpu_adam.py +# during metadata generation (before pip installs deepspeed's declared deps), which needs it. +# deepspeed is pinned separately from requirements/deepspeed.txt's <=0.18.4: that range's XPU +# accelerator imports IPEX's DpcppBuildExtension unconditionally, which no longer exists on a +# post-IPEX torch. Fixed upstream in 0.18.7 (plain torch.utils.cpp_extension.BuildExtension). +RUN pip install --no-cache-dir py-cpuinfo && \ + (DS_BUILD_OPS=0 pip install --no-cache-dir "deepspeed==0.19.6" \ + || echo "WARNING: deepspeed install failed - deepspeed training unavailable in this image") +RUN pip install --no-cache-dir -r requirements/bitsandbytes.txt \ + || echo "WARNING: bitsandbytes install failed - 4-bit quantization (QLoRA) unavailable in this image" + +# Source the oneAPI environment in every interactive shell +RUN echo "source /opt/intel/oneapi/setvars.sh --force" >> /root/.bashrc + +# Set up volumes +# VOLUME [ "/root/.cache/huggingface", "/app/shared_data", "/app/output" ] + +# Expose port 7860 for LLaMA Board +ENV GRADIO_SERVER_PORT=7860 +EXPOSE 7860 + +# Expose port 8000 for API service +ENV API_PORT=8000 +EXPOSE 8000 + +# Reset pip config +RUN pip config unset global.index-url && \ + pip config unset global.extra-index-url + +# oneAPI must be on the loader path before torch can open its XPU backend +CMD ["bash", "-c", "source /opt/intel/oneapi/setvars.sh --force && exec llamafactory-cli webui"] diff --git a/docker/docker-xpu/README.md b/docker/docker-xpu/README.md new file mode 100644 index 000000000..4f3d58845 --- /dev/null +++ b/docker/docker-xpu/README.md @@ -0,0 +1,206 @@ +# Docker Setup for Intel GPUs + +This directory contains Docker configuration files for running LLaMA Factory with Intel GPU (XPU) support. + +## Image Details + +| Component | Version | +|---|---| +| Base OS | Ubuntu 24.04 LTS (x86_64) | +| Intel DLE base | [intel/deep-learning-essentials:2026.1.0-devel-ubuntu24.04](https://hub.docker.com/r/intel/deep-learning-essentials) | +| Python | 3.12 | +| PyTorch | 2.13.0+xpu | +| Intel GPU runtime | Bundled inside DLE (libze-intel-gpu 26.18.x, oneAPI 2026.1) | + +The Intel compute runtime is bundled inside the DLE base image — no GPU compute packages are needed on the host, only the kernel driver. The bundled runtime must be compatible with the host kernel driver; verify with `dpkg -l libze-intel-gpu1 | grep -oP '\d+\.\d+\.\d+'`. + +## Prerequisites + +### 1. Docker & Docker Compose + +```bash +# Ubuntu/Debian +sudo apt-get update && sudo apt-get install docker.io docker-compose-v2 +``` + +See the [official Docker install docs](https://docs.docker.com/engine/install/) for other distros or newer versions. + +### 2. Intel GPU Kernel Driver (host only) + +```bash +# Add the Intel GPU PPA +sudo apt-get install -y gpg-agent wget +wget -qO - https://repositories.intel.com/gpu/intel-graphics.key | \ + sudo gpg --dearmor -o /usr/share/keyrings/intel-graphics.gpg +echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.gpg] \ + https://repositories.intel.com/gpu/ubuntu noble client" | \ + sudo tee /etc/apt/sources.list.d/intel-graphics.list +sudo apt-get update + +# Kernel driver only — no compute runtime packages needed on the host +sudo apt-get install -y intel-i915-dkms intel-fw-gpu +sudo reboot +``` + +After reboot, verify `/dev/dri` is populated: `ls /dev/dri/` (expect `card0`, `renderD128`, etc). + +See the [Intel GPU Installation Guide](https://dgpu-docs.intel.com/installation-guides/installing-packages-from-the-intel-ppa.html) for details. + +> [!IMPORTANT] +> Enable **Resizable BAR** in your system BIOS before proceeding, or you may see `Bus error (core dumped)` or degraded performance. See [Intel's guide](https://www.intel.com/content/www/us/en/support/articles/000090831/graphics.html). + +### 3. Add your user to the GPU groups + +```bash +sudo usermod -aG render,video $USER +# Log out and back in for group membership to take effect +``` + +(Optional) sanity-check the host side with `sudo apt-get install -y clinfo && clinfo --list | grep Device`. The check that actually matters — `torch.xpu.device_count()` — runs inside the container, in Usage below. + +## Usage + +### Using Docker Compose (Recommended) + +```bash +cd docker/docker-xpu/ +docker compose up -d +docker compose exec llamafactory bash +``` + +Verify GPU access inside the container: + +```bash +python3 -c "import torch; print(torch.xpu.device_count(), 'XPU device(s) found')" +``` + +### Using Docker Run + +```bash +# Build the image (from the repo root) +docker build -t llamafactory:xpu -f docker/docker-xpu/Dockerfile . + +# Run the container +docker run -it --rm \ + --device /dev/dri \ + -v /dev/dri/by-path:/dev/dri/by-path \ + --group-add $(getent group render | cut -d: -f3) \ + --group-add $(getent group video | cut -d: -f3) \ + --ipc=host \ + -p 7860:7860 \ + -p 8000:8000 \ + -v ~/.cache/huggingface:/root/.cache/huggingface \ + --name llamafactory \ + llamafactory:xpu bash +``` + +## Build arguments + +| Argument | Default | Purpose | +|---|---|---| +| `BASE_IMAGE` | `intel/deep-learning-essentials:2026.1.0-devel-ubuntu24.04` | oneAPI / GPU runtime version | +| `PIP_INDEX` | `https://pypi.org/simple` | PyPI mirror for everything except the torch wheels | +| `PYTORCH_INDEX` | `https://download.pytorch.org/whl/xpu` | where the `+xpu` torch wheels come from | +| `OCLOC_VERSION` | `26.18.38308.1` | pinned `intel-ocloc` build — must match `BASE_IMAGE`'s compute-runtime series | + +```bash +docker build -t llamafactory:xpu -f docker/docker-xpu/Dockerfile . \ + --build-arg PIP_INDEX=https://pypi.org/simple +``` + +## Design notes + +Why the image is built the way it is — skip to [Troubleshooting](#troubleshooting) if you just want to run it. + +### Pinned torch stack + +`torch`, `torchvision` and `torchaudio` install together from `requirements/xpu.txt` as `+xpu` builds, via `--index-url https://download.pytorch.org/whl/xpu` (not `--extra-index-url`, which would let a plain PyPI build win). Pinning all three avoids `RuntimeError: operator torchvision::nms does not exist` from an ABI-mismatched pair. + +No pip constraint file is needed: every other `torch` requirement in the dependency graph is a lower bound that `2.13.0+xpu` already satisfies, so nothing later swaps it out. + +If the XPU wheels ever get replaced with plain ones, `torch.xpu.device_count()` returns `0`. Restore with: + +```bash +pip install --force-reinstall -r requirements/xpu.txt \ + --index-url https://download.pytorch.org/whl/xpu +``` + +### How pip is installed + +The DLE base ships Python 3.12 with no pip and no `ensurepip` (Ubuntu strips it). pip comes from apt (`python3-pip`) and is **left at the distro version** rather than upgraded — apt's pip has no `RECORD` file, so `pip install --upgrade pip` fails with: + +``` +ERROR: Cannot uninstall pip 24.0, RECORD file not found. Hint: The package was installed by debian. +``` + +— taking `setuptools`/`wheel`/`hatchling` down with it in the same command. + +### Multi-GPU and `OPTIM_TORCH` + +The image sets `ENV OPTIM_TORCH=0`. The launcher's default (`OPTIM_TORCH=1`) sets `PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True`, whose allocator path trips a Level-Zero driver bug on 2-GPU XPU runs (DDP segfault, FSDP2 hang). Disabling it costs <1% throughput. Re-enable per-run with `OPTIM_TORCH=1 llamafactory-cli train ...` to test it. It is tracked in intel, will be update or remove this ENV when it is resolve with PyTorch or level zero etc fixes. + +### Optional accelerators + +`deepspeed` and `bitsandbytes` install on a best-effort basis — neither is required to train on XPU, so a build failure only prints a warning instead of failing the image: + +``` +WARNING: deepspeed install failed - deepspeed training unavailable in this image +WARNING: bitsandbytes install failed - 4-bit quantization (QLoRA) unavailable in this image +``` + +Both install and work on XPU today (verified: `torch.compile`, LoRA + DeepSpeed ZeRO-2 SFT, and bitsandbytes 4-bit all pass on real Arc Pro B60 hardware). `deepspeed` is pinned to `==0.19.6` here, separately from `requirements/deepspeed.txt`'s shared `<=0.18.4` cap: any version in that range detects XPU as the active accelerator and imports `torch.utils.cpp_extension.DpcppBuildExtension` — a class that used to be provided by `intel_extension_for_pytorch` (IPEX). IPEX is no longer part of this stack, so that import fails with `ImportError: cannot import name 'DpcppBuildExtension'`. deepspeed dropped the IPEX dependency and switched to the stock `BuildExtension` in `0.18.7`, so anything from there onward works; `0.19.6` is the latest release at time of writing. + +Check the build log, or verify inside the container with `python3 -c "import deepspeed"` / `import bitsandbytes`. Install manually if missing: + +```bash +DS_BUILD_OPS=0 pip install "deepspeed==0.19.6" +pip install -r requirements/bitsandbytes.txt +``` + +### `ocloc` and `torch.compile` + +`torch.compile` lowers to Intel Triton, which shells out to `ocloc` (the Intel offline GPU compiler) — not bundled in the DLE base, so compiled paths fail with `FileNotFoundError: 'ocloc'` without it. The Dockerfile installs a single pinned `.deb` from Intel's GitHub releases (`ARG OCLOC_VERSION`) rather than `apt-get install intel-ocloc`, because the Intel graphics PPA keeps only its newest revision and would drag `libze-intel-gpu1` (the host-driver-facing package) up with it. + +> [!IMPORTANT] +> If you override `ARG BASE_IMAGE`, move `OCLOC_VERSION` to match — check `docker run --rm dpkg -l libze-intel-gpu1 | tail -1` and pick the nearest [compute-runtime release](https://github.com/intel/compute-runtime/releases). + +## Troubleshooting + +### GPU Not Detected (`torch.xpu.device_count()` returns 0) + +1. **Kernel driver too old for the DLE runtime** — compare `dpkg -l libze-intel-gpu1` on the host vs. `docker run --rm llamafactory:xpu dpkg -l libze-intel-gpu1`. Update the host driver (`sudo apt-get install -y intel-i915-dkms intel-fw-gpu && sudo reboot`) or use an older `BASE_IMAGE`. +2. **Missing `/dev/dri` device** — pass `--device /dev/dri` (done automatically by `docker compose`). +3. **Missing group membership** — the container process needs the `render` and `video` groups; `docker compose` sets these via `group_add`. For manual `docker run`, pass `--group-add $(getent group render | cut -d: -f3) --group-add $(getent group video | cut -d: -f3)`. +4. **`by-path` mount missing** — required for multi-GPU Level-Zero IPC: `-v /dev/dri/by-path:/dev/dri/by-path` (included in `docker-compose.yml`). + +### `fatal error: Python.h: No such file or directory` + +Intel Triton JIT-compiles a C driver at the first XPU kernel launch, which needs matching `pythonX.Y-dev` headers and a C compiler present in the image — installed at build time even though the error would happen at runtime. If you've swapped the interpreter, reinstall matching headers: + +```bash +PY_MM="$(python3 -c 'import sys; print(f"{sys.version_info.major}.{sys.version_info.minor}")')" +apt-get update && apt-get install -y "python${PY_MM}-dev" build-essential +``` + +### Permission Denied on `/dev/dri` + +```bash +sudo usermod -aG render,video $USER +newgrp render # apply without logout +``` + +### `SYCL Backends mismatch` / `libsycl.so.N: cannot open shared object file` + +Two SYCL runtimes exist in this image by design — torch's own `intel-sycl-rt` (pip) and the DLE base's oneAPI — and normally share the same `libsycl.so.9` SONAME, so both coexist fine. A mismatch shows up only after changing `torch` or `BASE_IMAGE` independently. Compare versions: + +```bash +pip list | grep -E "intel-sycl-rt|dpcpp-cpp-rt" +ls /opt/intel/oneapi/*/lib/libsycl.so.* +``` + +The `libsycl.so.N` numbers on both sides must match. `ldconfig -p | grep libsycl` returning nothing is expected — neither runtime is in the ldconfig cache. + +## Additional Notes + +- The container automatically sources `/opt/intel/oneapi/setvars.sh` in every interactive shell (`~/.bashrc`). For non-interactive scripts, source it explicitly. +- For training, `llamafactory-cli train` dispatches automatically via `torchrun` for multi-GPU. diff --git a/docker/docker-xpu/docker-compose.yml b/docker/docker-xpu/docker-compose.yml new file mode 100644 index 000000000..7bf7a4bff --- /dev/null +++ b/docker/docker-xpu/docker-compose.yml @@ -0,0 +1,48 @@ +services: + llamafactory: + build: + dockerfile: ./docker/docker-xpu/Dockerfile + context: ../.. + args: + PIP_INDEX: https://pypi.org/simple + # The DLE base image determines which Intel GPU runtime (libze-intel-gpu) + # is shipped. The runtime version must be >= the host driver version, + # otherwise torch.xpu.device_count() returns 0: + # DLE 2025.3 → libze-intel-gpu 25.18.x (works on driver ≤26.09) + # DLE 2026.1 → libze-intel-gpu 26.18.x+ (works on driver 26.18/26.22) + # Verify host driver: dpkg -l libze-intel-gpu1 | grep -oP '\d+\.\d+\.\d+' + BASE_IMAGE: intel/deep-learning-essentials:2026.1.0-devel-ubuntu24.04 + container_name: llamafactory + image: llamafactory:xpu + ports: + - "7860:7860" + - "8000:8000" + ipc: host + tty: true + # shm_size: "16gb" # ipc: host is set + stdin_open: true + command: bash + devices: + # Intel GPU character devices (renderD* and card* under /dev/dri). + # renderD* nodes are owned by group `render`; card* by group `video`. + - /dev/dri:/dev/dri + volumes: + # by-path symlinks are required for oneCCL's ze_fd_manager (2-GPU IPC). + # `devices:` alone does not carry sub-directories — this volume does. + - /dev/dri/by-path:/dev/dri/by-path + - ~/.cache/huggingface:/root/.cache/huggingface + group_add: + # Grant access to the Intel GPU device nodes. + # renderD* nodes are owned by group `render`; card* nodes by `video`. + # Use numeric GIDs here — Docker compose resolves group names against + # the CONTAINER's /etc/group (not the host), and the DLE base image + # does not carry a `render` group entry. Standard Linux GIDs: + # If your host uses different GIDs, run `getent group render video` and + # update these values accordingly. + # example output. + # getent group render video + # render:x:992:root + # video:x:44:support + - "992" # render — owns /dev/dri/renderD* nodes + - "44" # video — owns /dev/dri/card* nodes + restart: unless-stopped diff --git a/requirements/xpu.txt b/requirements/xpu.txt new file mode 100644 index 000000000..a1880bd66 --- /dev/null +++ b/requirements/xpu.txt @@ -0,0 +1,3 @@ +torch==2.13.0+xpu +torchvision==0.28.0+xpu +torchaudio==2.11.0+xpu diff --git a/src/llamafactory/v1/accelerator/helper.py b/src/llamafactory/v1/accelerator/helper.py index 2c780aae2..b83e7f2f5 100644 --- a/src/llamafactory/v1/accelerator/helper.py +++ b/src/llamafactory/v1/accelerator/helper.py @@ -186,6 +186,8 @@ def get_process_group_backend() -> str: return "hccl" elif get_current_accelerator().type == DeviceType.CUDA: return "nccl" + elif get_current_accelerator().type == DeviceType.XPU: + return "xccl" else: return "gloo"