services: llamafactory: build: dockerfile: ./docker/docker-xpu/Dockerfile context: ../.. args: PIP_INDEX: https://pypi.org/simple # The DLE base image determines which Intel GPU runtime (libze-intel-gpu) # is shipped. The runtime version must be >= the host driver version, # otherwise torch.xpu.device_count() returns 0: # DLE 2025.3 → libze-intel-gpu 25.18.x (works on driver ≤26.09) # DLE 2026.1 → libze-intel-gpu 26.18.x+ (works on driver 26.18/26.22) # Verify host driver: dpkg -l libze-intel-gpu1 | grep -oP '\d+\.\d+\.\d+' BASE_IMAGE: intel/deep-learning-essentials:2026.1.0-devel-ubuntu24.04 container_name: llamafactory image: llamafactory:xpu ports: - "7860:7860" - "8000:8000" ipc: host tty: true # shm_size: "16gb" # ipc: host is set stdin_open: true command: bash devices: # Intel GPU character devices (renderD* and card* under /dev/dri). # renderD* nodes are owned by group `render`; card* by group `video`. - /dev/dri:/dev/dri volumes: # by-path symlinks are required for oneCCL's ze_fd_manager (2-GPU IPC). # `devices:` alone does not carry sub-directories — this volume does. - /dev/dri/by-path:/dev/dri/by-path - ~/.cache/huggingface:/root/.cache/huggingface group_add: # Grant access to the Intel GPU device nodes. # renderD* nodes are owned by group `render`; card* nodes by `video`. # Use numeric GIDs here — Docker compose resolves group names against # the CONTAINER's /etc/group (not the host), and the DLE base image # does not carry a `render` group entry. Standard Linux GIDs: # If your host uses different GIDs, run `getent group render video` and # update these values accordingly. # example output. # getent group render video # render:x:992:root # video:x:44:support - "992" # render — owns /dev/dri/renderD* nodes - "44" # video — owns /dev/dri/card* nodes restart: unless-stopped