392 lines
16 KiB
Bash
Executable File
392 lines
16 KiB
Bash
Executable File
#!/bin/bash
|
|
# Set up GPU passthrough for a native-profile guest, with the smallest change to
|
|
# the host that will do the job. A passed-through card gives the guest real
|
|
# silicon in place of an emulated display adapter.
|
|
#
|
|
# vm-native-gpu show GPUs, then ask and do it
|
|
# vm-native-gpu --help show GPUs and what to do, change nothing
|
|
# vm-native-gpu --single <domain> <pciid> [rom] pass the host's only GPU
|
|
# vm-native-gpu --dual <domain> <pciid> [rom] pass a GPU the host is not using
|
|
# vm-native-gpu --apply <domain> <pciid> [rom] wire the GPU into the domain XML only
|
|
# vm-native-gpu --xml <pciid> print the <hostdev> block, change nothing
|
|
# vm-native-gpu --revert undo the hook and modprobe file
|
|
#
|
|
# --single, --dual and --apply all write the domain XML: every function in the
|
|
# card's IOMMU group as a <hostdev>, an optional vBIOS rom file, the emulated
|
|
# display removed and any GL render node pointing at the card dropped. The
|
|
# domain is backed up first and a re-run replaces what an earlier run wrote.
|
|
#
|
|
# Most machines have one GPU, so single-GPU is the normal path. The host gives
|
|
# the card up while the guest runs: the display manager is stopped, the console
|
|
# and framebuffer are unbound, the GPU driver is unloaded and the card is bound
|
|
# to vfio-pci. All of it is reversed when the guest stops.
|
|
#
|
|
# The host has no display for as long as the guest runs. Have SSH working before
|
|
# you try it - that is your way back if a step fails.
|
|
#
|
|
# If you do have a second GPU the host is not using, --dual is strictly better:
|
|
# the card is bound to vfio-pci at boot, nothing is torn down at VM start, and
|
|
# the host keeps its display the whole time.
|
|
set -uo pipefail
|
|
|
|
MODPROBE=/etc/modprobe.d/vfio-native.conf
|
|
HOOK=/etc/libvirt/hooks/qemu.d/20-gpu-passthrough.sh
|
|
|
|
die() { echo "$*" >&2; exit 1; }
|
|
need_root() { [ "$(id -u)" = 0 ] || die "run this as root"; }
|
|
|
|
# --- discovery ---------------------------------------------------------------
|
|
gpus() {
|
|
lspci -Dnn | grep -E "VGA compatible controller|3D controller" |
|
|
sed 's/ (rev [0-9a-f]*)//'
|
|
}
|
|
|
|
iommu_group() {
|
|
local d="$1" g
|
|
g=$(readlink -f "/sys/bus/pci/devices/$d/iommu_group" 2>/dev/null)
|
|
echo "${g##*/}"
|
|
}
|
|
|
|
# every device sharing the GPU's IOMMU group must go to the guest with it
|
|
group_members() {
|
|
local d="$1" g
|
|
g=$(iommu_group "$d")
|
|
[ -n "$g" ] || return
|
|
for m in /sys/kernel/iommu_groups/"$g"/devices/*; do
|
|
[ -e "$m" ] && basename "$m"
|
|
done
|
|
}
|
|
|
|
driver_of() {
|
|
local l
|
|
l=$(readlink -f "/sys/bus/pci/devices/$1/driver" 2>/dev/null)
|
|
[ -n "$l" ] && echo "${l##*/}" || echo "(none)"
|
|
}
|
|
|
|
# does this card currently drive a connected display?
|
|
drives_display() {
|
|
local d="$1" c
|
|
for c in /sys/class/drm/card*/device; do
|
|
[ -e "$c" ] || continue
|
|
if [ "$(basename "$(readlink -f "$c")")" = "$d" ]; then
|
|
for s in "$(dirname "$c")"*/status; do
|
|
[ -e "$s" ] && grep -qx connected "$s" && return 0
|
|
done
|
|
fi
|
|
done
|
|
return 1
|
|
}
|
|
|
|
ids_of() { # vendor:device for vfio-pci binding
|
|
local d
|
|
for d in $(group_members "$1"); do
|
|
local cls; cls=$(cat "/sys/bus/pci/devices/$d/class" 2>/dev/null)
|
|
# only real functions of the card: skip bridges (class 0x0604xx)
|
|
case "$cls" in 0x0604*) continue;; esac
|
|
printf '%s:%s\n' \
|
|
"$(cut -c3- < "/sys/bus/pci/devices/$d/vendor")" \
|
|
"$(cut -c3- < "/sys/bus/pci/devices/$d/device")"
|
|
done | sort -u | paste -sd,
|
|
}
|
|
|
|
show() {
|
|
echo "GPUs in this machine:"
|
|
echo
|
|
while read -r line; do
|
|
d=${line%% *}
|
|
grp=$(iommu_group "$d"); drv=$(driver_of "$d")
|
|
if drives_display "$d"; then use="drives a connected display"; else use="no display attached"; fi
|
|
echo " $line"
|
|
echo " pci $d iommu group $grp driver $drv"
|
|
echo " $use"
|
|
echo " group members: $(group_members "$d" | paste -sd' ')"
|
|
echo
|
|
done < <(gpus)
|
|
|
|
local n; n=$(gpus | wc -l)
|
|
echo "What to do:"
|
|
local n cand=""
|
|
n=$(gpus | wc -l)
|
|
while read -r line; do
|
|
d=${line%% *}
|
|
drives_display "$d" || cand="$d"
|
|
done < <(gpus)
|
|
|
|
if [ "$n" -ge 2 ] && [ -n "$cand" ]; then
|
|
echo " You have a spare GPU ($cand, no display attached), which is the"
|
|
echo " easy case. Bind it to vfio-pci at boot and the host never touches"
|
|
echo " it - nothing to tear down at VM start, host keeps its display:"
|
|
echo
|
|
echo " sudo vm-native-gpu --dual <domain> $cand"
|
|
echo
|
|
echo " Check first that nothing on the host is using it:"
|
|
echo " sudo fuser -v /dev/dri/by-path/pci-$cand-*"
|
|
echo " A compositor often holds every DRM device even with no monitor on"
|
|
echo " it. Binding at boot fixes that; it needs an initramfs rebuild and"
|
|
echo " a reboot."
|
|
else
|
|
local only; only=$(gpus | head -1 | cut -d' ' -f1)
|
|
echo " One GPU, which is the normal case. The host hands it over while"
|
|
echo " the guest runs and takes it back afterwards:"
|
|
echo
|
|
echo " sudo vm-native-gpu --single <domain> $only"
|
|
echo
|
|
echo " The host has no display for as long as the guest runs. Get SSH"
|
|
echo " working first - that is your way back if a step fails."
|
|
fi
|
|
|
|
}
|
|
|
|
# --- dual GPU: bind at boot, no hooks ---------------------------------------
|
|
dual() {
|
|
need_root
|
|
local dom="$1" d="$2" rom="${3:-}" ids
|
|
[ -e "/sys/bus/pci/devices/$d" ] || die "no such PCI device: $d"
|
|
drives_display "$d" && die "$d is driving a connected display - move your monitors off it first"
|
|
ids=$(ids_of "$d")
|
|
[ -n "$ids" ] || die "could not read device ids for $d"
|
|
|
|
cat > "$MODPROBE" <<EOF
|
|
# Written by vfio-native. Binds the passthrough GPU to vfio-pci at boot so the
|
|
# host driver never claims it. Remove this file and rebuild the initramfs to undo.
|
|
options vfio-pci ids=$ids
|
|
softdep amdgpu pre: vfio-pci
|
|
softdep nvidia pre: vfio-pci
|
|
softdep nouveau pre: vfio-pci
|
|
EOF
|
|
echo "wrote $MODPROBE:"
|
|
sed 's/^/ /' "$MODPROBE"
|
|
echo
|
|
echo "vfio-pci must be in the initramfs so it claims the card before the"
|
|
echo "host driver does. On Arch, add it to MODULES in /etc/mkinitcpio.conf:"
|
|
echo
|
|
echo " MODULES=(vfio_pci vfio vfio_iommu_type1)"
|
|
echo " sudo mkinitcpio -P"
|
|
echo
|
|
echo "Then reboot and check with:"
|
|
echo " lspci -nnk -d ::0300 | grep -A2 $d"
|
|
echo "The driver in use should read vfio-pci."
|
|
echo
|
|
apply "$dom" "$d" "$rom"
|
|
}
|
|
|
|
# --- single GPU: dynamic, invasive ------------------------------------------
|
|
single() {
|
|
need_root
|
|
local dom="$1" d="$2" rom="${3:-}"
|
|
[ -e "/sys/bus/pci/devices/$d" ] || die "no such PCI device: $d"
|
|
virsh -c qemu:///system dominfo "$dom" >/dev/null 2>&1 || die "no such domain: $dom"
|
|
|
|
echo "This installs a libvirt hook that, every time the guest starts, will:"
|
|
echo " - stop your display manager and switch to multi-user.target"
|
|
echo " - unbind the virtual consoles and the EFI framebuffer"
|
|
echo " - unload the GPU driver and bind the card to vfio-pci"
|
|
echo "and reverse all of it when the guest stops."
|
|
echo
|
|
echo "The host has NO DISPLAY while the guest runs. If it fails part-way you"
|
|
echo "may be left at a black screen and need SSH to recover."
|
|
echo
|
|
echo "It fires ONLY for domain \"$dom\". Other VMs are untouched."
|
|
echo
|
|
read -rp "install it for domain $dom, PCI $d? [y/N]: " a
|
|
[ "$a" = y ] || { echo aborted; exit 1; }
|
|
|
|
sed -e "s|@@GPU@@|$d|g" -e "s|@@DOMAIN@@|$dom|g" > "$HOOK" <<'HOOKEOF'
|
|
#!/bin/bash
|
|
# Single-GPU passthrough for one domain, installed by vfio-native.
|
|
#
|
|
# Frees the GPU before the guest starts and gives it back afterwards. The actual
|
|
# vfio-pci bind/unbind is left to libvirt, because the <hostdev> is managed='yes'
|
|
# - this only has to make the card free for libvirt to take.
|
|
#
|
|
# Exits 0 on every path. libvirt treats a non-zero prepare hook as fatal, and a
|
|
# GPU helper must never be the reason a VM refuses to start.
|
|
GPU="@@GPU@@"
|
|
DOMAIN="@@DOMAIN@@"
|
|
LOG=/var/log/libvirt/gpu-passthrough.log
|
|
STATE=/run/vfio-native
|
|
|
|
log() { echo "$(date +%T) $*" >> "$LOG"; }
|
|
|
|
# Only ever act for the domain this was installed for.
|
|
[ "$1" = "$DOMAIN" ] || exit 0
|
|
|
|
mkdir -p "$STATE"
|
|
|
|
case "$2" in
|
|
prepare)
|
|
log "$DOMAIN starting: releasing $GPU"
|
|
|
|
# 1. stop whatever is holding the DRM device
|
|
dm=$(systemctl list-units --type=service --state=running --no-legend 2>/dev/null |
|
|
awk '{print $1}' | grep -xE '(gdm|sddm|lightdm|lxdm|greetd|display-manager)\.service' | head -1)
|
|
if [ -n "$dm" ]; then
|
|
echo "$dm" > "$STATE/dm"
|
|
log "stopping $dm"
|
|
systemctl stop "$dm"
|
|
# wait for it to actually let go, rather than racing it
|
|
for _ in $(seq 1 20); do systemctl is-active --quiet "$dm" || break; sleep 0.5; done
|
|
fi
|
|
|
|
# 2. release the console framebuffers
|
|
: > "$STATE/consoles"
|
|
for i in /sys/class/vtconsole/vtcon*; do
|
|
[ -e "$i/name" ] || continue
|
|
if grep -q "frame buffer" "$i/name" 2>/dev/null; then
|
|
echo 0 > "$i/bind" 2>/dev/null && { basename "$i" >> "$STATE/consoles"; log "unbound $(basename "$i")"; }
|
|
fi
|
|
done
|
|
[ -e /sys/bus/platform/drivers/efi-framebuffer/unbind ] &&
|
|
echo efi-framebuffer.0 > /sys/bus/platform/drivers/efi-framebuffer/unbind 2>/dev/null
|
|
[ -e /sys/bus/platform/drivers/simple-framebuffer/unbind ] &&
|
|
for fb in /sys/bus/platform/drivers/simple-framebuffer/simple-framebuffer.*; do
|
|
[ -e "$fb" ] && basename "$fb" > /sys/bus/platform/drivers/simple-framebuffer/unbind 2>/dev/null
|
|
done
|
|
sleep 1
|
|
|
|
# 3. unload the host GPU driver, remembering which one to put back
|
|
drv=$(basename "$(readlink -f "/sys/bus/pci/devices/$GPU/driver" 2>/dev/null)" 2>/dev/null)
|
|
case "$drv" in
|
|
amdgpu|radeon) echo "$drv" > "$STATE/driver"
|
|
modprobe -r amdgpu radeon 2>/dev/null ;;
|
|
nvidia*) echo nvidia > "$STATE/driver"
|
|
modprobe -r nvidia_uvm nvidia_drm nvidia_modeset nvidia 2>/dev/null ;;
|
|
nouveau) echo nouveau > "$STATE/driver"
|
|
modprobe -r nouveau 2>/dev/null ;;
|
|
vfio-pci) log "already on vfio-pci" ;;
|
|
*) log "unexpected driver '$drv', leaving it alone" ;;
|
|
esac
|
|
modprobe vfio-pci 2>/dev/null
|
|
log "released; libvirt will bind $GPU to vfio-pci"
|
|
;;
|
|
|
|
release)
|
|
log "$DOMAIN stopped: reclaiming $GPU"
|
|
|
|
drv=$(cat "$STATE/driver" 2>/dev/null)
|
|
[ -n "$drv" ] && { modprobe "$drv" 2>/dev/null; log "reloaded $drv"; }
|
|
sleep 1
|
|
|
|
while read -r c; do
|
|
[ -n "$c" ] && echo 1 > "/sys/class/vtconsole/$c/bind" 2>/dev/null
|
|
done < "$STATE/consoles" 2>/dev/null
|
|
|
|
dm=$(cat "$STATE/dm" 2>/dev/null)
|
|
if [ -n "$dm" ]; then
|
|
log "starting $dm"
|
|
systemctl start "$dm"
|
|
fi
|
|
rm -f "$STATE/dm" "$STATE/consoles" "$STATE/driver"
|
|
log "host display restored"
|
|
;;
|
|
esac
|
|
exit 0
|
|
HOOKEOF
|
|
chmod +x "$HOOK"
|
|
echo "installed $HOOK"
|
|
echo
|
|
apply "$dom" "$d" "$rom"
|
|
}
|
|
|
|
# Write the passthrough into the domain: the whole IOMMU group as <hostdev>s
|
|
# tagged with a user alias so a re-run replaces exactly these, the vBIOS rom if
|
|
# given, the emulated display gone, and no GL render node left on the card.
|
|
apply() {
|
|
local dom="$1" d="$2" rom="${3:-}" v
|
|
v=$(command -v virsh) || die "virsh not found"
|
|
"$v" -c qemu:///system dominfo "$dom" >/dev/null 2>&1 || die "no such domain: $dom"
|
|
[ "$("$v" -c qemu:///system domstate "$dom")" = "shut off" ] || die "shut $dom down first"
|
|
[ -z "$rom" ] || [ -f "$rom" ] || die "rom file not found: $rom"
|
|
local backup="${SUDO_USER:+/home/$SUDO_USER}"; backup="${backup:-$HOME}/vfio-native-backup"
|
|
mkdir -p "$backup"
|
|
# the revert target is the state before this tool first touched the domain
|
|
"$v" -c qemu:///system dumpxml --inactive "$dom" > "$backup/$dom.gpu-current.xml"
|
|
grep -q 'ua-vfionative-gpu' "$backup/$dom.gpu-current.xml" || cp "$backup/$dom.gpu-current.xml" "$backup/$dom.before-gpu.xml"
|
|
local devs="" m
|
|
for m in $( { group_members "$d"; ls -d "/sys/bus/pci/devices/${d%.*}".* | xargs -n1 basename; } | sort -u); do
|
|
case "$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)" in 0x0604*) continue;; esac
|
|
IFS=':.' read -r dm bs sl fn <<< "$m"
|
|
devs+=" <hostdev mode='subsystem' type='pci' managed='yes'>\n <source>\n"
|
|
devs+=" <address domain='0x$dm' bus='0x$bs' slot='0x$sl' function='0x$fn'/>\n </source>\n"
|
|
[ -n "$rom" ] && [ "$m" = "$d" ] && devs+=" <rom file='$rom'/>\n"
|
|
devs+=" <alias name='ua-vfionative-gpu-$dm-$bs-$sl-$fn'/>\n </hostdev>\n"
|
|
done
|
|
DEVS="$devs" GPU="$d" python3 - "$backup/$dom.gpu-current.xml" "$backup/$dom.gpu.xml" <<'PY'
|
|
import io, os, re, sys
|
|
s = io.open(sys.argv[1], encoding="utf-8").read()
|
|
s = re.sub(r"\s*<hostdev mode='subsystem' type='pci' managed='yes'>(?:(?!</hostdev>).)*<alias name='ua-vfionative-gpu-[^']*'/>(?:(?!</hostdev>).)*</hostdev>", "", s, flags=re.S)
|
|
s = re.sub(r"<video>.*?</video>", "<video>\n <model type='none'/>\n </video>", s, flags=re.S)
|
|
s = re.sub(r"\s*<gl enable='yes' rendernode='[^']*%s[^']*'/>" % re.escape(os.environ["GPU"]), "", s)
|
|
s = s.replace(" </devices>", os.environ["DEVS"].replace("\\n", "\n") + " </devices>", 1)
|
|
io.open(sys.argv[2], "w", encoding="utf-8").write(s)
|
|
PY
|
|
"$v" -c qemu:///system define "$backup/$dom.gpu.xml" >/dev/null || die "define failed"
|
|
echo "$dom: $(grep -c "ua-vfionative-gpu" "$backup/$dom.gpu.xml") hostdev(s) for $d${rom:+ with rom $rom}, emulated display removed."
|
|
echo "revert: virsh -c qemu:///system define $backup/$dom.before-gpu.xml"
|
|
}
|
|
|
|
xml() {
|
|
local d="$1"
|
|
echo "Add this to the domain, inside <devices>. Every device in the GPU's"
|
|
echo "IOMMU group has to go together:"
|
|
echo
|
|
for m in $(group_members "$d"); do
|
|
local cls; cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)
|
|
case "$cls" in 0x0604*) continue;; esac
|
|
IFS=':. ' read -r dom bus slot fn <<< "$(echo "$m" | tr ':.' ' ')"
|
|
printf " <hostdev mode='subsystem' type='pci' managed='yes'>\n"
|
|
printf " <source>\n"
|
|
printf " <address domain='0x%s' bus='0x%s' slot='0x%s' function='0x%s'/>\n" \
|
|
"$dom" "$bus" "$slot" "$fn"
|
|
printf " </source>\n </hostdev>\n"
|
|
done
|
|
echo
|
|
echo "Then remove the emulated display so the guest has only the real card:"
|
|
echo " <video><model type='none'/></video>"
|
|
echo "and drop any <graphics> node that points its GL rendernode at this card."
|
|
}
|
|
|
|
# Run bare: the same information, then the questions, then the matching command.
|
|
guided() {
|
|
show
|
|
[ -t 0 ] || return 0
|
|
local n cand="" only d dom mode rom a
|
|
n=$(gpus | wc -l)
|
|
while read -r line; do d=${line%% *}; drives_display "$d" || cand="$d"; done < <(gpus)
|
|
only=$(gpus | head -1 | cut -d' ' -f1)
|
|
echo "-----------------------------------------------------------------------"
|
|
read -rp "set up passthrough now? (y/n) [n]: " a; [ "$a" = y ] || return 0
|
|
read -rp "domain: " dom; [ -n "$dom" ] || die "need a domain"
|
|
if [ "$n" -ge 2 ] && [ -n "$cand" ]; then
|
|
read -rp "GPU to pass [$cand]: " d; d=${d:-$cand}
|
|
read -rp "mode - dual (bind at boot, host keeps its display) or single (host gives it up while the guest runs) [dual]: " mode; mode=${mode:-dual}
|
|
else
|
|
read -rp "GPU to pass [$only]: " d; d=${d:-$only}
|
|
mode=single
|
|
fi
|
|
read -rp "vBIOS rom file for the guest, or empty [none]: " rom
|
|
[ -z "$rom" ] || [ -f "$rom" ] || die "rom file not found: $rom"
|
|
[ "$(id -u)" = 0 ] || die "installing the ${mode} setup needs root: sudo $(basename "$0") --$mode $dom $d${rom:+ $rom}"
|
|
"$mode" "$dom" "$d" "$rom"
|
|
}
|
|
|
|
revert() {
|
|
need_root
|
|
local did=0
|
|
[ -e "$MODPROBE" ] && { rm -f "$MODPROBE"; echo "removed $MODPROBE"; did=1; }
|
|
[ -e "$HOOK" ] && { rm -f "$HOOK"; echo "removed $HOOK"; did=1; }
|
|
[ "$did" = 1 ] || echo "nothing installed by this script was found"
|
|
[ -e "$MODPROBE" ] || echo "rebuild the initramfs (sudo mkinitcpio -P) and reboot to release the card"
|
|
}
|
|
|
|
case "${1:-}" in
|
|
--dual) [ $# -ge 3 ] || die "usage: --dual <domain> <pciid> [romfile]"; dual "$2" "$3" "${4:-}";;
|
|
--single) [ $# -ge 3 ] || die "usage: --single <domain> <pciid> [romfile]"; single "$2" "$3" "${4:-}";;
|
|
--apply) [ $# -ge 3 ] || die "usage: --apply <domain> <pciid> [romfile]"; apply "$2" "$3" "${4:-}";;
|
|
--xml) [ $# -ge 2 ] || die "need a PCI id"; xml "$2";;
|
|
--revert) revert;;
|
|
"") guided;;
|
|
-h|--help) show;;
|
|
*) die "unknown option: $1";;
|
|
esac
|