5 Commits

3 changed files with 74 additions and 31 deletions

View File

@@ -8,7 +8,7 @@
# vfio-native-qemu QEMU 11.1.1 with the platform-identity patches, in /opt # vfio-native-qemu QEMU 11.1.1 with the platform-identity patches, in /opt
pkgname=vfio-native pkgname=vfio-native
pkgver=1.3.0 pkgver=1.3.2
pkgrel=1 pkgrel=1
pkgdesc="Present a libvirt guest as a self-consistent physical machine, and tune it" pkgdesc="Present a libvirt guest as a self-consistent physical machine, and tune it"
arch=('any') arch=('any')

View File

@@ -1,12 +1,18 @@
#!/bin/bash #!/bin/bash
# Waits for a just-started guest's network to actually come up, then turns # Keeps the kvm_amd cpuid_passthrough switch correct across a guest's whole life,
# kvm_amd cpuid_passthrough on. # in-guest reboots included.
# #
# The readiness signal is the guest's own tap RX counter: it is fresh every boot, # The switch must be OFF (N) while the guest enumerates CPUID at boot or Windows
# it needs nothing enabled inside the guest (no SSH, no RDP, no agent), and it only # hangs, and ON (Y) once it is up, where it clears the timer detection. The libvirt
# moves once the guest's NIC driver has really loaded - which is well past the CPU # hook only fires at VM start and stop, so a guest-initiated reboot would otherwise
# enumeration that the switch must not change under. A DHCP lease left over from a # re-enumerate with the switch still Y and hang. This watcher drops it to N on every
# previous boot cannot trip it early. # QEMU RESET and raises it again once the guest's NIC is back up.
#
# Readiness signal: the guest tap's rx_packets counter growing past a baseline. It
# only moves once the guest NIC driver has loaded, well past CPU enumeration, and it
# needs nothing enabled inside the guest. The counter is cumulative and does NOT
# reset on an in-guest reboot, so readiness is growth past the value captured at the
# reset, not an absolute threshold.
# #
# Launched as a transient systemd unit by the cpuid-passthrough hook, so it is free # Launched as a transient systemd unit by the cpuid-passthrough hook, so it is free
# to call virsh (the hook itself must not - that deadlocks libvirtd). # to call virsh (the hook itself must not - that deadlocks libvirtd).
@@ -18,24 +24,43 @@ V="virsh -c qemu:///system"
running() { [ "$($V domstate "$DOMAIN" 2>/dev/null)" = running ]; } running() { [ "$($V domstate "$DOMAIN" 2>/dev/null)" = running ]; }
tap="" tap=$($V domiflist "$DOMAIN" 2>/dev/null | awk '$1 ~ /^(vnet|tap|macvtap)/ {print $1; exit}')
i=0 RX="/sys/class/net/$tap/statistics/rx_packets"
for _ in $(seq 1 65); do # ~195 s cap, then flip anyway if still up rx() { cat "$RX" 2>/dev/null || echo 0; }
i=$((i + 1))
running || { sleep 3; continue; } # not "running" yet at prepare time - wait
[ -z "$tap" ] && tap=$($V domiflist "$DOMAIN" 2>/dev/null |
awk '$1 ~ /^(vnet|tap|macvtap)/ {print $1; exit}')
rx="/sys/class/net/$tap/statistics/rx_packets"
# the iteration floor keeps a stray pre-OS packet (a UEFI netboot attempt) from
# tripping the flip before the guest is even past its interrupt and timer setup
if [ "$i" -ge 4 ] && [ -n "$tap" ] && [ -r "$rx" ] &&
[ "$(cat "$rx" 2>/dev/null || echo 0)" -ge 4 ]; then
break
fi
sleep 3
done
running || exit 0 # guest went away before it came up set_N() { echo N > "$PARAM/cpuid_passthrough"; }
printf '%s' "$BRAND" > "$PARAM/brand_string" set_Y() { printf '%s' "$BRAND" > "$PARAM/brand_string"; echo Y > "$PARAM/cpuid_passthrough"; }
echo Y > "$PARAM/cpuid_passthrough"
# Wait until the tap rx counter grows at least 4 past $1 (guest NIC driver up again).
# The iteration floor keeps a stray pre-OS packet (a UEFI netboot attempt) from
# tripping the flip before the guest is even past its interrupt and timer setup.
wait_net_up() {
local base=$1 i=0
for _ in $(seq 1 100); do # ~300 s cap, then raise anyway if still up
i=$((i + 1))
running || return 1
[ "$i" -ge 4 ] && [ -n "$tap" ] && [ -r "$RX" ] &&
[ "$(rx)" -ge "$((base + 4))" ] && return 0
sleep 3
done
return 0
}
# initial cold boot: wait for the network, then harden
wait_net_up 0
running || exit 0
set_Y
logger -t vfio-cpuid "$DOMAIN network up: cpuid_passthrough=Y" logger -t vfio-cpuid "$DOMAIN network up: cpuid_passthrough=Y"
# every in-guest reboot fires a QEMU RESET: drop to N for the re-enumeration, then
# raise it again once the guest's NIC is back. --loop streams one line per reset.
$V qemu-monitor-event --domain "$DOMAIN" --event RESET --loop 2>/dev/null | while read -r _; do
running || continue
base=$(rx)
set_N
logger -t vfio-cpuid "$DOMAIN reset: cpuid_passthrough=N for re-enumeration"
wait_net_up "$base" || continue
running || continue
set_Y
logger -t vfio-cpuid "$DOMAIN back up: cpuid_passthrough=Y"
done

View File

@@ -57,6 +57,24 @@ group_members() {
done done
} }
# The functions to hand to the guest with GPU $1: every device in its IOMMU group
# (mandatory for vfio), plus any sibling function of the same PCI device that is
# itself a display or HDMI/DP audio controller. An APU parks its PSP and USB
# controllers on the same PCI device in separate IOMMU groups - those are the
# host's, so siblings are filtered by class and never pulled in blindly.
passthrough_devs() {
local d="$1" m cls
{
group_members "$d"
for m in "/sys/bus/pci/devices/${d%.*}".*; do
[ -e "$m" ] || continue
m=$(basename "$m")
cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)
case "$cls" in 0x03*|0x0403*) echo "$m";; esac
done
} | sort -u
}
driver_of() { driver_of() {
local l local l
l=$(readlink -f "/sys/bus/pci/devices/$1/driver" 2>/dev/null) l=$(readlink -f "/sys/bus/pci/devices/$1/driver" 2>/dev/null)
@@ -79,7 +97,7 @@ drives_display() {
ids_of() { # vendor:device for vfio-pci binding ids_of() { # vendor:device for vfio-pci binding
local d local d
for d in $(group_members "$1"); do for d in $(passthrough_devs "$1"); do
local cls; cls=$(cat "/sys/bus/pci/devices/$d/class" 2>/dev/null) local cls; cls=$(cat "/sys/bus/pci/devices/$d/class" 2>/dev/null)
# only real functions of the card: skip bridges (class 0x0604xx) # only real functions of the card: skip bridges (class 0x0604xx)
case "$cls" in 0x0604*) continue;; esac case "$cls" in 0x0604*) continue;; esac
@@ -219,7 +237,7 @@ prepare)
# 1. stop whatever is holding the DRM device # 1. stop whatever is holding the DRM device
dm=$(systemctl list-units --type=service --state=running --no-legend 2>/dev/null | dm=$(systemctl list-units --type=service --state=running --no-legend 2>/dev/null |
awk '{print $1}' | grep -xE '(gdm|sddm|lightdm|lxdm|greetd|display-manager)\.service' | head -1) awk '{print $1}' | grep -xE '((gdm|sddm|lightdm|lxdm|greetd|ly|emptty|lemurs)(@[a-z0-9-]+)?|display-manager)\.service' | head -1)
if [ -n "$dm" ]; then if [ -n "$dm" ]; then
echo "$dm" > "$STATE/dm" echo "$dm" > "$STATE/dm"
log "stopping $dm" log "stopping $dm"
@@ -303,7 +321,7 @@ apply() {
"$v" -c qemu:///system dumpxml --inactive "$dom" > "$backup/$dom.gpu-current.xml" "$v" -c qemu:///system dumpxml --inactive "$dom" > "$backup/$dom.gpu-current.xml"
grep -q 'ua-vfionative-gpu' "$backup/$dom.gpu-current.xml" || cp "$backup/$dom.gpu-current.xml" "$backup/$dom.before-gpu.xml" grep -q 'ua-vfionative-gpu' "$backup/$dom.gpu-current.xml" || cp "$backup/$dom.gpu-current.xml" "$backup/$dom.before-gpu.xml"
local devs="" m local devs="" m
for m in $( { group_members "$d"; ls -d "/sys/bus/pci/devices/${d%.*}".* | xargs -n1 basename; } | sort -u); do for m in $(passthrough_devs "$d"); do
case "$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)" in 0x0604*) continue;; esac case "$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)" in 0x0604*) continue;; esac
IFS=':.' read -r dm bs sl fn <<< "$m" IFS=':.' read -r dm bs sl fn <<< "$m"
devs+=" <hostdev mode='subsystem' type='pci' managed='yes'>\n <source>\n" devs+=" <hostdev mode='subsystem' type='pci' managed='yes'>\n <source>\n"
@@ -330,7 +348,7 @@ xml() {
echo "Add this to the domain, inside <devices>. Every device in the GPU's" echo "Add this to the domain, inside <devices>. Every device in the GPU's"
echo "IOMMU group has to go together:" echo "IOMMU group has to go together:"
echo echo
for m in $(group_members "$d"); do for m in $(passthrough_devs "$d"); do
local cls; cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null) local cls; cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)
case "$cls" in 0x0604*) continue;; esac case "$cls" in 0x0604*) continue;; esac
IFS=':. ' read -r dom bus slot fn <<< "$(echo "$m" | tr ':.' ' ')" IFS=':. ' read -r dom bus slot fn <<< "$(echo "$m" | tr ':.' ' ')"