Compare commits
7 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f99ff1400a | |||
| 4e0df16271 | |||
| 0c3425e5af | |||
| a956ececfd | |||
| 966a2a6390 | |||
| d3062f5e24 | |||
| 1e1078e16a |
@@ -1,6 +1,6 @@
|
||||
pkgbase = vfio-native
|
||||
pkgdesc = Present a libvirt guest as a self-consistent physical machine, and tune it
|
||||
pkgver = 1.1.1
|
||||
pkgver = 1.3.3
|
||||
pkgrel = 1
|
||||
url = https://git.archworks.co/sandwich/vfio-native
|
||||
install = vfio-native.install
|
||||
@@ -17,7 +17,7 @@ pkgbase = vfio-native
|
||||
optdepends = cpupower: set the host CPU governor
|
||||
optdepends = vfio-native-kvm-dkms: patched KVM modules for the full level
|
||||
optdepends = vfio-native-qemu: patched QEMU for the full level
|
||||
source = git+https://git.archworks.co/sandwich/vfio-native.git#tag=v1.1.1
|
||||
source = git+https://git.archworks.co/sandwich/vfio-native.git#tag=v1.3.3
|
||||
sha256sums = SKIP
|
||||
|
||||
pkgname = vfio-native
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
# vfio-native-qemu QEMU 11.1.1 with the platform-identity patches, in /opt
|
||||
|
||||
pkgname=vfio-native
|
||||
pkgver=1.3.0
|
||||
pkgver=1.3.3
|
||||
pkgrel=1
|
||||
pkgdesc="Present a libvirt guest as a self-consistent physical machine, and tune it"
|
||||
arch=('any')
|
||||
|
||||
@@ -1,12 +1,18 @@
|
||||
#!/bin/bash
|
||||
# Waits for a just-started guest's network to actually come up, then turns
|
||||
# kvm_amd cpuid_passthrough on.
|
||||
# Keeps the kvm_amd cpuid_passthrough switch correct across a guest's whole life,
|
||||
# in-guest reboots included.
|
||||
#
|
||||
# The readiness signal is the guest's own tap RX counter: it is fresh every boot,
|
||||
# it needs nothing enabled inside the guest (no SSH, no RDP, no agent), and it only
|
||||
# moves once the guest's NIC driver has really loaded - which is well past the CPU
|
||||
# enumeration that the switch must not change under. A DHCP lease left over from a
|
||||
# previous boot cannot trip it early.
|
||||
# The switch must be OFF (N) while the guest enumerates CPUID at boot or Windows
|
||||
# hangs, and ON (Y) once it is up, where it clears the timer detection. The libvirt
|
||||
# hook only fires at VM start and stop, so a guest-initiated reboot would otherwise
|
||||
# re-enumerate with the switch still Y and hang. This watcher drops it to N on every
|
||||
# QEMU RESET and raises it again once the guest's NIC is back up.
|
||||
#
|
||||
# Readiness signal: the guest tap's rx_packets counter growing past a baseline. It
|
||||
# only moves once the guest NIC driver has loaded, well past CPU enumeration, and it
|
||||
# needs nothing enabled inside the guest. The counter is cumulative and does NOT
|
||||
# reset on an in-guest reboot, so readiness is growth past the value captured at the
|
||||
# reset, not an absolute threshold.
|
||||
#
|
||||
# Launched as a transient systemd unit by the cpuid-passthrough hook, so it is free
|
||||
# to call virsh (the hook itself must not - that deadlocks libvirtd).
|
||||
@@ -17,25 +23,49 @@ PARAM=/sys/module/kvm_amd/parameters
|
||||
V="virsh -c qemu:///system"
|
||||
|
||||
running() { [ "$($V domstate "$DOMAIN" 2>/dev/null)" = running ]; }
|
||||
rx() { cat "$RX" 2>/dev/null || echo 0; }
|
||||
|
||||
tap=""
|
||||
i=0
|
||||
for _ in $(seq 1 65); do # ~195 s cap, then flip anyway if still up
|
||||
set_N() { echo N > "$PARAM/cpuid_passthrough"; }
|
||||
set_Y() { printf '%s' "$BRAND" > "$PARAM/brand_string"; echo Y > "$PARAM/cpuid_passthrough"; }
|
||||
|
||||
# Wait until the tap rx counter grows at least 4 past $1 (guest NIC driver up again).
|
||||
# The iteration floor keeps a stray pre-OS packet (a UEFI netboot attempt) from
|
||||
# tripping the flip before the guest is even past its interrupt and timer setup.
|
||||
wait_net_up() {
|
||||
local base=$1 i=0
|
||||
for _ in $(seq 1 100); do # ~300 s cap, then raise anyway if still up
|
||||
i=$((i + 1))
|
||||
running || { sleep 3; continue; } # not "running" yet at prepare time - wait
|
||||
[ -z "$tap" ] && tap=$($V domiflist "$DOMAIN" 2>/dev/null |
|
||||
awk '$1 ~ /^(vnet|tap|macvtap)/ {print $1; exit}')
|
||||
rx="/sys/class/net/$tap/statistics/rx_packets"
|
||||
# the iteration floor keeps a stray pre-OS packet (a UEFI netboot attempt) from
|
||||
# tripping the flip before the guest is even past its interrupt and timer setup
|
||||
if [ "$i" -ge 4 ] && [ -n "$tap" ] && [ -r "$rx" ] &&
|
||||
[ "$(cat "$rx" 2>/dev/null || echo 0)" -ge 4 ]; then
|
||||
break
|
||||
fi
|
||||
running || return 1
|
||||
[ "$i" -ge 4 ] && [ -n "$tap" ] && [ -r "$RX" ] &&
|
||||
[ "$(rx)" -ge "$((base + 4))" ] && return 0
|
||||
sleep 3
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
running || exit 0 # guest went away before it came up
|
||||
printf '%s' "$BRAND" > "$PARAM/brand_string"
|
||||
echo Y > "$PARAM/cpuid_passthrough"
|
||||
# The hook arms this watcher at prepare, before QEMU starts, so the domain is not
|
||||
# yet running and its tap does not exist. Wait for the domain, then read the tap;
|
||||
# otherwise the first running check in wait_net_up bails and the flip never happens.
|
||||
for _ in $(seq 1 120); do running && break; sleep 1; done
|
||||
running || exit 0
|
||||
tap=$($V domiflist "$DOMAIN" 2>/dev/null | awk '$1 ~ /^(vnet|tap|macvtap)/ {print $1; exit}')
|
||||
RX="/sys/class/net/$tap/statistics/rx_packets"
|
||||
|
||||
# initial cold boot: wait for the network, then harden
|
||||
wait_net_up 0
|
||||
running || exit 0
|
||||
set_Y
|
||||
logger -t vfio-cpuid "$DOMAIN network up: cpuid_passthrough=Y"
|
||||
|
||||
# every in-guest reboot fires a QEMU RESET: drop to N for the re-enumeration, then
|
||||
# raise it again once the guest's NIC is back. --loop streams one line per reset.
|
||||
$V qemu-monitor-event --domain "$DOMAIN" --event RESET --loop 2>/dev/null | while read -r _; do
|
||||
running || continue
|
||||
base=$(rx)
|
||||
set_N
|
||||
logger -t vfio-cpuid "$DOMAIN reset: cpuid_passthrough=N for re-enumeration"
|
||||
wait_net_up "$base" || continue
|
||||
running || continue
|
||||
set_Y
|
||||
logger -t vfio-cpuid "$DOMAIN back up: cpuid_passthrough=Y"
|
||||
done
|
||||
|
||||
@@ -57,6 +57,24 @@ group_members() {
|
||||
done
|
||||
}
|
||||
|
||||
# The functions to hand to the guest with GPU $1: every device in its IOMMU group
|
||||
# (mandatory for vfio), plus any sibling function of the same PCI device that is
|
||||
# itself a display or HDMI/DP audio controller. An APU parks its PSP and USB
|
||||
# controllers on the same PCI device in separate IOMMU groups - those are the
|
||||
# host's, so siblings are filtered by class and never pulled in blindly.
|
||||
passthrough_devs() {
|
||||
local d="$1" m cls
|
||||
{
|
||||
group_members "$d"
|
||||
for m in "/sys/bus/pci/devices/${d%.*}".*; do
|
||||
[ -e "$m" ] || continue
|
||||
m=$(basename "$m")
|
||||
cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)
|
||||
case "$cls" in 0x03*|0x0403*) echo "$m";; esac
|
||||
done
|
||||
} | sort -u
|
||||
}
|
||||
|
||||
driver_of() {
|
||||
local l
|
||||
l=$(readlink -f "/sys/bus/pci/devices/$1/driver" 2>/dev/null)
|
||||
@@ -79,7 +97,7 @@ drives_display() {
|
||||
|
||||
ids_of() { # vendor:device for vfio-pci binding
|
||||
local d
|
||||
for d in $(group_members "$1"); do
|
||||
for d in $(passthrough_devs "$1"); do
|
||||
local cls; cls=$(cat "/sys/bus/pci/devices/$d/class" 2>/dev/null)
|
||||
# only real functions of the card: skip bridges (class 0x0604xx)
|
||||
case "$cls" in 0x0604*) continue;; esac
|
||||
@@ -219,7 +237,7 @@ prepare)
|
||||
|
||||
# 1. stop whatever is holding the DRM device
|
||||
dm=$(systemctl list-units --type=service --state=running --no-legend 2>/dev/null |
|
||||
awk '{print $1}' | grep -xE '(gdm|sddm|lightdm|lxdm|greetd|display-manager)\.service' | head -1)
|
||||
awk '{print $1}' | grep -xE '((gdm|sddm|lightdm|lxdm|greetd|ly|emptty|lemurs)(@[a-z0-9-]+)?|display-manager)\.service' | head -1)
|
||||
if [ -n "$dm" ]; then
|
||||
echo "$dm" > "$STATE/dm"
|
||||
log "stopping $dm"
|
||||
@@ -303,7 +321,7 @@ apply() {
|
||||
"$v" -c qemu:///system dumpxml --inactive "$dom" > "$backup/$dom.gpu-current.xml"
|
||||
grep -q 'ua-vfionative-gpu' "$backup/$dom.gpu-current.xml" || cp "$backup/$dom.gpu-current.xml" "$backup/$dom.before-gpu.xml"
|
||||
local devs="" m
|
||||
for m in $( { group_members "$d"; ls -d "/sys/bus/pci/devices/${d%.*}".* | xargs -n1 basename; } | sort -u); do
|
||||
for m in $(passthrough_devs "$d"); do
|
||||
case "$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)" in 0x0604*) continue;; esac
|
||||
IFS=':.' read -r dm bs sl fn <<< "$m"
|
||||
devs+=" <hostdev mode='subsystem' type='pci' managed='yes'>\n <source>\n"
|
||||
@@ -330,7 +348,7 @@ xml() {
|
||||
echo "Add this to the domain, inside <devices>. Every device in the GPU's"
|
||||
echo "IOMMU group has to go together:"
|
||||
echo
|
||||
for m in $(group_members "$d"); do
|
||||
for m in $(passthrough_devs "$d"); do
|
||||
local cls; cls=$(cat "/sys/bus/pci/devices/$m/class" 2>/dev/null)
|
||||
case "$cls" in 0x0604*) continue;; esac
|
||||
IFS=':. ' read -r dom bus slot fn <<< "$(echo "$m" | tr ':.' ' ')"
|
||||
|
||||
Reference in New Issue
Block a user