855 lines
47 KiB
Bash
Executable File
855 lines
47 KiB
Bash
Executable File
#!/bin/bash
|
|
# Configure an existing libvirt domain for platform fidelity and low latency.
|
|
#
|
|
# @@SELFNAME@@ interactive
|
|
# @@SELFNAME@@ -d win11 -p full -c 8 -m 16 scripted
|
|
#
|
|
# -d domain (default win11)
|
|
# -p tuned | native | full (default: full)
|
|
# -c guest cores (SMT doubles this into vCPUs)
|
|
# -m guest RAM in GiB
|
|
# -s on | off Secure Boot with enrolled keys (default on)
|
|
# -u none | auto | <vid:pid,...|0000:bb:dd.f,...> USB passthrough
|
|
# -r randomize the hardware identity (serials, MAC, memory modules)
|
|
# -y no prompts
|
|
#
|
|
# Takes a plain libvirt domain to the tuned, corrected state in one pass:
|
|
# detects the host CPU layout and maps vCPUs onto real SMT pairs within one
|
|
# cache domain, keeps the emulator off the vCPU cores, moves the disk to
|
|
# emulated NVMe, replaces the virtio device set, wires Secure Boot with a
|
|
# generated key store, SMBIOS and ACPI identity, and the patched QEMU for the
|
|
# full level. Backs the domain up first; idempotent.
|
|
set -euo pipefail
|
|
|
|
# Run bare it is an interview: every choice is asked, with a default. Every flag
|
|
# answers one question in advance, and -y takes every default.
|
|
DOM=win11; PROFILE=""; CORES=""; RAMG=""; ASSUME=0; SECBOOT=""; USBSPEC=""; RANDOMIZE=""
|
|
while getopts "d:p:c:m:s:u:ryh" o; do case $o in
|
|
d) DOM=$OPTARG;; p) PROFILE=$OPTARG;; c) CORES=$OPTARG;;
|
|
m) RAMG=$OPTARG;; s) SECBOOT=$OPTARG;; u) USBSPEC=$OPTARG;; r) RANDOMIZE=1;; y) ASSUME=1;;
|
|
# print the header comment block, however long it grows
|
|
h) sed -n '2,/^[^#]/p' "$0" | sed "\$d; s/^# \?//; s|@@SELFNAME@@|$(basename "$0")|g"; exit 0;;
|
|
esac; done
|
|
|
|
C=(virsh -c qemu:///system)
|
|
BACKUP="$HOME/vfio-native-backup"
|
|
# Locate the project data (patches, benchmark sources), whether running from a
|
|
# checkout or installed as a package.
|
|
for d in "$(cd "$(dirname "$0")/.." 2>/dev/null && pwd)" /usr/share/vfio-native; do
|
|
if [ -d "$d/bench" ] || [ -d "$d/patches" ]; then SELF="$d"; break; fi
|
|
done
|
|
SELF="${SELF:-$(cd "$(dirname "$0")/.." && pwd)}"
|
|
|
|
command -v virsh >/dev/null || { echo "virsh not found"; exit 1; }
|
|
"${C[@]}" dominfo "$DOM" >/dev/null 2>&1 || { echo "no such domain: $DOM"; exit 1; }
|
|
[ "$("${C[@]}" domstate "$DOM")" = "shut off" ] || { echo "shut $DOM down first"; exit 1; }
|
|
case "$SECBOOT" in on|off|"") ;; *) echo "-s takes on or off"; exit 1;; esac
|
|
ask() { # question default -> answer (default when -y or empty input)
|
|
local a; [ "$ASSUME" = 1 ] && { echo "$2"; return; }
|
|
read -rp "$1 [$2]: " a; echo "${a:-$2}"
|
|
}
|
|
DATA="${XDG_DATA_HOME:-$HOME/.local/share}/vfio-native/$DOM"
|
|
|
|
# ---------------------------------------------------------------- host CPU ---
|
|
VENDOR=$(awk -F': ' '/vendor_id/{print $2; exit}' /proc/cpuinfo)
|
|
CPUNAME=$(awk -F': ' '/model name/{print $2; exit}' /proc/cpuinfo)
|
|
FAMILY=$(awk -F': ' '/^cpu family/{print $2; exit}' /proc/cpuinfo)
|
|
MODELNO=$(awk -F': ' '/^model\t/{print $2; exit}' /proc/cpuinfo)
|
|
# not nproc: it honours the shell's own affinity, which the isolation hook narrows while a guest runs
|
|
HOST_THREADS=$(ls -d /sys/devices/system/cpu/cpu[0-9]* | wc -l)
|
|
|
|
# Physical cores grouped by last-level cache: "l3id core l3sizeK". On Zen 3 and
|
|
# later each group is one CCD of 8 cores; on Zen 1/2 it is one CCX of 4, so
|
|
# there are two groups per CCD; on a monolithic chip there is a single group.
|
|
primaries() {
|
|
for d in /sys/devices/system/cpu/cpu[0-9]*; do
|
|
c=${d##*/cpu}
|
|
[ "$c" = "$(cut -d, -f1 < "$d/topology/thread_siblings_list")" ] || continue
|
|
l3=$(cat "$d/cache/index3/id" 2>/dev/null) || l3=0
|
|
sz=$(cat "$d/cache/index3/size" 2>/dev/null) || sz=0K
|
|
echo "$l3 $c ${sz%K}"
|
|
done | sort -n -k1,1 -k2,2
|
|
}
|
|
# The guest's domain is the one with the most L3. On a 3D V-Cache part one CCD
|
|
# has three times the other's, and that is the one a latency-sensitive guest
|
|
# wants; everywhere else the sizes tie and the lowest id wins.
|
|
BEST=$(primaries | sort -k3,3nr -k1,1n | head -1 | awk '{print $1}')
|
|
BEST_L3K=$(primaries | awk -v g="$BEST" '$1==g{print $3; exit}')
|
|
node_of() { local n; n=$(ls -d "/sys/devices/system/cpu/cpu$1"/node* 2>/dev/null | head -1); echo "${n##*/node}"; }
|
|
# Intel 12th gen and later mix P-cores and E-cores. A latency-sensitive guest
|
|
# thread landing on an E-core shows up as hitching, so the guest gets P-cores
|
|
# and the emulator gets the E-cores, which is exactly what they are good for.
|
|
HYBRID=0
|
|
if [ -d /sys/devices/cpu_core ] && [ -d /sys/devices/cpu_atom ]; then
|
|
HYBRID=1
|
|
expand() { tr ',' '\n' < "$1" | while read -r r; do
|
|
case $r in *-*) seq "${r%-*}" "${r#*-}";; *) echo "$r";; esac; done; }
|
|
mapfile -t PCPUS < <(expand /sys/devices/cpu_core/cpus)
|
|
mapfile -t ECPUS < <(expand /sys/devices/cpu_atom/cpus)
|
|
mapfile -t GRP0 < <(for c in "${PCPUS[@]}"; do
|
|
[ "$c" = "$(cut -d, -f1 < "/sys/devices/system/cpu/cpu$c/topology/thread_siblings_list")" ] && echo "$c"; done)
|
|
REST=("${ECPUS[@]}")
|
|
else
|
|
mapfile -t GRP0 < <(primaries | awk -v g="$BEST" '$1==g{print $2}')
|
|
mapfile -t REST < <(primaries | awk -v g="$BEST" '$1!=g{print $2}')
|
|
fi
|
|
NGROUPS=$(primaries | awk '{print $1}' | sort -u | wc -l)
|
|
NNODES=$(ls -d /sys/devices/system/node/node[0-9]* 2>/dev/null | wc -l)
|
|
GNODE=$(node_of "${GRP0[0]}")
|
|
SMT=$([ "$HOST_THREADS" -gt "$(primaries | wc -l)" ] && echo 2 || echo 1)
|
|
# SMT sibling of a core, empty when SMT is off; and the whole pair as a cpuset
|
|
sib() { cut -d, -f2 -s < "/sys/devices/system/cpu/cpu$1/topology/thread_siblings_list"; }
|
|
pair() { cat "/sys/devices/system/cpu/cpu$1/topology/thread_siblings_list"; }
|
|
|
|
echo "host: $CPUNAME"
|
|
if [ "$HYBRID" = 1 ]; then
|
|
echo "layout: Intel hybrid - P-cores ${GRP0[*]}, E-cores ${REST[*]}"
|
|
echo " guest gets P-cores; emulator and IO go on E-cores"
|
|
else
|
|
echo "layout: $(primaries | wc -l) cores / $HOST_THREADS threads, $NGROUPS cache domain(s), SMT $([ $SMT = 2 ] && echo on || echo off)"
|
|
echo " guest domain: cores ${GRP0[*]} (L3 $(( BEST_L3K / 1024 )) MiB)"
|
|
[ "${#REST[@]}" -gt 0 ] && echo " remaining: cores ${REST[*]}"
|
|
if [ "$(primaries | awk '{print $3}' | sort -u | wc -l)" -gt 1 ]; then
|
|
echo " L3 is asymmetric - picked the larger (3D V-Cache) domain for the guest"
|
|
fi
|
|
[ "$NNODES" -gt 1 ] && echo " $NNODES NUMA nodes - guest memory will be pinned to node $GNODE"
|
|
fi
|
|
HOST_RAM=$(( $(awk '/MemTotal/{print $2}' /proc/meminfo) / 1024 / 1024 ))
|
|
echo "ram: ${HOST_RAM} GiB total"
|
|
echo
|
|
|
|
# ---------------------------------------------------------------- profile ---
|
|
if [ -z "$PROFILE" ]; then
|
|
cat <<'EOF'
|
|
Fidelity levels. All three get the same performance tuning - the platform
|
|
corrections cost nothing measurable, so the level only changes how closely the
|
|
guest matches real hardware.
|
|
|
|
tuned Performance tuning only. Hypervisor visible, Hyper-V enlightenments
|
|
on.
|
|
|
|
native Tuning plus the domain-level platform corrections: hypervisor CPUID
|
|
bit cleared, KVM signature off, CPU feature and firmware identity
|
|
corrected. Needs no patched binaries, so it survives any host update.
|
|
|
|
full Tuning plus corrections plus the patched QEMU and patched KVM
|
|
modules. The lowest score. Costs a module rebuild after every kernel
|
|
upgrade.
|
|
|
|
EOF
|
|
read -rp "level [full]: " PROFILE; PROFILE=${PROFILE:-full}
|
|
fi
|
|
case "$PROFILE" in
|
|
tuned|native|full) ;;
|
|
performance) PROFILE=tuned;;
|
|
*) echo "unknown level: $PROFILE"; exit 1;;
|
|
esac
|
|
|
|
# ---------------------------------------------------------------- resources ---
|
|
MAXC=${#GRP0[@]}
|
|
# Keep two cores back for the emulator and IO threads. They can come from another
|
|
# cache domain if there is one, otherwise they come out of the guest's share.
|
|
if [ "${#REST[@]}" -ge 2 ]; then RESERVE=0; else RESERVE=2; fi
|
|
SUGGEST=$(( MAXC - RESERVE ))
|
|
|
|
if [ -z "$CORES" ]; then
|
|
echo
|
|
TOTAL=$(primaries | wc -l)
|
|
echo "Guest cores."
|
|
echo " $SUGGEST keeps the guest inside one cache domain - lowest memory"
|
|
echo " latency, best for latency-sensitive workloads."
|
|
echo " up to $(( TOTAL - 4 )) is fine for CPU-heavy work, at the cost of"
|
|
echo " higher L3 latency. Leave the host at least 4 cores either way."
|
|
read -rp "cores [$SUGGEST]: " CORES; CORES=${CORES:-$SUGGEST}
|
|
fi
|
|
[ "$CORES" -ge 1 ] 2>/dev/null || { echo "cores must be a number"; exit 1; }
|
|
VCPUS=$(( CORES * SMT ))
|
|
|
|
# Windows calibrates the TSC at boot, and that calibration is a race: if the
|
|
# host cannot schedule the vCPU threads cleanly through it, Windows gives up on
|
|
# the TSC and QueryPerformanceCounter costs ~1300 ns instead of ~15 for the life
|
|
# of that boot. Measured pass rate over 4 cold boots each, 16 physical cores:
|
|
# host keeps 4+ cores -> 4/4
|
|
# host keeps 2 cores -> 3/4
|
|
# host keeps 0 cores -> 2/4
|
|
# So the rule is headroom, not a vCPU ceiling.
|
|
HOSTCORES=$(( $(primaries | wc -l) - CORES ))
|
|
if [ "$HOSTCORES" -lt 4 ]; then
|
|
echo
|
|
echo "WARNING: this leaves the host only $HOSTCORES physical core(s)."
|
|
echo "Windows calibrates the TSC at boot and that calibration needs the host"
|
|
echo "able to schedule cleanly. With this little headroom it fails on some"
|
|
echo "boots, and when it does, every timing call in the guest costs ~1300 ns"
|
|
echo "instead of ~15 for the rest of that boot. Measured: 2 of 4 boots failed"
|
|
echo "with no headroom at all."
|
|
echo
|
|
echo "Leave 4 physical cores free and it passed 4 of 4. Check with"
|
|
echo "vm-native-verify after booting - if QPC reads over 1000 ns, reboot."
|
|
[ "$ASSUME" = 1 ] || { read -rp "continue anyway? [y/N]: " a; [ "$a" = y ] || exit 1; }
|
|
fi
|
|
if [ "$CORES" -gt "$(( MAXC - RESERVE ))" ]; then
|
|
echo "NOTE: $CORES cores spans more than one cache domain. Expect L3 latency"
|
|
echo " around 17 ns instead of 10. Worth it for throughput work, not for"
|
|
echo " latency-sensitive workloads."
|
|
fi
|
|
|
|
if [ -z "$RAMG" ]; then
|
|
DEF=$(( HOST_RAM / 2 )); [ "$DEF" -gt 32 ] && DEF=32
|
|
read -rp "guest RAM in GiB [$DEF]: " RAMG; RAMG=${RAMG:-$DEF}
|
|
fi
|
|
[ "$RAMG" -ge 2 ] 2>/dev/null || { echo "ram must be a number >= 2"; exit 1; }
|
|
|
|
# ------------------------------------------------------- guest CPU identity ---
|
|
# The declared part must really have this many threads, and it must be from the
|
|
# same generation as the host, or the brand string contradicts the family, model
|
|
# and cache leaves the guest reads straight from the silicon. Every name here is
|
|
# a real desktop SKU whose thread count VMAware's own database agrees with; a
|
|
# host generation with no matching SKU keeps its own name and says so.
|
|
sku() { # vendor family model threads vcache -> brand string, or empty
|
|
local gen="" n=$4
|
|
case "$1:$2:$3" in
|
|
AuthenticAMD:23:1|AuthenticAMD:23:17) gen=zen1;;
|
|
AuthenticAMD:23:8|AuthenticAMD:23:24) gen=zenp;;
|
|
AuthenticAMD:23:113) gen=zen2;;
|
|
AuthenticAMD:25:33) gen=zen3;;
|
|
AuthenticAMD:25:97) gen=zen4;;
|
|
AuthenticAMD:26:68) gen=zen5;;
|
|
GenuineIntel:6:165) gen=cml;;
|
|
GenuineIntel:6:167) gen=rkl;;
|
|
GenuineIntel:6:151|GenuineIntel:6:154) gen=adl;;
|
|
GenuineIntel:6:183|GenuineIntel:6:191) gen=rpl;;
|
|
esac
|
|
case "$gen:$n:$5" in
|
|
zen1:12:*) echo "AMD Ryzen 5 1600X Six-Core Processor";;
|
|
zen1:16:*) echo "AMD Ryzen 7 1800X Eight-Core Processor";;
|
|
zenp:12:*) echo "AMD Ryzen 5 2600X Six-Core Processor";;
|
|
zenp:16:*) echo "AMD Ryzen 7 2700X Eight-Core Processor";;
|
|
zen2:12:*) echo "AMD Ryzen 5 3600X 6-Core Processor";;
|
|
zen2:16:*) echo "AMD Ryzen 7 3700X 8-Core Processor";;
|
|
zen2:24:*) echo "AMD Ryzen 9 3900X 12-Core Processor";;
|
|
zen2:32:*) echo "AMD Ryzen 9 3950X 16-Core Processor";;
|
|
zen3:12:*) echo "AMD Ryzen 5 5600X 6-Core Processor";;
|
|
zen3:16:1) echo "AMD Ryzen 7 5800X3D 8-Core Processor";;
|
|
zen3:16:*) echo "AMD Ryzen 7 5800X 8-Core Processor";;
|
|
zen3:24:*) echo "AMD Ryzen 9 5900X 12-Core Processor";;
|
|
zen3:32:*) echo "AMD Ryzen 9 5950X 16-Core Processor";;
|
|
zen4:12:*) echo "AMD Ryzen 5 7600X 6-Core Processor";;
|
|
zen4:16:1) echo "AMD Ryzen 7 7800X3D 8-Core Processor";;
|
|
zen4:16:*) echo "AMD Ryzen 7 7700X 8-Core Processor";;
|
|
zen4:24:*) echo "AMD Ryzen 9 7900X 12-Core Processor";;
|
|
zen4:32:1) echo "AMD Ryzen 9 7950X3D 16-Core Processor";;
|
|
zen4:32:*) echo "AMD Ryzen 9 7950X 16-Core Processor";;
|
|
zen5:12:*) echo "AMD Ryzen 5 9600X 6-Core Processor";;
|
|
zen5:16:1) echo "AMD Ryzen 7 9800X3D 8-Core Processor";;
|
|
zen5:16:*) echo "AMD Ryzen 7 9700X 8-Core Processor";;
|
|
zen5:24:*) echo "AMD Ryzen 9 9900X 12-Core Processor";;
|
|
zen5:32:1) echo "AMD Ryzen 9 9950X3D 16-Core Processor";;
|
|
zen5:32:*) echo "AMD Ryzen 9 9950X 16-Core Processor";;
|
|
cml:12:*) echo "Intel(R) Core(TM) i5-10400 CPU @ 2.90GHz";;
|
|
cml:16:*) echo "Intel(R) Core(TM) i7-10700K CPU @ 3.80GHz";;
|
|
cml:20:*) echo "Intel(R) Core(TM) i9-10900K CPU @ 3.70GHz";;
|
|
rkl:12:*) echo "Intel(R) Core(TM) i5-11400 @ 2.60GHz";;
|
|
rkl:16:*) echo "Intel(R) Core(TM) i7-11700K @ 3.60GHz";;
|
|
adl:8:*) echo "12th Gen Intel(R) Core(TM) i3-12100";;
|
|
adl:12:*) echo "12th Gen Intel(R) Core(TM) i5-12400";;
|
|
rpl:8:*) echo "13th Gen Intel(R) Core(TM) i3-13100";;
|
|
esac
|
|
}
|
|
# A guest on a 3D V-Cache domain reads that L3 size straight from CPUID, so it
|
|
# has to claim the X3D part to stay consistent with it.
|
|
VCACHE=$([ "$BEST_L3K" -ge 65536 ] && echo 1 || echo 0)
|
|
MODEL=$(sku "$VENDOR" "$FAMILY" "$MODELNO" "$VCPUS" "$VCACHE")
|
|
if [ "$PROFILE" != tuned ] && [ -z "$MODEL" ]; then
|
|
echo
|
|
echo "NOTE: no desktop part of this CPU's generation (family $FAMILY, model $MODELNO)"
|
|
echo " has exactly $VCPUS threads, so the guest will keep the host's CPU name and"
|
|
echo " the thread count will not match it. Pick a core count that maps to a"
|
|
echo " real SKU to avoid that, or accept the mismatch."
|
|
fi
|
|
|
|
# ---------------------------------------------------------- host identity ---
|
|
# Every value the guest can read comes from one per-domain file, generated once
|
|
# from the host's own strings with fresh serials, so two people running this
|
|
# tool do not share a fingerprint. -r regenerates it. Manufacturer and product
|
|
# strings stay real; only serials, the MAC and the memory module change.
|
|
dmi() { cat "/sys/class/dmi/id/$1" 2>/dev/null || echo ""; }
|
|
rnd() { tr -dc "$2" < /dev/urandom | head -c "$1"; }
|
|
gen_identity() {
|
|
local bv; bv=$(dmi board_vendor)
|
|
local mem_man="Kingston" mem_part="KF556C40BB-16" mem_speed=5600
|
|
# the host's real module strings need dmidecode as root; take them when sudo is cached
|
|
if command -v dmidecode >/dev/null && sudo -n true 2>/dev/null; then
|
|
mem_man=$(sudo -n dmidecode -t 17 2>/dev/null | awk -F': ' '/Manufacturer:/{print $2; exit}')
|
|
mem_part=$(sudo -n dmidecode -t 17 2>/dev/null | awk -F': ' '/Part Number:/{gsub(/ +$/,"",$2); print $2; exit}')
|
|
mem_speed=$(sudo -n dmidecode -t 17 2>/dev/null | awk '/Configured Memory Speed:/{print $4; exit}')
|
|
fi
|
|
{
|
|
echo "# generated by vm-native-setup $(date -I); edit freely, -r regenerates"
|
|
echo "NVME_SERIAL='S6PXNS0W$(rnd 7 'A-Z0-9')'"
|
|
echo "MAC='a0:36:9f:$(rnd 2 'a-f0-9'):$(rnd 2 'a-f0-9'):$(rnd 2 'a-f0-9')'"
|
|
# ASUS DIY boards report the literal placeholders for system and chassis, and
|
|
# a 15-digit board serial; anything else gets a plain alphanumeric serial
|
|
if [ "$bv" = "ASUSTeK COMPUTER INC." ]; then
|
|
echo "SYS_SERIAL='System Serial Number'"
|
|
echo "BOARD_SERIAL='$(date +%y%m)$(rnd 11 '0-9')'"
|
|
echo "CHASSIS_SERIAL='Default string'"
|
|
else
|
|
echo "SYS_SERIAL='$(rnd 10 'A-Z0-9')'"
|
|
echo "BOARD_SERIAL='$(rnd 12 'A-Z0-9')'"
|
|
echo "CHASSIS_SERIAL='$(rnd 10 'A-Z0-9')'"
|
|
fi
|
|
echo "MEM_MANUFACTURER='${mem_man:-Kingston}'"
|
|
echo "MEM_PART='${mem_part:-KF556C40BB-16}'"
|
|
echo "MEM_SPEED='${mem_speed:-5600}'"
|
|
echo "MEM_SERIAL='$(rnd 8 'A-F0-9')'"
|
|
} > "$DATA/identity.env"
|
|
}
|
|
mkdir -p "$DATA"
|
|
if [ -z "$RANDOMIZE" ] && [ "$PROFILE" != tuned ]; then
|
|
echo
|
|
echo "Hardware identity. Everyone running this tool with the same fixed values shares one"
|
|
echo "fingerprint. -r writes this deployment its own serials, MAC and memory module to"
|
|
echo "$DATA/identity.env and puts them in the domain (SMBIOS, disk, NIC)."
|
|
if [ -f "$DATA/identity.env" ]; then
|
|
echo "An identity file exists from an earlier run; 'y' replaces it with fresh values."
|
|
fi
|
|
echo "Windows may ask to re-activate after the board serial and MAC change."
|
|
a=$(ask "randomize the hardware identity? (y/n)" n); RANDOMIZE=$([ "$a" = y ] && echo 1 || echo 0)
|
|
fi
|
|
RANDOMIZE=${RANDOMIZE:-0}
|
|
if [ "$RANDOMIZE" = 1 ] || [ ! -f "$DATA/identity.env" ]; then gen_identity; fi
|
|
# shellcheck disable=SC1091
|
|
. "$DATA/identity.env"
|
|
|
|
# --------------------------------------------------------- USB passthrough ---
|
|
# Passing a whole controller is cleaner than passing devices one by one: no
|
|
# emulated hub, no hotplug, nothing in the guest that says "redirected". It is
|
|
# only clean when everything behind that controller should go to the guest and
|
|
# its IOMMU group holds nothing else, so the default recommendation is: whole
|
|
# controller for a controller that carries only keyboard and mouse, individual
|
|
# devices otherwise.
|
|
usb_devices() { # busnum vid:pid kind product
|
|
local d v pr kind
|
|
for d in /sys/bus/usb/devices/[0-9]*-[0-9]*; do
|
|
[[ ${d##*/} == *:* ]] && continue
|
|
[ -f "$d/idVendor" ] || continue
|
|
v="$(cat "$d/idVendor"):$(cat "$d/idProduct")"
|
|
[ "$(cat "$d/bDeviceClass")" = 09 ] && continue # hubs stay
|
|
kind=other
|
|
for i in "$d"/*:*; do
|
|
[ -f "$i/bInterfaceProtocol" ] || continue
|
|
[ "$(cat "$i/bInterfaceClass")" = 03 ] || continue
|
|
case "$(cat "$i/bInterfaceProtocol")" in 01) kind=keyboard;; 02) [ "$kind" = keyboard ] || kind=mouse;; esac
|
|
done
|
|
pr=$(cat "$d/product" 2>/dev/null || echo "?")
|
|
echo "$(cat "$d/busnum") $v $kind $pr"
|
|
done
|
|
}
|
|
usb_ctrl_of_bus() { basename "$(readlink -f "/sys/bus/usb/devices/usb$1/..")"; }
|
|
group_clean() { # true if the IOMMU group of a PCI device holds only it and bridges
|
|
local g m; g=$(readlink -f "/sys/bus/pci/devices/$1/iommu_group") || return 1
|
|
for m in "$g"/devices/*; do
|
|
m=${m##*/}; [ "$m" = "$1" ] && continue
|
|
case "$(cat "/sys/bus/pci/devices/$m/class")" in 0x0604*) ;; *) return 1;; esac
|
|
done
|
|
}
|
|
USB_HOSTDEVS=""
|
|
usb_plan() { # prints the hostdev XML for a spec: auto | vid:pid,... | pci addrs
|
|
local spec="$1" item ctrl bus devs kinds
|
|
if [ "$spec" = auto ]; then
|
|
# controllers whose every device is a keyboard or mouse go whole; the
|
|
# rest of the input devices go one by one
|
|
for ctrl in $(usb_devices | while read -r bus _ _ _; do usb_ctrl_of_bus "$bus"; done | sort -u); do
|
|
devs=$(usb_devices | while read -r bus vp kind pr; do [ "$(usb_ctrl_of_bus "$bus")" = "$ctrl" ] && echo "$vp $kind $pr"; done)
|
|
kinds=$(echo "$devs" | awk '{print $2}' | sort -u | tr '\n' ' ')
|
|
case "$kinds" in
|
|
"keyboard "|"mouse "|"keyboard mouse ")
|
|
if group_clean "$ctrl"; then echo "pci $ctrl"; continue; fi;;
|
|
esac
|
|
echo "$devs" | awk '$2=="keyboard"||$2=="mouse"{print "usb", $1}'
|
|
done
|
|
else
|
|
for item in ${spec//,/ }; do
|
|
case "$item" in
|
|
0000:*) echo "pci $item";;
|
|
*:*) echo "usb $item";;
|
|
*) echo "unknown USB spec '$item' - want vid:pid or 0000:bb:dd.f" >&2; exit 1;;
|
|
esac
|
|
done
|
|
fi | sort -u | while read -r kind id; do
|
|
if [ "$kind" = usb ]; then
|
|
printf " <hostdev mode='subsystem' type='usb' managed='yes'>\n <source>\n <vendor id='0x%s'/>\n <product id='0x%s'/>\n </source>\n <alias name='ua-vfionative-usb-%s-%s'/>\n </hostdev>\n" "${id%:*}" "${id#*:}" "${id%:*}" "${id#*:}"
|
|
else
|
|
IFS=':.' read -r dm bs sl fn <<< "$id"
|
|
printf " <hostdev mode='subsystem' type='pci' managed='yes'>\n <source>\n <address domain='0x%s' bus='0x%s' slot='0x%s' function='0x%s'/>\n </source>\n <alias name='ua-vfionative-pci-%s-%s-%s-%s'/>\n </hostdev>\n" "$dm" "$bs" "$sl" "$fn" "$dm" "$bs" "$sl" "$fn"
|
|
fi
|
|
done
|
|
}
|
|
if [ -z "$USBSPEC" ] && [ "$ASSUME" != 1 ]; then
|
|
echo
|
|
echo "USB devices on this host:"
|
|
usb_devices | while read -r bus vp kind pr; do
|
|
ctrl=$(usb_ctrl_of_bus "$bus"); printf ' %-9s %-9s %-40s controller %s%s\n' "$vp" "$kind" "$pr" "$ctrl" "$(group_clean "$ctrl" || echo ' (shared IOMMU group)')"
|
|
done
|
|
echo " auto = keyboard and mouse; a whole controller when only they sit on it"
|
|
echo " none = no USB passthrough"
|
|
echo " or a list: vid:pid,vid:pid,0000:bb:dd.f (a PCI address passes that whole controller)"
|
|
echo "The host loses whatever is passed for as long as the guest runs."
|
|
read -rp "USB passthrough [none]: " USBSPEC; USBSPEC=${USBSPEC:-none}
|
|
fi
|
|
USBSPEC=${USBSPEC:-none}
|
|
if [ "$USBSPEC" != none ] && [ "$USBSPEC" != auto ]; then
|
|
for item in ${USBSPEC//,/ }; do
|
|
[[ $item =~ ^[0-9a-f]{4}:[0-9a-f]{4}$ ]] || [[ $item =~ ^[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-7]$ ]] ||
|
|
{ echo "bad -u item '$item': want vid:pid or 0000:bb:dd.f"; exit 1; }
|
|
done
|
|
fi
|
|
[ "$USBSPEC" = none ] || USB_HOSTDEVS=$(usb_plan "$USBSPEC")
|
|
|
|
# The patched QEMU refuses a virtio disk, so the full level always converts; the
|
|
# native level runs stock QEMU and gets the choice.
|
|
CONVERT=0
|
|
if [ "$PROFILE" = full ]; then CONVERT=1
|
|
elif [ "$PROFILE" = native ]; then
|
|
echo
|
|
echo "Disk and devices. Moving the disks to emulated NVMe and replacing the virtio device set"
|
|
echo "(balloon, RNG, agent channels, virtiofs, virtio NIC and inputs) removes every device that"
|
|
echo "names the emulator. The guest must have stornvme boot-start first; see docs/GUEST-SETUP.md."
|
|
a=$(ask "move disks to NVMe and replace the virtio devices? (y/n)" y); [ "$a" = y ] && CONVERT=1
|
|
fi
|
|
|
|
# ------------------------------------------------------------- pin mapping ---
|
|
if [ "$RESERVE" = 0 ]; then
|
|
# Helper threads go on the far end of what is left, on the guest's own NUMA
|
|
# node when the host has more than one.
|
|
mapfile -t NEAR < <(for c in "${REST[@]}"; do [ "$(node_of "$c")" = "$GNODE" ] && echo "$c"; done)
|
|
[ "${#NEAR[@]}" -ge 2 ] || NEAR=("${REST[@]}")
|
|
EMU=${NEAR[$(( ${#NEAR[@]} - 2 ))]}; IOC=${NEAR[$(( ${#NEAR[@]} - 1 ))]}
|
|
GUEST=("${GRP0[@]:0:$CORES}")
|
|
else
|
|
GUEST=("${GRP0[@]:0:$CORES}")
|
|
EMU=${GRP0[$CORES]}; IOC=${GRP0[$(( CORES + 1 ))]}
|
|
fi
|
|
[ "${#GUEST[@]}" -eq "$CORES" ] || { echo "not enough physical cores for $CORES"; exit 1; }
|
|
|
|
# The level decides which QEMU the domain runs. "full" needs the patched build;
|
|
# the other two must not silently depend on it.
|
|
SYSQEMU=$(command -v qemu-system-x86_64 2>/dev/null || echo /usr/bin/qemu-system-x86_64)
|
|
PATCHED=/opt/qemu-native/bin/qemu-system-x86_64
|
|
if [ "$PROFILE" = full ] && [ -x "$PATCHED" ]; then
|
|
EMULATOR="$PATCHED"
|
|
else
|
|
EMULATOR="$SYSQEMU"
|
|
fi
|
|
|
|
# --------------------------------------------------------------- firmware ---
|
|
# libvirt's firmware autoselection has no descriptor with enrolled keys on most
|
|
# distributions, so a Secure Boot store with real keys is generated once per
|
|
# domain (OEM-named PK/KEK, Microsoft db) and named explicitly. Changing the
|
|
# template resets the domain's EFI variable store, which BitLocker notices.
|
|
LOADER=/usr/share/edk2/x64/OVMF_CODE.secboot.4m.fd
|
|
NVRAM_TPL=""; NVRAM_PATH="/var/lib/libvirt/qemu/nvram/${DOM}_VARS.fd"; NVRAM_RESET=0
|
|
if [ -z "$SECBOOT" ] && [ "$PROFILE" != tuned ]; then
|
|
echo
|
|
echo "Secure Boot. A key store with an OEM-named PK/KEK and the Microsoft db is generated"
|
|
echo "for the domain, so the guest sees Secure Boot enforcing, which some software refuses to run without."
|
|
echo "Enrolling it resets the domain's EFI variable store; BitLocker will notice."
|
|
SECBOOT=$(ask "enable Secure Boot with enrolled keys? (on/off)" on)
|
|
fi
|
|
SECBOOT=${SECBOOT:-off}
|
|
case "$SECBOOT" in on|off) ;; *) echo "-s takes on or off"; exit 1;; esac
|
|
if [ "$SECBOOT" = on ] && [ "$PROFILE" != tuned ]; then
|
|
if ! "${C[@]}" dumpxml --inactive "$DOM" | grep -q '<loader'; then
|
|
# a BIOS-installed guest does not boot from UEFI firmware; leave it alone
|
|
echo "NOTE: $DOM boots with legacy BIOS, so Secure Boot does not apply. Reinstall the guest"
|
|
echo " as UEFI to use it."; SECBOOT=off
|
|
elif ! command -v virt-fw-vars >/dev/null; then
|
|
echo "NOTE: virt-fw-vars (python-virt-firmware) not found - Secure Boot keys cannot be"
|
|
echo " enrolled, the firmware block is left as it is."
|
|
SECBOOT=off
|
|
elif [ ! -f "$LOADER" ]; then
|
|
echo "NOTE: $LOADER not found - firmware block left as it is."; SECBOOT=off
|
|
else
|
|
NVRAM_TPL="$DATA/OVMF_VARS.enrolled.fd"
|
|
if [ ! -f "$NVRAM_TPL" ]; then
|
|
virt-fw-vars -i /usr/share/edk2/x64/OVMF_VARS.4m.fd -o "$NVRAM_TPL" \
|
|
--enroll-generate "$(dmi board_vendor)" --secure-boot >/dev/null 2>&1 || { echo "virt-fw-vars failed"; exit 1; }
|
|
fi
|
|
OLD_TPL=$("${C[@]}" dumpxml --inactive "$DOM" | grep -o "<nvram template='[^']*'" | head -1 | cut -d"'" -f2)
|
|
OLD_PATH=$("${C[@]}" dumpxml --inactive "$DOM" | grep -o "<nvram [^>]*>[^<]*</nvram>" | head -1 | sed 's/.*>\([^<]*\)<.*/\1/')
|
|
# A new template only takes effect on a store that does not exist yet. When
|
|
# the domain already has one at the standard path it has to go; a store
|
|
# another domain also points at is never touched - that domain keeps it.
|
|
if [ "$OLD_TPL" != "$NVRAM_TPL" ] && [ "$OLD_PATH" = "$NVRAM_PATH" ]; then
|
|
for other in $("${C[@]}" list --all --name); do
|
|
[ "$other" = "$DOM" ] && continue
|
|
if "${C[@]}" dumpxml --inactive "$other" 2>/dev/null | grep -q "<nvram [^>]*>$NVRAM_PATH<"; then
|
|
echo "ERROR: $NVRAM_PATH is also the firmware store of domain '$other'. Give $DOM its"
|
|
echo " own nvram path first, or run with -s off."; exit 1
|
|
fi
|
|
done
|
|
NVRAM_RESET=1
|
|
fi
|
|
fi
|
|
fi
|
|
|
|
THP=$(cat /sys/kernel/mm/transparent_hugepage/enabled 2>/dev/null || echo "")
|
|
|
|
echo
|
|
echo "will configure $DOM:"
|
|
echo " level $PROFILE"
|
|
echo " vCPUs $VCPUS ($CORES cores x $SMT threads) on host cores ${GUEST[*]}"
|
|
echo " emulator core $EMU | iothread core $IOC"
|
|
echo " binary $EMULATOR"
|
|
echo " memory ${RAMG} GiB"
|
|
[ -n "$MODEL" ] && [ "$PROFILE" != tuned ] && echo " declares $MODEL"
|
|
if [ "$PROFILE" != tuned ]; then
|
|
[ "$CONVERT" = 1 ] && echo " disk emulated NVMe, serial $NVME_SERIAL; virtio devices replaced"
|
|
[ "$SECBOOT" = on ] && echo " firmware Secure Boot, keys enrolled from $NVRAM_TPL"
|
|
[ "$RANDOMIZE" = 1 ] && echo " identity SMBIOS serials, MAC $MAC and memory module from $DATA/identity.env"
|
|
[ "$PROFILE" = full ] && echo " display none - the guest console goes dark, use SSH or RDP"
|
|
fi
|
|
[ -n "$USB_HOSTDEVS" ] && echo " usb $(echo "$USB_HOSTDEVS" | grep -c '<hostdev') hostdev(s): $USBSPEC"
|
|
case "$THP" in *"[never]"*) echo; echo "WARNING: transparent hugepages are disabled on the host (/sys/kernel/mm/transparent_hugepage/enabled)."; echo " The guest will run on 4 KiB pages. Set it to madvise or always.";; esac
|
|
FREEM=$(awk '/MemFree/{print int($2/1024)}' /proc/meminfo)
|
|
if [ "$FREEM" -lt $(( RAMG * 1024 )) ]; then
|
|
echo
|
|
echo "NOTE: the host has ${FREEM} MiB free against a ${RAMG} GiB guest. Hugepages are allocated when the"
|
|
echo " guest first touches its memory, and fragmented or cached memory means 4 KiB pages instead."
|
|
echo " Before starting it: sudo sync; echo 3 | sudo tee /proc/sys/vm/drop_caches; echo 1 | sudo tee /proc/sys/vm/compact_memory"
|
|
fi
|
|
if [ "$NVRAM_RESET" = 1 ]; then
|
|
echo
|
|
echo "WARNING: the firmware variable store of $DOM will be reset to enrol the Secure Boot"
|
|
echo " keys. Boot entries are rebuilt by the firmware. If BitLocker is on in the"
|
|
echo " guest, suspend it first or it will ask for the recovery key."
|
|
fi
|
|
echo
|
|
if [ "$ASSUME" != 1 ]; then read -rp "apply? [y/N]: " a; [ "$a" = y ] || { echo aborted; exit 1; }; fi
|
|
|
|
mkdir -p "$BACKUP"
|
|
"${C[@]}" dumpxml "$DOM" > "$BACKUP/$DOM.before-setup.xml"
|
|
|
|
PINS=""
|
|
for ((k=0;k<CORES;k++)); do
|
|
PINS+=" <vcpupin vcpu='$((k*SMT))' cpuset='${GUEST[$k]}'/>\n"
|
|
[ "$SMT" = 2 ] && PINS+=" <vcpupin vcpu='$((k*SMT+1))' cpuset='$(sib "${GUEST[$k]}")'/>\n"
|
|
done
|
|
|
|
PROFILE="$PROFILE" VENDOR="$VENDOR" VCPUS="$VCPUS" CORES="$CORES" SMT="$SMT" RAMG="$RAMG" \
|
|
MODEL="$MODEL" EMULATOR="$EMULATOR" PINS="$PINS" \
|
|
EMUSET="$(pair "$EMU")" IOCSET="$(pair "$IOC")" NUMANODE="$([ "$NNODES" -gt 1 ] && echo "$GNODE")" \
|
|
SECBOOT="$SECBOOT" LOADER="$LOADER" NVRAM_TPL="$NVRAM_TPL" NVRAM_PATH="$NVRAM_PATH" CONVERT="$CONVERT" \
|
|
RANDOMIZE="$RANDOMIZE" NVME_SERIAL="$NVME_SERIAL" MAC="$MAC" USB_HOSTDEVS="$USB_HOSTDEVS" \
|
|
SYS_SERIAL="$SYS_SERIAL" BOARD_SERIAL="$BOARD_SERIAL" CHASSIS_SERIAL="$CHASSIS_SERIAL" \
|
|
DMI_SYS_VENDOR="$(dmi sys_vendor)" DMI_PRODUCT="$(dmi product_name)" DMI_PRODUCT_VERSION="$(dmi product_version)" \
|
|
DMI_SKU="$(dmi product_sku)" DMI_FAMILY="$(dmi product_family)" DMI_BOARD_VENDOR="$(dmi board_vendor)" \
|
|
DMI_BOARD="$(dmi board_name)" DMI_BOARD_VERSION="$(dmi board_version)" DMI_CHASSIS_VENDOR="$(dmi chassis_vendor)" \
|
|
DMI_CHASSIS_VERSION="$(dmi chassis_version)" DMI_BIOS_VENDOR="$(dmi bios_vendor)" DMI_BIOS_VERSION="$(dmi bios_version)" \
|
|
DMI_BIOS_DATE="$(dmi bios_date)" \
|
|
python3 - "$BACKUP/$DOM.before-setup.xml" "$BACKUP/$DOM.setup.xml" <<'XMLGEN_END'
|
|
import io, os, re, sys
|
|
from xml.sax.saxutils import escape
|
|
src, out = sys.argv[1], sys.argv[2]
|
|
E = os.environ
|
|
prof, vcpus, cores, smt = E["PROFILE"], int(E["VCPUS"]), int(E["CORES"]), int(E["SMT"])
|
|
s = io.open(src, encoding="utf-8").read()
|
|
conformant = prof != "tuned"
|
|
|
|
kib = int(E["RAMG"]) * 1024 * 1024
|
|
s = re.sub(r"<memory unit='KiB'>\d+</memory>", "<memory unit='KiB'>%d</memory>" % kib, s, count=1)
|
|
s = re.sub(r"<currentMemory unit='KiB'>\d+</currentMemory>",
|
|
"<currentMemory unit='KiB'>%d</currentMemory>" % kib, s, count=1)
|
|
s = re.sub(r"<vcpu placement='static'>\d+</vcpu>",
|
|
"<vcpu placement='static'>%d</vcpu>" % vcpus, s, count=1)
|
|
s = re.sub(r"<topology sockets='1' dies='1' clusters='1' cores='\d+' threads='\d+'/>",
|
|
"<topology sockets='1' dies='1' clusters='1' cores='%d' threads='%d'/>" % (cores, smt),
|
|
s, count=1)
|
|
s = re.sub(r"<emulator>[^<]*</emulator>", "<emulator>%s</emulator>" % E["EMULATOR"], s, count=1)
|
|
|
|
block = (" <iothreads>1</iothreads>\n <cputune>\n" + E["PINS"].replace("\\n", "\n") +
|
|
" <emulatorpin cpuset='%s'/>\n" % E["EMUSET"] +
|
|
" <iothreadpin iothread='1' cpuset='%s'/>\n" % E["IOCSET"] +
|
|
" </cputune>\n")
|
|
if E["NUMANODE"]:
|
|
block += " <numatune>\n <memory mode='strict' nodeset='%s'/>\n </numatune>\n" % E["NUMANODE"]
|
|
s = re.sub(r"\s*<iothreads>\d+</iothreads>", "", s)
|
|
s = re.sub(r"\s*<cputune>.*?</cputune>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<numatune>.*?</numatune>", "", s, flags=re.S)
|
|
s = s.replace(" <resource>", block + " <resource>", 1)
|
|
|
|
feats = [" <feature policy='require' name='invtsc'/>"]
|
|
if E.get("VENDOR") == "AuthenticAMD":
|
|
feats.insert(0, " <feature policy='require' name='topoext'/>")
|
|
if conformant:
|
|
feats.insert(0, " <feature policy='disable' name='hypervisor'/>")
|
|
feats += [" <feature policy='disable' name='spec-ctrl'/>",
|
|
" <feature policy='disable' name='stibp'/>",
|
|
" <feature policy='disable' name='ssbd'/>"]
|
|
s = re.sub(r"\n\s*<feature policy='(disable|require)' name='[^']*'/>", "", s)
|
|
s = s.replace("threads='%d'/>" % smt, "threads='%d'/>\n" % smt + "\n".join(feats), 1)
|
|
|
|
s = re.sub(r"\s*<kvm>.*?</kvm>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<hyperv.*?</hyperv>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<timer name='hypervclock'[^/]*/>", "", s)
|
|
if conformant:
|
|
s = s.replace(" <vmport state='off'/>",
|
|
" <kvm>\n <hidden state='on'/>\n </kvm>\n <vmport state='off'/>", 1)
|
|
else:
|
|
hv = (" <hyperv mode='custom'>\n"
|
|
" <relaxed state='on'/>\n <vapic state='on'/>\n"
|
|
" <spinlocks state='on' retries='8191'/>\n <vpindex state='on'/>\n"
|
|
" <runtime state='on'/>\n <synic state='on'/>\n"
|
|
" <stimer state='on'>\n <direct state='on'/>\n </stimer>\n"
|
|
" <reset state='on'/>\n <frequencies state='on'/>\n"
|
|
" <tlbflush state='on'>\n <direct state='on'/>\n"
|
|
" <extended state='on'/>\n </tlbflush>\n"
|
|
" <ipi state='on'/>\n <avic state='on'/>\n </hyperv>\n")
|
|
s = s.replace(" <vmport state='off'/>", hv + " <vmport state='off'/>", 1)
|
|
s = s.replace(" <timer name='tsc' present='yes' mode='native'/>",
|
|
" <timer name='tsc' present='yes' mode='native'/>\n"
|
|
" <timer name='hypervclock' present='yes'/>", 1)
|
|
|
|
|
|
# memfd + shared memory backing is only needed for virtiofs, and it blocks
|
|
# transparent hugepages outright - shared mappings are not anonymous, so THP
|
|
# for anon never applies and the whole guest runs on 4 KiB pages. Measured on
|
|
# an 8 GiB guest: 0 MiB AnonHugePages with it, 8110 MiB without.
|
|
if "<filesystem" not in s:
|
|
s = re.sub(r"\s*<memoryBacking>.*?</memoryBacking>", "", s, flags=re.S)
|
|
if "<pm>" not in s:
|
|
s = s.replace(" <devices>", " <pm>\n <suspend-to-mem enabled='yes'/>\n"
|
|
" <suspend-to-disk enabled='yes'/>\n </pm>\n <devices>", 1)
|
|
|
|
if conformant and E["MODEL"]:
|
|
s = re.sub(r"version=AMD Ryzen [^,]*,", "version=%s," % E["MODEL"], s)
|
|
|
|
if conformant and E["CONVERT"] == "1":
|
|
# Disks go to emulated NVMe: the patched QEMU refuses virtio, Windows boots
|
|
# NVMe with its inbox driver, and libvirt adds the controller itself. The
|
|
# first disk gets the identity's serial, further disks a derived one.
|
|
n = [0]
|
|
if E["RANDOMIZE"] == "1":
|
|
s = re.sub(r"(<disk type='[^']*' device='disk'>(?:(?!</disk>).)*?)\s*<serial>[^<]*</serial>", r"\1", s, flags=re.S)
|
|
s = re.sub(r"(<controller type='nvme'[^>]*>\s*)<serial>[^<]*</serial>", r"\1<serial>%s</serial>" % E["NVME_SERIAL"], s)
|
|
def to_nvme(m):
|
|
d = m.group(0)
|
|
if "device='disk'" not in d or ("bus='nvme'" in d and "<serial>" in d):
|
|
return d
|
|
d = re.sub(r"<target dev='([^']*)' bus='(virtio|sata|scsi)'/>", r"<target dev='\1' bus='nvme'/>", d)
|
|
# cache='none' is O_DIRECT: on btrfs the guest can change a page while the
|
|
# write is in flight, so the stored checksum never matches and later reads
|
|
# fail with EIO. Buffered writes hand the filesystem a stable page.
|
|
d = re.sub(r"<driver name='qemu' type='([^']*)'[^/]*/>",
|
|
r"<driver name='qemu' type='\1' cache='writeback' io='threads' discard='unmap'/>", d)
|
|
d = re.sub(r"\s*<address type='(pci|drive)'[^/]*/>", "", d)
|
|
if "<serial>" not in d:
|
|
serial = E["NVME_SERIAL"] if n[0] == 0 else E["NVME_SERIAL"][:-1] + "0123456789ABCDEF"[n[0] % 16]
|
|
d = d.replace("</disk>", " <serial>%s</serial>\n </disk>" % serial)
|
|
n[0] += 1
|
|
return d
|
|
s = re.sub(r"<disk type='[^']*' device='disk'>.*?</disk>", to_nvme, s, flags=re.S)
|
|
# everything virtio or agent-shaped names the emulator; a real machine has none of it
|
|
s = re.sub(r"\s*<channel type='unix'>.*?guest_agent.*?</channel>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<channel type='spicevmc'>.*?</channel>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<controller type='virtio-serial'[^>]*>.*?</controller>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<controller type='virtio-serial'[^>]*/>", "", s)
|
|
s = re.sub(r"\s*<filesystem type=.*?</filesystem>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<memoryBacking>.*?</memoryBacking>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<rng model='virtio'>.*?</rng>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<input type='[^']*' bus='virtio'>.*?</input>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<input type='[^']*' bus='virtio'/>", "", s)
|
|
s = re.sub(r"<memballoon model='virtio'>.*?</memballoon>", "<memballoon model='none'/>", s, flags=re.S)
|
|
s = re.sub(r"<memballoon model='virtio'/>", "<memballoon model='none'/>", s)
|
|
# jumbo frames are the single biggest win on the host<->guest link: the emulated
|
|
# NIC is packet-rate bound, so 9000-byte frames cut the per-packet cost the guest
|
|
# pays on receive. Measured 2922 -> 14232 Mbit/s inbound on igb, byte-exact clean.
|
|
s = re.sub(r"<model type='virtio'/>(\s*<driver [^/]*/>)?",
|
|
"<model type='igb'/>\n <mtu size='9000'/>", s)
|
|
if prof == "full":
|
|
s = re.sub(r"<video>.*?</video>", "<video>\n <model type='none'/>\n </video>", s, flags=re.S)
|
|
|
|
if conformant:
|
|
if E["RANDOMIZE"] == "1":
|
|
s = re.sub(r"<mac address='[^']*'/>", "<mac address='%s'/>" % E["MAC"], s, count=1)
|
|
def ent(name, key):
|
|
v = E.get(key, "")
|
|
return " <entry name='%s'>%s</entry>\n" % (name, escape(v)) if v else ""
|
|
sysinfo = (" <sysinfo type='smbios'>\n <bios>\n" + ent("vendor", "DMI_BIOS_VENDOR") +
|
|
ent("version", "DMI_BIOS_VERSION") + ent("date", "DMI_BIOS_DATE") + " </bios>\n <system>\n" +
|
|
ent("manufacturer", "DMI_SYS_VENDOR") + ent("product", "DMI_PRODUCT") + ent("version", "DMI_PRODUCT_VERSION") +
|
|
ent("serial", "SYS_SERIAL") + ent("sku", "DMI_SKU") + ent("family", "DMI_FAMILY") +
|
|
" </system>\n <baseBoard>\n" + ent("manufacturer", "DMI_BOARD_VENDOR") + ent("product", "DMI_BOARD") +
|
|
ent("version", "DMI_BOARD_VERSION") + ent("serial", "BOARD_SERIAL") + " </baseBoard>\n <chassis>\n" +
|
|
ent("manufacturer", "DMI_CHASSIS_VENDOR") + ent("version", "DMI_CHASSIS_VERSION") + ent("serial", "CHASSIS_SERIAL") +
|
|
" </chassis>\n </sysinfo>\n")
|
|
s = re.sub(r"\s*<sysinfo type='smbios'>.*?</sysinfo>", "", s, flags=re.S)
|
|
s = s.replace(" <os", sysinfo + " <os", 1)
|
|
s = re.sub(r"<smbios mode='[^']*'/>", "<smbios mode='sysinfo'/>", s)
|
|
if "<smbios mode=" not in s:
|
|
s = s.replace(" </os>", " <smbios mode='sysinfo'/>\n </os>", 1)
|
|
|
|
if E["SECBOOT"] == "on":
|
|
s = re.sub(r"<os firmware='efi'>", "<os>", s, count=1)
|
|
s = re.sub(r"\s*<firmware>.*?</firmware>", "", s, flags=re.S)
|
|
s = re.sub(r"\s*<loader [^>]*>[^<]*</loader>", "", s)
|
|
s = re.sub(r"\s*<nvram [^>]*>[^<]*</nvram>", "", s)
|
|
s = re.sub(r"\s*<nvram [^>]*/>", "", s)
|
|
fw = ("\n <loader readonly='yes' secure='yes' type='pflash' format='raw'>%s</loader>"
|
|
"\n <nvram template='%s' templateFormat='raw' format='raw'>%s</nvram>" %
|
|
(E["LOADER"], E["NVRAM_TPL"], E["NVRAM_PATH"]))
|
|
s = re.sub(r"(<type arch='x86_64' machine='[^']*'>hvm</type>)", r"\1" + fw.replace("\\", "\\\\"), s, count=1)
|
|
if "<smm state='on'/>" not in s:
|
|
s = s.replace(" </features>", " <smm state='on'/>\n </features>", 1)
|
|
|
|
# USB passthrough: replace what this tool put in before, leave other hostdevs alone
|
|
s = re.sub(r"\s*<hostdev mode='subsystem' type='(usb|pci)' managed='yes'>(?:(?!</hostdev>).)*<alias name='ua-vfionative-(usb|pci)-[^']*'/>(?:(?!</hostdev>).)*</hostdev>", "", s, flags=re.S)
|
|
if E["USB_HOSTDEVS"]:
|
|
s = s.replace(" </devices>", E["USB_HOSTDEVS"] + " </devices>", 1)
|
|
|
|
io.open(out, "w", encoding="utf-8").write(s)
|
|
XMLGEN_END
|
|
|
|
# libvirt only copies the template when the domain's own store does not exist
|
|
# yet, so enrolling keys means dropping the old store; libvirt does that for us.
|
|
if [ "$NVRAM_RESET" = 1 ]; then
|
|
if [ "$ASSUME" != 1 ]; then read -rp "reset the firmware variable store of $DOM? [y/N]: " a; [ "$a" = y ] || { echo aborted; exit 1; }; fi
|
|
"${C[@]}" undefine "$DOM" --nvram >/dev/null
|
|
fi
|
|
"${C[@]}" define "$BACKUP/$DOM.setup.xml" >/dev/null
|
|
|
|
# ------------------------------------------------ qemu command-line args ---
|
|
# libvirt has no XML for the CPUID brand string, raw SMBIOS structures or extra
|
|
# ACPI tables, so they ride in as <qemu:commandline>. The generated files live
|
|
# in a per-domain directory, so a package upgrade or a moved checkout does not
|
|
# change a defined domain underneath it. Only the arguments this script owns
|
|
# are replaced; anything else already in the block is left alone.
|
|
QARGS=()
|
|
if [ "$PROFILE" != tuned ]; then
|
|
kb() { local s; s=$(cat "/sys/devices/system/cpu/cpu0/cache/index$1/size"); echo "${s%K}"; }
|
|
SPAN=$(( (CORES + MAXC - 1) / MAXC ))
|
|
python3 "$SELF/scripts/generate-tables.py" --output-dir "$DATA" \
|
|
--cache-l1 $(( CORES * ($(kb 0) + $(kb 1)) )) --cache-l2 $(( CORES * $(kb 2) )) \
|
|
--cache-l3 $(( SPAN * $(kb 3) )) >/dev/null
|
|
for f in "$DATA"/type*.bin; do QARGS+=("-smbios" "file=$f"); done
|
|
if [ -d "$SELF/acpi" ]; then
|
|
cp "$SELF"/acpi/*.aml "$DATA"/
|
|
for f in "$DATA"/*.aml; do QARGS+=("-acpitable" "file=$f"); done
|
|
fi
|
|
# A second -cpu REPLACES the first rather than merging, so it is built from
|
|
# libvirt's own generated line and never hand-written. It is incompatible
|
|
# with <hyperv>, which is why the tuned level skips it.
|
|
# libvirt's own -cpu comes first; a previous run's override sits at the
|
|
# end, quoted because the brand has spaces, so split like a shell would.
|
|
if [ -n "$MODEL" ]; then
|
|
python3 - "$BACKUP/$DOM.setup.xml" "$BACKUP/$DOM.nohostdev.xml" <<'STRIP_END'
|
|
import io, re, sys
|
|
s = io.open(sys.argv[1], encoding="utf-8").read()
|
|
io.open(sys.argv[2], "w", encoding="utf-8").write(re.sub(r"\s*<hostdev .*?</hostdev>", "", s, flags=re.S))
|
|
STRIP_END
|
|
GEN=$("${C[@]}" domxml-to-native --format qemu-argv --xml "$BACKUP/$DOM.nohostdev.xml" | python3 -c '
|
|
import shlex, sys
|
|
a = shlex.split(sys.stdin.read())
|
|
print(a[a.index("-cpu") + 1] if "-cpu" in a else "")')
|
|
if [ -n "$GEN" ] && [[ "$GEN" != *hv-* ]]; then QARGS+=("-cpu" "$GEN,model-id=$MODEL"); fi
|
|
# SMBIOS type 4 defaults to manufacturer "QEMU" and the machine name as
|
|
# the version; a comma inside a QEMU option value is written twice.
|
|
case "$VENDOR:$FAMILY" in
|
|
AuthenticAMD:23) SOCK=AM4; CPUVEND="Advanced Micro Devices,, Inc.";;
|
|
AuthenticAMD:*) SOCK=AM5; CPUVEND="Advanced Micro Devices,, Inc.";;
|
|
GenuineIntel:*) SOCK=$([ "$MODELNO" -ge 151 ] && echo LGA1700 || echo LGA1200); CPUVEND="Intel(R) Corporation";;
|
|
*) SOCK=CPU; CPUVEND="$VENDOR";;
|
|
esac
|
|
MAXMHZ=$(( $(cat /sys/devices/system/cpu/cpu0/cpufreq/cpuinfo_max_freq 2>/dev/null || echo 4500000) / 1000 ))
|
|
QARGS+=("-smbios" "type=4,sock_pfx=$SOCK,manufacturer=$CPUVEND,version=$MODEL,max-speed=$MAXMHZ,current-speed=$MAXMHZ")
|
|
fi
|
|
if [ "$RANDOMIZE" = 1 ]; then
|
|
QARGS+=("-smbios" "type=17,loc_pfx=DIMM,bank=P0 CHANNEL A,manufacturer=$MEM_MANUFACTURER,part=$MEM_PART,serial=$MEM_SERIAL,speed=$MEM_SPEED")
|
|
fi
|
|
fi
|
|
QARGS_NL="$(printf '%s\n' "${QARGS[@]}")" python3 - "$BACKUP/$DOM.setup.xml" <<'QARGS_END'
|
|
import io, os, re, sys
|
|
p = sys.argv[1]
|
|
args = [a for a in os.environ["QARGS_NL"].split("\n") if a]
|
|
s = io.open(p, encoding="utf-8").read()
|
|
if "xmlns:qemu=" not in s:
|
|
s = s.replace("<domain type='kvm'>",
|
|
"<domain type='kvm' xmlns:qemu='http://libvirt.org/schemas/domain/qemu/1.0'>", 1)
|
|
kept, envs = [], []
|
|
m = re.search(r"\n\s*<qemu:commandline>(.*?)</qemu:commandline>", s, re.S)
|
|
if m:
|
|
envs = re.findall(r"<qemu:env [^>]*/>", m.group(1))
|
|
vals = re.findall(r"<qemu:arg value='([^']*)'/>", m.group(1))
|
|
i = 0
|
|
while i < len(vals):
|
|
owned = vals[i] == "-cpu" or (vals[i] in ("-smbios", "-acpitable") and i + 1 < len(vals)
|
|
and vals[i + 1].startswith(("file=", "type=")))
|
|
if owned:
|
|
i += 2
|
|
else:
|
|
kept.append(vals[i]); i += 1
|
|
s = s[:m.start()] + s[m.end():]
|
|
vals = kept + args
|
|
if vals or envs:
|
|
block = (" <qemu:commandline>\n"
|
|
+ "".join(" <qemu:arg value='%s'/>\n" % v.replace("'", "'") for v in vals)
|
|
+ "".join(" %s\n" % e for e in envs)
|
|
+ " </qemu:commandline>\n")
|
|
s = s.replace("</domain>", block + "</domain>", 1)
|
|
io.open(p, "w", encoding="utf-8").write(s)
|
|
QARGS_END
|
|
"${C[@]}" define "$BACKUP/$DOM.setup.xml" >/dev/null
|
|
|
|
echo
|
|
if [ "$PROFILE" = full ]; then
|
|
ok=1
|
|
[ "$EMULATOR" = "$PATCHED" ] || { echo "MISSING: patched QEMU not found at $PATCHED"; ok=0; }
|
|
# DKMS puts them in updates/dkms/, the manual install script in updates/
|
|
ko=$(find "/usr/lib/modules/$(uname -r)/updates" -name 'kvm.ko*' 2>/dev/null | head -1)
|
|
if [ -n "$ko" ]; then
|
|
built=$(modinfo -F srcversion "$ko" 2>/dev/null)
|
|
live=$(cat /sys/module/kvm/srcversion 2>/dev/null)
|
|
if [ -z "$built" ] || [ "$built" != "$live" ]; then
|
|
echo "MISSING: patched KVM modules installed but not loaded."
|
|
echo " With all VMs off: sudo modprobe -r kvm_amd kvm && sudo modprobe kvm_amd"
|
|
ok=0
|
|
fi
|
|
else
|
|
echo "MISSING: no patched KVM modules - install vfio-native-kvm-dkms."
|
|
ok=0
|
|
fi
|
|
[ "$ok" = 1 ] && echo "patched QEMU and KVM modules both in place."
|
|
if [ -n "$MODEL" ] && [ -f /sys/module/kvm_amd/parameters/cpuid_passthrough ]; then
|
|
echo
|
|
echo "The TIMER check needs CPUID passthrough, which must be off while the guest cold"
|
|
echo "boots and on once it is up. Let the hook handle that around this guest:"
|
|
echo " sudo vm-native-cpuid enable $DOM"
|
|
echo "Or drive it by hand, after the guest has booted, on the host:"
|
|
echo " echo '$MODEL' | sudo tee /sys/module/kvm_amd/parameters/brand_string"
|
|
echo " echo Y | sudo tee /sys/module/kvm_amd/parameters/cpuid_passthrough # N again before the next boot"
|
|
fi
|
|
fi
|
|
|
|
if [ ! -e /usr/lib/udev/rules.d/99-vfio-native-vnet-offload.rules ] && [ ! -e /etc/udev/rules.d/99-vfio-native-vnet-offload.rules ]; then
|
|
echo "NOTE: the NIC offload udev rule is not installed. SSH into the guest will fail with"
|
|
echo " 'Corrupted MAC on input' until it is:"
|
|
echo " sudo install -Dm644 $SELF/scripts/99-vfio-native-vnet-offload.rules /etc/udev/rules.d/ && sudo udevadm control --reload-rules"
|
|
fi
|
|
|
|
NET=$("${C[@]}" dumpxml "$DOM" 2>/dev/null | sed -n "s/.*<source network='\([^']*\)'.*/\1/p" | head -1)
|
|
NETMTU=$([ -n "$NET" ] && "${C[@]}" net-dumpxml --inactive "$NET" 2>/dev/null | sed -n "s/.*<mtu size='\([0-9]*\)'.*/\1/p")
|
|
if [ -n "$NET" ] && [ "${NETMTU:-1500}" -lt 9000 ]; then
|
|
echo "NOTE: the guest interface asks for MTU 9000 but libvirt network '$NET' is at ${NETMTU:-1500}."
|
|
echo " Jumbo needs both ends; inbound throughput is ~5x with it. Add <mtu size='9000'/> to"
|
|
echo " the network and restart it: virsh net-edit $NET && virsh net-destroy $NET && virsh net-start $NET"
|
|
echo " Then in the guest: set the NIC's Jumbo Packet to 9014 and the interface MTU to 9000."
|
|
fi
|
|
gov=$(cat /sys/devices/system/cpu/cpu0/cpufreq/scaling_governor 2>/dev/null || echo unknown)
|
|
[ "$gov" = performance ] || echo "host governor is '$gov' - run: sudo cpupower frequency-set -g performance"
|
|
|
|
echo
|
|
echo "$DOM configured: $PROFILE, $VCPUS vCPU, ${RAMG} GiB."
|
|
echo "revert: virsh -c qemu:///system define $BACKUP/$DOM.before-setup.xml"
|
|
if command -v vm-native-verify >/dev/null; then echo "verify: vm-native-verify"
|
|
else echo "verify: $(dirname "$0")/verify-perf.sh"; fi
|