windows/pci: vIOMMU for nested passthrough, and a data disk built with the image
Three things, all in service of putting Proxmox in a VM that can still hand a
GPU to its own guests, and of a Windows VM whose profile survives its OS disk.
pci.viommu.enable emits `-device intel-iommu,intremap=on,caching-mode=on` and
forces kernel-irqchip=split, which interrupt remapping requires. Without an
IOMMU of its own a guest cannot bind a passed-through device to vfio-pci, so
it can never forward one on. The device leads the command line because QEMU
realizes devices in order and intel-iommu must precede what it translates.
pci.vgaPassthrough (default true, so nothing changes for existing VMs) makes
x-vga=on optional. It was forced on the first passthrough device, which is
wrong for a card the guest only forwards onward: it claims the VGA path the
emulated console adapter needs.
customizeImage gains extraDisk, a blank disk attached for the Audit Mode boot
and emitted as the derivation's `data` output. generalize uses it for dataDisk
and profilesDirectory, so the disk is partitioned and the profile relocated
under OOBE in the build VM. That is what removes the need for delayOobeRun --
previously the volume ProfilesDirectory names could not exist until the image
reached real hardware. A second output rather than a directory keeps ${image}
meaning the OS qcow2 for every existing consumer.
generalize also picks up staticIP and profilesDirectory.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_0117qMyjpuXsjpVAcpJbFD8g
This commit is contained in:
parent
9784736260
commit
c213fc4db9
4 changed files with 122 additions and 4 deletions
|
|
@ -29,6 +29,12 @@
|
|||
compact ? false,
|
||||
# QEMU timeout in seconds (default 30 min, increase for Windows Update)
|
||||
qemuTimeout ? 1800,
|
||||
# Blank disk attached for the Audit Mode boot, e.g. { size = "100G"; }.
|
||||
# Windows sees it as disk 1, which is what lets a template partition it and
|
||||
# relocate profiles onto it in that same boot instead of deferring OOBE to
|
||||
# real hardware. Emitted as the derivation's `data` output, so whatever the
|
||||
# template writes to it survives the build.
|
||||
extraDisk ? null,
|
||||
}:
|
||||
let
|
||||
originalImageName = lib.strings.removeSuffix "-vmix" (lib.strings.removeSuffix ".qcow2" originalImage.name);
|
||||
|
|
@ -67,6 +73,11 @@
|
|||
]);
|
||||
|
||||
cdromArgs = lib.concatMapStringsSep " \\\n " (cd: "-drive file=${cd},media=cdrom,readonly=on") cdroms;
|
||||
extraDiskImg = "./extra.qcow2";
|
||||
extraDiskArgs = lib.optionalString (extraDisk != null)
|
||||
(if isAHCI
|
||||
then "-drive file=${extraDiskImg},format=qcow2,if=none,id=disk1 -device ide-hd,drive=disk1"
|
||||
else "-drive file=${extraDiskImg},format=qcow2,if=virtio");
|
||||
|
||||
displayArg = if vncDisplay != null then "-vnc ${vncDisplay}" else null;
|
||||
|
||||
|
|
@ -109,6 +120,7 @@
|
|||
then "-drive file=${resultImg},format=qcow2,if=none,id=disk0 -device ide-hd,drive=disk0"
|
||||
else "-drive file=${resultImg},format=qcow2,if=virtio"} \
|
||||
${cdromArgs} \
|
||||
${extraDiskArgs} \
|
||||
-nic user,model=${if nicModel != null then nicModel else if isAHCI then "e1000" else "virtio-net-pci"}"
|
||||
|
||||
timeout ${toString qemuTimeout} qemu-system-x86_64 $VMIX_DISPLAY $QEMU_ARGS || \
|
||||
|
|
@ -127,6 +139,10 @@
|
|||
# create resulting image backed by original image
|
||||
qemu-img create -f qcow2 -b ${originalImage} -F qcow2 ${resultImg}
|
||||
[ -n "${diskSize}" ] && qemu-img resize ${resultImg} ${diskSize}
|
||||
${lib.optionalString (extraDisk != null) ''
|
||||
echo "=== vmix: creating extra disk (${extraDisk.size}) ==="
|
||||
qemu-img create -f qcow2 ${extraDiskImg} ${extraDisk.size}
|
||||
''}
|
||||
${virtWinRegMerge}
|
||||
${auditBootCommands}
|
||||
${lib.optionalString compact ''
|
||||
|
|
@ -135,10 +151,14 @@
|
|||
mv compact.qcow2 ${resultImg}
|
||||
''}
|
||||
mv ${resultImg} $out
|
||||
${lib.optionalString (extraDisk != null) "mv ${extraDiskImg} $data"}
|
||||
'';
|
||||
builtImage = pkgs.runCommand customImageName ({
|
||||
nativeBuildInputs = with pkgs; [ qemu perl guestfs-tools ];
|
||||
requiredSystemFeatures = [ "kvm" ];
|
||||
} // lib.optionalAttrs impure { __noChroot = true; }) builderCommand;
|
||||
} // lib.optionalAttrs impure { __noChroot = true; }
|
||||
# A second output rather than a directory, so ${image} keeps meaning the OS
|
||||
# qcow2 for every existing consumer and the fold can still back onto it.
|
||||
// lib.optionalAttrs (extraDisk != null) { outputs = [ "out" "data" ]; }) builderCommand;
|
||||
in
|
||||
builtImage // { _vmixOsType = "windows"; useAHCI = isAHCI; }
|
||||
|
|
|
|||
|
|
@ -23,6 +23,18 @@ in
|
|||
enableRDP ? false,
|
||||
# NIC model for the build VM (e.g. "e1000" for images without VirtIO drivers)
|
||||
nicModel ? null,
|
||||
# Static IPv4 for the guest's single NIC, applied from inside Windows:
|
||||
# { address = "10.10.10.26"; prefixLength = 24; gateway = "10.10.10.1";
|
||||
# dns = [ "10.10.10.1" ]; }
|
||||
staticIP ? null,
|
||||
# Relocate user profiles, e.g. "D:\\Users". Needs the volume to exist by the
|
||||
# time specialize runs, which is what dataDisk arranges.
|
||||
profilesDirectory ? null,
|
||||
# Partition the non-OS disk and relocate user profiles onto it, e.g.
|
||||
# { driveLetter = "D"; label = "data"; size = "100G"; }. The disk is attached
|
||||
# during the build (see extraDisk in the returned set), so this is done and
|
||||
# verified before the image ever reaches a host.
|
||||
dataDisk ? null,
|
||||
# delayOobeRun = true: sysprep only, OOBE + activation on real hardware
|
||||
# delayOobeRun = false: sysprep + OOBE + activation in build VM
|
||||
delayOobeRun ? false,
|
||||
|
|
@ -42,6 +54,45 @@ in
|
|||
stripHash = s: lib.removePrefix "#" s;
|
||||
bgRgb = if bgColor != null then hexToRgbStr (stripHash bgColor) else null;
|
||||
|
||||
staticDnsList = lib.optionalString (staticIP != null)
|
||||
(lib.concatMapStringsSep "," (s: "'${s}'") staticIP.dns);
|
||||
|
||||
dataDriveLetter = if dataDisk != null then (dataDisk.driveLetter or "D") else "D";
|
||||
dataLabel = if dataDisk != null then (dataDisk.label or "data") else "data";
|
||||
|
||||
# ProfilesDirectory is only honoured when the volume it names already exists,
|
||||
# and a freshly created zvol arrives RAW. Initializing the disk here, in the
|
||||
# same specialize pass, brings it up before oobeSystem creates any profile.
|
||||
#
|
||||
# Idempotent, because specialize runs again on every sysprep: a RAW disk gets
|
||||
# a GPT label, one full-size NTFS partition and the drive letter, while a disk
|
||||
# that already holds data keeps it and only has its letter re-asserted. The
|
||||
# OS disk is added to QEMU first and so is always disk 0.
|
||||
initDataDiskScript = pkgs.writeText "vmix-init-data-disk.cmd" ''
|
||||
@echo off
|
||||
powershell -NoProfile -ExecutionPolicy Bypass -Command "$ErrorActionPreference='Stop'; $d = Get-Disk | Where-Object Number -ne 0 | Sort-Object Number | Select-Object -First 1; if (-not $d) { exit 0 }; if ($d.PartitionStyle -eq 'RAW') { Initialize-Disk -Number $d.Number -PartitionStyle GPT -Confirm:$false; $p = New-Partition -DiskNumber $d.Number -UseMaximumSize -DriveLetter ${dataDriveLetter}; Format-Volume -Partition $p -FileSystem NTFS -NewFileSystemLabel '${dataLabel}' -Confirm:$false | Out-Null } else { $p = Get-Partition -DiskNumber $d.Number | Sort-Object Size -Descending | Select-Object -First 1; if ($p -and $p.DriveLetter -ne '${dataDriveLetter}') { Set-Partition -InputObject $p -NewDriveLetter ${dataDriveLetter} } }"
|
||||
'';
|
||||
|
||||
folderLocationsXml = lib.optionalString (profilesDirectory != null) ''
|
||||
<!-- Profiles live on the data disk, so the OS disk stays disposable
|
||||
and rebuilding it does not take the profile along -->
|
||||
<FolderLocations>
|
||||
<ProfilesDirectory>${profilesDirectory}</ProfilesDirectory>
|
||||
</FolderLocations>'';
|
||||
|
||||
dataDiskXml = lib.optionalString (dataDisk != null) ''
|
||||
<!-- Runs during specialize, before the first profile is created -->
|
||||
<component name="Microsoft-Windows-Deployment" processorArchitecture="amd64"
|
||||
publicKeyToken="31bf3856ad364e35" language="neutral" versionScope="nonSxS">
|
||||
<RunSynchronous>
|
||||
<RunSynchronousCommand wcm:action="add">
|
||||
<Order>1</Order>
|
||||
<Path>cmd /c C:\vmix-init-data-disk.cmd</Path>
|
||||
<Description>vmix: initialize the data disk</Description>
|
||||
</RunSynchronousCommand>
|
||||
</RunSynchronous>
|
||||
</component>'';
|
||||
|
||||
# Post-OOBE script: runs as the created user via FirstLogonCommands.
|
||||
postOobeScript = pkgs.writeText "post-oobe.cmd" ''
|
||||
@echo off
|
||||
|
|
@ -114,6 +165,13 @@ in
|
|||
reg add "HKLM\SYSTEM\CurrentControlSet\Services\TermService" /v Start /t REG_DWORD /d 2 /f
|
||||
''}
|
||||
|
||||
${lib.optionalString (staticIP != null) ''
|
||||
:: This VM's only NIC sits on a macvtap bridged to the host's LAN, so its
|
||||
:: address is a LAN address that nothing hands out -- the guest asserts it.
|
||||
:: Clearing first makes the command idempotent across re-runs.
|
||||
powershell -NoProfile -Command "$a = Get-NetAdapter -Physical | Sort-Object ifIndex | Select-Object -First 1; Remove-NetIPAddress -InterfaceIndex $a.ifIndex -AddressFamily IPv4 -Confirm:$false -ErrorAction SilentlyContinue; Remove-NetRoute -InterfaceIndex $a.ifIndex -AddressFamily IPv4 -Confirm:$false -ErrorAction SilentlyContinue; New-NetIPAddress -InterfaceIndex $a.ifIndex -IPAddress '${staticIP.address}' -PrefixLength ${toString staticIP.prefixLength} -DefaultGateway '${staticIP.gateway}' | Out-Null; Set-DnsClientServerAddress -InterfaceIndex $a.ifIndex -ServerAddresses ${staticDnsList}"
|
||||
''}
|
||||
|
||||
:: Clean up
|
||||
del /q C:\oobe-unattend.xml 2>nul
|
||||
del /q C:\vmix-audit-script.cmd 2>nul
|
||||
|
|
@ -135,7 +193,9 @@ in
|
|||
<Themes>
|
||||
<WindowColor>Automatic</WindowColor>
|
||||
</Themes>
|
||||
${folderLocationsXml}
|
||||
</component>
|
||||
${dataDiskXml}
|
||||
</settings>
|
||||
|
||||
<settings pass="oobeSystem">
|
||||
|
|
@ -195,11 +255,16 @@ in
|
|||
in {
|
||||
name = if delayOobeRun then "generalize-delay-oobe" else "generalize";
|
||||
inherit nicModel;
|
||||
# The blank disk is attached for the Audit Mode boot itself, so the disk-init
|
||||
# command and the profile relocation both happen under OOBE in the build VM.
|
||||
# That is what makes delayOobeRun unnecessary: nothing is left to do on real
|
||||
# hardware. The written disk comes back as this derivation's `data` output.
|
||||
extraDisk = if dataDisk != null then { size = dataDisk.size or "100G"; } else null;
|
||||
uploads = [
|
||||
{ source = oobeXml; dest = "/oobe-unattend.xml"; }
|
||||
{ source = postOobeScript; dest = "/post-oobe.cmd"; }
|
||||
{ source = masScript; dest = "/MAS_AIO.cmd"; }
|
||||
];
|
||||
] ++ lib.optional (dataDisk != null) { source = initDataDiskScript; dest = "/vmix-init-data-disk.cmd"; };
|
||||
# delayOobeRun: sysprep + shutdown — OOBE runs on real hardware
|
||||
# generalize: sysprep + reboot into OOBE in the same QEMU session
|
||||
auditScript = ''
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue