Files
3cloud-backend/scripts/release_gpus.py

67 lines
2.1 KiB
Python

#!/usr/bin/env python3
"""
Unbind all GPUs currently held by vfio-pci and restore them to the nvidia driver.
Must be run as root on the hypervisor host.
Usage:
sudo python3 scripts/release_gpus.py [--dry-run]
"""
import sys
import os
import argparse
import logging
from pathlib import Path
logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s")
log = logging.getLogger(__name__)
# Allow running from repo root or scripts/
sys.path.insert(0, str(Path(__file__).resolve().parent.parent / "worker"))
from dynamic_vfio_passthrough import DynamicVfioPassthroughController
NVIDIA_VENDOR_ID = "0x10de"
def main():
parser = argparse.ArgumentParser(description="Release all vfio-pci GPUs back to nvidia")
parser.add_argument("--dry-run", action="store_true", help="Print actions without executing them")
args = parser.parse_args()
if os.geteuid() != 0 and not args.dry_run:
sys.exit("Must be run as root (or use --dry-run)")
# Use a throwaway controller just for the list method
probe = DynamicVfioPassthroughController("0000:00:00.0", dry_run=True)
vfio_devices = probe.list_attached_devices()
gpus = [d for d in vfio_devices if d.vendor_id == NVIDIA_VENDOR_ID]
if not gpus:
log.info("No NVIDIA GPUs currently bound to vfio-pci.")
return
log.info("Found %d NVIDIA GPU(s) bound to vfio-pci:", len(gpus))
for d in gpus:
log.info(" %s (device %s, iommu group %s)", d.pci_address, d.device_id, d.iommu_group)
failed = []
for d in gpus:
log.info("Releasing %s -> nvidia ...", d.pci_address)
try:
ctrl = DynamicVfioPassthroughController(d.pci_address, dry_run=args.dry_run)
ctrl.detach_device_vfio(restore_driver="nvidia")
log.info(" OK")
except Exception as e:
log.error(" FAILED: %s", e)
failed.append(d.pci_address)
if failed:
log.error("Failed to release: %s", ", ".join(failed))
sys.exit(1)
else:
log.info("All GPUs released.")
if __name__ == "__main__":
main()