"""VM provisioning mixin for Proxmox — creates, destroys, and monitors VMs via Cloud-Init.""" from __future__ import annotations import logging import time import yaml from typing import Any, Dict, List from urllib.parse import quote from napalm_device_types.models import VMProvisionResultDict, VMStatusDict _logger = logging.getLogger(__name__) class ProxmoxVMProvisionMixin: """Mixin to add VM provisioning to ProxmoxDriver.""" def _wait_for_task(self, upid: str, timeout: int = 120) -> None: """ Poll a Proxmox task until completion. Polls /nodes/{node}/tasks/{upid}/status until status == 'stopped'. Raises RuntimeError if exitstatus != 'OK' or timeout exceeded. """ start_time = time.time() while True: elapsed = time.time() - start_time if elapsed > timeout: raise RuntimeError(f"Task {upid} timed out after {timeout}s") try: task_status = self._node_api().tasks(upid).status.get() except Exception as e: _logger.debug(f"Error polling task {upid}: {e}") time.sleep(2) continue if task_status.get("status") == "stopped": exitstatus = task_status.get("exitstatus", "UNKNOWN") if exitstatus != "OK": raise RuntimeError( f"Task {upid} failed with exitstatus='{exitstatus}': " f"{task_status.get('exitstatus_text', 'no error message')}" ) return time.sleep(2) def create_vm_from_cloud_init( self, name: str, *, template: str, cpu: int, memory: int, mgmt_bridge: str, mgmt_vlan_tag: int | None, capture_bridge: str, capture_vlan_tags: List[int], cloud_init_config: Dict[str, Any], ssh_public_keys: List[str] | None = None, timeout: int = 120, ) -> VMProvisionResultDict: """ Create a new VM from a Cloud-Init template via Proxmox API. Steps: 1. Get next available VMID from cluster 2. Clone template VM (full clone, new VMID) 3. Configure CPU, memory, and dual NICs (mgmt + capture) 4. Verify snippet storage exists 5. Render cloud-init config to YAML and upload 6. Set Cloud-Init config references and SSH keys 7. Start the VM 8. Return VMID, name, node Args: name: new VM display name template: template VMID/name to clone from cpu: number of vCPUs memory: RAM in MB mgmt_bridge: management bridge name mgmt_vlan_tag: VLAN tag for mgmt NIC (None = untagged) capture_bridge: packet capture bridge name (must be VLAN-aware) capture_vlan_tags: list of VLAN IDs for capture NIC (trunk) cloud_init_config: user-data dict (will be YAML-rendered) ssh_public_keys: SSH public keys to inject timeout: max seconds for provisioning Returns: {"vmid": str, "name": str, "node": str} Raises: RuntimeError: provisioning failure (clone, config, timeout, etc.) ValueError: invalid storage or configuration """ try: _logger.info(f"Creating VM '{name}' from template {template}") # Step 1: Get next VMID next_vmid = self._api.cluster.nextid.get() vmid = int(next_vmid) _logger.info(f"Allocated VMID {vmid}") # Step 2: Clone template _logger.info(f"Cloning template {template} → VMID {vmid}") clone_upid = self._node_api().qemu(template).clone.post( newid=vmid, full=1, name=name, ) self._wait_for_task(clone_upid, timeout=timeout) # Step 3: Configure CPU, memory, and dual NICs _logger.info(f"Configuring VM {vmid}: {cpu} CPU, {memory}MB RAM") # Build net0 (mgmt) config net0_config = f"virtio,bridge={mgmt_bridge}" if mgmt_vlan_tag is not None: net0_config += f",tag={mgmt_vlan_tag}" # Build net1 (capture trunk) config vlan_list = ";".join(str(v) for v in capture_vlan_tags) net1_config = f"virtio,bridge={capture_bridge},trunks={vlan_list}" self._node_api().qemu(vmid).config.post( cores=cpu, memory=memory, net0=net0_config, net1=net1_config, ) # Step 4: Verify snippet storage exists _logger.info("Checking for snippet storage...") storages = self._api.storage.get() snippet_storage = None for storage in storages: content = storage.get("content", "") if "snippets" in content and storage.get("enabled"): snippet_storage = storage["storage"] break if not snippet_storage: raise ValueError( "No storage with content='snippets' found. " "Configure a snippet-capable storage (e.g. local, nfs dir) " "and enable it." ) _logger.info(f"Using snippet storage: {snippet_storage}") # Step 5: Render and upload Cloud-Init config _logger.info(f"Rendering Cloud-Init config for VMID {vmid}") # Ensure cloud_init_config includes runcmd to bring up capture NIC if "runcmd" not in cloud_init_config: cloud_init_config["runcmd"] = [] if not any("eth1" in cmd if isinstance(cmd, str) else False for cmd in cloud_init_config.get("runcmd", [])): cloud_init_config["runcmd"].insert(0, "ip link set eth1 up") user_data_yaml = "#cloud-config\n" + yaml.dump( cloud_init_config, default_flow_style=False ) filename = f"{vmid}-user-data.yaml" _logger.debug(f"Uploading Cloud-Init snippet {filename} to {snippet_storage}") # Upload to snippet storage self._node_api().storage(snippet_storage).upload.post( content="snippets", filename=filename, data=user_data_yaml, ) # Step 6: Configure Cloud-Init references and SSH keys _logger.info(f"Setting Cloud-Init config for VM {vmid}") cloud_init_args = { "ide2": f"{snippet_storage}:cloudinit", "citype": "nocloud", "cicustom": f"user={snippet_storage}:snippets/{filename}", "ipconfig0": "ip=dhcp", # net0 gets DHCP (mgmt) # NO ipconfig1 for net1 (capture NIC stays unnummeriert) } if ssh_public_keys: # URL-encode SSH keys for Proxmox API sshkeys = ";".join(ssh_public_keys) cloud_init_args["sshkeys"] = quote(sshkeys) self._node_api().qemu(vmid).config.post(**cloud_init_args) # Step 7: Start the VM _logger.info(f"Starting VM {vmid}") start_upid = self._node_api().qemu(vmid).status.start.post() self._wait_for_task(start_upid, timeout=timeout) _logger.info(f"VM {vmid} ('{name}') provisioned successfully on {self._node_name}") return { "vmid": str(vmid), "name": name, "node": self._node_name, } except Exception as e: _logger.exception(f"Failed to create VM '{name}': {e}") raise def destroy_vm( self, vmid: str, *, remove_disk: bool = True, timeout: int = 60, ) -> None: """ Destroy a virtual machine and optionally remove its storage. Steps: 1. Stop the VM if running 2. Delete VM configuration and optionally disks 3. Clean up Cloud-Init snippets Args: vmid: hypervisor VMID (string, e.g. "101") remove_disk: if True, also delete disks and storage timeout: max seconds for stop/delete operations Raises: RuntimeError: VM doesn't exist or destruction fails """ try: vmid_int = int(vmid) _logger.info(f"Destroying VM {vmid}") # Step 1: Stop the VM if running try: _logger.debug(f"Stopping VM {vmid}") stop_upid = self._node_api().qemu(vmid_int).status.stop.post() self._wait_for_task(stop_upid, timeout=timeout) except Exception as e: _logger.debug(f"VM {vmid} stop failed (may already be stopped): {e}") # Step 2: Delete VM _logger.debug(f"Deleting VM {vmid} configuration and disks") self._node_api().qemu(vmid_int).delete( purge=1, destroy_unreferenced_disks=1 if remove_disk else 0, ) # Step 3: Clean up Cloud-Init snippets # (This is best-effort; snippet files may be unreachable if storage is unavailable) try: config = self._node_api().qemu(vmid_int).config.get() cicustom = config.get("cicustom", "") if "snippets/" in cicustom: parts = cicustom.split("=") if len(parts) >= 2: snippet_ref = parts[1] # e.g. "snippets:snippets/101-user-data.yaml" storage, filepath = snippet_ref.split(":", 1) _logger.debug(f"Deleting snippet {filepath} from {storage}") try: self._node_api().storage(storage).content(filepath).delete() except Exception as e: _logger.warning(f"Failed to delete snippet {filepath}: {e}") except Exception as e: _logger.debug(f"Could not clean up snippets for VM {vmid}: {e}") _logger.info(f"VM {vmid} destroyed successfully") except Exception as e: _logger.exception(f"Failed to destroy VM {vmid}: {e}") raise def get_vm_status( self, vmid: str, *, wait_for_ip: bool = False, timeout: int = 300, poll_interval: int = 5, ) -> VMStatusDict: """ Get the runtime status of a virtual machine. Optionally waits for the guest-agent to report an IP address on the management NIC (net0), useful after provisioning. Args: vmid: hypervisor VMID (string) wait_for_ip: if True, poll until IP appears on net0 timeout: max seconds to wait for IP (if wait_for_ip=True) poll_interval: seconds between status polls Returns: {"status": str, "ip_address": str, "hostname": str, "mac_address": str} (ip_address, hostname, mac_address only if VM is running and has network info) Raises: RuntimeError: VM doesn't exist or wait_for_ip times out """ try: vmid_int = int(vmid) _logger.debug(f"Getting status for VM {vmid}") # Get VM config to infer net0 MAC (for matching guest-agent results) try: config = self._node_api().qemu(vmid_int).config.get() except Exception: # VM may not exist yet or config not readable return {"status": "unknown"} # Parse net0 MAC from config (if present) net0_line = config.get("net0", "") expected_mac = None # Example: "virtio,bridge=vmbr0,tag=10" — no explicit MAC # Proxmox auto-generates MACs in a deterministic pattern, but we'll # match by looking for the first NIC's IP in guest-agent results # Polling loop start_time = time.time() while True: elapsed = time.time() - start_time if wait_for_ip and elapsed > timeout: raise RuntimeError( f"VM {vmid} failed to acquire IP within {timeout}s" ) try: # Query guest-agent network interfaces agent_info = self._node_api().qemu(vmid_int).agent.network_get_interfaces.get() interfaces = agent_info.get("result", []) # Find net0 (first interface with IP) if interfaces: net0_iface = interfaces[0] # Assumes net0 is first in list net0_mac = net0_iface.get("hardware-address", "") ip_addresses = net0_iface.get("ip-addresses", []) if ip_addresses: # Found IP ip_info = ip_addresses[0] ip_addr = ip_info.get("ip-address", "") if ip_addr: _logger.info(f"VM {vmid} acquired IP {ip_addr}") return { "status": "running", "ip_address": ip_addr, "hostname": net0_iface.get("name", ""), "mac_address": net0_mac, } except Exception as e: _logger.debug(f"Error querying guest-agent for VM {vmid}: {e}") if not wait_for_ip: # Return immediate status without IP return {"status": "running"} # Wait before next poll time.sleep(poll_interval) except RuntimeError: raise except Exception as e: _logger.exception(f"Error getting status for VM {vmid}: {e}") raise RuntimeError(f"Failed to get VM {vmid} status: {e}")