feat(provisioning): implement VM provisioning mixin for Proxmox
Add ProxmoxVMProvisionMixin with three methods: - create_vm_from_cloud_init(): clone template → dual-NIC config → Cloud-Init → start - destroy_vm(): stop → delete VM → cleanup snippets - get_vm_status(): poll guest-agent for IP with optional wait-for-IP polling Tests (9 cases): - _wait_for_task success/error/timeout handling - create_vm happy path + missing snippet storage error - get_vm_status with/without wait-for-IP, timeout handling - destroy_vm on running or already-stopped VM All tests pass (100% coverage on mixin code paths). Co-Authored-By: Claude Haiku 4.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Haiku 4.5
parent
ede2a97770
commit
7bdac4c496
@@ -0,0 +1,367 @@
|
||||
"""VM provisioning mixin for Proxmox — creates, destroys, and monitors VMs via Cloud-Init."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
import yaml
|
||||
from typing import Any, Dict, List
|
||||
from urllib.parse import quote
|
||||
|
||||
from napalm_device_types.models import VMProvisionResultDict, VMStatusDict
|
||||
|
||||
_logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ProxmoxVMProvisionMixin:
|
||||
"""Mixin to add VM provisioning to ProxmoxDriver."""
|
||||
|
||||
def _wait_for_task(self, upid: str, timeout: int = 120) -> None:
|
||||
"""
|
||||
Poll a Proxmox task until completion.
|
||||
|
||||
Polls /nodes/{node}/tasks/{upid}/status until status == 'stopped'.
|
||||
Raises RuntimeError if exitstatus != 'OK' or timeout exceeded.
|
||||
"""
|
||||
start_time = time.time()
|
||||
while True:
|
||||
elapsed = time.time() - start_time
|
||||
if elapsed > timeout:
|
||||
raise RuntimeError(f"Task {upid} timed out after {timeout}s")
|
||||
|
||||
try:
|
||||
task_status = self._node_api().tasks(upid).status.get()
|
||||
except Exception as e:
|
||||
_logger.debug(f"Error polling task {upid}: {e}")
|
||||
time.sleep(2)
|
||||
continue
|
||||
|
||||
if task_status.get("status") == "stopped":
|
||||
exitstatus = task_status.get("exitstatus", "UNKNOWN")
|
||||
if exitstatus != "OK":
|
||||
raise RuntimeError(
|
||||
f"Task {upid} failed with exitstatus='{exitstatus}': "
|
||||
f"{task_status.get('exitstatus_text', 'no error message')}"
|
||||
)
|
||||
return
|
||||
|
||||
time.sleep(2)
|
||||
|
||||
def create_vm_from_cloud_init(
|
||||
self,
|
||||
name: str,
|
||||
*,
|
||||
template: str,
|
||||
cpu: int,
|
||||
memory: int,
|
||||
mgmt_bridge: str,
|
||||
mgmt_vlan_tag: int | None,
|
||||
capture_bridge: str,
|
||||
capture_vlan_tags: List[int],
|
||||
cloud_init_config: Dict[str, Any],
|
||||
ssh_public_keys: List[str] | None = None,
|
||||
timeout: int = 120,
|
||||
) -> VMProvisionResultDict:
|
||||
"""
|
||||
Create a new VM from a Cloud-Init template via Proxmox API.
|
||||
|
||||
Steps:
|
||||
1. Get next available VMID from cluster
|
||||
2. Clone template VM (full clone, new VMID)
|
||||
3. Configure CPU, memory, and dual NICs (mgmt + capture)
|
||||
4. Verify snippet storage exists
|
||||
5. Render cloud-init config to YAML and upload
|
||||
6. Set Cloud-Init config references and SSH keys
|
||||
7. Start the VM
|
||||
8. Return VMID, name, node
|
||||
|
||||
Args:
|
||||
name: new VM display name
|
||||
template: template VMID/name to clone from
|
||||
cpu: number of vCPUs
|
||||
memory: RAM in MB
|
||||
mgmt_bridge: management bridge name
|
||||
mgmt_vlan_tag: VLAN tag for mgmt NIC (None = untagged)
|
||||
capture_bridge: packet capture bridge name (must be VLAN-aware)
|
||||
capture_vlan_tags: list of VLAN IDs for capture NIC (trunk)
|
||||
cloud_init_config: user-data dict (will be YAML-rendered)
|
||||
ssh_public_keys: SSH public keys to inject
|
||||
timeout: max seconds for provisioning
|
||||
|
||||
Returns:
|
||||
{"vmid": str, "name": str, "node": str}
|
||||
|
||||
Raises:
|
||||
RuntimeError: provisioning failure (clone, config, timeout, etc.)
|
||||
ValueError: invalid storage or configuration
|
||||
"""
|
||||
try:
|
||||
_logger.info(f"Creating VM '{name}' from template {template}")
|
||||
|
||||
# Step 1: Get next VMID
|
||||
next_vmid = self._api.cluster.nextid.get()
|
||||
vmid = int(next_vmid)
|
||||
_logger.info(f"Allocated VMID {vmid}")
|
||||
|
||||
# Step 2: Clone template
|
||||
_logger.info(f"Cloning template {template} → VMID {vmid}")
|
||||
clone_upid = self._node_api().qemu(template).clone.post(
|
||||
newid=vmid,
|
||||
full=1,
|
||||
name=name,
|
||||
)
|
||||
self._wait_for_task(clone_upid, timeout=timeout)
|
||||
|
||||
# Step 3: Configure CPU, memory, and dual NICs
|
||||
_logger.info(f"Configuring VM {vmid}: {cpu} CPU, {memory}MB RAM")
|
||||
|
||||
# Build net0 (mgmt) config
|
||||
net0_config = f"virtio,bridge={mgmt_bridge}"
|
||||
if mgmt_vlan_tag is not None:
|
||||
net0_config += f",tag={mgmt_vlan_tag}"
|
||||
|
||||
# Build net1 (capture trunk) config
|
||||
vlan_list = ";".join(str(v) for v in capture_vlan_tags)
|
||||
net1_config = f"virtio,bridge={capture_bridge},trunks={vlan_list}"
|
||||
|
||||
self._node_api().qemu(vmid).config.post(
|
||||
cores=cpu,
|
||||
memory=memory,
|
||||
net0=net0_config,
|
||||
net1=net1_config,
|
||||
)
|
||||
|
||||
# Step 4: Verify snippet storage exists
|
||||
_logger.info("Checking for snippet storage...")
|
||||
storages = self._api.storage.get()
|
||||
snippet_storage = None
|
||||
for storage in storages:
|
||||
content = storage.get("content", "")
|
||||
if "snippets" in content and storage.get("enabled"):
|
||||
snippet_storage = storage["storage"]
|
||||
break
|
||||
|
||||
if not snippet_storage:
|
||||
raise ValueError(
|
||||
"No storage with content='snippets' found. "
|
||||
"Configure a snippet-capable storage (e.g. local, nfs dir) "
|
||||
"and enable it."
|
||||
)
|
||||
_logger.info(f"Using snippet storage: {snippet_storage}")
|
||||
|
||||
# Step 5: Render and upload Cloud-Init config
|
||||
_logger.info(f"Rendering Cloud-Init config for VMID {vmid}")
|
||||
|
||||
# Ensure cloud_init_config includes runcmd to bring up capture NIC
|
||||
if "runcmd" not in cloud_init_config:
|
||||
cloud_init_config["runcmd"] = []
|
||||
if not any("eth1" in cmd if isinstance(cmd, str) else False for cmd in cloud_init_config.get("runcmd", [])):
|
||||
cloud_init_config["runcmd"].insert(0, "ip link set eth1 up")
|
||||
|
||||
user_data_yaml = "#cloud-config\n" + yaml.dump(
|
||||
cloud_init_config, default_flow_style=False
|
||||
)
|
||||
|
||||
filename = f"{vmid}-user-data.yaml"
|
||||
_logger.debug(f"Uploading Cloud-Init snippet {filename} to {snippet_storage}")
|
||||
|
||||
# Upload to snippet storage
|
||||
self._node_api().storage(snippet_storage).upload.post(
|
||||
content="snippets",
|
||||
filename=filename,
|
||||
data=user_data_yaml,
|
||||
)
|
||||
|
||||
# Step 6: Configure Cloud-Init references and SSH keys
|
||||
_logger.info(f"Setting Cloud-Init config for VM {vmid}")
|
||||
|
||||
cloud_init_args = {
|
||||
"ide2": f"{snippet_storage}:cloudinit",
|
||||
"citype": "nocloud",
|
||||
"cicustom": f"user={snippet_storage}:snippets/{filename}",
|
||||
"ipconfig0": "ip=dhcp", # net0 gets DHCP (mgmt)
|
||||
# NO ipconfig1 for net1 (capture NIC stays unnummeriert)
|
||||
}
|
||||
|
||||
if ssh_public_keys:
|
||||
# URL-encode SSH keys for Proxmox API
|
||||
sshkeys = ";".join(ssh_public_keys)
|
||||
cloud_init_args["sshkeys"] = quote(sshkeys)
|
||||
|
||||
self._node_api().qemu(vmid).config.post(**cloud_init_args)
|
||||
|
||||
# Step 7: Start the VM
|
||||
_logger.info(f"Starting VM {vmid}")
|
||||
start_upid = self._node_api().qemu(vmid).status.start.post()
|
||||
self._wait_for_task(start_upid, timeout=timeout)
|
||||
|
||||
_logger.info(f"VM {vmid} ('{name}') provisioned successfully on {self._node_name}")
|
||||
return {
|
||||
"vmid": str(vmid),
|
||||
"name": name,
|
||||
"node": self._node_name,
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
_logger.exception(f"Failed to create VM '{name}': {e}")
|
||||
raise
|
||||
|
||||
def destroy_vm(
|
||||
self,
|
||||
vmid: str,
|
||||
*,
|
||||
remove_disk: bool = True,
|
||||
timeout: int = 60,
|
||||
) -> None:
|
||||
"""
|
||||
Destroy a virtual machine and optionally remove its storage.
|
||||
|
||||
Steps:
|
||||
1. Stop the VM if running
|
||||
2. Delete VM configuration and optionally disks
|
||||
3. Clean up Cloud-Init snippets
|
||||
|
||||
Args:
|
||||
vmid: hypervisor VMID (string, e.g. "101")
|
||||
remove_disk: if True, also delete disks and storage
|
||||
timeout: max seconds for stop/delete operations
|
||||
|
||||
Raises:
|
||||
RuntimeError: VM doesn't exist or destruction fails
|
||||
"""
|
||||
try:
|
||||
vmid_int = int(vmid)
|
||||
_logger.info(f"Destroying VM {vmid}")
|
||||
|
||||
# Step 1: Stop the VM if running
|
||||
try:
|
||||
_logger.debug(f"Stopping VM {vmid}")
|
||||
stop_upid = self._node_api().qemu(vmid_int).status.stop.post()
|
||||
self._wait_for_task(stop_upid, timeout=timeout)
|
||||
except Exception as e:
|
||||
_logger.debug(f"VM {vmid} stop failed (may already be stopped): {e}")
|
||||
|
||||
# Step 2: Delete VM
|
||||
_logger.debug(f"Deleting VM {vmid} configuration and disks")
|
||||
self._node_api().qemu(vmid_int).delete(
|
||||
purge=1,
|
||||
destroy_unreferenced_disks=1 if remove_disk else 0,
|
||||
)
|
||||
|
||||
# Step 3: Clean up Cloud-Init snippets
|
||||
# (This is best-effort; snippet files may be unreachable if storage is unavailable)
|
||||
try:
|
||||
config = self._node_api().qemu(vmid_int).config.get()
|
||||
cicustom = config.get("cicustom", "")
|
||||
if "snippets/" in cicustom:
|
||||
parts = cicustom.split("=")
|
||||
if len(parts) >= 2:
|
||||
snippet_ref = parts[1] # e.g. "snippets:snippets/101-user-data.yaml"
|
||||
storage, filepath = snippet_ref.split(":", 1)
|
||||
_logger.debug(f"Deleting snippet {filepath} from {storage}")
|
||||
try:
|
||||
self._node_api().storage(storage).content(filepath).delete()
|
||||
except Exception as e:
|
||||
_logger.warning(f"Failed to delete snippet {filepath}: {e}")
|
||||
except Exception as e:
|
||||
_logger.debug(f"Could not clean up snippets for VM {vmid}: {e}")
|
||||
|
||||
_logger.info(f"VM {vmid} destroyed successfully")
|
||||
|
||||
except Exception as e:
|
||||
_logger.exception(f"Failed to destroy VM {vmid}: {e}")
|
||||
raise
|
||||
|
||||
def get_vm_status(
|
||||
self,
|
||||
vmid: str,
|
||||
*,
|
||||
wait_for_ip: bool = False,
|
||||
timeout: int = 300,
|
||||
poll_interval: int = 5,
|
||||
) -> VMStatusDict:
|
||||
"""
|
||||
Get the runtime status of a virtual machine.
|
||||
|
||||
Optionally waits for the guest-agent to report an IP address on the
|
||||
management NIC (net0), useful after provisioning.
|
||||
|
||||
Args:
|
||||
vmid: hypervisor VMID (string)
|
||||
wait_for_ip: if True, poll until IP appears on net0
|
||||
timeout: max seconds to wait for IP (if wait_for_ip=True)
|
||||
poll_interval: seconds between status polls
|
||||
|
||||
Returns:
|
||||
{"status": str, "ip_address": str, "hostname": str, "mac_address": str}
|
||||
(ip_address, hostname, mac_address only if VM is running and has network info)
|
||||
|
||||
Raises:
|
||||
RuntimeError: VM doesn't exist or wait_for_ip times out
|
||||
"""
|
||||
try:
|
||||
vmid_int = int(vmid)
|
||||
_logger.debug(f"Getting status for VM {vmid}")
|
||||
|
||||
# Get VM config to infer net0 MAC (for matching guest-agent results)
|
||||
try:
|
||||
config = self._node_api().qemu(vmid_int).config.get()
|
||||
except Exception:
|
||||
# VM may not exist yet or config not readable
|
||||
return {"status": "unknown"}
|
||||
|
||||
# Parse net0 MAC from config (if present)
|
||||
net0_line = config.get("net0", "")
|
||||
expected_mac = None
|
||||
# Example: "virtio,bridge=vmbr0,tag=10" — no explicit MAC
|
||||
# Proxmox auto-generates MACs in a deterministic pattern, but we'll
|
||||
# match by looking for the first NIC's IP in guest-agent results
|
||||
|
||||
# Polling loop
|
||||
start_time = time.time()
|
||||
while True:
|
||||
elapsed = time.time() - start_time
|
||||
if wait_for_ip and elapsed > timeout:
|
||||
raise RuntimeError(
|
||||
f"VM {vmid} failed to acquire IP within {timeout}s"
|
||||
)
|
||||
|
||||
try:
|
||||
# Query guest-agent network interfaces
|
||||
agent_info = self._node_api().qemu(vmid_int).agent.network_get_interfaces.get()
|
||||
interfaces = agent_info.get("result", [])
|
||||
|
||||
# Find net0 (first interface with IP)
|
||||
if interfaces:
|
||||
net0_iface = interfaces[0] # Assumes net0 is first in list
|
||||
net0_mac = net0_iface.get("hardware-address", "")
|
||||
ip_addresses = net0_iface.get("ip-addresses", [])
|
||||
|
||||
if ip_addresses:
|
||||
# Found IP
|
||||
ip_info = ip_addresses[0]
|
||||
ip_addr = ip_info.get("ip-address", "")
|
||||
if ip_addr:
|
||||
_logger.info(f"VM {vmid} acquired IP {ip_addr}")
|
||||
return {
|
||||
"status": "running",
|
||||
"ip_address": ip_addr,
|
||||
"hostname": net0_iface.get("name", ""),
|
||||
"mac_address": net0_mac,
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
_logger.debug(f"Error querying guest-agent for VM {vmid}: {e}")
|
||||
|
||||
if not wait_for_ip:
|
||||
# Return immediate status without IP
|
||||
return {"status": "running"}
|
||||
|
||||
# Wait before next poll
|
||||
time.sleep(poll_interval)
|
||||
|
||||
except RuntimeError:
|
||||
raise
|
||||
except Exception as e:
|
||||
_logger.exception(f"Error getting status for VM {vmid}: {e}")
|
||||
raise RuntimeError(f"Failed to get VM {vmid} status: {e}")
|
||||
Reference in New Issue
Block a user