Merge pull request 'feat(vm-provision): name a CPU model for every new VM, and list the choices' (#5) from feat/vm-cpu-type into master
This commit was merged in pull request #5.
This commit is contained in:
@@ -8,7 +8,7 @@ import logging
|
|||||||
import re
|
import re
|
||||||
import time
|
import time
|
||||||
import yaml
|
import yaml
|
||||||
from typing import Any, Dict, List
|
from typing import TYPE_CHECKING, Any, Dict, List
|
||||||
from urllib.parse import quote
|
from urllib.parse import quote
|
||||||
|
|
||||||
from napalm_device_types.models import (
|
from napalm_device_types.models import (
|
||||||
@@ -18,8 +18,53 @@ from napalm_device_types.models import (
|
|||||||
VMStatusDict,
|
VMStatusDict,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
# Type-only: VMCpuTypeDict is newer than the napalm_device_types floor in
|
||||||
|
# pyproject.toml, and nothing here needs it at runtime.
|
||||||
|
from napalm_device_types.models import VMCpuTypeDict
|
||||||
|
|
||||||
_logger = logging.getLogger(__name__)
|
_logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# The CPU models a new VM may be given. Each lists the /proc/cpuinfo flags it
|
||||||
|
# adds on top of QEMU's qemu64 baseline: what the node's CPU must have for the
|
||||||
|
# model to start at all, and what a guest can count on. The sets follow
|
||||||
|
# Proxmox's own x86-64-v* definitions (qemu-server, PVE/QemuServer/CPUConfig.pm).
|
||||||
|
#
|
||||||
|
# Leaving the model out of qemu.post is not neutral: Proxmox then falls back to
|
||||||
|
# kvm64, which lacks even AES-NI, let alone the AVX MongoDB 5.0+ needs
|
||||||
|
# (netOrk#494). The default below is what the Proxmox GUI picks since PVE 8.
|
||||||
|
_X86_64_V2_AES_FLAGS = ("aes", "popcnt", "pni", "sse4_1", "sse4_2", "ssse3")
|
||||||
|
_X86_64_V3_FLAGS = _X86_64_V2_AES_FLAGS + (
|
||||||
|
"avx",
|
||||||
|
"avx2",
|
||||||
|
"bmi1",
|
||||||
|
"bmi2",
|
||||||
|
"f16c",
|
||||||
|
"fma",
|
||||||
|
"abm",
|
||||||
|
"movbe",
|
||||||
|
"xsave",
|
||||||
|
)
|
||||||
|
_DEFAULT_CPU_TYPE = "x86-64-v2-AES"
|
||||||
|
_CPU_MODELS = (
|
||||||
|
(
|
||||||
|
"x86-64-v2-AES",
|
||||||
|
_X86_64_V2_AES_FLAGS,
|
||||||
|
"Proxmox's own default: runs on practically any x86-64 server CPU and "
|
||||||
|
"can live-migrate between different ones. No AVX.",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"x86-64-v3",
|
||||||
|
_X86_64_V3_FLAGS,
|
||||||
|
"Adds AVX and AVX2 (which MongoDB 5.0 and later need). Every node the VM "
|
||||||
|
"may run on needs an Intel Haswell or AMD Excavator CPU (2013) or newer.",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
_HOST_CPU_DESCRIPTION = (
|
||||||
|
"This node's CPU, passed through unchanged: fastest, with every feature it "
|
||||||
|
"has, but the VM can only live-migrate to nodes with the same CPU."
|
||||||
|
)
|
||||||
|
|
||||||
# Downloaded cloud images are cached here on the hypervisor node, keyed by
|
# Downloaded cloud images are cached here on the hypervisor node, keyed by
|
||||||
# filename, so provisioning multiple VMs from the same image only pays the
|
# filename, so provisioning multiple VMs from the same image only pays the
|
||||||
# download cost once.
|
# download cost once.
|
||||||
@@ -351,6 +396,56 @@ class ProxmoxVMProvisionMixin:
|
|||||||
)
|
)
|
||||||
return targets
|
return targets
|
||||||
|
|
||||||
|
def get_vm_cpu_types(self) -> list[VMCpuTypeDict]:
|
||||||
|
"""List the CPU models a new VM may be given, judged against this node's CPU."""
|
||||||
|
return self._cpu_types_for(self._node_cpu_flags())
|
||||||
|
|
||||||
|
def _node_cpu_flags(self) -> set[str]:
|
||||||
|
status = self._node_api().status.get() or {}
|
||||||
|
return set(str((status.get("cpuinfo") or {}).get("flags", "")).split())
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _cpu_types_for(node_flags: set[str]) -> list[VMCpuTypeDict]:
|
||||||
|
types: list[VMCpuTypeDict] = [
|
||||||
|
{
|
||||||
|
"name": name,
|
||||||
|
"description": description,
|
||||||
|
"features": list(flags),
|
||||||
|
"available": set(flags) <= node_flags,
|
||||||
|
"default": name == _DEFAULT_CPU_TYPE,
|
||||||
|
}
|
||||||
|
for name, flags, description in _CPU_MODELS
|
||||||
|
]
|
||||||
|
types.append(
|
||||||
|
{
|
||||||
|
"name": "host",
|
||||||
|
"description": _HOST_CPU_DESCRIPTION,
|
||||||
|
"features": sorted(node_flags),
|
||||||
|
"available": True,
|
||||||
|
"default": False,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
return types
|
||||||
|
|
||||||
|
def _resolve_cpu_type(self, cpu_type: str | None) -> str:
|
||||||
|
"""The model to create the VM with. A model asked for by name is checked
|
||||||
|
against this node's CPU first: one it cannot run would fail only at VM
|
||||||
|
start, after the disk import, leaving a half-built VM behind."""
|
||||||
|
if cpu_type is None:
|
||||||
|
return _DEFAULT_CPU_TYPE
|
||||||
|
node_flags = self._node_cpu_flags()
|
||||||
|
offered = {t["name"]: t for t in self._cpu_types_for(node_flags)}
|
||||||
|
entry = offered.get(cpu_type)
|
||||||
|
if entry is None:
|
||||||
|
raise ValueError(f"Unknown CPU type {cpu_type!r}; choose one of {', '.join(offered)}")
|
||||||
|
if not entry["available"]:
|
||||||
|
missing = sorted(set(entry["features"]) - node_flags)
|
||||||
|
raise ValueError(
|
||||||
|
f"CPU type {cpu_type!r} needs {', '.join(missing)}, which the CPU of "
|
||||||
|
f"node {self._node_name} does not have"
|
||||||
|
)
|
||||||
|
return cpu_type
|
||||||
|
|
||||||
def _wait_for_task(self, upid: str, timeout: int = 120) -> None:
|
def _wait_for_task(self, upid: str, timeout: int = 120) -> None:
|
||||||
"""
|
"""
|
||||||
Poll a Proxmox task until completion.
|
Poll a Proxmox task until completion.
|
||||||
@@ -395,6 +490,7 @@ class ProxmoxVMProvisionMixin:
|
|||||||
ssh_public_keys: List[str] | None = None,
|
ssh_public_keys: List[str] | None = None,
|
||||||
disk_resize_gb: int | None = None,
|
disk_resize_gb: int | None = None,
|
||||||
storage: str | None = None,
|
storage: str | None = None,
|
||||||
|
cpu_type: str | None = None,
|
||||||
download_timeout: int = 300,
|
download_timeout: int = 300,
|
||||||
timeout: int = 180,
|
timeout: int = 180,
|
||||||
) -> VMProvisionResultDict:
|
) -> VMProvisionResultDict:
|
||||||
@@ -427,6 +523,7 @@ class ProxmoxVMProvisionMixin:
|
|||||||
disk_resize_gb: resize root disk to this size (None = no resize)
|
disk_resize_gb: resize root disk to this size (None = no resize)
|
||||||
storage: storage pool for the root disk (None = auto-detect first
|
storage: storage pool for the root disk (None = auto-detect first
|
||||||
enabled, node-available storage with content='images')
|
enabled, node-available storage with content='images')
|
||||||
|
cpu_type: CPU model from get_vm_cpu_types() (None = x86-64-v2-AES)
|
||||||
download_timeout: max seconds for the image download (skipped if cached)
|
download_timeout: max seconds for the image download (skipped if cached)
|
||||||
timeout: max seconds for the remaining provisioning steps
|
timeout: max seconds for the remaining provisioning steps
|
||||||
|
|
||||||
@@ -435,10 +532,13 @@ class ProxmoxVMProvisionMixin:
|
|||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
RuntimeError: provisioning failure (download, import, config, timeout, etc.)
|
RuntimeError: provisioning failure (download, import, config, timeout, etc.)
|
||||||
ValueError: invalid storage or configuration
|
ValueError: invalid storage or configuration, or a cpu_type that is
|
||||||
|
unknown or that this node's CPU cannot run (raised before
|
||||||
|
anything is created)
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
_logger.info(f"Creating VM '{name}' from image {image_url}")
|
_logger.info(f"Creating VM '{name}' from image {image_url}")
|
||||||
|
cpu_model = self._resolve_cpu_type(cpu_type)
|
||||||
|
|
||||||
# Step 1: Get next VMID
|
# Step 1: Get next VMID
|
||||||
next_vmid = self._api.cluster.nextid.get()
|
next_vmid = self._api.cluster.nextid.get()
|
||||||
@@ -452,6 +552,7 @@ class ProxmoxVMProvisionMixin:
|
|||||||
name=name,
|
name=name,
|
||||||
memory=memory,
|
memory=memory,
|
||||||
cores=cpu,
|
cores=cpu,
|
||||||
|
cpu=cpu_model,
|
||||||
ostype="l26",
|
ostype="l26",
|
||||||
scsihw="virtio-scsi-pci",
|
scsihw="virtio-scsi-pci",
|
||||||
# Without this, Proxmox never attaches the virtio-serial
|
# Without this, Proxmox never attaches the virtio-serial
|
||||||
|
|||||||
@@ -0,0 +1,153 @@
|
|||||||
|
"""Which virtual CPU model a new VM gets.
|
||||||
|
|
||||||
|
Created without a ``cpu`` argument, a Proxmox VM falls back to ``kvm64``:
|
||||||
|
no AVX, no AES-NI. MongoDB 5.0 and later will not even start on it, which is
|
||||||
|
how a Graylog provisioned through netOrk failed (netOrk#494). The driver now
|
||||||
|
always names a model, defaulting to ``x86-64-v2-AES`` as the Proxmox GUI does,
|
||||||
|
and lists the alternatives with what each needs from the node's CPU.
|
||||||
|
|
||||||
|
The flag sets below are trimmed from real ``/nodes/{node}/status`` answers in a
|
||||||
|
mixed cluster: a Celeron J3455 (no AVX at all) and an i7-7700 (AVX2).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from napalm_proxmox.vm_provision_mixin import ProxmoxVMProvisionMixin
|
||||||
|
|
||||||
|
CELERON_J3455 = (
|
||||||
|
"fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush "
|
||||||
|
"mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc pni pclmulqdq "
|
||||||
|
"ssse3 cx16 sse4_1 sse4_2 x2apic movbe popcnt aes rdrand lahf_lm 3dnowprefetch "
|
||||||
|
"erms mpx rdseed smap clflushopt sha_ni xsaveopt xsavec xgetbv1"
|
||||||
|
)
|
||||||
|
CORE_I7_7700 = (
|
||||||
|
"fpu vme de pse tsc msr pae mce cx8 apic sep mtrr pge mca cmov pat pse36 clflush "
|
||||||
|
"mmx fxsr sse sse2 ss ht syscall nx pdpe1gb rdtscp lm constant_tsc pni pclmulqdq "
|
||||||
|
"ssse3 fma cx16 sse4_1 sse4_2 x2apic movbe popcnt aes xsave avx f16c rdrand "
|
||||||
|
"lahf_lm abm 3dnowprefetch fsgsbase bmi1 hle avx2 smep bmi2 erms invpcid rtm mpx "
|
||||||
|
"rdseed adx smap clflushopt xsaveopt xsavec xgetbv1 xsaves"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _mixin_on(flags: str) -> tuple[ProxmoxVMProvisionMixin, MagicMock, MagicMock]:
|
||||||
|
"""A mixin whose node reports `flags` and that can run a full create."""
|
||||||
|
mixin = ProxmoxVMProvisionMixin()
|
||||||
|
mixin._node_name = "pve1"
|
||||||
|
|
||||||
|
api = MagicMock()
|
||||||
|
api.cluster.nextid.get.return_value = 120
|
||||||
|
api.storage.return_value.get.return_value = {"path": "/var/lib/vz"}
|
||||||
|
|
||||||
|
node = MagicMock()
|
||||||
|
node.status.get.return_value = {"cpuinfo": {"model": "test", "flags": flags}}
|
||||||
|
node.storage.get.return_value = [
|
||||||
|
{"storage": "local-lvm", "type": "lvmthin", "content": "images,rootdir", "enabled": 1},
|
||||||
|
{"storage": "snippets", "type": "dir", "content": "snippets", "enabled": 1},
|
||||||
|
]
|
||||||
|
vm = MagicMock()
|
||||||
|
vm.config.get.return_value = {
|
||||||
|
"unused0": "local-lvm:vm-120-disk-0",
|
||||||
|
"scsi0": "local-lvm:vm-120-disk-0",
|
||||||
|
}
|
||||||
|
vm.status.start.post.return_value = "UPID:pve1:1:start"
|
||||||
|
node.qemu.return_value = vm
|
||||||
|
task = MagicMock()
|
||||||
|
task.status.get.return_value = {"status": "stopped", "exitstatus": "OK"}
|
||||||
|
node.tasks.return_value = task
|
||||||
|
|
||||||
|
mixin._api = api
|
||||||
|
mixin._node_api = MagicMock(return_value=node)
|
||||||
|
mixin._download_cloud_image = MagicMock(return_value="/var/lib/vz/template/x.qcow2")
|
||||||
|
mixin._run_node_command = MagicMock(return_value="")
|
||||||
|
return mixin, api, node
|
||||||
|
|
||||||
|
|
||||||
|
def _create(mixin: ProxmoxVMProvisionMixin, **kwargs):
|
||||||
|
with patch("time.sleep"):
|
||||||
|
return mixin.create_vm_from_cloud_init(
|
||||||
|
name="graylog-01",
|
||||||
|
image_url="https://cloud-images.ubuntu.com/noble/current/noble-server-cloudimg-amd64.img",
|
||||||
|
cpu=2,
|
||||||
|
memory=4096,
|
||||||
|
nics=[{"bridge": "vmbr0"}],
|
||||||
|
cloud_init_config={"hostname": "graylog-01"},
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _by_name(types):
|
||||||
|
return {t["name"]: t for t in types}
|
||||||
|
|
||||||
|
|
||||||
|
class TestListing:
|
||||||
|
def test_offers_the_three_models_in_order(self):
|
||||||
|
mixin, _, _ = _mixin_on(CORE_I7_7700)
|
||||||
|
assert [t["name"] for t in mixin.get_vm_cpu_types()] == [
|
||||||
|
"x86-64-v2-AES",
|
||||||
|
"x86-64-v3",
|
||||||
|
"host",
|
||||||
|
]
|
||||||
|
|
||||||
|
def test_exactly_one_default_and_it_is_what_the_gui_uses(self):
|
||||||
|
mixin, _, _ = _mixin_on(CORE_I7_7700)
|
||||||
|
defaults = [t["name"] for t in mixin.get_vm_cpu_types() if t["default"]]
|
||||||
|
assert defaults == ["x86-64-v2-AES"]
|
||||||
|
|
||||||
|
def test_v3_gives_avx_and_runs_on_a_core_i7(self):
|
||||||
|
mixin, _, _ = _mixin_on(CORE_I7_7700)
|
||||||
|
v3 = _by_name(mixin.get_vm_cpu_types())["x86-64-v3"]
|
||||||
|
assert v3["available"] is True
|
||||||
|
assert {"avx", "avx2"} <= set(v3["features"])
|
||||||
|
|
||||||
|
def test_v3_is_unavailable_on_a_celeron_without_avx(self):
|
||||||
|
mixin, _, _ = _mixin_on(CELERON_J3455)
|
||||||
|
types = _by_name(mixin.get_vm_cpu_types())
|
||||||
|
assert types["x86-64-v3"]["available"] is False
|
||||||
|
assert types["x86-64-v2-AES"]["available"] is True
|
||||||
|
|
||||||
|
def test_v2_aes_has_no_avx(self):
|
||||||
|
mixin, _, _ = _mixin_on(CORE_I7_7700)
|
||||||
|
assert "avx" not in _by_name(mixin.get_vm_cpu_types())["x86-64-v2-AES"]["features"]
|
||||||
|
|
||||||
|
def test_host_passes_the_nodes_own_flags_through(self):
|
||||||
|
"""On a CPU without AVX, `host` gives none either -- a role that needs
|
||||||
|
AVX must not be told `host` would help there."""
|
||||||
|
mixin, _, _ = _mixin_on(CELERON_J3455)
|
||||||
|
host = _by_name(mixin.get_vm_cpu_types())["host"]
|
||||||
|
assert host["available"] is True
|
||||||
|
assert "avx" not in host["features"]
|
||||||
|
assert "aes" in host["features"]
|
||||||
|
|
||||||
|
def test_every_entry_explains_itself(self):
|
||||||
|
mixin, _, _ = _mixin_on(CORE_I7_7700)
|
||||||
|
assert all(t["description"] for t in mixin.get_vm_cpu_types())
|
||||||
|
|
||||||
|
|
||||||
|
class TestCreate:
|
||||||
|
def test_names_the_default_model_instead_of_leaving_kvm64(self):
|
||||||
|
mixin, _, node = _mixin_on(CORE_I7_7700)
|
||||||
|
_create(mixin)
|
||||||
|
assert node.qemu.post.call_args[1]["cpu"] == "x86-64-v2-AES"
|
||||||
|
|
||||||
|
def test_passes_a_chosen_model_through(self):
|
||||||
|
mixin, _, node = _mixin_on(CORE_I7_7700)
|
||||||
|
_create(mixin, cpu_type="host")
|
||||||
|
assert node.qemu.post.call_args[1]["cpu"] == "host"
|
||||||
|
|
||||||
|
def test_refuses_a_model_the_node_cannot_run_before_creating_anything(self):
|
||||||
|
mixin, api, node = _mixin_on(CELERON_J3455)
|
||||||
|
with pytest.raises(ValueError, match="avx"):
|
||||||
|
_create(mixin, cpu_type="x86-64-v3")
|
||||||
|
api.cluster.nextid.get.assert_not_called()
|
||||||
|
node.qemu.post.assert_not_called()
|
||||||
|
|
||||||
|
def test_refuses_an_unknown_model_before_creating_anything(self):
|
||||||
|
mixin, api, node = _mixin_on(CORE_I7_7700)
|
||||||
|
with pytest.raises(ValueError, match="kvm64"):
|
||||||
|
_create(mixin, cpu_type="kvm64")
|
||||||
|
api.cluster.nextid.get.assert_not_called()
|
||||||
|
node.qemu.post.assert_not_called()
|
||||||
Reference in New Issue
Block a user