feat: NAPALM drivers for VMware ESXi and vCenter

Two drivers from one package, both in the hypervisor role:

- vmware_esxi talks to one host directly: facts, vmnics and vmkernel
  NICs, CDP/LLDP neighbours, sensors, VMs, datastores and port groups.
- vmware_vcenter talks to a vCenter: every VM of every host it manages,
  with the host as the VM's node, plus distributed port groups. It
  reports no interfaces of its own; the hosts' NICs belong to the hosts.

Both implement the HypervisorDriver VM contract: get_vms, get_vm_config,
start/stop/reboot/suspend_vm and the four snapshot methods, and emit raw
device warnings (maintenance mode, disconnected host, config issues,
host managed by a vCenter, free license making the API read-only).

Every read is a PropertyCollector query for the explicit paths in
paths.py, converted by to_plain() into dicts and lists; the parsers
only ever see that. tools/harvest.py dumps exactly those paths to JSON
and tools/sanitize.py scrubs the dump, so a real host can become a test
fixture without code changes. A VM's vmid is its instance UUID, which
survives vMotion and re-registration; a MoRef does not.

Tested against govmomi's vcsim in ESXi and vCenter mode, including real
power and snapshot tasks. Not yet tested against real hardware.
This commit is contained in:
Christian Manivong
2026-09-24 09:07:30 +02:00
commit c8e472b4c6
47 changed files with 14645 additions and 0 deletions
+79
View File
@@ -0,0 +1,79 @@
#!/usr/bin/env python3
"""Dump what the drivers read from an ESXi host or vCenter into one JSON file.
Usage::
tools/harvest.py <host> <user> <label> [--port 443] [--verify-ssl]
The password is prompted for (or taken from ``$VSPHERE_PASSWORD``). Output
goes to ``tools/harvest-out/<label>.json``, which is gitignored: it holds
serial numbers, MACs, IPs and names. Run ``tools/sanitize.py`` on it and read
the result before anything goes into ``tests/fixtures/``.
The file has the shape the test fake (``tests/fake_inventory.py``) serves::
{"about": {...}, "licenses": [...], "objects": {"HostSystem": [...], ...}}
It reads exactly the property paths in ``napalm_vmware.paths``, the same ones
the drivers ask for, so a fixture never contains a field no parser uses and
never lacks one a parser needs.
"""
from __future__ import annotations
import argparse
import getpass
import json
import os
import ssl
import sys
from pathlib import Path
from pyVim.connect import Disconnect, SmartConnect
from napalm_vmware import paths
from napalm_vmware._inventory import Inventory
OUT_DIR = Path(__file__).resolve().parent / "harvest-out"
def harvest(inventory: Inventory) -> dict:
"""Everything the drivers read, as one plain dict."""
return {
"about": inventory.about(),
"licenses": inventory.licenses(),
"objects": {name: inventory.collect(name, p) for name, p in paths.ALL.items()},
}
def main() -> int: # pragma: no cover - opens a live vSphere session
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
parser.add_argument("host")
parser.add_argument("user")
parser.add_argument("label")
parser.add_argument("--port", type=int, default=443)
parser.add_argument("--verify-ssl", action="store_true")
args = parser.parse_args()
password = os.environ.get("VSPHERE_PASSWORD") or getpass.getpass()
context = ssl.create_default_context()
if not args.verify_ssl:
context.check_hostname = False
context.verify_mode = ssl.CERT_NONE
si = SmartConnect(
host=args.host, port=args.port, user=args.user, pwd=password, sslContext=context
)
try:
data = harvest(Inventory(si.RetrieveContent()))
finally:
Disconnect(si)
OUT_DIR.mkdir(exist_ok=True)
out = OUT_DIR / f"{args.label}.json"
out.write_text(json.dumps(data, indent=1, sort_keys=True))
print(f"wrote {out} -- sanitise it before committing anything", file=sys.stderr)
return 0
if __name__ == "__main__": # pragma: no cover
sys.exit(main())
+121
View File
@@ -0,0 +1,121 @@
#!/usr/bin/env python3
"""Scrub a harvest dump so it can be committed as a test fixture.
Usage::
tools/sanitize.py tools/harvest-out/<label>.json > tests/fixtures/<label>.json
Two passes. The first walks the dump and collects every value stored under a
sensitive key -- names, serials, UUIDs, MACs, IPs, license keys -- and assigns
each a stable placeholder. The second replaces every occurrence of those
values anywhere in the dump, so a VM name embedded in a disk path
(``[ds1] payroll-db/payroll-db.vmdk``) is caught too, and references between
objects still line up afterwards.
Read the result before committing it. A key this script does not know about
is not scrubbed.
"""
from __future__ import annotations
import ipaddress
import json
import sys
from itertools import count
from typing import Any
_NAME_KEYS = {"name", "hostName", "domainName", "searchDomain"}
_SERIAL_KEYS = {"serialNumber", "identifierValue"}
_UUID_KEYS = {"uuid", "instanceUuid", "config.instanceUuid", "switchUuid"}
_MAC_KEYS = {"mac", "macAddress", "sourceMac"}
_IP_KEYS = {"ipAddress", "address", "managementServerIp"}
_LICENSE_KEYS = {"licenseKey"}
#: Values that identify nothing and must stay as they are.
_KEEP = {"", "0.0.0.0", "::", "127.0.0.1"}
class _Placeholders:
def __init__(self) -> None:
self.mapping: dict[str, str] = {}
self._n = count(1)
def add(self, value: str, make: Any) -> None:
if value not in _KEEP and value not in self.mapping and len(value) > 2:
self.mapping[value] = make(next(self._n))
def _ip(n: int, original: str) -> str:
if isinstance(ipaddress.ip_address(original), ipaddress.IPv6Address):
return f"2001:db8::{n:x}"
return f"192.0.2.{n % 254 + 1}"
def _add(ph: _Placeholders, key: str, value: str) -> None:
if key in _NAME_KEYS:
ph.add(value, lambda n: f"name{n:03d}")
elif key in _SERIAL_KEYS:
ph.add(value.strip(), lambda n: f"SERIAL{n:04d}")
elif key in _UUID_KEYS:
ph.add(value, lambda n: f"00000000-0000-0000-0000-{n:012d}")
elif key in _MAC_KEYS:
ph.add(value.lower(), lambda n: f"00:11:22:33:{n // 256:02x}:{n % 256:02x}")
elif key in _LICENSE_KEYS:
ph.add(value, lambda n: f"00000-00000-00000-00000-{n:05d}")
elif key in _IP_KEYS:
try:
ipaddress.ip_address(value)
except ValueError:
return
ph.add(value, lambda n, v=value: _ip(n, v))
def _collect(node: Any, ph: _Placeholders, key: str = "") -> None:
if isinstance(node, dict):
# A data object's type description ("Service tag") is not sensitive.
if node.get("_type") == "ElementDescription":
return
for k, v in node.items():
_collect(v, ph, k)
elif isinstance(node, list):
for item in node:
_collect(item, ph, key)
elif isinstance(node, str):
_add(ph, key, node)
def _replace(node: Any, mapping: list[tuple[str, str]]) -> Any:
if isinstance(node, dict):
return {k: _replace(v, mapping) for k, v in node.items()}
if isinstance(node, list):
return [_replace(v, mapping) for v in node]
if isinstance(node, str):
for original, placeholder in mapping:
if original in node or original in node.lower():
node = node.replace(original, placeholder)
node = node.replace(original.upper(), placeholder)
return node
return node
def sanitize(data: dict[str, Any]) -> dict[str, Any]:
"""A scrubbed copy of a harvest dump."""
ph = _Placeholders()
_collect(data, ph)
# Longest first, so "esx01.corp.example.com" is replaced before "esx01".
ordered = sorted(ph.mapping.items(), key=lambda kv: len(kv[0]), reverse=True)
return _replace(data, ordered)
def main() -> int: # pragma: no cover - file I/O wrapper
if len(sys.argv) != 2:
print(__doc__, file=sys.stderr)
return 2
with open(sys.argv[1]) as fh:
data = json.load(fh)
json.dump(sanitize(data), sys.stdout, indent=1, sort_keys=True)
print("\nread the output before committing it", file=sys.stderr)
return 0
if __name__ == "__main__": # pragma: no cover
sys.exit(main())