Two drivers from one package, both in the hypervisor role: - vmware_esxi talks to one host directly: facts, vmnics and vmkernel NICs, CDP/LLDP neighbours, sensors, VMs, datastores and port groups. - vmware_vcenter talks to a vCenter: every VM of every host it manages, with the host as the VM's node, plus distributed port groups. It reports no interfaces of its own; the hosts' NICs belong to the hosts. Both implement the HypervisorDriver VM contract: get_vms, get_vm_config, start/stop/reboot/suspend_vm and the four snapshot methods, and emit raw device warnings (maintenance mode, disconnected host, config issues, host managed by a vCenter, free license making the API read-only). Every read is a PropertyCollector query for the explicit paths in paths.py, converted by to_plain() into dicts and lists; the parsers only ever see that. tools/harvest.py dumps exactly those paths to JSON and tools/sanitize.py scrubs the dump, so a real host can become a test fixture without code changes. A VM's vmid is its instance UUID, which survives vMotion and re-registration; a MoRef does not. Tested against govmomi's vcsim in ESXi and vCenter mode, including real power and snapshot tasks. Not yet tested against real hardware.
122 lines
4.1 KiB
Python
Executable File
122 lines
4.1 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Scrub a harvest dump so it can be committed as a test fixture.
|
|
|
|
Usage::
|
|
|
|
tools/sanitize.py tools/harvest-out/<label>.json > tests/fixtures/<label>.json
|
|
|
|
Two passes. The first walks the dump and collects every value stored under a
|
|
sensitive key -- names, serials, UUIDs, MACs, IPs, license keys -- and assigns
|
|
each a stable placeholder. The second replaces every occurrence of those
|
|
values anywhere in the dump, so a VM name embedded in a disk path
|
|
(``[ds1] payroll-db/payroll-db.vmdk``) is caught too, and references between
|
|
objects still line up afterwards.
|
|
|
|
Read the result before committing it. A key this script does not know about
|
|
is not scrubbed.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import ipaddress
|
|
import json
|
|
import sys
|
|
from itertools import count
|
|
from typing import Any
|
|
|
|
_NAME_KEYS = {"name", "hostName", "domainName", "searchDomain"}
|
|
_SERIAL_KEYS = {"serialNumber", "identifierValue"}
|
|
_UUID_KEYS = {"uuid", "instanceUuid", "config.instanceUuid", "switchUuid"}
|
|
_MAC_KEYS = {"mac", "macAddress", "sourceMac"}
|
|
_IP_KEYS = {"ipAddress", "address", "managementServerIp"}
|
|
_LICENSE_KEYS = {"licenseKey"}
|
|
#: Values that identify nothing and must stay as they are.
|
|
_KEEP = {"", "0.0.0.0", "::", "127.0.0.1"}
|
|
|
|
|
|
class _Placeholders:
|
|
def __init__(self) -> None:
|
|
self.mapping: dict[str, str] = {}
|
|
self._n = count(1)
|
|
|
|
def add(self, value: str, make: Any) -> None:
|
|
if value not in _KEEP and value not in self.mapping and len(value) > 2:
|
|
self.mapping[value] = make(next(self._n))
|
|
|
|
|
|
def _ip(n: int, original: str) -> str:
|
|
if isinstance(ipaddress.ip_address(original), ipaddress.IPv6Address):
|
|
return f"2001:db8::{n:x}"
|
|
return f"192.0.2.{n % 254 + 1}"
|
|
|
|
|
|
def _add(ph: _Placeholders, key: str, value: str) -> None:
|
|
if key in _NAME_KEYS:
|
|
ph.add(value, lambda n: f"name{n:03d}")
|
|
elif key in _SERIAL_KEYS:
|
|
ph.add(value.strip(), lambda n: f"SERIAL{n:04d}")
|
|
elif key in _UUID_KEYS:
|
|
ph.add(value, lambda n: f"00000000-0000-0000-0000-{n:012d}")
|
|
elif key in _MAC_KEYS:
|
|
ph.add(value.lower(), lambda n: f"00:11:22:33:{n // 256:02x}:{n % 256:02x}")
|
|
elif key in _LICENSE_KEYS:
|
|
ph.add(value, lambda n: f"00000-00000-00000-00000-{n:05d}")
|
|
elif key in _IP_KEYS:
|
|
try:
|
|
ipaddress.ip_address(value)
|
|
except ValueError:
|
|
return
|
|
ph.add(value, lambda n, v=value: _ip(n, v))
|
|
|
|
|
|
def _collect(node: Any, ph: _Placeholders, key: str = "") -> None:
|
|
if isinstance(node, dict):
|
|
# A data object's type description ("Service tag") is not sensitive.
|
|
if node.get("_type") == "ElementDescription":
|
|
return
|
|
for k, v in node.items():
|
|
_collect(v, ph, k)
|
|
elif isinstance(node, list):
|
|
for item in node:
|
|
_collect(item, ph, key)
|
|
elif isinstance(node, str):
|
|
_add(ph, key, node)
|
|
|
|
|
|
def _replace(node: Any, mapping: list[tuple[str, str]]) -> Any:
|
|
if isinstance(node, dict):
|
|
return {k: _replace(v, mapping) for k, v in node.items()}
|
|
if isinstance(node, list):
|
|
return [_replace(v, mapping) for v in node]
|
|
if isinstance(node, str):
|
|
for original, placeholder in mapping:
|
|
if original in node or original in node.lower():
|
|
node = node.replace(original, placeholder)
|
|
node = node.replace(original.upper(), placeholder)
|
|
return node
|
|
return node
|
|
|
|
|
|
def sanitize(data: dict[str, Any]) -> dict[str, Any]:
|
|
"""A scrubbed copy of a harvest dump."""
|
|
ph = _Placeholders()
|
|
_collect(data, ph)
|
|
# Longest first, so "esx01.corp.example.com" is replaced before "esx01".
|
|
ordered = sorted(ph.mapping.items(), key=lambda kv: len(kv[0]), reverse=True)
|
|
return _replace(data, ordered)
|
|
|
|
|
|
def main() -> int: # pragma: no cover - file I/O wrapper
|
|
if len(sys.argv) != 2:
|
|
print(__doc__, file=sys.stderr)
|
|
return 2
|
|
with open(sys.argv[1]) as fh:
|
|
data = json.load(fh)
|
|
json.dump(sanitize(data), sys.stdout, indent=1, sort_keys=True)
|
|
print("\nread the output before committing it", file=sys.stderr)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__": # pragma: no cover
|
|
sys.exit(main())
|