"""Cloud images (qcow2/raw) converted to streamOptimized VMDKs, cached by URL. ESXi cannot boot a qcow2 and cannot download one itself, so the conversion runs where the driver runs: download, verify, ``qemu-img convert``. The result is kept, keyed by URL, so the next VM from the same image skips both steps. Only the converted disk is kept; the download is deleted. """ from __future__ import annotations import hashlib import json import os import shutil import subprocess import tempfile from collections.abc import Callable from pathlib import Path import requests _CHUNK = 1024 * 1024 DEFAULT_CACHE_DIR = Path(tempfile.gettempdir()) / "napalm-vmware-images" Fetch = Callable[[str, Path, float], None] Convert = Callable[[Path, Path], None] VirtualSize = Callable[[Path], int] def _qemu_img() -> str: path = shutil.which("qemu-img") if path is None: raise RuntimeError( "qemu-img is not installed where netOrk runs the provisioning job; " "it is needed to convert cloud images for VMware (package qemu-utils)" ) return path def fetch(url: str, dest: Path, timeout: float) -> None: # pragma: no cover - network with requests.get(url, stream=True, timeout=timeout) as response: response.raise_for_status() with dest.open("wb") as out: for chunk in response.iter_content(_CHUNK): out.write(chunk) def convert_to_vmdk(src: Path, dst: Path) -> None: subprocess.run( [ _qemu_img(), "convert", "-O", "vmdk", "-o", "subformat=streamOptimized", str(src), str(dst), ], check=True, capture_output=True, ) def virtual_size(path: Path) -> int: """The disk size the image describes, in bytes (not the file size).""" out = subprocess.run( [_qemu_img(), "info", "--output", "json", str(path)], check=True, capture_output=True, text=True, ).stdout return int(json.loads(out)["virtual-size"]) def _digest(path: Path, algorithm: str) -> str: h = hashlib.new(algorithm) with path.open("rb") as fh: for chunk in iter(lambda: fh.read(_CHUNK), b""): h.update(chunk) return h.hexdigest() class ImageCache: """Converted images in ``directory``, one ``.vmdk`` plus ``.json`` each.""" def __init__( self, directory: Path = DEFAULT_CACHE_DIR, *, fetch: Fetch = fetch, convert: Convert = convert_to_vmdk, virtual_size: VirtualSize = virtual_size, ) -> None: self._dir = Path(directory) self._fetch = fetch self._convert = convert self._virtual_size = virtual_size def vmdk(self, url: str, checksum: str | None, timeout: float) -> tuple[Path, int]: """``(path to the VMDK, virtual disk size in bytes)`` for ``url``.""" self._dir.mkdir(parents=True, exist_ok=True) key = hashlib.sha256(url.encode()).hexdigest()[:16] vmdk, meta = self._dir / f"{key}.vmdk", self._dir / f"{key}.json" if vmdk.exists() and meta.exists(): return vmdk, int(json.loads(meta.read_text())["virtual_size"]) download = self._dir / f"{key}.download" try: self._download_verified(url, checksum, download, timeout) size = self._virtual_size(download) partial = self._dir / f"{key}.vmdk.partial" self._convert(download, partial) # Rename last: a VMDK that exists is a VMDK that is complete, even # when two jobs convert the same image at once. os.replace(partial, vmdk) meta.write_text(json.dumps({"url": url, "virtual_size": size})) finally: download.unlink(missing_ok=True) return vmdk, size def _download_verified( self, url: str, checksum: str | None, dest: Path, timeout: float ) -> None: algorithm, _, expected = (checksum or "").rpartition(":") algorithm = (algorithm or "sha256").lower() for _attempt in (1, 2): self._fetch(url, dest, timeout) if not checksum: return actual = _digest(dest, algorithm) if actual.lower() == expected.lower(): return dest.unlink(missing_ok=True) raise RuntimeError(f"Checksum mismatch for {url}: expected {expected}, got {actual}")