# Copyright 2025 The NetOrk Authors # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. # You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, software # distributed under the License is distributed on an "AS IS" BASIS, # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. """SDN, VLAN and network-instance NAPALM getters for Proxmox VE nodes.""" from __future__ import annotations import logging import re from typing import Any from napalm.base.exceptions import ConnectionException from proxmoxer.core import ResourceException from napalm_proxmox import utils logger = logging.getLogger(__name__) _JsonDict = dict[str, Any] class ProxmoxSDNMixin: """Mixin providing SDN, VLAN and network-instance NAPALM methods.""" def _get_sdn_zones(self) -> list[_JsonDict]: try: return self._api.cluster.sdn.zones.get() or [] # type: ignore[union-attr] except Exception as exc: logger.debug("Failed to fetch SDN zones: %s", exc) return [] def _get_sdn_vnets(self) -> list[_JsonDict]: try: return self._api.cluster.sdn.vnets.get() or [] # type: ignore[union-attr] except Exception as exc: logger.debug("Failed to fetch SDN VNets: %s", exc) return [] def _get_sdn_subnets(self, vnet: str) -> list[_JsonDict]: try: return self._api.cluster.sdn.vnets(vnet).subnets.get() or [] except Exception as exc: logger.debug("Failed to fetch SDN subnets for %s: %s", vnet, exc) return [] def get_vlans(self) -> dict[str, _JsonDict]: """Return VLAN table. For OVS+SDN nodes: reads SDN VNets for VLAN IDs/names, then maps OVSIntPort (access ports with ovs_tag) → untagged membership, and OVSPort / OVSBridge (trunk ports) → tagged membership. Falls back to ``bridge vlan show`` for classic Linux-bridge nodes. """ result: dict[str, _JsonDict] = {} node_network = self._get_node_network() for vnet in self._get_sdn_vnets(): tag = vnet.get("tag") vnet_id = vnet.get("vnet", "") if tag is None: continue try: tag_int = int(tag) except (ValueError, TypeError): continue result[str(tag_int)] = { "name": vnet_id, "tagged": [], "untagged": [], } trunk_ports: list[str] = [] access_by_vlan: dict[str, list[str]] = {} for iface in node_network: ovs_type = iface.get("ovs_type", "") iface_name = iface.get("iface", "") if not iface_name: continue if ovs_type == "OVSIntPort": ovs_tag = iface.get("ovs_tag") if ovs_tag is not None: vid = str(int(ovs_tag)) access_by_vlan.setdefault(vid, []).append(iface_name) elif ovs_type in ("OVSPort", "OVSBridge"): trunk_ports.append(iface_name) if trunk_ports or access_by_vlan: for vid, vlan_entry in result.items(): vlan_entry["tagged"] = list(trunk_ports) vlan_entry["untagged"] = list(access_by_vlan.get(vid, [])) return result for entry in result.values(): entry.setdefault("tagged", []) entry.setdefault("untagged", []) for tag, bridges in self._get_vm_vlan_tags().items(): entry = result.setdefault(tag, {"name": "", "tagged": [], "untagged": []}) for bridge in bridges: if bridge not in entry["untagged"]: entry["untagged"].append(bridge) return { vid: entry for vid, entry in result.items() if entry.get("tagged") or entry.get("untagged") } def _get_vm_vlan_tags(self) -> dict[str, set[str]]: """Return ``{vlan_tag: {bridge_names}}`` derived from VM/container net configs. Scans every QEMU VM and LXC container on this node for ``netN`` config entries of the form ``bridge=vmbrX,tag=N,...`` and groups the bridges each VLAN tag is used on. """ tags: dict[str, set[str]] = {} net_re = re.compile(r"^net\d+$") def _collect(vmid: int, config: _JsonDict) -> None: for key, val in config.items(): if not net_re.match(key): continue bridge = "" tag: int | None = None for part in str(val).split(","): if "=" not in part: continue k, v = part.split("=", 1) k = k.strip().lower() if k == "bridge": bridge = v.strip() elif k == "tag": try: tag = int(v.strip()) except ValueError: pass if tag is not None and bridge: tags.setdefault(str(tag), set()).add(bridge) try: for vm in (self._node_api().qemu.get() or []): vmid = int(vm.get("vmid", 0)) try: config = self._node_api().qemu(vmid).config.get() or {} _collect(vmid, config) except Exception as exc: logger.debug("_get_vm_vlan_tags: QEMU %s config failed: %s", vmid, exc) except Exception as exc: logger.warning("_get_vm_vlan_tags: failed to list QEMU VMs: %s", exc) try: for ct in (self._node_api().lxc.get() or []): vmid = int(ct.get("vmid", 0)) try: config = self._node_api().lxc(vmid).config.get() or {} _collect(vmid, config) except Exception as exc: logger.debug("_get_vm_vlan_tags: LXC %s config failed: %s", vmid, exc) except Exception as exc: logger.warning("_get_vm_vlan_tags: failed to list LXC containers: %s", exc) return tags def get_network_instances(self, name: str = "") -> dict[str, _JsonDict]: """Return SDN zones as network instances.""" result: dict[str, _JsonDict] = {} result["default"] = { "name": "default", "type": "DEFAULT_INSTANCE", "state": {"route_distinguisher": None}, "interfaces": {"interface": {}}, } for iface in self._get_node_network(): iface_name = iface.get("iface", "") if iface_name: result["default"]["interfaces"]["interface"][iface_name] = {} for zone in self._get_sdn_zones(): zone_id = zone.get("zone", zone.get("name", "")) if not zone_id: continue if name and zone_id != name: continue instance = utils.sdn_zone_to_network_instance(zone) for vnet in self._get_sdn_vnets(): if vnet.get("zone") == zone_id: vnet_id = vnet.get("vnet", "") if vnet_id: instance["interfaces"]["interface"][vnet_id] = {} result[zone_id] = instance if name: return {k: v for k, v in result.items() if k == name} return result def _is_physical_uplink(self, iface_name: str, network: dict) -> bool: """Return True if *iface_name* is a physical Ethernet port usable as uplink. Rules: - Must not match any known virtual interface name prefix. - Must appear in the Proxmox node network config (runtime-only virtual interfaces such as ``fwpr*`` or ``tap*`` will not be listed there). - Must have a physical-compatible type: - ``"eth"`` — regular physical NIC - ``"OVSPort"`` — physical NIC attached directly to an OVS bridge - ``""`` — untyped (e.g. OVS bond slave, still physical) """ if any(iface_name.startswith(p) for p in self._VIRTUAL_IFACE_PREFIXES): return False iface_info = network.get(iface_name) if iface_info is None: return False return iface_info.get("type", "") in ("eth", "OVSPort", "") def _find_switch_uplink(self) -> str | None: """Return the name of the physical interface connected to a switch. Detection order: 1. LLDP detailed: physical port whose neighbour advertises Bridge capability. 2. LLDP basic fallback: first physical port with any LLDP neighbour. """ network = { iface["iface"]: iface for iface in self._get_node_network() if iface.get("iface") } try: for iface_name, neighbour_list in self.get_lldp_neighbors_detail().items(): if not self._is_physical_uplink(iface_name, network): continue for nb in neighbour_list: caps = nb.get("remote_system_capab", []) if any("bridge" in str(c).lower() for c in caps): return iface_name except Exception as exc: logger.debug("LLDP detailed neighbor discovery failed: %s", exc) try: for iface_name, neighbour_list in self.get_lldp_neighbors().items(): if self._is_physical_uplink(iface_name, network) and neighbour_list: return iface_name except Exception as exc: logger.debug("LLDP basic neighbor discovery failed: %s", exc) return None def _get_ovs_bridge_for_port(self, port_name: str) -> str | None: """Return the OVS bridge name that *port_name* belongs to, or ``None``. Checks (in order): 1. Port listed in an OVSBridge's ``ovs_ports``. 2. Port is a slave of an OVSBond which has an ``ovs_bridge`` reference. 3. Port itself carries an ``ovs_bridge`` field. """ network = self._get_node_network() by_name: dict[str, _JsonDict] = { iface["iface"]: iface for iface in network if iface.get("iface") } for iface in network: if iface.get("type") == "OVSBridge": ports = (iface.get("ovs_ports") or "").split() if port_name in ports: return iface["iface"] for iface in network: if iface.get("type") == "OVSBond": slaves = (iface.get("slaves") or "").split() if port_name in slaves: bridge = iface.get("ovs_bridge", "") if bridge: return bridge port_info = by_name.get(port_name, {}) return port_info.get("ovs_bridge") or None def _is_cluster_master(self) -> bool: """Return ``True`` if this node is the Corosync quorum coordinator. The coordinator is the online cluster node with the lowest ``nodeid``. On standalone (non-clustered) nodes this always returns ``True``. """ try: status = self._api.cluster.status.get() or [] node_entries = [e for e in status if e.get("type") == "node"] if not node_entries: return True online_nodes = [n for n in node_entries if n.get("online", 0)] if not online_nodes: return True min_id = min(int(n.get("nodeid", 9999)) for n in online_nodes) for n in online_nodes: if ( n.get("name") == self._node_name and int(n.get("nodeid", 9999)) == min_id ): return True return False except Exception as exc: logger.warning("Cannot determine cluster status, assuming standalone: %s", exc) return True def _get_sdn_zone_for_bridge(self, bridge_name: str) -> str | None: """Return the SDN zone ID whose ``bridge`` field matches *bridge_name*. In Proxmox SDN each zone is linked to exactly one OVS bridge via the ``bridge`` property. We look for that mapping so the VNet is always created in the correct zone instead of guessing by type order. Falls back to the first zone if no bridge match is found. """ try: zones = self._get_sdn_zones() for zone in zones: if zone.get("bridge") == bridge_name: return zone.get("zone") if zones: return zones[0].get("zone") except Exception as exc: logger.debug("Failed to get SDN zone for bridge: %s", exc) return None def set_vlan(self, vlan_id: int, config) -> None: """Create (or update) a VLAN via an SDN VNet on this Proxmox node. Pre-flight checks (all must pass to proceed): 1. Finds the physical uplink port connected to a switch via LLDP. Physical ports are those with type ``eth``, ``OVSPort``, or ``""`` in the Proxmox network config (excludes runtime virtuals like ``fwpr*``, ``tap*``, etc.). 2. Verifies that uplink is part of an OVS bridge or OVS bond. 3. Confirms this node is the Corosync quorum master (lowest node-id). Non-master nodes return silently — the master handles VNet creation. The SDN VNet is named ``vlan{vid:04d}`` (e.g. ``vlan0007`` for VID 7). If the VNet already exists its alias is updated. After creating / updating the VNet the SDN configuration is reloaded via ``PUT /cluster/sdn``. Args: vlan_id: VLAN identifier (1-4094). config: Dict that may contain ``"name"`` for the VLAN alias. """ name: str = ( (config.get("name") or f"VLAN{vlan_id}") if config else f"VLAN{vlan_id}" ) vnet_id = f"vlan{vlan_id:04d}" uplink = self._find_switch_uplink() if uplink is None: raise ConnectionException( f"set_vlan({vlan_id}): no LLDP-detected switch uplink found" f" on node {self._node_name!r}" ) ovs_bridge = self._get_ovs_bridge_for_port(uplink) if ovs_bridge is None: raise ConnectionException( f"set_vlan({vlan_id}): uplink {uplink!r} is not part of an OVS bridge" ) if not self._is_cluster_master(): return zone = self._get_sdn_zone_for_bridge(ovs_bridge) if not zone: raise ConnectionException( f"set_vlan({vlan_id}): no SDN zone found for bridge {ovs_bridge!r}" ) try: self._api.cluster.sdn.vnets.post( vnet=vnet_id, zone=zone, tag=vlan_id, alias=name, ) except ResourceException as exc: err_str = str(exc).lower() if "already exists" in err_str or "duplicate" in err_str or "500" in err_str: try: self._api.cluster.sdn.vnets(vnet_id).put(alias=name) except Exception as exc: logger.debug("Best-effort VNet alias update failed: %s", exc) else: raise ConnectionException( f"set_vlan({vlan_id}): failed to create VNet {vnet_id!r}: {exc}" ) from exc try: self._api.cluster.sdn.put() except Exception as exc: logger.debug("Best-effort SDN reload failed (may not be needed on older PVE): %s", exc) def delete_vlan(self, vlan_id: int) -> None: """Delete the SDN VNet corresponding to *vlan_id*. The VNet is identified by the canonical name ``vlan{vid:04d}``. Only the Corosync quorum master performs the deletion — non-master nodes return silently. After deletion the SDN configuration is reloaded via ``PUT /cluster/sdn``. Args: vlan_id: VLAN identifier to delete. """ if not self._is_cluster_master(): return vnet_id = f"vlan{vlan_id:04d}" try: self._api.cluster.sdn.vnets(vnet_id).delete() except ResourceException as exc: err_str = str(exc).lower() if "does not exist" in err_str or "404" in str(exc): return raise ConnectionException( f"delete_vlan({vlan_id}): failed to delete VNet {vnet_id!r}: {exc}" ) from exc try: self._api.cluster.sdn.put() except Exception as exc: logger.debug("Best-effort SDN reload after delete failed: %s", exc)