"""Parsers for the HPE OfficeConnect web UI. These switches have no CLI and no REST API — the only machine-readable surface is the web UI itself. Every list page (ports, VLANs, MAC table, trunks, counters) is a jQuery DataTable whose payload is embedded in the page as two JavaScript declarations:: var aDataSet = [ ['', '1', '', 'Enabled', ..., '1518'] , ['', '2', '', 'Enabled', ..., '1518'] ]; var aColumns = [ { "sTitle": "Interface", "sType": "intf-sort", "sWidth": "8%" }, ... ]; The array literal happens to be valid Python syntax, so it parses with ``ast.literal_eval`` — no HTML parser and no extra dependency needed. Note that this is a property of firmware PT.02.xx and later. Older 1820 firmware server-rendered the same data as ````/```` markup, which is why the (unmaintained) hp1820-cli scraper does not work here. """ from __future__ import annotations import ast import html as html_module import re _DATASET_RE = re.compile(r"var\s+aDataSet\s*=\s*\[(.*?)\n\s*\];", re.DOTALL) _STITLE_RE = re.compile(r"\"sTitle\"\s*:\s*(['\"])(.*?)\1", re.DOTALL) _TAG_RE = re.compile(r"<[^>]*>") _UPTIME_RE = re.compile( r"(\d+)\s*days?,\s*(\d+)\s*hours?,\s*(\d+)\s*mins?,\s*(\d+)\s*secs?", re.IGNORECASE, ) def strip_markup(value: str) -> str: """Reduce a DataTable cell to its plain text. The leading column of most tables is a row-selection checkbox whose "value" is an ```` element; it collapses to an empty string. """ text = _TAG_RE.sub("", value) text = html_module.unescape(text) return re.sub(r"\s+", " ", text).strip() def extract_data_set(page: str) -> list[list[str]]: """Return the rows of the page's ``aDataSet`` as lists of raw cell strings. Cells are returned verbatim, markup included — callers decide per column whether to run them through :func:`strip_markup`. Returns an empty list if the page carries no table (e.g. an error or login-redirect page). """ match = _DATASET_RE.search(page) if not match: return [] body = match.group(1) # Fast path: the whole literal at once. try: rows = ast.literal_eval("[" + body + "]") except (SyntaxError, ValueError): # Fall back to row-at-a-time so a single malformed row — an escape # sequence that is valid in JS but not in Python, say — costs us that # row rather than the entire table. rows = [] for line in body.splitlines(): line = line.strip().rstrip(",") if not line.startswith("["): continue try: rows.append(ast.literal_eval(line)) except (SyntaxError, ValueError): continue return [[str(cell) for cell in row] for row in rows if isinstance(row, (list, tuple))] def extract_column_titles(page: str) -> list[str]: """Return the text column titles declared in ``aColumns``. Purely informational columns whose title is markup rather than text — the select-all checkbox — are dropped, so the result lines up with the data columns a caller actually reads. """ titles = [strip_markup(m.group(2)) for m in _STITLE_RE.finditer(page)] return [t for t in titles if t] def parse_uptime(value: str) -> int: """Convert ``"0 days, 18 hours, 44 mins, 40 secs"`` to seconds. Returns -1 when the string cannot be parsed, which is NAPALM's convention for an unknown uptime. """ match = _UPTIME_RE.search(value or "") if not match: return -1 days, hours, mins, secs = (int(g) for g in match.groups()) return days * 86400 + hours * 3600 + mins * 60 + secs def _text_by_id(page: str, element_id: str) -> str: """Text content of the ```` carrying a dashboard value.""" match = re.search(rf'id="{re.escape(element_id)}"[^>]*>(.*?) str: """Value of the ```` carrying an editable dashboard field. The attribute often wraps across lines, hence the DOTALL search. """ match = re.search(rf'id="{re.escape(element_id)}"[^>]*?value="([^"]*)"', page, re.DOTALL) return html_module.unescape(match.group(1)).strip() if match else "" def parse_facts(page: str) -> dict[str, object]: """Extract system facts from ``dashboard.lsp``. The System Description is a comma-separated tuple of model, firmware, kernel and bootloader:: HPE OfficeConnect Switch 1820 8G PoE+ (65W) J9982A, PT.02.19, Linux 3.6.5, U-Boot 2012.10-00116 (Jul 30 2014 - 10:52:01) """ descr = _text_by_id(page, "sys_descr") parts = [p.strip() for p in descr.split(",")] if descr else [] model = parts[0] if parts else "" os_version = _text_by_id(page, "sw_version") or (parts[1] if len(parts) > 1 else "") return { "hostname": _input_value_by_id(page, "sys_name"), "model": model, "serial_number": _text_by_id(page, "serial_number"), "os_version": os_version, "sys_object_id": _text_by_id(page, "sys_obj_id"), "uptime": parse_uptime(_text_by_id(page, "sys_up_time")), "system_description": descr, }