"""Generic parser for the switch firmware's auto-generated ("XE") pages. Every data page (SysDescription.html, PortSummary.html, BufferedLogs.html, ...) renders one or more elements in one of two shapes: - "scalar": no is a label/value pair via per data row, each
headers, each
(label text) + . e.g. Network.html's "Network Details" table. - "tabular": column headers followed by one
holding the cell value inside a hidden . e.g. PortSummary.html's one row per switch port. This module parses either shape without needing per-page-specific code; each extractors/.py module just calls fetch_tables() and picks out the table(s)/fields it cares about by caption. """ from bs4 import BeautifulSoup import switch_client def extract_cell_value(td) -> str: inp = td.find("input") if inp is not None: return inp.get("value", "") return td.get_text(strip=True) def parse_table(table) -> dict: caption_tag = table.find("caption") caption = caption_tag.get_text(strip=True) if caption_tag else None headers = [th.get_text(strip=True) for th in table.find_all("th", recursive=False)] trs = table.find_all("tr", recursive=False) if headers: rows = [] for tr in trs: tds = tr.find_all("td", recursive=False) if len(tds) != len(headers): continue rows.append({h: extract_cell_value(td) for h, td in zip(headers, tds)}) return {"caption": caption, "kind": "tabular", "rows": rows} record = {} field_names = {} for tr in trs: defleft = tr.find("td", class_="defleft") defright = tr.find("td", class_="defright") if defleft is None or defright is None: continue label = defleft.get_text(strip=True) if not label: continue record[label] = extract_cell_value(defright) inp = defright.find("input") if inp is not None and inp.get("name"): field_names[label] = inp["name"] # field_names lets a write action look up the real for a # label instead of hardcoding a guessed "v_R_C_1"-style field name. return {"caption": caption, "kind": "scalar", "record": record, "field_names": field_names} def parse_xe_tables(html: str) -> list[dict]: soup = BeautifulSoup(html, "html.parser") return [parse_table(t) for t in soup.find_all("table")] def fetch_tables(session, path: str) -> dict[str, dict]: """Fetch path and return {caption: table} for every captioned table. Tables with no/blank caption (usually just a submit button row) are dropped since there's nothing useful to key them by.""" html = switch_client.fetch(session, path) tables = parse_xe_tables(html) return {t["caption"]: t for t in tables if t["caption"]}