per data row, each
+ | holding the cell value inside a hidden .
+ e.g. PortSummary.html's one row per switch port.
+
+This module parses either shape without needing per-page-specific code; each
+extractors/.py module just calls fetch_tables() and picks out the
+table(s)/fields it cares about by caption.
+"""
+
+from bs4 import BeautifulSoup
+
+import switch_client
+
+
+def extract_cell_value(td) -> str:
+ inp = td.find("input")
+ if inp is not None:
+ return inp.get("value", "")
+ return td.get_text(strip=True)
+
+
+def parse_table(table) -> dict:
+ caption_tag = table.find("caption")
+ caption = caption_tag.get_text(strip=True) if caption_tag else None
+ headers = [th.get_text(strip=True) for th in table.find_all("th", recursive=False)]
+ trs = table.find_all("tr", recursive=False)
+
+ if headers:
+ rows = []
+ for tr in trs:
+ tds = tr.find_all("td", recursive=False)
+ if len(tds) != len(headers):
+ continue
+ rows.append({h: extract_cell_value(td) for h, td in zip(headers, tds)})
+ return {"caption": caption, "kind": "tabular", "rows": rows}
+
+ record = {}
+ field_names = {}
+ for tr in trs:
+ defleft = tr.find("td", class_="defleft")
+ defright = tr.find("td", class_="defright")
+ if defleft is None or defright is None:
+ continue
+ label = defleft.get_text(strip=True)
+ if not label:
+ continue
+ record[label] = extract_cell_value(defright)
+ inp = defright.find("input")
+ if inp is not None and inp.get("name"):
+ field_names[label] = inp["name"]
+ # field_names lets a write action look up the real for a
+ # label instead of hardcoding a guessed "v_R_C_1"-style field name.
+ return {"caption": caption, "kind": "scalar", "record": record, "field_names": field_names}
+
+
+def parse_xe_tables(html: str) -> list[dict]:
+ soup = BeautifulSoup(html, "html.parser")
+ return [parse_table(t) for t in soup.find_all("table")]
+
+
+def fetch_tables(session, path: str) -> dict[str, dict]:
+ """Fetch path and return {caption: table} for every captioned table.
+
+ Tables with no/blank caption (usually just a submit button row) are
+ dropped since there's nothing useful to key them by."""
+ html = switch_client.fetch(session, path)
+ tables = parse_xe_tables(html)
+ return {t["caption"]: t for t in tables if t["caption"]}
diff --git a/menu.py b/menu.py
new file mode 100644
index 0000000..86ba476
--- /dev/null
+++ b/menu.py
@@ -0,0 +1,101 @@
+from dataclasses import dataclass
+from urllib.parse import urljoin
+
+from bs4 import BeautifulSoup
+
+import switch_client
+
+
+@dataclass
+class MenuItem:
+ label: str
+ href: str | None
+ parent: str | None
+ depth: int
+
+
+def find_frame_src(html: str, frame_id_or_name: str) -> str | None:
+ """Search for a / |