Files
hp-iac/extractors/xe_page.py
T

101 lines
3.9 KiB
Python

"""Generic parser for the switch firmware's auto-generated ("XE") pages.
Every data page (SysDescription.html, PortSummary.html, BufferedLogs.html, ...)
renders one or more <TABLE> elements in one of two shapes:
- "scalar": no <TH> headers, each <TR> is a label/value pair via
<TD class=defleft> (label text) + <TD class=defright><INPUT VALUE=...>.
e.g. Network.html's "Network Details" table.
- "tabular": <TH> column headers followed by one <TR> per data row, each
<TD> holding the cell value inside a hidden <INPUT VALUE=...>.
e.g. PortSummary.html's one row per switch port.
This module parses either shape without needing per-page-specific code; each
extractors/<name>.py module just calls fetch_tables() and picks out the
table(s)/fields it cares about by caption.
"""
from bs4 import BeautifulSoup
import switch_client
def extract_cell_value(td) -> str:
inp = td.find("input")
if inp is not None:
return inp.get("value", "")
return td.get_text(strip=True)
def parse_table(table) -> dict:
table_id = table.get("id")
caption_tag = table.find("caption")
caption = caption_tag.get_text(strip=True) if caption_tag else None
headers = [th.get_text(strip=True) for th in table.find_all("th", recursive=False)]
trs = table.find_all("tr", recursive=False)
if headers:
# Some pages repeat a label across a hidden raw-value column and a
# visible display column (e.g. PortSummary's/MAC Table's two
# "Interface"/"MAC Address" headers). Building the row dict via zip()
# means the later (visible) column wins on a duplicate label -- that
# has matched the raw column's value in every case seen so far, but
# if a future page's hidden/visible pair ever genuinely differs,
# only the visible one survives here.
rows = []
for tr in trs:
tds = tr.find_all("td", recursive=False)
if len(tds) != len(headers):
continue
rows.append({h: extract_cell_value(td) for h, td in zip(headers, tds)})
return {"caption": caption, "kind": "tabular", "rows": rows, "id": table_id}
record = {}
field_names = {}
for tr in trs:
defleft = tr.find("td", class_="defleft")
defright = tr.find("td", class_="defright")
if defleft is None or defright is None:
continue
label = defleft.get_text(strip=True)
if not label:
continue
record[label] = extract_cell_value(defright)
inp = defright.find("input")
if inp is not None and inp.get("name"):
field_names[label] = inp["name"]
# field_names lets a write action look up the real <INPUT NAME=...> for a
# label instead of hardcoding a guessed "v_R_C_1"-style field name.
return {
"caption": caption,
"kind": "scalar",
"record": record,
"field_names": field_names,
"id": table_id,
}
def parse_xe_tables(html: str) -> list[dict]:
soup = BeautifulSoup(html, "html.parser")
return [parse_table(t) for t in soup.find_all("table")]
def fetch_tables(session, path: str) -> dict[str, dict]:
"""Fetch path and return {caption: table} for every captioned table.
Tables with no/blank caption (usually just a submit button row, but
occasionally a real data table -- e.g. FDBSearch.html's MAC entries
table has no caption) are dropped since there's nothing useful to key
them by. Use fetch_all_tables() + table["id"] for those instead."""
html = switch_client.fetch(session, path)
tables = parse_xe_tables(html)
return {t["caption"]: t for t in tables if t["caption"]}
def fetch_all_tables(session, path: str) -> list[dict]:
"""Fetch path and return every table, including uncaptioned ones, in
document order. Look up by table["id"] (the <TABLE ID=...> attribute)
when a table has no caption to key it by."""
html = switch_client.fetch(session, path)
return parse_xe_tables(html)