102 lines
3.7 KiB
Python
102 lines
3.7 KiB
Python
from dataclasses import dataclass
|
|
from urllib.parse import urljoin
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
import switch_client
|
|
|
|
|
|
@dataclass
|
|
class MenuItem:
|
|
label: str
|
|
href: str | None
|
|
parent: str | None
|
|
depth: int
|
|
|
|
|
|
def find_frame_src(html: str, frame_id_or_name: str) -> str | None:
|
|
"""Search for a <frame>/<iframe> tag by id or name attr, return its src."""
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
tag = soup.find(["frame", "iframe"], id=frame_id_or_name) or soup.find(
|
|
["frame", "iframe"], attrs={"name": frame_id_or_name}
|
|
)
|
|
if tag is None:
|
|
found = [
|
|
{"tag": t.name, "id": t.get("id"), "name": t.get("name"), "src": t.get("src")}
|
|
for t in soup.find_all(["frame", "iframe"])
|
|
]
|
|
print(f"WARNING: frame/iframe '{frame_id_or_name}' not found. Frames present: {found}")
|
|
return None
|
|
return tag.get("src")
|
|
|
|
|
|
def get_treeframe_url(session) -> str:
|
|
html = getattr(session, "frameset_html", None)
|
|
if html is None:
|
|
print("WARNING: session has no cached frameset_html from login(), falling back to GET /.")
|
|
html = switch_client.fetch(session, "/")
|
|
src = find_frame_src(html, "treeframe")
|
|
if src is None:
|
|
raise RuntimeError("Could not locate iframe#treeframe on root page.")
|
|
return urljoin(switch_client.BASE_URL + "/", src)
|
|
|
|
|
|
def get_site_links_html(session, treeframe_url: str) -> str:
|
|
html = switch_client.fetch(session, treeframe_url)
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
|
|
container = None
|
|
body = soup.find("body", id="sitemap")
|
|
if body is not None:
|
|
sidebar = body.find("div", id="sidebar")
|
|
if sidebar is not None:
|
|
container = sidebar.find("div", id="site_links")
|
|
|
|
if container is None:
|
|
print(
|
|
"WARNING: body#sitemap > div#sidebar > div#site_links not found via strict path, "
|
|
"falling back to searching for div#site_links anywhere in the page."
|
|
)
|
|
container = soup.find("div", id="site_links")
|
|
|
|
if container is None:
|
|
top_level_ids = [d.get("id") for d in soup.find_all("div") if d.get("id")]
|
|
print(f"WARNING: div#site_links not found at all. Top-level div ids seen: {top_level_ids}")
|
|
raise RuntimeError("Could not locate div#site_links on treeframe page.")
|
|
|
|
return str(container)
|
|
|
|
|
|
def parse_menu_tree(site_links_html: str, base_href: str) -> list[MenuItem]:
|
|
soup = BeautifulSoup(site_links_html, "html.parser")
|
|
items: list[MenuItem] = []
|
|
|
|
def walk(ul, parent_label, depth):
|
|
for li in ul.find_all("li", recursive=False):
|
|
a = li.find("a", recursive=False)
|
|
nested_ul = li.find("ul", recursive=False)
|
|
if a is not None:
|
|
label = a.get_text(strip=True)
|
|
href = urljoin(base_href, a["href"]) if a.get("href") else None
|
|
else:
|
|
label = li.get_text(strip=True).split("\n")[0].strip()
|
|
href = None
|
|
if not label:
|
|
print(f"WARNING: <li> with no <a> and no text found at depth {depth}: {li}")
|
|
items.append(MenuItem(label=label, href=href, parent=parent_label, depth=depth))
|
|
if nested_ul is not None:
|
|
walk(nested_ul, label, depth + 1)
|
|
|
|
top_ul = soup.find("ul", recursive=False) or soup.find("ul")
|
|
if top_ul is None:
|
|
print(f"WARNING: no <ul> found inside site_links container: {site_links_html[:500]!r}")
|
|
return items
|
|
walk(top_ul, None, 0)
|
|
return items
|
|
|
|
|
|
def get_menu_items(session) -> list[MenuItem]:
|
|
treeframe_url = get_treeframe_url(session)
|
|
site_links_html = get_site_links_html(session, treeframe_url)
|
|
return parse_menu_tree(site_links_html, base_href=treeframe_url)
|