initial commit

This commit is contained in:
Joachim Friberg committed 2026-08-25 14:10:21 +02:00
1 parent af59a90c31
commit ef498e4a72
114 files changed
+11047

No files matched your search

+101
View File
@@ -0,0 +1,101 @@
from dataclasses import dataclass
from urllib.parse import urljoin
from bs4 import BeautifulSoup
import switch_client
@dataclass
class MenuItem:
label: str
href: str | None
parent: str | None
depth: int
def find_frame_src(html: str, frame_id_or_name: str) -> str | None:
"""Search for a <frame>/<iframe> tag by id or name attr, return its src."""
soup = BeautifulSoup(html, "html.parser")
tag = soup.find(["frame", "iframe"], id=frame_id_or_name) or soup.find(
["frame", "iframe"], attrs={"name": frame_id_or_name}
)
if tag is None:
found = [
{"tag": t.name, "id": t.get("id"), "name": t.get("name"), "src": t.get("src")}
for t in soup.find_all(["frame", "iframe"])
]
print(f"WARNING: frame/iframe '{frame_id_or_name}' not found. Frames present: {found}")
return None
return tag.get("src")
def get_treeframe_url(session) -> str:
html = getattr(session, "frameset_html", None)
if html is None:
print("WARNING: session has no cached frameset_html from login(), falling back to GET /.")
html = switch_client.fetch(session, "/")
src = find_frame_src(html, "treeframe")
if src is None:
raise RuntimeError("Could not locate iframe#treeframe on root page.")
return urljoin(switch_client.BASE_URL + "/", src)
def get_site_links_html(session, treeframe_url: str) -> str:
html = switch_client.fetch(session, treeframe_url)
soup = BeautifulSoup(html, "html.parser")
container = None
body = soup.find("body", id="sitemap")
if body is not None:
sidebar = body.find("div", id="sidebar")
if sidebar is not None:
container = sidebar.find("div", id="site_links")
if container is None:
print(
"WARNING: body#sitemap > div#sidebar > div#site_links not found via strict path, "
"falling back to searching for div#site_links anywhere in the page."
)
container = soup.find("div", id="site_links")
if container is None:
top_level_ids = [d.get("id") for d in soup.find_all("div") if d.get("id")]
print(f"WARNING: div#site_links not found at all. Top-level div ids seen: {top_level_ids}")
raise RuntimeError("Could not locate div#site_links on treeframe page.")
return str(container)
def parse_menu_tree(site_links_html: str, base_href: str) -> list[MenuItem]:
soup = BeautifulSoup(site_links_html, "html.parser")
items: list[MenuItem] = []
def walk(ul, parent_label, depth):
for li in ul.find_all("li", recursive=False):
a = li.find("a", recursive=False)
nested_ul = li.find("ul", recursive=False)
if a is not None:
label = a.get_text(strip=True)
href = urljoin(base_href, a["href"]) if a.get("href") else None
else:
label = li.get_text(strip=True).split("\n")[0].strip()
href = None
if not label:
print(f"WARNING: <li> with no <a> and no text found at depth {depth}: {li}")
items.append(MenuItem(label=label, href=href, parent=parent_label, depth=depth))
if nested_ul is not None:
walk(nested_ul, label, depth + 1)
top_ul = soup.find("ul", recursive=False) or soup.find("ul")
if top_ul is None:
print(f"WARNING: no <ul> found inside site_links container: {site_links_html[:500]!r}")
return items
walk(top_ul, None, 0)
return items
def get_menu_items(session) -> list[MenuItem]:
treeframe_url = get_treeframe_url(session)
site_links_html = get_site_links_html(session, treeframe_url)
return parse_menu_tree(site_links_html, base_href=treeframe_url)