import re from dataclasses import dataclass from menu import MenuItem TODO_PATH = "TODO.md" _HEADER = """# HPE 1810 switch scrape TODO Regenerate with `python discover.py`. Edit the "Want it?" column by hand; "Implemented?" is updated by the assistant when an extractor exists. """ _TABLE_HEADER = "| Menu Item | URL | Want it? | Implemented? |\n|---|---|---|---|\n" _ROW_RE = re.compile(r"^\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*$") @dataclass class TodoRow: label: str url: str want: str = "" implemented: str = "" def parse_existing(path: str = TODO_PATH) -> list[TodoRow]: """Read existing TODO.md's table into a list, preserving duplicate urls (e.g. two menu entries, "Home" and "System Description", pointing at the same page) so a round trip through refresh_todo()/mark_implemented() never silently drops one of them.""" try: with open(path, encoding="utf-8") as f: text = f.read() except FileNotFoundError: return [] rows: list[TodoRow] = [] for line in text.splitlines(): m = _ROW_RE.match(line) if not m: continue label, url, want, implemented = m.groups() if label in ("Menu Item", "---"): continue rows.append(TodoRow(label=label, url=url, want=want, implemented=implemented)) return rows def merge(discovered: list[MenuItem], existing: list[TodoRow]) -> list[TodoRow]: # Keyed by (label, url): a url can legitimately appear under more than # one label (see docstring above), so url alone isn't a safe lookup key. prior_by_key = {(row.label, row.url): row for row in existing} seen_keys = set() rows: list[TodoRow] = [] for item in discovered: if not item.href: continue key = (item.label, item.href) seen_keys.add(key) prior = prior_by_key.get(key) if prior is not None: rows.append(TodoRow(label=item.label, url=item.href, want=prior.want, implemented=prior.implemented)) else: rows.append(TodoRow(label=item.label, url=item.href)) for row in existing: if (row.label, row.url) not in seen_keys: rows.append(row) return rows def render(rows: list[TodoRow]) -> str: lines = [_HEADER, _TABLE_HEADER] for row in rows: lines.append(f"| {row.label} | {row.url} | {row.want} | {row.implemented} |\n") return "".join(lines) def refresh_todo(discovered: list[MenuItem], path: str = TODO_PATH) -> None: existing = parse_existing(path) rows = merge(discovered, existing) with open(path, "w", encoding="utf-8") as f: f.write(render(rows)) def mark_implemented(url: str, path: str = TODO_PATH) -> None: """Flip Implemented? to 'yes' for every row matching url (there may be more than one label pointing at the same page, e.g. Home/System Description both -> SysDescription.html).""" existing = parse_existing(path) matched = [row for row in existing if row.url == url] if not matched: raise KeyError(f"No TODO row found for url {url!r}") for row in matched: row.implemented = "yes" with open(path, "w", encoding="utf-8") as f: f.write(render(existing))