100 lines
3.2 KiB
Python
100 lines
3.2 KiB
Python
import re
|
|
from dataclasses import dataclass
|
|
|
|
from menu import MenuItem
|
|
|
|
TODO_PATH = "TODO.md"
|
|
|
|
_HEADER = """# HPE 1810 switch scrape TODO
|
|
|
|
Regenerate with `python discover.py`. Edit the "Want it?" column by hand;
|
|
"Implemented?" is updated by the assistant when an extractor exists.
|
|
|
|
"""
|
|
_TABLE_HEADER = "| Menu Item | URL | Want it? | Implemented? |\n|---|---|---|---|\n"
|
|
_ROW_RE = re.compile(r"^\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*(.*?)\s*\|\s*$")
|
|
|
|
|
|
@dataclass
|
|
class TodoRow:
|
|
label: str
|
|
url: str
|
|
want: str = ""
|
|
implemented: str = ""
|
|
|
|
|
|
def parse_existing(path: str = TODO_PATH) -> list[TodoRow]:
|
|
"""Read existing TODO.md's table into a list, preserving duplicate urls
|
|
(e.g. two menu entries, "Home" and "System Description", pointing at the
|
|
same page) so a round trip through refresh_todo()/mark_implemented()
|
|
never silently drops one of them."""
|
|
try:
|
|
with open(path, encoding="utf-8") as f:
|
|
text = f.read()
|
|
except FileNotFoundError:
|
|
return []
|
|
|
|
rows: list[TodoRow] = []
|
|
for line in text.splitlines():
|
|
m = _ROW_RE.match(line)
|
|
if not m:
|
|
continue
|
|
label, url, want, implemented = m.groups()
|
|
if label in ("Menu Item", "---"):
|
|
continue
|
|
rows.append(TodoRow(label=label, url=url, want=want, implemented=implemented))
|
|
return rows
|
|
|
|
|
|
def merge(discovered: list[MenuItem], existing: list[TodoRow]) -> list[TodoRow]:
|
|
# Keyed by (label, url): a url can legitimately appear under more than
|
|
# one label (see docstring above), so url alone isn't a safe lookup key.
|
|
prior_by_key = {(row.label, row.url): row for row in existing}
|
|
seen_keys = set()
|
|
rows: list[TodoRow] = []
|
|
|
|
for item in discovered:
|
|
if not item.href:
|
|
continue
|
|
key = (item.label, item.href)
|
|
seen_keys.add(key)
|
|
prior = prior_by_key.get(key)
|
|
if prior is not None:
|
|
rows.append(TodoRow(label=item.label, url=item.href, want=prior.want, implemented=prior.implemented))
|
|
else:
|
|
rows.append(TodoRow(label=item.label, url=item.href))
|
|
|
|
for row in existing:
|
|
if (row.label, row.url) not in seen_keys:
|
|
rows.append(row)
|
|
|
|
return rows
|
|
|
|
|
|
def render(rows: list[TodoRow]) -> str:
|
|
lines = [_HEADER, _TABLE_HEADER]
|
|
for row in rows:
|
|
lines.append(f"| {row.label} | {row.url} | {row.want} | {row.implemented} |\n")
|
|
return "".join(lines)
|
|
|
|
|
|
def refresh_todo(discovered: list[MenuItem], path: str = TODO_PATH) -> None:
|
|
existing = parse_existing(path)
|
|
rows = merge(discovered, existing)
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write(render(rows))
|
|
|
|
|
|
def mark_implemented(url: str, path: str = TODO_PATH) -> None:
|
|
"""Flip Implemented? to 'yes' for every row matching url (there may be
|
|
more than one label pointing at the same page, e.g. Home/System
|
|
Description both -> SysDescription.html)."""
|
|
existing = parse_existing(path)
|
|
matched = [row for row in existing if row.url == url]
|
|
if not matched:
|
|
raise KeyError(f"No TODO row found for url {url!r}")
|
|
for row in matched:
|
|
row.implemented = "yes"
|
|
with open(path, "w", encoding="utf-8") as f:
|
|
f.write(render(existing))
|