"""Check a[href] and img[src] in a root-hosted static site. Python 3.10+."""
import argparse
from html.parser import HTMLParser
from pathlib import Path
import sys
from urllib.parse import quote, unquote, urljoin, urlsplit


class References(HTMLParser):
    def __init__(self):
        super().__init__()
        self.items = []
        self.has_base = False

    def handle_starttag(self, tag, attrs):
        attrs = dict(attrs)
        if tag == "base" and "href" in attrs:
            self.has_base = True
        key = {"a": "href", "img": "src"}.get(tag)
        if key and attrs.get(key):
            self.items.append((self.getpos()[0], attrs[key].strip()))


def local_target(root, page, raw, host):
    if not raw or raw.startswith("#"):
        return None
    base = "https://" + host + "/" + quote(page.relative_to(root).as_posix())
    url = urlsplit(urljoin(base, raw))
    if url.scheme not in {"http", "https"} or url.hostname != host:
        return None
    path = unquote(url.path, errors="strict")
    if "\x00" in path or "\\" in path:
        raise ValueError("unsupported path character")
    target = (root / path.lstrip("/")).resolve()
    if not target.is_relative_to(root):
        raise ValueError("target leaves output directory")
    if target.is_dir() or path.endswith("/"):
        target = (target / "index.html").resolve()
    if not target.is_relative_to(root):
        raise ValueError("target leaves output directory")
    return target


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("root", type=Path, help="Hugo output directory")
    parser.add_argument("--host", required=True, help="site hostname, without scheme or port")
    parser.add_argument("--page", action="append", help="HTML file relative to root; repeatable")
    args = parser.parse_args()
    root = args.root.resolve()
    host = args.host.lower()
    if not root.is_dir() or any(c in host for c in "/:@") or not host:
        parser.error("provide an existing root and a hostname without scheme or port")
    try:
        pages = sorted(set((root / p).resolve() for p in args.page)) if args.page else sorted(set(p.resolve() for p in root.rglob("*.html")))
        if not pages:
            raise ValueError("no HTML pages found")
        if any(not p.is_relative_to(root) or p.suffix != ".html" or not p.is_file() for p in pages):
            raise ValueError("each --page must be an existing HTML file inside root")
        checked = problems = 0
        for page in pages:
            refs = References()
            refs.feed(page.read_text(encoding="utf-8"))
            refs.close()
            if refs.has_base:
                raise ValueError(f"{page.relative_to(root)}: <base href> is unsupported")
            for line, raw in refs.items:
                try:
                    target = local_target(root, page, raw, host)
                except (ValueError, OSError) as exc:
                    problems += 1
                    print(f"INVALID {page.relative_to(root)}:{line} {raw!r}: {exc}")
                    continue
                if target is None:
                    continue
                checked += 1
                if not target.is_file():
                    problems += 1
                    print(f"MISSING {page.relative_to(root)}:{line} {raw!r} -> {target.relative_to(root)}")
        print(f"pages={len(pages)} checked={checked} problems={problems}")
        return 1 if problems else 0
    except (OSError, UnicodeError, ValueError) as exc:
        print(f"ERROR: {exc}", file=sys.stderr)
        return 2


if __name__ == "__main__":
    raise SystemExit(main())
