#!/usr/bin/env python3
"""Offline check for a flat UTF-8 HTML site. Python 3.9+, standard library only.

Usage: python check-static-site.py ./public --origin https://example.com
Reads top-level *.html and sitemap.xml; never writes files or sends requests.
Does not validate HTML conformance, HTTP routing, security, or AdSense eligibility.
"""

import argparse
from collections import Counter
from html.parser import HTMLParser
from pathlib import Path
from urllib.parse import unquote, urljoin, urlsplit
import xml.etree.ElementTree as ET


class Page(HTMLParser):
    def __init__(self, source):
        super().__init__(convert_charrefs=True)
        self.titles, self.headings = [], []
        self.descriptions, self.canonicals, self.ids, self.links = [], [], [], []
        self.capture = None
        self.noindex = False
        self.has_base = False
        self.feed(source)
        self.close()

    def handle_starttag(self, tag, attrs):
        a = dict(attrs)
        if tag in ('title', 'h1'):
            self.capture = self.titles if tag == 'title' else self.headings
            self.capture.append('')
        if a.get('id'):
            self.ids.append(a['id'])
        if tag == 'base':
            self.has_base = True
        if tag == 'meta':
            name = (a.get('name') or '').lower()
            content = a.get('content') or ''
            if name == 'description':
                self.descriptions.append(content)
            if name == 'robots' and 'noindex' in content.lower().replace(',', ' ').split():
                self.noindex = True
        if tag == 'link' and 'canonical' in (a.get('rel') or '').lower().split():
            self.canonicals.append(a.get('href') or '')
            return
        attribute = 'href' if tag in ('a', 'link') else 'src'
        if tag in ('a', 'link', 'img', 'script') and a.get(attribute):
            self.links.append((a[attribute], self.getpos()[0]))

    def handle_endtag(self, tag):
        if tag in ('title', 'h1'):
            self.capture = None

    def handle_data(self, data):
        if self.capture is not None:
            self.capture[-1] += data


def audit(root, origin):
    issues, pages = [], {}

    def report(name, code, message):
        issues.append(f'{name}: {code} {message}')

    for path in sorted(root.glob('*.html')):
        if not path.is_file() or path.is_symlink():
            report(path.name, 'INPUT', 'expected a regular file, not a symlink')
            continue
        try:
            pages[path.name] = Page(path.read_text(encoding='utf-8-sig'))
        except (OSError, UnicodeError):
            report(path.name, 'INPUT', 'cannot read as UTF-8')
    if not pages:
        report('site', 'INPUT', 'no readable top-level HTML files')
    if 'index.html' not in pages:
        report('site', 'INPUT', 'index.html is missing')

    expected_urls = []
    titles = Counter()
    for name, page in pages.items():
        route = '/' if name == 'index.html' else '/' + name[:-5]
        for field, values in [('title', page.titles), ('h1', page.headings),
                              ('description', page.descriptions)]:
            if len(values) != 1 or not values[0].strip():
                report(name, 'META', f'expected one non-empty {field}')
        if len(page.ids) != len(set(page.ids)):
            report(name, 'ID', 'duplicate id')
        if page.has_base:
            report(name, 'UNSUPPORTED', 'base element changes relative URL resolution')
        if name != '404.html' and not page.noindex:
            expected = origin + route
            expected_urls.append(expected)
            if page.canonicals != [expected]:
                report(name, 'CANONICAL', 'expected this page URL, not another URL')
            if page.titles:
                titles[page.titles[0].strip()] += 1

        for link, line in page.links:
            try:
                url = urlsplit(urljoin(origin + route, link))
                if url.scheme not in ('http', 'https') or url.netloc != urlsplit(origin).netloc:
                    continue
                relative = unquote(url.path).lstrip('/')
                target = root / (relative or 'index.html')
                if not target.suffix:
                    target = target.with_suffix('.html')
                # Stay inside the selected folder, including symlink targets.
                if not target.resolve().is_relative_to(root):
                    report(name, 'LINK', f'target is outside the selected folder (line {line})')
                    continue
                if not target.is_file():
                    report(name, 'LINK', f'local target does not exist (line {line})')
                elif url.fragment and target.parent == root and target.name in pages:
                    if unquote(url.fragment) not in pages[target.name].ids:
                        report(name, 'ANCHOR', f'target id does not exist (line {line})')
            except (ValueError, OSError, RuntimeError):
                report(name, 'LINK', f'invalid local URL or file path (line {line})')

    for name, page in pages.items():
        if page.titles and titles[page.titles[0].strip()] > 1:
            report(name, 'TITLE', 'title is shared by indexable pages')
    sitemap = root / 'sitemap.xml'
    try:
        if sitemap.is_symlink():
            raise ValueError('symlink')
        tree = ET.parse(sitemap).getroot()
        urls = [node.text for node in tree.findall('{*}url/{*}loc')]
        if tree.tag.split('}')[-1] != 'urlset':
            report('sitemap.xml', 'UNSUPPORTED', 'expected urlset, not sitemap index')
        if len(urls) != len(set(urls)):
            report('sitemap.xml', 'SITEMAP', 'duplicate URL')
        if set(urls) != set(expected_urls):
            report('sitemap.xml', 'SITEMAP', 'URLs differ from indexable top-level HTML')
    except (OSError, ET.ParseError, ValueError):
        report('sitemap.xml', 'SITEMAP', 'missing, unreadable, symlink, or invalid XML')
    return len(pages), issues


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('folder', type=Path, help='published flat HTML folder (not credentials)')
    parser.add_argument('--origin', required=True, help='public HTTPS origin, without a path')
    args = parser.parse_args()
    try:
        base = urlsplit(args.origin)
        valid = (base.scheme == 'https' and base.hostname and not base.username
                 and not base.password and base.path in ('', '/')
                 and not base.query and not base.fragment and base.port is None)
        root = args.folder.resolve()
        if not valid or not root.is_dir():
            raise ValueError('invalid input')
    except (ValueError, OSError, RuntimeError):
        parser.error('provide an existing folder and an HTTPS origin without credentials, port or path')
    count, issues = audit(root, args.origin.rstrip('/'))
    for issue in issues:
        print(issue)
    print(f'Checked {count} HTML files; {len(issues)} issue(s).')
    print('HTTP, JavaScript rendering, external links and content quality: NOT CHECKED.')
    return 1 if issues else 0


if __name__ == '__main__':
    raise SystemExit(main())
