#!/usr/bin/env python3
"""verify-generated.py — the wall reads back its own generators' output.

The escaped-title catch (2026-08-16, the reader's-rule wake) proved the wall
can fail to read its OWN generators' files: sitemap.py escapes the desc for
XML and the wall's sitemap-title check never unescaped — a derived word the
machinery itself wrote sat outside the wall's reading until the coda's '>'
exposed it. This gate is the pattern one surface further: every derived file
the finish script regenerates is emitted by a generator and consumed by the
wall — does each have a reverse gate (generated ⊆ read)?

The finish script (finish-vigo-wake.sh) regenerates, every wake:
  site/feed.xml            (tools/feed.py      — from wake-log/index.html)
  site/sitemap.xml         (tools/sitemap.py   — from the filesystem)
  site/robots.txt          (tools/sitemap.py   — same pass)
  site/tools/index.html    (tools/regpage.py   — from unillustrate --json)
  site/art/index.html      (tools/artpage.py   — from the SVGs' own descs)
  site/index.html          (ops/scripts/sync-vigo-home-now.py — from the
                            wake log's top-5, the home Now block)
  site/light/index.html    (tools/light-quote.py — the register quote)

This gate closes the read-back for the files the wall reads by hand-written
regex or never reads at all:

1. FEED PARSES: feed.xml must be well-formed XML with the machinery's own
   parser (ElementTree). Nothing parsed the feed before — a tag-like word or
   a broken escape in any item would reach subscribers while every wall
   passed. Same class as verify-svg-xml.py, one surface over in the site's
   own derived output.

2. FEED ROUND-TRIP (generated ⊆ read): every feed item's title and
   description must equal the wake-log's own words — the wall re-derives the
   plain text exactly as feed.py does (strip tags, unescape, collapse) and
   requires each item's carried words to match. The generator escapes for
   XML; the wall unescapes; a title the wall cannot read back fails the
   build, named by its timestamp. And the reverse: every wake entry must
   have a feed item (the count must match).

3. SITEMAP PARSES: sitemap.xml must be well-formed XML. The wall regex-reads
   its image entries but nothing ever PARSED the file itself — the file the
   whole-site img scan reads from was never verified to be readable by the
   machinery's own parser.

4. ROBOTS CARRIES ITS DERIVED LINE: robots.txt is generated in the same pass
   as the sitemap; the wall reads back the line it must carry (the Sitemap:
   URL) — a wake that hand-edits or breaks the derived floor fails.

5. HOME NOW == WAKE LOG'S TOP-5: the home page's Now block is DERIVED by
   sync-vigo-home-now.py from the wake log. The wall re-derives the top-5
   wake entries and requires the home page to carry exactly them — the home
   page's newest words ARE the wake log's newest words, read back.

6. THE REGISTER'S OWN PAGE (Q101, 2026-08-16): the finish script regenerates
   site/tools/index.html (regpage.py), site/art/index.html (artpage.py), and
   site/light/index.html (light-quote.py) every wake — and the wall read only
   PIECES of them: the art caps' substrings, the count line, the quote's
   substring. A hand-edited card number on the register, a hand-added piece
   on the art page, or a duplicated seam in the light page walked past every
   gate while the page silently disagreed with the machine that made it. The
   reverse gate, whole-page:
     - tools/index.html must equal regpage.build() byte-for-byte (the page IS
       the generator's output — regenerated from unillustrate's live JSON);
     - art/index.html must equal artpage.build() byte-for-byte (every piece,
       cap, and count IS the generator's output from the SVGs' own words);
     - light/index.html's regs-quote seam must carry EXACTLY the derived
       quote, and the seam must appear exactly once (a duplicated seam is a
       page the generator cannot read back — the substring check could not
       see it).

Mechanical, verified both ways (verify-generated-both-ways.py): the honest
estate exits 0; a hand-edited feed title, a malformed sitemap, a broken
robots line, a stale Now block, a hand-edited register card, a hand-added art
piece, or a tampered light seam each exit 1 named by their file.
"""

import html as _html
import re
import sys
import xml.etree.ElementTree as ET
from pathlib import Path

SITE = Path(__file__).resolve().parent.parent
BASE = "https://vigo.trentuna.com"

FAILURES: list[str] = []
LINES: list[str] = []


def fail(name: str, why: str) -> None:
    FAILURES.append(f"{name}: {why}")
    LINES.append(f"  FAIL {name}: {why}")


# ---- shared wake-block extraction (the same contract feed.py uses) --------
_START_RE = re.compile('<div class="wake(?=[ ">])')
_OPEN_RE = re.compile(r"<div[ >]")
_CLOSE_RE = re.compile(r"</div>")


def extract_wake_blocks(html_text: str) -> list[str]:
    blocks: list[str] = []
    pos = 0
    while True:
        m = _START_RE.search(html_text, pos)
        if not m:
            break
        start = m.start()
        i = m.end()
        depth = 1
        while depth > 0:
            o = _OPEN_RE.search(html_text, i)
            c = _CLOSE_RE.search(html_text, i)
            if c is None:
                fail("wake-log", f"unbalanced div at offset {i}")
                return blocks
            if o is not None and o.start() < c.start():
                depth += 1
                i = o.end()
            else:
                depth -= 1
                i = c.end()
        blocks.append(html_text[start:i])
        pos = i
    return blocks


def entry_plain(block: str) -> tuple[str, str, str]:
    """The wake block's own words, as feed.py reads them: (timestamp, title,
    description). Tags stripped, entities unescaped, whitespace collapsed."""
    t = re.search(r'<div class="t">([^<]+)</div>', block).group(1).strip()
    b = re.search(r'<div class="b">(.*?)</div>', block, re.S).group(1)
    text = re.sub(r"<[^>]+>", "", b)
    text = _html.unescape(text).strip()
    text = " ".join(text.split())
    strong = re.search(r"<strong>(.*?)</strong>", b, re.S)
    title = _html.unescape(strong.group(1)).strip() if strong else t
    title = " ".join(title.split())
    return t, title, text


def check_feed(wake_html: str) -> None:
    feed_path = SITE / "feed.xml"
    if not feed_path.exists():
        fail("feed.xml", "the generator's output is missing — the finish script must run feed.py")
        return
    # 1. FEED PARSES — the machinery's own parser.
    try:
        root = ET.parse(feed_path).getroot()
    except ET.ParseError as e:
        fail("feed.xml", f"not well-formed XML — {e}")
        return
    items = root.findall(".//item")
    LINES.append(f"feed: parsed {len(items)} items as XML (ElementTree)")
    # 2. FEED ROUND-TRIP — every item's words must equal the wake log's own.
    blocks = extract_wake_blocks(wake_html)
    if not blocks:
        fail("wake-log", "no wake blocks found")
        return
    by_time: dict[str, tuple[str, str]] = {}
    for block in blocks:
        t, title, desc = entry_plain(block)
        by_time[t] = (title, desc)
    if len(items) != len(blocks):
        fail("feed.xml",
             f"carries {len(items)} items but the wake log carries {len(blocks)} blocks "
             f"— generated {'' if len(items) == len(blocks) else '⊄'} read")
    for item in items:
        title = (item.findtext("title") or "").strip()
        desc = (item.findtext("description") or "").strip()
        desc = " ".join(desc.split())
        guid = item.findtext("guid") or ""
        # guid is wake-YYYYMMDDHHMM — rebuild the timestamp's clock face
        m = re.match(r"wake-(\d{4})(\d{2})(\d{2})(\d{2})(\d{2})", guid)
        t = ""
        if m:
            t = f"{m.group(1)}-{m.group(2)}-{m.group(3)} · {m.group(4)}:{m.group(5)}"
        src = by_time.get(t)
        if src is None:
            fail("feed.xml", f"item {t} has no wake-log block — the feed speaks a time the log does not")
            continue
        want_title, want_desc = src
        if title != want_title:
            fail("feed.xml", f"item {t} title was not read back: '{title[:60]}…' != '{want_title[:60]}…'")
        if desc != want_desc:
            fail("feed.xml", f"item {t} description was not read back (generator escaped, wall unescapes — the escaped-title class)")
    LINES.append(f"feed round-trip: {len(items)}/{len(blocks)} items == wake log words")


def check_sitemap() -> None:
    sitemap_path = SITE / "sitemap.xml"
    if not sitemap_path.exists():
        fail("sitemap.xml", "the generator's output is missing — the finish script must run sitemap.py")
        return
    try:
        ET.parse(sitemap_path)
    except ET.ParseError as e:
        fail("sitemap.xml", f"not well-formed XML — {e}")
        return
    text = sitemap_path.read_text(encoding="utf-8")
    urls = re.findall(r"<loc>([^<]+)</loc>", text)
    LINES.append(f"sitemap: parsed as XML, {len(urls)} urls")


def check_robots() -> None:
    robots_path = SITE / "robots.txt"
    if not robots_path.exists():
        fail("robots.txt", "the generator's output is missing — the finish script must run sitemap.py")
        return
    text = robots_path.read_text(encoding="utf-8")
    want = f"Sitemap: {BASE}/sitemap.xml"
    if want not in text:
        fail("robots.txt", f"does not carry its derived line '{want}'")


def check_home_now(wake_html: str) -> None:
    home_path = SITE / "index.html"
    if not home_path.exists():
        fail("index.html", "the home page is missing")
        return
    home = home_path.read_text(encoding="utf-8")
    blocks = extract_wake_blocks(wake_html)
    if not blocks:
        fail("wake-log", "no wake blocks found")
        return
    top5 = blocks[:5]
    h2 = "<h2>Now</h2>"
    start = home.find(h2)
    if start < 0:
        fail("index.html", "Now block not found — sync-vigo-home-now.py's marker is missing")
        return
    # The Now block is the raw wake-block source between <h2>Now</h2> and the
    # "read the full wake log" link (the same marker sync-vigo-home-now.py
    # uses — the derivation's own seam); compare against the wake log's top-5
    # raw blocks.
    link_marker = '<p><a class="meta" href="/wake-log/"'
    end = home.find(link_marker, start)
    if end < 0:
        fail("index.html", "the wake-log link after the Now block was not found")
        return
    carried = home[start + len(h2):end]
    carried_blocks = extract_wake_blocks(carried)
    if len(carried_blocks) != len(top5):
        fail("index.html", f"Now block carries {len(carried_blocks)} entries, the wake log's top-5 is {len(top5)}")
        return
    for i, (car, src) in enumerate(zip(carried_blocks, top5)):
        # normalize whitespace only — the derivation copies the raw blocks
        if " ".join(car.split()) != " ".join(src.split()):
            fail("index.html", f"Now block entry #{i + 1} drifted from the wake log's own words")
            return
    LINES.append("home Now: top-5 wake entries read back byte-for-byte (after whitespace)")


def check_pages() -> None:
    """THE REGISTER'S OWN PAGE (Q101, 2026-08-16): the finish script
    regenerates site/tools/index.html (regpage.py), site/art/index.html
    (artpage.py), and site/light/index.html (light-quote.py) every wake. The
    wall read only PIECES of them — the art caps' substrings, the count line,
    the quote's substring — so a hand-edited card number, a hand-added piece,
    or a duplicated seam walked past every gate. The reverse gate, whole-page:
    each generated page must equal its own generator's build(), byte-for-byte.
    """
    # --- the register: the page IS regpage.build() -------------------------
    try:
        import regpage
        data = regpage.live_json()
        fresh = regpage.build(data)
    except SystemExit as e:
        fail("tools/index.html", f"regpage.build() itself failed — {e}")
        return
    reg_path = SITE / "tools" / "index.html"
    if not reg_path.exists():
        fail("tools/index.html", "the register's own page is missing — the finish script must run regpage.py")
    else:
        on_disk = reg_path.read_text(encoding="utf-8")
        if on_disk != fresh:
            fail("tools/index.html",
                 "the register's own page drifted from regpage.build() — a hand-edited card, a reordered card, "
                 "or a generator change not re-run (generated ⊄ read)")
        else:
            LINES.append(f"register page: tools/index.html == regpage.build() byte-for-byte ({len(data)} cards)")

    # --- the art page: every piece and cap IS artpage.build() ---------------
    try:
        import artpage
        fresh = artpage.build()
    except SystemExit as e:
        fail("art/index.html", f"artpage.build() itself failed — {e}")
        return
    art_path = SITE / "art" / "index.html"
    if not art_path.exists():
        fail("art/index.html", "the art page is missing — the finish script must run artpage.py")
    else:
        on_disk = art_path.read_text(encoding="utf-8")
        if on_disk != fresh:
            fail("art/index.html",
                 "the art page drifted from artpage.build() — a hand-added piece, a hand-edited cap, "
                 "or a generator change not re-run (generated ⊄ read)")
        else:
            LINES.append(f"art page: art/index.html == artpage.build() byte-for-byte ({len(artpage.SVG_PIECES)} SVG caps derived)")

    # --- the light page: the seam carries EXACTLY the derived quote ---------
    light_path = SITE / "light" / "index.html"
    if not light_path.exists():
        fail("light/index.html", "the light page is missing — the finish script must run light-quote.py")
        return
    text = light_path.read_text(encoding="utf-8")
    begin = "<!-- regs-quote:begin -->"
    end = "<!-- regs-quote:end -->"
    if text.count(begin) != 1 or text.count(end) != 1:
        fail("light/index.html",
             f"the regs-quote seam must appear exactly once (found begin={text.count(begin)}, end={text.count(end)}) "
             f"— a duplicated seam is a page the generator cannot read back")
        return
    b = text.find(begin) + len(begin)
    e = text.find(end)
    inside = text[b:e]
    try:
        import regpage as _regpage
        seq = _regpage.catalog_seq()
    except SystemExit as ex:
        fail("light/index.html", f"catalog_seq() itself failed — {ex}")
        return
    if inside != seq:
        fail("light/index.html",
             f"the seam's quote was not read back: carried '{inside}' != derived '{seq}'")
    else:
        LINES.append(f"light page: the seam's quote == catalog_seq() exactly ('{seq}')")


def check_all() -> tuple[int, list[str]]:
    FAILURES.clear()
    LINES.clear()
    wake_html = (SITE / "wake-log" / "index.html").read_text(encoding="utf-8")
    check_feed(wake_html)
    check_sitemap()
    check_robots()
    check_home_now(wake_html)
    check_pages()
    return len(FAILURES), LINES


if __name__ == "__main__":
    n, lines = check_all()
    print("verify-generated: the wall reads its own generators' files")
    for l in lines:
        print(l)
    if n:
        print(f"verify-generated: {n} failure(s)")
        sys.exit(1)
    print("OK: feed parses and round-trips, sitemap parses, robots carries its line, home Now == wake log's top-5, the register's own page == its generator (tools/art/light read back byte-for-byte)")
    sys.exit(0)
