#!/usr/bin/env python3
"""unillustrate.py — the illustration's numbers become its derivation card.

The inverse of illustrate.py. The garden's register claims that every
generator drawing is born from the essay's own numbers; this tool reads a
garden SVG back and recovers those numbers from the image alone — count the
life-lines, measure the ticks, recover the hour boundaries from block widths,
recover the seed. Then it round-trips: it re-runs illustrate.py with the
recovered parameters and the file's own captions, and compares geometry.

If the round-trip comes back identical, the image *contains its own
derivation*: an essay, its illustration, and the arithmetic that binds them,
each readable from the others. The register becomes self-proving.

Usage:
  python3 unillustrate.py assets/*.svg         # derivation cards for each
  python3 unillustrate.py --json assets/ten-essays.svg
  python3 unillustrate.py --verify-only assets/*.svg   # round-trip status only

Motif detection is structural (fingerprints, not filenames): lifelines,
ticks, boundary, scatter, hours, else hand-drawn (pre-generator register).

Derivation is honest about what it cannot recover: hand-drawn pieces that
predate the generator are reported as such, not forced into a card.

Since 2026-08-06 the check list is a *rendering*, not a second record. The
recoverers return only numbers (fields); the human-readable checks are
derived from those numbers afterwards — every note an f-string of the field
it checks. The text card can never disagree with the JSON card, because the
JSON card is the only ingredient of the text card. A note cannot drift from
its number; the number is its entire body.
"""

import argparse
import json
import re
import subprocess
import sys
from pathlib import Path

TOOLS_DIR = Path(__file__).resolve().parent
ILLUSTRATE = TOOLS_DIR / "illustrate.py"

W, H = 1280.0, 480.0
INDIGO = "#7c7cf0"
MUTED = "#5a5a74"
GOLD = "#d9a441"
BLUE = "#5b7fd4"
SEED = 31


# ---------------------------------------------------------------- parsing

def num(s):
    """Tolerant float parse of an SVG attribute value."""
    return float(s.rstrip("px"))


def grab_text(svg):
    """Return (title, desc, label, bottom) — the four text registers."""
    def first(pattern, group=1):
        m = re.search(pattern, svg, re.S)
        return m.group(group).strip() if m else ""

    title = first(r'<title id="t">(.*?)</title>')
    desc = first(r'<desc id="d">(.*?)</desc>')
    label = first(r'<text x="90" y="64"[^>]*>(.*?)</text>')
    bottom = first(r'<text x="90" y="440"[^>]*>(.*?)</text>')
    return title, desc, label, bottom


def get_rects(svg, y_tol=2.0, y_target=None):
    """All <rect> elements with x,y,width,height; optionally filtered by y."""
    out = []
    for m in re.finditer(
            r'<rect x="([0-9.]+)" y="([0-9.]+)" width="([0-9.]+)" height="([0-9.]+)"',
            svg):
        x, y, w, h = map(num, m.groups())
        if y_target is None or abs(y - y_target) <= y_tol:
            out.append((x, y, w, h))
    return out


def get_circles(svg, cx_tol=100000, cy_tol=100000):
    out = []
    for m in re.finditer(r'<circle cx="([0-9.]+)" cy="([0-9.]+)" r="([0-9.]+)"', svg):
        cx, cy, r = map(num, m.groups())
        out.append((cx, cy, r))
    return out


# ------------------------------------------------------------ fingerprints

def detect_motif(svg):
    if 'fill="url(#life)"' in svg:
        return "lifelines"
    if 'pattern id="ticks"' in svg:
        return "ticks"
    if "translate(700,268)" in svg and 'x1="820" y1="130"' in svg:
        return "boundary"
    if re.search(r'<line x1="69[0-9.]+" y1="265"', svg) and 'the invariant line' in svg:
        return "scatter"
    if "the thirty-first hour" in svg:
        return "hours"
    if 'x1="961" y1="280"' in svg and "the enforced process" in svg:
        return "wall"
    if "the vocabulary page" in svg and 'x1="640" y1="140"' in svg:
        return "tokens"
    if 'pattern id="rec-ticks"' in svg and "the ledger" in svg:
        return "recognition"
    if "Two rails on a dark field" in svg and 'id="t">The Family Line</title>' in svg:
        return "rails"
    return "hand-drawn"


# ------------------------------------------------------------ recovery

def recover_lifelines(svg):
    """Only the numbers: line count, diamond count, T-mark position, spacing.

    The semantic proof, spread from recognition/tokens: the thread labels
    (s1…sN) are *read*, not just counted — the <text> nodes at the line row
    come back as names, and the desc's own named threads are parsed too, so
    the render can state exactly which names the caption names and which it
    leaves drawn-but-unnamed. who-made-the-mark's desc names all six (as a
    range, "s1 to s6" -> full identity); ten-essays' desc names only the
    count (10 drawn labels read back, none captioned — the partial).

    The check list that used to be built here is now a rendering of these
    fields (see render_checks); the raw per-gap steps are kept as a private
    field (`_steps`) so the evenness check can still see every gap, not just
    the first.
    """
    lines = [x for (x, y, w, h) in get_rects(svg, y_tol=2.0, y_target=200)
             if abs(w - 8.0) < 0.5 and abs(h - 80.0) < 0.5]
    diamonds = re.findall(r'transform="rotate\(45 ([0-9.]+) 300\)"', svg)
    mark = re.search(r'<circle cx="([0-9.]+)" cy="280" r="48"', svg)
    n = len(lines)
    step = expected_step = None
    steps = []
    if n >= 2 and len(lines) > 1:
        steps = [lines[i + 1] - lines[i] for i in range(len(lines) - 1)]
        span = (1140 - 48 - 20) - 180
        expect = min(160, span / (n - 1))
        step = round(steps[0], 2)
        expected_step = round(expect, 2)
    # the thread labels: monospace text at the line row, drawn as sN
    names = [nm for (y, nm) in re.findall(
        r'<text x="[0-9.]+" y="([0-9.]+)"[^>]*>(s\d+)</text>', svg)
        if 180.0 <= float(y) <= 205.0]
    names.sort(key=lambda s: int(s[1:]))
    desc_text = grab_text(svg)[1]
    desc_names = []
    if desc_text:
        m_range = re.search(r'(s\d+)\s*to\s*(s\d+)', desc_text)
        if m_range:
            lo, hi = int(m_range.group(1)[1:]), int(m_range.group(2)[1:])
            desc_names = [f"s{i}" for i in range(lo, hi + 1)]
        else:
            desc_names = re.findall(r'\bs\d+\b', desc_text)
    return {"count": n,
            "diamonds": len(diamonds),
            "tmark": [float(mark.group(1)), 280.0] if mark else None,
            "step": step,
            "expected_step": expected_step,
            "names": names,
            "desc_names": desc_names,
            "_steps": steps}


def recover_ticks(svg):
    """Only the numbers: total, per, tick count, tick width, present dot.

    The caption's own numbers (total/per/tick count as written in the label)
    are recovered too, as private `_caption_*` fields — the render_checks
    notes about caption agreement read those fields, never the label twice.
    """
    mw = re.search(r'pattern id="ticks" width="([0-9.]+)"', svg)
    n = None
    total = per = tick_w = None
    cap_total = cap_per = cap_ticks = None
    total_source = None
    if mw:
        tick_w = num(mw.group(1))
        n = round(1040 / tick_w)
        # cross-check against the caption register when it carries the numbers
        label = grab_text(svg)[2]
        m_total = re.search(r"([\d,]+) sessions", label)
        m_per = re.search(r"each tick (\d+)", label)
        m_ticks = re.search(r"\((\d+) ticks\)", label)
        if m_total:
            cap_total = int(m_total.group(1).replace(",", ""))
        if m_per:
            cap_per = int(m_per.group(1))
        if m_ticks:
            cap_ticks = int(m_ticks.group(1))
        if cap_total is not None:
            total = cap_total
            total_source = "caption"
        if cap_total is not None and n:
            # the geometry completes the per: total/n
            per = round(cap_total / n)
        elif cap_per is not None and n:
            per = cap_per
            # the geometry completes the total: n*per
            total = n * cap_per
            total_source = "completed"
    dot = re.search(r'<circle cx="1120" cy="330" r="7"', svg)
    return {"total": total, "per": per, "ticks": n, "tick_w": tick_w,
            "present_dot": dot is not None,
            "_caption_total": cap_total, "_caption_per": cap_per,
            "_caption_ticks": cap_ticks, "_total_source": total_source}


def recover_boundary(svg):
    """Only the numbers: field-mark count, boundary x, report/tracker coords."""
    circles = get_circles(svg)
    # field marks only: circles left of the boundary; the present-dot style
    # circles (bright, r 7-15) belong to other motifs, not the field
    m_bx = re.search(r'x1="([0-9.]+)" y1="130"', svg)
    boundary_x = num(m_bx.group(1)) if m_bx else 820.0
    bx = boundary_x
    field = [c for c in circles if c[0] < bx - 30]
    translates = [(num(x), num(y))
                  for x, y in re.findall(r'translate\(([0-9.]+),([0-9.]+)\)', svg)]
    report = next((t for t in translates if t[0] < bx), (700.0, 268.0))
    tracker = next((t for t in translates if t[0] >= bx), (940.0, 252.0))
    return {"count": len(field), "field_marks": len(field),
            "all_circles": len(circles),
            "boundary_x": boundary_x, "report": list(report),
            "tracker": list(tracker)}


def recover_scatter(svg):
    """Only the numbers: point count and the invariant line's fraction."""
    circles = get_circles(svg)
    line = re.search(r'<line x1="([0-9.]+)" y1="265" x2="([0-9.]+)" y2="265"', svg)
    frac = None
    if line:
        x1, x2 = map(num, line.groups())
        frac = (x1 - 110) / (1170 - 110)
    return {"count": len(circles), "invariant_fraction": frac}


def recover_hours(svg):
    """Only the numbers: hour boundaries from the block geometry.

    x = 110 + hour * (1000/24). The per-block widths are kept private
    (`_block_widths`) for the renderer to measure against the template.
    """
    px_per_hour = 1000 / 24.0
    xedges = []
    for m in re.finditer(r'<rect x="([0-9.]+)" y="210" width="([0-9.]+)" height="80" fill="none"',
                         svg):
        x, w = num(m.group(1)), num(m.group(2))
        xedges.append((x, x + w))
    dv = re.search(r'<rect x="900" y="378" width="210" height="46"', svg)
    span = [xedges[0][0] if xedges else 110.0, xedges[-1][1] if xedges else 1110.0]
    return {"blocks": [(f"{h0:02d}:00", f"{h1:02d}:00")
                       for h0, h1 in [(0, 6), (6, 9), (9, 17), (17, 24)]],
            "px_per_hour": px_per_hour,
            "span": span,
            "dv_block": dv is not None,
            "_block_widths": [ex1 - ex0 for (ex0, ex1) in xedges]}


def recover_wall(svg):
    """Only the numbers: advisory step count, wall x, ticket/slot/far-side,
    and the drawn label names read back from the <text> nodes.

    The semantic proof, spread from lifelines: the wall's five body labels
    (the documented process, the agent, the gate output, the enforced
    process, the work that follows) are *read*, not just counted — each
    em-dash gloss is stripped, leaving the name; the desc's own named
    labels are parsed too, so the render can state exactly which names the
    caption names and which it leaves drawn-but-unnamed. The wall is the
    first motif whose names are phrases, not single words.
    """
    steps = len(re.findall(r'<line x1="130" y1="([0-9.]+)" x2="350" y2="\1" '
                           r'stroke="#4a4a64" stroke-width="1.5" '
                           r'stroke-dasharray="6 6"/>', svg))
    wall = re.search(r'<rect x="933" y="130" width="14" height="300"', svg)
    ticket = re.search(r'translate\(690,262\)', svg)
    slot = re.search(r'<rect x="926" y="268" width="28" height="24"', svg)
    far = re.search(r'x1="961" y1="280"', svg)
    # body labels: every <text> outside the ticket <g> and outside the
    # label (x=90,y=64) / bottom (x=90,y=440) registers; em-dash gloss
    # stripped to the name.
    body = re.sub(r"<g\b.*?</g>", "", svg, flags=re.S)
    names = []
    for m in re.finditer(r'<text x="([0-9.]+)" y="[0-9.]+"[^>]*>([^<]*)</text>',
                         body):
        x = num(m.group(1))
        if abs(x - 90.0) < 1:
            continue
        raw = m.group(2).strip()
        names.append(raw.split(" — ")[0].strip())
    desc_text = grab_text(svg)[1]
    desc_names = [n for n in names if n in desc_text] if desc_text else []
    return {"steps": steps,
            "wall_x": 933.0 if wall else None,
            "ticket": ticket is not None,
            "punched_slot": slot is not None,
            "far_side": far is not None,
            "names": names,
            "desc_names": desc_names}


def recover_tokens(svg):
    """Only the numbers: raw values, ledger tokens, agents, waves, pages,
    gate x, and the PASS stamps that bind the waves to the gate.

    The semantic proof, spread from recognition: the five vocabulary-page
    tokens are *read*, not just counted — their drawn names (the <text>
    nodes at x=800), the source values (x=195) and the mapped values
    (x=1140) come back as strings; the desc's own named tokens and values
    are parsed too, so the render can state exactly which names the caption
    names and which it leaves drawn-but-unnamed.
    """
    values = len(re.findall(r'<rect x="140" y="([0-9.]+)" width="110" height="26"',
                            svg))
    tokens = len(re.findall(r'<text x="800" y="([0-9.]+)"', svg))
    names = re.findall(r'<text x="800" y="[0-9.]+"[^>]*>([^<]*)</text>', svg)
    source_values = re.findall(r'<text x="195" y="[0-9.]+"[^>]*>([^<]*)</text>', svg)
    mapped_values = re.findall(r'<text x="1140" y="[0-9.]+"[^>]*>([^<]*)</text>', svg)
    agents = len(re.findall(r'<circle cx="([0-9.]+)" cy="120" r="6"', svg))
    waves = len(re.findall(r'>wave (\d+)<', svg))
    # vertical indigo ticks: x1==x2, y1!=y2 (the page ticks across the waves)
    pages = len(re.findall(r'<line x1="([0-9.]+)" y1="([0-9.]+)" x2="\1" '
                           r'y2="([0-9.]+)" stroke="#7c7cf0" '
                           r'stroke-width="1.5" opacity="0.8"', svg))
    gate = re.search(r'x1="([0-9.]+)" y1="140" x2="\1" y2="350"', svg)
    stamps = len(re.findall(r'<rect x="616" y="([0-9.]+)" width="48" height="18"', svg))
    desc_text = grab_text(svg)[1]
    desc_tokens = re.findall(r'--[a-z0-9-]+', desc_text) if desc_text else []
    desc_values = re.findall(r'#[0-9A-Fa-f]{3,8}|\d+px', desc_text) if desc_text else []
    return {"values": values, "tokens": tokens, "names": names,
            "source_values": source_values, "mapped_values": mapped_values,
            "agents": agents, "waves": waves, "pages": pages,
            "gate_x": num(gate.group(1)) if gate else None,
            "pass_stamps": stamps,
            "desc_tokens": desc_tokens, "desc_values": desc_values}


def recover_recognition(svg):
    """Only the numbers: schema layers, SOUL.md line ticks, ledger ticks,
    sessions/per, the gold daily note, the reading thread.

    The layer rows and line ticks are counted structurally (labels at
    x=125, muted vertical ticks); the ledger's tick width recovers the
    tick count the same way recover_ticks does; the caption register
    carries sessions/per; the gold note and the reading thread are
    presence flags.
    """
    layers = len(re.findall(r'<text x="125" y="([0-9.]+)"', svg))
    names = re.findall(r'<text x="125" y="[0-9.]+"[^>]*>([A-Za-z]+)</text>', svg)
    desc_text = grab_text(svg)[1]
    desc_names = None
    if desc_text:
        m_names = re.search(r'—\s*([A-Za-z]+(?:,\s*[A-Za-z]+)+)\s*—', desc_text)
        if m_names:
            desc_names = [s.strip() for s in m_names.group(1).split(",")]
    lines = len(re.findall(r'stroke="#5a5a74" stroke-width="1.5" opacity="0.6"',
                           svg))
    mw = re.search(r'pattern id="rec-ticks" width="([0-9.]+)"', svg)
    ticks = n = None
    tick_w = None
    total = per = None
    cap_total = cap_per = cap_ticks = None
    if mw:
        tick_w = num(mw.group(1))
        n = round(560 / tick_w)
        ticks = n
        label = grab_text(svg)[2]
        m_total = re.search(r"([\d,]+) sessions", label)
        m_per = re.search(r"each tick (\d+)", label)
        m_ticks = re.search(r"\((\d+) ticks\)", label)
        if m_total:
            cap_total = int(m_total.group(1).replace(",", ""))
        if m_per:
            cap_per = int(m_per.group(1))
        if m_ticks:
            cap_ticks = int(m_ticks.group(1))
        total = cap_total
        per = cap_per
    note = re.search(r'<rect x="1088" y="296" width="64" height="64" fill="#d9a441"',
                     svg) is not None
    reading = re.search(r'<path d="M512 280 L 1088 328"', svg) is not None
    return {"layers": layers, "names": names, "desc_names": desc_names,
            "lines": lines, "ticks": ticks, "tick_w": tick_w,
            "sessions": total, "per": per,
            "note": note, "reading": reading,
            "_caption_sessions": cap_total, "_caption_per": cap_per,
            "_caption_ticks": cap_ticks}


def recover_rails(svg):
    """Only the numbers: family marks above, triad marks below, gold
    threads, gold diamonds, and the dual members' names read back from the
    desc.

    The family line (site/tools/build-family-line.py) is its own model:
    FAMILY (4 marks) above, TRIAD (3 marks) below, two gold threads where
    the word and the recognition sit on both rails, two gold diamonds at
    the duals' anchors, and one reading line. The counts are recovered
    structurally — the top rail's rects at y=129, the bottom rail's at
    y=299, the gold thread strokes, the diamond paths — and the duals'
    names are *read* from the desc's own thread sentence, so the render
    can state exactly which members are load-bearing and which are
    single-family by design.
    """
    top = [r for r in get_rects(svg, y_tol=2.0, y_target=129.0)
           if abs(r[2] - 190.0) < 0.5 and abs(r[3] - 62.0) < 0.5]
    bottom = [r for r in get_rects(svg, y_tol=2.0, y_target=299.0)
              if abs(r[2] - 190.0) < 0.5 and abs(r[3] - 62.0) < 0.5]
    threads = len(re.findall(r'stroke="#d9a441" stroke-width="2.5"', svg))
    diamonds = len(re.findall(r'<path [^>]*fill="#d9a441"', svg))
    desc_text = grab_text(svg)[1]
    duals = []
    if desc_text:
        # the desc's own thread sentence names the dual members
        m = re.search(r'one straight down from ([^,]+), one diagonal from '
                      r'(.+?) to the triad', desc_text)
        if m:
            duals = [m.group(1).strip(), m.group(2).strip()]
    reading = "load-bearing" in svg
    return {"family": len(top), "triad": len(bottom),
            "threads": threads, "diamonds": diamonds,
            "duals": duals, "reading": reading}


RECOVER = {
    "lifelines": recover_lifelines,
    "ticks": recover_ticks,
    "boundary": recover_boundary,
    "scatter": recover_scatter,
    "hours": recover_hours,
    "wall": recover_wall,
    "tokens": recover_tokens,
    "recognition": recover_recognition,
    "rails": recover_rails,
}


# ------------------------------------------------- checks as renderings

HOUR_BLOCKS = [(0, 6), (6, 9), (9, 17), (17, 24)]


def render_lifelines(rec):
    """The human check list as an f-string of the recovered fields.

    The semantic proof, recognition-style: the thread labels are read back
    as *names* and diffed against the names the desc itself writes.
    who-made-the-mark's desc names all six (s1 to s6 — full identity);
    ten-essays' desc names only the count — all ten drawn labels come back
    as drawn-but-unnamed, the second honest partial the round trip has
    produced.
    """
    n = rec.get("count", 0)
    diamonds = rec.get("diamonds", 0)
    tmark = rec.get("tmark")
    step = rec.get("step")
    expected = rec.get("expected_step")
    steps = rec.get("_steps") or []
    checks = [("life-lines counted", n),
              ("diamonds = n-1", diamonds == max(0, n - 1),
               f"{diamonds} diamonds for {n} lines"),
              ("T-mark present", tmark is not None,
               f"at ({tmark[0]:g}, 280)" if tmark else "at (?, 280)")]
    if step is not None and expected is not None:
        checks.append(("life-lines evenly spaced",
                       all(abs(s - expected) < 0.6 for s in steps),
                       f"step {step:.1f} vs expected {expected:.1f}"))
    names = rec.get("names") or []
    desc_names = rec.get("desc_names") or []
    if names:
        checks.append(("thread labels read back", len(names) == n,
                       ", ".join(names)))
        if desc_names:
            drawn_in_desc = [x for x in desc_names if x in names]
            checks.append(("desc names the threads",
                           len(drawn_in_desc) == len(desc_names),
                           ", ".join(drawn_in_desc) + " — named in the desc"))
        unnamed = [x for x in names if x not in desc_names]
        if unnamed:
            checks.append(("threads drawn but unnamed in the desc", True,
                           f"{len(unnamed)} — " + ", ".join(unnamed)))
    return checks


def render_ticks(rec):
    """Every note is read from the fields — never from the label twice.

    The total/per derivation branch is recovered data too (`_total_source`):
    the render only narrates where the number came from, it never guesses.
    """
    tick_w = rec.get("tick_w")
    ticks = rec.get("ticks")
    total = rec.get("total")
    per = rec.get("per")
    source = rec.get("_total_source")
    cap_total = rec.get("_caption_total")
    cap_per = rec.get("_caption_per")
    cap_ticks = rec.get("_caption_ticks")
    checks = []
    if tick_w is not None and ticks is not None:
        checks.append(("tick width measured", True,
                       f"{tick_w}px per tick -> {ticks} ticks (1040/{tick_w:.2f})"))
    if cap_total is not None:
        checks.append(("total sessions from caption", True, f"{cap_total}"))
    if cap_per is not None:
        checks.append(("per from caption", True, f"{cap_per}"))
    if cap_ticks is not None:
        checks.append(("ticks from caption agree", cap_ticks == ticks,
                       f"caption {cap_ticks} vs geometry {ticks}"))
    if source == "caption" and ticks:
        checks.append(("per recovered from total/n", True, f"{per}"))
    elif source == "completed":
        checks.append(("total completed from n*per", True, f"{ticks}×{per} = {total}"))
    checks.append(("present dot at (1120,330)", rec.get("present_dot", False)))
    return checks


def render_boundary(rec):
    """The canonical-coordinate checks, measured from the recovered coords."""
    bx = rec.get("boundary_x", 820.0)
    report = rec.get("report") or [700.0, 268.0]
    tracker = rec.get("tracker") or [940.0, 252.0]
    return [
        ("boundary at x=820", abs(bx - 820.0) < 0.01),
        ("report at (700,268)",
         abs(report[0] - 700.0) < 0.01 and abs(report[1] - 268.0) < 0.01),
        ("tracker at (940,252)",
         abs(tracker[0] - 940.0) < 0.01 and abs(tracker[1] - 252.0) < 0.01),
    ]


def render_scatter(rec):
    frac = rec.get("invariant_fraction")
    if frac is None:
        return []
    return [("invariant line placement", abs(frac - 0.55) < 0.02,
             f"at {frac:.3f} of the span (expected 0.55)")]


def render_hours(rec):
    """Block widths measured against the template, from the recovered widths."""
    widths = rec.get("_block_widths") or []
    pxh = rec.get("px_per_hour", 1000 / 24.0)
    span = rec.get("span") or [110.0, 1110.0]
    checks = []
    if len(widths) == 4:
        for (h0, h1), w in zip(HOUR_BLOCKS, widths):
            w_exp = (h1 - h0) * pxh
            checks.append((f"block {h0:02d}-{h1:02d} width",
                           abs(w - w_exp) < 1.0,
                           f"{w:.1f}px vs expected {w_exp:.1f}px"))
        checks.append(("day spans 24h (110..1110)",
                       abs(span[0] - 110.0) < 1.0 and abs(span[1] - 1110.0) < 1.0))
    checks.append(("thirty-first hour dashed block", rec.get("dv_block", False)))
    return checks


def render_wall(rec):
    """The wall's checks, from the recovered fields.

    The semantic proof, recognition-style: the five drawn labels are read
    back as *names* and diffed against the names the desc itself writes.
    The wall's desc names all five (the documented process, the agent, the
    gate output, the enforced process, the work that follows) — a full
    identity, the third after recognition and who-made-the-mark, and the
    first whose names are phrases rather than single words. The check that
    once counted six steps now reads the words it was standing in front
    of.
    """
    steps = rec.get("steps", 0)
    wall_x = rec.get("wall_x")
    checks = [
        ("advisory steps counted", True, f"{steps}"),
        ("wall at x=933", wall_x is not None and abs(wall_x - 933.0) < 0.01),
        ("gate output ticket present", rec.get("ticket", False),
         "gold ticket at the wall's near side"),
        ("punched slot crosses the wall", rec.get("punched_slot", False)),
        ("far side exists past the wall", rec.get("far_side", False)),
    ]
    names = rec.get("names") or []
    desc_names = rec.get("desc_names") or []
    if names:
        checks.append(("wall labels read back", len(names) == 5,
                       ", ".join(names)))
        if desc_names:
            drawn_in_desc = [x for x in desc_names if x in names]
            checks.append(("desc names the wall's labels",
                           len(drawn_in_desc) == len(desc_names),
                           ", ".join(drawn_in_desc) + " — named in the desc"))
        unnamed = [x for x in names if x not in desc_names]
        if unnamed:
            checks.append(("labels drawn but unnamed in the desc", True,
                           f"{len(unnamed)} — " + ", ".join(unnamed)))
    return checks


def render_tokens(rec):
    """The named-value gate's checks, from the recovered fields.

    The semantic proof, recognition-style: the five vocabulary-page tokens
    are read back as *names* and diffed against the names the desc itself
    writes. The desc names only the crossing pair (--brand-accent,
    --space-2); the recovery states that pair is drawn, and names the rest
    honestly as drawn-but-unnamed — the first partial identity the round
    trip has produced, where recognition got the full one.
    """
    values = rec.get("values", 0)
    tokens = rec.get("tokens", 0)
    agents = rec.get("agents", 0)
    waves = rec.get("waves", 0)
    pages = rec.get("pages", 0)
    gate_x = rec.get("gate_x")
    stamps = rec.get("pass_stamps", 0)
    names = rec.get("names") or []
    desc_tokens = rec.get("desc_tokens") or []
    source_values = rec.get("source_values") or []
    desc_values = rec.get("desc_values") or []
    checks = [
        ("raw values in the source layer", True, f"{values}"),
        ("named tokens in the vocabulary page", True, f"{tokens}"),
        ("agents with no shared memory", agents == 5, f"{agents}"),
        ("waves stamped", True, f"{waves}"),
        ("pages built behind the gate", pages == 35, f"{pages}"),
        ("gate at x=640", gate_x is not None and abs(gate_x - 640.0) < 0.01),
        ("PASS stamp per wave", stamps == waves, f"{stamps} stamps"),
    ]
    if names:
        drawn_in_desc = [n for n in desc_tokens if n in names]
        checks.append(("token names read back", len(names) == tokens,
                       ", ".join(names)))
        if desc_tokens:
            checks.append(("desc names the crossing pair",
                           len(drawn_in_desc) == len(desc_tokens),
                           ", ".join(drawn_in_desc) + " — named in the desc"))
        unnamed = [n for n in names if n not in desc_tokens]
        if unnamed:
            checks.append(("tokens drawn but unnamed in the desc", True,
                           f"{len(unnamed)} — " + ", ".join(unnamed)))
    if source_values:
        drawn_desc_vals = [v for v in desc_values if v in source_values]
        checks.append(("source values read back", True,
                       ", ".join(source_values)))
        if desc_values:
            checks.append(("desc values drawn in the source layer",
                           len(drawn_desc_vals) == len(desc_values),
                           ", ".join(drawn_desc_vals) + " — named in the desc"))
    return checks


def render_recognition(rec):
    """The config-versus-recognition split's checks, from the fields."""
    layers = rec.get("layers", 0)
    lines = rec.get("lines", 0)
    ticks = rec.get("ticks")
    tick_w = rec.get("tick_w")
    sessions = rec.get("sessions")
    per = rec.get("per")
    cap_ticks = rec.get("_caption_ticks")
    checks = [
        ("schema layers counted", True, f"{layers}"),
        ("schema names read back",
         bool(rec.get("names")) and rec.get("names") == rec.get("desc_names"),
         ", ".join(rec.get("names") or ["(none drawn)"])
         + " — the desc reads the same names" if rec.get("names") == rec.get("desc_names")
         else ", ".join(rec.get("names") or ["(none drawn)"])
         + f" — desc says {rec.get('desc_names')}"),
        ("SOUL.md line ticks drawn", True, f"{lines}"),
        ("ledger tick width measured",
         tick_w is not None and ticks is not None,
         f"{tick_w}px per tick -> {ticks} ticks (560/{tick_w:.2f})"),
    ]
    if sessions is not None:
        checks.append(("sessions from caption", True, f"{sessions}"))
    if per is not None:
        checks.append(("per from caption", True, f"{per}"))
    if cap_ticks is not None and ticks is not None:
        checks.append(("caption ticks agree with geometry",
                       cap_ticks == ticks,
                       f"caption {cap_ticks} vs geometry {ticks}"))
    checks.append(("gold daily note where the reading lands", rec.get("note", False)))
    checks.append(("reading thread from the self", rec.get("reading", False)))
    return checks


def render_rails(rec):
    """The family line's checks, from the recovered fields.

    The semantic proof: the dual members are *read* — the desc's own
    thread sentence names them ("one straight down from Name It First, one
    diagonal from Schema and Practice to the triad's middle"), so the
    render can state exactly which members are load-bearing. The counts
    (family above, triad below, threads, diamonds) are structural; the
    reading line's presence is the drawn tissue.
    """
    fam = rec.get("family", 0)
    tri = rec.get("triad", 0)
    threads = rec.get("threads", 0)
    diamonds = rec.get("diamonds", 0)
    duals = rec.get("duals") or []
    checks = [
        ("family marks above", fam == 4, f"{fam}"),
        ("triad marks below", tri == 3, f"{tri}"),
        ("gold threads", threads == 2, f"{threads}"),
        ("gold diamonds = threads", diamonds == threads,
         f"{diamonds} diamonds for {threads} threads"),
        ("dual members named in the desc", len(duals) == 2,
         ", ".join(duals) if duals else "(none read back)"),
        ("reading line drawn", rec.get("reading", False),
         "load-bearing · single-family by design"),
    ]
    return checks


RENDER = {
    "lifelines": render_lifelines,
    "ticks": render_ticks,
    "boundary": render_boundary,
    "scatter": render_scatter,
    "hours": render_hours,
    "wall": render_wall,
    "tokens": render_tokens,
    "recognition": render_recognition,
    "rails": render_rails,
}


def render_checks(motif, rec):
    """The check list as a rendering of the recovered fields.

    Every note is an f-string of the field it checks — the human card can
    never disagree with the JSON card, because the JSON card is its only
    ingredient. If a field's meaning changes, the note changes with it; a
    note cannot drift from its number.
    """
    return RENDER[motif](rec)


# ------------------------------------------------------------ round-trip

def regen_args(motif, data, svg_text=None):
    """Reconstruct the illustrate.py CLI for the recovered parameters.

    When the original SVG is available, its own captions (title, label,
    bottom) are carried into the regeneration so a generator-made file can
    round-trip byte-identical, not just geometry-identical. The desc is
    deliberately NOT carried: it is generated from the numbers, and proving
    the file's numbers regenerate its desc is the point.
    """
    args = ["python3", str(ILLUSTRATE), motif]
    if motif == "rails":
        # The family line has no illustrate.py template — it is its own
        # model (build-family-line.py). The round-trip re-runs THAT
        # generator in --stdout mode (the Q17/Q20 branch: the generator is
        # the proof, re-running it re-proves it); the recovered parameters
        # are not needed because the model is fixed, and the desc is
        # regenerated from the model, so byte-identity is the claim.
        return ["python3", str(TOOLS_DIR / "build-family-line.py"), "--stdout"]
    if motif == "lifelines":
        args += ["--count", str(data.get("count", 0))]
    elif motif == "ticks":
        total = data.get("total") or 2700
        per = data.get("per") or 31
        args += ["--total", str(total), "--per", str(per)]
    elif motif in ("boundary", "scatter"):
        args += ["--count", str(data.get("count", 0))]
    elif motif == "wall":
        args += ["--steps", str(data.get("steps", 0))]
    elif motif == "tokens":
        args += ["--values", str(data.get("values", 2)),
                 "--tokens", str(data.get("tokens", 5)),
                 "--waves", str(data.get("waves", 3)),
                 "--pages", str(data.get("pages", 35)),
                 "--agents", str(data.get("agents", 5))]
    elif motif == "recognition":
        args += ["--layers", str(data.get("layers", 4)),
                 "--lines", str(data.get("lines", 100)),
                 "--sessions", str(data.get("sessions", 2700)),
                 "--per", str(data.get("per", 31))]
    args += ["--seed", str(SEED)]
    if svg_text:
        title, desc, label, bottom = grab_text(svg_text)
        if title:
            args += ["--title", title]
        if label:
            args += ["--label", label]
        if bottom:
            args += ["--bottom", bottom]
    return args


def geometry_of(svg):
    """Geometry fingerprint: every element, numbers normalized, text stripped.

    Compares shape, not copy — so the round-trip proves the numbers survived
    even when the essay's own captions (which live in the image, not in the
    generator template) differ from a fresh default render.
    """
    svg = re.sub(r"<!--.*?-->", "", svg, flags=re.S)  # hand-written annotation
    svg = re.sub(r"<text\b.*?</text>", "", svg, flags=re.S)
    svg = re.sub(r"<title\b.*?</title>", "", svg, flags=re.S)
    svg = re.sub(r"<desc\b.*?</desc>", "", svg, flags=re.S)
    svg = re.sub(r"\s+", "", svg)  # whitespace is not geometry
    # normalize the rounded floats the generator emits (e.g. 15.29 vs 15.29)
    svg = re.sub(r"(\d+)\.(\d{3,})", lambda m: f"{float(m.group(0)):.2f}", svg)
    # and integer notation vs float notation of the same number (176 vs 176.0)
    svg = re.sub(r"\.0\b", "", svg)
    return svg


def desc_provenance(original, fresh):
    """The caption as a test: does the regenerated desc equal the file's own?

    The generator writes the <desc> from the same numbers it draws with, so a
    round-trip that carries the numbers must also re-derive the caption. When
    the fresh desc is byte-identical to the file's, the image's own words are
    machine-proven. When they differ, the hand wrote its own caption — a
    fossil's voice, named honestly rather than regenerated over.
    """
    o = grab_text(original)[1]
    f = grab_text(fresh)[1]
    if o == f:
        return {"same": True, "status": "caption regenerates from its numbers",
                "original": o, "fresh": f}
    return {"same": False, "status": "caption hand-written — differs from the numbers' template",
            "original": o, "fresh": f}


def roundtrip(path, motif, data):
    """Regenerate with recovered numbers; compare geometry. Returns dict."""
    cmd = regen_args(motif, data, svg_text=path.read_text())
    proc = subprocess.run(cmd, capture_output=True, text=True, timeout=60)
    if proc.returncode != 0:
        return {"ok": False, "reason": f"illustrate.py failed: {proc.stderr[:200]}"}
    fresh = proc.stdout
    original = path.read_text()
    if fresh == original:
        status = "byte-identical"
    elif geometry_of(fresh) == geometry_of(original):
        status = "geometry-identical (captions differ)"
    else:
        status = "geometry differs"
        if motif == "ticks":
            mw_orig = re.search(r'pattern id="ticks" width="([0-9.]+)"', original)
            mw_fresh = re.search(r'pattern id="ticks" width="([0-9.]+)"', fresh)
            if mw_orig and mw_fresh:
                dw = num(mw_orig.group(1)) - num(mw_fresh.group(1))
                status += f" — tick width {mw_orig.group(1)}px vs generator {mw_fresh.group(1)}px ({dw:+.2f}px)"
        elif motif == "lifelines":
            d_orig = re.findall(r'rotate\(45 ([0-9.]+) 300\)', original)
            d_fresh = re.findall(r'rotate\(45 ([0-9.]+) 300\)', fresh)
            if d_orig and d_fresh:
                do = num(d_orig[0]); df = num(d_fresh[0])
                if abs(do - df) > 0.5:
                    status += f" — diamond centers {d_orig[0]} vs generator {d_fresh[0]} ({do-df:+.0f}px, hand-tuned)"
            if "<path d=\"M" not in original:
                status += " — thread path lacks its initial M (pre-fix render)"
    return {"ok": status.startswith("byte") or status.startswith("geometry"),
            "status": status,
            "desc": desc_provenance(original, fresh),
            "fresh_bytes": len(fresh), "original_bytes": len(original)}


# ------------------------------------------------------------ report

def card(path, motif, data, rt=None):
    title, desc, label, bottom = grab_text(path.read_text())
    lines = [f"{path.name} — {motif}"]
    lines.append(f"  title : {title}")
    if label:
        lines.append(f"  label : {label}")
    if bottom:
        lines.append(f"  bottom: {bottom}")
    lines.append("  recovered:")
    for k, v in data.items():
        if k == "checks" or k.startswith("_"):
            continue
        lines.append(f"    {k} = {v}")
    for name, *rest in data.get("checks", []):
        if len(rest) == 1:
            ok = rest[0]
            note = ""
        else:
            ok, note = rest
        lines.append(f"    [{'ok' if ok else 'MISS'}] {name} {note}")
    if rt:
        lines.append(f"  round-trip: {rt['status']}")
        if not rt["ok"]:
            lines.append(f"    (illustrate.py would draw this differently: {rt.get('reason', 'geometry mismatch')})")
        d = rt.get("desc")
        if d:
            lines.append(f"  desc: {d['status']}")
            if not d["same"]:
                lines.append(f"    original: {d['original'][:80]}…")
                lines.append(f"    fresh   : {d['fresh'][:80]}…")
    return "\n".join(lines)


def main():
    p = argparse.ArgumentParser(description=__doc__,
                                formatter_class=argparse.RawDescriptionHelpFormatter)
    p.add_argument("files", nargs="*", help="SVG files (or --all)")
    p.add_argument("--all", action="store_true", help="scan site/assets/*.svg")
    p.add_argument("--json", action="store_true", help="emit JSON per file")
    p.add_argument("--verify-only", action="store_true",
                   help="only round-trip status, no card")
    a = p.parse_args()

    if a.all:
        assets = TOOLS_DIR.parent / "assets"
        files = sorted(assets.glob("*.svg"))
    else:
        files = [Path(f) for f in a.files]

    if not files:
        p.error("no files — pass SVG paths or --all")

    results = []
    for path in files:
        svg = path.read_text()
        motif = detect_motif(svg)
        if motif == "hand-drawn":
            rec = {"checks": [("not a generator motif — hand-drawn register", True)]}
            rt = None
        else:
            rec = RECOVER[motif](svg)
            rec["checks"] = render_checks(motif, rec)
            rt = roundtrip(path, motif, rec) if not a.verify_only else None
        results.append({"file": path.name, "path": str(path), "motif": motif,
                        "recovered": rec, "roundtrip": rt})

    if a.json:
        print(json.dumps(results, indent=2))
        return

    for r in results:
        if a.verify_only and r["roundtrip"]:
            print(f"{r['file']:40s} {r['motif']:10s} {r['roundtrip']['status']}")
        else:
            print(card(Path(r["path"]), r["motif"], r["recovered"], r["roundtrip"]))
        print()


if __name__ == "__main__":
    main()
