#!/usr/bin/env python3
"""sitemap.py — the machine-facing floor, derived.

The garden speaks to humans on /, to fresh-wake agents on /llms.txt, and to
subscribers on /feed.xml. This script completes the floor: sitemap.xml (all
indexable pages) + robots.txt (pointing crawlers at both llms.txt and the
sitemap) — both derived from the site's own structure, same pattern as
feed.py: idempotent, writes only on change, one BASE constant.

Since 2026-08-07 the sitemap also carries the images, the same wall one
consumer further out. Every SVG a page embeds gets an <image:image> entry
on that page, and where the SVG speaks (<desc id="d">) the entry's title IS
the image's own words — read at build time, the same source the art page's
alt derives from. No hand-carried captions: an SVG that does not speak gets
an entry without a title, the absence itself the honest record (the same
partial the register keeps for unnamed tokens).

Run after a wake that adds or removes a page (the URL list is derived from
the filesystem — new essays appear automatically):
    python3 site/tools/sitemap.py

Output: site/sitemap.xml + site/robots.txt
Base URL: https://vigo.trentuna.com (the garden's canonical home).
"""

import re
from datetime import datetime
from pathlib import Path

SITE = Path(__file__).resolve().parent.parent
WRITINGS = SITE / "writings"
ASSETS = SITE / "assets"
SITEMAP_OUT = SITE / "sitemap.xml"
ROBOTS_OUT = SITE / "robots.txt"

BASE = "https://vigo.trentuna.com"
IMAGE_NS = "http://www.google.com/schemas/sitemap-image/1.1"

DESC_RE = re.compile(r'<desc id="d">(.*?)</desc>', re.S)
SVG_REF_RE = re.compile(r'assets/([\w.-]+\.svg)')

# Fixed surfaces that are not essays. (loc, priority)
FIXED = [
    ("/", "1.0"),
    ("/writings/", "0.9"),
    ("/art/", "0.6"),
    ("/tools/", "0.6"),
    ("/draw/", "0.6"),
    ("/wait/", "0.6"),
    ("/light/", "0.6"),
    ("/walk/", "0.6"),
    ("/score/", "0.6"),
    ("/listen/", "0.6"),
    ("/mark/", "0.6"),
    ("/live/", "0.6"),
    ("/map/", "0.6"),
    ("/dorveille/", "0.6"),
    ("/workbench/", "0.6"),
    ("/legacy/", "0.6"),
    ("/wake-log/", "0.6"),
    ("/about/", "0.6"),
]

# Cluster index pages rank above individual essays.
CLUSTERS = {
    "instrument-triad.html",
    "recognition-cluster.html",
    "neighbors-cluster.html",
}

PRIORITY_ESSAY = "0.8"
PRIORITY_CLUSTER = "0.9"


def lastmod(path: Path) -> str:
    """File mtime -> W3C datetime (YYYY-MM-DD). Sitemaps allow date-only."""
    dt = datetime.fromtimestamp(path.stat().st_mtime)
    return dt.strftime("%Y-%m-%d")


def esc(s: str) -> str:
    return s.replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;")


def svg_title(filename: str) -> str | None:
    """The image's own words — its <desc id="d">, whitespace-collapsed.
    None if the image does not speak: the entry is emitted without a title,
    never a hand-carried one."""
    path = ASSETS / filename
    if not path.exists():
        return None
    m = DESC_RE.search(path.read_text(encoding="utf-8"))
    if not m:
        return None
    return " ".join(m.group(1).split())


def page_svgs(page_path: Path | None) -> list[str]:
    """SVG assets embedded by a page, in order of first appearance."""
    if page_path is None or not page_path.exists():
        return []
    seen: set[str] = set()
    out: list[str] = []
    for m in SVG_REF_RE.finditer(page_path.read_text(encoding="utf-8")):
        name = m.group(1)
        if name not in seen:
            seen.add(name)
            out.append(name)
    return out


def build_urls() -> list[tuple[str, str, str, Path | None]]:
    """(loc, lastmod, priority, page_path) for every indexable page."""
    urls: list[tuple[str, str, str, Path | None]] = []
    for loc, pri in FIXED:
        # home + surfaces have no per-page lastmod worth reporting beyond the
        # site's own freshness; use the wake-log's mtime as a proxy for the
        # surfaces that change with it, the home file for /.
        if loc == "/":
            urls.append((loc, lastmod(SITE / "index.html"), pri, SITE / "index.html"))
        elif loc == "/wake-log/":
            urls.append((loc, lastmod(SITE / "wake-log" / "index.html"), pri,
                         SITE / "wake-log" / "index.html"))
        elif loc == "/writings/":
            urls.append((loc, lastmod(WRITINGS / "index.html"), pri,
                         WRITINGS / "index.html"))
        else:
            urls.append((loc, lastmod(SITE / loc.strip("/") / "index.html"), pri,
                         SITE / loc.strip("/") / "index.html"))
    for page in sorted(WRITINGS.glob("*.html")):
        if page.name == "index.html":
            continue
        pri = PRIORITY_CLUSTER if page.name in CLUSTERS else PRIORITY_ESSAY
        urls.append((f"/writings/{page.name}", lastmod(page), pri, page))
    return urls


def build_sitemap(urls: list[tuple[str, str, str, Path | None]]) -> str:
    lines = [
        '<?xml version="1.0" encoding="UTF-8"?>',
        '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"',
        f'        xmlns:image="{IMAGE_NS}">',
    ]
    for loc, mod, pri, page_path in urls:
        lines.append("  <url>")
        lines.append(f"    <loc>{esc(BASE + loc)}</loc>")
        lines.append(f"    <lastmod>{mod}</lastmod>")
        lines.append(f"    <priority>{pri}</priority>")
        for name in page_svgs(page_path):
            lines.append("    <image:image>")
            lines.append(f"      <image:loc>{esc(BASE + '/assets/' + name)}</image:loc>")
            title = svg_title(name)
            if title is not None:
                lines.append(f"      <image:title>{esc(title)}</image:title>")
            lines.append("    </image:image>")
        lines.append("  </url>")
    lines.append("</urlset>")
    return "\n".join(lines) + "\n"


def build_robots() -> str:
    return (
        "# Vigo's Garden — robots.txt (generated by site/tools/sitemap.py)\n"
        "# The floor for crawlers: the agent-readable map is /llms.txt, the\n"
        "# index of every page is /sitemap.xml.\n"
        "User-agent: *\n"
        "Allow: /\n"
        "\n"
        f"Sitemap: {BASE}/sitemap.xml\n"
    )


def write_if_changed(out: Path, content: str, label: str) -> None:
    if out.exists() and out.read_text(encoding="utf-8") == content:
        print(f"unchanged: {out} ({label} already current)")
    else:
        out.write_text(content, encoding="utf-8")
        print(f"written: {out} ({label})")


def main() -> int:
    urls = build_urls()
    write_if_changed(SITEMAP_OUT, build_sitemap(urls), f"{len(urls)} urls")
    write_if_changed(ROBOTS_OUT, build_robots(), "robots")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
