#!/usr/bin/env python3
"""Measures the figures optionalindustries.com presents as evidence.

Why: the studio page does not claim, it SHOWS — under every number stands the command
that produced it. For that to stay true, none of these numbers may be maintained by
hand: the build measures them, and a number that cannot be measured does not go on
the page.

Usage:  python3 web/receipts.py                 (readable, for looking things up)
        python3 web/receipts.py --json          (for web/build.py)
        python3 web/receipts.py --rev <commit>  (against one specific commit)

Comments here are English because this file is published — the page links to it, so a
reader lands in it. The rest of the repository comments in German; this file is the
deliberate exception.

And because it is published, it is also the ONE file in the repository that carries no
internal notes: no tool or script names, no paths outside the ones already printed on the
page, nothing about how the work is organised. What explains a measurement belongs here;
everything else belongs in a file that stays private.

RULE 1 — everything here must be measurable FROM THE REPOSITORY.
The first draft carried two figures that described the machine the work happened on
rather than the project itself. On another clone they differ, and the page would have
claimed "measured at build time" about something the build cannot see. Before adding a
field, ask: is it in git?

RULE 2 — everything against ONE commit, never against HEAD and never against the working
tree. `git rev-list --count HEAD` once moved by three within minutes while the page was
being built, because work landed in between. Two numbers on the same page came from two
different states. So the commit is resolved ONCE here and every measurement — including
every file measurement — runs through `git show`/`git ls-tree` against exactly that commit. A dirty
working tree can no longer distort the result.

RULE 3 — the number and its label must mean the same thing.
The page once printed "38 gate tests" under a command that counted FILES; those 38 files
hold 367 tests. This script therefore returns both numbers separately, and the page labels
what it shows.
"""

from __future__ import annotations

import argparse
import json
import re
import subprocess
import sys
from collections import Counter
from datetime import date, timedelta
from pathlib import Path

ROOT = Path(__file__).resolve().parent.parent

# The day the closed alpha opened on iOS and Android. NOT a measurement — an event that is
# not recorded in the repository. It sits here as the single constant so that the page does
# not carry "154 days to the alpha" by hand and quietly go stale on the next build.
ALPHA_DATE = date(2026, 8, 22)

# Backlog sources: the active file AND the archives. Closed entries are moved into archives
# as the active file grows — counting only docs/BACKLOG.md reports the current stock, not the
# total. The page says "bugs written down ... none of them forgotten"; archived is not
# forgotten, so the archives count too.
BACKLOG_GLOBS = ["docs/BACKLOG.md", "docs/backlog-archive-*.md"]


class MeasurementError(RuntimeError):
    """A measurement failed.

    Deliberately hard: a fallback value would be an unevidenced number on a page whose
    only point is that its numbers are evidenced. Better to break the build.
    """


def _git(*args: str) -> str:
    """Run git in the repository root and return stdout."""
    result = subprocess.run(
        ["git", "-C", str(ROOT), *args],
        capture_output=True,
        text=True,
        check=False,
    )
    if result.returncode != 0:
        raise MeasurementError(
            f"git {' '.join(args)} -> {result.returncode}: {result.stderr.strip()}"
        )
    return result.stdout.strip()


def _blob(rev: str, path: str) -> str:
    """Read file contents FROM THE COMMIT, not from the working tree — rule 2."""
    return _git("show", f"{rev}:{path}")


def _paths(rev: str, prefix: str, pattern: str) -> list[str]:
    """List files in the commit that match `pattern` — rule 2."""
    names = _git("ls-tree", "-r", "--name-only", rev, prefix).splitlines()
    rx = re.compile(pattern)
    return sorted(n for n in names if rx.search(n))


# ------------------------------------------------------------------ measurements
# Every function carries the command that stands under its number on the page.
# Change the measurement here and the footnote there changes with it — both together.


def commit_days(rev: str) -> list[date]:
    """git log --format=%ad --date=short — one date per commit, oldest first."""
    raw = _git("log", "--format=%ad", "--date=short", rev)
    days = [date.fromisoformat(line) for line in raw.splitlines() if line]
    if not days:
        raise MeasurementError("no commits found")
    return sorted(days)


def co_author_commits(rev: str) -> int:
    """git log --format=%b | grep -c 'Co-Authored-By: Claude'

    COMMITS are counted, not lines: `git log --grep` filters at commit level. Measured
    both ways on 2026-08-26 — they agree, because no commit carries two Claude trailers.
    The commit-level path stays the right one anyway: it also holds if one ever does.
    """
    raw = _git("log", "--format=%H", "--grep=Co-Authored-By: Claude", rev)
    return len({line for line in raw.splitlines() if line})


def nonmerge_co_author(rev: str) -> dict:
    """Trailer share WITHOUT merge commits — the denominator the page names.

    Why not simply `co_author_commits / commits`: a `git merge --no-ff` writes no trailer,
    and a merge is not a typed line either. Leaving merges in the denominator makes several
    hundred commits look like unevidenced handwork that never was any.

    This number stays a LOWER BOUND too: commits written by scripts carry no trailer either.
    Only the trailer is evidence — what happened without one is in no repository.
    """
    total = int(_git("rev-list", "--count", "--no-merges", rev))
    raw = _git("log", "--no-merges", "--format=%H", "--grep=Co-Authored-By: Claude", rev)
    with_trailer = len({line for line in raw.splitlines() if line})
    if with_trailer > total:
        raise MeasurementError(f"more trailers ({with_trailer}) than non-merge commits ({total})")
    return {
        "total": total,
        "with_trailer": with_trailer,
        "merges": int(_git("rev-list", "--count", "--merges", rev)),
    }


def co_author_by_month(rev: str) -> list[dict]:
    """git log --no-merges --date=format:%Y-%m --format=%ad --grep='Co-Authored-By: Claude'

    The same share as nonmerge_co_author, but resolved by month instead of summed over
    the whole history. One number for the whole repository hides the thing that actually
    happened: the share was barely half at the start and is near-total now. That movement
    is the claim this page makes, so it has to be measured, not asserted.

    Merges stay out of both numerator and denominator, for the reason given above. The
    LAST entry is almost always a partial month — the page has to label it as one, so the
    flag travels with the data rather than being re-derived in the template.
    """

    def by_month(*extra: str) -> Counter:
        raw = _git("log", "--no-merges", "--date=format:%Y-%m", "--format=%ad", *extra, rev)
        return Counter(line.strip() for line in raw.splitlines() if line.strip())

    total = by_month()
    trailer = by_month("--grep=Co-Authored-By: Claude")
    if not total:
        raise MeasurementError("no non-merge commits found — cannot build the monthly share")

    measured = _git("show", "-s", "--format=%cs", rev).strip()[:7]
    out = []
    for month in sorted(total):
        n, t = total[month], trailer.get(month, 0)
        if t > n:
            raise MeasurementError(f"{month}: more trailers ({t}) than non-merge commits ({n})")
        out.append(
            {
                "month": month,
                "label": date.fromisoformat(month + "-01").strftime("%b"),
                "total": n,
                "trailer": t,
                "pct": round(t * 100 / n),
                "partial": month == measured,
            }
        )
    return out


def authors(rev: str) -> dict:
    """git shortlog -sn — how many names stand on the commits?

    Only COUNTS leave this function. An earlier version also returned the names themselves,
    which no caller ever printed — the page needs to know how many names there are, not who
    they are, and this file is published.
    """
    # `-e` appends the e-mail — only that makes the name unambiguously delimitable (it may
    # contain spaces; the address always follows in angle brackets).
    raw = _git("shortlog", "-sn", "-e", rev)
    counts: Counter = Counter()
    for line in raw.splitlines():
        m = re.match(r"\s*(\d+)\s+(.*?)\s*<", line)
        if m:
            counts[m.group(2)] += int(m.group(1))
    if not counts:
        raise MeasurementError("git shortlog returned no authors")
    _top, top_n = counts.most_common(1)[0]
    return {"top": top_n, "total": sum(counts.values()), "names": len(counts)}


def bugs(rev: str) -> dict:
    """Numbered entries from BACKLOG.md + archives, classified the way the repo does it.

    The status convention is imported rather than copied, so that one definition of "open"
    holds everywhere instead of each reader inventing its own. A tripwire counts as open: a
    tripwire is something still being watched.
    """
    sys.path.insert(0, str(ROOT / "tools"))
    try:
        import backlog_audit
    except Exception as exc:  # pragma: no cover
        raise MeasurementError(f"backlog_audit not importable: {exc}") from exc

    files: list[str] = []
    for pattern in BACKLOG_GLOBS:
        prefix, _, tail = pattern.rpartition("/")
        files += _paths(
            rev, prefix or ".", "^" + re.escape(prefix + "/") + tail.replace("*", "[^/]*") + "$"
        )
    if not files:
        raise MeasurementError(f"no backlog file in the commit: {BACKLOG_GLOBS}")

    states: Counter = Counter()
    seen: set[str] = set()
    for path in files:
        entries, order, _dupes, _wrong = backlog_audit.parse_entries(_blob(rev, path))
        for key in order:
            if key in seen:  # same entry number in active AND archive: count it once
                continue
            seen.add(key)
            # eigene Nummer mitgeben: ohne sie blendet classify keine Fremd-Marken aus (BG-1513)
            state, _has_field = backlog_audit.classify(entries[key], key)
            states["OPEN" if state == "TRIPWIRE" else state] += 1

    total = sum(states.values())
    if total <= 0:
        raise MeasurementError("backlog holds no numbered entries")
    return {
        "total": total,
        "closed": states.get("DONE", 0),
        "partial": states.get("PARTIAL", 0),
        "open": states.get("OPEN", 0),
        "files": len(files),
    }


def tests(rev: str) -> dict:
    """ls tools/tests/test_*.py | wc -l  AND  grep -h '^def test_' ... | wc -l

    Two numbers, because they are two different things (rule 3): the files are the modules
    in the merge gate, the functions are the checks inside them.
    """
    files = _paths(rev, "tools/tests", r"/test_[^/]*\.py$")
    if not files:
        raise MeasurementError("no test modules found in the commit")
    n_funcs = 0
    for path in files:
        n_funcs += len(re.findall(r"^[ \t]*def test_", _blob(rev, path), re.MULTILINE))
    if n_funcs <= 0:
        raise MeasurementError("test modules without a single test function")
    return {"files": len(files), "functions": n_funcs}


def migrations(rev: str) -> int:
    """ls code/supabase/migrations/*.sql | wc -l"""
    return len(_paths(rev, "code/supabase/migrations", r"\.sql$"))


def curve(days: list[date], start: date, end: date) -> list[int]:
    """Cumulative commits per CALENDAR DAY from start to end, without gaps.

    Required for the page: the graph has to come from the same measurement as the tiles.
    Maintained by hand it ages — and an outdated graph is worse on an evidence page than
    no graph at all.
    """
    per_day = Counter(days)
    series, running, cursor = [], 0, start
    while cursor <= end:
        running += per_day.get(cursor, 0)
        series.append(running)
        cursor += timedelta(days=1)
    return series


# ------------------------------------------------------------------ aggregate


def build_label(rev: str) -> dict:
    """version/code + version/name from code/export_presets.cfg.

    The page once carried this number typed in by hand. The export presets are versioned and
    therefore evidence; anything read from outside the commit is not.

    Both presets must carry the same number; if they drift apart, an export stamped only one
    platform and the page would claim a state that never existed.
    """
    blob = _blob(rev, "code/export_presets.cfg")
    codes = re.findall(r"^version/code=(\d+)", blob, re.M)
    names = re.findall(r'^version/name="([^"]+)"', blob, re.M)
    if not codes or not names:
        raise MeasurementError("export_presets.cfg carries no version/code + version/name")
    if len(set(codes)) != 1:
        raise MeasurementError(f"version/code drifts between presets: {codes}")
    if len(set(names)) != 1:
        raise MeasurementError(f"version/name drifts between presets: {names}")
    return {"code": int(codes[0]), "name": names[0], "presets": len(codes)}


def month_axis(start: date, end: date) -> list[dict]:
    """Curve index AND label of the first of each month — one list, so the tick marks drawn
    on the canvas and the words written under it cannot say different things.

    They did: the marks came from here and grew with the history, while the labels were six
    hand-typed spans in the template. At the seventh month the chart drew a mark September
    had no name for (BG-828). Same failure the draft had — see month_starts below — one step
    further down the page.

    Index 0 is the first commit day itself, so the axis has a left mark.
    """
    out = [{"idx": 0, "label": start.strftime("%b")}]
    y, m = start.year, start.month
    while True:
        m += 1
        if m > 12:
            y, m = y + 1, 1
        d = date(y, m, 1)
        if d > end:
            break
        out.append({"idx": (d - start).days, "label": d.strftime("%b")})
    return out


def month_starts(start: date, end: date) -> list[int]:
    """Curve indices of the first of each month. The draft carried [0,11,41,72,102,133] as a
    literal list in the script — one that goes quietly wrong at the first month boundary
    after the build. Derived from month_axis so marks and labels share one source.
    """
    return [p["idx"] for p in month_axis(start, end)]


def collect(rev_arg: str = "HEAD") -> dict:
    """Measure every figure against ONE commit. Raises on any failed measurement."""
    rev = _git("rev-parse", rev_arg)  # rule 2: resolve once, then hold on to it
    short = _git("rev-parse", "--short", rev)

    days_list = commit_days(rev)
    start, head_day = days_list[0], days_list[-1]
    span = (head_day - start).days
    if span <= 0:
        raise MeasurementError(f"implausible span: {start} .. {head_day}")

    n_commits = len(days_list)
    series = curve(days_list, start, head_day)
    deltas = [b - a for a, b in zip(series, series[1:])]
    peak_i = max(range(len(deltas)), key=lambda i: deltas[i]) if deltas else 0

    return {
        # time
        "days": span,
        "alpha_days": (ALPHA_DATE - start).days,
        "alpha_date": ALPHA_DATE.isoformat(),
        "alpha_day_commits": sum(1 for d in days_list if d == ALPHA_DATE),
        "first_commit": start.isoformat(),
        # evidence
        "commits": n_commits,
        "co_author_commits": co_author_commits(rev),
        "nonmerge": nonmerge_co_author(rev),
        "co_author_by_month": co_author_by_month(rev),
        "authors": authors(rev),
        "bugs": bugs(rev),
        "tests": tests(rev),
        "migrations": migrations(rev),
        "build": build_label(rev),
        "people": 1,
        # operations
        "commits_per_day": round(n_commits / span, 1),
        "peak_commits": max(deltas) if deltas else 0,
        "peak_date": (start + timedelta(days=peak_i + 1)).isoformat(),
        "silent_days": sum(1 for d in deltas if d == 0),
        "curve": series,
        "month_axis": month_axis(start, head_day),
        "month_starts": month_starts(start, head_day),
        "months_span": len(month_starts(start, head_day)) - 1,
        # provenance
        "source_sha": short,
        "measured_on": head_day.isoformat(),
    }


def thousands(n: int) -> str:
    """7274 -> '7,274'. The page is English, so a comma."""
    return f"{n:,}"


def main() -> None:
    parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
    parser.add_argument("--json", action="store_true", help="machine-readable, for web/build.py")
    parser.add_argument("--rev", default="HEAD", help="commit to measure against")
    args = parser.parse_args()

    data = collect(args.rev)

    if args.json:
        print(json.dumps(data, indent=2, sort_keys=True))
        return

    b, t, a = data["bugs"], data["tests"], data["authors"]
    rows = [
        (thousands(data["alpha_days"]), "days from the first commit to the closed alpha"),
        (thousands(data["days"]), "days from the first commit to this snapshot"),
        (thousands(data["commits"]), "commits"),
        (thousands(data["co_author_commits"]), "of those with Claude as co-author"),
        (
            thousands(b["total"]),
            f"bugs written down ({b['closed']} closed, {b['partial']} partly, {b['open']} open)",
        ),
        (thousands(t["functions"]), f"automated checks, in {t['files']} files"),
        (thousands(a["top"]), f"commits by one author name (of {a['total']})"),
        ("", ""),
        (f"{data['commits_per_day']} / day", "commits, sustained"),
        (thousands(data["peak_commits"]), f"commits on the busiest day ({data['peak_date']})"),
        (thousands(data["silent_days"]), "days with no commit"),
        (thousands(data["migrations"]), "database migrations"),
    ]
    width = max(len(value) for value, _ in rows)
    for value, label in rows:
        print(f"  {value:>{width}}  {label}" if value else "")
    print(f"\n  {len(data['curve'])} calendar days in the curve")
    print(f"  measured at {data['source_sha']} ({data['measured_on']})")


if __name__ == "__main__":
    main()
