#!/usr/bin/env python3
"""Generate/refresh per-task reports under reports/task-by-task/.

One report per task in results/results.json (87 of them). Each report is level 3 of the
drill-down: level 1 is the dashboard row on benchmark-history, level 2 is ui/heatmap.html,
level 3 is "what did we change on this task, what worked, what didn't, what can we learn".

The point of a script here is that the *numbers* in each report stay derived from
results.json rather than transcribed by hand (the bug class that produced the 63/87-vs-64/87
heatmap drift and summary.md's wrong 0.825 mean), while the *narrative* stays hand-written.

So each file has an auto-generated block:

    <!-- BEGIN:auto ... -->   ... numbers, provenance, links to material ...   <!-- END:auto -->

and hand-written sections below it. Re-running this script rewrites ONLY the auto block of
an existing file and never touches anything outside it. New files get the full skeleton with
`_Not yet analysed._` narrative placeholders.

Usage:
  python3 scripts/build_task_reports.py            # create missing, refresh auto blocks
  python3 scripts/build_task_reports.py --check     # exit 1 if anything would change
  python3 scripts/build_task_reports.py --stats     # coverage report, write nothing
"""
import json
import re
import sys
from pathlib import Path

ROOT = Path(__file__).resolve().parent.parent
RESULTS_JSON = ROOT / "results" / "results.json"
MANIFEST_JSON = ROOT / "artifacts" / "task-by-task" / "MANIFEST.json"
REPORTS_DIR = ROOT / "reports" / "task-by-task"
ARTIFACTS_DIR = ROOT / "artifacts" / "task-by-task"
PER_TASK_LOGS = ROOT / "results" / "task-by-task-43" / "per-task-logs"
EVIDENCE_DIR = ROOT / "evidence"

CAND_KEYS = ["seed", "cand_0001", "cand_0002", "cand_0003", "cand_0004"]

# The 10 NO_SIGNAL tasks split into infra failures vs genuine optimizer failures. Source of
# truth is results/task-by-task-87/summary.md's "NO_SIGNAL tasks (10)" section, transcribed
# here because the split cannot be derived from the data: it came from inspecting whether the
# agent recorded any tool_calls. Do NOT try to infer it from MANIFEST run_dir names — 3 of
# these 4 carry an INFRA_BROKEN marker but `seismic-phase-picking`'s run_dir says `DONE`.
NO_SIGNAL_INFRA = {
    "earthquake-phase-association",
    "fix-build-google-auto",
    "fix-visual-stability",
    "seismic-phase-picking",
}

AUTO_BEGIN = "<!-- BEGIN:auto — regenerated by scripts/build_task_reports.py; do not hand-edit inside this block -->"
AUTO_END = "<!-- END:auto -->"
PLACEHOLDER = "_Not yet analysed._"

AUTO_BLOCK_RE = re.compile(
    re.escape(AUTO_BEGIN) + r".*?" + re.escape(AUTO_END),
    re.DOTALL,
)


def fmt(v):
    return "—" if v is None else f"{v:.3f}"


def find_evidence(task: str) -> str | None:
    """Evidence dirs are named after the task, sometimes with a suffix describing the
    finding (e.g. shock-analysis-demand-optimizer-regression). Match longest-prefix."""
    if not EVIDENCE_DIR.is_dir():
        return None
    for d in sorted(EVIDENCE_DIR.iterdir(), key=lambda p: -len(p.name)):
        if d.is_dir() and (d.name == task or d.name.startswith(task + "-")):
            return d.name
    return None


def build_auto_block(t: dict, man: dict) -> str:
    task = t["task"]
    has_best = (ARTIFACTS_DIR / task / "best").is_dir()
    process_md = ARTIFACTS_DIR / task / "best" / "PROCESS.md"
    log_md = PER_TASK_LOGS / f"{task}.md"
    evidence = find_evidence(task)

    sub = t.get("subcategory")
    cat = t["category"] + (f" / {sub}" if sub else "")
    run_dir = man.get("run_dir", "—")
    iters = man.get("iterations")
    iters_s = "—" if iters is None else str(iters)

    ft = t.get("final_test")
    ft_s = "not run" if ft is None else fmt(ft)

    lines = [
        AUTO_BEGIN,
        "",
        f"**Status:** `{t['status']}` · **Category:** {cat} · **Source run:** `{man.get('source','—')}` (`{run_dir}`)",
        "",
        f"**Best:** `{t.get('best_tag','—')}` @ {fmt(t.get('best'))} · "
        f"**Δ vs seed:** {('+' if (t.get('delta') or 0) >= 0 else '')}{fmt(t.get('delta'))} · "
        f"**Held-out test:** {ft_s} · **Iterations:** {iters_s}",
        "",
        "| " + " | ".join(CAND_KEYS) + " | best |",
        "|" + "---|" * (len(CAND_KEYS) + 1),
        "| " + " | ".join(fmt(t.get(k)) for k in CAND_KEYS) + f" | **{fmt(t.get('best'))}** |",
        "",
        "_Numbers above are generated from `results/results.json`. Val rewards unless labelled test._",
        "",
    ]

    a0f = t.get("a0f")
    if a0f is not None:
        a0f_delta = t.get("a0f_delta") or 0
        lines += [
            f"**A0f (196-skill vocabulary):** {fmt(a0f)} · "
            f"**Δ vs A0 (seed):** {('+' if a0f_delta >= 0 else '')}{fmt(a0f_delta)} · "
            "10 trials · see "
            "[`results/a0f-full-vocab/summary.md`](../../results/a0f-full-vocab/summary.md)",
            "",
        ]

    lines += ["**Material:**"]

    mat = []
    if process_md.is_file():
        mat.append(
            f"- [`best/PROCESS.md`](../../artifacts/task-by-task/{task}/best/PROCESS.md) "
            "— the optimizer's own write-up of the winning iteration: per-trial ground truth, "
            "ranked failure clusters with root causes, the kept edit, and what it deliberately skipped"
        )
    if has_best:
        mat.append(
            f"- [`best/`](../../artifacts/task-by-task/{task}/best/) vs "
            f"[`seed/`](../../artifacts/task-by-task/{task}/seed/) — diff these two trees to see exactly what changed"
        )
    else:
        mat.append(
            f"- [`seed/`](../../artifacts/task-by-task/{task}/seed/) — seed skill package "
            "(no `best/`: no candidate was ever accepted for this task)"
        )
    if log_md.is_file():
        mat.append(
            f"- [`per-task-logs/{task}.md`](../../results/task-by-task-43/per-task-logs/{task}.md) "
            "— per-trial reward vectors, from the earlier 43-task sweep (numbers may differ from "
            "above: different run)"
        )
    if evidence:
        mat.append(
            f"- [`evidence/{evidence}/`](../../evidence/{evidence}/) — full evidence bundle "
            "(journal, diffs, run reports)"
        )

    lines += mat
    lines += ["", AUTO_END]
    return "\n".join(lines)


def build_skeleton(t: dict, man: dict) -> str:
    task = t["task"]
    auto = build_auto_block(t, man)

    seed = t.get("seed")
    best = t.get("best")
    status = t["status"]
    saturated = seed is not None and seed >= 0.999
    no_signal = status == "NO_SIGNAL"
    moved = seed is not None and best is not None and best > seed + 1e-6

    # Three cases can be answered fully from the data alone, so their reports ship complete
    # rather than as placeholders. Everything else needs a human to read PROCESS.md.
    if saturated:
        headline = "already solved at seed — no room to optimize"
        changed = (
            "Nothing. The seed skill already scored 1.0 on val, so the run was stopped "
            "(`KILLED_saturated`) rather than spending budget on a task with no headroom."
        )
        worked = "N/A — the seed capability was already sufficient."
        didnt = "N/A — no candidate was attempted."
        learn = (
            "**Nothing about the optimizer.** This task tells us the seed skill was already "
            "adequate; it is a ceiling artifact, not a result. Tasks like this inflate a "
            "pass-rate headline while carrying no optimizer signal — which is why the "
            "saturated set is called out separately in "
            "[`results/task-by-task-87/summary.md`](../../results/task-by-task-87/summary.md)."
        )
    elif no_signal and task in NO_SIGNAL_INFRA:
        headline = "no signal — infrastructure failure, not an optimizer result"
        changed = (
            "Nothing, and nothing could have. The container built but the agent recorded "
            "**empty `tool_calls`** — the harness ran, the agent never did."
        )
        worked = "N/A — no agent activity was recorded, so nothing was exercised."
        didnt = (
            "The task never reached the optimizer. Whatever the seed skill's quality, it was "
            "never tested here."
        )
        learn = (
            "**Nothing about cap-evolve or the skill.** This is a harness/environment failure "
            "and its 0.0 is an absence of measurement, not a measurement of failure. Counting "
            "it as an optimizer failure would understate results; silently dropping it from "
            "the denominator to reach a nicer pass rate would overstate them — which is why "
            "[`summary.md`](../../results/task-by-task-87/summary.md) refuses to quote the "
            "no-signal-excluded cut against EvoSkill's number."
            + (
                " `fix-visual-stability` specifically is a benchmark-side broken "
                "`docker-compose.yml`."
                if task == "fix-visual-stability"
                else ""
            )
        )
    elif no_signal:
        headline = "no signal — ran to completion, never moved off zero"
        changed = PLACEHOLDER
        worked = "Nothing measurable: every candidate scored 0.0, same as seed."
        didnt = PLACEHOLDER
        learn = (
            "This one is a **genuine optimizer failure**, not an infra artifact: cap-evolve ran "
            "to completion and no candidate moved the reward off zero. That makes it real "
            "signal and worth analysing — a task where the optimizer had no lever it could "
            "find is more informative than one it solved. Read "
            f"[`best/PROCESS.md`](../../artifacts/task-by-task/{task}/best/PROCESS.md) for what "
            "it tried, if present."
            if (ARTIFACTS_DIR / task / "best" / "PROCESS.md").is_file()
            else "This one is a **genuine optimizer failure**, not an infra artifact: cap-evolve "
            "ran to completion and no candidate moved the reward off zero. That makes it real "
            "signal and worth analysing — a task where the optimizer had no lever it could find "
            "is more informative than one it solved. No `PROCESS.md` was produced (no candidate "
            "was ever accepted), so reconstructing what it tried needs the original run store."
        )
    else:
        headline = PLACEHOLDER.strip("_")
        changed = PLACEHOLDER
        worked = PLACEHOLDER
        didnt = PLACEHOLDER
        learn = PLACEHOLDER

    title = f"# {task}" + (f" — {headline}" if headline != PLACEHOLDER.strip("_") else "")

    body = f"""{title}

{auto}

## What changed

{changed}

## What worked

{worked}

## What didn't

{didnt}

_A rejected candidate is not a regression: it reverts to the champion, so a low score on a
later candidate does not undo an earlier accepted fix._

## What we can (or can't) learn

{learn}
"""
    return body


def main() -> int:
    check_only = "--check" in sys.argv
    stats_only = "--stats" in sys.argv

    results = json.loads(RESULTS_JSON.read_text())
    tasks = results["tasks"]
    manifest = {m["task"]: m for m in json.loads(MANIFEST_JSON.read_text())}

    if stats_only:
        n_best = sum(1 for t in tasks if (ARTIFACTS_DIR / t["task"] / "best").is_dir())
        n_proc = sum(1 for t in tasks if (ARTIFACTS_DIR / t["task"] / "best" / "PROCESS.md").is_file())
        n_log = sum(1 for t in tasks if (PER_TASK_LOGS / f"{t['task']}.md").is_file())
        n_ev = sum(1 for t in tasks if find_evidence(t["task"]))
        n_sat = sum(1 for t in tasks if (t.get("seed") or 0) >= 0.999)
        n_ns = sum(1 for t in tasks if t["status"] == "NO_SIGNAL")
        written = sorted(p.stem for p in REPORTS_DIR.glob("*.md")) if REPORTS_DIR.is_dir() else []
        analysed = [
            p.stem for p in (REPORTS_DIR.glob("*.md") if REPORTS_DIR.is_dir() else [])
            if PLACEHOLDER not in p.read_text()
        ]
        print(f"tasks:                    {len(tasks)}")
        print(f"  with best/:             {n_best}")
        print(f"  with best/PROCESS.md:   {n_proc}")
        print(f"  with per-task-log:      {n_log}")
        print(f"  with evidence bundle:   {n_ev}")
        print(f"  saturated at seed:      {n_sat}")
        print(f"  NO_SIGNAL:              {n_ns}")
        print(f"reports written:          {len(written)}")
        print(f"  fully analysed:         {len(analysed)}")
        return 0

    REPORTS_DIR.mkdir(parents=True, exist_ok=True)

    created, refreshed, unchanged = [], [], []
    for t in tasks:
        task = t["task"]
        man = manifest.get(task, {})
        path = REPORTS_DIR / f"{task}.md"

        if not path.is_file():
            if check_only:
                created.append(task)
                continue
            path.write_text(build_skeleton(t, man))
            created.append(task)
            continue

        # Existing file: rewrite ONLY the auto block, preserve all hand-written prose.
        old = path.read_text()
        new_auto = build_auto_block(t, man)
        if AUTO_BEGIN not in old or AUTO_END not in old:
            print(f"WARNING: {path.name} has no auto block — left untouched", file=sys.stderr)
            unchanged.append(task)
            continue
        new = AUTO_BLOCK_RE.sub(lambda _: new_auto, old, count=1)
        if new == old:
            unchanged.append(task)
        elif check_only:
            refreshed.append(task)
        else:
            path.write_text(new)
            refreshed.append(task)

    if check_only:
        if created or refreshed:
            print(
                f"STALE: {len(created)} report(s) missing, {len(refreshed)} auto block(s) out of date.",
                file=sys.stderr,
            )
            for t in created:
                print(f"  missing: {t}", file=sys.stderr)
            for t in refreshed:
                print(f"  stale:   {t}", file=sys.stderr)
            return 1
        print(f"All {len(unchanged)} reports up to date.")
        return 0

    print(f"created {len(created)}, refreshed {len(refreshed)}, unchanged {len(unchanged)}")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
