"""TechFuelHQ coding-eval v1 — repeatable local-LLM coding tasks on THIS machine.

Eight auto-scored tasks (bugfix, spec-implement, test-writing, refactor,
explain-legacy, SQL, regex, agentic tool-calling), run against local models via
ollama's /api/chat. Every task is pass/fail by MECHANICAL check — executed
code, exact row comparison, fullmatch lists, tool-call JSON inspection. No
judge model anywhere.

Method pins (mirrors llm-throughput-ollama-v1 discipline):
  - temperature 0, seed 42, num_ctx 8192, num_predict 4096, 2 reps/task
  - model identity pinned by digest from /api/tags
  - generated code runs in a subprocess with a hard timeout, in a temp dir
  - per-request decode tok/s recorded from ollama's own eval counters
  - GPU/CPU layer split recorded from /api/ps after each model's first call
  - raw per-request JSON preserved under ops/audit/evidence/<date>/coding-eval/

Usage:
  python run_coding_eval.py --models qwen2.5-coder:7b gpt-oss:20b
  python run_coding_eval.py --export-tasks   # writes static/data/coding-eval-tasks-v1.json
"""

from __future__ import annotations

import argparse
import json
import re
import sqlite3
import statistics
import subprocess
import sys
import tempfile
import urllib.request
from datetime import datetime, timezone
from pathlib import Path
from typing import Any

OLLAMA = "http://127.0.0.1:11434"
EVAL_VERSION = "techfuelhq-coding-eval-v1"
_HERE = Path(__file__).resolve()
# In the site repository this file sits at ops/bench/coding-eval/. A standalone
# copy (the one published at /data/coding-eval-v1/) writes under the working directory.
REPO = (_HERE.parents[3]
        if len(_HERE.parents) > 3 and (_HERE.parents[3] / "hugo.toml").is_file()
        else Path.cwd())
PYTHON = sys.executable
REQ_TIMEOUT = 900
SUBPROC_TIMEOUT = 30
OPTIONS = {"temperature": 0, "seed": 42, "num_ctx": 8192, "num_predict": 4096}
REPS = 2

# ---------------------------------------------------------------- fixtures

T1_BUGGY = '''def find_first_ge(arr, target):
    """Return the index of the FIRST element in sorted list arr that is
    greater than or equal to target, or len(arr) if no such element exists."""
    lo, hi = 0, len(arr)
    while lo < hi:
        mid = (lo + hi) // 2
        if arr[mid] <= target:
            lo = mid + 1
        else:
            hi = mid
    return lo
'''

T1_CHECKER = '''import solution
cases = [
    (([1, 2, 2, 3], 2), 1),
    (([1, 2, 2, 3], 4), 4),
    (([1, 3, 5], 0), 0),
    (([], 7), 0),
    (([5], 5), 0),
    (([1, 1, 1], 1), 0),
    (([2, 4, 6, 8], 5), 2),
    (([2, 4, 6, 8], 8), 3),
]
for (arr, target), want in cases:
    got = solution.find_first_ge(arr, target)
    assert got == want, f"find_first_ge({arr}, {target}) = {got}, want {want}"
print("OK")
'''

T2_CHECKER = '''import solution
ok = [("2h45m", 9900), ("90s", 90), ("1h", 3600), ("1h30m15s", 5415),
      ("0s", 0), ("59m", 3540), ("10h0m0s", 36000)]
for text, want in ok:
    got = solution.parse_duration(text)
    assert got == want, f"parse_duration({text!r}) = {got}, want {want}"
bad = ["", "45", "m30", "30x", "15s1h", "1x30m", "h", "1.5h", "-10s", "1h1h"]
for text in bad:
    try:
        solution.parse_duration(text)
    except ValueError:
        continue
    raise AssertionError(f"parse_duration({text!r}) should raise ValueError")
print("OK")
'''

T3_GOOD_IMPL = '''class LRUCache:
    def __init__(self, capacity):
        self.capacity = capacity
        self._data = {}

    def get(self, key):
        if key not in self._data:
            return -1
        value = self._data.pop(key)
        self._data[key] = value
        return value

    def put(self, key, value):
        if self.capacity <= 0:
            return
        if key in self._data:
            self._data.pop(key)
        elif len(self._data) >= self.capacity:
            oldest = next(iter(self._data))
            self._data.pop(oldest)
        self._data[key] = value
'''

# Mutant: evicts the MOST recently used entry instead of the least.
T3_MUTANT_IMPL = T3_GOOD_IMPL.replace(
    "oldest = next(iter(self._data))",
    "oldest = next(reversed(self._data))")

T3_DRIVER = '''import solution
from impl import LRUCache
solution.check(LRUCache)
print("OK")
'''

T4_ORIGINAL = '''def summarize(readings_a, readings_b, readings_c):
    """Return a dict of per-sensor stats for three reading lists."""
    result = {}
    total_a = 0
    for value in readings_a:
        total_a += value
    if len(readings_a) > 0:
        mean_a = total_a / len(readings_a)
        peak_a = max(readings_a)
    else:
        mean_a = 0.0
        peak_a = None
    result["sensor_a"] = {"mean": round(mean_a, 2), "peak": peak_a,
                          "count": len(readings_a)}
    total_b = 0
    for value in readings_b:
        total_b += value
    if len(readings_b) > 0:
        mean_b = total_b / len(readings_b)
        peak_b = max(readings_b)
    else:
        mean_b = 0.0
        peak_b = None
    result["sensor_b"] = {"mean": round(mean_b, 2), "peak": peak_b,
                          "count": len(readings_b)}
    total_c = 0
    for value in readings_c:
        total_c += value
    if len(readings_c) > 0:
        mean_c = total_c / len(readings_c)
        peak_c = max(readings_c)
    else:
        mean_c = 0.0
        peak_c = None
    result["sensor_c"] = {"mean": round(mean_c, 2), "peak": peak_c,
                          "count": len(readings_c)}
    return result
'''

T4_CHECKER = '''import re
import solution
source = open("solution.py", encoding="utf-8").read()
defs = re.findall(r"^\\s*def \\w+", source, flags=re.M)
assert len(defs) >= 2, "refactor must extract a helper function (found %d defs)" % len(defs)
cases = [
    ([1, 2, 3], [10.5, 20.5], []),
    ([], [], []),
    ([5], [5], [5]),
    ([0, 0, 7], [1.1, 2.2, 3.3, 4.4], [100]),
    ([-5, 5], [0.0], [2, 4, 6, 8]),
]
def reference(a, b, c):
    out = {}
    for name, readings in (("sensor_a", a), ("sensor_b", b), ("sensor_c", c)):
        if readings:
            out[name] = {"mean": round(sum(readings) / len(readings), 2),
                         "peak": max(readings), "count": len(readings)}
        else:
            out[name] = {"mean": 0.0, "peak": None, "count": 0}
    return out
for a, b, c in cases:
    got = solution.summarize(list(a), list(b), list(c))
    want = reference(a, b, c)
    assert got == want, f"summarize({a}, {b}, {c}) = {got}, want {want}"
print("OK")
'''

T5_CODE = '''def f(s):
    d = [int(c) for c in s if c.isdigit()][::-1]
    t = 0
    for i, x in enumerate(d):
        if i % 2:
            x *= 2
            if x > 9:
                x -= 9
        t += x
    return t % 10 == 0
'''

T6_SCHEMA = '''CREATE TABLE customers (id INTEGER PRIMARY KEY, name TEXT, country TEXT);
CREATE TABLE orders (id INTEGER PRIMARY KEY, customer_id INTEGER REFERENCES customers(id),
                     amount_cents INTEGER, created_at TEXT);'''

T6_SEED = [
    "INSERT INTO customers VALUES (1,'Ada','DE'),(2,'Ben','US'),(3,'Cho','US'),"
    "(4,'Dee','JP'),(5,'Eli','DE'),(6,'Fay','BR')",
    "INSERT INTO orders VALUES "
    "(1,1,30000,'2026-07-03'),(2,1,25000,'2026-07-28'),"
    "(3,2,45000,'2026-07-15'),(4,3,5000,'2026-07-31'),"
    "(5,4,50000,'2026-07-10'),"
    "(6,5,80000,'2026-06-30'),"
    "(7,2,120000,'2026-08-01'),"
    "(8,6,49999,'2026-07-20')",
]
T6_EXPECTED = [("DE", 55000), ("US", 50000), ("JP", 50000)]

T7_SHOULD_MATCH = ["1.0.0", "0.1.0", "10.20.30", "1.0.0-alpha", "1.0.0-alpha.1",
                   "2.0.0-rc.1", "1.2.3-0valid"]
T7_SHOULD_NOT = ["1.0", "01.0.0", "1.0.0.", "1.0.0-", "a.b.c", "v1.0.0",
                 "1.0.0-alpha_beta", "", "1.0.0 ", "1..0"]

T8_TOOLS = [
    {"type": "function", "function": {
        "name": "create_directory",
        "description": "Create a directory (and any missing parents) at the given path.",
        "parameters": {"type": "object",
                       "properties": {"path": {"type": "string",
                                               "description": "Directory path to create"}},
                       "required": ["path"]}}},
    {"type": "function", "function": {
        "name": "write_file",
        "description": "Write text content to a file, overwriting if it exists.",
        "parameters": {"type": "object",
                       "properties": {"path": {"type": "string"},
                                      "content": {"type": "string"}},
                       "required": ["path", "content"]}}},
]

# ---------------------------------------------------------------- tasks

TASKS: list[dict[str, Any]] = [
    {
        "id": "t1-bugfix-binary-search",
        "family": "bugfix",
        "prompt": (
            "The following Python function is supposed to do what its docstring says, "
            "but it has a bug.\n\n```python\n" + T1_BUGGY + "```\n\n"
            "Fix the bug. Reply with the complete corrected function in a single "
            "```python code block. Keep the same function name and signature."),
    },
    {
        "id": "t2-implement-parse-duration",
        "family": "implement-from-spec",
        "prompt": (
            "Write a Python function `parse_duration(text)` that converts a duration "
            "string into total seconds (int).\n\nSpec:\n"
            "- Components: hours `h`, minutes `m`, seconds `s`, each an integer with no sign.\n"
            "- Any subset may appear, but each at most once, and they must appear in h, m, s order. "
            "Examples: `\"2h45m\"` -> 9900, `\"90s\"` -> 90, `\"1h30m15s\"` -> 5415.\n"
            "- At least one component is required.\n"
            "- For anything else (empty string, missing unit, unknown unit, wrong order, "
            "decimals, negatives), raise `ValueError`.\n\n"
            "Reply with the complete function in a single ```python code block."),
    },
    {
        "id": "t3-write-tests-lru",
        "family": "test-writing",
        "prompt": (
            "Here is an LRU cache implementation:\n\n```python\n" + T3_GOOD_IMPL + "```\n\n"
            "Write a thorough test for it as a single Python function "
            "`check(cache_class)` that instantiates `cache_class` and raises "
            "`AssertionError` if the implementation is wrong. Cover eviction order, "
            "the recency effect of `get`, updating an existing key, and the missing-key "
            "return value. A weak test that only checks basic put/get will not count.\n\n"
            "Reply with only the `check` function (plus imports if needed) in a single "
            "```python code block. Do not call `check` at module level."),
    },
    {
        "id": "t4-refactor-summarize",
        "family": "refactor",
        "prompt": (
            "Refactor this Python function to remove the copy-paste duplication by "
            "extracting the repeated per-sensor logic into ONE helper function. "
            "The public function `summarize(readings_a, readings_b, readings_c)` must "
            "keep exactly the same name, signature, and return value for every input.\n\n"
            "```python\n" + T4_ORIGINAL + "```\n\n"
            "Reply with the complete refactored code (helper + `summarize`) in a single "
            "```python code block."),
    },
    {
        "id": "t5-explain-legacy",
        "family": "explain-legacy",
        "prompt": (
            "What algorithm does this function implement, and what is it commonly used "
            "for? Answer in 2-3 sentences.\n\n```python\n" + T5_CODE + "```"),
    },
    {
        "id": "t6-sql-revenue",
        "family": "sql",
        "prompt": (
            "Given this SQLite schema:\n\n```sql\n" + T6_SCHEMA + "\n```\n\n"
            "Write ONE SQL query that returns total order revenue per customer country "
            "for orders created in July 2026 (created_at is an ISO `YYYY-MM-DD` text "
            "column), only including countries whose July total is at least 50000 cents, "
            "sorted by total descending (ties: any order). Return columns: country, "
            "total_cents.\n\nReply with only the query in a single ```sql code block."),
    },
    {
        "id": "t7-regex-semver",
        "family": "regex",
        "prompt": (
            "Write a single Python `re` pattern that matches semantic version strings of "
            "the form MAJOR.MINOR.PATCH with an optional -prerelease suffix:\n"
            "- MAJOR, MINOR, PATCH: non-negative integers, no leading zeros (a lone `0` is fine)\n"
            "- optional prerelease: `-` followed by one or more dot-separated identifiers "
            "of letters, digits, or hyphens (at least one character each)\n"
            "- no `v` prefix, no build metadata, nothing else before or after\n"
            "The pattern will be used with `re.fullmatch`.\n\n"
            "Reply with ONLY the regex pattern inside a single fenced code block."),
    },
    {
        "id": "t8-agentic-tool-calls",
        "family": "tool-calling",
        "prompt": (
            "Create a directory at `reports/2026`, then write a file at "
            "`reports/2026/status.txt` whose content is exactly `OK`. "
            "Use the available tools."),
        "tools": T8_TOOLS,
    },
]

# ---------------------------------------------------------------- plumbing


def utc_now() -> str:
    return datetime.now(timezone.utc).isoformat()


def api(path: str, payload: dict[str, Any] | None = None,
        timeout: int = REQ_TIMEOUT) -> dict[str, Any]:
    req = urllib.request.Request(
        OLLAMA + path,
        data=None if payload is None else json.dumps(payload).encode("utf-8"),
        headers={"Content-Type": "application/json"},
        method="GET" if payload is None else "POST",
    )
    with urllib.request.urlopen(req, timeout=timeout) as response:
        return json.load(response)


def chat(model: str, prompt: str, tools: list | None = None,
         messages: list | None = None) -> dict[str, Any]:
    msgs = messages if messages is not None else [{"role": "user", "content": prompt}]
    payload: dict[str, Any] = {"model": model, "messages": msgs,
                               "stream": False, "options": dict(OPTIONS)}
    if tools:
        payload["tools"] = tools
    # One retry on server-side 5xx: an ollama transient is not model behavior.
    # The retry is visible in the raw log timing; a second 5xx propagates.
    import time as _time
    import urllib.error
    try:
        return api("/api/chat", payload)
    except urllib.error.HTTPError as exc:
        if exc.code >= 500:
            _time.sleep(10)
            return api("/api/chat", payload)
        raise


def extract_block(text: str, lang: str | None) -> str | None:
    """Last fenced block preferring the requested language tag."""
    blocks = re.findall(r"```([A-Za-z0-9_+-]*)[ \t]*\r?\n(.*?)```", text, flags=re.S)
    if not blocks:
        return None
    if lang:
        tagged = [body for tag, body in blocks if tag.lower() == lang]
        if tagged:
            return tagged[-1].strip()
    return blocks[-1][1].strip()


def run_python(workdir: Path, main_file: str) -> tuple[bool, str]:
    try:
        # -E (not -I): PYTHON* environment variables are ignored, and the
        # script directory stays on sys.path so `import solution` resolves.
        proc = subprocess.run(
            [PYTHON, "-E", main_file], cwd=workdir, capture_output=True,
            text=True, timeout=SUBPROC_TIMEOUT, encoding="utf-8", errors="replace")
    except subprocess.TimeoutExpired:
        return False, "TIMEOUT"
    out = (proc.stdout or "") + (proc.stderr or "")
    return proc.returncode == 0 and "OK" in (proc.stdout or ""), out[-2000:]


def score_python_task(response: str, checker: str,
                      extra_files: dict[str, str] | None = None) -> tuple[bool, str]:
    code = extract_block(response, "python")
    if not code:
        return False, "no python code block found"
    with tempfile.TemporaryDirectory() as tmp:
        workdir = Path(tmp)
        (workdir / "solution.py").write_text(code, encoding="utf-8")
        for name, content in (extra_files or {}).items():
            (workdir / name).write_text(content, encoding="utf-8")
        (workdir / "checker.py").write_text(checker, encoding="utf-8")
        return run_python(workdir, "checker.py")


def score_t3(response: str) -> tuple[bool, str]:
    code = extract_block(response, "python")
    if not code:
        return False, "no python code block found"
    results = {}
    for label, impl in (("good", T3_GOOD_IMPL), ("mutant", T3_MUTANT_IMPL)):
        with tempfile.TemporaryDirectory() as tmp:
            workdir = Path(tmp)
            (workdir / "solution.py").write_text(code, encoding="utf-8")
            (workdir / "impl.py").write_text(impl, encoding="utf-8")
            (workdir / "driver.py").write_text(T3_DRIVER, encoding="utf-8")
            ok, detail = run_python(workdir, "driver.py")
            results[label] = (ok, detail)
    good_ok = results["good"][0]
    mutant_caught = not results["mutant"][0]
    detail = f"good_impl_passes={good_ok} mutant_caught={mutant_caught}"
    return good_ok and mutant_caught, detail


def score_t5(response: str) -> tuple[bool, str]:
    low = response.lower()
    has_luhn = "luhn" in low
    has_context = any(term in low for term in
                      ("card", "checksum", "check digit", "mod 10", "mod-10"))
    return has_luhn and has_context, f"luhn={has_luhn} context={has_context}"


def score_t6(response: str) -> tuple[bool, str]:
    sql = extract_block(response, "sql")
    if not sql:
        return False, "no sql code block found"
    if re.search(r"\b(insert|update|delete|drop|alter|attach|pragma)\b", sql, re.I):
        return False, "non-select statement refused"
    conn = sqlite3.connect(":memory:")
    try:
        conn.executescript(T6_SCHEMA)
        for stmt in T6_SEED:
            conn.execute(stmt)
        try:
            rows = conn.execute(sql).fetchall()
        except sqlite3.Error as exc:
            return False, f"sqlite error: {exc}"
    finally:
        conn.close()
    got = [(str(r[0]), int(r[1])) for r in rows]
    want = [(c, t) for c, t in T6_EXPECTED]
    if sorted(got) != sorted(want):
        return False, f"rows {got} != expected {want}"
    totals = [t for _, t in got]
    if totals != sorted(totals, reverse=True):
        return False, f"rows not sorted desc: {got}"
    return True, f"rows={got}"


def score_t7(response: str) -> tuple[bool, str]:
    pattern = extract_block(response, None)
    if pattern is None:
        return False, "no code block found"
    pattern = pattern.strip().strip("`").strip()
    if pattern.lower().startswith(("r'", 'r"')):
        pattern = pattern[2:-1]
    elif pattern.startswith(("'", '"')) and pattern.endswith(("'", '"')):
        pattern = pattern[1:-1]
    try:
        compiled = re.compile(pattern)
    except re.error as exc:
        return False, f"invalid regex: {exc}"
    misses = [s for s in T7_SHOULD_MATCH if not compiled.fullmatch(s)]
    false_hits = [s for s in T7_SHOULD_NOT if compiled.fullmatch(s)]
    ok = not misses and not false_hits
    return ok, f"missed={misses} false_hits={false_hits}"


def _norm_path(value: Any) -> str:
    return str(value).replace("\\", "/").strip("./ ")


def _calls(message: dict[str, Any]) -> list[dict[str, Any]]:
    out = []
    for call in message.get("tool_calls") or []:
        fn = call.get("function") or {}
        args = fn.get("arguments")
        if isinstance(args, str):
            try:
                args = json.loads(args)
            except (ValueError, TypeError):
                args = {}
        out.append({"name": fn.get("name"), "arguments": args or {}})
    return out


def run_t8(model: str, task: dict[str, Any]) -> tuple[bool, str, dict[str, Any]]:
    """Two-turn agentic exchange; returns (pass, detail, telemetry)."""
    messages = [{"role": "user", "content": task["prompt"]}]
    try:
        first = chat(model, "", tools=task["tools"], messages=messages)
    except Exception as exc:
        detail = str(exc)
        if "400" in detail or "does not support tools" in detail.lower():
            # A runner-level refusal is a real, publishable result for this
            # task (the stock template cannot express tools) — not a bench error.
            return False, f"runner refuses tools for this model ({detail[:120]})", {}
        raise
    telemetry = {"eval_count": first.get("eval_count"),
                 "eval_duration": first.get("eval_duration")}
    msg1 = first.get("message") or {}
    calls1 = _calls(msg1)
    if not calls1:
        return False, "no structured tool_calls emitted (prose only)", telemetry
    made_dir = any(c["name"] == "create_directory"
                   and "reports/2026" in _norm_path(c["arguments"].get("path", ""))
                   for c in calls1)
    wrote = any(c["name"] == "write_file"
                and "reports/2026/status.txt" in _norm_path(c["arguments"].get("path", ""))
                and str(c["arguments"].get("content", "")).strip() == "OK"
                for c in calls1)
    if made_dir and wrote:
        return True, f"both calls in one turn: {calls1}", telemetry
    if not made_dir:
        return False, f"first turn calls wrong: {calls1}", telemetry
    # Feed tool results back, expect write_file on turn 2.
    messages.append(msg1)
    for call in calls1:
        messages.append({"role": "tool", "content": json.dumps({"status": "success"}),
                         "tool_name": call["name"]})
    second = chat(model, "", tools=task["tools"], messages=messages)
    msg2 = second.get("message") or {}
    calls2 = _calls(msg2)
    wrote2 = any(c["name"] == "write_file"
                 and "reports/2026/status.txt" in _norm_path(c["arguments"].get("path", ""))
                 and str(c["arguments"].get("content", "")).strip() == "OK"
                 for c in calls2)
    detail = f"turn1={calls1} turn2={calls2}"
    return wrote2, detail, telemetry


SCORERS = {
    "t1-bugfix-binary-search": lambda r: score_python_task(r, T1_CHECKER),
    "t2-implement-parse-duration": lambda r: score_python_task(r, T2_CHECKER),
    "t3-write-tests-lru": score_t3,
    "t4-refactor-summarize": lambda r: score_python_task(r, T4_CHECKER),
    "t5-explain-legacy": score_t5,
    "t6-sql-revenue": score_t6,
    "t7-regex-semver": score_t7,
}

# ---------------------------------------------------------------- runner


def model_digest(model: str) -> str:
    for entry in (api("/api/tags").get("models") or []):
        if entry.get("name") == model or entry.get("model") == model:
            return str(entry.get("digest") or "")[:12]
    return ""


def gpu_split(model: str) -> dict[str, Any]:
    try:
        for proc in (api("/api/ps").get("models") or []):
            if proc.get("name") == model or proc.get("model") == model:
                size = proc.get("size") or 0
                vram = proc.get("size_vram") or 0
                pct = round(100.0 * vram / size, 1) if size else None
                return {"size_bytes": size, "size_vram_bytes": vram,
                        "gpu_resident_pct": pct}
    except Exception as exc:
        return {"error": str(exc)}
    return {}


def eval_model(model: str, out_dir: Path) -> dict[str, Any]:
    print(f"== {model}", flush=True)
    digest = model_digest(model)
    block: dict[str, Any] = {"model": model, "digest12": digest,
                             "started_at": utc_now(), "tasks": [], "split": None}
    for task in TASKS:
        for rep in range(REPS):
            started = utc_now()
            try:
                if task["id"] == "t8-agentic-tool-calls":
                    passed, detail, telemetry = run_t8(model, task)
                    response_text = detail
                else:
                    res = chat(model, task["prompt"])
                    msg = res.get("message") or {}
                    response_text = msg.get("content") or ""
                    telemetry = {"eval_count": res.get("eval_count"),
                                 "eval_duration": res.get("eval_duration")}
                    passed, detail = SCORERS[task["id"]](response_text)
            except Exception as exc:
                passed, detail, telemetry, response_text = False, f"ERROR: {exc}", {}, ""
            if block["split"] is None:
                block["split"] = gpu_split(model)
            ec, ed = telemetry.get("eval_count"), telemetry.get("eval_duration")
            decode = round(ec / (ed / 1e9), 1) if ec and ed else None
            row = {"task": task["id"], "rep": rep, "passed": passed,
                   "detail": detail[:500], "decode_tok_s": decode,
                   "output_tokens": ec, "started_at": started,
                   "hit_num_predict_cap": bool(ec) and ec >= OPTIONS["num_predict"]}
            block["tasks"].append(row)
            raw = out_dir / f"{model.replace(':', '_').replace('/', '_')}-{task['id']}-rep{rep}.json"
            raw.write_text(json.dumps({"row": row, "response": response_text},
                                      indent=1), encoding="utf-8")
            print(f"  {task['id']} rep{rep}: {'PASS' if passed else 'FAIL'}"
                  f" ({detail[:80]})", flush=True)
    decodes = [r["decode_tok_s"] for r in block["tasks"] if r["decode_tok_s"]]
    per_task: dict[str, list[bool]] = {}
    for row in block["tasks"]:
        per_task.setdefault(row["task"], []).append(row["passed"])
    block["summary"] = {
        "stable_passes": sum(1 for oks in per_task.values() if all(oks)),
        "flaky": sum(1 for oks in per_task.values() if any(oks) and not all(oks)),
        "tasks_total": len(per_task),
        "median_decode_tok_s": round(statistics.median(decodes), 1) if decodes else None,
        "per_task": {k: ["pass" if p else "fail" for p in v] for k, v in per_task.items()},
    }
    block["finished_at"] = utc_now()
    print(f"  -> {block['summary']['stable_passes']}/{len(per_task)} stable, "
          f"{block['summary']['flaky']} flaky, "
          f"median decode {block['summary']['median_decode_tok_s']} tok/s", flush=True)
    return block


def export_tasks(path: Path) -> None:
    doc = {
        "name": "TechFuelHQ coding-eval",
        "version": EVAL_VERSION,
        "license": "https://creativecommons.org/licenses/by/4.0/",
        "canonical": "https://techfuelhq.com/data/rtx-5080-llm-throughput/",
        "method": {"api": "ollama /api/chat", "options": OPTIONS, "reps": REPS,
                   "scoring": "mechanical (executed checkers, exact rows, "
                              "fullmatch lists, tool-call JSON); no judge model"},
        "tasks": [{"id": t["id"], "family": t["family"], "prompt": t["prompt"],
                   **({"tools": t["tools"]} if "tools" in t else {})}
                  for t in TASKS],
    }
    path.write_text(json.dumps(doc, indent=1), encoding="utf-8")
    print(f"exported {len(TASKS)} tasks -> {path}")


def main(argv: list[str] | None = None) -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--models", nargs="+")
    parser.add_argument("--num-ctx", type=int, default=OPTIONS["num_ctx"],
                        help="context length for this invocation's models "
                             "(recorded per run; use for documented exceptions "
                             "like a model whose weights+8k KV spill the card)")
    parser.add_argument("--summary-name", default="summary.json")
    parser.add_argument("--tasks", nargs="+",
                        help="run only these task ids (for surgical re-runs "
                             "after a fixture fix; merge per-task downstream)")
    parser.add_argument("--export-tasks", action="store_true")
    parser.add_argument("--out-dir", type=Path, default=(
        REPO / "ops" / "audit" / "evidence"
        / datetime.now(timezone.utc).strftime("%Y-%m-%d") / "coding-eval"))
    args = parser.parse_args(argv)

    if args.export_tasks:
        export_tasks(REPO / "static" / "data" / "coding-eval-tasks-v1.json")
        if not args.models:
            return 0
    if not args.models:
        parser.error("--models required unless --export-tasks")

    OPTIONS["num_ctx"] = args.num_ctx
    if args.tasks:
        global TASKS
        TASKS = [t for t in TASKS if t["id"] in set(args.tasks)]
        if not TASKS:
            parser.error("no tasks matched --tasks")
    args.out_dir.mkdir(parents=True, exist_ok=True)
    capture = {"schema": "techfuelhq:coding-eval:1", "eval_version": EVAL_VERSION,
               "captured_at": utc_now(),
               "environment": {"ollama_version": api("/api/version").get("version")},
               "options": OPTIONS, "reps": REPS, "models": []}
    for model in args.models:
        capture["models"].append(eval_model(model, args.out_dir))
        summary_path = args.out_dir / args.summary_name
        summary_path.write_text(json.dumps(capture, indent=1), encoding="utf-8")
    print(f"summary -> {args.out_dir / args.summary_name}")
    return 0


if __name__ == "__main__":
    sys.exit(main())
