#!/usr/bin/env python3
"""Minimal runnable agent-eval suite: execute cli scenarios, record manual ones.

Usage (from the checkout root):
    python3 docs/agent-engineering/examples/evals/run_evals.py [--out report.json]

Reads scenarios.json next to this script. For kind=cli, runs the command,
checks exit code, expected stdout substrings, and — where the scenario
declares stdout_json — per-row values of the parsed JSON document, then
records pass/fail with an output excerpt. For kind=manual, records not_run with reproduction steps.
Also reads behavior.json (agent-behavior runs) when present and records each
as not_run with its criteria: behavior runs need a live model session plus a
human judge, so the runner never executes or passes them. Writes a JSON
report and prints a one-line-per-scenario summary. Exit code is 0 only when
every cli scenario passes (manual and behavior scenarios never fail the run).

No third-party dependencies. Launches the ./law entry point, which rebuilds
its engine on first use, so the first scenario takes the longest.
"""

from __future__ import annotations

import argparse
import json
import subprocess
import sys
import time
from datetime import datetime, timezone
from pathlib import Path

HERE = Path(__file__).resolve().parent
EXCERPT_CHARS = 1200


def check_json_rows(stdout: str, requirements: list) -> tuple:
    """Match per-row requirements against a queryResult document.

    Each requirement names a row by the tail of its ``producer``
    value (``producer_suffix``) and pins the row's other ``values``
    fields (``strength`` here). Returns (matched, missing): matched
    requirement labels, and missing labels with the reason — no row,
    or the row's field value differing. Independent substrings
    cannot do this: one ``strict`` anywhere would satisfy two rules.
    """
    matched = []
    missing = []
    try:
        document = json.loads(stdout)
    except json.JSONDecodeError as error:
        return matched, [
            "%s: stdout is not JSON (%s)" % (r["producer_suffix"], error)
            for r in requirements
        ]
    rows = [
        r.get("values", {}) for r in document.get("rows", [])
        if isinstance(r, dict)
    ]
    for req in requirements:
        label = req["producer_suffix"]
        pinned = {k: v for k, v in req.items() if k != "producer_suffix"}
        hits = [
            row for row in rows
            if str(row.get("producer", "")).endswith(label)
        ]
        if not hits:
            missing.append("%s: no row" % label)
            continue
        bad = [
            "%s is %r, want %r" % (k, hit.get(k), v)
            for hit in hits
            for k, v in pinned.items()
            if hit.get(k) != v
        ]
        if bad:
            missing.append("%s: %s" % (label, "; ".join(bad)))
        else:
            matched.append(
                "%s: %s"
                % (label, ", ".join("%s=%r" % kv for kv in pinned.items()))
            )
    return matched, missing


def run_cli(scenario: dict) -> dict:
    started = time.monotonic()
    try:
        completed = subprocess.run(
            scenario["command"],
            cwd=Path.cwd(),
            capture_output=True,
            text=True,
            timeout=scenario.get("timeout_s", 600),
        )
        duration_ms = int((time.monotonic() - started) * 1000)
        stdout = completed.stdout or ""
        expect = scenario.get("expect", {})
        missing = [
            s for s in expect.get("stdout_contains", []) if s not in stdout
        ]
        exit_ok = completed.returncode == expect.get("exit", 0)
        result = {
            "id": scenario["id"],
            "kind": "cli",
            "status": "pass",  # narrowed below
            "command": scenario["command"],
            "exit": completed.returncode,
            "expected_exit": expect.get("exit", 0),
            "matched": [
                s
                for s in expect.get("stdout_contains", [])
                if s not in missing
            ],
            "missing": missing,
            "duration_ms": duration_ms,
            "output_excerpt": stdout[:EXCERPT_CHARS],
            "stderr_excerpt": (completed.stderr or "")[:400],
        }
        if "stdout_json" in expect:
            json_matched, json_missing = check_json_rows(
                stdout, expect["stdout_json"]
            )
            result["json_matched"] = json_matched
            result["json_missing"] = json_missing
            if json_missing:
                missing = missing + json_missing
        result["status"] = (
            "pass" if exit_ok and not missing else "fail"
        )
        return result
    except subprocess.TimeoutExpired:
        return {
            "id": scenario["id"],
            "kind": "cli",
            "status": "fail",
            "command": scenario["command"],
            "error": f"timeout after {scenario.get('timeout_s', 600)}s",
            "duration_ms": int((time.monotonic() - started) * 1000),
        }
    except FileNotFoundError as error:
        return {
            "id": scenario["id"],
            "kind": "cli",
            "status": "fail",
            "command": scenario["command"],
            "error": f"cannot launch: {error}",
            "duration_ms": int((time.monotonic() - started) * 1000),
        }


def record_manual(scenario: dict) -> dict:
    return {
        "id": scenario["id"],
        "kind": "manual",
        "status": "not_run",
        "reason": "needs an MCP session or a human reader; the runner cannot execute it",
        "reproduction": scenario.get("reproduction", []),
        "checks": scenario.get("checks", ""),
    }


def record_behavior(scenario: dict) -> dict:
    return {
        "id": scenario["id"],
        "kind": "behavior",
        "status": "not_run",
        "reason": "OPEN: needs a live model run plus a human judge; the runner cannot execute it",
        "input_task": scenario.get("input_task", ""),
        "model_version": scenario.get("model_version", "OPEN"),
        "criteria": scenario.get("criteria", []),
    }


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--out", default="agent-eval-report.json")
    args = parser.parse_args()

    scenarios = json.loads((HERE / "scenarios.json").read_text())
    results = []
    for scenario in scenarios["scenarios"]:
        if scenario["kind"] == "cli":
            results.append(run_cli(scenario))
        else:
            results.append(record_manual(scenario))

    behavior_file = HERE / "behavior.json"
    if behavior_file.exists():
        behavior = json.loads(behavior_file.read_text())
        for scenario in behavior["scenarios"]:
            results.append(record_behavior(scenario))

    report = {
        "suite": scenarios.get("suite", "agent-evals"),
        "suite_version": scenarios.get("version", "0.1.0"),
        "started_at": datetime.now(timezone.utc).isoformat(),
        "results": results,
        "summary": {
            status: sum(1 for r in results if r["status"] == status)
            for status in ("pass", "fail", "not_run")
        },
    }
    out = Path(args.out)
    out.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")

    for result in results:
        if result["status"] == "pass":
            print(f"PASS    {result['id']} ({result['duration_ms']} ms)")
        elif result["status"] == "fail":
            detail = result.get("error") or f"missing={result.get('missing')}"
            print(f"FAIL    {result['id']} ({detail})")
        else:
            print(f"NOT_RUN {result['id']} ({result['reason']})")
    print(f"summary={report['summary']} report={out}")
    return 0 if report["summary"]["fail"] == 0 else 1


if __name__ == "__main__":
    sys.exit(main())
