"""Normalize benchmark summary JSON into a stable schema.

Ensures each benchmark entry includes required fields:
- name
- status
- pass
- runtime_sec
- metric
"""

from __future__ import annotations

import argparse
import json
from pathlib import Path
from typing import Any, Dict


def _coerce_runtime(value: Any) -> float:
    try:
        return float(value)
    except (TypeError, ValueError):
        return 0.0


def _coerce_pass(value: Any, *, status: Any = None) -> int:
    """Coerce benchmark pass values into stable 0/1 integer semantics."""
    true_values = {"1", "true", "yes", "on", "passed", "pass", "ok", "success"}
    false_values = {"0", "false", "no", "off", "failed", "fail", "error", "timed_out", "invalid", "timeout", "unknown"}

    if isinstance(value, bool):
        return int(value)

    if isinstance(value, (int, float)):
        try:
            return 1 if float(value) != 0 else 0
        except (TypeError, ValueError):
            return 0

    if value is None:
        if status is not None:
            status_text = str(status).strip().lower()
            if status_text == "passed":
                return 1
        return 0

    if isinstance(value, (bytes, bytearray)):
        try:
            value = value.decode("utf-8")
        except Exception:
            value = ""

    if isinstance(value, str):
        normalized = value.strip().lower()
        if not normalized:
            return 0
        if normalized in true_values:
            return 1
        if normalized in false_values:
            return 0
        try:
            return 1 if float(normalized) != 0 else 0
        except ValueError:
            return 0

    return 1 if bool(value) else 0


def _normalize_benchmark_entry(entry: Dict[str, Any]) -> Dict[str, Any]:
    raw = entry if isinstance(entry, dict) else {}
    metric = raw.get("metric")
    if metric is None:
        metric = {"metric_unavailable": True}

    name = raw.get("name")
    if not name:
        name = raw.get("id", "unknown")

    status = raw.get("status")
    if not status:
        status = "unknown"

    normalized = dict(raw)
    normalized["name"] = str(name)
    normalized["status"] = str(status)
    normalized["pass"] = _coerce_pass(raw.get("pass"), status=normalized["status"])
    normalized["runtime_sec"] = _coerce_runtime(raw.get("runtime_sec", 0.0))
    normalized["metric"] = metric
    return normalized


def normalize_summary(payload: Dict[str, Any]) -> Dict[str, Any]:
    raw = payload.get("benchmarks", [])
    if not isinstance(raw, list):
        raw = []

    payload = dict(payload)
    payload["benchmarks"] = [_normalize_benchmark_entry(b) for b in raw]
    return payload


def parse_args() -> argparse.Namespace:
    parser = argparse.ArgumentParser(description="Normalize benchmark summary JSON.")
    parser.add_argument("summary_path", type=Path)
    parser.add_argument("--inplace", action="store_true", help="Overwrite file with normalized version")
    parser.add_argument("--output", type=Path)
    return parser.parse_args()


def main() -> int:
    args = parse_args()
    input_path: Path = args.summary_path

    if not input_path.exists():
        raise SystemExit(f"summary path not found: {input_path}")

    payload = json.loads(input_path.read_text(encoding="utf-8"))
    normalized = normalize_summary(payload)

    output_path = args.output or (input_path if args.inplace else None)
    if output_path is None:
        raise SystemExit("either --inplace or --output required")

    output_path.write_text(json.dumps(normalized, indent=2), encoding="utf-8")
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
