#!/usr/bin/env python3
"""Selective, opt-in compatibility checks for OpenAI-shaped API routes.

`--self-test` uses local fixtures only. `--live` requires explicit surface/check
selection, sends no automatic retries, and may incur provider usage charges.
"""
import argparse
from datetime import datetime, timezone
import hashlib
import json
import os
from pathlib import Path
import sys
import tempfile
from urllib.error import HTTPError, URLError
from urllib.parse import urlsplit
from urllib.request import Request, urlopen


SURFACES = ("chat", "responses")
CHECKS = ("basic", "stream", "tools", "json")
SCHEMA = {
    "type": "object",
    "properties": {
        "ok": {"type": "boolean"},
        "count": {"type": "integer"},
        "label": {"type": "string"},
    },
    "required": ["ok", "count", "label"],
    "additionalProperties": False,
}


class CheckFailure(ValueError):
    """A local pass criterion was not met."""
    def __init__(self, message, metadata=None):
        super().__init__(message)
        self.metadata = metadata or {}


class RouteFailure(RuntimeError):
    def __init__(self, message, metadata=None):
        super().__init__(message)
        self.metadata = metadata or {}


def parse_sse(text):
    """Yield JSON events only from blank-line-terminated SSE frames."""
    normalized = text.replace("\r\n", "\n")
    frames = normalized.split("\n\n")
    # The final segment is pending data at EOF unless a blank line dispatched it.
    if not normalized.endswith("\n\n"):
        frames = frames[:-1]
    for block in frames:
        data = "\n".join(line[5:].lstrip() for line in block.splitlines()
                           if line.startswith("data:"))
        if data and data != "[DONE]":
            try:
                event = json.loads(data)
            except json.JSONDecodeError as exc:
                raise CheckFailure("SSE data was not valid JSON") from exc
            if not isinstance(event, dict):
                raise CheckFailure("SSE data was not a JSON object")
            yield event


def request_metadata(status, headers):
    headers = headers or {}
    request_id = next((headers.get(key) for key in
                       ("x-request-id", "openai-request-id", "request-id")
                       if headers.get(key)), None)
    return {"http_status": status, "request_id": request_id}


def request_json(url, key, payload, timeout, want_stream=False):
    req = Request(url, data=json.dumps(payload).encode("utf-8"), method="POST", headers={
        "Authorization": "Bearer " + key,
        "Content-Type": "application/json",
        "Accept": "text/event-stream" if want_stream else "application/json",
    })
    try:
        with urlopen(req, timeout=timeout) as response:
            status = response.status
            meta = request_metadata(status, response.headers)
            try:
                body = response.read().decode("utf-8", errors="replace")
            except Exception as exc:
                # Headers are already observed; retain their safe metadata without the body/error text.
                raise RouteFailure(f"Response body read failed ({type(exc).__name__})", meta) from None
    except HTTPError as exc:
        meta = request_metadata(exc.code, exc.headers)
        # Never echo error bodies: they may contain prompt fragments or account data.
        raise RouteFailure(f"HTTP {exc.code}", meta) from None
    except URLError as exc:
        # Do not include the URL or upstream body in the report.
        raise RouteFailure(f"Network/transport error ({type(exc.reason).__name__})") from None
    if status < 200 or status >= 300:
        raise RouteFailure(f"HTTP {status}", meta)
    try:
        parsed = list(parse_sse(body)) if want_stream else json.loads(body)
    except (json.JSONDecodeError, CheckFailure):
        raise CheckFailure("Response body did not match the requested JSON/SSE format", meta) from None
    if not want_stream and not isinstance(parsed, dict):
        raise CheckFailure("Response body was not a JSON object", meta)
    return parsed, meta


def check_chat_text_response(response):
    if not isinstance(response, dict):
        raise CheckFailure("Chat response was not an object")
    choices = response.get("choices")
    if not isinstance(choices, list) or not choices or not isinstance(choices[0], dict):
        raise CheckFailure("Chat response had no first choice")
    choice = choices[0]
    message = choice.get("message") or {}
    if not isinstance(message, dict):
        raise CheckFailure("Chat response message was malformed")
    if message.get("refusal"):
        raise CheckFailure("Chat fixture was refused")
    if message.get("tool_calls"):
        raise CheckFailure("Chat fixture returned a tool handoff instead of final text")
    if choice.get("finish_reason") != "stop":
        raise CheckFailure(f"Chat text did not finish normally ({choice.get('finish_reason')!r})")
    content = message.get("content")
    if not isinstance(content, str) or not content.strip():
        raise CheckFailure("Chat response had no non-empty assistant text")
    return content


def check_chat_stream(events):
    chunks, terminal_reasons, refusal, errors = [], set(), False, []
    for event in events:
        if "error" in event or event.get("type") == "error":
            errors.append("stream error event")
        for choice in event.get("choices", []):
            delta = choice.get("delta") or {}
            content = delta.get("content")
            if isinstance(content, str):
                chunks.append(content)
            if delta.get("refusal"):
                refusal = True
            reason = choice.get("finish_reason")
            if reason is not None:
                terminal_reasons.add(reason)
    if errors:
        raise CheckFailure("Chat stream contained an explicit error event")
    if refusal:
        raise CheckFailure("Chat stream was refused")
    if terminal_reasons != {"stop"}:
        raise CheckFailure(f"Chat stream did not finish with one unambiguous normal stop ({sorted(map(str, terminal_reasons))!r})")
    if not "".join(chunks).strip():
        raise CheckFailure("Chat stream had no non-empty assistant text")
    return "".join(chunks)


def response_parts(response):
    """Extract text, refusals, and function calls from raw Responses JSON wire data."""
    if not isinstance(response, dict):
        raise CheckFailure("Responses body was not a JSON object")
    output = response.get("output")
    if not isinstance(output, list):
        raise CheckFailure("Responses body had no output item array")
    text_parts, refusals, function_calls = [], [], []
    for item in output:
        if not isinstance(item, dict):
            continue
        if item.get("type") == "function_call":
            function_calls.append(item)
        if item.get("type") != "message":
            continue
        for content in item.get("content", []) or []:
            if not isinstance(content, dict):
                continue
            if content.get("type") == "output_text" and isinstance(content.get("text"), str):
                text_parts.append(content["text"])
            elif content.get("type") == "refusal":
                refusals.append(content.get("refusal") or content.get("text") or "refused")
    return {"text": "".join(text_parts), "refusals": refusals,
            "function_calls": function_calls, "output": output}


def check_responses_text_response(response):
    if response.get("status") != "completed":
        raise CheckFailure(f"Responses status was not completed ({response.get('status')!r})")
    parts = response_parts(response)
    if parts["refusals"]:
        raise CheckFailure("Responses fixture was refused")
    if parts["function_calls"]:
        raise CheckFailure("Responses fixture returned an unresolved function call instead of final text")
    if not parts["text"].strip():
        raise CheckFailure("Responses body had no non-empty output_text content item")
    return parts["text"]


def check_responses_stream(events):
    chunks, completed, refusal = [], False, False
    for event in events:
        kind = event.get("type")
        if kind in {"error", "response.failed", "response.incomplete"} or "error" in event:
            raise CheckFailure(f"Responses stream contained {kind or 'an error'}")
        if kind in {"response.refusal.delta", "response.refusal.done"}:
            refusal = True
        if kind == "response.output_text.delta":
            delta = event.get("delta")
            if isinstance(delta, str):
                chunks.append(delta)
        if kind == "response.completed":
            wire_response = event.get("response")
            if not isinstance(wire_response, dict) or wire_response.get("status") != "completed":
                raise CheckFailure("Responses completed event lacked a completed wire response")
            if "output" in wire_response and response_parts(wire_response)["refusals"]:
                refusal = True
            completed = True
    if refusal:
        raise CheckFailure("Responses stream was refused")
    if not completed:
        raise CheckFailure("Responses stream ended without response.completed")
    if not "".join(chunks).strip():
        raise CheckFailure("Responses stream had no non-empty output-text delta")
    return "".join(chunks)


def validate_schema_result(value):
    if not isinstance(value, dict) or set(value) != {"ok", "count", "label"}:
        raise CheckFailure("JSON output must contain exactly ok, count, and label")
    if not isinstance(value["ok"], bool):
        raise CheckFailure("ok must be a boolean")
    if not isinstance(value["count"], int) or isinstance(value["count"], bool):
        raise CheckFailure("count must be an integer")
    if not isinstance(value["label"], str):
        raise CheckFailure("label must be a string")
    return value


def check_surface_and_checks(surfaces, checks):
    if not surfaces or any(item not in SURFACES for item in surfaces):
        raise CheckFailure("Select at least one surface: chat and/or responses")
    if not checks or any(item not in CHECKS for item in checks):
        raise CheckFailure("Select at least one check: basic, stream, tools, and/or json")
    return list(dict.fromkeys(surfaces)), list(dict.fromkeys(checks))


def initial_report(base, model, surfaces, checks):
    now = datetime.now(timezone.utc).isoformat()
    status = {surface: {check: {"status": "unknown", "reason": "not selected"}
                       for check in CHECKS} for surface in SURFACES}
    for surface in surfaces:
        for check in checks:
            status[surface][check] = {"status": "pending"}
    return {
        "mode": "live-opt-in",
        "checked_at_utc": now,
        "base_url": base,
        "model": model,
        "checks": status,
        "billing_reconciliation": {
            "status": "unknown",
            "reason": "This script cannot query the provider account ledger or invoice.",
        },
    }


def write_report(path, report, reserve=False):
    path = Path(path)
    path.parent.mkdir(parents=True, exist_ok=True)
    data = json.dumps(report, indent=2, sort_keys=True) + "\n"
    if reserve:
        with path.open("x", encoding="utf-8") as output:
            output.write(data)
        return
    # Atomic replacement avoids a half-written JSON report after interruption.
    fd, temp_name = tempfile.mkstemp(prefix=path.name + ".", suffix=".tmp", dir=path.parent)
    try:
        with os.fdopen(fd, "w", encoding="utf-8") as output:
            output.write(data)
        os.replace(temp_name, path)
    finally:
        if os.path.exists(temp_name):
            os.unlink(temp_name)


def usage_record(response, surface):
    usage = response.get("usage")
    if not isinstance(usage, dict) or not usage:
        return {"status": "unknown", "reason": "usage object not reported"}
    return {"status": "reported", "fields": usage,
            "surface": surface,
            "note": "Not a billing receipt; reconcile with the account ledger separately."}


def route_endpoint(base, surface):
    return base + ("/chat/completions" if surface == "chat" else "/responses")


def run_selected(base, model, key, timeout, surfaces, checks, report_path,
                 request_fn=request_json):
    surfaces, checks = check_surface_and_checks(surfaces, checks)
    parsed_base = urlsplit(base)
    if parsed_base.scheme != "https" or not parsed_base.hostname or parsed_base.username \
            or parsed_base.password or parsed_base.query or parsed_base.fragment:
        raise CheckFailure("Base URL must be HTTPS without embedded credentials, query, or fragment")
    base = base.rstrip("/")
    report_path = Path(report_path)
    report = initial_report(base, model, surfaces, checks)
    write_report(report_path, report, reserve=True)

    def call(surface, payload, stream=False, metadata=None):
        endpoint = route_endpoint(base, surface)
        try:
            value, meta = request_fn(endpoint, key, payload, timeout, want_stream=stream)
        except RouteFailure as exc:
            (metadata if metadata is not None else []).append({
                "endpoint": endpoint.rsplit("/", 1)[-1], **exc.metadata
            })
            raise
        except CheckFailure as exc:
            (metadata if metadata is not None else []).append({
                "endpoint": endpoint.rsplit("/", 1)[-1], **exc.metadata
            })
            raise
        (metadata if metadata is not None else []).append({
            "endpoint": endpoint.rsplit("/", 1)[-1], **meta
        })
        return value

    for surface in surfaces:
        for check in checks:
            row = report["checks"][surface][check]
            row["status"] = "running"
            row["requests"] = []
            write_report(report_path, report)
            try:
                if surface == "chat" and check == "basic":
                    response = call(surface, {
                        "model": model,
                        "messages": [{"role": "user", "content": "Reply with the word ready."}],
                    }, metadata=row["requests"])
                    check_chat_text_response(response)
                    row["usage"] = usage_record(response, surface)
                elif surface == "chat" and check == "stream":
                    events = call(surface, {
                        "model": model, "stream": True,
                        "messages": [{"role": "user", "content": "Reply with the word streamed."}],
                    }, stream=True, metadata=row["requests"])
                    check_chat_stream(events)
                elif surface == "chat" and check == "tools":
                    tools = [{"type": "function", "function": {
                        "name": "lookup_demo", "description": "Return a fixed local test value.",
                        "parameters": {"type": "object", "properties": {"key": {"type": "string"}},
                                        "required": ["key"], "additionalProperties": False},
                    }}]
                    user = {"role": "user", "content": "Call lookup_demo with key alpha."}
                    first = call(surface, {
                        "model": model, "messages": [user], "tools": tools, "tool_choice": "required",
                    }, metadata=row["requests"])
                    choice = first.get("choices", [{}])[0]
                    message = choice.get("message") or {}
                    if choice.get("finish_reason") != "tool_calls":
                        raise CheckFailure("Chat tool request did not end with finish_reason=tool_calls")
                    calls = message.get("tool_calls") or []
                    if not calls:
                        raise CheckFailure("Chat tool response contained no tool_calls")
                    follow = [user, message]
                    for tool_call in calls:
                        function = tool_call.get("function") or {}
                        if function.get("name") != "lookup_demo" or json.loads(function.get("arguments", "{}")) != {"key": "alpha"}:
                            raise CheckFailure("Chat returned an unexpected tool name or arguments")
                        follow.append({"role": "tool", "tool_call_id": tool_call["id"],
                                       "content": json.dumps({"value": "fixture-alpha"})})
                    final = call(surface, {"model": model, "messages": follow}, metadata=row["requests"])
                    check_chat_text_response(final)
                elif surface == "chat" and check == "json":
                    response = call(surface, {
                        "model": model,
                        "messages": [{"role": "user", "content":
                                      'Return JSON only: {"ok":true,"count":3,"label":"sample"}'}],
                        "response_format": {"type": "json_object"},
                    }, metadata=row["requests"])
                    content = check_chat_text_response(response)
                    validate_schema_result(json.loads(content))
                    row["usage"] = usage_record(response, surface)
                    row["qualification"] = "JSON mode plus local schema validation; not server-side JSON Schema enforcement."
                elif surface == "responses" and check == "basic":
                    response = call(surface, {"model": model, "input": "Reply with the word ready."},
                                    metadata=row["requests"])
                    check_responses_text_response(response)
                    row["usage"] = usage_record(response, surface)
                elif surface == "responses" and check == "stream":
                    events = call(surface, {
                        "model": model, "input": "Reply with the word streamed.", "stream": True,
                    }, stream=True, metadata=row["requests"])
                    check_responses_stream(events)
                elif surface == "responses" and check == "tools":
                    user_input = [{"role": "user", "content": "Call lookup_demo with key alpha."}]
                    first = call(surface, {
                        "model": model, "input": user_input,
                        "tools": [{"type": "function", "name": "lookup_demo",
                                   "description": "Return a fixed local test value.",
                                   "parameters": {"type": "object", "properties": {"key": {"type": "string"}},
                                                  "required": ["key"], "additionalProperties": False}}],
                        "tool_choice": "required",
                    }, metadata=row["requests"])
                    if first.get("status") != "completed":
                        raise CheckFailure("Responses tool call did not have status=completed")
                    first_parts = response_parts(first)
                    calls = first_parts["function_calls"]
                    if first_parts["refusals"] or not calls:
                        raise CheckFailure("Responses tool request had no usable function_call")
                    tool_outputs = []
                    for tool_call in calls:
                        if tool_call.get("name") != "lookup_demo" or json.loads(tool_call.get("arguments", "{}")) != {"key": "alpha"}:
                            raise CheckFailure("Responses returned an unexpected tool name or arguments")
                        tool_outputs.append({"type": "function_call_output", "call_id": tool_call["call_id"],
                                             "output": json.dumps({"value": "fixture-alpha"})})
                    second = call(surface, {
                        "model": model,
                        "input": user_input + first_parts["output"] + tool_outputs,
                    }, metadata=row["requests"])
                    check_responses_text_response(second)
                    row["usage"] = usage_record(second, surface)
                elif surface == "responses" and check == "json":
                    response = call(surface, {
                        "model": model,
                        "input": 'Return {"ok":true,"count":3,"label":"sample"}.',
                        "text": {"format": {"type": "json_schema", "name": "demo",
                                             "strict": True, "schema": SCHEMA}},
                    }, metadata=row["requests"])
                    content = check_responses_text_response(response)
                    validate_schema_result(json.loads(content))
                    row["usage"] = usage_record(response, surface)
                    row["qualification"] = "Strict JSON Schema request plus local validation; confirm supported schema subset."
                row["status"] = "pass"
                row["reason"] = "Fixture-specific pass criteria met."
            except Exception as exc:
                row["status"] = "fail"
                row["reason"] = str(exc) if isinstance(exc, (CheckFailure, RouteFailure)) else type(exc).__name__
                if isinstance(exc, (RouteFailure, CheckFailure)) and exc.metadata and not row["requests"]:
                    row["requests"].append({"endpoint": route_endpoint(base, surface).rsplit("/", 1)[-1],
                                            **exc.metadata})
            write_report(report_path, report)
    report["summary"] = {
        status: sum(1 for surface in report["checks"].values() for row in surface.values()
                    if row["status"] == status)
        for status in ("pass", "fail", "unknown")
    }
    write_report(report_path, report)
    return report


def _expect_failure(fn, *args):
    try:
        fn(*args)
    except (CheckFailure, json.JSONDecodeError):
        return
    raise AssertionError(f"{fn.__name__} unexpectedly passed")


def self_test(results_path=None):
    fixture_path = Path(__file__).with_name("compat-fixtures-v3.json")
    fixtures = json.loads(fixture_path.read_text(encoding="utf-8"))
    results = []

    def assert_equal(actual, expected):
        if actual != expected:
            raise AssertionError(f"expected {expected!r}, got {actual!r}")

    def passes(name, fn):
        fn()
        results.append({"case": name, "status": "pass"})

    def fails(name, fn):
        try:
            fn()
        except (CheckFailure, json.JSONDecodeError):
            results.append({"case": name, "status": "pass", "expected": "rejected"})
            return
        raise AssertionError(name + " unexpectedly passed")

    passes("chat_basic_normal_stop", lambda: check_chat_text_response(fixtures["chat_basic_pass"]))
    fails("chat_basic_refusal", lambda: check_chat_text_response(fixtures["chat_refusal"]))
    for reason in fixtures["chat_reasons_not_plain_text_pass"]:
        fixture = {"choices": [{"message": {"content": "partial"}, "finish_reason": reason}]}
        fails("chat_nonstream_reason_" + str(reason), lambda fixture=fixture: check_chat_text_response(fixture))
    passes("chat_stream_normal_stop", lambda: check_chat_stream(fixtures["chat_stream_pass"]))
    for reason in fixtures["chat_stream_reasons_not_pass"]:
        event = [{"choices": [{"delta": {"content": "partial"}, "finish_reason": reason}]}]
        fails("chat_stream_reason_" + str(reason), lambda event=event: check_chat_stream(event))
    fails("chat_stream_explicit_error", lambda: check_chat_stream(fixtures["chat_stream_error"]))
    fails("chat_stream_conflicting_terminal_reasons",
          lambda: check_chat_stream(fixtures["chat_stream_conflicting_reasons"]))

    wire = fixtures["responses_raw_basic_multi_item"]
    passes("responses_raw_wire_multiple_output_items",
           lambda: assert_equal(check_responses_text_response(wire), "ready now"))
    fails("responses_missing_status", lambda: check_responses_text_response(fixtures["responses_missing_status"]))
    fails("responses_incomplete", lambda: check_responses_text_response(fixtures["responses_incomplete"]))
    fails("responses_refusal", lambda: check_responses_text_response(fixtures["responses_refusal"]))
    fails("responses_text_with_unresolved_function_call",
          lambda: check_responses_text_response(fixtures["responses_basic_with_unresolved_function_call"]))
    tool_first = fixtures["responses_tool_first"]
    passes("responses_raw_function_call_item",
           lambda: assert_equal(response_parts(tool_first)["function_calls"][0]["call_id"], "call_fixture"))
    passes("responses_raw_tool_followup_message",
           lambda: assert_equal(check_responses_text_response(fixtures["responses_tool_followup"]), "fixture-alpha"))
    passes("responses_stream_completed", lambda: check_responses_stream(fixtures["responses_stream_pass"]))
    for key in ("responses_stream_error_then_completed", "responses_stream_failed_then_completed",
                "responses_stream_incomplete_then_completed", "responses_stream_malformed_completed",
                "responses_stream_missing_completed"):
        fails(key, lambda key=key: check_responses_stream(fixtures[key]))
    fails("responses_stream_unterminated_completion_frame",
          lambda: check_responses_stream(list(parse_sse(fixtures["responses_stream_unterminated_completion"]))))
    parsed = list(parse_sse('data: {"type":"response.completed"}\n\ndata: [DONE]\n\n'))
    passes("sse_json_data_and_done_marker", lambda: assert_equal(parsed, [{"type": "response.completed"}]))
    passes("schema_valid_fixture", lambda: validate_schema_result(fixtures["valid_schema_result"]))
    for index, invalid in enumerate(fixtures["invalid_schema_results"]):
        fails("schema_invalid_fixture_" + str(index), lambda invalid=invalid: validate_schema_result(invalid))
    passes("raw_responses_structured_output",
           lambda: assert_equal(validate_schema_result(json.loads(check_responses_text_response(
               fixtures["responses_json_pass"])))['count'], 3))
    passes("request_id_header_capture",
           lambda: assert_equal(request_metadata(200, {"x-request-id": "fixture-header-id"})["request_id"],
                                "fixture-header-id"))

    def sequence_request(sequence, calls):
        def request(url, key, payload, timeout, want_stream=False):
            calls.append({"endpoint": url.rsplit("/", 1)[-1], "want_stream": want_stream})
            result = sequence.pop(0)
            return result, fixtures["http_success_metadata"]
        return request

    import tempfile
    with tempfile.TemporaryDirectory() as temp_dir:
        # Selective basic route: exactly one synthetic call; other cells remain unknown.
        calls = []
        basic_path = Path(temp_dir) / "responses-basic.json"
        report = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                              ["responses"], ["basic"], basic_path,
                              request_fn=sequence_request([wire], calls))
        saved = json.loads(basic_path.read_text())
        assert_equal(len(calls), 1)
        assert_equal(report["checks"]["responses"]["basic"]["status"], "pass")
        assert_equal(report["checks"]["responses"]["basic"]["requests"][0]["request_id"], "fixture-id")
        assert_equal(saved["checks"]["chat"]["basic"]["status"], "unknown")
        assert_equal(saved["billing_reconciliation"]["status"], "unknown")
        results.append({"case": "selective_responses_basic_one_call_and_unknown_cells", "status": "pass"})

        # Raw function call and output message are both exercised through the live-suite code path.
        tool_calls = []
        tool_path = Path(temp_dir) / "responses-tools.json"
        tool_report = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                                   ["responses"], ["tools"], tool_path,
                                   request_fn=sequence_request([tool_first, fixtures["responses_tool_followup"]], tool_calls))
        assert_equal(len(tool_calls), 2)
        assert_equal(tool_report["checks"]["responses"]["tools"]["status"], "pass")
        assert_equal(tool_report["checks"]["responses"]["tools"]["requests"][1]["request_id"], "fixture-id")
        results.append({"case": "selective_responses_tool_raw_followup_two_calls", "status": "pass"})

        unresolved_calls = []
        unresolved_path = Path(temp_dir) / "responses-tools-unresolved-followup.json"
        unresolved_report = run_selected(
            "https://example.invalid/v1", "fixture-model", "not-used", 1,
            ["responses"], ["tools"], unresolved_path,
            request_fn=sequence_request([tool_first, fixtures["responses_basic_with_unresolved_function_call"]], unresolved_calls))
        assert_equal(len(unresolved_calls), 2)
        assert_equal(unresolved_report["checks"]["responses"]["tools"]["status"], "fail")
        results.append({"case": "responses_tool_followup_rejects_unresolved_function_call", "status": "pass"})

        chat_tool_calls = []
        chat_tool_path = Path(temp_dir) / "chat-tools.json"
        chat_tool_report = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                                        ["chat"], ["tools"], chat_tool_path,
                                        request_fn=sequence_request([fixtures["chat_tool_first"],
                                                                     fixtures["chat_tool_followup"]], chat_tool_calls))
        assert_equal(len(chat_tool_calls), 2)
        assert_equal(chat_tool_report["checks"]["chat"]["tools"]["status"], "pass")
        results.append({"case": "selective_chat_tool_round_trip_two_calls", "status": "pass"})

        # Both JSON surfaces parse raw text and apply client-side schema validation.
        json_calls = []
        json_path = Path(temp_dir) / "selected-json.json"
        json_report = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                                   ["chat", "responses"], ["json"], json_path,
                                   request_fn=sequence_request([fixtures["chat_json_pass"], fixtures["responses_json_pass"]], json_calls))
        assert_equal(len(json_calls), 2)
        assert_equal(json_report["checks"]["chat"]["json"]["status"], "pass")
        assert_equal(json_report["checks"]["responses"]["json"]["status"], "pass")
        results.append({"case": "selective_chat_and_responses_json_checks", "status": "pass"})

        def failing_fixture(url, key, payload, timeout, want_stream=False):
            raise RouteFailure("HTTP 400", fixtures["http_error_metadata"])
        fail_path = Path(temp_dir) / "failed-report.json"
        failed = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                              ["chat"], ["basic"], fail_path, request_fn=failing_fixture)
        assert_equal(failed["checks"]["chat"]["basic"]["status"], "fail")
        assert_equal(json.loads(fail_path.read_text())["checks"]["chat"]["basic"]["requests"][0]["request_id"], "fixture-error-id")
        results.append({"case": "failed_check_persists_status_and_request_id", "status": "pass"})

        class ReadTimeoutResponse:
            status = 200
            headers = {"x-request-id": fixtures["http_read_timeout_metadata"]["request_id"]}
            def __enter__(self): return self
            def __exit__(self, *args): return None
            def read(self): raise TimeoutError("synthetic fixture")

        timeout_path = Path(temp_dir) / "read-timeout-report.json"
        saved_urlopen = globals()["urlopen"]
        try:
            globals()["urlopen"] = lambda *args, **kwargs: ReadTimeoutResponse()
            timed_out = run_selected("https://example.invalid/v1", "fixture-model", "not-used", 1,
                                     ["responses"], ["stream"], timeout_path, request_fn=request_json)
        finally:
            globals()["urlopen"] = saved_urlopen
        timeout_row = json.loads(timeout_path.read_text())["checks"]["responses"]["stream"]
        assert_equal(timed_out["checks"]["responses"]["stream"]["status"], "fail")
        assert_equal(timeout_row["requests"][0]["http_status"], 200)
        assert_equal(timeout_row["requests"][0]["request_id"], fixtures["http_read_timeout_metadata"]["request_id"])
        results.append({"case": "body_read_timeout_retains_observed_status_and_request_id", "status": "pass"})

    if results_path:
        results_path = Path(results_path)
        if results_path.exists():
            raise FileExistsError("refusing to overwrite offline test results")
        payload = {
            "executed_at_utc": datetime.now(timezone.utc).isoformat(),
            "script_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(),
            "fixture_sha256": hashlib.sha256(fixture_path.read_bytes()).hexdigest(),
            "network_requests": 0,
            "result": "pass" if all(item["status"] == "pass" for item in results) else "fail",
            "cases": results,
        }
        results_path.parent.mkdir(parents=True, exist_ok=True)
        results_path.write_text(json.dumps(payload, indent=2, sort_keys=True) + "\n", encoding="utf-8")
        print("offline test results saved:", results_path)
    print(f"offline self-tests: PASS ({len(results)} fixture assertions; no network)")
    return results


def default_report_path():
    stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
    return Path(f"compat-report-{stamp}.json")


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--self-test", action="store_true", help="run synthetic local fixtures; no network")
    parser.add_argument("--live", action="store_true", help="send the explicitly selected small test requests")
    parser.add_argument("--base-url", default=os.getenv("LLM_BASE_URL"))
    parser.add_argument("--model", default=os.getenv("LLM_MODEL"))
    parser.add_argument("--key-env", default="LLM_API_KEY", help="environment variable name for the API key")
    parser.add_argument("--surface", choices=SURFACES, action="append", help="repeat for each surface to test")
    parser.add_argument("--check", choices=CHECKS, action="append", help="repeat for each feature check")
    parser.add_argument("--timeout", type=int, default=40)
    parser.add_argument("--report", type=Path, help="new JSON report path; existing files are never overwritten")
    parser.add_argument("--results", type=Path, help="new JSON file for offline self-test results")
    args = parser.parse_args()
    if args.self_test and args.live:
        parser.error("--self-test and --live are separate modes; do not combine them")
    if args.self_test:
        self_test(args.results)
    if args.live:
        if not args.base_url or not args.model or not os.getenv(args.key_env):
            parser.error("--live requires --base-url, --model, and a key in the --key-env variable")
        if not args.surface or not args.check:
            parser.error("--live requires explicit --surface and --check selection; no tests run by default")
        surfaces, checks = check_surface_and_checks(args.surface, args.check)
        report_path = args.report or default_report_path()
        if report_path.exists():
            parser.error(f"report path already exists; choose a new path: {report_path}")
        report = run_selected(args.base_url, args.model, os.getenv(args.key_env), args.timeout,
                              surfaces, checks, report_path)
        print(json.dumps(report, indent=2, sort_keys=True))
        print(f"report saved: {report_path}")
        return 1 if report["summary"]["fail"] else 0
    if not args.self_test:
        parser.error("choose --self-test (offline) or --live (network and usage charges may occur)")
    return 0


if __name__ == "__main__":
    try:
        sys.exit(main())
    except Exception as exc:
        print(f"FAIL: {exc}", file=sys.stderr)
        sys.exit(1)
