# /// script
# requires-python = ">=3.12,<3.13"
# dependencies = ["tiktoken==0.14.0", "toon-format==1.1.0"]
# ///
"""Recount the current local capture and verify the frozen historical baseline."""

import argparse
import json
from hashlib import sha256
from pathlib import Path

import tiktoken
from toon_format import dumps, loads

ROOT = Path(__file__).resolve().parents[1]
DIRECTORY = ROOT / "public/blog/evidence"
if not DIRECTORY.is_dir():
    DIRECTORY = Path(__file__).resolve().parent
FIXTURES = DIRECTORY / "mcp-current-fixtures-2026-10-07.json"
PUBLIC_RESULTS = DIRECTORY / "mcp-current-measurements-2026-10-07.json"
RESULTS = ROOT / "src/data/mcp-current-measurements.json"
if not RESULTS.parent.is_dir():
    RESULTS = PUBLIC_RESULTS


def compact(value):
    return json.dumps(value, ensure_ascii=False, allow_nan=False, separators=(",", ":"))


def reduction(before, after):
    return round(100 * (1 - after / before), 2)


def measure(fixtures_path):
    fixtures = json.loads(fixtures_path.read_text())
    baseline_path = fixtures_path.parent / fixtures["baseline_file"]
    assert sha256(baseline_path.read_bytes()).hexdigest() == fixtures["baseline_sha256"]
    baseline = json.loads(baseline_path.read_text())
    assert all(
        len(json.loads(row["discovery_request"])["tools"]) == 2
        for row in baseline["initial_context"]
    )
    assert [tool["function"]["name"] for tool in fixtures["helpers"]] == [
        "aixy_search_tools",
        "aixy_execute_tool",
        "aixy_connections",
        "aixy_connect",
    ]
    report = {
        key: value
        for key, value in fixtures.items()
        if key not in {"helpers", "samples", "initial_context"}
    }
    report.update(
        helper_count=4,
        historical_helper_count=2,
        historical_label_correction="The 6 October initial control contains search and execute only, not four helpers. Its frozen files and counts are unchanged.",
        sequence_baseline="Previous verbose discovery replies and legacy search/execute helpers from the frozen capture, with the two current connection helpers added to every previous request. Current requests contain all four actual gateway helpers. Native catalogs, user prompt, execution arguments and fixed successful result are unchanged. Schema lookup rounds follow the captured primary schema.",
        encoding_baseline="Same decoded response value and schema constraints; current compact JSON versus offline TOON. This control is distinct from the previous discovery sequence.",
        limitations="Local serialized input estimates with simulated model steps. No live accounts, native execution, provider framing, inference, quality, retries, latency, prompt caching or billed cost. Initial eager requests contain 100 native definitions; discovery requests contain four gateway helpers. Percentages have different denominators and cannot be added.",
        tokenizers={},
    )
    for tokenizer in ["o200k_base", "cl100k_base"]:
        encoding = tiktoken.get_encoding(tokenizer)

        def count(value, encoding=encoding):
            return len(encoding.encode_ordinary(value))

        cases = []
        for sample in fixtures["samples"]:
            value, exact = (
                json.loads(sample["response"]),
                json.loads(sample["selected_definition"]),
            )
            assert sample["response"] == compact(value)
            assert sample["selected_definition"] == compact(exact)
            assert loads(sample["same_data_toon"]) == value
            assert loads(dumps(value)) == value
            schemas = {
                tool["display_name"]: tool["inputSchema"]
                for tool in sample["native_tools"]
            }
            for tool in (
                value.get("tools", [value]) + value.get("context_tools", []) + [exact]
            ):
                if "inputSchema" in tool:
                    assert tool["inputSchema"] == schemas[tool["name"].split(".", 1)[1]]
            row = {key: sample[key] for key in ["id", "shape", "candidates", "mode"]}
            row["schema_in_search"] = any(
                tool.get("name") == exact["name"] and "inputSchema" in tool
                for tool in value.get("tools", [value])
            )
            row["context_helpers"] = len(value.get("context_tools", []))
            before, after = count(sample["same_data_toon"]), count(sample["response"])
            row["response"] = {
                "compact_json": after,
                "same_data_toon": before,
                "json_reduction_percent": reduction(before, after),
            }
            if "historical_toon_response" in sample:
                row["same_value_as_historical"] = value == loads(
                    sample["historical_toon_response"]
                )
            if "current_requests" in sample:
                current, control, previous = (
                    sample["current_requests"],
                    sample["same_data_toon_requests"],
                    sample["previous_requests"],
                )
                assert len(current) == len(control)
                for raw, alternate in zip(current, control, strict=True):
                    request, other = json.loads(raw), json.loads(alternate)
                    assert request["tools"] == other["tools"] == fixtures["helpers"]
                    for message in other["messages"]:
                        if message.get("role") == "tool" and message.get(
                            "tool_call_id"
                        ) in {"search", "definition"}:
                            message["content"] = compact(loads(message["content"]))
                    assert request == other
                for raw in previous:
                    assert len(json.loads(raw)["tools"]) == 4
                old, new, toon = (
                    sum(map(count, previous)),
                    sum(map(count, current)),
                    sum(map(count, control)),
                )
                row["sequence"] = {
                    "previous": old,
                    "current_json": new,
                    "same_data_toon": toon,
                    "previous_rounds": len(previous),
                    "current_rounds": len(current),
                    "combined_reduction_percent": reduction(old, new),
                    "json_format_reduction_percent": reduction(toon, new),
                }
            cases.append(row)
        initial = []
        for sample in fixtures["initial_context"]:
            request = json.loads(sample["discovery_request"])
            assert request["tools"] == fixtures["helpers"]
            assert (
                len(json.loads(sample["eager_request"])["tools"])
                == sample["catalog_tools"]
                == 100
            )
            eager, discovery = (
                count(sample["eager_request"]),
                count(sample["discovery_request"]),
            )
            initial.append(
                {
                    "shape": sample["shape"],
                    "catalog_tools": sample["catalog_tools"],
                    "helper_count": 4,
                    "eager": eager,
                    "discovery": discovery,
                    "reduction_percent": reduction(eager, discovery),
                }
            )
        report["tokenizers"][tokenizer] = {"cases": cases, "initial_context": initial}
    return report


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--fixtures", type=Path, default=FIXTURES)
    parser.add_argument("--results", type=Path, default=RESULTS)
    parser.add_argument("--write", action="store_true")
    args = parser.parse_args()
    report = measure(args.fixtures)
    if args.write:
        content = json.dumps(report, ensure_ascii=False, indent=2) + "\n"
        args.results.write_text(content)
        if args.results == RESULTS:
            PUBLIC_RESULTS.write_text(content)
    else:
        assert json.loads(args.results.read_text()) == report, (
            "Current measurements differ; inspect and recount."
        )
        if args.results == RESULTS:
            assert PUBLIC_RESULTS.read_bytes() == RESULTS.read_bytes()
    for tokenizer, data in report["tokenizers"].items():
        print(tokenizer, "initial", data["initial_context"])
        for case in data["cases"]:
            print(case["id"], case["response"], case.get("sequence", {}))
    print(
        "Verified 12 same-data response pairs, 10 deterministic sequences and four literal gateway helpers with two tokenizers."
    )


if __name__ == "__main__":
    main()
