"""Reproduce the September 30 editorial tables; Python standard library only.

Run: python analysis.py (from any directory). Reads the two retained exports
beside this file. No network, model inference, or upstream raw-score claims.
"""
import collections
import hashlib
import json
import math
import pathlib
import statistics

BASE = pathlib.Path(__file__).resolve().parent
models = json.loads((BASE / "models.json").read_text(encoding="utf-8"))
evidence = json.loads((BASE / "evidence.json").read_text(encoding="utf-8"))
assert models["generatedAt"] == evidence["generatedAt"]
lookup = {row["modelId"]: row for row in models["models"]}
current = sorted((row for row in evidence["models"] if row["rank"]["current"] is not None), key=lambda row: row["rank"]["current"])
assert [row["rank"]["current"] for row in current] == list(range(1, len(current) + 1))

def price(row):
    prices = lookup[row["modelId"]]["pricingUsdPerMillionTokens"]
    if prices["input"] is None or prices["output"] is None:
        return None
    return 0.75 * prices["input"] + 0.25 * prices["output"]

def describe(row):
    meta = lookup[row["modelId"]]
    return dict(name=row["name"], slug=row["slug"], rank=row["rank"]["current"], score=row["score"], support=row["confidence"], interval=row["interval"], state=meta["modelCap"]["evidenceState"], access=meta["weightAccess"], price=price(row), endpoints=meta["currentProviderEndpoints"], releasedAt=meta["releasedAt"], families=row["families"], boards=[obs["board"] for obs in row["observations"]])

def med(values):
    return statistics.median(values) if values else None

def frontier(rows):
    # Equal prices: higher score first, then current rank. The same-price
    # lower-scoring row is dominated. A zero list price remains a list price.
    ordered = sorted((row for row in rows if price(row) is not None), key=lambda row: (price(row), -row["score"], row["rank"]["current"]))
    best, out = -math.inf, []
    for row in ordered:
        if row["score"] > best:
            out.append(describe(row))
            best = row["score"]
    return out

def obs(row, board):
    return next((o for o in row["observations"] if o["board"] == board), None)

pairs = []
for row in current:
    arena, aa = obs(row, "Arena Overall"), obs(row, "Artificial Analysis Intelligence Index")
    if arena and aa:
        pairs.append(dict(name=row["name"], rank=row["rank"]["current"], arena=arena["sourceScore"], aa=aa["sourceScore"], difference=round(aa["sourceScore"] - arena["sourceScore"], 10), arenaConfiguration=arena["configuration"], aaConfiguration=aa["configuration"]))
gaps = [abs(pair["difference"]) for pair in pairs]
states = collections.defaultdict(list)
access = collections.defaultdict(list)
publishers = collections.defaultdict(list)
publisher_projection = json.loads((BASE / "publisher-identities.json").read_text(encoding="utf-8"))
assert publisher_projection["generatedAt"] == evidence["generatedAt"]
publisher_ids = {row["modelId"]: publisher_projection["canonicalAliases"].get(row["publisherId"],row["publisherId"]) for row in publisher_projection["models"]}
assert set(publisher_ids) == {row["modelId"] for row in current}
canonical_publishers = collections.defaultdict(list)
for row in current:
    meta = lookup[row["modelId"]]
    states[meta["modelCap"]["evidenceState"]].append(row)
    access[meta["weightAccess"]].append(row)
    publishers[meta["publisher"]].append(row)
    canonical_publishers[publisher_ids[row["modelId"]]].append(row)
established = [row for row in current if row["basis"] == "measured" and row["fusion"] is None and row["confidence"] >= 60]
coding = sorted((row for row in current if obs(row, "Arena Coding")), key=lambda row: (-obs(row, "Arena Coding")["sourceScore"], row["rank"]["current"]))
caps = []
support_errors = []
for row in evidence["models"]:
    if row["basis"] != "measured" or row["fusion"]:
        continue
    support = sum(evidence["method"]["familyWeights"][o["family"]] * (o["share"] or 0) * o["confidence"] / 100 for o in row["observations"])
    support_errors.append(abs(support - row["confidence"]))
    if row["rank"]["current"] is None:
        continue
    for family in ("coding", "agent", "reasoning"):
        value = row["families"][family]
        if value is not None and abs(value - row["families"]["general"]) > evidence["method"]["specialistDeltaBound"]:
            caps.append(dict(name=row["name"], family=family, delta=value - row["families"]["general"]))
contexts = [lookup[row["modelId"]]["contextTokens"] for row in current if lookup[row["modelId"]]["contextTokens"] is not None]
output = dict(
    generatedAt=evidence["generatedAt"], method=evidence["methodology"], currentCount=len(current), allVersionsCount=len(evidence["models"]),
    pricedCount=sum(price(row) is not None for row in current), medianPrice=med([price(row) for row in current if price(row) is not None]),
    establishedCount=len(established), topTen=[describe(row) for row in current[:10]],
    frontiers=dict(all=frontier(current), support25=frontier([row for row in current if row["confidence"] >= 25]), measured60=frontier(established)),
    states={state: dict(count=len(rows), supportMedian=med([row["confidence"] for row in rows]), intervalWidthMedian=med([row["interval"]["upper"] - row["interval"]["lower"] for row in rows]), top25=sum(row["rank"]["current"] <= 25 for row in rows)) for state, rows in sorted(states.items())},
    access={label: dict(count=len(rows), measured=sum(row["basis"] == "measured" for row in rows), supportMedian=med([row["confidence"] for row in rows]), priced=sum(price(row) is not None for row in rows), priceMedian=med([price(row) for row in rows if price(row) is not None]), served=sum(lookup[row["modelId"]]["currentProviderEndpoints"] > 0 for row in rows), servedMeasured=sum(row["basis"] == "measured" and lookup[row["modelId"]]["currentProviderEndpoints"] > 0 for row in rows)) for label, rows in sorted(access.items())},
    establishedTop=[describe(row) for row in established[:12]],
    establishedAccessLeaders={label: describe(next(row for row in established if lookup[row["modelId"]]["weightAccess"] == label)) for label in access if any(lookup[row["modelId"]]["weightAccess"] == label for row in established)},
    codingCount=len(coding), codingTop=[dict(**describe(row), coding=obs(row,"Arena Coding")) for row in coding[:12]],
    noCodingTop20=[row["name"] for row in current[:20] if (row["families"] or {}).get("coding") is None],
    pairs=pairs, pairSummary=dict(n=len(pairs), medianAbsoluteGap=med(gaps), meanAbsoluteGap=statistics.mean(gaps), over10=sum(gap > 10 for gap in gaps), within3=sum(gap <= 3 for gap in gaps), medianSignedGap=med([pair["difference"] for pair in pairs]), aaHigher=sum(pair["difference"] > 0 for pair in pairs), arenaHigher=sum(pair["difference"] < 0 for pair in pairs)),
    caps=caps, supportCheck=dict(n=len(support_errors), maximumAbsoluteError=max(support_errors)),
    boardCoverage=[dict(**board, currentCount=sum(obs(row,board["board"]) is not None for row in current), newestAdmittedPublication=max((o["publishedAt"] for row in evidence["models"] for o in row["observations"] if o["board"] == board["board"]),default=None)) for board in evidence["method"]["boards"]],
    publisherNames=dict(count=len(publishers), note="Exact exported publisher display names; no unverified alias merging.", rows=[dict(name=name,count=len(rows),top25=sum(row["rank"]["current"] <= 25 for row in rows),top50=sum(row["rank"]["current"] <= 50 for row in rows)) for name, rows in sorted(publishers.items(),key=lambda item:(-len(item[1]),item[0]))]),
    canonicalPublishers=dict(count=len(canonical_publishers), note="Recorded publisher keys joined by exact reviewed site/Hub authority aliases, not fuzzy names or legal entity ownership.",rows=[dict(publisherId=publisherId,count=len(rows),top25=sum(row["rank"]["current"] <= 25 for row in rows),top50=sum(row["rank"]["current"] <= 50 for row in rows)) for publisherId,rows in sorted(canonical_publishers.items(),key=lambda item:(-len(item[1]),item[0]))]),
    septemberCount=sum((lookup[row["modelId"]]["releasedAt"] or "").startswith("2026-09") for row in current),
    context=dict(count=len(contexts),median=med(contexts),atLeastMillion=sum(value >= 1000000 for value in contexts),maximum=max(contexts)),
    launches=[describe(row) for row in evidence["models"] if row["name"] in ("Claude Opus 5.5", "GPT-6 Sol", "Grok 4.7", "MiMo-V2.6-Pro")],
    estimatesTop25=[dict(**describe(row),fusion=row["fusion"]) for row in current[:25] if row["fusion"]],
    workedExamples={row["name"]:row for row in current if row["name"] in ("Claude Fable 5.1","Gemini 3.1 Pro Preview","MiMo-V2.6-Flash")},
)
launch = json.loads((BASE / "launch-outcomes.json").read_text(encoding="utf-8"))
report = json.loads((BASE / "launch-validation.json").read_text(encoding="utf-8"))
records = launch["records"]
assert report["generatedAt"] == evidence["generatedAt"] == launch["generatedAt"]
assert all(row["revision"] == 0 and row["predictedAt"] < row["measuredAt"] for row in records)
assert all(len(row["boards"]) >= 2 and ("Arena Overall" in row["boards"] or "Artificial Analysis Intelligence Index" in row["boards"]) for row in records)
assert all(abs(round(row["predicted"] - row["measured"], 1) - row["signedError"]) < 1e-8 for row in records)
def summary(rows):
    errors = sorted(abs(row["signedError"]) for row in rows)
    # Producer uses Math.round, including half-point ties; Python round uses
    # ties-to-even, which changes the within-card channel's 6.45 median.
    round1 = lambda value: math.floor(value * 10 + 0.5) / 10
    return dict(n=len(rows), mae=round1(statistics.mean(errors)), signedMedian=round1(med([row["signedError"] for row in rows])), p80=errors[math.ceil(len(errors)*0.8)-1], coverage80=round(sum(row["covered80"] for row in rows)/len(rows),3))
assert summary(records) == report["prospective"]
for channel, expected in report["channels"].items():
    channel_rows = [row for row in records if any(term["channel"] == channel and term["share"] >= 0.25 for term in row["terms"])]
    assert summary(channel_rows) == expected
correct = total = 0
for i, first in enumerate(records):
    for second in records[i+1:]:
        gap = first["measured"] - second["measured"]
        if abs(gap) < 5:
            continue
        total += 1
        predicted_gap = first["predicted"] - second["predicted"]
        correct += (gap > 0 and predicted_gap > 0) or (gap < 0 and predicted_gap < 0)
assert dict(n=total,accuracy=round(correct/total,3),minimumGap=5) == report["pairwise"]
output["launchValidationCheck"] = dict(prospective=summary(records),correctPairs=correct,totalPairs=total,chronology="All retained revision-zero prediction times precede settled ledger outcome times. This does not prove the upstream result was unavailable at prediction time.")
(BASE / "analysis.json").write_text(json.dumps(output, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
print(json.dumps({key: output[key] for key in ("generatedAt","currentCount","allVersionsCount","pricedCount","establishedCount","codingCount","pairSummary","supportCheck")},indent=2))
