"""
calibrate.py -- [2026-10-01] Are the playbook references' odds right? Every 5 minutes from
9:45 to 15:00, for every stock-day of the test range, match the whole day so far (as
reference_trader.py: 150 closest EARLIER library days), record the references' forecast
(share "up" = +2% before -1%, share "down", median best gain) and what the stock really did
next. Then compare forecast buckets with reality.

    python3 calibrate.py 2026-08-27..2026-09-30
"""
import gzip
import json
import sys
from pathlib import Path

import numpy as np

sys.path.insert(0, str(Path(__file__).resolve().parent))
import reference_trader as RT


def main(rng):
    a_, b_ = rng.split("..")
    lib = RT.Lib()
    days = sorted({k.split("|")[0] for k in lib.keys if a_ <= k.split("|")[0] <= b_})
    rows = []
    for day in days:
        d = json.load(gzip.open(RT.BARS / f"bars_{day}.json.gz", "rt"))
        for s in d["top30_current"]:
            ref = RT.REF.load(day, s)
            if not ref:
                continue
            a = RT.day_arrays(d["symbols"][s]["bars"], ref)
            if a is None:
                continue
            lib.start(day)
            for m in range(15, 331, 5):
                idx, _ = lib.match(a["F"], m, day, 150)
                outc = lib.outcomes(idx, m)
                p_up = sum(o[0] for o in outc) / 150
                p_dn = sum(o[1] for o in outc) / 150
                med = float(np.median([o[2] for o in outc]))
                up, dn, best = RT.outcomes_from(a["C"], a["H"], a["L"], m)
                rows.append((day, s, m, p_up, p_dn, med, up, dn, best, (a["C"][385] / a["C"][m] - 1) * 100))
        RT.REF._CACHE.clear()
        print(day, len(rows), flush=True)
    R = np.array([r[3:] for r in rows], dtype=float)
    p_up, p_dn, med, up, dn, best, close = R.T
    h1 = np.array([r[0] for r in rows]) < days[len(days) // 2]
    print(f"\n{len(rows)} forecasts, {len(days)} days. Overall real: up first {up.mean():.0%}, down first {dn.mean():.0%}")
    print(f"\n{'refs say up':>12} {'n':>6} {'real up':>8} {'real down':>10} {'to close':>9} {'1st half up':>12} {'2nd half up':>12}")
    for lo, hi in ((0, .2), (.2, .25), (.25, .3), (.3, .35), (.35, .4), (.4, 1.01)):
        k = (p_up >= lo) & (p_up < hi)
        if k.sum() < 20:
            print(f"{lo:>5.0%}-{min(hi, 1):<5.0%} {k.sum():6}")
            continue
        print(f"{lo:>5.0%}-{min(hi, 1):<5.0%} {k.sum():6} {up[k].mean():8.0%} {dn[k].mean():10.0%} {close[k].mean():+8.2f}% "
              f"{up[k & h1].mean():11.0%} {up[k & ~h1].mean():11.0%}")
    edge = p_up - p_dn
    print(f"\n{'refs edge (up-down)':>20} {'n':>6} {'real up':>8} {'real down':>10} {'to close':>9}")
    for lo, hi in ((-1, -.4), (-.4, -.3), (-.3, -.2), (-.2, -.1), (-.1, 0), (0, 1)):
        k = (edge >= lo) & (edge < hi)
        if k.sum() >= 20:
            print(f"{lo:>+8.1f} .. {hi:<+5.1f}    {k.sum():6} {up[k].mean():8.0%} {dn[k].mean():10.0%} {close[k].mean():+8.2f}%")
    # rank check: does a higher forecast mean a better outcome? (AUC of p_up for real "up")
    pos, neg = p_up[up == 1], p_up[up == 0]
    order = np.argsort(np.concatenate([pos, neg]))
    ranks = np.empty(len(order)); ranks[order] = np.arange(1, len(order) + 1)
    auc = (ranks[:len(pos)].sum() - len(pos) * (len(pos) + 1) / 2) / (len(pos) * len(neg))
    print(f"\nAUC of the references' 'up' share for the real outcome: {auc:.3f}  (0.5 = no information)")
    print(f"target check: references' median best gain vs real best gain: correlation "
          f"{np.corrcoef(med, best)[0, 1]:.2f}; median ratio real/forecast {np.median(best / np.maximum(med, .1)):.2f}")
    json.dump([list(map(lambda x: x if isinstance(x, str) else float(x), r)) for r in rows],
              open(Path(__file__).resolve().parent / f"calibration_{a_}_{b_}.json", "w"))


if __name__ == "__main__":
    main(sys.argv[1])
