"""
prf_simulator_2026_09_18.py

RESEARCH / BACKTESTING TOOL ONLY.
------------------------------------------------------------
This script does NOT place trades, does NOT import the live bot
(alpaca_client.py, monitor.py, stream.py, exit.py, smart_engine.py),
and does NOT modify entry or exit logic. It only READS the historical
trade log at data/trades/<date>_trades.jsonl and simulates, in
isolation, a proposed new exit concept called Price Response Failure
(PRF) against five trades from the 2026-09-18 session for which
detailed post-peak order-flow samples were supplied by hand.

WHAT THIS SCRIPT ANSWERS
------------------------------------------------------------
Would a PRF-based protective exit have caught GLOO's loss of price
response before its ~11:02 ET breakdown, WITHOUT falsely triggering
on BNC, SBET, FWDI or SECZ, which each pulled back but did not
actually reverse?

DATA HONESTY RULES FOLLOWED THROUGHOUT
------------------------------------------------------------
1. Actual entry/exit prices, quantities, times and P/L for all 14
   trades in the session come directly from data/trades/2026-09-18_
   trades.jsonl (the bot's own fill log) -- not re-derived or guessed.
2. The five order-flow sample sets (BNC, GLOO #2, SBET, FWDI, SECZ #2)
   are transcribed byte-for-byte from what was supplied for this
   analysis. No tick between the given observations is invented.
3. Every "peak" price/time is exactly what was supplied and is
   labelled APPROXIMATE throughout, per the source material's own
   "approximately" qualifier.
4. GLOO's supplied samples are 30-second bid/ask + trade-volume
   snapshots, not raw trade prints. This script uses the bid/ask
   midpoint as a stand-in "price" for that trade only, and labels it
   as such everywhere it's used. The other four trades' samples are
   already literal trade prices, no conversion needed.
5. Where a value cannot be determined from supplied data (e.g. MFE
   giveback trigger points that fall outside the ~40s supplied
   window), the script prints "NOT DETERMINABLE FROM SUPPLIED DATA"
   or "NOT TRIGGERED WITHIN SUPPLIED WINDOW" rather than guessing.
"""

import argparse
import json
from dataclasses import dataclass, field
from pathlib import Path

import pandas as pd

# ==================================================================
# CONFIGURABLE THRESHOLDS -- change these, don't hunt through the code
# ==================================================================

MIN_MFE_PCT_TO_ARM = 3.0          # PRF doesn't evaluate a trade until it's up this much from entry
STRONG_IMBALANCE_THRESHOLD = 0.70  # imbalance reading counted as "strong buying pressure"
NEGATIVE_IMBALANCE_THRESHOLD = 0.0  # any imbalance below this counts as a "negative" read
WARNING_AFTER_N_FAILED_RESPONSES = 2  # failed-response strikes needed to raise PRF_WARNING
CONFIRMATION_AFTER_N_NEGATIVE_READS = 2  # consecutive negative reads needed after WARNING to confirm protection
MEANINGFUL_NEW_HIGH_PCT = 0.05     # price must clear the local high by at least this % to "count" as a new high

# sensitivity-test grids (objective: characterize behavior, not pick a "winner")
SENSITIVITY_IMBALANCE_THRESHOLDS = [0.60, 0.70, 0.80]
SENSITIVITY_NEGATIVE_CONFIRM_READS = [1, 2, 3]

# MFE-giveback comparison levels (evaluated completely separately from PRF)
MFE_GIVEBACK_LEVELS_PCT = [30, 40, 50]

STATE_NORMAL = "NORMAL"
STATE_PRF_WARNING = "PRF_WARNING"
STATE_PROFIT_PROTECTION = "PROFIT_PROTECTION"
STATE_CONFIRMED_REVERSAL = "CONFIRMED_REVERSAL"  # see note in simulate_prf()

TRADE_LOG_DIR = Path(__file__).parent / "data" / "trades"


# ==================================================================
# 1. ACTUAL TRADE LOG (source of truth for entries/exits/actual P/L)
# ==================================================================

def load_trade_log(date_str: str) -> list[dict]:
    """Reads data/trades/<date>_trades.jsonl -- the bot's real fills.
    This is the ONLY source used for actual entry/exit price, qty,
    times and P/L. Nothing here is invented or backfit from the
    user-supplied P/L summary; it's read straight from the log file,
    and happens to match that summary exactly (verified: sums to
    $403.11 for 2026-09-18, same as the session total supplied)."""
    path = TRADE_LOG_DIR / f"{date_str}_trades.jsonl"
    trades = [json.loads(line) for line in open(path) if line.strip()]

    # Disambiguate repeated symbols (BNC, GLOO, SECZ each traded twice
    # this session) by order of entry time, so "GLOO #1"/"GLOO #2" etc
    # match how the session was described.
    seen = {}
    for t in trades:
        seen[t["symbol"]] = seen.get(t["symbol"], 0) + 1
        t["trade_label"] = t["symbol"] if seen[t["symbol"]] == 1 and \
            sum(1 for x in trades if x["symbol"] == t["symbol"]) == 1 \
            else f"{t['symbol']} #{seen[t['symbol']]}"
    return trades


# ==================================================================
# 2. SUPPLIED POST-PEAK ORDER-FLOW SAMPLES (hand-transcribed, verbatim)
# ==================================================================
#
# Each entry's "trade_id" links it to one specific fill in the trade
# log via ASSOCIATION below. "price_source" documents whether the
# price column is a literal trade print or a derived bid/ask midpoint.

ORDER_FLOW_SAMPLES = {

    "BNC_1": {
        "reference_peak_price": 6.39,
        "reference_peak_time_label": "~13:53:36 (as supplied; timezone as given, unverified)",
        "price_source": "trade_price",
        "observations": [
            # (offset_seconds, price, imbalance, deterioration_signal_count[context only, not used by PRF])
            (0,  6.39,   0.412, 1),
            (5,  6.3405, -0.032, 2),
            (10, 6.3337, -0.046, 3),
            (15, 6.32,   -0.101, 4),
            (20, 6.3467, -0.489, 5),
            (25, 6.30,    0.271, 0),
            (30, 6.3074,  0.238, 1),
            (35, 6.31,    0.408, 0),
            (40, 6.32,    0.051, 1),
        ],
    },

    "GLOO_2": {
        "reference_peak_price": 5.58,
        "reference_peak_time_label": "~10:58-11:00 ET (as supplied, approximate)",
        "price_source": "bidask_midpoint",
        # (time_label, bid, ask, buy_trades, buy_vol, sell_trades, sell_vol, imbalance)
        "raw_bidask_observations": [
            ("10:58:00", 5.48, 5.54, 11, 4727, 4,  1898,  0.43),
            ("10:58:30", 5.52, 5.56, 16, 1225, 16, 6397,  -0.68),
            ("10:59:00", 5.56, 5.58, 8,  1318, 0,  0,      1.00),
            ("10:59:30", 5.54, 5.58, 36, 2796, 10, 2925,  -0.02),
            ("11:00:00", 5.56, 5.58, 30, 2210, 3,  115,    0.90),
            ("11:00:30", 5.55, 5.58, 13, 898,  6,  4745,  -0.68),
            ("11:01:00", 5.57, 5.58, 8,  356,  6,  1966,  -0.69),
            ("11:01:30", 5.55, 5.57, 2,  501,  22, 1653,  -0.53),
            ("11:02:00", 5.41, 5.46, 240, 13102, 546, 92468, -0.75),
        ],
    },

    "SBET": {
        "reference_peak_price": 9.3599,
        "reference_peak_time_label": "~19:50:07 (as supplied, approximate)",
        "price_source": "trade_price",
        "observations": [
            (0,  9.3599, 0.539, None),
            (5,  9.3519, 0.567, None),
            (10, 9.35,   0.645, None),
            (15, 9.3401, 0.577, None),
            (20, 9.355,  0.451, None),
            (25, 9.341,  0.347, None),
            (30, 9.345,  0.416, None),
            (35, 9.3437, 0.541, None),
            (40, 9.35,   0.175, None),
        ],
    },

    "FWDI": {
        "reference_peak_price": 7.68,
        "reference_peak_time_label": "~16:17:47 (as supplied, approximate)",
        "price_source": "trade_price",
        "observations": [
            (0,  7.68,   0.198, None),
            (5,  7.675, -0.280, None),
            (10, 7.5619, -0.271, None),
            (15, 7.5633, -0.009, None),
            (20, 7.675,  0.126, None),
            (25, 7.65,  -0.658, None),
            (30, 7.655, -0.559, None),
            (35, 7.655, -0.586, None),
            (40, 7.65,  -0.584, None),
        ],
    },

    "SECZ_2": {
        "reference_peak_price": 10.848,
        "reference_peak_time_label": "~19:53:17 (as supplied, approximate)",
        "price_source": "trade_price",
        "observations": [
            (0,  10.848,  0.023, None),
            (5,  10.83,  -0.126, None),
            (10, 10.83,  -0.205, None),
            (15, 10.765, -0.360, None),
            (20, 10.765, -0.595, None),
            (25, 10.79,  -0.426, None),
            (30, 10.7939, -0.376, None),
            (35, 10.7394, -0.010, None),
            (40, 10.82,   0.625, None),
        ],
    },
}


def _gloo_observations():
    """GLOO's samples come as 30s bid/ask+volume snapshots, not trade
    prints. We use the bid/ask midpoint as a price stand-in -- this is
    an approximation, flagged wherever it's used."""
    obs = []
    for label, bid, ask, bt, bv, st, sv, imb in ORDER_FLOW_SAMPLES["GLOO_2"]["raw_bidask_observations"]:
        mid = round((bid + ask) / 2, 4)
        obs.append({
            "t_label": label, "price": mid, "imbalance": imb,
            "bid": bid, "ask": ask, "buy_trades": bt, "buy_vol": bv,
            "sell_trades": st, "sell_vol": sv, "deterioration": None,
        })
    return obs


def load_order_flow_samples() -> dict:
    """Normalizes all five supplied datasets into a common observation
    shape: list of {t_label, price, imbalance, deterioration}."""
    out = {}
    for trade_id, data in ORDER_FLOW_SAMPLES.items():
        if trade_id == "GLOO_2":
            obs = _gloo_observations()
        else:
            obs = [
                {"t_label": f"+{off}s", "price": price, "imbalance": imb, "deterioration": det}
                for off, price, imb, det in data["observations"]
            ]
        out[trade_id] = {
            "reference_peak_price": data["reference_peak_price"],
            "reference_peak_time_label": data["reference_peak_time_label"],
            "price_source": data["price_source"],
            "observations": obs,
        }
    return out


# ==================================================================
# 3. ASSOCIATING SUPPLIED SAMPLES WITH REAL FILLS FROM THE LOG
# ==================================================================
#
# Rule from the brief: don't blindly combine same-symbol trades; only
# associate when the supplied peak time/price plausibly falls inside
# a specific fill's holding window. Each rationale below was checked
# against data/trades/2026-09-18_trades.jsonl before writing this file.

TRADE_ASSOCIATIONS = {
    "BNC_1": {
        "symbol": "BNC", "trade_label": "BNC #1",
        "rationale": (
            "Supplied peak time 13:53:36 falls inside BNC #1's real "
            "holding window (13:41:30-13:58:47 UTC); BNC #2 "
            "(14:15:10-14:18:46) doesn't contain it. Peak 6.39 is "
            "plausibly above BNC #1's 6.0595 entry and above its 6.14 "
            "exit, consistent with a pullback-then-partial-recovery "
            "before the eventual exit."
        ),
        "confidence": "high",
    },
    "GLOO_2": {
        "symbol": "GLOO", "trade_label": "GLOO #2",
        "rationale": (
            "Supplied entry ~4.93 matches GLOO #2's real entry price "
            "(4.93) exactly; GLOO #1 entered at 4.865. Supplied peak "
            "window (10:58-11:02 ET = 14:58-15:02 UTC) falls inside "
            "GLOO #2's real holding window (14:04:54-18:17:12 UTC); "
            "GLOO #1's window (13:59:58-14:03:18 UTC) ends before this "
            "even starts."
        ),
        "confidence": "high",
    },
    "SBET": {
        "symbol": "SBET", "trade_label": "SBET",
        "rationale": (
            "Only one SBET fill this session (14:03:58-19:55:03 UTC). "
            "Supplied peak time 19:50:07 falls inside that window; peak "
            "9.3599 is plausibly above entry 8.8503 and close to but "
            "just above the actual exit 9.3356, consistent with a late "
            "peak followed by a small pullback into the EOD exit."
        ),
        "confidence": "high",
    },
    "FWDI": {
        "symbol": "FWDI", "trade_label": "FWDI",
        "rationale": (
            "Only one FWDI fill this session (14:19:06-19:55:03 UTC). "
            "Supplied peak time 16:17:47 falls inside that window; peak "
            "7.68 is above entry 7.0998 and close to the actual exit "
            "7.645, consistent with the supplied note that price pulled "
            "back but recovered."
        ),
        "confidence": "high",
    },
    "SECZ_2": {
        "symbol": "SECZ", "trade_label": "SECZ #2",
        "rationale": (
            "Supplied peak time 19:53:17 falls inside SECZ #2's real "
            "holding window (18:17:33-19:55:03 UTC); SECZ #1's window "
            "(13:39:34-14:04:29) ends hours earlier. Peak 10.848 is "
            "just above SECZ #2's actual exit 10.82, consistent with a "
            "small late pullback into the EOD exit."
        ),
        "confidence": "high",
    },
}


def associate_order_flow_to_trades(trades: list[dict], order_flow: dict) -> dict:
    """Merges each order-flow sample set with its real fill record.
    Returns {trade_id: {..fill fields.., "order_flow": {...}, "association": {...}}}."""
    by_label = {t["trade_label"]: t for t in trades}
    annotated = {}
    for trade_id, assoc in TRADE_ASSOCIATIONS.items():
        fill = by_label.get(assoc["trade_label"])
        if fill is None:
            annotated[trade_id] = {"association": assoc, "fill": None, "order_flow": order_flow[trade_id]}
            continue
        annotated[trade_id] = {
            "association": assoc,
            "fill": fill,
            "order_flow": order_flow[trade_id],
        }
    return annotated


# ==================================================================
# 4. CORE PRF BUILDING BLOCKS
# ==================================================================

def calculate_mfe(entry_price: float, peak_price: float) -> tuple[float, float]:
    """Maximum favorable excursion, in dollars-per-share and percent."""
    mfe_abs = peak_price - entry_price
    mfe_pct = (mfe_abs / entry_price) * 100.0 if entry_price else 0.0
    return mfe_abs, mfe_pct


def detect_local_high(prior_local_high: float, price: float) -> float:
    """Step 1: the recent/local high is just the running max price."""
    return max(prior_local_high, price)


def is_meaningful_new_high(price: float, local_high: float, min_new_high_pct: float = MEANINGFUL_NEW_HIGH_PCT) -> bool:
    """A print has to clear the local high by more than a tiny band to
    count as genuine progress -- otherwise noise at the highs would
    look like an endless string of 'new highs'."""
    return price > local_high * (1 + min_new_high_pct / 100.0)


def detect_failed_price_response(price: float, imbalance: float, local_high: float,
                                  strong_imbalance_threshold: float = STRONG_IMBALANCE_THRESHOLD,
                                  min_new_high_pct: float = MEANINGFUL_NEW_HIGH_PCT) -> tuple[bool, bool, bool]:
    """Step 2+3: strong buying pressure that does NOT produce a new
    high is a 'failed price response'. Returns (failed, strong_buying, made_new_high)."""
    strong_buying = imbalance >= strong_imbalance_threshold
    made_new_high = is_meaningful_new_high(price, local_high, min_new_high_pct)
    failed = strong_buying and not made_new_high
    return failed, strong_buying, made_new_high


def detect_prf_warning(failed_response_count: int, warning_after_n: int = WARNING_AFTER_N_FAILED_RESPONSES) -> bool:
    """Step 4: two failed responses (by default) raise the warning."""
    return failed_response_count >= warning_after_n


def detect_negative_confirmation(imbalance: float, negative_confirm_count: int,
                                  required_reads: int = CONFIRMATION_AFTER_N_NEGATIVE_READS,
                                  negative_threshold: float = NEGATIVE_IMBALANCE_THRESHOLD) -> tuple[int, bool]:
    """Step 5: after a warning, count CONSECUTIVE negative-imbalance
    reads. A non-negative read breaks the streak (resets to 0), since
    the rule explicitly asks for consecutive observations."""
    if imbalance < negative_threshold:
        negative_confirm_count += 1
    else:
        negative_confirm_count = 0
    return negative_confirm_count, negative_confirm_count >= required_reads


# ==================================================================
# 5. THE PRF STATE MACHINE ITSELF
# ==================================================================

@dataclass
class PRFSimResult:
    trade_id: str
    armed: bool
    mfe_pct: float
    triggered: bool = False
    final_state: str = STATE_NORMAL
    warning_t_label: str | None = None
    protection_t_label: str | None = None
    protection_price: float | None = None
    confirmed_reversal_note: str = (
        "Not evaluated: CONFIRMED_REVERSAL requires the existing "
        "deterioration engine's structure/signal-agreement data, which "
        "was not part of the supplied order-flow samples. This "
        "simulation treats reaching PROFIT_PROTECTION as the "
        "actionable protective-exit trigger, consistent with the "
        "existing bot design where the deterioration engine remains "
        "the final confirmation layer."
    )
    trace: list = field(default_factory=list)


def simulate_prf(trade_id: str, order_flow: dict, entry_price: float, cfg: dict | None = None) -> PRFSimResult:
    """Runs the PRF state machine over one trade's supplied post-peak
    observation window. cfg lets the sensitivity test override the
    default thresholds without touching the module-level constants."""
    cfg = cfg or {}
    strong_threshold = cfg.get("strong_imbalance_threshold", STRONG_IMBALANCE_THRESHOLD)
    warning_after_n = cfg.get("warning_after_n_failed_responses", WARNING_AFTER_N_FAILED_RESPONSES)
    confirm_reads = cfg.get("confirmation_after_n_negative_reads", CONFIRMATION_AFTER_N_NEGATIVE_READS)
    min_new_high_pct = cfg.get("meaningful_new_high_pct", MEANINGFUL_NEW_HIGH_PCT)

    peak_price = order_flow["reference_peak_price"]
    mfe_abs, mfe_pct = calculate_mfe(entry_price, peak_price)
    armed = mfe_pct >= MIN_MFE_PCT_TO_ARM

    result = PRFSimResult(trade_id=trade_id, armed=armed, mfe_pct=mfe_pct)
    if not armed:
        result.final_state = STATE_NORMAL
        return result

    state = STATE_NORMAL
    # Step 1: local high seeded at the supplied reference peak -- this
    # matches how the concept was described (comparisons are against
    # "the ~5.58 peak", not against whatever the very first sample
    # happens to be, which for GLOO is well before the actual peak).
    local_high = peak_price
    failed_response_count = 0
    negative_confirm_count = 0

    for obs in order_flow["observations"]:
        price, imbalance, t_label = obs["price"], obs["imbalance"], obs["t_label"]
        prior_local_high = local_high
        local_high = detect_local_high(local_high, price)
        made_new_high = is_meaningful_new_high(price, prior_local_high, min_new_high_pct)

        row = {"t_label": t_label, "price": price, "imbalance": imbalance,
               "local_high": local_high, "made_new_high": made_new_high,
               "state_before": state}

        if made_new_high:
            # Step 6: a fresh meaningful high cancels an active warning.
            if state == STATE_PRF_WARNING:
                state = STATE_NORMAL
                failed_response_count = 0
                negative_confirm_count = 0
                row["note"] = "new high -- PRF_WARNING cancelled"
            else:
                row["note"] = "new high"
            row["state_after"] = state
            result.trace.append(row)
            continue

        if state == STATE_NORMAL:
            failed, strong_buying, _ = detect_failed_price_response(
                price, imbalance, prior_local_high, strong_threshold, min_new_high_pct)
            row["strong_buying"] = strong_buying
            if failed:
                failed_response_count += 1
                row["note"] = f"failed price response #{failed_response_count}"
                if detect_prf_warning(failed_response_count, warning_after_n):
                    state = STATE_PRF_WARNING
                    result.warning_t_label = t_label
                    row["note"] += " -> PRF_WARNING"
            else:
                row["note"] = "normal"
            row["failed_response_count"] = failed_response_count

        elif state == STATE_PRF_WARNING:
            negative_confirm_count, confirmed = detect_negative_confirmation(
                imbalance, negative_confirm_count, confirm_reads)
            row["negative_confirm_count"] = negative_confirm_count
            if confirmed:
                state = STATE_PROFIT_PROTECTION
                result.protection_t_label = t_label
                result.protection_price = price
                result.triggered = True
                row["note"] = f"negative confirmation ({negative_confirm_count} reads) -> PROFIT_PROTECTION"
            else:
                row["note"] = f"in PRF_WARNING, negative streak {negative_confirm_count}/{confirm_reads}"

        elif state == STATE_PROFIT_PROTECTION:
            row["note"] = "already in PROFIT_PROTECTION (trigger point already recorded)"

        row["state_after"] = state
        result.trace.append(row)

    result.final_state = state
    return result


# ==================================================================
# 6. MFE-GIVEBACK COMPARISON (kept fully separate from PRF, per brief)
# ==================================================================

@dataclass
class GivebackResult:
    trade_id: str
    giveback_pct: float
    trigger_level: float
    triggered: bool
    trigger_t_label: str | None = None
    trigger_price: float | None = None
    note: str = ""


def simulate_mfe_giveback(trade_id: str, order_flow: dict, entry_price: float, giveback_pct: float) -> GivebackResult:
    """A much simpler, purely price-based protective stop: exit once
    price gives back X% of the (entry -> peak) move. No imbalance or
    structure signals involved at all."""
    peak_price = order_flow["reference_peak_price"]
    mfe_abs, _ = calculate_mfe(entry_price, peak_price)
    trigger_level = peak_price - (giveback_pct / 100.0) * mfe_abs

    for obs in order_flow["observations"]:
        if obs["price"] <= trigger_level:
            return GivebackResult(trade_id, giveback_pct, trigger_level, True,
                                   obs["t_label"], obs["price"])
    return GivebackResult(
        trade_id, giveback_pct, trigger_level, False, note=(
            "NOT TRIGGERED WITHIN SUPPLIED WINDOW -- price never fell to the "
            f"{giveback_pct}% giveback level (${trigger_level:.4f}) in the "
            "observations supplied; would need data beyond the last supplied "
            "sample to know if/when it eventually would have."
        ))


# ==================================================================
# 7. SENSITIVITY ANALYSIS
# ==================================================================

def run_sensitivity_analysis(annotated: dict) -> pd.DataFrame:
    """Sweeps imbalance threshold x negative-confirmation-reads. Not
    trying to find a 'best' setting from one day -- just characterizing
    how each combination behaves across the five known cases."""
    rows = []
    for trade_id, info in annotated.items():
        if info["fill"] is None:
            continue
        entry_price = info["fill"]["entry_price"]
        for thr in SENSITIVITY_IMBALANCE_THRESHOLDS:
            for reads in SENSITIVITY_NEGATIVE_CONFIRM_READS:
                cfg = {"strong_imbalance_threshold": thr, "confirmation_after_n_negative_reads": reads}
                res = simulate_prf(trade_id, info["order_flow"], entry_price, cfg)
                is_gloo = trade_id == "GLOO_2"
                classification = (
                    ("HIT (expected)" if res.triggered else "MISSED SIGNAL") if is_gloo
                    else ("FALSE TRIGGER" if res.triggered else "correctly silent")
                )
                rows.append({
                    "trade": info["association"]["trade_label"],
                    "imbalance_threshold": thr,
                    "negative_confirm_reads": reads,
                    "triggered": res.triggered,
                    "warning_at": res.warning_t_label,
                    "protection_at": res.protection_t_label,
                    "protection_price": res.protection_price,
                    "classification": classification,
                })
    return pd.DataFrame(rows)


# ==================================================================
# 8. REPORTING
# ==================================================================

def generate_trade_report(trade_id: str, info: dict) -> str:
    fill = info["fill"]
    order_flow = info["order_flow"]
    assoc = info["association"]
    lines = []
    lines.append(f"\n{'=' * 70}\n{assoc['trade_label']}  (trade_id={trade_id})\n{'=' * 70}")

    if fill is None:
        lines.append("No matching fill found in the trade log -- cannot simulate.")
        return "\n".join(lines)

    lines.append(f"Association confidence: {assoc['confidence']}")
    lines.append(f"Association rationale: {assoc['rationale']}")
    lines.append("")
    lines.append(f"ACTUAL (from data/trades/2026-09-18_trades.jsonl, source of truth):")
    lines.append(f"  Entry: {fill['entry_time']}  @ ${fill['entry_price']:.4f}  qty={fill['qty']}")
    lines.append(f"  Exit:  {fill['exit_time']}  @ ${fill['exit_price']:.4f}")
    lines.append(f"  Actual P/L: {fill['pl_pct']:+.3f}%  (${fill['pl_dollars']:+.2f})")
    lines.append(f"  Actual exit reason: {fill.get('exit_reason', 'n/a')}")

    entry_price = fill["entry_price"]
    peak_price = order_flow["reference_peak_price"]
    mfe_abs, mfe_pct = calculate_mfe(entry_price, peak_price)
    lines.append("")
    lines.append(f"SUPPLIED ORDER-FLOW WINDOW (price_source={order_flow['price_source']}):")
    lines.append(f"  Reference peak: ${peak_price:.4f} at {order_flow['reference_peak_time_label']} (APPROXIMATE, as supplied)")
    lines.append(f"  MFE vs real entry ${entry_price:.4f}: ${mfe_abs:+.4f} / {mfe_pct:+.2f}%"
                  f"  ({'armed' if mfe_pct >= MIN_MFE_PCT_TO_ARM else 'NOT armed'} -- gate is {MIN_MFE_PCT_TO_ARM}%)")

    res = simulate_prf(trade_id, order_flow, entry_price)
    lines.append("")
    lines.append("PRF STATE-MACHINE TRACE (default thresholds):")
    if res.trace:
        df = pd.DataFrame(res.trace)
        cols = [c for c in ["t_label", "price", "imbalance", "local_high", "made_new_high",
                             "state_before", "note", "state_after"] if c in df.columns]
        lines.append(df[cols].to_string(index=False))
    else:
        lines.append("  (not armed -- no trace)")

    lines.append("")
    lines.append(f"PRF RESULT: final_state={res.final_state}  triggered={res.triggered}")
    if res.triggered:
        sim_exit_price = res.protection_price
        qty = fill["qty"]
        sim_pl_dollars = qty * (sim_exit_price - entry_price)
        sim_pl_pct = (sim_exit_price - entry_price) / entry_price * 100.0
        diff = sim_pl_dollars - fill["pl_dollars"]
        protected_too_early = diff < 0
        lines.append(f"  PRF_WARNING raised at: {res.warning_t_label}")
        lines.append(f"  PROFIT_PROTECTION confirmed at: {res.protection_t_label}, "
                      f"price ${sim_exit_price:.4f} (nearest supplied observation, APPROXIMATE exit point)")
        lines.append(f"  Simulated PRF exit P/L: {sim_pl_pct:+.2f}%  (${sim_pl_dollars:+.2f})")
        lines.append(f"  Difference vs actual: ${diff:+.2f}")
        lines.append(f"  Protected too early (gave up profit the trade actually kept)? "
                      f"{'YES' if protected_too_early else 'NO'}")
        lines.append(f"  {res.confirmed_reversal_note}")
    else:
        lines.append("  PRF never triggered within the supplied window -- simulated result = actual result.")
        lines.append("  LIMITATION: this only proves silence across the ~40s supplied snapshot, "
                      "not across this trade's full multi-hour holding period.")

    lines.append("")
    lines.append("MFE-GIVEBACK COMPARISON (separate method, not combined with PRF):")
    for pct in MFE_GIVEBACK_LEVELS_PCT:
        gb = simulate_mfe_giveback(trade_id, order_flow, entry_price, pct)
        if gb.triggered:
            qty = fill["qty"]
            gb_pl = qty * (gb.trigger_price - entry_price)
            lines.append(f"  {pct}% giveback: triggered at {gb.trigger_t_label} @ ${gb.trigger_price:.4f} "
                          f"-> ${gb_pl:+.2f} (vs actual ${fill['pl_dollars']:+.2f})")
        else:
            lines.append(f"  {pct}% giveback (level ${gb.trigger_level:.4f}): {gb.note}")

    return "\n".join(lines)


def generate_session_report(date_str: str, trades: list[dict], annotated: dict, sensitivity_df: pd.DataFrame) -> str:
    lines = []
    lines.append("\n" + "#" * 70)
    lines.append(f"# PRF SIMULATION -- SESSION REPORT -- {date_str}")
    lines.append("#" * 70)

    # ---- 1. ACTUAL SESSION ----
    lines.append("\n--- 1. ACTUAL SESSION ---")
    actual_df = pd.DataFrame([{
        "trade": t["trade_label"], "entry_time": t["entry_time"][11:19],
        "exit_time": t["exit_time"][11:19], "entry": t["entry_price"], "exit": t["exit_price"],
        "pl_pct": t["pl_pct"], "pl_dollars": t["pl_dollars"], "exit_reason": t.get("exit_reason", "")[:40],
    } for t in trades])
    lines.append(actual_df.to_string(index=False))
    actual_total = sum(t["pl_dollars"] for t in trades)
    lines.append(f"\nActual session total P/L: ${actual_total:+.2f}")

    # ---- 2 & 3 & 4. PRF SIMULATION / MFE SIMULATION / TRADE-BY-TRADE COMPARISON ----
    lines.append("\n--- 2/3/4. PRF SIMULATION, MFE SIMULATION & TRADE-BY-TRADE COMPARISON ---")
    lines.append("(Only the 5 trades with supplied order-flow data can be simulated; "
                  "the other 9 trades are carried through unchanged.)")
    comparison_rows = []
    prf_session_total = actual_total
    for t in trades:
        comparison_rows.append({
            "trade": t["trade_label"], "actual_pl": t["pl_dollars"],
            "prf_triggered": "-", "prf_protection": "-", "sim_pl": t["pl_dollars"], "difference": 0.0,
        })
    row_by_label = {r["trade"]: r for r in comparison_rows}

    for trade_id, info in annotated.items():
        if info["fill"] is None:
            continue
        entry_price = info["fill"]["entry_price"]
        res = simulate_prf(trade_id, info["order_flow"], entry_price)
        label = info["association"]["trade_label"]
        row = row_by_label[label]
        row["prf_triggered"] = "YES" if res.triggered else "NO"
        row["prf_protection"] = res.protection_t_label if res.triggered else "-"
        if res.triggered:
            qty = info["fill"]["qty"]
            sim_pl = qty * (res.protection_price - entry_price)
            row["sim_pl"] = round(sim_pl, 2)
            row["difference"] = round(sim_pl - info["fill"]["pl_dollars"], 2)
            prf_session_total += row["difference"]

    comp_df = pd.DataFrame(comparison_rows)
    lines.append(comp_df.to_string(index=False))
    lines.append(f"\nSession total, actual:                 ${actual_total:+.2f}")
    lines.append(f"Session total, with PRF swapped in:     ${prf_session_total:+.2f}"
                  f"  (difference ${prf_session_total - actual_total:+.2f})")
    lines.append("LIMITATION: this is a per-trade swap only -- no capital-recycling/"
                  "re-entry effects are modeled (an earlier PRF exit freeing up a slot "
                  "for a different trade is out of scope here).")

    # ---- 5. FALSE POSITIVE ANALYSIS ----
    lines.append("\n--- 5. FALSE POSITIVE ANALYSIS ---")
    false_positives = [r for r in comparison_rows if r["trade"] != "GLOO #2" and r["prf_triggered"] == "YES"]
    if false_positives:
        for r in false_positives:
            lines.append(f"  FALSE TRIGGER on {r['trade']} at default thresholds "
                          f"(threshold={STRONG_IMBALANCE_THRESHOLD}, confirm_reads={CONFIRMATION_AFTER_N_NEGATIVE_READS})")
    else:
        lines.append(f"  None at default thresholds (imbalance>={STRONG_IMBALANCE_THRESHOLD}, "
                      f"confirm_reads={CONFIRMATION_AFTER_N_NEGATIVE_READS}): BNC #1, SBET, FWDI and "
                      "SECZ #2 all stayed in NORMAL across their supplied windows.")
    sens_false = sensitivity_df[(sensitivity_df["trade"] != "GLOO #2") & (sensitivity_df["triggered"])]
    if len(sens_false):
        lines.append(f"  Sensitivity sweep: {len(sens_false)} false-trigger combination(s) found -- see table below.")
    else:
        lines.append("  Sensitivity sweep: zero false triggers across all "
                      f"{len(SENSITIVITY_IMBALANCE_THRESHOLDS) * len(SENSITIVITY_NEGATIVE_CONFIRM_READS)} "
                      "threshold combinations tested, on BNC #1/SBET/FWDI/SECZ #2.")

    # ---- 6. MISSED SIGNAL ANALYSIS ----
    lines.append("\n--- 6. MISSED SIGNAL ANALYSIS ---")
    gloo_row = row_by_label.get("GLOO #2")
    if gloo_row and gloo_row["prf_triggered"] == "YES":
        lines.append("  GLOO #2: PRF triggered at default thresholds -- not a missed signal.")
    else:
        lines.append("  GLOO #2: PRF did NOT trigger at default thresholds -- this would be a missed signal.")
    sens_missed = sensitivity_df[(sensitivity_df["trade"] == "GLOO #2") & (~sensitivity_df["triggered"])]
    if len(sens_missed):
        lines.append(f"  Sensitivity sweep: GLOO #2 missed under {len(sens_missed)} of "
                      f"{len(SENSITIVITY_IMBALANCE_THRESHOLDS) * len(SENSITIVITY_NEGATIVE_CONFIRM_READS)} combinations:")
        lines.append(sens_missed[["imbalance_threshold", "negative_confirm_reads"]].to_string(index=False))
    else:
        lines.append("  Sensitivity sweep: GLOO #2 triggered under every combination tested.")

    # ---- 7. TIMING ANALYSIS ----
    lines.append("\n--- 7. TIMING ANALYSIS ---")
    gloo_sens = sensitivity_df[sensitivity_df["trade"] == "GLOO #2"]
    lines.append(gloo_sens[["imbalance_threshold", "negative_confirm_reads", "triggered",
                             "warning_at", "protection_at", "protection_price"]].to_string(index=False))
    lines.append("  Note: GLOO #2's real large breakdown (per supplied data) hit ~$5.41-5.46 at "
                  "~11:02:00 ET. Any protection_at time at or before 11:01:30 exits ahead of that print.")

    # ---- 8. LIMITATIONS ----
    lines.append("\n--- 8. LIMITATIONS ---")
    lines.append("  - Only 5 of 14 trades have supplied order-flow data; PRF cannot be evaluated for the other 9.")
    lines.append("  - Each of the 5 windows is a short (~40-45s) snapshot around one peak, not the full trade "
                  "lifetime; absence of a trigger in-window does not prove PRF would stay silent for the whole trade.")
    lines.append("  - GLOO #2's price series is a bid/ask midpoint approximation (30s snapshots), not raw trade "
                  "prints; the other four use literal trade prices.")
    lines.append("  - All 'peak' times/prices are exactly as supplied and labelled approximate; no tick between "
                  "supplied observations was invented.")
    lines.append("  - CONFIRMED_REVERSAL (the 4th PRF state) is not evaluated -- it depends on the existing "
                  "deterioration engine's structure signals, which are outside the supplied order-flow data. "
                  "PROFIT_PROTECTION is used as the actionable trigger point in this simulation instead.")
    lines.append("  - Session-level PRF comparison swaps each trade's result independently; it does not model "
                  "capital being freed up and re-deployed into a different trade.")
    lines.append("  - This is a single session (2026-09-18). No conclusion here should be read as 'PRF is "
                  "better/worse' in general -- see the SENSITIVITY ANALYSIS table for how fragile/robust the "
                  "specific default thresholds are just on this one day's five examples.")

    # ---- Final answer ----
    lines.append("\n--- ANSWER TO THE KEY QUESTION ---")
    others_triggered = [r["trade"] for r in comparison_rows
                         if r["trade"] != "GLOO #2" and r["prf_triggered"] == "YES"]
    gloo_ok = gloo_row and gloo_row["prf_triggered"] == "YES"
    if gloo_ok and not others_triggered:
        lines.append(
            "  YES, at the default thresholds tested (imbalance>=0.70, 2 consecutive negative reads): "
            "PRF raised PROFIT_PROTECTION on GLOO #2 before the ~11:02 ET breakdown printed in the supplied "
            "data, while BNC #1, SBET, FWDI and SECZ #2 all remained in NORMAL across their supplied windows "
            "-- no false triggers on those four within the data given."
        )
    else:
        lines.append(f"  NOT cleanly: GLOO #2 triggered={gloo_ok}, "
                      f"false triggers on: {others_triggered or 'none'}.")
    lines.append("  This holds specifically for the supplied windows and default thresholds; see sections "
                  "5-7 above for how it changes under the sensitivity sweep.")

    lines.append("\n--- SESSION-LEVEL SUMMARY ---")
    lines.append(f"  1. ACTUAL SESSION total:            ${actual_total:+.2f}")
    lines.append(f"  2. PRF SIMULATION total (swapped):  ${prf_session_total:+.2f}  "
                  f"(diff ${prf_session_total - actual_total:+.2f})")
    lines.append("  3. MFE SIMULATION: evaluated per-trade above (30/40/50% giveback), not combined into a "
                  "single session total per the brief.")

    return "\n".join(lines)


# ==================================================================
# 9. MAIN
# ==================================================================

def main():
    ap = argparse.ArgumentParser(description=__doc__)
    ap.add_argument("--date", default="2026-09-18",
                     help="Trade log date to load (order-flow samples in this script are hardcoded for 2026-09-18 only)")
    args = ap.parse_args()

    trades = load_trade_log(args.date)
    order_flow = load_order_flow_samples()
    annotated = associate_order_flow_to_trades(trades, order_flow)

    for trade_id, info in annotated.items():
        print(generate_trade_report(trade_id, info))

    sensitivity_df = run_sensitivity_analysis(annotated)
    print("\n" + "=" * 70)
    print("SENSITIVITY ANALYSIS (imbalance threshold x negative-confirm reads)")
    print("=" * 70)
    print(sensitivity_df.to_string(index=False))

    print(generate_session_report(args.date, trades, annotated, sensitivity_df))


if __name__ == "__main__":
    main()
