"""
scanner.py

The candidate-scoring model from docs/breakout_bot_design_conversation.
pdf (Part II, sections 2-4). Ported from screener/premarket's
breakout_scanner.py (which ran this same scoring as a shadow/comparison
scanner alongside sip_bot) -- here it's the primary and only scanner
for this project, not a comparison run.

Deliberately scoped to ONLY the scanner piece, per "build piece by
piece, checking performance as we go": the full breakout_bot design
also calls for a 1-min-bar / 5-min-rolling-window state machine (WATCH
-> BUILDING -> PRE_BREAKOUT -> BREAKOUT -> CONFIRMED) driven by live
streaming data, plus independent entry/exit modules -- none of that
exists yet. This module only answers "which stocks deserve attention
today," never "which stock will go up" or "buy this."

CANDIDATE SCORE weighting is the PDF's own (Part II section 3): 30%
relative volume, 20% premarket volume, 20% gap/price movement, 15%
premarket price strength, 15% distance to important resistance.

Deliberately NO hard reject gates beyond universe membership (see
universe.py) and a minimum-bars data-availability check -- every
prefiltered symbol with enough bars gets ranked, top N taken. This is
an intentional design choice, not an oversight: sip_bot's own
premarket_scanner.py hard-rejects on RVOL/spread/volatility floors,
which on a genuinely thin market-wide day (2026-09-16) rejected all
540 prefiltered symbols and produced ZERO candidates for the session.
A purely continuous score never has that failure mode.
"""

from datetime import datetime, timedelta, timezone

from config_loader import get_config
from logger_setup import get_logger
from indicators import relative_volume
import universe
import data_store
import volatility

log = get_logger("scanner")


def _clamp(x, lo=0.0, hi=1.0):
    return max(lo, min(hi, x))


def compute_range_20d_high(client, symbols: list, daily_atr_out: dict = None) -> dict:
    """
    {symbol: 20-day high} -- one bulk multi-symbol daily-bars call.
    Feeds the distance-to-resistance sub-score's third priority level
    (premarket high -> previous-day high -> 20-day high, per the
    design doc's resistance priority order, Part I page 16 / Part II
    section 10).

    A symbol absent from the result (too new, delisted, or the request
    failed) is simply missing -- _distance_to_resistance_pct() treats a
    missing level the same as one that's below current price: skip to
    the next priority level, never reject the symbol.
    """
    now = datetime.now(timezone.utc)
    start = now - timedelta(days=40)
    daily_bars = client.get_daily_bars_bulk(symbols, start, now)

    out = {}
    today = datetime.now(timezone.utc).date()
    for symbol, bars in daily_bars.items():
        if daily_atr_out is not None:
            # [2026-09-23] Same fetch, reused for volatility.py's stop floor --
            # completed sessions only (a mid-session call would otherwise
            # include today's partial bar).
            done = [b for b in bars if _bar_date(b) < today]
            datr = volatility.daily_atr(done)
            if datr:
                daily_atr_out[symbol] = datr
        bars = bars[-20:]
        if not bars:
            continue
        out[symbol] = max(b["h"] for b in bars)
    return out


def _is_premarket(b) -> bool:
    from zoneinfo import ZoneInfo
    sched = get_config()["schedule"]
    t = b["t"] if hasattr(b["t"], "astimezone") else datetime.fromisoformat(str(b["t"]))
    t_et = t.astimezone(ZoneInfo(sched["timezone"]))
    oh, om, _ = (int(x) for x in sched["market_open_time"].split(":"))
    return (t_et.hour, t_et.minute) < (oh, om)


def _bar_date(b):
    t = b.get("t")
    return t.date() if hasattr(t, "date") else datetime.fromisoformat(str(t)).date()


def _gap_pct(price: float, prev_close: float) -> float:
    """(premarket_last_price - prev_close) / prev_close."""
    if not prev_close:
        return 0.0
    return (price - prev_close) / prev_close * 100.0


def _premarket_price_strength(bars: list) -> float:
    """
    Splits the premarket bar window into thirds (early/mid/late) and
    compares the late segment's average close against the early
    segment's -- a steady upward trend across the whole premarket
    session scores well; flat or declining scores near zero.

    Deliberately does NOT try to distinguish a steady climb from a
    dip-then-reverse-up (V) shape -- both trend up from early to late
    and would score similarly. This matches the design doc's decision
    to drop setup-type classification entirely (sip_bot's own
    backtesting found the elaborate composite/setup-aware score had
    near-zero correlation with outcomes).

    Returns a % change (early-segment avg -> late-segment avg), NOT
    yet normalized to a 0-1 sub-score.
    """
    n = len(bars)
    if n < 3:
        return 0.0
    third = max(1, n // 3)
    early = bars[:third]
    late = bars[-third:]
    early_avg = sum(b["c"] for b in early) / len(early)
    late_avg = sum(b["c"] for b in late) / len(late)
    if not early_avg:
        return 0.0
    return (late_avg - early_avg) / early_avg * 100.0


def _distance_to_resistance_pct(price: float, pm_high: float, prev_day_high: float,
                                 range_20d_high: float):
    """
    Nearest overhead level still ABOVE current price, checked in
    priority order (premarket high -> previous-day high -> 20-day
    high). Returns (distance_pct, level_used) -- distance_pct is None
    if price has already cleared every known level ("clear air"): a
    breakout already in progress shouldn't be penalized for having no
    ceiling left to measure against, so that case is full credit at
    the call site rather than a zero/undefined distance.

    More room below the nearest ceiling scores BETTER here, not worse
    -- sitting right at a resistance level is the risky spot, not the
    strong one.
    """
    for level in (pm_high, prev_day_high, range_20d_high):
        if level and level > price:
            return (level - price) / level * 100.0, level
    return None, None


def score_breakout_candidate(symbol: str, bars: list, pm_high: float, pm_low: float,
                              avg_vol_baseline: float, prev_close: float,
                              prev_day_high: float, range_20d_high: float) -> dict:
    """
    bars: 1-min premarket bars, oldest first, each {"t","o","h","l","c","v"}

    Returns {"symbol", "candidate_score", "breakdown", "metrics"}.
    candidate_score is 0-100.
    """
    cfg = get_config()["scanner"]
    w = cfg["weights"]

    closes = [b["c"] for b in bars]
    volumes = [b["v"] for b in bars]
    price = closes[-1]
    premarket_volume = sum(volumes)

    rvol = relative_volume(premarket_volume, avg_vol_baseline)
    gap_pct = _gap_pct(price, prev_close)
    price_strength_pct = _premarket_price_strength(bars)
    dist_pct, resistance_level = _distance_to_resistance_pct(
        price, pm_high, prev_day_high, range_20d_high)

    # gap_or_price_movement / premarket_price_strength clamp to [-1, 1],
    # not [0, 1] -- a decline scores the full NEGATIVE weight, symmetric
    # with how a matching gain earns full positive credit. A stock
    # actively falling should score WORSE than a flat one, not the same.
    breakdown = {
        "relative_volume": w["relative_volume"] * _clamp(rvol / cfg["rvol_full_credit_multiple"]),
        "premarket_volume": w["premarket_volume"] * _clamp(premarket_volume / cfg["premarket_volume_full_credit"]),
        "gap_or_price_movement": w["gap_or_price_movement"] * _clamp(
            gap_pct / cfg["gap_full_credit_pct"], -1.0, 1.0),
        "premarket_price_strength": w["premarket_price_strength"] * _clamp(
            price_strength_pct / cfg["price_strength_full_credit_pct"], -1.0, 1.0),
    }
    if dist_pct is None:
        breakdown["distance_to_resistance"] = w["distance_to_resistance"]  # clear air -- full credit
    else:
        breakdown["distance_to_resistance"] = w["distance_to_resistance"] * _clamp(
            dist_pct / cfg["resistance_full_credit_distance_pct"])

    total = sum(breakdown.values())
    max_possible = sum(w.values())
    candidate_score = _clamp(total / max_possible * 100.0, 0, 100) if max_possible else 0.0

    return {
        "symbol": symbol,
        "candidate_score": round(candidate_score, 2),
        "breakdown": {k: round(v, 2) for k, v in breakdown.items()},
        "metrics": {
            "price": price,
            "previous_close": prev_close,
            "gap_pct": round(gap_pct, 2),
            "premarket_volume": premarket_volume,
            "avg_daily_volume": avg_vol_baseline,
            "rvol": round(rvol, 2),
            "premarket_high": pm_high,
            "premarket_low": pm_low,
            "previous_day_high": prev_day_high,
            "range_20d_high": range_20d_high,
            "price_strength_pct": round(price_strength_pct, 2),
            "distance_to_resistance_pct": round(dist_pct, 2) if dist_pct is not None else None,
            "resistance_level_used": resistance_level,
        },
    }


def scan(client, prefiltered: list, volume_baselines: dict, prev_day_highs: dict,
         prev_closes: dict, range_20d_highs: dict, lookback_hours: float = 6,
         candidate_count: int = None, candidates_file_suffix: str = "") -> list:
    """
    One-shot scan over an already-prefiltered symbol list. No hard
    reject gates beyond a minimum-bars data-availability check and
    requiring a real prev_close (gap_pct's denominator) -- every other
    prefiltered symbol gets ranked, top candidate_count taken purely by
    candidate_score.
    """
    cfg = get_config()["scanner"]
    candidate_count = candidate_count or cfg.get("top_candidate_count", 30)

    now = datetime.now(timezone.utc)
    start = now - timedelta(hours=lookback_hours)

    scored = []
    skipped_no_prev_close = 0
    for symbol in prefiltered:
        bars = universe._bars_to_dicts(client.get_minute_bars(symbol, start, now))
        if len(bars) < 3:
            continue
        prev_close = prev_closes.get(symbol)
        if not prev_close:
            skipped_no_prev_close += 1
            continue

        pm_high = max(b["h"] for b in bars)
        pm_low = min(b["l"] for b in bars)
        avg_vol_baseline = volume_baselines.get(symbol) or get_config()["universe"]["min_avg_daily_volume"]

        result = score_breakout_candidate(
            symbol, bars, pm_high, pm_low, avg_vol_baseline, prev_close,
            prev_day_highs.get(symbol), range_20d_highs.get(symbol))
        if cfg.get("true_premarket_high"):
            # [2026-09-24] After the open, the 6h lookback includes regular-
            # session bars, so pm_high above is really "high so far today" --
            # a rescan then saved that as premarket_high (58 of 68 rescanned
            # symbols on 9/24; DNA's 'premarket high' read 9.35-9.78 vs a true
            # 8.68), handing the trade plan a fake resistance at the current
            # high. The saved LEVEL now uses premarket bars only; scoring above
            # is unchanged.
            pm_bars = [b for b in bars if _is_premarket(b)]
            # [2026-09-25] No premarket trades -> no premarket level at all
            # (WNC/MAAS/RES on 9/25 fell back to the high-so-far-today).
            result.setdefault("metrics", {})["premarket_high"] = (
                max(b["h"] for b in pm_bars) if pm_bars else None)
        scored.append(result)

    scored.sort(key=lambda r: r["candidate_score"], reverse=True)
    top = scored[:candidate_count]

    log.info(f"[SCANNER] Selected {len(top)} candidates (of {len(scored)} scored, "
             f"{len(prefiltered)} prefiltered, {skipped_no_prev_close} skipped for missing prev_close)")
    data_store.write_candidates(top, suffix=candidates_file_suffix)
    return top
