"""
premarket_scanner.py

Runs at ~09:00 ET. Scans the tradable universe ($5-$15, liquid enough)
and produces a ranked list of candidates using scorer.py. Called by
monitor.py, but also runnable standalone for testing:

    python premarket_scanner.py

It does NOT buy anything — this module only discovers and ranks.
"""

import sys
from datetime import datetime, timedelta, timezone

from config_loader import get_config
from logger_setup import get_logger
from alpaca_client import get_client
from scorer import score_premarket_candidate
from indicators import atr
import symbol_filters
import data_store

log = get_logger("premarket_scanner")


def _bars_to_dicts(bars) -> list:
    out = []
    for b in bars:
        out.append({
            "t": b.timestamp, "o": float(b.open), "h": float(b.high),
            "l": float(b.low), "c": float(b.close), "v": float(b.volume),
        })
    return out


def get_universe_symbols(client) -> list:
    """
    Pulls active, tradable US equities and applies:
      1. the include/exclude_symbols lists from config
      2. ETF/fund/trust exclusion (symbol_filters.is_etf_or_fund)
      3. leveraged/inverse ("multiplier") product exclusion
      4. company-name syllable-count cap

    Alpaca's asset_class=US_EQUITY includes ETFs/ETNs, so filters 2-4
    are done on the asset's `name` field via symbol_filters.py.
    """
    cfg = get_config()["universe"]
    if cfg.get("include_only_symbols"):
        return cfg["include_only_symbols"]

    assets = client.get_tradable_assets()
    excluded = set(cfg.get("exclude_symbols", []))

    candidates = [a for a in assets if getattr(a, "tradable", False) and a.symbol not in excluded]
    before = len(candidates)
    survivors = symbol_filters.filter_symbol_list(candidates, log_rejections=True)
    log.info(f"[PREMARKET] Symbol filters (ETF/leveraged/name-complexity) removed "
             f"{before - len(survivors)} of {before} candidates")
    return survivors


def prefilter_by_snapshot(client, symbols: list, baseline_out: dict = None,
                           prev_day_high_out: dict = None, prev_close_out: dict = None) -> list:
    """
    Cheap first pass using snapshot data to cut the universe down to a
    manageable set before pulling minute bars for full scoring. Filters
    on price band and a basic volume floor.

    baseline_out: [Phase 5 -- 2026-09-08, BUGFIXED 2026-09-15] if a dict is
        passed, this function populates it with {symbol: real_prior_day_
        volume} as a side effect (mutated in place, return value
        unchanged). Originally sourced from `snap.daily_bar.volume` --
        that was itself a fix for an older bug (every RVOL calculation
        comparing against the same flat universe.min_avg_daily_volume
        constant regardless of a symbol's real normal volume) but
        introduced a worse one: `daily_bar` is the CURRENT session's bar,
        which Alpaca already starts accumulating from 4:00am ET extended
        hours -- so by the ~09:00 scan time it's not a historical
        baseline at all, it's just today's premarket volume so far. A few
        minutes later scan() sums those same premarket bars into
        total_volume, so RVOL ended up comparing today's premarket volume
        to itself (~1.0x for nearly everything, real catalyst or not),
        silently killing the scanner's main volume-interest signal from
        2026-09-08 onward. Now sourced from `snap.previous_daily_bar.
        volume` instead (the prior COMPLETE session's real volume) --
        same fix pattern already used for prev_day_high_out below. Omit
        baseline_out (the default) to reproduce this function's exact
        pre-Phase-5 behavior -- callers that don't pass it are completely
        unaffected.

    prev_day_high_out: [FEATURE 2026-09-11, attribute name BUGFIXED
        2026-09-15] same pattern as baseline_out above, for
        fast_prediction_engine.py's resistance-context layer. Populated
        with {symbol: previous session's high} from Alpaca's snapshot --
        reads snap.previous_daily_bar (unambiguous, always "the prior
        completed session") rather than snap.daily_bar (which silently
        shifts to mean TODAY once regular hours start -- see baseline_
        out's history above for what trusting daily_bar here actually
        costs). Originally read via getattr(snap, "prev_daily_bar", None)
        -- alpaca-py's actual Snapshot field is `previous_daily_bar`, so
        that getattr's default silently fired every single time and this
        was empty from the day it shipped, with nothing to surface the
        typo since a missing value here fails soft by design (see below).
        Purely additive/optional context, never a required dependency --
        per explicit instruction, its absence must never block or reject
        a symbol; fast_prediction_engine.py falls back to premarket high/
        session high/detected support-resistance when this is missing.

    prev_close_out: [FEATURE 2026-09-16] same pattern as prev_day_high_out
        above -- {symbol: previous session's close} for breakout_scanner.
        py's gap_pct (needs the prior close as its denominator; unlike
        prev_day_high_out, breakout_scanner.py treats a missing value here
        as "can't score this symbol" rather than failing open, since gap%
        is a required, weighted input there -- see breakout_scanner.scan()).
    """
    cfg = get_config()["universe"]
    scoring_cfg = get_config()["premarket_scoring"]
    survivors = []

    snapshots = client.get_snapshots(symbols)
    if not snapshots:
        return survivors

    for symbol, snap in snapshots.items():
        try:
            latest_trade = snap.latest_trade
            daily_bar = snap.daily_bar
            if latest_trade is None:
                continue
            price = float(latest_trade.price)
            if not (cfg["price_min"] <= price <= cfg["price_max"]):
                continue
            today_vol_so_far = float(daily_bar.volume) if daily_bar else 0
            if today_vol_so_far and today_vol_so_far < cfg["min_avg_daily_volume"] * 0.05:
                # extremely thin so far today even accounting for premarket partial data
                continue
            prev_bar = getattr(snap, "previous_daily_bar", None)
            if prev_bar is not None and prev_bar.volume:
                prev_volume = float(prev_bar.volume)
                # [BUGFIX 2026-09-16] universe.min_avg_daily_volume was never
                # actually enforced against a real historical value -- only
                # used as an RVOL denominator fallback when a baseline was
                # MISSING. A symbol with a real but tiny previous-day volume
                # (e.g. BACC/ACGC on 2026-09-16: ~177/~208 shares) slipped
                # through as "eligible" and then produced nonsense 100x-2500x
                # RVOL readings off that near-zero denominator, floating to
                # the top of quiet-morning scans. A confirmed real reading
                # below the floor now rejects outright, same as any other
                # universe filter -- a MISSING baseline (prev_bar.volume
                # falsy) still fails open and reaches survivors below,
                # unchanged.
                if prev_volume < cfg["min_avg_daily_volume"]:
                    continue
                if baseline_out is not None:
                    baseline_out[symbol] = prev_volume
            if prev_day_high_out is not None and prev_bar is not None and prev_bar.high:
                prev_day_high_out[symbol] = float(prev_bar.high)
            if prev_close_out is not None and prev_bar is not None and prev_bar.close:
                prev_close_out[symbol] = float(prev_bar.close)
            survivors.append(symbol)
        except (AttributeError, TypeError):
            continue

    return survivors


def compute_volatility_baseline(client, symbols: list) -> dict:
    """
    [FEATURE 2026-09-15] {symbol: trailing-N-day average daily true
    range, as a % of price} -- one bulk multi-symbol daily-bars call
    (get_daily_bars_bulk, chunked) instead of one REST call per symbol.
    Feeds scorer.py's daily_atr_pct / universe.min_daily_atr_pct floor.

    Uses DAILY bars specifically, not the 1-min premarket/intraday bars
    already being pulled elsewhere -- real-world calibration (2026-09-15)
    found per-1-minute-bar ATR% does NOT separate a structurally dead
    BDC/SPAC from a real momentum name (both read ~0.1-0.3%, dominated
    by bar-granularity noise); only the multi-day range history does
    (WHF 2.23%, APXT 0.09% vs. 4.3%-21.9% for real same-day candidates).

    A symbol absent from the result (too new, delisted, or the request
    failed) is simply missing -- callers must treat that as "unknown,"
    never as "confirmed low volatility" (see scorer.py's daily_atr_pct
    fail-open contract).
    """
    cfg = get_config()["universe"]
    lookback_days = cfg.get("volatility_lookback_days", 30)
    now = datetime.now(timezone.utc)
    # [calendar days, not trading days] pad well past lookback_days to
    # comfortably cover weekends/holidays and still get enough real
    # trading sessions for atr()'s 14-period window.
    start = now - timedelta(days=lookback_days * 2 + 10)
    daily_bars = client.get_daily_bars_bulk(symbols, start, now)

    out = {}
    for symbol, bars in daily_bars.items():
        bars = bars[-lookback_days:]
        if len(bars) < 10:
            continue  # too little history to call this "known" -- stays absent, fails open
        price = bars[-1]["c"]
        if not price:
            continue
        atr_val = atr(bars, period=min(14, len(bars) - 1))
        out[symbol] = atr_val / price * 100.0
    return out


def scan(candidate_count: int = None, prefiltered: list = None, lookback_hours: float = 6,
         volume_baselines: dict = None, prev_day_highs: dict = None,
         volatility_baselines: dict = None) -> list:
    """
    Full premarket scan. Returns a ranked list of scored candidate dicts,
    truncated to `candidate_count` (defaults to config's
    premarket_candidate_count, i.e. 20).

    [FEATURE 2026-08-17] `prefiltered` lets a caller (monitor.py) supply
    an already-computed universe+snapshot-filtered symbol list instead
    of this function pulling and filtering the full tradable-asset list
    again. monitor.py caches this list once at 09:00 and reuses it for
    every subsequent 30-minute intraday full rescan, so the expensive
    get_tradable_assets() + bulk snapshot call only ever happens once
    per day rather than once per rescan. `lookback_hours` defaults to 6
    (covers the overnight/premarket session) but monitor.py's intraday
    rescans pass a much shorter window (intraday_health.
    full_rescan_lookback_hours, default 1) since a 6-hour window is
    unnecessary once the regular session is already underway.

    volume_baselines: [Phase 5 -- new, optional] {symbol: real_prior_day_
        volume} from prefilter_by_snapshot()'s baseline_out -- when a
        symbol has a real entry here, its own historical volume is used
        for RVOL instead of the flat universe.min_avg_daily_volume
        constant. Omit (the default) to reproduce this function's exact
        pre-Phase-5 behavior.

    prev_day_highs: [FEATURE 2026-09-11] {symbol: previous session's high}
        from prefilter_by_snapshot()'s prev_day_high_out -- threaded
        through to each result dict as result["prev_day_high"] for
        fast_prediction_engine.py's resistance context. Optional, same
        fail-soft contract as volume_baselines above.

    volatility_baselines: [FEATURE 2026-09-15] {symbol: trailing-N-day
        avg daily ATR%} from compute_volatility_baseline() -- feeds
        scorer.py's daily_atr_pct floor (universe.min_daily_atr_pct),
        which hard-rejects structurally low-volatility names (BDCs,
        pre-merger SPACs) below. A symbol missing from this dict is
        "unknown," not "confirmed low volatility" -- never rejected on
        that basis alone, same fail-open contract as the other two
        baseline dicts above.
    """
    cfg = get_config()
    candidate_count = candidate_count or cfg["candidates"]["premarket_candidate_count"]
    client = get_client()

    log.info("[PREMARKET] Starting scan")
    if prefiltered is None:
        universe = get_universe_symbols(client)
        log.info(f"[PREMARKET] Universe size after asset filtering: {len(universe)}")
        prefiltered = prefilter_by_snapshot(client, universe)
        log.info(f"[PREMARKET] Prefiltered to {len(prefiltered)} symbols in price/volume band")

    now = datetime.now(timezone.utc)
    start = now - timedelta(hours=lookback_hours)

    scored = []
    rvol_near_miss = []  # [FEATURE 2026-09-16] see starvation-fallback block below
    for symbol in prefiltered:
        bars = _bars_to_dicts(client.get_minute_bars(symbol, start, now))
        if len(bars) < 3:
            log.debug(f"[PREMARKET] {symbol} rejected: insufficient bar data ({len(bars)} bars)")
            continue

        pm_high = max(b["h"] for b in bars)
        pm_low = min(b["l"] for b in bars)
        quote = client.get_latest_quote(symbol)
        bid = float(quote.bid_price) if quote and quote.bid_price else None
        ask = float(quote.ask_price) if quote and quote.ask_price else None

        # [Phase 5] Real per-symbol historical volume when available
        # (see volume_baselines docstring above), falling back to the old
        # flat universe constant only for a symbol this scan has no real
        # baseline for yet (e.g. a fresh replacement candidate).
        avg_vol_baseline = (volume_baselines or {}).get(symbol) or get_config()["universe"]["min_avg_daily_volume"]

        daily_atr_pct = (volatility_baselines or {}).get(symbol)
        result = score_premarket_candidate(symbol, bars, pm_high, pm_low, avg_vol_baseline, bid, ask,
                                            daily_atr_pct=daily_atr_pct)
        result["pm_high"] = pm_high
        result["pm_low"] = pm_low
        result["prev_day_high"] = (prev_day_highs or {}).get(symbol)

        if not result["flags"].get("meets_min_volume", False):
            log.debug(f"[PREMARKET] {symbol} rejected: below min premarket volume")
            continue
        # [FEATURE 2026-09-15] Structural volatility floor -- see
        # volatility_baselines docstring above / config.json's
        # universe.min_daily_atr_pct note for the WHF/APXT real cases
        # this catches (a BDC and a pre-merger SPAC, both of which
        # cleared every other filter here).
        if not result["flags"].get("meets_min_volatility", True):
            log.debug(f"[PREMARKET] {symbol} rejected: below min daily ATR% "
                      f"(structurally low volatility, e.g. BDC/SPAC)")
            continue
        if not result["flags"].get("spread_ok", True):
            log.debug(f"[PREMARKET] {symbol} rejected: spread too wide")
            continue
        # [BUGFIX 2026-09-15, reordered 2026-09-16] meets_min_rvol was
        # computed by scorer.py and exposed in flags, but never actually
        # checked here -- a quiet stock with no real volume interest today
        # could still rank into the top 20 purely on non-volume factors
        # (structure/tightness/consistency). Enforced as a hard gate, same
        # pattern as the volume/spread checks around it. Checked LAST among
        # the hard gates (moved 2026-09-16) so rvol_near_miss below only
        # ever holds symbols that already cleared every OTHER real
        # liquidity/volatility/spread requirement -- a genuine "everything
        # else checks out, just short on RVOL today" pool, not a dumping
        # ground for symbols that were going to be rejected anyway.
        if not result["flags"].get("meets_min_rvol", False):
            log.debug(f"[PREMARKET] {symbol} rejected: below min RVOL")
            rvol_near_miss.append(result)
            continue

        scored.append(result)
        data_store.append_premarket_snapshot(symbol, result)

    # [FEATURE 2026-09-16] Starvation fallback -- see 2026-09-16 session
    # review: on a premarket morning with almost no unusual volume anywhere
    # (2026-09-16 itself: the single highest-RVOL real stock in the entire
    # universe read 0.92x, everything else under 0.5x), the hard meets_min_
    # rvol gate rejected all 540 prefiltered symbols and the scanner
    # produced zero candidates for the whole session -- the same
    # hard-floor-with-no-fallback starvation shape this project hit before
    # with the old structure_score/entry_score engines. Confirmed no single
    # static threshold fixes this (checked thresholds down to 0.1 that
    # day -- still only ~10 names, dominated by data artifacts) since the
    # problem is genuinely thin premarket activity market-wide, not a
    # miscalibrated cutoff. Rather than lower min_rvol (which would only
    # matter on already-thin days -- every normal day so far clears 1.5x
    # comfortably with room to spare), backfill with the best-available
    # candidates BY SCORE from rvol_near_miss when the real qualified pool
    # is too small, capped at rvol_starvation_fallback.min_candidates.
    # These are explicitly flagged (flags.rvol_floor_waived = True) rather
    # than silently blended in -- downstream code and log review can always
    # tell a genuine RVOL-qualified candidate from a starvation backfill.
    # This only widens the MONITORING pool; fast_entry_gate.py's live
    # confirmation checks still gate any actual trade, so a backfilled
    # candidate still has to prove real live momentum before an entry
    # fires.
    fallback_cfg = get_config().get("rvol_starvation_fallback", {})
    if fallback_cfg.get("enabled", True) and rvol_near_miss:
        min_candidates = fallback_cfg.get("min_candidates", 5)
        if len(scored) < min_candidates:
            rvol_near_miss.sort(key=lambda r: r["total_score"], reverse=True)
            needed = min_candidates - len(scored)
            backfilled = rvol_near_miss[:needed]
            for r in backfilled:
                r["flags"]["rvol_floor_waived"] = True
                scored.append(r)
                data_store.append_premarket_snapshot(r["symbol"], r)
            log.warning(
                f"[PREMARKET] RVOL starvation fallback: only "
                f"{len(scored) - len(backfilled)} candidate(s) cleared min_rvol; "
                f"backfilled {len(backfilled)} more by score (rvol_floor_waived=True) "
                f"to reach {len(scored)}"
            )

    scored.sort(key=lambda r: r["total_score"], reverse=True)
    top = scored[:candidate_count]

    for r in top:
        log.info(f"[PREMARKET] {r['symbol']} score={r['total_score']}")

    log.info(f"[PREMARKET] Selected {len(top)} candidates")
    data_store.write_premarket_candidates(top)
    return top


if __name__ == "__main__":
    results = scan()
    for r in results:
        print(f"{r['symbol']:6s} score={r['total_score']:6.2f} "
              f"price={r['metrics']['price']:.2f} rvol={r['metrics']['rvol']:.2f}")
