#!/usr/bin/env python3
"""Backtest of the strategies taught by "Stock Burner" (Dinesh Kirola), for the stockburner-audit reel.

Rules live in projects/stockburner-audit/sb_rules.json (edit there, not here). Every rule there carries the
quote + timestamp it comes from (see BACKTEST_PLAN.md). Dated copies in projects/stockburner-audit/prereg/.
Nothing in here is tuned on results.

Data (all free; nseindia.com / niftyindices.com are geo-blocked from this network, see projects/.research/india_data.md):
  * Spot index 1-minute (NIFTY 50, NIFTY BANK, NIFTY FIN SERVICE), 2015-01-09 -> 2026-05-15:
    Kaggle debashis74017/nifty-50-minute-data  (`build`)  -> data_raw/<SYM>_m1.npz
  * Second source for the spot index (2021-05 -> 2026-07): Hugging Face thetrademarkk/india-index-options-1m
    index/*.parquet (`hfindex`, needs pyarrow)  -> data_raw/<SYM>_hf_m1.npz
  * NIFTY weekly options 1-minute (same HF set, 2021-05 -> 2026-08) for the option-execution check (`options`,
    needs pyarrow). CC BY-NC 4.0: research input / cross-check only, aggregate results only.

Execution model:
  * Signals on candles of the SPOT index built from 1-minute bars, aligned to 09:15 IST and never spanning two
    days (what Kite / Dhan / TradingView show). EMAs on candle closes, continuous across days (TradingView style).
  * He trades index OPTIONS but draws the setup, stop and 1:3 target on the spot chart ("मैं स्पॉट में 1:3 लगा
    रहा", Ma1HnAXxww4 18:56). So the win/loss of a trade is decided on spot: entry at the close of the entry
    candle, stop/target followed minute by minute from the next minute. Stop first when stop and target share a
    minute; on a gap the stop fills at the minute's open. Intraday only: no entry after 15:00 IST, flat at 15:20
    (our assumption for an intraday option buyer, stated in the plan).
  * Costs: per trade, the statutory charges in force on that date (STT, NSE transaction charge, GST, stamp duty,
    SEBI fee) on an ATM premium plus a stated slippage per side, in option-premium points per unit, converted to
    spot points with delta 0.5 (ATM); plus the flat ₹20-per-order brokerage (+GST) as a share of the fixed ₹1,000
    risk. Zero-cost run for every headline variant.
    The real option P&L (actual 1-minute premiums) is computed separately for NIFTY 2021-05 -> 2026-05 (`options`).
  * Money: ₹1,00,000 account, a fixed ₹1,000 (1% of the start) risked on every trade, no compounding; if the balance
    hits 0 we report the blown date instead of negative money. A compounded line is reported separately.

Usage:
  .venv/bin/python tools/sb_backtest.py build --symbols NIFTY,BANKNIFTY,FINNIFTY
  .venv/bin/python tools/sb_backtest.py run --symbols NIFTY,BANKNIFTY [--strategies nine20]
  .venv/bin/python tools/sb_backtest.py spot --symbols NIFTY --variant "<name>" --n 5
  .venv/bin/python tools/sb_backtest.py premise --symbols NIFTY,BANKNIFTY
  uv run --with pyarrow --with numba --with pandas python tools/sb_backtest.py hfindex|crosscheck|options ...
  .venv/bin/python tools/sb_backtest.py report --tag NIFTY_BANKNIFTY
"""
from __future__ import annotations

import argparse
import hashlib
import json
import math
import random
import sys
from dataclasses import dataclass
from pathlib import Path

import numpy as np
import pandas as pd

try:
    from numba import njit
except ImportError:  # pragma: no cover - plain Python fallback (slow but identical)
    def njit(*a, **k):
        if a and callable(a[0]):
            return a[0]
        return lambda f: f

ROOT = Path(__file__).resolve().parents[1]
PROJ = ROOT / "projects/stockburner-audit"
RAW = PROJ / "data_raw"
RULES = PROJ / "sb_rules.json"
OUT_DIR = ROOT / "engine/public/projects/stockburner-audit/data"
MD = PROJ / "BACKTEST.md"
IST = "Asia/Kolkata"
KAGGLE_FILES = {"NIFTY": ["NIFTY_50_minute.csv.gz", "NIFTY_50_minute.csv"],
                "BANKNIFTY": ["NIFTY_BANK_minute.csv.gz", "NIFTY_BANK_minute.csv"],
                "FINNIFTY": ["NIFTY_FIN_SERVICE_minute.csv.gz", "NIFTY_FIN_SERVICE_minute.csv"]}
KAGGLE_DIRS = [RAW / "kaggle", ROOT / "projects/.research/data_samples/kaggle"]
HF_BASE = "https://huggingface.co/datasets/thetrademarkk/india-index-options-1m/resolve/main"


# ------------------------------------------------------------------ data ---
@dataclass
class Market:
    symbol: str
    t: np.ndarray        # int64 unix seconds of the minute start (UTC); IST = UTC + 5:30, no DST
    o: np.ndarray; h: np.ndarray; l: np.ndarray; c: np.ndarray
    day: np.ndarray      # int64 trading-day id (days since epoch; IST date for NSE, UTC date for crypto)
    mod: np.ndarray      # int16 minute of day (IST for NSE: 09:15 = 555; UTC for crypto)
    source: str = "kaggle"
    kind: str = "nse"    # "nse" (09:15-15:30 IST session) or "crypto" (24/7, candles aligned to UTC)

    @property
    def base_mod(self) -> int:
        return 555 if self.kind == "nse" else 0

    @property
    def tzoff(self) -> int:
        return 19800 if self.kind == "nse" else 0

    def __len__(self) -> int:
        return len(self.t)


def _to_market(sym: str, ts_ist: pd.Series, o, h, l, c, source: str) -> Market:
    ts = pd.to_datetime(ts_ist)
    if ts.dt.tz is not None:
        ts = ts.dt.tz_convert(IST).dt.tz_localize(None)
    loc = ts.to_numpy("datetime64[s]").astype(np.int64)            # IST wall clock as if UTC
    t = loc - 19800                                                   # -> real UTC
    day = loc // 86400
    mod = ((loc % 86400) // 60).astype(np.int16)
    return Market(sym, t, np.asarray(o, float), np.asarray(h, float), np.asarray(l, float), np.asarray(c, float),
                  day, mod, source)


def cmd_build(a, rules: dict) -> None:
    """Kaggle CSV -> data_raw/<SYM>_m1.npz (regular session 09:15-15:29 only; special sessions flagged) + gap report."""
    RAW.mkdir(parents=True, exist_ok=True)
    for sym in a.symbols.split(","):
        src = next((d / f for f in KAGGLE_FILES[sym] for d in KAGGLE_DIRS if (d / f).exists()), None)
        if src is None:
            print(f"{sym}: no Kaggle file found in {KAGGLE_DIRS}")
            continue
        df = pd.read_csv(src)
        df["date"] = pd.to_datetime(df["date"])
        nan_rows = int(df[["open", "high", "low", "close"]].isna().any(axis=1).sum())
        df = df.dropna(subset=["open", "high", "low", "close"])
        df = df.drop_duplicates("date").sort_values("date")
        n0 = len(df)
        hm = df["date"].dt.hour * 60 + df["date"].dt.minute
        df = df[(hm >= 555) & (hm <= 929)]
        bad = (df["high"] < df[["open", "close"]].max(axis=1)) | (df["low"] > df[["open", "close"]].min(axis=1)) | (df["low"] <= 0)
        df = df[~bad]
        cnt = df.groupby(df["date"].dt.date).size()
        m = _to_market(sym, df["date"], df["open"], df["high"], df["low"], df["close"], "kaggle")
        np.savez_compressed(RAW / f"{sym}_m1.npz", t=m.t, o=m.o, h=m.h, l=m.l, c=m.c)
        rep = {"source_file": str(src.relative_to(ROOT)), "rows_in": int(n0), "rows_kept": int(len(df)),
               "dropped_bad_ohlc": int(bad.sum()), "dropped_nan_rows": nan_rows, "first": str(df["date"].iloc[0]), "last": str(df["date"].iloc[-1]),
               "days": int(len(cnt)), "median_minutes_per_day": float(cnt.median()),
               "short_days_lt300": {str(k): int(v) for k, v in cnt[cnt < 300].items()},
               "days_300_369": {str(k): int(v) for k, v in cnt[(cnt >= 300) & (cnt < 370)].items()}}
        (RAW / f"{sym}_gaps.json").write_text(json.dumps(rep, indent=1))
        print(f"{sym}: {len(df):,} minutes, {len(cnt)} days, {rep['first']} .. {rep['last']}; "
              f"{len(rep['short_days_lt300'])} short days (<300 min) excluded from entries")


def cmd_hfindex(a, rules: dict) -> None:
    """HF trademarkk spot index parquet -> data_raw/<SYM>_hf_m1.npz (second source). Needs pyarrow."""
    import urllib.request
    for sym in a.symbols.split(","):
        p = RAW / "hf_trademarkk/index" / f"{sym}.parquet"
        if not p.exists():
            p.parent.mkdir(parents=True, exist_ok=True)
            urllib.request.urlretrieve(f"{HF_BASE}/index/{sym}.parquet", p)
        df = pd.read_parquet(p)
        tcol = next(c for c in df.columns if c.lower() in ("timestamp", "datetime", "date", "time", "ts"))
        ts = pd.to_datetime(df[tcol])
        if ts.dt.tz is None:
            ts = ts.dt.tz_localize(IST)
        df = df.assign(_ts=ts.dt.tz_convert(IST).dt.tz_localize(None)).drop_duplicates("_ts").sort_values("_ts")
        hm = df["_ts"].dt.hour * 60 + df["_ts"].dt.minute
        df = df[(hm >= 555) & (hm <= 929)]
        m = _to_market(sym, df["_ts"], df["open"], df["high"], df["low"], df["close"], "hf_trademarkk")
        np.savez_compressed(RAW / f"{sym}_hf_m1.npz", t=m.t, o=m.o, h=m.h, l=m.l, c=m.c)
        print(f"{sym}: HF {len(df):,} minutes {df['_ts'].iloc[0]} .. {df['_ts'].iloc[-1]}")


CRYPTO = ("BTCUSDT", "ETHUSDT")


def load_market(sym: str, source: str = "kaggle", start: str | None = None, end: str | None = None) -> Market:
    kind = "crypto" if sym in CRYPTO else "nse"
    if kind == "crypto":
        f = RAW / f"{sym}_m1.npz"
        source = "binance"
    else:
        f = RAW / (f"{sym}_m1.npz" if source == "kaggle" else f"{sym}_hf_m1.npz")
    z = np.load(f)
    t = z["t"]
    keep = np.ones(len(t), bool)
    if start:
        keep &= t >= int(pd.Timestamp(start, tz=IST).timestamp())
    if end:
        keep &= t < int((pd.Timestamp(end, tz=IST) + pd.Timedelta(days=1)).timestamp())
    off = 19800 if kind == "nse" else 0
    loc = t[keep] + off
    return Market(sym, t[keep], z["o"][keep], z["h"][keep], z["l"][keep], z["c"][keep],
                  loc // 86400, ((loc % 86400) // 60).astype(np.int16), source, kind)


def cmd_binance(a, rules: dict) -> None:
    """Binance spot 5-minute klines (data.binance.vision monthly zips in data_raw/binance) -> data_raw/<SYM>_m1.npz (5-min bars; positions are followed bar by bar, stop first)."""
    import zipfile
    for sym in a.symbols.split(","):
        parts = []
        for z in sorted((RAW / "binance").glob(f"{sym}-5m-*.zip")):
            with zipfile.ZipFile(z) as zf:
                with zf.open(zf.namelist()[0]) as fh:
                    df = pd.read_csv(fh, header=None, usecols=[0, 1, 2, 3, 4])
            ot = df[0].astype(np.int64)
            ot = np.where(ot > 10**14, ot // 1000, ot)          # 2025+ files are in microseconds
            parts.append(pd.DataFrame({"t": ot // 1000, "o": df[1], "h": df[2], "l": df[3], "c": df[4]}))
        d = pd.concat(parts).drop_duplicates("t").sort_values("t")
        np.savez_compressed(RAW / f"{sym}_m1.npz", t=d["t"].to_numpy(np.int64), o=d["o"].to_numpy(float),
                            h=d["h"].to_numpy(float), l=d["l"].to_numpy(float), c=d["c"].to_numpy(float))
        gaps = np.diff(d["t"].to_numpy()) // 60
        print(f"{sym}: {len(d):,} minutes {pd.Timestamp(int(d.t.iat[0]), unit='s')} .. {pd.Timestamp(int(d.t.iat[-1]), unit='s')}; "
              f"gaps > 60 min: {int((gaps > 60).sum())}")


def candles(m: Market, minutes: int) -> pd.DataFrame:
    """Candles of `minutes` length aligned to 09:15 IST, never spanning days. i0/i1 = [start, end) minute indices."""
    b = m.base_mod
    key = m.day * 10000 + (m.mod.astype(np.int64) - b) // minutes
    brk = np.flatnonzero(np.diff(key)) + 1
    i0 = np.concatenate([[0], brk])
    i1 = np.concatenate([brk, [len(m.t)]])
    return pd.DataFrame({
        "t0": m.day[i0] * 86400 + b * 60 + ((m.mod[i0].astype(np.int64) - b) // minutes) * minutes * 60 - m.tzoff,
        "i0": i0, "i1": i1, "day": m.day[i0],
        "o": m.o[i0], "c": m.c[i1 - 1], "h": np.maximum.reduceat(m.h, i0), "l": np.minimum.reduceat(m.l, i0),
        "end_mod": m.mod[i1 - 1] + 1,
    })


def ema(x: np.ndarray, n: int) -> np.ndarray:
    """TradingView ta.ema: alpha 2/(n+1), seeded with the first value."""
    out = np.empty(len(x))
    a = 2.0 / (n + 1)
    acc = x[0]
    for i in range(len(x)):
        acc = a * x[i] + (1 - a) * acc if i else x[0]
        out[i] = acc
    return out


def exit_horizon(m: Market, flat_mod: int, hold_days: int = 0, max_hold_min: int | None = None) -> np.ndarray:
    """For every minute, the index of the forced-exit minute of a position opened there.
    NSE: the last minute before flat_mod (15:20) on the same trading day (hold_days=0) or on the hold_days-th next
    trading day. Crypto (24/7): the last minute within max_hold_min (None = to the end of the data)."""
    n = len(m.t)
    if m.kind == "crypto":
        if not max_hold_min:
            return np.full(n, n - 1, np.int64)
        return np.clip(np.searchsorted(m.t, m.t + max_hold_min * 60, side="right") - 1, 0, n - 1).astype(np.int64)
    brk = np.flatnonzero(np.diff(m.day)) + 1
    starts = np.concatenate([[0], brk])
    ends = np.concatenate([brk, [n]])
    dlast = np.empty(len(starts), np.int64)
    for q, (s, e) in enumerate(zip(starts, ends)):
        idx = s + np.flatnonzero(m.mod[s:e] < flat_mod)
        dlast[q] = idx[-1] if len(idx) else e - 1
    last = np.empty(n, np.int64)
    for q, (s, e) in enumerate(zip(starts, ends)):
        last[s:e] = dlast[min(q + hold_days, len(dlast) - 1)]
    return last


# ------------------------------------------------------------- costs ---
def _sched(table: list, date: str) -> float:
    v = table[0][1]
    for d, x in table:
        if date >= d:
            v = x
    return v


def cost_points(rules: dict, sym: str, date: str, spot: float, zero: bool = False) -> float:
    if not zero and sym in CRYPTO:
        # Delta Exchange perpetual futures: taker fee per side (+18% GST on the fee) + slippage per side, % of price
        K = rules["costs_crypto"]
        return spot * 2 * (K["taker_fee"] * (1 + K["gst"]) + K["slippage_pct_per_side"] * rules["costs"].get("slippage_mult", 1.0))
    return _cost_points_nse(rules, sym, date, spot, zero)


def _cost_points_nse(rules: dict, sym: str, date: str, spot: float, zero: bool = False) -> float:
    """Per-UNIT round-trip cost of an ATM option trade, in SPOT index points (premium points / delta): STT, NSE
    transaction charge (+GST), stamp duty, SEBI fee on an ATM premium, plus the stated slippage per side.
    Brokerage is a flat ₹20 per order whatever the quantity, so it is charged separately in rupees (brokerage_R)."""
    if zero:
        return 0.0
    C = rules["costs"]
    gst = C["gst"]
    prem = C["atm_premium_pct_of_spot"][sym] * spot
    pct = (_sched(C["stt_option_sell"], date) + 2 * _sched(C["nse_txn_option"], date) * (1 + gst)
           + C["stamp_buy"] + 2 * C["sebi_fee"] * (1 + gst))
    slip = 2 * C["slippage_premium_pts_per_side"][sym] * C.get("slippage_mult", 1.0)
    return (prem * pct + slip) / C["delta"]


def brokerage_R(rules: dict, zero: bool = False, sym: str = "") -> float:
    """Flat brokerage (2 orders x ₹20 + GST) as a fraction of the fixed rupee risk per trade."""
    if zero or sym in CRYPTO:
        return 0.0
    C = rules["costs"]
    return 2 * C["brokerage_per_order_inr"] * (1 + C["gst"]) / rules["money"]["fixed_risk_inr"]


# ------------------------------------------------------------ execution ---
@njit(cache=True)
def manage_nb(h, l, c, o, i0, last_i, d, stop, target, cend, cclose, cema, exit_ema, optimistic):
    """Follow one position from minute i0 (the minute after the entry candle) to last_i.
    d=+1 long / -1 short; target<=0: none. cend/cclose/cema: end minute index (exclusive), close and EMA20 of
    the following candles, for the 'exit on a close beyond the 20 EMA' rule.
    Returns (exit_i, exit_px, reason) reason 0=stop 1=target 2=ema-exit 3=time."""
    k = 0
    nk = len(cend)
    for j in range(i0, last_i + 1):
        if d == 1:
            st = l[j] <= stop
            tg = target > 0 and h[j] >= target
        else:
            st = h[j] >= stop
            tg = target > 0 and l[j] <= target
        if st and (not tg or not optimistic):
            px = min(stop, o[j]) if d == 1 else max(stop, o[j])
            return j, px, 0
        if tg:
            return j, target, 1
        if exit_ema:
            while k < nk and cend[k] - 1 < j:
                k += 1
            if k < nk and cend[k] - 1 == j:
                if (d == 1 and cclose[k] < cema[k]) or (d == -1 and cclose[k] > cema[k]):
                    return j, c[j], 2
    return last_i, c[last_i], 3


REASONS = {0: "stop", 1: "target", 2: "ema20_exit", 3: "time"}


# ------------------------------------------------------------- statistics ---
def wilson(k: int, n: int, z: float = 1.96) -> tuple[float, float]:
    if n == 0:
        return (float("nan"), float("nan"))
    p = k / n
    den = 1 + z * z / n
    centre = (p + z * z / (2 * n)) / den
    half = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / den
    return centre - half, centre + half


def summarize(tr: pd.DataFrame, weeks: float, rules: dict, claim_wr: float) -> dict:
    n = len(tr)
    if n == 0:
        return {"trades": 0}
    tr = tr.sort_values("entry_t")
    R = tr["R"].to_numpy()
    wins = R > 0
    cum = np.cumsum(R)
    dd_r = float(np.max(np.maximum.accumulate(np.concatenate([[0], cum]))[1:] - cum))
    cap, risk_inr = rules["money"]["start_equity_inr"], rules["money"]["fixed_risk_inr"]
    bal = cap + risk_inr * cum
    blown = np.flatnonzero(bal <= 0)
    ruin_i = int(blown[0]) if len(blown) else None
    if ruin_i is not None:
        bal = bal.copy()
        bal[ruin_i:] = 0.0
    peak = np.maximum.accumulate(np.concatenate([[cap], bal]))[1:]
    eq_c = cap * np.cumprod(np.maximum(1 + rules["money"]["compound_risk_pct"] / 100 * R, 0))
    gp, gl = R[R > 0].sum(), -R[R < 0].sum()
    avg_win = float(R[wins].mean()) if wins.any() else 0.0
    avg_loss = float(-R[~wins].mean()) if (~wins).any() else 0.0
    lo, hi = wilson(int(wins.sum()), n)
    tg = (tr["reason"] == "target").to_numpy()
    tlo, thi = wilson(int(tg.sum()), n)
    se = R.std(ddof=1) / math.sqrt(n) if n > 1 else float("nan")
    years = []
    for y, g in tr.groupby(tr["entry_t"].str[:4]):
        r = g["R"].to_numpy()
        years.append({"year": int(y), "trades": int(len(r)), "win_rate": round(100 * (r > 0).mean(), 1),
                      "target_hit_rate": round(100 * (g["reason"] == "target").mean(), 1),
                      "total_R": round(float(r.sum()), 1), "avg_R": round(float(r.mean()), 3)})
    streak = best = 0
    for w in wins:
        streak = 0 if w else streak + 1
        best = max(best, streak)
    months = max(weeks / 4.348, 1e-9)
    return {
        "trades": n, "trades_per_week": round(n / weeks, 2),
        "win_rate_pct": round(100 * wins.mean(), 1),
        "win_rate_95ci_pct": [round(100 * lo, 1), round(100 * hi, 1)],
        "target_hit_rate_pct": round(100 * tg.mean(), 1),
        "target_hit_95ci_pct": [round(100 * tlo, 1), round(100 * thi, 1)],
        "stop_rate_pct": round(100 * (tr["reason"] == "stop").mean(), 1),
        "ema_exit_rate_pct": round(100 * tr["reason"].isin(["ema20_exit", "ema9_exit"]).mean(), 1),
        "time_exit_rate_pct": round(100 * (tr["reason"] == "time").mean(), 1),
        "avg_R": round(float(R.mean()), 3),
        "avg_R_95ci": [round(float(R.mean() - 1.96 * se), 3), round(float(R.mean() + 1.96 * se), 3)],
        "avg_win_R": round(avg_win, 3), "avg_loss_R": round(avg_loss, 3),
        "breakeven_win_rate_pct": round(100 * avg_loss / (avg_win + avg_loss), 1) if (avg_win + avg_loss) else None,
        "profit_factor": round(float(gp / gl), 3) if gl > 0 else None,
        "total_R": round(float(R.sum()), 1), "max_drawdown_R": round(dd_r, 1),
        "longest_losing_streak": int(best),
        "fixed_final_balance_inr": round(float(bal[-1]), 0),
        "fixed_account_blown_at_trade": (ruin_i + 1) if ruin_i is not None else None,
        "fixed_account_blown_date": (str(tr["entry_t"].iloc[ruin_i])[:10]) if ruin_i is not None else None,
        "fixed_total_inr_if_refilled": round(float(risk_inr * R.sum()), 0),
        "fixed_profit_per_month_inr": round(float(risk_inr * R.sum() / months), 0),
        "fixed_max_drawdown_inr": round(float(np.max(peak - bal)), 0),
        "compounded_final_balance_inr": round(float(eq_c[-1]), 0),
        "median_risk_points": round(float(tr["risk"].median()), 2),
        "median_cost_points": round(float(tr["cost_pts"].median()), 2),
        "median_cost_share_of_risk_pct": round(float((tr["cost_pts"] / tr["risk"] + tr["brokerage_R"]).median() * 100), 1),
        "years": years,
        "years_at_or_above_claim": [y["year"] for y in years if y["target_hit_rate"] >= claim_wr],
        "years_win_at_or_above_claim": [y["year"] for y in years if y["win_rate"] >= claim_wr],
    }


# ------------------------------------------------------------ 9-20 EMA module ---
def _ts(t: int) -> str:
    return pd.Timestamp(int(t), unit="s", tz="UTC").tz_convert(IST).isoformat()


def _entry_gate(m: Market, v: dict, rules: dict):
    """Returns (ok(k_end_minute_index, candle_end_mod, candle_day) -> reason|None, horizon array)."""
    S = rules["session"]
    if m.kind == "nse":
        dcount = pd.Series(1, index=m.day).groupby(level=0).size()
        good_day = set(dcount[dcount >= S["min_minutes_for_entries"]].index)
        first_mod = v.get("first_entry_mod", S["first_entry_mod"])
        last_mod = v.get("last_entry_mod", S["last_entry_mod"])
        hz = exit_horizon(m, S["flat_mod"], v.get("hold_days", 0))

        def ok(ia, endmod, cday):
            if cday not in good_day:
                return "skip_bad_day"
            if endmod < first_mod or endmod > last_mod:
                return "skip_time"
            if ia >= len(m.t) or m.day[ia] != cday:
                return "skip_time"
            return None
        return ok, hz
    hz = exit_horizon(m, 0, 0, v.get("max_hold_min", rules.get("crypto", {}).get("max_hold_min")))
    win = v.get("ist_window")

    def ok(ia, endmod, cday):
        if ia >= len(m.t):
            return "skip_time"
        if win:
            im = ((m.t[ia] + 19800) % 86400) // 60
            if not (win[0] <= im < win[1]):
                return "skip_time"
        return None
    return ok, hz


def _htf_state(m: Market, htf_min: int, i1s: np.ndarray) -> np.ndarray:
    hc = candles(m, htf_min)
    hs = np.sign(ema(hc["c"].to_numpy(), 9) - ema(hc["c"].to_numpy(), 20))
    hend = hc["i1"].to_numpy()
    pos = np.searchsorted(hend, i1s, side="right") - 1        # last COMPLETED higher-tf candle
    return np.where(pos >= 0, hs[np.clip(pos, 0, None)], 0)


def _record(m, cd, k, armed_at, ia, j, px, reason, d, fill, stop, target, risk, rules, zero) -> dict:
    date = _ts(m.t[ia])[:10]
    cp = cost_points(rules, m.symbol, date, fill, zero)
    bR = brokerage_R(rules, zero, m.symbol)
    pts = (px - fill) * d - cp
    return {"entry_i": int(ia), "exit_i": int(j), "dir": int(d), "entry_px": round(float(fill), 4),
            "stop": round(float(stop), 4), "target": round(float(target), 4) if target else None,
            "exit_px": round(float(px), 4), "risk": round(float(risk), 4), "cost_pts": round(float(cp), 4),
            "gross_pts": round(float((px - fill) * d), 4), "pts": round(float(pts), 4), "brokerage_R": round(bR, 4),
            "R": float(pts / risk - bR), "reason": REASONS[int(reason)],
            "signal_candle_t": _ts(cd["t0"].iat[k]), "entry_t": _ts(m.t[ia]), "exit_t": _ts(m.t[j]),
            "cross_t": _ts(cd["t0"].iat[armed_at]) if armed_at >= 0 else None}


def strat_nine20(m: Market, v: dict, rules: dict, zero: bool = False, optimistic: bool = False) -> tuple[list[dict], dict]:
    """The 9-20 EMA setup on `tf_min` candles (sb_rules.json -> nine20 for every rule's source).
    Bias: long only while EMA9 > EMA20, short only while EMA9 < EMA20. A 9/20 crossover arms the setup.
      range_filter='breakout': the cross must come with a range breakout in the same direction = a candle close
        beyond the extreme of the previous `range_n` candles, within `breakout_window` candles before the cross or
        after it (before the entry). [Ma1HnAXxww4: range -> breakout -> crossover]
      nochop: skip a cross that comes after >= 2 other crosses in the previous `chop_n` candles (EMAs intertwined).
      need_departure: after the cross, price must first leave the 9 EMA (a candle entirely beyond it) and then come
        back for the pullback. [uK7epk8-r9o 0:08:42 'निकला निकला निकला। और फर्स्ट टाइम पुलबैक']
      max_entries_per_cross: 1 = first pullback only; null = every pullback while the trend lasts.
    Entry candle (long; short mirrored): low <= EMA9 (it comes to the 9 EMA) and close > max(EMA9, EMA20), and with
      entry='touch9_bull' a green candle. Entry at its close.
    Stop: 'entry_candle' low/high or 'swing5' = extreme of the last 5 candles. Target rr x risk (null = none).
    exit_ema20: exit at the close of a candle that closes beyond the 20 EMA against the trade."""
    tf = v.get("tf_min", 5)
    cd = candles(m, tf)
    C = cd["c"].to_numpy(); H = cd["h"].to_numpy(); L = cd["l"].to_numpy(); O = cd["o"].to_numpy()
    e9, e20 = ema(C, 9), ema(C, 20)
    state = np.sign(e9 - e20)
    i1s = cd["i1"].to_numpy(); endmod = cd["end_mod"].to_numpy(); cday = cd["day"].to_numpy()
    ok, hz = _entry_gate(m, v, rules)
    rn = v.get("range_n", 12)
    hh = pd.Series(H).rolling(rn).max().shift(1).to_numpy()
    ll = pd.Series(L).rolling(rn).min().shift(1).to_numpy()
    bo_up = C > hh
    bo_dn = C < ll
    crosses = np.concatenate([[False], (state[1:] != state[:-1]) & (state[1:] != 0)])
    htf = _htf_state(m, v["htf_min"], i1s) if v.get("htf_min") else None
    fl = {"crosses": 0, "skip_chop": 0, "skip_htf": 0, "skip_time": 0, "skip_bad_day": 0,
          "skip_nonpositive_risk": 0, "skip_busy": 0}
    trades = []
    armed_dir, armed_at, departed, n_entries, busy_until = 0, -1, False, 0, -1
    bw = v.get("breakout_window", 12)
    rf = v.get("range_filter", "breakout")
    ent = v.get("entry", "touch9_bull")
    mpc = v.get("max_entries_per_cross", 1)
    chop_n = v.get("chop_n", 12)
    for k in range(25, len(cd)):
        if crosses[k]:
            fl["crosses"] += 1
            armed_dir, armed_at, departed, n_entries = int(state[k]), k, False, 0
            if v.get("nochop") and crosses[max(0, k - chop_n):k].sum() >= 2:
                armed_dir = 0
                fl["skip_chop"] += 1
                continue
        if armed_dir == 0 or state[k] != armed_dir:
            continue
        d = armed_dir
        touch = (L[k] <= e9[k]) if d == 1 else (H[k] >= e9[k])
        if not touch and k > armed_at and ((L[k] > e9[k]) if d == 1 else (H[k] < e9[k])):
            departed = True
            continue
        if mpc and n_entries >= mpc:
            continue
        if v.get("need_departure") and not departed:
            continue
        if rf == "breakout":
            seg = bo_up[max(0, armed_at - bw):k + 1] if d == 1 else bo_dn[max(0, armed_at - bw):k + 1]
            if not seg.any():
                continue
        beyond = C[k] > max(e9[k], e20[k]) if d == 1 else C[k] < min(e9[k], e20[k])
        colour = (C[k] > O[k] if d == 1 else C[k] < O[k]) if ent == "touch9_bull" else True
        if not (touch and beyond and colour):
            continue
        ia = int(i1s[k])
        why = ok(ia, endmod[k], cday[k])
        if why:
            fl[why] += 1
            continue
        if htf is not None and htf[k] != d:
            fl["skip_htf"] += 1
            continue
        if ia <= busy_until:
            fl["skip_busy"] += 1
            continue
        fill = C[k]
        if v.get("stop", "entry_candle") == "entry_candle":
            stop = L[k] if d == 1 else H[k]
        else:
            stop = L[max(0, k - 4):k + 1].min() if d == 1 else H[max(0, k - 4):k + 1].max()
        risk = (fill - stop) * d
        if risk <= 0:
            fl["skip_nonpositive_risk"] += 1
            continue
        rr = v.get("rr")
        target = fill + d * rr * risk if rr else 0.0
        nxt = slice(k + 1, min(len(cd), k + 1 + 4000))
        j, px, reason = manage_nb(m.h, m.l, m.c, m.o, ia, int(hz[ia]), d, stop, target,
                                  i1s[nxt].astype(np.int64), C[nxt], e20[nxt], bool(v.get("exit_ema20", False)),
                                  bool(optimistic))
        trades.append(_record(m, cd, k, armed_at, ia, j, px, reason, d, fill, stop, target, risk, rules, zero))
        n_entries += 1
        busy_until = int(j)
    return trades, fl


# ------------------------------------------------------------ 9 EMA scalping module ---
def strat_ema9(m: Market, v: dict, rules: dict, zero: bool = False, optimistic: bool = False) -> tuple[list[dict], dict]:
    """9 EMA scalping (AO5tS1hqMao, Gc-xw57Zrg8) on 5-minute candles of the spot index.
    Bias: the previous candle closed below the 9 EMA -> puts only (short); above -> calls only (long).
    Entry candle (short; long mirrored): it comes up to the 9 EMA (high >= EMA9), is rejected and closes back below
      it as a red candle; entry at its close. Stop: its high ('entry_candle') or 'swing5'.
    Exits: target rr x risk (null = none); a candle CLOSE back above the 9 EMA ('exit_ema9'); time stop after
      `max_candles` candles. Daily: at most `max_trades_per_day`, stop after `stop_after_losses` stop-losses in a row."""
    tf = v.get("tf_min", 5)
    cd = candles(m, tf)
    C = cd["c"].to_numpy(); H = cd["h"].to_numpy(); L = cd["l"].to_numpy(); O = cd["o"].to_numpy()
    e9 = ema(C, 9)
    i1s = cd["i1"].to_numpy(); endmod = cd["end_mod"].to_numpy(); cday = cd["day"].to_numpy()
    ok, hz = _entry_gate(m, v, rules)
    fl = {"skip_time": 0, "skip_bad_day": 0, "skip_nonpositive_risk": 0, "skip_busy": 0, "skip_daily_limit": 0}
    trades, busy_until = [], -1
    day_n, day_sl, cur_day = 0, 0, -1
    for k in range(25, len(cd)):
        if cday[k] != cur_day:
            cur_day, day_n, day_sl = cday[k], 0, 0
        d = 1 if C[k - 1] > e9[k - 1] else -1
        if d == 1:
            sig = L[k] <= e9[k] and C[k] > e9[k] and C[k] > O[k]
        else:
            sig = H[k] >= e9[k] and C[k] < e9[k] and C[k] < O[k]
        if not sig:
            continue
        ia = int(i1s[k])
        why = ok(ia, endmod[k], cday[k])
        if why:
            fl[why] += 1
            continue
        if ia <= busy_until:
            fl["skip_busy"] += 1
            continue
        if (v.get("max_trades_per_day") and day_n >= v["max_trades_per_day"]) or \
           (v.get("stop_after_losses") and day_sl >= v["stop_after_losses"]):
            fl["skip_daily_limit"] += 1
            continue
        fill = C[k]
        stop = (L[k] if d == 1 else H[k]) if v.get("stop", "entry_candle") == "entry_candle" else \
            (L[max(0, k - 4):k + 1].min() if d == 1 else H[max(0, k - 4):k + 1].max())
        risk = (fill - stop) * d
        if risk <= 0:
            fl["skip_nonpositive_risk"] += 1
            continue
        rr = v.get("rr")
        target = fill + d * rr * risk if rr else 0.0
        mc = v.get("max_candles", 3)
        nxt = slice(k + 1, min(len(cd), k + 1 + mc))
        last_i = min(int(hz[ia]), int(i1s[min(len(cd) - 1, k + mc)]) - 1)
        j, px, reason = manage_nb(m.h, m.l, m.c, m.o, ia, last_i, d, stop, target,
                                  i1s[nxt].astype(np.int64), C[nxt], e9[nxt], bool(v.get("exit_ema9", True)),
                                  bool(optimistic))
        trades.append(_record(m, cd, k, -1, ia, j, px, reason, d, fill, stop, target, risk, rules, zero))
        if trades[-1]["reason"] == "ema20_exit":
            trades[-1]["reason"] = "ema9_exit"
        busy_until = int(j)
        day_n += 1
        day_sl = day_sl + 1 if trades[-1]["reason"] == "stop" else 0
    return trades, fl


# ------------------------------------------------------------ Big Bar module ---
def bigbar_flags(cd: pd.DataFrame, e9: np.ndarray, v: dict) -> tuple[np.ndarray, np.ndarray]:
    """Big bars (long = +1 / short = -1 / 0). A big bar: range >= k_size x the mean range of the previous 3
    candles ('प्रीवियस वन तू थ्री कैंडल से काफी हेल्दी कैंडल', tLQuEUQAvYE 0:10:31), body >= body_min of its
    range, closes beyond the 9 EMA in its direction (3kHLk5tWBRg 0:09:20-0:09:53); with require_touch the bar or
    the candle before it touched the 9 EMA ('9 EMA support then a big bar', 3kHL 0:06:22, LJR 0:04:02);
    range <= cap_pct of price ('15 पॉइंट से ऊपर ... बिग बार में नहीं', 3kHL 0:06:05). Returns (direction, range)."""
    C = cd["c"].to_numpy(); H = cd["h"].to_numpy(); L = cd["l"].to_numpy(); O = cd["o"].to_numpy()
    rng = H - L
    prev = pd.Series(rng).shift(1).rolling(3).mean().to_numpy()
    body = np.abs(C - O)
    big = (rng >= v.get("k_size", 2.0) * prev) & (body >= v.get("body_min", 0.6) * rng) & (rng > 0)
    if v.get("cap_pct"):
        big &= rng <= v["cap_pct"] / 100 * C
    up = big & (C > O) & (C > e9)
    dn = big & (C < O) & (C < e9)
    if v.get("require_touch", True):
        Lp = np.concatenate([[np.inf], L[:-1]]); Hp = np.concatenate([[-np.inf], H[:-1]])
        e9p = np.concatenate([[np.nan], e9[:-1]])
        up &= (L <= e9) | (Lp <= e9p)
        dn &= (H >= e9) | (Hp >= e9p)
    d = np.where(up, 1, np.where(dn, -1, 0))
    d[:4] = 0
    return d, rng


def strat_bigbar(m: Market, v: dict, rules: dict, zero: bool = False, optimistic: bool = False) -> tuple[list[dict], dict]:
    """Big Bar scalping, tested on the SPOT index chart (he says the premium chart must show the big bar too:
    3kHLk5tWBRg 0:06:57 / 0:08:48 - not coded, see the plan). Entry at the big bar's close ('क्लोजिंग पे तुरंत
    एंट्री', 3kHL 0:07:19); stop at its low/high (3kHL 0:11:48); target rr x risk or a premium-% target converted to
    spot (target_prem_pct x ATM premium / delta); time stop `max_candles` candles (15 min, KpOwBYb3H7c 0:15:41);
    no entries in the first 15 minutes (KpOw 0:16:40) or after 15:00 (3kHL 0:14:05); max trades/day."""
    tf = v.get("tf_min", 5)
    cd = candles(m, tf)
    C = cd["c"].to_numpy(); H = cd["h"].to_numpy(); L = cd["l"].to_numpy()
    e9 = ema(C, 9)
    dirs, rng = bigbar_flags(cd, e9, v)
    i1s = cd["i1"].to_numpy(); endmod = cd["end_mod"].to_numpy(); cday = cd["day"].to_numpy()
    v2 = dict(v)
    v2.setdefault("first_entry_mod", 555 + 15 + tf)
    ok, hz = _entry_gate(m, v2, rules)
    fl = {"big_bars": int((dirs != 0).sum()), "skip_time": 0, "skip_bad_day": 0, "skip_busy": 0, "skip_daily_limit": 0}
    trades, busy_until, day_n, cur_day = [], -1, 0, -1
    C_ = rules["costs"]
    for k in np.flatnonzero(dirs):
        if cday[k] != cur_day:
            cur_day, day_n = cday[k], 0
        d = int(dirs[k])
        ia = int(i1s[k])
        why = ok(ia, endmod[k], cday[k])
        if why:
            fl[why] += 1
            continue
        if ia <= busy_until:
            fl["skip_busy"] += 1
            continue
        if v.get("max_trades_per_day") and day_n >= v["max_trades_per_day"]:
            fl["skip_daily_limit"] += 1
            continue
        fill = C[k]
        stop = L[k] if d == 1 else H[k]
        risk = (fill - stop) * d
        if v.get("target_prem_pct"):
            sym = m.symbol if m.symbol in C_["atm_premium_pct_of_spot"] else "NIFTY"
            tgt_pts = v["target_prem_pct"] / 100 * C_["atm_premium_pct_of_spot"][sym] * fill / C_["delta"]
            target = fill + d * tgt_pts
        else:
            target = fill + d * v.get("rr", 2) * risk
        mc = v.get("max_candles", 3)
        last_i = min(int(hz[ia]), int(i1s[min(len(cd) - 1, k + mc)]) - 1)
        j, px, reason = manage_nb(m.h, m.l, m.c, m.o, ia, last_i, d, stop, target, np.zeros(0, np.int64),
                                  np.zeros(0), np.zeros(0), False, bool(optimistic))
        trades.append(_record(m, cd, k, -1, ia, j, px, reason, d, fill, stop, target, risk, rules, zero))
        busy_until = int(j)
        day_n += 1
    return trades, fl


def bigbar_premise(m: Market, v: dict) -> dict:
    """'10 में से आठ बार ... बिग बार के बाद से डायरेक्शन होता है' (KpOwBYb3H7c 0:18:52): after a big bar, how often
    does price continue in its direction? Compared with ordinary candles of the same colour on the same side of
    the 9 EMA (the baseline)."""
    cd = candles(m, v.get("tf_min", 5))
    C = cd["c"].to_numpy(); H = cd["h"].to_numpy(); L = cd["l"].to_numpy(); O = cd["o"].to_numpy()
    e9 = ema(C, 9)
    dirs, rng = bigbar_flags(cd, e9, v)
    day = cd["day"].to_numpy()
    n = len(cd)
    base = np.where((C > O) & (C > e9), 1, np.where((C < O) & (C < e9), -1, 0))
    base[dirs != 0] = 0

    def measure(sig):
        nc, ext, cnt = 0, 0, 0
        for k in np.flatnonzero(sig):
            if k + 3 >= n or day[k + 3] != day[k]:
                continue
            d = int(sig[k]); cnt += 1
            nc += (C[k + 1] - C[k]) * d > 0
            goal = C[k] + d * 0.5 * rng[k]; bad = L[k] if d == 1 else H[k]
            for q in range(k + 1, k + 4):
                hitg = H[q] >= goal if d == 1 else L[q] <= goal
                hitb = L[q] <= bad if d == 1 else H[q] >= bad
                if hitb:
                    break
                if hitg:
                    ext += 1
                    break
        return {"bars": int(cnt), "next_candle_continues_pct": round(100 * nc / cnt, 1) if cnt else None,
                "extends_half_bar_before_its_stop_within_3_candles_pct": round(100 * ext / cnt, 1) if cnt else None}
    return {"definition": bigbar_flags.__doc__, "big_bars": measure(dirs), "baseline_ordinary_candles": measure(base)}


STRATEGIES = {"nine20": strat_nine20, "ema9": strat_ema9, "bigbar": strat_bigbar}


def apply_day_cap(trades: list[dict], cap: int | None) -> list[dict]:
    if not cap or not trades:
        return trades
    df = pd.DataFrame(trades).sort_values("entry_t")
    df = df[df.groupby(df["entry_t"].str[:10]).cumcount() < cap]
    return df.to_dict("records")


def run_strategy(strategy: str, sym: str, m: Market, rules: dict, zero=False, optimistic=False,
                 variants: list[dict] | None = None) -> dict:
    res = {}
    weeks = (m.t[-1] - m.t[0]) / (7 * 86400)
    claim = rules["claims"]["headline_accuracy_pct"]
    for v in variants or rules[strategy]["variants"]:
        if m.kind not in v.get("markets", ["nse", "crypto"]):
            continue
        tr, fl = STRATEGIES[strategy](m, v, rules, zero, optimistic)
        tr = apply_day_cap(tr, v.get("max_trades_per_day"))
        df = pd.DataFrame(tr)
        s = summarize(df, weeks, rules, claim) if len(df) else {"trades": 0}
        s.update({"variant": v, "flags": fl})
        res[v["name"]] = {"summary": s, "trades": df}
        if len(df):
            print(f"  [{sym}] {strategy}:{v['name'][:60]:60s} n={s['trades']:5d} win={s['win_rate_pct']:5.1f}% "
                  f"tgt={s['target_hit_rate_pct']:5.1f}% avgR={s['avg_R']:+.3f} PF={s['profit_factor']} "
                  f"totR={s['total_R']:+.1f}", flush=True)
        else:
            print(f"  [{sym}] {strategy}:{v['name'][:60]:60s} no trades {fl}", flush=True)
    return res


# -------------------------------------------------------------------- run ---
LABELS = {"nine20": "9-20 EMA strategy", "ema9": "9 EMA scalping", "bigbar": "Big Bar scalping (spot chart)"}
KEEP = ("trades", "win_rate_pct", "win_rate_95ci_pct", "target_hit_rate_pct", "target_hit_95ci_pct", "avg_R",
        "profit_factor", "total_R", "breakeven_win_rate_pct", "years_at_or_above_claim", "years", "longest_losing_streak")


def cmd_run(a, rules: dict) -> None:
    syms = a.symbols.split(",")
    strategies = a.strategies.split(",")
    final = a.source == "kaggle" and not a.start and not a.end and not rules.get("_preliminary", True)
    out = {"title": "Stock Burner strategies - backtest of the rules as taught" + ("" if final else " (PRELIMINARY)"),
           "preliminary": not final, "source": a.source, "claims": rules["claims"], "symbols": {},
           "generated_utc": pd.Timestamp.now(tz="UTC").isoformat(), "rules_sha1": hashlib.sha1(RULES.read_bytes()).hexdigest()}
    tag = ("_".join(syms) if a.tag == "" else a.tag) + ("" if a.source == "kaggle" else f"_{a.source}")
    tdir = RAW / "trades" / tag
    tdir.mkdir(parents=True, exist_ok=True)
    for sym in syms:
        m = load_market(sym, a.source, start=a.start, end=a.end)
        span = f"{pd.Timestamp(int(m.t[0]), unit='s', tz='UTC').tz_convert(IST).date()} .. {pd.Timestamp(int(m.t[-1]), unit='s', tz='UTC').tz_convert(IST).date()}"
        print(f"{sym}: {len(m):,} minutes {span} source={m.source}", flush=True)
        S = {"source": m.source, "span": span, "minutes": len(m), "strategies": {}}
        for st in strategies:
            if not any(m.kind in v.get("markets", ["nse", "crypto"]) for v in rules[st]["variants"]):
                continue
            res = run_strategy(st, sym, m, rules)
            block = {"variants": {}, "no_cost": {}, "optimistic": {}}
            for name, r in res.items():
                block["variants"][name] = r["summary"]
                if len(r["trades"]):
                    slug = f"{sym}_{st}_{hashlib.md5(name.encode()).hexdigest()[:8]}"
                    r["trades"].assign(variant=name).to_csv(tdir / f"{slug}.csv.gz", index=False)
                    block["variants"][name]["trades_file"] = f"data_raw/trades/{tag}/{slug}.csv.gz"
            heads = [v for v in rules[st]["variants"] if v["name"] in rules[st].get("headline", [])
                     and m.kind in v.get("markets", ["nse", "crypto"])]
            if heads and not a.fast:
                z = run_strategy(st, sym, m, rules, zero=True, variants=heads)
                o = run_strategy(st, sym, m, rules, optimistic=True, variants=heads)
                g = run_strategy(st, sym, m, rules, zero=True, optimistic=True, variants=heads)
                block["no_cost"] = {k: {x: r["summary"].get(x) for x in KEEP} for k, r in z.items()}
                block["optimistic"] = {k: {x: r["summary"].get(x) for x in KEEP} for k, r in o.items()}
                block["most_generous"] = {k: {x: r["summary"].get(x) for x in KEEP} for k, r in g.items()}
                # slippage sensitivity (x0.5, x2) on the headline variants
                sens = {}
                for mult in (0.5, 2.0):
                    r2 = json.loads(json.dumps(rules))
                    r2["costs"]["slippage_mult"] = mult
                    s2 = run_strategy(st, sym, m, r2, variants=heads)
                    sens[str(mult)] = {k: {x: r["summary"].get(x) for x in ("trades", "win_rate_pct", "avg_R", "total_R")}
                                       for k, r in s2.items()}
                block["slippage_sensitivity"] = sens
            S["strategies"][st] = block
        out["symbols"][sym] = S
        del m
    OUT_DIR.mkdir(parents=True, exist_ok=True)
    path = OUT_DIR / f"sb_results_{tag}.json"
    path.write_text(json.dumps(out, default=lambda x: x.item() if hasattr(x, "item") else str(x), separators=(",", ":")))
    print(f"wrote {path.relative_to(ROOT)}")


# ---------------------------------------------------------------- premises ---
def cmd_premise(a, rules: dict) -> None:
    """His premises measured directly: (P2) 'bigger timeframe = more accurate' vs '5 minute only';
    (P3) 'in index only 3-4 days out of 10 trend'. Writes sb_premises_<tag>.json."""
    out = {}
    pv = rules["nine20"].get("premise_timeframes", [])
    for sym in a.symbols.split(","):
        m = load_market(sym, "kaggle")
        weeks = (m.t[-1] - m.t[0]) / (7 * 86400)
        P = {"P2_timeframe": {}}
        if m.kind != "nse":
            continue
        for v in pv:
            tr, fl = strat_nine20(m, v, rules, zero=True)
            df = pd.DataFrame(tr)
            s = summarize(df, weeks, rules, rules["claims"]["headline_accuracy_pct"]) if len(df) else {"trades": 0}
            P["P2_timeframe"][v["name"]] = {x: s.get(x) for x in KEEP}
            print(f"  [{sym}] P2 {v['name']}: n={s.get('trades')} tgt={s.get('target_hit_rate_pct')} avgR={s.get('avg_R')}")
        # P3: share of 'trend days' = days whose open->close move is >= 60% of the day's high-low range
        d1 = pd.DataFrame({"day": m.day, "o": m.o, "h": m.h, "l": m.l, "c": m.c}).groupby("day").agg(
            o=("o", "first"), h=("h", "max"), l=("l", "min"), c=("c", "last"), n=("o", "size"))
        d1 = d1[d1["n"] >= rules["session"]["min_minutes_for_entries"]]
        trendy = (d1["c"] - d1["o"]).abs() >= 0.6 * (d1["h"] - d1["l"])
        P["P3_trend_days"] = {"definition": "open-to-close move >= 60% of the day's high-low range (our definition)",
                              "days": int(len(d1)), "trend_days_pct": round(100 * float(trendy.mean()), 1)}
        if "bigbar" in rules:
            P["PB_bigbar_direction"] = {v["name"]: bigbar_premise(m, v) for v in rules["bigbar"].get("premise", [])}
        out[sym] = P
    OUT_DIR.mkdir(parents=True, exist_ok=True)
    p = OUT_DIR / f"sb_premises_{'_'.join(a.symbols.split(','))}.json"
    p.write_text(json.dumps(out, indent=1, default=str))
    print("wrote", p.relative_to(ROOT))


# ------------------------------------------------------------- spot check ---
def cmd_spot(a, rules: dict) -> None:
    """Print random trades of one variant candle by candle so they can be recomputed by hand."""
    st, name = a.strategies.split(",")[0], a.variant
    v = [x for x in rules[st]["variants"] if x["name"] == name][0]
    m = load_market(a.symbols.split(",")[0], a.source, start=a.start, end=a.end)
    tr, fl = STRATEGIES[st](m, v, rules)
    rng = random.Random(a.seed)
    pick = sorted(rng.sample(range(len(tr)), min(a.n, len(tr))))
    tf = v.get("tf_min", 5)
    cd = candles(m, tf)
    e9, e20 = ema(cd["c"].to_numpy(), 9), ema(cd["c"].to_numpy(), 20)
    print(f"{m.symbol} {st} '{name}' source={m.source} trades={len(tr)} flags={fl}")
    for ix in pick:
        t = tr[ix]
        print("=" * 110)
        print({k: (round(x, 3) if isinstance(x, float) else x) for k, x in t.items() if k not in ("entry_i", "exit_i")})
        s0 = int(pd.Timestamp(t.get("cross_t") or t["signal_candle_t"]).timestamp()) - 14 * tf * 60
        e = int(pd.Timestamp(t["exit_t"]).timestamp())
        sel = np.flatnonzero((cd["t0"].to_numpy() >= s0) & (cd["t0"].to_numpy() <= e))
        print(f"  {tf}-min candles from 14 before the cross to the exit (time o h l c | EMA9 EMA20):")
        for k in sel[:60]:
            r = cd.iloc[k]
            print(f"   {pd.Timestamp(int(r.t0), unit='s', tz='UTC').tz_convert(IST):%Y-%m-%d %H:%M} {r.o:9.2f} {r.h:9.2f} {r.l:9.2f} {r.c:9.2f} | {e9[k]:9.2f} {e20[k]:9.2f}")
        print(f"  check: R = ((exit {t['exit_px']} - entry {t['entry_px']}) x dir {t['dir']} - cost {t['cost_pts']}) / risk {t['risk']} = {t['R']:+.3f}")


# ------------------------------------------------------- second data source ---
def cmd_crosscheck(a, rules: dict) -> None:
    """Headline variants, zero cost, Kaggle vs HF trademarkk spot over their common span: trade-level agreement."""
    out = {}
    for sym in a.symbols.split(","):
        mk = load_market(sym, "kaggle")
        mh = load_market(sym, "hf")
        lo = max(mk.t[0], mh.t[0]); hi = min(mk.t[-1], mh.t[-1])
        s = str(pd.Timestamp(int(lo), unit="s", tz="UTC").tz_convert(IST).date())
        e = str(pd.Timestamp(int(hi), unit="s", tz="UTC").tz_convert(IST).date() - pd.Timedelta(days=1))
        mk = load_market(sym, "kaggle", s, e); mh = load_market(sym, "hf", s, e)
        common = np.intersect1d(mk.t, mh.t)
        ik = np.searchsorted(mk.t, common); ih = np.searchsorted(mh.t, common)
        diff = np.abs(mk.c[ik] - mh.c[ih])
        res = {"span": f"{s} .. {e}", "minutes_kaggle": len(mk), "minutes_hf": len(mh), "common_minutes": int(len(common)),
               "close_abs_diff_median": round(float(np.median(diff)), 3), "close_abs_diff_p99": round(float(np.percentile(diff, 99)), 3),
               "variants": {}}
        heads = [v for v in rules["nine20"]["variants"] if v["name"] in rules["nine20"]["headline"]
                 and "nse" in v.get("markets", ["nse"])]
        for v in heads:
            tk, _ = strat_nine20(mk, v, rules, zero=True)
            th, _ = strat_nine20(mh, v, rules, zero=True)
            dk = pd.DataFrame(tk); dh = pd.DataFrame(th)
            same = set(dk["entry_t"]) & set(dh["entry_t"]) if len(dk) and len(dh) else set()
            both = 0
            if same:
                mk_ = dk.set_index("entry_t")["reason"]; mh_ = dh.set_index("entry_t")["reason"]
                both = int(sum(mk_[x] == mh_[x] for x in same))
            res["variants"][v["name"]] = {
                "kaggle_trades": len(dk), "hf_trades": len(dh), "same_entry_minute": len(same), "same_entry_and_exit_type": both,
                "kaggle_target_hit": round(100 * float((dk["reason"] == "target").mean()), 1) if len(dk) else None,
                "hf_target_hit": round(100 * float((dh["reason"] == "target").mean()), 1) if len(dh) else None,
                "kaggle_avg_R": round(float(dk["R"].mean()), 3) if len(dk) else None,
                "hf_avg_R": round(float(dh["R"].mean()), 3) if len(dh) else None}
            print(sym, v["name"][:50], res["variants"][v["name"]])
        out[sym] = res
    OUT_DIR.mkdir(parents=True, exist_ok=True)
    p = OUT_DIR / "sb_crosscheck.json"
    p.write_text(json.dumps(out, indent=1))
    print("wrote", p.relative_to(ROOT))


# ------------------------------------------------- option execution check ---
def _expiry_list() -> list[str]:
    import urllib.request
    f = RAW / "hf_trademarkk/nifty_expiries.json"
    if not f.exists():
        url = "https://huggingface.co/api/datasets/thetrademarkk/india-index-options-1m/tree/main/options/NIFTY"
        d = json.loads(urllib.request.urlopen(url, timeout=60).read())
        f.parent.mkdir(parents=True, exist_ok=True)
        f.write_text(json.dumps(sorted(x["path"].split("/")[-1][:10] for x in d if x["path"].endswith(".parquet"))))
    return json.loads(f.read_text())


def _option_file(expiry: str) -> Path:
    import urllib.request
    p = RAW / "hf_trademarkk/options/NIFTY" / f"{expiry}.parquet"
    if not p.exists():
        p.parent.mkdir(parents=True, exist_ok=True)
        tmp = p.with_suffix(".part")
        urllib.request.urlretrieve(f"{HF_BASE}/options/NIFTY/{expiry}.parquet", tmp)
        tmp.rename(p)
    return p


def cmd_options(a, rules: dict) -> None:
    """Real NIFTY weekly option premiums for every headline NIFTY trade inside the HF option span: buy the ATM
    (and 1-strike ITM) CE for longs / PE for shorts of the nearest weekly expiry at the entry minute, sell when the
    spot trade exits. Costs on actual premiums. Needs pyarrow."""
    tag = a.tag or "NIFTY"
    res = json.loads((OUT_DIR / f"sb_results_{tag}.json").read_text())
    B = res["symbols"]["NIFTY"]["strategies"]["nine20"]
    heads = rules["nine20"]["headline"]
    exps = _expiry_list()
    C = rules["costs"]
    out = {"_method": cmd_options.__doc__, "sample": f"every {getattr(a, 'every', 1)}th weekly NIFTY expiry from {exps[0]}", "variants": {}}
    cache: dict[str, pd.DataFrame] = {}
    for name in heads:
        x = B["variants"].get(name, {})
        if not x.get("trades_file"):
            continue
        tr = pd.read_csv(PROJ / x["trades_file"])
        tr = tr[(tr["entry_t"] >= exps[0]) & (tr["entry_t"] <= "2026-07-20")]
        rows = []
        every = max(1, int(getattr(a, "every", 1) or 1))
        for _, t in tr.iterrows():
            day = t["entry_t"][:10]
            exp = next((e for e in exps if e >= day), None)
            if exp is None:
                continue
            if exps.index(exp) % every:
                continue                     # systematic sample: every N-th weekly expiry (decided before any result)
            if exp not in cache:
                try:
                    df = pd.read_parquet(_option_file(exp))
                except Exception as ex:  # noqa: BLE001
                    print("  option file failed", exp, ex)
                    cache[exp] = pd.DataFrame()
                    continue
                tcol = next(c for c in df.columns if c.lower() in ("timestamp", "datetime", "date", "time", "ts"))
                ts = pd.to_datetime(df[tcol])
                if ts.dt.tz is None:
                    ts = ts.dt.tz_localize(IST)
                df["_ts"] = ts.dt.tz_convert(IST).dt.strftime("%Y-%m-%dT%H:%M")
                kc = next(c for c in df.columns if c.lower() in ("strike", "strike_price"))
                oc = next(c for c in df.columns if c.lower() in ("option_type", "type", "right", "opt_type", "instrument_type"))
                df = df.drop_duplicates(subset=[kc, oc, "_ts"], keep="last")   # a few minutes appear twice in the source
                cache[exp] = df
            df = cache[exp]
            if df.empty:
                continue
            kcol = next(c for c in df.columns if c.lower() in ("strike", "strike_price"))
            ocol = next(c for c in df.columns if c.lower() in ("option_type", "type", "right", "opt_type", "instrument_type"))
            for itm in (0, 1):
                otype = "CE" if t["dir"] == 1 else "PE"
                atm = round(t["entry_px"] / 50) * 50
                strike = atm - 50 * itm if otype == "CE" else atm + 50 * itm
                sel = df[(df[kcol] == strike) & (df[ocol].astype(str).str.upper().str[:2] == otype)].set_index("_ts")
                # entry = close of the minute the entry candle closed in (entry_t is the next minute's start)
                te = (pd.Timestamp(t["entry_t"]) - pd.Timedelta(minutes=1)).strftime("%Y-%m-%dT%H:%M")
                tx = pd.Timestamp(t["exit_t"]).strftime("%Y-%m-%dT%H:%M")
                if te not in sel.index or tx not in sel.index:
                    rows.append({"entry_t": t["entry_t"], "itm": itm, "missing": True})
                    continue
                pe, px = float(sel.at[te, "close"]), float(sel.at[tx, "close"])
                lot = _sched(C["lot_size"]["NIFTY"], day)
                slip = C["slippage_premium_pts_per_side"]["NIFTY"]
                pct = (_sched(C["stt_option_sell"], day) * px + _sched(C["nse_txn_option"], day) * (pe + px) * (1 + C["gst"])
                       + C["stamp_buy"] * pe)
                cost = 2 * C["brokerage_per_order_inr"] * (1 + C["gst"]) / lot + pct + 2 * slip
                rows.append({"entry_t": t["entry_t"], "exit_t": t["exit_t"], "itm": itm, "expiry": exp, "strike": strike,
                             "type": otype, "spot_reason": t["reason"], "spot_R": t["R"], "prem_in": pe, "prem_out": px,
                             "gross_prem_pts": px - pe, "cost_prem_pts": cost, "net_prem_pts": px - pe - cost,
                             "net_inr_per_lot": (px - pe - cost) * lot, "lot": lot, "expiry_day": exp == day, "missing": False})
        dfr = pd.DataFrame(rows)
        summ = {}
        for itm, g in dfr.groupby("itm") if len(dfr) else []:
            g2 = g[~g["missing"]]
            if not len(g2):
                continue
            summ["ATM" if itm == 0 else "1 strike ITM"] = {
                "trades": int(len(g2)), "missing": int(g["missing"].sum()),
                "span": f"{g2['entry_t'].min()[:10]} .. {g2['entry_t'].max()[:10]}",
                "win_rate_pct_net": round(100 * float((g2["net_prem_pts"] > 0).mean()), 1),
                "spot_target_hit_pct": round(100 * float((g2["spot_reason"] == "target").mean()), 1),
                "avg_net_prem_pts": round(float(g2["net_prem_pts"].mean()), 2),
                "avg_gross_prem_pts": round(float(g2["gross_prem_pts"].mean()), 2),
                "avg_cost_prem_pts": round(float(g2["cost_prem_pts"].mean()), 2),
                "total_net_inr_1lot": round(float(g2["net_inr_per_lot"].sum()), 0),
                "total_gross_inr_1lot": round(float((g2["gross_prem_pts"] * g2["lot"]).sum()), 0),
                "expiry_day_trades": int(g2["expiry_day"].sum()),
                "by_year": {y: {"trades": int(len(h)), "win_rate_pct_net": round(100 * float((h["net_prem_pts"] > 0).mean()), 1),
                                "total_net_inr_1lot": round(float(h["net_inr_per_lot"].sum()), 0)}
                            for y, h in g2.groupby(g2["entry_t"].str[:4])}}
        out["variants"][name] = summ
        if len(dfr):
            tdir = RAW / "trades" / "options"
            tdir.mkdir(parents=True, exist_ok=True)
            dfr.to_csv(tdir / f"NIFTY_{hashlib.md5(name.encode()).hexdigest()[:8]}_options.csv.gz", index=False)
        print(name[:60], json.dumps(summ)[:400])
        if len(cache) > 40:
            cache.clear()
    p = OUT_DIR / "sb_options_NIFTY.json"
    p.write_text(json.dumps(out, indent=1, default=str))
    print("wrote", p.relative_to(ROOT))


# ------------------------------------------------------------- his record ---
SENSIBULL_CSV = PROJ / "evidence/snapshots/pnl/sensibull_daily.csv"
CLAIMS = PROJ / "record_claims.json"


def _sessions() -> pd.DatetimeIndex:
    """NSE regular sessions (>= 300 one-minute bars of NIFTY) - the trading calendar."""
    z = np.load(RAW / "NIFTY_m1.npz")
    days = pd.Series(pd.to_datetime((z["t"] + 19800) // 86400, unit="D")).value_counts()
    return pd.DatetimeIndex(sorted(days[days >= 300].index))


def _bar(m: Market, ts_ist: str, minutes: int = 0) -> dict:
    """OHLC of the 1-minute bar at an IST time (or the window [ts, ts+minutes])."""
    t0 = int(pd.Timestamp(ts_ist, tz=IST).timestamp())
    sel = (m.t >= t0) & (m.t <= t0 + minutes * 60)
    if not sel.any():
        return {}
    return {"open": float(m.o[sel][0]), "high": float(m.h[sel].max()), "low": float(m.l[sel].min()), "close": float(m.c[sel][-1])}


def cmd_record(a, rules: dict) -> None:
    """His PUBLIC RECORD: the Sensibull-verified daily P&L (Zerodha, 17 May 2023 - 4 Sep 2024) against his own
    claims for the same periods (record_claims.json), coverage of trading sessions, and market-data checks of the
    trades he showed. Writes sb_record_audit.json."""
    C = json.loads(CLAIMS.read_text())
    d = pd.read_csv(SENSIBULL_CSV)
    d["date"] = pd.to_datetime(d["date_ist"])
    sess = _sessions()
    span = sess[(sess >= d["date"].min()) & (sess <= d["date"].max())]
    shared = set(d["date"])
    unshared = [x for x in span if x not in shared]
    runs, cur = [], []
    for x in span:
        if x in shared:
            if cur:
                runs.append(cur)
            cur = []
        else:
            cur.append(x)
    if cur:
        runs.append(cur)
    runs = sorted(runs, key=len, reverse=True)
    t = d[~d["no_trade_day"]].copy()
    v = t["booked_pnl_inr"]
    monthly = []
    for per, g in t.groupby(t["date"].dt.to_period("M")):
        ms = span[(span >= per.start_time) & (span <= per.end_time)]
        monthly.append({"month": str(per), "sessions": int(len(ms)), "shared_trade_days": int(len(g)),
                        "shared_no_trade_days": int(((d["date"].dt.to_period("M") == per) & d["no_trade_day"]).sum()),
                        "unshared_sessions": int(sum(1 for x in ms if x not in shared)),
                        "total_inr": round(float(g["booked_pnl_inr"].sum()), 0),
                        "green_days": int((g["booked_pnl_inr"] > 0).sum()), "red_days": int((g["booked_pnl_inr"] < 0).sum()),
                        "best_inr": round(float(g["booked_pnl_inr"].max()), 0), "worst_inr": round(float(g["booked_pnl_inr"].min()), 0)})
    years = {str(y): {"shared_trade_days": int(len(g)), "total_inr": round(float(g["booked_pnl_inr"].sum()), 0)}
             for y, g in t.groupby(t["date"].dt.year)}

    def window(lo: str, hi: str) -> dict:
        ms = span[(span >= pd.Timestamp(lo)) & (span <= pd.Timestamp(hi))]
        g = t[(t["date"] >= pd.Timestamp(lo)) & (t["date"] <= pd.Timestamp(hi))]
        return {"window": f"{lo} .. {hi}", "sessions_in_window": int(len(ms)), "shared_trade_days": int(len(g)),
                "unshared_sessions": [x.strftime("%Y-%m-%d") for x in ms if x not in shared],
                "verified_total_inr": round(float(g["booked_pnl_inr"].sum()), 0),
                "verified_worst_day_inr": round(float(g["booked_pnl_inr"].min()), 0) if len(g) else None}
    W = {"R1": ("2023-05-01", "2023-05-31"), "R2": ("2023-06-01", "2023-06-30"), "R3": ("2023-07-01", "2023-07-31"),
         "R4": ("2023-08-03", "2023-08-03"), "R5": ("2023-01-01", "2023-12-31"), "R6": ("2024-04-01", "2024-04-28"),
         "R7": ("2024-04-01", "2024-04-30"), "R8": ("2024-05-01", "2024-05-31"), "R9": ("2024-06-19", "2024-06-20"),
         "R10": ("2024-01-01", "2024-06-30")}
    claims = []
    for c in C["claims"]:
        lo, hi = W[c["id"]]
        claims.append({**c, "verified": window(lo, hi)})
    # market-data checks of trades he showed
    checks = []
    try:
        m = load_market("NIFTY")
        b19 = _bar(m, "2024-06-19 15:29"); o20 = _bar(m, "2024-06-20 09:15"); d20 = _bar(m, "2024-06-20 09:15", 374)
        s20 = _bar(m, "2024-06-20 15:29"); d19 = _bar(m, "2024-06-19 09:15", 374)
        checks.append({"what": "20 Jun 2024 NIFTY 23700 CE, -₹13.18L (Sensibull card 'Taken @ 20 Jun 2024, 3:56 PM', NIFTY 23567.00)",
                       "nifty_19jun": d19, "nifty_19jun_close_1529": b19.get("close"), "nifty_20jun": d20,
                       "gap_20jun_open_vs_19jun_close_pts": round(o20.get("open", np.nan) - b19.get("close", np.nan), 2),
                       "nifty_20jun_1529_close": s20.get("close"),
                       "reading": "23700 CE expired out of the money if NIFTY stayed below 23,700 on expiry day (20 Jun 2024, Thursday)"})
        b03 = _bar(m, "2023-08-03 09:15", 374)
        checks.append({"what": "3 Aug 2023 (Thursday, NIFTY weekly expiry) - the disputed '+5 lakh, 150% ROI' vlog day", "nifty_3aug2023": b03,
                       "reading": "the day is not in his Sensibull-verified record (unshared run 6 Jul - 11 Aug 2023)"})
    except FileNotFoundError:
        pass
    # the option itself (needs pyarrow + the HF expiry file data_raw/hf_trademarkk/options/NIFTY/2024-06-20.parquet)
    try:
        f = RAW / "hf_trademarkk/options/NIFTY/2024-06-20.parquet"
        oc = pd.read_parquet(f)
        oc = oc[(oc["strike"] == 23700) & (oc["option_type"] == "CE")].copy()
        oc["ist"] = pd.to_datetime(oc["timestamp"]).dt.tz_convert(IST)
        prem = {}
        for day in ("2024-06-18", "2024-06-19", "2024-06-20"):
            g = oc[oc["ist"].dt.strftime("%Y-%m-%d") == day].sort_values("ist")
            if len(g):
                prem[day] = {"open": float(g["open"].iloc[0]), "high": float(g["high"].max()), "low": float(g["low"].min()),
                             "close": float(g["close"].iloc[-1]),
                             "first_15min_low_high": [float(g["low"].iloc[:15].min()), float(g["high"].iloc[:15].max())]}
        checks.append({"what": "NIFTY 20 Jun 2024 23700 CE premiums (HF thetrademarkk 1-minute, CC BY-NC cross-check)", "premium_by_day": prem,
                       "his_story": "bought at ~50, ~40, ~30 (average ~33-36), held overnight, gap-up ~70 points, exits ~17-20, last ~13; ~₹26L position; ~₹13L loss (rMkxKGIBFR4)",
                       "implied_units": "₹13.18L / (~34 - ~17) ≈ 77,000 units ≈ ₹26L at ₹34 - consistent with his '26 lakh' position"})
    except Exception as ex:  # noqa: BLE001
        checks.append({"what": "NIFTY 20 Jun 2024 23700 CE premiums", "status": f"not run here ({type(ex).__name__}); needs pyarrow"})
    for sym, day, spot, lbl in (("BANKNIFTY", "2023-12-06", 46812.85, "6 Dec 2023 BANKNIFTY 46800 PE +₹10.67L (card 'Taken @ 6 Dec 2023, 3:30 PM', BANKNIFTY 46812.85)"),
                                ("FINNIFTY", "2023-08-14", 19652.50, "14 Aug 2023 FINNIFTY 19500 PE / 19600 CE / 19650 PE +₹1.15L (card 'Taken @ 14 Aug 2023, 8:50 PM', FINNIFTY 19652.50)")):
        try:
            mm = load_market(sym)
            dd = _bar(mm, f"{day} 09:15", 374)
            checks.append({"what": lbl, f"{sym.lower()}_day": dd, "card_spot": spot,
                           "close_minus_card_spot": round(dd.get("close", np.nan) - spot, 2) if dd else None})
        except FileNotFoundError:
            checks.append({"what": lbl, "status": f"{sym} 1-minute data not built"})
    for sym, lo, hi, lbl, lev in (("BTCUSDT", "2024-11-20", "2024-11-25", "Nov 2024 BTC long shown live: entry 97,594, mark 98,736 (DLOxd7t8nQs, 25 Nov 2024)", (97594, 98736)),
                                  ("BTCUSDT", "2025-04-13", "2025-04-23", "Apr 2025 BTC short from ~13 Apr, scaled 2->12 BTC into a spike on 21 Apr, closed 22 Apr; -$34,000 (jnLWteKE5SQ)", None),
                                  ("ETHUSDT", "2026-09-21", "2026-09-21", "21 Sep 2026: 150 ETH short, -$12,000 = -$80/ETH needed (gpflJzHCY5Q)", None),
                                  ("BTCUSDT", "2026-09-21", "2026-09-21", "21 Sep 2026: 2 BTC short stopped, -$4,290 = -$2,145/BTC needed (gpflJzHCY5Q)", None)):
        try:
            mm = load_market(sym, start=lo, end=hi)
            r = {"what": lbl, "window_ist": f"{lo} .. {hi}", "low": float(mm.l.min()), "high": float(mm.h.max()),
                 "first_open": float(mm.o[0]), "last_close": float(mm.c[-1]), "source": "Binance spot 5-minute (proxy for Delta perpetuals)"}
            if lev:
                r["levels_inside_range"] = [bool(mm.l.min() <= x <= mm.h.max()) for x in lev]
            checks.append(r)
        except FileNotFoundError:
            pass
    out = {"_title": "Stock Burner - his public record vs his claims" + (" (PRELIMINARY)" if rules.get("_preliminary", True) else ""), "_method": cmd_record.__doc__,
           "sensibull": C["sensibull"],
           "coverage": {"span": f"{span[0].date()} .. {span[-1].date()}", "nse_sessions": int(len(span)),
                        "sessions_shared": int(len(span) - len(unshared)), "sessions_unshared": int(len(unshared)),
                        "shared_trade_days": int(len(t)), "shared_no_trade_days": int(d["no_trade_day"].sum()),
                        "longest_unshared_runs": [{"from": r_[0].strftime("%Y-%m-%d"), "to": r_[-1].strftime("%Y-%m-%d"), "sessions": len(r_)} for r_ in runs[:4]],
                        "disputed_days_present": {"2023-08-03": pd.Timestamp("2023-08-03") in shared, "2024-06-19": pd.Timestamp("2024-06-19") in shared}},
           "totals": {"verified_total_inr": round(float(v.sum()), 0), "green_days": int((v > 0).sum()), "red_days": int((v < 0).sum()),
                      "green_day_pct": round(100 * float((v > 0).mean()), 1), "avg_green_inr": round(float(v[v > 0].mean()), 0),
                      "avg_red_inr": round(float(v[v < 0].mean()), 0),
                      "best_day": [t.loc[v.idxmax(), "date_ist"], round(float(v.max()), 0)],
                      "worst_day": [t.loc[v.idxmin(), "date_ist"], round(float(v.min()), 0)],
                      "_gross_note": "position P&L at the time of sharing; brokerage, STT, exchange charges, GST and stamp duty are not deducted on the share cards"},
           "years": years, "monthly": monthly, "claims_vs_verified": claims, "roi_framing": C["roi_framing"], "market_checks": checks}
    OUT_DIR.mkdir(parents=True, exist_ok=True)
    p = OUT_DIR / "sb_record_audit.json"
    p.write_text(json.dumps(out, indent=1, ensure_ascii=False, default=lambda x: x.item() if hasattr(x, "item") else str(x)))
    print("wrote", p.relative_to(ROOT))
    print(json.dumps({k: out[k] for k in ("coverage", "totals", "years")}, indent=1, default=str))


# ------------------------------------------------------------- proof export ---
def cmd_export(a, rules: dict) -> None:
    """Trade CSVs of the headline readings for the proof page + sb_headline.json (the numbers a script may quote)."""
    tdir = OUT_DIR / "trades"
    tdir.mkdir(parents=True, exist_ok=True)
    cols = ["entry_t", "exit_t", "dir", "entry_px", "stop", "target", "exit_px", "risk", "cost_pts", "R", "reason"]
    head = {"_note": ("PRELIMINARY. " if rules.get("_preliminary", True) else "") + "Target hit = his 'accuracy' (1:3 / stated target reached on the spot chart before the stop). "
                     "Avg R after costs (statutory option charges + stated slippage) and with zero costs.", "claims": rules["claims"], "rows": []}
    for tag in a.tag.split(","):
        R_ = json.loads((OUT_DIR / f"sb_results_{tag}.json").read_text())
        for sym, S in R_["symbols"].items():
            for st, B in S["strategies"].items():
                for name in rules[st].get("headline", []):
                    x = B["variants"].get(name)
                    if not x or not x.get("trades"):
                        continue
                    slug = f"{sym}_{st}_{hashlib.md5(name.encode()).hexdigest()[:8]}"
                    csv = None
                    if st != "ema9":                     # 25k+ trades per ema9 variant: kept in data_raw/trades only
                        df = pd.read_csv(PROJ / x["trades_file"])[cols]
                        df.to_csv(tdir / f"{slug}.csv", index=False, float_format="%.2f")
                        csv = f"trades/{slug}.csv"
                    z = (B.get("no_cost") or {}).get(name, {})
                    head["rows"].append({"symbol": sym, "strategy": st, "variant": name, "span": S["span"], "trades": x["trades"],
                                         "target_hit_pct": x["target_hit_rate_pct"], "target_hit_95ci": x["target_hit_95ci_pct"],
                                         "win_rate_pct": x["win_rate_pct"], "avg_R": x["avg_R"], "avg_R_95ci": x["avg_R_95ci"],
                                         "avg_R_zero_cost": z.get("avg_R"), "target_hit_zero_cost_pct": z.get("target_hit_rate_pct"),
                                         "years_at_or_above_60pct": f"{len(x['years_at_or_above_claim'])} of {len(x['years'])}",
                                         "best_year": max(x["years"], key=lambda y: y["target_hit_rate"]) if x["years"] else None,
                                         "fixed_1L_account": x.get("fixed_account_blown_date") and f"blown {x['fixed_account_blown_date']}" or x.get("fixed_final_balance_inr"),
                                         "trades_csv": csv or x["trades_file"]})
    (OUT_DIR / "sb_headline.json").write_text(json.dumps(head, indent=1, ensure_ascii=False))
    print("wrote", len(head["rows"]), "headline rows and CSVs to", tdir.relative_to(ROOT))


# ----------------------------------------------------------------- report ---
def cmd_report(a, rules: dict) -> None:
    """Write projects/stockburner-audit/BACKTEST.md from results files (--tag a,b)."""
    nm = lambda k: k.replace("|", "·")
    runs = [json.loads((OUT_DIR / f"sb_results_{t}.json").read_text()) for t in a.tag.split(",")]
    claim = rules["claims"]["headline_accuracy_pct"]
    pre = any(r.get("preliminary") for r in runs)
    L = [f"# Stock Burner strategies: backtest results{' (PRELIMINARY)' if pre else ''}", "",
         "_Generated by `tools/sb_backtest.py report`. Rules, sources and method: `BACKTEST_PLAN.md` and `sb_rules.json`. "
         "Target hit = the 1:3 (or stated) target reached on the spot chart before the stop. Win = trade closed with net R > 0 "
         "after costs. Money = fixed ₹1,000 risk per trade on a ₹1,00,000 account, no compounding._", ""]
    for r in runs:
        for sym, S in r["symbols"].items():
            if S["source"] == "binance":
                L.append(f"- **{sym}**: Binance spot 5-minute candles (proxy for his Delta Exchange perpetuals), {S['span']}, {S['minutes']:,} bars.")
            else:
                L.append(f"- **{sym}**: spot index 1-minute ({S['source']}), {S['span']}, {S['minutes']:,} minutes.")
    for st in [x for x in ("nine20", "ema9", "bigbar") if x in rules]:
        heads = rules[st].get("headline", [])
        L += ["", f"## {LABELS[st]}", "",
              "| Symbol | Variant | Trades | /week | Target hit (95% CI) | Win rate (net R>0) | Avg R | PF | Break-even WR | Total R | Max DD (R) | Longest losing run | ₹1L fixed ₹1k/trade | Years ≥" + f"{claim}% target hit |",
              "|---|---|---|---|---|---|---|---|---|---|---|---|---|---|"]
        for r in runs:
            for sym, S in r["symbols"].items():
                B = S["strategies"].get(st)
                if not B:
                    continue
                for name in heads:
                    x = B["variants"].get(name)
                    if not x or not x.get("trades"):
                        continue
                    money = (f"blown {x['fixed_account_blown_date']}" if x.get("fixed_account_blown_date")
                             else f"₹{x['fixed_final_balance_inr']:,.0f}")
                    L.append(f"| {sym} | {nm(name)} | {x['trades']} | {x['trades_per_week']} | {x['target_hit_rate_pct']}% ({x['target_hit_95ci_pct'][0]}–{x['target_hit_95ci_pct'][1]}) "
                             f"| {x['win_rate_pct']}% | {x['avg_R']:+.3f} | {x['profit_factor']} | {x['breakeven_win_rate_pct']}% | {x['total_R']:+.0f} "
                             f"| {x['max_drawdown_R']} | {x['longest_losing_streak']} | {money} | {len(x['years_at_or_above_claim'])} of {len(x['years'])} |")
        for r in runs:
            for sym, S in r["symbols"].items():
                B = S["strategies"].get(st)
                if not B:
                    continue
                V = {k: x for k, x in B["variants"].items() if x.get("trades")}
                if not V:
                    continue
                best = max(V.items(), key=lambda kv: kv[1]["target_hit_rate_pct"] if kv[1]["variant"].get("rr") else -1)
                bestR = max(V.items(), key=lambda kv: kv[1]["avg_R"])
                vy = sum(len(x["years_at_or_above_claim"]) for x in V.values() if x["variant"].get("rr"))
                ny = sum(len(x["years"]) for x in V.values() if x["variant"].get("rr"))
                L += ["", f"**{sym}, all {len(B['variants'])} variants:** highest target-hit rate {best[1]['target_hit_rate_pct']}% ({nm(best[0])}); "
                          f"best average {bestR[1]['avg_R']:+.3f}R ({nm(bestR[0])}); variant-years with target hit ≥{claim}% (fixed-target variants): {vy} of {ny}; "
                          f"variants with a positive average R after costs: {sum(1 for x in V.values() if x['avg_R'] > 0)}.", "",
                      "<details><summary>Full grid</summary>\n",
                      "| Variant | Trades | Target | Win | Stop | EMA exit | Time | Avg win R | Avg loss R | BE WR | Avg R | PF | Total R | Median SL (pts) | Median cost (pts) | Cost / risk | Flags |",
                      "|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|"]
                for k, x in B["variants"].items():
                    if not x.get("trades"):
                        L.append(f"| {nm(k)} | 0 | | | | | | | | | | | | | | | {x.get('flags')} |")
                        continue
                    L.append(f"| {nm(k)} | {x['trades']} | {x['target_hit_rate_pct']}% | {x['win_rate_pct']}% | {x['stop_rate_pct']}% | {x['ema_exit_rate_pct']}% | "
                             f"{x['time_exit_rate_pct']}% | {x['avg_win_R']} | {x['avg_loss_R']} | {x['breakeven_win_rate_pct']}% | {x['avg_R']:+.3f} | {x['profit_factor']} | "
                             f"{x['total_R']:+.0f} | {x['median_risk_points']} | {x['median_cost_points']} | {x['median_cost_share_of_risk_pct']}% | {x.get('flags')} |")
                L.append("\n</details>")
                zc, op, mg = B.get("no_cost") or {}, B.get("optimistic") or {}, B.get("most_generous") or {}
                if zc:
                    L += ["", f"{sym}, headline variants with **zero costs**, the **most generous intrabar order**, and both:", "",
                          "| Variant | Target hit (costs) | Avg R (costs) | Target hit (zero cost) | Win (zero cost) | Avg R (zero cost) | PF (zero cost) | Avg R (generous intrabar) | Zero cost + generous: target / avg R |",
                          "|---|---|---|---|---|---|---|---|---|"]
                    for k in heads:
                        x, z, o, g = B["variants"].get(k, {}), zc.get(k, {}), op.get(k, {}), mg.get(k, {})
                        if not x.get("trades"):
                            continue
                        L.append(f"| {nm(k)} | {x['target_hit_rate_pct']}% | {x['avg_R']:+.3f} | {z.get('target_hit_rate_pct')}% | {z.get('win_rate_pct')}% | {z.get('avg_R')} | "
                                 f"{z.get('profit_factor')} | {o.get('avg_R')} | {g.get('target_hit_rate_pct')}% / {g.get('avg_R')} |")
                sens = B.get("slippage_sensitivity") or {}
                if sens:
                    L += ["", f"{sym}, slippage sensitivity (avg R after costs): ×0.5 / ×1 / ×2 the stated slippage:", "",
                          "| Variant | ×0.5 | ×1 | ×2 |", "|---|---|---|---|"]
                    for k in heads:
                        x = B["variants"].get(k, {})
                        if not x.get("trades"):
                            continue
                        L.append(f"| {nm(k)} | {sens.get('0.5', {}).get(k, {}).get('avg_R')} | {x['avg_R']:+.3f} | {sens.get('2.0', {}).get(k, {}).get('avg_R')} |")
                hv = [k for k in heads if B["variants"].get(k, {}).get("trades")]
                yrs = sorted({y["year"] for k in hv for y in B["variants"][k]["years"]})
                L += ["", f"{sym}, by year, headline variants (cell = target hit / win rate, trades, total R):", "",
                      "| Year | " + " | ".join(nm(k) for k in hv) + " |", "|---|" + "---|" * len(hv)]
                for yv in yrs:
                    cells = []
                    for k in hv:
                        row = next((y for y in B["variants"][k]["years"] if y["year"] == yv), None)
                        cells.append(f"{row['target_hit_rate']}% / {row['win_rate']}% ({row['trades']}, {row['total_R']:+.0f}R)" if row else "–")
                    L.append(f"| {yv} | " + " | ".join(cells) + " |")
    T = rules["nine20"]
    rd = T.get("readings", {})
    if rd:
        L += ["", "## 9-20: three readings side by side (for critics)", "",
              f"- **Exact** = {nm(rd.get('exact', ''))}", f"- **All conditions** = {nm(rd.get('all_conditions', ''))}",
              f"- **Most generous** = {nm(rd.get('generous', ''))}; also run with zero costs AND the most favourable intrabar order.",
              f"- {rd.get('_note', '')}"]
        for r in runs:
            for sym, S in r["symbols"].items():
                B = S["strategies"].get("nine20")
                if not B:
                    continue
                pre_ = "crypto_" if sym in CRYPTO else ""
                L += ["", f"**{sym}**", "", "| Reading | Trades | Target hit after costs (95% CI) | Win (net R>0) | Avg R after costs | PF | Target hit zero cost | Avg R zero cost | Zero cost + generous intrabar |",
                      "|---|---|---|---|---|---|---|---|---|"]
                for key, lab in (("exact", "Exact"), ("all_conditions", "All conditions"), ("generous", "Most generous")):
                    k = rd.get(pre_ + key)
                    x = B["variants"].get(k, {})
                    z = (B.get("no_cost") or {}).get(k, {})
                    g = (B.get("most_generous") or {}).get(k, {})
                    if not x.get("trades"):
                        continue
                    L.append(f"| {lab} | {x['trades']} | {x['target_hit_rate_pct']}% ({x['target_hit_95ci_pct'][0]}–{x['target_hit_95ci_pct'][1]}) | {x['win_rate_pct']}% | "
                             f"{x['avg_R']:+.3f} | {x['profit_factor']} | {z.get('target_hit_rate_pct')}% | {z.get('avg_R')} | "
                             f"{g.get('target_hit_rate_pct')}% / {g.get('avg_R')}R |")
    flags = T.get("interpretation_flags", {})
    if flags:
        L += ["", "## What is his definition vs our interpretation", "",
              "**Defined explicitly in his words (safe to show as 'his rule'):** " + "; ".join(flags["explicit_in_his_words"]) + ".", "",
              "**Our interpretation (he names it but never defines it; label it on screen or leave it out):** " + "; ".join(flags["our_interpretation"]) + ".", "",
              "**Not coded (pure discretion or undefined):** " + "; ".join(flags["not_coded"]) + "."]
    for f in sorted(OUT_DIR.glob("sb_premises_*.json")):
        P = json.loads(f.read_text())
        L += ["", f"## His premises ({f.name})"]
        for sym, x in P.items():
            L += ["", f"**{sym}** — 'bigger timeframe = more accurate' (lp57qG4Evj0) vs '5 minute only' (Ma1HnAXxww4); same rules, zero cost:", "",
                  "| Timeframe variant | Trades | Target hit | Win | Avg R |", "|---|---|---|---|---|"]
            for k, y in x["P2_timeframe"].items():
                L.append(f"| {nm(k)} | {y.get('trades')} | {y.get('target_hit_rate_pct')}% | {y.get('win_rate_pct')}% | {y.get('avg_R')} |")
            p3 = x.get("P3_trend_days", {})
            L += ["", f"Trend days ({p3.get('definition')}): {p3.get('trend_days_pct')}% of {p3.get('days')} days "
                      "(he says 3–4 days in 10 trend, r96JUvEANlE 15:25)."]
            pb = x.get("PB_bigbar_direction") or {}
            if pb:
                L += ["", f"**{sym}** — Big Bar: '10 में से आठ बार … बिग बार के बाद से डायरेक्शन होता है' (KpOwBYb3H7c 0:18:52), measured:", "",
                      "| Definition | Big bars | Next candle continues | Goes ½ bar further before its stop (≤3 candles) | Ordinary candles: next continues / ½-bar |",
                      "|---|---|---|---|---|"]
                for k, y in pb.items():
                    bb, bs = y["big_bars"], y["baseline_ordinary_candles"]
                    L.append(f"| {nm(k)} | {bb['bars']} | {bb['next_candle_continues_pct']}% | {bb['extends_half_bar_before_its_stop_within_3_candles_pct']}% | "
                             f"{bs['next_candle_continues_pct']}% / {bs['extends_half_bar_before_its_stop_within_3_candles_pct']}% |")
    f = OUT_DIR / "sb_crosscheck.json"
    if f.exists():
        X = json.loads(f.read_text())
        L += ["", "## Second data source (sb_crosscheck.json): Kaggle spot vs Hugging Face spot, same rules, zero costs"]
        for sym, y in X.items():
            L += ["", f"**{sym}** {y['span']}: {y['common_minutes']:,} common minutes; 1-min close difference median {y['close_abs_diff_median']} pt, p99 {y['close_abs_diff_p99']} pt.", "",
                  "| Variant | Kaggle trades | HF trades | Same entry minute | Same entry + exit type | Target hit K / H | Avg R K / H |", "|---|---|---|---|---|---|---|"]
            for k, z in y["variants"].items():
                L.append(f"| {nm(k)} | {z['kaggle_trades']} | {z['hf_trades']} | {z['same_entry_minute']} | {z['same_entry_and_exit_type']} | "
                         f"{z['kaggle_target_hit']}% / {z['hf_target_hit']}% | {z['kaggle_avg_R']} / {z['hf_avg_R']} |")
    f = OUT_DIR / "sb_options_NIFTY.json"
    if f.exists():
        X = json.loads(f.read_text())
        L += ["", "## Real option premiums (sb_options_NIFTY.json): NIFTY weekly options, 1-minute, Hugging Face (CC BY-NC, cross-check only)", "",
              "| Variant | Strike | Trades | Span | Spot target hit | Option trades won (net) | Avg gross / cost / net (premium pts) | Total net ₹ (1 lot) | Total gross ₹ (1 lot) |",
              "|---|---|---|---|---|---|---|---|---|"]
        for k, y in X["variants"].items():
            for sk, z in y.items():
                L.append(f"| {nm(k)} | {sk} | {z['trades']} | {z['span']} | {z['spot_target_hit_pct']}% | {z['win_rate_pct_net']}% | "
                         f"{z['avg_gross_prem_pts']} / {z['avg_cost_prem_pts']} / {z['avg_net_prem_pts']} | ₹{z['total_net_inr_1lot']:,.0f} | ₹{z['total_gross_inr_1lot']:,.0f} |")
    f = OUT_DIR / "sb_record_audit.json"
    if f.exists():
        X = json.loads(f.read_text())
        cv, tt = X["coverage"], X["totals"]
        L += ["", "## His public record: Sensibull-verified daily P&L (sb_record_audit.json)", "",
              f"Source: {X['sensibull']['profile']} (linked from his descriptions as {X['sensibull']['short_link_in_his_descriptions']}), "
              f"Zerodha positions, retrieved {X['sensibull']['retrieved_utc'][:10]}. Sensibull: \"{X['sensibull']['sensibull_says']}\"", "",
              f"- Span {cv['span']}: {cv['nse_sessions']} NSE sessions; {cv['sessions_shared']} shared ({cv['shared_trade_days']} trade days, "
              f"{cv['shared_no_trade_days']} 'no trade day' cards); **{cv['sessions_unshared']} sessions never shared**.",
              "- Longest unshared runs: " + "; ".join(f"{r['from']} → {r['to']} ({r['sessions']} sessions)" for r in cv["longest_unshared_runs"][:2]) + ".",
              f"- Disputed days present in the verified record: " + ", ".join(f"{k}: {'yes' if v else 'no'}" for k, v in cv["disputed_days_present"].items()) + ".",
              f"- Shared trade days: total **₹{tt['verified_total_inr']:,.0f}** position P&L (before charges); {tt['green_days']} green / {tt['red_days']} red "
              f"({tt['green_day_pct']}% green); average green day ₹{tt['avg_green_inr']:,.0f}, average red day ₹{tt['avg_red_inr']:,.0f}; "
              f"best {tt['best_day'][0]} ₹{tt['best_day'][1]:,.0f}; worst {tt['worst_day'][0]} ₹{tt['worst_day'][1]:,.0f}.", "",
              "| Month | Sessions | Shared trade days | Unshared | Total (₹) | Green / red | Worst day (₹) |", "|---|---|---|---|---|---|---|"]
        for mo in X["monthly"]:
            L.append(f"| {mo['month']} | {mo['sessions']} | {mo['shared_trade_days']} | {mo['unshared_sessions']} | {mo['total_inr']:,.0f} | "
                     f"{mo['green_days']} / {mo['red_days']} | {mo['worst_inr']:,.0f} |")
        L += ["", "**His statements vs the verified record (same period):**", "",
              "| ID | Period | His words (video, time) | He said | Verified shared days | Verified total (₹) | Unshared sessions in the period |", "|---|---|---|---|---|---|---|"]
        for c in X["claims_vs_verified"]:
            v = c["verified"]
            said = f"₹{c['claimed_inr']:,.0f}" if c.get("claimed_inr") is not None else "see words"
            uns = v["unshared_sessions"]
            L.append(f"| {c['id']} | {c['period']} | {c['english']} ({c['video']} {c.get('ts', '')}) | {said}{' (ASR uncertain)' if c.get('asr_uncertain') else ''} | "
                     f"{v['shared_trade_days']} of {v['sessions_in_window']} | {v['verified_total_inr']:,.0f} | {len(uns)}{(': ' + ', '.join(uns[:6]) + ('…' if len(uns) > 6 else '')) if uns else ''} |")
        L += ["", "**ROI framing (his own words):**", ""]
        for r_ in X["roi_framing"]:
            L.append(f"- {r_['video']} {r_['ts']}: «{r_['quote']}» ({r_['english']}) — {r_['but']}")
        L += ["", "**Market-data checks of trades he showed:**", ""]
        for mc in X["market_checks"]:
            L.append(f"- {mc['what']}: " + json.dumps({k: v for k, v in mc.items() if k != 'what'}, ensure_ascii=False))
    MD.write_text("\n".join(L) + "\n")
    print("wrote", MD.relative_to(ROOT), len(L), "lines")


def main() -> None:
    ap = argparse.ArgumentParser()
    ap.add_argument("cmd", choices=["build", "hfindex", "binance", "run", "spot", "crosscheck", "premise", "options", "record", "export", "report"])
    ap.add_argument("--symbols", default="NIFTY")
    ap.add_argument("--strategies", default="nine20")
    ap.add_argument("--source", default="kaggle")
    ap.add_argument("--start", default=None)
    ap.add_argument("--end", default=None)
    ap.add_argument("--tag", default="")
    ap.add_argument("--fast", action="store_true", help="skip the zero-cost / optimistic re-runs")
    ap.add_argument("--variant", default="")
    ap.add_argument("--n", type=int, default=5)
    ap.add_argument("--seed", type=int, default=7)
    ap.add_argument("--every", type=int, default=1, help="options: use every N-th weekly expiry (systematic sample)")
    a = ap.parse_args()
    rules = json.loads(RULES.read_text())
    {"build": cmd_build, "hfindex": cmd_hfindex, "binance": cmd_binance, "run": cmd_run, "spot": cmd_spot, "crosscheck": cmd_crosscheck,
     "premise": cmd_premise, "options": cmd_options, "record": cmd_record, "export": cmd_export, "report": cmd_report}[a.cmd](a, rules)


if __name__ == "__main__":
    sys.exit(main())
