"""Run a Combine backtest over YOUR OHLCV data (CSV or Parquet -> pandas).

    uv run python examples/run_real_data.py data/sample_mnq_1m.csv \\
        --symbol MNQ --stamp open --unit minute --unit-number 1

A databento-data-playground export needs NO declaration flags — the file is
self-describing (see SELF-DESCRIBING EXPORTS below):

    uv run python examples/run_real_data.py data/MNQ.ohlcv-1m.parquet

Expects columns (case-insensitive): a timestamp column under any accepted
alias (timestamp / ts / time / datetime / ts_event — Databento uses
ts_event), plus open, high, low, close, volume; Parquet may instead carry
the timestamp as a DatetimeIndex. Timestamps must be tz-aware
(Databento is UTC; localized-to-ET works too). Four arguments are REQUIRED
with no defaults, because guessing any of them silently corrupts the run:

* ``--symbol`` declares the product (tick economics). It is cross-checked
  against the filename: a known product symbol appearing as a token of the
  file's basename that contradicts ``--symbol`` refuses the run — an ES file
  run as MNQ passes every other check and produces exactly 25x-wrong P&L.
  ``--force-symbol`` overrides a genuinely misleading filename. ``--contract``
  is cross-checked too: an id naming a different product than ``--symbol`` is
  refused outright, with no override — the id is stamped on every bar.
* ``--stamp`` declares what the timestamp MEANS: "open" = bar-open time
  (Databento's OHLCV convention), "close" = bar-close time. Declared wrong,
  every signal shifts by one bar — the classic silent look-ahead.
* ``--unit`` / ``--unit-number`` declare the bar span. A default would
  mis-stamp every non-1-minute bar's close (a 5-minute bar stamped with a
  60s span acts 4 minutes early against the session clock).

SELF-DESCRIBING EXPORTS: a Parquet file produced by databento-data-playground's
convert.py carries all of those declarations in its embedded metadata, so this
script loads it with ``topstep_backtest.data.loaders.load_bars`` instead — and
REFUSES the flags above: passing them would re-declare what the file already
states, and a contradiction between the two is exactly the silent corruption
the flags exist to prevent. The contract-roll warning below does not apply to
an export either — it is already a stitched, back-adjusted front-month series.

NO HOLIDAY FILTERING: this package ships no exchange calendar, so bars on
market holidays and bars completing after an early-close halt pass straight
through. Neither the validator nor this script will flag them. If your vendor
emits such rows and you do not want them, filter them upstream.

CONTRACT-ROLL WARNING: this script stamps the ENTIRE file with ONE contract
id, so a multi-month export spanning a quarterly roll is merged into one
fictitious contract. That does NOT corrupt P&L — the 16:10 flatten means no
position survives a roll seam — but the price jump at each seam corrupts
INDICATOR state, producing a spurious cross, an inflated ATR and a false
breakout at every roll.

For a multi-contract range, slice the export per expiry and stitch it first::

    from topstep_backtest.core.instruments import spec_for_symbol
    from topstep_backtest.data.continuous import stitch_continuous
    from topstep_backtest.data.wrangler import bars_from_dataframe

    spec = spec_for_symbol("MNQ")
    per_expiry = {
        cid: bars_from_dataframe(
            frame, contract_id=cid, spec=spec,
            unit=AggregateBarUnit.MINUTE, unit_number=1, stamp="open",
        )
        for cid, frame in frames.items()          # one frame per expiry
    }
    series = stitch_continuous(per_expiry, spec=spec)
    report = Backtest(series.bars, MyStrategy(series.symbol)).run()

``stitch_continuous`` back-adjusts additively, which is exactly P&L-neutral
here, and labels the result with the bare ticker (``"MNQ"``).

SESSIONS: if your export is a 24h tape (Globex, 18:00-17:00 ET) rather than
RTH-only, every indicator sees Asia and London by default. That is often not
what you want. ``SymbolStrategy`` takes two independent switches for it —
``use(indicator, session=NEW_YORK)`` scopes an indicator's input DATA, and
``trade_sessions=(NEW_YORK,)`` scopes when ``on_bar`` may fire — so a
continuous trend filter can sit beside a session-scoped volatility measure
while trading stays in one session. See ``examples/session_scoped.py`` and
``docs/INDICATORS.md`` §7b. Note the default ``SmaCross`` below does neither:
it trades the whole tape you hand it.

The frame is wrangled to exact on-grid Decimal bars and validated strictly by
default: ERROR-level data problems raise ``DataValidationError`` before a
single bar is traded. Swap ``SmaCross`` for your own ``SymbolStrategy``
subclass when ready.
"""

from __future__ import annotations

import argparse
import sys
from pathlib import Path
from typing import Any

from sma_cross import SmaCross
from topstep_sdk import AggregateBarUnit

from topstep_backtest import AccountSize, Backtest, DataValidationError
from topstep_backtest.core.instruments import SPECS, spec_for_symbol, symbol_of_contract_id
from topstep_backtest.data.clean import (
    AmbiguousSymbolError,
    infer_symbol_from_filename,
)
from topstep_backtest.data.loaders import load_bars
from topstep_backtest.data.wrangler import bars_from_dataframe
from topstep_backtest.fills.bar_fill import BarFillConfig

_UNITS: dict[str, AggregateBarUnit] = {
    "second": AggregateBarUnit.SECOND,
    "minute": AggregateBarUnit.MINUTE,
    "hour": AggregateBarUnit.HOUR,
}

# The wrangler's accepted timestamp spellings (data/wrangler.py), matched
# case-insensitively — Databento OHLCV exports use ``ts_event``.
_TS_ALIASES = ("timestamp", "ts", "time", "datetime", "ts_event")

_ROLL_WARNING = (
    "WARNING: the whole file is stamped with ONE contract id. A multi-month "
    "export spanning a roll keeps its P&L correct (positions never survive the "
    "16:10 flatten) but the seam corrupts INDICATOR state — a spurious cross "
    "and an inflated ATR at every roll. Slice per expiry and stitch first: "
    "data.continuous.stitch_continuous (see this file's docstring)."
)


def is_playground_export(path: Path) -> bool:
    """True when the file is a self-describing databento-data-playground export.

    The key matches what convert.py writes (and loaders.load_bars reads); a
    Parquet file without it is treated as plain OHLCV and goes down the manual,
    declare-everything path.
    """
    if path.suffix not in (".parquet", ".pq") or not path.exists():
        return False
    import pyarrow.parquet as pq  # pyright: ignore[reportMissingTypeStubs]

    return b"databento_playground" in (pq.read_schema(path).metadata or {})


def load_frame(path: Path) -> Any:
    import pandas as pd  # pyright: ignore[reportMissingTypeStubs]

    if path.suffix in (".parquet", ".pq"):
        return pd.read_parquet(path)  # pyright: ignore[reportUnknownMemberType]
    frame = pd.read_csv(path)
    # Resolve the timestamp column the way the wrangler does (case-insensitive,
    # same aliases) and parse it to datetimes here: the wrangler takes ints or
    # tz-aware datetimes, not the strings read_csv produces. tz-aware source
    # strings stay tz-aware; naive ones stay naive so the wrangler can reject
    # them with the fix named. A missing column falls through untouched — the
    # wrangler's error lists the accepted spellings.
    names: list[str] = [str(c) for c in frame.columns]  # pyright: ignore[reportUnknownVariableType, reportUnknownArgumentType]
    ts_column = next((n for n in names if n.lower() in _TS_ALIASES), None)
    if ts_column is not None:
        frame[ts_column] = pd.to_datetime(frame[ts_column])  # pyright: ignore[reportUnknownMemberType, reportUnknownArgumentType]
    return frame


def main() -> int:
    ap = argparse.ArgumentParser(
        description=__doc__,
        epilog=_ROLL_WARNING,
        formatter_class=argparse.RawDescriptionHelpFormatter,
    )
    ap.add_argument("path", type=Path, help="CSV or Parquet file of OHLCV bars")
    # The four declarations are REQUIRED for a plain file and REFUSED for a
    # self-describing export — enforced in main(), where the file has been seen,
    # rather than by argparse, which cannot know which mode it is in.
    ap.add_argument(
        "--stamp",
        choices=["open", "close"],
        help="what the source timestamp MEANS (Databento OHLCV = open); "
        "declared wrong, every signal shifts by one bar",
    )
    ap.add_argument(
        "--unit",
        choices=sorted(_UNITS),
        help="the bar span's unit (with --unit-number: 1-minute bars = "
        "--unit minute --unit-number 1)",
    )
    ap.add_argument(
        "--unit-number",
        type=int,
        help="the bar span's length in --unit units",
    )
    ap.add_argument(
        "--symbol",
        help="product symbol for the tick-economics lookup (e.g. MNQ, ES) — "
        "cross-checked against the filename: a known symbol in the basename "
        "that contradicts this refuses the run (--force-symbol overrides)",
    )
    ap.add_argument(
        "--force-symbol",
        action="store_true",
        help="skip the filename/--symbol cross-check (use only when the "
        "filename genuinely misleads about the product inside)",
    )
    ap.add_argument(
        "--contract",
        default=None,
        help="contract id stamped on EVERY row of the file (default "
        "CON.F.US.<SYMBOL>.X); must name the --symbol product — see the "
        "contract-roll warning below",
    )
    ap.add_argument("--size", choices=["50K", "100K", "150K"], default="50K")
    ap.add_argument("--qty", type=int, default=2, help="contracts per entry")
    ap.add_argument(
        "--touch-fills",
        action="store_true",
        help="optimistic limit fills on touch (default: trade-through)",
    )
    args = ap.parse_args()

    declared: dict[str, object] = {
        "--symbol": args.symbol,
        "--stamp": args.stamp,
        "--unit": args.unit,
        "--unit-number": args.unit_number,
        "--contract": args.contract,
        "--force-symbol": args.force_symbol or None,
    }
    if is_playground_export(args.path):
        passed = [flag for flag, value in declared.items() if value is not None]
        if passed:
            ap.error(
                f"{args.path.name} is a self-describing databento-data-playground "
                f"export — {', '.join(passed)} would re-declare what the file's own "
                "metadata already states, and a contradiction between the two is the "
                "silent corruption those flags exist to prevent; drop them"
            )
        bars, spec, meta = load_bars(args.path)
        contract_id = spec.symbol  # the bare root — how the export labels its bars
        rolls: dict[str, str] = meta.get("roll_days", {})
        # No _ROLL_WARNING here: the export is already a stitched, back-adjusted
        # front-month series, so there is no seam for indicators to trip on.
        print(
            f"self-describing export: {meta['bar_schema']}, stamp={meta['stamp']}, "
            f"{len(rolls)} roll(s) already stitched and back-adjusted",
            file=sys.stderr,
        )
        return run_backtest(bars, contract_id, args)

    missing = [f for f in ("--symbol", "--stamp", "--unit", "--unit-number") if declared[f] is None]
    if missing:
        ap.error(
            f"the following arguments are required: {', '.join(missing)} (only a "
            "databento-data-playground Parquet export is self-describing)"
        )

    symbol: str = args.symbol.upper()
    if symbol not in SPECS:
        ap.error(
            f"unknown product symbol {args.symbol!r}; known products: {', '.join(sorted(SPECS))}"
        )
    if not args.force_symbol:
        # Filename cross-check: an ES file run under MNQ economics passes every
        # other validation and produces exactly 25x-wrong P&L.
        try:
            inferred = infer_symbol_from_filename(args.path.name)
        except AmbiguousSymbolError as error:
            ap.error(f"{error}; pass --force-symbol to run it as {symbol} anyway")
        if inferred is not None and inferred != symbol:
            ap.error(
                f"--symbol {symbol} contradicts the filename {args.path.name!r}, which "
                f"names {inferred} — the wrong symbol means the wrong tick economics "
                "(silently wrong P&L); pass --force-symbol if the filename is misleading"
            )

    contract_id: str = args.contract or f"CON.F.US.{symbol}.X"
    # --contract cross-check: the id is stamped on every bar, so an id naming a
    # different product would trade one instrument's file under another's tick
    # economics — the same silent 25x-wrong-P&L failure the filename check
    # refuses. Unlike a filename, an id/--symbol mismatch is never merely
    # "misleading": --force-symbol does NOT bypass this.
    contract_symbol = symbol_of_contract_id(contract_id)
    if contract_symbol != symbol:
        ap.error(
            f"--contract {contract_id!r} names {contract_symbol!r}, contradicting "
            f"--symbol {symbol} — the wrong product means the wrong tick economics "
            "(silently wrong P&L); pass a contract id for the --symbol product"
        )
    spec = spec_for_symbol(symbol)
    print(_ROLL_WARNING, file=sys.stderr)

    # Wrangle (exact Decimal, on-grid, tz-checked), then Backtest(bars, ...)
    # validates strictly and assembles the sim stack.
    #
    # NOTE: no exchange-holiday filtering happens here. This package ships no
    # holiday calendar, so bars on market holidays and past early-close halts
    # pass through untouched — filter them upstream if your vendor emits them.
    bars = bars_from_dataframe(
        load_frame(args.path),
        contract_id=contract_id,
        spec=spec,
        unit=_UNITS[args.unit],
        unit_number=args.unit_number,
        stamp=args.stamp,
    )
    return run_backtest(bars, contract_id, args)


def run_backtest(bars: Any, contract_id: str, args: argparse.Namespace) -> int:
    try:
        backtest = Backtest(
            bars,
            SmaCross(contract_id, size=args.qty),  # <- swap in your strategy here
            account=AccountSize(args.size),
            fill_config=BarFillConfig(fill_limit_on_touch=args.touch_fills),
        )
    except DataValidationError as error:
        print(f"\n{error}", file=sys.stderr)
        print(
            "\nrefusing to run on ERROR-level data problems — fix the feed first", file=sys.stderr
        )
        return 1

    print(backtest.run())
    return 0


if __name__ == "__main__":
    sys.exit(main())
