#!/usr/bin/env python3
"""Coverage-only audit for authentic D48 raw entry signals.

This script deliberately stops before the portfolio, order, fill, exit, and PnL
stages.  It loads the frozen strategy class and calls Freqtrade's authentic
indicator and entry wrappers:

    strategy.advise_indicators(...) -> strategy.advise_entry(...)

This preserves ``populate_indicators`` and ``populate_entry_trend`` semantics
while avoiding backtesting's slot, lock, protection, confirmation, and order
paths.  Freqtrade 2026.6 ``backtesting --export signals`` is not a complete raw
signal source: it exports trade-producing signals and max-open-trades
rejections, but omits several other pre-order rejection paths.

The output contains no candidate returns or portfolio result.  It evaluates
only:

* point-in-time listing and complete 29-daily-candle XSMOM eligibility;
* the frozen 28-calendar-day log-return rank and top4/bottom4 decision;
* availability of the latest three finalized funding events.

The 5m signal is attached to its source candle timestamp.  Its causal eligible
time is one 5m interval later, after that candle has finalized.
"""

from __future__ import annotations

import argparse
import csv
import hashlib
import importlib.util
import json
import math
import platform
import socket
import sys
from dataclasses import dataclass
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Iterable

import pandas as pd

SCRIPT_VERSION = "1.0"
LOOKBACK_DAYS = 28
MIN_ELIGIBLE_UNIVERSE = 10
BUCKET_SIZE = 4
SIGNAL_TIMEFRAME = pd.Timedelta(minutes=5)
MAX_FUNDING_EVENT_GAP = pd.Timedelta(hours=8)

OUTPUT_FIELDS = (
    "signal_candle_time",
    "signal_eligible_time",
    "pair",
    "side",
    "strategy_class",
    "strategy_version",
    "listing_time",
    "listing_eligible",
    "first_full_listing_day",
    "last_complete_utc_day",
    "daily_start_endpoint",
    "daily_end_endpoint",
    "daily_window_after_listing",
    "daily_29_complete",
    "daily_29_positive_finite",
    "pair_rank_eligible",
    "eligible_universe_count",
    "eligible_universe",
    "rank_snapshot_valid",
    "xsmom_score",
    "xsmom_rank",
    "xsmom_bucket",
    "xsmom_decision",
    "xsmom_reason",
    "funding_event_1_time",
    "funding_event_1_rate",
    "funding_event_2_time",
    "funding_event_2_rate",
    "funding_event_3_time",
    "funding_event_3_rate",
    "funding_event_intervals_seconds",
    "funding_last3_present",
    "funding_last3_finite",
    "funding_last3_gap_le_8h",
    "funding_coverage_valid",
)


@dataclass(frozen=True)
class DailyEligibility:
    pair: str
    listing_eligible: bool
    window_after_listing: bool
    complete: bool
    positive_finite: bool
    score: float | None
    start_endpoint: pd.Timestamp
    end_endpoint: pd.Timestamp

    @property
    def eligible(self) -> bool:
        return (
            self.listing_eligible
            and self.complete
            and self.positive_finite
            and self.score is not None
        )


def utc_now() -> str:
    return datetime.now(timezone.utc).isoformat()


def sha256_file(path: Path, chunk_size: int = 1024 * 1024) -> str:
    digest = hashlib.sha256()
    with path.open("rb") as handle:
        for chunk in iter(lambda: handle.read(chunk_size), b""):
            digest.update(chunk)
    return digest.hexdigest()


def pair_file(data_dir: Path, pair: str, suffix: str) -> Path:
    contract, *settle_parts = pair.split(":", 1)
    base, quote = contract.split("/", 1)
    settle = settle_parts[0] if settle_parts else quote
    return data_dir / f"{base}_{quote}_{settle}-{suffix}.feather"


def parse_timestamp(value: str) -> pd.Timestamp:
    stamp = pd.Timestamp(value)
    return stamp.tz_localize("UTC") if stamp.tz is None else stamp.tz_convert("UTC")


def parse_pairs(raw_values: Iterable[str]) -> list[str]:
    result: list[str] = []
    for raw in raw_values:
        for item in raw.split(","):
            pair = item.strip()
            if pair and pair not in result:
                result.append(pair)
    return result


def read_config(path: Path) -> dict[str, Any]:
    payload = json.loads(path.read_text(encoding="utf-8"))
    if not isinstance(payload, dict):
        raise ValueError("Freqtrade config must contain a JSON object")
    return payload


def load_pairs(config: dict[str, Any], raw_pairs: list[str]) -> list[str]:
    if raw_pairs:
        return parse_pairs(raw_pairs)
    pairs = config.get("exchange", {}).get("pair_whitelist")
    if not isinstance(pairs, list) or not all(isinstance(pair, str) for pair in pairs):
        raise ValueError("exchange.pair_whitelist must be a string list")
    return parse_pairs(pairs)


def load_listing_dates(path: Path, pairs: list[str]) -> dict[str, pd.Timestamp]:
    payload = json.loads(path.read_text(encoding="utf-8"))
    if not isinstance(payload, dict):
        raise ValueError("listing dates must be a JSON object")
    missing = [pair for pair in pairs if pair not in payload]
    if missing:
        raise ValueError(f"listing dates missing pairs: {', '.join(missing)}")
    return {pair: parse_timestamp(str(payload[pair])) for pair in pairs}


def load_feather(path: Path) -> pd.DataFrame:
    if not path.is_file():
        raise FileNotFoundError(path)
    frame = pd.read_feather(path)
    if "date" not in frame.columns:
        raise ValueError(f"{path}: required date column missing")
    frame = frame.copy()
    frame["date"] = pd.to_datetime(frame["date"], utc=True, errors="raise")
    frame = frame.sort_values("date", kind="stable").reset_index(drop=True)
    if frame["date"].duplicated().any():
        raise ValueError(f"{path}: duplicate timestamps")
    return frame


def indexed_values(frame: pd.DataFrame, value_column: str) -> pd.Series:
    if value_column not in frame.columns:
        raise ValueError(f"frame has no {value_column} column")
    return pd.Series(
        pd.to_numeric(frame[value_column], errors="coerce").to_numpy(),
        index=pd.DatetimeIndex(frame["date"]),
        dtype="float64",
    )


def last_complete_utc_day(signal_eligible_time: pd.Timestamp) -> pd.Timestamp:
    return signal_eligible_time.normalize() - pd.Timedelta(days=1)


def first_full_listing_day(listing_time: pd.Timestamp) -> pd.Timestamp:
    """Return the first 1d candle timestamp that is not a partial listing day."""
    day = listing_time.normalize()
    return day if listing_time == day else day + pd.Timedelta(days=1)


def daily_eligibility(
    *,
    pair: str,
    daily_close: pd.Series,
    listing_time: pd.Timestamp,
    signal_eligible_time: pd.Timestamp,
) -> DailyEligibility:
    end = last_complete_utc_day(signal_eligible_time)
    start = end - pd.Timedelta(days=LOOKBACK_DAYS)
    expected = pd.date_range(start, end, freq="1D", tz="UTC")
    window = daily_close.reindex(expected)
    window_after_listing = bool(
        len(expected) == LOOKBACK_DAYS + 1
        and (expected >= first_full_listing_day(listing_time)).all()
    )
    complete = bool(
        window_after_listing
        and len(window) == LOOKBACK_DAYS + 1
        and window.notna().all()
    )
    values = window.to_numpy(dtype=float)
    positive_finite = bool(
        complete and (values > 0).all() and pd.Series(values).map(math.isfinite).all()
    )
    score = float(math.log(values[-1] / values[0])) if positive_finite else None
    return DailyEligibility(
        pair=pair,
        listing_eligible=bool(listing_time <= signal_eligible_time),
        window_after_listing=window_after_listing,
        complete=complete,
        positive_finite=positive_finite,
        score=score,
        start_endpoint=start,
        end_endpoint=end,
    )


def rank_snapshot(
    eligibility: dict[str, DailyEligibility],
) -> tuple[list[str], dict[str, int], bool]:
    eligible = [item for item in eligibility.values() if item.eligible]
    ordered = sorted(eligible, key=lambda item: (-float(item.score), item.pair))
    universe = [item.pair for item in ordered]
    ranks = {pair: index + 1 for index, pair in enumerate(universe)}
    return universe, ranks, len(universe) >= MIN_ELIGIBLE_UNIVERSE


def xsmom_decision(
    *,
    pair: str,
    side: str,
    item: DailyEligibility,
    universe: list[str],
    ranks: dict[str, int],
    snapshot_valid: bool,
) -> tuple[str, str, int | None, str | None]:
    if not snapshot_valid:
        return "block", "xsmom_snapshot_unavailable", None, None
    if not item.eligible:
        return "block", "xsmom_missing_pair", None, None

    rank = ranks[pair]
    count = len(universe)
    if rank <= BUCKET_SIZE:
        bucket = "top4"
    elif rank >= count - BUCKET_SIZE + 1:
        bucket = "bottom4"
    else:
        bucket = "middle"

    if side == "long":
        passed = bucket == "top4"
        return (
            "pass" if passed else "block",
            "xsmom_top4" if passed else "xsmom_not_top4",
            rank,
            bucket,
        )
    passed = bucket == "bottom4"
    return (
        "pass" if passed else "block",
        "xsmom_bottom4" if passed else "xsmom_not_bottom4",
        rank,
        bucket,
    )


def funding_coverage(
    funding_close: pd.Series, signal_eligible_time: pd.Timestamp
) -> dict[str, Any]:
    finalized = funding_close.loc[funding_close.index <= signal_eligible_time].tail(3)
    present = len(finalized) == 3
    times = list(finalized.index)
    rates = [float(value) for value in finalized.to_numpy(dtype=float)]
    finite = bool(present and all(math.isfinite(value) for value in rates))
    intervals = [
        int((times[index] - times[index - 1]).total_seconds())
        for index in range(1, len(times))
    ]
    gap_valid = bool(
        present
        and all(
            0 < interval <= int(MAX_FUNDING_EVENT_GAP.total_seconds())
            for interval in intervals
        )
    )
    result: dict[str, Any] = {
        "funding_last3_present": present,
        "funding_last3_finite": finite,
        "funding_last3_gap_le_8h": gap_valid,
        "funding_coverage_valid": bool(present and finite and gap_valid),
        "funding_event_intervals_seconds": intervals,
    }
    for index in range(3):
        event_number = index + 1
        result[f"funding_event_{event_number}_time"] = (
            times[index].isoformat() if index < len(times) else None
        )
        result[f"funding_event_{event_number}_rate"] = (
            rates[index] if index < len(rates) else None
        )
    return result


class HistoricalDataProvider:
    """Small DataProvider surface used by the frozen D48 indicator method."""

    def __init__(self, frames: dict[tuple[str, str], pd.DataFrame]):
        self.frames = frames

    def get_pair_dataframe(
        self, pair: str, timeframe: str | None = None, candle_type: str = ""
    ) -> pd.DataFrame:
        del candle_type
        key = (pair, timeframe or "5m")
        if key not in self.frames:
            raise KeyError(f"informative dataframe unavailable: {key}")
        return self.frames[key].copy()


def load_strategy_class(strategy_file: Path, class_name: str):
    strategy_dir = strategy_file.parent.resolve()
    if str(strategy_dir) not in sys.path:
        sys.path.insert(0, str(strategy_dir))
    spec = importlib.util.spec_from_file_location(
        f"_coverage_strategy_{strategy_file.stem}", strategy_file
    )
    if spec is None or spec.loader is None:
        raise ImportError(f"cannot load strategy module: {strategy_file}")
    module = importlib.util.module_from_spec(spec)
    spec.loader.exec_module(module)
    cls = getattr(module, class_name, None)
    if cls is None:
        raise ImportError(f"{class_name} not found in {strategy_file}")
    return cls


def strategy_source_hashes(strategy_file: Path) -> dict[str, str]:
    hashes: dict[str, str] = {}
    for path in sorted(strategy_file.parent.glob("*.py")):
        hashes[path.name] = sha256_file(path)
    return hashes


def extract_raw_signals(
    *,
    strategy: Any,
    main_frames: dict[str, pd.DataFrame],
    window_start: pd.Timestamp,
    window_end: pd.Timestamp,
) -> list[dict[str, Any]]:
    signals: list[dict[str, Any]] = []
    for pair, frame in main_frames.items():
        analyzed = strategy.advise_indicators(frame.copy(), {"pair": pair})
        analyzed = strategy.advise_entry(analyzed, {"pair": pair})
        for side, column in (("long", "enter_long"), ("short", "enter_short")):
            if column not in analyzed.columns:
                continue
            selected = analyzed.loc[analyzed[column].fillna(0).astype(bool), ["date"]]
            for candle_time in selected["date"]:
                eligible_time = pd.Timestamp(candle_time) + SIGNAL_TIMEFRAME
                if window_start <= eligible_time < window_end:
                    signals.append(
                        {
                            "signal_candle_time": pd.Timestamp(candle_time).isoformat(),
                            "signal_eligible_time": eligible_time.isoformat(),
                            "pair": pair,
                            "side": side,
                        }
                    )
    return sorted(
        signals,
        key=lambda row: (row["signal_eligible_time"], row["pair"], row["side"]),
    )


def json_csv_value(value: Any) -> Any:
    if isinstance(value, (dict, list)):
        return json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
    return value


def write_csv(path: Path, rows: list[dict[str, Any]]) -> None:
    with path.open("w", encoding="utf-8", newline="") as handle:
        writer = csv.DictWriter(handle, fieldnames=OUTPUT_FIELDS, extrasaction="ignore")
        writer.writeheader()
        for row in rows:
            writer.writerow({field: json_csv_value(row.get(field)) for field in OUTPUT_FIELDS})


def grouped_counts(rows: list[dict[str, Any]], keys: tuple[str, ...]) -> list[dict[str, Any]]:
    counts: dict[tuple[Any, ...], dict[str, int]] = {}
    for row in rows:
        group = tuple(row[key] for key in keys)
        item = counts.setdefault(
            group,
            {
                "signals": 0,
                "rank_snapshot_valid": 0,
                "pair_rank_eligible": 0,
                "xsmom_pass": 0,
                "funding_coverage_valid": 0,
            },
        )
        item["signals"] += 1
        item["rank_snapshot_valid"] += int(bool(row["rank_snapshot_valid"]))
        item["pair_rank_eligible"] += int(bool(row["pair_rank_eligible"]))
        item["xsmom_pass"] += int(row["xsmom_decision"] == "pass")
        item["funding_coverage_valid"] += int(bool(row["funding_coverage_valid"]))
    result = []
    for group, item in sorted(counts.items(), key=lambda entry: tuple(map(str, entry[0]))):
        record = {key: value for key, value in zip(keys, group, strict=True)}
        record.update(item)
        denominator = item["signals"]
        record["rank_coverage_ratio"] = (
            item["pair_rank_eligible"] / denominator if denominator else None
        )
        record["funding_coverage_ratio"] = (
            item["funding_coverage_valid"] / denominator if denominator else None
        )
        result.append(record)
    return result


def readme_text(summary: dict[str, Any]) -> str:
    return f"""# Authentic D48 raw-signal coverage audit

Generated: `{summary["generated_at"]}`

Window: `[{summary["window_start"]}, {summary["window_end"]})`

Strategy: `{summary["strategy_class"]}` / `{summary["strategy_version"]}`

This is a coverage-only artifact. It contains no candidate PnL, portfolio,
order, fill, exit, or subsequent-price result.

## Authenticity

Raw signals were generated before portfolio simulation by calling the frozen
strategy source through Freqtrade's `advise_indicators` and `advise_entry`
wrappers. Freqtrade 2026.6 `backtesting --export signals` was not used as the
raw denominator because it omits several non-trade signal paths.

The source candle time and causal signal eligible time are stored separately;
the latter is exactly five minutes after the completed 5m candle timestamp.

## Frozen XSMOM rule

- 28 completed UTC calendar days, requiring all 29 positive finite closes.
- If onboarding is intraday, its partial UTC daily candle is excluded; the
  first usable daily timestamp is the following 00:00 UTC.
- At least 10 rank-eligible pairs.
- Descending score, canonical pair string as the exact-tie breaker.
- Long passes top4; short passes bottom4.

## Funding interpretation

Funding coverage means that the three latest finalized event records at or
before signal eligibility exist, contain finite rates, and have no interval
larger than eight hours. It does not reproduce a live predicted funding rate
and is not evidence of funding alpha.
"""


def build_parser() -> argparse.ArgumentParser:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--config", type=Path, required=True)
    parser.add_argument("--strategy-file", type=Path, required=True)
    parser.add_argument("--strategy-class", default="VolatilityBreakoutRiskCap")
    parser.add_argument("--data-dir", type=Path, required=True)
    parser.add_argument("--listing-dates", type=Path, required=True)
    parser.add_argument("--pairs", action="append", default=[])
    parser.add_argument("--window-start", required=True)
    parser.add_argument("--window-end", required=True)
    parser.add_argument("--output-dir", type=Path, required=True)
    parser.add_argument(
        "--source-role",
        choices=("research_cache", "production_runtime_export", "other"),
        default="research_cache",
    )
    return parser


def main() -> int:
    args = build_parser().parse_args()
    for path, label in (
        (args.config, "config"),
        (args.strategy_file, "strategy file"),
        (args.listing_dates, "listing dates"),
    ):
        if not path.is_file():
            raise SystemExit(f"{label} does not exist: {path}")
    if not args.data_dir.is_dir():
        raise SystemExit(f"data directory does not exist: {args.data_dir}")

    window_start = parse_timestamp(args.window_start)
    window_end = parse_timestamp(args.window_end)
    if window_start >= window_end:
        raise SystemExit("--window-start must be earlier than --window-end")

    config = read_config(args.config)
    pairs = load_pairs(config, args.pairs)
    listing_dates = load_listing_dates(args.listing_dates, pairs)

    daily_frames = {
        pair: load_feather(pair_file(args.data_dir, pair, "1d-futures"))
        for pair in pairs
    }
    daily_closes = {
        pair: indexed_values(frame, "close") for pair, frame in daily_frames.items()
    }
    funding_rates = {
        pair: indexed_values(
            load_feather(pair_file(args.data_dir, pair, "1h-funding_rate")),
            "open",
        )
        for pair in pairs
    }
    main_frames = {
        pair: load_feather(pair_file(args.data_dir, pair, "5m-futures"))
        for pair in pairs
    }

    strategy_class = load_strategy_class(args.strategy_file, args.strategy_class)
    strategy_config = dict(config)
    # Raw signal methods do not use runmode-dependent order callbacks. The enum
    # is supplied when Freqtrade is importable, preserving wrapper expectations.
    try:
        from freqtrade.enums import RunMode

        strategy_config["runmode"] = RunMode.BACKTEST
    except ImportError:
        strategy_config["runmode"] = "backtest"
    strategy_config["timeframe"] = "5m"
    strategy = strategy_class(strategy_config)
    strategy.dp = HistoricalDataProvider({("BTC/USDT:USDT", "1d"): daily_frames["BTC/USDT:USDT"]})
    strategy_version = str(strategy.version())

    raw_signals = extract_raw_signals(
        strategy=strategy,
        main_frames=main_frames,
        window_start=window_start,
        window_end=window_end,
    )

    rows: list[dict[str, Any]] = []
    snapshot_cache: dict[str, tuple[dict[str, DailyEligibility], list[str], dict[str, int], bool]] = {}
    for signal in raw_signals:
        eligible_time = parse_timestamp(signal["signal_eligible_time"])
        # Listing eligibility can change within a UTC day, so a daily-only
        # cache key is not causally sufficient for intraday onboarding.
        snapshot_key = eligible_time.isoformat()
        if snapshot_key not in snapshot_cache:
            eligibility = {
                pair: daily_eligibility(
                    pair=pair,
                    daily_close=daily_closes[pair],
                    listing_time=listing_dates[pair],
                    signal_eligible_time=eligible_time,
                )
                for pair in pairs
            }
            universe, ranks, snapshot_valid = rank_snapshot(eligibility)
            snapshot_cache[snapshot_key] = (
                eligibility,
                universe,
                ranks,
                snapshot_valid,
            )
        eligibility, universe, ranks, snapshot_valid = snapshot_cache[snapshot_key]
        pair = signal["pair"]
        side = signal["side"]
        item = eligibility[pair]
        decision, reason, rank, bucket = xsmom_decision(
            pair=pair,
            side=side,
            item=item,
            universe=universe,
            ranks=ranks,
            snapshot_valid=snapshot_valid,
        )
        funding = funding_coverage(funding_rates[pair], eligible_time)
        rows.append(
            {
                **signal,
                "strategy_class": args.strategy_class,
                "strategy_version": strategy_version,
                "listing_time": listing_dates[pair].isoformat(),
                "listing_eligible": item.listing_eligible,
                "first_full_listing_day": first_full_listing_day(
                    listing_dates[pair]
                ).isoformat(),
                "last_complete_utc_day": item.end_endpoint.isoformat(),
                "daily_start_endpoint": item.start_endpoint.isoformat(),
                "daily_end_endpoint": item.end_endpoint.isoformat(),
                "daily_window_after_listing": item.window_after_listing,
                "daily_29_complete": item.complete,
                "daily_29_positive_finite": item.positive_finite,
                "pair_rank_eligible": item.eligible,
                "eligible_universe_count": len(universe),
                "eligible_universe": universe,
                "rank_snapshot_valid": snapshot_valid,
                "xsmom_score": item.score,
                "xsmom_rank": rank,
                "xsmom_bucket": bucket,
                "xsmom_decision": decision,
                "xsmom_reason": reason,
                **funding,
            }
        )

    total = len(rows)
    rank_covered = sum(
        bool(row["rank_snapshot_valid"]) and bool(row["pair_rank_eligible"])
        for row in rows
    )
    funding_covered = sum(bool(row["funding_coverage_valid"]) for row in rows)
    summary = {
        "script_version": SCRIPT_VERSION,
        "generated_at": utc_now(),
        "generated_host": socket.gethostname(),
        "source_role": args.source_role,
        "window_start": window_start.isoformat(),
        "window_end": window_end.isoformat(),
        "strategy_class": args.strategy_class,
        "strategy_version": strategy_version,
        "strategy_sources_sha256": strategy_source_hashes(args.strategy_file),
        "config_sha256": sha256_file(args.config),
        "listing_dates_sha256": sha256_file(args.listing_dates),
        "data_sha256": {
            pair: {
                suffix: sha256_file(pair_file(args.data_dir, pair, suffix))
                for suffix in ("5m-futures", "1d-futures", "1h-funding_rate")
            }
            for pair in pairs
        },
        "environment": {
            "python": sys.version,
            "platform": platform.platform(),
            "pandas": pd.__version__,
        },
        "raw_signal_count": total,
        "raw_signal_long_count": sum(row["side"] == "long" for row in rows),
        "raw_signal_short_count": sum(row["side"] == "short" for row in rows),
        "rank_covered_signal_count": rank_covered,
        "rank_coverage_ratio": rank_covered / total if total else None,
        "rank_coverage_gate_98pct": bool(total and rank_covered / total >= 0.98),
        "funding_covered_signal_count": funding_covered,
        "funding_coverage_ratio": funding_covered / total if total else None,
        "funding_coverage_gate_98pct": bool(total and funding_covered / total >= 0.98),
        "xsmom_decisions": {
            decision: sum(row["xsmom_decision"] == decision for row in rows)
            for decision in ("pass", "block")
        },
        "xsmom_reasons": dict(
            sorted(
                {
                    reason: sum(row["xsmom_reason"] == reason for row in rows)
                    for reason in {row["xsmom_reason"] for row in rows}
                }.items()
            )
        ),
        "coverage_by_pair_side": grouped_counts(rows, ("pair", "side")),
        "coverage_by_month": grouped_counts(
            [
                {
                    **row,
                    "month": row["signal_eligible_time"][:7],
                }
                for row in rows
            ],
            ("month",),
        ),
    }

    args.output_dir.mkdir(parents=True, exist_ok=True)
    write_csv(args.output_dir / "raw_signal_coverage.csv", rows)
    (args.output_dir / "summary.json").write_text(
        json.dumps(summary, indent=2, ensure_ascii=False, sort_keys=True) + "\n",
        encoding="utf-8",
    )
    (args.output_dir / "README.md").write_text(readme_text(summary), encoding="utf-8")
    print(
        json.dumps(
            {
                "raw_signals": total,
                "rank_coverage_ratio": summary["rank_coverage_ratio"],
                "funding_coverage_ratio": summary["funding_coverage_ratio"],
            },
            sort_keys=True,
        )
    )
    return 0


if __name__ == "__main__":
    raise SystemExit(main())
