"""
Robustness / parameter-sensitivity testing (StockLab overhaul, Part 21).

Same split as performance.py: the STATISTICAL technique here is real and tested against
hand-computed values. What it does NOT have yet is a real backtest to run it against (see
replay.py) — so `assess_robustness()` takes a plain `{parameter_set_label: metric_value}` mapping
rather than running anything itself, and this module has no opinion on where that mapping came
from. Once `replay.py` exists, the natural caller is: run the same strategy through the replay
loop under N small perturbations of its tunable parameters (e.g. margin-of-safety threshold ±2pp,
score weight ±5%) and pass the resulting {perturbation_label: CAGR} mapping in here.
"""
from __future__ import annotations

from dataclasses import dataclass
from typing import Optional

import numpy as np


@dataclass(frozen=True)
class RobustnessReport:
    parameter_sets_tested: int
    metric_mean: Optional[float]
    metric_stdev: Optional[float]
    coefficient_of_variation: Optional[float]
    best_case: Optional[float]
    worst_case: Optional[float]
    overfit_risk: str  # "LOW" | "MEDIUM" | "HIGH" | "INSUFFICIENT_DATA"
    overfit_risk_reason: str


MIN_PARAMETER_SETS = 4
# Coefficient-of-variation thresholds — how much the metric swings, relative to its own mean,
# across small parameter perturbations. A strategy whose CAGR halves when a threshold moves by a
# couple of percentage points is much more likely to be overfit to its exact backtest window than
# one whose performance is stable under similar perturbations. Configurable, not a hard science —
# documented as a heuristic, not a statistically-derived cutoff.
COV_LOW_RISK_MAX = 0.15
COV_MEDIUM_RISK_MAX = 0.35


def assess_robustness(metric_by_parameter_set: dict[str, float]) -> RobustnessReport:
    n = len(metric_by_parameter_set)
    if n < MIN_PARAMETER_SETS:
        return RobustnessReport(
            parameter_sets_tested=n, metric_mean=None, metric_stdev=None,
            coefficient_of_variation=None, best_case=None, worst_case=None,
            overfit_risk="INSUFFICIENT_DATA",
            overfit_risk_reason=f"Need at least {MIN_PARAMETER_SETS} parameter sets to assess "
                                 f"sensitivity; got {n}.",
        )
    values = np.asarray(list(metric_by_parameter_set.values()), dtype=float)
    mean = float(np.mean(values))
    stdev = float(np.std(values, ddof=1))
    cov = abs(stdev / mean) if mean != 0 else None

    if cov is None:
        risk, reason = "INSUFFICIENT_DATA", "Mean metric value is 0 — coefficient of variation is undefined."
    elif cov <= COV_LOW_RISK_MAX:
        risk, reason = "LOW", f"Metric varies only {cov:.1%} (relative to its mean) across {n} parameter sets."
    elif cov <= COV_MEDIUM_RISK_MAX:
        risk, reason = "MEDIUM", f"Metric varies {cov:.1%} across {n} parameter sets — moderately sensitive to tuning."
    else:
        risk, reason = "HIGH", (f"Metric varies {cov:.1%} across {n} parameter sets — small parameter "
                                 f"changes swing the result substantially, consistent with overfitting "
                                 f"to the specific backtest window rather than a robust edge.")

    return RobustnessReport(
        parameter_sets_tested=n, metric_mean=mean, metric_stdev=stdev, coefficient_of_variation=cov,
        best_case=float(np.max(values)), worst_case=float(np.min(values)),
        overfit_risk=risk, overfit_risk_reason=reason,
    )
