"""
Model versioning and reproducibility (final master pass, §39).

## The problem this solves

Every score, valuation and recommendation StockLab persists is the output of a *model*: a specific
set of metric formulas, pillar weights, scoring thresholds, peer-group rules, DCF and WACC
assumptions, margin-of-safety bands and recommendation rules. Change any one of them and every
number the platform has ever produced means something slightly different — but the stored rows
look identical. A historical score of 72 from March and a score of 72 from September are not
comparable if the weights changed in between, and nothing in the data says so.

`METHODOLOGY_VERSION` already existed and is stamped on every `MetricResult`, but it covers only
the metric formulas. Nothing versioned the scoring or valuation layers at all.

## Two numbers, and why both are needed

- **`MODEL_VERSION`** — a hand-maintained semantic version. It is a *claim*: "the model changed
  in this way". Humans read it, and it goes in the changelog.
- **`model_fingerprint()`** — a deterministic hash over the actual constants the model is made of.
  It is *evidence*: it changes automatically whenever a weight, threshold or assumption changes,
  whether or not anyone remembered to bump `MODEL_VERSION`.

A version string on its own is only as reliable as the discipline of the person editing the
constants. The fingerprint is what makes the claim checkable, and
`tests/test_model_version.py::test_fingerprint_changes_when_a_weight_changes` is what proves the
fingerprint is not decorative.

## What is versioned, and what is deliberately not

**Versioned** (any change alters the fingerprint): the metric formula version, the five pillar
weights, the renormalisation cap, the minimum peer-group size, the industry-multiple minimum
group size, the multiples reference preference, the DCF explicit window, terminal-growth default,
fade behaviour and scenario mode, the WACC defaults, and the margin-of-safety bands.

**Not versioned**: anything read from the environment per deployment that does not change the
*meaning* of a number — database URLs, pool sizes, rate limits, Celery concurrency, log levels.
Including them would make the fingerprint change for reasons that have nothing to do with the
model, which would destroy its usefulness as a comparability signal.

## What this module does NOT claim

It does not make historical results reproducible on its own. Reproducing a score also needs the
*input snapshot* that produced it — the exact line items known at that date. `metric_history`
stores per-metric values with an `as_of` date, and `financial_periods` stores the filings, so the
inputs are recoverable in principle; no code path in this repository currently re-runs a scoring
pass against a historical snapshot and asserts it reproduces the stored number. That is the
missing half of reproducibility and it is recorded as such in `docs/AUDIT_PILLARS.md` —
**NOT IMPLEMENTED**, not "supported".
"""
from __future__ import annotations

import hashlib
import json
from typing import Any

#: Bump the MINOR component when a weight, threshold or assumption changes in a way that alters
#: scores; bump MAJOR when the framework itself changes (a pillar added or removed, a metric's
#: formula redefined). The fingerprint below is what actually detects a change — this string is
#: the human-readable claim about it.
MODEL_VERSION = "2.0.0"

#: What changed, newest first. Keep this honest: it is the only place a reader can learn why two
#: historical scores are not comparable.
MODEL_VERSION_HISTORY: tuple[tuple[str, str], ...] = (
    (
        "2.0.0",
        "Final master engineering pass. Scores from this version are NOT comparable with 1.x: "
        "peer groups are now actually computed (before, every percentile was a neutral 50), the "
        "Competitive Advantage pillar is actually wired (before, it was INSUFFICIENT_DATA for "
        "every security and its 15% weight was silently renormalised away), a metric with no "
        "peers is now excluded rather than scored 50, ROIC no longer discards a genuine 0% tax "
        "rate, the DCF refuses a missing share count instead of dividing by one share, and the "
        "multiples side of the Fair Value blend uses a median rather than an outlier-sensitive "
        "mean.",
    ),
    (
        "1.0.0",
        "The 0.2.0 overhaul baseline: 30 audited metrics, five pillars, WACC/DCF/multiples, "
        "recommendation engine. Retrospective label — nothing was stamped with it at the time, "
        "which is precisely the gap this module closes.",
    ),
)


def model_configuration() -> dict[str, Any]:
    """The constants that define the model, gathered from the modules that own them.

    Imported inside the function rather than at module scope so that importing
    `model_version` stays dependency-free — `app.core.config` pulls in `pydantic_settings`, and a
    caller that only wants `MODEL_VERSION` should not need it.
    """
    from app.core.config import get_settings
    from app.engines.metrics.core import METHODOLOGY_VERSION
    from app.engines.scoring.overall import MAX_MISSING_WEIGHT_FOR_RENORMALIZATION
    from app.engines.valuation.dcf import EXPLICIT_YEARS

    s = get_settings()
    return {
        "metric_formula_version": METHODOLOGY_VERSION,
        "pillar_weights": {
            "quality": s.SCORE_WEIGHT_QUALITY,
            "financial_health": s.SCORE_WEIGHT_FINANCIAL_HEALTH,
            "growth": s.SCORE_WEIGHT_GROWTH,
            "competitive_advantage": s.SCORE_WEIGHT_COMPETITIVE_ADVANTAGE,
            "valuation": s.SCORE_WEIGHT_VALUATION,
        },
        "max_missing_weight_for_renormalization": MAX_MISSING_WEIGHT_FOR_RENORMALIZATION,
        "min_peer_group_size": s.MIN_PEER_GROUP_SIZE,
        "industry_multiple_min_group_size": s.INDUSTRY_MULTIPLE_MIN_GROUP_SIZE,
        "multiples_reference_preference": s.MULTIPLES_REFERENCE_PREFERENCE,
        "dcf": {
            "explicit_years": EXPLICIT_YEARS,
            "terminal_growth": s.DEFAULT_DCF_TERMINAL_GROWTH,
            "revenue_growth_fallback": s.DEFAULT_DCF_REVENUE_GROWTH_FALLBACK,
            "reinvestment_rate_fallback": s.DEFAULT_DCF_REINVESTMENT_RATE_FALLBACK,
            "corporate_tax_rate": s.DEFAULT_CORPORATE_TAX_RATE,
            "growth_fade": s.DCF_GROWTH_FADE,
            "growth_fade_years": s.DCF_GROWTH_FADE_YEARS,
            "scenario_mode": s.DCF_SCENARIO_MODE,
        },
        "wacc": {
            "risk_free_rate": s.DEFAULT_RISK_FREE_RATE,
            "equity_risk_premium": s.DEFAULT_EQUITY_RISK_PREMIUM,
            "beta_if_missing": s.DEFAULT_BETA_IF_MISSING,
            "min_cost_of_debt": s.MIN_COST_OF_DEBT,
        },
        "margin_of_safety": {
            "strong_buy": s.DEFAULT_STRONG_BUY_MOS,
            "buy": s.DEFAULT_BUY_MOS,
            "overvalued_premium": s.DEFAULT_OVERVALUED_PREMIUM,
        },
        "reporting_basis": s.FINANCIAL_BASIS,
    }


def model_fingerprint(configuration: dict[str, Any] | None = None) -> str:
    """A 16-character hash of the model configuration.

    Deterministic across processes and machines: the configuration is serialised with sorted keys
    so dictionary ordering cannot change the result. Truncated to 16 hex characters — enough to
    make an accidental collision implausible, short enough to sit in a log line or a JSON payload
    without dominating it.
    """
    config = model_configuration() if configuration is None else configuration
    payload = json.dumps(config, sort_keys=True, separators=(",", ":"), default=str)
    return hashlib.sha256(payload.encode("utf-8")).hexdigest()[:16]


def model_stamp() -> dict[str, Any]:
    """What gets attached to a persisted score, valuation or recommendation."""
    config = model_configuration()
    return {
        "model_version": MODEL_VERSION,
        "model_fingerprint": model_fingerprint(config),
        "metric_formula_version": config["metric_formula_version"],
    }
