"""
Confidence Score (StockLab overhaul, Part 6/7): a 0-100 measure of how much weight to put on the
Overall Score, kept strictly separate from it. Overall Score answers "how good is this company by
the framework's own metrics"; Confidence answers "how much of the framework could actually be
computed, from what quality of inputs, and how much of it is real vs. assumed" — a company can
have a high Overall Score built from thin, mostly-assumed data (low Confidence) or a mediocre
Overall Score built from deep, fully-verified data (high Confidence). Collapsing these into one
number is exactly the failure mode `overall.py`'s renormalization audit note describes.

Every component below is computed from data this codebase actually has (pillar presence from
`Scorecard`, per-metric `DataQualityStatus` from `MetricResult`, a caller-supplied history-depth
count, and — optionally — whether DCF and multiples valuation agree). Nothing here is invented;
where an input isn't available the component is simply omitted and the remaining weights are
renormalized, the same discipline used everywhere else in this codebase (never silently 0).
"""
from __future__ import annotations

from dataclasses import dataclass, field
from typing import Optional

from app.engines.types import DataQualityStatus

# Component weights sum to 1.0 when every component is available. Configurable — spec §57.
DEFAULT_COMPONENT_WEIGHTS = {
    "pillar_completeness": 0.35,   # how many of the 5 pillars actually produced a score
    "metric_completeness": 0.25,   # how many of the ~30 metrics were computable at all
    "source_quality": 0.20,        # VERIFIED/CALCULATED vs ESTIMATED/ASSUMPTION/CONFLICTING/STALE mix
    "historical_depth": 0.10,      # years of trailing history actually available vs a 5Y target
    "valuation_agreement": 0.10,   # do DCF and multiples-based fair value roughly agree
}

_STATUS_QUALITY_SCORE = {
    DataQualityStatus.VERIFIED: 100.0,
    DataQualityStatus.CALCULATED: 90.0,
    DataQualityStatus.ESTIMATED: 55.0,
    DataQualityStatus.ASSUMPTION: 40.0,
    DataQualityStatus.STALE: 30.0,
    DataQualityStatus.CONFLICTING: 20.0,
    DataQualityStatus.MISSING: 0.0,
}

TARGET_HISTORY_YEARS = 5


@dataclass
class ConfidenceComponent:
    name: str
    weight: float
    raw_score: float  # 0-100
    contribution: float  # weight * raw_score (already renormalized weight)


@dataclass
class ConfidenceResult:
    value: Optional[float]  # 0-100, None only if literally nothing could be assessed
    components: list[ConfidenceComponent] = field(default_factory=list)


def compute_confidence_score(
    *,
    pillars_present: int,
    pillars_total: int = 5,
    metric_statuses: list[DataQualityStatus],
    metrics_missing_count: int = 0,
    years_of_history: Optional[int] = None,
    dcf_fair_value: Optional[float] = None,
    multiples_fair_value: Optional[float] = None,
    weights: Optional[dict[str, float]] = None,
) -> ConfidenceResult:
    weights = dict(weights or DEFAULT_COMPONENT_WEIGHTS)
    raw: dict[str, float] = {}

    raw["pillar_completeness"] = 100.0 * pillars_present / pillars_total if pillars_total else 0.0

    total_metrics = len(metric_statuses) + metrics_missing_count
    if total_metrics > 0:
        raw["metric_completeness"] = 100.0 * len(metric_statuses) / total_metrics

    if metric_statuses:
        raw["source_quality"] = sum(_STATUS_QUALITY_SCORE.get(s, 0.0) for s in metric_statuses) / len(metric_statuses)

    if years_of_history is not None:
        raw["historical_depth"] = 100.0 * min(years_of_history, TARGET_HISTORY_YEARS) / TARGET_HISTORY_YEARS

    if dcf_fair_value is not None and multiples_fair_value is not None and (dcf_fair_value or multiples_fair_value):
        base = max(abs(dcf_fair_value), abs(multiples_fair_value)) or 1.0
        pct_diff = abs(dcf_fair_value - multiples_fair_value) / base
        # 0% difference -> 100, >=50% difference -> 0 (linear in between). Two independent
        # valuation methods that agree closely is itself evidence the inputs are sound.
        raw["valuation_agreement"] = max(0.0, 100.0 * (1.0 - min(pct_diff / 0.50, 1.0)))

    available = {k: v for k, v in raw.items() if k in weights}
    if not available:
        return ConfidenceResult(value=None, components=[])
    total_w = sum(weights[k] for k in available)
    components = [
        ConfidenceComponent(name=k, weight=weights[k] / total_w, raw_score=v,
                             contribution=(weights[k] / total_w) * v)
        for k, v in available.items()
    ]
    value = sum(c.contribution for c in components)
    return ConfidenceResult(value=value, components=components)
