"""
Data Quality Score (StockLab overhaul, Part 6/g): a 0-100 measure of the reliability of the raw
and derived data behind a security, independent of both Overall Score and Confidence.

Distinct from Confidence (confidence.py) on purpose, even though both read from the same
DataQualityStatus population and there is deliberate overlap: Confidence asks "how much of the
scoring framework could be computed and how much should I trust the resulting Score", scoped to
one calculation run; Data Quality asks "how reliable is this company's underlying data,
independent of what today's scoring run happened to need", scoped to the data itself, and is the
number that should stay comparable across companies/runs even as the scoring framework's metric
list evolves. It is driven by the provider source-tier hierarchy
(`app/models/governance.py::Source.provider_tier` — OFFICIAL_FILING > PRIMARY > SECONDARY >
CALCULATED, docs/DATA_SOURCES.md §4) and the same per-field DataQualityStatus used everywhere
else, never a separate invented taxonomy.

Kept dependency-free like every other engine module: callers pass in the tier/status lists they
already have (from `Source`/`DataQuality` rows in a real deployment, or from fixtures in tests) —
this module does no I/O.
"""
from __future__ import annotations

from dataclasses import dataclass, field
from typing import Optional

from app.engines.types import DataQualityStatus

SOURCE_TIER_SCORE = {
    "OFFICIAL_FILING": 100.0,
    "PRIMARY": 85.0,
    "SECONDARY": 60.0,
    "CALCULATED": 40.0,
}

_STATUS_SCORE = {
    DataQualityStatus.VERIFIED: 100.0,
    DataQualityStatus.CALCULATED: 90.0,
    DataQualityStatus.ESTIMATED: 55.0,
    DataQualityStatus.ASSUMPTION: 40.0,
    DataQualityStatus.STALE: 25.0,
    DataQualityStatus.CONFLICTING: 15.0,
    DataQualityStatus.MISSING: 0.0,
}

DEFAULT_WEIGHTS = {"source_tier": 0.6, "field_status": 0.4}


@dataclass
class DataQualityComponent:
    name: str
    weight: float
    raw_score: float
    contribution: float


@dataclass
class DataQualityResult:
    value: Optional[float]
    components: list[DataQualityComponent] = field(default_factory=list)
    conflicting_field_count: int = 0
    unknown_source_tiers: list[str] = field(default_factory=list)


def compute_data_quality_score(
    *,
    source_tiers: list[str],
    field_statuses: list[DataQualityStatus],
    weights: Optional[dict[str, float]] = None,
) -> DataQualityResult:
    weights = dict(weights or DEFAULT_WEIGHTS)
    raw: dict[str, float] = {}
    unknown_tiers = sorted({t for t in source_tiers if t not in SOURCE_TIER_SCORE})

    if source_tiers:
        raw["source_tier"] = sum(SOURCE_TIER_SCORE.get(t, 0.0) for t in source_tiers) / len(source_tiers)
    if field_statuses:
        raw["field_status"] = sum(_STATUS_SCORE.get(s, 0.0) for s in field_statuses) / len(field_statuses)

    available = {k: v for k, v in raw.items() if k in weights}
    conflicting = sum(1 for s in field_statuses if s == DataQualityStatus.CONFLICTING)
    if not available:
        return DataQualityResult(value=None, conflicting_field_count=conflicting,
                                  unknown_source_tiers=unknown_tiers)
    total_w = sum(weights[k] for k in available)
    components = [
        DataQualityComponent(name=k, weight=weights[k] / total_w, raw_score=v,
                              contribution=(weights[k] / total_w) * v)
        for k, v in available.items()
    ]
    value = sum(c.contribution for c in components)
    return DataQualityResult(value=value, components=components, conflicting_field_count=conflicting,
                              unknown_source_tiers=unknown_tiers)
