"""Source hierarchy conflict resolution (spec §8/§9, docs/DATA_SOURCES.md §4)."""
from __future__ import annotations

from dataclasses import dataclass
from typing import Optional

# Lower index = higher priority.
SOURCE_HIERARCHY = ["OFFICIAL_FILING", "PRIMARY", "SECONDARY", "CALCULATED"]


@dataclass(frozen=True)
class ConflictResolution:
    selected_value: Optional[float]
    selected_source_tier: Optional[str]
    selection_reason: str
    is_conflicting: bool


def resolve_conflict(candidates: list[tuple[str, Optional[float]]], tolerance_pct: float = 0.005) -> ConflictResolution:
    """
    `candidates` = [(source_tier, value), ...] for the same (security, period, field).
    Selects the highest-priority tier's value; flags CONFLICTING if a lower-priority source
    disagrees by more than `tolerance_pct` (default 0.5%) — small rounding/currency-conversion
    differences are not treated as a real conflict.
    """
    present = [(tier, val) for tier, val in candidates if val is not None]
    if not present:
        return ConflictResolution(None, None, "No source provided a value", is_conflicting=False)

    present.sort(key=lambda c: SOURCE_HIERARCHY.index(c[0]) if c[0] in SOURCE_HIERARCHY else 99)
    selected_tier, selected_value = present[0]

    conflicting = False
    for tier, val in present[1:]:
        if selected_value == 0:
            if val != 0:
                conflicting = True
        elif abs(val - selected_value) / abs(selected_value) > tolerance_pct:
            conflicting = True

    reason = f"Highest-priority available source: {selected_tier}"
    if conflicting:
        reason += f"; disagreement > {tolerance_pct:.1%} with a lower-priority source — see data_quality row"
    return ConflictResolution(selected_value, selected_tier, reason, is_conflicting=conflicting)


@dataclass(frozen=True)
class FieldConflict:
    field: str
    value_a: Optional[float]
    value_b: Optional[float]
    selected_value: Optional[float]
    selection_reason: str


def resolve_line_items(
    primary: dict, secondary: Optional[dict], fields: list[str],
    primary_tier: str, secondary_tier: Optional[str] = None,
) -> tuple[dict, list[FieldConflict]]:
    """
    StockLab overhaul, final engineering pass, Part A6 -- the statement-level counterpart to
    `resolve_conflict()` above, which only ever resolves ONE field at a time. This applies it
    across a whole LineItems-shaped dict (`app/adapters/schemas.py`) so a real ingestion call site
    can resolve an entire income statement / balance sheet / cash flow / shares record between two
    providers in one call, without each call site re-implementing the per-field loop.

    `primary`/`secondary` are dicts of {field_name: value} for the SAME (security, period) from
    two different provider adapters -- the same shape `ingest.py`'s existing single-source loop
    already reads from `period.line_items`. `fields` is the list of field names to consider (the
    caller passes the exact tuple it already iterates for that statement type, e.g. ingest.py's
    existing `("revenue", "cogs", ...)` tuples -- kept as an explicit argument here rather than
    hardcoded, so this function has no per-statement-type knowledge of its own).

    Returns `(resolved, conflicts)`:
    - `resolved` contains exactly the fields at least one source provided a value for (mirrors the
      existing `if field in li: setattr(...)` "only touch fields that were actually provided"
      contract every single-source call site already relies on -- a field neither source reported
      is simply absent from `resolved`, never defaulted to 0/None-overwrite).
    - `conflicts` lists only the fields where `resolve_conflict()` flagged real (> tolerance_pct)
      disagreement -- the caller persists these as `DataQuality` rows; the vast majority of fields
      on any real statement are expected to end up here with zero conflicts.

    When `secondary` is None (no second source was fetched for this call -- the default,
    `Settings.PROVIDER_SECONDARY = "NONE"`), this degrades to exactly "use primary's value for
    every field primary provided", byte-identical in output to what single-source ingestion
    already does. There is deliberately no behavior change for that, the common, case -- dual-
    source resolution only activates when a caller actually has two dicts to pass in.
    """
    resolved: dict = {}
    conflicts: list[FieldConflict] = []
    for field in fields:
        val_a = primary.get(field)
        val_b = secondary.get(field) if secondary is not None else None
        if val_a is None and val_b is None:
            continue
        candidates = [(primary_tier, val_a)]
        if secondary is not None and secondary_tier is not None:
            candidates.append((secondary_tier, val_b))
        result = resolve_conflict(candidates)
        if result.selected_value is not None:
            resolved[field] = result.selected_value

        # Secondary-only provenance must be observable. A NULL in the
        # primary provider followed by a value from the secondary provider
        # is not a numeric conflict, but it is still a cross-provider merge.
        if (
            secondary is not None
            and secondary_tier is not None
            and val_a is None
            and val_b is not None
        ):
            conflicts.append(
                FieldConflict(
                    field,
                    val_a,
                    val_b,
                    result.selected_value,
                    "Primary field missing; value supplied by secondary source",
                )
            )
        elif result.is_conflicting:
            conflicts.append(
                FieldConflict(
                    field,
                    val_a,
                    val_b,
                    result.selected_value,
                    result.selection_reason,
                )
            )
    return resolved, conflicts


def merge_statement_line_items(
    income: dict,
    balance: Optional[dict] = None,
    cash_flow: Optional[dict] = None,
) -> dict:
    """Merge one provider's three statement responses for the SAME reporting period into the one
    flat line-items dict `ingest_security()` writes to the four statement tables.

    AUDIT FIX (StockLab final engineering pass, Part A9 — a critical bug found while wiring TTM).
    Before this existed, `ingest_security()` called ONLY `adapter.get_income_statements()` and then
    wrote `BalanceSheet`, `CashFlow` and `Shares` rows out of that single response's line items.
    `adapter.get_balance_sheets()` and `adapter.get_cash_flows()` were never called anywhere in the
    application (verified by grep across `app/` and `scripts/`: the only remaining matches were
    comments). Against `DemoDataAdapter` this was invisible, because its three statement methods
    all return the SAME fully-merged dict containing every field. Against `FMPAdapter` — the real
    primary provider — `get_income_statements()` maps only `_INCOME_MAP`, so every balance-sheet,
    cash-flow and share-count column would have been written NULL for real data. See
    docs/AUDIT_INGEST_STATEMENTS_A9.md.

    Merge rule: first non-None value wins, in income -> balance -> cash-flow order. The income
    statement is therefore authoritative for fields more than one endpoint reports (FMP's
    cash-flow response also carries `netIncome`, for example), and a `None` from a later statement
    can never erase a real value from an earlier one.
    """
    merged: dict = {}
    for source in (income, balance, cash_flow):
        if not source:
            continue
        for key, value in source.items():
            if value is None:
                continue
            if merged.get(key) is None:
                merged[key] = value
    return merged
