"""
Shared provider-response validation (StockLab final engineering pass, Part D).

## Why this module exists

Writing `tests/test_eodhd_adapter.py` in Part A7 pinned four behaviours of the adapter layer that
are individually defensible and collectively incoherent:

1. `get_prices()` does **no type validation**: a JSON body with `"close": "not-a-number"` is
   passed straight through into `ProviderPriceBar.close` as a `str`. It travels all the way to a
   metric calculation before anything notices.
2. `get_prices()` fails the **whole batch** on one malformed bar: the list comprehension raises
   `KeyError` on a row missing `"date"`, so two well-formed bars are discarded with the third.
3. `list_universe()` does the **opposite** — silently drops rows without a `Code` key and returns
   the rest, with no record that anything was dropped.
4. Nothing anywhere checks whether a parsed value is *possible*: a negative price, a `high` below
   its `low`, a `close` outside the day's range, a negative share count.

So the same layer both over-reacts and under-reacts, and in neither case does the caller learn
what happened. Part A7 deliberately left this alone — fixing one adapter method while its
counterpart stayed inconsistent would have been worse than the gap — and recorded it as Part D's
job. This module is that job.

## The policy this module implements

**Reject the row, not the batch, and always say so.** A malformed bar is dropped with a recorded
`ValidationIssue` naming the field, the offending value and the reason. The caller gets the good
rows AND a `ValidationReport` it can log, count, or refuse to proceed on. That is strictly better
than both existing behaviours: no silent data loss, and no batch thrown away over one bad row.

**Coerce narrowly, never creatively.** A numeric string becomes a float, because providers really
do return `"12.5"`. Everything else — `None`, `""`, `"n/a"`, a bool, a list, `NaN`, `Infinity` —
becomes `None` with an issue recorded. `True` is explicitly not `1.0`: Python would happily do that
arithmetic and produce a silently wrong number.

**Sanity-check what is checkable, and nothing else.** A negative price is impossible; a `high`
below a `low` is impossible; a `close` outside `[low, high]` is impossible. Those are checked. What
a "reasonable" P/E or revenue growth is, is a judgment about companies, not about data integrity —
this module does not have opinions about those, deliberately.

Pure, dependency-free (stdlib only), so it is genuinely testable in this environment.
"""
from __future__ import annotations

import math
from dataclasses import dataclass, field
from typing import Any, Optional

SEVERITY_ERROR = "ERROR"      # the row cannot be used
SEVERITY_WARNING = "WARNING"  # the row is usable but something is off


@dataclass(frozen=True)
class ValidationIssue:
    field_name: str
    value: Any
    reason: str
    severity: str = SEVERITY_ERROR
    row_key: Optional[str] = None  # e.g. the bar's date, so an issue can be traced to a row

    def __str__(self) -> str:  # pragma: no cover - logging aid
        where = f" [{self.row_key}]" if self.row_key else ""
        return f"{self.severity}{where} {self.field_name}={self.value!r}: {self.reason}"


@dataclass
class ValidationReport:
    """Accumulates issues across a batch. Never raises — the caller decides what to do."""

    issues: list[ValidationIssue] = field(default_factory=list)
    rows_seen: int = 0
    rows_accepted: int = 0

    def add(self, issue: ValidationIssue) -> None:
        self.issues.append(issue)

    @property
    def rows_rejected(self) -> int:
        return self.rows_seen - self.rows_accepted

    @property
    def errors(self) -> list[ValidationIssue]:
        return [i for i in self.issues if i.severity == SEVERITY_ERROR]

    @property
    def warnings(self) -> list[ValidationIssue]:
        return [i for i in self.issues if i.severity == SEVERITY_WARNING]

    @property
    def ok(self) -> bool:
        return not self.errors

    @property
    def rejection_rate(self) -> float:
        """0.0-1.0. A caller should treat a high rate as a provider/mapping problem, not as data:
        losing 40% of a price history silently is how a backtest ends up quietly wrong."""
        return (self.rows_rejected / self.rows_seen) if self.rows_seen else 0.0

    def summary(self) -> str:
        return (
            f"{self.rows_accepted}/{self.rows_seen} rows accepted, "
            f"{len(self.errors)} errors, {len(self.warnings)} warnings"
        )


def coerce_float(
    value: Any, field_name: str, report: Optional[ValidationReport] = None,
    row_key: Optional[str] = None, required: bool = False,
) -> Optional[float]:
    """Return `value` as a float, or `None` with an issue recorded.

    Accepts int, float and numeric strings (providers really do return `"12.5"`). Rejects
    everything else, including `bool` — `True` would otherwise become `1.0` and produce a silently
    wrong number that no downstream check could catch.
    """
    if value is None or value == "":
        if required and report is not None:
            report.add(ValidationIssue(field_name, value, "required field is missing", row_key=row_key))
        return None
    if isinstance(value, bool):
        if report is not None:
            report.add(ValidationIssue(field_name, value, "boolean is not a number", row_key=row_key))
        return None
    if isinstance(value, (int, float)):
        number = float(value)
    elif isinstance(value, str):
        try:
            number = float(value.strip())
        except ValueError:
            if report is not None:
                report.add(ValidationIssue(field_name, value, "not parseable as a number", row_key=row_key))
            return None
    else:
        if report is not None:
            report.add(ValidationIssue(field_name, value, f"unsupported type {type(value).__name__}", row_key=row_key))
        return None

    if math.isnan(number) or math.isinf(number):
        if report is not None:
            report.add(ValidationIssue(field_name, value, "NaN/Infinity is not a usable value", row_key=row_key))
        return None
    return number


def validate_price_bar(
    raw: dict, report: ValidationReport, date_key: str = "date", row_key: Optional[str] = None,
) -> Optional[dict]:
    """Validate one OHLCV row. Returns a clean dict of floats, or `None` if the row is unusable.

    `raw` uses this codebase's canonical keys (`open`/`high`/`low`/`close`/`adjusted_close`/
    `volume`); an adapter maps its provider's names before calling.

    The row is REJECTED when: the date is missing, `close` is missing or unparseable, or `close`
    is not strictly positive. Everything else produces a WARNING and a `None` for that field — an
    unusable `volume` is not a reason to throw away a valid price.
    """
    report.rows_seen += 1
    key = row_key or str(raw.get(date_key))

    if not raw.get(date_key):
        report.add(ValidationIssue(date_key, raw.get(date_key), "row has no date", row_key=key))
        return None

    close = coerce_float(raw.get("close"), "close", report, key, required=True)
    if close is None:
        return None
    if close <= 0:
        report.add(ValidationIssue("close", close, "price must be strictly positive", row_key=key))
        return None

    out: dict = {date_key: raw[date_key], "close": close}
    for name in ("open", "high", "low", "adjusted_close"):
        value = coerce_float(raw.get(name), name, report, key)
        if value is not None and value <= 0:
            report.add(ValidationIssue(name, value, "price must be strictly positive",
                                       SEVERITY_WARNING, key))
            value = None
        out[name] = value

    volume = coerce_float(raw.get("volume"), "volume", report, key)
    if volume is not None and volume < 0:
        report.add(ValidationIssue("volume", volume, "volume cannot be negative", SEVERITY_WARNING, key))
        volume = None
    out["volume"] = volume

    high, low = out["high"], out["low"]
    if high is not None and low is not None and high < low:
        # Impossible, and it means the two are swapped or mismapped. Both are dropped rather than
        # guessing which one is wrong; `close` is still usable, which is what matters downstream.
        report.add(ValidationIssue("high/low", (high, low), "high is below low", SEVERITY_WARNING, key))
        out["high"] = out["low"] = high = low = None
    if high is not None and close > high:
        report.add(ValidationIssue("close", close, f"close is above high ({high})", SEVERITY_WARNING, key))
    if low is not None and close < low:
        report.add(ValidationIssue("close", close, f"close is below low ({low})", SEVERITY_WARNING, key))

    report.rows_accepted += 1
    return out


#: Line items that cannot be negative in any real filing. Kept deliberately short: every entry is
#: an arithmetic impossibility, not a judgment about what a healthy company looks like. Equity,
#: net income, operating income, FCF and retained earnings are all legitimately negative and are
#: NOT here.
NON_NEGATIVE_LINE_ITEMS = (
    "revenue", "total_assets", "current_assets", "current_liabilities", "cash_and_equivalents",
    "short_term_investments", "total_debt", "short_term_debt", "long_term_debt",
    "lease_liabilities", "goodwill", "intangible_assets", "inventory", "receivables",
    "shares_outstanding", "diluted_shares",
)


def validate_line_items(
    line_items: dict, report: ValidationReport, row_key: Optional[str] = None,
) -> dict:
    """Coerce and sanity-check one period's line items. Returns the cleaned dict.

    Unlike a price bar, a financial period is never rejected wholesale: one impossible field does
    not invalidate the other thirty, and a period with a bad `inventory` still supports every
    metric that does not use inventory. Bad fields are dropped individually with an issue recorded.
    """
    report.rows_seen += 1
    out: dict = {}
    for name, value in line_items.items():
        number = coerce_float(value, name, report, row_key)
        if number is None:
            continue
        if name in NON_NEGATIVE_LINE_ITEMS and number < 0:
            report.add(ValidationIssue(name, number, "cannot be negative in a real filing",
                                       SEVERITY_WARNING, row_key))
            continue
        out[name] = number

    # Cross-field checks: each is an identity that must hold, not a heuristic.
    revenue, gross_profit = out.get("revenue"), out.get("gross_profit")
    if revenue is not None and gross_profit is not None and gross_profit > revenue:
        report.add(ValidationIssue("gross_profit", gross_profit,
                                   f"exceeds revenue ({revenue})", SEVERITY_WARNING, row_key))
    total_assets, current_assets = out.get("total_assets"), out.get("current_assets")
    if total_assets is not None and current_assets is not None and current_assets > total_assets:
        report.add(ValidationIssue("current_assets", current_assets,
                                   f"exceeds total assets ({total_assets})", SEVERITY_WARNING, row_key))
    diluted, outstanding = out.get("diluted_shares"), out.get("shares_outstanding")
    if diluted is not None and outstanding is not None and diluted < outstanding:
        # Diluted counts every share that COULD exist, so it can never be below basic/outstanding.
        report.add(ValidationIssue("diluted_shares", diluted,
                                   f"below shares outstanding ({outstanding})", SEVERITY_WARNING, row_key))

    report.rows_accepted += 1
    return out
