"""
Peer-group construction for percentile scoring and industry-median reference multiples.

## Why this module exists

Two separate places in this codebase documented "a cross-security aggregate query this pass
doesn't build" as a known gap, and both were more serious than the wording suggested:

1. **`app/workers/recompute.py::recompute_security(db, security_id, peer_metric_values=None)`**
   — the only call site in the application is `recompute_security(db, security_id)`, with no third
   argument. So `peer_metric_values` was `{}` on every real run. `compute_generic_subscore()` then
   handed an empty list to `percentile_rank()`, which returns a neutral `50.0` for an empty peer
   set — meaning **every metric scored exactly 50, every one of the five pillar subscores came out
   50, and every security's Overall Score was 50.** The scoring engine's arithmetic was correct
   and unit-tested throughout; it was simply never given anything to compare against.

2. **`app/engines/valuation/multiples.py`** — `ReferenceSource.INDUSTRY_MEDIAN` and `PEER_MEDIAN`
   are modelled in the enum but nothing computed them, so multiples fair value fell back to each
   security's own historical median (or `None`), never to an industry reference.

Both need the same primitive: group every security's metric values by industry / cap bucket /
sector, then read the group off. This module is that primitive, kept pure and dependency-free
(stdlib only, no sqlalchemy) so it can be genuinely tested here; `app/workers/peer_groups.py`
supplies it with rows from the database and nothing else.

## Deliberate design decisions, stated rather than implied

- **A security is excluded from its own peer set.** "The percentage of peers this company beats"
  is a statement about other companies. Including itself drags every percentile toward the middle
  by `1/n` and makes a one-company industry score 50 by construction rather than by evidence.
- **A metric with an empty peer set is reported as having no peers**, never silently given a
  neutral 50 — see `PeerLookup.is_empty`. The consumer decides what to do; this module refuses to
  invent a comparison that does not exist.
- **`None` metric values are dropped, not coerced.** A missing ROIC is not a ROIC of zero, and a
  peer group padded with zeros would push every real company's percentile up.
- **Grouping is by exact key.** No fuzzy industry matching, no "similar sector" heuristics. If two
  industries should be one peer group, that belongs in the industry taxonomy, not hidden here.
"""
from __future__ import annotations

from dataclasses import dataclass, field
from statistics import median
from typing import Iterable, Optional

#: The three widening tiers, narrowest first. `peer_values_with_fallback()` in percentile.py
#: consumes them in this order.
TIER_INDUSTRY_BUCKET = "industry_bucket"
TIER_INDUSTRY = "industry"
TIER_SECTOR = "sector"
TIERS = (TIER_INDUSTRY_BUCKET, TIER_INDUSTRY, TIER_SECTOR)


@dataclass(frozen=True)
class PeerMetricRow:
    """One security's value for one metric, plus the classification needed to group it.

    Deliberately flat and primitive-only: the worker turns database rows into these, and this
    module never sees an ORM object.
    """

    security_id: str
    metric_key: str
    value: float
    industry_id: Optional[str]
    sector_id: Optional[str]
    market_cap_bucket: Optional[str]


@dataclass
class PeerLookup:
    """`{metric_key: {tier: [values]}}` plus the counts needed to report honestly."""

    by_metric: dict[str, dict[str, list[float]]] = field(default_factory=dict)

    def as_dict(self) -> dict[str, dict[str, list[float]]]:
        """The exact shape `compute_generic_subscore()` expects."""
        return self.by_metric

    def is_empty(self, metric_key: str) -> bool:
        tiers = self.by_metric.get(metric_key)
        if not tiers:
            return True
        return not any(tiers.get(t) for t in TIERS)

    def widest_available_size(self, metric_key: str) -> int:
        tiers = self.by_metric.get(metric_key, {})
        return max((len(tiers.get(t, [])) for t in TIERS), default=0)


def build_peer_metric_values(
    rows: Iterable[PeerMetricRow],
    target_security_id: str,
    target_industry_id: Optional[str],
    target_sector_id: Optional[str],
    target_market_cap_bucket: Optional[str],
) -> PeerLookup:
    """Partition `rows` into the three peer tiers for one target security.

    `rows` is the metric universe — typically every security's current metric values. The target
    itself is excluded from all three tiers.
    """
    lookup = PeerLookup()
    for row in rows:
        if row.security_id == target_security_id:
            continue
        if row.value is None:
            continue
        tiers = lookup.by_metric.setdefault(row.metric_key, {t: [] for t in TIERS})

        in_sector = target_sector_id is not None and row.sector_id == target_sector_id
        in_industry = target_industry_id is not None and row.industry_id == target_industry_id
        in_bucket = (
            in_industry
            and target_market_cap_bucket is not None
            and row.market_cap_bucket == target_market_cap_bucket
        )

        # A row belongs to every tier it qualifies for: the industry+bucket group is a subset of
        # the industry group, which is (normally) a subset of the sector group. Building them as
        # nested supersets is what makes peer_values_with_fallback()'s widening meaningful --
        # if the tiers were disjoint, widening would swap one small group for a different small
        # group rather than growing the sample.
        if in_bucket:
            tiers[TIER_INDUSTRY_BUCKET].append(row.value)
        if in_industry:
            tiers[TIER_INDUSTRY].append(row.value)
        if in_sector:
            tiers[TIER_SECTOR].append(row.value)
    return lookup


def industry_median_multiples(
    rows: Iterable[PeerMetricRow],
    metric_keys: Iterable[str],
    min_group_size: int = 5,
) -> dict[str, dict[str, float]]:
    """`{industry_id: {metric_key: median}}` for the requested valuation multiples.

    This is what `ReferenceSource.INDUSTRY_MEDIAN` needs and never had.

    `min_group_size` is a real refusal, not a soft preference: a "median P/E" computed from two
    companies is not an industry reference, and publishing it as one would be exactly the kind of
    number that looks authoritative and is not. Industries below the threshold are simply absent
    from the result, so the caller falls back to the self-historical reference (or to `None`) by
    the path it already has.

    Non-positive multiples are dropped. A negative P/E is not a cheap company, it is a company
    with negative earnings, and including it would drag an industry median toward or below zero
    and produce a negative "fair value".
    """
    wanted = set(metric_keys)
    buckets: dict[str, dict[str, list[float]]] = {}
    for row in rows:
        if row.metric_key not in wanted or row.industry_id is None:
            continue
        if row.value is None or row.value <= 0:
            continue
        buckets.setdefault(row.industry_id, {}).setdefault(row.metric_key, []).append(row.value)

    out: dict[str, dict[str, float]] = {}
    for industry_id, by_metric in buckets.items():
        medians = {
            key: median(values)
            for key, values in by_metric.items()
            if len(values) >= min_group_size
        }
        if medians:
            out[industry_id] = medians
    return out
