"""
Pillar wiring and membership (final master pass, §8/§9).

Two jobs:

1. **Pin the metric→pillar mapping** so `docs/SCORING.md` and the code cannot drift apart again.
   The audit that produced this file found four substantive mismatches between the two, including
   a documented methodology (a 50/50 self-historical/industry blend in Valuation) that does not
   exist in the code at all.

2. **Prove the Competitive Advantage pillar is actually wired.** `recompute_security()` called
   `compute_competitive_advantage_score({})` — an empty proxy dict — so that pillar was
   INSUFFICIENT_DATA for every security, always, and 15% of the Overall Score weight was
   permanently absent while `compute_overall_score()` quietly renormalised the rest. This is the
   third defect of the same family, after Confidence/Data Quality (Part A1) and peer groups
   (Part B2): an engine that was built, tested, and never called.

Pure, dependency-free — genuinely executed. Dual-mode: pytest, or
`PYTHONPATH=. python3 tests/test_pillar_wiring.py`.
"""
from __future__ import annotations

from app.adapters.demo import DemoDataAdapter
from app.engines.metrics.core import invested_capital
from app.engines.metrics.formula_utils import safe_div
from app.engines.normalize import build_snapshot
from app.engines.scoring.persistence import build_competitive_advantage_proxies
from app.engines.scoring.subscores import (
    COMPETITIVE_ADVANTAGE_METRICS,
    FINANCIAL_HEALTH_METRICS,
    GROWTH_METRICS,
    MIN_PROXIES_FOR_COMPETITIVE_ADVANTAGE,
    QUALITY_METRICS,
    VALUATION_METRICS,
    compute_competitive_advantage_score,
)


# --- 1. membership, pinned ---


def test_pillar_membership_matches_the_documented_mapping():
    """`docs/SCORING.md` §2 is the authority; this asserts the code agrees with it, key by key.
    If a metric is added to a pillar, this test fails until the document is updated too."""
    assert set(QUALITY_METRICS) == {
        "roic", "roic_minus_wacc", "gross_margin", "operating_margin", "net_margin",
        "fcf_margin", "fcf_growth_cagr_3y", "revenue_growth_cagr_3y", "eps_growth_cagr_3y",
        "share_count_growth_cagr_3y",
    }
    assert set(FINANCIAL_HEALTH_METRICS) == {
        "debt_to_ebitda", "net_debt_to_ebitda", "debt_to_equity", "interest_coverage",
        "current_ratio", "fcf",
    }
    assert set(GROWTH_METRICS) == {
        "revenue_growth_cagr_3y", "revenue_cagr_5y", "eps_growth_cagr_3y",
        "fcf_growth_cagr_3y", "revenue_growth_trend", "fcf_growth_trend",
    }
    assert set(VALUATION_METRICS) == {
        "pe", "forward_pe", "peg", "ev_to_ebitda", "p_fcf", "ev_to_fcf", "fcf_yield",
    }
    # Competitive Advantage is deliberately NOT metric-based: it is built from persistence
    # proxies, so its metric dict is empty by design and must stay that way.
    assert COMPETITIVE_ADVANTAGE_METRICS == {}


def test_lower_is_better_flags_are_correct_for_every_metric():
    """A flipped flag inverts a percentile silently: a company with the worst leverage in its
    industry would score 100 for Financial Health."""
    for key in ("debt_to_ebitda", "net_debt_to_ebitda", "debt_to_equity"):
        assert FINANCIAL_HEALTH_METRICS[key] is True, f"{key}: lower leverage must score higher"
    for key in ("interest_coverage", "current_ratio", "fcf"):
        assert FINANCIAL_HEALTH_METRICS[key] is False
    for key in ("pe", "forward_pe", "peg", "ev_to_ebitda", "p_fcf", "ev_to_fcf"):
        assert VALUATION_METRICS[key] is True, f"{key}: a lower multiple is cheaper"
    assert VALUATION_METRICS["fcf_yield"] is False, "a HIGHER FCF yield is cheaper"
    assert QUALITY_METRICS["share_count_growth_cagr_3y"] is True, "shrinking share count is good"
    for key in ("roic", "gross_margin", "operating_margin", "net_margin", "fcf_margin"):
        assert QUALITY_METRICS[key] is False


def test_the_three_growth_metrics_counted_in_two_pillars_are_known_and_pinned():
    """FINDING (§8 "провери double counting"), documented rather than silently changed.

    `revenue_growth_cagr_3y`, `eps_growth_cagr_3y` and `fcf_growth_cagr_3y` are members of BOTH
    Quality (weight 0.25) and Growth (weight 0.20). A fast-growing company therefore has its
    growth rewarded across 45% of the total weight, while Quality is nominally about returns and
    margins. That is a real methodology question, not a bug — it is what `docs/SCORING.md` has
    always described — so this pass PINS it rather than changing every score in the platform on
    its own judgment. See docs/AUDIT_PILLARS.md for the argument on both sides."""
    overlap = set(QUALITY_METRICS) & set(GROWTH_METRICS)
    assert overlap == {"revenue_growth_cagr_3y", "eps_growth_cagr_3y", "fcf_growth_cagr_3y"}


def test_no_metric_appears_in_three_or_more_pillars():
    pillars = [QUALITY_METRICS, FINANCIAL_HEALTH_METRICS, GROWTH_METRICS, VALUATION_METRICS]
    counts: dict[str, int] = {}
    for p in pillars:
        for key in p:
            counts[key] = counts.get(key, 0) + 1
    assert max(counts.values()) <= 2, {k: v for k, v in counts.items() if v > 2}


def test_no_metric_is_in_both_valuation_and_financial_health():
    """A multiple and a leverage ratio measure different things; sharing one would double-count
    the balance sheet into the cheapness score."""
    assert not (set(VALUATION_METRICS) & set(FINANCIAL_HEALTH_METRICS))


# --- 2. Competitive Advantage is really wired now ---


def _proxies_for(snapshot, tax_rate: float = 0.25) -> dict:
    """Mirrors exactly what `app/workers/recompute.py` now does. Kept in the test so the wiring
    itself is exercised, not just the engine it calls."""
    c = snapshot.current
    periods = [c] + list(snapshot.history)
    return build_competitive_advantage_proxies(
        roic_history=[
            safe_div(p.ebit * (1 - tax_rate), invested_capital(p))
            if (p.ebit is not None and invested_capital(p)) else None
            for p in periods
        ],
        gross_margin_history=[safe_div(p.gross_profit, p.revenue) for p in periods],
        operating_margin_history=[safe_div(p.operating_income, p.revenue) for p in periods],
        revenue_history=[p.revenue for p in periods],
        fcf_history=[
            (p.operating_cash_flow - p.capital_expenditure)
            if (p.operating_cash_flow is not None and p.capital_expenditure is not None) else None
            for p in periods
        ],
        invested_capital_to_revenue=safe_div(invested_capital(c), c.revenue),
        customer_concentration_pct=None,
    )


def _demo_snapshot(ticker: str):
    a = DemoDataAdapter()
    periods = a.get_income_statements(ticker)
    profile = a.get_company_profile(ticker)
    return build_snapshot("S", "IND", profile.sector or "DEFAULT", periods, current_price=100.0)


def test_an_empty_proxy_dict_is_always_insufficient_data():
    """The exact call recompute_security() used to make."""
    r = compute_competitive_advantage_score({})
    assert r.value is None
    assert r.status == "INSUFFICIENT_DATA"


def test_real_multi_year_history_produces_a_real_competitive_advantage_score():
    a = DemoDataAdapter()
    scored = 0
    for ticker in sorted(a.list_universe())[:6]:
        snapshot = _demo_snapshot(ticker)
        assert len(snapshot.history) >= 4, f"{ticker}: too little history to test persistence"
        result = compute_competitive_advantage_score(_proxies_for(snapshot))
        assert result.value is not None, f"{ticker}: still INSUFFICIENT_DATA after wiring"
        assert 0.0 <= result.value <= 100.0
        scored += 1
    assert scored == 6


def test_competitive_advantage_scores_differ_across_companies():
    """A pillar that returns the same number for every company is not measuring anything."""
    a = DemoDataAdapter()
    values = {t: compute_competitive_advantage_score(_proxies_for(_demo_snapshot(t))).value
              for t in sorted(a.list_universe())[:6]}
    assert len(set(round(v, 3) for v in values.values())) > 1, values


def test_a_short_history_still_yields_insufficient_data_rather_than_a_guess():
    """`stability_score` needs at least 4 points. A company with 2 years of history must NOT get a
    persistence score invented from 2 observations — the honest answer is INSUFFICIENT_DATA."""
    snapshot = _demo_snapshot(sorted(DemoDataAdapter().list_universe())[0])
    truncated = type(snapshot)(
        security_id=snapshot.security_id, industry_id=snapshot.industry_id,
        sector_id=snapshot.sector_id, market_cap_bucket=snapshot.market_cap_bucket,
        calculation_date=snapshot.calculation_date, current=snapshot.current,
        history=snapshot.history[:1],
    )
    proxies = _proxies_for(truncated)
    live = [k for k, v in proxies.items() if v is not None]
    assert len(live) < MIN_PROXIES_FOR_COMPETITIVE_ADVANTAGE
    assert compute_competitive_advantage_score(proxies).status == "INSUFFICIENT_DATA"


def test_customer_concentration_is_absent_not_defaulted():
    """No adapter ingests it. A missing proxy must be dropped, never given a neutral value."""
    proxies = _proxies_for(_demo_snapshot(sorted(DemoDataAdapter().list_universe())[0]))
    assert "low_customer_concentration" not in proxies


ALL_TESTS = [v for k, v in sorted(globals().items()) if k.startswith("test_")]

if __name__ == "__main__":
    passed = failed = 0
    for t in ALL_TESTS:
        try:
            t()
            print(f"PASS  {t.__name__}")
            passed += 1
        except Exception as exc:  # noqa: BLE001
            print(f"FAIL  {t.__name__}: {exc}")
            failed += 1
    print(f"\n{passed}/{passed + failed} passed")
    raise SystemExit(1 if failed else 0)
