From 6533c75e4effb6acefa22a27a47f1e7536a7920e Mon Sep 17 00:00:00 2001 From: Shivam Lalakiya <50960482+shivamlalakiya@users.noreply.github.com> Date: Fri, 25 Sep 2026 17:07:52 -0500 Subject: [PATCH] Fix score_upgrade_prospects crashes and its fixed top-N validation report A donor population with no historical upgrades at all (or none left below the leadership threshold) crashed with an opaque IndexError deep inside predict_proba, and a thin training history crashed the same way inside sklearn's internal cross-validated calibration. Both now raise a clear, documented ValueError instead. fiscal_year was also leaking in as a model feature even though it is not donor-specific and always sits outside the training range at scoring time; it stays as an output column but is no longer used to train the model or picked for top_reasons. The validation report's top-N lift used a fixed top_n=10, which read as a perfect 1.0 lift on a small validation fold and an uninformative one on a large real fiscal year. top_n now scales with the fold size (configurable), and the report adds a per-decile breakdown, roc_auc, average_precision, and a second naive baseline ("gave >= X last FY") alongside the existing "top N by FY total" one. --- CHANGELOG.md | 43 ++++-- philanthropy/models/_upgrade.py | 207 ++++++++++++++++++++++---- tests/test_cli.py | 4 +- tests/test_score_upgrade_prospects.py | 116 ++++++++++++++- 4 files changed, 330 insertions(+), 40 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ea9fe81..fb69a63 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -45,18 +45,21 @@ Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) `FiscalYearGroupedSplitter`. - `philanthropy.models.score_upgrade_prospects(gifts, *, activities=None, donors=None, threshold=1000.0, band=(100.0, 999.0), fiscal_year_start=7, - as_of=None, random_state=None)`: the fit-and-score entry point over - `build_upgrade_snapshots`. Trains a `MajorGiftClassifier` on every - fully-resolved historical fiscal year, validated with a walk-forward - `FiscalYearGroupedSplitter` fold, then scores today's band-qualifying - donors (cut at `as_of`, never at a future fiscal-year end) with a model - refit on all history. Returns a `(scores, report)` pair: `scores` has - `affinity_score`, `rank`, `decile`, a per-donor `top_reasons` heuristic - built from global permutation importance, and a `suggested_ask` left - `NaN` (no ask-amount label exists yet to train one honestly); `report` - carries training-row counts, a low-data warning under ~500 rows, the - `activities_to_features` id-match warning, and a top-N upgrade-rate lift - over the naive "highest FY total" rule. Wired into the CLI as + as_of=None, top_n=None, baseline_giving_threshold=None, random_state=None)`: + the fit-and-score entry point over `build_upgrade_snapshots`. Trains a + `MajorGiftClassifier` on every fully-resolved historical fiscal year + (excluding `fiscal_year` itself from the feature set), validated with a + walk-forward `FiscalYearGroupedSplitter` fold, then scores today's + band-qualifying donors (cut at `as_of`, never at a future fiscal-year end) + with a model refit on all history. Returns a `(scores, report)` pair: + `scores` has `affinity_score`, `rank`, `decile`, a per-donor `top_reasons` + heuristic built from global permutation importance, and a `suggested_ask` + left `NaN` (no ask-amount label exists yet to train one honestly); + `report` carries training-row counts, a low-data warning under ~500 rows, + the `activities_to_features` id-match warning, a per-decile breakdown of + the held-out fold (`deciles`), `roc_auc`/`average_precision`, and two + named-baseline upgrade rates and lifts ("gave >= X last FY" and "top N by + FY total", `top_n` defaulting to ~10% of the fold). Wired into the CLI as `philanthropy train --task upgrade`; `philanthropy features` gained repeated `--activity TYPE=PATH` and `--as-of` flags to fold engagement data into the feature table the same way. @@ -65,6 +68,22 @@ Format: [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) `MajorGiftClassifier`, validated with a fiscal-year walk-forward split and compared against a naive "gave $500+ last FY" rule on top-N upgrade rate. +### Fixed +- `score_upgrade_prospects` no longer crashes on a single-class historical + target (no donor ever upgraded, or every one did) or on a training set too + small for its internal 5-fold calibrated classifier; both now raise a + clear `ValueError` instead of an opaque one from deep inside + `CalibratedClassifierCV`. +- `score_upgrade_prospects` dropped `fiscal_year` from the model's own + feature columns (it isn't donor-specific, and the scored row's year always + sits outside the training range); it's still returned as an output column. +- `score_upgrade_prospects`'s validation report no longer uses a fixed + top-10 count, which read as a 1.0 lift on a large real validation year + whose true top-1% lift was 2.24x. `top_n` is now a parameter (default + ~10% of the held-out fold), and the report adds `deciles`, `roc_auc`, + `average_precision`, and two named baselines ("gave >= X last FY" and + "top N by FY total") with their own rates and lifts. + ## [0.8.0] - 2026-09-24 The first release with a Raiser's Edge on-ramp and `as_of` scoring cutoffs on the diff --git a/philanthropy/models/_upgrade.py b/philanthropy/models/_upgrade.py index c3d8977..699cc5c 100644 --- a/philanthropy/models/_upgrade.py +++ b/philanthropy/models/_upgrade.py @@ -29,6 +29,7 @@ import numpy as np import pandas as pd +from sklearn.metrics import average_precision_score, roc_auc_score from philanthropy.ingest import build_upgrade_snapshots from philanthropy.ingest._upgrade_snapshots import ( @@ -44,9 +45,14 @@ __all__ = ["score_upgrade_prospects"] -_TOP_N = 10 _TOP_REASONS = 3 _LOW_DATA_ROWS = 500 +# CalibratedClassifierCV (inside MajorGiftClassifier) defaults to a 5-fold +# cross-validation internally, which needs at least this many historical +# rows to split at all; fewer crashes deep inside sklearn ("Cannot have +# number of splits n_splits=5 greater than the number of samples") instead of +# failing with a message that points at the actual cause. +_MIN_TRAINING_ROWS = 5 def score_upgrade_prospects( @@ -58,6 +64,8 @@ def score_upgrade_prospects( band: Tuple[float, float] = (100.0, 999.0), fiscal_year_start: int = 7, as_of: Optional[Union[str, pd.Timestamp]] = None, + top_n: Optional[int] = None, + baseline_giving_threshold: Optional[float] = None, random_state: Optional[int] = None, ) -> Tuple[pd.DataFrame, Dict[str, Any]]: """Fit an upgrade model on history, then score today's band-qualifying donors. @@ -101,6 +109,18 @@ def score_upgrade_prospects( the historical training rows or the current scored row. Defaults to the latest gift date in ``gifts``, the same leakage-free default every other ``reference_date``-style parameter in this package uses. + top_n : int, optional + How many of the held-out validation fold's highest-scored donors + count as "the top" for ``model_upgrade_rate_top_n`` and the "top N + by FY total" baseline. Defaults to roughly 10% of the validation + fold (at least 1), rather than a fixed count: a fixed ``top_n`` reads + as a near-perfect rate on a small fold and an uninformative one on a + large real-world validation year (e.g. reporting 1.0 on a fixed + top-10 against a 64,178-row fold whose true top-1% lift was 2.24x). + Clipped to the fold size if larger. + baseline_giving_threshold : float, optional + The FY T giving level the "gave >= X last FY" naive baseline uses. + Defaults to ``threshold / 2``. random_state : int, optional Seed forwarded to the classifier fits and to the permutation importance call, for reproducible scores and reasons. @@ -110,7 +130,8 @@ def score_upgrade_prospects( scores : pandas.DataFrame One row per currently band-qualifying donor, indexed by ``donor_id``, sorted by ``affinity_score`` descending. Columns: ``fiscal_year`` (the - current, possibly still-open FY); ``affinity_score`` (0-100, see + current, possibly still-open FY, reported for reference only -- + it is not a model feature, see Notes); ``affinity_score`` (0-100, see :meth:`~philanthropy.models.MajorGiftClassifier.predict_affinity_score`); ``rank`` (1 = highest score); ``decile`` (1 = top 10% by rank, 10 = bottom); ``top_reasons`` (a tuple of up to 3 ``(feature_name, @@ -123,18 +144,45 @@ def score_upgrade_prospects( ``500`` training rows); ``activity_id_match_warnings`` (list of str, captured from ``activities_to_features``'s own low-match-rate warning, not recomputed here); ``validated`` (whether a walk-forward - held-out fold existed at all), and, when it did, - ``validation_fiscal_year``, ``n_validation_rows``, ``top_n``, - ``model_upgrade_rate_top_n``, ``baseline_upgrade_rate_top_n`` - (the naive "highest FY total" rule, same ``top_n``), - ``overall_upgrade_rate``, and ``lift_over_baseline`` (the two rates' - ratio; ``None`` if the baseline rate is 0). + held-out fold existed at all), and, when it did: + + - ``validation_fiscal_year``, ``n_validation_rows``: which FY the + held-out fold is, and how many rows it has. + - ``top_n``: how many of the fold's highest-scored donors count as + "the top", see the ``top_n`` parameter. + - ``model_upgrade_rate_top_n``: the actual upgrade rate among the + model's own top ``top_n`` donors by predicted score. + - ``baseline_topn_fy_total_upgrade_rate`` / ``lift_topn_fy_total``: + the naive "top N by FY total" rule's upgrade rate over the same + ``top_n``, and the model rate's ratio to it (``None`` if the + baseline rate is 0). + - ``baseline_giving_threshold``, ``baseline_gave_threshold_upgrade_rate`` + / ``lift_over_gave_threshold``: the naive "gave >= X last FY" rule + (``X`` is ``baseline_giving_threshold``), its upgrade rate among + donors who cleared it, and the model rate's ratio to it (``None`` + if nobody in the fold cleared it, or the resulting rate is 0). + - ``overall_upgrade_rate``: the fold's overall positive rate. + - ``deciles``: a list of 10 dicts, one per predicted-score decile (1 + = highest-scored 10% of the fold, 10 = lowest), each with + ``decile``, ``n`` (rows in that decile), ``actual_rate`` (observed + upgrade rate) and ``mean_predicted`` (mean predicted probability). + A finer-grained, fixed-``top_n``-independent view of the same + ranking; see the ``top_n`` parameter for why a single fixed count + is misleading on its own. + - ``roc_auc``, ``average_precision``: from :mod:`sklearn.metrics` + on the held-out fold; ``None`` if the fold has only one class. Raises ------ ValueError - If ``band[0] > band[1]``, or if there is not one historical - ``(donor, fiscal year)`` row to train on as of ``as_of``. + If ``band[0] > band[1]``; if there is not one historical + ``(donor, fiscal year)`` row to train on as of ``as_of``; if there + are fewer than 5 historical rows (too few for + :class:`~philanthropy.models.MajorGiftClassifier`'s internal 5-fold + calibration to split at all); or if a walk-forward training fold, or + the full historical training set when a current donor still needs + scoring, has only one target class (nothing to learn or nothing to + score: no donor in it ever upgraded, or every one did). KeyError If ``gifts`` is missing ``donor_id``, ``gift_date`` or ``gift_amount``. @@ -147,7 +195,13 @@ def score_upgrade_prospects( ``donors`` column (e.g. a wealth-rating letter grade) is still joined and returned for reference but dropped before fitting, the simplest rule that needs no per-column encoding policy for a feature nobody asked this - function to build. A current-row feature column absent from history (or + function to build. ``fiscal_year`` is excluded from the model features + even though it is numeric: it is not a donor-specific signal, and the + current row's fiscal year always sits outside the range the model was + trained on (every historical row is a strictly earlier FY), so it can + only ever generalise as noise or, worse, an ordering artefact. It is + still returned as an output column and left out of ``top_reasons`` for + the same reason. A current-row feature column absent from history (or vice versa), e.g. an activity type that only shows up in one window, is reindexed to 0.0 rather than dropped, matching ``activities_to_features``'s own "no rows of that type -> 0" rule. @@ -273,7 +327,8 @@ def score_upgrade_prospects( feature_cols = [ c for c in historical_snap.columns - if c != "target" and pd.api.types.is_numeric_dtype(historical_snap[c]) + if c not in ("target", "fiscal_year") + and pd.api.types.is_numeric_dtype(historical_snap[c]) ] X = historical_snap[feature_cols].to_numpy(dtype="float64") y = historical_snap["target"].to_numpy() @@ -290,6 +345,16 @@ def score_upgrade_prospects( ) warnings.warn(low_data_message, UserWarning, stacklevel=2) + # Too few historical rows crashes deep inside CalibratedClassifierCV + # (every downstream fit needs at least this many rows to split), with a + # confusing error if left unchecked. + _check_min_rows(len(historical_snap), as_of_ts) + + gave_threshold = ( + float(baseline_giving_threshold) + if baseline_giving_threshold is not None else threshold / 2.0 + ) + report: Dict[str, Any] = { "n_training_rows": int(len(historical_snap)), "n_training_fiscal_years": n_unique_fys, @@ -303,9 +368,15 @@ def score_upgrade_prospects( "n_validation_rows": 0, "top_n": None, "model_upgrade_rate_top_n": None, - "baseline_upgrade_rate_top_n": None, + "baseline_topn_fy_total_upgrade_rate": None, + "lift_topn_fy_total": None, + "baseline_giving_threshold": gave_threshold, + "baseline_gave_threshold_upgrade_rate": None, + "lift_over_gave_threshold": None, "overall_upgrade_rate": None, - "lift_over_baseline": None, + "deciles": None, + "roc_auc": None, + "average_precision": None, } if n_unique_fys >= 2: @@ -316,27 +387,53 @@ def score_upgrade_prospects( eval_model = MajorGiftClassifier(random_state=random_state).fit( X[train_idx], y[train_idx] ) + # A single-class training fold fits fine but its predict_proba comes + # back with one column, not two: guard the call site rather than let + # the [:, 1] below raise a confusing IndexError. + _check_two_classes(eval_model.classes_, as_of_ts, "The walk-forward training fold") y_test = y[test_idx] proba_test = eval_model.predict_proba(X[test_idx])[:, 1] fy_total_test = historical_snap["fy_total"].to_numpy()[test_idx] - top_n = min(_TOP_N, len(test_idx)) - model_top_n = np.argsort(-proba_test)[:top_n] - baseline_top_n = np.argsort(-fy_total_test)[:top_n] + n_val = len(test_idx) + resolved_top_n = top_n if top_n is not None else max(1, round(0.1 * n_val)) + top_n_eff = min(resolved_top_n, n_val) + model_top_n = np.argsort(-proba_test)[:top_n_eff] + baseline_top_n = np.argsort(-fy_total_test)[:top_n_eff] baseline_rate = float(y_test[baseline_top_n].mean()) + model_rate = float(y_test[model_top_n].mean()) + + gave_threshold_mask = fy_total_test >= gave_threshold + n_gave_threshold = int(gave_threshold_mask.sum()) + gave_threshold_rate = ( + float(y_test[gave_threshold_mask].mean()) if n_gave_threshold > 0 else None + ) + + multiclass_fold = np.unique(y_test).size >= 2 + roc_auc = float(roc_auc_score(y_test, proba_test)) if multiclass_fold else None + average_precision = ( + float(average_precision_score(y_test, proba_test)) if multiclass_fold else None + ) report.update({ "validated": True, "validation_fiscal_year": int(fys[test_idx][0]), - "n_validation_rows": int(len(test_idx)), - "top_n": int(top_n), - "model_upgrade_rate_top_n": float(y_test[model_top_n].mean()), - "baseline_upgrade_rate_top_n": baseline_rate, - "overall_upgrade_rate": float(y_test.mean()), - "lift_over_baseline": ( - float(y_test[model_top_n].mean() / baseline_rate) - if baseline_rate > 0 else None + "n_validation_rows": int(n_val), + "top_n": int(top_n_eff), + "model_upgrade_rate_top_n": model_rate, + "baseline_topn_fy_total_upgrade_rate": baseline_rate, + "lift_topn_fy_total": ( + float(model_rate / baseline_rate) if baseline_rate > 0 else None + ), + "baseline_gave_threshold_upgrade_rate": gave_threshold_rate, + "lift_over_gave_threshold": ( + float(model_rate / gave_threshold_rate) + if gave_threshold_rate else None ), + "overall_upgrade_rate": float(y_test.mean()), + "deciles": _decile_report(proba_test, y_test), + "roc_auc": roc_auc, + "average_precision": average_precision, }) importance_df = donor_feature_importance( eval_model, X[test_idx], y[test_idx], feature_names=feature_cols, @@ -362,6 +459,10 @@ def score_upgrade_prospects( if current_snap is None or current_snap.empty: scores = _empty_scores_frame() else: + # Same single-class guard as the walk-forward fold above: only + # needed here (not unconditionally), since with no current row to + # score there is nothing that would call predict_proba at all. + _check_two_classes(model.classes_, as_of_ts, "The historical training set") X_current = current_snap.reindex(columns=feature_cols, fill_value=0.0) affinity = model.predict_affinity_score(X_current.to_numpy(dtype="float64")) reasons = _top_reasons(X_current, importance_df) @@ -389,6 +490,62 @@ def score_upgrade_prospects( # --------------------------------------------------------------------------- # # Internals # --------------------------------------------------------------------------- # +def _check_min_rows(n_rows: int, as_of_ts: pd.Timestamp) -> None: + """Raise a clear ``ValueError`` instead of letting a too-small training + set crash inside ``MajorGiftClassifier``'s internal + ``CalibratedClassifierCV(cv=5)``. See ``score_upgrade_prospects``'s + ``Raises`` section.""" + if n_rows < _MIN_TRAINING_ROWS: + raise ValueError( + f"Only {n_rows} historical donor-year row(s) as of " + f"{as_of_ts.date()} (need at least {_MIN_TRAINING_ROWS}): " + "MajorGiftClassifier calibrates its probabilities with an " + "internal 5-fold cross-validation, which needs at least that " + "many rows to split at all. Gather more historical data, or " + "widen `band`/lower `threshold` to enlarge the candidate " + "population." + ) + + +def _check_two_classes(classes: np.ndarray, as_of_ts: pd.Timestamp, scope: str) -> None: + """Raise a clear ``ValueError`` instead of letting a single-class fit + crash a caller's later ``predict_proba(...)[:, 1]`` (a single-class fit + succeeds, but its ``predict_proba`` only has one column). See + ``score_upgrade_prospects``'s ``Raises`` section.""" + if classes.size < 2: + outcome = "an upgrade (target=1)" if classes[0] == 1 else "not an upgrade (target=0)" + raise ValueError( + f"{scope} has only one class as of {as_of_ts.date()}: every row " + f"is {outcome}. There is nothing to learn from a single-class " + "history; this can happen with too few fiscal years of data, " + "or a `threshold`/`band` that no historical donor ever crossed " + "(or that every one did)." + ) + + +def _decile_report(proba: np.ndarray, y_true: np.ndarray) -> list: + """Ten dicts, one per predicted-score decile of a held-out fold (1 = + highest-scored 10%, 10 = lowest), each with ``decile``, ``n``, + ``actual_rate`` and ``mean_predicted``. See ``score_upgrade_prospects``'s + ``Returns`` section.""" + n = len(proba) + ranks = np.empty(n, dtype="int64") + ranks[np.argsort(-proba, kind="stable")] = np.arange(1, n + 1) + deciles = np.ceil(ranks / n * 10).clip(max=10).astype("int64") + + rows = [] + for d in range(1, 11): + mask = deciles == d + n_d = int(mask.sum()) + rows.append({ + "decile": d, + "n": n_d, + "actual_rate": float(y_true[mask].mean()) if n_d > 0 else None, + "mean_predicted": float(proba[mask].mean()) if n_d > 0 else None, + }) + return rows + + def _top_reasons( X: pd.DataFrame, importance_df: pd.DataFrame, top_k: int = _TOP_REASONS ) -> list: diff --git a/tests/test_cli.py b/tests/test_cli.py index 1892d8a..d0088d1 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -477,12 +477,14 @@ def test_python_and_cli_upgrade_paths_produce_identical_scores(tmp_path): ]) cli_scores = pd.read_csv(scored_path).set_index("donor_id").sort_index() direct_sorted = direct_scores.sort_index() + cli_scores.index = cli_scores.index.astype(str) + direct_sorted.index = direct_sorted.index.astype(str) assert list(cli_scores.index) == list(direct_sorted.index) pd.testing.assert_series_equal( cli_scores["affinity_score"].astype(float), direct_sorted["affinity_score"].astype(float), - check_names=False, check_exact=False, check_index_type=False, + check_names=False, check_exact=False, ) assert list(cli_scores["rank"]) == list(direct_sorted["rank"]) assert list(cli_scores["decile"]) == list(direct_sorted["decile"]) diff --git a/tests/test_score_upgrade_prospects.py b/tests/test_score_upgrade_prospects.py index a8cfe63..b99e8e4 100644 --- a/tests/test_score_upgrade_prospects.py +++ b/tests/test_score_upgrade_prospects.py @@ -88,8 +88,11 @@ def test_report_keys_present(): "n_training_rows", "n_training_fiscal_years", "low_data_warning", "low_data_message", "activity_id_match_warnings", "current_fiscal_year", "n_scored", "validated", "validation_fiscal_year", "n_validation_rows", - "top_n", "model_upgrade_rate_top_n", "baseline_upgrade_rate_top_n", - "overall_upgrade_rate", "lift_over_baseline", + "top_n", "model_upgrade_rate_top_n", + "baseline_topn_fy_total_upgrade_rate", "lift_topn_fy_total", + "baseline_giving_threshold", "baseline_gave_threshold_upgrade_rate", + "lift_over_gave_threshold", "overall_upgrade_rate", "deciles", + "roc_auc", "average_precision", ): assert key in report @@ -269,3 +272,112 @@ def test_as_of_defaults_to_latest_gift_date(): explicit, _ = score_upgrade_prospects(gifts, as_of="2024-08-01", random_state=0) default, _ = score_upgrade_prospects(gifts, random_state=0) pd.testing.assert_frame_equal(explicit.sort_index(), default.sort_index()) + + +# --------------------------------------------------------------------------- # +# F5: no upgraders / all upgraders in history (previously an IndexError deep +# inside predict_proba) +# --------------------------------------------------------------------------- # +def test_single_class_history_raises_clear_value_error(): + # Every donor stays flat below the band ceiling for every year: target is + # 0 for every historical row, so there is nothing to learn. + years = ["2020-08-01", "2021-08-01", "2022-08-01", "2023-08-01", "2024-08-01"] + rows = [ + {"donor_id": f"flat_low_{i}", "gift_date": year, "gift_amount": 400} + for i in range(20) + for year in years + ] + gifts = pd.DataFrame(rows) + with pytest.raises(ValueError, match="only one class"): + score_upgrade_prospects(gifts, random_state=0) + + +# --------------------------------------------------------------------------- # +# F5: tiny training sets (previously an sklearn ValueError from deep inside +# CalibratedClassifierCV instead of a clear, documented one) +# --------------------------------------------------------------------------- # +def test_tiny_training_set_raises_clear_value_error(): + years = ["2020-08-01", "2021-08-01", "2022-08-01"] + rows = [ + {"donor_id": "a", "gift_date": years[0], "gift_amount": 400}, + {"donor_id": "a", "gift_date": years[1], "gift_amount": 1200}, + {"donor_id": "a", "gift_date": years[2], "gift_amount": 400}, + {"donor_id": "b", "gift_date": years[0], "gift_amount": 400}, + {"donor_id": "b", "gift_date": years[1], "gift_amount": 400}, + {"donor_id": "b", "gift_date": years[2], "gift_amount": 400}, + ] + gifts = pd.DataFrame(rows) + with pytest.raises(ValueError, match="5-fold cross-validation"): + score_upgrade_prospects(gifts, random_state=0) + + +# --------------------------------------------------------------------------- # +# F5: fiscal_year is not donor-specific and sits outside the training range +# at scoring time, so it must not be a model feature (still an output column) +# --------------------------------------------------------------------------- # +def test_fiscal_year_excluded_from_top_reasons(): + scores, _ = score_upgrade_prospects(_archetype_gifts(), random_state=0) + for reasons in scores["top_reasons"]: + assert all(feature != "fiscal_year" for feature, _ in reasons) + + +# --------------------------------------------------------------------------- # +# F5: top_n defaults to ~10% of the validation fold, not a fixed count, and +# the report carries deciles, roc_auc, average_precision and two baselines +# --------------------------------------------------------------------------- # +def test_default_top_n_is_roughly_ten_percent_of_validation_fold(): + _, report = score_upgrade_prospects(_archetype_gifts(), random_state=0) + n_val = report["n_validation_rows"] + assert report["top_n"] == max(1, round(0.1 * n_val)) + + +def test_top_n_parameter_overrides_default(): + _, report = score_upgrade_prospects(_archetype_gifts(), top_n=3, random_state=0) + assert report["top_n"] == 3 + + +def test_larger_validation_fold_does_not_use_a_fixed_top_n(): + # With more rows per archetype, ~10% of the fold should exceed the old + # hardcoded top_n of 10, proving it now scales with fold size. + _, report = score_upgrade_prospects(_archetype_gifts(n_per_group=200), random_state=0) + assert report["top_n"] > 10 + + +def test_deciles_report_shape_and_coverage(): + _, report = score_upgrade_prospects(_archetype_gifts(), random_state=0) + deciles = report["deciles"] + assert len(deciles) == 10 + assert [d["decile"] for d in deciles] == list(range(1, 11)) + assert sum(d["n"] for d in deciles) == report["n_validation_rows"] + for d in deciles: + if d["n"] > 0: + assert 0.0 <= d["actual_rate"] <= 1.0 + assert 0.0 <= d["mean_predicted"] <= 1.0 + + +def test_roc_auc_and_average_precision_bounded(): + _, report = score_upgrade_prospects(_archetype_gifts(), random_state=0) + assert report["roc_auc"] is None or 0.0 <= report["roc_auc"] <= 1.0 + assert report["average_precision"] is None or 0.0 <= report["average_precision"] <= 1.0 + + +def test_baseline_giving_threshold_defaults_to_half_threshold(): + _, report = score_upgrade_prospects(_archetype_gifts(), threshold=1000.0, random_state=0) + assert report["baseline_giving_threshold"] == 500.0 + + +def test_baseline_giving_threshold_parameter_overrides_default(): + _, report = score_upgrade_prospects( + _archetype_gifts(), baseline_giving_threshold=300.0, random_state=0 + ) + assert report["baseline_giving_threshold"] == 300.0 + + +def test_two_named_baselines_and_lifts_reported(): + _, report = score_upgrade_prospects(_archetype_gifts(), random_state=0) + for rate_key in ( + "baseline_topn_fy_total_upgrade_rate", "baseline_gave_threshold_upgrade_rate", + ): + assert report[rate_key] is None or 0.0 <= report[rate_key] <= 1.0 + for lift_key in ("lift_topn_fy_total", "lift_over_gave_threshold"): + assert report[lift_key] is None or report[lift_key] >= 0.0