diff --git a/CHANGELOG.md b/CHANGELOG.md index 9e18ece..714c647 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,6 +23,8 @@ All notable changes to this project will be documented in this file. ### Fixed +- `stack_proportional()` now uses direction-aware exponential utilities with an explicit score-unit `temperature`, preserving shared weight for near-tied losses as well as rewards; validates input and handles signed/extreme scores (#140). + - Strict mypy failed on `run_adaptive`'s optuna `directions` argument with current optuna stubs; directions are now typed as `Literal["minimize", "maximize"]`. - `reduce_factors()` no longer lets a NaN-valued observable (e.g. a Type-I rate that's legitimately undefined outside null regimes) silently corrupt every other observable's importance for the same factor. It aggregated via `np.maximum`, which propagates NaN (`np.maximum(4.05, nan) == nan`); one conditionally-undefined observable could erase a real, significant importance value found via a *different* observable, dropping the factor with no warning or error. Now uses NaN-safe `np.fmax`, and warns when a factor's importance is NaN across *every* observable (dropped for lack of data, not confirmed unimportance) (#119). diff --git a/src/trade_study/stacking.py b/src/trade_study/stacking.py index c8269dd..af1c0aa 100644 --- a/src/trade_study/stacking.py +++ b/src/trade_study/stacking.py @@ -82,42 +82,47 @@ def stack_proportional( score_matrix: NDArray[np.floating[Any]], *, maximize: bool = False, + temperature: float = 1.0, ) -> NDArray[np.floating[Any]]: - """Weight models in direct proportion to their mean score. - - Unlike :func:`stack_scores` (a linear program that puts *all* weight - on the single best-performing model whenever there's any nonzero gap, - even a noise-scale one), this scales smoothly with relative - performance -- useful when "how much better" should matter, not just - "which one is best". Two near-tied models get near-equal weights - here; under :func:`stack_scores` the same tiny gap can flip the - result between an even split and 100% on one model, since a linear - objective over a simplex has no reason to split weight once any - model is even infinitesimally ahead. + """Weight models proportionally to exponential score utilities. + + Mean scores are transformed using a direction-aware softmax. Near-tied + models receive comparable weights in either direction, including when + scores are zero or negative. These are heuristic ensemble weights, + rather than Bayesian stacking weights or calibrated probabilities. Args: - score_matrix: Array of shape (n_models, n_test_points) where each - entry is the score of model i on test point j. - maximize: If True, higher scores are better. If False, lower - scores are better; scores are inverted internally - (``max - score``) rather than divided, so a score of exactly - 0 doesn't produce an unbounded weight. + score_matrix: Finite array of shape (n_models, n_test_points). + maximize: Whether higher scores are better. Default is minimize. + temperature: Positive, finite scale in score units. Larger values + spread weight more evenly; smaller values favor the best model. + The default is 1.0. Scale this with the scores when changing + their units to preserve the same weights. Returns: - Array of weights, shape (n_models,), summing to 1, each >= 0. - Falls back to a uniform split if every model scores identically - (nothing to distinguish them by). - """ - mean_scores = np.mean(score_matrix, axis=1) - if not maximize: - mean_scores = np.max(mean_scores) - mean_scores - mean_scores = np.clip(mean_scores, a_min=0.0, a_max=None) + Non-negative float64 weights summing to 1, shape (n_models,). + Exactly equal mean scores receive uniform weights. - total = mean_scores.sum() - n_models = score_matrix.shape[0] - if total <= 0.0: - return np.full(n_models, 1.0 / n_models, dtype=np.float64) - return np.asarray(mean_scores / total, dtype=np.float64) + Raises: + ValueError: If scores are empty, non-finite, or not two-dimensional, + or temperature is not positive and finite. + """ + scores = np.asarray(score_matrix, dtype=np.float64) + if scores.ndim != 2 or 0 in scores.shape or not np.all(np.isfinite(scores)): + msg = "score_matrix must be a non-empty, finite two-dimensional array" + raise ValueError(msg) + if not np.isfinite(temperature) or temperature <= 0: + msg = "temperature must be positive and finite" + raise ValueError(msg) + + # Divide before summing to avoid overflowing the mean of large scores. + means = np.sum(scores / scores.shape[1], axis=1) + utility = means if maximize else -means + # Overflow in a very large unfavorable gap means a zero softmax weight. + with np.errstate(over="ignore"): + logits = (utility - np.max(utility)) / temperature + weights = np.exp(logits) + return np.asarray(weights / weights.sum(), dtype=np.float64) def ensemble_predict( diff --git a/tests/test_stacking.py b/tests/test_stacking.py index bef8d44..c165f60 100644 --- a/tests/test_stacking.py +++ b/tests/test_stacking.py @@ -266,3 +266,46 @@ def test_ensemble_weights_normalised() -> None: result = ensemble_predict([p1, p2], weights) expected = (p1 + p2) / 2.0 np.testing.assert_allclose(result, expected) + + +@pytest.mark.parametrize("maximize", [False, True]) +def test_proportional_two_model_near_tie(*, maximize: bool) -> None: + weights = stack_proportional(np.array([[1.0], [1.001]]), maximize=maximize) + np.testing.assert_allclose(weights, [0.5, 0.5], atol=0.001) + assert (weights[1] > weights[0]) == maximize + + +def test_proportional_signed_scores_and_direction_reversal() -> None: + scores = np.array([[-3.0], [-2.0], [0.0]]) + low = stack_proportional(scores) + high = stack_proportional(-scores, maximize=True) + np.testing.assert_allclose(low, high) + assert low[0] > low[1] > low[2] > 0 + np.testing.assert_allclose(low, stack_proportional(scores + 100)) + + +def test_proportional_temperature_tracks_score_units() -> None: + scores = np.array([[0.0], [1.0]]) + np.testing.assert_allclose( + stack_proportional(scores), + stack_proportional(scores * 100, temperature=100), + ) + assert stack_proportional(scores, temperature=0.1)[0] > 0.99 + assert stack_proportional(scores, temperature=100)[0] < 0.51 + + +def test_proportional_extreme_scores_are_finite() -> None: + scores = np.array([[1e308, 1e308], [-1e308, -1e308]]) + np.testing.assert_array_equal(stack_proportional(scores), [0.0, 1.0]) + + +@pytest.mark.parametrize("temperature", [0.0, -1.0, np.nan, np.inf]) +def test_proportional_rejects_invalid_temperature(temperature: float) -> None: + with pytest.raises(ValueError, match="temperature"): + stack_proportional(np.array([[1.0]]), temperature=temperature) + + +@pytest.mark.parametrize("scores", [[], [[]], [1.0], [[np.nan]], [[np.inf]]]) +def test_proportional_rejects_invalid_scores(scores: list) -> None: + with pytest.raises(ValueError, match="score_matrix"): + stack_proportional(np.asarray(scores))