Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,10 @@ All notable changes to this project will be documented in this file.

### Fixed

- Confidence constraints now reject non-finite means and unknown/invalid standard errors instead of treating missing uncertainty as zero. Adaptive tells require constraint scores and validate them before updating storage; rejected tells can be corrected (#141).

- `stack_proportional()` now uses direction-aware exponential utilities with an explicit score-unit `temperature`, preserving shared weight for near-tied losses as well as rewards; validates input and handles signed/extreme scores (#140).

- Strict mypy failed on `run_adaptive`'s optuna `directions` argument with current optuna stubs; directions are now typed as `Literal["minimize", "maximize"]`.
- `reduce_factors()` no longer lets a NaN-valued observable (e.g. a Type-I rate that's legitimately undefined outside null regimes) silently corrupt every other observable's importance for the same factor. It aggregated via `np.maximum`, which propagates NaN (`np.maximum(4.05, nan) == nan`); one conditionally-undefined observable could erase a real, significant importance value found via a *different* observable, dropping the factor with no warning or error. Now uses NaN-safe `np.fmax`, and warns when a factor's importance is NaN across *every* observable (dropped for lack of data, not confirmed unimportance) (#119).

Expand Down
12 changes: 12 additions & 0 deletions src/trade_study/protocols.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,7 @@
import operator as _operator
from dataclasses import dataclass, field
from enum import Enum
from math import isfinite
from typing import TYPE_CHECKING, Any, Protocol, runtime_checkable

if TYPE_CHECKING:
Expand Down Expand Up @@ -107,11 +108,22 @@ def bound(self, mean: float, standard_error: float) -> float:
Returns:
``mean`` without ``confidence``; otherwise the one-sided bound
on the unfavourable side of ``threshold``.

Raises:
ValueError: If a confidence bound has a non-finite mean or a
non-finite or negative standard error.
"""
if self.confidence is None:
return mean
from statistics import NormalDist

if not isfinite(mean) or not isfinite(standard_error) or standard_error < 0:
msg = (
f"Constraint {self.name!r}: confidence bound needs a finite mean "
"and a finite non-negative standard error; report at least "
"two finite replicates"
)
raise ValueError(msg)
z = NormalDist().inv_cdf(self.confidence)
return (
mean + z * standard_error
Expand Down
23 changes: 20 additions & 3 deletions src/trade_study/session.py
Original file line number Diff line number Diff line change
Expand Up @@ -132,9 +132,15 @@ def _constraint_values(self, trial: optuna.trial.FrozenTrial) -> list[float]:
errors = trial.user_attrs.get("standard_error", {})
values = []
for c in self.constraints:
mean = float(scores.get(c.observable, np.inf))
mean = float(scores.get(c.observable, np.nan))
error = float(errors.get(c.observable, np.nan))
bound = c.bound(mean, error if np.isfinite(error) else 0.0)
unknown = not np.isfinite(mean) or (
c.confidence is not None and (not np.isfinite(error) or error < 0)
)
if unknown:
values.append(float("inf"))
continue
bound = c.bound(mean, error)
values.append(_constraint_value(c, bound))
return values

Expand Down Expand Up @@ -180,7 +186,9 @@ def tell(

Raises:
ValueError: If the trial id is unknown, already told, or an
objective is missing.
objective or constraint score is missing, or a constraint
has no finite mean/required standard error. Rejected tells
leave the trial pending and may be corrected.
"""
import optuna as _optuna

Expand All @@ -199,6 +207,15 @@ def tell(
msg = f"Missing objective scores: {missing}"
raise ValueError(msg)
summary = {name: _summarize(values) for name, values in scores.items()}
for constraint in self.constraints:
if constraint.observable not in summary:
msg = f"Missing constraint score: {constraint.observable!r}"
raise ValueError(msg)
mean, standard_error, _count = summary[constraint.observable]
if not np.isfinite(mean):
msg = f"Constraint {constraint.name!r} needs a finite reported mean"
raise ValueError(msg)
constraint.bound(mean, standard_error)
self._storage.set_trial_user_attr(
internal, "scores", {k: v[0] for k, v in summary.items()}
)
Expand Down
65 changes: 35 additions & 30 deletions src/trade_study/stacking.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,42 +82,47 @@ def stack_proportional(
score_matrix: NDArray[np.floating[Any]],
*,
maximize: bool = False,
temperature: float = 1.0,
) -> NDArray[np.floating[Any]]:
"""Weight models in direct proportion to their mean score.

Unlike :func:`stack_scores` (a linear program that puts *all* weight
on the single best-performing model whenever there's any nonzero gap,
even a noise-scale one), this scales smoothly with relative
performance -- useful when "how much better" should matter, not just
"which one is best". Two near-tied models get near-equal weights
here; under :func:`stack_scores` the same tiny gap can flip the
result between an even split and 100% on one model, since a linear
objective over a simplex has no reason to split weight once any
model is even infinitesimally ahead.
"""Weight models proportionally to exponential score utilities.

Mean scores are transformed using a direction-aware softmax. Near-tied
models receive comparable weights in either direction, including when
scores are zero or negative. These are heuristic ensemble weights,
rather than Bayesian stacking weights or calibrated probabilities.

Args:
score_matrix: Array of shape (n_models, n_test_points) where each
entry is the score of model i on test point j.
maximize: If True, higher scores are better. If False, lower
scores are better; scores are inverted internally
(``max - score``) rather than divided, so a score of exactly
0 doesn't produce an unbounded weight.
score_matrix: Finite array of shape (n_models, n_test_points).
maximize: Whether higher scores are better. Default is minimize.
temperature: Positive, finite scale in score units. Larger values
spread weight more evenly; smaller values favor the best model.
The default is 1.0. Scale this with the scores when changing
their units to preserve the same weights.

Returns:
Array of weights, shape (n_models,), summing to 1, each >= 0.
Falls back to a uniform split if every model scores identically
(nothing to distinguish them by).
"""
mean_scores = np.mean(score_matrix, axis=1)
if not maximize:
mean_scores = np.max(mean_scores) - mean_scores
mean_scores = np.clip(mean_scores, a_min=0.0, a_max=None)
Non-negative float64 weights summing to 1, shape (n_models,).
Exactly equal mean scores receive uniform weights.

total = mean_scores.sum()
n_models = score_matrix.shape[0]
if total <= 0.0:
return np.full(n_models, 1.0 / n_models, dtype=np.float64)
return np.asarray(mean_scores / total, dtype=np.float64)
Raises:
ValueError: If scores are empty, non-finite, or not two-dimensional,
or temperature is not positive and finite.
"""
scores = np.asarray(score_matrix, dtype=np.float64)
if scores.ndim != 2 or 0 in scores.shape or not np.all(np.isfinite(scores)):
msg = "score_matrix must be a non-empty, finite two-dimensional array"
raise ValueError(msg)
if not np.isfinite(temperature) or temperature <= 0:
msg = "temperature must be positive and finite"
raise ValueError(msg)

# Divide before summing to avoid overflowing the mean of large scores.
means = np.sum(scores / scores.shape[1], axis=1)
utility = means if maximize else -means
# Overflow in a very large unfavorable gap means a zero softmax weight.
with np.errstate(over="ignore"):
logits = (utility - np.max(utility)) / temperature
weights = np.exp(logits)
return np.asarray(weights / weights.sum(), dtype=np.float64)


def ensemble_predict(
Expand Down
43 changes: 43 additions & 0 deletions tests/test_stacking.py
Original file line number Diff line number Diff line change
Expand Up @@ -266,3 +266,46 @@ def test_ensemble_weights_normalised() -> None:
result = ensemble_predict([p1, p2], weights)
expected = (p1 + p2) / 2.0
np.testing.assert_allclose(result, expected)


@pytest.mark.parametrize("maximize", [False, True])
def test_proportional_two_model_near_tie(*, maximize: bool) -> None:
weights = stack_proportional(np.array([[1.0], [1.001]]), maximize=maximize)
np.testing.assert_allclose(weights, [0.5, 0.5], atol=0.001)
assert (weights[1] > weights[0]) == maximize


def test_proportional_signed_scores_and_direction_reversal() -> None:
scores = np.array([[-3.0], [-2.0], [0.0]])
low = stack_proportional(scores)
high = stack_proportional(-scores, maximize=True)
np.testing.assert_allclose(low, high)
assert low[0] > low[1] > low[2] > 0
np.testing.assert_allclose(low, stack_proportional(scores + 100))


def test_proportional_temperature_tracks_score_units() -> None:
scores = np.array([[0.0], [1.0]])
np.testing.assert_allclose(
stack_proportional(scores),
stack_proportional(scores * 100, temperature=100),
)
assert stack_proportional(scores, temperature=0.1)[0] > 0.99
assert stack_proportional(scores, temperature=100)[0] < 0.51


def test_proportional_extreme_scores_are_finite() -> None:
scores = np.array([[1e308, 1e308], [-1e308, -1e308]])
np.testing.assert_array_equal(stack_proportional(scores), [0.0, 1.0])


@pytest.mark.parametrize("temperature", [0.0, -1.0, np.nan, np.inf])
def test_proportional_rejects_invalid_temperature(temperature: float) -> None:
with pytest.raises(ValueError, match="temperature"):
stack_proportional(np.array([[1.0]]), temperature=temperature)


@pytest.mark.parametrize("scores", [[], [[]], [1.0], [[np.nan]], [[np.inf]]])
def test_proportional_rejects_invalid_scores(scores: list) -> None:
with pytest.raises(ValueError, match="score_matrix"):
stack_proportional(np.asarray(scores))
55 changes: 55 additions & 0 deletions tests/test_uncertainty_aware.py
Original file line number Diff line number Diff line change
Expand Up @@ -144,3 +144,58 @@ def test_session_constraints_use_the_confidence_bound() -> None:
error = meta["standard_error"]["cost"]
expected = 0.475 + 1.6448536269514722 * error - 0.5
assert meta["constraints"] == pytest.approx([expected])


@pytest.mark.parametrize("op", ["<=", ">="])
@pytest.mark.parametrize("values", [0.4, [0.4, np.nan], [np.nan], [np.inf]])
def test_confident_session_rejects_unknown_uncertainty_and_allows_correction(
op: str, values: float | list[float]
) -> None:
session = AdaptiveSession(
FACTORS,
[Observable("cost", Direction.MINIMIZE)],
constraints=[Constraint("cap", "cost", op, 0.5, confidence=0.95)],
)
((trial_id, _config),) = session.ask(1)
with pytest.raises(ValueError, match="finite"):
session.tell(trial_id, {"cost": values})
assert session.results().configs == []
session.tell(trial_id, {"cost": [0.4, 0.45]})
assert len(session.results().configs) == 1


@pytest.mark.parametrize("op", ["<=", ">="])
def test_session_requires_non_objective_constraint_score(op: str) -> None:
session = AdaptiveSession(
FACTORS,
[Observable("loss", Direction.MINIMIZE)],
constraints=[Constraint("cap", "cost", op, 0.5)],
)
((trial_id, _config),) = session.ask(1)
with pytest.raises(ValueError, match="Missing constraint score"):
session.tell(trial_id, {"loss": 0.1})
session.tell(trial_id, {"loss": 0.1, "cost": 0.4})
assert session.results().metadata[0]["scores"]["cost"] == pytest.approx(0.4)


@pytest.mark.parametrize("error", [np.nan, np.inf, -0.1])
def test_table_rejects_invalid_confidence_standard_error(error: float) -> None:
table = ResultsTable(
configs=[{"x": 0}],
scores=np.array([[0.4]]),
observable_names=["cost"],
metadata=[{"standard_error": {"cost": error}}],
)
with pytest.raises(ValueError, match="standard error"):
table.feasible([Constraint("cap", "cost", "<=", 0.5, confidence=0.95)])


def test_confident_constraint_accepts_zero_variance_with_replicates() -> None:
session = AdaptiveSession(
FACTORS,
[Observable("cost", Direction.MINIMIZE)],
constraints=[Constraint("cap", "cost", "<=", 0.5, confidence=0.95)],
)
((trial_id, _config),) = session.ask(1)
session.tell(trial_id, {"cost": [0.4, 0.4]})
assert session.results().metadata[0]["constraints"] == pytest.approx([-0.1])
Loading