Skip to content
Closed
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
47 changes: 22 additions & 25 deletions tests/parity/test_boost.py
Original file line number Diff line number Diff line change
Expand Up @@ -55,56 +55,53 @@ def test_nns_boost_ivs_test_none_matches_r() -> None:
depth=None,
features_only=False,
)
# random_seed is pinned for determinism. The deterministic feature-set path
# still draws from the CV-split RNG for iterations above n_rows/4, so an
# unseeded call left this assertion theoretically seed-sensitive even though
# the boosted result is empirically seed-invariant here (see
# test_nns_boost_ivs_test_none_is_seed_invariant). Pinning the seed removes
# any residual flakiness without altering the matched values.
# Use the default seed 123 to match R NNS.boost and its delegated NNS.stack
# folds.
actual = nns_boost(
variable,
y,
learner_trials=10,
cv_size=0.25,
feature_importance=False,
random_seed=123,
)

_assert_boost_matches(actual, expected)


@pytest.mark.parity
def test_nns_boost_ivs_test_none_is_seed_invariant() -> None:
# Regression guard for the previously reported cache-parity failure: the
# depth=None / feature_importance=False boosted result must be identical
# across seeds (and an unseeded call), so the parity comparison cannot be
# destabilised by RNG draws on the CV-split path.
@pytest.mark.parametrize("seed", [0, 1, 4, 42, 1234])
def test_nns_boost_ivs_test_none_is_seed_reproducible(seed: int) -> None:
"""Reusing a seed reproduces the delegated NNS.stack estimate."""
x = np.linspace(-2.0, 2.0, 24)
variable = np.column_stack((x, np.sin(x), np.cos(x)))
y = x + np.sin(x) + 0.25 * np.cos(x)

baseline = np.asarray(
first = np.asarray(
nns_boost(
variable,
y,
learner_trials=10,
cv_size=0.25,
feature_importance=False,
random_seed=seed,
)["results"],
dtype=np.float64,
)
for seed in (None, 0, 1, 4, 42, 1234):
result = np.asarray(
nns_boost(
variable,
y,
learner_trials=10,
cv_size=0.25,
feature_importance=False,
random_seed=seed,
)["results"],
dtype=np.float64,
)
np.testing.assert_array_equal(result, baseline)

second = np.asarray(
nns_boost(
variable,
y,
learner_trials=10,
cv_size=0.25,
feature_importance=False,
random_seed=seed,
)["results"],
dtype=np.float64,
)

np.testing.assert_array_equal(first, second)


@pytest.mark.parity
Expand Down
Loading