Files
trueskill-tt/tests/degenerate_inputs.rs
T
logaritmiskandClaude Opus 5 8dff7513f7 fix!: seal ConstantDrift's field so gamma can be validated
`gamma` enters only as `gamma * gamma`, so the sign was squared away:
measured against the old public-field form, `ConstantDrift(-0.0833)`
produced results bit identical to `ConstantDrift(0.0833)`. The sign was
neither rejected nor honoured — it vanished.

It could not be checked while the field was a public tuple position,
because there was nothing to intercept. Validating inside
`variance_for_elapsed` would have been worse: it runs in the sweep, so a
construction-time mistake would panic mid-inference, and `Gaussian::from_ms`
is a worked example of why that is the wrong place — rejecting NaN there
turned the NonFiniteResult reporting path into a crash.

So `ConstantDrift::new` is the only way in and it checks, with `gamma()`
to read the value back. 129 call sites rewritten across src, tests,
benches, examples and the README. The dated plan and spec documents under
docs/superpowers are left alone: they record what was built at the time,
and rewriting them would falsify that.

tests/constructor_validation.rs is the more valuable half. This defect
class was closed three times in one session and reopened twice, because
each fix validated the layer it had just touched and inferred the rest —
`HistoryBuilder`, then `Game`'s own entry points, then the constructors
beneath both. A per-site fix cannot notice the site nobody thought of, so
that file enumerates every public entry point taking a magnitude and
asserts each refuses negative and non-finite values.

It found an eleventh defect on its first run: `HistoryBuilder::score_sigma`
accepted infinity, because `inf > 0.0` is true and the assert only tested
positivity. Fixed, and its own `should_panic` message updated to match.

`Gaussian::from_ms` is deliberately exempt from the non-finite half, for
the reason above: a broken fit produces a NaN sigma legitimately and
`converge` must be allowed to report it.

The convergence-level drift-variance check stays and is now tested through
a custom `Drift` implementation, since `ConstantDrift` can no longer reach
it. That check is the only thing standing between a third-party `Drift`
and a NaN fit.

BREAKING CHANGE: `ConstantDrift`'s field is private. Replace
`ConstantDrift(x)` with `ConstantDrift::new(x)`, and `drift().0` with
`drift().gamma()`. `HistoryBuilder::score_sigma` now rejects infinity.

Closes #65

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011hcFjNDmHXZF8URGLku5zZ
2026-09-09 19:11:57 +02:00

433 lines
12 KiB
Rust

//! Degenerate, boundary, and error-path coverage.
//!
//! These run in both debug and release: the defects they pin were all
//! guarded only by `debug_assert!`, so a debug-only suite never saw them.
mod common;
use common::assert_finite;
use trueskill_tt::{
ConstantDrift, ConvergenceOptions, Game, GameOptions, Gaussian, History, InferenceError,
NullObserver, Outcome, Rating,
};
type R = Rating<i64, ConstantDrift>;
fn rating() -> R {
R::new(
Gaussian::from_ms(25.0, 25.0 / 3.0),
25.0 / 6.0,
ConstantDrift::new(25.0 / 300.0),
)
}
#[test]
fn record_draw_without_draw_probability_is_rejected() {
let mut h = History::default();
let err = h.record_draw(&"a", &"b", 1).unwrap_err();
assert!(matches!(
err,
InferenceError::TieWithoutDrawProbability { .. }
));
}
#[test]
fn builder_draw_without_draw_probability_is_rejected() {
let mut h = History::default();
let err = h
.event(1)
.team(["a"])
.team(["b"])
.draw()
.commit()
.unwrap_err();
assert!(matches!(
err,
InferenceError::TieWithoutDrawProbability { .. }
));
}
#[test]
fn draw_with_positive_draw_probability_is_finite() {
let mut h = History::builder().p_draw(0.25).build();
h.record_draw(&"a", &"b", 1).unwrap();
let report = h.converge().unwrap();
assert_finite(h.current_skill("a").unwrap(), "drawn competitor skill");
assert_finite(h.current_skill("b").unwrap(), "drawn competitor skill");
assert!(report.log_evidence.is_finite());
assert!(report.converged);
}
#[test]
fn game_ranked_rejects_tie_without_draw_probability() {
let a = [rating()];
let b = [rating()];
let teams: Vec<&[R]> = vec![&a, &b];
let err = Game::ranked(&teams, Outcome::draw(2), &GameOptions::default()).unwrap_err();
assert!(matches!(
err,
InferenceError::TieWithoutDrawProbability { .. }
));
}
/// `Outcome::winner(w, n)` ties every loser, so any n >= 3 free-for-all hits
/// the tie path even though the caller never asked for a draw.
#[test]
fn winner_of_three_or_more_requires_draw_probability() {
let a = [rating()];
let b = [rating()];
let c = [rating()];
let teams: Vec<&[R]> = vec![&a, &b, &c];
let err = Game::ranked(&teams, Outcome::winner(0, 3), &GameOptions::default()).unwrap_err();
assert!(matches!(
err,
InferenceError::TieWithoutDrawProbability { .. }
));
let opts = GameOptions {
p_draw: 0.1,
..GameOptions::default()
};
let game = Game::ranked(&teams, Outcome::winner(0, 3), &opts).unwrap();
for team in game.posteriors() {
for skill in team {
assert_finite(skill, "3-team winner posterior");
}
}
}
#[test]
fn full_ranking_without_ties_needs_no_draw_probability() {
let a = [rating()];
let b = [rating()];
let c = [rating()];
let teams: Vec<&[R]> = vec![&a, &b, &c];
let game = Game::ranked(&teams, Outcome::ranking([0, 1, 2]), &GameOptions::default()).unwrap();
for team in game.posteriors() {
for skill in team {
assert_finite(skill, "strict ranking posterior");
}
}
}
#[test]
fn empty_history_converges_trivially() {
let mut h = History::default();
let report = h.converge().unwrap();
assert_eq!(report.iterations, 0);
assert!(report.converged);
}
/// Issue #27's exact reproduction: a non-default key type reaching `converge`
/// with no events at all. The underflow it reported trapped in debug and
/// indexed out of bounds in release, so this must run in both profiles.
#[test]
fn converge_on_an_empty_history_with_owned_keys() {
let mut history: History<i64, ConstantDrift, NullObserver, String> =
History::builder_with_key().score_sigma(5.0).build();
let report = history.converge().unwrap();
assert_eq!(report.iterations, 0);
assert!(report.converged);
}
/// A weights/team length mismatch used to be a `debug_assert!`, so release
/// builds ingested the event with the weights silently unapplied. This file's
/// CI job runs in release too, which is the point of pinning it here.
#[test]
fn event_builder_rejects_a_weights_length_mismatch() {
let mut h = History::default();
let err = h
.event(1)
.team(["a"])
.weights([1.0, 2.0])
.team(["b"])
.winner(0)
.commit()
.unwrap_err();
assert!(
matches!(
err,
InferenceError::MismatchedShape {
kind: "weights",
expected: 1,
got: 2,
}
),
"expected a weights MismatchedShape, got {err:?}"
);
}
/// The mismatch must not be applied even partially — a half-weighted team
/// reaching the history would be worse than the error.
#[test]
fn event_builder_weights_mismatch_leaves_the_history_untouched() {
let mut h = History::default();
// Two teams, so ingestion would otherwise succeed. A one-team event is
// rejected as `NotEnoughTeams` before the weights are ever examined, so
// building this with one team would pass vacuously.
let _ = h
.event(1)
.team(["a"])
.weights([1.0, 2.0])
.team(["b"])
.winner(0)
.commit();
assert!(h.learning_curve("a").is_empty());
}
#[test]
fn empty_event_stream_then_converge() {
let mut h = History::default();
h.add_events(std::iter::empty()).unwrap();
let report = h.converge().unwrap();
assert_eq!(report.iterations, 0);
}
#[test]
fn empty_history_queries_do_not_panic() {
let h = History::default();
assert!(h.learning_curves().is_empty());
assert!(h.learning_curve("nobody").is_empty());
assert!(h.current_skill("nobody").is_none());
}
#[test]
fn single_event_history_converges() {
let mut h = History::default();
h.record_winner(&"a", &"b", 1).unwrap();
let report = h.converge().unwrap();
assert!(report.converged);
assert_finite(h.current_skill("a").unwrap(), "single-event skill");
}
#[test]
fn scored_event_rejects_non_positive_sigma() {
let mut h = History::builder().score_sigma(2.0).build();
let err = h
.event(1)
.team(["a"])
.team(["b"])
.scores_with_sigma([3.0, 1.0], f64::NAN)
.commit()
.unwrap_err();
assert!(matches!(
err,
InferenceError::InvalidParameter {
name: "score_sigma",
..
}
));
}
#[test]
fn convergence_reports_are_finite_across_many_teams() {
let opts = GameOptions {
p_draw: 0.1,
convergence: ConvergenceOptions::default(),
..GameOptions::default()
};
let holders: Vec<[R; 1]> = (0..12).map(|_| [rating()]).collect();
let teams: Vec<&[R]> = holders.iter().map(|t| t.as_slice()).collect();
let game = Game::ranked(&teams, Outcome::ranking(0..12), &opts).unwrap();
assert!(
game.log_evidence().is_finite(),
"12-team log-evidence must be finite, got {}",
game.log_evidence()
);
for team in game.posteriors() {
for skill in team {
assert_finite(skill, "12-team posterior");
}
}
}
/// A long diff chain underflows a linear evidence product: each link
/// contributes a probability in (0, 1], so ~1000 links flush the product to
/// exactly 0.0 and `ln(0.0)` is `-inf`. Accumulating in log space keeps it
/// finite.
#[test]
fn log_evidence_survives_a_long_diff_chain() {
let holders: Vec<[R; 1]> = (0..1200).map(|_| [rating()]).collect();
let teams: Vec<&[R]> = holders.iter().map(|t| t.as_slice()).collect();
let game = Game::ranked(
&teams,
Outcome::ranking(0..holders.len() as u32),
&GameOptions::default(),
)
.unwrap();
let log_evidence = game.log_evidence();
assert!(
log_evidence.is_finite(),
"1200-team log-evidence must be finite, got {log_evidence}"
);
assert!(
log_evidence < 0.0,
"log-evidence of a probability must be negative, got {log_evidence}"
);
}
/// A near-certain outcome rounds the losing tail to exactly zero in the
/// `erfc` approximation; the evidence floor keeps `ln` finite.
#[test]
fn log_evidence_finite_for_near_certain_outcome() {
let overwhelming = R::new(
Gaussian::from_ms(5_000.0, 0.5),
1.0,
ConstantDrift::new(0.0),
);
let hopeless = R::new(
Gaussian::from_ms(-5_000.0, 0.5),
1.0,
ConstantDrift::new(0.0),
);
let a = [overwhelming];
let b = [hopeless];
let teams: Vec<&[R]> = vec![&a, &b];
let game = Game::ranked(&teams, Outcome::winner(0, 2), &GameOptions::default()).unwrap();
assert!(
game.log_evidence().is_finite(),
"got {}",
game.log_evidence()
);
// And the reverse — a colossal upset — must also stay finite.
let upset = Game::ranked(&teams, Outcome::winner(1, 2), &GameOptions::default()).unwrap();
assert!(
upset.log_evidence().is_finite(),
"upset log-evidence must be finite, got {}",
upset.log_evidence()
);
}
#[test]
fn empty_history_has_no_filtered_estimates() {
let history: History = History::builder().build();
assert_eq!(history.filtered_log_evidence(), 0.0);
assert!(history.filtered_learning_curves().is_empty());
assert!(history.filtered_learning_curve("nobody").is_empty());
}
// --- Boundary inputs (#26) ----------------------------------------------
fn tight() -> ConvergenceOptions {
ConvergenceOptions {
max_iter: 2_000,
epsilon: 1e-12,
..ConvergenceOptions::default()
}
}
fn assert_curve_finite(h: &History, keys: &[&str], what: &str) {
for key in keys {
for (time, g) in h.learning_curve(*key) {
assert!(
g.mu().is_finite() && g.sigma().is_finite(),
"{what}: non-finite posterior for {key} at t={time} (mu={} sigma={})",
g.mu(),
g.sigma()
);
}
}
}
/// A zero weight reaches `(m - performance.exclude(..)) * (1.0 / w)`, i.e. a
/// division by zero. The commit is accepted today, so this pins that the
/// resulting posterior is still finite rather than quietly NaN.
#[test]
fn zero_weight_does_not_produce_a_non_finite_posterior() {
let mut h = History::builder().build();
h.event(1)
.team(["a"])
.weights([0.0])
.team(["b"])
.winner(0)
.commit()
.expect("a zero weight is accepted today; update this test if that changes");
let _ = h.converge().unwrap();
assert_curve_finite(&h, &["a", "b"], "zero weight");
}
#[test]
fn negative_weight_does_not_produce_a_non_finite_posterior() {
let mut h = History::builder().build();
h.event(1)
.team(["a"])
.weights([-1.0])
.team(["b"])
.winner(0)
.commit()
.expect("a negative weight is accepted today; update this test if that changes");
let _ = h.converge().unwrap();
assert_curve_finite(&h, &["a", "b"], "negative weight");
}
/// Events supplied newest-first must land in the same slices as oldest-first:
/// ingestion sorts by time rather than trusting arrival order.
#[test]
fn out_of_order_timestamps_converge_to_the_same_answer() {
fn build(descending: bool) -> History {
let mut h = History::builder().convergence(tight()).build();
let mut times: Vec<i64> = (1..=6).collect();
if descending {
times.reverse();
}
for time in times {
h.record_winner(&"a", &"b", time).unwrap();
}
let _ = h.converge().unwrap();
h
}
let ascending = build(false);
let descending = build(true);
let one = ascending.current_skill("a").unwrap();
let other = descending.current_skill("a").unwrap();
assert!(
(one.mu() - other.mu()).abs() < 1e-8 && (one.sigma() - other.sigma()).abs() < 1e-8,
"arrival order changed the answer: ascending mu={} sigma={}, descending mu={} sigma={}",
one.mu(),
one.sigma(),
other.mu(),
other.sigma()
);
}
#[test]
fn extreme_beta_and_sigma_stay_finite() {
for (beta, sigma) in [(1e-6, 1e-6), (1e6, 1e6), (1e-6, 1e6), (1e6, 1e-6)] {
let mut h = History::builder().beta(beta).sigma(sigma).build();
h.record_winner(&"a", &"b", 1).unwrap();
h.record_winner(&"a", &"b", 2).unwrap();
let _ = h.converge().unwrap();
assert_curve_finite(&h, &["a", "b"], &format!("beta={beta} sigma={sigma}"));
}
}