Files
trueskill-tt/tests/game.rs
T
logaritmiskandClaude Opus 5 eebf8aacd3 fix!: reject malformed games at the Game boundary too
I fixed this at `History`'s ingestion chokepoint and said the boundary
was complete. It was not. `Game` is a separate public entry point that
does not pass through that chokepoint, and every one of the same four
defects was still live there:

  Game::ranked(&[&[a]], ..)  -> PANIC at src/game.rs:317
  Game::scored(&[&[a]], ..)  -> PANIC at src/game.rs:317
  Game::ranked(&[&[], &[a]]) -> Ok, finite posterior for the opponent
  Game::scored(.., [NaN, 1]) -> Ok

The same panic, from safe API, in release. Fixing one path and
generalising from it is exactly the mistake that produced the
latest-slice joint bug: validating on the shape that cannot expose the
problem, then reporting the property as held.

`Game::validate_teams` is shared by `ranked` and `scored`, with the
non-finite score check in `scored` alongside it. Ranks need no equivalent
— they are `u32`.

`one_v_one` and `free_for_all` build their teams internally and are
unaffected; a test asserts all three well-formed constructors still
succeed, so the check cannot quietly widen.

BREAKING CHANGE: `Game::ranked` and `Game::scored` return
`NotEnoughTeams`, `EmptyTeam` or `InvalidParameter` for inputs they
previously panicked on or silently accepted.

Refs #18, #26

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011hcFjNDmHXZF8URGLku5zZ
2026-09-08 21:13:02 +02:00

253 lines
7.6 KiB
Rust

use trueskill_tt::{
ConstantDrift, ConvergenceOptions, Game, GameOptions, Gaussian, InferenceError, Outcome, Rating,
};
type R = Rating<i64, ConstantDrift>;
fn default_rating() -> R {
R::new(
Gaussian::from_ms(25.0, 25.0 / 3.0),
25.0 / 6.0,
ConstantDrift(25.0 / 300.0),
)
}
#[test]
fn game_ranked_1v1_golden() {
let a = default_rating();
let b = default_rating();
let g = Game::<i64, _>::ranked(
&[&[a], &[b]],
Outcome::winner(0, 2),
&GameOptions::default(),
)
.unwrap();
let p = g.posteriors();
assert!(p[0][0].mu() > 25.0);
assert!(p[1][0].mu() < 25.0);
assert!((p[0][0].sigma() - p[1][0].sigma()).abs() < 1e-6);
}
#[test]
fn game_one_v_one_shortcut() {
let a = default_rating();
let b = default_rating();
let (a_post, b_post) =
Game::<i64, _>::one_v_one(&a, &b, Outcome::winner(0, 2), &GameOptions::default()).unwrap();
assert!(a_post.mu() > 25.0);
assert!(b_post.mu() < 25.0);
}
#[test]
fn game_ranked_rejects_bad_p_draw() {
let a = R::new(Gaussian::default(), 1.0, ConstantDrift(0.0));
let err = Game::<i64, _>::ranked(
&[&[a], &[a]],
Outcome::winner(0, 2),
&GameOptions {
p_draw: 1.5,
score_sigma: 1.0,
convergence: ConvergenceOptions::default(),
},
)
.unwrap_err();
assert!(matches!(err, InferenceError::InvalidProbability { .. }));
}
#[test]
fn game_ranked_rejects_mismatched_ranks() {
let a = R::new(Gaussian::default(), 1.0, ConstantDrift(0.0));
let err = Game::<i64, _>::ranked(
&[&[a], &[a]],
Outcome::ranking([0, 1, 2]),
&GameOptions::default(),
)
.unwrap_err();
assert!(matches!(err, InferenceError::MismatchedShape { .. }));
}
#[test]
fn game_free_for_all_three_players() {
let a = default_rating();
let b = default_rating();
let c = default_rating();
let g = Game::<i64, _>::free_for_all(
&[&a, &b, &c],
Outcome::ranking([0, 1, 2]),
&GameOptions::default(),
)
.unwrap();
let p = g.posteriors();
assert_eq!(p.len(), 3);
assert!(p[0][0].mu() > p[1][0].mu());
assert!(p[1][0].mu() > p[2][0].mu());
}
#[test]
fn game_log_evidence_is_finite() {
let a = default_rating();
let b = default_rating();
let g = Game::<i64, _>::ranked(
&[&[a], &[b]],
Outcome::winner(0, 2),
&GameOptions::default(),
)
.unwrap();
assert!(g.log_evidence().is_finite());
assert!(g.log_evidence() < 0.0);
}
/// `one_v_one` used to hardcode `GameOptions::default()`, so a 1v1 could
/// never set `p_draw` and a drawn 1v1 was unreachable through it.
#[test]
fn one_v_one_honours_the_draw_probability_it_is_given() {
let a = default_rating();
let b = default_rating();
// Default options still reject a draw, because the default p_draw is zero.
let err = Game::<i64, _>::one_v_one(&a, &b, Outcome::draw(2), &GameOptions::default())
.expect_err("a draw needs a positive p_draw");
assert!(matches!(
err,
InferenceError::TieWithoutDrawProbability { .. }
));
// With a draw probability supplied it succeeds — which was impossible
// before the signature took options.
let options = GameOptions {
p_draw: 0.25,
..GameOptions::default()
};
let (a_post, b_post) = Game::<i64, _>::one_v_one(&a, &b, Outcome::draw(2), &options)
.expect("a draw is representable once p_draw is positive");
// A symmetric draw leaves the means alone and sharpens both sides.
assert!((a_post.mu() - b_post.mu()).abs() < 1e-9);
assert!(a_post.sigma() < 25.0 / 3.0);
}
/// Convergence options reach the 1v1 path too, not just `p_draw`.
#[test]
fn one_v_one_honours_convergence_options() {
let a = default_rating();
let b = default_rating();
let options = GameOptions {
convergence: ConvergenceOptions::default(),
..GameOptions::default()
};
let (a_post, _) = Game::<i64, _>::one_v_one(&a, &b, Outcome::winner(0, 2), &options).unwrap();
assert!(a_post.mu() > 25.0);
}
/// `Game` is a public entry point that does not pass through `History`'s
/// ingestion chokepoint, so it needs its own boundary — and did not have one.
///
/// A one-team game panicked at `src/game.rs:317` with "range start index 1 out
/// of range for slice of length 0", in release, from safe API. This is the
/// same defect `tests/ingestion_shape.rs` covers for `History`; fixing that
/// path left this one open, because they share no validation.
mod malformed_games {
use super::*;
#[test]
fn a_one_team_ranked_game_is_an_error_not_a_panic() {
let a = default_rating();
let err = Game::<i64, _>::ranked(&[&[a]], Outcome::winner(0, 1), &GameOptions::default())
.unwrap_err();
assert!(
matches!(err, InferenceError::NotEnoughTeams { got: 1 }),
"{err:?}"
);
}
#[test]
fn a_one_team_scored_game_is_an_error_not_a_panic() {
let a = default_rating();
let err = Game::<i64, _>::scored(
&[&[a]],
Outcome::scores([1.0]),
&GameOptions {
score_sigma: 1.0,
..GameOptions::default()
},
)
.unwrap_err();
assert!(
matches!(err, InferenceError::NotEnoughTeams { got: 1 }),
"{err:?}"
);
}
#[test]
fn a_zero_team_game_is_an_error() {
let err =
Game::<i64, ConstantDrift>::ranked(&[], Outcome::ranking([]), &GameOptions::default())
.unwrap_err();
assert!(
matches!(err, InferenceError::NotEnoughTeams { got: 0 }),
"{err:?}"
);
}
/// The quiet half: an empty team contributed no performance, so the game
/// returned a finite posterior for its opponent as though it had won one.
#[test]
fn an_empty_team_is_an_error() {
let a = default_rating();
let err =
Game::<i64, _>::ranked(&[&[], &[a]], Outcome::winner(0, 2), &GameOptions::default())
.unwrap_err();
assert!(
matches!(err, InferenceError::EmptyTeam { team: 0 }),
"{err:?}"
);
}
#[test]
fn a_non_finite_score_is_an_error() {
let a = default_rating();
for bad in [f64::NAN, f64::INFINITY, f64::NEG_INFINITY] {
let err = Game::<i64, _>::scored(
&[&[a], &[a]],
Outcome::scores([bad, 1.0]),
&GameOptions {
score_sigma: 1.0,
..GameOptions::default()
},
)
.unwrap_err();
assert!(
matches!(err, InferenceError::InvalidParameter { name: "score", .. }),
"{bad}: {err:?}"
);
}
}
/// `free_for_all` and `one_v_one` build their teams internally, so they
/// must keep working — the check must not catch well-formed games.
#[test]
fn well_formed_games_are_untouched() {
let a = default_rating();
assert!(
Game::<i64, _>::ranked(
&[&[a], &[a]],
Outcome::winner(0, 2),
&GameOptions::default()
)
.is_ok()
);
assert!(
Game::<i64, _>::free_for_all(
&[&a, &a, &a],
Outcome::ranking([0, 1, 2]),
&GameOptions::default()
)
.is_ok()
);
assert!(
Game::<i64, _>::one_v_one(&a, &a, Outcome::winner(0, 2), &GameOptions::default())
.is_ok()
);
}
}