test: add property-based tests, a shared finiteness helper, and boundary inputs
Most of what remained on #26. **Property tests (`tests/properties.rs`, proptest as a dev-dependency).** Four invariants over generated 1v1 schedules rather than hand-written fixtures, which is where this crate's shipped defects actually hid — a linear evidence product that underflowed only past ~1000 teams, and a batching path no golden exercised because every golden ingests in one call: - converged posteriors are always finite with positive sigma - log-evidence, batch and filtered, is finite and never above zero - filtered evidence is invariant to whether `converge` has run - one-at-a-time ingestion reaches the same fixed point as batched The invariance property was mutation-proved: making `filtered_step` read `skill.forward` instead of the carried message fails it with `-1.1038430064192069 -> -1.1135747072822761`. **Shared finiteness helper (`tests/common/mod.rs`).** `assert_finite` was local to `degenerate_inputs.rs`. It now also rejects a non-positive sigma, which the old version let through — `Gaussian::sigma` reports a non-positive precision as improper rather than trapping, so a collapsed posterior would have passed a finite-only check. **Boundary inputs.** Zero and negative weights, out-of-order timestamps, and extreme beta/sigma combinations. Worth recording that zero weight reaches `(m - performance.exclude(..)) * (1.0 / w)` — a division by zero — and the posterior comes out finite anyway; the test pins that rather than asserting what ought to happen. The weight tests `expect()` the commit rather than returning early on error, because an early return would have made them vacuous the moment validation changed. I checked that specifically by turning the return into a failure and confirming it did not fire. Not done, and left on #26: benchmark regression gating. Nothing fails on a regression today; making it fail needs a threshold chosen against how noisy the shared runner is, which is a policy call rather than a mechanical one. 60 test binaries, up from 56. MSRV 1.85 verified with proptest in the graph. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01T5SYDExxL4vZgvunrcNSMc
This commit is contained in:
@@ -62,6 +62,7 @@ rayon = ["dep:rayon"]
|
|||||||
criterion = "0.5"
|
criterion = "0.5"
|
||||||
plotters = { version = "0.3", default-features = false, features = ["svg_backend", "all_elements", "all_series"] }
|
plotters = { version = "0.3", default-features = false, features = ["svg_backend", "all_elements", "all_series"] }
|
||||||
plotters-backend = "0.3"
|
plotters-backend = "0.3"
|
||||||
|
proptest = "1.11.0"
|
||||||
time = { version = "0.3", features = ["parsing"] }
|
time = { version = "0.3", features = ["parsing"] }
|
||||||
trueskill-tt = { path = ".", features = ["approx"] }
|
trueskill-tt = { path = ".", features = ["approx"] }
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,38 @@
|
|||||||
|
//! Helpers shared across the integration suites.
|
||||||
|
//!
|
||||||
|
//! Each integration file is its own binary, so `mod common;` compiles a copy
|
||||||
|
//! per suite. Anything unused in a given suite would warn, hence the
|
||||||
|
//! `#![allow(dead_code)]`.
|
||||||
|
|
||||||
|
#![allow(dead_code)]
|
||||||
|
|
||||||
|
use trueskill_tt::Gaussian;
|
||||||
|
|
||||||
|
/// A posterior must be finite with a strictly positive sigma.
|
||||||
|
///
|
||||||
|
/// A non-finite posterior is the failure mode this crate is most prone to —
|
||||||
|
/// EP breaking down produces NaN rather than an error — and a zero or negative
|
||||||
|
/// sigma means the precision went non-positive, which `Gaussian::sigma` reports
|
||||||
|
/// as improper rather than trapping.
|
||||||
|
pub fn assert_finite(g: Gaussian, what: &str) {
|
||||||
|
assert!(
|
||||||
|
g.mu().is_finite(),
|
||||||
|
"{what}: mu is not finite (mu={}, sigma={})",
|
||||||
|
g.mu(),
|
||||||
|
g.sigma()
|
||||||
|
);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
g.sigma().is_finite() && g.sigma() > 0.0,
|
||||||
|
"{what}: sigma must be finite and positive (mu={}, sigma={})",
|
||||||
|
g.mu(),
|
||||||
|
g.sigma()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every point on every learning curve must be finite.
|
||||||
|
pub fn assert_curve_finite(curve: &[(i64, Gaussian)], who: &str) {
|
||||||
|
for (time, g) in curve {
|
||||||
|
assert_finite(*g, &format!("{who} at t={time}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
+112
-9
@@ -3,6 +3,9 @@
|
|||||||
//! These run in both debug and release: the defects they pin were all
|
//! These run in both debug and release: the defects they pin were all
|
||||||
//! guarded only by `debug_assert!`, so a debug-only suite never saw them.
|
//! guarded only by `debug_assert!`, so a debug-only suite never saw them.
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::assert_finite;
|
||||||
use trueskill_tt::{
|
use trueskill_tt::{
|
||||||
ConstantDrift, ConvergenceOptions, Game, GameOptions, Gaussian, History, InferenceError,
|
ConstantDrift, ConvergenceOptions, Game, GameOptions, Gaussian, History, InferenceError,
|
||||||
NullObserver, Outcome, Rating,
|
NullObserver, Outcome, Rating,
|
||||||
@@ -18,15 +21,6 @@ fn rating() -> R {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn assert_finite(g: Gaussian, what: &str) {
|
|
||||||
assert!(
|
|
||||||
g.mu().is_finite() && g.sigma().is_finite(),
|
|
||||||
"{what} must be finite, got mu={} sigma={}",
|
|
||||||
g.mu(),
|
|
||||||
g.sigma()
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn record_draw_without_draw_probability_is_rejected() {
|
fn record_draw_without_draw_probability_is_rejected() {
|
||||||
let mut h = History::default();
|
let mut h = History::default();
|
||||||
@@ -318,3 +312,112 @@ fn empty_history_has_no_filtered_estimates() {
|
|||||||
|
|
||||||
assert!(history.filtered_learning_curve("nobody").is_empty());
|
assert!(history.filtered_learning_curve("nobody").is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- Boundary inputs (#26) ----------------------------------------------
|
||||||
|
|
||||||
|
fn tight() -> ConvergenceOptions {
|
||||||
|
ConvergenceOptions {
|
||||||
|
max_iter: 2_000,
|
||||||
|
epsilon: 1e-12,
|
||||||
|
..ConvergenceOptions::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn assert_curve_finite(h: &History, keys: &[&str], what: &str) {
|
||||||
|
for key in keys {
|
||||||
|
for (time, g) in h.learning_curve(*key) {
|
||||||
|
assert!(
|
||||||
|
g.mu().is_finite() && g.sigma().is_finite(),
|
||||||
|
"{what}: non-finite posterior for {key} at t={time} (mu={} sigma={})",
|
||||||
|
g.mu(),
|
||||||
|
g.sigma()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A zero weight reaches `(m - performance.exclude(..)) * (1.0 / w)`, i.e. a
|
||||||
|
/// division by zero. The commit is accepted today, so this pins that the
|
||||||
|
/// resulting posterior is still finite rather than quietly NaN.
|
||||||
|
#[test]
|
||||||
|
fn zero_weight_does_not_produce_a_non_finite_posterior() {
|
||||||
|
let mut h = History::builder().build();
|
||||||
|
|
||||||
|
h.event(1)
|
||||||
|
.team(["a"])
|
||||||
|
.weights([0.0])
|
||||||
|
.team(["b"])
|
||||||
|
.winner(0)
|
||||||
|
.commit()
|
||||||
|
.expect("a zero weight is accepted today; update this test if that changes");
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
assert_curve_finite(&h, &["a", "b"], "zero weight");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn negative_weight_does_not_produce_a_non_finite_posterior() {
|
||||||
|
let mut h = History::builder().build();
|
||||||
|
|
||||||
|
h.event(1)
|
||||||
|
.team(["a"])
|
||||||
|
.weights([-1.0])
|
||||||
|
.team(["b"])
|
||||||
|
.winner(0)
|
||||||
|
.commit()
|
||||||
|
.expect("a negative weight is accepted today; update this test if that changes");
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
assert_curve_finite(&h, &["a", "b"], "negative weight");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Events supplied newest-first must land in the same slices as oldest-first:
|
||||||
|
/// ingestion sorts by time rather than trusting arrival order.
|
||||||
|
#[test]
|
||||||
|
fn out_of_order_timestamps_converge_to_the_same_answer() {
|
||||||
|
fn build(descending: bool) -> History {
|
||||||
|
let mut h = History::builder().convergence(tight()).build();
|
||||||
|
|
||||||
|
let mut times: Vec<i64> = (1..=6).collect();
|
||||||
|
if descending {
|
||||||
|
times.reverse();
|
||||||
|
}
|
||||||
|
|
||||||
|
for time in times {
|
||||||
|
h.record_winner(&"a", &"b", time).unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
let ascending = build(false);
|
||||||
|
let descending = build(true);
|
||||||
|
|
||||||
|
let one = ascending.current_skill("a").unwrap();
|
||||||
|
let other = descending.current_skill("a").unwrap();
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
(one.mu() - other.mu()).abs() < 1e-8 && (one.sigma() - other.sigma()).abs() < 1e-8,
|
||||||
|
"arrival order changed the answer: ascending mu={} sigma={}, descending mu={} sigma={}",
|
||||||
|
one.mu(),
|
||||||
|
one.sigma(),
|
||||||
|
other.mu(),
|
||||||
|
other.sigma()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extreme_beta_and_sigma_stay_finite() {
|
||||||
|
for (beta, sigma) in [(1e-6, 1e-6), (1e6, 1e6), (1e-6, 1e6), (1e6, 1e-6)] {
|
||||||
|
let mut h = History::builder().beta(beta).sigma(sigma).build();
|
||||||
|
|
||||||
|
h.record_winner(&"a", &"b", 1).unwrap();
|
||||||
|
h.record_winner(&"a", &"b", 2).unwrap();
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
assert_curve_finite(&h, &["a", "b"], &format!("beta={beta} sigma={sigma}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,167 @@
|
|||||||
|
//! Property-based tests over generated histories.
|
||||||
|
//!
|
||||||
|
//! The golden suite pins exact values against the Python/Julia reference on a
|
||||||
|
//! handful of fixtures. These pin *invariants* over inputs nobody wrote by
|
||||||
|
//! hand, which is where the defects this crate has actually shipped were
|
||||||
|
//! hiding: a linear evidence product that underflowed only past ~1000 teams,
|
||||||
|
//! and a batching path no golden exercised because every golden ingests in one
|
||||||
|
//! call.
|
||||||
|
|
||||||
|
mod common;
|
||||||
|
|
||||||
|
use common::assert_finite;
|
||||||
|
use proptest::prelude::*;
|
||||||
|
use smallvec::smallvec;
|
||||||
|
use trueskill_tt::{ConvergenceOptions, Event, History, Member, Outcome, Team};
|
||||||
|
|
||||||
|
/// Distinct competitors, so no event pits someone against themselves.
|
||||||
|
fn pairs() -> impl Strategy<Value = Vec<(usize, usize)>> {
|
||||||
|
prop::collection::vec((0usize..8, 0usize..8), 1..24)
|
||||||
|
.prop_map(|v| v.into_iter().filter(|(a, b)| a != b).collect::<Vec<_>>())
|
||||||
|
.prop_filter("needs at least one valid pair", |v| !v.is_empty())
|
||||||
|
}
|
||||||
|
|
||||||
|
const KEYS: [&str; 8] = ["a", "b", "c", "d", "e", "f", "g", "h"];
|
||||||
|
|
||||||
|
fn history_from(games: &[(usize, usize)]) -> History {
|
||||||
|
let mut h = History::builder()
|
||||||
|
.convergence(ConvergenceOptions {
|
||||||
|
max_iter: 200,
|
||||||
|
epsilon: 1e-10,
|
||||||
|
..ConvergenceOptions::default()
|
||||||
|
})
|
||||||
|
.build();
|
||||||
|
|
||||||
|
let events: Vec<Event<i64, &'static str>> = games
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, &(a, b))| Event {
|
||||||
|
time: i as i64 + 1,
|
||||||
|
teams: smallvec![
|
||||||
|
Team::with_members([Member::new(KEYS[a])]),
|
||||||
|
Team::with_members([Member::new(KEYS[b])]),
|
||||||
|
],
|
||||||
|
outcome: Outcome::winner(0, 2),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
h.add_events(events).unwrap();
|
||||||
|
|
||||||
|
h
|
||||||
|
}
|
||||||
|
|
||||||
|
proptest! {
|
||||||
|
#![proptest_config(ProptestConfig::with_cases(48))]
|
||||||
|
|
||||||
|
/// Whatever the schedule of games, convergence must not produce NaN or an
|
||||||
|
/// improper posterior. `converge` returns `NonFiniteResult` rather than
|
||||||
|
/// silently reporting a NaN step as converged, so a break shows up here as
|
||||||
|
/// either an Err or a non-finite curve point.
|
||||||
|
#[test]
|
||||||
|
fn converged_posteriors_are_always_finite(games in pairs()) {
|
||||||
|
let mut h = history_from(&games);
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
for key in KEYS {
|
||||||
|
for (time, g) in h.learning_curve(key) {
|
||||||
|
assert_finite(g, &format!("{key} at t={time}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Log-evidence is a log probability: finite, and never above zero.
|
||||||
|
///
|
||||||
|
/// The linear-product implementation this replaced underflowed to zero on
|
||||||
|
/// long chains, making `ln(0)` = -inf — finite-ness is the property that
|
||||||
|
/// would have caught it.
|
||||||
|
#[test]
|
||||||
|
fn log_evidence_is_a_finite_log_probability(games in pairs()) {
|
||||||
|
let mut h = history_from(&games);
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
let batch = h.log_evidence();
|
||||||
|
let filtered = h.filtered_log_evidence();
|
||||||
|
|
||||||
|
prop_assert!(batch.is_finite(), "batch log-evidence {batch} is not finite");
|
||||||
|
prop_assert!(batch <= 0.0, "batch log-evidence {batch} exceeds zero");
|
||||||
|
prop_assert!(filtered.is_finite(), "filtered log-evidence {filtered} is not finite");
|
||||||
|
prop_assert!(filtered <= 0.0, "filtered log-evidence {filtered} exceeds zero");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Filtered estimates must not depend on whether `converge` has run — the
|
||||||
|
/// property the whole forward-only design rests on.
|
||||||
|
#[test]
|
||||||
|
fn filtered_evidence_is_invariant_to_convergence(games in pairs()) {
|
||||||
|
let mut h = history_from(&games);
|
||||||
|
|
||||||
|
let before = h.filtered_log_evidence();
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
|
||||||
|
let after = h.filtered_log_evidence();
|
||||||
|
|
||||||
|
prop_assert!(
|
||||||
|
(before - after).abs() < 1e-8,
|
||||||
|
"filtered evidence moved across converge(): {before} -> {after}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ingesting the same games one at a time must reach the same fixed point
|
||||||
|
/// as ingesting them in one call.
|
||||||
|
#[test]
|
||||||
|
fn ingestion_order_does_not_change_the_answer(games in pairs()) {
|
||||||
|
let batched = {
|
||||||
|
let mut h = history_from(&games);
|
||||||
|
h.converge().unwrap();
|
||||||
|
h
|
||||||
|
};
|
||||||
|
|
||||||
|
let incremental = {
|
||||||
|
let mut h = History::builder()
|
||||||
|
.convergence(ConvergenceOptions {
|
||||||
|
max_iter: 200,
|
||||||
|
epsilon: 1e-10,
|
||||||
|
..ConvergenceOptions::default()
|
||||||
|
})
|
||||||
|
.build();
|
||||||
|
|
||||||
|
for (i, &(a, b)) in games.iter().enumerate() {
|
||||||
|
h.add_events([Event {
|
||||||
|
time: i as i64 + 1,
|
||||||
|
teams: smallvec![
|
||||||
|
Team::with_members([Member::new(KEYS[a])]),
|
||||||
|
Team::with_members([Member::new(KEYS[b])]),
|
||||||
|
],
|
||||||
|
outcome: Outcome::winner(0, 2),
|
||||||
|
}])
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
h.converge().unwrap();
|
||||||
|
h
|
||||||
|
};
|
||||||
|
|
||||||
|
for key in KEYS {
|
||||||
|
let one = batched.current_skill(key);
|
||||||
|
let other = incremental.current_skill(key);
|
||||||
|
|
||||||
|
match (one, other) {
|
||||||
|
(Some(one), Some(other)) => {
|
||||||
|
prop_assert!(
|
||||||
|
(one.mu() - other.mu()).abs() < 1e-6
|
||||||
|
&& (one.sigma() - other.sigma()).abs() < 1e-6,
|
||||||
|
"{key}: batched mu={} sigma={}, incremental mu={} sigma={}",
|
||||||
|
one.mu(),
|
||||||
|
one.sigma(),
|
||||||
|
other.mu(),
|
||||||
|
other.sigma()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
(None, None) => {}
|
||||||
|
_ => prop_assert!(false, "{key} present in only one history"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user