midi-harbor/crates/daemon/tests/recovery_timing.rs
2026-09-28 13:59:10 -05:00

143 lines
6.2 KiB
Rust

//! The two numbers that define self-healing.
//!
//! SC-003 and SC-004 are the product's central claim: MIDI resumes within ten seconds of an
//! interruption ending, and within fifteen of a wake. Everything else in this system is in
//! service of those.
//!
//! These are simulations over the real constants and the real backoff, not over a real network.
//! What they guard is the part of recovery that is scheduling (when the next attempt happens),
//! because that is the part a change to a constant can quietly blow, and the part that was
//! blowing it. Measured end-to-end figures live in research R-034.
#![allow(
clippy::expect_used,
clippy::indexing_slicing,
clippy::panic,
clippy::unwrap_used
)]
use midi_harbor_core::backoff::{Backoff, BackoffPolicy};
use midi_harbor_daemon::session::TICK_INTERVAL;
use midi_harbor_daemon::state::SYSTEM_EVENT_INTERVAL;
use midi_harbor_platform::sysevents::SAMPLE_INTERVAL;
use midi_harbor_rtpmidi::clock::LIVENESS_TIMEOUT;
use midi_harbor_rtpmidi::session::{INVITE_TIMEOUT, PROBE_TIMEOUT};
use std::time::Duration;
/// SC-003: MIDI resumes within ten seconds of an interruption ending.
const RESUME_BUDGET: Duration = Duration::from_secs(10);
/// SC-004: sessions resume within fifteen seconds of the machine waking.
const WAKE_BUDGET: Duration = Duration::from_secs(15);
/// How long an invitation takes to be answered, at worst.
///
/// One lost invitation is assumed rather than the happy path, because a budget that only holds
/// when nothing goes wrong is not a budget.
const HANDSHAKE: Duration = INVITE_TIMEOUT;
/// Interruptions to induce, per SC-003's "at least 99 of 100".
const TRIALS: usize = 100;
/// How long each simulated interruption lasts.
///
/// Longer than the liveness timeout on purpose: a shorter outage never reaches the retry loop at
/// all, because the session does not notice it and the MIDI simply resumes. The interesting case
/// is the one where the session gave up and is waiting.
const OUTAGE: Duration = Duration::from_secs(300);
/// Returns when each retry would be attempted through one simulated outage.
///
/// Uses the real backoff with real jitter, so the schedule is the one a session would follow.
fn retry_schedule(policy: BackoffPolicy) -> Vec<Duration> {
let mut backoff = Backoff::new(policy);
let mut attempts = Vec::new();
let mut at = Duration::ZERO;
while at < OUTAGE {
at = at.saturating_add(backoff.next_delay_random());
attempts.push(at);
}
attempts
}
/// Proves SC-003: at least 99 of 100 interruptions are recovered within ten seconds of ending.
///
/// Nothing reports that an interruption ended when this machine's own address never changed (an
/// upstream router rebooting looks exactly like silence), so the only thing that finds out is the
/// next scheduled retry. The worst case is the responsive retry ceiling of 5 s, plus a lost
/// invitation's 2 s `INVITE_TIMEOUT`, plus one 250 ms session tick: 7.25 s, inside the 10 s budget.
/// The default ceiling of 30 s plus the same 2 s handshake is 32 s, which is why sessions carry a
/// policy of their own; if the ceiling is ever raised back towards the default, the ceiling
/// assertion says why before the simulation does.
#[test]
fn an_interruption_that_ends_is_recovered_within_ten_seconds() {
// Bound the worst case from the ceiling.
let ceiling = BackoffPolicy::responsive().max;
let worst = ceiling
.saturating_add(HANDSHAKE)
.saturating_add(TICK_INTERVAL);
assert!(
worst <= RESUME_BUDGET,
"a retry ceiling of {ceiling:?} leaves {worst:?} in the worst case, over the {RESUME_BUDGET:?} budget"
);
assert!(
BackoffPolicy::default().max.saturating_add(HANDSHAKE) > RESUME_BUDGET,
"the default ceiling is supposed to be the one that cannot meet this, or sessions need no policy of their own"
);
// Simulate interruptions over the real backoff with real jitter.
let mut breaches = Vec::new();
for trial in 0..TRIALS {
let schedule = retry_schedule(BackoffPolicy::responsive());
// The network returns at a point of its own choosing, which lands inside whichever retry
// interval happens to cover it.
let returns_at = OUTAGE.mul_f64(rand::random::<f64>());
let next_attempt = schedule
.iter()
.find(|at| **at >= returns_at)
.copied()
.expect("the session never stops retrying, so an attempt follows the network's return");
let recovery = (next_attempt - returns_at)
.saturating_add(HANDSHAKE)
.saturating_add(TICK_INTERVAL);
if recovery > RESUME_BUDGET {
breaches.push((trial, recovery));
}
}
assert!(
breaches.len() <= 1,
"{} of {TRIALS} interruptions took longer than {RESUME_BUDGET:?} to recover: {breaches:?}",
breaches.len()
);
}
/// Proves SC-004: a session suspended with the machine is recovered within fifteen seconds of the
/// wake.
///
/// After a suspend the session is still Connected, because nothing ran while the machine was
/// away. Recovery is: notice the machine came back (the 1 s clock `SAMPLE_INTERVAL` plus the 1 s
/// `SYSTEM_EVENT_INTERVAL` watcher pass), ask the peer whether the link survived and give up on
/// the answer (2 s `PROBE_TIMEOUT`), wait for the next 250 ms tick, then reconnect through a lost
/// invitation (2 s). That is 6.25 s, inside the 15 s budget. The ordinary 35 s `LIVENESS_TIMEOUT`
/// would miss it, which is the reason the probe exists: every build before the probe waited for
/// it, because a suspended session looks exactly like a working one until it expires.
#[test]
fn a_wake_is_recovered_within_fifteen_seconds() {
let notice = SAMPLE_INTERVAL.saturating_add(SYSTEM_EVENT_INTERVAL);
let worst = notice
.saturating_add(PROBE_TIMEOUT)
.saturating_add(TICK_INTERVAL)
.saturating_add(HANDSHAKE);
assert!(
worst <= WAKE_BUDGET,
"waking takes {worst:?} in the worst case, over the {WAKE_BUDGET:?} budget"
);
assert!(
LIVENESS_TIMEOUT > WAKE_BUDGET,
"if the liveness timeout fits inside the budget, the probe is no longer load-bearing and this reasoning needs revisiting"
);
}