This is an automated email from the ASF dual-hosted git repository. krishvishal pushed a commit to branch sim-workload-faults in repository https://gitbox.apache.org/repos/asf/iggy.git
commit 1771001c876a4d78c20a9164df61744326f75d74 Author: Krishna Vishal <[email protected]> AuthorDate: Sat Aug 15 14:45:30 2026 +0530 feat(simulator): re-establish a session after the cluster evicts a client Session bindings live in the per-connection `SessionManager` and do not survive a replica restart, so a client that had a session there becomes unbound and its next replicated request is refused with `Eviction(NoSession)`. The driver treated that as unrecoverable and stopped the run, which made the dispatch shell and crash/restart mutually exclusive: every combination of the two failed. A real client reconnects and logs in, so the driver does too. The client table is replicated metadata and survives the restart, so the re-login rebinds the existing entry rather than creating one, and request numbering continues instead of restarting. The login rotates targets across retries for the same reason the workload's resends do: a register has to reach the metadata primary, and retrying one replica is useless once it is partitioned or crashed mid-handshake. Fixing that alone took the sweep from 4 of 9 to 11 of 12. Outstanding requests are forgotten rather than resent, since the message the retry buffer holds carries the old session id and would be refused again. That forgetting disarms the strict outcome oracle for the rest of the run. The refusal proves only that the ATTEMPT which drew it did not commit, and that attempt may have been a resend whose original had already committed with its reply lost; the shadow is then missing an effect that really happened. Keeping the oracle armed would turn a known unknown into a spurious failure, so it stands down and says why. Shell with faults and crash/restart now drains and converges 11 of 12 across four op mixes, where it previously could not run at all. The raw path is unchanged at 9 of 9. --- core/simulator/src/bin/workload-fuzz.rs | 3 +- core/simulator/src/lib.rs | 7 +++ core/simulator/src/workload/auditor.rs | 12 +++++ core/simulator/src/workload/mod.rs | 93 +++++++++++++++++++++++++-------- core/simulator/src/workload/oracle.rs | 8 +++ 5 files changed, 101 insertions(+), 22 deletions(-) diff --git a/core/simulator/src/bin/workload-fuzz.rs b/core/simulator/src/bin/workload-fuzz.rs index dd0720cf4..6638cf0d4 100644 --- a/core/simulator/src/bin/workload-fuzz.rs +++ b/core/simulator/src/bin/workload-fuzz.rs @@ -526,13 +526,14 @@ fn print_coverage(workload: &Workload) { let stats = workload.auditor.stats(); println!( "coverage: replies_seen={} replies_unknown={} committed_rejections={} \ - samples_none={} resends={} denials={}", + samples_none={} resends={} denials={} evictions={}", stats.replies_seen, stats.replies_unknown, stats.committed_rejections, workload.samples_none(), workload.resends(), stats.denials, + workload.evictions(), ); for action in Action::iter() { let commits = stats.commits(action); diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 4a0b3a08d..61b8cca2d 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -606,9 +606,16 @@ impl Simulator { message: &Message<GenericHeader>, label: &str, ) -> Message<ReplyHeader> { + let mut target = target; for step in 0..SETUP_TOTAL_STEPS { if step % SETUP_RETRY_STEPS == 0 { self.submit_request(client_id, target, message.deep_copy()); + // Rotate, exactly as the workload's resend path does. Retrying the + // same replica forever is enough on a perfect network and useless + // once one is partitioned or crashed mid-handshake: a register has + // to reach the metadata primary, and the replica that looked live + // when this started may be neither reachable nor able to forward. + target = (target + 1) % self.replica_count.max(1); } if let Some(reply) = self.step().into_iter().next() { return reply; diff --git a/core/simulator/src/workload/auditor.rs b/core/simulator/src/workload/auditor.rs index 96cd09258..44fd724d4 100644 --- a/core/simulator/src/workload/auditor.rs +++ b/core/simulator/src/workload/auditor.rs @@ -209,6 +209,18 @@ impl ServerAuditor { &self.stats } + /// Drop every in-flight expectation for `client`, returning how many went. + /// + /// For a client the cluster evicted: its session is gone, so nothing it had + /// outstanding will ever be answered and an expectation left behind would + /// wait forever. The requests themselves were refused before commit, so + /// forgetting them loses no committed state. + pub fn forget_client(&mut self, client: u128) -> usize { + let before = self.in_flight.len(); + self.in_flight.retain(|&(owner, _), _| owner != client); + before - self.in_flight.len() + } + /// The action of an outstanding request, if one is recorded for `key`. /// Diagnostic only: names what a stalled run is waiting on, which the bare /// `(client, request)` pair cannot. diff --git a/core/simulator/src/workload/mod.rs b/core/simulator/src/workload/mod.rs index 9e25f9f82..95ca36c63 100644 --- a/core/simulator/src/workload/mod.rs +++ b/core/simulator/src/workload/mod.rs @@ -93,6 +93,9 @@ pub struct Workload { now: u64, /// Total resends issued, for the run summary. resends: u64, + /// Client evictions survived, for the run summary. Only the dispatch shell + /// produces them, and in practice only once replicas restart under it. + evictions: u64, /// Debug counter for `sample()` returning `None` (a targeted outcome whose /// shadow precondition is unmet). Flags PRNG-trace drift during development. samples_none: u64, @@ -121,6 +124,7 @@ impl Workload { outstanding: BTreeMap::new(), now: 0, resends: 0, + evictions: 0, samples_none: 0, strict_outcome_oracle, } @@ -151,6 +155,42 @@ impl Workload { self.resends } + /// Total client evictions survived. + #[must_use] + pub const fn evictions(&self) -> u64 { + self.evictions + } + + /// Forget everything outstanding for a client the cluster evicted. + /// + /// An eviction is session-terminal: the server refused the request BEFORE + /// commit (`Eviction(NoSession)` from an unbound transport, which is what a + /// replica restart leaves behind, since session bindings live in the + /// per-connection `SessionManager` and do not survive it). The client logs in + /// again and carries on. + /// + /// Resending is not an option: the retained message carries the old session + /// id, so it would be refused again. A fresh sample after the re-login + /// carries the new one. + /// + /// The forgotten request's fate is genuinely unknown, which is why this + /// disarms the strict outcome oracle. The refusal proves only that the + /// ATTEMPT that drew it did not commit, and that attempt may have been a + /// resend of a request whose original had already committed with its reply + /// lost. The shadow is then missing an effect that did happen, and every + /// later outcome targeted against it can disagree with what commits. Losing + /// the oracle for the rest of the run is the honest price; claiming the + /// shadow is still authoritative would turn a known unknown into a spurious + /// failure. + /// + /// Returns how many requests were forgotten. + pub fn forget_evicted_client(&mut self, client_id: u128) -> usize { + self.evictions += 1; + self.strict_outcome_oracle = false; + self.outstanding.retain(|&(owner, _), _| owner != client_id); + self.auditor.forget_client(client_id) + } + /// Requests whose reply has not arrived within /// [`WorkloadOptions::request_timeout_ticks`], each paired with the replica /// to retry it against. Callers must submit every returned message. @@ -541,7 +581,7 @@ pub fn run_with_faults( apply_sim_commands(sim, &cmds); replies_seen += 1; } - assert_no_evictions(sim); + recover_evicted_clients(sim, workload, clients); invariants.check(sim, workload); if replies_seen >= replies_target { break; @@ -688,30 +728,41 @@ impl FaultInjector { } } -/// Fail loudly if the cluster evicted a client, which the workload cannot yet -/// survive. +/// Log any evicted client back in, which is what a real client does. /// -/// An eviction ends the session: the client's outstanding requests become -/// unanswerable and it must log in again before submitting anything. Modelling -/// that means re-establishing the session mid-run and discarding the auditor's -/// expectations for it, which the driver does not do. Until it does, an eviction -/// presents as a client that has silently stopped making progress, so name it -/// here rather than let the run time out with no explanation. +/// A replica restart drops its `SessionManager`, since bindings are +/// per-connection and volatile, so a client that had a session there is unbound +/// and its next replicated request is refused with `Eviction(NoSession)`. The +/// client table itself is replicated metadata and survives, so the re-login +/// rebinds the existing entry (bumping its fence epoch) and request numbering +/// continues rather than restarting. /// -/// Only reachable through the dispatch shell, and in practice only once crashes -/// and restarts are also in play. +/// Outstanding requests are forgotten rather than resent: the refused request +/// never committed, and the message the retry buffer holds carries the old +/// session id, so resending it would only be refused again. See +/// [`Workload::forget_evicted_client`]. /// /// # Panics -/// If any client was evicted since the last step. -fn assert_no_evictions(sim: &mut Simulator) { - let evicted = sim.take_evictions(); - assert!( - evicted.is_empty(), - "cluster evicted client(s) {evicted:?}: their sessions are gone, so their \ - outstanding requests can never be answered and their next request is \ - refused. The workload does not re-establish a session, so the run cannot \ - continue" - ); +/// If an evicted client id is not one the driver knows about, which would mean +/// the simulator and the driver disagree about who is connected. +fn recover_evicted_clients(sim: &mut Simulator, workload: &mut Workload, clients: &[SimClient]) { + for client_id in sim.take_evictions() { + let client = clients + .iter() + .find(|client| client.client_id() == client_id) + .unwrap_or_else(|| { + panic!("cluster evicted unknown client {client_id}: not one the driver drives") + }); + workload.forget_evicted_client(client_id); + // Any live replica: a client dialing a backup is a supported path (the + // backup forwards the register), and the default target may itself be the + // replica whose restart caused the eviction, in which case the login just + // times out. + let Some(target) = (0..sim.replica_count).find(|idx| !sim.is_crashed(*idx)) else { + continue; + }; + sim.shell_login_via(client, target); + } } /// Submit every request whose reply is overdue (see [`Workload::due_resends`]). diff --git a/core/simulator/src/workload/oracle.rs b/core/simulator/src/workload/oracle.rs index b1eba8e5d..63daa2d72 100644 --- a/core/simulator/src/workload/oracle.rs +++ b/core/simulator/src/workload/oracle.rs @@ -116,6 +116,14 @@ pub fn drive_to_quiesce(sim: &mut Simulator, workload: &mut Workload, max_ticks: let cmds = workload.on_reply(&reply); apply_sim_commands(sim, &cmds); } + // A resend can land on a transport a restart left unbound, which the + // server refuses with an eviction. No re-login here, unlike the active + // driver: the drain submits nothing new, and the refused request was + // rejected before commit, so forgetting it is what "drained" means for a + // request that can never be answered. + for client_id in sim.take_evictions() { + workload.forget_evicted_client(client_id); + } if workload.total_in_flight() == 0 { drained = true; break;
