This is an automated email from the ASF dual-hosted git repository.

krishvishal pushed a commit to branch sim-workload-faults
in repository https://gitbox.apache.org/repos/asf/iggy.git

commit 1771001c876a4d78c20a9164df61744326f75d74
Author: Krishna Vishal <[email protected]>
AuthorDate: Sat Aug 15 14:45:30 2026 +0530

    feat(simulator): re-establish a session after the cluster evicts a client
    
    Session bindings live in the per-connection `SessionManager` and do not
    survive a replica restart, so a client that had a session there becomes
    unbound and its next replicated request is refused with
    `Eviction(NoSession)`. The driver treated that as unrecoverable and
    stopped the run, which made the dispatch shell and crash/restart mutually
    exclusive: every combination of the two failed.
    
    A real client reconnects and logs in, so the driver does too. The client
    table is replicated metadata and survives the restart, so the re-login
    rebinds the existing entry rather than creating one, and request numbering
    continues instead of restarting. The login rotates targets across retries
    for the same reason the workload's resends do: a register has to reach the
    metadata primary, and retrying one replica is useless once it is
    partitioned or crashed mid-handshake. Fixing that alone took the sweep
    from 4 of 9 to 11 of 12.
    
    Outstanding requests are forgotten rather than resent, since the message
    the retry buffer holds carries the old session id and would be refused
    again. That forgetting disarms the strict outcome oracle for the rest of
    the run. The refusal proves only that the ATTEMPT which drew it did not
    commit, and that attempt may have been a resend whose original had already
    committed with its reply lost; the shadow is then missing an effect that
    really happened. Keeping the oracle armed would turn a known unknown into
    a spurious failure, so it stands down and says why.
    
    Shell with faults and crash/restart now drains and converges 11 of 12
    across four op mixes, where it previously could not run at all. The raw
    path is unchanged at 9 of 9.
---
 core/simulator/src/bin/workload-fuzz.rs |  3 +-
 core/simulator/src/lib.rs               |  7 +++
 core/simulator/src/workload/auditor.rs  | 12 +++++
 core/simulator/src/workload/mod.rs      | 93 +++++++++++++++++++++++++--------
 core/simulator/src/workload/oracle.rs   |  8 +++
 5 files changed, 101 insertions(+), 22 deletions(-)

diff --git a/core/simulator/src/bin/workload-fuzz.rs 
b/core/simulator/src/bin/workload-fuzz.rs
index dd0720cf4..6638cf0d4 100644
--- a/core/simulator/src/bin/workload-fuzz.rs
+++ b/core/simulator/src/bin/workload-fuzz.rs
@@ -526,13 +526,14 @@ fn print_coverage(workload: &Workload) {
     let stats = workload.auditor.stats();
     println!(
         "coverage: replies_seen={} replies_unknown={} committed_rejections={} \
-         samples_none={} resends={} denials={}",
+         samples_none={} resends={} denials={} evictions={}",
         stats.replies_seen,
         stats.replies_unknown,
         stats.committed_rejections,
         workload.samples_none(),
         workload.resends(),
         stats.denials,
+        workload.evictions(),
     );
     for action in Action::iter() {
         let commits = stats.commits(action);
diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs
index 4a0b3a08d..61b8cca2d 100644
--- a/core/simulator/src/lib.rs
+++ b/core/simulator/src/lib.rs
@@ -606,9 +606,16 @@ impl Simulator {
         message: &Message<GenericHeader>,
         label: &str,
     ) -> Message<ReplyHeader> {
+        let mut target = target;
         for step in 0..SETUP_TOTAL_STEPS {
             if step % SETUP_RETRY_STEPS == 0 {
                 self.submit_request(client_id, target, message.deep_copy());
+                // Rotate, exactly as the workload's resend path does. 
Retrying the
+                // same replica forever is enough on a perfect network and 
useless
+                // once one is partitioned or crashed mid-handshake: a 
register has
+                // to reach the metadata primary, and the replica that looked 
live
+                // when this started may be neither reachable nor able to 
forward.
+                target = (target + 1) % self.replica_count.max(1);
             }
             if let Some(reply) = self.step().into_iter().next() {
                 return reply;
diff --git a/core/simulator/src/workload/auditor.rs 
b/core/simulator/src/workload/auditor.rs
index 96cd09258..44fd724d4 100644
--- a/core/simulator/src/workload/auditor.rs
+++ b/core/simulator/src/workload/auditor.rs
@@ -209,6 +209,18 @@ impl ServerAuditor {
         &self.stats
     }
 
+    /// Drop every in-flight expectation for `client`, returning how many went.
+    ///
+    /// For a client the cluster evicted: its session is gone, so nothing it 
had
+    /// outstanding will ever be answered and an expectation left behind would
+    /// wait forever. The requests themselves were refused before commit, so
+    /// forgetting them loses no committed state.
+    pub fn forget_client(&mut self, client: u128) -> usize {
+        let before = self.in_flight.len();
+        self.in_flight.retain(|&(owner, _), _| owner != client);
+        before - self.in_flight.len()
+    }
+
     /// The action of an outstanding request, if one is recorded for `key`.
     /// Diagnostic only: names what a stalled run is waiting on, which the bare
     /// `(client, request)` pair cannot.
diff --git a/core/simulator/src/workload/mod.rs 
b/core/simulator/src/workload/mod.rs
index 9e25f9f82..95ca36c63 100644
--- a/core/simulator/src/workload/mod.rs
+++ b/core/simulator/src/workload/mod.rs
@@ -93,6 +93,9 @@ pub struct Workload {
     now: u64,
     /// Total resends issued, for the run summary.
     resends: u64,
+    /// Client evictions survived, for the run summary. Only the dispatch shell
+    /// produces them, and in practice only once replicas restart under it.
+    evictions: u64,
     /// Debug counter for `sample()` returning `None` (a targeted outcome whose
     /// shadow precondition is unmet). Flags PRNG-trace drift during 
development.
     samples_none: u64,
@@ -121,6 +124,7 @@ impl Workload {
             outstanding: BTreeMap::new(),
             now: 0,
             resends: 0,
+            evictions: 0,
             samples_none: 0,
             strict_outcome_oracle,
         }
@@ -151,6 +155,42 @@ impl Workload {
         self.resends
     }
 
+    /// Total client evictions survived.
+    #[must_use]
+    pub const fn evictions(&self) -> u64 {
+        self.evictions
+    }
+
+    /// Forget everything outstanding for a client the cluster evicted.
+    ///
+    /// An eviction is session-terminal: the server refused the request BEFORE
+    /// commit (`Eviction(NoSession)` from an unbound transport, which is what 
a
+    /// replica restart leaves behind, since session bindings live in the
+    /// per-connection `SessionManager` and do not survive it). The client 
logs in
+    /// again and carries on.
+    ///
+    /// Resending is not an option: the retained message carries the old 
session
+    /// id, so it would be refused again. A fresh sample after the re-login
+    /// carries the new one.
+    ///
+    /// The forgotten request's fate is genuinely unknown, which is why this
+    /// disarms the strict outcome oracle. The refusal proves only that the
+    /// ATTEMPT that drew it did not commit, and that attempt may have been a
+    /// resend of a request whose original had already committed with its reply
+    /// lost. The shadow is then missing an effect that did happen, and every
+    /// later outcome targeted against it can disagree with what commits. 
Losing
+    /// the oracle for the rest of the run is the honest price; claiming the
+    /// shadow is still authoritative would turn a known unknown into a 
spurious
+    /// failure.
+    ///
+    /// Returns how many requests were forgotten.
+    pub fn forget_evicted_client(&mut self, client_id: u128) -> usize {
+        self.evictions += 1;
+        self.strict_outcome_oracle = false;
+        self.outstanding.retain(|&(owner, _), _| owner != client_id);
+        self.auditor.forget_client(client_id)
+    }
+
     /// Requests whose reply has not arrived within
     /// [`WorkloadOptions::request_timeout_ticks`], each paired with the 
replica
     /// to retry it against. Callers must submit every returned message.
@@ -541,7 +581,7 @@ pub fn run_with_faults(
             apply_sim_commands(sim, &cmds);
             replies_seen += 1;
         }
-        assert_no_evictions(sim);
+        recover_evicted_clients(sim, workload, clients);
         invariants.check(sim, workload);
         if replies_seen >= replies_target {
             break;
@@ -688,30 +728,41 @@ impl FaultInjector {
     }
 }
 
-/// Fail loudly if the cluster evicted a client, which the workload cannot yet
-/// survive.
+/// Log any evicted client back in, which is what a real client does.
 ///
-/// An eviction ends the session: the client's outstanding requests become
-/// unanswerable and it must log in again before submitting anything. Modelling
-/// that means re-establishing the session mid-run and discarding the auditor's
-/// expectations for it, which the driver does not do. Until it does, an 
eviction
-/// presents as a client that has silently stopped making progress, so name it
-/// here rather than let the run time out with no explanation.
+/// A replica restart drops its `SessionManager`, since bindings are
+/// per-connection and volatile, so a client that had a session there is 
unbound
+/// and its next replicated request is refused with `Eviction(NoSession)`. The
+/// client table itself is replicated metadata and survives, so the re-login
+/// rebinds the existing entry (bumping its fence epoch) and request numbering
+/// continues rather than restarting.
 ///
-/// Only reachable through the dispatch shell, and in practice only once 
crashes
-/// and restarts are also in play.
+/// Outstanding requests are forgotten rather than resent: the refused request
+/// never committed, and the message the retry buffer holds carries the old
+/// session id, so resending it would only be refused again. See
+/// [`Workload::forget_evicted_client`].
 ///
 /// # Panics
-/// If any client was evicted since the last step.
-fn assert_no_evictions(sim: &mut Simulator) {
-    let evicted = sim.take_evictions();
-    assert!(
-        evicted.is_empty(),
-        "cluster evicted client(s) {evicted:?}: their sessions are gone, so 
their \
-         outstanding requests can never be answered and their next request is \
-         refused. The workload does not re-establish a session, so the run 
cannot \
-         continue"
-    );
+/// If an evicted client id is not one the driver knows about, which would mean
+/// the simulator and the driver disagree about who is connected.
+fn recover_evicted_clients(sim: &mut Simulator, workload: &mut Workload, 
clients: &[SimClient]) {
+    for client_id in sim.take_evictions() {
+        let client = clients
+            .iter()
+            .find(|client| client.client_id() == client_id)
+            .unwrap_or_else(|| {
+                panic!("cluster evicted unknown client {client_id}: not one 
the driver drives")
+            });
+        workload.forget_evicted_client(client_id);
+        // Any live replica: a client dialing a backup is a supported path (the
+        // backup forwards the register), and the default target may itself be 
the
+        // replica whose restart caused the eviction, in which case the login 
just
+        // times out.
+        let Some(target) = (0..sim.replica_count).find(|idx| 
!sim.is_crashed(*idx)) else {
+            continue;
+        };
+        sim.shell_login_via(client, target);
+    }
 }
 
 /// Submit every request whose reply is overdue (see 
[`Workload::due_resends`]).
diff --git a/core/simulator/src/workload/oracle.rs 
b/core/simulator/src/workload/oracle.rs
index b1eba8e5d..63daa2d72 100644
--- a/core/simulator/src/workload/oracle.rs
+++ b/core/simulator/src/workload/oracle.rs
@@ -116,6 +116,14 @@ pub fn drive_to_quiesce(sim: &mut Simulator, workload: 
&mut Workload, max_ticks:
             let cmds = workload.on_reply(&reply);
             apply_sim_commands(sim, &cmds);
         }
+        // A resend can land on a transport a restart left unbound, which the
+        // server refuses with an eviction. No re-login here, unlike the active
+        // driver: the drain submits nothing new, and the refused request was
+        // rejected before commit, so forgetting it is what "drained" means 
for a
+        // request that can never be answered.
+        for client_id in sim.take_evictions() {
+            workload.forget_evicted_client(client_id);
+        }
         if workload.total_in_flight() == 0 {
             drained = true;
             break;

Reply via email to