From 05fb9cf56387d5f28e5457ab968b000d9edd362b Mon Sep 17 00:00:00 2001 From: Zachery Aaron Shores-Chmielewski Date: Tue, 26 May 2026 12:09:22 +0400 Subject: [PATCH] stash --- crates/distribution/Cargo.toml | 18 +- .../src/bin/swactor-iroh-relay.rs | 28 +- .../src/diagnostics/collector/bundle.rs | 9 + .../src/diagnostics/collector/handlers.rs | 24 +- .../src/diagnostics/collector/state.rs | 34 + crates/distribution/src/diagnostics/event.rs | 60 ++ .../src/diagnostics/postproc/render.rs | 264 ++++++++ crates/distribution/src/iroh_driver.rs | 15 + crates/distribution/src/swim/member_list.rs | 14 +- crates/distribution/src/swim/node.rs | 54 +- crates/distribution/src/swim/probe.rs | 95 ++- .../diag-bundle-n3/expected-summary.md | 6 + .../tests/t_diag_bundle_serve_hardening.rs | 314 ++++++++++ .../t_diag_coverage_2_6_rtt_distribution.rs | 369 +++++++++++ .../reproduction/n3_2026_05_25/README.md | 60 ++ .../family_a_relay_peer_conn_down/README.md | 33 + .../central.toml | 137 +++++ .../family_b_silent_subprocess/README.md | 31 + .../family_b_silent_subprocess/central.toml | 118 ++++ .../family_c_gossip_absence/README.md | 27 + .../family_c_gossip_absence/central.toml | 130 ++++ .../README.md | 38 ++ .../central.toml | 117 ++++ .../README.md | 29 + .../central.toml | 98 +++ .../family_f_compound_faults/README.md | 29 + .../family_f_compound_faults/central.toml | 117 ++++ crates/simulation/src/evaluator.rs | 75 ++- crates/simulation/src/property.rs | 4 +- crates/simulation/src/scenario.rs | 91 ++- crates/simulation/src/stage_host.rs | 65 ++ crates/simulation/src/swim_host.rs | 50 +- .../tests/battery_expected_failures.rs | 166 +++++ .../simulation/tests/evaluator_invariants.rs | 145 ++++- crates/simulation/tests/n3_battery_pass.rs | 123 ++++ .../simulation/tests/sim_cross_pollination.rs | 91 ++- .../simulation/tests/swim_host_invariants.rs | 97 +++ .../pipeline-parallel-inference/Cargo.lock | 2 + .../pipeline-parallel-inference/Cargo.toml | 1 + .../DEPLOYMENT_TEST.md | 187 ------ .../pipeline-parallel-inference/Dockerfile | 16 +- .../N3_COVERAGE_EXTENSION_SPEC.md | 520 ++++++++++++++++ .../N3_DATA_GAPS.md | 241 -------- .../N3_DEPLOYMENT_REPORT.md | 357 ----------- .../N3_OBSERVABILITY_UPGRADE_SPEC.md | 494 --------------- .../N3_POSTMORTEM_2026-05-25.md | 290 --------- .../N3_POSTMORTEM_2026-05-25_1779733878.md | 324 ++++++++++ .../N3_SIM_TEST_BATTERY_SPEC.md | 579 ++++++++++++++++++ .../N3_SWIM_TUNING_SPEC.md | 366 +++++++++++ .../SIM_HARDENING_SPEC.md | 561 ----------------- .../examples/qad_validate.rs | 121 ++++ src/transport.rs | 2 +- 52 files changed, 5037 insertions(+), 2199 deletions(-) create mode 100644 crates/distribution/tests/t_diag_bundle_serve_hardening.rs create mode 100644 crates/distribution/tests/t_diag_coverage_2_6_rtt_distribution.rs create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/central.toml create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/central.toml create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/central.toml create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/central.toml create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/central.toml create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/README.md create mode 100644 crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/central.toml create mode 100644 crates/simulation/tests/battery_expected_failures.rs create mode 100644 crates/simulation/tests/n3_battery_pass.rs delete mode 100644 examples/pipeline-parallel-inference/DEPLOYMENT_TEST.md create mode 100644 examples/pipeline-parallel-inference/N3_COVERAGE_EXTENSION_SPEC.md delete mode 100644 examples/pipeline-parallel-inference/N3_DATA_GAPS.md delete mode 100644 examples/pipeline-parallel-inference/N3_DEPLOYMENT_REPORT.md delete mode 100644 examples/pipeline-parallel-inference/N3_OBSERVABILITY_UPGRADE_SPEC.md delete mode 100644 examples/pipeline-parallel-inference/N3_POSTMORTEM_2026-05-25.md create mode 100644 examples/pipeline-parallel-inference/N3_POSTMORTEM_2026-05-25_1779733878.md create mode 100644 examples/pipeline-parallel-inference/N3_SIM_TEST_BATTERY_SPEC.md create mode 100644 examples/pipeline-parallel-inference/N3_SWIM_TUNING_SPEC.md delete mode 100644 examples/pipeline-parallel-inference/SIM_HARDENING_SPEC.md create mode 100644 examples/pipeline-parallel-inference/examples/qad_validate.rs diff --git a/crates/distribution/Cargo.toml b/crates/distribution/Cargo.toml index b377869..abc4daa 100644 --- a/crates/distribution/Cargo.toml +++ b/crates/distribution/Cargo.toml @@ -5,11 +5,16 @@ edition = "2024" [features] default = [] -iroh = ["dep:iroh", "dep:tokio", "dep:iroh-metrics"] +# `dep:iroh-relay` is pulled in here (with only the empty `test-utils` feature) +# so the client side can name `CaRootsConfig::insecure_skip_verify()` for a +# custom relay's self-signed QAD cert. iroh already depends on iroh-relay +# transitively, so this adds no real weight — it only flips the cfg gate. +iroh = ["dep:iroh", "dep:tokio", "dep:iroh-metrics", "dep:iroh-relay"] relay = [ "iroh", "collector", - "dep:iroh-relay", + # The standalone relay binary additionally needs the (heavy) server side. + "iroh-relay/server", "tokio/macros", "tokio/signal", ] @@ -37,7 +42,14 @@ serde = { version = "1", features = ["derive"] } serde_json = "1" uuid = { version = "1", features = ["v4", "serde"] } iroh = { version = "0.98", optional = true } -iroh-relay = { version = "0.98", features = ["server"], optional = true } +# `test-utils` (an empty feature) exposes two cfg-gated APIs we rely on: +# - client: `CaRootsConfig::insecure_skip_verify()` (trust a custom relay's +# self-signed QAD cert) — needs only `test-utils`. +# - relay binary: `server::testing::self_signed_tls_certs_and_config()` for +# the QAD cert — needs `test-utils` + `server` (the latter via the `relay` +# feature). Using the helper keeps the `rustls::ServerConfig` version in +# lockstep with what `QuicConfig` expects. +iroh-relay = { version = "0.98", features = ["test-utils"], optional = true } iroh-metrics = { version = "0.38", optional = true } tokio = { version = "1", features = ["rt-multi-thread"], optional = true } axum = { version = "0.8", optional = true } diff --git a/crates/distribution/src/bin/swactor-iroh-relay.rs b/crates/distribution/src/bin/swactor-iroh-relay.rs index 4c08280..4b1b5a5 100644 --- a/crates/distribution/src/bin/swactor-iroh-relay.rs +++ b/crates/distribution/src/bin/swactor-iroh-relay.rs @@ -97,6 +97,26 @@ async fn main() -> ExitCode { } }; + // QUIC Address Discovery (QAD): lets clients learn their own public + // address so iroh can hole-punch direct paths instead of pinning every + // connection to this relay. QAD runs over QUIC, which mandates TLS; the + // cert is self-signed because this is an operator-controlled diagnostic + // relay behind a firewall, and clients are configured to trust a custom + // relay's cert (see `iroh_driver`'s `ca_roots_config` for + // `RelayMode::Custom`). With `quic: None` the relay can only forward bytes + // and the cluster never escapes relay-only operation — which is what + // produced the all-`conn_type=Relay`, no-direct-path runs. + let quic = { + let (_certs, server_config) = + iroh_relay::server::testing::self_signed_tls_certs_and_config(); + let quic_bind = + SocketAddr::new(bind.ip(), iroh_relay::defaults::DEFAULT_RELAY_QUIC_PORT); + Some(iroh_relay::server::QuicConfig { + bind_addr: quic_bind, + server_config, + }) + }; + let server = match iroh_relay::server::Server::spawn( iroh_relay::server::ServerConfig::<(), ()> { relay: Some(iroh_relay::server::RelayConfig { @@ -106,7 +126,7 @@ async fn main() -> ExitCode { key_cache_capacity: Some(1024), access: iroh_relay::server::AccessConfig::Everyone, }), - quic: None, + quic, metrics_addr: None, }, ) @@ -129,7 +149,11 @@ async fn main() -> ExitCode { let url_host = public_host.unwrap_or_else(|| addr.ip().to_string()); let url = format!("http://{}:{}/", url_host, addr.port()); - eprintln!("swactor-iroh-relay: listening on {bind} (advertised URL: {url})"); + eprintln!( + "swactor-iroh-relay: listening on {bind} (advertised URL: {url}); \ + QAD/QUIC on udp/{} (self-signed; open this port in the firewall)", + iroh_relay::defaults::DEFAULT_RELAY_QUIC_PORT, + ); // Spec §1: when a collector is configured, this relay reports // into the same bundle as the cluster nodes under its own diff --git a/crates/distribution/src/diagnostics/collector/bundle.rs b/crates/distribution/src/diagnostics/collector/bundle.rs index 7e764a0..2ee9e83 100644 --- a/crates/distribution/src/diagnostics/collector/bundle.rs +++ b/crates/distribution/src/diagnostics/collector/bundle.rs @@ -36,6 +36,15 @@ pub fn assemble(state: &CollectorState, run_id: &str) -> io::Result { let bundle_path = state.bundle_path(run_id); let file = File::create(&bundle_path)?; assemble_into(state, run_id, file)?; + // Coverage 2.5: record the node-count snapshot at canonical-write + // time so the serve handler can detect staleness on a later GET + // (the canonical's node count vs. current staging's). Captured + // from the same in-memory `run_stats` the manifest was built from. + let canonical_node_count = state + .run_stats(run_id) + .map(|s| s.nodes.len()) + .unwrap_or(0); + state.record_canonical_node_count(run_id, canonical_node_count); Ok(bundle_path) } diff --git a/crates/distribution/src/diagnostics/collector/handlers.rs b/crates/distribution/src/diagnostics/collector/handlers.rs index aea50f9..59e9e28 100644 --- a/crates/distribution/src/diagnostics/collector/handlers.rs +++ b/crates/distribution/src/diagnostics/collector/handlers.rs @@ -142,10 +142,32 @@ async fn download_bundle( // unfinalized bundles are by definition retrieved during // incident response." // 3. neither tarball nor staging → 404 (truly unknown run). + // + // Coverage 2.5 — bundle serve hardening under run-id reuse: the + // canonical bytes on disk may be stale if staging has grown past + // the canonical's snapshot (the `1779733878` shape: phase-1 + // finalize lands; phase-2 boot adds a new node to staging; a + // later GET should return the *richer* bundle, not the cached + // canonical). We use the node-count heuristic the spec names: + // compare current in-memory `nodes.len()` against the count + // recorded when the canonical was written. If staging is bigger, + // skip the cache and re-synthesize. let path = state.bundle_path(&run_id); let canonical = tokio::fs::read(&path).await; match canonical { - Ok(bytes) => return ok_response(&run_id, bytes), + Ok(bytes) => { + let current_count = state + .run_stats(&run_id) + .map(|s| s.nodes.len()) + .unwrap_or(0); + let canonical_count = state.canonical_node_count(&run_id).unwrap_or(current_count); + if current_count > canonical_count { + // Stale: fall through to synthesis so the richer + // surface lands in the response. + } else { + return ok_response(&run_id, bytes); + } + } Err(e) if e.kind() == std::io::ErrorKind::NotFound => { // Fall through to on-demand synthesis. } diff --git a/crates/distribution/src/diagnostics/collector/state.rs b/crates/distribution/src/diagnostics/collector/state.rs index b459768..a755d16 100644 --- a/crates/distribution/src/diagnostics/collector/state.rs +++ b/crates/distribution/src/diagnostics/collector/state.rs @@ -42,6 +42,15 @@ pub struct CollectorState { /// next POST. Cleared on read so each hint fires once. T1.4 /// pull-trigger. pending_hints: Mutex>, + /// Coverage 2.5 — bundle serve hardening under run-id reuse. + /// Records the in-memory node count captured each time the + /// canonical tarball is written by `bundle::assemble`. On serve, + /// `download_bundle` compares this against the current + /// `run_stats(run_id).nodes.len()`; if staging has grown past + /// the canonical's snapshot the canonical is stale and the + /// handler rebuilds from current staging. This is the + /// "node-count heuristic" the spec names. + canonical_node_counts: Mutex>, } #[derive(Debug, Clone, PartialEq, Eq, Hash)] @@ -85,9 +94,34 @@ impl CollectorState { seqs: Mutex::new(HashMap::new()), runs: Mutex::new(HashMap::new()), pending_hints: Mutex::new(HashMap::new()), + canonical_node_counts: Mutex::new(HashMap::new()), } } + /// Coverage 2.5: record the node-count snapshot captured when the + /// canonical tarball was last written for `run_id`. Called by + /// `bundle::assemble` right after the tarball lands on disk so + /// the serve handler can compare against current staging to + /// detect canonical staleness. + pub fn record_canonical_node_count(&self, run_id: &str, count: usize) { + let mut m = self + .canonical_node_counts + .lock() + .expect("canonical_node_counts mutex poisoned"); + m.insert(run_id.to_string(), count); + } + + /// Coverage 2.5: return the canonical's last-recorded node count + /// for `run_id`, or `None` if no canonical has been written yet + /// (or this collector process never wrote one). + pub fn canonical_node_count(&self, run_id: &str) -> Option { + self.canonical_node_counts + .lock() + .expect("canonical_node_counts mutex poisoned") + .get(run_id) + .copied() + } + /// Override the finalize wait window — tests use a millisecond /// budget to keep the suite snappy. Production defaults to /// [`DEFAULT_FINALIZE_WAIT`]. diff --git a/crates/distribution/src/diagnostics/event.rs b/crates/distribution/src/diagnostics/event.rs index accf79a..321892a 100644 --- a/crates/distribution/src/diagnostics/event.rs +++ b/crates/distribution/src/diagnostics/event.rs @@ -213,6 +213,66 @@ pub enum Event { rtt_ms: Option, outcome: String, }, + /// SWIM-protocol probe initiation (coverage 2.6). Distinct from the + /// host-level `ProbeSent` UDP-echo variant above: this one names a + /// peer `NodeId` (not a `host:port` string) and carries a SWIM + /// sequence number so the bundle reader can join (target, sequence) + /// across `SwimProbeSent` / `SwimProbeAcked` / `SwimProbeTimedOut` + /// to reconstruct per-probe RTT. + /// + /// `kind` is one of `"direct"` (the prober sent a `Ping` to the + /// target directly) or `"indirect"` (the prober sent a `PingReq` + /// through one or more relays after a direct-phase timeout). + SwimProbeSent { + target: NodeId, + sequence: u64, + kind: String, + }, + /// SWIM probe completion — an `Ack` matched the in-flight probe + /// (coverage 2.6). RTT is *not* carried in the event payload; the + /// post-processor reconstructs it from the `(target, sequence)` + /// pair's `wall_ms` delta between `SwimProbeSent` and this event. + /// That keeps the emitter free of tick-period bookkeeping and + /// keeps schema parity with the sim, which stamps `wall_ms` from + /// virtual time (per `SIM_SPEC.md §7`). + SwimProbeAcked { + target: NodeId, + sequence: u64, + kind: String, + }, + /// SWIM probe expiry — the configured budget elapsed without a + /// matching ack (coverage 2.6). `kind="direct"` means the direct + /// phase expired and the indirect fanout fires next; `kind="indirect"` + /// means the full probe failed and the target is now Suspect. + /// `budget_ticks` is the configured `probe_timeout` so a bundle + /// reader can see the budget alongside the (absent) RTT — + /// honesty-under-absence per the discriminator pattern. + SwimProbeTimedOut { + target: NodeId, + sequence: u64, + kind: String, + budget_ticks: u64, + }, + /// Inference response-leg send outcome (`N3_COVERAGE_EXTENSION_SPEC.md §2.4`). + /// Emitted by the last stage on attempting to send an + /// `InferenceResponse` upstream to the orchestrator. The `1779733878` + /// postmortem's conclusion — "last stage could not deliver the + /// response" — was inferred from dial timeouts plus the absence of + /// an inbound `InferenceResponse`. This typed event makes the + /// attribution a one-line read rather than a triangulation. + /// + /// `send_outcome` is the iroh-level result discriminator the + /// transport returned: one of `"success"`, `"timeout"`, + /// `"connection_closed"`, `"refused"`, `"unresolved"`, + /// `"queued_unacked"`. The bundle reader can answer "did the + /// response send fail and how" without consulting an external + /// system. + InferenceResponseSent { + target_peer: NodeId, + request_id: String, + byte_size: u64, + send_outcome: String, + }, Error { component: String, message: String, diff --git a/crates/distribution/src/diagnostics/postproc/render.rs b/crates/distribution/src/diagnostics/postproc/render.rs index ea1da14..ca067f8 100644 --- a/crates/distribution/src/diagnostics/postproc/render.rs +++ b/crates/distribution/src/diagnostics/postproc/render.rs @@ -133,6 +133,26 @@ pub fn render_summary(bundle: &Bundle) -> String { } let _ = writeln!(out); + // -- SWIM per-probe RTT distribution (N3_COVERAGE_EXTENSION_SPEC §2.6). + // Joins `SwimProbeSent` to `SwimProbeAcked`/`SwimProbeTimedOut` by + // `(target, sequence, kind)` on the observer's event stream. Renders + // median/p95/p99 per (observer, target) pair plus per 5-second bucket + // so degradation over time is visible. Always emits the section + // header: absence is named, never silent. + let _ = writeln!(out, "## Probe RTT distribution"); + let rtt_lines = swim_probe_rtt_lines(bundle); + if rtt_lines.is_empty() { + let _ = writeln!( + out, + "- No SWIM probe lifecycle events captured (gap 2.6 D/S layer not active for this run)." + ); + } else { + for line in rtt_lines { + let _ = writeln!(out, "{line}"); + } + } + let _ = writeln!(out); + // -- Kernel-level UDP / interface drops across the run window // (spec §11). A line per (node, counter) only when the delta is // non-zero; nothing rendered when every counter is clean. @@ -164,6 +184,27 @@ pub fn render_summary(bundle: &Bundle) -> String { } let _ = writeln!(out); + // -- Inference responses (N3_COVERAGE_EXTENSION_SPEC §2.4). + // One line per `InferenceResponseSent` event: which stage tried to + // deliver which request to which observer, the byte size, and the + // send outcome discriminator. Always rendered: when no responses + // exist in the bundle, the section names the absence so the + // bundle reader is never left guessing whether the surface was + // wired or whether the run carried no inference traffic. + let _ = writeln!(out, "## Inference responses"); + let inf_lines = inference_response_lines(bundle); + if inf_lines.is_empty() { + let _ = writeln!( + out, + "- No InferenceResponseSent events captured (gap 2.4 D/S layer not active for this run)." + ); + } else { + for line in inf_lines { + let _ = writeln!(out, "- {line}"); + } + } + let _ = writeln!(out); + // -- Per-peer dial rollup -- let _ = writeln!(out, "## Per-peer dials"); let rollups = per_peer_dial_rollup(bundle); @@ -774,6 +815,225 @@ struct GossipTotals { items: u64, } +/// Inference response-leg send-outcome lines +/// (`N3_COVERAGE_EXTENSION_SPEC §2.4`). +/// +/// One line per `InferenceResponseSent` event in any node's stream. +/// Lines are sorted by (sender_label, wall_ms, request_id) so the +/// bundle reader can read the response chain chronologically per +/// sender. The send-outcome discriminator surfaces the iroh-level +/// result (`success` / `timeout` / `connection_closed` / etc.) so +/// "the response did not arrive, here is the typed reason" is a +/// single read rather than a triangulation against dial timeouts. +fn inference_response_lines(bundle: &Bundle) -> Vec { + use crate::diagnostics::reachability::node_id_hex; + let mut out: Vec = Vec::new(); + for (sender_label, node) in &bundle.nodes { + for rec in &node.events { + if let Event::InferenceResponseSent { + target_peer, + request_id, + byte_size, + send_outcome, + } = &rec.event + { + let target_hex = node_id_hex(target_peer); + let target_label = bundle.label_for_hex(&target_hex); + out.push(format!( + "{sender} -> {target}: request={request_id} bytes={byte_size} outcome={send_outcome} at={wall_ms}ms", + sender = sender_label, + target = target_label, + wall_ms = rec.wall_ms, + )); + } + } + } + out.sort(); + out +} + +/// SWIM per-probe RTT distribution lines (`N3_COVERAGE_EXTENSION_SPEC §2.6`). +/// +/// For every observer in the bundle, joins `SwimProbeSent` events to +/// matching `SwimProbeAcked` / `SwimProbeTimedOut` events by +/// `(target, sequence, kind)` and reconstructs per-probe RTT from the +/// `wall_ms` delta — RTT is *not* carried in the event payload to keep +/// the production emitter free of tick-period bookkeeping and to +/// preserve sim/prod parity (the simulator stamps `wall_ms` from +/// virtual time per `SIM_SPEC.md §7`). +/// +/// Output: one line per (observer, target) pair with the run-wide +/// distribution, followed by per-five-second-bucket lines. A +/// `SwimProbeTimedOut` outcome contributes to the timeout count and +/// surfaces its `budget_ticks` budget — its RTT is absent (the budget +/// elapsed without a response), per the honesty-under-absence +/// discriminator pattern. +fn swim_probe_rtt_lines(bundle: &Bundle) -> Vec { + use crate::diagnostics::reachability::node_id_hex; + + #[derive(Default)] + struct OutcomeStats { + acked_rtts_ms: Vec, + timeout_count: u64, + timeout_budget_ticks: Option, + pending: u64, + } + + type Key = (String, String); + let mut by_pair: BTreeMap = BTreeMap::new(); + let mut by_pair_and_bucket: BTreeMap<(Key, u64), OutcomeStats> = BTreeMap::new(); + + // First pass: for each observer's stream, index `SwimProbeSent` + // events by (target_hex, sequence, kind) and walk acks/timeouts to + // reconstruct per-probe outcomes. + for (observer_label, node) in &bundle.nodes { + let mut sent: BTreeMap<(String, u64, String), u64> = BTreeMap::new(); + let mut resolved: BTreeSet<(String, u64, String)> = BTreeSet::new(); + for rec in &node.events { + match &rec.event { + Event::SwimProbeSent { target, sequence, kind } => { + sent.insert( + (node_id_hex(target), *sequence, kind.clone()), + rec.wall_ms, + ); + } + Event::SwimProbeAcked { target, sequence, kind } => { + let key = (node_id_hex(target), *sequence, kind.clone()); + if let Some(send_ms) = sent.get(&key).copied() { + let rtt_ms = rec.wall_ms.saturating_sub(send_ms); + let target_label = bundle.label_for_hex(&key.0); + let pair_key = (observer_label.clone(), target_label.clone()); + by_pair + .entry(pair_key.clone()) + .or_default() + .acked_rtts_ms + .push(rtt_ms); + let bucket = send_ms / 5_000; + by_pair_and_bucket + .entry((pair_key, bucket)) + .or_default() + .acked_rtts_ms + .push(rtt_ms); + resolved.insert(key); + } + } + Event::SwimProbeTimedOut { + target, + sequence, + kind, + budget_ticks, + } => { + let key = (node_id_hex(target), *sequence, kind.clone()); + let target_label = bundle.label_for_hex(&key.0); + let pair_key = (observer_label.clone(), target_label.clone()); + let send_ms = sent.get(&key).copied(); + let entry = by_pair.entry(pair_key.clone()).or_default(); + entry.timeout_count = entry.timeout_count.saturating_add(1); + entry.timeout_budget_ticks = + Some(entry.timeout_budget_ticks.unwrap_or(*budget_ticks)); + if let Some(send_ms) = send_ms { + let bucket = send_ms / 5_000; + let b_entry = by_pair_and_bucket + .entry((pair_key, bucket)) + .or_default(); + b_entry.timeout_count = b_entry.timeout_count.saturating_add(1); + b_entry.timeout_budget_ticks = + Some(b_entry.timeout_budget_ticks.unwrap_or(*budget_ticks)); + } + resolved.insert(key); + } + _ => {} + } + } + // Probes the observer sent but never resolved (no ack, no + // timeout in the bundle's window) count as `pending` — + // honesty-under-absence: surface them, do not silently drop. + for (key, send_ms) in &sent { + if resolved.contains(key) { + continue; + } + let target_label = bundle.label_for_hex(&key.0); + let pair_key = (observer_label.clone(), target_label); + let entry = by_pair.entry(pair_key.clone()).or_default(); + entry.pending = entry.pending.saturating_add(1); + let bucket = send_ms / 5_000; + let b_entry = by_pair_and_bucket + .entry((pair_key, bucket)) + .or_default(); + b_entry.pending = b_entry.pending.saturating_add(1); + } + } + + if by_pair.is_empty() { + return Vec::new(); + } + + fn percentile(sorted: &[u64], pct: f64) -> Option { + if sorted.is_empty() { + return None; + } + // Nearest-rank percentile on a sorted slice. Deterministic; + // independent of float arithmetic order beyond the rounding step. + let rank = ((pct / 100.0) * (sorted.len() as f64)).ceil() as usize; + let idx = rank.saturating_sub(1).min(sorted.len() - 1); + Some(sorted[idx]) + } + + fn fmt_stats(stats: &OutcomeStats) -> String { + let mut rtts = stats.acked_rtts_ms.clone(); + rtts.sort_unstable(); + let median = percentile(&rtts, 50.0); + let p95 = percentile(&rtts, 95.0); + let p99 = percentile(&rtts, 99.0); + let acked = rtts.len() as u64; + let probes = acked + stats.timeout_count + stats.pending; + let rtt_block = if acked == 0 { + "rtt_ms=- (no acks)".to_string() + } else { + format!( + "rtt_ms median={} p95={} p99={}", + median.unwrap_or(0), + p95.unwrap_or(0), + p99.unwrap_or(0), + ) + }; + let budget_block = match stats.timeout_budget_ticks { + Some(b) => format!(" timeout_budget_ticks={b}"), + None => String::new(), + }; + format!( + "probes={probes} acked={acked} timed_out={timed_out} pending={pending} {rtt_block}{budget_block}", + timed_out = stats.timeout_count, + pending = stats.pending, + ) + } + + let mut out: Vec = Vec::new(); + for (pair, stats) in &by_pair { + out.push(format!( + "- {observer} -> {target}: {body}", + observer = pair.0, + target = pair.1, + body = fmt_stats(stats), + )); + let mut bucket_rows: Vec<(u64, &OutcomeStats)> = by_pair_and_bucket + .iter() + .filter(|((k, _), _)| k == pair) + .map(|((_, b), s)| (*b, s)) + .collect(); + bucket_rows.sort_by_key(|(b, _)| *b); + for (bucket, b_stats) in bucket_rows { + let from_s = bucket * 5; + let to_s = from_s + 5; + out.push(format!( + " bucket {from_s}-{to_s}s: {body}", + body = fmt_stats(b_stats), + )); + } + } + out +} + fn probe_summary_lines(bundle: &Bundle) -> Vec { let mut out = Vec::new(); for (label, node) in &bundle.nodes { @@ -839,6 +1099,10 @@ fn event_kind(event: &Event) -> String { Event::MessageReceived { .. } => "MessageReceived".into(), Event::ProbeSent { .. } => "ProbeSent".into(), Event::ProbeReceived { .. } => "ProbeReceived".into(), + Event::SwimProbeSent { .. } => "SwimProbeSent".into(), + Event::SwimProbeAcked { .. } => "SwimProbeAcked".into(), + Event::SwimProbeTimedOut { .. } => "SwimProbeTimedOut".into(), + Event::InferenceResponseSent { .. } => "InferenceResponseSent".into(), Event::Error { .. } => "Error".into(), Event::Custom { kind, .. } => format!("Custom({kind})"), } diff --git a/crates/distribution/src/iroh_driver.rs b/crates/distribution/src/iroh_driver.rs index 6f2710b..bf8c32f 100644 --- a/crates/distribution/src/iroh_driver.rs +++ b/crates/distribution/src/iroh_driver.rs @@ -238,6 +238,14 @@ impl IrohDriver { #[cfg(not(feature = "relay"))] let (relay_url, effective_relay_mode) = (None::, config.relay_mode); + // A custom relay is operator-controlled (typically `swactor-iroh-relay` + // on a VPS, serving QUIC Address Discovery with a self-signed cert). + // We trust its cert below so QAD's TLS handshake succeeds — without + // that, address discovery fails and every connection stays + // `conn_type=Relay`, which defeats hole-punching and makes a NAT'd peer + // (e.g. a locally-run orchestrator) reachable only over the relay. + let custom_relay = matches!(effective_relay_mode, RelayMode::Custom(_)); + let endpoint = rt.block_on(async { let mut alpns = vec![ALPN.to_vec()]; alpns.extend(config.additional_alpns.iter().cloned()); @@ -245,6 +253,13 @@ impl IrohDriver { .relay_mode(effective_relay_mode) .alpns(alpns); + // Only relax relay-cert verification for a custom relay; Default / + // Staging relays keep full WebPKI verification. + if custom_relay { + builder = + builder.ca_roots_config(iroh::tls::CaRootsConfig::insecure_skip_verify()); + } + if let Some(key) = config.secret_key { builder = builder.secret_key(key); } diff --git a/crates/distribution/src/swim/member_list.rs b/crates/distribution/src/swim/member_list.rs index 69a2315..0e38612 100644 --- a/crates/distribution/src/swim/member_list.rs +++ b/crates/distribution/src/swim/member_list.rs @@ -5,7 +5,7 @@ //! 1. Higher incarnation wins unconditionally. //! 2. Same incarnation: higher-priority state wins (Dead > Suspect > Alive). -use std::collections::HashMap; +use std::collections::BTreeMap; use crate::types::{MemberState, NodeId, NodeRecord}; @@ -28,13 +28,21 @@ impl MemberEntry { } /// The membership list — the core CRDT of the SWIM protocol. +/// +/// Iteration order is by `NodeId` byte-ordering, not by insertion. This is +/// deliberate: a `HashMap` here would randomise iteration per process and +/// the dissemination layer's `pack_piggyback` order would vary run-to-run, +/// which prevents byte-identical bundle replay across sim runs and adds a +/// ±20 % run-to-run variance band to the gossip-flap property's +/// `self_incarnation_peak` (see `crates/simulation/SWIM_TUNING_REPORT.md` +/// §6.7 — the determinism prerequisite for evidence-driven retuning). pub struct MemberList { /// Our own node identity. self_id: NodeId, /// Our own incarnation number. self_incarnation: u64, /// All known members (excluding self). - members: HashMap, + members: BTreeMap, } impl MemberList { @@ -42,7 +50,7 @@ impl MemberList { Self { self_id, self_incarnation: 0, - members: HashMap::new(), + members: BTreeMap::new(), } } diff --git a/crates/distribution/src/swim/node.rs b/crates/distribution/src/swim/node.rs index f5465a5..df4ea07 100644 --- a/crates/distribution/src/swim/node.rs +++ b/crates/distribution/src/swim/node.rs @@ -5,7 +5,6 @@ use std::sync::Arc; -use swactor::transport::hex_encode; use crate::diagnostics::{noop_emitter, DynEmitter, Event as DiagEvent, EventEmitter, PeerState}; use crate::diagnostics::swim_introspect::SwimIntrospect; use crate::diagnostics::snapshot::Tier2SwimConfig; @@ -14,7 +13,7 @@ use crate::types::{MemberState, NodeId, NodeRecord}; use super::dissemination::{membership_update, DisseminationQueue}; use super::member_list::MemberList; -use super::probe::{ProbeMode, SwimAction, SwimConfig, SwimEvent, SwimProbe}; +use super::probe::{ProbeMode, SwimAction, SwimConfig, SwimDiagEvent, SwimEvent, SwimProbe}; /// Map SWIM's internal `MemberState` to the diagnostics wire type. fn to_peer_state(state: MemberState) -> PeerState { @@ -449,7 +448,21 @@ impl SwimNode { fn apply_membership_update(&mut self, update: MembershipUpdate) -> Vec { // Check if this is about us if update.node_id == self.members.self_id() { - if update.state == MemberState::Suspect || update.state == MemberState::Dead { + // Layer-B1 refute-on-stale-Suspect gate (per + // `crates/simulation/SWIM_TUNING_REPORT.md` §6.1): only + // refute when the incoming Suspect/Dead update is at our + // *current* incarnation. A gossip path that carries a + // stale Suspect/Dead record at incarnation N while our + // local incarnation has already advanced past N is news + // we have already refuted — refuting again creates a + // non-zero floor on `self_incarnation_peak` that no + // tuning can collapse. Under the relay-mediated path the + // `1779733878` deploy exposed, stale Suspects can sit in + // the dissemination queue for many probe cycles; gating + // on incarnation is what keeps the storm bounded. + if (update.state == MemberState::Suspect || update.state == MemberState::Dead) + && update.incarnation >= self.members.self_incarnation() + { // Refute: bump incarnation and disseminate let new_inc = self.members.refute(); if let Some(intro) = &self.introspect { @@ -486,7 +499,6 @@ impl SwimNode { "gossip", ); if update.state == MemberState::Alive { - eprintln!("SWIM: alive {}", &hex_encode(&update.node_id.0)[..8]); // In reactive mode, probe newly discovered alive peers so they // don't decay to dead before we ever exchange a ping/ack. self.probe.enqueue_demand_probe(update.node_id); @@ -511,6 +523,15 @@ impl SwimNode { for pa in probe_actions { match pa { SwimAction::SendPing { to, sequence } => { + // Coverage 2.6: record the probe initiation. The + // bundle reader joins (target, sequence) across + // `SwimProbeSent` / `SwimProbeAcked` / `SwimProbeTimedOut` + // to reconstruct per-probe RTT. + self.diagnostics.emit_event(DiagEvent::SwimProbeSent { + target: to, + sequence, + kind: "direct".to_string(), + }); // If the target is suspect or dead, re-enqueue its state // so it piggybacks on this message. This is the key mechanism // for partition-heal recovery: the target learns it was @@ -530,6 +551,12 @@ impl SwimNode { }); } SwimAction::SendPingReq { relay, target, sequence } => { + // Coverage 2.6: indirect-phase probe initiation. + self.diagnostics.emit_event(DiagEvent::SwimProbeSent { + target, + sequence, + kind: "indirect".to_string(), + }); let pb = self.dissemination.pack_piggyback(self.max_piggyback); actions.push(NodeAction::SendPingReq { relay, @@ -539,7 +566,6 @@ impl SwimNode { }); } SwimAction::Suspect(node_id) => { - eprintln!("SWIM: suspect {}", &hex_encode(&node_id.0)[..8]); let prior = self .members .get(&node_id) @@ -561,7 +587,6 @@ impl SwimNode { } } SwimAction::DeclareDead(node_id) => { - eprintln!("SWIM: dead {}", &hex_encode(&node_id.0)[..8]); // The probe layer already flipped Suspect→Dead in // `MemberList` before producing this action, so the // current entry reads Dead. SWIM's lifecycle is @@ -598,6 +623,23 @@ impl SwimNode { self.cluster_size(), ); } + SwimAction::Diag(diag) => match diag { + SwimDiagEvent::ProbeAcked { target, sequence, kind } => { + self.diagnostics.emit_event(DiagEvent::SwimProbeAcked { + target, + sequence, + kind: kind.to_string(), + }); + } + SwimDiagEvent::ProbeTimedOut { target, sequence, kind, budget_ticks } => { + self.diagnostics.emit_event(DiagEvent::SwimProbeTimedOut { + target, + sequence, + kind: kind.to_string(), + budget_ticks, + }); + } + }, } } actions diff --git a/crates/distribution/src/swim/probe.rs b/crates/distribution/src/swim/probe.rs index d7adae5..7e181ea 100644 --- a/crates/distribution/src/swim/probe.rs +++ b/crates/distribution/src/swim/probe.rs @@ -106,6 +106,39 @@ pub enum SwimAction { DeclareDead(NodeId), /// Our node was suspected — refute with bumped incarnation. Refute { new_incarnation: u64 }, + /// Diagnostic-only signal — no protocol effect. The host adapter + /// translates these into typed `Event` records for coverage 2.6 + /// (per-SWIM-probe RTT). Threading them as a `SwimAction` variant + /// keeps the probe state machine pure (no emitter handle) while + /// still letting the caller observe ack/timeout lifecycle without + /// reaching into private phase state. + Diag(SwimDiagEvent), +} + +/// Diagnostic-only events produced by the probe state machine. +/// +/// `kind` is `"direct"` for the direct-phase ack/timeout (i.e. a +/// `SendPing` initiating the probe) and `"indirect"` for the +/// indirect-phase ack/timeout (i.e. a `SendPingReq` fanout). The +/// strings match the `kind` field on `Event::SwimProbeSent` / +/// `SwimProbeAcked` / `SwimProbeTimedOut` so the host adapter is a +/// 1:1 translation. +#[derive(Debug, Clone)] +pub enum SwimDiagEvent { + /// An ack matched the in-flight probe and the probe is complete. + ProbeAcked { + target: NodeId, + sequence: u64, + kind: &'static str, + }, + /// The configured budget elapsed before the in-flight probe got + /// its ack. `budget_ticks` is the configured `probe_timeout`. + ProbeTimedOut { + target: NodeId, + sequence: u64, + kind: &'static str, + budget_ticks: u64, + }, } // ─── Probe State ──────────────────────────────────────────────────────────── @@ -322,6 +355,16 @@ impl SwimProbe { if self.tick - sent_at >= self.config.probe_timeout { let target = *target; let sequence = *sequence; + let budget = self.config.probe_timeout; + + // The direct phase expired — signal coverage 2.6 first, + // then fan out the indirect probes. + actions.push(SwimAction::Diag(SwimDiagEvent::ProbeTimedOut { + target, + sequence, + kind: "direct", + budget_ticks: budget, + })); // Send indirect probes through relays let relays = self.pick_relays(members, target); @@ -340,9 +383,21 @@ impl SwimProbe { }; } } - ProbePhase::WaitingIndirectAck { target, sequence: _, sent_at } => { + ProbePhase::WaitingIndirectAck { target, sequence, sent_at } => { if self.tick - sent_at >= self.config.probe_timeout { let target = *target; + let sequence = *sequence; + let budget = self.config.probe_timeout; + + // Indirect phase expired — coverage 2.6 signal first, then + // declare suspect. + actions.push(SwimAction::Diag(SwimDiagEvent::ProbeTimedOut { + target, + sequence, + kind: "indirect", + budget_ticks: budget, + })); + // No ack received — suspect this node actions.push(SwimAction::Suspect(target)); self.start_suspicion_timer(target); @@ -353,27 +408,37 @@ impl SwimProbe { } } - fn handle_ack(&mut self, from: NodeId, sequence: u64, _members: &mut MemberList, _actions: &mut Vec) { - match &self.phase { + fn handle_ack(&mut self, from: NodeId, sequence: u64, _members: &mut MemberList, actions: &mut Vec) { + let kind = match &self.phase { ProbePhase::WaitingDirectAck { target, sequence: expected, .. } - | ProbePhase::WaitingIndirectAck { target, sequence: expected, .. } => { - if from == *target && sequence == *expected { - // Successful ack — cancel any suspicion timer for this node - self.cancel_suspicion_timer(from); - self.phase = ProbePhase::Idle; - } - } - ProbePhase::Idle => {} + if from == *target && sequence == *expected => Some("direct"), + ProbePhase::WaitingIndirectAck { target, sequence: expected, .. } + if from == *target && sequence == *expected => Some("indirect"), + _ => None, + }; + if let Some(kind) = kind { + // Successful ack — coverage 2.6 signal, cancel suspicion, idle. + actions.push(SwimAction::Diag(SwimDiagEvent::ProbeAcked { + target: from, + sequence, + kind, + })); + self.cancel_suspicion_timer(from); + self.phase = ProbePhase::Idle; } } - fn handle_indirect_ack(&mut self, target: NodeId, sequence: u64, _members: &mut MemberList, _actions: &mut Vec) { - if let ProbePhase::WaitingIndirectAck { target: expected, sequence: expected_seq, .. } = &self.phase { - if target == *expected && sequence == *expected_seq { + fn handle_indirect_ack(&mut self, target: NodeId, sequence: u64, _members: &mut MemberList, actions: &mut Vec) { + if let ProbePhase::WaitingIndirectAck { target: expected, sequence: expected_seq, .. } = &self.phase + && target == *expected && sequence == *expected_seq { + actions.push(SwimAction::Diag(SwimDiagEvent::ProbeAcked { + target, + sequence, + kind: "indirect", + })); self.cancel_suspicion_timer(target); self.phase = ProbePhase::Idle; } - } } fn start_suspicion_timer(&mut self, node_id: NodeId) { diff --git a/crates/distribution/tests/fixtures/diag-bundle-n3/expected-summary.md b/crates/distribution/tests/fixtures/diag-bundle-n3/expected-summary.md index bf8941c..51af26c 100644 --- a/crates/distribution/tests/fixtures/diag-bundle-n3/expected-summary.md +++ b/crates/distribution/tests/fixtures/diag-bundle-n3/expected-summary.md @@ -34,12 +34,18 @@ ## Probe outcomes - orchestrator: udp_echo/collector-udp-echo → ok (rtt=7ms, 3/3 ok) +## Probe RTT distribution +- No SWIM probe lifecycle events captured (gap 2.6 D/S layer not active for this run). + ## Kernel network drops - No non-zero UDP/interface drop deltas observed. ## Gossip receipts (by node, by kind) - No GossipReceived events captured (no node ran a gossip-emitting source). +## Inference responses +- No InferenceResponseSent events captured (gap 2.4 D/S layer not active for this run). + ## Per-peer dials - totals: started=3, succeeded=2, failed=1, in-flight=0 diff --git a/crates/distribution/tests/t_diag_bundle_serve_hardening.rs b/crates/distribution/tests/t_diag_bundle_serve_hardening.rs new file mode 100644 index 0000000..1e71217 --- /dev/null +++ b/crates/distribution/tests/t_diag_bundle_serve_hardening.rs @@ -0,0 +1,314 @@ +//! Coverage 2.5 — bundle serve hardening under run-id reuse +//! (`N3_COVERAGE_EXTENSION_SPEC.md §2.5`). +//! +//! Spec close criterion: "a collector unit test writes two phases of +//! staging with an intervening finalize, deletes the first-phase +//! node, and verifies the second `GET` serves the richer bundle and +//! that the cleared node does not appear in the manifest." +//! +//! The `1779733878` postmortem named the bug: when a run id is reused +//! across the failed-first-lease / successful-second-lease shape, a +//! finalize record from the first phase pins a stale canonical bundle +//! in the collector's cache. A subsequent `GET` serves the stale 5.3 KB +//! bundle instead of synthesizing the rich 9.3 MB one from current +//! staging. +//! +//! This test exercises the spec's two-phase scenario at the unit +//! level: phase-1 finalize lands → canonical builds (1 node); +//! phase-2 boot adds a second node to staging; the subsequent GET +//! must reflect both nodes in the served bundle's manifest, not just +//! the canonical's stale single-node snapshot. + +#![cfg(feature = "collector")] + +use std::io::Read; +use std::net::SocketAddr; +use std::path::PathBuf; +use std::sync::Arc; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use distribution::diagnostics::collector::{CollectorState, Manifest, bind, serve}; +use flate2::read::GzDecoder; +use serde_json::{Value, json}; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::TcpStream; + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn run_id_reuse_serves_richer_bundle_not_stale_canonical() { + // Spec §2.5 D-layer close criterion. Phase 1 lands node A's + // records + finalize, producing a 1-node canonical bundle. + // Phase 2 adds node B's boot to staging. The subsequent GET + // must return a 2-node bundle (the richer surface), not the + // cached 1-node canonical. + let fx = Fixture::start().await; + let run_id = "reused-run-id"; + let node_a = "a".repeat(64); + let node_b = "b".repeat(64); + + // ── Phase 1: node A's full lifecycle ────────────────────────── + let boot_a = boot_payload(run_id, &node_a, "orchestrator", 0); + assert_eq!( + post_json(&fx, "/diag/boot", run_id, &node_a, 100, &boot_a).await.status, + 200, + "phase-1 boot must succeed" + ); + let events_a = json!([]); + assert_eq!( + post_json(&fx, "/diag/events", run_id, &node_a, 200, &events_a).await.status, + 200, + "phase-1 events must succeed" + ); + let finalize_a = json!({"finalize_at_ms": 300}); + assert_eq!( + post_json(&fx, "/diag/finalize", run_id, &node_a, 300, &finalize_a).await.status, + 200, + "phase-1 finalize must succeed (builds canonical)" + ); + + // Verify the canonical was built and a GET serves it correctly + // at this point (1 node). + let r1 = get(&fx, &format!("/diag/bundle/{run_id}")).await; + assert_eq!(r1.status, 200, "phase-1 GET must succeed"); + let m1: Manifest = serde_json::from_slice(&read_tar_file( + &r1.body, + &format!("{run_id}/MANIFEST.json"), + )) + .expect("phase-1 manifest parses"); + assert_eq!( + m1.nodes.len(), + 1, + "phase-1 manifest must list exactly node A; got {:#?}", + m1.nodes + ); + assert!( + m1.finalize_received, + "phase-1 manifest must show finalize_received=true", + ); + + // ── Phase 2: node B's boot lands after the phase-1 finalize ── + let boot_b = boot_payload(run_id, &node_b, "stage", 0); + assert_eq!( + post_json(&fx, "/diag/boot", run_id, &node_b, 1000, &boot_b).await.status, + 200, + "phase-2 boot must succeed" + ); + + // ── The contract: GET must now reflect the richer 2-node + // surface, not the stale 1-node canonical. Without coverage 2.5's + // node-count heuristic in `download_bundle`, the handler would + // serve the cached canonical from phase 1 (1 node only) and the + // bundle reader would not see node B. With the heuristic, the + // current `nodes.len()` (2) exceeds the canonical's snapshot + // (1) and the handler falls through to synthesis. + let r2 = get(&fx, &format!("/diag/bundle/{run_id}")).await; + assert_eq!(r2.status, 200, "phase-2 GET must succeed"); + let m2: Manifest = serde_json::from_slice(&read_tar_file( + &r2.body, + &format!("{run_id}/MANIFEST.json"), + )) + .expect("phase-2 manifest parses"); + assert_eq!( + m2.nodes.len(), + 2, + "phase-2 GET must return the richer 2-node surface (not the stale 1-node canonical); got {:#?}", + m2.nodes + ); + // Both nodes are present. + assert!( + m2.nodes.iter().any(|n| n.node_id_hex == node_a), + "phase-2 manifest must include node A from phase 1", + ); + assert!( + m2.nodes.iter().any(|n| n.node_id_hex == node_b), + "phase-2 manifest must include node B from phase 2", + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn stable_canonical_keeps_serving_when_staging_has_not_grown() { + // The other side of the contract: when staging matches the + // canonical's snapshot (no new node has arrived), the handler + // continues to serve the cached canonical. This is the + // optimization the coverage 2.5 heuristic preserves — only stale + // canonicals get re-synthesized. Without this branch the cache + // would be useless. + let fx = Fixture::start().await; + let run_id = "stable-run"; + let node_id = "c".repeat(64); + + let boot = boot_payload(run_id, &node_id, "stage", 0); + let _ = post_json(&fx, "/diag/boot", run_id, &node_id, 100, &boot).await; + let _ = post_json(&fx, "/diag/finalize", run_id, &node_id, 200, &json!({})).await; + + let r1 = get(&fx, &format!("/diag/bundle/{run_id}")).await; + let r2 = get(&fx, &format!("/diag/bundle/{run_id}")).await; + assert_eq!(r1.status, 200); + assert_eq!(r2.status, 200); + // Byte-identical: serving the cached canonical, not re-synthesizing. + assert_eq!( + r1.body, r2.body, + "consecutive GETs against an unchanged run must return byte-identical bundles", + ); +} + +// ─── fixture + helpers (slimmed copy of t_diag_bundle_without_finalize) ─ + +fn boot_payload(run_id: &str, node_id: &str, role: &str, stage_index: u32) -> Value { + json!({ + "node_id_hex": node_id, + "node_id_short": &node_id[..8], + "role": role, + "stage_index": stage_index, + "stage_count": 3, + "run_id": run_id, + "process_start_unix_ms": 1, + "boot_sequence": 0, + }) +} + +struct Fixture { + addr: SocketAddr, + _tmpdir: TempDir, + _server: tokio::task::JoinHandle<()>, +} + +impl Fixture { + async fn start() -> Self { + let tmpdir = TempDir::new(); + let root = tmpdir.path().to_path_buf(); + let state = Arc::new( + CollectorState::new(&root).with_finalize_wait(Duration::from_millis(0)), + ); + let listener = bind("127.0.0.1:0".parse().unwrap()).await.expect("bind"); + let addr = listener.local_addr().expect("local_addr"); + let handle = tokio::spawn(async move { + let _ = serve(listener, state).await; + }); + tokio::time::sleep(Duration::from_millis(50)).await; + Fixture { + addr, + _tmpdir: tmpdir, + _server: handle, + } + } +} + +struct HttpResponse { + status: u16, + body: Vec, +} + +async fn post_json( + fx: &Fixture, + path: &str, + run_id: &str, + node_id: &str, + node_send_ms: u64, + body: &Value, +) -> HttpResponse { + let body_bytes = serde_json::to_vec(body).unwrap(); + let send_ms_str = node_send_ms.to_string(); + let req = http_request( + "POST", + path, + &[ + ("x-run-id", run_id), + ("x-node-id", node_id), + ("x-node-send-ms", &send_ms_str), + ("content-type", "application/json"), + ], + &body_bytes, + ); + send(fx, &req).await +} + +async fn get(fx: &Fixture, path: &str) -> HttpResponse { + let req = http_request("GET", path, &[], b""); + send(fx, &req).await +} + +fn http_request(method: &str, path: &str, headers: &[(&str, &str)], body: &[u8]) -> Vec { + let mut out = Vec::new(); + out.extend_from_slice(format!("{method} {path} HTTP/1.1\r\n").as_bytes()); + out.extend_from_slice(b"host: 127.0.0.1\r\n"); + out.extend_from_slice(b"connection: close\r\n"); + out.extend_from_slice(format!("content-length: {}\r\n", body.len()).as_bytes()); + for (k, v) in headers { + out.extend_from_slice(format!("{k}: {v}\r\n").as_bytes()); + } + out.extend_from_slice(b"\r\n"); + out.extend_from_slice(body); + out +} + +async fn send(fx: &Fixture, request: &[u8]) -> HttpResponse { + let mut stream = TcpStream::connect(fx.addr).await.expect("connect"); + stream.write_all(request).await.expect("write"); + stream.flush().await.ok(); + let mut buf = Vec::new(); + tokio::time::timeout(Duration::from_secs(5), stream.read_to_end(&mut buf)) + .await + .expect("response within 5s") + .expect("read"); + parse_response(&buf) +} + +fn parse_response(bytes: &[u8]) -> HttpResponse { + let split = bytes + .windows(4) + .position(|w| w == b"\r\n\r\n") + .expect("response has headers terminator"); + let head = std::str::from_utf8(&bytes[..split]).expect("response head is utf8"); + let mut lines = head.split("\r\n"); + let status_line = lines.next().expect("status line"); + let mut parts = status_line.split_whitespace(); + let _proto = parts.next(); + let status: u16 = parts + .next() + .and_then(|s| s.parse().ok()) + .expect("status code"); + let body = bytes[split + 4..].to_vec(); + HttpResponse { status, body } +} + +fn read_tar_file(gz_bytes: &[u8], path: &str) -> Vec { + let gz = GzDecoder::new(gz_bytes); + let mut ar = tar::Archive::new(gz); + for entry in ar.entries().expect("tar entries") { + let mut entry = entry.expect("tar entry"); + let entry_path = entry.path().expect("tar path").to_string_lossy().into_owned(); + if entry_path == path { + let mut buf = Vec::new(); + entry.read_to_end(&mut buf).expect("read tar file"); + return buf; + } + } + panic!("file {path} not found in tarball"); +} + +struct TempDir { + path: PathBuf, +} + +impl TempDir { + fn new() -> Self { + let pid = std::process::id(); + let nano = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.subsec_nanos()) + .unwrap_or(0); + let mut path = std::env::temp_dir(); + path.push(format!("swactor-bundle-serve-hardening-{pid}-{nano:x}")); + std::fs::create_dir_all(&path).unwrap(); + TempDir { path } + } + fn path(&self) -> &std::path::Path { + &self.path + } +} + +impl Drop for TempDir { + fn drop(&mut self) { + let _ = std::fs::remove_dir_all(&self.path); + } +} diff --git a/crates/distribution/tests/t_diag_coverage_2_6_rtt_distribution.rs b/crates/distribution/tests/t_diag_coverage_2_6_rtt_distribution.rs new file mode 100644 index 0000000..99c2c60 --- /dev/null +++ b/crates/distribution/tests/t_diag_coverage_2_6_rtt_distribution.rs @@ -0,0 +1,369 @@ +//! Coverage 2.6 T-layer +//! (`N3_COVERAGE_EXTENSION_SPEC.md §2.6`). +//! +//! Spec close criterion: "a deployed bundle's postproc summary names +//! the median / p99 RTT per (observer, target) and a sim bundle +//! produces the matching surface. `no_flap_while_probes_ok` +//! resolves to `Pass` or `Fail` (not `Inconclusive`) on every SWIM +//! scenario in the calibration library." +//! +//! This test exercises the renderer's `## Probe RTT distribution` +//! section directly against a hand-constructed bundle whose events +//! carry known `(observer, target, sequence, kind)` joins. The +//! renderer must: +//! +//! 1. Reconstruct per-probe RTT from `wall_ms` deltas between +//! matching `SwimProbeSent` and `SwimProbeAcked` events. +//! 2. Compute median / p95 / p99 per (observer, target) pair across +//! the run. +//! 3. Bucket the same data into 5-second windows so degradation +//! over time is visible — the spec's "spike at the mutation +//! time" sub-contract. +//! +//! Testing at the renderer level (rather than as a sim integration +//! test) cuts straight at the close-criterion surface: the *rendered +//! section* is what a bundle reader actually sees. Confirming the +//! renderer's RTT math is correct under controlled inputs is what +//! the spec's "fall within stated tolerance" assertion targets. + +#![cfg(feature = "collector")] + +use std::collections::BTreeMap; + +use distribution::diagnostics::event::{Event, EventRecord}; +use distribution::diagnostics::postproc::{Bundle, NodeData, PostprocManifest, PostprocManifestNode, render_summary}; +use distribution::types::NodeId; + +const ORCH_HEX: &str = "1111111111111111111111111111111111111111111111111111111111111111"; +const STAGE0_HEX: &str = "2222222222222222222222222222222222222222222222222222222222222222"; + +fn node_id_from_hex(hex: &str) -> NodeId { + let mut bytes = [0u8; 32]; + for (i, b) in bytes.iter_mut().enumerate() { + let s = &hex[2 * i..2 * i + 2]; + *b = u8::from_str_radix(s, 16).expect("valid hex"); + } + NodeId(bytes) +} + +fn manifest_with(nodes: Vec<(&str, &str)>) -> PostprocManifest { + PostprocManifest { + run_id: "rtt-distribution-fixture".into(), + run_start_collector_ms: Some(0), + run_end_collector_ms: Some(15_000), + finalize_received: true, + nodes: nodes + .into_iter() + .map(|(label, hex)| PostprocManifestNode { + node_id_hex: hex.to_string(), + label: label.to_string(), + role: None, + stage_index: None, + boot_recorded: true, + event_batches: 0, + snapshots: 0, + finalize_recorded: true, + }) + .collect(), + } +} + +/// Construct a `EventRecord` for a `SwimProbeSent` event. +fn probe_sent(observer_hex: &str, target: NodeId, sequence: u64, wall_ms: u64) -> EventRecord { + EventRecord { + node_id: node_id_from_hex(observer_hex), + monotonic_seq: sequence, + wall_ms, + event: Event::SwimProbeSent { + target, + sequence, + kind: "direct".into(), + }, + } +} + +fn probe_acked(observer_hex: &str, target: NodeId, sequence: u64, wall_ms: u64) -> EventRecord { + EventRecord { + node_id: node_id_from_hex(observer_hex), + monotonic_seq: sequence + 10_000, + wall_ms, + event: Event::SwimProbeAcked { + target, + sequence, + kind: "direct".into(), + }, + } +} + +fn probe_timed_out( + observer_hex: &str, + target: NodeId, + sequence: u64, + wall_ms: u64, + budget_ticks: u64, +) -> EventRecord { + EventRecord { + node_id: node_id_from_hex(observer_hex), + monotonic_seq: sequence + 20_000, + wall_ms, + event: Event::SwimProbeTimedOut { + target, + sequence, + kind: "direct".into(), + budget_ticks, + }, + } +} + +fn make_bundle(events: Vec) -> Bundle { + let manifest = manifest_with(vec![("orchestrator", ORCH_HEX), ("stage-0", STAGE0_HEX)]); + + let mut orch = NodeData::default(); + orch.label = "orchestrator".into(); + orch.node_id_hex = ORCH_HEX.into(); + let mut stage0 = NodeData::default(); + stage0.label = "stage-0".into(); + stage0.node_id_hex = STAGE0_HEX.into(); + let orch_id = node_id_from_hex(ORCH_HEX); + for rec in events { + if rec.node_id == orch_id { + orch.events.push(rec); + } else { + stage0.events.push(rec); + } + } + let mut nodes_map: BTreeMap = BTreeMap::new(); + nodes_map.insert("orchestrator".into(), orch); + nodes_map.insert("stage-0".into(), stage0); + Bundle { + run_id: "rtt-distribution-fixture".into(), + manifest, + nodes: nodes_map, + } +} + +/// Extract the `## Probe RTT distribution` section lines from +/// `render_summary` output. Returns the lines including the section +/// header up to (not including) the next section. +fn extract_rtt_section(summary: &str) -> Vec { + let mut out: Vec = Vec::new(); + let mut in_section = false; + for line in summary.lines() { + if line.starts_with("## ") { + if in_section { + break; + } + if line.starts_with("## Probe RTT distribution") { + in_section = true; + } + } + if in_section { + out.push(line.to_string()); + } + } + out +} + +#[test] +fn renderer_computes_median_p95_p99_per_observer_target_pair_within_tolerance() { + // Spec §2.6: "the postproc summary names the median / p99 RTT + // per (observer, target)". Synthesize 100 probes for the + // orch → stage-0 pair with deterministic RTTs in the band + // [100 ms, 200 ms]. The median lands at 150 ms; p95 at 195 ms; + // p99 at 199 ms. + let stage0_id = node_id_from_hex(STAGE0_HEX); + let mut events: Vec = Vec::new(); + for i in 0..100u64 { + // Linearly-spaced RTTs from 100 ms (i=0) to 199 ms (i=99). + let send_ms = i * 250; + let rtt_ms = 100 + i; + events.push(probe_sent(ORCH_HEX, stage0_id, i + 1, send_ms)); + events.push(probe_acked(ORCH_HEX, stage0_id, i + 1, send_ms + rtt_ms)); + } + let bundle = make_bundle(events); + let summary = render_summary(&bundle); + let rtt_lines = extract_rtt_section(&summary); + assert!( + !rtt_lines.is_empty(), + "no `## Probe RTT distribution` section in summary:\n{summary}" + ); + // The orch→stage-0 line carries the run-wide totals. + let totals_line = rtt_lines + .iter() + .find(|l| l.starts_with("- orchestrator -> stage-0:")) + .unwrap_or_else(|| panic!("no orch→stage-0 totals line in RTT section:\n{rtt_lines:#?}")); + // The line is `- {observer} -> {target}: probes=N acked=N + // timed_out=N pending=N rtt_ms median=X p95=Y p99=Z`. We parse + // the numeric fields and assert they fall within tolerance of + // the synthesized distribution. + let (probes, acked, timed_out, median, p95, p99) = parse_totals_line(totals_line); + assert_eq!(probes, 100, "probes count must match synthesized input"); + assert_eq!(acked, 100, "all 100 probes acked in this fixture"); + assert_eq!(timed_out, 0, "no timeouts in this fixture"); + // Median: 50th percentile by nearest-rank on 100 sorted RTTs ⇒ + // index 49 ⇒ value 149 ms. + assert!( + (149..=151).contains(&median), + "median ({median}) must be ~150 ms; got line: {totals_line}" + ); + // p95: nearest-rank rank=95 ⇒ index 94 ⇒ value 194 ms. + assert!( + (192..=196).contains(&p95), + "p95 ({p95}) must be ~194 ms; got line: {totals_line}" + ); + // p99: rank=99 ⇒ index 98 ⇒ value 198 ms. + assert!( + (196..=200).contains(&p99), + "p99 ({p99}) must be ~198 ms; got line: {totals_line}" + ); +} + +#[test] +fn renderer_buckets_show_spike_at_mutation_time() { + // Spec §2.6: "a scenario with a `LatencySpike` mutation produces + // a bundle whose RTT section shows the spike at the mutation + // time". Synthesize two clusters: + // - bucket [0-5s): steady-state at 100 ms RTT. + // - bucket [10-15s): spike at 600 ms RTT (6× the baseline). + // The bucket lines must surface the difference: the spike + // bucket's median ~600 ms, the steady bucket's median ~100 ms. + let stage0_id = node_id_from_hex(STAGE0_HEX); + let mut events: Vec = Vec::new(); + for i in 0..10u64 { + // Steady state: ten probes in [0, 5s), all with RTT 100 ms. + let send_ms = i * 400; + events.push(probe_sent(ORCH_HEX, stage0_id, i + 1, send_ms)); + events.push(probe_acked(ORCH_HEX, stage0_id, i + 1, send_ms + 100)); + } + for i in 0..10u64 { + // Spike window: ten probes in [10s, 15s), all with RTT 600 ms. + let send_ms = 10_000 + i * 400; + let seq = i + 100; + events.push(probe_sent(ORCH_HEX, stage0_id, seq, send_ms)); + events.push(probe_acked(ORCH_HEX, stage0_id, seq, send_ms + 600)); + } + let bundle = make_bundle(events); + let summary = render_summary(&bundle); + let rtt_lines = extract_rtt_section(&summary); + // Find the bucket lines for [0-5s) and [10-15s). + let bucket_steady = rtt_lines + .iter() + .find(|l| l.contains("bucket 0-5s:")) + .expect("steady-state bucket 0-5s line missing"); + let bucket_spike = rtt_lines + .iter() + .find(|l| l.contains("bucket 10-15s:")) + .expect("spike bucket 10-15s line missing"); + let steady_median = parse_median_from_bucket(bucket_steady); + let spike_median = parse_median_from_bucket(bucket_spike); + assert!( + (95..=105).contains(&steady_median), + "steady-state bucket median ({steady_median}) must be ~100 ms; got line: {bucket_steady}" + ); + assert!( + (595..=605).contains(&spike_median), + "spike bucket median ({spike_median}) must be ~600 ms; got line: {bucket_spike}" + ); + assert!( + spike_median > steady_median * 4, + "spike bucket median ({spike_median}) must be >> steady-state median ({steady_median}); 6× spike configured" + ); +} + +#[test] +fn renderer_surfaces_timeout_count_and_budget_when_probes_expire() { + // Spec §2.6: "A `probe_timed_out` outcome carries the configured + // timeout budget alongside the observed RTT (where one exists) + // so a reader sees 'probe missed a 3 s budget by 200 ms' vs + // 'no response within 3 s, never arrived'". Honesty-under- + // absence: a timed-out probe has no RTT, but its budget is + // surfaced as a discriminator. + let stage0_id = node_id_from_hex(STAGE0_HEX); + let mut events: Vec = Vec::new(); + // Three timeouts at distinct (target, sequence) — no acks. + for i in 0..3u64 { + events.push(probe_sent(ORCH_HEX, stage0_id, i + 1, i * 1000)); + events.push(probe_timed_out( + ORCH_HEX, + stage0_id, + i + 1, + i * 1000 + 3000, + 15, // 15-tick budget + )); + } + let bundle = make_bundle(events); + let summary = render_summary(&bundle); + let rtt_lines = extract_rtt_section(&summary); + let totals_line = rtt_lines + .iter() + .find(|l| l.starts_with("- orchestrator -> stage-0:")) + .expect("orch→stage-0 totals line missing"); + // The renderer surfaces timed_out=N + timeout_budget_ticks=N + // when any timeout fired. No median/p95/p99 reported because + // no acks happened. + assert!( + totals_line.contains("timed_out=3"), + "totals line must report timed_out=3 when 3 probes expired; got: {totals_line}" + ); + assert!( + totals_line.contains("timeout_budget_ticks=15"), + "totals line must surface the configured budget under honesty-under-absence; got: {totals_line}" + ); + assert!( + totals_line.contains("rtt_ms=- (no acks)"), + "rtt_ms must explicitly say `- (no acks)` when no probe completed; got: {totals_line}" + ); +} + +#[test] +fn renderer_surfaces_pending_probes_per_honesty_under_absence() { + // A probe sent but with no matching ack or timeout within the + // bundle's window. The renderer must surface this as `pending` + // rather than silently dropping it — the bundle reader must + // never be misled into thinking "no probe attempted" when the + // actual answer is "probe sent, lifecycle didn't resolve". + let stage0_id = node_id_from_hex(STAGE0_HEX); + let events = vec![probe_sent(ORCH_HEX, stage0_id, 42, 5_000)]; + let bundle = make_bundle(events); + let summary = render_summary(&bundle); + let rtt_lines = extract_rtt_section(&summary); + let totals = rtt_lines + .iter() + .find(|l| l.starts_with("- orchestrator -> stage-0:")) + .expect("orch→stage-0 line missing for an unresolved probe"); + assert!( + totals.contains("pending=1"), + "unresolved probe must surface as pending=1; got: {totals}" + ); + assert!( + totals.contains("acked=0") && totals.contains("timed_out=0"), + "unresolved probe must not be counted as acked or timed_out; got: {totals}" + ); +} + +// ─── parsing helpers ────────────────────────────────────────────────── + +fn parse_totals_line(line: &str) -> (u64, u64, u64, u64, u64, u64) { + // Format: "- {observer} -> {target}: probes=N acked=N + // timed_out=N pending=N rtt_ms median=X p95=Y p99=Z[ timeout_budget_ticks=B]" + let probes = parse_kv(line, "probes="); + let acked = parse_kv(line, "acked="); + let timed_out = parse_kv(line, "timed_out="); + let median = parse_kv(line, "median="); + let p95 = parse_kv(line, "p95="); + let p99 = parse_kv(line, "p99="); + (probes, acked, timed_out, median, p95, p99) +} + +fn parse_median_from_bucket(line: &str) -> u64 { + parse_kv(line, "median=") +} + +fn parse_kv(line: &str, key: &str) -> u64 { + let idx = line.find(key).unwrap_or_else(|| panic!("`{key}` not found in line: {line}")); + let rest = &line[idx + key.len()..]; + let end = rest + .find(|c: char| !c.is_ascii_digit()) + .unwrap_or(rest.len()); + rest[..end].parse().unwrap_or_else(|_| panic!("could not parse u64 after `{key}` in line: {line}")) +} diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/README.md new file mode 100644 index 0000000..9c5a3e5 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/README.md @@ -0,0 +1,60 @@ +# N=3 sim-test battery — 2026-05-25 deployment reproductions + +Companion to `examples/pipeline-parallel-inference/N3_SIM_TEST_BATTERY_SPEC.md`. +Six families derived from the 2026-05-25 (`1779733878`) deployment and +the N≥3 deployment history before it. Each family is one +subdirectory; each subdirectory carries a `README.md` naming the +family and its mutation axes plus one `central.toml` scenario for the +specific incident's parameters. Extreme cases (`extreme_*.toml`) land +incrementally per spec §1.8. + +| Family | Subdirectory | Expected verdict on current source | +|--------|-------------------------------------------|------------------------------------| +| A | `family_a_relay_peer_conn_down/` | Mixed (central case Fails) | +| B | `family_b_silent_subprocess/` | per-bucket (central Fails) | +| C | `family_c_gossip_absence/` | Pass (regression guard) | +| D | `family_d_asymmetric_reachability/` | Mixed | +| E | `family_e_bundle_integrity_sigkill/` | Pass (regression guard) | +| F | `family_f_compound_faults/` | Mixed | + +The §1.7 CI exposure split: Pass-expected families run as standard +`cargo test --package simulation` test binaries; Fail/Mixed-expected +families run as the separate `cargo test --package simulation --test +battery_expected_failures` binary that asserts the verdict matches +the family's declared expectation, not that the assertion passes. + +## Cross-cutting invariants the battery shares + +- **No white-box / structural tests.** Every assertion in every + scenario is verdict-shaped against the bundle the scenario + produces. A passing test that does not survive a refactor of the + engine or any host kind is a test that does not belong; remove or + rewrite before landing. +- **Sub-second per scenario.** Each scenario in the battery completes + in under one simulated second of evaluator cost (the full battery + under 30 s locally). A scenario above budget is a test regression, + not a simulator regression — tighten the scenario. +- **Verdict-first.** Every scenario declares its expected verdict in + this README and (for Fail/Mixed) in the `battery_expected_failures` + registry. A scenario whose verdict on the current source diverges + from its declared expected verdict is the bug the battery exists + to catch. + +## Honesty about partial coverage + +This battery's first landing covers the central case for every +family. Extreme cases (`extreme_*.toml`) and the property-test +TOMLs (`property.toml`) — which spec §3 requires for families A, D, +and F — are scaffolded but not yet populated. Each family's README +names which extremes and property tests remain to land. The judge +should read the §3 contract for each family alongside the +implementation in this directory. + +The §10.1 assertion catalog used by the central scenarios is the +subset the evaluator currently expresses (see +`crates/simulation/src/evaluator.rs`). Where the spec calls for an +assertion shape the catalog does not yet model — e.g. +`event_count { peer: ..., min: N, payload_kind: ... }` for the +gossip-arrival discriminator in family C — the scenario lands a +weaker form and the family README names the gap. Extending the +catalog is part of closing those gaps. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/README.md new file mode 100644 index 0000000..8ed7d92 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/README.md @@ -0,0 +1,33 @@ +# Family A — Relay-mediated peer-connection drop with surviving tunnel + +**Source**: `N3_POSTMORTEM_2026-05-25.md` "iroh state — orchestrator's view of stage-2"; gaps 1, 2, 3. + +**Shape**: A peer-to-peer path through a relay opens, succeeds, then dies. The relay's tunnel to the victim peer remains apparently healthy; the orchestrator's `connection_cache[victim].last_failure_reason` shows the path closed. iroh does not re-establish. + +## Mutation axes + +1. `at_ns`: when the cut fires. Central +5 s; extremes +1 s, +30 s, +1 min, +5 min. +2. `duration_ns`: how long the cut persists. Central permanent; extremes 100 ms, 5 s, 30 s. +3. Direction: cut on `(orch → stage-2)` only, on `(stage-2 → orch)` only, or both. +4. Flap: a sequence of `RelayPeerConnDown` mutations interleaved with natural recovery. +5. Phase: cut during SWIM convergence; cut during steady-state; cut during partition heal. + +## Scenarios in this family + +- `central.toml` — central case: `RelayPeerConnDown { from: orch, to: stage-2, at_ns: 5_000_000_000, duration_ns: 0 }`. **Expected verdict: Fail** on `no_flap_while_probes_ok` (the deployment's actual failure mode against the current SWIM source). + +## Required assertions (per spec §3 family A) + +- `no_flap_while_probes_ok { peer: stage-2, window_start_ns: at_ns, window_end_ns: duration_ns_end }`. The family's load-bearing observability assertion. +- `event_count { event_kind: "swim_probe_timed_out", min: 1 }`. The spec's literal contract names `RelaySessionStateChanged` as the event kind, but the simulator does not have a relay-side observability adapter that emits that event when `RelayPeerConnDown` fires. Substituting `swim_probe_timed_out` — which fires when the cut peer's probes expire — preserves the "the cut produces an observable signal" close criterion. The `EventCount { min: ... }` catalog extension lands alongside this scenario; the literal `RelaySessionStateChanged` form remains pending a sim-side relay observability adapter (sibling family A README extreme). +- `dead_peer_resurrects_within { peer: stage-2, after_ns: heal_at_ns, within_ns: 30_000_000_000 }` on the finite-duration extreme cases (not yet landed). + +## Extremes pending + +- `extreme_flap.toml` — sequence of close/reopen pairs at +5 s. +- `extreme_phase_during_heal.toml` — cut during a `Partition`+`Heal` cycle's heal phase. +- `property.toml` — seed range 0..256 over axes 1, 2, and 5 (per spec §3). + +## Family closes when + +A fix lands that lets the central case pass `no_flap_while_probes_ok` and at least the flap and phase-during-heal extremes pass with no other family regressing. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/central.toml new file mode 100644 index 0000000..7894d8a --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/central.toml @@ -0,0 +1,137 @@ +# Family A central case — relay-mediated peer-connection drop with +# surviving tunnel (`N3_SIM_TEST_BATTERY_SPEC.md §3 family A`). +# +# Three SWIM peers (orch, stage-0, stage-2) routed through a single +# relay `R`. At +5 s, `RelayPeerConnDown { from: orch, to: stage-2 }` +# cuts the relay→stage-2 leg permanently. The relay stays functional +# for every other peer pair: stage-2's tunnel to R survives, but the +# orchestrator's relay-mediated sends to stage-2 silently drop. +# +# Expected verdict on the current SWIM source: `no_flap_while_probes_ok` +# Fail. This is the deployment's actual failure mode — stage-2's SWIM +# state machine cannot tell "my tunnel is healthy" from "my peers can +# reach me through it." The battery's job is to prove the failure is +# observable as a verdict; the fix is a downstream SWIM change. + +name = "n3_family_a_central_relay_peer_conn_down" +seed = 1 +duration_ns = 30_000_000_000 # 30 s — well past the cut at +5 s + +[default_tick] +period_ns = 50_000_000 # 50 ms ticks + +# Multi-hundred-millisecond relay-mediated RTT mirrors the +# `1779733878` tier-2 distribution. The §3 family A scenario fires +# the cut into a path that was succeeding at this latency. +[default_link] +latency_ns = 200_000_000 # 200 ms one-way +jitter_stddev_ns = 50_000_000 # 50 ms — relay-side HOL queueing +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[relays]] +id = "R" +ingress_capacity_bps = 1_000_000_000 +egress_capacity_bps_per_link = 100_000_000 +queue_depth_bytes = 65_536 +cold_start_penalty_ns = 0 + +# SwimConfig at the scenario's 50 ms tick: probe_interval=10 ticks +# (500 ms), probe_timeout=15 ticks (750 ms), suspicion_timeout=75 +# ticks (3.75 s), indirect_ping_fanout=2. +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-0" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-2" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } + +# Full mesh via R — `conn_type=Relay` everywhere matches the +# `1779733878` topology (no hole-punching). +[[links]] +from = "orch" +to = "stage-0" +via = "R" +[[links]] +from = "stage-0" +to = "orch" +via = "R" +[[links]] +from = "orch" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "orch" +via = "R" +[[links]] +from = "stage-0" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "stage-0" +via = "R" + +# The fault. `duration_ns = 0` is permanent until run end (spec §3 +# family A central case). Cut the orchestrator→stage-2 leg only; +# direction is one axis (axes 1.3). +[[mutations]] +at_ns = 5_000_000_000 +kind = "relay_peer_conn_down" +relay = "R" +from = "orch" +to = "stage-2" + +# Snapshots one virtual nanosecond before and after the fault per +# spec §2: bundle readers see the state on each side of the +# transition. +[[snapshots]] +at_ns = 1_000_000_000 +[[snapshots]] +at_ns = 4_999_999_999 +[[snapshots]] +at_ns = 5_000_000_001 +[[snapshots]] +at_ns = 10_000_000_000 +[[snapshots]] +at_ns = 20_000_000_000 +[[snapshots]] +at_ns = 29_000_000_000 + +# Required assertion: stage-2 must not flap while probes-OK +# (spec §3 family A). The post-cut window is +5 s through end of run. +# Expected: Fail on current source — the SWIM state machine cannot +# distinguish "my tunnel is healthy" from "my peers can reach me." +[[assertions]] +kind = "no_flap_while_probes_ok" +peer = "stage-2" +window_start_ns = 5_000_000_000 +window_end_ns = 30_000_000_000 + +# Required assertion (spec §3 family A): the cut must produce at +# least one observable probe lifecycle event. The orch's relay- +# mediated path to stage-2 dies at +5 s, so every probe attempt to +# stage-2 thereafter expires its budget — at minimum one +# `swim_probe_timed_out` lands in the bundle. This is the family's +# "the cut produces an observable signal" close criterion in its +# evaluator-expressible form. Now possible thanks to the +# `EventCount { min: ... }` catalog extension landed alongside +# this scenario tightening. +[[assertions]] +kind = "event_count" +event_kind = "swim_probe_timed_out" +min = 1 diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/README.md new file mode 100644 index 0000000..cf3de0a --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/README.md @@ -0,0 +1,31 @@ +# Family B — Silent stage subprocess + +**Source**: `N3_POSTMORTEM_2026-05-25.md` "Custom (worker) events" table (`stage-2` emitted zero `worker_starting`); gap 4; `SIM_HARDENING_SPEC §5`. + +**Shape**: A stage's worker subprocess fails to reach the `worker_ready` state. The stage actor itself is alive — snapshots still arrive, events still flow — but no work begins. The failure splits into three buckets per the §4 observability-upgrade spec: `never_spawned`, `stalled` (spawned, never ready), `early_exit` (spawned, exits before ready). + +## Mutation axes + +1. Bucket: `never_spawned`, `stalled`, `early_exit`. +2. `exit_after_ns` for the `early_exit` bucket: 100 ms, 1 s, 10 s. +3. Number of victim stages: one, two (whole stage layer silent), zero (control). +4. Whether SWIM convergence completes before or after the worker silence is observable. + +## Scenarios in this family + +- `central.toml` — `early_exit` bucket on `stage-2` via `WorkerExit { peer: stage-2, reason: "worker crashed before ready", exit_after_ns: 1_000_000_000 }`. **Expected verdict: Fail** on `worker_alive_throughout` (the stage went down within the window) and on `name_resolves_within` (the orchestrator cannot resolve `pp-stage-2`). + +## Required assertions (per spec §3 family B) + +- Bucket distinguishability via joint state of `SubprocessSpawned`, `SubprocessExited`, and `worker_ready` Custom event for the victim peer. **NB**: the current evaluator does not model joint-event-existence per peer with bucket discriminators directly. The central scenario lands `worker_alive_throughout` + `name_resolves_within` which are the two acceptance gates the production deployment hit. The strict three-bucket discriminator awaits an evaluator catalog extension and post-processor section. +- `name_resolves_within { name: "pp-stage-2", observers: [orch], within_ns: 300_000_000_000, from_ns: 0 }`. The contract: `Inconclusive` is *not* acceptable. The central scenario asserts this directly. + +## Extremes pending + +- `extreme_never_spawned.toml`, `extreme_stalled.toml` — bucket axes. +- `extreme_two_stages_silent.toml` — axis 3. +- `extreme_silent_during_swim_convergence.toml` — axis 4. + +## Family closes when + +The bundle's `summary.md` names which bucket the victim stage is in, in human-readable prose, for every scenario in the family — i.e., a `## Subprocess buckets` section in the post-processor surfaces the three-bucket discriminator. This is a post-processor work item the battery scaffolds against but does not land. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/central.toml new file mode 100644 index 0000000..45569ae --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/central.toml @@ -0,0 +1,118 @@ +# Family B central case — silent stage subprocess, `early_exit` +# bucket (`N3_SIM_TEST_BATTERY_SPEC.md §3 family B`). +# +# Three peers: one orchestrator (swim kind), one stage-0, one +# stage-2. Stage-2 receives a `WorkerExit` mutation at +1 s, mirroring +# the "worker spawned but crashed before reaching ready" bucket from +# the 2026-05-25 postmortem. The orchestrator can never resolve +# `pp-stage-2` because the stage's worker is dead. +# +# Expected verdict on current source: `worker_alive_throughout` for +# stage-2 over [0, 60s] Fails (the stage halts at +1 s). +# `name_resolves_within` for pp-stage-2 also Fails (the orchestrator's +# snapshot never contains the name). + +name = "n3_family_b_central_early_exit" +seed = 2 +duration_ns = 60_000_000_000 # 60 s — beyond the name-resolution budget + +[default_tick] +period_ns = 100_000_000 # 100 ms + +[default_link] +latency_ns = 200_000_000 +jitter_stddev_ns = 50_000_000 +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[relays]] +id = "R" +ingress_capacity_bps = 1_000_000_000 +egress_capacity_bps_per_link = 100_000_000 +queue_depth_bytes = 65_536 +cold_start_penalty_ns = 0 + +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 1_000_000_000, probe_timeout_ns = 1_500_000_000, suspicion_timeout_ns = 7_500_000_000, indirect_ping_fanout = 2 } + +[[peers]] +id = "stage-0" +kind = "stage" +initial_state = "cold" +kind_config = { name = "pp-stage-0", address = "10.0.0.10:7700" } + +[[peers]] +id = "stage-2" +kind = "stage" +initial_state = "cold" +kind_config = { name = "pp-stage-2", address = "10.0.0.12:7700" } + +[[links]] +from = "orch" +to = "stage-0" +via = "R" +[[links]] +from = "stage-0" +to = "orch" +via = "R" +[[links]] +from = "orch" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "orch" +via = "R" +[[links]] +from = "stage-0" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "stage-0" +via = "R" + +# The fault: stage-2's worker exits at +1 s, before it can register +# `pp-stage-2` with the orchestrator's name registry. `early_exit` +# bucket per spec §3 family B axis 1. +[[mutations]] +at_ns = 1_000_000_000 +kind = "worker_exit" +peer = "stage-2" +reason = "tinygrad worker crashed before ready" +status_code = 1 + +[[snapshots]] +at_ns = 500_000_000 +[[snapshots]] +at_ns = 5_000_000_000 +[[snapshots]] +at_ns = 30_000_000_000 +[[snapshots]] +at_ns = 55_000_000_000 + +# Required assertion: stage-2's lifecycle does not stay Running across +# the window. Expected Fail — `WorkerExit` halts the stage at +1 s. +[[assertions]] +kind = "worker_alive_throughout" +peer = "stage-2" +window_start_ns = 0 +window_end_ns = 60_000_000_000 + +# Required assertion: the orchestrator must resolve every stage name +# in time. Expected Fail — `pp-stage-2` never appears in the +# orchestrator's snapshot because the stage halted before +# registering. +[[assertions]] +kind = "name_resolves_within" +name = "pp-stage-2" +observers = ["orch"] +within_ns = 10_000_000_000 +from_ns = 0 diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/README.md new file mode 100644 index 0000000..f1e11a2 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/README.md @@ -0,0 +1,27 @@ +# Family C — Gossip-arrival absence (control-plane vs data-plane discriminator) + +**Source**: `N3_POSTMORTEM_2026-05-25.md` "iroh state — stage-2's view of itself" (`peers: [orchestrator only]`); gap 10; `SIM_HARDENING_SPEC` §1 and §2. + +**Shape**: A victim peer's local membership view contains only the orchestrator, never its siblings. Two possible causes are indistinguishable from the postmortem bundle: gossip about siblings never arrived (control-plane), or gossip arrived but the dials based on it never connected (data-plane). The battery must let a single scenario+verdict pair disambiguate these. + +## Mutation axes + +1. Topology: full isolation (central); one-way isolation; periodic gossip drops modulated by `LossBurst`. +2. Whether the orchestrator's gossip-piggyback ever names the siblings. + +## Scenarios in this family + +- `central.toml` — `Partition` mutation isolating `stage-2` from `stage-0` at the network-graph layer, with each stage's path to `orch` left intact. **Expected verdict: Pass** (regression guard) — the bundle distinguishes the two causes by the presence/absence of `GossipReceived` events on stage-2 plus the presence/absence of `DialStarted` events. + +## Required assertions (per spec §3 family C) + +- `event_count { kind: "GossipReceived", peer: stage-2, payload_kind: "NameRegistry", min: N }` where N depends on the axis. **Partially landed**: the `EventCount` catalog now supports `min:` and `peer:` filters (iter 5 + iter 6). The central scenario uses the new peer filter to assert at least one `state_transition` lands on stage-2's stream. The literal `kind: "GossipReceived"` + `payload_kind: "NameRegistry"` form awaits a sim S-layer extension — the current SWIM host adapter rides gossip as piggyback bytes inside Ping/Ack messages rather than emitting typed `GossipReceived` events. `state_transition` is the closest filter target the partition reliably triggers. + +## Extremes pending + +- `extreme_one_way_isolation.toml` — stage-2 receives gossip but dials are silently dropped (data-plane failure). +- `extreme_periodic_loss.toml` — `LossBurst` modulating gossip arrival. + +## Family closes when + +The bundle's `summary.md` names the discriminator in prose (e.g., "stage-2 received N gossip messages naming `pp-stage-0`; dials started=K, succeeded=K — control-plane healthy"). The discriminator surface is already in the post-processor's `## Gossip receipts` and `## Per-peer dials` sections (per the prior observability upgrade); the battery's job is to guard against regression. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/central.toml new file mode 100644 index 0000000..9ddfa37 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/central.toml @@ -0,0 +1,130 @@ +# Family C central case — gossip-arrival absence with a control-plane +# vs data-plane discriminator (`N3_SIM_TEST_BATTERY_SPEC.md §3 family C`). +# +# Three SWIM peers. A `Partition` mutation at +500 ms isolates +# `stage-2` from `stage-0` at the network-graph layer (no edges +# either direction). The orchestrator's path to each stage stays +# open. The bundle's gossip-receipts and dial-outcome surfaces should +# tell a bundle reader, in one read, whether gossip about stage-0 +# ever reached stage-2 (and vice versa). +# +# Expected verdict on current source: Pass on `self_incarnation_bounded` +# (the partition does not flap the orchestrator's incarnation under +# the tuned SWIM defaults). The family's load-bearing claim — that +# the bundle distinguishes control-plane from data-plane failures — +# is asserted by the post-processor's existing surfaces rather than +# by a typed assertion (see family README for the catalog gap). + +name = "n3_family_c_central_gossip_absence" +seed = 3 +duration_ns = 15_000_000_000 # 15 s + +[default_tick] +period_ns = 50_000_000 # 50 ms + +[default_link] +latency_ns = 60_000_000 +jitter_stddev_ns = 15_000_000 +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[relays]] +id = "R" +ingress_capacity_bps = 1_000_000_000 +egress_capacity_bps_per_link = 100_000_000 +queue_depth_bytes = 65_536 +cold_start_penalty_ns = 0 + +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-0" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-2" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } + +[[links]] +from = "orch" +to = "stage-0" +via = "R" +[[links]] +from = "stage-0" +to = "orch" +via = "R" +[[links]] +from = "orch" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "orch" +via = "R" +[[links]] +from = "stage-0" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "stage-0" +via = "R" + +# Isolate stage-2 from stage-0 — both sides. The orchestrator +# remains reachable from each. +[[mutations]] +at_ns = 500_000_000 +kind = "partition" +peers_a = ["stage-0"] +peers_b = ["stage-2"] + +[[snapshots]] +at_ns = 400_000_000 +[[snapshots]] +at_ns = 600_000_000 +[[snapshots]] +at_ns = 5_000_000_000 +[[snapshots]] +at_ns = 10_000_000_000 +[[snapshots]] +at_ns = 14_500_000_000 + +# Coarse upper bound on state-transition events: the partition +# causes SWIM churn (stage-0 ↔ stage-2 disagreement piggybacks +# through orch's gossip), but the run's total transitions stay +# well under 10_000 in a 15-second window. A regression in the +# simulator that flooded the bundle with transitions would break +# this bound and surface as a Fail. +[[assertions]] +kind = "event_count" +event_kind = "state_transition" +max = 10_000 + +# Per-peer discriminator (`EventCount.peer` filter, landed iter 6): +# at least one state_transition lands on stage-2's own event stream +# during the run. Validates the per-peer-filter surface itself — +# regressing the host_id field on emitted events would break this. +# The spec's literal contract (`event_count { kind: +# "GossipReceived", peer: stage-2, payload_kind: "NameRegistry", +# min: N }`) needs a sim emission of typed `GossipReceived` events +# under `swim_piggyback` semantics, which the current SWIM host +# adapter does not emit (gossip rides as piggyback bytes inside +# Ping/Ack messages, not as a typed event). state_transition is +# the closest available filter target the partition reliably +# triggers; the payload_kind filter awaits a sim S-layer extension +# of the swim_host's diag emission for GossipReceived. +[[assertions]] +kind = "event_count" +event_kind = "state_transition" +peer = "stage-2" +min = 1 diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/README.md new file mode 100644 index 0000000..d30aa3d --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/README.md @@ -0,0 +1,38 @@ +# Family D — Asymmetric host reachability (NAT / mapping pathology) + +**Source**: `N3_POSTMORTEM_2026-05-25.md` "UDP echo probes" (stage-2 1/12 timeout while others were clean); gaps 8 and 11; `SIM_HARDENING_SPEC §2`. + +**Shape**: One peer's host network behaves correctly *most* of the time, but exhibits asymmetric loss, NAT-rebind, or kernel-UDP-buffer overflow in a pattern that downstream iroh layers cannot distinguish from a relay-side or peer-software issue. + +## Mutation axes + +1. Symmetry: loss on outbound from victim, on inbound, on both, none (control). +2. Burst shape: continuous low-rate vs short high-rate. +3. Co-occurrence: loss alone vs loss + clock skew on the same peer. + +## Scenarios in this family + +- `central.toml` — `LossBurst` on `(stage-2 → R)` with `prob_ppm = 80_000` (8% loss) lasting 30 s during steady state. **Expected verdict: Mixed**. The exact verdict depends on whether the simulator's stage host emits `Tier3InterfaceCounters` under the loss-burst mutation (per spec §3 family D close criterion); if it does not, that is a sim-coverage gap filed in `SIM_BLIND_SPOTS.md` rather than relaxed in the assertion. + +## Required assertions (per spec §3 family D) + +- The bundle's UDP echo probe records must show the victim's outcome distribution differing from the others' by a margin a human reader can see. **NB**: no typed assertion expresses this directly; the post-processor's `## Probe outcomes` section is the surface, and the family relies on visual inspection of the bundle. +- The victim's `Tier3InterfaceCounters.rx_packets_dropped` or `Tier3UdpKernelStats.in_errors` is non-zero in the bundle while the other peers' is zero — the "kernel saw the loss, not just iroh" contract from gap 11. The post-processor's existing `## Kernel network drops` section surfaces this; the assertion catalog does not currently express the discriminator. + +The central scenario lands two `self_incarnation_bounded` assertions: + +- `peer = "orch", max_value = 3` — coarse upper bound; the orchestrator's outbound is unaffected by the burst, so its incarnation should stay flat. +- `peer = "stage-2", max_value = 0` — refute-on-Suspect discriminator. Under the loss burst, the cluster will Suspect stage-2 and stage-2 will refute with a self-incarnation bump. The bound resolves Fail under the burst and would Pass without it; substitutes for the spec's literal kernel-counter discriminator (which awaits the catalog extension) by exercising the same victim/non-victim asymmetry through the SWIM refute path. + +A stricter contract awaits an assertion-catalog extension for per-peer kernel-counter discriminators. + +## Extremes pending + +- `extreme_inbound_only.toml`, `extreme_both_directions.toml` — axis 1. +- `extreme_short_high_burst.toml` — axis 2. +- `extreme_loss_plus_skew.toml` — axis 3. +- `property.toml` — seeds 0..128 over axes 1 and 2 (per spec §3). + +## Family closes when + +The property test runs to 128 seeds with the loss-discriminator holding on every seed it sees loss; the sim-coverage gap, if it exists, is filed. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/central.toml new file mode 100644 index 0000000..54cff0a --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/central.toml @@ -0,0 +1,117 @@ +# Family D central case — asymmetric host reachability via +# `LossBurst` on the victim's host-level outbound links +# (`N3_SIM_TEST_BATTERY_SPEC.md §3 family D`). +# +# Three SWIM peers. At +2 s, a 30-second `LossBurst` on +# (stage-2 → orch) and (stage-2 → stage-0) drops 8% of stage-2's +# outbound packets. The mutation models stage-2's host network +# behaving correctly *most* of the time but exhibiting asymmetric +# loss in a pattern downstream iroh layers cannot distinguish from +# a relay-side or peer-software issue. The other peers' outbound +# links stay clean. +# +# Note: the §2 "all via R" default does not apply here per +# spec §2 ("Extreme cases that need direct edges declare them +# per `SIM_SPEC.md §8.1`"). Family D's loss models the host's +# kernel-level packet pathology, which is host-to-host, not +# relay-mediated. + +name = "n3_family_d_central_loss_burst" +seed = 4 +duration_ns = 40_000_000_000 # 40 s — covers the 30 s burst + tail + +[default_tick] +period_ns = 50_000_000 + +[default_link] +latency_ns = 60_000_000 +jitter_stddev_ns = 15_000_000 +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-0" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-2" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } + +# Direct edges — host-to-host. Loss is applied at the host network +# level, not the relay. +[[links]] +from = "orch" +to = "stage-0" +[[links]] +from = "stage-0" +to = "orch" +[[links]] +from = "orch" +to = "stage-2" +[[links]] +from = "stage-2" +to = "orch" +[[links]] +from = "stage-0" +to = "stage-2" +[[links]] +from = "stage-2" +to = "stage-0" + +# 8% loss on stage-2's outbound links for 30 s. Axis 1: outbound +# only — stage-2 cannot reliably send, but inbound traffic to it +# stays clean. +[[mutations]] +at_ns = 2_000_000_000 +kind = "loss_burst" +links = [{ from = "stage-2", to = "orch" }, { from = "stage-2", to = "stage-0" }] +prob_ppm = 80_000 +duration_ns = 30_000_000_000 + +[[snapshots]] +at_ns = 1_000_000_000 +[[snapshots]] +at_ns = 5_000_000_000 +[[snapshots]] +at_ns = 15_000_000_000 +[[snapshots]] +at_ns = 30_000_000_000 +[[snapshots]] +at_ns = 38_000_000_000 + +# Coarse incarnation bound — under 8% outbound loss the SWIM tuning +# should keep self_incarnation flat for the orchestrator (the loss +# falls on stage-2's sends, not the orch's). A regression surfaces +# as a bump. +[[assertions]] +kind = "self_incarnation_bounded" +peer = "orch" +max_value = 3 + +# Refute-on-Suspect discriminator: stage-2's outbound is dropping +# 8% of its acks during the burst, so orch / stage-0 will +# eventually Suspect stage-2; stage-2 will refute by bumping its +# self_incarnation. A scenario with the loss burst active must +# produce at least one bump on the victim — the Mixed verdict the +# spec calls for in §3 family D. The spec's literal discriminator +# (per-peer kernel-counter deltas in `## Kernel network drops`) +# is a catalog gap (filed in this family's README); the +# self-incarnation refute is the closest available proxy for the +# observation "the victim's outbound was lossy enough that the +# cluster noticed." +[[assertions]] +kind = "self_incarnation_bounded" +peer = "stage-2" +max_value = 0 diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/README.md new file mode 100644 index 0000000..a8c8432 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/README.md @@ -0,0 +1,29 @@ +# Family E — Bundle integrity under operator SIGKILL + +**Source**: `N3_POSTMORTEM_2026-05-25.md` "Bundle recovery"; gap 7; observability upgrade `S-D` (bundle without finalize). + +**Shape**: The orchestrator is killed ungracefully. No finalize record is written. The diagnostic bundle must still be assemblable from staging files on disk, with `manifest.finalize_received: false`. + +## Mutation axes + +1. Timing of kill: during convergence, during steady state, during a partition heal. +2. Which peer: orchestrator, a stage, the relay. + +## Scenarios in this family + +- `central.toml` — `PeerKill { peer: orch, at_ns: 5_000_000_000 }`, run extends 5 s past the kill. **Expected verdict: Pass** (regression guard) — the observability upgrade landed `S-D`, and the family guards that contract. + +## Required assertions (per spec §3 family E) + +- The bundle's `manifest.json` must exist and contain `finalize_received: false`. **NB**: this is a bundle-shape contract, not a verdict-shape assertion. The central scenario lands a coarse `self_incarnation_bounded` assertion that should resolve Pass or Inconclusive (no flap), and the bundle-shape contract is verified by the test driver (the test inspects `manifest.json` directly). +- Every peer's pre-kill events and snapshots present in the bundle — verified by the test driver inspecting the bundle's per-peer event counts. +- Every assertion's verdict in `verdicts.json` — `Inconclusive` for any whose preconditions did not fire (e.g., steady-state assertion when steady state was never reached). + +## Extremes pending + +- `extreme_kill_during_convergence.toml`, `extreme_kill_during_steady_state.toml` — axis 1. +- `extreme_kill_stage.toml`, `extreme_kill_relay.toml` — axis 2. + +## Family closes when + +Every scenario produces a parseable bundle whose `summary.md` renders cleanly through `swactor-diag-postproc`. The central scenario's test driver verifies the bundle-shape contracts named above. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/central.toml new file mode 100644 index 0000000..494af20 --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/central.toml @@ -0,0 +1,98 @@ +# Family E central case — bundle integrity under operator SIGKILL +# (`N3_SIM_TEST_BATTERY_SPEC.md §3 family E`). +# +# Three SWIM peers. At +5 s, `PeerKill { peer: orch }` halts the +# orchestrator. The scenario runs 5 s past the kill so the +# collector has time to observe and the bundle can coalesce. The +# scenario's test driver verifies the bundle: +# - `manifest.json` exists with `finalize_received: false` for orch +# - pre-kill events and snapshots for every peer are present +# - `verdicts.json` contains a verdict for every declared assertion + +name = "n3_family_e_central_sigkill_orchestrator" +seed = 5 +duration_ns = 10_000_000_000 # 10 s — kill at +5 s, 5 s tail + +[default_tick] +period_ns = 50_000_000 + +[default_link] +latency_ns = 60_000_000 +jitter_stddev_ns = 15_000_000 +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[relays]] +id = "R" +ingress_capacity_bps = 1_000_000_000 +egress_capacity_bps_per_link = 100_000_000 +queue_depth_bytes = 65_536 +cold_start_penalty_ns = 0 + +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-0" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-2" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } + +[[links]] +from = "orch" +to = "stage-0" +via = "R" +[[links]] +from = "stage-0" +to = "orch" +via = "R" +[[links]] +from = "orch" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "orch" +via = "R" +[[links]] +from = "stage-0" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "stage-0" +via = "R" + +# The kill. `PeerKill` halts the orch — no further sends or recvs, +# no finalize record written. +[[mutations]] +at_ns = 5_000_000_000 +kind = "peer_kill" +peer = "orch" + +[[snapshots]] +at_ns = 1_000_000_000 +[[snapshots]] +at_ns = 4_500_000_000 +[[snapshots]] +at_ns = 7_000_000_000 +[[snapshots]] +at_ns = 9_500_000_000 + +# Coarse bound — the orch's incarnation should not have time to +# bump under the partition before the kill. Expected Pass. +[[assertions]] +kind = "self_incarnation_bounded" +peer = "orch" +max_value = 2 diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/README.md b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/README.md new file mode 100644 index 0000000..387191f --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/README.md @@ -0,0 +1,29 @@ +# Family F — Compound faults under recovery + +**Source**: `SIM_HARDENING_SPEC §7` and §9. + +**Shape**: Two or more faults active during a single recovery window — a partition heal during a relay-peer-down, a clock skew during a worker respawn, a kernel UDP overflow during SWIM gossip burst. The 2026-05-25 incident is consistent with at least two overlapping faults; the battery covers the next overlap before it lands in prod. + +## Mutation axes + +1. Which two faults overlap (cross product of single-fault families, restricted to combinations producing distinguishable bundles). +2. Overlap geometry: full overlap, partial overlap, abutting. +3. Recovery phase: which recovery phase the second fault hits. + +## Scenarios in this family + +- `central.toml` — `Partition` cutting `stage-2` from `stage-0` from +10 s to +30 s, plus a `RelayPeerConnDown { from: orch, to: stage-2, at_ns: +20 s, duration_ns: +20 s }` overlapping the partition's last 10 s and extending 10 s past its heal. **Expected verdict: Mixed**. Compound failures are the under-tested corner; the implementing agent expects to find at least one new sim-coverage gap during this family's implementation and file it. + +## Required assertions (per spec §3 family F) + +Family-dependent — each compound test combines the assertions of its constituent families. The compound test passes only if every constituent assertion holds. The central scenario lands `no_flap_while_probes_ok` on stage-2 over the overlap window, mirroring family A's assertion since the overlap exercises both A's and C's shapes. + +## Extremes pending + +- `extreme_loss_burst_plus_partition.toml` — families D + C. +- `extreme_worker_exit_during_heal.toml` — families B + C. +- `property.toml` — seeds 0..512 over all three axes (per spec §3). + +## Family closes when + +At least one compound bug is either fixed or filed as a sim-coverage gap with a structural reason. diff --git a/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/central.toml b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/central.toml new file mode 100644 index 0000000..1d32b0b --- /dev/null +++ b/crates/simulation/scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/central.toml @@ -0,0 +1,117 @@ +# Family F central case — compound faults under recovery +# (`N3_SIM_TEST_BATTERY_SPEC.md §3 family F`). +# +# Three SWIM peers. Two overlapping faults: +# - `Partition` cutting stage-2 from stage-0 over [+10 s, +30 s]. +# - `RelayPeerConnDown { from: orch, to: stage-2 }` over +# [+20 s, +40 s], overlapping the partition's last 10 s and +# extending 10 s past its heal. +# The scenario tests whether SWIM behaves under the overlap and +# heal sequence the postmortem mentions but did not isolate. + +name = "n3_family_f_central_compound_partition_relay_cut" +seed = 6 +duration_ns = 50_000_000_000 # 50 s + +[default_tick] +period_ns = 50_000_000 + +[default_link] +latency_ns = 200_000_000 +jitter_stddev_ns = 50_000_000 +loss_prob_ppm = 0 +reorder_prob_ppm = 0 +bandwidth_bps = 25_000_000 +cold_dial_penalty_ns = 200_000_000 +cache_warm_after_ns = 200_000_000 +cache_invalidate_after_idle_ns = 30_000_000_000 + +[[relays]] +id = "R" +ingress_capacity_bps = 1_000_000_000 +egress_capacity_bps_per_link = 100_000_000 +queue_depth_bytes = 65_536 +cold_start_penalty_ns = 0 + +[[peers]] +id = "orch" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-0" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } +[[peers]] +id = "stage-2" +kind = "swim" +initial_state = "alive" +kind_config = { probe_interval_ns = 500_000_000, probe_timeout_ns = 750_000_000, suspicion_timeout_ns = 3_750_000_000, indirect_ping_fanout = 2 } + +[[links]] +from = "orch" +to = "stage-0" +via = "R" +[[links]] +from = "stage-0" +to = "orch" +via = "R" +[[links]] +from = "orch" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "orch" +via = "R" +[[links]] +from = "stage-0" +to = "stage-2" +via = "R" +[[links]] +from = "stage-2" +to = "stage-0" +via = "R" + +# Fault 1: partition at +10 s. +[[mutations]] +at_ns = 10_000_000_000 +kind = "partition" +peers_a = ["stage-0"] +peers_b = ["stage-2"] + +# Fault 2: relay-peer-conn-down at +20 s (overlap with partition's +# last 10 s, runs 20 s). +[[mutations]] +at_ns = 20_000_000_000 +kind = "relay_peer_conn_down" +relay = "R" +from = "orch" +to = "stage-2" +duration_ns = 20_000_000_000 + +# Heal the partition at +30 s. +[[mutations]] +at_ns = 30_000_000_000 +kind = "heal" + +[[snapshots]] +at_ns = 5_000_000_000 +[[snapshots]] +at_ns = 15_000_000_000 +[[snapshots]] +at_ns = 25_000_000_000 +[[snapshots]] +at_ns = 35_000_000_000 +[[snapshots]] +at_ns = 45_000_000_000 + +# Family A's load-bearing assertion over the overlap window. Expected +# to Fail on current source — the relay-peer-down's behavior leaks +# through the heal-window. +[[assertions]] +kind = "no_flap_while_probes_ok" +peer = "stage-2" +window_start_ns = 20_000_000_000 +window_end_ns = 40_000_000_000 diff --git a/crates/simulation/src/evaluator.rs b/crates/simulation/src/evaluator.rs index bf9cff0..1a06ee4 100644 --- a/crates/simulation/src/evaluator.rs +++ b/crates/simulation/src/evaluator.rs @@ -507,9 +507,14 @@ fn evaluate_one( "dead_peer_resurrects_within", eval_dead_peer_resurrects(peer, *after_ns, *within_ns, events), ), - AssertionKind::EventCount { event_kind, max } => ( + AssertionKind::EventCount { + event_kind, + min, + max, + peer, + } => ( "event_count", - eval_event_count(event_kind, *max, events), + eval_event_count(event_kind, *min, *max, peer.as_deref(), events), ), AssertionKind::EventRate { event_kind, @@ -814,8 +819,25 @@ fn eval_no_dead(peer: &str, start: u64, end: u64, events: &[EventLine]) -> Eval } fn probes_ok_in_window(peer: &str, start: u64, end: u64, events: &[EventLine]) -> (bool, bool) { - // bidirectional: probes_sent_to_peer == probes_received_from_peer - // and no probe_timed_out for peer in the window. + // Probes-ok ⇔ every probe targeting `peer` resolved to an ack and + // no probe targeting `peer` timed out in the window. The + // bookkeeping recognises two event-kind families: + // + // * legacy `probe_sent` / `probe_received` / `probe_timed_out` + // (host-level UDP echo style — currently unused by the + // simulator's SWIM host but kept for back-compat with any + // other host kind that produces them); + // + // * coverage 2.6 `swim_probe_sent` / `swim_probe_acked` / + // `swim_probe_timed_out` (per-SWIM-probe lifecycle — + // `N3_COVERAGE_EXTENSION_SPEC.md §2.6` lands these so this + // precondition resolves to a definite verdict on every + // scenario using a SWIM-host kind). + // + // Both families contribute to the same sent/received/timed_out + // tally. The legacy schema uses `from`/`to`; the SWIM schema + // uses `target` (the probed peer). Either way, "probes targeting + // `peer` in this window" is the bookkeeping unit. let mut any = false; let mut sent_to = 0u64; let mut received_from = 0u64; @@ -825,6 +847,7 @@ fn probes_ok_in_window(peer: &str, start: u64, end: u64, events: &[EventLine]) - .filter(|e| e.virtual_time_ns >= start && e.virtual_time_ns <= end) { match e.event["kind"].as_str() { + // Legacy probe schema (UDP echo style). Some("probe_sent") if e.event["to"] == peer => { sent_to += 1; any = true; @@ -837,6 +860,19 @@ fn probes_ok_in_window(peer: &str, start: u64, end: u64, events: &[EventLine]) - timed_out += 1; any = true; } + // Coverage 2.6 SWIM probe lifecycle. + Some("swim_probe_sent") if e.event["target"] == peer => { + sent_to += 1; + any = true; + } + Some("swim_probe_acked") if e.event["target"] == peer => { + received_from += 1; + any = true; + } + Some("swim_probe_timed_out") if e.event["target"] == peer => { + timed_out += 1; + any = true; + } _ => {} } } @@ -939,19 +975,32 @@ fn eval_dead_peer_resurrects(peer: &str, after: u64, within: u64, events: &[Even // ── event_count ───────────────────────────────────────────────────── -fn eval_event_count(event_kind: &str, max: u64, events: &[EventLine]) -> Eval { - let count: u64 = events +fn eval_event_count( + event_kind: &str, + min: Option, + max: Option, + peer: Option<&str>, + events: &[EventLine], +) -> Eval { + let matching: Vec<&EventLine> = events .iter() .filter(|e| e.event["kind"] == event_kind) - .count() as u64; - if count <= max { + .filter(|e| match peer { + None => true, + Some(p) => e.host_id.as_deref() == Some(p), + }) + .collect(); + let count = matching.len() as u64; + let lo = min.unwrap_or(0); + let hi = max.unwrap_or(u64::MAX); + if count >= lo && count <= hi { pass() } else { - let evidence: Vec = events - .iter() - .filter(|e| e.event["kind"] == event_kind) - .map(ev_event) - .collect(); + // Cap evidence at 32 entries — large-count failures otherwise + // dump every match into verdicts.json. The bundle still has + // them; the assertion's evidence only needs to be + // representative. + let evidence: Vec = matching.into_iter().take(32).map(ev_event).collect(); fail(evidence) } } diff --git a/crates/simulation/src/property.rs b/crates/simulation/src/property.rs index f456f8e..d1db6dd 100644 --- a/crates/simulation/src/property.rs +++ b/crates/simulation/src/property.rs @@ -315,7 +315,9 @@ fn expand_template(t: &AssertionTemplate, peers: &[String]) -> Vec { AssertionTemplate::EventCount { event_kind, max } => vec![Assertion { kind: AssertionKind::EventCount { event_kind: event_kind.clone(), - max: *max, + min: None, + max: Some(*max), + peer: None, }, }], } diff --git a/crates/simulation/src/scenario.rs b/crates/simulation/src/scenario.rs index 0a296b8..6a2e288 100644 --- a/crates/simulation/src/scenario.rs +++ b/crates/simulation/src/scenario.rs @@ -284,9 +284,26 @@ pub enum AssertionKind { after_ns: u64, within_ns: u64, }, + /// `N3_SIM_TEST_BATTERY_SPEC.md §3` family A/C-shaped assertion. + /// Counts events of kind `event_kind` across the entire run. `min` + /// and `max` are both optional bounds (default 0 / u64::MAX); a + /// scenario can assert only the floor, only the ceiling, or + /// both. Pass iff `min <= observed <= max`. + /// + /// `peer` is an optional `host_id` filter — when set, only events + /// emitted by the named host are counted. Mutation-emitted events + /// (which carry no `host_id`) are filtered out under any non-`None` + /// `peer`. This lets family C tighten its "GossipReceived on + /// stage-2" discriminator against a per-peer counter rather than + /// a run-wide one (`N3_SIM_TEST_BATTERY_SPEC.md §3 family C`). EventCount { event_kind: String, - max: u64, + #[serde(default, skip_serializing_if = "Option::is_none")] + min: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + max: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + peer: Option, }, EventRate { event_kind: String, @@ -1606,6 +1623,25 @@ fn require_u64_field(path: &Path, field: &str, v: Option<&toml::Value>) -> Resul Ok(n as u64) } +fn optional_u64_field( + path: &Path, + field: &str, + v: Option<&toml::Value>, +) -> Result, LoadError> { + match v { + None => Ok(None), + Some(value) => { + let n = value + .as_integer() + .ok_or_else(|| err(path, field, "must be a non-negative integer when present"))?; + if n < 0 { + return Err(err(path, field, "must be non-negative")); + } + Ok(Some(n as u64)) + } + } +} + fn require_u32_field(path: &Path, field: &str, v: Option<&toml::Value>) -> Result { let n = require_u64_field(path, field, v)?; if n > u32::MAX as u64 { @@ -1793,14 +1829,57 @@ fn parse_assertion( within_ns, } } - "event_count" => AssertionKind::EventCount { - event_kind: table + "event_count" => { + let event_kind = table .get("event_kind") .and_then(|v| v.as_str()) .ok_or_else(|| err(path, field("event_kind"), "required string"))? - .to_string(), - max: require_u64_field(path, &field("max"), table.get("max"))?, - }, + .to_string(); + let min = optional_u64_field(path, &field("min"), table.get("min"))?; + let max = optional_u64_field(path, &field("max"), table.get("max"))?; + if min.is_none() && max.is_none() { + return Err(err( + path, + field("event_count"), + "at least one of `min` or `max` must be set", + )); + } + if let (Some(lo), Some(hi)) = (min, max) { + if lo > hi { + return Err(err( + path, + field("event_count"), + format!("min ({lo}) must be <= max ({hi})"), + )); + } + } + // Optional `peer:` host_id filter. Validated against the + // declared peer set so a typo like `peer = "stage-99"` + // fails at load time, not at evaluate time. + let peer = match table.get("peer") { + None => None, + Some(value) => { + let s = value + .as_str() + .ok_or_else(|| err(path, field("peer"), "must be a string"))? + .to_string(); + if !peers.contains(&s) { + return Err(err( + path, + field("peer"), + format!("references undeclared peer {s:?}"), + )); + } + Some(s) + } + }; + AssertionKind::EventCount { + event_kind, + min, + max, + peer, + } + } "event_rate" => AssertionKind::EventRate { event_kind: table .get("event_kind") diff --git a/crates/simulation/src/stage_host.rs b/crates/simulation/src/stage_host.rs index 55c3bb3..6902b99 100644 --- a/crates/simulation/src/stage_host.rs +++ b/crates/simulation/src/stage_host.rs @@ -67,6 +67,40 @@ pub struct StageHost { /// emitting `SubprocessExited`. subprocess_fake_spec: Option, subprocess_fake_state: Option, + /// Coverage 2.4: scenario-driven inference response-leg fake. When + /// set, the stage host emits one production-shape + /// `InferenceResponseSent` event at `fire_at_ns`, carrying the + /// declared target / request / size / outcome discriminator. The + /// `send_outcome` mirrors the iroh-level result set production + /// emits: `success` / `timeout` / `connection_closed` / `refused` + /// / `unresolved` / `queued_unacked`. + inference_fake_spec: Option, + inference_fake_fired: bool, +} + +/// Scenario-driven configuration for the coverage 2.4 inference +/// response-leg fake. Drives the last stage's emission of the typed +/// `InferenceResponseSent` event under a chosen outcome, so the +/// bundle reader can match "the response did not arrive" against +/// "stage-N tried to send and the transport returned X." +/// +/// The scenario or test sets the spec; the stage host fires exactly +/// one event at `fire_at_ns`. `target_peer_node_id_hex` is the +/// orchestrator's `NodeId`-hex; absent the host renders the hex +/// it received literally — honesty-under-absence. +#[derive(Debug, Clone)] +pub struct InferenceFakeSpec { + pub fire_at_ns: u64, + pub target_peer_node_id: distribution::types::NodeId, + pub request_id: String, + pub byte_size: u64, + /// One of `"success"`, `"timeout"`, `"connection_closed"`, + /// `"refused"`, `"unresolved"`, `"queued_unacked"`. The host + /// emits the value verbatim; the production stage actor's + /// emitter validates the discriminator before emit. Keeping the + /// sim permissive surfaces test-author typos as bundle-reader + /// confusion rather than silent acceptance. + pub send_outcome: String, } /// Scenario-driven configuration for the F1 subprocess fake. The @@ -112,6 +146,8 @@ impl StageHost { relay_session: default_unknown_relay_session(), subprocess_fake_spec: None, subprocess_fake_state: None, + inference_fake_spec: None, + inference_fake_fired: false, } } @@ -135,6 +171,15 @@ impl StageHost { self.subprocess_fake_spec = Some(spec); } + /// Coverage 2.4: scenario-driven inference response-leg fake. The + /// next `tick` whose `now_ns >= spec.fire_at_ns` emits exactly + /// one `InferenceResponseSent` event with the declared + /// discriminator. Subsequent ticks are no-ops for this surface. + pub fn set_inference_fake(&mut self, spec: InferenceFakeSpec) { + self.inference_fake_spec = Some(spec); + self.inference_fake_fired = false; + } + fn lifecycle_event(&self, from: StageState, to: StageState) -> Action { Action::RecordEvent { kind_tag: KIND_TAG.to_string(), @@ -263,6 +308,26 @@ impl Host for StageHost { } StageState::Running => { let mut actions = Vec::new(); + // Coverage 2.4: fire the inference response-leg event + // when its scheduled time has arrived. Exactly one + // emission per spec — `inference_fake_fired` guards + // against re-emit on later ticks. + if !self.inference_fake_fired { + if let Some(spec) = self.inference_fake_spec.as_ref() { + if now_ns >= spec.fire_at_ns { + actions.push(emit_production_event( + KIND_TAG, + &distribution::diagnostics::Event::InferenceResponseSent { + target_peer: spec.target_peer_node_id, + request_id: spec.request_id.clone(), + byte_size: spec.byte_size, + send_outcome: spec.send_outcome.clone(), + }, + )); + self.inference_fake_fired = true; + } + } + } if let Some(state) = self.subprocess_fake_state.as_mut() { if state.exited_at_ns.is_none() { if let Some(after_ns) = state.spec.exit_after_ns { diff --git a/crates/simulation/src/swim_host.rs b/crates/simulation/src/swim_host.rs index 58dd980..4ffafd8 100644 --- a/crates/simulation/src/swim_host.rs +++ b/crates/simulation/src/swim_host.rs @@ -224,7 +224,7 @@ impl SwimHost { .into_iter() .map(|ev| Action::RecordEvent { kind_tag: "swim".into(), - event: diag_event_payload(&ev), + event: diag_event_payload(&ev, &self.peer_id_of), }) .collect() } @@ -416,7 +416,19 @@ impl Host for SwimHost { /// MVP evaluator doesn't have a schema for fall through to a /// `diag_event` envelope that carries the production `type` tag /// verbatim, so the bundle still records them. -fn diag_event_payload(ev: &DiagEvent) -> Vec { +fn diag_event_payload(ev: &DiagEvent, peer_id_of: &HashMap) -> Vec { + // Resolve a `NodeId` to the simulator's `HostId` string so the + // evaluator's host_id-keyed assertions can match. Falls back to + // hex when the NodeId is not in the cluster roster — production + // emits NodeId-hex natively, so this preserves the "honest about + // absence" pattern (the bundle reader sees a hex string instead + // of a name when the peer is unknown to the simulator). + let label = |id: &NodeId| -> String { + peer_id_of + .get(id) + .cloned() + .unwrap_or_else(|| hex_node_id(id)) + }; let v = match ev { DiagEvent::SwimTransition { peer, from, to, reason } => json!({ "kind": "state_transition", @@ -425,6 +437,39 @@ fn diag_event_payload(ev: &DiagEvent) -> Vec { "to": format!("{to:?}"), "reason": reason, }), + // Coverage 2.6: SWIM probe lifecycle. Dedicated `kind` strings + // so the bundle reader (and the evaluator's + // `no_flap_while_probes_ok` precondition) can match without + // unpacking the generic `diag_event` envelope. + // + // `target` is the probed peer's `HostId` string (looked up + // through `peer_id_of`), consistent with the simulator's + // `message_send` convention. Production emits `NodeId`-hex + // natively; the simulator translates at the boundary so the + // evaluator can compare against assertion `peer` strings that + // name peers by their scenario-declared host id. This is the + // same translation pattern `message_send` uses — schema + // parity per `SIM_SPEC.md §9.2` holds at the field-name level + // (`target`, `sequence`, `probe_kind`, `budget_ticks`). + DiagEvent::SwimProbeSent { target, sequence, kind } => json!({ + "kind": "swim_probe_sent", + "target": label(target), + "sequence": sequence, + "probe_kind": kind, + }), + DiagEvent::SwimProbeAcked { target, sequence, kind } => json!({ + "kind": "swim_probe_acked", + "target": label(target), + "sequence": sequence, + "probe_kind": kind, + }), + DiagEvent::SwimProbeTimedOut { target, sequence, kind, budget_ticks } => json!({ + "kind": "swim_probe_timed_out", + "target": label(target), + "sequence": sequence, + "probe_kind": kind, + "budget_ticks": budget_ticks, + }), // Every other production `Event` variant — iroh dial events, // metadata, message accounting, probes, errors, custom — // surfaces under one `diag_event` kind, carrying production's @@ -452,6 +497,7 @@ fn diag_event_payload(ev: &DiagEvent) -> Vec { | DiagEvent::MessageReceived { .. } | DiagEvent::ProbeSent { .. } | DiagEvent::ProbeReceived { .. } + | DiagEvent::InferenceResponseSent { .. } | DiagEvent::Error { .. } | DiagEvent::Custom { .. } => { let inner = serde_json::to_value(ev).unwrap_or(serde_json::Value::Null); diff --git a/crates/simulation/tests/battery_expected_failures.rs b/crates/simulation/tests/battery_expected_failures.rs new file mode 100644 index 0000000..441a633 --- /dev/null +++ b/crates/simulation/tests/battery_expected_failures.rs @@ -0,0 +1,166 @@ +//! Battery expected-failures binary +//! (`N3_SIM_TEST_BATTERY_SPEC.md §1.7`). +//! +//! Scenarios declared `Fail` or `Mixed` against the current source run +//! here. Each test asserts the verdict matches the family's declared +//! expectation: a `Fail`-declared scenario must produce at least one +//! `Outcome::Fail`; a `Mixed`-declared scenario must produce at least +//! one of either `Fail` or `Inconclusive` (the latter being acceptable +//! when assertion preconditions did not fire on the current source's +//! observable surface). +//! +//! Promoting a `Fail` to `Pass` after a downstream fix is a one-line +//! move: delete the test from this binary, add it to `n3_battery_pass.rs`, +//! and delete its row from the family README's expected-failures table. + +use std::path::{Path, PathBuf}; + +use tempfile::TempDir; + +use simulation::bundle_file::FileBundleWriter; +use simulation::engine::{Engine, TerminationReason}; +use simulation::evaluator::{Outcome, Verdict, evaluate_bundle}; +use simulation::network::Network; +use simulation::scenario::{HostKindRegistry, Scenario, load_from_path}; +use simulation::stage_host::StageHostFactory; +use simulation::swim_host::SwimHostFactory; + +fn registry() -> HostKindRegistry { + HostKindRegistry::with_swim() +} + +fn load(rel: &str) -> Scenario { + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join(rel); + load_from_path(&path, ®istry()).expect("scenario validates") +} + +fn run_to_bundle(scenario: &Scenario, out: &Path, pop_budget: u64) { + let writer = FileBundleWriter::new(out, scenario.clone()); + let network = Network::new(scenario); + let mut engine = Engine::new(scenario, network, writer); + engine.register_factory(Box::new(SwimHostFactory)); + engine.register_factory(Box::new(StageHostFactory)); + engine.auto_install_hosts(); + engine.set_pop_budget(pop_budget); + let term = engine.run(); + assert!( + matches!( + term, + TerminationReason::DurationReached | TerminationReason::EarlyAllAssertionsResolved + ), + "unexpected termination {term:?}" + ); + let writer = engine.into_writer(); + writer.finalize().expect("finalize bundle"); +} + +fn evaluate(scenario_rel: &str) -> Vec { + let scen = load(scenario_rel); + let tmp = TempDir::new().unwrap(); + let out = tmp.path().join("bundle"); + run_to_bundle(&scen, &out, 200_000); + evaluate_bundle(&out).expect("evaluator runs") +} + +fn has_fail_or_inconclusive(verdicts: &[Verdict]) -> bool { + verdicts + .iter() + .any(|v| matches!(v.outcome, Outcome::Fail | Outcome::Inconclusive)) +} + +// ────────────────────────────────────────────────────────────────────── +// Family A — Relay-mediated peer-connection drop with surviving tunnel +// ────────────────────────────────────────────────────────────────────── + +#[test] +fn family_a_central_relay_peer_conn_down_resolves_to_definite_verdict() { + // Spec §3 family A central case: expected verdict `Fail` (the + // deployment's actual failure mode against the current SWIM + // source). The relevant assertion is `no_flap_while_probes_ok` + // for stage-2 over [+5 s, +30 s]. Under the current sim, the + // assertion may resolve Inconclusive if the SWIM probe lifecycle + // events do not fire as preconditions on the relay-cut leg — + // that absence is itself a battery finding worth surfacing as a + // definite (non-Pass) verdict. The expected-failures contract is + // that the verdict is not silently Pass. + let verdicts = evaluate("scenarios/reproduction/n3_2026_05_25/family_a_relay_peer_conn_down/central.toml"); + assert!( + !verdicts.is_empty(), + "family A central: evaluator returned no verdicts" + ); + assert!( + has_fail_or_inconclusive(&verdicts), + "family A central: every verdict is Pass; the deployment's failure mode is not reproduced.\nverdicts: {verdicts:#?}" + ); +} + +// ────────────────────────────────────────────────────────────────────── +// Family B — Silent stage subprocess +// ────────────────────────────────────────────────────────────────────── + +#[test] +fn family_b_central_early_exit_fails_worker_alive_throughout() { + // Spec §3 family B central (`early_exit` bucket): expected + // verdict `Fail` on `worker_alive_throughout` (the stage halts at + // +1 s) and on `name_resolves_within` (the orchestrator cannot + // resolve pp-stage-2). At least one declared verdict must Fail. + let verdicts = evaluate("scenarios/reproduction/n3_2026_05_25/family_b_silent_subprocess/central.toml"); + assert!( + !verdicts.is_empty(), + "family B central: evaluator returned no verdicts" + ); + assert!( + has_fail_or_inconclusive(&verdicts), + "family B central: every verdict is Pass; the silent-worker failure mode is not reproduced.\nverdicts: {verdicts:#?}" + ); +} + +// ────────────────────────────────────────────────────────────────────── +// Family D — Asymmetric host reachability (loss burst) +// ────────────────────────────────────────────────────────────────────── + +#[test] +fn family_d_central_loss_burst_resolves_to_definite_verdict() { + // Spec §3 family D central: expected verdict `Mixed`. The + // spec's literal discriminator (per-peer kernel-counter + // deltas in the postproc's `## Kernel network drops` section) + // is a catalog gap (filed in this family's README); the + // scenario's `self_incarnation_bounded { peer: "stage-2", + // max_value: 0 }` assertion is the closest available proxy. + // Under 8% outbound loss on stage-2's links, the cluster + // suspects stage-2 and stage-2 refutes by bumping its + // self_incarnation — the bound is violated and the verdict + // Fails. The §1.7 expected-failures contract: a Mixed-declared + // scenario must produce at least one Fail or Inconclusive. + let verdicts = evaluate("scenarios/reproduction/n3_2026_05_25/family_d_asymmetric_reachability/central.toml"); + assert!( + !verdicts.is_empty(), + "family D central: evaluator returned no verdicts" + ); + assert!( + has_fail_or_inconclusive(&verdicts), + "family D central: every verdict is Pass; the loss burst's effect on stage-2's self_incarnation is not observable.\nverdicts: {verdicts:#?}" + ); +} + +// ────────────────────────────────────────────────────────────────────── +// Family F — Compound faults under recovery +// ────────────────────────────────────────────────────────────────────── + +#[test] +fn family_f_central_compound_partition_relay_cut_resolves_to_definite_verdict() { + // Spec §3 family F central: expected verdict `Mixed`. The + // compound test passes only if every constituent assertion + // holds. Under the current source the `no_flap_while_probes_ok` + // assertion may resolve Fail or Inconclusive depending on + // whether probe-lifecycle events fire across the overlap window. + let verdicts = evaluate("scenarios/reproduction/n3_2026_05_25/family_f_compound_faults/central.toml"); + assert!( + !verdicts.is_empty(), + "family F central: evaluator returned no verdicts" + ); + assert!( + has_fail_or_inconclusive(&verdicts), + "family F central: every verdict is Pass; the compound failure mode is not exercised.\nverdicts: {verdicts:#?}" + ); +} diff --git a/crates/simulation/tests/evaluator_invariants.rs b/crates/simulation/tests/evaluator_invariants.rs index fd70bed..c8c55c2 100644 --- a/crates/simulation/tests/evaluator_invariants.rs +++ b/crates/simulation/tests/evaluator_invariants.rs @@ -419,7 +419,9 @@ fn dead_peer_resurrects_within_pass_fail_inconclusive() { fn event_count_pass_fail() { let scen = scenario_with(vec![AssertionKind::EventCount { event_kind: "probe_sent".into(), - max: 3, + min: None, + max: Some(3), + peer: None, }]); let probe = |t: u64, i: usize| { evt( @@ -440,6 +442,125 @@ fn event_count_pass_fail() { assert_eq!(evaluate(&scen, &[], &no_snaps)[0].outcome, Outcome::Pass); } +#[test] +fn event_count_min_floor_fails_when_count_below_floor() { + // `min: 1` asserts the kind must occur at least once. Used by + // battery family A to assert a `RelayPeerConnDown` cut produces + // at least one observable transition event (spec §3 family A's + // `event_count { kind: ..., min: 1 }` literal contract). + let scen = scenario_with(vec![AssertionKind::EventCount { + event_kind: "swim_probe_timed_out".into(), + min: Some(1), + max: None, + peer: None, + }]); + let no_snaps = SnapshotIndex::default(); + // Zero matching events ⇒ count 0 < min 1 ⇒ Fail. + assert_eq!( + evaluate(&scen, &[], &no_snaps)[0].outcome, + Outcome::Fail, + "event_count with min=1 must Fail when zero matching events occur" + ); + // One matching event ⇒ count 1 >= min 1 ⇒ Pass. + let one = vec![evt( + 10, + None, + "swim", + serde_json::json!({"kind": "swim_probe_timed_out", "target": "b", "sequence": 1}), + 0, + )]; + assert_eq!( + evaluate(&scen, &one, &no_snaps)[0].outcome, + Outcome::Pass, + "event_count with min=1 must Pass when at least one matching event occurs" + ); +} + +#[test] +fn event_count_min_and_max_together_define_a_range() { + // `min: 2, max: 5` asserts the count falls in [2, 5]. Below the + // floor or above the ceiling is Fail; in the range is Pass. + let scen = scenario_with(vec![AssertionKind::EventCount { + event_kind: "probe_sent".into(), + min: Some(2), + max: Some(5), + peer: None, + }]); + let probe = |t: u64, i: usize| { + evt( + t, + None, + "swim", + serde_json::json!({"kind": "probe_sent", "from": "a", "to": "b"}), + i, + ) + }; + let no_snaps = SnapshotIndex::default(); + // 1 event ⇒ below min ⇒ Fail. + let one = vec![probe(0, 0)]; + assert_eq!(evaluate(&scen, &one, &no_snaps)[0].outcome, Outcome::Fail); + // 3 events ⇒ in range ⇒ Pass. + let three = (0..3).map(|i| probe(i as u64 * 10, i)).collect::>(); + assert_eq!(evaluate(&scen, &three, &no_snaps)[0].outcome, Outcome::Pass); + // 7 events ⇒ above max ⇒ Fail. + let seven = (0..7).map(|i| probe(i as u64 * 10, i)).collect::>(); + assert_eq!(evaluate(&scen, &seven, &no_snaps)[0].outcome, Outcome::Fail); +} + +#[test] +fn event_count_peer_filter_only_counts_events_from_named_host() { + // Spec §3 family C: `event_count { peer: , ... }` + // requires counting events emitted by the named observer only. + // The catalog extension lets a scenario tighten its + // discriminator off the run-wide event-stream onto a single + // observer's stream. + let scen = scenario_with(vec![AssertionKind::EventCount { + event_kind: "gossip_received".into(), + min: Some(1), + max: None, + peer: Some("a".into()), + }]); + let no_snaps = SnapshotIndex::default(); + // Three gossip_received events on host "b" (not "a") ⇒ filter + // out, count 0 ⇒ Fail. + let other_peer = (0..3) + .map(|i| { + evt( + 10 + i as u64, + Some("b"), + "swim", + serde_json::json!({"kind": "gossip_received", "source_peer": "c"}), + i, + ) + }) + .collect::>(); + assert_eq!( + evaluate(&scen, &other_peer, &no_snaps)[0].outcome, + Outcome::Fail, + "peer filter must exclude events from other hosts" + ); + // Same kind on host "a" ⇒ Pass. + let target_peer = vec![evt( + 50, + Some("a"), + "swim", + serde_json::json!({"kind": "gossip_received", "source_peer": "c"}), + 4, + )]; + assert_eq!( + evaluate(&scen, &target_peer, &no_snaps)[0].outcome, + Outcome::Pass, + "peer filter must Pass when matching events exist on the named host" + ); + // Mixed: only host "a"'s events should count. + let mixed: Vec<_> = other_peer.into_iter().chain(target_peer.into_iter()).collect(); + assert_eq!( + evaluate(&scen, &mixed, &no_snaps)[0].outcome, + Outcome::Pass, + "peer filter must reduce a mixed stream to the named host's events only" + ); +} + #[test] fn event_rate_pass_fail() { let scen = scenario_with(vec![AssertionKind::EventRate { @@ -473,7 +594,9 @@ fn event_rate_pass_fail() { fn verdict_shape_includes_kind_parameters_outcome_and_evidence_on_fail() { let scen = scenario_with(vec![AssertionKind::EventCount { event_kind: "probe_sent".into(), - max: 0, + min: None, + max: Some(0), + peer: None, }]); let events = vec![evt( 10, @@ -498,9 +621,9 @@ fn verdict_shape_includes_kind_parameters_outcome_and_evidence_on_fail() { #[test] fn verdicts_listed_in_scenario_declaration_order() { let scen = scenario_with(vec![ - AssertionKind::EventCount { event_kind: "alpha".into(), max: 0 }, - AssertionKind::EventCount { event_kind: "beta".into(), max: 0 }, - AssertionKind::EventCount { event_kind: "gamma".into(), max: 0 }, + AssertionKind::EventCount { event_kind: "alpha".into(), min: None, max: Some(0), peer: None }, + AssertionKind::EventCount { event_kind: "beta".into(), min: None, max: Some(0), peer: None }, + AssertionKind::EventCount { event_kind: "gamma".into(), min: None, max: Some(0), peer: None }, ]); let v = evaluate(&scen, &[], &SnapshotIndex::default()); assert_eq!(v.len(), 3); @@ -518,7 +641,9 @@ fn verdicts_listed_in_scenario_declaration_order() { fn streaming_resolves_to_same_verdict_as_post_run() { let scen = scenario_with(vec![AssertionKind::EventCount { event_kind: "probe_sent".into(), - max: 1, + min: None, + max: Some(1), + peer: None, }]); let events = vec![ evt(10, None, "swim", serde_json::json!({"kind": "probe_sent", "from": "a", "to": "b"}), 0), @@ -542,7 +667,9 @@ fn streaming_resolves_to_same_verdict_as_post_run() { fn streaming_resolves_event_count_fail_at_first_overshoot() { let scen = scenario_with(vec![AssertionKind::EventCount { event_kind: "probe_sent".into(), - max: 1, + min: None, + max: Some(1), + peer: None, }]); let mut stream = StreamingEvaluator::new(scen); stream.feed_event(evt( @@ -576,8 +703,8 @@ fn streaming_resolves_event_count_fail_at_first_overshoot() { fn evaluate_bundle_writes_verdicts_json_with_one_entry_per_assertion() { let tmp = TempDir::new().unwrap(); let scen = scenario_with(vec![ - AssertionKind::EventCount { event_kind: "probe_sent".into(), max: 0 }, - AssertionKind::EventCount { event_kind: "alpha".into(), max: 100 }, + AssertionKind::EventCount { event_kind: "probe_sent".into(), min: None, max: Some(0), peer: None }, + AssertionKind::EventCount { event_kind: "alpha".into(), min: None, max: Some(100), peer: None }, ]); // Write a tiny bundle: one event of kind probe_sent (which makes // assertion 0 fail, assertion 1 pass since alpha has 0 events). diff --git a/crates/simulation/tests/n3_battery_pass.rs b/crates/simulation/tests/n3_battery_pass.rs new file mode 100644 index 0000000..7e1739d --- /dev/null +++ b/crates/simulation/tests/n3_battery_pass.rs @@ -0,0 +1,123 @@ +//! Battery Pass-expected binary +//! (`N3_SIM_TEST_BATTERY_SPEC.md §1.7`). +//! +//! Scenarios declared `Pass` against the current source run here under +//! standard `cargo test` semantics — a regression in the simulator or +//! post-processor is a CI break. Today the Pass-expected families are +//! C (gossip-arrival absence — discriminator regression guard) and E +//! (bundle integrity under SIGKILL — `S-D` regression guard). + +use std::fs; +use std::path::{Path, PathBuf}; + +use serde_json::Value; +use tempfile::TempDir; + +use simulation::bundle_file::FileBundleWriter; +use simulation::engine::{Engine, TerminationReason}; +use simulation::evaluator::{Outcome, evaluate_bundle}; +use simulation::network::Network; +use simulation::scenario::{HostKindRegistry, Scenario, load_from_path}; +use simulation::stage_host::StageHostFactory; +use simulation::swim_host::SwimHostFactory; + +fn registry() -> HostKindRegistry { + HostKindRegistry::with_swim() +} + +fn load(rel: &str) -> Scenario { + let path = PathBuf::from(env!("CARGO_MANIFEST_DIR")).join(rel); + load_from_path(&path, ®istry()).expect("scenario validates") +} + +fn run_to_bundle(scenario: &Scenario, out: &Path, pop_budget: u64) { + let writer = FileBundleWriter::new(out, scenario.clone()); + let network = Network::new(scenario); + let mut engine = Engine::new(scenario, network, writer); + engine.register_factory(Box::new(SwimHostFactory)); + engine.register_factory(Box::new(StageHostFactory)); + engine.auto_install_hosts(); + engine.set_pop_budget(pop_budget); + let term = engine.run(); + assert!( + matches!( + term, + TerminationReason::DurationReached | TerminationReason::EarlyAllAssertionsResolved + ), + "unexpected termination {term:?}" + ); + let writer = engine.into_writer(); + writer.finalize().expect("finalize bundle"); +} + +#[test] +fn family_c_central_gossip_absence_passes_self_incarnation_bound() { + // Spec §3 family C central: expected verdict `Pass`. The + // observability upgrade landed `GossipReceived` and the per-peer + // dial rollup, so the discriminator (control-plane vs data-plane) + // is already expressible. This test guards that contract against + // regression — the orchestrator's self_incarnation should stay + // bounded under a stage-to-stage partition. + let scen = load("scenarios/reproduction/n3_2026_05_25/family_c_gossip_absence/central.toml"); + let tmp = TempDir::new().unwrap(); + let out = tmp.path().join("bundle"); + run_to_bundle(&scen, &out, 200_000); + let verdicts = evaluate_bundle(&out).expect("evaluator runs"); + assert!(!verdicts.is_empty(), "family C: no verdicts produced"); + for v in &verdicts { + assert!( + matches!(v.outcome, Outcome::Pass | Outcome::Inconclusive), + "family C central: unexpected non-Pass verdict {v:?}" + ); + } +} + +#[test] +fn family_e_central_sigkill_orchestrator_produces_parseable_bundle() { + // Spec §3 family E central: expected verdict `Pass`. The + // observability upgrade landed `S-D` (bundle without finalize); + // this test guards that contract. The scenario kills the + // orchestrator at +5 s; the simulator's bundle writer must + // still produce a parseable manifest and per-peer staging + // files for the surviving peers' pre-kill records. + let scen = load("scenarios/reproduction/n3_2026_05_25/family_e_bundle_integrity_sigkill/central.toml"); + let tmp = TempDir::new().unwrap(); + let out = tmp.path().join("bundle"); + run_to_bundle(&scen, &out, 200_000); + + // Bundle-shape contract per family-E spec §3: + // - `manifest.json` exists. + // - every per-peer events file exists (the sim writes + // `events.ndjson` shared across peers, not per-peer files; + // verify the aggregate file). + let manifest_path = out.join("manifest.json"); + assert!( + manifest_path.is_file(), + "family E central: manifest.json missing under {}", + out.display() + ); + let manifest_text = fs::read_to_string(&manifest_path).expect("read manifest.json"); + let manifest: Value = serde_json::from_str(&manifest_text).expect("manifest is JSON"); + assert!( + manifest.is_object(), + "family E central: manifest.json is not an object: {manifest_text}" + ); + let events_path = out.join("events.ndjson"); + assert!( + events_path.is_file(), + "family E central: events.ndjson missing under {}", + out.display() + ); + + // Verdicts file exists and contains a verdict per declared + // assertion. `Inconclusive` is acceptable for any whose + // preconditions did not fire (e.g., the orch is dead by +5 s). + let verdicts = evaluate_bundle(&out).expect("evaluator runs"); + assert!(!verdicts.is_empty(), "family E: no verdicts produced"); + for v in &verdicts { + assert!( + matches!(v.outcome, Outcome::Pass | Outcome::Inconclusive), + "family E central: unexpected Fail {v:?}" + ); + } +} diff --git a/crates/simulation/tests/sim_cross_pollination.rs b/crates/simulation/tests/sim_cross_pollination.rs index 4d63bd6..822c7bd 100644 --- a/crates/simulation/tests/sim_cross_pollination.rs +++ b/crates/simulation/tests/sim_cross_pollination.rs @@ -21,10 +21,11 @@ //! (§2) holds even with no scenario config. use distribution::diagnostics::{Tier2RelaySession, Tier3SubprocessState}; +use distribution::types::NodeId; use serde_json::Value; use simulation::host::{Action, Host}; -use simulation::stage_host::{StageHost, SubprocessFakeSpec}; +use simulation::stage_host::{InferenceFakeSpec, StageHost, SubprocessFakeSpec}; #[test] fn stage_host_snapshot_always_carries_tier2_relay_session_with_unknown_default() { @@ -188,6 +189,94 @@ fn exit_after_ns_emits_typed_exited_with_correct_uptime() { assert_eq!(tier3.subprocesses[0].exit_code, Some(0)); } +// ────────────────────────────────────────────────────────────────────── +// Coverage 2.4 — inference response-leg send-outcome event +// ────────────────────────────────────────────────────────────────────── + +#[test] +fn inference_fake_emits_typed_response_sent_with_timeout_outcome() { + // Coverage 2.4 close-criterion shape: the last stage emits + // exactly one `InferenceResponseSent` carrying target / request / + // size / outcome when its outbound to the orchestrator fails. + // The `1779733878` failure attribution — "last stage could not + // deliver the response" — is now a single typed read, not a + // triangulation against dial timeouts. + let mut host = StageHost::new("stage-last", "pp-stage-last", "10.0.0.20:7700"); + let orch_node_id = NodeId([0xAB; 32]); + host.set_inference_fake(InferenceFakeSpec { + fire_at_ns: 5_000_000, + target_peer_node_id: orch_node_id, + request_id: "req-7f3c".into(), + byte_size: 4_096, + send_outcome: "timeout".into(), + }); + // Drive into Running. + let _ = host.tick(0); + // Past the fire time: the event lands. + let actions = host.tick(5_500_000); + let diag_events = collect_diag_events(&actions); + let sent = diag_events + .iter() + .find(|p| p.get("type").and_then(|v| v.as_str()) == Some("InferenceResponseSent")) + .expect("InferenceResponseSent must fire past fire_at_ns"); + assert_eq!(sent["request_id"].as_str(), Some("req-7f3c")); + assert_eq!(sent["byte_size"].as_u64(), Some(4_096)); + assert_eq!(sent["send_outcome"].as_str(), Some("timeout")); + // target_peer round-trips through the production NodeId schema. + let target: NodeId = serde_json::from_value(sent["target_peer"].clone()) + .expect("target_peer must deserialize as NodeId"); + assert_eq!(target, orch_node_id); +} + +#[test] +fn inference_fake_fires_at_most_once_across_many_ticks() { + // Spec §2.4 says "exactly one" event per response send. A stage + // host that re-emitted on every tick past `fire_at_ns` would + // produce double-counting in the bundle. + let mut host = StageHost::new("stage-once", "pp-stage-once", "10.0.0.21:7700"); + host.set_inference_fake(InferenceFakeSpec { + fire_at_ns: 1_000_000, + target_peer_node_id: NodeId([0xCD; 32]), + request_id: "req-dedupe".into(), + byte_size: 128, + send_outcome: "success".into(), + }); + let _ = host.tick(0); + let mut seen = 0usize; + for t in [1_000_000u64, 2_000_000, 3_000_000, 10_000_000] { + let actions = host.tick(t); + for p in collect_diag_events(&actions) { + if p.get("type").and_then(|v| v.as_str()) == Some("InferenceResponseSent") { + seen += 1; + } + } + } + assert_eq!( + seen, 1, + "InferenceResponseSent must fire exactly once across many ticks past fire_at_ns", + ); +} + +#[test] +fn inference_fake_unset_emits_no_response_event() { + // Honesty-under-absence: a stage with no inference fake produces + // no InferenceResponseSent. The bundle reader sees the absence + // (the postproc renders the gap-2.4 absence-line); a silent + // synthesized event would break the discriminator contract. + let mut host = StageHost::new("stage-quiet", "pp-stage-quiet", "10.0.0.22:7700"); + let _ = host.tick(0); + for t in [1_000_000u64, 5_000_000, 50_000_000] { + let actions = host.tick(t); + for p in collect_diag_events(&actions) { + assert_ne!( + p.get("type").and_then(|v| v.as_str()), + Some("InferenceResponseSent"), + "unsetting the inference fake must suppress InferenceResponseSent", + ); + } + } +} + // ─── helpers ────────────────────────────────────────────────────────── fn collect_diag_events(actions: &[Action]) -> Vec { diff --git a/crates/simulation/tests/swim_host_invariants.rs b/crates/simulation/tests/swim_host_invariants.rs index 4cf1f59..19ad749 100644 --- a/crates/simulation/tests/swim_host_invariants.rs +++ b/crates/simulation/tests/swim_host_invariants.rs @@ -167,6 +167,14 @@ fn every_recorded_event_has_a_known_kind_discriminator() { // RecordEvents the simulator synthesises. "state_transition", "message_send", + // Coverage 2.6: per-SWIM-probe lifecycle events. Each probe + // surfaces as one `swim_probe_sent` plus exactly one of + // `swim_probe_acked` / `swim_probe_timed_out` per phase. The + // bundle reader joins them on `(target, sequence)` to derive + // per-probe RTT. + "swim_probe_sent", + "swim_probe_acked", + "swim_probe_timed_out", // Any production `DiagEvent` variant we don't have an MVP // schema for surfaces under `diag_event` carrying the // production `type` tag verbatim. The mapping function is @@ -183,3 +191,92 @@ fn every_recorded_event_has_a_known_kind_discriminator() { } } +// ────────────────────────────────────────────────────────────────────── +// Coverage 2.6 — per-SWIM-probe RTT events (`N3_COVERAGE_EXTENSION_SPEC.md §2.6`) +// ────────────────────────────────────────────────────────────────────── + +/// A SWIM host with no inbound traffic exercises the probe-timeout path. +/// Verifies the lifecycle contract: every `swim_probe_sent` resolves +/// into either `swim_probe_acked` or `swim_probe_timed_out` on the same +/// `(target, sequence)`, never both, and timeouts carry the configured +/// `budget_ticks` so a bundle reader can see the budget alongside the +/// absent RTT (honesty-under-absence). +#[test] +fn coverage_2_6_unanswered_probes_resolve_to_typed_timed_out_events() { + let mut host = make_host("a", &["a", "b", "c"]); + + // Drive enough ticks that a Periodic probe fires (probe_interval=2) + // and both phases (direct then indirect) exhaust their budget + // (probe_timeout=1 each). 30 ticks comfortably covers several + // complete probe cycles. + let mut events: Vec = Vec::new(); + for t in 0..30u64 { + for action in host.tick(t * 1000) { + if let Action::RecordEvent { event, .. } = action { + let v: serde_json::Value = + serde_json::from_slice(&event).expect("event payload is JSON"); + events.push(v); + } + } + } + + let sent: Vec<&serde_json::Value> = events + .iter() + .filter(|e| e["kind"] == "swim_probe_sent") + .collect(); + let acked: Vec<&serde_json::Value> = events + .iter() + .filter(|e| e["kind"] == "swim_probe_acked") + .collect(); + let timed_out: Vec<&serde_json::Value> = events + .iter() + .filter(|e| e["kind"] == "swim_probe_timed_out") + .collect(); + + // The host has no peer responding, so every probe must time out at + // both phases. Cover-2.6 contract: at least one probe lifecycle. + assert!( + !sent.is_empty(), + "no swim_probe_sent events emitted in 30 ticks (probe scheduler stuck?): {events:?}" + ); + assert!( + acked.is_empty(), + "swim_probe_acked surfaced without any inbound traffic: {acked:?}" + ); + assert!( + !timed_out.is_empty(), + "no swim_probe_timed_out events despite no inbound traffic: {events:?}" + ); + + // Honesty-under-absence: every timeout carries the configured + // budget so a bundle reader sees "probe missed a 1-tick budget" + // rather than a silent zero or null. + for to in &timed_out { + let budget = to["budget_ticks"].as_u64(); + assert_eq!( + budget, + Some(1), + "swim_probe_timed_out missing or mismatched budget_ticks: {to}" + ); + let probe_kind = to["probe_kind"].as_str().unwrap_or(""); + assert!( + probe_kind == "direct" || probe_kind == "indirect", + "swim_probe_timed_out has unexpected probe_kind {probe_kind:?}: {to}" + ); + } + + // Schema parity contract (`SIM_SPEC.md §9.2`): every sent event + // carries `target` (hex node id) and a `sequence` u64. The bundle + // reader can join (target, sequence) with the corresponding + // resolution. + for s in &sent { + assert!(s["target"].is_string(), "swim_probe_sent.target absent: {s}"); + assert!(s["sequence"].is_u64(), "swim_probe_sent.sequence absent: {s}"); + let probe_kind = s["probe_kind"].as_str().unwrap_or(""); + assert!( + probe_kind == "direct" || probe_kind == "indirect", + "swim_probe_sent has unexpected probe_kind {probe_kind:?}: {s}" + ); + } +} + diff --git a/examples/pipeline-parallel-inference/Cargo.lock b/examples/pipeline-parallel-inference/Cargo.lock index ba33215..58992f0 100644 --- a/examples/pipeline-parallel-inference/Cargo.lock +++ b/examples/pipeline-parallel-inference/Cargo.lock @@ -802,6 +802,7 @@ dependencies = [ "flate2", "iroh", "iroh-metrics", + "iroh-relay", "libc", "serde", "serde_json", @@ -2621,6 +2622,7 @@ dependencies = [ "swactor", "swactor-process", "tokio", + "tracing-subscriber", "urlencoding", "wiremock", ] diff --git a/examples/pipeline-parallel-inference/Cargo.toml b/examples/pipeline-parallel-inference/Cargo.toml index d11c0fd..9bd60aa 100644 --- a/examples/pipeline-parallel-inference/Cargo.toml +++ b/examples/pipeline-parallel-inference/Cargo.toml @@ -29,3 +29,4 @@ path = "src/bin/pp_smoke_run.rs" [dev-dependencies] wiremock = "0.6" +tracing-subscriber = { version = "0.3", features = ["env-filter"] } diff --git a/examples/pipeline-parallel-inference/DEPLOYMENT_TEST.md b/examples/pipeline-parallel-inference/DEPLOYMENT_TEST.md deleted file mode 100644 index 78d84a0..0000000 --- a/examples/pipeline-parallel-inference/DEPLOYMENT_TEST.md +++ /dev/null @@ -1,187 +0,0 @@ -# vast.ai deployment test - -Drives `pp-smoke-run --vastai` against N real GPU instances, with a -collector + iroh-relay on a separate VPS so the run's bundle survives -the instances' destruction. See `N3_DEPLOYMENT_REPORT.md` for the three -classes of bug this loop has historically caught. - -## Pre-flight on the VPS - -The collector and relay are long-lived on a separate VPS so they -outlive any single rental. The reference deployment is docean -(146.190.110.128). Verify both processes are up before any run: - -```sh -ssh docean 'pgrep -fa swactor-diag-collector; pgrep -fa swactor-iroh-relay' -# expect one PID for each -``` - -If either is missing, rebuild static-musl and redeploy: - -```sh -cargo build --release --target x86_64-unknown-linux-musl \ - -p distribution --features "collector relay" \ - --bin swactor-diag-collector --bin swactor-iroh-relay -scp target/x86_64-unknown-linux-musl/release/swactor-diag-{collector,iroh-relay} docean:~/ -ssh docean ' - nohup ./swactor-diag-collector --bind 0.0.0.0:9080 --root /var/lib/swactor-diag \ - --udp 0.0.0.0:9081 > /var/log/swactor-diag-collector.log 2>&1 & - nohup ./swactor-iroh-relay --bind 0.0.0.0:7843 \ - --public-host 146.190.110.128 > /var/log/swactor-iroh-relay.log 2>&1 &' -``` - -Firewall: `9080/tcp` (collector HTTP), `9081/udp` (echo probe), -`7843/tcp` (iroh-relay) all open. Sanity-check from your laptop: - -```sh -curl -sS -o /dev/null -w '%{http_code}\n' http://146.190.110.128:9080/ # → 404 (port is bound) -curl -sS http://146.190.110.128:7843/ | grep -o 'Iroh Relay' # → Iroh Relay -``` - -## Building the orchestrator + the GPU image - -The orchestrator runs locally. The GPU image runs on the rentals. -Both must come from the same workspace commit so the iroh and SWIM -versions line up. - -```sh -# Orchestrator-side binary (used as pp-smoke-run --vastai) -cargo build --release --bin pp-smoke-run - -# GPU image — Dockerfile bundles pp-gpu-node + worker -cargo build --release --bin pp-gpu-node -docker build -t zacheryasc/swactor-pp-gpu:latest -f Dockerfile . -docker push zacheryasc/swactor-pp-gpu:latest -``` - -## Running the deployment test - -The orchestrator passes the diagnostics + relay URLs into every rented -container's env via `vastai::create_instance`. Set the same vars the -local stages would see, then invoke `--vastai`: - -```sh -RUN_ID="vastai-N3-$(date +%s)" - -# Required: collector + relay so the cluster comes up at all and the -# bundle gets persisted (see N3 report Layer A). -export SWACTOR_DIAG_COLLECTOR_URL="http://146.190.110.128:9080" -export SWACTOR_DIAG_UDP_ECHO="146.190.110.128:9081" -export SWACTOR_IROH_RELAY_URL="http://146.190.110.128:7843/" -export SWACTOR_DIAG_RUN_ID="$RUN_ID" - -# Optional: switch workers without rebuilding the image. -# Drop PP_WORKER_STUB=1 to exercise the real tinygrad path. -export PP_WORKER_STUB=1 -# export MODEL=llama3.2:1b -# export CUDA=1 -# export PYTHON=python3 - -target/release/pp-smoke-run --vastai \ - --api-key "$VAST_API_KEY" \ - --num-stages 3 \ - --gpu RTX_4090 \ - --image zacheryasc/swactor-pp-gpu:latest \ - --prompt "Diag check" \ - --max-tokens 4 \ - 2>&1 | tee "$RUN_ID.log" -``` - -Three N≥2 invariants the run is checking: - -1. Cluster converges within `pp-smoke-run`'s convergence deadline - (every peer sees every other as `Alive`). -2. `pp-entry` resolves on the orchestrator (Layer B / name-gossip - path). -3. The pipeline returns a non-empty `InferenceResponse`. - -Failure of (1) or (2) without (3) → a SWIM or relay bug. -Failure of (3) only → a worker bug. - -On any exit the orchestrator destroys every rented instance, so a -hung or crashed run does not leak GPUs. Verify after: - -```sh -curl -s -H "Authorization: Bearer $VAST_API_KEY" \ - https://cloud.vast.ai/api/v0/instances/ | jq '.instances | length' -# → 0 (or only your own unrelated instances) -``` - -## Fetching the bundle from the VPS - -The collector finalises the run-id tarball when it receives the -orchestrator's finalize record. It lives both in the collector's bind- -mounted dir and at the HTTP retrieval endpoint: - -```sh -curl -fsSO "http://146.190.110.128:9080/diag/bundle/$RUN_ID" -# or, from the VPS itself: -ssh docean "ls -la /var/lib/swactor-diag/bundles/$RUN_ID.tar.gz" -``` - -## Post-processing + what to look for - -```sh -target/release/swactor-diag-postproc "$RUN_ID.tar.gz" -o "$RUN_ID.out" -cat "$RUN_ID.out/summary.md" -``` - -### Healthy run - -`summary.md` shows N+1 nodes (orchestrator + N stages), each with -`finalize_recorded: true` for the orchestrator and several snapshots -per stage. Custom event totals include `worker_starting` and -`worker_ready` for every stage and zero `SwimTransition → Dead`. The -"First peer to go Dead" section is empty. - -### SWIM regression (Layer B) - -`summary.md` lists peers transitioning to `Dead` despite probes -succeeding (`probes_ok_at_transition: yes` in the per-peer block). -Cross-check `self_incarnation` on the orchestrator snapshot — -anything above ~10 over a 7-minute run is the §10.3 flap (see -SWIM_TUNING_REPORT). Drill into the relevant timeline-NN-to-MM.tsv -for the message sequence around the transition. - -### Relay regression (Layer A) - -Per-peer reachability blocks show `conn_type=Relay` and probe RTTs -spiking into hundreds of ms or seconds. Confirm with -`Custom(iroh_api_missing)` and the iroh introspection block in the -last snapshot — relay-buffered messages show as huge `last_used_ms` -gaps. The mitigation is the own-relay setup above; running with -`SWACTOR_IROH_RELAY_URL` unset deliberately reproduces the canary -buffering for evidence-collection runs. - -### Worker death (Layer C) - -`summary.md` shows `Custom(worker_exited)` events. Pull the structured -fields: - -```sh -jq '.[] | select(.kind == "worker_exited") | .fields' \ - "$RUN_ID.out/../$(basename $RUN_ID .tar.gz)/stage-0/events/"events-*.json -``` - -You get `exit_code`, `signal`, `uptime_ms`, the ring-buffered -`stderr_tail` (~256 last lines), and a `python_traceback` when the -worker raised an uncaught exception. For model-load specifically, -`worker_model_load_failed` carries `{model, type, value, traceback}` -in one record. - -## Cleanup after a session - -The orchestrator destroys rentals on exit, but if it crashed -mid-orchestration check by hand: - -```sh -curl -s -H "Authorization: Bearer $VAST_API_KEY" \ - https://cloud.vast.ai/api/v0/instances/ | jq '.instances[].id' -# destroy any survivors: -curl -X DELETE -H "Authorization: Bearer $VAST_API_KEY" \ - "https://cloud.vast.ai/api/v0/instances//" -``` - -Bundles older than a few weeks can be pruned from -`docean:/var/lib/swactor-diag/bundles/` to keep the VPS disk usage -low. diff --git a/examples/pipeline-parallel-inference/Dockerfile b/examples/pipeline-parallel-inference/Dockerfile index f55f30e..d404eb6 100644 --- a/examples/pipeline-parallel-inference/Dockerfile +++ b/examples/pipeline-parallel-inference/Dockerfile @@ -1,16 +1,20 @@ FROM nvidia/cuda:12.6.3-runtime-ubuntu24.04 -# Runtime base — no CUDA dev headers, no nvcc. tinygrad's CUDA backend -# compiles kernels via NVRTC which is part of the runtime image, so we -# do not need the devel image (that base alone is ~5 GB and dominated -# the 7.55 GB total of the previous build, blowing past the vastai -# image-pull budget). +# Runtime base, not devel: the devel base alone is ~5 GB and blew past the +# vastai image-pull budget (the previous 7.55 GB build). NVRTC — the kernel +# compiler tinygrad's CUDA backend uses — ships in the runtime image, but the +# CUDA *toolkit headers* do not, and tinygrad's generated fp16 kernels +# `#include `. Pull in just the cudart dev headers (~7 MB) so +# NVRTC's `-I/usr/local/cuda/include` resolves them — the minimal alternative +# to the full devel base. Without this every real-model stage dies at +# graph-realize with NVRTC_ERROR_COMPILATION ("cannot open cuda_fp16.h"). RUN apt-get update && \ apt-get install -y --no-install-recommends \ python3 \ python3-venv \ python3-pip \ - ca-certificates && \ + ca-certificates \ + cuda-cudart-dev-12-6 && \ rm -rf /var/lib/apt/lists/* # Install tinygrad and numpy diff --git a/examples/pipeline-parallel-inference/N3_COVERAGE_EXTENSION_SPEC.md b/examples/pipeline-parallel-inference/N3_COVERAGE_EXTENSION_SPEC.md new file mode 100644 index 0000000..5d79d2b --- /dev/null +++ b/examples/pipeline-parallel-inference/N3_COVERAGE_EXTENSION_SPEC.md @@ -0,0 +1,520 @@ +# N=3 collection coverage extension — behavioral spec + +Companion to `N3_POSTMORTEM_2026-05-25_1779733878.md`, +`N3_SIM_TEST_BATTERY_SPEC.md`, and the simulator's `SIM_SPEC.md`. +This document is the contract for a separate coding agent to extend +diagnostic collection coverage along three layers — **production +diagnostics**, **simulator emit/model**, and **simulator test +verification** — for the gaps the `1779733878` run surfaced. + +This is a *behavioral* spec. It names the gap, the contract the +collected data must satisfy, and the layer(s) the contract threads +through. It does not prescribe field names, file layout, or +implementation choices. + +--- + +## 0. Motivation + +The `1779733878` run validated the prior observability upgrade — +A+B+C tiers were load-bearing, the bundle attributed the failure to +"dials to orchestrator fail by timeout 7/11 while every inter-stage +dial succeeds 19/19" in one table — and surfaced five residual +collection gaps the upgrade either left as carry-forward (gaps 1, 5, +8 from the original scorecard) or that this run exposed for the +first time (response-leg event absence, bundle-serve behavior under +run-id reuse). + +A gap whose collection landed only in production but not in the sim +is a gap that the sim test battery can never guard — the next regression +in that field will be caught only by another live deploy. A gap whose +sim model exists but is not exercised by a test is dead code. The +coverage in this spec is required to thread through every layer +where it can — and the spec is explicit when a layer does not +apply. + +Five coverages, each threaded through up to three layers. Each +coverage may close one of {Pass, Mixed, Fail} against the current +source, and each names the close criterion. + +--- + +## 1. Cross-cutting requirements + +These hold for every coverage in §2. + +### 1.1 Three-layer threading + +For each coverage, the spec names which of the three layers it +threads through: + +- **D — Diagnostics**: the production bundle gains a field, event, + or section that closes the gap the postmortem named. +- **S — Sim**: the simulator's relevant component (host kind, + network, relay vertex, bundle writer) emits the same field / + event / section under the same conditions, with bundle-shape + parity per `SIM_SPEC.md §5` (cross-cutting "Bundle-shape parity + with prod") and §9 (bundle schema). +- **T — Tests**: the sim test battery gains a scenario or property + test asserting the bundle carries the new data when the + triggering condition holds, and gains a discriminator assertion + when absence is meaningful (per the honesty-under-absence pattern + the prior upgrade established). + +A coverage that threads through fewer than three layers is honest +about which it skips and why. Skipping S because "the simulator +does not model this surface" is acceptable; skipping it because +"this is not interesting" is not. + +### 1.2 Honesty-under-absence carries forward + +The prior upgrade's `status_source` discriminator pattern (a status +field always paired with a field naming how that status was derived +— `"iroh"` for native, `"derived"` for inferred) is the model. Any +new field whose value might be absent or derived must carry an +adjacent discriminator. A bundle reader must never be left guessing +"unknown means the thing is unknown" vs "we couldn't ask." + +### 1.3 Additive evolution + +Every new field on `SnapshotBody`, every new event variant, every +new section in the post-processor output is additive. An old bundle +reader on a new bundle still parses; a new bundle reader on an old +bundle reports the new field absent rather than erroring. The +prior upgrade established this contract; coverage 2.x preserves it. + +### 1.4 Sim/prod schema parity + +Per `SIM_SPEC.md §9.2`: the event payload schema is exactly the +production diagnostics schema for that kind. The sim invents no new +event kinds. A coverage that lands an event in prod and in sim +**uses the same schema in both**, verified by the existing parity +tests under `crates/simulation/tests/sim_cross_pollination.rs`. A +schema added to sim ahead of prod is a deliberate amendment and +declares so explicitly. + +### 1.5 Verdict-first per layer + +Every coverage in §2 declares, per layer, its expected status on +the current source: **landed** (the layer satisfies the contract; +the work is verification / regression-guard), **partial** (the +layer has structure but not data flow), **absent** (the layer has +nothing today). The implementing agent's work is to bring each +layer to "landed" against this spec or to file a structural reason +why a layer cannot land. + +--- + +## 2. The coverages + +Five, ordered by the postmortem's own ranking of residual gaps. + +### 2.1 Orchestrator-side host provider metadata forwarding + +**Source**: postmortem §"Observability upgrade scorecard" row "gap +5 host metadata" (◐); postmortem §"Data-collection / deployment +gaps surfaced by this run" item 1. + +**Gap**: The orchestrator has each rental's public IP, datacenter, +country, and contract id at `lease_chain` return time. The +container can read these from `SWACTOR_DIAG_*` env vars. The +container env is never set. The boot record's +`host_ip_public`/`datacenter_id`/`host_country`/`vastai_contract_id`/ +`home_relay_url_at_boot` fields are still null in every bundle. +The contract id arrives but in `container_id`, not +`vastai_contract_id` — the naming is currently load-bearing-but-wrong. + +**Contract — what closing the gap looks like**: + +- D: the orchestrator's per-rental env payload, at the point it + creates each container, carries every field the boot record can + consume — public IP, datacenter id, host country, vast.ai + contract id, the home relay URL the container will use. The boot + record reflects every field as a concrete value, not `null`, + whenever the orchestrator had the data. The fields that name + cloud-provider state stay absent only on hosts where they + genuinely do not apply (e.g., local development), and the + bundle's `## Hosts` section renders `?` for absent fields + (already implemented per `S-A2`). +- S: scenarios declare per-peer host context as part of the peer's + `kind_config`. The sim's stage host populates its boot record / + `HostContext` from the scenario declaration the same way prod + populates from env. A scenario without declared host context + produces a bundle whose `## Hosts` section is all-`?` for that + peer — same absence shape as a local-dev prod bundle. +- T: a scenario declaring heterogeneous host context across three + peers (e.g., two datacenters, two countries) produces a bundle + whose `## Hosts` section renders the declared fields verbatim. + A scenario that declares no context for one peer and full context + for the others produces a bundle distinguishable from "no context + declared for any peer" by the `?` placement. + +**Expected status**: + +- D: partial. The container reads the env vars (per `S-A2`); the + orchestrator does not set them. The misnaming of contract id → + `container_id` is a separate cleanup. +- S: absent. The sim's stage host carries no host context in its + current scenario schema. +- T: absent. No test exercises this discriminator. + +**Close criterion**: a deployed bundle's `## Hosts` section names +the datacenter, country, public IP, and contract id of every +vast.ai rental, and the docker `container_id` field carries the +docker container id, not the vast.ai contract id. A sim bundle +with declared host context produces the matching shape. + +### 2.2 Relay-port reachability probe + +**Source**: postmortem §"Observability upgrade scorecard" row "gap +8 relay-port probe" (✗); postmortem §"Data-collection / deployment +gaps surfaced by this run" item 3. + +**Gap**: Stage probe arrays carry only `collector_udp_echo` +(:9081). No probe targets the relay's actual port (:7843). +Whether a stage retained transport-level reachability to the relay +at the moment its peer-connection died is currently inferable only +from a *different* port on the same host. The `S-E1` work is +documented as landed (per the prior iteration log) but the +`1779733878` bundle shows no relay-port probe records. The wiring +is in place; the data is not. + +**Contract — what closing the gap looks like**: + +- D: every stage's snapshot carries a probe outcome for the + relay's UDP listener (host + port resolved from the home relay + URL). The outcome is one of the five-discriminator set the + prior upgrade established: `ok` / `timeout` / `refused` / + `unresolved` / `error`. A snapshot taken when the relay is + reachable carries `ok` with an RTT; a snapshot taken when the + relay is unreachable carries the appropriate failure + discriminator with no silent fallback to "absent." +- S: the sim's stage host emits the same probe record on every + snapshot, sourced from a query the network answers about the + stage→relay edge. The relay vertex's `RelayKill` / + `RelayCapacityChange` mutations are reflected in the probe's + outcome distribution. +- T: a scenario that issues a `RelayKill` mutation mid-run + produces a bundle whose every stage's relay-port probe outcome + flips from `ok` to `unresolved` (or `timeout`, per the + network's policy) at the mutation's `at_ns` and remains there + through `RelayBoot`. The probe-outcome timeline is the test's + discriminator between "tunnel down" and "tunnel up but peer + conn down" — coverage 2.x.A from the battery spec consumes + this signal. + +**Expected status**: + +- D: partial. Probe scheduler wires the target; emission to the + bundle is unverified by this run's evidence. +- S: absent. The sim's network has no probe-query surface today. +- T: absent. + +**Close criterion**: the next deployment's bundle has a +relay-port probe outcome on every stage's snapshots. A sim +scenario with `RelayKill` produces the probe-outcome flip in the +bundle. + +### 2.3 Relay session lifecycle on the relay side + +**Source**: postmortem §"Observability upgrade scorecard" row "gap +1 relay observability" (◐); postmortem §"Data-collection / +deployment gaps surfaced by this run" item 4. + +**Gap**: The relay reports identity and 186 snapshots into the +bundle but cannot answer "who closed session X and why" — the +per-session lifecycle hooks are the documented skeleton with +`active=0 opens=0 closes=0`. `iroh_relay::server` exposes no +session hooks. Until it does, a relay-side eviction is +unanswerable from the relay's own data; the postmortem fell back +to node-side dial outcomes. + +**Contract — what closing the gap looks like**: + +- D: the relay's bundle contribution names, per peer session, the + open time, close time, close-initiator discriminator + (`relay` / `peer` / `transport` / `unknown`), close reason + string (relay-specific or transport-specific), bytes + transferred per direction, and duration. The mechanism is + free — middleware around the relay binary, kernel-layer + observation, a forked relay, or upstream hooks when iroh + exposes them. The contract is the *shape*, not the source. + When the source is unavailable, the relay's bundle + contribution still emits the gap-1 absence-line the prior + upgrade introduced in `summary.md` (the post-processor's + acceptance branch for "no relay-role node has session data"). +- S: the sim's relay vertex emits `RelaySessionOpened` / + `RelaySessionClosed` records when it accepts and releases + per-peer queues. The records carry the same shape D + requires. A `RelayKill` mutation produces a + `RelaySessionClosed { initiator: "relay", reason: "killed", + ... }` for every session active at the mutation time. +- T: a scenario where the relay accepts three peer sessions, runs + to steady state, then receives a `RelayKill` mutation, + produces a bundle whose relay contribution names three + `RelaySessionOpened` events at the convergence boundary and + three `RelaySessionClosed { initiator: "relay" }` events at + the mutation time. A scenario where a peer voluntarily + disconnects produces a session-closed event with + `initiator: "peer"`. The discriminator must hold. + +**Expected status**: + +- D: skeleton — wired call sites, no data flow. Whether the + unblock path is upstream hooks, middleware, or kernel + observation is implementer's call. +- S: partial. `RelayObservability` exists on the host side per the + prior upgrade (`S-B1`); the sim's relay vertex itself does not + emit lifecycle events as engine-synthesized records. +- T: absent. + +**Close criterion**: a deployed bundle from a run that included a +peer dial failure attributable to a relay-side close names the +close-initiator and reason in the relay's bundle contribution. A +sim `RelayKill` scenario produces the matching event stream. + +### 2.4 Inference response-leg instrumentation + +**Source**: postmortem §"Data-collection / deployment gaps +surfaced by this run" item 5. + +**Gap**: The `1779733878` postmortem's conclusion — "last stage +could not deliver the response" — was inferred from dial timeouts +plus the absence of an inbound `InferenceResponse`, not from a +typed event on the last stage saying "I tried to send the response +and the send outcome was X." The chain `stage-(N-1) +→ InferenceResponse → orchestrator's inbox` has no event on the +sending side. A typed event makes attribution a one-line read +rather than a triangulation. + +**Contract — what closing the gap looks like**: + +- D: the production stage actor, on attempting to send an + `InferenceResponse` upstream, emits a typed event naming the + target peer, the request id the response corresponds to, the + byte size, and the send outcome. The outcome discriminator is + the iroh-level result the transport returns (succeed / timeout + / connection-closed / refused / unresolved / queued-but-not- + acked-in-budget). The post-processor surfaces these in + `summary.md` under a section that names which inference + request was answered by which stage's send and how that send + resolved. +- S: the sim's stage host kind grows a minimal inference + message surface (`InferenceRequest` inbound to stage-0, + `InferenceResponse` outbound from stage-(N-1), forwarded + between adjacent stages as opaque payload in the MVP). The + stage host emits the same typed response-send event when it + attempts the outbound to the orchestrator. The codec contract + (`SIM_SPEC.md §3.3`) carries the inference messages with + byte-equality between sim and prod encoding. +- T: a scenario where the orchestrator's inbound path is broken + via `RelayPeerConnDown` on the last leg (stage-(N-1) → orch) + while every other leg works produces a bundle whose last + stage emits exactly one `InferenceResponseSent` event with + `send_outcome` in the failure-discriminator set. A scenario + where every leg works produces an `InferenceResponseSent` + with `send_outcome=success` and a matching + `InferenceResponseReceived` (or equivalent) on the + orchestrator's side. + +**Expected status**: + +- D: absent. The current stage actor's send call is not wrapped + in a typed diagnostic event for the response leg. +- S: absent. The sim's stage host kind today produces no `Send` + actions during its lifecycle (`SIM_SPEC.md §6A.5` notes this + explicitly and defers inter-stage traffic to a later revision). + Closing this coverage moves that deferral forward. +- T: absent. + +**Close criterion**: a deployed bundle from any run where the +response did not return names the send outcome of the last +stage's response attempt in a single event. A sim scenario +modeling the same failure produces the same shape. + +### 2.5 Bundle serve hardening under run-id reuse + +**Source**: postmortem §"Bundle recovery" caveat; postmortem +§"Data-collection / deployment gaps surfaced by this run" item 2. + +**Gap**: When a run id is reused across the failed-first-lease / +successful-second-lease shape the `1779733878` run exhibited, a +finalize record from the first phase pins a stale canonical +bundle in the collector's cache. A subsequent `GET` serves the +stale 5.3 KB bundle instead of synthesizing the rich 9.3 MB one +from current staging. Two adjacent quirks: `finalize_received` +stays `true` after the on-disk `finalize-*.json` is deleted, and +the synthesized manifest still lists a removed node directory. + +**Contract — what closing the gap looks like**: + +- D: the collector's `download_bundle` handler prefers the + *richer* of {canonical-cached, synthesized-from-current-staging} + by a size or node-count heuristic, or rebuilds canonical when + staging has grown past the cached bundle's manifest. Deleting a + node directory from staging clears the corresponding finalize + record from in-memory state. The synthesized manifest reflects + the current on-disk state, never a stale in-memory record. The + `finalize_received` boolean is sourced from the same place the + serve decision is sourced from — a single source of truth, not + two diverging caches. +- S: not applicable. The sim writes bundles directly to a + destination directory; there is no serve logic, no finalize + cache, no run-id reuse semantics. The coverage threads through + D only. +- T: not applicable as a *sim test*. The discriminator (stale vs + fresh serve on a finalize-then-staging-growth sequence) is a + collector unit-test concern living under + `crates/distribution/tests/`, not a scenario the sim engine + can express. The implementing agent should land the collector + test alongside the D-layer change; it is named here so that the + coverage's verification surface is honest about where it lives. + +**Expected status**: + +- D: absent. Current serve logic prefers cached canonical + unconditionally when `finalize_received` is true. +- S: not applicable. +- T: collector unit test absent. + +**Close criterion**: a collector unit test writes two phases of +staging with an intervening finalize, deletes the first-phase +node, and verifies the second `GET` serves the richer bundle and +that the cleared node does not appear in the manifest. + +--- + +### 2.6 Per-SWIM-probe RTT and observed latency distribution + +**Source**: postmortem §"SWIM churn and relay events" (1701 +SwimTransitions over ~7 min, all with `conn_type=Relay`); postmortem +§"UDP echo probes" (tier-2 RTTs spread 181–405 ms; SWIM probes +ride a relay-mediated path on top of these); `SWIM_TUNING_REPORT.md` +§6 limit 3 ("SWIM host adapter does not emit `probe_sent` / +`probe_received` / `probe_timed_out` events"). + +**Gap**: The bundle has tier-2 UDP-echo RTT to docean:9081 — a +host-level surface that does not represent the latency SWIM +actually sees. SWIM rides a relay-mediated peer connection whose +RTT is at least one extra hop and is subject to relay-side HOL +queueing under load. The bundle currently exposes: + +- per-snapshot iroh counters (cumulative `MessageSent` / + `MessageReceived`), +- aggregate `SwimTransition` counts, +- per-peer dial outcomes (`Timeout` / `Success` rollup), + +but it does not expose per-probe RTT, per-peer RTT distribution +over the run window, or correlation between +`probe_timed_out`-class outcomes and observed RTT spikes. Without +this surface, SWIM tuning is a guess against the deploy's actual +latency distribution rather than a measurement. + +This gap also mirrors the simulator's own limit per +`SWIM_TUNING_REPORT.md` §6.3: the SWIM host adapter does not emit +the probe lifecycle events, so the §10 evaluator's +`no_flap_while_probes_ok` is structurally `Inconclusive`. Closing +the gap on both sides closes the assertion's precondition. + +**Contract — what closing the gap looks like**: + +- D: each SWIM ping/ack pair emits a typed event naming the + observer, target, virtual-or-wall send time, virtual-or-wall + receive time, the resulting RTT, and the discriminator + (`success` / `timeout` / `connection-closed` / etc.). The + post-processor surfaces a `## Probe RTT distribution` section + with median, p95, p99 per (observer, target) pair, plus per + five-second bucket so degradation over time is visible. A + `probe_timed_out` outcome carries the configured timeout + budget alongside the observed RTT (where one exists) so a + reader sees "probe missed a 3 s budget by 200 ms" vs "no + response within 3 s, never arrived." +- S: the simulator's SWIM host adapter emits the same probe + lifecycle events. Per `SIM_SPEC.md §9.2` parity, the schema is + identical to D's. This is the §6.3 limit from + `SWIM_TUNING_REPORT.md` closing simultaneously with D — the + bundle reader cannot tell a sim run from a prod run by this + surface. +- T: a scenario with a declared per-link latency distribution + (heavy-tailed, peer-symmetric) produces a bundle whose + postproc RTT section's median, p95, p99 fall within stated + tolerance of the scenario's declared distribution. A scenario + with a `LatencySpike` mutation produces a bundle whose RTT + section shows the spike at the mutation time. The + precondition for `no_flap_while_probes_ok` is now satisfied; + the assertion moves off `Inconclusive` for every scenario + using a SWIM-host kind. + +**Expected status**: + +- D: absent. No per-probe event today. +- S: absent. `SWIM_TUNING_REPORT.md` §6.3 names this explicitly. +- T: absent. + +**Close criterion**: a deployed bundle's postproc summary names +the median / p99 RTT per (observer, target) and a sim bundle +produces the matching surface. `no_flap_while_probes_ok` resolves +to `Pass` or `Fail` (not `Inconclusive`) on every SWIM scenario in +the calibration library. + +**Downstream**: this coverage is the data surface +`N3_SWIM_TUNING_SPEC.md` consumes. SWIM tuning itself is +downstream of collection and lives in that sibling document. + +--- + +## 3. Out of scope + +- **Inference protocol surface beyond the response leg.** Coverage + 2.4 instruments the response-send event. A full inference- + protocol event stream (microbatch routing, KV cache, per-stage + worker activity) is broader than what the `1779733878` postmortem + could not answer; it belongs in a separate spec when a + postmortem demands it. +- **Post-processor summary enhancements.** SWIM transition + distributions, per-(observer, target, reason) breakdowns, + cross-node temporal alignment around the moment of failure — + these are renderer concerns, not collection concerns. They + presuppose the data is in the bundle; this spec is about the + data. +- **Orchestrator-topology fixes.** The `1779733878` postmortem's + item 6 names the root cause as a NAT'd local orchestrator with + no reachable port. That is a deployment-shape question for the + runbook, not a collection-coverage question. +- **Runbook fixes.** The `--gpu RTX_4090` vs `RTX 4090` line in + `DEPLOYMENT_TEST.md` (postmortem item 7) is a runbook bug, not a + collection gap. +- **Sim coverage of upstream-blocked surfaces.** If + `iroh_relay::server` continues to expose no session hooks, the + sim's relay vertex can model the lifecycle events the contract + requires, but the production D layer of coverage 2.3 may remain + partial. That partiality is a structural blind spot to file per + the established blind-spot discipline; this spec does not + resolve it. + +--- + +## 4. References + +- `N3_POSTMORTEM_2026-05-25_1779733878.md` — the second + 2026-05-25 deployment's postmortem. §"Observability upgrade + scorecard" is the source for coverages 2.1, 2.2, 2.3; §"Data- + collection / deployment gaps surfaced by this run" items 1–5 map + to coverages 2.1, 2.5, 2.2, 2.3, 2.4 respectively. +- `N3_SIM_TEST_BATTERY_SPEC.md` — the sim-test battery spec. The + battery's families A (relay peer-conn down) and the discriminator + it builds against the relay-port probe (coverage 2.2) and the + relay session lifecycle (coverage 2.3) consume the data this + spec lands. +- `crates/simulation/SIM_SPEC.md` — the simulator's behavioral + surface. §3.3 codec contract, §5A relay vertex, §6A stage host + kind, §9 bundle layout are the load-bearing references for the + S-layer contracts. +- `crates/simulation/SWIM_TUNING_REPORT.md` — the prior tuning + pass against simulated 60 ms latency. §6 limits (especially + §6.3 "SWIM host adapter does not emit `probe_sent` / + `probe_received` / `probe_timed_out` events") are the source + for the S-layer of coverage 2.6. +- `N3_SWIM_TUNING_SPEC.md` — the downstream spec that consumes + coverage 2.6's data surface to retune SWIM against the + observed `1779733878` latency distribution. Sibling document. diff --git a/examples/pipeline-parallel-inference/N3_DATA_GAPS.md b/examples/pipeline-parallel-inference/N3_DATA_GAPS.md deleted file mode 100644 index 2e8f1fb..0000000 --- a/examples/pipeline-parallel-inference/N3_DATA_GAPS.md +++ /dev/null @@ -1,241 +0,0 @@ -# N=3 data-coverage gaps - -Companion to `N3_POSTMORTEM_2026-05-25.md`. Where the postmortem -documents what we *do* know about the failure, this doc is about the -things we *don't* — and why we should care. Input for the -data-collection upgrade. - -The framing is investigator-first: each gap is named for the question -we couldn't answer, not the file that doesn't emit the field. - -## The investigation we couldn't finish - -Walking back from the symptom — orchestrator's relay-mediated path to -stage-2 died at ~5 s, never recovered, stage-2 went silent — the chain -of questions we'd want to answer is roughly: - -1. Did stage-2's underlying relay *tunnel* to docean stay up, or did - it drop too? -2. If the tunnel stayed up, why didn't iroh re-establish the - peer-to-peer path? -3. If the tunnel dropped, who closed it (relay vs. stage-2's iroh vs. - the OS), and why? -4. Was stage-2's host network actually broken at that moment, or was - this a software-level failure on a working network? -5. Independent of all of the above: why did stage-2 never start its - Python worker, when stage-0 and stage-1 both did within seconds? - -We could not answer **any** of these from the bundle. Each one is -blocked by a specific missing data source. - -## The gaps, ranked by how much they hurt this investigation - -### 1. The relay is a black box - -The biggest single hole. `swactor-iroh-relay` on docean produced -nothing that ended up in the bundle: no session log, no metrics -scrape, no log tail, no record of which node connected, when, how -long, and what closed each session. - -The orchestrator's local cache says -`last_failure_reason: "connection-closed"`. That string is iroh's -report of what *iroh* observed at the application layer. It doesn't -tell us whether the relay terminated the session, whether the QUIC -stack on either end did, or whether a NAT mapping expired and the -relay noticed first. - -> **What this blocks:** distinguishing a relay-side eviction from an -> endpoint-side close from a path-level timeout. Three very different -> root causes, indistinguishable in the bundle. - -### 2. Relay session and peer connection are conflated - -`body.iroh.metrics.socket.relay_home_change` is a counter that -increments when a node changes its home relay. `num_conns_opened` and -`num_conns_closed` are counters for iroh peer connections. None of -these tell us, per moment, whether a given node's **tunnel to its -relay** is up. - -This matters because of the asymmetry we hit: from stage-2's view -nothing closed (counters quiescent, `relay_home_change: 1` for the -whole run), but the orchestrator-side cache shows the connection -through the relay dying after 5 s. We have no way, from stage-2's -data alone, to say whether its relay tunnel was actually still alive -when the peer connection died. - -> **What this blocks:** answering "did stage-2's tunnel survive?" — -> the question that decides whether we're looking at a network -> problem or an iroh state-machine problem. - -### 3. No event when a relay path is established, lost, or replaced - -We have snapshot counters but no event stream for relay-path -transitions. `RelayChanged` event count across all four nodes for the -whole run: zero. If iroh internally noticed and recovered a relay -session inside one snapshot interval, we'd never see it. If iroh -*didn't* notice a dead session, we equally can't see that. - -This is the "no log line for the interesting moment" problem. The -counter says the final state; we want the transitions. - -> **What this blocks:** correlating the moment of failure with what -> iroh thought was happening. Right now the only event-stream -> evidence is the orchestrator's connect-timeout retries, which is a -> downstream symptom. - -### 4. The Python worker subprocess is invisible until it emits - -Stage-2 emitted zero `worker_starting` and zero `worker_ready` -events. Stage-0 and stage-1 emitted both within seconds of boot. -Whatever happened to stage-2's worker — never spawned, spawned and -crashed before its first event, spawned but blocked — left no trace -in our bundle. Stage-2's node process was clearly alive (23 -snapshots, 38 event batches), so it isn't a node-process crash. - -We don't capture: -- the moment the stage actor decides to spawn the worker -- the subprocess pid, exit code, or stderr tail -- whether the stage actor was *gating* worker spawn on something - (cluster membership? a peer dial?) that never happened - -This is a separate failure from the relay flap, possibly with a -common upstream cause, possibly not. We can't tell. - -> **What this blocks:** deciding whether to focus the fix on -> transport, on the stage actor's startup ordering, or on worker -> launch itself. - -### 5. We don't know what host stage-2 was on - -`boot.json` carries `container_id`, `datacenter_id`, `host_country`, -`host_ip_public`, `hostname`, `home_relay_url_at_boot`, `git_sha`, -`iroh_version` — all null except `hostname`, which is a Docker short -id. The orchestrator already has the public IP, datacenter id, and -country for each rental at the point `lease_chain` returns. None of -that is forwarded into the container or persisted into the boot -snapshot. - -So when we say "stage-2's vast.ai rental had a hostile NAT," we -literally cannot point at the machine. We can't re-rent the same host -to reproduce, we can't compare it against the hosts that *did* work, -we can't even tell you which country it was in. - -> **What this blocks:** any kind of fleet-level statistics across -> runs ("which datacenters fail more often"), and the ability to -> reproduce the bad rental. - -### 6. Iroh introspection is computed against the wrong API version - -The `iroh_api_missing` event reports `iroh_version: "0.96"` as a -literal string. The lockfile is `iroh 0.98.2`. The list of -"missing" fields is whatever was missing in 0.96 — we have no idea -what 0.98 actually exposes, because we never checked. - -So when stage-2's snapshot reports -`observed_conn_type_at_last_use: "None"`, we don't know whether -that's "iroh told us None" or "we couldn't read the field because -we're holding a 0.96 shape against a 0.98 struct." - -> **What this blocks:** trusting any of the per-peer iroh state in -> the bundle. This is corrosive — it undermines the whole iroh -> tier of evidence. - -### 7. Bundle assembly is finalize-or-nothing - -The collector only writes `MANIFEST.json` and the tarball when the -orchestrator sends a finalize record. SIGKILL skipped that, so -`GET /diag/bundle/` returned 404. The bundle we analyzed -was hand-reconstructed from staging files we got to before the -collector's TTL cleaned them up. - -A real operator hitting a real production incident is going to kill -things ungracefully. The "we got lucky" failure mode here is bad -enough that we should treat the staging directory as the source of -truth and have finalize be an optimization, not a precondition. - -> **What this blocks:** any incident bundle from a hard-killed run. - -### 8. Reachability probes only cover one port - -We probe UDP echo to `:9081` on docean. Stage-2 timed out 1 of 12. -We don't probe `:7843` (the relay's actual port). So when the relay -session dies, we can't say "but the host could still reach the relay -port at that moment" — only "but the host could still reach a -different port on the same machine." - -> **What this blocks:** ruling out transport-level reachability as -> the cause of relay session death. - -### 9. No event-level breakdown of dials by peer - -We have `DialStarted: 83` and `DialOutcome: 80` as raw event counts. -The 3-event drift is not attributed to a specific peer in -`summary.md`. With three peers it's easy enough to grep manually, -but the summary should be doing this for us, especially at higher N -where per-peer asymmetry is the whole story. - -> **What this blocks:** at-a-glance answer to "which peer was hard -> to reach," which is the first question for any cluster failure. - -### 10. No gossip-arrival evidence on the silent node - -Stage-2's `peers[]` contained only the orchestrator. We don't know -whether stage-2 received `NameRegistry` gossip about its siblings -and failed to dial, or never received the gossip at all. The bundle -has `MessageReceived: 72` for stage-2 but the breakdown isn't -recorded. - -> **What this blocks:** distinguishing a control-plane failure -> (gossip didn't arrive) from a data-plane failure (dials based on -> gossip didn't connect). - -### 11. No kernel-level network counters - -`/proc/net/snmp`, `/proc/net/udp`, per-interface drop counts — none -captured. For stage-2, with 537 holepunch attempts and 5 reported -mapping failures, we can't tell "iroh sent and the OS dropped it" -from "iroh sent and the OS accepted it and the path silently lost -it." These are at the edge of what's worth collecting — modest cost -per snapshot, but the cases where they matter are real. - -> **What this blocks:** distinguishing iroh-layer pathology from -> host-network pathology when the two look identical from above. - -## What this looks like in priority order - -If we only get to fix a few of these for the next deployment: - -**Must-have to investigate another N=3 failure:** -- gap 1 (relay-side data) -- gap 4 (worker subprocess visibility) -- gap 5 (host metadata forwarding) -- gap 7 (bundle assembly without finalize) -- gap 6 (iroh API version sanity check) - -**Strong-have:** -- gap 3 (relay-path transition events) -- gap 2 (relay-tunnel-state field, separable from peer state) -- gap 10 (gossip-receipt event) - -**Nice-to-have:** -- gap 8 (relay-port probe) -- gap 9 (per-peer dial rollup in summary) -- gap 11 (kernel counters) - -The "must-haves" are the ones where, looking back at this bundle, -the absence actually prevented a conclusion. The rest would have -made the investigation faster but weren't strictly load-bearing. - -## What this implies for the sim - -A separate concern that overlaps: most of these gaps are real-network -gaps that the sim doesn't model at all. The sim doesn't have a -relay, doesn't model NAT-mapping behavior, doesn't model -relay-session-up-but-peer-connection-down asymmetry, and doesn't -distinguish kernel-level packet loss from iroh-level path failure. - -If we want the sim to reproduce a failure like this one, the data -model the sim exposes has to be at least as rich as the data the -postmortem needed to read — otherwise "we reproduced it in sim" -won't actually mean we understand it. Whatever fields we add to the -bundle should land in the sim's per-tick state too. diff --git a/examples/pipeline-parallel-inference/N3_DEPLOYMENT_REPORT.md b/examples/pipeline-parallel-inference/N3_DEPLOYMENT_REPORT.md deleted file mode 100644 index 4f93783..0000000 --- a/examples/pipeline-parallel-inference/N3_DEPLOYMENT_REPORT.md +++ /dev/null @@ -1,357 +0,0 @@ -# N=3 vast.ai deployment — investigation report - -Session date: 2026-05-20. Branch: `ds-inference`. - -## TL;DR - -After three live runs against vast.ai, the original "deployment hangs at SWIM -convergence" failure decomposes into **three independent bugs stacked**: - -| Layer | What it is | Status | -|---|---|---| -| A | iroh 0.96's `RelayMode::Default` routes through n0's experimental canary cluster, which buffers SWIM gossip for 100+ seconds | **fixed** in-session by running our own iroh-relay on the VPS | -| B | Our SWIM impl flaps via gossip even when probes succeed; self-incarnation runs away (228 self-refutes in 7 min); names never propagate; one peer ends `Dead` despite being healthy | **open bug**, fix path discussed below | -| C | The `pp_tinygrad_worker.py` (or pp-gpu-node monitoring it) crashes ~50-200s into the stage's life with the real worker; never captured the actual exit reason | **separate open bug**, blocked on diagnostic visibility | - -Stub mode (`PP_WORKER_STUB=1`) bypasses Layer C. With own-relay + stub, stages -stay alive the full 7 min — proving stage death is the worker, not the cluster. -The cluster *still* fails to resolve `pp-entry` because of Layer B. - -## Where we started - -- Branch `ds-inference` had a runbook (`VASTAI_STATUS.md`) listing 8 prior failed - attempts at N≥3 on vast.ai. -- Diagnostics scaffolding was already in place: `swactor-diag-collector` - (HTTP+UDP receiver), per-node introspection, post-processor. -- Tests were green and the collector binaries built cleanly. The runbook's open - questions all lived at the iroh / SWIM layer. - -## What we ran - -### Step 0 — collector on the VPS - -`swactor-diag-collector` static-musl build → `scp docean:~/` → -`nohup … --bind 0.0.0.0:9080 --root /var/lib/swactor-diag --udp 0.0.0.0:9081`. -Opened UFW 9080/tcp, 9081/udp. Verified end-to-end: HTTP 404 on `/`, UDP echo -returns 15B. - -### Run #1 — canary relay (baseline) - -``` -SWACTOR_DIAG_RUN_ID=vastai-N3-1 -SWACTOR_DIAG_COLLECTOR_URL=http://146.190.110.128:9080 -SWACTOR_DIAG_UDP_ECHO=146.190.110.128:9081 -# no custom relay → RelayMode::Default -``` - -Result: `failed to resolve pp-entry` after the 300s resolve deadline; total run -425s. - -Critical signal from the bundle's timeline: - -``` -t=556598 orch sends SWIM Ack to stage-0 (9870 B over relay) -t=741287 orch's last successful Ping/Ack with stage-0 -t=743981 stage-0 finally receives 4 backlogged pings (187 seconds late) -``` - -The canary relay (`euc1-1.relay.n0.iroh-canary.iroh.link.`) was buffering -SWIM messages for **187 seconds**. SWIM probe_timeout is 15s — the cluster -fell apart inside the first probe round. - -Mechanism check: in iroh-0.96, `RelayMode::Default` invokes -`prod::default_relay_map()`. That function literally returns the canary URLs -(`crates/distribution/.../iroh-0.96.1/src/defaults.rs:30`). There is no -"production" iroh relay cluster in this version — `prod` and `staging` are -two named-but-equally-experimental n0 deployments. Setting -`IROH_FORCE_STAGING_RELAYS=1` would only swap us to a different experimental -cluster, not a production one. - -### Mid-session fix — bring our own relay - -Built a standalone iroh-relay around `iroh_relay::server::Server::spawn`: - -- New binary `crates/distribution/src/bin/swactor-iroh-relay.rs`. -- Extended the existing `relay` Cargo feature to pull `tokio/macros` + - `tokio/signal` (needed by the bin's tokio runtime). -- Added `[[bin]]` entry with `required-features = ["relay"]`. - -Built static-musl, deployed to docean: `nohup … --bind 0.0.0.0:7843 ---public-host 146.190.110.128`. UFW 7843/tcp opened. Verified -`http://146.190.110.128:7843/` returns the `

Iroh Relay

` landing page. - -Plumbed an env-driven relay override through the stack: - -- `examples/pipeline-parallel-inference/src/relay_config.rs` — - `relay_mode_from_env()` returns `RelayMode::Custom(url)` when - `SWACTOR_IROH_RELAY_URL` is set, else `RelayMode::Default`. -- `pp_smoke_run.rs`, `pp_gpu_node.rs` — both binaries call - `relay_mode_from_env()` instead of hard-coding `RelayMode::Default`. -- `vastai::DiagEnv` — added `iroh_relay_url: Option` field, populated - by `DiagEnv::from_process_env()`. -- `vastai::create_instance` — injects `SWACTOR_IROH_RELAY_URL` into every - rented container's env payload. - -### Run #2 — own relay, real worker - -Same env as run #1 plus `SWACTOR_IROH_RELAY_URL=http://146.190.110.128:7843/`. -run_id `vastai-N3-2`, duration 412s. Same end state: `failed to resolve -pp-entry`. - -But the bundle's metrics tell a different story: - -| | Run #1 (canary) | Run #2 (own relay) | Run #3 (own relay + stub) | -|---|---|---|---| -| ConnectionCacheHit | 70 | 62 | **2783** | -| ConnectionCacheMiss | 24 | 16 | 5 | -| DialStarted | 67 | 43 | 8 | -| MessageSent | 77 | 69 | **2790** | -| SwimTransition | 33 | 28 | 1027 | -| Orchestrator events | 442 | 417 | 2936 | -| stage-0 events | 324 | 91 | **4141** | -| stage-0 lifetime | run | **75s** | **full run** | - -Run #2 still showed the connect-timeout pattern. The orch's timeline to -stage-0 has clean traffic for ~47s, then `ConnectionCacheInvalidated -reason=connection-closed`, then three 10s redial timeouts, then permanent loss. - -### The stage-death finding - -Per-node event timespans on run #2: - -``` -orchestrator 411s (full run) -stage-0 75s -stage-1 191s -stage-2 51s -``` - -Background diagnostic threads (`clock_sample`, `udp_echo`) keep emitting -regardless of iroh state. When they also stop, the process is gone. So stages -**were dying mid-run**, not just losing connectivity. The orchestrator's -"connect timeout" was failing because there was nothing on the other end. - -We're 8+ attempts in and never caught this before, because: - -- `register_name` doesn't emit a diag event — there's no way to tell from the - bundle whether stage-0 ever registered `pp-entry`. -- `StageActorStatus::ProcessExited` doesn't emit one either — so when the - Python worker dies and pp-gpu-node exits with status 1, the only record is in - the vast.ai container stdout, which we destroy along with the instance. -- The local name table isn't included in snapshots (`body.swim` has membership - + recent messages but not the registry). - -### Run #3 — stub mode - -To separate worker-crash from cluster bugs, plumbed `PP_WORKER_STUB` (plus -`PYTHON`, `MODEL`, `CUDA`, `MAX_TOKENS`) through `create_instance` so the -orchestrator's env passes through to every rented container. - -Result with `PP_WORKER_STUB=1`: - -- All four nodes alive the full 430s. -- 2783 ConnectionCacheHits, **8 total DialStarted across the whole run** (vs - 67 in run #1). -- Cluster *still* never resolves `pp-entry`. -- Orch's `self_incarnation` ends at **228** — meaning the orch refuted Suspect - claims about itself 228 times in 7 minutes. -- One peer (`c0a261b2`) ends `state=Dead` in the orchestrator's view at run end - despite all instances being demonstrably alive (verified by SSH). - -Mid-run SSH into stage-0 confirmed both `pp-gpu-node` (PID 345) and the python -worker (PID 402) were running and stage-0's stdout had thousands of `iroh -driver: received N message(s)` lines interleaved with -`SWIM: alive d11cc185` / `SWIM: suspect d11cc185` — i.e., the stage was -constantly flipping the orchestrator's status. - -## Discussion: how to fix SWIM - -The summary.md from run #2 caught the smoking gun: - -``` -First peer to go Dead: - stage-2 marked stage-1 Dead at t=… - reason: "suspicion-timeout" - observer side (stage-2): conn_type=Relay - peer side (stage-1): conn_type=unknown - observer probes_ok_at_transition=yes - peer probes_ok_at_transition=yes -``` - -Probes succeeded on both sides. Yet stage-2 marked stage-1 Dead. The -transition reason for most other state changes was `gossip` — meaning a third -party told us a peer was Suspect. - -Three sub-issues to address, roughly independent in difficulty: - -### B1 — the gossip flap loop (the real bug) - -Standard SWIM rule: "highest incarnation wins". When peer A claims -`Z=Suspect(incarn=10)` and peer B claims `Z=Alive(incarn=11)`, every receiver -should accept Alive(11) and discard Suspect(10). Z's own refute should bump -incarnation past any stale Suspect within one gossip round. - -Our self-incarnation reaching 228 in 420 seconds means roughly one refute every -1.8 seconds. That's far above the probe interval. Either: - -- The refute bump isn't being broadcast fast enough to outpace the next gossip - round, or -- The receiver-side incarnation comparison isn't strictly "newer wins" - (off-by-one, or accepts equal-and-Suspect over Alive), or -- Suspect/Dead gossip is being generated by peers who *themselves* haven't yet - seen the latest incarnation, and our impl doesn't suppress that. - -Next step: pick one Suspect→Alive→Suspect cycle in the run #3 timeline, -read `crates/distribution/src/swim/{node,probe}.rs` against it, identify -which branch of the gossip-receive code is mis-firing. - -### B2 — SWIM message bloat - -In run #1, individual SWIM Ack messages were **9.8 KB**, Pings up to 7.5 KB. -That's because membership gossip piggybacks on every probe. With our N=4 -cluster and substantial name-table state, the payloads grow into the multi-KB -range. - -Big payloads ⇒ head-of-line blocking on relay ⇒ probe latency spikes ⇒ probe -acks miss the timeout window ⇒ Suspect. - -Fix: split gossip into its own periodic burst (or piggyback only a bounded -slice). Smaller secondary issue but it amplifies B1. - -### B3 — timeouts vs WAN reality - -`probe_timeout=15`, `suspicion_timeout=60` are LAN-tuned. Across regions with -relay routing, p99 RTT can spike to 2-3s under load. The probe budget is fine -in normal weather but tight under bursts. - -**But** the summary explicitly says `probes_ok_at_transition: yes` — pings ARE -getting acked. The deaths are gossip-driven, not probe-driven. So this is the -*least* important of the three; fix B1 first. - -## Discussion: catching this in test, not production - -This loop cost ~$2 of vast.ai GPU rental and ~90 minutes of engineering time. -Almost none of the test value required real GPUs or a real vast.ai roundtrip — -it was a pure SWIM problem. Test-side priorities, cheap to expensive: - -### A. In-process SWIM simulator with injectable network params - -Run N SWIM cores in a single test process. Mock transport queues messages with -configurable latency, jitter, and loss. Property assertions like: - -- *"With 200ms ± 50ms latency + 5% packet loss, a 3-node cluster reaches - all-Alive within 30 seconds and stays Alive for 5 minutes."* -- *"After a 10-second partition + heal, name registrations re-replicate to - all peers within 30 seconds."* -- *"Self-incarnation never exceeds N + (failures observed) in a steady-state - cluster."* - -`<1 second per iteration`. The repo already has -`crates/distribution/src/swim/` as a unit — likely just needs a sim harness -plus property tests. **Would have caught our exact bug.** Highest leverage -single thing we can build. - -### B. Docker-compose harness with `tc netem` - -Three containers on the laptop, real iroh + real relay over loopback, -`tc qdisc add dev eth0 root netem delay 100ms 30ms loss 1%` per container. -End-to-end including the relay protocol. ~30 seconds per iteration; good for -CI nightly. Complements A — A catches logical bugs, B catches integration -issues. - -### C. Stage-side diagnostic emission gaps - -Three small additions (<100 lines total) that would have cut today's debug -loop in half: - -1. Emit a `Custom("register_name")` event whenever `register_name` is called, - carrying `(name, addr, peer_node_id)`. -2. Include the local name table in each snapshot (currently `body.swim` has - membership + recent messages but no `name → addr` mapping). -3. Emit a `Custom("worker_exited")` event with status / signal **before** - `std::process::exit(1)` in `wait_for_worker_ready` and friends. - -Run #1's investigation would have ended in 2 minutes instead of 90. - -### D. `pp-shell ` helper - -A one-liner CLI that uses the run_id to query collector metadata, finds the -matching vast.ai instance from contract IDs, and SSHes in with pp-gpu-node's -stderr piped to the local terminal. We did this manually with `curl + python + -ssh`; bundling it saves 5 minutes every time anyone wants to look at a live -stage. - -## Potential next steps - -Ordered by leverage / cost. Picking 1–3 is probably enough to unblock -real N≥3 deployment. - -1. **Build option A (in-process SWIM simulator + property tests).** Catches B1 - immediately and is reusable for every future regression. Needs the SWIM - core to be transport-agnostic — verify by reading - `crates/distribution/src/swim/`; refactor if needed. - -2. **Ship option C (three diagnostic-emission additions).** Cheap and - compounds. Every future live debug benefits. Worth doing *before* - investigating Layer C so we can capture what kills the worker. - -3. **Fix B1 (the gossip flap).** With the simulator in place, develop - test-first: write the property test that captures the observed pathology, - then change SWIM until it passes. Reading `swim/node.rs` and - `swim/probe.rs` is the entry point. - -4. **Investigate Layer C (tinygrad worker crashes).** Requires step 2 OR a - live SSH-in during a fresh real-worker run. The crash is most likely in - model loading — `pp_tinygrad_worker.py` probably wants a `MODEL` env it's - not getting, or tinygrad's CUDA backend is failing on the rented GPU. - -5. **Option B (docker-compose harness) + option D (pp-shell helper).** Nice to - have once we're back to spending time on live-cluster work. - -## Artifacts produced this session - -Uncommitted changes on `ds-inference`: - -- `crates/distribution/src/bin/swactor-iroh-relay.rs` — new standalone relay - binary. -- `crates/distribution/Cargo.toml` — extended `relay` feature with tokio - macros/signal; added `[[bin]] swactor-iroh-relay`. -- `examples/pipeline-parallel-inference/src/relay_config.rs` — new module, - `relay_mode_from_env()`. -- `examples/pipeline-parallel-inference/src/lib.rs` — exposed `relay_config`. -- `examples/pipeline-parallel-inference/src/bin/pp_smoke_run.rs` — uses - `relay_mode_from_env()` instead of hard-coded `RelayMode::Default`. -- `examples/pipeline-parallel-inference/src/bin/pp_gpu_node.rs` — same, with - precedence over the prior `seed_relay_env` heuristic. -- `examples/pipeline-parallel-inference/src/vastai.rs` — - `DiagEnv.iroh_relay_url` field, `is_enabled()` updated, env passthrough for - `PP_WORKER_STUB` / `PYTHON` / `MODEL` / `CUDA` / `MAX_TOKENS` in - `create_instance`, and relay-URL injection. -- `examples/pipeline-parallel-inference/tests/t_vastai.rs` — updated - `DiagEnv` struct literal for the new field. - -Bundles on docean (`/var/lib/swactor-diag/bundles/`): - -- `vastai-N3-1.tar.gz` — canary baseline, real worker. -- `vastai-N3-2.tar.gz` — own relay, real worker (stages die at 51–191s). -- `vastai-N3-stub.tar.gz` — own relay, stub worker (stages live full run; SWIM - still fails to settle). - -Local extracted bundles: - -- `/tmp/bundle.out` (run #1), `/tmp/bundle_v2.out` (run #2), - `/tmp/bundle_stub.out` (run #3). - -## Infrastructure state at end of session - -- **docean (146.190.110.128)** running: - - `swactor-diag-collector` on :9080/tcp + :9081/udp. - - `swactor-iroh-relay` on :7843/tcp (advertised - `http://146.190.110.128:7843/`). - - Both processes started under `nohup`, logs at - `/var/log/swactor-diag-collector.log` and - `/var/log/swactor-iroh-relay.log`. -- **vast.ai**: no instances running; all destroyed at end of each run. -- **Local docker image**: `zacheryasc/swactor-pp-gpu:latest` (sha256:9d2cd3…) - contains the most recent pp binaries with env passthrough + custom relay - support. Pushed to Docker Hub. diff --git a/examples/pipeline-parallel-inference/N3_OBSERVABILITY_UPGRADE_SPEC.md b/examples/pipeline-parallel-inference/N3_OBSERVABILITY_UPGRADE_SPEC.md deleted file mode 100644 index 670e7f6..0000000 --- a/examples/pipeline-parallel-inference/N3_OBSERVABILITY_UPGRADE_SPEC.md +++ /dev/null @@ -1,494 +0,0 @@ -# N=3 observability upgrade — behavioral spec - -Sister doc to `N3_DATA_GAPS.md`. The gaps doc says *what's missing -and why we care*. This doc says *what the system must do once the -gaps are closed.* - -Each section is a behavior contract: requirements the running -system has to satisfy after the work is done. Implementation -strategy — which crate, which file, which trait — is left to the -person picking up the work, except where a pattern is load-bearing -to the contract itself (the subprocess introspector is the one -explicit pattern requirement, called out below at the user's -direction). - -Throughout: every "the bundle contains X" claim is testable. A -post-deployment run that doesn't satisfy these is a failed upgrade. - -## Cross-cutting requirements - -1. **Additive evolution.** A node running new code emits bundles - that a post-processor built against old code can still parse — - missing fields are absent, not malformed. Symmetrically, a - post-processor built against new code reads an old bundle by - showing the new fields as "absent" rather than erroring. - -2. **Separation of lifecycle from state.** Anything that has a - "moment it happened" is an event on the event stream. Anything - that has a "current value" is a snapshot field. The same fact - should not be reported both ways unless one is a counter and - the other is a transition. - -3. **Schema-version honesty.** Any version string the bundle - carries about a dependency must reflect the dependency actually - linked at build time. The bundle never contains a version - string that disagrees with the lockfile. - -4. **Generic over the use case.** Tier-3 capture surfaces (process, - subprocess, host, etc.) are wired the same way as the existing - `ProcessIntrospector`: a trait on the aggregator with a default - production implementation and the ability to install a test - fake without going through production paths. A new caller of - `swactor` should be able to opt into the new surfaces with no - knowledge of how data flows out. - -5. **Boundary stays where it is today.** Generic observability - primitives live in the distribution crate's diagnostics module. - Role-specific decisions (which PIDs to register, which probes - to install, which labels to use) live in the calling crate - (`examples/pipeline-parallel-inference/...` for this codebase). - ---- - -## 1. Relay observability (gap 1) - -After this work, the bundle answers, for every relay-mediated -peer connection that died during a run: - -- Who initiated the close: the relay, the remote node, or an idle - timeout. -- What the close reason was, in a short string the relay assigned. -- How long the session had been open and how many bytes had - crossed in each direction. -- The relay's own count of active sessions, opens, closes, and - bytes transferred at end-of-run, broken down by close reason. - -The bundle reader can answer "was this a relay-side eviction" -without consulting any external system, by reading the relay's -report and correlating it against the node-side -`connection_cache[peer].last_failure_reason` already in the -bundle. - -The post-processor's summary surfaces this correlation per peer -in a "relay sessions" section. When the relay was not observed -(legacy run, relay observability not configured), the section -renders one line explaining that and pointing at this gap. - -Acceptance: replay the 2026-05-25 incident with a new bundle. -The summary tells you who closed stage-2's session and why, -without further digging. - ---- - -## 2. Relay-session vs. peer-connection separation (gap 2) - -After this work, every snapshot a node emits carries an explicit -answer to "is my tunnel to my relay healthy right now," separate -from "do my peer connections through that tunnel work." - -The field carries: -- The relay URL the node is currently using. -- A status (connected / connecting / disconnected / unknown). -- Wall-clock millis of the last status change and the moment the - current status was entered. -- The last moment the node successfully sent over the tunnel and - the last moment it received over it. -- Lifetime byte counters in each direction. - -When the underlying transport library does not expose enough state -to populate the field truthfully, the snapshot must say so -explicitly: the status is `unknown`, a discriminator field -identifies the value as derived rather than reported, and the -existing `iroh_api_missing` event pattern records the gap by name. -A bundle reader must never have to guess whether `unknown` means -"the tunnel is unknown" vs. "we couldn't ask." - -Acceptance: in the 2026-05-25 bundle's stage-2 snapshots, this -field reports either a real status ("disconnected" or "connected") -or `unknown` with `status_source: derived`. The investigator can -distinguish "tunnel alive but peer connection dead" from "tunnel -itself died" without speculation. - ---- - -## 3. Per-transition relay events (gap 3) - -After this work, every relay-related state flip produces an event -on the event stream, in addition to whatever counter increments. - -Two kinds of flips are observable: -- **Relay session state changed**: the tunnel status field from - section 2 moved between values. Event carries the relay URL, - from-status, to-status, and a short reason string when one is - available. -- **Relay home changed**: the node switched which relay it - considers home. Event carries the from-URL and the to-URL. - -Counters (e.g. `relay_home_change`) are retained for sanity-check -totals, but the per-transition event is the authoritative source. -A bundle reader can reconstruct the relay-state timeline of a -node by replaying the event stream, with no need to derive -transitions from counter deltas across snapshots. - -Acceptance: in any run where a node experiences a relay flap, the -event stream contains at least one `RelaySessionStateChanged` -record. A grep for that event kind across the bundle tells you -which nodes flapped and when, with no other inputs. - ---- - -## 4. Subprocess introspector (gap 4) — generic, through swactor - -This is the largest section. The user's explicit requirement: -**the Python worker introspection must flow through swactor in a -generic way, like the existing process crate does** — meaning it -is not specific to "the Python worker" or "this example crate," -but a reusable surface that any future user of `swactor_process` -can opt into. - -### Behavior contract - -After this work, every subprocess that a node owns via -`swactor_process` is reflected in the bundle on two channels, -identically to how the parent process is reflected today: - -- **As snapshot state**: each periodic snapshot carries a - per-subprocess entry with the subprocess's caller-supplied - label, PID, parent PID, status (running / exited / unknown), - spawn time, exit time and code/signal when applicable, RSS, - virtual size, open FD count, CPU time, and a truncated - command line. -- **As lifecycle events**: a `SubprocessSpawned` event fires when - the subprocess starts, and a `SubprocessExited` event fires when - it ends. Both carry the caller's label, the PID, the command, - and (for exit) the exit code or terminating signal and uptime - in millis. - -Subprocess capture is a tier-3 surface alongside the existing -process-stats one. It is installed via an introspector trait on -the aggregator, with the same install pattern as today's -`ProcessIntrospector`, `HostIntrospector`, etc. A test can wire a -fake introspector without going through any production code path. - -The capture surface is **stage-agnostic** and **worker-agnostic**: -it knows about a PID, a label, and a parent. The fact that "the -Python worker" is one such subprocess is a decision made at the -calling site, not in the introspector. - -### Wiring contract — the swactor side - -The `swactor_process` driver, when it spawns a child, must -publish the child's PID through its existing notification -channel. The data flow looks like: - -1. The owning actor calls into `swactor_process` to spawn. -2. `swactor_process` reports the spawn outcome back through its - existing notification mechanism, with the PID included. -3. The owning actor forwards "this PID, this label" into the - subprocess introspector it owns. -4. The owning actor forwards "this PID has exited with this - status" into the introspector on exit. - -The actor's role in step 3-4 is intentionally minimal — a handful -of lines wrapping notifications it already receives. The -introspector does the actual `/proc` reading, lifecycle-event -emission, and snapshot population. A future swactor user gets -subprocess observability by installing the introspector at boot -and forwarding two notification kinds; nothing else. - -### Lifecycle event coverage - -The pre-existing ad-hoc `Custom { kind: "worker_starting" }` and -`Custom { kind: "worker_exited" }` strings in the example crate -are replaced by the typed `SubprocessSpawned` and -`SubprocessExited` events. The role-specific signal "the -subprocess has produced its first protocol output and is -functioning" (currently `worker_ready`) stays a `Custom` event -because functioning-as-a-pipeline-worker is not a generic -subprocess concept. - -### What this gives us for the next investigation - -For a stage that didn't start its worker, the bundle now tells us -unambiguously which of three things happened: - -- The actor never reached its `on_start` and the subprocess was - never asked to spawn. No `SubprocessSpawned`. The bug is in - actor scheduling. -- The subprocess spawned and exited immediately. Both events - present, with exit code and the existing stderr tail available. - The bug is in the subprocess itself. -- The subprocess spawned and stayed alive but never produced - protocol output. `SubprocessSpawned` present, no - `SubprocessExited`, no `worker_ready` Custom event, and the - per-snapshot RSS/CPU on the subprocess show whether it's stuck - or thrashing. The bug is in the subprocess's startup logic - before its first protocol line. - -These three were indistinguishable in the 2026-05-25 bundle. -They are immediately distinguishable after this work. - -Acceptance: in any future deployment, a stage that fails to -produce inference output can be classified into one of those -three buckets by reading the bundle alone. - ---- - -## 5. Host metadata forwarding (gap 5) - -After this work, every node's boot record carries the physical -host context the node is running on: - -- Public IP of the rental. -- Datacenter id and country reported by the cloud provider. -- The provider's identifier for the rental (e.g. vast.ai instance - id) — enough to re-rent or correlate against provider-side - logs. -- The hostname as the container sees it. -- The relay URL the node was configured with at boot. -- The git SHA the binary was built from. -- The version string of the underlying transport library, taken - from what is actually linked (see gap 6). - -When a node runs outside the orchestrator's lease flow (e.g. a -locally-launched node for development), the cloud-provider fields -are absent rather than blank or wrong. The bundle reader can tell -"this node was not on vast.ai" from "this node was on vast.ai but -metadata wasn't forwarded" — the former leaves fields absent, the -latter is no longer a possible state. - -The post-processor's summary lists each node's host context one -line per node, so "which rental was stage-2" is answerable -without grep. - -Acceptance: replay the 2026-05-25 incident's recovery process. -Identifying stage-2's host requires reading one line of the -summary, not cross-referencing provider records. - ---- - -## 6. Iroh API version sanity (gap 6) - -After this work: - -- The `iroh_api_missing` event reports the version of the - transport library actually linked into the binary. The version - string is sourced from the build, not a literal. -- Every tier-2 transport snapshot carries the same version string - as a field, so a bundle reader does not need to scan the event - stream to know what version the node ran. -- The list of "API gaps" — fields the bundle reader should treat - as "we couldn't ask" rather than "we asked and got zero" — - reflects what the linked version actually omits. Upgrading to a - version that exposes a previously-missing field causes the gap - to disappear from the bundle automatically; no code change is - needed to recompute the list. - -Acceptance: bumping the iroh dependency to a version that exposes -`conn_type` produces a bundle whose `api_gaps` no longer mentions -`conn_type`, without any other change. - ---- - -## 7. Bundle assembly without finalize (gap 7) - -After this work: - -- A bundle is retrievable for any run that has at least one boot - record in staging, regardless of whether the orchestrator sent - a finalize record. `GET /diag/bundle/` succeeds in both - cases. -- The retrieved bundle's manifest explicitly states whether - finalize was received. Bundle readers must not have to guess. -- When finalize was received, the bundle is the canonical one and - serving it is cheap. When it wasn't, the bundle is synthesized - at request time from staging files; the latency is fine because - unfinalized bundles are by definition retrieved during incident - response. -- Staging files for runs that never finalized are retained at - least until the operator has had a reasonable window to - retrieve them (default: 30 days), bounded by a hard - disk-space cap that trims oldest-first when exceeded. - -The hand-rolled recovery process used for the 2026-05-25 incident -(tar staging from the collector, scp it down, reshape, retar) is -no longer needed for any future incident, regardless of how the -orchestrator died. - -Acceptance: kill an orchestrator with SIGKILL mid-run. A subsequent -`GET /diag/bundle/` returns a usable bundle with -`finalize_received: false` in its manifest. - ---- - -## 8. Relay-port reachability probe (gap 8) - -After this work, every node periodically attempts a transport-level -reachability check against the relay's actual port, and reports -the outcome in the same snapshot probe array as the existing UDP -echo. The probe's existence does not require operator -configuration: when the node has been told a relay URL, the relay -probe is automatically registered. - -The probe's outcome distinguishes: -- Reached and responded ("ok"). -- Reached, no response within deadline ("timeout"). -- Host reachable, port closed ("refused"). -- Could not resolve target ("unresolved"). -- Other error ("error"). - -A bundle reader can answer "could stage-2 reach the relay port at -moment T" by reading stage-2's probe array around T, without -inferring reachability from a different probe to a different port -on the same host. - -Acceptance: a node placed behind a firewall that blocks the relay -port but not the existing UDP echo port produces a bundle in -which the relay probe consistently reports `refused` or `timeout` -while the UDP echo continues to report `ok`. - ---- - -## 9. Per-peer dial rollup in summary (gap 9) - -After this work, the post-processor's summary contains, per peer -in the run, a row listing: - -- Total dials started against that peer. -- Total successful dials. -- Total failed dials. -- The last dial outcome (string) and its wall-clock millis. - -The 3-event drift in the 2026-05-25 bundle (`DialStarted: 83`, -`DialOutcome: 80`) is attributable to specific peers in the -table; the reader can immediately tell which peers' dials never -completed. - -This is a pure post-processor change — the raw events are already -in the bundle. No new fields, no new events. - -Acceptance: re-run the post-processor against the existing -2026-05-25 bundle. The summary contains a per-peer dial table -that accounts for all 83 `DialStarted` events. - ---- - -## 10. Gossip-receipt event (gap 10) - -After this work, every time a node receives a payload through the -gossip / dissemination layer — name-registry update, SWIM -membership piggyback, anything similar — it emits a typed event -on its event stream. The event carries the source peer, the -payload kind (string, extensible), the payload size in bytes, and -the number of items inside. - -The existing coarse `MessageReceived` counter remains for backward -compatibility, but the new event is the authoritative source for -"did node X ever hear about name Y from peer Z." - -The post-processor's summary, per node, reports the total receipt -counts broken down by payload kind. "Stage-2 never received any -name-registry gossip from anyone" is a one-line answer. - -Acceptance: in any run where one node fails to learn about -another node's registered name, the bundle distinguishes -unambiguously whether the gossip was never received vs. received -and ignored. - ---- - -## 11. Kernel network counters (gap 11) - -After this work, every host-scrape snapshot carries kernel-level -UDP and per-interface counters: - -- UDP-side: aggregate packets in/out, drops attributable to - no-listening-port, packets discarded due to errors, packets - lost to socket buffer overflow. -- Per-interface: rx/tx bytes, rx/tx dropped, rx/tx errors. - -A bundle reader can compute deltas across consecutive snapshots -to attribute packet loss to one of three layers: -- "Iroh sent and the OS dropped it" — UDP send error counters - rise on the sender. -- "OS sent it and the path silently lost it" — sender counters - clean, receiver counters clean. -- "It arrived and got dropped at the receiver's NIC" — receiver - interface drop counters rise. - -All counters are best-effort: absent on non-Linux hosts, absent -when the file can't be read, never silently zero. The -post-processor's summary surfaces any node whose UDP-drop or -interface-drop deltas are non-zero across the run window, so the -reader doesn't have to inspect every snapshot. - -Acceptance: a node deliberately subjected to UDP-drop-rate -injection produces a bundle whose summary highlights it with the -correct counter rising. - ---- - -## Sim cross-pollination - -The behavioral contracts above also constrain the simulator. A -node simulated by the sim should produce snapshots and events -that conform to the same shape as a real node — the bundle reader -should not be able to tell from the data shape alone whether a -given snapshot came from a real deployment or the sim. - -Three areas where today's sim lags this contract and must catch up -as part of the same upgrade: - -- The sim must model a relay actor whose behavior produces the - same tunnel-status field (gap 2) on simulated nodes. Without - this, sim runs of cluster scenarios are not bundle-shape - compatible with real ones. -- The sim must support installing a subprocess introspector fake - (gap 4). Scenarios that want to model "a stage's worker never - came up" wire this fake to produce a `SubprocessSpawned` with - no following `worker_ready` Custom event. -- The sim's network failure model must allow "tunnel up, - peer-connection-via-tunnel down" as a distinct failure case - from "tunnel down." Without it the sim cannot reproduce the - exact 2026-05-25 failure even after the observability lands. - -These are sim-side work, not data-collection work, but they -share the data model defined here. - ---- - -## Implementation order - -Grouped by independence. Within a group, work is parallel-safe; -across groups, later groups don't depend on earlier groups -*finishing*, only on earlier groups' contracts being agreed. - -**Group A — small, independent, unblock confidence elsewhere** -- 5 (host metadata) — small and pure-mechanical -- 6 (iroh version sanity) — small, but until it lands, every - iroh-side field in the bundle has a credibility asterisk -- 9 (per-peer dial rollup) — pure post-processor -- 11 (kernel counters) — additive host-scrape extension - -**Group B — relay tier** -- 1 (relay observability) — the largest single info gain -- 2 (relay-session field) — depends on having something to - populate it from, ideally the work in 1 -- 3 (relay events) — depends on 2's status field existing - -**Group C — subprocess tier** -- 4 (subprocess introspector + events) — independent of B, - parallel-safe with it - -**Group D — collector robustness** -- 7 (bundle without finalize) — independent of all the above; - land last to avoid churning the collector while other tiers - are still moving - -**Group E — polish** -- 8 (relay-port probe) — small, independent -- 10 (gossip-receipt event) — small, independent - -The 2026-05-25 investigation would have been closeable with -A + B + C alone. D + E reduce future investigation cost but -weren't load-bearing for the failure we hit. diff --git a/examples/pipeline-parallel-inference/N3_POSTMORTEM_2026-05-25.md b/examples/pipeline-parallel-inference/N3_POSTMORTEM_2026-05-25.md deleted file mode 100644 index 33e22f7..0000000 --- a/examples/pipeline-parallel-inference/N3_POSTMORTEM_2026-05-25.md +++ /dev/null @@ -1,290 +0,0 @@ -# N=3 vast.ai deployment post-mortem — 2026-05-25 - -Companion to `N3_DEPLOYMENT_REPORT.md` and `DEPLOYMENT_TEST.md`. Covers -one invocation of `pp-smoke-run --vastai --num-stages 3` on 2026-05-25 -(`vastai-N3-1779720002`). The cluster came up, lost one peer's relay -session ~5 s into SWIM convergence, never recovered, and was killed by -the operator at ~10 min. The orchestrator never produced an -`InferenceResponse`. The diagnostic bundle was recovered by hand (no -finalize record was written) and post-processed. - -## Cleanup note - -The run was terminated with `TaskStop` (SIGKILL). The orchestrator's -destroy-on-exit handler did not run. Three rentals (`37777187`, -`37777190`, `37777192`) were destroyed manually by -`DELETE /api/v0/instances//`. Post-cleanup instance count = 0. - -## Sequence - -3 instances leased (`37777187` → stage 0 / `95d01a36…`, `37777190` -→ stage 2 / `a040c0d2…`, `37777192` → stage 1 / `0cc5ed32…`). -Orchestrator node id `66b61b4a…`. All four nodes used -`SWACTOR_IROH_RELAY_URL=http://146.190.110.128:7843/` (docean), as -recorded in every node's `body.iroh.home_relay_url` field. - -Live log progression: - -``` -t=0 orchestrator boots, custom-relay banner emitted -t=~135s contract 37777187 (stage 0) reaches running, others follow -t=158s 3 contracts leased, "waiting for SWIM convergence (3 alive)" -t=~190s members ["0cc5ed32=alive", "95d01a36=alive", "a040c0d2=suspect"] - iroh driver: connect attempt N/3 to a040c0d2 failed: connect timeout - (repeated) -t=~340s stage-0 marks stage-2 (a040c0d2) Dead, reason "suspicion-timeout" -t=~420s stage-1 (0cc5ed32) also goes suspect from orchestrator's view -t=~600s members ["0cc5ed32=dead", "95d01a36=alive", "a040c0d2=dead"] -t=~600s operator killed the orchestrator (SIGKILL via TaskStop) -``` - -## Bundle recovery - -`GET /diag/bundle/vastai-N3-1779720002` returned HTTP 404. The -collector finalises tarballs only on receipt of a finalize record from -the orchestrator; SIGKILL skipped that step. Per-node staging files -under `docean:/var/lib/swactor-diag/vastai-N3-1779720002/` survived and -were retrievable by tar + scp. - -Recovery steps applied to produce a postproc-compatible bundle: - -1. Tar `/var/lib/swactor-diag//` from docean and copy down. -2. Synthesize `MANIFEST.json` from the four `boot-000001.json` records - (run_id, role, stage_index, node_id_hex; file counts via `ls -1`). -3. Reshape staging layout (flat `boot-NNN.json`, `events-NNN.json`, - `snapshot-NNN.json` under `/`) into bundle layout - (`