swactor/crates/simulation/scenarios/topology/partition_heal.toml
Zachery Aaron Shores-Chmielewski ae9ca3bcf3 feat: datastream feature cleaning
Promote pipeline-parallel-inference to a first-class app and consolidate observability on the datastream wire, decoupling the dashboard crate from `distribution`.

- apps/pipeline-parallel-inference: move the example out of `examples/` into `apps/` as its own workspace, rename binaries to `pp-worker`/`pp-orchestrator`, and strip release binaries
- cluster: add `ClusterNode`, a synchronous facade over the actorized distribution protocol (IrohDriver + per-node Runtime hosting Swim/Registry/Metadata/Directory actors with a `MembershipFanout`), replacing ad-hoc `driver.node()`/`tick()` call sites
- fleet: add per-node fleet telemetry that ships identity/resource records as `DatastreamFrame`s over the cluster transport to the orchestrator's `DatastreamSink`, folded into a `FleetView` on a 3s tick
- provision: add best-effort, opt-in SSH boot-phase telemetry (`PP_DEPLOY_KEY`) that streams rented-node boot logs onto the orchestrator's datastream as `proc.boot.<stage>.*`
- dashboard: rewire the crate dependency from `distribution` to `datastream`, drop the standalone `swactor-datastream-dashboard` binary, and rewrite `datastream_source.rs` to demux per-node frames into Overview/Distribution/Fleet views with live-node TTL filtering
- distribution: refresh dist/netmap plugin copy and README from "Kademlia routing" to gossip-directory terminology

Signed-off-by: Zachery Aaron Shores-Chmielewski <zacheryasc@gmail.com>
2026-06-09 13:29:07 +04:00

92 lines
2.3 KiB
TOML

# Partition the cluster into {a} | {b, c}, then heal. Asserts the cluster
# converges back to all-Alive within a bounded window AND that the isolated
# node's death is *resolved* — the Dead→Alive resurrection arc (Goal 3, "death
# is provisional"), not just the post-heal endpoint. `dead_reprobe_interval_ns`
# is non-zero so the partition-heal detector (SWIM_ACTOR_SPEC §9.8) actually
# re-probes the isolated node and lets it refute back to Alive.
# Expected verdict: pass-now.
name = "partition_heal"
seed = 7
duration_ns = 90_000_000_000
[default_tick]
period_ns = 200_000_000
[default_link]
latency_ns = 5_000_000
jitter_stddev_ns = 1_000_000
loss_prob_ppm = 0
reorder_prob_ppm = 0
bandwidth_bps = 100_000_000
cold_dial_penalty_ns = 50_000_000
cache_warm_after_ns = 100_000_000
cache_invalidate_after_idle_ns = 30_000_000_000
[[peers]]
id = "a"
kind = "swim"
initial_state = "alive"
kind_config = { probe_interval_ns = 1_000_000_000, suspicion_timeout_ns = 5_000_000_000, dead_reprobe_interval_ns = 1_000_000_000 }
[[peers]]
id = "b"
kind = "swim"
initial_state = "alive"
kind_config = { probe_interval_ns = 1_000_000_000, suspicion_timeout_ns = 5_000_000_000, dead_reprobe_interval_ns = 1_000_000_000 }
[[peers]]
id = "c"
kind = "swim"
initial_state = "alive"
kind_config = { probe_interval_ns = 1_000_000_000, suspicion_timeout_ns = 5_000_000_000, dead_reprobe_interval_ns = 1_000_000_000 }
[[links]]
from = "a"
to = "b"
[[links]]
from = "b"
to = "a"
[[links]]
from = "a"
to = "c"
[[links]]
from = "c"
to = "a"
[[links]]
from = "b"
to = "c"
[[links]]
from = "c"
to = "b"
[[mutations]]
kind = "partition"
at_ns = 10_000_000_000
peers_a = ["a"]
peers_b = ["b", "c"]
[[mutations]]
kind = "heal"
at_ns = 30_000_000_000
# Snapshots spanning the post-heal convergence window so `convergence_after`
# has data to evaluate (snapshots are only emitted at these declared times).
[[snapshots]]
at_ns = 45_000_000_000
[[snapshots]]
at_ns = 55_000_000_000
[[snapshots]]
at_ns = 60_000_000_000
[[assertions]]
kind = "convergence_after"
after_ns = 30_000_000_000
within_ns = 30_000_000_000
peers = ["a", "b", "c"]
# Goal 3 — death is provisional: the isolated node "a" is detected Dead during
# the partition, then resurrected to Alive after the heal (within the window).
[[assertions]]
kind = "dead_peer_resurrects_within"
peer = "a"
after_ns = 10_000_000_000
within_ns = 60_000_000_000