This commit is contained in:
Zachery Aaron Shores-Chmielewski 2026-05-30 11:51:22 +04:00
parent bd179dda2c
commit 2ea137a66f
34 changed files with 1517 additions and 615 deletions

View file

@ -4,7 +4,7 @@
!target/release/swactor-diag-postproc
!examples/single-gpu-inference/target/release/gpu-node
!examples/single-gpu-inference/tinygrad_worker.py
!examples/pipeline-parallel-inference/target/release/pp-gpu-node
!examples/pipeline-parallel-inference/target/release/pp-smoke-run
!examples/pipeline-parallel-inference/target/release/pp-worker
!examples/pipeline-parallel-inference/target/release/pp-orchestrator
!examples/pipeline-parallel-inference/pp_tinygrad_worker.py
!examples/pipeline-parallel-inference/pp_entrypoint.sh

View file

@ -125,6 +125,7 @@ pub const ACTOR_DETAIL_HTML: &str = r##"<!DOCTYPE html>
<a href="/actors" class="nav-link">Actors</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/datastore" class="nav-link">Datastore</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
</div>
@ -569,6 +570,7 @@ pub const ACTORS_HTML: &str = r##"<!DOCTYPE html>
<a href="/actors" class="nav-link active">Actors</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/datastore" class="nav-link">Datastore</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
<div class="header-right">
@ -1377,6 +1379,7 @@ pub const DASHBOARD_HTML: &str = r##"<!DOCTYPE html>
<a href="/actors" class="nav-link">Actors</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/datastore" class="nav-link">Datastore</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
<div class="header-right">
@ -1934,6 +1937,7 @@ pub const TOPOLOGY_HTML: &str = r##"<!DOCTYPE html>
<a href="/topology" class="nav-link active">Topology</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/datastore" class="nav-link">Datastore</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
</div>

View file

@ -229,13 +229,9 @@ impl IrohDriver {
match rt.block_on(start_embedded_relay(bind_addr, config.relay_public_ip)) {
Ok((server, url)) => {
let url_str = url.to_string();
eprintln!("Relay: embedded relay started at {url}");
(Some(server), Some(url_str), RelayMode::Custom(url.into()))
}
Err(e) => {
eprintln!("Relay: failed to start embedded relay: {e}, falling back");
(None, None, config.relay_mode)
}
Err(_) => (None, None, config.relay_mode),
}
}
None => (None, None, config.relay_mode),
@ -301,33 +297,18 @@ impl IrohDriver {
Some(auth) => auth.lock().unwrap().is_allowed(&node_id),
};
if !allowed {
eprintln!(
"iroh driver: rejected connection from unauthorized peer {}",
swactor::transport::hex_encode(&node_id.0[..4])
);
conn.close(0u32.into(), b"unauthorized");
continue;
}
// Route by negotiated ALPN
let negotiated_alpn = conn.alpn();
if negotiated_alpn == ALPN {
eprintln!(
"iroh driver: accepted SWIM connection from {}",
swactor::transport::hex_encode(&node_id.0[..4])
);
swim_buf.lock().unwrap().push((node_id, conn));
} else {
eprintln!(
"iroh driver: accepted non-SWIM connection from {} (ALPN: {})",
swactor::transport::hex_encode(&node_id.0[..4]),
String::from_utf8_lossy(negotiated_alpn),
);
other_buf.lock().unwrap().push((node_id, conn));
}
}
Err(e) => {
eprintln!("iroh driver: incoming connection error: {e}");
}
Err(_) => {}
},
None => break, // endpoint closed
}
@ -659,7 +640,6 @@ impl IrohDriver {
});
}
eprintln!("iroh driver: join attempt {attempt}/{max_attempts} connecting to {}...", seed_addr.id);
diagnostics.emit_event(DiagEvent::DialStarted {
peer: seed_node_id,
attempt,
@ -726,7 +706,6 @@ impl IrohDriver {
});
}
eprintln!("iroh driver: join attempt {attempt}/{max_attempts} connected to {}, sending...", seed_addr.id);
let send_result: Result<(), String> = async {
let mut send = conn.open_uni().await.map_err(|e| e.to_string())?;
let tag_len = (tag.len() as u32).to_be_bytes();
@ -740,7 +719,6 @@ impl IrohDriver {
match send_result {
Ok(()) => {
eprintln!("iroh driver: join attempt {attempt}/{max_attempts} sent to {}", seed_addr.id);
diagnostics.emit_event(DiagEvent::MessageSent {
peer: seed_node_id,
kind: tag.to_string(),
@ -764,10 +742,6 @@ impl IrohDriver {
return;
}
Err(e) => {
eprintln!(
"iroh driver: join attempt {attempt}/{max_attempts} send error to {}: {e}",
seed_addr.id
);
diagnostics.emit_event(DiagEvent::Error {
component: "iroh_driver".into(),
message: format!("join send error: {e}"),
@ -778,10 +752,6 @@ impl IrohDriver {
}
}
Ok(Err(e)) => {
eprintln!(
"iroh driver: join attempt {attempt}/{max_attempts} connect error to {}: {e}",
seed_addr.id
);
let outcome = classify_dial_error_str(&e.to_string());
diagnostics.emit_event(DiagEvent::DialOutcome {
peer: seed_node_id,
@ -792,10 +762,6 @@ impl IrohDriver {
continue;
}
Err(_) => {
eprintln!(
"iroh driver: join attempt {attempt}/{max_attempts} connect timeout to {}",
seed_addr.id
);
diagnostics.emit_event(DiagEvent::DialOutcome {
peer: seed_node_id,
attempt,
@ -817,7 +783,6 @@ impl IrohDriver {
updated_at: Instant::now(),
});
}
eprintln!("iroh driver: join failed after {max_attempts} attempts to {}", seed_addr.id);
});
}
@ -832,9 +797,6 @@ impl IrohDriver {
// Collect completed background join connections
{
let mut pending = self.pending_joins.lock().unwrap();
if !pending.is_empty() {
eprintln!("iroh driver: collecting {} pending join connection(s)", pending.len());
}
for result in pending.drain(..) {
self.connection_cache_tracker
.note_dial_success(result.node_id, wall_ms_now());
@ -864,7 +826,6 @@ impl IrohDriver {
let mut failure_targets: Vec<NodeId> = Vec::new();
for action in actions {
if let Err(e) = self.send_action(action) {
eprintln!("iroh driver: send error: {e}");
let target = action_target(action);
self.diagnostics.emit_event(DiagEvent::Error {
component: "iroh_driver".into(),
@ -882,9 +843,7 @@ impl IrohDriver {
let probe_actions = self.node.report_send_failure(target);
// Best-effort send of probe actions — no recursion on failure
for action in &probe_actions {
if let Err(e) = self.send_action(action) {
eprintln!("iroh driver: probe send error: {e}");
}
let _ = self.send_action(action);
}
}
}
@ -1230,10 +1189,6 @@ impl IrohDriver {
self.read_streams(&conn, remote_id, &mut messages).await;
}
if !messages.is_empty() {
eprintln!("iroh driver: received {} message(s)", messages.len());
}
(messages, new_connections)
}
@ -1250,8 +1205,7 @@ impl IrohDriver {
Ok((tag, payload)) => {
messages.push((tag, payload, remote_id));
}
Err(e) => {
eprintln!("iroh driver: read error: {e}");
Err(_) => {
break;
}
}
@ -1276,18 +1230,12 @@ impl IrohDriver {
match tag {
"swactor_dist::Ping" => match serde_json::from_slice::<Ping>(payload) {
Ok(msg) => self.node.handle_ping(msg.from, msg.sequence, &msg.piggyback),
Err(e) => {
eprintln!("iroh driver: decode Ping: {e}");
Vec::new()
}
Err(_) => Vec::new(),
},
"swactor_dist::Ack" => match serde_json::from_slice::<Ack>(payload) {
Ok(msg) => self.node.handle_ack(msg.from, msg.sequence, &msg.piggyback),
Err(e) => {
eprintln!("iroh driver: decode Ack: {e}");
Vec::new()
}
Err(_) => Vec::new(),
},
"swactor_dist::PingReq" => match serde_json::from_slice::<PingReq>(payload) {
@ -1295,44 +1243,29 @@ impl IrohDriver {
self.node
.handle_ping_req(msg.from, msg.target, msg.sequence, &msg.piggyback)
}
Err(e) => {
eprintln!("iroh driver: decode PingReq: {e}");
Vec::new()
}
Err(_) => Vec::new(),
},
"swactor_dist::JoinRequest" => {
match serde_json::from_slice::<JoinRequest>(payload) {
Ok(msg) => self.node.handle_join_request(msg.from),
Err(e) => {
eprintln!("iroh driver: decode JoinRequest: {e}");
Vec::new()
}
Err(_) => Vec::new(),
}
}
"swactor_dist::JoinResponse" => {
match serde_json::from_slice::<JoinResponse>(payload) {
Ok(msg) => self.node.handle_join_response(msg.members),
Err(e) => {
eprintln!("iroh driver: decode JoinResponse: {e}");
Vec::new()
}
Err(_) => Vec::new(),
}
}
"swactor_dist::IndirectAck" => match serde_json::from_slice::<IndirectAck>(payload) {
Ok(msg) => self.node.handle_indirect_ack(msg.target, msg.sequence, &msg.piggyback),
Err(e) => {
eprintln!("iroh driver: decode IndirectAck: {e}");
Vec::new()
}
Err(_) => Vec::new(),
},
other => {
eprintln!("iroh driver: unknown message type: {other}");
Vec::new()
}
_ => Vec::new(),
}
}

View file

@ -634,6 +634,7 @@ dependencies = [
"axum",
"crossbeam-queue",
"ctrlc",
"distribution",
"serde",
"serde_json",
"swactor",
@ -2655,6 +2656,7 @@ dependencies = [
"base64",
"dashboard",
"distribution",
"futures-util",
"iroh",
"libc",
"reqwest 0.12.28",
@ -2935,6 +2937,7 @@ dependencies = [
"bytes",
"encoding_rs",
"futures-core",
"futures-util",
"h2",
"http",
"http-body",
@ -2956,12 +2959,14 @@ dependencies = [
"sync_wrapper",
"tokio",
"tokio-native-tls",
"tokio-util",
"tower",
"tower-http",
"tower-service",
"url",
"wasm-bindgen",
"wasm-bindgen-futures",
"wasm-streams 0.4.2",
"web-sys",
]
@ -2998,7 +3003,7 @@ dependencies = [
"url",
"wasm-bindgen",
"wasm-bindgen-futures",
"wasm-streams",
"wasm-streams 0.5.0",
"web-sys",
]
@ -4227,6 +4232,19 @@ dependencies = [
"wasmparser",
]
[[package]]
name = "wasm-streams"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "15053d8d85c7eccdbefef60f06769760a563c7f0a9d6902a13d35c7800b0ad65"
dependencies = [
"futures-util",
"js-sys",
"wasm-bindgen",
"wasm-bindgen-futures",
"web-sys",
]
[[package]]
name = "wasm-streams"
version = "0.5.0"

View file

@ -11,24 +11,27 @@ swactor = { path = "../..", features = ["transport", "serde", "std"] }
swactor-process = { path = "../../crates/process" }
serde = { version = "1", features = ["derive"] }
serde_json = "1"
reqwest = { version = "0.12", features = ["json"] }
reqwest = { version = "0.12", features = ["json", "stream"] }
futures-util = "0.3"
tokio = { version = "1", features = ["full"] }
distribution = { path = "../../crates/distribution", features = ["iroh", "collector"] }
# Live runtime dashboard (default features = HTTP overview/actors/topology pages,
# served on localhost when PP_DASHBOARD is set). No tui/replay-viewer pulled in.
dashboard = { path = "../../crates/dashboard" }
# Live runtime dashboard (HTTP overview/actors/topology pages, served on
# localhost when PP_DASHBOARD is set). The `live-collector` feature pulls in the
# server-side vast.ai fold (`VastaiLivePlugin`/`fold`) so the orchestrator can
# fold the collector's record stream into the Fleet tab. No tui/replay pulled in.
dashboard = { path = "../../crates/dashboard", features = ["live-collector"] }
iroh = "0.98"
urlencoding = "2"
base64 = "0.22"
libc = "0.2"
[[bin]]
name = "pp-gpu-node"
path = "src/bin/pp_gpu_node.rs"
name = "pp-worker"
path = "src/bin/pp_worker.rs"
[[bin]]
name = "pp-smoke-run"
path = "src/bin/pp_smoke_run.rs"
name = "pp-orchestrator"
path = "src/bin/pp_orchestrator.rs"
[dev-dependencies]
wiremock = "0.6"

View file

@ -23,8 +23,8 @@ FROM ${BASE_IMAGE}
# Pipeline binaries (this crate's target/) + diagnostics binaries (the
# workspace-root target/). All are statically linked enough that the base
# stage's libc is all they need; the worker is pure Python.
COPY examples/pipeline-parallel-inference/target/release/pp-gpu-node /usr/local/bin/pp-gpu-node
COPY examples/pipeline-parallel-inference/target/release/pp-smoke-run /usr/local/bin/pp-smoke-run
COPY examples/pipeline-parallel-inference/target/release/pp-worker /usr/local/bin/pp-worker
COPY examples/pipeline-parallel-inference/target/release/pp-orchestrator /usr/local/bin/pp-orchestrator
COPY target/release/swactor-diag-collector /usr/local/bin/swactor-diag-collector
COPY target/release/swactor-diag-postproc /usr/local/bin/swactor-diag-postproc
COPY examples/pipeline-parallel-inference/pp_tinygrad_worker.py /usr/local/share/pp_tinygrad_worker.py

View file

@ -1,49 +0,0 @@
# PP N=12 Deploy — Session Report
## Problems encountered (trivial → blocking)
- **GPU filter too narrow** — exact `RTX 3060` match: stage 5 hit "no offers" mid-chain. Fixed by VRAM-based selection (`PP_GPU_MIN_RAM_MB`).
- **Offer churn** — `no_such_ask`: offers vanish between search and create.
- **API rate-limit (HTTP 429)** — `provision_stage` retried with zero backoff + no inter-stage pacing, burning the candidate pool. Fixed: 429 backoff + `PP_LEASE_PACE_MS`.
- **Worker crash `Code(2)` on all 12 stages** — two root causes, both invisible at first:
- `Tensor(str)` rejected by bundled tinygrad → needed `Tensor(Path(...))`.
- Image shipped only 4 CUDA headers; NVRTC needs the full set (`vector_types.h`). Dockerfile copied a hand-picked subset.
- **Worker stderr swallowed** — StageActor buffers it into a `worker_exit_detail` diag event that needs a collector; none configured → error vanished.
- **SSH auth** — `publickey` denied initially (propagation lag), then worked on 5/12 but **persistently failed on 7/12** (pp-gpu-node holds PID 1 via `exec`; vast key-injection never ran). `vastai attach`/reboot didn't fix it.
- **`vastai execute` unusable** — "Invalid command given" (restricted command set).
- **`--redeploy` blocked** — uses the same SSH/scp, so unusable on the 7 unreachable nodes.
- **Slow/stalled image pulls** — cheap Korea GTX-10-series hosts; one fully stalled (0 bytes), triggering Phase-2 **autoreplace churn**.
- **Autoreplace not disableable** in the running binary.
- **429 on relaunch** — teardown's 12 destroys consumed the budget; absorbed by the new backoff.
## Where we spent the most time
1. **~20 min blind on the silent resolve loop** — connect-timeout SWIM noise looked like the problem but was a red herring; workers had actually crashed instantly.
2. **Getting on a node to see the real error** — SSH flakiness, restricted `execute`, local docker repro, then manual on-node run.
3. **Run #2 image-pull waiting** — many heartbeat ticks on slow/stalled pulls + the re-download after teardown.
## Observability that was clunky / insufficient
- Worker stderr + Python traceback never reach the container log (no collector) — had to reproduce locally and SSH a node to see `Code(2)`'s cause. *(Fixed: pp-gpu-node now prints abnormal-exit stderr.)*
- Resolve loop emits **nothing per-stage** — orchestrator log is just SWIM gossip for up to 20 min; no per-stage worker-ready/download visibility.
- The rich SSE diag stream (`pp_download_progress`, `worker_exit_detail`) was dead — `SWACTOR_DIAG_COLLECTOR_URL` unset.
- vast exposes **no docker-pull %** — `status_msg` only says "Pulling from"; `disk_usage` = -1.
- Connect-timeout logs were prominent but cosmetic — actively misleading.
- Node-id→stage mapping had to be derived by hand from `stage_secrets`.
## Where interaction with the live deployment was limited
- SSH worked on only 5/12 nodes; no reliable shell on the rest.
- `vastai execute` restricted; couldn't run arbitrary diagnostics via API.
- `--redeploy` (the intended fix-forward path) depends on the same broken SSH → fix-forward on live nodes was effectively impossible; had to rebuild the image + re-lease.
- Couldn't pause/disable autoreplace or see/intervene in image-pull progress.
- During the "loading" (pull) phase there's no container, so no SSH at all on the node that mattered most.
## Other notes
- **Sharded fetch works** (~1.8 GB/stage, not 18 GB) — but a stale code comment claims the full GGUF is pulled, which misled diagnosis.
- Core bugs are fixed + validated on a real GPU (stage 0 → `ready`) and in the pushed image; the remaining blocker is purely **host quality** (slow-pull hosts), not code.
- Re-leasing fresh always re-pulls image + re-downloads model; the in-place cache advantage is lost on every teardown.
- Highest-leverage follow-ups: (1) configure a diagnostics collector, (2) emit per-stage resolve/download progress to the orchestrator log, (3) host-throughput preflight or stalled-pull fast-replace, (4) fix the onstart so vast SSH-key injection survives (don't `exec` over it).
## Fixes shipped this session
- `vastai.rs`: 429 backoff in `provision_stage` (find + create paths) and inter-stage pacing (`PP_LEASE_PACE_MS`, default 600ms).
- `pp_tinygrad_worker.py`: `Tensor(gguf_path)` → `Tensor(Path(gguf_path))`.
- `Dockerfile`: copy the full CUDA include set (with a `test -f vector_types.h` build guard) instead of 4 hand-picked headers.
- `stage_actor.rs`: mirror an abnormal worker exit's stderr tail + Python traceback to pp-gpu-node's own stderr (→ container log, collector-independent).
- Image rebuilt + pushed (`zacheryasc/swactor-pp-gpu:latest`, digest `ca373d02…`); both bug fixes validated on a real GPU node (stage 0 reached `ready`).

View file

@ -3,14 +3,14 @@
# The collector service runs in its own container with the bundles dir
# bind-mounted from the host so the harness can read the finalized
# tarball. Both the collector and the pp processes use the host network
# namespace, so the stage children (spawned by `pp-smoke-run` via
# namespace, so the stage children (spawned by `pp-orchestrator` via
# `docker-gpu-node.sh`) share localhost reachability with the
# collector — `SWACTOR_DIAG_COLLECTOR_URL=http://127.0.0.1:9080` works
# uniformly from every actor in the run.
#
# The host network choice mirrors the existing `docker-e2e.sh` shape:
# pp-smoke-run runs on the host (orchestrator) and each
# `pp-gpu-node` runs in its own container under `--network host`. The
# pp-orchestrator runs on the host (orchestrator) and each
# `pp-worker` runs in its own container under `--network host`. The
# collector container just adds one more service to that arrangement.
#
# Used by `scripts/docker-diag-e2e.sh`. Direct `docker compose up`
@ -24,7 +24,7 @@ services:
container_name: ${PP_DIAG_COLLECTOR_NAME:-pp-diag-collector}
network_mode: host
# Override the image's default entrypoint (pp_entrypoint.sh, which runs
# pp-gpu-node) so this container runs the collector instead. The
# pp-worker) so this container runs the collector instead. The
# diagnostics binaries ship in the same unified code image.
entrypoint: /usr/local/bin/swactor-diag-collector
command:

View file

@ -1,7 +1,7 @@
#!/usr/bin/env bash
# PID-1 supervisor for the pipeline-parallel runtime container.
#
# The worker used to run as PID 1 (`exec pp-gpu-node`), so SSH depended on
# The worker used to run as PID 1 (`exec pp-worker`), so SSH depended on
# vast's host-side helper winning a race against our exec, and a worker crash
# killed the whole container — no shell left to read the traceback. This script
# instead owns PID 1: it brings up sshd deterministically, runs the worker as a
@ -41,9 +41,9 @@ echo "pp-entrypoint: sshd up on :22" >&2
# ── Worker: run as a child, tee output to a file readable over SSH ───────────
WORKER_LOG=/var/log/pp-worker.log
echo "pp-entrypoint: launching pp-gpu-node (log -> $WORKER_LOG)" >&2
echo "pp-entrypoint: launching pp-worker (log -> $WORKER_LOG)" >&2
set -o pipefail
/usr/local/bin/pp-gpu-node 2>&1 | tee "$WORKER_LOG"
/usr/local/bin/pp-worker 2>&1 | tee "$WORKER_LOG"
code=${PIPESTATUS[0]}
# ── Operator runbooks (manual, over SSH) ─────────────────────────────────────
@ -51,23 +51,23 @@ code=${PIPESTATUS[0]}
#
# Worker hot-reload (no restart) — edit the Python in place, then SIGHUP:
# scp -P <port> pp_tinygrad_worker.py root@<host>:/usr/local/share/pp_tinygrad_worker.py
# ssh <host> 'kill -HUP $(pidof pp-gpu-node)'
# pp-gpu-node tears down its worker and re-execs the on-disk script; the
# ssh <host> 'kill -HUP $(pidof pp-worker)'
# pp-worker tears down its worker and re-execs the on-disk script; the
# swactor process (and SWIM membership) stays up across the swap.
#
# Swactor-binary swap — stop the binary, stage the new one, re-exec under
# PID 1's env (preserves STAGE/SEED_ADDR/PP_STAGE_SECRET so the node id is
# unchanged). `.new` staging avoids ETXTBSY on the mapped ELF:
# ssh <host> 'pkill -x pp-gpu-node' # drops to the hold below
# scp -P <port> pp-gpu-node root@<host>:/usr/local/bin/pp-gpu-node.new
# ssh <host> 'mv -f /usr/local/bin/pp-gpu-node.new /usr/local/bin/pp-gpu-node && \
# chmod +x /usr/local/bin/pp-gpu-node && \
# ssh <host> 'pkill -x pp-worker' # drops to the hold below
# scp -P <port> pp-worker root@<host>:/usr/local/bin/pp-worker.new
# ssh <host> 'mv -f /usr/local/bin/pp-worker.new /usr/local/bin/pp-worker && \
# chmod +x /usr/local/bin/pp-worker && \
# setsid bash -c "while IFS= read -r -d \"\" kv; do export \"\$kv\"; done \
# < /proc/1/environ; exec /usr/local/bin/pp-gpu-node" \
# < /proc/1/environ; exec /usr/local/bin/pp-worker" \
# >/var/log/pp-restart.log 2>&1 </dev/null &'
#
# ── Crash policy: do NOT restart. Keep PID 1 / sshd alive for postmortem. ────
echo "pp-entrypoint: pp-gpu-node exited with code $code; NOT restarting (node held for postmortem)" >&2
echo "pp-entrypoint: pp-worker exited with code $code; NOT restarting (node held for postmortem)" >&2
echo "pp-entrypoint: --- last 40 lines of $WORKER_LOG ---" >&2
tail -n 40 "$WORKER_LOG" >&2 || true
exec sleep infinity

View file

@ -4,7 +4,7 @@
# across the binaries and one operator's shell. Copy it to set up a run:
#
# cp profiles/example.env profiles/local.env # untracked; put real secrets here
# PP_PROFILE=profiles/local.env cargo run --bin pp-smoke-run -- --seed ...
# PP_PROFILE=profiles/local.env cargo run --bin pp-orchestrator -- --seed ...
#
# When PP_PROFILE is unset, profiles/local.env is loaded automatically if it
# exists. The run scripts pick this up too (the binaries load it at startup).
@ -25,27 +25,27 @@
#SWACTOR_IROH_RELAY_URL=https://relay.example.com
# --- Topology ---------------------------------------------------------------
# Number of pipeline stages (>= 2). pp-smoke-run also accepts --num-stages.
# Number of pipeline stages (>= 2). pp-orchestrator also accepts --num-stages.
#NUM_STAGES=2
# --- Compute target (vast.ai / docker) --------------------------------------
# Container image to run on each node. The run scripts already read PP_IMAGE;
# pp-smoke-run's --image default now reads it too. Set to your registry tag.
# pp-orchestrator's --image default now reads it too. Set to your registry tag.
#PP_IMAGE=swactor-pp-gpu:latest
# GPU class requested when leasing on vast.ai (pp-smoke-run --gpu overrides).
# GPU class requested when leasing on vast.ai (pp-orchestrator --gpu overrides).
#PP_GPU=RTX 3060
# --- Workload (what each stage computes) ------------------------------------
# Model identifier handed to the worker.
#MODEL=
# Python worker script + interpreter (per-stage compute). WORKER_SCRIPT is the
# pp-gpu-node default; pp-smoke-run --worker ships a script to remote nodes.
# pp-worker default; pp-orchestrator --worker ships a script to remote nodes.
#WORKER_SCRIPT=./pp_tinygrad_worker.py
#WORKER_CMD=python3
#PYTHON=python3
# Stub mode: skip the real worker, echo activations (fast local smoke runs).
#PP_WORKER_STUB=1
# Inference request prompt + token budget (pp-smoke-run --prompt/--max-tokens).
# Inference request prompt + token budget (pp-orchestrator --prompt/--max-tokens).
#MAX_TOKENS=64
# --- Timeouts (seconds; sane defaults baked in — override only if needed) ---

View file

@ -1,7 +1,7 @@
# Local seed-mode smoke run: everything on loopback, stub workers, no relay.
# Drives one request through a 2-stage chain without a GPU or external relay.
#
# PP_PROFILE=profiles/local-seed.env cargo run --bin pp-smoke-run -- --seed
# PP_PROFILE=profiles/local-seed.env cargo run --bin pp-orchestrator -- --seed
#
# (No SWACTOR_IROH_RELAY_URL: seed mode wires nodes via direct loopback addrs.)

View file

@ -1,7 +1,7 @@
# WAN run on leased vast.ai GPUs. Copy to profiles/local.env and fill in the
# relay URL + image; keep that copy untracked.
#
# PP_PROFILE=profiles/local.env cargo run --bin pp-smoke-run -- \
# PP_PROFILE=profiles/local.env cargo run --bin pp-orchestrator -- \
# --vastai --api-key "$VASTAI_API_KEY"
#
# All nodes home onto one operator-controlled relay so they can hole-punch /
@ -13,6 +13,10 @@ SWACTOR_IROH_RELAY_URL=https://relay.example.com
NUM_STAGES=2
PP_IMAGE=swactor-pp-gpu:latest
PP_GPU=RTX 3060
MODEL=
# GPU filters are optional. Leave both unset to let any verified 1-GPU offer
# qualify (a 1B model fits anywhere). Set PP_GPU="RTX 3060" to pin a model, and/or
# PP_GPU_MIN_RAM_MB=8000 to require a VRAM floor.
# llama3.2:1b -> HF Llama-3.2-1B-Instruct-Q6_K.gguf (16 blocks, sharded across
# the N stages by compute_layer_range; validated key baked into the image).
MODEL=llama3.2:1b
MAX_TOKENS=64

View file

@ -1,17 +1,23 @@
#!/usr/bin/env bash
# demo-fleet.sh — one-command local mock of a vast.ai fleet, watchable live.
#
# Brings up, from a single command, a self-contained demo that streams the
# REAL host metrics each container measures (no synthetic data) into a live
# browser dashboard:
# Brings up, from a single command, a self-contained demo that mirrors the
# production topology: a diagnostics collector running "off-box" (in prod, a
# VPS) and the orchestrator running locally and hosting the FULL swactor
# dashboard. The orchestrator's dashboard shows its own live swactor process
# info (overview / actors / topology / distribution / netmap) and a Fleet tab
# that pulls the vast.ai + host metrics remotely from the collector:
#
# - swactor-diag-collector on the HOST (HTTP 9080 + UDP 9081), serving the
# live fleet board at http://127.0.0.1:9080/dashboard
# - pp-smoke-run on the HOST in --seed mode, spawning N pp-gpu-node
# containers (one per stage) via docker-gpu-node.sh, each on --network host
# - swactor-diag-collector on the HOST (HTTP 9080 + UDP 9081) — the "VPS"
# sink. Each stage's in-VM monitor ships REAL host_sample + log records
# (no synthetic data) here; the orchestrator pushes its distribution
# snapshot here too. Its own fleet board stays at http://127.0.0.1:9080/dashboard
# - pp-orchestrator on the HOST in --seed mode (PP_DASHBOARD on), spawning N
# pp-worker containers (one per stage) via docker-gpu-node.sh, each on
# --network host, and serving the full dashboard at http://127.0.0.1:9095/
# - PP_HOLD=1 keeps the cluster up after the first drive, so every stage's
# in-VM monitor keeps shipping host_sample + log records (~every 5s) and
# the dashboard animates in real time.
# in-VM monitor keeps shipping records (~every 5s) and the dashboard
# animates in real time.
#
# Ctrl+C (or any exit) tears everything down: stage containers, collector,
# orchestrator, and all temp files.
@ -31,6 +37,7 @@
# PP_MAX_TOKENS decode token cap (default: 4)
# PP_BIND_HOST collector bind host (default: 127.0.0.1)
# PP_PORT collector HTTP port (default: 9080)
# PP_DASHBOARD_PORT orchestrator dashboard HTTP port (default: 9095)
# PP_NO_OPEN if set, don't try to open the dashboard in a browser
# PP_DIAG_NETWORK docker network for stages (default: host)
set -euo pipefail
@ -43,9 +50,13 @@ MAX_TOKENS="${PP_MAX_TOKENS:-4}"
BIND_HOST="${PP_BIND_HOST:-127.0.0.1}"
PORT="${PP_PORT:-9080}"
UDP_PORT=$((PORT + 1))
DASH_PORT="${PP_DASHBOARD_PORT:-9095}"
CONTAINER_PREFIX="demo-fleet-stage"
RUN_ID="demo-fleet-$(date +%s)"
DASH_URL="http://${BIND_HOST}:${PORT}/dashboard"
# The full swactor dashboard is served by the orchestrator at "/"; the collector
# keeps its own standalone fleet board at :PORT/dashboard.
DASH_URL="http://${BIND_HOST}:${DASH_PORT}/"
COLLECTOR_URL="http://${BIND_HOST}:${PORT}"
if ! [[ "$NUM_STAGES" =~ ^[0-9]+$ ]] || [ "$NUM_STAGES" -lt 2 ]; then
echo "demo-fleet: N must be an integer >= 2 (seed mode needs >=2 stages), got '$NUM_STAGES'" >&2
@ -57,26 +68,36 @@ fi
if ! docker info >/dev/null 2>&1; then
echo "demo-fleet: docker daemon unreachable" >&2; exit 2
fi
# Fail loudly on a clash for the orchestrator dashboard port. Its HTTP server is
# spawned on the driver's tokio runtime and `.expect()`s its bind; a collision
# panics that task silently and the run carries on with no dashboard. Catch it
# here so the user can pick a free one.
if (exec 3<>"/dev/tcp/${BIND_HOST}/${DASH_PORT}") 2>/dev/null; then
exec 3>&- 3<&-
echo "demo-fleet: dashboard port ${DASH_PORT} is already in use." \
"Pick a free one: PP_DASHBOARD_PORT=9096 $0 ${NUM_STAGES}" >&2
exit 2
fi
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
CRATE_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
WORKSPACE_DIR="$(cd "$CRATE_DIR/../.." && pwd)"
SMOKE_RUN_BIN="$CRATE_DIR/target/release/pp-smoke-run"
GPU_NODE_BIN="$CRATE_DIR/target/release/pp-gpu-node"
ORCHESTRATOR_BIN="$CRATE_DIR/target/release/pp-orchestrator"
WORKER_BIN="$CRATE_DIR/target/release/pp-worker"
WORKER_PY="$CRATE_DIR/pp_tinygrad_worker.py"
COLLECTOR_BIN="$WORKSPACE_DIR/target/release/swactor-diag-collector"
# ── Step 1: build release artifacts ───────────────────────────────────────
if [ -z "${PP_SKIP_BUILD:-}" ]; then
echo "demo-fleet: cargo build pp-smoke-run + pp-gpu-node (release)"
echo "demo-fleet: cargo build pp-orchestrator + pp-worker (release)"
cargo build --manifest-path "$CRATE_DIR/Cargo.toml" --release \
--bin pp-gpu-node --bin pp-smoke-run
--bin pp-worker --bin pp-orchestrator
echo "demo-fleet: cargo build swactor-diag-collector (release, --features collector)"
cargo build --manifest-path "$WORKSPACE_DIR/Cargo.toml" --release \
-p distribution --features collector --bin swactor-diag-collector
fi
for f in "$SMOKE_RUN_BIN" "$GPU_NODE_BIN" "$WORKER_PY" "$COLLECTOR_BIN"; do
for f in "$ORCHESTRATOR_BIN" "$WORKER_BIN" "$WORKER_PY" "$COLLECTOR_BIN"; do
[ -f "$f" ] || { echo "demo-fleet: missing $f (run without PP_SKIP_BUILD)" >&2; exit 1; }
done
@ -129,7 +150,7 @@ cleanup() {
trap cleanup EXIT
trap 'exit 130' INT TERM
# ── Step 3: collector on the host (serves the dashboard) ───────────────────
# ── Step 3: collector on the host (the off-box "VPS" metrics sink) ─────────
echo "demo-fleet: starting collector on ${BIND_HOST}:${PORT} (root=$COLLECTOR_ROOT)"
"$COLLECTOR_BIN" --bind "${BIND_HOST}:${PORT}" --root "$COLLECTOR_ROOT" \
--udp "${BIND_HOST}:${UDP_PORT}" >"$COLLECTOR_LOG" 2>&1 &
@ -148,38 +169,31 @@ until (echo > "/dev/tcp/${BIND_HOST}/${PORT}") >/dev/null 2>&1; do
done
echo "demo-fleet: collector ready"
# ── Step 4: announce + open the dashboard ──────────────────────────────────
echo
echo " ┌─────────────────────────────────────────────────────────────┐"
echo " │ Live fleet dashboard: $DASH_URL"
echo " └─────────────────────────────────────────────────────────────┘"
echo
if [ -z "${PP_NO_OPEN:-}" ]; then
if command -v xdg-open >/dev/null 2>&1; then (xdg-open "$DASH_URL" >/dev/null 2>&1 &) || true
elif command -v open >/dev/null 2>&1; then (open "$DASH_URL" >/dev/null 2>&1 &) || true
fi
fi
# ── Step 5: orchestrator (holds the cluster open, real metrics enabled) ─────
# ── Step 4: orchestrator (hosts the full dashboard, holds the cluster open) ─
# stdin is the FIFO; we hold its write end open on fd 3 so hold_open() never
# sees EOF and the cluster stays up until we tear down.
exec 3<>"$FIFO"
echo "demo-fleet: launching orchestrator + ${NUM_STAGES} stage containers (run_id=$RUN_ID)"
# The orchestrator (and, via inheritance, docker-gpu-node.sh) read these from
# the environment. PP_HOLD keeps the cluster up; the SWACTOR_DIAG_* vars enable
# each stage's in-VM monitor so real host metrics ship to the collector.
# the environment. PP_HOLD keeps the cluster up; PP_DASHBOARD makes the
# orchestrator host the full swactor dashboard locally; the SWACTOR_DIAG_* vars
# point each stage's in-VM monitor at the collector (the off-box sink) and give
# the orchestrator the same URL to push its distribution snapshot to and to pull
# the fleet model from for its Fleet tab.
export PP_HOLD=1
export PP_WORKER_STUB=1
export PP_DEV=CPU
export PP_IMAGE="$IMAGE"
export PP_CONTAINER_PREFIX="$CONTAINER_PREFIX"
export SWACTOR_DIAG_COLLECTOR_URL="http://${BIND_HOST}:${PORT}"
export PP_DASHBOARD=1
export PP_DASHBOARD_PORT="$DASH_PORT"
export SWACTOR_DIAG_COLLECTOR_URL="$COLLECTOR_URL"
export SWACTOR_DIAG_RUN_ID="$RUN_ID"
export SWACTOR_DIAG_SPOOL_DIR="$SPOOL_DIR"
export SWACTOR_DIAG_UDP_ECHO="${BIND_HOST}:${UDP_PORT}"
[ -n "${PP_GPUS:-}" ] && export PP_GPUS
[ -n "${PP_DIAG_NETWORK:-}" ] && export PP_DIAG_NETWORK
"$SMOKE_RUN_BIN" \
"$ORCHESTRATOR_BIN" \
--seed \
--num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \
@ -189,6 +203,31 @@ export SWACTOR_DIAG_UDP_ECHO="${BIND_HOST}:${UDP_PORT}"
<"$FIFO" >"$ORCH_LOG" 2>&1 &
ORCH_PID=$!
# ── Step 5: wait for the orchestrator's dashboard to bind, then announce + open
WAITED=0
until (echo > "/dev/tcp/${BIND_HOST}/${DASH_PORT}") >/dev/null 2>&1; do
if ! kill -0 "$ORCH_PID" >/dev/null 2>&1; then
echo "demo-fleet: orchestrator exited before its dashboard came up." >&2
tail -n 40 "$ORCH_LOG" >&2 || true
exit 1
fi
WAITED=$((WAITED + 1))
[ "$WAITED" -ge 30 ] && { echo "demo-fleet: orchestrator dashboard did not bind :${DASH_PORT} in 30s" >&2; tail -n 40 "$ORCH_LOG" >&2; exit 1; }
sleep 1
done
echo
echo " ┌─────────────────────────────────────────────────────────────┐"
echo " │ Full swactor dashboard: $DASH_URL"
echo " │ (overview / actors / topology / distribution / netmap / fleet)"
echo " │ Collector fleet board: ${COLLECTOR_URL}/dashboard"
echo " └─────────────────────────────────────────────────────────────┘"
echo
if [ -z "${PP_NO_OPEN:-}" ]; then
if command -v xdg-open >/dev/null 2>&1; then (xdg-open "$DASH_URL" >/dev/null 2>&1 &) || true
elif command -v open >/dev/null 2>&1; then (open "$DASH_URL" >/dev/null 2>&1 &) || true
fi
fi
# ── Step 6: wait until the cluster is converged + held open ────────────────
echo "demo-fleet: waiting for the cluster to converge (first inference drive)…"
WAITED=0

View file

@ -2,7 +2,7 @@
# docker-dashboard-e2e.sh — the docker-e2e run, held open under the live
# swactor dashboard, one dashboard PER STAGE.
#
# Brings up `N` stub-mode `pp-gpu-node` containers on localhost and drives
# Brings up `N` stub-mode `pp-worker` containers on localhost and drives
# one InferenceRequest through them, exactly like `docker-e2e.sh` — but each
# stage serves the live swactor dashboard (PP_STAGE_DASHBOARD) and the
# orchestrator HOLDS after the drive (PP_HOLD). The stage containers run on
@ -78,8 +78,8 @@ done
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
CRATE_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
WORKSPACE_DIR="$(cd "$CRATE_DIR/../.." && pwd)"
SMOKE_RUN_BIN="$CRATE_DIR/target/release/pp-smoke-run"
GPU_NODE_BIN="$CRATE_DIR/target/release/pp-gpu-node"
ORCHESTRATOR_BIN="$CRATE_DIR/target/release/pp-orchestrator"
WORKER_BIN="$CRATE_DIR/target/release/pp-worker"
WORKER_PY="$CRATE_DIR/pp_tinygrad_worker.py"
COLLECTOR_BIN="$WORKSPACE_DIR/target/release/swactor-diag-collector"
POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
@ -87,20 +87,20 @@ POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
# Step 1: build the release artifacts the docker image packages (same set
# docker-e2e.sh builds — the image's COPY needs the diag binaries present).
if [ -z "${PP_SKIP_BUILD:-}" ]; then
echo "docker-dashboard-e2e: building pp-gpu-node + pp-smoke-run (release)"
echo "docker-dashboard-e2e: building pp-worker + pp-orchestrator (release)"
cargo build --manifest-path "$CRATE_DIR/Cargo.toml" --release \
--bin pp-gpu-node --bin pp-smoke-run
--bin pp-worker --bin pp-orchestrator
echo "docker-dashboard-e2e: building swactor-diag-{collector,postproc} (release)"
cargo build --manifest-path "$WORKSPACE_DIR/Cargo.toml" --release \
-p distribution --features collector \
--bin swactor-diag-collector --bin swactor-diag-postproc
fi
for f in "$SMOKE_RUN_BIN" "$GPU_NODE_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
for f in "$ORCHESTRATOR_BIN" "$WORKER_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
[ -f "$f" ] || { echo "docker-dashboard-e2e: missing $f" >&2; exit 1; }
done
# Step 2: build the layered image (heavy CUDA base, then thin code layer).
# The stage dashboard lives in the pp-gpu-node binary baked into this image,
# The stage dashboard lives in the pp-worker binary baked into this image,
# so a stale image without it will show nothing — rebuild unless you know the
# current image already carries the dashboard-enabled binary.
if [ -z "${PP_SKIP_IMAGE_BUILD:-}" ]; then
@ -131,7 +131,7 @@ for ((k = 0; k < NUM_STAGES; k++)); do
echo " stage ${k}: http://localhost:$((PORT_BASE + k))"
done
# Step 4: drive pp-smoke-run with the docker shim. The orchestrator serves its
# Step 4: drive pp-orchestrator with the docker shim. The orchestrator serves its
# own dashboard (PP_DASHBOARD) — including the live SWIM distribution graph and
# message tallies — and each stage serves its own (PP_STAGE_DASHBOARD). PP_HOLD
# makes the orchestrator block at the end, ticking the driver so the
@ -146,7 +146,7 @@ PP_DASHBOARD=1 \
PP_DASHBOARD_PORT="$ORCH_PORT" \
PP_STAGE_DASHBOARD=1 \
PP_STAGE_DASHBOARD_PORT_BASE="$PORT_BASE" \
"$SMOKE_RUN_BIN" \
"$ORCHESTRATOR_BIN" \
--seed \
--num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \

View file

@ -3,7 +3,7 @@
#
# Brings up:
# - `swactor-diag-collector` in a container (HTTP 9080 + UDP 9081)
# - `pp-smoke-run` on the host, in seed mode with N stub-mode stage
# - `pp-orchestrator` on the host, in seed mode with N stub-mode stage
# children spawned via `docker-gpu-node.sh` (each its own
# container on `--network host`)
#
@ -95,8 +95,8 @@ CRATE_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
WORKSPACE_DIR="$(cd "$CRATE_DIR/../.." && pwd)"
COMPOSE_FILE="$CRATE_DIR/docker-compose.diag.yml"
SMOKE_RUN_BIN="$CRATE_DIR/target/release/pp-smoke-run"
GPU_NODE_BIN="$CRATE_DIR/target/release/pp-gpu-node"
ORCHESTRATOR_BIN="$CRATE_DIR/target/release/pp-orchestrator"
WORKER_BIN="$CRATE_DIR/target/release/pp-worker"
WORKER_PY="$CRATE_DIR/pp_tinygrad_worker.py"
COLLECTOR_BIN="$WORKSPACE_DIR/target/release/swactor-diag-collector"
POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
@ -104,15 +104,15 @@ POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
# Step 1: build release artifacts the image will package. The pp binaries
# live in their own workspace; the distribution binaries live at the top.
if [ -z "${PP_SKIP_BUILD:-}" ]; then
echo "docker-diag-e2e: cargo build pp-smoke-run + pp-gpu-node (release)"
echo "docker-diag-e2e: cargo build pp-orchestrator + pp-worker (release)"
cargo build --manifest-path "$CRATE_DIR/Cargo.toml" --release \
--bin pp-gpu-node --bin pp-smoke-run
--bin pp-worker --bin pp-orchestrator
echo "docker-diag-e2e: cargo build swactor-diag-{collector,postproc} (release, --features collector)"
cargo build --manifest-path "$WORKSPACE_DIR/Cargo.toml" --release \
-p distribution --features collector \
--bin swactor-diag-collector --bin swactor-diag-postproc
fi
for f in "$SMOKE_RUN_BIN" "$GPU_NODE_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
for f in "$ORCHESTRATOR_BIN" "$WORKER_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
[ -f "$f" ] || { echo "docker-diag-e2e: missing $f" >&2; exit 1; }
done
@ -190,7 +190,7 @@ if [ "$USE_COMPOSE" = 1 ]; then
"${COMPOSE[@]}" -f "$COMPOSE_FILE" up -d --remove-orphans collector
else
# --entrypoint runs the collector directly; the default image entrypoint
# (pp_entrypoint.sh) would ignore these args and launch pp-gpu-node.
# (pp_entrypoint.sh) would ignore these args and launch pp-worker.
docker run -d --rm \
--name "$COLLECTOR_NAME" \
--network "$DIAG_NETWORK" \
@ -223,7 +223,7 @@ until (echo > /dev/tcp/127.0.0.1/9080) >/dev/null 2>&1; do
done
echo "docker-diag-e2e: collector ready"
# Step 5: drive pp-smoke-run with diagnostics env vars set. Stage children
# Step 5: drive pp-orchestrator with diagnostics env vars set. Stage children
# pick up the same vars via docker-gpu-node.sh's `-e` forwarders.
OUTPUT_DIR="$(mktemp -d)"
STDOUT_LOG="$OUTPUT_DIR/stdout.log"
@ -244,7 +244,7 @@ SWACTOR_DIAG_COLLECTOR_URL="http://127.0.0.1:9080" \
SWACTOR_DIAG_RUN_ID="$RUN_ID" \
SWACTOR_DIAG_SPOOL_DIR="$OUTPUT_DIR/spool" \
SWACTOR_DIAG_UDP_ECHO="127.0.0.1:9081" \
"$SMOKE_RUN_BIN" \
"$ORCHESTRATOR_BIN" \
--seed \
--num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \
@ -252,20 +252,20 @@ SWACTOR_DIAG_UDP_ECHO="127.0.0.1:9081" \
--prompt "$PROMPT" \
--max-tokens "$MAX_TOKENS" \
>"$STDOUT_LOG" 2>"$STDERR_LOG"
SMOKE_STATUS=$?
ORCH_STATUS=$?
set -e
if [ "$SMOKE_STATUS" -ne 0 ]; then
echo "docker-diag-e2e: pp-smoke-run exited $SMOKE_STATUS" >&2
if [ "$ORCH_STATUS" -ne 0 ]; then
echo "docker-diag-e2e: pp-orchestrator exited $ORCH_STATUS" >&2
echo "----- stdout -----" >&2
cat "$STDOUT_LOG" >&2
echo "----- stderr (last 60) -----" >&2
tail -n 60 "$STDERR_LOG" >&2
exit 1
fi
echo "docker-diag-e2e: pp-smoke-run exited 0"
echo "docker-diag-e2e: pp-orchestrator exited 0"
if [ -n "${PP_DIAG_VERBOSE:-}" ]; then
echo "----- pp-smoke-run stderr (last 30) -----"
echo "----- pp-orchestrator stderr (last 30) -----"
tail -n 30 "$STDERR_LOG"
fi
@ -295,7 +295,7 @@ while true; do
else
docker logs "$COLLECTOR_NAME" 2>&1 | tail -n 50 >&2 || true
fi
echo "----- pp-smoke-run stderr (last 60) -----" >&2
echo "----- pp-orchestrator stderr (last 60) -----" >&2
tail -n 60 "$STDERR_LOG" >&2 || true
exit 1
fi

View file

@ -1,8 +1,8 @@
#!/usr/bin/env bash
# docker-e2e.sh — the Stage 11 pre-deploy gate.
#
# Brings up `N` stub-mode `pp-gpu-node` containers on localhost, drives
# one InferenceRequest through them via `pp-smoke-run --seed`, and tears
# Brings up `N` stub-mode `pp-worker` containers on localhost, drives
# one InferenceRequest through them via `pp-orchestrator --seed`, and tears
# everything down. The image is built locally from the workspace's
# release artifacts; no GPU, no tinygrad, no GGUF required.
#
@ -63,8 +63,8 @@ fi
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
CRATE_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
WORKSPACE_DIR="$(cd "$CRATE_DIR/../.." && pwd)"
SMOKE_RUN_BIN="$CRATE_DIR/target/release/pp-smoke-run"
GPU_NODE_BIN="$CRATE_DIR/target/release/pp-gpu-node"
ORCHESTRATOR_BIN="$CRATE_DIR/target/release/pp-orchestrator"
WORKER_BIN="$CRATE_DIR/target/release/pp-worker"
WORKER_PY="$CRATE_DIR/pp_tinygrad_worker.py"
COLLECTOR_BIN="$WORKSPACE_DIR/target/release/swactor-diag-collector"
POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
@ -75,15 +75,15 @@ POSTPROC_BIN="$WORKSPACE_DIR/target/release/swactor-diag-postproc"
# present even for this stub run. The pp binaries live in this crate's
# workspace; the diagnostics binaries live at the repo root.
if [ -z "${PP_SKIP_BUILD:-}" ]; then
echo "docker-e2e: building pp-gpu-node + pp-smoke-run (release)"
echo "docker-e2e: building pp-worker + pp-orchestrator (release)"
cargo build --manifest-path "$CRATE_DIR/Cargo.toml" --release \
--bin pp-gpu-node --bin pp-smoke-run
--bin pp-worker --bin pp-orchestrator
echo "docker-e2e: building swactor-diag-{collector,postproc} (release, --features collector)"
cargo build --manifest-path "$WORKSPACE_DIR/Cargo.toml" --release \
-p distribution --features collector \
--bin swactor-diag-collector --bin swactor-diag-postproc
fi
for f in "$SMOKE_RUN_BIN" "$GPU_NODE_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
for f in "$ORCHESTRATOR_BIN" "$WORKER_BIN" "$WORKER_PY" "$COLLECTOR_BIN" "$POSTPROC_BIN"; do
[ -f "$f" ] || { echo "docker-e2e: missing $f" >&2; exit 1; }
done
@ -117,7 +117,7 @@ cleanup_containers() {
}
cleanup_containers
# Step 4: drive pp-smoke-run with the docker shim as its --gpu-node.
# Step 4: drive pp-orchestrator with the docker shim as its --gpu-node.
# The shim consults PP_IMAGE / PP_CONTAINER_PREFIX / PP_DEV from its env.
OUTPUT_DIR="$(mktemp -d)"
STDOUT_LOG="$OUTPUT_DIR/stdout.log"
@ -135,7 +135,7 @@ if [ -n "${PP_REAL:-}" ]; then
PP_DEV=CUDA \
PP_GPUS="${PP_GPUS:-all}" \
PP_MODEL_CACHE_DIR="$PP_MODEL_CACHE_DIR" \
"$SMOKE_RUN_BIN" \
"$ORCHESTRATOR_BIN" \
--seed \
--num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \
@ -148,7 +148,7 @@ else
PP_IMAGE="$IMAGE" \
PP_CONTAINER_PREFIX="$PREFIX" \
PP_DEV=CPU \
"$SMOKE_RUN_BIN" \
"$ORCHESTRATOR_BIN" \
--seed \
--num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \
@ -157,11 +157,11 @@ else
--max-tokens "$MAX_TOKENS" \
>"$STDOUT_LOG" 2>"$STDERR_LOG"
fi
SMOKE_STATUS=$?
ORCH_STATUS=$?
set -e
if [ $SMOKE_STATUS -ne 0 ]; then
echo "docker-e2e: pp-smoke-run exited $SMOKE_STATUS" >&2
if [ $ORCH_STATUS -ne 0 ]; then
echo "docker-e2e: pp-orchestrator exited $ORCH_STATUS" >&2
echo "----- stdout -----" >&2
cat "$STDOUT_LOG" >&2
echo "----- stderr (last 60) -----" >&2

View file

@ -1,9 +1,9 @@
#!/usr/bin/env bash
# docker-gpu-node.sh — shim that pp-smoke-run can spawn instead of the
# pp-gpu-node binary directly. Boots one pp-gpu-node container per stage
# docker-gpu-node.sh — shim that pp-orchestrator can spawn instead of the
# pp-worker binary directly. Boots one pp-worker container per stage
# on the host network so iroh can dial without NAT.
#
# Required env (forwarded by pp-smoke-run):
# Required env (forwarded by pp-orchestrator):
# STAGE, NUM_STAGES, SEED_ADDR, SEED_DIRECT, MAX_TOKENS
# Optional env (forwarded if present):
# MODEL, PP_WORKER_STUB, WORKER_CMD,
@ -80,10 +80,10 @@ mkdir -p "$CACHE_DIR"
#
# --entrypoint runs the binary directly, bypassing the image's default
# pp_entrypoint.sh (sshd + postmortem hold). That supervisor is for remote
# vast.ai nodes; a local docker stage should exit cleanly when pp-gpu-node
# vast.ai nodes; a local docker stage should exit cleanly when pp-worker
# does so --rm reaps it and the E2E's no-leftover-container check holds.
exec docker run --rm --init \
--entrypoint /usr/local/bin/pp-gpu-node \
--entrypoint /usr/local/bin/pp-worker \
--name "$NAME" \
--network "${PP_DIAG_NETWORK:-host}" \
"${GPU_ARGS[@]}" \

View file

@ -0,0 +1,179 @@
#!/usr/bin/env bash
# remote-collector-fleet.sh — like demo-fleet.sh, but the diagnostics collector
# lives OFF-box on the real VPS (prod topology) instead of on localhost.
#
# Brings up, on THIS machine, the docker stage fleet you would normally deploy
# (N stub-mode pp-worker containers on --network host) plus pp-orchestrator
# hosting the FULL swactor dashboard locally. Every stage's in-VM monitor ships
# its VastaiSample/VastaiLogs records to the REMOTE collector, and the
# orchestrator's Fleet tab subscribes to that same remote collector's SSE
# stream (/diag/stream/<run_id>) and folds the records — so the Fleet tab is
# exercised end-to-end against the production collector over the public WAN.
#
# The collector is NOT started here; it must already be running on the VPS and
# its port reachable (ufw). This is the "test the prod collector locally" path.
#
# Usage:
# examples/pipeline-parallel-inference/scripts/remote-collector-fleet.sh [N]
#
# Environment overrides:
# PP_COLLECTOR_HOST VPS host running swactor-diag-collector (default 139.59.195.69)
# PP_COLLECTOR_PORT collector HTTP port (default 9080)
# PP_COLLECTOR_UDP collector UDP echo port (default 9081)
# PP_RUN_ID diagnostics run id (default remote-fleet-<epoch>)
# PP_DIAG_IMAGE code image tag (default swactor-pp-gpu:latest)
# PP_BASE_IMAGE base image tag (default swactor-pp-base:cuda12.6)
# PP_SKIP_BUILD skip cargo build (reuse target/)
# PP_SKIP_IMAGE_BUILD skip docker image build (reuse tag)
# PP_DASHBOARD_PORT orchestrator dashboard port (default 9095)
# PP_PROMPT inference prompt (default "remote fleet demo")
# PP_MAX_TOKENS decode token cap (default 4)
# PP_NO_OPEN if set, don't open a browser
set -euo pipefail
NUM_STAGES="${1:-3}"
COLLECTOR_HOST="${PP_COLLECTOR_HOST:-139.59.195.69}"
COLLECTOR_PORT="${PP_COLLECTOR_PORT:-9080}"
COLLECTOR_UDP="${PP_COLLECTOR_UDP:-9081}"
RUN_ID="${PP_RUN_ID:-remote-fleet-$(date +%s)}"
IMAGE="${PP_DIAG_IMAGE:-swactor-pp-gpu:latest}"
BASE_IMAGE="${PP_BASE_IMAGE:-swactor-pp-base:cuda12.6}"
PROMPT="${PP_PROMPT:-remote fleet demo}"
MAX_TOKENS="${PP_MAX_TOKENS:-4}"
DASH_PORT="${PP_DASHBOARD_PORT:-9095}"
CONTAINER_PREFIX="remote-fleet-stage"
COLLECTOR_URL="http://${COLLECTOR_HOST}:${COLLECTOR_PORT}"
DASH_URL="http://127.0.0.1:${DASH_PORT}/"
if ! [[ "$NUM_STAGES" =~ ^[0-9]+$ ]] || [ "$NUM_STAGES" -lt 2 ]; then
echo "remote-fleet: N must be an integer >= 2 (seed mode needs >=2 stages), got '$NUM_STAGES'" >&2
exit 2
fi
if ! command -v docker >/dev/null 2>&1; then echo "remote-fleet: docker not on PATH" >&2; exit 2; fi
if ! docker info >/dev/null 2>&1; then echo "remote-fleet: docker daemon unreachable" >&2; exit 2; fi
if (exec 3<>"/dev/tcp/127.0.0.1/${DASH_PORT}") 2>/dev/null; then
exec 3>&- 3<&-
echo "remote-fleet: dashboard port ${DASH_PORT} in use. Pick another: PP_DASHBOARD_PORT=9096 $0 ${NUM_STAGES}" >&2
exit 2
fi
# Preflight: the remote collector must be reachable, else the Fleet tab and the
# stages' shipping will silently get nothing. Fail loudly here instead.
echo "remote-fleet: checking remote collector at ${COLLECTOR_URL} …"
if ! curl -fsS -m 8 -o /dev/null "${COLLECTOR_URL}/diag/runs"; then
echo "remote-fleet: cannot reach ${COLLECTOR_URL}/diag/runs — is the collector up and the port open (ufw)?" >&2
exit 1
fi
echo "remote-fleet: remote collector reachable."
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
CRATE_DIR="$(cd "$SCRIPT_DIR/.." && pwd)"
WORKSPACE_DIR="$(cd "$CRATE_DIR/../.." && pwd)"
ORCHESTRATOR_BIN="$CRATE_DIR/target/release/pp-orchestrator"
WORKER_BIN="$CRATE_DIR/target/release/pp-worker"
WORKER_PY="$CRATE_DIR/pp_tinygrad_worker.py"
if [ -z "${PP_SKIP_BUILD:-}" ]; then
echo "remote-fleet: cargo build pp-orchestrator + pp-worker (release)"
cargo build --manifest-path "$CRATE_DIR/Cargo.toml" --release --bin pp-worker --bin pp-orchestrator
fi
for f in "$ORCHESTRATOR_BIN" "$WORKER_BIN" "$WORKER_PY"; do
[ -f "$f" ] || { echo "remote-fleet: missing $f" >&2; exit 1; }
done
if [ -z "${PP_SKIP_IMAGE_BUILD:-}" ]; then
echo "remote-fleet: docker build $BASE_IMAGE (base)"
docker build -f "$CRATE_DIR/Dockerfile.base" -t "$BASE_IMAGE" "$WORKSPACE_DIR"
echo "remote-fleet: docker build $IMAGE (code)"
docker build -f "$CRATE_DIR/Dockerfile" --build-arg "BASE_IMAGE=$BASE_IMAGE" -t "$IMAGE" "$WORKSPACE_DIR"
fi
WORKDIR="$(mktemp -d -t remote-fleet.XXXXXX)"
SPOOL_DIR="$WORKDIR/spool"
ORCH_LOG="$WORKDIR/orch.log"
FIFO="$WORKDIR/orch.stdin"
mkdir -p "$SPOOL_DIR"
mkfifo "$FIFO"
ORCH_PID=""
CLEANED=""
cleanup() {
[ -n "$CLEANED" ] && return 0
CLEANED=1
set +e
echo; echo "remote-fleet: tearing down…"
local ids
ids=$(docker ps -aq --filter "name=^${CONTAINER_PREFIX}-[0-9]+$")
[ -n "$ids" ] && docker rm -f $ids >/dev/null 2>&1
[ -n "$ORCH_PID" ] && kill "$ORCH_PID" >/dev/null 2>&1
exec 3>&- 2>/dev/null
[ -d "$WORKDIR" ] && rm -rf "$WORKDIR"
set -e
echo "remote-fleet: done."
}
trap cleanup EXIT
trap 'exit 130' INT TERM
exec 3<>"$FIFO"
echo "remote-fleet: launching orchestrator + ${NUM_STAGES} stage containers"
echo "remote-fleet: collector = ${COLLECTOR_URL} run_id = ${RUN_ID}"
export PP_HOLD=1
export PP_WORKER_STUB=1
export PP_DEV=CPU
export PP_IMAGE="$IMAGE"
export PP_CONTAINER_PREFIX="$CONTAINER_PREFIX"
export PP_DASHBOARD=1
export PP_DASHBOARD_PORT="$DASH_PORT"
export SWACTOR_DIAG_COLLECTOR_URL="$COLLECTOR_URL"
export SWACTOR_DIAG_RUN_ID="$RUN_ID"
export SWACTOR_DIAG_SPOOL_DIR="$SPOOL_DIR"
export SWACTOR_DIAG_UDP_ECHO="${COLLECTOR_HOST}:${COLLECTOR_UDP}"
"$ORCHESTRATOR_BIN" \
--seed --num-stages "$NUM_STAGES" \
--gpu-node "$SCRIPT_DIR/docker-gpu-node.sh" \
--worker "$WORKER_PY" \
--prompt "$PROMPT" --max-tokens "$MAX_TOKENS" \
<"$FIFO" >"$ORCH_LOG" 2>&1 &
ORCH_PID=$!
WAITED=0
until (echo > "/dev/tcp/127.0.0.1/${DASH_PORT}") >/dev/null 2>&1; do
if ! kill -0 "$ORCH_PID" >/dev/null 2>&1; then
echo "remote-fleet: orchestrator exited before its dashboard came up." >&2
tail -n 40 "$ORCH_LOG" >&2 || true
exit 1
fi
WAITED=$((WAITED + 1))
[ "$WAITED" -ge 30 ] && { echo "remote-fleet: dashboard did not bind :${DASH_PORT} in 30s" >&2; tail -n 40 "$ORCH_LOG" >&2; exit 1; }
sleep 1
done
echo
echo " ┌─────────────────────────────────────────────────────────────┐"
echo " │ Full swactor dashboard: $DASH_URL"
echo " │ Fleet tab pulls from remote collector: ${COLLECTOR_URL}/diag/stream/${RUN_ID}"
echo " │ Remote collector board: ${COLLECTOR_URL}/dashboard?run=${RUN_ID}"
echo " └─────────────────────────────────────────────────────────────┘"
echo
if [ -z "${PP_NO_OPEN:-}" ]; then
if command -v xdg-open >/dev/null 2>&1; then (xdg-open "$DASH_URL" >/dev/null 2>&1 &) || true
elif command -v open >/dev/null 2>&1; then (open "$DASH_URL" >/dev/null 2>&1 &) || true
fi
fi
echo "remote-fleet: waiting for the cluster to converge (first inference drive)…"
WAITED=0
until grep -q "holding cluster open" "$ORCH_LOG" 2>/dev/null; do
if ! kill -0 "$ORCH_PID" >/dev/null 2>&1; then
echo "remote-fleet: orchestrator exited before holding — drive failed." >&2
tail -n 40 "$ORCH_LOG" >&2 || true
exit 1
fi
WAITED=$((WAITED + 1))
[ "$WAITED" -ge 180 ] && { echo "remote-fleet: cluster did not converge within 180s" >&2; tail -n 40 "$ORCH_LOG" >&2; exit 1; }
sleep 1
done
RUNNING=$(docker ps -q --filter "name=^${CONTAINER_PREFIX}-[0-9]+$" | wc -l | tr -d ' ')
echo
echo "remote-fleet: ✅ fleet up — ${RUNNING}/${NUM_STAGES} stage containers shipping to ${COLLECTOR_URL}."
echo "remote-fleet: watch the Fleet tab live at $DASH_URL"
echo "remote-fleet: press Ctrl+C to tear everything down."
echo
wait "$ORCH_PID"

View file

@ -1,4 +1,4 @@
//! pp-gpu-node — pipeline-parallel GPU inference node.
//! pp-worker — pipeline-parallel GPU inference node.
//!
//! Boots one stage of a pipeline-parallel inference run. Reads its
//! configuration from the environment (set at instance-create time on
@ -68,7 +68,7 @@ use swactor_process::{ProcessMode, ProcessSpec};
/// SWIM name the orchestrator uses to publish the address of its
/// `InferenceResponse` inbox. The last stage resolves this name to learn
/// where to send the final response. Defined here (and re-declared in
/// `pp-smoke-run`) so the topology module stays test-shaped; the binary
/// `pp-orchestrator`) so the topology module stays test-shaped; the binary
/// is the only place that cares about this name.
const ORCHESTRATOR_NAME: &str = "pp-orchestrator";
@ -88,7 +88,7 @@ fn parse_hex_node_id(s: &str) -> [u8; 32] {
fn require_env(name: &str) -> String {
std::env::var(name)
.unwrap_or_else(|_| {
eprintln!("pp-gpu-node: env {name} is required");
eprintln!("pp-worker: env {name} is required");
std::process::exit(2);
})
.trim()
@ -98,7 +98,7 @@ fn require_env(name: &str) -> String {
fn require_u32(name: &str) -> u32 {
let raw = require_env(name);
raw.parse::<u32>().unwrap_or_else(|_| {
eprintln!("pp-gpu-node: env {name}={raw:?} must be a u32");
eprintln!("pp-worker: env {name}={raw:?} must be a u32");
std::process::exit(2);
})
}
@ -118,7 +118,7 @@ fn stage_secret_from_env() -> Option<SecretKey> {
}
if hex.len() != 64 {
eprintln!(
"pp-gpu-node: PP_STAGE_SECRET must be 64 hex chars, got {}",
"pp-worker: PP_STAGE_SECRET must be 64 hex chars, got {}",
hex.len()
);
std::process::exit(2);
@ -126,7 +126,7 @@ fn stage_secret_from_env() -> Option<SecretKey> {
let mut bytes = [0u8; 32];
for (i, b) in bytes.iter_mut().enumerate() {
*b = u8::from_str_radix(&hex[i * 2..i * 2 + 2], 16).unwrap_or_else(|_| {
eprintln!("pp-gpu-node: PP_STAGE_SECRET is not valid hex");
eprintln!("pp-worker: PP_STAGE_SECRET is not valid hex");
std::process::exit(2);
});
}
@ -266,7 +266,7 @@ fn build_route(
fn register_name(driver: &mut IrohDriver, name: &str, addr: ActorAddress, stage: u32) {
driver.node_mut().register_name(name.into(), addr);
diag::emit_register_name(driver, name, addr, Some(stage));
eprintln!("pp-gpu-node: registered {name} -> {addr:?}");
eprintln!("pp-worker: registered {name} -> {addr:?}");
}
fn resolve_or_die(
@ -274,10 +274,10 @@ fn resolve_or_die(
name: &str,
timeout: Duration,
) -> (ActorAddress, String) {
eprintln!("pp-gpu-node: resolving {name}...");
eprintln!("pp-worker: resolving {name}...");
resolve_name(driver, name, timeout).unwrap_or_else(|| {
eprintln!(
"pp-gpu-node: failed to resolve {name} in {:.0}s",
"pp-worker: failed to resolve {name} in {:.0}s",
timeout.as_secs_f32()
);
std::process::exit(1);
@ -294,16 +294,16 @@ fn add_route_or_die(
match build_route(driver, node_hex) {
Ok(t) => router.add_route(addr, t),
Err(e) => {
eprintln!("pp-gpu-node: route to {label} failed: {e}");
eprintln!("pp-worker: route to {label} failed: {e}");
std::process::exit(1);
}
}
}
/// Ask the kernel to deliver `SIGTERM` to this process when its parent dies.
/// Without this, a `SIGKILL` to `pp-smoke-run` would orphan its children to
/// Without this, a `SIGKILL` to `pp-orchestrator` would orphan its children to
/// pid 1 and leave them running — the orchestrator's `ChainGuard::drop` runs
/// only on graceful exit. With it, each `pp-gpu-node` dies seconds after its
/// only on graceful exit. With it, each `pp-worker` dies seconds after its
/// orchestrator does, which is the `binary_e2e_orchestrator_sigkilled_*`
/// contract from TEST_SPEC §13.2. Linux-only; other platforms are no-ops.
#[cfg(target_os = "linux")]
@ -322,7 +322,7 @@ fn install_parent_death_signal() {}
/// Set by the `SIGHUP` handler; polled by the pump loops to drive an
/// in-place worker hot-reload (re-exec the on-disk worker script). An
/// operator pushes a new `pp_tinygrad_worker.py` over the running one and
/// `kill -HUP $(pidof pp-gpu-node)` to pick it up without re-leasing.
/// `kill -HUP $(pidof pp-worker)` to pick it up without re-leasing.
static RELOAD_REQUESTED: AtomicBool = AtomicBool::new(false);
#[cfg(target_os = "linux")]
@ -352,7 +352,7 @@ fn install_sighup_handler() {}
/// pump so both honour reloads with the same latency.
fn drain_reload_request(rt: &Runtime, stage_actor_addr: ActorAddress) {
if RELOAD_REQUESTED.swap(false, Ordering::SeqCst) {
eprintln!("pp-gpu-node: SIGHUP — reloading worker");
eprintln!("pp-worker: SIGHUP — reloading worker");
let _ = rt.send_to(stage_actor_addr, StageMsg::ReloadWorker);
}
}
@ -366,7 +366,7 @@ fn maybe_simulate_boot_delay(stage: u32) {
let secs = std::env::var("PP_BOOT_DELAY_SECS").ok().and_then(|s| s.trim().parse::<u64>().ok());
if let (Some(target), Some(secs)) = (target, secs) {
if target == stage && secs > 0 {
eprintln!("pp-gpu-node: simulated boot delay of {secs}s on stage {stage}");
eprintln!("pp-worker: simulated boot delay of {secs}s on stage {stage}");
std::thread::sleep(Duration::from_secs(secs));
}
}
@ -378,7 +378,7 @@ fn main() {
let num_stages = require_u32("NUM_STAGES");
if num_stages < 2 || stage >= num_stages {
eprintln!(
"pp-gpu-node: invalid STAGE={stage} for NUM_STAGES={num_stages} \
"pp-worker: invalid STAGE={stage} for NUM_STAGES={num_stages} \
(need NUM_STAGES >= 2 and STAGE < NUM_STAGES; N=1 is not supported)"
);
std::process::exit(2);
@ -441,7 +441,7 @@ fn main() {
let vastai_forwarder = _vastai_in_vm.as_ref().map(|m| m.forwarder());
// Stamp the bundle the moment this process announces itself, so a
// bundle reader can tell two pp-gpu-node incarnations of the same
// bundle reader can tell two pp-worker incarnations of the same
// stage apart: a manual binary swap (the operator runbook) pkills the
// old process and setsid's a new one under the same PID-1 env, which
// means the same run_id + node_id, but the pid differs. The event carries that
@ -468,7 +468,7 @@ fn main() {
.map(|sa| sa.to_string())
.collect();
eprintln!(
"pp-gpu-node: stage {stage}/{num_stages} ({role:?}) started (node_id: {my_hex})"
"pp-worker: stage {stage}/{num_stages} ({role:?}) started (node_id: {my_hex})"
);
// PP_GPU_NODE_ADDR is printed to stdout (flushed) so a parent process
// capturing this child's stdout can extract our addressing. The orchestrator
@ -485,7 +485,7 @@ fn main() {
let mut seed_addr = iroh::EndpointAddr::from(seed_key);
if let Some(relay) = seed_relay_env.as_deref() {
if let Ok(relay_url) = relay.trim().parse::<iroh::RelayUrl>() {
eprintln!("pp-gpu-node: using seed relay {relay}");
eprintln!("pp-worker: using seed relay {relay}");
seed_addr = seed_addr.with_relay_url(relay_url);
}
}
@ -525,7 +525,7 @@ fn main() {
peer_addr = peer_addr.with_ip_addr(sa);
}
}
eprintln!("pp-gpu-node: also joining {label} {peer_hex}");
eprintln!("pp-worker: also joining {label} {peer_hex}");
targets.push(peer_addr);
}
};
@ -545,7 +545,7 @@ fn main() {
"first-stage peer",
);
eprintln!("pp-gpu-node: joining seed {seed_hex}");
eprintln!("pp-worker: joining seed {seed_hex}");
driver.join(&join_targets);
// If we ended up with a relay (vast.ai / WAN), publish it via SWIM
@ -556,7 +556,7 @@ fn main() {
// the autoregressive feedback edge (last → first) because SWIM has
// not yet probed that specific pair.
if let Some(home) = driver.home_relay_url() {
eprintln!("pp-gpu-node: publishing home relay {home} to SWIM gossip");
eprintln!("pp-worker: publishing home relay {home} to SWIM gossip");
driver.node_mut().set_relay_url(Some(home.to_string()));
}
@ -578,10 +578,10 @@ fn main() {
.and_then(|s| s.trim().parse().ok())
.unwrap_or(1200);
if !wait_for_cluster(&mut driver, Duration::from_secs(converge_secs)) {
eprintln!("pp-gpu-node: cluster did not converge in {converge_secs}s");
eprintln!("pp-worker: cluster did not converge in {converge_secs}s");
std::process::exit(1);
}
eprintln!("pp-gpu-node: cluster converged");
eprintln!("pp-worker: cluster converged");
// Create the actor runtime, codec registry, and transport router.
let mut rt = Runtime::new(RuntimeConfig::default());
@ -617,7 +617,7 @@ fn main() {
if let Some((handle, collector, port)) = &stage_dash {
handle.set_runtime(Arc::clone(&rt), Arc::clone(collector));
handle.start_http(driver.tokio_handle());
eprintln!("pp-gpu-node: stage {stage} dashboard on http://localhost:{port}");
eprintln!("pp-worker: stage {stage} dashboard on http://localhost:{port}");
}
run_stage(
@ -655,17 +655,17 @@ fn hold_until_worker_ready(
if let Some(status) = status_inbox.try_recv() {
match status {
StageActorStatus::WorkerReady { pid } => {
eprintln!("pp-gpu-node: worker ready (pid: {pid:?})");
eprintln!("pp-worker: worker ready (pid: {pid:?})");
return;
}
StageActorStatus::ProcessStarted => {
eprintln!("pp-gpu-node: worker process started");
eprintln!("pp-worker: worker process started");
}
StageActorStatus::ProcessExited { status } => {
eprintln!(
"pp-gpu-node: stage-{stage} worker exited during startup: \
"pp-worker: stage-{stage} worker exited during startup: \
{status:?}; holding (SWIM alive) — push a fixed worker.py \
and `kill -HUP $(pidof pp-gpu-node)` to reload"
and `kill -HUP $(pidof pp-worker)` to reload"
);
last_warn = Instant::now();
}
@ -674,7 +674,7 @@ fn hold_until_worker_ready(
if last_warn.elapsed() >= warn_after {
eprintln!(
"pp-gpu-node: stage-{stage} worker still not ready after {}s; \
"pp-worker: stage-{stage} worker still not ready after {}s; \
holding — SIGHUP to reload the worker script",
warn_after.as_secs()
);
@ -881,7 +881,7 @@ fn run_stage(
let (next_addr, next_hex) =
resolve_or_die(&mut driver, &next_name, neighbor_resolve_timeout);
eprintln!(
"pp-gpu-node: resolved {next_name} -> {next_addr:?} on {next_hex}"
"pp-worker: resolved {next_name} -> {next_addr:?} on {next_hex}"
);
add_route_or_die(&driver, &router, next_addr, &next_hex, &next_name);
rt.send_to(
@ -900,7 +900,7 @@ fn run_stage(
let (next_addr, next_hex) =
resolve_or_die(&mut driver, &next_name, neighbor_resolve_timeout);
eprintln!(
"pp-gpu-node: resolved {next_name} -> {next_addr:?} on {next_hex}"
"pp-worker: resolved {next_name} -> {next_addr:?} on {next_hex}"
);
add_route_or_die(&driver, &router, next_addr, &next_hex, &next_name);
rt.send_to(
@ -927,7 +927,7 @@ fn run_stage(
let (orch_addr, orch_hex) =
resolve_or_die(&mut driver, ORCHESTRATOR_NAME, neighbor_resolve_timeout);
eprintln!(
"pp-gpu-node: resolved {feedback_name}={feedback_addr:?} on \
"pp-worker: resolved {feedback_name}={feedback_addr:?} on \
{feedback_hex}, orch={orch_addr:?} on {orch_hex}"
);
add_route_or_die(
@ -978,7 +978,7 @@ fn main_pump(
msg_pump: ActorMessagePump,
stage_actor_addr: ActorAddress,
) {
eprintln!("pp-gpu-node: entering main pump loop");
eprintln!("pp-worker: entering main pump loop");
loop {
driver.recv();
driver.tick();
@ -989,10 +989,10 @@ fn main_pump(
if let Some(status) = status_inbox.try_recv() {
match status {
StageActorStatus::ProcessExited { status } => {
eprintln!("pp-gpu-node: worker exited: {status:?}");
eprintln!("pp-gpu-node: keeping SWIM alive for diagnostics");
eprintln!("pp-worker: worker exited: {status:?}");
eprintln!("pp-worker: keeping SWIM alive for diagnostics");
}
other => eprintln!("pp-gpu-node: status: {other:?}"),
other => eprintln!("pp-worker: status: {other:?}"),
}
}
std::thread::sleep(Duration::from_millis(20));

View file

@ -1,5 +1,5 @@
//! Wire `crates/distribution` diagnostics into `pp-smoke-run` and
//! `pp-gpu-node` from environment variables.
//! Wire `crates/distribution` diagnostics into `pp-orchestrator` and
//! `pp-worker` from environment variables.
//!
//! Reading `SWACTOR_DIAG_COLLECTOR_URL` is the opt-in switch. When it is
//! unset (or empty) `install_from_env` returns `None` and the binary

View file

@ -7,10 +7,14 @@
<style>
* { margin: 0; padding: 0; box-sizing: border-box; }
body { font-family: 'Menlo','Consolas','Monaco',monospace; background:#0f1117; color:#e0e0e0; font-size:13px; }
.header { display:flex; align-items:center; gap:16px; padding:10px 16px; background:#161922; border-bottom:1px solid #262b38; }
.header h1 { font-size:15px; font-weight:600; color:#7aa2f7; }
.nav a { color:#9aa5b1; text-decoration:none; margin-right:14px; }
.nav a:hover, .nav a.active { color:#7aa2f7; }
.header { display:flex; align-items:center; justify-content:space-between; padding:12px 20px; background:#161822; border-bottom:1px solid #2a2d3e; }
.header h1 { font-size:16px; font-weight:600; color:#fff; }
.header-left { display:flex; align-items:center; }
.nav-links { display:flex; gap:4px; margin-left:20px; }
.nav-link { color:#888; text-decoration:none; font-size:12px; padding:4px 10px; border-radius:3px; transition:color 0.2s; }
.nav-link:hover { color:#e0e0e0; }
.nav-link.active { color:#fff; background:#2a2d3e; }
.header-right { display:flex; align-items:center; gap:12px; }
.wrap { padding:16px; display:grid; grid-template-columns: 1fr 1fr; gap:16px; }
.card { background:#161922; border:1px solid #262b38; border-radius:6px; padding:12px; }
.card h2 { font-size:13px; color:#c0caf5; margin-bottom:10px; border-bottom:1px solid #262b38; padding-bottom:6px; }
@ -33,13 +37,18 @@
</head>
<body>
<div class="header">
<h1>Orchestrator · Distribution</h1>
<div class="nav">
<a href="/">Overview</a>
<a href="/actors">Actors</a>
<a href="/plugin/distribution" class="active">Distribution</a>
<div class="header-left">
<h1>Swactor Runtime Dashboard</h1>
<nav class="nav-links">
<a href="/" class="nav-link">Overview</a>
<a href="/actors" class="nav-link">Actors</a>
<a href="/topology" class="nav-link">Topology</a>
<a href="/plugin/distribution" class="nav-link active">Distribution</a>
<a href="/plugin/netmap" class="nav-link">Net map</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
<span id="node" class="mono" style="margin-left:auto"></span>
<div class="header-right"><span id="node" class="mono"></span></div>
</div>
<div class="wrap">

View file

@ -0,0 +1,206 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<title>Swactor Runtime Dashboard — Fleet</title>
<style>
/* Reuses the Swactor Runtime Dashboard layout/CSS (panels + stat-cards). */
* { margin: 0; padding: 0; box-sizing: border-box; }
body { font-family: 'Menlo', 'Consolas', 'Monaco', monospace; background: #0f1117; color: #e0e0e0; font-size: 13px; }
.header {
display: flex; align-items: center; justify-content: space-between;
padding: 12px 20px; background: #161822; border-bottom: 1px solid #2a2d3e;
}
.header h1 { font-size: 16px; font-weight: 600; color: #fff; }
.header-left { display: flex; align-items: center; }
.nav-links { display: flex; gap: 4px; margin-left: 20px; }
.nav-link {
color: #888; text-decoration: none; font-size: 12px;
padding: 4px 10px; border-radius: 3px; transition: color 0.2s;
}
.nav-link:hover { color: #e0e0e0; }
.nav-link.active { color: #fff; background: #2a2d3e; }
.header-right { display: flex; align-items: center; gap: 12px; }
.mono { color: #9aa5b1; }
.muted { color: #5c6370; font-style: italic; }
.grid { display: grid; grid-template-columns: 1fr; gap: 12px; padding: 12px; }
.panel {
background: #161822; border: 1px solid #2a2d3e; border-radius: 6px;
padding: 14px; overflow: hidden;
}
.panel h2 { font-size: 12px; color: #888; text-transform: uppercase; letter-spacing: 1px; margin-bottom: 10px; }
.stats-cards { display: grid; grid-template-columns: repeat(4, 1fr); gap: 10px; }
.stat-card { background: #1c1f2e; border-radius: 4px; padding: 10px; text-align: center; }
.stat-card .value { font-size: 22px; font-weight: 700; color: #fff; }
.stat-card .label { font-size: 10px; color: #888; text-transform: uppercase; margin-top: 2px; }
/* Per-stage panels laid out in a responsive grid. */
.stages-grid { display: grid; grid-template-columns: repeat(auto-fill, minmax(320px, 1fr)); gap: 12px; }
.stage-head { display: flex; align-items: center; justify-content: space-between; margin-bottom: 4px; }
.stage-head .name { font-size: 14px; font-weight: 700; color: #fff; }
.stage-sub { font-size: 11px; margin-bottom: 10px; }
.stage .stats-cards { grid-template-columns: repeat(3, 1fr); gap: 8px; }
.stage .stat-card .value { font-size: 18px; }
.pill { display: inline-flex; align-items: center; gap: 5px; padding: 2px 9px; border-radius: 20px; font-size: 11px; font-weight: 600; background: #0f1117; }
.pill .dot { width: 7px; height: 7px; border-radius: 50%; }
.statusmsg { margin-top: 9px; font-size: 11px; color: #fbbf24; min-height: 0; }
.statusmsg.err { color: #f87171; }
.log-wrap { max-height: 320px; overflow-y: auto; }
.log-wrap table { width: 100%; border-collapse: collapse; }
.log-wrap td { padding: 3px 8px; text-align: left; border-bottom: 1px solid #1c1f2e; font-size: 11px; vertical-align: top; }
.log-wrap td.t { color: #5c6370; white-space: nowrap; }
.log-wrap td.s { font-weight: 600; white-space: nowrap; }
.log-wrap td.msg { white-space: pre-wrap; word-break: break-word; }
.log-wrap tr.stderr td.msg { color: #f87171; }
::-webkit-scrollbar { width: 6px; }
::-webkit-scrollbar-track { background: #0f1117; }
::-webkit-scrollbar-thumb { background: #2a2d3e; border-radius: 3px; }
</style>
</head>
<body>
<div class="header">
<div class="header-left">
<h1>Swactor Runtime Dashboard</h1>
<nav class="nav-links">
<a href="/" class="nav-link">Overview</a>
<a href="/actors" class="nav-link">Actors</a>
<a href="/topology" class="nav-link">Topology</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/netmap" class="nav-link">Net map</a>
<a href="/plugin/vastai" class="nav-link active">Fleet</a>
</nav>
</div>
<div class="header-right"><span id="run" class="mono"></span></div>
</div>
<div class="grid">
<div class="panel">
<h2>Fleet</h2>
<div class="stats-cards">
<div class="stat-card"><div class="value" id="s-running">—</div><div class="label">Running</div></div>
<div class="stat-card"><div class="value" id="s-gpu">—</div><div class="label">Fleet GPU</div></div>
<div class="stat-card"><div class="value" id="s-cost">—</div><div class="label">Spend</div></div>
<div class="stat-card"><div class="value" id="s-stages">—</div><div class="label">Stages</div></div>
</div>
</div>
<div class="panel">
<h2>Stages</h2>
<div class="stages-grid" id="stages"><div class="muted">waiting for vast.ai records…</div></div>
</div>
<div class="panel">
<h2>Logs</h2>
<div class="log-wrap"><table><tbody id="logs"></tbody></table></div>
</div>
</div>
<script>
const $ = (id) => document.getElementById(id);
const STAGE_COLORS = ["#4a90e2","#f06292","#ffb74d","#81c784","#ba68c8","#4dd0e1","#aed581","#ff8a65"];
function stageColor(idx){ return STAGE_COLORS[idx % STAGE_COLORS.length]; }
function statusInfo(status){
switch(status){
case "running": return {color:"#4ade80",label:"running",err:false};
case "loading": return {color:"#fbbf24",label:"loading",err:false};
case "created": return {color:"#fbbf24",label:"created",err:false};
case "offline": return {color:"#9ca3af",label:"offline",err:true};
case "exited": return {color:"#f87171",label:"exited",err:true};
default: return {color:"#6b7280",label:status||"—",err:false};
}
}
function fmtClock(ms){ const s=Math.floor(ms/1000),m=Math.floor(s/60); return `${String(m).padStart(2,"0")}:${String(s%60).padStart(2,"0")}`; }
function fmtBytesRate(bps){ if(bps==null)return "—"; const u=["B/s","KB/s","MB/s","GB/s"]; let i=0,v=bps; while(v>=1024&&i<u.length-1){v/=1024;i++;} return `${v.toFixed(v<10?1:0)} ${u[i]}`; }
function fmtUSD(v){ if(v==null)return "—"; return `$${v.toFixed(v<1?4:2)}`; }
function escapeHtml(s){ return s.replace(/[&<>]/g,(c)=>({"&":"&amp;","<":"&lt;",">":"&gt;"}[c])); }
// View-time selectors over the already-folded model (always at the latest time).
function stageStateAt(stage, t){
const lastAtOrBefore=(arr)=>{ let r=null; for(const x of arr){ if(x.at<=t) r=x; else break; } return r; };
const inst=lastAtOrBefore(stage.instances);
const samp=lastAtOrBefore(stage.samples);
let activeContract=null;
for(const c of (stage.contractList||[])){ if((c.leasedAt??0)<=t && (c.endAt==null||c.endAt>t)) activeContract=c; }
return {inst,samp,activeContract};
}
function sumCost(s,t){ const byC=new Map(); for(const o of s.instances){ if(o.at<=t && o.cost!=null) byC.set(o.contractId,o.cost); } let c=0; for(const v of byC.values()) c+=v; return c; }
function card(label,val){ return `<div class="stat-card"><div class="value">${val}</div><div class="label">${label}</div></div>`; }
let model=null;
function renderStages(){
const t = model.tEnd || 0;
let running=0,gpuSum=0,gpuN=0,costSum=0;
const panels=model.stages.map((s)=>{
const {inst,samp,activeContract}=stageStateAt(s,t);
const status=inst?inst.status:(activeContract?"loading":"—");
const si=statusInfo(status);
if(si.label==="running") running++;
const util=samp?samp.util:(inst?inst.gpuUtil:null);
const temp=samp?samp.temp:(inst?inst.gpuTemp:null);
const vram=samp?samp.vram:null;
const vramTot=samp?samp.vramTotal:(inst&&inst.gpuRam?inst.gpuRam*1024:null);
const cpu=samp?samp.cpu:(inst?inst.cpuUtil:null);
if(si.label==="running"&&util!=null){gpuSum+=util;gpuN++;}
const cost=sumCost(s,t); costSum+=cost;
const vramTxt=(vram!=null&&vramTot)?(vram/1024).toFixed(1)+"/"+(vramTot/1024).toFixed(0)+"G":(vram!=null?(vram/1024).toFixed(1)+"G":"—");
return `<div class="panel stage">
<div class="stage-head">
<span class="name" style="color:${s.color}">${s.label}</span>
<span class="pill"><span class="dot" style="background:${si.color}"></span>${si.label}</span>
</div>
<div class="stage-sub muted">${s.gpuName||"CPU"}${s.geo?" · "+s.geo:""} · ${fmtUSD(cost)} @ ${fmtUSD(s.dph)}/h</div>
<div class="stats-cards">
${card("GPU", util!=null?util.toFixed(0)+"%":"—")}
${card("VRAM", vramTxt)}
${card("Temp", temp!=null?temp.toFixed(0)+"°C":"—")}
${card("CPU", cpu!=null?cpu.toFixed(0)+"%":"—")}
${card("Power", samp&&samp.power!=null?samp.power.toFixed(0)+"W":"—")}
${card("Net↓", samp?fmtBytesRate(samp.netRx):"—")}
</div>
<div class="statusmsg ${si.err?"err":""}">${inst&&inst.statusMsg?escapeHtml(inst.statusMsg):""}</div>
</div>`;
});
$("stages").innerHTML=panels.join("")||'<div class="muted">no stages yet</div>';
$("s-running").textContent=`${running}/${model.stages.length}`;
$("s-stages").textContent=model.stages.length;
$("s-cost").textContent=fmtUSD(costSum);
$("s-gpu").textContent=gpuN?(gpuSum/gpuN).toFixed(0)+"%":"—";
}
function renderLogs(){
const t = model.tEnd || 0;
const tail=(model.allLogs||[]).filter(l=>l.at<=t).slice(-80);
$("logs").innerHTML=tail.map((l)=>
`<tr class="${l.stream==="stderr"?"stderr":""}">
<td class="t">${fmtClock(l.at)}</td>
<td class="s" style="color:${stageColor(l.stage)}">stage-${l.stage}</td>
<td class="msg">${escapeHtml(l.text)}</td>
</tr>`).join("");
}
function onFleetModel(m){
model=m;
$("run").textContent=(m.meta&&m.meta.run_id)?("run "+m.meta.run_id):"";
renderStages();
renderLogs();
}
// ── Data source ──────────────────────────────────────────────────────────────
// The orchestrator's RemoteVastaiPlugin re-serves the collector's folded fleet
// model. Seed from the one-shot API, then stream live updates over SSE (`vastai`).
fetch('/api/plugin/vastai/model').then(r=>r.json()).then(function(m){ if(m) onFleetModel(m); }).catch(function(){});
var es=new EventSource('/events');
es.addEventListener('vastai', function(e){ try{ onFleetModel(JSON.parse(e.data)); }catch(err){} });
es.addEventListener('done', function(){ es.close(); });
window.addEventListener('beforeunload', function(){ es.close(); });
</script>
</body>
</html>

View file

@ -1,71 +1,135 @@
//! Orchestrator "fleet" dashboard plugin — a remote-sourced vast.ai view.
//!
//! In production the diagnostics collector runs on a VPS: the stage containers
//! ship their vast.ai/host-metric records to it, and it folds them into a fleet
//! board model. The orchestrator runs locally and hosts the *full* swactor
//! dashboard (overview / actors / topology / distribution / netmap). To surface
//! the fleet alongside those live views, this plugin **pulls** the collector's
//! already-folded model (`GET {collector}/api/plugin/vastai/model`) ~1/s and
//! re-serves it verbatim under the same `"vastai"` name — so the Fleet page
//! renders identically to the standalone collector board, with no extra ingest.
//! ship their vast.ai/host-metric records to it. The orchestrator runs locally
//! and hosts the *full* swactor dashboard (overview / actors / topology /
//! distribution / netmap). To surface the fleet alongside those live views, this
//! plugin **subscribes** to the collector's raw record stream
//! (`GET {collector}/diag/stream/{run_id}`, an SSE feed of `LiveRecord`s) and
//! folds the vast.ai records locally — reusing the dashboard's own server-side
//! [`VastaiLivePlugin`] fold — then serves the result same-origin at
//! `/api/plugin/vastai/model` and over `/events` (the `vastai` event), so the
//! Fleet page renders without any cross-origin calls to the collector.
//!
//! It is the read mirror of the orchestrator's distribution broadcaster
//! ([`crate::dist_broadcast`]), which *pushes* its snapshot to the same
//! collector. The poll loop tolerates an unreachable collector: a failed fetch
//! leaves the last good model in place, so a transient blip never blanks the UI.
//! collector. The stream loop reconnects on drops and tolerates an unreachable
//! collector: until records arrive, the Fleet tab simply waits.
use std::collections::HashMap;
use std::sync::{Arc, Mutex};
use std::sync::Arc;
use std::time::Duration;
use dashboard::live_collector::VastaiLivePlugin;
use dashboard::plugin::{DashboardPlugin, PluginResponse};
use distribution::diagnostics::collector::protocol::{LiveRecord, RecordKind};
use futures_util::StreamExt;
/// The Fleet page. Reuses the standalone collector board's Fleet renderer but
/// with the orchestrator dashboard's nav, so it sits beside the live tabs.
/// Rebuild a [`LiveRecord`] from one stream frame's `data:` JSON. `LiveRecord` is
/// a serialize-only wire type, so we parse the fields by hand. Returns `None` for
/// non-vast.ai kinds (e.g. the pushed dist snapshot) so we skip them cheaply.
fn parse_vastai_record(data: &str) -> Option<LiveRecord> {
let v: serde_json::Value = serde_json::from_str(data).ok()?;
let kind = RecordKind::parse(v.get("kind")?.as_str()?)?;
if !kind.is_vastai() {
return None;
}
Some(LiveRecord {
run_id: v.get("run_id").and_then(|x| x.as_str()).unwrap_or("").to_string(),
node_id: v.get("node_id").and_then(|x| x.as_str()).unwrap_or("").to_string(),
kind,
recv_ms: v.get("recv_ms").and_then(|x| x.as_u64()).unwrap_or(0),
seq: v.get("seq").and_then(|x| x.as_u64()).unwrap_or(0),
body: v.get("body").cloned().unwrap_or(serde_json::Value::Null),
})
}
/// The Fleet page. Reuses the Swactor Runtime Dashboard layout/CSS so it sits
/// beside the live tabs (no scrubber / Live control — this is a live-only view).
const FLEET_HTML: &str = include_str!("fleet_page.html");
/// Read-only dashboard plugin backing `/plugin/vastai`. Holds the last good fleet
/// model JSON pulled from the remote collector; serves it to the SSE stream
/// (event `vastai`) and the seed API (`/api/plugin/vastai/model`).
/// Dashboard plugin (name `"vastai"`) backing `/plugin/vastai`. Wraps the
/// dashboard's [`VastaiLivePlugin`] (which buffers records and serves the fold)
/// and feeds it from the remote collector's record stream; overrides only the
/// HTML page so the Fleet tab wears the dashboard chrome.
pub struct RemoteVastaiPlugin {
/// Last successfully fetched fleet model JSON (the collector's folded board).
cached: Arc<Mutex<Option<String>>>,
inner: Arc<VastaiLivePlugin>,
}
impl RemoteVastaiPlugin {
pub fn new() -> Self {
Self {
cached: Arc::new(Mutex::new(None)),
inner: Arc::new(VastaiLivePlugin::new()),
}
}
/// Spawn the poll loop on `rt`. Every ~1s it GETs
/// `{collector_url}/api/plugin/vastai/model`; a successful response with a
/// non-`null` body replaces the cache, anything else (error, non-2xx, empty,
/// `null`) leaves the last good model untouched. The task ends when `rt`'s
/// runtime is dropped at end of run. Cadence matches [`crate::dist_broadcast`].
pub fn spawn_poller(&self, rt: &tokio::runtime::Handle, collector_url: String) {
/// Subscribe to the collector's `/diag/stream/{run_id}` SSE feed and fold each
/// vast.ai `LiveRecord` into the inner plugin. Reconnects every ~2s on drop or
/// while the collector is unreachable. Ends when `rt`'s runtime is dropped at
/// end of run. `run_id` must match what the stages ship under
/// (`SWACTOR_DIAG_RUN_ID`).
pub fn spawn_stream(&self, rt: &tokio::runtime::Handle, collector_url: String, run_id: String) {
let url = format!(
"{}/api/plugin/vastai/model",
collector_url.trim_end_matches('/')
"{}/diag/stream/{}",
collector_url.trim_end_matches('/'),
run_id
);
let cached = Arc::clone(&self.cached);
let inner = Arc::clone(&self.inner);
eprintln!("pp-orchestrator: fleet tab streaming vast.ai records from {url}");
rt.spawn(async move {
let http = reqwest::Client::new();
let mut ticker = tokio::time::interval(Duration::from_secs(1));
ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip);
// Log the first record and the first error once, so a blank Fleet tab
// is easy to localize (collector empty vs. stream unreachable) without
// spamming the orchestrator log.
let mut logged_first = false;
let mut logged_err = false;
loop {
ticker.tick().await;
let body = match http.get(&url).send().await {
Ok(resp) if resp.status().is_success() => resp.text().await.ok(),
_ => None,
};
if let Some(body) = body {
let trimmed = body.trim();
if !trimmed.is_empty() && trimmed != "null" {
*cached.lock().unwrap() = Some(body);
match http.get(&url).send().await {
Ok(resp) if resp.status().is_success() => {
let mut stream = resp.bytes_stream();
let mut buf: Vec<u8> = Vec::new();
while let Some(chunk) = stream.next().await {
let Ok(chunk) = chunk else { break };
buf.extend_from_slice(&chunk);
// SSE frames are separated by a blank line. Parse on
// byte boundaries so a chunk split mid-frame is safe.
while let Some(idx) = buf.windows(2).position(|w| w == b"\n\n") {
let frame: Vec<u8> = buf.drain(..idx + 2).collect();
let Ok(text) = std::str::from_utf8(&frame) else {
continue;
};
for line in text.lines() {
let Some(data) = line.strip_prefix("data:") else {
continue;
};
let data = data.trim_start();
if let Some(rec) = parse_vastai_record(data) {
inner.ingest(&rec);
if !logged_first {
logged_first = true;
eprintln!(
"pp-orchestrator: fleet tab is receiving vast.ai records from the collector"
);
}
}
}
}
}
}
Ok(resp) => {
if !logged_err {
logged_err = true;
eprintln!("pp-orchestrator: fleet stream got HTTP {}", resp.status());
}
}
Err(e) => {
if !logged_err {
logged_err = true;
eprintln!("pp-orchestrator: fleet stream error (collector reachable?): {e}");
}
}
}
tokio::time::sleep(Duration::from_secs(2)).await;
}
});
}
@ -83,28 +147,17 @@ impl DashboardPlugin for RemoteVastaiPlugin {
}
fn snapshot_json(&self) -> Option<String> {
self.cached.lock().unwrap().clone()
self.inner.snapshot_json()
}
fn handle_request(
&self,
method: &str,
path: &str,
_query: &HashMap<String, String>,
_body: &[u8],
query: &HashMap<String, String>,
body: &[u8],
) -> PluginResponse {
// The page seeds from `/api/plugin/vastai/model` (the route requires a
// non-empty trailing segment), matching the standalone collector board.
match (method, path) {
("GET", "" | "model") => PluginResponse::json(
self.cached
.lock()
.unwrap()
.clone()
.unwrap_or_else(|| "null".into()),
),
_ => PluginResponse::not_found(),
}
self.inner.handle_request(method, path, query, body)
}
fn html_page(&self) -> Option<&str> {

View file

@ -1,6 +1,7 @@
pub mod diag;
pub mod dist_broadcast;
pub mod dist_plugin;
pub mod fleet_plugin;
pub mod messages;
pub mod netmap_plugin;
pub mod profile;

View file

@ -7,10 +7,14 @@
<style>
* { margin: 0; padding: 0; box-sizing: border-box; }
body { font-family: 'Menlo','Consolas','Monaco',monospace; background:#0f1117; color:#e0e0e0; font-size:13px; }
.header { display:flex; align-items:center; gap:16px; padding:10px 16px; background:#161922; border-bottom:1px solid #262b38; }
.header h1 { font-size:15px; font-weight:600; color:#7aa2f7; }
.nav a { color:#9aa5b1; text-decoration:none; margin-right:14px; }
.nav a:hover, .nav a.active { color:#7aa2f7; }
.header { display:flex; align-items:center; justify-content:space-between; padding:12px 20px; background:#161822; border-bottom:1px solid #2a2d3e; }
.header h1 { font-size:16px; font-weight:600; color:#fff; }
.header-left { display:flex; align-items:center; }
.nav-links { display:flex; gap:4px; margin-left:20px; }
.nav-link { color:#888; text-decoration:none; font-size:12px; padding:4px 10px; border-radius:3px; transition:color 0.2s; }
.nav-link:hover { color:#e0e0e0; }
.nav-link.active { color:#fff; background:#2a2d3e; }
.header-right { display:flex; align-items:center; gap:12px; }
.wrap { padding:16px; display:grid; grid-template-columns: 1fr; gap:16px; }
.card { background:#161922; border:1px solid #262b38; border-radius:6px; padding:12px; }
.card h2 { font-size:13px; color:#c0caf5; margin-bottom:10px; border-bottom:1px solid #262b38; padding-bottom:6px; }
@ -33,14 +37,18 @@
</head>
<body>
<div class="header">
<h1>Orchestrator · Net map</h1>
<div class="nav">
<a href="/">Overview</a>
<a href="/actors">Actors</a>
<a href="/plugin/distribution">Distribution</a>
<a href="/plugin/netmap" class="active">Net map</a>
<div class="header-left">
<h1>Swactor Runtime Dashboard</h1>
<nav class="nav-links">
<a href="/" class="nav-link">Overview</a>
<a href="/actors" class="nav-link">Actors</a>
<a href="/topology" class="nav-link">Topology</a>
<a href="/plugin/distribution" class="nav-link">Distribution</a>
<a href="/plugin/netmap" class="nav-link active">Net map</a>
<a href="/plugin/vastai" class="nav-link">Fleet</a>
</nav>
</div>
<span id="node" class="mono" style="margin-left:auto"></span>
<div class="header-right"><span id="node" class="mono"></span></div>
</div>
<div class="wrap">

View file

@ -1,4 +1,4 @@
//! Orchestrator helpers for `pp-smoke-run`.
//! Orchestrator helpers for `pp-orchestrator`.
//!
//! Extracted from the binary so the spawn-chain and convergence-wait logic
//! can be unit-tested without provisioning child processes or driving a
@ -17,7 +17,7 @@
//!
//! Tests inject a fake command builder (e.g. `sh -c "echo PP_GPU_NODE_ADDR
//! <hex> <direct>; sleep 60"`) so the chain can be exercised end-to-end
//! without `pp-gpu-node` on disk.
//! without `pp-worker` on disk.
use std::io::{BufRead, BufReader};
use std::process::{Child, ChildStdout, Command, Stdio};
@ -118,7 +118,7 @@ impl Drop for ChainGuard {
// to the container's PID 1 and only then does the container exit
// and `--rm` clean up. A bare SIGKILL bypasses that proxy and
// orphans the container. For the no-wrapper host case the cost is
// ~tens of ms — pp-gpu-node has no SIGTERM handler so it exits
// ~tens of ms — pp-worker has no SIGTERM handler so it exits
// immediately on receipt.
//
// The budget is *per stage*, not shared across the whole chain.
@ -148,7 +148,7 @@ impl Drop for ChainGuard {
// child already exited (we ignore the error either way).
let _ = stage.child.kill();
let _ = stage.child.wait();
eprintln!("pp-smoke-run: stopped stage {} child pid {pid}", stage.stage);
eprintln!("pp-orchestrator: stopped stage {} child pid {pid}", stage.stage);
}
}
}
@ -307,7 +307,7 @@ fn read_stage_address(
});
rx.recv_timeout(timeout).map_err(|_| {
eprintln!(
"pp-smoke-run: stage {stage} did not announce PP_GPU_NODE_ADDR within {:.0}s",
"pp-orchestrator: stage {stage} did not announce PP_GPU_NODE_ADDR within {:.0}s",
timeout.as_secs_f32()
);
})

View file

@ -1,6 +1,6 @@
//! Picks the iroh `RelayMode` from process environment.
//!
//! `SWACTOR_IROH_RELAY_URL` — when set, both pp-smoke-run and pp-gpu-node
//! `SWACTOR_IROH_RELAY_URL` — when set, both pp-orchestrator and pp-worker
//! use `RelayMode::Custom(<url>)` instead of the canary default. The
//! orchestrator's [`DiagEnv`](crate::vastai::DiagEnv) propagates the same
//! var into every rented container so the whole cluster homes onto one

View file

@ -121,7 +121,7 @@ pub enum StageActorStatus {
/// this enum by the bridge actors; `Process` is adapted from the worker
/// subprocess; `SetNeighbors` and `Reset` are control messages.
///
/// `SetNeighbors` is a one-shot setup message used by the `pp-gpu-node`
/// `SetNeighbors` is a one-shot setup message used by the `pp-worker`
/// binary to inject the resolved addresses of neighbouring stages and the
/// orchestrator after SWIM gossip has propagated them. Each field is
/// optional; only the fields relevant to the actor's role need to be set.
@ -143,7 +143,7 @@ pub enum StageMsg {
/// Tear down the running worker subprocess and spawn a fresh one,
/// re-exec'ing the on-disk worker script so an edited
/// `pp_tinygrad_worker.py` is picked up without restarting the node.
/// Driven by a `SIGHUP` to `pp-gpu-node`.
/// Driven by a `SIGHUP` to `pp-worker`.
ReloadWorker,
}
@ -1184,18 +1184,18 @@ impl ActorInterface for StageActor {
// stderr. The worker's stderr is otherwise consumed here and
// only re-emitted on the `worker_exit_detail` diagnostics
// event, which is invisible when no collector is configured
// (the common bare-deploy case). pp-gpu-node's stderr is
// (the common bare-deploy case). pp-worker's stderr is
// captured by the container log, so this makes a crashed
// worker self-diagnosing without a diagnostics backend.
if !normal_exit {
eprintln!(
"pp-gpu-node: worker exited abnormally (code={exit_code:?} signal={signal:?}); stderr tail:"
"pp-worker: worker exited abnormally (code={exit_code:?} signal={signal:?}); stderr tail:"
);
for line in &stderr_tail {
eprintln!(" worker| {line}");
}
if let Some(tb) = traceback.as_deref() {
eprintln!("pp-gpu-node: worker python traceback:\n{tb}");
eprintln!("pp-worker: worker python traceback:\n{tb}");
}
}

View file

@ -15,6 +15,7 @@
//!
//! All functions accept a `base_url` so tests can point at a wiremock server.
use std::io::{IsTerminal, Write};
use std::time::Duration;
use reqwest::Client;
@ -28,7 +29,7 @@ pub struct InstanceInfo {
/// Diagnostics env-var bundle forwarded to rented stage containers.
///
/// `pp-gpu-node::diag::install_from_env` reads `SWACTOR_DIAG_*` on boot
/// `pp-worker::diag::install_from_env` reads `SWACTOR_DIAG_*` on boot
/// inside each container to decide whether to enable the aggregator and
/// where to ship to. The orchestrator-side caller of [`lease_chain`]
/// builds this from its own process env (typically the same vars the
@ -111,6 +112,12 @@ pub struct Offer {
/// that gouge on bandwidth, since bandwidth price is a per-host policy.
#[serde(default)]
pub host_id: Option<u64>,
/// vast.ai host verification state: `"verified"`, `"unverified"` (never
/// tested), or `"deverified"` (was verified, then failed vast's checks).
/// Deverified hosts recurrently fail CDI GPU-device injection at container
/// start despite a high `reliability2`, so they are dropped by default.
#[serde(default)]
pub verification: Option<String>,
}
/// Connection details for a running instance.
@ -237,6 +244,118 @@ fn next_eligible_offer<'a>(
})
}
/// The offers [`lease_chain`] would rent on the happy path: the cheapest
/// `num_stages` on distinct hosts, in stage order (stage 0 = cheapest). Mirrors
/// the distinct-host draw in [`next_eligible_offer`] (a `None` host id is never
/// deduped, matching that helper). Actual picks can differ only if a create
/// fails and the chain falls through to the next survivor — so the confirmed
/// cost is the floor, not a ceiling.
fn plan_picks(pool: &[Offer], num_stages: u32) -> Vec<&Offer> {
let mut picks: Vec<&Offer> = Vec::with_capacity(num_stages as usize);
let mut used: std::collections::HashSet<u64> = std::collections::HashSet::new();
for o in pool {
if picks.len() == num_stages as usize {
break;
}
if let Some(h) = o.host_id {
if !used.insert(h) {
continue; // host already claimed by an earlier pick
}
}
picks.push(o);
}
picks
}
/// `PP_ASSUME_YES`: skip the interactive lease confirmation (for scripted / CI
/// runs that intend to rent without a human at the keyboard). Truthy = any
/// non-empty value other than `0` / `false` / `no`.
fn assume_yes() -> bool {
std::env::var("PP_ASSUME_YES")
.ok()
.map(|v| {
let v = v.trim().to_ascii_lowercase();
!v.is_empty() && v != "0" && v != "false" && v != "no"
})
.unwrap_or(false)
}
/// Print the planned lease + its hourly cost and, when running interactively,
/// require an explicit `y`/`N` before any instance is created.
///
/// This is the cost guardrail: the offer search can legitimately land on an
/// expensive card when the cheap pool is thin, and an unconfirmed lease once put
/// a 3-stage run onto an A100. The prompt is the last gate before money is spent.
///
/// To keep scripted runs and the (HTTP-mocked) test suite unaffected, the
/// confirmation is *skipped* (the lease proceeds) when `PP_ASSUME_YES` is set or
/// when stdin is not a TTY — there is no human to answer in those cases. The cost
/// summary is always logged regardless.
fn confirm_lease(pool: &[Offer], num_stages: u32, cost: &CostModel) -> Result<(), String> {
let picks = plan_picks(pool, num_stages);
let total_dph: f64 = picks.iter().map(|o| o.dph_total).sum();
let total_eff: f64 = picks.iter().map(|o| cost.effective_price(o)).sum();
eprintln!(
"pp-orchestrator: lease plan — {num_stages} stage(s), cheapest on distinct hosts:"
);
for (i, o) in picks.iter().enumerate() {
eprintln!(
" stage {i} {:<14} {:>8} ${:.3}/hr [{}] host {}",
o.gpu_name,
o.gpu_ram
.map(|r| format!("{:.0}MB", r))
.unwrap_or_else(|| "?MB".into()),
o.dph_total,
o.geolocation.as_deref().unwrap_or("?"),
o.host_id
.map(|h| h.to_string())
.unwrap_or_else(|| "?".into()),
);
}
if picks.len() < num_stages as usize {
eprintln!(
" WARNING: only {} distinct-host offer(s) available for {num_stages} stage(s) — \
the lease will likely fail to fill the chain.",
picks.len(),
);
}
let eff_note = if (total_eff - total_dph).abs() > 1e-6 {
format!(" (image-pull priced in: ${total_eff:.3}/hr eff)")
} else {
String::new()
};
eprintln!(
" TOTAL ${total_dph:.3}/hr (~${:.2}/day){eff_note}",
total_dph * 24.0,
);
if assume_yes() {
eprintln!("pp-orchestrator: PP_ASSUME_YES set — proceeding without confirmation");
return Ok(());
}
if !std::io::stdin().is_terminal() {
eprintln!(
"pp-orchestrator: stdin is not a TTY — proceeding without interactive confirmation \
(set PP_ASSUME_YES=1 to silence this)"
);
return Ok(());
}
eprint!("Proceed with renting these {num_stages} instance(s)? [y/N]: ");
let _ = std::io::stderr().flush();
let mut line = String::new();
std::io::stdin()
.read_line(&mut line)
.map_err(|e| format!("failed to read lease confirmation: {e}"))?;
let ans = line.trim().to_ascii_lowercase();
if ans == "y" || ans == "yes" {
Ok(())
} else {
Err("operator declined the lease (cost not confirmed); no instances were created".into())
}
}
/// Build the ranked survivor pool for a heterogeneous PP lease in a single
/// query. Replaces the old per-stage `find_offer` / `find_offer_chain`.
///
@ -253,40 +372,55 @@ pub async fn select_offer_pool(
gpu_name: &str,
num_stages: u32,
) -> Result<Vec<Offer>, String> {
// Hard gates expressed server-side. reliability2 >= 0.995 and
// cuda_max_good >= 12.6 (our CUDA-12.6 base image) drop hosts that
// recurrently fail container init / CDI device injection; num_gpus == 1
// keeps us from renting a multi-GPU rig per stage. Network speed can't be
// probed before renting, so we trust vast.ai's measured inet figures and
// gate on a configurable minimum (PP_MIN_INET_DOWN_MBPS, default 100; the
// upload gate is off by default).
// Hard gates expressed server-side. reliability2 (PP_MIN_RELIABILITY,
// default 0.95 — "semi-reliable", NOT near-perfect) and cuda_max_good >=
// 12.6 (our CUDA-12.6 base image) drop hosts that recurrently fail
// container init / CDI device injection; num_gpus == 1 keeps us from
// renting a multi-GPU rig per stage. Network speed can't be probed before
// renting, so we trust vast.ai's measured inet figures and gate on a
// configurable minimum (PP_MIN_INET_DOWN_MBPS, default 100; the upload
// gate is off by default).
let mut query = serde_json::json!({
"rentable": {"eq": true},
"rented": {"eq": false},
"reliability2": {"gte": 0.995},
"reliability2": {"gte": env_min_reliability()},
"cuda_max_good": {"gte": 12.6},
"verified": {"eq": true},
"direct_port_count": {"gte": 1},
"num_gpus": {"eq": 1},
"inet_down": {"gte": env_min_inet_down_mbps()},
// Cap the response so one query covers N distinct hosts even after the
// per-model cheap drop, without paging.
"limit": 512,
// vast.ai treats `limit` as a SCAN BUDGET (machines examined in the
// engine's default high-perf-first order), NOT a result cap: a small
// limit returns *fewer* matches because it never reaches the cheap
// commodity hosts that rank low. Empirically `limit:512` returned ~184
// ram>=8000 offers while `limit:5000` returned ~1639 — the missing
// ~1450 included the cheap 3090/3060 supply, so a small limit alone
// skews the pool toward datacenter cards. Set high so the survivor pool
// reflects the whole market.
"limit": 5000,
});
if let Some(up) = env_min_inet_up_mbps() {
query["inet_up"] = serde_json::json!({"gte": up});
}
// VRAM mode spans a heterogeneous card set (each PP stage is an independent
// process exchanging fp16 hidden state, so stages need not share a model —
// only enough VRAM for their block slice). Model-name mode is the historical
// default and implicitly bounds cost to that one cheap model.
match env_min_gpu_ram_mb() {
Some(min_ram) => {
query["gpu_ram"] = serde_json::json!({"gte": min_ram});
}
None => {
query["gpu_name"] = serde_json::json!({"eq": gpu_name});
}
// vast.ai's `verified` flag means the host passed vast's own datacenter
// vetting. AND'd with the other gates it discarded ~90% of supply — almost
// every cheap consumer 3090/3060 is unverified — so it is OFF by default and
// reliability2 (above) carries the quality floor. Opt back in with
// PP_REQUIRE_VERIFIED=1 for a vetted-hosts-only pool.
if env_require_verified() {
query["verified"] = serde_json::json!({"eq": true});
}
// GPU selection is two independent, optional filters — neither is required.
// A VRAM floor (PP_GPU_MIN_RAM_MB) spans a heterogeneous card set (each PP
// stage is an independent process exchanging fp16 hidden state, so stages
// need not share a model — only enough VRAM for their block slice). A model
// pin (PP_GPU / `gpu_name`) restricts to one model. Unset both → the GPU
// itself isn't filtered and the quality gates above + cost ranking pick the
// host.
if let Some(min_ram) = env_min_gpu_ram_mb() {
query["gpu_ram"] = serde_json::json!({"gte": min_ram});
}
if !gpu_name.is_empty() {
query["gpu_name"] = serde_json::json!({"eq": gpu_name});
}
let url = format!(
"{base_url}/api/v0/bundles/?q={}",
@ -311,8 +445,16 @@ pub async fn select_offer_pool(
.map_err(|e| format!("select_offer_pool parse failed: {e}"))?;
// Post-filter in Rust: drop unknown/Chinese geolocations (Docker Hub and
// iroh relays are unreachable from behind the Great Firewall) and any
// blacklisted host (providers caught gouging on bandwidth).
// iroh relays are unreachable from behind the Great Firewall), any
// blacklisted host (providers caught gouging on bandwidth), and — by
// default — `deverified` hosts. vast.ai deverifies a host after it fails
// vast's own checks; in practice these recurrently fail CDI GPU-device
// injection at container start ("unresolvable CDI devices …/gpu=0") even
// though their `reliability2` stays ~0.99, which is why the reliability
// gate alone does not catch them. `unverified` (never-tested) hosts are
// kept — they hold the cheap consumer-GPU supply and usually start fine.
// PP_REQUIRE_VERIFIED already restricts the query to verified-only, in
// which case this filter is a no-op.
let blacklist = blacklisted_host_ids();
let reachable: Vec<Offer> = body
.offers
@ -323,6 +465,7 @@ pub async fn select_offer_pool(
.map_or(false, |g| !g.to_uppercase().contains("CN"))
})
.filter(|o| o.host_id.map_or(true, |h| !blacklist.contains(&h)))
.filter(|o| o.verification.as_deref() != Some("deverified"))
.collect();
let cost = CostModel::from_env();
@ -348,10 +491,10 @@ pub async fn select_offer_pool(
Ok(pool)
}
/// `PP_GPU_MIN_RAM_MB`: when set to a positive integer, the offer search
/// selects cards by VRAM (`gpu_ram >= N` MB) instead of by exact GPU model,
/// enabling a heterogeneous cluster. Unset / blank / zero → model-name mode
/// (the historical default, unchanged).
/// `PP_GPU_MIN_RAM_MB`: when set to a positive integer, adds a VRAM floor
/// (`gpu_ram >= N` MB) to the offer search, enabling a heterogeneous cluster.
/// Independent of the optional `PP_GPU` model pin; unset / blank / zero → no
/// VRAM filter.
fn env_min_gpu_ram_mb() -> Option<u64> {
std::env::var("PP_GPU_MIN_RAM_MB")
.ok()
@ -371,6 +514,31 @@ fn env_min_inet_down_mbps() -> f64 {
.unwrap_or(100.0)
}
/// `PP_MIN_RELIABILITY`: minimum vast.ai `reliability2` an offer must carry.
/// The old hardcoded 0.995, combined with the `verified` gate, admitted almost
/// only datacenter rigs (A100/H100) — every cheap consumer 3090/3060 sits at
/// 0.95–0.99 and/or is unverified, so the two gates AND'd together left zero
/// cheap cards and the lease was forced onto an expensive datacenter card.
/// Default 0.95 ("semi-reliable"); clamped to [0, 1]; 0 disables the gate.
fn env_min_reliability() -> f64 {
std::env::var("PP_MIN_RELIABILITY")
.ok()
.and_then(|s| s.trim().parse::<f64>().ok())
.filter(|&v| (0.0..=1.0).contains(&v))
.unwrap_or(0.95)
}
/// `PP_REQUIRE_VERIFIED`: when truthy (`1`/`true`/`yes`, case-insensitive),
/// restrict the search to vast.ai-verified hosts. Default off — the verified
/// flag AND'd with the other gates excluded nearly all cheap consumer GPUs, so
/// reliability2 carries the quality floor and unvetted hosts are admitted.
fn env_require_verified() -> bool {
std::env::var("PP_REQUIRE_VERIFIED")
.ok()
.map(|s| matches!(s.trim().to_ascii_lowercase().as_str(), "1" | "true" | "yes"))
.unwrap_or(false)
}
/// `PP_MIN_INET_UP_MBPS`: optional minimum reported upload speed (Mbps).
/// Default unset / 0 → no upload-speed gate.
fn env_min_inet_up_mbps() -> Option<f64> {
@ -677,7 +845,7 @@ pub async fn create_instance(
// Pin this stage's iroh identity so it survives a restart: re-read from
// PID 1's env on restart, the stage keeps the same node id and the
// pipeline name registry stays valid. See
// pp-gpu-node::stage_secret_from_env.
// pp-worker::stage_secret_from_env.
if let Some(secret) = stage_secret {
env["PP_STAGE_SECRET"] = serde_json::Value::String(secret.to_string());
}
@ -717,7 +885,7 @@ pub async fn create_instance(
// Run the PID-1 supervisor (not the worker directly): it brings up
// sshd deterministically and keeps the container — and the shell —
// alive if the worker crashes. No `exec` of the worker: the supervisor
// owns PID 1 and runs pp-gpu-node as a child.
// owns PID 1 and runs pp-worker as a child.
"onstart": "/usr/local/bin/pp_entrypoint.sh 2>&1",
// Every stage fetch()s the FULL gguf (whole file mmap'd by
// from_gguf), regardless of which layers it runs. qwen3:30b-a3b
@ -1135,6 +1303,11 @@ pub async fn lease_chain(
.await
.map_err(|e| format!("lease_chain: {e}"))?;
// Cost guardrail: show what we're about to rent and (interactively) require
// a y/N before spending money. Runs before any create_instance, so a decline
// is a clean abort with nothing leased. Skipped for non-TTY / PP_ASSUME_YES.
confirm_lease(&pool, num_stages, &CostModel::from_env())?;
let mut tried_offer_ids: Vec<u64> = Vec::new();
let mut created: Vec<InstanceInfo> = Vec::with_capacity(num_stages as usize);
// Host ids leased by this chain, owned here so it survives across both
@ -1297,6 +1470,7 @@ mod tests {
inet_down_cost_per_tb: 0.0,
inet_up_cost_per_tb: 0.0,
host_id: Some(host),
verification: Some("verified".to_string()),
}
}

View file

@ -1,4 +1,4 @@
//! T-binary: drive the actual `pp-smoke-run --seed` and `pp-gpu-node`
//! T-binary: drive the actual `pp-orchestrator --seed` and `pp-worker`
//! binaries as child processes. TEST_SPEC §13 (stub workers) and §14
//! (real tinygrad workers, `#[ignore]`).
//!
@ -18,8 +18,8 @@ use std::sync::{Arc, Mutex};
use std::thread;
use std::time::{Duration, Instant};
const SMOKE_RUN_BIN: &str = env!("CARGO_BIN_EXE_pp-smoke-run");
const GPU_NODE_BIN: &str = env!("CARGO_BIN_EXE_pp-gpu-node");
const ORCHESTRATOR_BIN: &str = env!("CARGO_BIN_EXE_pp-orchestrator");
const WORKER_BIN: &str = env!("CARGO_BIN_EXE_pp-worker");
/// Default per-test budget for an N=5 stub-mode happy path: cluster build +
/// SWIM convergence + sequential worker boots + a short decode loop. Tests
@ -156,7 +156,7 @@ fn wait_for_n_children(parent_pid: u32, expected: usize, timeout: Duration) -> O
}
}
/// Extract the response text from `pp-smoke-run`'s stdout. The orchestrator
/// Extract the response text from `pp-orchestrator`'s stdout. The orchestrator
/// prints the response between two banner lines:
///
/// ```text
@ -194,14 +194,14 @@ struct SmokeRunOpts {
boot_delay_secs: Option<u32>,
}
/// Spawn `pp-smoke-run --seed` with the given options. Returns the spawned
/// Spawn `pp-orchestrator --seed` with the given options. Returns the spawned
/// process plus a shared log buffer that captures every stdout / stderr
/// line from `pp-smoke-run` AND every `pp-gpu-node` child (children inherit
/// line from `pp-orchestrator` AND every `pp-worker` child (children inherit
/// the orchestrator's stderr fd, so their messages land in the same buffer).
fn spawn_smoke_run(opts: &SmokeRunOpts) -> (Child, LogBuffer, LogBuffer) {
assert!(opts.num_stages >= 2);
let worker = worker_script();
let mut cmd = Command::new(SMOKE_RUN_BIN);
let mut cmd = Command::new(ORCHESTRATOR_BIN);
cmd.arg("--seed")
.arg("--num-stages")
.arg(opts.num_stages.to_string())
@ -210,7 +210,7 @@ fn spawn_smoke_run(opts: &SmokeRunOpts) -> (Child, LogBuffer, LogBuffer) {
.arg("--max-tokens")
.arg(opts.max_tokens.to_string())
.arg("--gpu-node")
.arg(GPU_NODE_BIN)
.arg(WORKER_BIN)
.arg("--worker")
.arg(&worker)
.stdout(Stdio::piped())
@ -228,13 +228,13 @@ fn spawn_smoke_run(opts: &SmokeRunOpts) -> (Child, LogBuffer, LogBuffer) {
cmd.env_remove("PP_BOOT_DELAY_SECS");
}
let mut child = cmd.spawn().expect("spawn pp-smoke-run");
let mut child = cmd.spawn().expect("spawn pp-orchestrator");
let stdout = child.stdout.take().expect("piped stdout");
let stderr = child.stderr.take().expect("piped stderr");
let stdout_buf = LogBuffer::default();
let stderr_buf = LogBuffer::default();
drain_stdout(stdout, "pp-smoke-run", stdout_buf.clone());
drain_stderr(stderr, "pp-smoke-run", stderr_buf.clone());
drain_stdout(stdout, "pp-orchestrator", stdout_buf.clone());
drain_stderr(stderr, "pp-orchestrator", stderr_buf.clone());
(child, stdout_buf, stderr_buf)
}
@ -251,13 +251,13 @@ fn run_to_completion(opts: &SmokeRunOpts) -> RunOutcome {
.unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not spawn {n} pp-gpu-node children within 60s")
panic!("pp-orchestrator did not spawn {n} pp-worker children within 60s")
});
let status = wait_with_timeout(&mut smoke, HAPPY_PATH_TIMEOUT).unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not exit within {:?}", HAPPY_PATH_TIMEOUT);
panic!("pp-orchestrator did not exit within {:?}", HAPPY_PATH_TIMEOUT);
});
RunOutcome {
status,
@ -278,7 +278,7 @@ impl RunOutcome {
fn require_success(&self) {
assert!(
self.status.success(),
"pp-smoke-run exited with {:?}\n--- stdout ---\n{}\n--- stderr (last 40 lines) ---\n{}",
"pp-orchestrator exited with {:?}\n--- stdout ---\n{}\n--- stderr (last 40 lines) ---\n{}",
self.status,
self.stdout.join("\n"),
self.stderr
@ -295,7 +295,7 @@ impl RunOutcome {
fn require_response_non_empty(&self) -> String {
let response = extract_response(&self.stdout).unwrap_or_else(|| {
panic!(
"pp-smoke-run stdout missing response banner; got:\n{}",
"pp-orchestrator stdout missing response banner; got:\n{}",
self.stdout.join("\n")
)
});
@ -310,7 +310,7 @@ impl RunOutcome {
for pid in &self.pre_exit_pids {
assert!(
!pid_alive(*pid),
"pp-gpu-node child pid {pid} still alive after pp-smoke-run exit"
"pp-worker child pid {pid} still alive after pp-orchestrator exit"
);
}
}
@ -436,7 +436,7 @@ fn binary_e2e_all_stages_register_pp_stage_index_names() {
// §13.2 — Failure / cleanup paths
// ───────────────────────────────────────────────────────────────────────
/// Spawn `pp-smoke-run` at `num_stages`, wait until every stage child is
/// Spawn `pp-orchestrator` at `num_stages`, wait until every stage child is
/// visible, kill `stage_to_kill`, and assert the orchestrator exits
/// non-zero with no surviving stage children.
fn kill_stage_and_expect_failure(num_stages: u32, stage_to_kill: u32) {
@ -448,7 +448,7 @@ fn kill_stage_and_expect_failure(num_stages: u32, stage_to_kill: u32) {
.unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not spawn {num_stages} children within 60s");
panic!("pp-orchestrator did not spawn {num_stages} children within 60s");
});
let victim = pids[stage_to_kill as usize];
@ -457,16 +457,16 @@ fn kill_stage_and_expect_failure(num_stages: u32, stage_to_kill: u32) {
let exit = wait_with_timeout(&mut smoke, Duration::from_secs(120)).unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not exit within 120s after killing stage {stage_to_kill}");
panic!("pp-orchestrator did not exit within 120s after killing stage {stage_to_kill}");
});
assert!(
!exit.success(),
"pp-smoke-run should fail when stage {stage_to_kill} (pid {victim}) is killed, got {exit:?}"
"pp-orchestrator should fail when stage {stage_to_kill} (pid {victim}) is killed, got {exit:?}"
);
for pid in &pids {
assert!(
!pid_alive(*pid),
"stage child pid {pid} still alive after pp-smoke-run exit"
"stage child pid {pid} still alive after pp-orchestrator exit"
);
}
}
@ -532,9 +532,9 @@ fn binary_e2e_no_orphaned_processes_after_failed_exit_3() {
#[test]
#[ignore]
fn binary_e2e_orchestrator_sigkilled_children_die_within_timeout() {
// SIGKILL `pp-smoke-run` itself once its children are up. The kernel
// delivers `SIGTERM` to each pp-gpu-node (via PR_SET_PDEATHSIG, set in
// pp-gpu-node's main), and each pp-gpu-node then dies — which also
// SIGKILL `pp-orchestrator` itself once its children are up. The kernel
// delivers `SIGTERM` to each pp-worker (via PR_SET_PDEATHSIG, set in
// pp-worker's main), and each pp-worker then dies — which also
// closes its Python worker's stdin, making the worker exit on EOF.
let opts = stub_opts(3);
let (mut smoke, _stdout, _stderr) = spawn_smoke_run(&opts);
@ -543,7 +543,7 @@ fn binary_e2e_orchestrator_sigkilled_children_die_within_timeout() {
.unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not spawn 3 children within 60s");
panic!("pp-orchestrator did not spawn 3 children within 60s");
});
sigkill(smoke_pid);
@ -554,7 +554,7 @@ fn binary_e2e_orchestrator_sigkilled_children_die_within_timeout() {
});
assert!(
cleaned_up,
"pp-gpu-node children {pids:?} still alive 10s after pp-smoke-run SIGKILL; \
"pp-worker children {pids:?} still alive 10s after pp-orchestrator SIGKILL; \
per-pid alive states: {:?}",
pids.iter().map(|p| (p, pid_alive(*p))).collect::<Vec<_>>()
);
@ -595,7 +595,7 @@ fn binary_e2e_orchestrator_can_resolve_pp_entry_after_n_stages_register() {
#[test]
#[ignore]
fn binary_e2e_pp_smoke_run_handles_slow_middle_stage_boot() {
fn binary_e2e_pp_orchestrator_handles_slow_middle_stage_boot() {
let opts = SmokeRunOpts {
num_stages: 4,
prompt: "Say hello".into(),
@ -612,7 +612,7 @@ fn binary_e2e_pp_smoke_run_handles_slow_middle_stage_boot() {
#[test]
#[ignore]
fn binary_e2e_pp_smoke_run_handles_slow_last_stage_boot() {
fn binary_e2e_pp_orchestrator_handles_slow_last_stage_boot() {
let opts = SmokeRunOpts {
num_stages: 4,
prompt: "Say hello".into(),
@ -675,18 +675,18 @@ fn real_tinygrad_at(num_stages: u32) {
.unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not spawn {num_stages} children within 120s");
panic!("pp-orchestrator did not spawn {num_stages} children within 120s");
});
let status = wait_with_timeout(&mut smoke, Duration::from_secs(1200)).unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not finish within 20m");
panic!("pp-orchestrator did not finish within 20m");
});
let stdout_lines = stdout.lines();
let stderr_joined = stderr.joined();
assert!(
status.success(),
"pp-smoke-run exited {status:?}\n--- stdout ---\n{}\n--- stderr (tail) ---\n{}",
"pp-orchestrator exited {status:?}\n--- stdout ---\n{}\n--- stderr (tail) ---\n{}",
stdout_lines.join("\n"),
stderr_joined
.lines()
@ -708,7 +708,7 @@ fn real_tinygrad_at(num_stages: u32) {
for pid in pre_exit_pids {
assert!(
!pid_alive(pid),
"pp-gpu-node child pid {pid} still alive after pp-smoke-run exit"
"pp-worker child pid {pid} still alive after pp-orchestrator exit"
);
}
}
@ -764,9 +764,9 @@ fn binary_e2e_real_tinygrad_response_matches_single_node_for_say_hello() {
let status = wait_with_timeout(&mut smoke, Duration::from_secs(1200)).unwrap_or_else(|| {
let _ = smoke.kill();
let _ = smoke.wait();
panic!("pp-smoke-run did not finish within 20m");
panic!("pp-orchestrator did not finish within 20m");
});
assert!(status.success(), "pp-smoke-run exited {status:?}");
assert!(status.success(), "pp-orchestrator exited {status:?}");
let response = extract_response(&stdout.lines()).expect("missing response banner");
assert_eq!(
response.trim(),

View file

@ -3,7 +3,7 @@
//! and verify clean teardown.
//!
//! TEST_SPEC §13b. Mirrors §13's happy path and at least one failure
//! scenario, but with each `pp-gpu-node` running inside its own
//! scenario, but with each `pp-worker` running inside its own
//! container instead of as a host process.
//!
//! Every test in this file is `#[ignore]`d and requires a working
@ -144,7 +144,7 @@ fn wait_until_prefix_drains(prefix: &str, timeout: Duration) {
}
/// Poll for "no leftover stage containers" with a short grace window.
/// A clean `pp-smoke-run` exit triggers `docker run --rm` teardown on
/// A clean `pp-orchestrator` exit triggers `docker run --rm` teardown on
/// each shim, but the daemon-side delete is not synchronous with the
/// CLI's exit, so we give the daemon a moment to catch up before we
/// call the run dirty.
@ -359,9 +359,9 @@ fn cargo_build_release_once() {
.arg(crate_dir().join("Cargo.toml"))
.arg("--release")
.arg("--bin")
.arg("pp-gpu-node")
.arg("pp-worker")
.arg("--bin")
.arg("pp-smoke-run")
.arg("pp-orchestrator")
.status()
.expect("invoke cargo build");
assert!(status.success(), "cargo build --release failed");
@ -570,7 +570,7 @@ fn docker_e2e_premature_container_exit_fails_fast() {
.expect("docker build (code)");
assert!(build.success(), "docker build (code) failed");
let smoke_bin = crate_dir().join("target/release/pp-smoke-run");
let smoke_bin = crate_dir().join("target/release/pp-orchestrator");
let worker_py = crate_dir().join("pp_tinygrad_worker.py");
assert!(smoke_bin.exists() && worker_py.exists());
@ -593,10 +593,10 @@ fn docker_e2e_premature_container_exit_fails_fast() {
.stdout(Stdio::piped())
.stderr(Stdio::piped())
.spawn()
.expect("spawn pp-smoke-run with docker shim");
.expect("spawn pp-orchestrator with docker shim");
// Drain the child's stderr (and stdout) into in-memory buffers via
// background reader threads. Without this any pp-smoke-run / shim
// background reader threads. Without this any pp-orchestrator / shim
// diagnostic message is swallowed and the test gives the reader
// nothing actionable on failure.
let stderr_buf = spawn_stream_collector(child.stderr.take().expect("child stderr piped"));
@ -610,8 +610,8 @@ fn docker_e2e_premature_container_exit_fails_fast() {
let _ = child.wait();
panic!(
"3 stage containers did not start within 180s\n\
--- pp-smoke-run stderr (tail) ---\n{}\n\
--- pp-smoke-run stdout (tail) ---\n{}",
--- pp-orchestrator stderr (tail) ---\n{}\n\
--- pp-orchestrator stdout (tail) ---\n{}",
tail_lines(&stderr_buf.snapshot(), 80),
tail_lines(&stdout_buf.snapshot(), 40),
);
@ -634,15 +634,15 @@ fn docker_e2e_premature_container_exit_fails_fast() {
// the docker-shim child exit promptly).
let deadline = Instant::now() + Duration::from_secs(60);
let status = loop {
match child.try_wait().expect("try_wait pp-smoke-run") {
match child.try_wait().expect("try_wait pp-orchestrator") {
Some(s) => break s,
None if Instant::now() >= deadline => {
let _ = child.kill();
let _ = child.wait();
panic!(
"pp-smoke-run did not exit within 60s of killing a container\n\
--- pp-smoke-run stderr (tail) ---\n{}\n\
--- pp-smoke-run stdout (tail) ---\n{}",
"pp-orchestrator did not exit within 60s of killing a container\n\
--- pp-orchestrator stderr (tail) ---\n{}\n\
--- pp-orchestrator stdout (tail) ---\n{}",
tail_lines(&stderr_buf.snapshot(), 80),
tail_lines(&stdout_buf.snapshot(), 40),
);
@ -653,7 +653,7 @@ fn docker_e2e_premature_container_exit_fails_fast() {
assert!(
!status.success(),
"expected non-zero exit after container kill, got {status:?}\n\
--- pp-smoke-run stderr (tail) ---\n{}",
--- pp-orchestrator stderr (tail) ---\n{}",
tail_lines(&stderr_buf.snapshot(), 80),
);

View file

@ -10,7 +10,7 @@
//! to stage 0, and so on until EOS or `max_tokens`. The last stage emits the
//! final `InferenceResponse` back to the orchestrator.
//!
//! The same shape the binary uses (see `pp_smoke_run.rs`), minus the child
//! The same shape the binary uses (see `pp_orchestrator.rs`), minus the child
//! subprocesses — actors live in this test process and addresses are wired up
//! directly without SWIM resolution.

View file

@ -2,7 +2,7 @@
//! lease chain (mocked HTTP), and the convergence-wait helper.
//!
//! None of these tests touch real iroh, real vast.ai, or the
//! `pp-gpu-node` binary. The spawn chain is exercised with a `sh -c`
//! `pp-worker` binary. The spawn chain is exercised with a `sh -c`
//! "fake child" that prints `PP_GPU_NODE_ADDR ...` and then sleeps; the
//! vast.ai chain uses `wiremock`; the convergence-wait helper is a
//! pure function fed a closure.
@ -151,7 +151,7 @@ fn spawn_chain_propagates_each_stage_peer_direct_to_successor() {
}
#[test]
fn spawn_chain_reads_pp_gpu_node_addr_in_order() {
fn spawn_chain_reads_addr_announcement_in_order() {
let tmp = TempDir::new("reads-in-order");
let pids_file = tmp.child("pids");