feat: data streams primitive #48

Merged
zacheryasc merged 1 commit from streams into master 2026-02-23 04:47:55 +00:00
34 changed files with 5647 additions and 11 deletions
Showing only changes of commit eea507e389 - Show all commits

21
Cargo.lock generated
View file

@ -5693,14 +5693,17 @@ dependencies = [
"dashboard",
"distribution",
"getrandom 0.2.17",
"iroh",
"proptest",
"serde",
"serde_json",
"shared-types",
"swactor",
"swactor-std",
"swactor-streams",
"tempfile",
"tiny_http",
"tokio",
"ureq",
]
@ -5718,6 +5721,7 @@ dependencies = [
"swactor",
"swactor-datastore",
"swactor-std",
"swactor-streams",
"toml 0.8.23",
]
@ -5744,6 +5748,23 @@ dependencies = [
"swactor",
]
[[package]]
name = "swactor-streams"
version = "0.1.0"
dependencies = [
"blake3",
"crossbeam-queue",
"distribution",
"getrandom 0.2.17",
"iroh",
"proptest",
"serde",
"shared-types",
"swactor",
"swactor-std",
"tokio",
]
[[package]]
name = "syn"
version = "2.0.116"

View file

@ -12,6 +12,7 @@ members = [
"crates/datastore",
"crates/shared-types",
"crates/swactor-node",
"crates/streams",
"tests/docker",
"crates/ci",
"xtask",

View file

@ -221,6 +221,7 @@ fn run_iroh(
relay_mode: RelayMode::Default,
node: node_config,
peer_auth: None,
additional_alpns: vec![],
};
let mut driver = IrohDriver::new(iroh_config).expect("failed to create iroh driver");

View file

@ -16,12 +16,19 @@ ureq = { version = "2", features = ["json"], optional = true }
getrandom = { version = "0.2", optional = true }
dashboard = { path = "../dashboard" }
swactor-std = { path = "../std" }
swactor-streams = { path = "../streams" }
tokio = { version = "1", features = ["sync", "rt", "time"] }
[dev-dependencies]
serde_json = "1"
proptest = "1"
tempfile = "3"
swactor = { path = "../.." }
swactor-streams = { path = "../streams" }
swactor-std = { path = "../std" }
distribution = { path = "../distribution", features = ["iroh"] }
tokio = { version = "1", features = ["rt-multi-thread", "macros", "io-util", "time"] }
iroh = "0.96"
ureq = { version = "2", features = ["json"] }
[features]

View file

@ -5,17 +5,27 @@
//! Delete, List, Status) and receive responses. Also routes incoming network
//! protocol messages to the appropriate internal actors.
use std::collections::BTreeMap;
use std::collections::{BTreeMap, HashMap};
use std::sync::Arc;
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
use swactor::runtime::Runtime;
use distribution::types::NodeId;
use crate::actors::stream_downloader::StreamDownloader;
use crate::actors::stream_server::StreamServer;
use crate::chunking::chunk_blob;
use crate::messages::{
BlobStoreMsg, DatastoreNodeMsg, DatastoreResponse, MetadataMsg,
};
use crate::types::{ContentHash, DatastoreConfig, ObjectEntry};
use crate::types::{ContentHash, DatastoreConfig, ObjectEntry, ObjectManifest};
/// Progress from a partially-completed stream download, used for resume.
struct PartialDownload {
_source_node: [u8; 32],
chunks_completed: u64,
}
/// Top-level coordinator actor for the datastore.
///
@ -26,6 +36,12 @@ pub struct DatastoreNode {
blob_store: ActorAddress,
metadata: ActorAddress,
config: DatastoreConfig,
// Stream support (configured lazily via ConfigureStreams)
runtime: Option<Arc<Runtime>>,
tokio_handle: Option<tokio::runtime::Handle>,
stream_manager: Option<ActorAddress>,
// Resume state for interrupted stream downloads
partial_downloads: HashMap<ContentHash, PartialDownload>,
}
impl DatastoreNode {
@ -40,6 +56,10 @@ impl DatastoreNode {
blob_store,
metadata,
config,
runtime: None,
tokio_handle: None,
stream_manager: None,
partial_downloads: HashMap::new(),
}
}
@ -231,6 +251,138 @@ impl DatastoreNode {
},
);
}
// ── Stream-based transfer handlers ───────────────────────────────────
fn handle_configure_streams(
&mut self,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
) {
self.stream_manager = Some(stream_manager);
self.tokio_handle = Some(tokio_handle);
self.runtime = Some(runtime);
}
fn handle_download_via_stream(
&self,
ctx: &Ctx,
content_hash: ContentHash,
source_node: [u8; 32],
reply_to: ActorAddress,
) {
let (stream_manager, tokio_handle, runtime) =
match (&self.stream_manager, &self.tokio_handle, &self.runtime) {
(Some(sm), Some(th), Some(rt)) => (*sm, th.clone(), Arc::clone(rt)),
_ => {
let _ = ctx.send(
reply_to,
DatastoreResponse::TransferFailed {
reason: "stream support not configured".into(),
},
);
return;
}
};
// Check for partial progress from a previous attempt
let skip_chunks = self
.partial_downloads
.get(&content_hash)
.map(|p| p.chunks_completed)
.unwrap_or(0);
let downloader = StreamDownloader::new(
content_hash,
source_node,
ctx.self_addr(),
self.blob_store,
reply_to,
stream_manager,
tokio_handle,
runtime,
skip_chunks,
);
let _ = ctx.spawn(downloader);
}
fn handle_stream_offer(
&self,
ctx: &Ctx,
stream_id: swactor_streams::types::StreamId,
content_hash: ContentHash,
_from_node: [u8; 32],
stream_manager: ActorAddress,
resume_from_chunk: u64,
) {
let (tokio_handle, runtime) = match (&self.tokio_handle, &self.runtime) {
(Some(th), Some(rt)) => (th.clone(), Arc::clone(rt)),
_ => return,
};
let server = StreamServer::new(
stream_id,
content_hash,
self.blob_store,
stream_manager,
tokio_handle,
runtime,
resume_from_chunk,
);
let _ = ctx.spawn(server);
}
fn handle_stream_download_complete(
&mut self,
ctx: &Ctx,
content_hash: ContentHash,
manifest: ObjectManifest,
reply_to: ActorAddress,
) {
// Clear any partial progress now that download is complete
self.partial_downloads.remove(&content_hash);
let entry = ObjectEntry {
content_hash,
name: None,
node_id: self.node_id,
tags: BTreeMap::new(),
size_bytes: manifest.total_size,
created_at: 0,
};
let _ = ctx.send(
self.metadata,
MetadataMsg::PutObject {
entry,
manifest,
reply_to,
},
);
}
fn handle_stream_download_failed(
&mut self,
ctx: &Ctx,
content_hash: ContentHash,
reason: String,
chunks_completed: u64,
source_node: [u8; 32],
reply_to: ActorAddress,
) {
// Store partial progress so next attempt can resume
if chunks_completed > 0 {
self.partial_downloads.insert(
content_hash,
PartialDownload {
_source_node: source_node,
chunks_completed,
},
);
}
let _ = ctx.send(reply_to, DatastoreResponse::TransferFailed { reason });
}
}
impl ActorInterface for DatastoreNode {
@ -277,6 +429,34 @@ impl ActorInterface for DatastoreNode {
DatastoreNodeMsg::IncomingListObjects { request, reply_to } => {
self.handle_incoming_list_objects(ctx, request, reply_to)
}
DatastoreNodeMsg::DownloadViaStream {
content_hash,
source_node,
reply_to,
} => self.handle_download_via_stream(ctx, content_hash, source_node, reply_to),
DatastoreNodeMsg::HandleStreamOffer {
stream_id,
content_hash,
from_node,
stream_manager,
resume_from_chunk,
} => self.handle_stream_offer(ctx, stream_id, content_hash, from_node, stream_manager, resume_from_chunk),
DatastoreNodeMsg::StreamDownloadComplete {
content_hash,
manifest,
reply_to,
} => self.handle_stream_download_complete(ctx, content_hash, manifest, reply_to),
DatastoreNodeMsg::StreamDownloadFailed {
content_hash,
reason,
chunks_completed,
reply_to,
} => self.handle_stream_download_failed(ctx, content_hash, reason, chunks_completed, [0; 32], reply_to),
DatastoreNodeMsg::ConfigureStreams {
stream_manager,
tokio_handle,
runtime,
} => self.handle_configure_streams(stream_manager, tokio_handle, runtime),
}
}
}

View file

@ -2,6 +2,9 @@ pub mod blob_store;
pub mod datastore_node;
pub mod gateway;
pub mod metadata;
pub mod stream_downloader;
pub mod stream_listener;
pub mod stream_server;
pub mod transfer;
pub use blob_store::BlobStoreActor;

View file

@ -0,0 +1,171 @@
//! StreamDownloader — opens a stream to a remote node and downloads a blob.
//!
//! Lifecycle:
//! 1. on_start: sends Open to StreamManager
//! 2. StreamReady: spawns a tokio task for I/O, then stops self
//! 3. tokio task: recv_blob, write chunks to BlobStore, notify DatastoreNode
use std::sync::Arc;
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
use swactor::runtime::Runtime;
use swactor_streams::messages::{StreamManagerMsg, StreamNotification};
use swactor_streams::types::{StreamConfig, StreamMode};
use crate::blob_transfer::{encode_metadata, recv_blob, BlobTransferMetadata};
use crate::messages::{BlobStoreMsg, DatastoreNodeMsg};
use crate::types::ContentHash;
pub struct StreamDownloader {
content_hash: ContentHash,
source_node: [u8; 32],
datastore_node: ActorAddress,
blob_store: ActorAddress,
reply_to: ActorAddress,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
skip_chunks: u64,
}
impl StreamDownloader {
#[allow(clippy::too_many_arguments)]
pub fn new(
content_hash: ContentHash,
source_node: [u8; 32],
datastore_node: ActorAddress,
blob_store: ActorAddress,
reply_to: ActorAddress,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
skip_chunks: u64,
) -> Self {
Self {
content_hash,
source_node,
datastore_node,
blob_store,
reply_to,
stream_manager,
tokio_handle,
runtime,
skip_chunks,
}
}
}
impl ActorInterface for StreamDownloader {
type Incoming = StreamNotification;
type Response = ();
fn on_start(&mut self, ctx: &Ctx) {
let resume = if self.skip_chunks > 0 {
Some(self.skip_chunks)
} else {
None
};
let meta = BlobTransferMetadata {
content_hash: self.content_hash,
resume_from_chunk: resume,
};
let config = StreamConfig {
metadata: encode_metadata(&meta),
stripe_count: 1, // blob transfer is sequential — one stripe avoids empty-stripe Closed races
..Default::default()
};
let _ = ctx.send(
self.stream_manager,
StreamManagerMsg::Open {
target_node: self.source_node,
mode: StreamMode::BlobTransfer,
config,
reply_to: ctx.self_addr(),
},
);
}
fn handle(&mut self, ctx: &Ctx, msg: StreamNotification) {
match msg {
StreamNotification::StreamReady { handle, .. } => {
let stream_handle = match handle.take() {
Some(h) => h,
None => return,
};
let runtime = Arc::clone(&self.runtime);
let blob_store = self.blob_store;
let datastore_node = self.datastore_node;
let reply_to = self.reply_to;
let content_hash = self.content_hash;
let skip_chunks = self.skip_chunks;
self.tokio_handle.spawn(async move {
let (_send, mut recv) = (stream_handle.send, stream_handle.recv);
match recv_blob(&mut recv, skip_chunks).await {
Ok(received) => {
// Write chunks to BlobStore (fire-and-forget)
for (hash, data) in &received.chunks {
let _ = runtime.send_to(
blob_store,
BlobStoreMsg::WriteChunk {
hash: *hash,
data: data.clone(),
reply_to: datastore_node, // response ignored
},
);
}
// Write manifest to BlobStore (fire-and-forget)
let _ = runtime.send_to(
blob_store,
BlobStoreMsg::WriteManifest {
manifest: received.manifest.clone(),
reply_to: datastore_node, // response ignored
},
);
// Notify DatastoreNode of completion
let _ = runtime.send_to(
datastore_node,
DatastoreNodeMsg::StreamDownloadComplete {
content_hash,
manifest: received.manifest,
reply_to,
},
);
}
Err(e) => {
let _ = runtime.send_to(
datastore_node,
DatastoreNodeMsg::StreamDownloadFailed {
content_hash,
reason: e.to_string(),
chunks_completed: skip_chunks,
reply_to,
},
);
}
}
});
ctx.stop_self();
}
StreamNotification::StreamFailed { error, .. } => {
let _ = ctx.send(
self.datastore_node,
DatastoreNodeMsg::StreamDownloadFailed {
content_hash: self.content_hash,
reason: error.to_string(),
chunks_completed: self.skip_chunks,
reply_to: self.reply_to,
},
);
ctx.stop_self();
}
_ => {}
}
}
}

View file

@ -0,0 +1,75 @@
//! StreamListener — listens for incoming BlobTransfer stream offers and
//! forwards them to DatastoreNode for handling.
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
use swactor_streams::messages::{StreamManagerMsg, StreamNotification};
use swactor_streams::types::StreamMode;
use crate::blob_transfer::parse_metadata;
use crate::messages::DatastoreNodeMsg;
pub struct StreamListener {
datastore_node: ActorAddress,
stream_manager: ActorAddress,
}
impl StreamListener {
pub fn new(datastore_node: ActorAddress, stream_manager: ActorAddress) -> Self {
Self {
datastore_node,
stream_manager,
}
}
}
impl ActorInterface for StreamListener {
type Incoming = StreamNotification;
type Response = ();
fn on_start(&mut self, ctx: &Ctx) {
let _ = ctx.send(
self.stream_manager,
StreamManagerMsg::Listen {
mode: StreamMode::BlobTransfer,
listener: ctx.self_addr(),
},
);
}
fn handle(&mut self, ctx: &Ctx, msg: StreamNotification) {
match msg {
StreamNotification::StreamOffer {
stream_id,
metadata,
from_node,
..
} => {
// Parse metadata (supports both legacy 32-byte and new versioned format)
let meta = match parse_metadata(&metadata) {
Some(m) => m,
None => {
// Reject malformed offer
let _ = ctx.send(self.stream_manager, StreamManagerMsg::Reject { stream_id });
return;
}
};
let stream_manager = self.stream_manager;
let _ = ctx.send(
self.datastore_node,
DatastoreNodeMsg::HandleStreamOffer {
stream_id,
content_hash: meta.content_hash,
from_node,
stream_manager,
resume_from_chunk: meta.resume_from_chunk.unwrap_or(0),
},
);
}
// Ignore other notifications
_ => {}
}
}
}

View file

@ -0,0 +1,151 @@
//! StreamServer — accepts an incoming stream and serves blob data.
//!
//! Lifecycle:
//! 1. on_start: sends Accept to StreamManager
//! 2. StreamReady: spawns a tokio task for I/O, then stops self
//! 3. tokio task: reads manifest from BlobStore, streams each chunk on-demand
use std::sync::Arc;
use std::time::Duration;
use swactor::actor::{ActorAddress, ActorInterface, Ctx};
use swactor::runtime::Runtime;
use swactor_streams::messages::{StreamManagerMsg, StreamNotification};
use swactor_streams::types::StreamId;
use crate::blob_transfer::{poll_inbox, send_blob, BlobTransferError};
use crate::messages::{BlobStoreMsg, DatastoreResponse};
use crate::types::ContentHash;
const INBOX_TIMEOUT: Duration = Duration::from_secs(10);
pub struct StreamServer {
stream_id: StreamId,
content_hash: ContentHash,
blob_store: ActorAddress,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
skip_chunks: u64,
}
impl StreamServer {
pub fn new(
stream_id: StreamId,
content_hash: ContentHash,
blob_store: ActorAddress,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
skip_chunks: u64,
) -> Self {
Self {
stream_id,
content_hash,
blob_store,
stream_manager,
tokio_handle,
runtime,
skip_chunks,
}
}
}
impl ActorInterface for StreamServer {
type Incoming = StreamNotification;
type Response = ();
fn on_start(&mut self, ctx: &Ctx) {
let _ = ctx.send(
self.stream_manager,
StreamManagerMsg::Accept {
stream_id: self.stream_id,
reply_to: ctx.self_addr(),
},
);
}
fn handle(&mut self, ctx: &Ctx, msg: StreamNotification) {
match msg {
StreamNotification::StreamReady { handle, .. } => {
let stream_handle = match handle.take() {
Some(h) => h,
None => return,
};
let runtime = Arc::clone(&self.runtime);
let blob_store = self.blob_store;
let content_hash = self.content_hash;
let skip_chunks = self.skip_chunks;
self.tokio_handle.spawn(async move {
let (mut send, _recv) = (stream_handle.send, stream_handle.recv);
// Read manifest from BlobStore
let manifest_inbox = match runtime.new_inbox::<DatastoreResponse>() {
Ok(inbox) => inbox,
Err(_) => return,
};
let _ = runtime.send_to(
blob_store,
BlobStoreMsg::ReadManifest {
hash: content_hash,
reply_to: *manifest_inbox.addr(),
},
);
let manifest = match poll_inbox(&manifest_inbox, INBOX_TIMEOUT).await {
Some(DatastoreResponse::ManifestOk { manifest }) => manifest,
other => {
// Manifest not found or timeout — close stream
eprintln!("StreamServer: manifest read failed: {other:?}");
let _ = send.close();
return;
}
};
// Stream each chunk on-demand (one at a time)
let rt = Arc::clone(&runtime);
let bs = blob_store;
let result = send_blob(&mut send, &manifest, |chunk_hash| {
let rt = Arc::clone(&rt);
async move {
let chunk_inbox = rt
.new_inbox::<DatastoreResponse>()
.map_err(|e| {
BlobTransferError::Storage(format!(
"failed to create inbox: {e}"
))
})?;
let _ = rt.send_to(
bs,
BlobStoreMsg::ReadChunk {
hash: chunk_hash,
reply_to: *chunk_inbox.addr(),
},
);
match poll_inbox(&chunk_inbox, INBOX_TIMEOUT).await {
Some(DatastoreResponse::ChunkOk { data, .. }) => Ok(data),
_ => Err(BlobTransferError::Storage(
"chunk not found or timeout".into(),
)),
}
}
}, skip_chunks)
.await;
if let Err(e) = result {
eprintln!("StreamServer: send_blob failed: {e}");
}
});
ctx.stop_self();
}
StreamNotification::StreamFailed { .. } => {
ctx.stop_self();
}
_ => {}
}
}
}

View file

@ -0,0 +1,787 @@
//! BlobTransfer wire protocol — async functions for sending and receiving
//! content-addressed blobs over a QUIC stream (SendHalf / RecvHalf).
//!
//! Wire format:
//! ```text
//! [4B manifest_json_length (u32 BE)]
//! [N bytes manifest JSON]
//! [chunk_0 raw bytes] ← size from manifest.chunks[0].size
//! [chunk_1 raw bytes]
//! ...
//! ```
//!
//! These run inside tokio tasks (NOT actor handlers).
use std::future::Future;
use std::time::Duration;
use serde::{Deserialize, Serialize};
use swactor::runtime::Inbox;
use swactor::actor::Message;
use swactor_streams::handle::{RecvHalf, SendHalf};
use crate::types::{ContentHash, ObjectManifest};
// ─── Metadata encoding ──────────────────────────────────────────────────
/// Version tag for new-format metadata (byte 0).
const METADATA_VERSION_1: u8 = 0x01;
/// Structured metadata sent in `StreamConfig.metadata` for blob transfers.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
pub struct BlobTransferMetadata {
pub content_hash: ContentHash,
#[serde(default)]
pub resume_from_chunk: Option<u64>,
}
/// Encode metadata into the wire format: `[0x01][JSON bytes]`.
pub fn encode_metadata(meta: &BlobTransferMetadata) -> Vec<u8> {
let json = serde_json::to_vec(meta).expect("BlobTransferMetadata serialization cannot fail");
let mut buf = Vec::with_capacity(1 + json.len());
buf.push(METADATA_VERSION_1);
buf.extend_from_slice(&json);
buf
}
/// Parse metadata from either the legacy 32-byte format or the new versioned format.
pub fn parse_metadata(data: &[u8]) -> Option<BlobTransferMetadata> {
if data.len() == 32 {
// Legacy format: raw 32-byte ContentHash
let mut hash_bytes = [0u8; 32];
hash_bytes.copy_from_slice(data);
return Some(BlobTransferMetadata {
content_hash: ContentHash(hash_bytes),
resume_from_chunk: None,
});
}
if data.len() > 1 && data[0] == METADATA_VERSION_1 {
return serde_json::from_slice(&data[1..]).ok();
}
None
}
// ─── Error type ──────────────────────────────────────────────────────────
#[derive(Debug, Clone)]
pub enum BlobTransferError {
/// The stream was closed or disconnected before the transfer completed.
IncompleteTransfer(String),
/// A chunk failed blake3 verification.
ChunkVerificationFailed {
index: usize,
expected: ContentHash,
actual: ContentHash,
},
/// Manifest JSON could not be parsed.
InvalidManifest(String),
/// An error from the underlying storage layer.
Storage(String),
}
impl std::fmt::Display for BlobTransferError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
BlobTransferError::IncompleteTransfer(msg) => {
write!(f, "incomplete transfer: {msg}")
}
BlobTransferError::ChunkVerificationFailed {
index,
expected,
actual,
} => write!(
f,
"chunk {index} verification failed: expected {expected}, got {actual}"
),
BlobTransferError::InvalidManifest(msg) => {
write!(f, "invalid manifest: {msg}")
}
BlobTransferError::Storage(msg) => write!(f, "storage error: {msg}"),
}
}
}
impl std::error::Error for BlobTransferError {}
// ─── ReceivedBlob ────────────────────────────────────────────────────────
/// Result of a successful `recv_blob` call.
#[derive(Debug)]
pub struct ReceivedBlob {
pub manifest: ObjectManifest,
pub chunks: Vec<(ContentHash, Vec<u8>)>,
}
// ─── Core protocol ───────────────────────────────────────────────────────
/// Send a blob over a stream. Chunks are read on-demand via `read_chunk`.
///
/// `read_chunk` is called once per chunk — at most one chunk is in memory
/// at a time on the sender side.
///
/// `skip_chunks` allows resuming a previous transfer: the first `skip_chunks`
/// chunks are not read or written. The manifest preamble is always sent so the
/// receiver can verify integrity.
pub async fn send_blob<F, Fut>(
send: &mut SendHalf,
manifest: &ObjectManifest,
read_chunk: F,
skip_chunks: u64,
) -> Result<(), BlobTransferError>
where
F: Fn(ContentHash) -> Fut,
Fut: Future<Output = Result<Vec<u8>, BlobTransferError>>,
{
// Serialize manifest
let manifest_json = serde_json::to_vec(manifest)
.map_err(|e| BlobTransferError::InvalidManifest(e.to_string()))?;
// Write manifest preamble: [4B length BE] [manifest JSON]
let len_bytes = (manifest_json.len() as u32).to_be_bytes();
write_all(send, &len_bytes).await?;
write_all(send, &manifest_json).await?;
// Write chunks, skipping already-transferred ones
for (i, chunk_ref) in manifest.chunks.iter().enumerate() {
if (i as u64) < skip_chunks {
continue;
}
let data = read_chunk(chunk_ref.hash).await?;
write_all(send, &data).await?;
}
// Flush and close
send.flush()
.map_err(|e| BlobTransferError::IncompleteTransfer(e.to_string()))?;
send.close()
.map_err(|e| BlobTransferError::IncompleteTransfer(e.to_string()))?;
Ok(())
}
/// Receive a blob from a stream. Reads manifest, then reads and verifies
/// each chunk via blake3.
///
/// `skip_chunks` allows resuming: the sender skipped the first `skip_chunks`
/// chunks, so the receiver only reads chunks from `skip_chunks` onward.
/// The manifest preamble is always read.
pub async fn recv_blob(
recv: &mut RecvHalf,
skip_chunks: u64,
) -> Result<ReceivedBlob, BlobTransferError> {
// Read manifest preamble
let mut len_buf = [0u8; 4];
read_exact(recv, &mut len_buf).await?;
let manifest_len = u32::from_be_bytes(len_buf) as usize;
// Read manifest JSON
let mut manifest_buf = vec![0u8; manifest_len];
read_exact(recv, &mut manifest_buf).await?;
let manifest: ObjectManifest = serde_json::from_slice(&manifest_buf)
.map_err(|e| BlobTransferError::InvalidManifest(e.to_string()))?;
// Read and verify each chunk (only those the sender actually sent)
let total = manifest.chunks.len();
let start = (skip_chunks as usize).min(total);
let mut chunks = Vec::with_capacity(total - start);
for (i, chunk_ref) in manifest.chunks.iter().enumerate().skip(start) {
let mut chunk_data = vec![0u8; chunk_ref.size as usize];
read_exact(recv, &mut chunk_data).await?;
// Verify blake3
let actual_hash = ContentHash::of(&chunk_data);
if actual_hash != chunk_ref.hash {
return Err(BlobTransferError::ChunkVerificationFailed {
index: i,
expected: chunk_ref.hash,
actual: actual_hash,
});
}
chunks.push((chunk_ref.hash, chunk_data));
}
Ok(ReceivedBlob { manifest, chunks })
}
// ─── Helpers ─────────────────────────────────────────────────────────────
/// Write all bytes to a SendHalf, yielding when backpressured.
async fn write_all(send: &mut SendHalf, data: &[u8]) -> Result<(), BlobTransferError> {
let mut offset = 0;
while offset < data.len() {
match send.try_write(&data[offset..]) {
Ok(0) => {
// Backpressure — yield and retry
tokio::task::yield_now().await;
}
Ok(n) => {
offset += n;
}
Err(e) => {
return Err(BlobTransferError::IncompleteTransfer(e.to_string()));
}
}
}
Ok(())
}
/// Read exactly `buf.len()` bytes from a RecvHalf, yielding when no data.
async fn read_exact(recv: &mut RecvHalf, buf: &mut [u8]) -> Result<(), BlobTransferError> {
let mut offset = 0;
while offset < buf.len() {
match recv.try_read(&mut buf[offset..]) {
Ok(0) => {
// No data available — yield and retry
tokio::task::yield_now().await;
}
Ok(n) => {
offset += n;
}
Err(e) => {
return Err(BlobTransferError::IncompleteTransfer(e.to_string()));
}
}
}
Ok(())
}
/// Async version of `bridge.rs:poll_response` — yields instead of thread::sleep.
pub async fn poll_inbox<M: Message>(inbox: &Inbox<M>, timeout: Duration) -> Option<M> {
let start = tokio::time::Instant::now();
loop {
if let Some(msg) = inbox.try_recv() {
return Some(msg);
}
if start.elapsed() > timeout {
return None;
}
tokio::task::yield_now().await;
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::chunking::chunk_blob;
use swactor_streams::handle::create_stream_handle;
use swactor_streams::types::{StreamConfig, StreamId};
/// Helper: create a pair of (SendHalf, RecvHalf) connected via tokio tasks
/// that relay data through a DuplexStream.
fn create_test_pair() -> (SendHalf, RecvHalf) {
let stream_id = StreamId::new_random();
let config = StreamConfig {
stripe_count: 1,
frame_size: 256 * 1024,
metadata: Vec::new(),
};
let (handle_a, endpoints_a) = create_stream_handle(stream_id, &config, 128, 64);
let (handle_b, endpoints_b) = create_stream_handle(stream_id, &config, 128, 64);
// Wire a's send → b's recv via a DuplexStream
let (client, server) = tokio::io::duplex(1024 * 1024);
let (client_read, client_write) = tokio::io::split(client);
let (server_read, server_write) = tokio::io::split(server);
// a's send data-plane task: read from cmd_rx, write to client_write
spawn_send_task(endpoints_a.send_cmd_rx, endpoints_a.send_evt_tx, endpoints_a.pool.clone(), client_write);
// b's recv data-plane task: read from server_read, push to evt_tx
spawn_recv_task(server_read, endpoints_b.recv_evt_tx, endpoints_b.pool.clone());
// b's send data-plane task: for the other direction (not used in basic tests)
spawn_send_task(endpoints_b.send_cmd_rx, endpoints_b.send_evt_tx, endpoints_b.pool.clone(), server_write);
// a's recv data-plane task
spawn_recv_task(client_read, endpoints_a.recv_evt_tx, endpoints_a.pool.clone());
// Return a's send half and b's recv half for unidirectional testing
(handle_a.send, handle_b.recv)
}
fn spawn_send_task(
mut cmd_rx: tokio::sync::mpsc::Receiver<swactor_streams::channel::SendCommand>,
evt_tx: tokio::sync::mpsc::Sender<swactor_streams::channel::SendEvent>,
pool: swactor_streams::BufferPool,
mut writer: tokio::io::WriteHalf<tokio::io::DuplexStream>,
) {
use tokio::io::AsyncWriteExt;
tokio::spawn(async move {
while let Some(cmd) = cmd_rx.recv().await {
match cmd {
swactor_streams::channel::SendCommand::Data(buf) => {
let data = buf.written();
// Write length-prefixed frame
let len = (data.len() as u32).to_be_bytes();
if writer.write_all(&len).await.is_err() {
pool.checkin(buf);
let _ = evt_tx.send(swactor_streams::channel::SendEvent::Error(
swactor_streams::StreamError::Disconnected,
)).await;
return;
}
if writer.write_all(data).await.is_err() {
pool.checkin(buf);
let _ = evt_tx.send(swactor_streams::channel::SendEvent::Error(
swactor_streams::StreamError::Disconnected,
)).await;
return;
}
pool.checkin(buf);
}
swactor_streams::channel::SendCommand::Flush => {
let _ = writer.flush().await;
}
swactor_streams::channel::SendCommand::Close => {
let _ = writer.shutdown().await;
break;
}
}
}
});
}
fn spawn_recv_task(
mut reader: tokio::io::ReadHalf<tokio::io::DuplexStream>,
evt_tx: tokio::sync::mpsc::Sender<swactor_streams::channel::RecvEvent>,
pool: swactor_streams::BufferPool,
) {
use tokio::io::AsyncReadExt;
tokio::spawn(async move {
loop {
// Read length-prefixed frame
let mut len_buf = [0u8; 4];
match reader.read_exact(&mut len_buf).await {
Ok(_) => {}
Err(_) => {
let _ = evt_tx.send(swactor_streams::channel::RecvEvent::Closed).await;
return;
}
}
let len = u32::from_be_bytes(len_buf) as usize;
let mut data = vec![0u8; len];
match reader.read_exact(&mut data).await {
Ok(_) => {}
Err(_) => {
let _ = evt_tx.send(swactor_streams::channel::RecvEvent::Closed).await;
return;
}
}
// Write data into FrameBufs and send
let mut offset = 0;
while offset < data.len() {
let mut buf = match pool.checkout() {
Some(b) => b,
None => {
let _ = evt_tx.send(swactor_streams::channel::RecvEvent::Error(
swactor_streams::StreamError::BufferExhausted,
)).await;
return;
}
};
let written = buf.write(&data[offset..]);
offset += written;
let _ = evt_tx.send(swactor_streams::channel::RecvEvent::Data(buf)).await;
}
}
});
}
#[tokio::test]
async fn small_blob_round_trips() {
let data = b"hello, world!";
let (_, manifest, chunks) = chunk_blob(data, 1024);
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, 0)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
assert_eq!(received.manifest, manifest);
assert_eq!(received.chunks.len(), chunks.len());
for (i, (hash, data)) in received.chunks.iter().enumerate() {
assert_eq!(*hash, chunks[i].0);
assert_eq!(*data, chunks[i].1);
}
}
#[tokio::test]
async fn multi_chunk_round_trips() {
// 4MB with 256KB chunks = 16 chunks
let data: Vec<u8> = (0..4 * 1024 * 1024).map(|i| (i % 251) as u8).collect();
let (_, manifest, chunks) = chunk_blob(&data, 256 * 1024);
assert_eq!(chunks.len(), 16);
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, 0)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
assert_eq!(received.manifest, manifest);
assert_eq!(received.chunks.len(), 16);
// Verify all chunk data matches
for (i, (hash, cdata)) in received.chunks.iter().enumerate() {
assert_eq!(*hash, chunks[i].0);
assert_eq!(cdata.len(), chunks[i].1.len());
}
}
#[tokio::test]
async fn corrupted_chunk_detected() {
let data = b"integrity test data here";
let (_, manifest, mut chunks) = chunk_blob(data, 1024);
// Flip a byte in the chunk data
chunks[0].1[0] ^= 0xFF;
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
// Note: we send the corrupted data (hash won't match)
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, 0)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await
});
send_task.await.unwrap();
let result = recv_task.await.unwrap();
match result {
Err(BlobTransferError::ChunkVerificationFailed { index: 0, .. }) => {}
other => panic!("expected ChunkVerificationFailed, got: {other:?}"),
}
}
#[tokio::test]
async fn truncated_stream_detected() {
// Create a blob with multiple chunks
let data: Vec<u8> = vec![42u8; 4096];
let (_, manifest, chunks) = chunk_blob(&data, 1024);
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
// Only send the manifest + first chunk, then close
let send_task = tokio::spawn(async move {
let manifest_json = serde_json::to_vec(&manifest_clone).unwrap();
let len_bytes = (manifest_json.len() as u32).to_be_bytes();
write_all(&mut send, &len_bytes).await.unwrap();
write_all(&mut send, &manifest_json).await.unwrap();
// Write first chunk
write_all(&mut send, &chunks[0].1).await.unwrap();
// Close without writing remaining chunks
send.flush().unwrap();
send.close().unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await
});
send_task.await.unwrap();
let result = recv_task.await.unwrap();
match result {
Err(BlobTransferError::IncompleteTransfer(_)) => {}
other => panic!("expected IncompleteTransfer, got: {other:?}"),
}
}
// ── Resume token tests ──────────────────────────────────────────────
#[tokio::test]
async fn resume_skips_first_n_chunks() {
// 16 chunks, resume from chunk 8 → only chunks 8-15 transferred
let data: Vec<u8> = (0..4 * 1024 * 1024).map(|i| (i % 251) as u8).collect();
let (_, manifest, chunks) = chunk_blob(&data, 256 * 1024);
assert_eq!(chunks.len(), 16);
let skip = 8u64;
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, skip)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, skip).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
assert_eq!(received.manifest, manifest);
assert_eq!(received.chunks.len(), 8); // only chunks 8-15
for (j, (hash, cdata)) in received.chunks.iter().enumerate() {
let orig_idx = skip as usize + j;
assert_eq!(*hash, chunks[orig_idx].0);
assert_eq!(*cdata, chunks[orig_idx].1);
}
}
#[tokio::test]
async fn resume_from_zero_is_full_transfer() {
let data: Vec<u8> = (0..4 * 1024 * 1024).map(|i| (i % 251) as u8).collect();
let (_, manifest, chunks) = chunk_blob(&data, 256 * 1024);
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, 0)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
assert_eq!(received.chunks.len(), chunks.len());
}
#[tokio::test]
async fn resume_from_last_chunk() {
// 16 chunks, skip=15 → only the final chunk transfers
let data: Vec<u8> = (0..4 * 1024 * 1024).map(|i| (i % 251) as u8).collect();
let (_, manifest, chunks) = chunk_blob(&data, 256 * 1024);
assert_eq!(chunks.len(), 16);
let skip = 15u64;
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("chunk not found".into()))
}
}, skip)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, skip).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
assert_eq!(received.chunks.len(), 1);
assert_eq!(received.chunks[0].0, chunks[15].0);
assert_eq!(received.chunks[0].1, chunks[15].1);
}
#[test]
fn metadata_encoding_round_trip() {
let meta = BlobTransferMetadata {
content_hash: ContentHash([42u8; 32]),
resume_from_chunk: Some(50),
};
let encoded = encode_metadata(&meta);
assert_eq!(encoded[0], 0x01);
let decoded = parse_metadata(&encoded).unwrap();
assert_eq!(decoded, meta);
}
#[test]
fn metadata_backward_compat() {
// Legacy 32-byte format → parsed as no resume offset
let raw_hash = [0xABu8; 32];
let decoded = parse_metadata(&raw_hash).unwrap();
assert_eq!(decoded.content_hash, ContentHash(raw_hash));
assert_eq!(decoded.resume_from_chunk, None);
}
mod proptests {
use super::*;
use proptest::prelude::*;
proptest! {
#[test]
fn arbitrary_blob_round_trips(
data in proptest::collection::vec(any::<u8>(), 1..=128 * 1024),
chunk_size in 256u32..=64 * 1024,
) {
let rt = tokio::runtime::Runtime::new().unwrap();
rt.block_on(async {
let (_, manifest, chunks) = chunk_blob(&data, chunk_size);
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("not found".into()))
}
}, 0)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, 0).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
prop_assert_eq!(received.manifest, manifest);
prop_assert_eq!(received.chunks.len(), chunks.len());
for (i, (hash, cdata)) in received.chunks.iter().enumerate() {
prop_assert_eq!(*hash, chunks[i].0);
prop_assert_eq!(cdata, &chunks[i].1);
}
Ok(())
})?;
}
#[test]
fn arbitrary_resume_offset_round_trips(
data in proptest::collection::vec(any::<u8>(), 1..=128 * 1024),
chunk_size in 256u32..=64 * 1024,
skip_frac in 0.0f64..1.0,
) {
let rt = tokio::runtime::Runtime::new().unwrap();
rt.block_on(async {
let (_, manifest, chunks) = chunk_blob(&data, chunk_size);
let total = chunks.len() as u64;
let skip = (skip_frac * total as f64).floor() as u64;
let (mut send, mut recv) = create_test_pair();
let manifest_clone = manifest.clone();
let chunks_clone = chunks.clone();
let send_task = tokio::spawn(async move {
send_blob(&mut send, &manifest_clone, |hash| {
let chunks = chunks_clone.clone();
async move {
chunks
.iter()
.find(|(h, _)| *h == hash)
.map(|(_, d)| d.clone())
.ok_or_else(|| BlobTransferError::Storage("not found".into()))
}
}, skip)
.await
.unwrap();
});
let recv_task = tokio::spawn(async move {
recv_blob(&mut recv, skip).await.unwrap()
});
send_task.await.unwrap();
let received = recv_task.await.unwrap();
let expected_count = total - skip;
prop_assert_eq!(received.chunks.len() as u64, expected_count);
for (j, (hash, cdata)) in received.chunks.iter().enumerate() {
let orig_idx = skip as usize + j;
prop_assert_eq!(*hash, chunks[orig_idx].0);
prop_assert_eq!(cdata, &chunks[orig_idx].1);
}
Ok(())
})?;
}
}
}
}

View file

@ -381,6 +381,7 @@ pub struct DatastoreAuthConfig {
/// Owns the full lifecycle of a datastore actor group: BlobStore, Metadata,
/// DatastoreNode, and optional GatewayActor.
pub struct DatastoreGroup {
datastore_addr: ActorAddress,
metadata_addr: ActorAddress,
gateway_addr: Option<ActorAddress>,
bridge: Arc<dyn DatastoreStatsProvider>,
@ -472,6 +473,7 @@ impl DatastoreGroup {
}
Ok(Self {
datastore_addr,
metadata_addr,
gateway_addr,
bridge,
@ -500,6 +502,30 @@ impl DatastoreGroup {
pub fn bridge(&self) -> &Arc<dyn DatastoreStatsProvider> {
&self.bridge
}
/// Configure stream support: sends ConfigureStreams to DatastoreNode and
/// spawns a StreamListener actor.
pub fn configure_streams(
&self,
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
) {
let _ = self.runtime.send_to(
self.datastore_addr,
DatastoreNodeMsg::ConfigureStreams {
stream_manager,
tokio_handle,
runtime: Arc::clone(&self.runtime),
},
);
// Spawn StreamListener
use crate::actors::stream_listener::StreamListener;
use swactor_std::RuntimeNaming;
if let Ok(addr) = self.runtime.spawn(StreamListener::new(self.datastore_addr, stream_manager)) {
let _ = self.runtime.register_name("StreamListener", addr);
}
}
}
fn generate_node_id() -> NodeId {

View file

@ -8,6 +8,7 @@ pub mod cli;
pub mod metrics;
pub mod api;
pub mod ui_html;
pub mod blob_transfer;
pub mod bridge;
pub use types::{ChunkRef, ContentHash, DatastoreConfig, ObjectEntry, ObjectManifest};

View file

@ -9,12 +9,15 @@
use std::collections::BTreeMap;
use std::collections::HashSet;
use std::net::SocketAddr;
use std::sync::Arc;
use serde::{Deserialize, Serialize};
use swactor::actor::ActorAddress;
use swactor::runtime::Runtime;
use swactor::transport::NetworkMessage;
use distribution::types::NodeId;
use swactor_streams::types::StreamId;
use crate::auth::{AccessRequestInfo, AuthorizedKeyInfo, DeniedReason, SignedRequest};
use crate::types::{ContentHash, ObjectEntry, ObjectManifest};
@ -276,7 +279,7 @@ pub enum TransferMsg {
// ─── DatastoreNodeMsg ───────────────────────────────────────────────────────
/// Messages handled by the `DatastoreNode` coordinator actor.
#[derive(Debug, Clone)]
#[derive(Clone)]
pub enum DatastoreNodeMsg {
// ── User-facing commands ────────────────────────────────────────────
/// Store a blob with optional name and tags.
@ -333,6 +336,64 @@ pub enum DatastoreNodeMsg {
request: ListObjectsRequest,
reply_to: ActorAddress,
},
// ── Stream-based blob transfer ─────────────────────────────────────
/// Download a blob via QUIC stream from a remote node.
DownloadViaStream {
content_hash: ContentHash,
source_node: [u8; 32],
reply_to: ActorAddress,
},
/// Handle an incoming stream offer (from StreamListener).
HandleStreamOffer {
stream_id: StreamId,
content_hash: ContentHash,
from_node: [u8; 32],
stream_manager: ActorAddress,
resume_from_chunk: u64,
},
/// A stream download completed successfully.
StreamDownloadComplete {
content_hash: ContentHash,
manifest: ObjectManifest,
reply_to: ActorAddress,
},
/// A stream download failed.
StreamDownloadFailed {
content_hash: ContentHash,
reason: String,
chunks_completed: u64,
reply_to: ActorAddress,
},
/// Configure stream support (StreamManager address + tokio handle).
ConfigureStreams {
stream_manager: ActorAddress,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
},
}
impl std::fmt::Debug for DatastoreNodeMsg {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::Put { name, .. } => f.debug_struct("Put").field("name", name).finish_non_exhaustive(),
Self::Get { content_hash, .. } => f.debug_struct("Get").field("content_hash", content_hash).finish_non_exhaustive(),
Self::Delete { content_hash, .. } => f.debug_struct("Delete").field("content_hash", content_hash).finish_non_exhaustive(),
Self::List { name_filter, all, .. } => f.debug_struct("List").field("name_filter", name_filter).field("all", all).finish_non_exhaustive(),
Self::Status { .. } => write!(f, "Status"),
Self::ReadChunk { hash, .. } => f.debug_struct("ReadChunk").field("hash", hash).finish_non_exhaustive(),
Self::IncomingGetChunk { .. } => write!(f, "IncomingGetChunk"),
Self::IncomingGetManifest { .. } => write!(f, "IncomingGetManifest"),
Self::IncomingStoreObject { .. } => write!(f, "IncomingStoreObject"),
Self::IncomingFindObject { .. } => write!(f, "IncomingFindObject"),
Self::IncomingListObjects { .. } => write!(f, "IncomingListObjects"),
Self::DownloadViaStream { content_hash, .. } => f.debug_struct("DownloadViaStream").field("content_hash", content_hash).finish_non_exhaustive(),
Self::HandleStreamOffer { stream_id, content_hash, resume_from_chunk, .. } => f.debug_struct("HandleStreamOffer").field("stream_id", stream_id).field("content_hash", content_hash).field("resume_from_chunk", resume_from_chunk).finish_non_exhaustive(),
Self::StreamDownloadComplete { content_hash, .. } => f.debug_struct("StreamDownloadComplete").field("content_hash", content_hash).finish_non_exhaustive(),
Self::StreamDownloadFailed { content_hash, reason, chunks_completed, .. } => f.debug_struct("StreamDownloadFailed").field("content_hash", content_hash).field("reason", reason).field("chunks_completed", chunks_completed).finish_non_exhaustive(),
Self::ConfigureStreams { stream_manager, .. } => f.debug_struct("ConfigureStreams").field("stream_manager", stream_manager).finish_non_exhaustive(),
}
}
}
// ─── DatastoreResponse ──────────────────────────────────────────────────────

View file

@ -0,0 +1,415 @@
//! End-to-end test: store a blob on node A, download via QUIC stream on node B.
use std::sync::Arc;
use std::time::{Duration, Instant};
use distribution::iroh_driver::{IrohDriver, IrohDriverConfig};
use distribution::node::DistributedNodeConfig;
use iroh::RelayMode;
use swactor::config::RuntimeConfig;
use swactor::runtime::Runtime;
use swactor_datastore::bridge::{DatastoreGroup, DatastoreGroupConfig};
use swactor_datastore::messages::{DatastoreNodeMsg, DatastoreResponse};
use swactor_std::RuntimeNaming;
fn make_driver_with_streams() -> IrohDriver {
IrohDriver::new(IrohDriverConfig {
secret_key: None,
relay_mode: RelayMode::Disabled,
node: DistributedNodeConfig::default(),
peer_auth: None,
additional_alpns: vec![swactor_streams::ALPN.to_vec()],
})
.expect("create iroh driver")
}
fn make_runtime() -> Arc<Runtime> {
let rt = Runtime::new(RuntimeConfig {
num_threads: 2,
max_actors: 256,
channel_buffer_size: 2000,
..Default::default()
})
.with_extension(Arc::new(swactor_std::StdExtension::new()));
let handle = rt.run().expect("start runtime");
handle.runtime
}
fn spawn_stream_manager(
runtime: &Arc<Runtime>,
driver: &IrohDriver,
) -> swactor::actor::ActorAddress {
let mgr = swactor_streams::StreamManager::new(
driver.endpoint().clone(),
driver.tokio_handle(),
Arc::clone(runtime),
);
let addr = runtime.spawn(mgr).expect("spawn StreamManager");
runtime
.register_name(swactor_streams::STREAM_MANAGER_NAME, addr)
.expect("register StreamManager");
addr
}
fn spawn_datastore(
runtime: &Arc<Runtime>,
driver: &IrohDriver,
mgr_addr: swactor::actor::ActorAddress,
) -> DatastoreGroup {
let node_id = driver.node_id();
let group = DatastoreGroup::spawn(
Arc::clone(runtime),
DatastoreGroupConfig {
node_id,
node_id_hex: format!("{:?}", node_id),
chunk_size: 256,
storage_path: None, // in-memory
auth: None,
gc_interval: u64::MAX,
disseminate_interval: u64::MAX,
},
)
.expect("spawn datastore");
group.configure_streams(mgr_addr, driver.tokio_handle());
group
}
/// Poll for a response from the inbox, routing stream connections between
/// the two nodes. Actor message processing is handled by worker threads.
fn pump_until_response(
rt_a: &Arc<Runtime>,
rt_b: &Arc<Runtime>,
driver_a: &mut IrohDriver,
driver_b: &mut IrohDriver,
mgr_a: swactor::actor::ActorAddress,
mgr_b: swactor::actor::ActorAddress,
inbox: &swactor::runtime::Inbox<DatastoreResponse>,
timeout: Duration,
) -> Option<DatastoreResponse> {
let deadline = Instant::now() + timeout;
let tokio_handle = driver_a.tokio_handle();
loop {
// SWIM protocol ticks
driver_a.recv();
driver_a.tick();
driver_b.recv();
driver_b.tick();
// Route incoming stream connections on node A
for (node_id, conn) in driver_a.drain_other_connections() {
let rt = Arc::clone(rt_a);
let mgr = mgr_a;
let node_bytes = node_id.0;
tokio_handle.spawn(async move {
let _ = swactor_streams::accept::handle_incoming(node_bytes, conn, &rt, mgr).await;
});
}
// Route incoming stream connections on node B
for (node_id, conn) in driver_b.drain_other_connections() {
let rt = Arc::clone(rt_b);
let mgr = mgr_b;
let node_bytes = node_id.0;
tokio_handle.spawn(async move {
let _ = swactor_streams::accept::handle_incoming(node_bytes, conn, &rt, mgr).await;
});
}
// Check for completion
if let Some(resp) = inbox.try_recv() {
return Some(resp);
}
if Instant::now() >= deadline {
return None;
}
std::thread::sleep(Duration::from_millis(10));
}
}
#[test]
fn small_blob_transfers_between_two_nodes_via_stream() {
let mut driver_a = make_driver_with_streams();
let mut driver_b = make_driver_with_streams();
let rt_a = make_runtime();
let rt_b = make_runtime();
let mgr_a = spawn_stream_manager(&rt_a, &driver_a);
let mgr_b = spawn_stream_manager(&rt_b, &driver_b);
let _ds_a = spawn_datastore(&rt_a, &driver_a, mgr_a);
let ds_b = spawn_datastore(&rt_b, &driver_b, mgr_b);
let _ = &ds_b; // keep alive
// Have both nodes discover each other via SWIM
let addr_a = driver_a.endpoint_addr();
let addr_b = driver_b.endpoint_addr();
driver_a.join(&[addr_b]);
driver_b.join(&[addr_a]);
// Pump until SWIM membership converges
let swim_deadline = Instant::now() + Duration::from_secs(10);
loop {
driver_a.recv();
driver_a.tick();
driver_b.recv();
driver_b.tick();
let snap_a = driver_a.snapshot();
let snap_b = driver_b.snapshot();
if snap_a.alive_count >= 1 && snap_b.alive_count >= 1 {
break;
}
if Instant::now() >= swim_deadline {
panic!("SWIM convergence timed out");
}
std::thread::sleep(Duration::from_millis(50));
}
// Give worker threads time to process spawned actors
std::thread::sleep(Duration::from_millis(100));
// Store a small blob on Node A
let test_data = b"Hello from node A! This is a stream integration test.";
let put_inbox = rt_a
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let ds_a_addr = rt_a.where_is("Datastore").expect("Datastore registered on A");
let _ = rt_a.send_to(
ds_a_addr,
DatastoreNodeMsg::Put {
data: test_data.to_vec(),
name: Some("test-blob".into()),
tags: Default::default(),
reply_to: *put_inbox.addr(),
},
);
// Wait for PutOk (worker threads process messages)
let content_hash = loop {
if let Some(resp) = put_inbox.try_recv() {
match resp {
DatastoreResponse::PutOk { content_hash } => break content_hash,
other => panic!("expected PutOk, got: {other:?}"),
}
}
std::thread::sleep(Duration::from_millis(10));
};
// PutOk comes from MetadataActor; BlobStore writes are fire-and-forget.
// Give BlobStore time to finish writing chunks + manifest.
std::thread::sleep(Duration::from_millis(200));
// Download the blob on Node B via stream
let download_inbox = rt_b
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let ds_b_addr = rt_b.where_is("Datastore").expect("Datastore registered on B");
let _ = rt_b.send_to(
ds_b_addr,
DatastoreNodeMsg::DownloadViaStream {
content_hash,
source_node: driver_a.node_id().0,
reply_to: *download_inbox.addr(),
},
);
// Pump loop until we get a response
let resp = pump_until_response(
&rt_a,
&rt_b,
&mut driver_a,
&mut driver_b,
mgr_a,
mgr_b,
&download_inbox,
Duration::from_secs(15),
);
match resp {
Some(DatastoreResponse::PutOk { content_hash: h }) => {
assert_eq!(h, content_hash, "downloaded blob hash should match");
}
other => panic!("expected PutOk from download, got: {other:?}"),
}
// Verify: read the blob back from Node B's datastore
let verify_inbox = rt_b
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let _ = rt_b.send_to(
ds_b_addr,
DatastoreNodeMsg::Get {
content_hash,
reply_to: *verify_inbox.addr(),
},
);
let verify_deadline = Instant::now() + Duration::from_secs(5);
loop {
if let Some(resp) = verify_inbox.try_recv() {
match resp {
DatastoreResponse::GetOk { entry, manifest } => {
assert_eq!(entry.content_hash, content_hash);
assert_eq!(manifest.total_size, test_data.len() as u64);
break;
}
other => panic!("expected GetOk, got: {other:?}"),
}
}
if Instant::now() >= verify_deadline {
panic!("verify timed out — blob not found on Node B");
}
std::thread::sleep(Duration::from_millis(10));
}
driver_a.shutdown();
driver_b.shutdown();
rt_a.shutdown();
rt_b.shutdown();
}
#[test]
fn multi_chunk_blob_transfers_between_two_nodes_via_stream() {
let mut driver_a = make_driver_with_streams();
let mut driver_b = make_driver_with_streams();
let rt_a = make_runtime();
let rt_b = make_runtime();
let mgr_a = spawn_stream_manager(&rt_a, &driver_a);
let mgr_b = spawn_stream_manager(&rt_b, &driver_b);
let _ds_a = spawn_datastore(&rt_a, &driver_a, mgr_a);
let ds_b = spawn_datastore(&rt_b, &driver_b, mgr_b);
let _ = &ds_b;
// Discover each other via SWIM
let addr_a = driver_a.endpoint_addr();
let addr_b = driver_b.endpoint_addr();
driver_a.join(&[addr_b]);
driver_b.join(&[addr_a]);
let swim_deadline = Instant::now() + Duration::from_secs(10);
loop {
driver_a.recv();
driver_a.tick();
driver_b.recv();
driver_b.tick();
let snap_a = driver_a.snapshot();
let snap_b = driver_b.snapshot();
if snap_a.alive_count >= 1 && snap_b.alive_count >= 1 {
break;
}
if Instant::now() >= swim_deadline {
panic!("SWIM convergence timed out");
}
std::thread::sleep(Duration::from_millis(50));
}
// Give worker threads time to process spawned actors
std::thread::sleep(Duration::from_millis(100));
// 4 chunks at 256 bytes each = 1024 bytes
let test_data: Vec<u8> = (0..1024).map(|i| (i % 251) as u8).collect();
let put_inbox = rt_a
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let ds_a_addr = rt_a.where_is("Datastore").expect("Datastore registered on A");
let _ = rt_a.send_to(
ds_a_addr,
DatastoreNodeMsg::Put {
data: test_data.clone(),
name: Some("multi-chunk".into()),
tags: Default::default(),
reply_to: *put_inbox.addr(),
},
);
let content_hash = loop {
if let Some(resp) = put_inbox.try_recv() {
match resp {
DatastoreResponse::PutOk { content_hash } => break content_hash,
other => panic!("expected PutOk, got: {other:?}"),
}
}
std::thread::sleep(Duration::from_millis(10));
};
// PutOk comes from MetadataActor; BlobStore writes are fire-and-forget.
std::thread::sleep(Duration::from_millis(200));
// Download on Node B
let download_inbox = rt_b
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let ds_b_addr = rt_b.where_is("Datastore").expect("Datastore registered on B");
let _ = rt_b.send_to(
ds_b_addr,
DatastoreNodeMsg::DownloadViaStream {
content_hash,
source_node: driver_a.node_id().0,
reply_to: *download_inbox.addr(),
},
);
let resp = pump_until_response(
&rt_a,
&rt_b,
&mut driver_a,
&mut driver_b,
mgr_a,
mgr_b,
&download_inbox,
Duration::from_secs(15),
);
match resp {
Some(DatastoreResponse::PutOk { content_hash: h }) => {
assert_eq!(h, content_hash);
}
other => panic!("expected PutOk from download, got: {other:?}"),
}
// Verify the data on Node B by reading each chunk
let verify_inbox = rt_b
.new_inbox::<DatastoreResponse>()
.expect("create inbox");
let _ = rt_b.send_to(
ds_b_addr,
DatastoreNodeMsg::Get {
content_hash,
reply_to: *verify_inbox.addr(),
},
);
let verify_deadline = Instant::now() + Duration::from_secs(5);
loop {
if let Some(resp) = verify_inbox.try_recv() {
match resp {
DatastoreResponse::GetOk { entry, manifest } => {
assert_eq!(entry.content_hash, content_hash);
assert_eq!(manifest.total_size, test_data.len() as u64);
assert!(manifest.chunks.len() > 1, "should be multi-chunk");
break;
}
other => panic!("expected GetOk, got: {other:?}"),
}
}
if Instant::now() >= verify_deadline {
panic!("verify timed out — blob not found on Node B");
}
std::thread::sleep(Duration::from_millis(10));
}
driver_a.shutdown();
driver_b.shutdown();
rt_a.shutdown();
rt_b.shutdown();
}

View file

@ -40,6 +40,8 @@ pub struct IrohDriverConfig {
pub node: DistributedNodeConfig,
/// Optional peer allow-list. If provided, only allowed peers can connect.
pub peer_auth: Option<Arc<Mutex<PeerAllowList>>>,
/// Additional ALPNs to register beyond SWIM. Opaque to the driver.
pub additional_alpns: Vec<Vec<u8>>,
/// If set, start an embedded relay server on this address.
/// Requires the `relay` feature. On success, the driver uses the embedded
/// relay for `RelayMode::Custom`; on failure, falls back to `relay_mode`.
@ -74,8 +76,10 @@ pub struct IrohDriver {
peer_auth: Option<Arc<Mutex<PeerAllowList>>>,
/// Collects connections from background join tasks.
pending_joins: Arc<Mutex<Vec<JoinResult>>>,
/// Connections accepted by the background accept loop.
/// Connections accepted by the background accept loop (SWIM ALPN).
accepted_conns: Arc<Mutex<Vec<(NodeId, Connection)>>>,
/// Connections accepted on non-SWIM ALPNs (streams, etc.).
other_accepted_conns: Arc<Mutex<Vec<(NodeId, Connection)>>>,
/// Relay URLs learned from join seeds, used for reconnection.
peer_relay_urls: HashMap<NodeId, iroh::RelayUrl>,
/// Embedded relay server (if started).
@ -122,8 +126,10 @@ impl IrohDriver {
let (relay_url, effective_relay_mode) = (None::<String>, config.relay_mode);
let endpoint = rt.block_on(async {
let mut alpns = vec![ALPN.to_vec()];
alpns.extend(config.additional_alpns.iter().cloned());
let mut builder = Endpoint::empty_builder(effective_relay_mode)
.alpns(vec![ALPN.to_vec()]);
.alpns(alpns);
if let Some(key) = config.secret_key {
builder = builder.secret_key(key);
@ -141,10 +147,13 @@ impl IrohDriver {
// Spawn background accept loop so incoming connections are never missed
let accepted_conns: Arc<Mutex<Vec<(NodeId, Connection)>>> =
Arc::new(Mutex::new(Vec::new()));
let other_accepted_conns: Arc<Mutex<Vec<(NodeId, Connection)>>> =
Arc::new(Mutex::new(Vec::new()));
{
let ep = endpoint.clone();
let peer_auth = config.peer_auth.clone();
let buf = Arc::clone(&accepted_conns);
let swim_buf = Arc::clone(&accepted_conns);
let other_buf = Arc::clone(&other_accepted_conns);
rt.spawn(async move {
loop {
match ep.accept().await {
@ -165,11 +174,22 @@ impl IrohDriver {
conn.close(0u32.into(), b"unauthorized");
continue;
}
eprintln!(
"iroh driver: accepted connection from {}",
crate::identity::hex_encode(&node_id.0[..4])
);
buf.lock().unwrap().push((node_id, conn));
// Route by negotiated ALPN
let negotiated_alpn = conn.alpn();
if negotiated_alpn == ALPN {
eprintln!(
"iroh driver: accepted SWIM connection from {}",
crate::identity::hex_encode(&node_id.0[..4])
);
swim_buf.lock().unwrap().push((node_id, conn));
} else {
eprintln!(
"iroh driver: accepted non-SWIM connection from {} (ALPN: {})",
crate::identity::hex_encode(&node_id.0[..4]),
String::from_utf8_lossy(&negotiated_alpn),
);
other_buf.lock().unwrap().push((node_id, conn));
}
}
Err(e) => {
eprintln!("iroh driver: incoming connection error: {e}");
@ -189,6 +209,7 @@ impl IrohDriver {
peer_auth: config.peer_auth,
pending_joins: Arc::new(Mutex::new(Vec::new())),
accepted_conns,
other_accepted_conns,
peer_relay_urls: HashMap::new(),
#[cfg(feature = "relay")]
relay_server,
@ -201,6 +222,16 @@ impl IrohDriver {
self.rt.handle().clone()
}
/// Get a reference to the iroh endpoint (for creating outbound connections).
pub fn endpoint(&self) -> &Endpoint {
&self.endpoint
}
/// Drain connections accepted on non-SWIM ALPNs.
pub fn drain_other_connections(&self) -> Vec<(NodeId, Connection)> {
self.other_accepted_conns.lock().unwrap().drain(..).collect()
}
/// The node's identity.
pub fn node_id(&self) -> NodeId {
self.node.node_id()

View file

@ -22,6 +22,7 @@ pub fn make_driver() -> IrohDriver {
relay_mode: RelayMode::Disabled,
node: test_config(),
peer_auth: None,
additional_alpns: vec![],
#[cfg(feature = "relay")]
embedded_relay_bind: None,
#[cfg(feature = "relay")]
@ -36,6 +37,7 @@ pub fn make_driver_with_auth(auth: Arc<Mutex<PeerAllowList>>) -> IrohDriver {
relay_mode: RelayMode::Disabled,
node: test_config(),
peer_auth: Some(auth),
additional_alpns: vec![],
#[cfg(feature = "relay")]
embedded_relay_bind: None,
#[cfg(feature = "relay")]
@ -50,6 +52,7 @@ pub fn make_driver_with_relay(relay_url: iroh::RelayUrl) -> IrohDriver {
relay_mode: RelayMode::Custom(relay_url.into()),
node: test_config(),
peer_auth: None,
additional_alpns: vec![],
#[cfg(feature = "relay")]
embedded_relay_bind: None,
#[cfg(feature = "relay")]

20
crates/streams/Cargo.toml Normal file
View file

@ -0,0 +1,20 @@
[package]
name = "swactor-streams"
version = "0.1.0"
edition = "2024"
[dependencies]
swactor = { path = "../..", features = ["serde"] }
swactor-std = { path = "../std" }
shared-types = { path = "../shared-types" }
distribution = { path = "../distribution" }
crossbeam-queue = "0.3.12"
tokio = { version = "1", features = ["sync", "io-util"] }
iroh = { version = "0.96" }
blake3 = "1"
serde = { version = "1", features = ["derive"] }
getrandom = "0.2"
[dev-dependencies]
proptest = "1"
tokio = { version = "1", features = ["rt-multi-thread", "macros", "test-util", "io-util"] }

View file

@ -0,0 +1,74 @@
use std::sync::Arc;
use iroh::endpoint::Connection;
use swactor::actor::ActorAddress;
use swactor::runtime::Runtime;
use crate::messages::{OneShot, StreamManagerMsg};
use crate::wire;
/// Spawn a bridge task that processes incoming stream connections and
/// forwards them to the StreamManager actor.
///
/// For each `(node_id, conn)` received:
/// 1. Accept the control bi-stream
/// 2. Read the stream header
/// 3. Send `StreamManagerMsg::IncomingConnection` to the StreamManager
pub fn spawn_accept_bridge(
accepted_rx: tokio::sync::mpsc::Receiver<([u8; 32], Connection)>,
runtime: Arc<Runtime>,
manager_addr: ActorAddress,
) -> tokio::task::JoinHandle<()> {
tokio::spawn(async move {
accept_bridge_loop(accepted_rx, runtime, manager_addr).await;
})
}
async fn accept_bridge_loop(
mut accepted_rx: tokio::sync::mpsc::Receiver<([u8; 32], Connection)>,
runtime: Arc<Runtime>,
manager_addr: ActorAddress,
) {
while let Some((node_id, conn)) = accepted_rx.recv().await {
let runtime = Arc::clone(&runtime);
let manager_addr = manager_addr;
tokio::spawn(async move {
if let Err(e) = handle_incoming(node_id, conn, &runtime, manager_addr).await
{
eprintln!("accept bridge: failed to handle incoming connection: {e}");
}
});
}
}
/// Handle a single incoming stream connection: accept the control bi-stream,
/// read the header, and forward to the StreamManager.
pub async fn handle_incoming(
node_id: [u8; 32],
conn: Connection,
runtime: &Runtime,
manager_addr: ActorAddress,
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
// Accept the control bi-stream (opener sends header here)
let (_, mut recv_ctrl) = conn.accept_bi().await?;
// Read the full header into a buffer. The opener finishes the send side
// after writing the header, so read_to_end collects all header bytes.
let header_bytes = recv_ctrl.read_to_end(4096).await?;
let header = wire::decode_header(&header_bytes)
.map_err(|e| format!("invalid stream header: {e}"))?;
let msg = StreamManagerMsg::IncomingConnection {
node_id,
stream_id: header.stream_id,
mode: header.mode,
config: header.config,
conn: OneShot::new(conn),
};
runtime
.send_to(manager_addr, msg)
.map_err(|e| format!("send to StreamManager failed: {e}"))?;
Ok(())
}

View file

@ -0,0 +1,252 @@
use std::sync::Arc;
use crossbeam_queue::ArrayQueue;
/// A pre-allocated frame buffer with read/write cursors for zero-alloc recycling.
///
/// Data is written starting at `write_pos` and read starting at `read_pos`.
/// When recycled via `reset()`, only the cursors are zeroed -- no memset.
pub struct FrameBuf {
data: Box<[u8]>,
read_pos: usize,
write_pos: usize,
}
impl FrameBuf {
/// Create a new buffer with the given capacity.
pub fn new(capacity: usize) -> Self {
FrameBuf {
data: vec![0u8; capacity].into_boxed_slice(),
read_pos: 0,
write_pos: 0,
}
}
/// Write data into the buffer. Returns the number of bytes written.
pub fn write(&mut self, src: &[u8]) -> usize {
let available = self.data.len() - self.write_pos;
let n = src.len().min(available);
self.data[self.write_pos..self.write_pos + n].copy_from_slice(&src[..n]);
self.write_pos += n;
n
}
/// Read data from the buffer. Returns the number of bytes read.
pub fn read(&mut self, dst: &mut [u8]) -> usize {
let available = self.write_pos - self.read_pos;
let n = dst.len().min(available);
dst[..n].copy_from_slice(&self.data[self.read_pos..self.read_pos + n]);
self.read_pos += n;
n
}
/// Reset cursors for reuse. Does NOT zero the data.
pub fn reset(&mut self) {
self.read_pos = 0;
self.write_pos = 0;
}
/// Number of unread bytes in the buffer.
pub fn remaining(&self) -> usize {
self.write_pos - self.read_pos
}
/// Available space for writing.
pub fn available(&self) -> usize {
self.data.len() - self.write_pos
}
/// Whether the buffer is full (no more write space).
pub fn is_full(&self) -> bool {
self.write_pos == self.data.len()
}
/// Whether all written data has been read.
pub fn is_empty(&self) -> bool {
self.read_pos == self.write_pos
}
/// Total capacity of the buffer.
pub fn capacity(&self) -> usize {
self.data.len()
}
/// The written portion of the buffer as a slice.
pub fn written(&self) -> &[u8] {
&self.data[..self.write_pos]
}
/// Load data directly into the buffer, replacing any existing content.
pub fn load(&mut self, src: &[u8]) {
assert!(
src.len() <= self.data.len(),
"source data exceeds buffer capacity"
);
self.data[..src.len()].copy_from_slice(src);
self.read_pos = 0;
self.write_pos = src.len();
}
}
/// A fixed-size lock-free pool of `FrameBuf`s backed by `crossbeam::ArrayQueue`.
///
/// Supports concurrent checkout/checkin between actor threads and tokio tasks.
#[derive(Clone)]
pub struct BufferPool {
inner: Arc<ArrayQueue<FrameBuf>>,
buf_capacity: usize,
}
impl BufferPool {
/// Create a new pool with `count` buffers, each of `buf_capacity` bytes.
pub fn new(count: usize, buf_capacity: usize) -> Self {
let queue = ArrayQueue::new(count);
for _ in 0..count {
let _ = queue.push(FrameBuf::new(buf_capacity));
}
BufferPool {
inner: Arc::new(queue),
buf_capacity,
}
}
/// Check out a buffer from the pool. Returns `None` if exhausted.
pub fn checkout(&self) -> Option<FrameBuf> {
self.inner.pop()
}
/// Return a buffer to the pool. The buffer is reset before being made available.
pub fn checkin(&self, mut buf: FrameBuf) {
buf.reset();
// If push fails (pool full), the buffer is dropped -- this is fine.
let _ = self.inner.push(buf);
}
/// Number of buffers currently available in the pool.
pub fn available(&self) -> usize {
self.inner.len()
}
/// The capacity of each buffer in the pool.
pub fn buf_capacity(&self) -> usize {
self.buf_capacity
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn frame_buf_write_then_read_returns_same_data() {
let mut buf = FrameBuf::new(64);
let data = b"hello, streams!";
let written = buf.write(data);
assert_eq!(written, data.len());
assert_eq!(buf.remaining(), data.len());
let mut out = vec![0u8; data.len()];
let read = buf.read(&mut out);
assert_eq!(read, data.len());
assert_eq!(&out, data);
assert!(buf.is_empty());
}
#[test]
fn frame_buf_partial_write_when_full() {
let mut buf = FrameBuf::new(8);
let written = buf.write(b"twelve chars");
assert_eq!(written, 8);
assert!(buf.is_full());
assert_eq!(buf.available(), 0);
}
#[test]
fn frame_buf_reset_allows_reuse() {
let mut buf = FrameBuf::new(16);
buf.write(b"first");
buf.reset();
assert!(buf.is_empty());
assert_eq!(buf.remaining(), 0);
assert_eq!(buf.available(), 16);
let written = buf.write(b"second");
assert_eq!(written, 6);
let mut out = vec![0u8; 6];
buf.read(&mut out);
assert_eq!(&out, b"second");
}
#[test]
fn frame_buf_load_replaces_content() {
let mut buf = FrameBuf::new(32);
buf.write(b"old data");
buf.load(b"new data here");
assert_eq!(buf.remaining(), 13);
let mut out = vec![0u8; 13];
buf.read(&mut out);
assert_eq!(&out, b"new data here");
}
#[test]
fn pool_checkout_checkin_cycle() {
let pool = BufferPool::new(4, 1024);
assert_eq!(pool.available(), 4);
let b1 = pool.checkout().unwrap();
let b2 = pool.checkout().unwrap();
assert_eq!(pool.available(), 2);
pool.checkin(b1);
assert_eq!(pool.available(), 3);
pool.checkin(b2);
assert_eq!(pool.available(), 4);
}
#[test]
fn pool_exhausted_returns_none() {
let pool = BufferPool::new(2, 64);
let _b1 = pool.checkout().unwrap();
let _b2 = pool.checkout().unwrap();
assert!(pool.checkout().is_none());
}
#[test]
fn pool_checkin_after_exhaustion_restores_availability() {
let pool = BufferPool::new(1, 64);
let buf = pool.checkout().unwrap();
assert!(pool.checkout().is_none());
pool.checkin(buf);
assert!(pool.checkout().is_some());
}
#[test]
fn pool_checkin_resets_buffer() {
let pool = BufferPool::new(1, 64);
let mut buf = pool.checkout().unwrap();
buf.write(b"dirty data");
assert_eq!(buf.remaining(), 10);
pool.checkin(buf);
let recycled = pool.checkout().unwrap();
assert!(recycled.is_empty());
assert_eq!(recycled.available(), 64);
}
#[test]
fn pool_clone_shares_same_backing() {
let pool = BufferPool::new(3, 128);
let pool2 = pool.clone();
let _b = pool.checkout().unwrap();
assert_eq!(pool2.available(), 2);
}
}

View file

@ -0,0 +1,41 @@
use crate::buffer::FrameBuf;
use crate::types::StreamError;
/// Commands sent from the actor to the send-side data-plane task.
pub enum SendCommand {
/// A buffer of data to write to the wire.
Data(FrameBuf),
/// Flush any partially-filled buffers.
Flush,
/// Gracefully close the send side.
Close,
}
/// Events sent from the send-side data-plane task back to the actor.
#[derive(Debug, Clone)]
pub enum SendEvent {
/// The data-plane task is ready to accept more data.
WriteReady,
/// An error occurred on the send side.
Error(StreamError),
/// The send side has been closed.
Closed,
}
/// Commands sent from the actor to the recv-side data-plane task.
pub enum RecvCommand {
/// Return a consumed buffer to the pool.
Consumed(FrameBuf),
/// Close the receive side.
Close,
}
/// Events sent from the recv-side data-plane task to the actor.
pub enum RecvEvent {
/// A buffer of received data.
Data(FrameBuf),
/// An error occurred on the recv side.
Error(StreamError),
/// The recv side has been closed (all stripes finished).
Closed,
}

View file

@ -0,0 +1,65 @@
use std::collections::HashMap;
use iroh::endpoint::Connection;
use iroh::{Endpoint, PublicKey};
use crate::types::StreamError;
use crate::wire::ALPN;
/// Cache of QUIC connections used for stream data transfer.
///
/// Separate from the SWIM connection pool in IrohDriver. All connections
/// are established using the stream ALPN (`swactor/stream/1`).
pub struct StreamConnectionCache {
connections: HashMap<[u8; 32], Connection>,
}
impl StreamConnectionCache {
pub fn new() -> Self {
StreamConnectionCache {
connections: HashMap::new(),
}
}
/// Get an existing healthy connection or establish a new one.
pub async fn get_or_connect(
&mut self,
endpoint: &Endpoint,
node_id: [u8; 32],
) -> Result<Connection, StreamError> {
// Check for cached connection that's still open
if let Some(conn) = self.connections.get(&node_id) {
if conn.close_reason().is_none() {
return Ok(conn.clone());
}
// Connection closed, remove it
self.connections.remove(&node_id);
}
let key = PublicKey::from_bytes(&node_id)
.map_err(|e| StreamError::BrokenPipe(format!("invalid public key: {e}")))?;
let conn = endpoint
.connect(key, ALPN)
.await
.map_err(|e| StreamError::BrokenPipe(format!("connect failed: {e}")))?;
self.connections.insert(node_id, conn.clone());
Ok(conn)
}
/// Remove dead connections from the cache.
pub fn prune_closed(&mut self) {
self.connections.retain(|_, conn| conn.close_reason().is_none());
}
/// Remove a specific connection.
pub fn remove(&mut self, node_id: &[u8; 32]) {
self.connections.remove(node_id);
}
/// Insert a connection into the cache.
pub fn insert(&mut self, node_id: [u8; 32], conn: Connection) {
self.connections.insert(node_id, conn);
}
}

View file

@ -0,0 +1,201 @@
//! Convenience extension traits for actors that use streams.
//!
//! `CtxStreams` wraps StreamManager message construction for use inside actor
//! handlers (defaults `reply_to`/`listener` to `ctx.self_addr()`).
//!
//! `RuntimeStreams` provides the same operations from a `Runtime` handle,
//! requiring explicit addresses since there's no implicit "self".
use swactor::actor::{ActorAddress, Ctx};
use swactor::runtime::Runtime;
use swactor_std::CtxNaming;
use swactor_std::RuntimeNaming;
use crate::messages::StreamManagerMsg;
use crate::types::{StreamConfig, StreamError, StreamId, StreamMode};
fn mgr_not_found() -> StreamError {
StreamError::BrokenPipe("StreamManager not found in name registry".into())
}
/// Stream operations available inside actor handlers via `Ctx`.
///
/// All methods default `reply_to` / `listener` to `ctx.self_addr()`.
pub trait CtxStreams {
/// Open a new stream to a remote node.
fn stream_open(
&self,
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
) -> Result<(), StreamError>;
/// Register as a stream listener for the given mode.
fn stream_listen(&self, mode: StreamMode) -> Result<(), StreamError>;
/// Accept an offered incoming stream.
fn stream_accept(&self, stream_id: StreamId) -> Result<(), StreamError>;
/// Reject an offered incoming stream.
fn stream_reject(&self, stream_id: StreamId) -> Result<(), StreamError>;
/// Close a stream.
fn stream_close(&self, stream_id: StreamId) -> Result<(), StreamError>;
}
impl CtxStreams for Ctx<'_> {
fn stream_open(
&self,
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send(
mgr,
StreamManagerMsg::Open {
target_node,
mode,
config,
reply_to: self.self_addr(),
},
)
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_listen(&self, mode: StreamMode) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send(
mgr,
StreamManagerMsg::Listen {
mode,
listener: self.self_addr(),
},
)
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_accept(&self, stream_id: StreamId) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send(
mgr,
StreamManagerMsg::Accept {
stream_id,
reply_to: self.self_addr(),
},
)
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_reject(&self, stream_id: StreamId) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send(mgr, StreamManagerMsg::Reject { stream_id })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_close(&self, stream_id: StreamId) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send(mgr, StreamManagerMsg::Close { stream_id })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
}
/// Stream operations available from a `Runtime` handle (outside actor handlers).
///
/// Requires explicit `reply_to` / `listener` addresses.
pub trait RuntimeStreams {
/// Open a new stream to a remote node.
fn stream_open(
&self,
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
reply_to: ActorAddress,
) -> Result<(), StreamError>;
/// Register an address as a stream listener for the given mode.
fn stream_listen(&self, mode: StreamMode, listener: ActorAddress) -> Result<(), StreamError>;
/// Accept an offered incoming stream.
fn stream_accept(
&self,
stream_id: StreamId,
reply_to: ActorAddress,
) -> Result<(), StreamError>;
/// Reject an offered incoming stream.
fn stream_reject(&self, stream_id: StreamId) -> Result<(), StreamError>;
/// Close a stream.
fn stream_close(&self, stream_id: StreamId) -> Result<(), StreamError>;
}
impl RuntimeStreams for Runtime {
fn stream_open(
&self,
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
reply_to: ActorAddress,
) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send_to(
mgr,
StreamManagerMsg::Open {
target_node,
mode,
config,
reply_to,
},
)
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_listen(&self, mode: StreamMode, listener: ActorAddress) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send_to(mgr, StreamManagerMsg::Listen { mode, listener })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_accept(
&self,
stream_id: StreamId,
reply_to: ActorAddress,
) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send_to(mgr, StreamManagerMsg::Accept { stream_id, reply_to })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_reject(&self, stream_id: StreamId) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send_to(mgr, StreamManagerMsg::Reject { stream_id })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
fn stream_close(&self, stream_id: StreamId) -> Result<(), StreamError> {
let mgr = self
.where_is(crate::manager::STREAM_MANAGER_NAME)
.ok_or_else(mgr_not_found)?;
self.send_to(mgr, StreamManagerMsg::Close { stream_id })
.map_err(|e| StreamError::BrokenPipe(e.to_string()))
}
}

View file

@ -0,0 +1,499 @@
use tokio::io::{AsyncReadExt, AsyncWriteExt};
use tokio::sync::mpsc;
use crate::buffer::BufferPool;
use crate::channel::{RecvEvent, SendCommand};
use crate::notify::NotifySink;
use crate::types::StreamError;
use crate::wire;
/// A send-side data-plane task for a single stripe.
///
/// Reads `SendCommand`s from the actor's channel, encodes them as wire
/// frames, and writes them to the underlying transport. Returns consumed
/// buffers to the pool.
///
/// Generic over `AsyncWrite` so it can be tested with `DuplexStream`.
pub async fn send_stripe_task<W>(
mut writer: W,
mut cmd_rx: mpsc::Receiver<SendCommand>,
pool: BufferPool,
notify: Option<NotifySink>,
) -> Result<(), StreamError>
where
W: AsyncWriteExt + Unpin + Send + 'static,
{
while let Some(cmd) = cmd_rx.recv().await {
match cmd {
SendCommand::Data(buf) => {
let frame = wire::encode_data_frame(buf.written());
writer
.write_all(&frame)
.await
.map_err(|e| StreamError::BrokenPipe(e.to_string()))?;
// Return buffer to pool
pool.checkin(buf);
// Signal write ready
if let Some(ref sink) = notify {
sink.write_ready();
}
}
SendCommand::Flush => {
writer
.flush()
.await
.map_err(|e| StreamError::BrokenPipe(e.to_string()))?;
}
SendCommand::Close => {
// Write end-of-stripe sentinel
let sentinel = wire::encode_end_of_stripe();
writer
.write_all(&sentinel)
.await
.map_err(|e| StreamError::BrokenPipe(e.to_string()))?;
writer
.flush()
.await
.map_err(|e| StreamError::BrokenPipe(e.to_string()))?;
break;
}
}
}
Ok(())
}
/// A recv-side data-plane task for a single stripe.
///
/// Reads wire-encoded frames from the transport, fills `FrameBuf`s from
/// the pool, and sends them to the actor via the event channel.
///
/// Generic over `AsyncRead` so it can be tested with `DuplexStream`.
pub async fn recv_stripe_task<R>(
mut reader: R,
evt_tx: mpsc::Sender<RecvEvent>,
pool: BufferPool,
notify: Option<NotifySink>,
) -> Result<(), StreamError>
where
R: AsyncReadExt + Unpin + Send + 'static,
{
loop {
// Read the 4-byte length prefix
let mut len_buf = [0u8; 4];
match reader.read_exact(&mut len_buf).await {
Ok(_) => {}
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => {
// Connection closed
let _ = evt_tx.send(RecvEvent::Closed).await;
if let Some(ref sink) = notify {
sink.closed();
}
return Ok(());
}
Err(e) => {
let err = StreamError::BrokenPipe(e.to_string());
let _ = evt_tx.send(RecvEvent::Error(err.clone())).await;
if let Some(ref sink) = notify {
sink.error();
}
return Err(err);
}
}
let payload_len = u32::from_be_bytes(len_buf) as usize;
// End-of-stripe sentinel
if payload_len == 0 {
let _ = evt_tx.send(RecvEvent::Closed).await;
if let Some(ref sink) = notify {
sink.closed();
}
return Ok(());
}
// Read the payload into a buffer from the pool
let mut buf = match pool.checkout() {
Some(b) => b,
None => {
let err = StreamError::BufferExhausted;
let _ = evt_tx.send(RecvEvent::Error(err.clone())).await;
if let Some(ref sink) = notify {
sink.error();
}
return Err(err);
}
};
let mut temp = vec![0u8; payload_len];
match reader.read_exact(&mut temp).await {
Ok(_) => {}
Err(e) => {
pool.checkin(buf);
let err = StreamError::BrokenPipe(e.to_string());
let _ = evt_tx.send(RecvEvent::Error(err.clone())).await;
if let Some(ref sink) = notify {
sink.error();
}
return Err(err);
}
}
buf.load(&temp);
// Send to actor
if evt_tx.send(RecvEvent::Data(buf)).await.is_err() {
return Err(StreamError::Disconnected);
}
if let Some(ref sink) = notify {
sink.data_ready();
}
}
}
/// Spawn a complete set of send-side stripe tasks.
///
/// Returns a Vec of `mpsc::Sender<SendCommand>` -- one per stripe.
/// The caller assigns chunks round-robin: chunk `i` goes to stripe `i % stripe_count`.
pub fn spawn_send_stripes<W, F>(
stripe_count: usize,
pool: BufferPool,
_notify: Option<NotifySink>,
mut writer_factory: F,
channel_capacity: usize,
) -> Vec<mpsc::Sender<SendCommand>>
where
W: AsyncWriteExt + Unpin + Send + 'static,
F: FnMut(usize) -> W,
{
let mut senders = Vec::with_capacity(stripe_count);
for i in 0..stripe_count {
let (tx, rx) = mpsc::channel(channel_capacity);
let writer = writer_factory(i);
let pool = pool.clone();
tokio::spawn(async move {
let _ = send_stripe_task(writer, rx, pool, None).await;
});
senders.push(tx);
}
senders
}
/// Spawn a complete set of recv-side stripe tasks.
///
/// Returns a single `mpsc::Receiver<RecvEvent>` that merges events from all stripes.
pub fn spawn_recv_stripes<R, F>(
stripe_count: usize,
pool: BufferPool,
_notify: Option<NotifySink>,
mut reader_factory: F,
channel_capacity: usize,
) -> mpsc::Receiver<RecvEvent>
where
R: AsyncReadExt + Unpin + Send + 'static,
F: FnMut(usize) -> R,
{
// All stripes feed into a single merged channel
let (merged_tx, merged_rx) = mpsc::channel(channel_capacity * stripe_count);
for i in 0..stripe_count {
let reader = reader_factory(i);
let pool = pool.clone();
let tx = merged_tx.clone();
tokio::spawn(async move {
let _ = recv_stripe_task(reader, tx, pool, None).await;
});
}
merged_rx
}
#[cfg(test)]
mod tests {
use super::*;
use crate::buffer::BufferPool;
use crate::types::StreamId;
fn make_test_pool(count: usize, capacity: usize) -> BufferPool {
BufferPool::new(count, capacity)
}
/// End-to-end: send data through a single stripe, receive it back.
#[tokio::test]
async fn single_stripe_transfer() {
let (client, server) = tokio::io::duplex(64 * 1024);
let pool = make_test_pool(16, 1024);
let (cmd_tx, cmd_rx) = mpsc::channel(16);
let (evt_tx, mut evt_rx) = mpsc::channel(16);
let send_pool = pool.clone();
let recv_pool = pool.clone();
let send_handle = tokio::spawn(async move {
send_stripe_task(client, cmd_rx, send_pool, None).await
});
let recv_handle = tokio::spawn(async move {
recv_stripe_task(server, evt_tx, recv_pool, None).await
});
// Send some data
let test_data = b"hello, streams!";
let mut buf = pool.checkout().unwrap();
buf.write(test_data);
cmd_tx.send(SendCommand::Data(buf)).await.unwrap();
// Send close
cmd_tx.send(SendCommand::Close).await.unwrap();
// Receive data
let evt = evt_rx.recv().await.unwrap();
match evt {
RecvEvent::Data(mut buf) => {
let mut out = vec![0u8; test_data.len()];
let n = buf.read(&mut out);
assert_eq!(n, test_data.len());
assert_eq!(&out, test_data);
}
other => panic!("expected Data, got {:?}", std::mem::discriminant(&other)),
}
// Receive close
let evt = evt_rx.recv().await.unwrap();
assert!(matches!(evt, RecvEvent::Closed));
send_handle.await.unwrap().unwrap();
recv_handle.await.unwrap().unwrap();
}
/// Multiple chunks through a single stripe.
#[tokio::test]
async fn multiple_chunks_single_stripe() {
let (client, server) = tokio::io::duplex(256 * 1024);
let pool = make_test_pool(32, 1024);
let (cmd_tx, cmd_rx) = mpsc::channel(32);
let (evt_tx, mut evt_rx) = mpsc::channel(32);
let sp = pool.clone();
let rp = pool.clone();
tokio::spawn(async move { send_stripe_task(client, cmd_rx, sp, None).await });
tokio::spawn(async move { recv_stripe_task(server, evt_tx, rp, None).await });
let chunk_count = 20;
for i in 0..chunk_count {
let mut buf = pool.checkout().unwrap();
let data = format!("chunk-{i:04}");
buf.write(data.as_bytes());
cmd_tx.send(SendCommand::Data(buf)).await.unwrap();
}
cmd_tx.send(SendCommand::Close).await.unwrap();
let mut received = Vec::new();
loop {
match evt_rx.recv().await.unwrap() {
RecvEvent::Data(mut buf) => {
let mut out = vec![0u8; buf.remaining()];
buf.read(&mut out);
received.push(String::from_utf8(out).unwrap());
}
RecvEvent::Closed => break,
RecvEvent::Error(e) => panic!("unexpected error: {e}"),
}
}
assert_eq!(received.len(), chunk_count);
for (i, chunk) in received.iter().enumerate() {
assert_eq!(chunk, &format!("chunk-{i:04}"));
}
}
/// Multi-stripe transfer with round-robin assignment.
#[tokio::test]
async fn multi_stripe_round_robin() {
let stripe_count = 4;
let chunk_count = 100;
// Pool needs enough buffers for in-flight data on both sides
let pool = make_test_pool(256, 256);
// Create duplex pairs for each stripe
let mut send_writers = Vec::new();
let mut recv_readers = Vec::new();
for _ in 0..stripe_count {
let (client, server) = tokio::io::duplex(64 * 1024);
send_writers.push(Some(client));
recv_readers.push(Some(server));
}
// Spawn recv stripe tasks
let (merged_tx, mut merged_rx) = mpsc::channel(chunk_count * 2);
for i in 0..stripe_count {
let reader = recv_readers[i].take().unwrap();
let p = pool.clone();
let tx = merged_tx.clone();
tokio::spawn(async move {
recv_stripe_task(reader, tx, p, None).await
});
}
drop(merged_tx); // so merged_rx closes when all tasks finish
// Spawn send stripe tasks
let mut stripe_txs = Vec::new();
for i in 0..stripe_count {
let (tx, rx) = mpsc::channel(32);
let writer = send_writers[i].take().unwrap();
let p = pool.clone();
tokio::spawn(async move {
send_stripe_task(writer, rx, p, None).await
});
stripe_txs.push(tx);
}
// Send chunks round-robin
for i in 0..chunk_count {
let stripe_idx = i % stripe_count;
let mut buf = pool.checkout().unwrap();
let data = format!("chunk-{i:04}");
buf.write(data.as_bytes());
stripe_txs[stripe_idx]
.send(SendCommand::Data(buf))
.await
.unwrap();
}
// Close all stripes
for tx in &stripe_txs {
tx.send(SendCommand::Close).await.unwrap();
}
// Collect all received data (order may differ per stripe)
let mut received = Vec::new();
let mut closed_count = 0;
while let Some(evt) = merged_rx.recv().await {
match evt {
RecvEvent::Data(mut buf) => {
let mut out = vec![0u8; buf.remaining()];
buf.read(&mut out);
received.push(String::from_utf8(out).unwrap());
}
RecvEvent::Closed => {
closed_count += 1;
if closed_count == stripe_count {
break;
}
}
RecvEvent::Error(e) => panic!("unexpected error: {e}"),
}
}
// All chunks should have arrived (order may vary across stripes)
assert_eq!(received.len(), chunk_count);
received.sort();
for (i, chunk) in received.iter().enumerate() {
assert_eq!(chunk, &format!("chunk-{i:04}"));
}
}
/// Graceful close: writer closes, receiver sees end-of-stripe then Closed.
#[tokio::test]
async fn graceful_close() {
let (client, server) = tokio::io::duplex(64 * 1024);
let pool = make_test_pool(8, 256);
let (cmd_tx, cmd_rx) = mpsc::channel(8);
let (evt_tx, mut evt_rx) = mpsc::channel(8);
let sp = pool.clone();
let rp = pool.clone();
tokio::spawn(async move { send_stripe_task(client, cmd_rx, sp, None).await });
tokio::spawn(async move { recv_stripe_task(server, evt_tx, rp, None).await });
// Close immediately without sending data
cmd_tx.send(SendCommand::Close).await.unwrap();
// Should receive Closed
let evt = evt_rx.recv().await.unwrap();
assert!(matches!(evt, RecvEvent::Closed));
}
/// Notification coalescing through NotifySink.
#[tokio::test]
async fn notification_coalescing() {
use crate::notify::*;
use std::sync::atomic::{AtomicUsize, Ordering};
use std::sync::Arc;
let flag = Arc::new(NotifyFlag::new());
let inject_count = Arc::new(AtomicUsize::new(0));
let count_clone = inject_count.clone();
let stream_id = StreamId::new_random();
let sink = NotifySink::new(flag.clone(), stream_id, move |_evt| {
count_clone.fetch_add(1, Ordering::SeqCst);
});
// First notification should inject
sink.data_ready();
assert_eq!(inject_count.load(Ordering::SeqCst), 1);
// Duplicate should coalesce (no inject)
sink.data_ready();
assert_eq!(inject_count.load(Ordering::SeqCst), 1);
// Clear and re-notify
flag.clear(DATA_READY);
sink.data_ready();
assert_eq!(inject_count.load(Ordering::SeqCst), 2);
// Different flag should still inject independently
sink.write_ready();
assert_eq!(inject_count.load(Ordering::SeqCst), 3);
}
/// Backpressure: when channel and active buffer are saturated, try_write returns 0.
#[tokio::test]
async fn send_backpressure() {
use crate::handle::create_stream_handle;
use crate::types::StreamConfig;
let config = StreamConfig {
stripe_count: 1,
frame_size: 64,
metadata: vec![],
};
// Pool of 4, channel of 2 -- we can fill both quickly
let (mut handle, _endpoints) = create_stream_handle(
StreamId::new_random(),
&config,
4,
2,
);
let data = vec![0xAA; 64]; // exactly fills one buffer
// Write 1: checks out buf, fills it (64 bytes), buf is full -> try_send succeeds
let n1 = handle.send.try_write(&data).unwrap();
assert_eq!(n1, 64);
// Write 2: checks out new buf, fills it, buf is full -> try_send succeeds
let n2 = handle.send.try_write(&data).unwrap();
assert_eq!(n2, 64);
// Write 3: checks out new buf, fills it, buf is full -> try_send fails (channel full)
// Data IS in the buffer (written=64), buffer kept locally
let n3 = handle.send.try_write(&data).unwrap();
assert_eq!(n3, 64);
// Write 4: active buf still full, buf.write() returns 0 (no space),
// try_send fails again -> returns 0 signaling backpressure
let n4 = handle.send.try_write(&data).unwrap();
assert_eq!(n4, 0); // backpressure!
}
}

View file

@ -0,0 +1,248 @@
use tokio::sync::mpsc;
use crate::buffer::{BufferPool, FrameBuf};
use crate::channel::{RecvCommand, RecvEvent, SendCommand, SendEvent};
use crate::types::{StreamConfig, StreamError, StreamId};
/// Sending half of a stream. Owned by the actor that sends data.
///
/// Uses `try_send`/`try_recv` for non-blocking operation on the actor thread.
pub struct SendHalf {
stream_id: StreamId,
cmd_tx: mpsc::Sender<SendCommand>,
evt_rx: mpsc::Receiver<SendEvent>,
pool: BufferPool,
active_buf: Option<FrameBuf>,
}
impl SendHalf {
pub fn stream_id(&self) -> StreamId {
self.stream_id
}
/// Write data into the stream. Returns the number of bytes consumed.
///
/// Fills the active buffer and sends full buffers to the data-plane task.
/// Returns 0 if the channel is full (backpressure) or the buffer pool
/// is exhausted. Does NOT block.
pub fn try_write(&mut self, data: &[u8]) -> Result<usize, StreamError> {
if data.is_empty() {
return Ok(0);
}
// Ensure we have an active buffer
if self.active_buf.is_none() {
self.active_buf = self.pool.checkout();
if self.active_buf.is_none() {
return Err(StreamError::BufferExhausted);
}
}
let buf = self.active_buf.as_mut().unwrap();
let written = buf.write(data);
// If the buffer is full, send it to the data-plane task
if buf.is_full() {
let full_buf = self.active_buf.take().unwrap();
match self.cmd_tx.try_send(SendCommand::Data(full_buf)) {
Ok(()) => {}
Err(mpsc::error::TrySendError::Full(cmd)) => {
// Put the buffer back -- channel is full (backpressure).
// Data is already written into the buffer. Next call to
// try_write will attempt try_send again.
if let SendCommand::Data(buf) = cmd {
self.active_buf = Some(buf);
}
return Ok(written);
}
Err(mpsc::error::TrySendError::Closed(_)) => {
return Err(StreamError::Disconnected);
}
}
}
Ok(written)
}
/// Flush any partially-filled buffer to the data-plane task.
pub fn flush(&mut self) -> Result<(), StreamError> {
if let Some(buf) = self.active_buf.take() {
if buf.remaining() > 0 || buf.written().len() > 0 {
self.cmd_tx
.try_send(SendCommand::Data(buf))
.map_err(|_| StreamError::Disconnected)?;
} else {
self.pool.checkin(buf);
}
}
self.cmd_tx
.try_send(SendCommand::Flush)
.map_err(|_| StreamError::Disconnected)?;
Ok(())
}
/// Close the send side of the stream.
pub fn close(&mut self) -> Result<(), StreamError> {
if let Some(buf) = self.active_buf.take() {
if buf.written().len() > 0 {
let _ = self.cmd_tx.try_send(SendCommand::Data(buf));
} else {
self.pool.checkin(buf);
}
}
self.cmd_tx
.try_send(SendCommand::Close)
.map_err(|_| StreamError::Disconnected)?;
Ok(())
}
/// Poll for events from the data-plane task (non-blocking).
pub fn try_recv_event(&mut self) -> Option<SendEvent> {
self.evt_rx.try_recv().ok()
}
}
/// Receiving half of a stream. Owned by the actor that receives data.
pub struct RecvHalf {
stream_id: StreamId,
evt_rx: mpsc::Receiver<RecvEvent>,
cmd_tx: mpsc::Sender<RecvCommand>,
pool: BufferPool,
active_buf: Option<FrameBuf>,
}
impl RecvHalf {
pub fn stream_id(&self) -> StreamId {
self.stream_id
}
/// Read data from the stream. Returns the number of bytes read.
///
/// Drains the active buffer, then pulls new buffers from the channel.
/// Returns 0 if no data is currently available. Does NOT block.
pub fn try_read(&mut self, dst: &mut [u8]) -> Result<usize, StreamError> {
if dst.is_empty() {
return Ok(0);
}
// Drain any active buffer first
if let Some(buf) = &mut self.active_buf {
if buf.remaining() > 0 {
let read = buf.read(dst);
if buf.is_empty() {
let buf = self.active_buf.take().unwrap();
self.pool.checkin(buf);
}
return Ok(read);
} else {
let buf = self.active_buf.take().unwrap();
self.pool.checkin(buf);
}
}
// Try to pull a new buffer from the channel
match self.evt_rx.try_recv() {
Ok(RecvEvent::Data(mut buf)) => {
let read = buf.read(dst);
if buf.is_empty() {
self.pool.checkin(buf);
} else {
self.active_buf = Some(buf);
}
Ok(read)
}
Ok(RecvEvent::Error(e)) => Err(e),
Ok(RecvEvent::Closed) => Err(StreamError::Closed),
Err(mpsc::error::TryRecvError::Empty) => Ok(0),
Err(mpsc::error::TryRecvError::Disconnected) => Err(StreamError::Disconnected),
}
}
/// Check if data is available without consuming it.
pub fn has_data(&self) -> bool {
if let Some(buf) = &self.active_buf {
if buf.remaining() > 0 {
return true;
}
}
!self.evt_rx.is_empty()
}
/// Close the receive side of the stream.
pub fn close(&mut self) -> Result<(), StreamError> {
if let Some(buf) = self.active_buf.take() {
self.pool.checkin(buf);
}
self.cmd_tx
.try_send(RecvCommand::Close)
.map_err(|_| StreamError::Disconnected)?;
Ok(())
}
}
/// Combined stream handle with both send and receive halves.
///
/// `Send` but NOT `Clone` (mpsc::Receiver is not Clone).
pub struct StreamHandle {
pub send: SendHalf,
pub recv: RecvHalf,
}
/// Channel endpoints for the data-plane tasks.
pub struct DataPlaneEndpoints {
/// Receive send commands from the actor.
pub send_cmd_rx: mpsc::Receiver<SendCommand>,
/// Send events back to the actor.
pub send_evt_tx: mpsc::Sender<SendEvent>,
/// Send received data to the actor.
pub recv_evt_tx: mpsc::Sender<RecvEvent>,
/// Receive consume/close commands from the actor.
pub recv_cmd_rx: mpsc::Receiver<RecvCommand>,
/// Shared buffer pool.
pub pool: BufferPool,
}
/// Create a stream handle and its corresponding data-plane channel endpoints.
///
/// `channel_capacity` controls how many FrameBufs can be in-flight between
/// the actor and the data-plane tasks.
pub fn create_stream_handle(
stream_id: StreamId,
config: &StreamConfig,
pool_size: usize,
channel_capacity: usize,
) -> (StreamHandle, DataPlaneEndpoints) {
let pool = BufferPool::new(pool_size, config.frame_size as usize);
let (send_cmd_tx, send_cmd_rx) = mpsc::channel(channel_capacity);
let (send_evt_tx, send_evt_rx) = mpsc::channel(channel_capacity);
let (recv_evt_tx, recv_evt_rx) = mpsc::channel(channel_capacity);
let (recv_cmd_tx, recv_cmd_rx) = mpsc::channel(channel_capacity);
let handle = StreamHandle {
send: SendHalf {
stream_id,
cmd_tx: send_cmd_tx,
evt_rx: send_evt_rx,
pool: pool.clone(),
active_buf: None,
},
recv: RecvHalf {
stream_id,
evt_rx: recv_evt_rx,
cmd_tx: recv_cmd_tx,
pool: pool.clone(),
active_buf: None,
},
};
let endpoints = DataPlaneEndpoints {
send_cmd_rx,
send_evt_tx,
recv_evt_tx,
recv_cmd_rx,
pool,
};
(handle, endpoints)
}

23
crates/streams/src/lib.rs Normal file
View file

@ -0,0 +1,23 @@
pub mod accept;
pub mod buffer;
pub mod channel;
pub mod connection;
pub mod ctx_ext;
pub mod data_plane;
pub mod handle;
pub mod manager;
pub mod messages;
pub mod notify;
pub mod types;
pub mod wire;
pub use buffer::{BufferPool, FrameBuf};
pub use channel::{RecvCommand, RecvEvent, SendCommand, SendEvent};
pub use connection::StreamConnectionCache;
pub use ctx_ext::{CtxStreams, RuntimeStreams};
pub use handle::{create_stream_handle, DataPlaneEndpoints, RecvHalf, SendHalf, StreamHandle};
pub use manager::{StreamManager, STREAM_MANAGER_NAME};
pub use messages::{OneShot, StreamManagerMsg, StreamNotification};
pub use notify::{NotifyFlag, NotifySink, StreamEvent, StreamEventKind};
pub use types::{ResumeToken, StreamConfig, StreamError, StreamId, StreamMode};
pub use wire::{StreamHeader, ALPN, MAGIC, VERSION};

View file

@ -0,0 +1,622 @@
use std::collections::HashMap;
use std::sync::Arc;
use iroh::{Endpoint, PublicKey};
use tokio::io::AsyncWriteExt;
use swactor::actor::{ActorAddress, ActorInterface, Ctx, Down};
use swactor::runtime::Runtime;
use crate::connection::StreamConnectionCache;
use crate::data_plane;
use crate::handle::{create_stream_handle, StreamHandle};
use crate::messages::{OneShot, StreamManagerMsg, StreamNotification};
use crate::types::{StreamConfig, StreamError, StreamId, StreamMode};
use crate::wire;
/// Well-known name for the StreamManager actor in the name registry.
pub const STREAM_MANAGER_NAME: &str = "StreamManager";
/// Accept byte sent back on control stream to indicate stream acceptance.
const ACCEPT_BYTE: u8 = 0x01;
/// Reject byte sent back on control stream to indicate stream rejection.
const REJECT_BYTE: u8 = 0x00;
struct StreamState {
_stream_id: StreamId,
owner: ActorAddress,
_mode: StreamMode,
_remote_node: [u8; 32],
}
struct PendingIncoming {
_node_id: [u8; 32],
_stream_id: StreamId,
_mode: StreamMode,
config: StreamConfig,
conn: OneShot<iroh::endpoint::Connection>,
}
pub struct StreamManager {
streams: HashMap<StreamId, StreamState>,
pending_incoming: HashMap<StreamId, PendingIncoming>,
listeners: HashMap<StreamMode, Vec<ActorAddress>>,
#[allow(dead_code)]
conn_cache: StreamConnectionCache,
endpoint: Endpoint,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
self_addr: Option<ActorAddress>,
}
impl StreamManager {
pub fn new(
endpoint: Endpoint,
tokio_handle: tokio::runtime::Handle,
runtime: Arc<Runtime>,
) -> Self {
StreamManager {
streams: HashMap::new(),
pending_incoming: HashMap::new(),
listeners: HashMap::new(),
conn_cache: StreamConnectionCache::new(),
endpoint,
tokio_handle,
runtime,
self_addr: None,
}
}
fn handle_open(
&mut self,
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
reply_to: ActorAddress,
) {
let stream_id = StreamId::new_random();
let endpoint = self.endpoint.clone();
let runtime = Arc::clone(&self.runtime);
let self_addr = self.self_addr.expect("StreamManager not started");
self.tokio_handle.spawn(async move {
let result = open_stream_async(endpoint, target_node, stream_id, mode, &config).await;
let msg = StreamManagerMsg::OpenCompleted {
stream_id,
reply_to,
result: OneShot::new(result),
};
let _ = runtime.send_to(self_addr, msg);
});
}
fn handle_open_completed(
&mut self,
ctx: &Ctx,
stream_id: StreamId,
reply_to: ActorAddress,
result: OneShot<Result<StreamHandle, StreamError>>,
) {
match result.take() {
Some(Ok(handle)) => {
self.streams.insert(
stream_id,
StreamState {
_stream_id: stream_id,
owner: reply_to,
_mode: StreamMode::BlobTransfer,
_remote_node: [0; 32],
},
);
let notif = StreamNotification::StreamReady {
stream_id,
handle: OneShot::new(handle),
};
let _ = ctx.send(reply_to, notif);
}
Some(Err(err)) => {
let notif = StreamNotification::StreamFailed {
stream_id,
error: err,
};
let _ = ctx.send(reply_to, notif);
}
None => {
// OneShot already consumed — should not happen
eprintln!("StreamManager: OpenCompleted result already consumed");
}
}
}
fn handle_incoming_connection(
&mut self,
ctx: &Ctx,
node_id: [u8; 32],
stream_id: StreamId,
mode: StreamMode,
config: StreamConfig,
conn: OneShot<iroh::endpoint::Connection>,
) {
// Notify listeners for this mode
if let Some(listeners) = self.listeners.get(&mode) {
let notif = StreamNotification::StreamOffer {
stream_id,
mode,
metadata: config.metadata.clone(),
from_node: node_id,
};
for listener in listeners {
let _ = ctx.send(*listener, notif.clone());
}
}
// Store pending incoming for Accept/Reject
self.pending_incoming.insert(
stream_id,
PendingIncoming {
_node_id: node_id,
_stream_id: stream_id,
_mode: mode,
config,
conn,
},
);
}
fn handle_accept(&mut self, stream_id: StreamId, reply_to: ActorAddress) {
let pending = match self.pending_incoming.remove(&stream_id) {
Some(p) => p,
None => {
eprintln!("StreamManager: Accept for unknown stream {stream_id}");
return;
}
};
let conn = match pending.conn.take() {
Some(c) => c,
None => {
eprintln!("StreamManager: Accept connection already consumed for {stream_id}");
return;
}
};
let config = pending.config;
let runtime = Arc::clone(&self.runtime);
let self_addr = self.self_addr.expect("StreamManager not started");
self.tokio_handle.spawn(async move {
let result = accept_stream_async(conn, stream_id, &config).await;
let msg = StreamManagerMsg::AcceptCompleted {
stream_id,
reply_to,
result: OneShot::new(result),
};
let _ = runtime.send_to(self_addr, msg);
});
}
fn handle_accept_completed(
&mut self,
ctx: &Ctx,
stream_id: StreamId,
reply_to: ActorAddress,
result: OneShot<Result<StreamHandle, StreamError>>,
) {
match result.take() {
Some(Ok(handle)) => {
self.streams.insert(
stream_id,
StreamState {
_stream_id: stream_id,
owner: reply_to,
_mode: StreamMode::BlobTransfer,
_remote_node: [0; 32],
},
);
let notif = StreamNotification::StreamReady {
stream_id,
handle: OneShot::new(handle),
};
let _ = ctx.send(reply_to, notif);
}
Some(Err(err)) => {
let notif = StreamNotification::StreamFailed {
stream_id,
error: err,
};
let _ = ctx.send(reply_to, notif);
}
None => {
eprintln!("StreamManager: AcceptCompleted result already consumed");
}
}
}
fn handle_reject(&mut self, stream_id: StreamId) {
if let Some(pending) = self.pending_incoming.remove(&stream_id) {
// If we have the connection, send reject and close
if let Some(conn) = pending.conn.take() {
let runtime = Arc::clone(&self.runtime);
self.tokio_handle.spawn(async move {
// Best-effort: send reject on any open bi-stream, then close
let _ = reject_stream_async(&conn).await;
drop(conn);
drop(runtime);
});
}
}
}
fn handle_listen(&mut self, mode: StreamMode, listener: ActorAddress) {
self.listeners
.entry(mode)
.or_insert_with(Vec::new)
.push(listener);
}
fn handle_close(&mut self, stream_id: StreamId) {
// Remove stream state; data-plane tasks terminate when channels drop
self.streams.remove(&stream_id);
}
}
impl ActorInterface for StreamManager {
type Incoming = StreamManagerMsg;
type Response = ();
fn on_start(&mut self, ctx: &Ctx) {
self.self_addr = Some(ctx.self_addr());
}
fn handle(&mut self, ctx: &Ctx, msg: StreamManagerMsg) {
match msg {
StreamManagerMsg::Open {
target_node,
mode,
config,
reply_to,
} => self.handle_open(target_node, mode, config, reply_to),
StreamManagerMsg::Accept {
stream_id,
reply_to,
} => self.handle_accept(stream_id, reply_to),
StreamManagerMsg::Reject { stream_id } => self.handle_reject(stream_id),
StreamManagerMsg::Listen { mode, listener } => self.handle_listen(mode, listener),
StreamManagerMsg::Close { stream_id } => self.handle_close(stream_id),
StreamManagerMsg::IncomingConnection {
node_id,
stream_id,
mode,
config,
conn,
} => self.handle_incoming_connection(ctx, node_id, stream_id, mode, config, conn),
StreamManagerMsg::OpenCompleted {
stream_id,
reply_to,
result,
} => self.handle_open_completed(ctx, stream_id, reply_to, result),
StreamManagerMsg::AcceptCompleted {
stream_id,
reply_to,
result,
} => self.handle_accept_completed(ctx, stream_id, reply_to, result),
}
}
fn handle_down(&mut self, _ctx: &Ctx, down: Down) {
// Clean up streams owned by the dead actor
let dead_addr = down.addr;
self.streams.retain(|_, state| state.owner != dead_addr);
// Remove from listeners
for listeners in self.listeners.values_mut() {
listeners.retain(|addr| *addr != dead_addr);
}
}
}
// ─── Async helpers (run inside tokio tasks) ─────────────────────────────
/// Open a stream to a remote node: connect, send header on control bi-stream,
/// wait for accept/reject, then spawn data-plane tasks.
async fn open_stream_async(
endpoint: Endpoint,
target_node: [u8; 32],
stream_id: StreamId,
mode: StreamMode,
config: &StreamConfig,
) -> Result<StreamHandle, StreamError> {
let key = PublicKey::from_bytes(&target_node)
.map_err(|e| StreamError::BrokenPipe(format!("invalid public key: {e}")))?;
let conn = endpoint
.connect(key, wire::ALPN)
.await
.map_err(|e| StreamError::BrokenPipe(format!("connect failed: {e}")))?;
// Open control bi-stream and send header
let (mut send_ctrl, _recv_ctrl) = conn
.open_bi()
.await
.map_err(|e| StreamError::BrokenPipe(format!("open_bi failed: {e}")))?;
let header = wire::StreamHeader {
stream_id,
mode,
config: config.clone(),
};
let header_bytes = wire::encode_header(&header);
send_ctrl
.write_all(&header_bytes)
.await
.map_err(|e| StreamError::BrokenPipe(format!("write header failed: {e}")))?;
send_ctrl
.finish()
.map_err(|e| StreamError::BrokenPipe(format!("finish control send failed: {e}")))?;
// Wait for accept/reject response on a uni-stream opened by the acceptor.
// (The bi-stream's send half was dropped by the accept bridge after reading
// the header, so the acceptor responds via a separate uni-stream.)
let mut response_recv = conn
.accept_uni()
.await
.map_err(|e| StreamError::BrokenPipe(format!("accept response stream failed: {e}")))?;
let mut response = [0u8; 1];
response_recv
.read_exact(&mut response)
.await
.map_err(|e| StreamError::BrokenPipe(format!("read accept/reject failed: {e}")))?;
if response[0] != ACCEPT_BYTE {
return Err(StreamError::BrokenPipe("stream rejected by remote".into()));
}
// Create StreamHandle and spawn data-plane tasks
let stripe_count = config.stripe_count as usize;
let (handle, endpoints) = create_stream_handle(stream_id, config, 32, 16);
// Spawn send stripe tasks with QUIC uni-streams
{
let pool = endpoints.pool.clone();
let mut cmd_rx = endpoints.send_cmd_rx;
let evt_tx = endpoints.send_evt_tx;
let conn_clone = conn.clone();
let sc = stripe_count;
tokio::spawn(async move {
// Open uni-streams for each stripe
let mut writers = Vec::with_capacity(sc);
for _ in 0..sc {
match conn_clone.open_uni().await {
Ok(send_stream) => writers.push(send_stream),
Err(e) => {
let _ = evt_tx
.send(crate::channel::SendEvent::Error(StreamError::BrokenPipe(
format!("open_uni failed: {e}"),
)))
.await;
return;
}
}
}
// Simple single-task approach: round-robin commands across stripes
let mut stripe_idx = 0;
while let Some(cmd) = cmd_rx.recv().await {
match cmd {
crate::channel::SendCommand::Data(buf) => {
let frame = wire::encode_data_frame(buf.written());
let writer = &mut writers[stripe_idx];
if let Err(e) = writer.write_all(&frame).await {
pool.checkin(buf);
let _ = evt_tx
.send(crate::channel::SendEvent::Error(
StreamError::BrokenPipe(e.to_string()),
))
.await;
return;
}
pool.checkin(buf);
stripe_idx = (stripe_idx + 1) % sc;
}
crate::channel::SendCommand::Flush => {
for writer in &mut writers {
let _ = writer.flush().await;
}
}
crate::channel::SendCommand::Close => {
let sentinel = wire::encode_end_of_stripe();
for writer in &mut writers {
let _ = writer.write_all(&sentinel).await;
let _ = writer.finish();
}
break;
}
}
}
});
}
// Spawn recv stripe tasks with QUIC uni-streams (accepted from remote)
{
let pool = endpoints.pool.clone();
let evt_tx = endpoints.recv_evt_tx;
let conn_clone = conn.clone();
let sc = stripe_count;
tokio::spawn(async move {
// Accept uni-streams for each recv stripe
let mut closed_count = 0;
loop {
match conn_clone.accept_uni().await {
Ok(recv_stream) => {
let pool = pool.clone();
let tx = evt_tx.clone();
tokio::spawn(async move {
let _ =
data_plane::recv_stripe_task(recv_stream, tx, pool, None).await;
});
closed_count += 1;
if closed_count >= sc {
// We only expect stripe_count recv streams
// but keep accepting in case more arrive
}
}
Err(_) => break,
}
}
});
}
Ok(handle)
}
/// Accept a stream: send accept byte on control stream, spawn data-plane tasks.
async fn accept_stream_async(
conn: iroh::endpoint::Connection,
stream_id: StreamId,
config: &StreamConfig,
) -> Result<StreamHandle, StreamError> {
// Open a uni-stream to send the accept byte back
// (The opener reads from the recv side of the bi-stream they opened.
// We need to open our own bi-stream to send the response.)
// Actually, the opener opened a bi-stream - we need to accept it and
// respond on it. But IncomingConnection already accepted the bi-stream
// and read the header. We need the send half of that bi-stream.
//
// Since the accept bridge consumed the bi-stream to read the header,
// we send the accept response on a new uni-stream that the opener
// will accept_uni on. But the plan says "1-byte accept/reject response"
// on the same control bi-stream.
//
// The design: the accept bridge reads the header from the bi-stream
// (recv side), and the StreamManager sends accept/reject on the
// send side. Since the accept bridge consumed the Connection but not
// the bi-stream send half, we need a different approach.
//
// Simpler: use a uni-stream for the response.
let mut response_stream = conn
.open_uni()
.await
.map_err(|e| StreamError::BrokenPipe(format!("open response stream failed: {e}")))?;
response_stream
.write_all(&[ACCEPT_BYTE])
.await
.map_err(|e| StreamError::BrokenPipe(format!("write accept byte failed: {e}")))?;
response_stream
.finish()
.map_err(|e| StreamError::BrokenPipe(format!("finish response stream failed: {e}")))?;
let stripe_count = config.stripe_count as usize;
let (handle, endpoints) = create_stream_handle(stream_id, config, 32, 16);
// Spawn send stripe tasks — we open uni-streams to write
{
let pool = endpoints.pool.clone();
let mut cmd_rx = endpoints.send_cmd_rx;
let evt_tx = endpoints.send_evt_tx;
let conn_clone = conn.clone();
let sc = stripe_count;
tokio::spawn(async move {
let mut writers = Vec::with_capacity(sc);
for _ in 0..sc {
match conn_clone.open_uni().await {
Ok(send_stream) => writers.push(send_stream),
Err(e) => {
let _ = evt_tx
.send(crate::channel::SendEvent::Error(StreamError::BrokenPipe(
format!("open_uni failed: {e}"),
)))
.await;
return;
}
}
}
let mut stripe_idx = 0;
while let Some(cmd) = cmd_rx.recv().await {
match cmd {
crate::channel::SendCommand::Data(buf) => {
let frame = wire::encode_data_frame(buf.written());
let writer = &mut writers[stripe_idx];
if let Err(e) = writer.write_all(&frame).await {
pool.checkin(buf);
let _ = evt_tx
.send(crate::channel::SendEvent::Error(
StreamError::BrokenPipe(e.to_string()),
))
.await;
return;
}
pool.checkin(buf);
stripe_idx = (stripe_idx + 1) % sc;
}
crate::channel::SendCommand::Flush => {
for writer in &mut writers {
let _ = writer.flush().await;
}
}
crate::channel::SendCommand::Close => {
let sentinel = wire::encode_end_of_stripe();
for writer in &mut writers {
let _ = writer.write_all(&sentinel).await;
let _ = writer.finish();
}
break;
}
}
}
});
}
// Spawn recv stripe tasks — accept uni-streams from remote
{
let pool = endpoints.pool.clone();
let evt_tx = endpoints.recv_evt_tx;
let conn_clone = conn.clone();
tokio::spawn(async move {
loop {
match conn_clone.accept_uni().await {
Ok(recv_stream) => {
let pool = pool.clone();
let tx = evt_tx.clone();
tokio::spawn(async move {
let _ =
data_plane::recv_stripe_task(recv_stream, tx, pool, None).await;
});
}
Err(_) => break,
}
}
});
}
Ok(handle)
}
/// Send reject on a connection (best-effort).
async fn reject_stream_async(
conn: &iroh::endpoint::Connection,
) -> Result<(), StreamError> {
let mut response_stream = conn
.open_uni()
.await
.map_err(|e| StreamError::BrokenPipe(format!("open response stream failed: {e}")))?;
response_stream
.write_all(&[REJECT_BYTE])
.await
.map_err(|e| StreamError::BrokenPipe(format!("write reject byte failed: {e}")))?;
response_stream
.finish()
.map_err(|e| StreamError::BrokenPipe(format!("finish response stream failed: {e}")))?;
Ok(())
}

View file

@ -0,0 +1,176 @@
use std::fmt;
use std::sync::{Arc, Mutex};
use iroh::endpoint::Connection;
use swactor::actor::ActorAddress;
use crate::handle::StreamHandle;
use crate::types::{StreamConfig, StreamError, StreamId, StreamMode};
// ─── OneShot ────────────────────────────────────────────────────────────
/// Clone-friendly wrapper for non-Clone data (StreamHandle, Connection).
///
/// The first `.take()` extracts the value; subsequent clones/takes get `None`.
/// This allows non-Clone payloads to live inside Clone message enums required
/// by the actor system's `Message` trait.
pub struct OneShot<T>(Arc<Mutex<Option<T>>>);
impl<T> OneShot<T> {
pub fn new(val: T) -> Self {
OneShot(Arc::new(Mutex::new(Some(val))))
}
/// Extract the value. Returns `Some` exactly once; all subsequent calls
/// (including from clones) return `None`.
pub fn take(&self) -> Option<T> {
self.0.lock().unwrap().take()
}
}
impl<T> Clone for OneShot<T> {
fn clone(&self) -> Self {
OneShot(Arc::clone(&self.0))
}
}
impl<T> fmt::Debug for OneShot<T> {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
let has_value = self.0.lock().unwrap().is_some();
write!(f, "OneShot({})", if has_value { "Some" } else { "None" })
}
}
// SAFETY: OneShot<T> is Send+Sync because access is guarded by Mutex,
// and Arc provides shared ownership.
unsafe impl<T: Send> Send for OneShot<T> {}
unsafe impl<T: Send> Sync for OneShot<T> {}
// ─── StreamManagerMsg ───────────────────────────────────────────────────
/// Messages sent TO the StreamManager actor.
#[derive(Clone, Debug)]
pub enum StreamManagerMsg {
/// Open a new stream to a remote node.
Open {
target_node: [u8; 32],
mode: StreamMode,
config: StreamConfig,
/// The requesting actor's address; receives StreamNotification.
reply_to: ActorAddress,
},
/// Accept an offered incoming stream.
Accept {
stream_id: StreamId,
/// Receives StreamNotification::StreamReady.
reply_to: ActorAddress,
},
/// Reject an offered incoming stream.
Reject {
stream_id: StreamId,
},
/// Register as a stream listener for a given mode.
Listen {
mode: StreamMode,
/// Receives StreamNotification::StreamOffer.
listener: ActorAddress,
},
/// Close a stream.
Close {
stream_id: StreamId,
},
// -- Internal (from tokio tasks back to StreamManager) --
/// Incoming connection from the accept bridge task.
IncomingConnection {
node_id: [u8; 32],
stream_id: StreamId,
mode: StreamMode,
config: StreamConfig,
/// QUIC connection for data stripes.
conn: OneShot<Connection>,
},
/// Async open task completed.
OpenCompleted {
stream_id: StreamId,
reply_to: ActorAddress,
result: OneShot<Result<StreamHandle, StreamError>>,
},
/// Async accept task completed (data-plane tasks spawned).
AcceptCompleted {
stream_id: StreamId,
reply_to: ActorAddress,
result: OneShot<Result<StreamHandle, StreamError>>,
},
}
// ─── StreamNotification ─────────────────────────────────────────────────
/// Notifications sent FROM StreamManager TO user actors.
#[derive(Clone, Debug)]
pub enum StreamNotification {
/// A stream is ready for use (open or accept completed successfully).
StreamReady {
stream_id: StreamId,
handle: OneShot<StreamHandle>,
},
/// A remote node is offering a new stream.
StreamOffer {
stream_id: StreamId,
mode: StreamMode,
metadata: Vec<u8>,
from_node: [u8; 32],
},
/// A stream was closed.
StreamClosed {
stream_id: StreamId,
reason: Option<StreamError>,
},
/// A stream open/accept failed.
StreamFailed {
stream_id: StreamId,
error: StreamError,
},
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn oneshot_take_once_semantics() {
let os = OneShot::new(42u64);
assert_eq!(os.take(), Some(42));
assert_eq!(os.take(), None);
}
#[test]
fn oneshot_clone_shares_value() {
let os = OneShot::new("hello".to_string());
let clone = os.clone();
// First take from clone succeeds
assert_eq!(clone.take(), Some("hello".to_string()));
// Original now gets None
assert_eq!(os.take(), None);
}
#[test]
fn oneshot_debug_format() {
let os = OneShot::new(1);
assert_eq!(format!("{os:?}"), "OneShot(Some)");
os.take();
assert_eq!(format!("{os:?}"), "OneShot(None)");
}
fn assert_message<T: 'static + Clone + Send + Sync>() {}
#[test]
fn stream_manager_msg_is_message() {
assert_message::<StreamManagerMsg>();
}
#[test]
fn stream_notification_is_message() {
assert_message::<StreamNotification>();
}
}

View file

@ -0,0 +1,172 @@
use std::sync::atomic::{AtomicU8, Ordering};
use std::sync::Arc;
/// Bit positions for notification flags.
pub const DATA_READY: u8 = 0b0000_0001;
pub const WRITE_READY: u8 = 0b0000_0010;
pub const CLOSED: u8 = 0b0000_0100;
pub const ERROR: u8 = 0b0000_1000;
/// Atomic bitflags for coalescing notifications to an actor.
///
/// Multiple data-plane tasks may set flags concurrently. The actor clears
/// flags after handling them. If a flag is already set when a task tries
/// to set it, the notification is coalesced (deduplicated).
pub struct NotifyFlag {
flags: AtomicU8,
}
impl NotifyFlag {
pub fn new() -> Self {
NotifyFlag {
flags: AtomicU8::new(0),
}
}
/// Set a flag bit. Returns `true` if the bit was previously clear
/// (i.e., this is a new notification that should trigger an inject).
/// Returns `false` if already set (coalesced, no inject needed).
pub fn set(&self, kind: u8) -> bool {
let prev = self.flags.fetch_or(kind, Ordering::AcqRel);
(prev & kind) == 0
}
/// Clear a flag bit. Called by the actor after handling.
pub fn clear(&self, kind: u8) {
self.flags.fetch_and(!kind, Ordering::AcqRel);
}
/// Read all currently-set flags.
pub fn read(&self) -> u8 {
self.flags.load(Ordering::Acquire)
}
/// Check if a specific flag is set.
pub fn is_set(&self, kind: u8) -> bool {
(self.read() & kind) != 0
}
}
impl Default for NotifyFlag {
fn default() -> Self {
Self::new()
}
}
/// The kind of stream event delivered to an actor.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum StreamEventKind {
DataReady,
WriteReady,
Closed,
Error,
}
/// A lightweight notification message injected into an actor's mailbox.
#[derive(Debug, Clone)]
pub struct StreamEvent {
pub stream_id: crate::types::StreamId,
pub kind: StreamEventKind,
}
/// Sink that data-plane tasks use to inject notifications into the actor system.
///
/// Holds the shared `NotifyFlag` for coalescing, and an inject closure
/// that sends a `StreamEvent` into the actor's mailbox when a truly new
/// notification needs to fire.
pub struct NotifySink {
flag: Arc<NotifyFlag>,
inject: Box<dyn Fn(StreamEvent) + Send + Sync>,
stream_id: crate::types::StreamId,
}
impl NotifySink {
pub fn new(
flag: Arc<NotifyFlag>,
stream_id: crate::types::StreamId,
inject: impl Fn(StreamEvent) + Send + Sync + 'static,
) -> Self {
NotifySink {
flag,
inject: Box::new(inject),
stream_id,
}
}
/// Notify the actor of a stream event. Coalesces duplicate notifications.
pub fn notify(&self, kind_flag: u8, kind: StreamEventKind) {
if self.flag.set(kind_flag) {
(self.inject)(StreamEvent {
stream_id: self.stream_id,
kind,
});
}
}
/// Convenience: notify data ready.
pub fn data_ready(&self) {
self.notify(DATA_READY, StreamEventKind::DataReady);
}
/// Convenience: notify write ready.
pub fn write_ready(&self) {
self.notify(WRITE_READY, StreamEventKind::WriteReady);
}
/// Convenience: notify closed.
pub fn closed(&self) {
self.notify(CLOSED, StreamEventKind::Closed);
}
/// Convenience: notify error.
pub fn error(&self) {
self.notify(ERROR, StreamEventKind::Error);
}
/// Access the shared flag for the actor side to clear bits.
pub fn flag(&self) -> &Arc<NotifyFlag> {
&self.flag
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn set_returns_true_first_time_false_on_duplicate() {
let flag = NotifyFlag::new();
assert!(flag.set(DATA_READY));
assert!(!flag.set(DATA_READY));
}
#[test]
fn clear_allows_re_notification() {
let flag = NotifyFlag::new();
assert!(flag.set(DATA_READY));
flag.clear(DATA_READY);
assert!(flag.set(DATA_READY));
}
#[test]
fn independent_flags_do_not_interfere() {
let flag = NotifyFlag::new();
assert!(flag.set(DATA_READY));
assert!(flag.set(WRITE_READY));
assert!(!flag.set(DATA_READY)); // still set
flag.clear(DATA_READY);
assert!(flag.is_set(WRITE_READY)); // unaffected
assert!(!flag.is_set(DATA_READY));
}
#[test]
fn read_shows_all_set_flags() {
let flag = NotifyFlag::new();
flag.set(DATA_READY);
flag.set(ERROR);
let bits = flag.read();
assert_eq!(bits & DATA_READY, DATA_READY);
assert_eq!(bits & ERROR, ERROR);
assert_eq!(bits & WRITE_READY, 0);
}
}

148
crates/streams/src/types.rs Normal file
View file

@ -0,0 +1,148 @@
use std::fmt;
use serde::{Deserialize, Serialize};
/// Unique identifier for a stream, generated randomly.
#[derive(Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
pub struct StreamId(pub [u8; 16]);
impl StreamId {
pub fn new_random() -> Self {
let mut bytes = [0u8; 16];
getrandom::getrandom(&mut bytes).expect("getrandom failed");
StreamId(bytes)
}
}
impl fmt::Debug for StreamId {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(f, "StreamId(")?;
for b in &self.0[..4] {
write!(f, "{b:02x}")?;
}
write!(f, "\u{2026})")
}
}
impl fmt::Display for StreamId {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
for b in &self.0[..8] {
write!(f, "{b:02x}")?;
}
write!(f, "\u{2026}")
}
}
/// Mode of a stream -- what kind of data flows through it.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
pub enum StreamMode {
BlobTransfer,
}
/// Configuration for a stream.
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct StreamConfig {
/// Number of parallel QUIC stripes for data transfer.
pub stripe_count: u8,
/// Maximum frame payload size in bytes.
pub frame_size: u32,
/// Opaque metadata attached to the stream negotiation.
pub metadata: Vec<u8>,
}
impl Default for StreamConfig {
fn default() -> Self {
StreamConfig {
stripe_count: 4,
frame_size: 256 * 1024, // 256 KB
metadata: Vec::new(),
}
}
}
/// Errors that can occur during stream operations.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum StreamError {
Closed,
BrokenPipe(String),
Disconnected,
BufferExhausted,
InvalidHeader(String),
ChunkVerificationFailed {
chunk_index: usize,
expected: [u8; 32],
actual: [u8; 32],
},
}
impl fmt::Display for StreamError {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
StreamError::Closed => write!(f, "stream closed"),
StreamError::BrokenPipe(msg) => write!(f, "broken pipe: {msg}"),
StreamError::Disconnected => write!(f, "disconnected"),
StreamError::BufferExhausted => write!(f, "buffer pool exhausted"),
StreamError::InvalidHeader(msg) => write!(f, "invalid header: {msg}"),
StreamError::ChunkVerificationFailed {
chunk_index,
expected,
actual,
} => {
write!(f, "chunk {chunk_index} verification failed: expected ")?;
for b in &expected[..4] {
write!(f, "{b:02x}")?;
}
write!(f, "\u{2026}, got ")?;
for b in &actual[..4] {
write!(f, "{b:02x}")?;
}
write!(f, "\u{2026}")
}
}
}
}
impl std::error::Error for StreamError {}
/// Token that allows resuming an interrupted stream transfer.
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
pub struct ResumeToken {
pub stream_id: StreamId,
pub mode: StreamMode,
pub chunks_completed: u64,
pub bytes_transferred: u64,
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn stream_id_random_is_unique() {
let a = StreamId::new_random();
let b = StreamId::new_random();
assert_ne!(a, b);
}
#[test]
fn stream_id_debug_shows_prefix() {
let id = StreamId([0xab, 0xcd, 0xef, 0x01, 0x23, 0x45, 0x67, 0x89, 0, 0, 0, 0, 0, 0, 0, 0]);
let dbg = format!("{id:?}");
assert_eq!(dbg, "StreamId(abcdef01\u{2026})");
}
#[test]
fn stream_id_display_shows_8_bytes() {
let id = StreamId([0xab, 0xcd, 0xef, 0x01, 0x23, 0x45, 0x67, 0x89, 0xaa, 0xbb, 0, 0, 0, 0, 0, 0]);
let disp = format!("{id}");
assert_eq!(disp, "abcdef0123456789\u{2026}");
}
#[test]
fn stream_config_defaults() {
let cfg = StreamConfig::default();
assert_eq!(cfg.stripe_count, 4);
assert_eq!(cfg.frame_size, 256 * 1024);
assert!(cfg.metadata.is_empty());
}
}

306
crates/streams/src/wire.rs Normal file
View file

@ -0,0 +1,306 @@
use crate::types::{StreamConfig, StreamError, StreamId, StreamMode};
/// Magic bytes identifying the swactor stream protocol.
pub const MAGIC: [u8; 2] = [0x53, 0x57];
/// Wire protocol version.
pub const VERSION: u8 = 0x01;
/// ALPN protocol identifier for QUIC negotiation.
pub const ALPN: &[u8] = b"swactor/stream/1";
/// Header sent at the beginning of a stream connection.
///
/// Wire layout:
/// ```text
/// [2B magic] [1B version] [16B stream_id] [1B mode] [1B stripe_count]
/// [4B frame_size] [4B metadata_len] [metadata_len B metadata]
/// ```
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct StreamHeader {
pub stream_id: StreamId,
pub mode: StreamMode,
pub config: StreamConfig,
}
/// Fixed portion of the header (before variable-length metadata).
const HEADER_FIXED_SIZE: usize = 2 + 1 + 16 + 1 + 1 + 4 + 4; // 29 bytes
/// Encode a stream header into bytes.
pub fn encode_header(header: &StreamHeader) -> Vec<u8> {
let meta_len = header.config.metadata.len() as u32;
let total = HEADER_FIXED_SIZE + header.config.metadata.len();
let mut buf = Vec::with_capacity(total);
// Magic + version
buf.extend_from_slice(&MAGIC);
buf.push(VERSION);
// Stream ID
buf.extend_from_slice(&header.stream_id.0);
// Mode
let mode_byte = match header.mode {
StreamMode::BlobTransfer => 0x01,
};
buf.push(mode_byte);
// Config: stripe_count, frame_size, metadata
buf.push(header.config.stripe_count);
buf.extend_from_slice(&header.config.frame_size.to_be_bytes());
buf.extend_from_slice(&meta_len.to_be_bytes());
buf.extend_from_slice(&header.config.metadata);
buf
}
/// Decode a stream header from bytes.
pub fn decode_header(data: &[u8]) -> Result<StreamHeader, StreamError> {
if data.len() < HEADER_FIXED_SIZE {
return Err(StreamError::InvalidHeader(format!(
"too short: {} bytes, need at least {HEADER_FIXED_SIZE}",
data.len()
)));
}
// Magic
if data[0..2] != MAGIC {
return Err(StreamError::InvalidHeader(format!(
"bad magic: [{:#04x}, {:#04x}]",
data[0], data[1]
)));
}
// Version
if data[2] != VERSION {
return Err(StreamError::InvalidHeader(format!(
"unsupported version: {}",
data[2]
)));
}
// Stream ID
let mut id_bytes = [0u8; 16];
id_bytes.copy_from_slice(&data[3..19]);
let stream_id = StreamId(id_bytes);
// Mode
let mode = match data[19] {
0x01 => StreamMode::BlobTransfer,
other => {
return Err(StreamError::InvalidHeader(format!(
"unknown mode: {other:#04x}"
)));
}
};
// Config
let stripe_count = data[20];
let frame_size = u32::from_be_bytes([data[21], data[22], data[23], data[24]]);
let meta_len = u32::from_be_bytes([data[25], data[26], data[27], data[28]]) as usize;
if data.len() < HEADER_FIXED_SIZE + meta_len {
return Err(StreamError::InvalidHeader(format!(
"metadata truncated: have {} bytes after fixed header, need {meta_len}",
data.len() - HEADER_FIXED_SIZE
)));
}
let metadata = data[HEADER_FIXED_SIZE..HEADER_FIXED_SIZE + meta_len].to_vec();
Ok(StreamHeader {
stream_id,
mode,
config: StreamConfig {
stripe_count,
frame_size,
metadata,
},
})
}
/// Encode a data frame: `[4B payload_len (big-endian)] [payload]`.
/// A payload length of 0 signals end-of-stripe.
pub fn encode_data_frame(payload: &[u8]) -> Vec<u8> {
let len = payload.len() as u32;
let mut buf = Vec::with_capacity(4 + payload.len());
buf.extend_from_slice(&len.to_be_bytes());
buf.extend_from_slice(payload);
buf
}
/// End-of-stripe sentinel: a frame with zero-length payload.
pub fn encode_end_of_stripe() -> [u8; 4] {
[0, 0, 0, 0]
}
/// Result of decoding a data frame from a byte slice.
#[derive(Debug, PartialEq, Eq)]
pub enum DataFrameDecoded<'a> {
/// A data frame with payload.
Data(&'a [u8]),
/// End-of-stripe sentinel.
EndOfStripe,
}
/// Decode a data frame from a byte slice.
/// Returns the decoded frame and the number of bytes consumed.
pub fn decode_data_frame(data: &[u8]) -> Result<(DataFrameDecoded<'_>, usize), StreamError> {
if data.len() < 4 {
return Err(StreamError::InvalidHeader(
"data frame too short for length prefix".into(),
));
}
let len = u32::from_be_bytes([data[0], data[1], data[2], data[3]]) as usize;
if len == 0 {
return Ok((DataFrameDecoded::EndOfStripe, 4));
}
if data.len() < 4 + len {
return Err(StreamError::InvalidHeader(format!(
"data frame truncated: need {len} bytes, have {}",
data.len() - 4
)));
}
Ok((DataFrameDecoded::Data(&data[4..4 + len]), 4 + len))
}
#[cfg(test)]
mod tests {
use super::*;
use proptest::prelude::*;
#[test]
fn header_round_trip_basic() {
let header = StreamHeader {
stream_id: StreamId([1; 16]),
mode: StreamMode::BlobTransfer,
config: StreamConfig {
stripe_count: 4,
frame_size: 262144,
metadata: vec![10, 20, 30],
},
};
let encoded = encode_header(&header);
let decoded = decode_header(&encoded).unwrap();
assert_eq!(header, decoded);
}
#[test]
fn header_rejects_bad_magic() {
let mut encoded = encode_header(&StreamHeader {
stream_id: StreamId([0; 16]),
mode: StreamMode::BlobTransfer,
config: StreamConfig::default(),
});
encoded[0] = 0xFF;
assert!(matches!(
decode_header(&encoded),
Err(StreamError::InvalidHeader(_))
));
}
#[test]
fn header_rejects_bad_version() {
let mut encoded = encode_header(&StreamHeader {
stream_id: StreamId([0; 16]),
mode: StreamMode::BlobTransfer,
config: StreamConfig::default(),
});
encoded[2] = 0xFF;
assert!(matches!(
decode_header(&encoded),
Err(StreamError::InvalidHeader(_))
));
}
#[test]
fn header_rejects_truncated() {
let encoded = encode_header(&StreamHeader {
stream_id: StreamId([0; 16]),
mode: StreamMode::BlobTransfer,
config: StreamConfig {
metadata: vec![1, 2, 3],
..StreamConfig::default()
},
});
// Chop off the metadata
let truncated = &encoded[..HEADER_FIXED_SIZE];
assert!(matches!(
decode_header(truncated),
Err(StreamError::InvalidHeader(_))
));
}
#[test]
fn data_frame_round_trip() {
let payload = b"hello world";
let encoded = encode_data_frame(payload);
let (decoded, consumed) = decode_data_frame(&encoded).unwrap();
assert_eq!(decoded, DataFrameDecoded::Data(b"hello world"));
assert_eq!(consumed, encoded.len());
}
#[test]
fn end_of_stripe_sentinel() {
let sentinel = encode_end_of_stripe();
assert_eq!(sentinel, [0, 0, 0, 0]);
let (decoded, consumed) = decode_data_frame(&sentinel).unwrap();
assert_eq!(decoded, DataFrameDecoded::EndOfStripe);
assert_eq!(consumed, 4);
}
#[test]
fn data_frame_rejects_truncated() {
let encoded = encode_data_frame(b"hello");
// Only give the length prefix + partial payload
let truncated = &encoded[..6];
assert!(matches!(
decode_data_frame(truncated),
Err(StreamError::InvalidHeader(_))
));
}
proptest! {
#[test]
fn header_round_trip_arbitrary(
id_bytes in prop::array::uniform16(any::<u8>()),
stripe_count in 1u8..=16,
frame_size in 1024u32..=1_048_576,
metadata in prop::collection::vec(any::<u8>(), 0..256),
) {
let header = StreamHeader {
stream_id: StreamId(id_bytes),
mode: StreamMode::BlobTransfer,
config: StreamConfig {
stripe_count,
frame_size,
metadata,
},
};
let encoded = encode_header(&header);
let decoded = decode_header(&encoded).unwrap();
prop_assert_eq!(header, decoded);
}
#[test]
fn data_frame_round_trip_arbitrary(
payload in prop::collection::vec(any::<u8>(), 0..262144),
) {
if payload.is_empty() {
// Empty payload encodes as end-of-stripe
let encoded = encode_data_frame(&payload);
let (decoded, _) = decode_data_frame(&encoded).unwrap();
prop_assert_eq!(decoded, DataFrameDecoded::EndOfStripe);
} else {
let encoded = encode_data_frame(&payload);
let (decoded, consumed) = decode_data_frame(&encoded).unwrap();
prop_assert_eq!(decoded, DataFrameDecoded::Data(&payload));
prop_assert_eq!(consumed, 4 + payload.len());
}
}
}
}

View file

@ -9,6 +9,7 @@ swactor-std = { path = "../std" }
dashboard = { path = "../dashboard", features = ["distribution"] }
swactor-datastore = { path = "../datastore" }
distribution = { path = "../distribution" }
swactor-streams = { path = "../streams" }
clap = { version = "4", features = ["derive"] }
ctrlc = "3"
iroh = { version = "0.96", optional = true }

View file

@ -713,6 +713,7 @@ fn run_iroh(
) {
use distribution::iroh_driver::{IrohDriver, IrohDriverConfig};
use iroh::{RelayMode, SecretKey};
use swactor_std::RuntimeNaming;
// Evaluate relay candidacy and determine embedded relay bind address
#[cfg(feature = "relay")]
@ -775,6 +776,7 @@ fn run_iroh(
relay_mode,
node: node_config,
peer_auth: Some(peer_auth.clone()),
additional_alpns: vec![swactor_streams::ALPN.to_vec()],
#[cfg(feature = "relay")]
embedded_relay_bind,
#[cfg(feature = "relay")]
@ -782,6 +784,26 @@ fn run_iroh(
};
let mut driver = IrohDriver::new(iroh_config).expect("failed to create iroh driver");
// Spawn StreamManager actor
let stream_mgr = swactor_streams::StreamManager::new(
driver.endpoint().clone(),
driver.tokio_handle(),
Arc::clone(&handle.runtime),
);
let stream_mgr_addr = handle
.runtime
.spawn(stream_mgr)
.expect("spawn StreamManager");
handle
.runtime
.register_name(swactor_streams::STREAM_MANAGER_NAME, stream_mgr_addr)
.expect("register StreamManager");
// Wire streams into the datastore
if let Some(group) = ds_group {
group.configure_streams(stream_mgr_addr, driver.tokio_handle());
}
eprintln!("Node {} started (iroh)", hex(&driver.node_id().0[..4]));
// Join seed if provided (accepts hex or base58)
@ -835,6 +857,25 @@ fn run_iroh(
driver.recv();
driver.tick();
// Forward incoming stream connections to StreamManager
for (node_id, conn) in driver.drain_other_connections() {
let rt_clone = Arc::clone(&handle.runtime);
let mgr_addr = stream_mgr_addr;
let node_bytes = node_id.0;
driver.tokio_handle().spawn(async move {
match swactor_streams::accept::handle_incoming(
node_bytes, conn, &rt_clone, mgr_addr,
)
.await
{
Ok(()) => {}
Err(e) => {
eprintln!("stream accept: failed to handle incoming: {e}");
}
}
});
}
// Drain discovered peers (dashboard "Add Peer") and auto-join them
{
let mut new_peers = Vec::new();

View file

@ -0,0 +1,475 @@
Swactor Stream Primitive -- Architectural Design
Context
Swactor has a distributed actor runtime with SWIM membership, Kademlia routing, and a content-addressed datastore. The current datastore
transfers blobs one chunk at a time via actor message round-trips -- extremely slow for large objects. Beyond the datastore, the system
needs a general-purpose bulk data transfer primitive for ML workloads (training data, weight checkpoints, gradient exchange), real-time
media (video/voice), and future game state replication.
The stream primitive is a high-performance data channel between nodes that actors negotiate and manage but do not sit on the data path of.
It should achieve top-class throughput by leveraging QUIC's multiplexed streams directly, bypassing the actor mailbox system for data
transfer.
Decisions made:
- Data path: StreamHandle with try_read/try_write; actors receive lightweight notification messages but data bypasses mailboxes
- Reliability: Reliable-only MVP; abstraction designed so unreliable (QUIC datagrams) can be added later
- Locality: Cross-node only; same-node actors use regular messages
- Crate: New crates/streams/ crate
---
1. Core Concept: Control Plane vs Data Plane
The fundamental architecture separates stream management (control plane) from data transfer (data plane).
Control plane -- actor messages through normal mailboxes:
- Stream negotiation (open, accept, reject)
- Parameter configuration (buffer sizes, chunk sizes, parallelism)
- Lifecycle events (established, closed, error)
- Progress/health notifications
Data plane -- bypasses actors entirely:
- Raw bytes flow through QUIC streams on the iroh transport
- Managed by async tasks on the IrohDriver's tokio runtime
- Actors interact via StreamHandle objects (try_read/try_write), not mailbox messages
- QUIC's built-in flow control handles backpressure
CONTROL PLANE (actor messages, mailboxes, worker ticks)
+--------+ StreamOpen +-----------+ StreamAccept +--------+
| Actor | -----------> | Stream | <------------- | Actor |
| (nodeA)| | Manager | |(nodeB) |
+--------+ +-----------+ +--------+
| | |
| StreamReady(handle) | | StreamReady(handle)
v v v
DATA PLANE (tokio tasks, QUIC streams, pre-allocated buffers)
+----------+ bytes +----------+ bytes +----------+
| SendHalf | =========> | QUIC | =========> | RecvHalf |
| (writer) | N parallel| streams | N parallel | (reader) |
+----------+ stripes +----------+ stripes +----------+
---
2. Stream Identity and Addressing
StreamId: A 16-byte random identifier, generated by the initiator during negotiation. Deliberately not an ActorAddress -- streams are not
actors, are not placed on workers, and are not discoverable via Kademlia. Keeping them out of the AddressMap avoids polluting the actor
routing hot path.
Full stream address: The tuple (NodeId, StreamId) is globally unique. A node can host many concurrent streams to many peers.
ALPN separation: Streams use a new protocol identifier swactor/stream/1, separate from the existing swactor/swim/1 used for membership.
This means:
- The iroh accept loop can distinguish stream connections from protocol messages immediately
- Stream data never blocks or interferes with cluster heartbeats
- Stream connections can have different tuning in the future
---
3. QUIC Stream Utilization
Parallel Stripes for Blob Transfers
For a single large transfer, multiple QUIC streams are opened in parallel on the same QUIC connection. Each stream carries a disjoint
range of the data. This is the stripe count, negotiated during handshake (default: 4).
Why multiple streams? A single QUIC stream can be limited by per-stream receive-window backpressure. Multiple streams allow the sender to
push data into QUIC's send buffer more aggressively, keeping the congestion window filled. Measurements from quinn/s2n-quic show 2-8
parallel streams can improve throughput 2-4x on high-bandwidth-delay-product links.
Stream layout per transfer:
- Stream 0 (control stream): Bidirectional QUIC stream. Carries the handshake header and out-of-band signals (completion, cancel, errors,
health). Stays open for the transfer's lifetime.
- Streams 1..N (data stripes): Unidirectional QUIC streams, each carrying sequential chunks. Stripe assignment is round-robin by chunk
index.
Connection Reuse
Multiple concurrent streams between the same two nodes share one QUIC connection (on the stream ALPN). QUIC multiplexing handles this
natively. The streams crate maintains a connection cache separate from the SWIM connection cache.
---
4. Wire Format
Two layers of wire format: the stream-level protocol (negotiation + data framing) and the blob transfer application protocol that rides on
top of it.
Control Stream Header (stream-level)
[2B magic: 0x53 0x57] -- "SW"
[1B version: 0x01]
[16B StreamId]
[1B mode] -- 0x01=BlobTransfer, 0x02=ContinuousStream (future)
[1B stripe_count] -- parallel data stripes (1-255)
[4B frame_size (BE u32)] -- maximum frame payload size in bytes
[4B metadata_len (BE u32)]
[N bytes metadata] -- negotiation payload (e.g., ContentHash for blob transfer)
Data Stripe Frame Format (stream-level)
[4B frame_len (BE u32)] -- 0 = end-of-stripe
[N bytes payload] -- raw data bytes
Deliberately minimal. No per-frame type tags (QUIC provides ordered reliable delivery), no per-frame checksums on the wire (QUIC provides
TLS integrity for transport), no per-frame metadata. Every byte of overhead on the hot path costs throughput.
BlobTransfer Application Protocol
For blob transfers, the `StreamConfig.metadata` carries the 32-byte `ContentHash` of the requested blob (so the serve side knows what to
send). The actual blob data flows over the StreamHandle with this application-level framing:
[4B manifest_json_length (u32 BE)]
[N bytes manifest JSON] -- serialized ObjectManifest
[chunk_0 raw bytes] -- size from manifest.chunks[0].size
[chunk_1 raw bytes] -- size from manifest.chunks[1].size
...
The receiver knows each chunk's expected size and blake3 hash from the manifest. Each chunk is verified individually on arrival:
blake3(chunk_data) == chunk_ref.hash. Corrupted chunks cause immediate transfer failure. This is implemented by the `send_blob` and
`recv_blob` async functions in `crates/datastore/src/blob_transfer.rs`.
Note: the blob transfer protocol sends chunks sequentially through the StreamHandle, which distributes data frames across stripes via the
data-plane layer's round-robin. Individual chunks are not split across stripes -- the stripe layer is transparent to the application
protocol.
---
5. Buffering Strategy
Pre-allocated Sliding Window (Zero Allocation on Hot Path)
The buffer pool is a sliding window, not a store. It never holds the entire blob in memory -- data flows through it like water through a
pipe. A 1TB transfer uses the same ~4MB of buffer memory as a 1MB transfer; only the duration changes.
All buffers are allocated during stream setup, not per-frame.
Sender pipeline (per stripe, double-buffered):
Source (disk/memory/computation)
→ [Buffer A: being filled from source]
→ [Buffer B: being written to QUIC]
→ Buffer B recycled → becomes the next Buffer A
→ repeat until source exhausted
One buffer is being filled while the other is being sent. When QUIC accepts Buffer B's bytes, it's recycled and refilled from the source.
The source can be disk I/O, a computation producing data, or anything that yields bytes.
Receiver pipeline (per stripe, double-buffered):
QUIC recv stream
→ [Buffer A: being filled from QUIC]
→ [Buffer B: being written to disk/consumed]
→ Buffer B recycled → becomes the next Buffer A
→ repeat until stream ends
The receiver reads from QUIC into one buffer while the previous buffer is being written to disk (for blob transfer) or consumed by the
application. Buffers are recycled, never allocated mid-transfer.
Backpressure chain (end-to-end):
Source read speed
→ fills sender buffer pool (2 per stripe)
→ QUIC congestion window
→ network bandwidth
→ QUIC receive window
→ fills receiver buffer pool (2 per stripe)
→ sink write speed (disk I/O, consumer processing)
If ANY link is slow, pressure propagates backward automatically.
No custom flow control needed -- QUIC handles it.
Sizing:
- Pool: stripe_count * 2 buffers per side = 8 buffers (at 4 stripes)
- Frame size: 256KB per frame (separate from the datastore's 1MB storage chunk size)
- Total memory per stream direction: 8 x 256KB = 2MB
- Total for a bidirectional transfer: ~4MB, regardless of blob size
- At ~1200 bytes per QUIC packet, 256KB = ~213 packets. Smaller blast radius on packet loss than 1MB, better interleaving across stripes,
aligns with OS page sizes.
TB-Scale Considerations
For very large transfers (100GB+ ML weights, TB-scale training data), additional design considerations apply:
Manifest streaming: At 1MB datastore chunks, a 1TB blob has ~1M chunks. Each ChunkRef is ~40 bytes, so the manifest is ~40MB. This is too
large for a single negotiation payload. The current implementation sends the manifest as a JSON preamble on the data stream itself (not in
the negotiation metadata). For TB-scale, the manifest could be streamed progressively instead of loaded all at once.
Per-chunk verification on arrival: The receiver verifies each chunk individually as it arrives: blake3(chunk_data) == chunk_ref.hash.
This is implemented in `recv_blob`. A corrupted chunk causes immediate transfer failure. This catches problems early rather than waiting
for full reassembly.
Progressive resume tokens: Resume tokens are emitted periodically (e.g., every 1000 chunks or every 256MB, whichever comes first), not
just on failure. The sender acknowledges receipt of resume tokens. On connection loss, the receiver persists the latest resume token, and
a new stream can resume from that point. For a 1TB transfer, a resume token with a 1M-bit BitVec is ~125KB -- cheap to exchange.
(Not yet implemented -- the ResumeToken type exists but nothing emits or consumes it.)
Disk I/O as the bottleneck: For TB-scale over fast networks (10Gbps+), disk I/O often becomes the bottleneck rather than the network. The
buffering strategy handles this naturally: when disk writes slow down, the receiver's buffer pool fills, QUIC backpressure kicks in, and
the sender slows to match. No special handling needed -- the pipeline self-regulates. For maximum disk throughput, the receiver can use
O_DIRECT or memory-mapped writes, but this is an implementation optimization, not an architectural decision.
Stripe count scaling: For very high bandwidth links, the default 4 stripes may not be enough to saturate the connection. The stripe count
should be configurable up to 16, negotiated during handshake based on the expected transfer size and link characteristics. A heuristic:
min(16, max(4, total_chunks / 1000)) -- more stripes for larger transfers.
---
6. StreamHandle -- The Actor-Facing API
The StreamHandle is a lightweight, Send (but not Clone) object that actors store in their state. It communicates with the data-plane tokio
tasks via channels internally.
Writer interface:
- try_write(data: &[u8]) -> Result<usize, StreamError> -- Non-blocking. Returns bytes accepted.
- flush() -- Signal that buffered data should be sent.
- close() -- Graceful close.
Reader interface:
- try_read(buf: &mut [u8]) -> Result<usize, StreamError> -- Non-blocking. Returns bytes read, 0 if none available.
- has_data() -> bool -- Check if data is available without consuming it.
BlobTransfer Async Functions
Rather than a wrapper object, blob transfer uses standalone async functions that run inside tokio tasks (spawned after StreamReady). These
functions loop over try_write/try_read with tokio::task::yield_now() for cooperative scheduling:
- send_blob(send, manifest, read_chunk) -- Writes the manifest preamble, then calls read_chunk(hash) for each chunk on-demand and writes
it. At most one chunk is in memory at a time on the sender side. The read_chunk callback allows any data source (BlobStore via Inbox,
in-memory, etc).
- recv_blob(recv) -- Reads the manifest, then reads and blake3-verifies each chunk. Returns ReceivedBlob { manifest, chunks }.
- poll_inbox(inbox, timeout) -- Async version of the bridge.rs poll_response pattern. Yields instead of thread::sleep.
These live in crates/datastore/src/blob_transfer.rs. The key insight: since actors can't await futures, the pattern is for the actor to
receive StreamReady, extract the StreamHandle via OneShot::take(), spawn a tokio task for the I/O loop, then stop itself. The tokio task
sends results back to other actors via runtime.send_to().
Why non-blocking? Actor handlers are synchronous (fn handle(&mut self, ctx: &Ctx, msg)). They cannot await futures. The try_read/try_write
pattern fits naturally. The tokio task bridge is the mechanism for async I/O.
---
7. Actor Integration: Negotiation Protocol
Opening a Stream (Initiator)
1. Actor sends a StreamOpen control message (through normal actor mailbox routing) to a StreamManager system actor. Contains: target_node:
NodeId, mode, metadata (e.g., ContentHash + manifest for blob transfer), reply_to: ActorAddress.
2. StreamManager validates the request, allocates a StreamId, and posts an async task to the tokio runtime that:
- Opens a QUIC connection to the target (stream ALPN)
- Opens the control bidirectional stream
- Sends the stream header
- Waits for accept/reject
3. On accept: StreamManager sends StreamReady { stream_id, handle: StreamHandle } back to the requesting actor.
Accepting a Stream (Receiver)
1. IrohDriver's accept loop receives connection on stream ALPN.
2. Reads control stream header, extracts StreamId + mode + metadata.
3. Sends StreamIncoming actor message to local StreamManager.
4. StreamManager routes to registered stream acceptors (actors that called StreamListen).
5. Matching actor receives StreamOffer { stream_id, mode, metadata } in its mailbox.
6. Actor sends StreamAccept or StreamReject back to StreamManager.
7. On accept: StreamManager allocates buffers, spawns data-plane tasks, sends StreamReady { handle } to the accepting actor.
Notification Model (Hybrid)
Stream data bypasses mailboxes, but actors need to know when data is available:
- The data-plane tasks inject lightweight StreamEvent sentinel messages into the owning actor's mailbox when state changes: DataReady,
WriteReady, Closed, Error.
- Coalescing: An atomic flag prevents duplicate notifications. Set when notification posted, cleared when actor handles it. A
high-throughput stream generates at most one DataReady per actor tick, not one per frame.
- The actor's handle_any dispatches StreamEvent via downcast (same mechanism as Down and ActorExited today -- no core trait changes
needed).
- Actors can also proactively call handle.try_read() from any handler, not just in response to DataReady.
---
8. The StreamManager Actor
A system actor spawned alongside the IrohDriver, registered under a well-known name. It is the bridge between the actor world and the
stream data plane.
Responsibilities:
- Registry of active streams: StreamId -> StreamState
- Handle StreamOpen / StreamAccept / StreamReject / StreamListen / StreamClose messages
- Spawn and supervise data-plane tokio tasks
- Monitor stream-holding actors; clean up streams when actors die
- Expose stream metrics (active streams, throughput, errors) for the dashboard
Communication with tokio runtime: Uses tokio::sync::mpsc and tokio::sync::oneshot channels. Posts commands to async tasks, receives
results as actor messages (via the Inbox pattern already used by DatastoreBridge).
---
9. Flow Control and Backpressure
Three layers, all leveraging what QUIC already provides:
1. QUIC-level: Per-stream and per-connection flow control (receive window, congestion window). This is the primary mechanism. Not
duplicated.
2. Buffer pool saturation: When receiver's pre-allocated buffer pool is full, the recv-side tokio task stops reading from QUIC. QUIC's
receive window closes, sender stops transmitting. Natural backpressure without custom protocol.
3. StreamHandle backpressure: try_write() returns 0 bytes accepted when the send buffer is full. The actor knows to back off or buffer
internally.
No custom flow control protocol. QUIC's congestion control (Cubic/BBR) is battle-tested. Adding application-level flow control would add
complexity and latency without benefit.
Cancellation
- Cooperative: StreamCancel signal on the control stream
- Abrupt: reset() on the QUIC streams
- Nuclear: close the QUIC connection (node shutdown only)
---
10. Error Handling and Recovery
Failure Modes
┌──────────────────┬──────────────────────┬────────────────────────────────────────────────┐
│ Failure │ Detection │ Behavior │
├──────────────────┼──────────────────────┼────────────────────────────────────────────────┤
│ Frame corruption │ QUIC TLS + checksums │ Automatic retransmit │
├──────────────────┼──────────────────────┼────────────────────────────────────────────────┤
│ Stream reset │ QUIC RST_STREAM │ StreamEvent::Error to owning actor │
├──────────────────┼──────────────────────┼────────────────────────────────────────────────┤
│ Connection loss │ QUIC timeout │ StreamEvent::Error on all streams to that node │
├──────────────────┼──────────────────────┼────────────────────────────────────────────────┤
│ Node death │ SWIM declares Dead │ StreamEvent::Error on all streams to that node │
├──────────────────┼──────────────────────┼────────────────────────────────────────────────┤
│ Owner actor dies │ Worker cleanup phase │ Stream closed, remote side notified │
└──────────────────┴──────────────────────┴────────────────────────────────────────────────┘
Resume Tokens for Blob Transfers (Not Yet Implemented)
For large transfers, the receiver periodically emits a ResumeToken on the control channel:
ResumeToken {
stream_id: StreamId,
manifest_hash: ContentHash,
chunks_received: BitVec, -- which chunks confirmed stored
}
On failure, the initiator can open a new stream with the ResumeToken. The sender skips confirmed chunks. This avoids retransmitting
terabytes when a checkpoint transfer fails near completion. Leverages the existing ObjectManifest/ChunkRef model.
The ResumeToken type is defined in crates/streams/src/types.rs but emission/consumption logic is deferred to a future stage.
---
11. Integration with Existing Datastore
The stream primitive adds a parallel transfer path to the datastore. The existing chunk-at-a-time TransferActor is preserved for
compatibility; the new stream path is used when stream support is configured.
Architecture:
DOWNLOAD SIDE: SERVE SIDE:
DatastoreNode StreamListener
│ DownloadViaStream (Incoming = StreamNotification)
│ ctx.spawn(StreamDownloader) │ on StreamOffer → HandleStreamOffer
▼ ▼
StreamDownloader DatastoreNode
(Incoming = StreamNotification) │ HandleStreamOffer
│ on_start: Open → StreamManager │ ctx.spawn(StreamServer)
│ StreamReady → tokio task: ▼
│ recv_blob → verify → write chunks StreamServer
│ send completion to DatastoreNode (Incoming = StreamNotification)
▼ │ on_start: Accept → StreamManager
DatastoreNode │ StreamReady → tokio task:
│ StreamDownloadComplete │ read manifest from BlobStore (Inbox)
│ persist metadata, reply to caller │ for each chunk: read from BlobStore,
│ write to stream (one at a time)
│ close stream
Design principles:
- No bridge/shim actors. Stream-facing actors use Incoming = StreamNotification directly.
- No preloading all chunks into memory. Chunks flow on-demand: storage → network.
- DatastoreNode stays simple (fire-and-forget coordination). The stream actors own the full I/O lifecycle.
- After receiving StreamReady, actors spawn tokio tasks for I/O. Tokio tasks communicate results back via runtime.send_to().
- StreamServer reads chunks on-demand — at most one chunk in memory at a time.
Flow:
1. DatastoreNode receives DownloadViaStream { content_hash, source_node, reply_to }.
2. Spawns a StreamDownloader actor, which sends Open to StreamManager with metadata = content_hash.0 (32 bytes).
3. Remote StreamListener receives StreamOffer, extracts ContentHash from metadata, sends HandleStreamOffer to DatastoreNode.
4. Remote DatastoreNode spawns a StreamServer actor, which sends Accept to StreamManager.
5. StreamServer receives StreamReady, spawns tokio task: reads manifest from BlobStore, then streams each chunk on-demand via send_blob.
6. StreamDownloader receives StreamReady, spawns tokio task: calls recv_blob, writes chunks to BlobStore (fire-and-forget), notifies
DatastoreNode of completion.
7. DatastoreNode creates ObjectEntry and persists via MetadataActor, which sends PutOk to the original caller.
This eliminates the round-trip-per-chunk bottleneck. A 1GB object with 1MB chunks currently requires 1,024 sequential round-trips. With
streams and 4 parallel stripes, the entire blob flows in a single burst limited only by network bandwidth.
---
12. Growth Path
Phase 1 (MVP): Reliable Ordered Blob Transfer — IMPLEMENTED
- StreamConfig with BlobTransfer mode only
- New ALPN swactor/stream/1 handler
- StreamOpen/StreamAccept handshake
- Parallel striped data transfer
- StreamHandle with try_read/try_write
- StreamManager actor
- Datastore integration (StreamListener, StreamDownloader, StreamServer actors)
- BlobTransfer application protocol (send_blob/recv_blob with per-chunk blake3 verification)
Remaining MVP work:
- Two-node integration test (real QUIC, full download flow)
- Resume tokens (ResumeToken type exists, emission/consumption not yet wired)
Phase 2: Continuous Streams
- ContinuousStream mode (no total size known)
- Single bidirectional QUIC stream (no striping)
- Variable-sized message frames
- Bounded ring buffer backpressure
- Enables: federated learning gradient streams, data pipelines
Phase 3: Unreliable Datagrams
- UnreliableSequenced reliability mode using QUIC datagrams
- Sequence-based frame dropping (latest-wins)
- Receiver-side jitter buffer
- Advisory StreamThrottle on control channel
- Enables: voice/video, game entity state replication
Phase 4: Priority and QoS
- priority: u8 in StreamConfig
- Priority-aware write scheduler across concurrent streams
- QUIC stream priority hints
- Per-stream health reporting and dashboard integration
- Enables: simultaneous video + checkpoint without starvation
Phase 5: Parallel Unordered Transfer
- ReliableUnordered mode: parallel QUIC streams per chunk, independent delivery
- Configurable parallelism
- Enables: gradient exchange for distributed ML (any chunk consumable independently)
---
13. Key Design Decisions Summary
┌───────────────────┬───────────────────────────────────────────────┬────────────────────────────────────────────────────────────┐
│ Decision │ Choice │ Rationale │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Data path │ StreamHandle bypass, tokio task bridge │ Max throughput; actors manage, don't bottleneck │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Stream identity │ 16-byte StreamId, not ActorAddress │ Streams are not actors; avoid polluting address space │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ ALPN │ Separate swactor/stream/1 │ Isolate from SWIM; no interference with heartbeats │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Parallel stripes │ 4 QUIC streams per blob transfer │ Saturate congestion window on high-BDP links │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Flow control │ QUIC built-in only │ Don't duplicate what the transport does well │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Wire format │ 4-byte length prefix, no type tags │ Minimal per-frame overhead │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Stream chunk size │ 256KB │ Better packet-loss resilience, page-aligned │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Buffering │ Pre-allocated slab per stream │ Zero allocation on hot path │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Blob protocol │ Async functions, not wrapper object │ Simpler; tokio tasks own the I/O loop after StreamReady │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Actor pattern │ Spawn actor → StreamReady → tokio task → stop │ Clean separation; actor negotiates, task does I/O │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Chunk I/O │ On-demand via Inbox polling (poll_inbox) │ At most 1 chunk in memory; no preloading │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Crate │ New crates/streams/ │ Optional, clean dependency graph │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ MVP scope │ Reliable ordered only │ Covers ML + datastore; unreliable added later │
├───────────────────┼───────────────────────────────────────────────┼────────────────────────────────────────────────────────────┤
│ Locality │ Cross-node only │ Focused scope; same-node uses regular messages │
└───────────────────┴───────────────────────────────────────────────┴────────────────────────────────────────────────────────────┘

View file

@ -0,0 +1,338 @@
# Swactor Streams -- Implementation Status
## What Was Built
Stages 1-4 are implemented. Stages 1-3 built the stream primitive in `crates/streams/` (the `swactor-streams` crate). Stage 4 connected streams to the datastore so blob transfers use QUIC streams instead of sequential actor-message round-trips. All 42 tests pass (37 streams + 5 blob_transfer).
### Stage 1: Types, Wire Format, and Buffer Pool
Pure Rust -- no tokio, no iroh, no network. Compiles and tests in isolation.
#### `src/types.rs`
Core domain types for the stream system.
- **`StreamId([u8; 16])`** -- 16-byte random identifier. `Copy`, `Hash`, `Eq`, `Serialize`/`Deserialize`. Custom `Debug` (4-byte hex prefix) and `Display` (8-byte hex prefix) following the codebase's ID conventions. Not an `ActorAddress` -- streams are not actors and don't pollute the address space.
- **`StreamMode`** -- enum with `BlobTransfer` variant. Extensible for future modes (continuous streams, datagrams).
- **`StreamConfig`** -- negotiation parameters: `stripe_count` (default 4), `frame_size` (default 256KB), `metadata` (opaque bytes for application-level negotiation payloads like ContentHash).
- **`StreamError`** -- error enum covering `Closed`, `BrokenPipe`, `Disconnected`, `BufferExhausted`, `InvalidHeader`, and `ChunkVerificationFailed` (with expected/actual hashes for diagnostics).
- **`ResumeToken`** -- checkpoint for resuming interrupted transfers, carrying stream identity and progress counters.
#### `src/wire.rs`
Binary wire format for stream headers and data frames. Pure functions, no I/O.
- **Constants**: `MAGIC: [0x53, 0x57]` ("SW"), `VERSION: 0x01`, `ALPN: b"swactor/stream/1"`.
- **`StreamHeader`** -- the negotiation header sent at connection establishment. Wire layout: `[2B magic][1B version][16B stream_id][1B mode][1B stripe_count][4B frame_size][4B metadata_len][N metadata]`.
- **`encode_header` / `decode_header`** -- round-trippable serialization with validation (magic, version, mode, truncation checks).
- **Data frame format**: `[4B payload_len (big-endian)][payload]`. Deliberately minimal -- no per-frame type tags or checksums (QUIC provides TLS integrity). A zero-length payload signals end-of-stripe.
- **`encode_data_frame` / `decode_data_frame` / `encode_end_of_stripe`** -- frame-level codec.
#### `src/buffer.rs`
Pre-allocated buffer pool for zero-allocation data transfer.
- **`FrameBuf`** -- a `Box<[u8]>` with read/write cursors. `write(&[u8]) -> usize` fills from the write cursor, `read(&mut [u8]) -> usize` drains from the read cursor. `reset()` zeroes only the cursors (not the data) for fast recycling. `load(&[u8])` replaces content directly.
- **`BufferPool`** -- a fixed-size pool backed by `crossbeam::ArrayQueue<FrameBuf>` (lock-free MPMC). `checkout() -> Option<FrameBuf>` and `checkin(buf)` enable concurrent use between actor threads and tokio tasks without locks. `Clone` shares the underlying `Arc` so send/recv sides reference the same pool.
### Stage 2: StreamHandle, Channels, and Data Plane
Introduces tokio channels and async tasks but NOT iroh. Data-plane tasks are generic over `AsyncRead`/`AsyncWrite`, fully testable with `tokio::io::DuplexStream`.
#### `src/channel.rs`
Typed channel messages that move `FrameBuf`s by ownership (zero-copy handoff).
- **`SendCommand`** -- `Data(FrameBuf)`, `Flush`, `Close`. Actor -> send task.
- **`SendEvent`** -- `WriteReady`, `Error(StreamError)`, `Closed`. Send task -> actor.
- **`RecvCommand`** -- `Consumed(FrameBuf)`, `Close`. Actor -> recv task.
- **`RecvEvent`** -- `Data(FrameBuf)`, `Error(StreamError)`, `Closed`. Recv task -> actor.
#### `src/notify.rs`
Notification coalescing to prevent flooding actor mailboxes.
- **`NotifyFlag`** -- `AtomicU8` bitflags (`DATA_READY`, `WRITE_READY`, `CLOSED`, `ERROR`). `set(kind) -> bool` returns true only if the bit was previously clear, signaling a new notification should be injected. `clear(kind)` is called by the actor after handling.
- **`StreamEvent`** / **`StreamEventKind`** -- the lightweight sentinel message injected into actor mailboxes. Carries `stream_id` and `kind` (DataReady, WriteReady, Closed, Error).
- **`NotifySink`** -- held by data-plane tasks. Combines the shared `NotifyFlag` with an inject closure. Convenience methods: `data_ready()`, `write_ready()`, `closed()`, `error()`.
#### `src/handle.rs`
The actor-facing API for reading and writing stream data.
- **`SendHalf`** -- owns `mpsc::Sender<SendCommand>`, `mpsc::Receiver<SendEvent>`, a `BufferPool` clone, and an active `FrameBuf`. `try_write(&[u8]) -> Result<usize>` fills the active buffer and sends full buffers via `try_send` (non-blocking). Returns 0 on backpressure. `flush()` sends partial buffers. `close()` flushes remaining data and sends the Close command.
- **`RecvHalf`** -- owns `mpsc::Receiver<RecvEvent>`, `mpsc::Sender<RecvCommand>`, a `BufferPool` clone, and an active `FrameBuf`. `try_read(&mut [u8]) -> Result<usize>` drains the active buffer then pulls new buffers from the channel. Returns 0 when no data is available. `has_data()` peeks without consuming.
- **`StreamHandle`** -- combines `SendHalf` and `RecvHalf`. `Send` but not `Clone` (the mpsc receivers are not cloneable).
- **`create_stream_handle(stream_id, config, pool_size, channel_capacity)`** -- factory that returns `(StreamHandle, DataPlaneEndpoints)`. The handle goes to the actor; the endpoints go to the data-plane tasks.
#### `src/data_plane.rs`
Async tasks that bridge `StreamHandle` channels to actual byte streams.
- **`send_stripe_task<W: AsyncWrite>`** -- reads `SendCommand`s from the channel, wire-encodes them as data frames, writes to the transport, returns consumed buffers to the pool, and optionally notifies the actor via `NotifySink`.
- **`recv_stripe_task<R: AsyncRead>`** -- reads wire-encoded frames from the transport, loads payloads into `FrameBuf`s from the pool, sends `RecvEvent::Data` to the actor channel. Handles end-of-stripe sentinel and connection closure.
- **`spawn_send_stripes` / `spawn_recv_stripes`** -- spawn a set of stripe tasks from a writer/reader factory. The recv spawner merges all stripe outputs into a single `mpsc::Receiver<RecvEvent>`.
Generic over `AsyncRead + AsyncWrite + Send + Unpin + 'static`, so tests use `tokio::io::DuplexStream` with no network stack.
### Stage 3: QUIC Integration and StreamManager Actor
Connects the data-plane tasks to real QUIC streams via iroh. Introduces the `StreamManager` system actor with full open/accept/reject lifecycle. Modifies `IrohDriver` for generic ALPN routing and bootstraps the StreamManager in `swactor-node`.
#### `src/messages.rs`
Protocol types for the stream control plane.
- **`OneShot<T>`** -- Clone-friendly wrapper for non-Clone data (`StreamHandle`, `Connection`). Uses `Arc<Mutex<Option<T>>>` internally. First `.take()` extracts the value; subsequent calls (including from clones) return `None`. This allows non-Clone payloads inside Clone message enums required by the actor system's `Message` trait.
- **`StreamManagerMsg`** -- 8-variant enum for messages sent TO the StreamManager actor:
- `Open { target_node, mode, config, reply_to }` -- Request a new stream to a remote node.
- `Accept { stream_id, reply_to }` -- Accept an offered incoming stream.
- `Reject { stream_id }` -- Reject an offered incoming stream.
- `Listen { mode, listener }` -- Register as a stream listener for a given mode.
- `Close { stream_id }` -- Close a stream.
- `IncomingConnection { node_id, stream_id, mode, config, conn }` -- Internal: from accept bridge to StreamManager.
- `OpenCompleted { stream_id, reply_to, result }` -- Internal: async open task completed.
- `AcceptCompleted { stream_id, reply_to, result }` -- Internal: async accept task completed.
- **`StreamNotification`** -- 4-variant enum for notifications sent FROM StreamManager TO user actors:
- `StreamReady { stream_id, handle }` -- Stream is ready for use (open or accept completed).
- `StreamOffer { stream_id, mode, metadata, from_node }` -- A remote node is offering a stream.
- `StreamClosed { stream_id, reason }` -- A stream was closed.
- `StreamFailed { stream_id, error }` -- A stream open/accept failed.
#### `src/connection.rs`
Async connection cache for stream QUIC connections, separate from SWIM connections.
- **`StreamConnectionCache`** -- `HashMap<[u8; 32], Connection>` with health-check-on-access. `get_or_connect()` checks `conn.close_reason().is_none()` before reuse and falls back to connecting via `endpoint.connect(key, ALPN)`. `prune_closed()` for bulk cleanup. Uses the stream ALPN (`swactor/stream/1`).
#### `src/manager.rs`
The core StreamManager system actor.
- **`StreamManager`** -- implements `ActorInterface<Incoming = StreamManagerMsg>`. Manages active streams, pending incoming offers, listener registrations, and a connection cache. Holds an `Endpoint`, `tokio::runtime::Handle`, and `Arc<Runtime>` for spawning async tasks and sending messages back to itself.
- **`STREAM_MANAGER_NAME`** -- well-known name `"StreamManager"` for the name registry.
- **Open flow**: Generates `StreamId`, spawns a tokio task that connects, sends header on a control bi-stream, waits for a 1-byte accept/reject response, then creates `StreamHandle` + data-plane tasks, and sends `OpenCompleted` back to the StreamManager. StreamManager then delivers `StreamNotification::StreamReady` to the requesting actor.
- **Incoming flow**: Accept bridge reads header, sends `IncomingConnection` to StreamManager. StreamManager stores as pending, notifies matching listeners with `StreamOffer`.
- **Accept flow**: Takes pending connection, spawns tokio task that sends accept byte, creates `StreamHandle` + data-plane tasks, sends `AcceptCompleted` back. StreamManager delivers `StreamReady` to accepting actor.
- **Reject flow**: Sends reject byte on a uni-stream, drops the connection.
- **Close flow**: Removes stream state; data-plane tasks terminate when channels drop.
- **`handle_down`**: Cleans up streams owned by dead actors and removes dead listeners.
- **Data-plane spawning**: For each stream direction, a single tokio task opens N uni-streams and round-robins data frames across them. Recv tasks accept incoming uni-streams and dispatch each to a `recv_stripe_task`.
#### `src/accept.rs`
Bridge between incoming QUIC connections and the StreamManager actor.
- **`spawn_accept_bridge`** -- spawns a tokio task that reads from a channel of `(node_id, Connection)` pairs, accepting the control bi-stream, reading the stream header via `read_to_end` + `decode_header`, and forwarding `StreamManagerMsg::IncomingConnection` to the StreamManager via `runtime.send_to()`.
- **`handle_incoming`** -- public async function for per-connection header processing. Can also be called directly from the main loop (used by `swactor-node`).
#### Modified: `crates/distribution/src/iroh_driver.rs`
Generic ALPN support to route stream connections separately from SWIM.
- **`IrohDriverConfig`**: Added `additional_alpns: Vec<Vec<u8>>` field. All existing call sites updated with `additional_alpns: vec![]`.
- **Endpoint creation**: ALPNs now include both SWIM and any additional ALPNs (`vec![ALPN.to_vec()] + additional_alpns`).
- **Accept loop**: After accepting a connection, checks `conn.alpn()`. SWIM ALPN routes to `accepted_conns` (existing behavior). All other ALPNs route to `other_accepted_conns` (new buffer).
- **New field**: `other_accepted_conns: Arc<Mutex<Vec<(NodeId, Connection)>>>`.
- **New methods**: `endpoint() -> &Endpoint` (for outbound stream connections), `drain_other_connections() -> Vec<(NodeId, Connection)>` (polled from main loop).
#### Modified: `crates/streams/src/types.rs`
- Added `Hash` derive to `StreamMode` (needed as `HashMap` key in listeners registry).
#### Modified: `crates/streams/src/lib.rs`
- Added module declarations and re-exports for `accept`, `connection`, `manager`, `messages`.
- Re-exports: `StreamConnectionCache`, `StreamManager`, `STREAM_MANAGER_NAME`, `OneShot`, `StreamManagerMsg`, `StreamNotification`.
#### Modified: `crates/streams/Cargo.toml`
- Added `swactor-std` dependency (for `CtxMonitoring`, `RuntimeNaming`).
- Added `io-util` feature to `tokio` (for `AsyncWriteExt::flush`).
#### Modified: `crates/swactor-node/src/main.rs`
Bootstrap integration in `run_iroh()`.
- Passes `swactor_streams::ALPN.to_vec()` in `IrohDriverConfig::additional_alpns`.
- After driver creation, spawns `StreamManager::new(endpoint, tokio_handle, runtime)` as a named actor under `"StreamManager"`.
- In the main loop, drains `driver.drain_other_connections()` and spawns `handle_incoming` tasks for each, forwarding to the StreamManager.
#### Modified: `crates/swactor-node/Cargo.toml`
- Added `swactor-streams` dependency.
#### Modified: `crates/distribution/tests/common/iroh.rs`, `crates/dashboard/src/bin/swactor-node.rs`
- Updated all `IrohDriverConfig` construction sites with `additional_alpns: vec![]`.
### Stage 4: Datastore Stream Integration
Connects the stream system to the datastore so blob transfers flow over QUIC streams instead of sequential per-chunk actor-message round-trips. A 1GB blob with 1MB chunks that previously required 1,024 round-trips now flows in a single burst.
#### `crates/datastore/src/blob_transfer.rs` (NEW)
Async functions for sending/receiving blobs over StreamHandle. Runs inside tokio tasks, NOT actor handlers.
- **`BlobTransferError`** -- enum: `IncompleteTransfer(String)`, `ChunkVerificationFailed { expected, actual }`, `InvalidManifest(String)`, `Storage(String)`.
- **`ReceivedBlob`** -- `{ manifest: ObjectManifest, chunks: Vec<(ContentHash, Vec<u8>)> }`.
- **`send_blob(send, manifest, read_chunk)`** -- generic over an async callback `F: Fn(ContentHash) -> Future<Output = Result<Vec<u8>>>`. Writes `[4B manifest_json_len][manifest JSON]` preamble, then for each chunk in the manifest calls `read_chunk(hash)` and writes the raw bytes. Chunks are NOT preloaded -- the callback reads one at a time.
- **`recv_blob(recv)`** -- reads manifest preamble, deserializes JSON, then reads + blake3-verifies each chunk against the manifest's `ChunkRef` entries. Returns `ReceivedBlob`.
- **`poll_inbox(inbox, timeout)`** -- async version of `bridge.rs:poll_response`. Yields (`tokio::task::yield_now`) instead of `thread::sleep`, polling the swactor `Inbox` until a message arrives or timeout.
- **Internal helpers**: `write_all` (loops `try_write` + `yield_now`), `read_exact` (loops `try_read` + `yield_now`).
Wire format:
```
[4B manifest_json_length (u32 BE)]
[N bytes manifest JSON]
[chunk_0 raw bytes] <- size from manifest.chunks[0].size
[chunk_1 raw bytes]
...
```
#### `crates/datastore/src/actors/stream_listener.rs` (NEW)
Listens for incoming BlobTransfer stream offers and routes them to DatastoreNode.
- **`StreamListener`** -- `Incoming = StreamNotification`. State: `datastore_node: ActorAddress`, `stream_manager: Option<ActorAddress>`.
- `on_start`: looks up `"StreamManager"` via `ctx.where_is()`, sends `StreamManagerMsg::Listen { mode: BlobTransfer }`.
- `handle(StreamOffer)`: extracts 32-byte ContentHash from `metadata`, sends `DatastoreNodeMsg::HandleStreamOffer` to DatastoreNode. Rejects if metadata != 32 bytes.
#### `crates/datastore/src/actors/stream_downloader.rs` (NEW)
Opens a stream to a remote node and downloads a blob.
- **`StreamDownloader`** -- `Incoming = StreamNotification`. Constructor takes: `content_hash`, `source_node`, `datastore_node`, `blob_store`, `reply_to`, `stream_manager`, `tokio_handle`, `runtime`.
- `on_start`: sends `StreamManagerMsg::Open { target_node, mode: BlobTransfer, config.metadata: content_hash.0.to_vec() }`.
- `handle(StreamReady)`: takes handle via `OneShot::take()`, spawns tokio task:
- Calls `recv_blob(&mut recv_half)`.
- Writes each chunk to BlobStore via `runtime.send_to(blob_store, WriteChunk)` (fire-and-forget).
- Writes manifest via `runtime.send_to(blob_store, WriteManifest)` (fire-and-forget).
- Sends `DatastoreNodeMsg::StreamDownloadComplete` to DatastoreNode.
- On error: sends `DatastoreNodeMsg::StreamDownloadFailed`.
- Actor calls `ctx.stop_self()` after spawning the task.
- `handle(StreamFailed)`: sends `StreamDownloadFailed`, stops self.
#### `crates/datastore/src/actors/stream_server.rs` (NEW)
Serves a blob to a requesting node over a stream, reading chunks on-demand.
- **`StreamServer`** -- `Incoming = StreamNotification`. Constructor takes: `stream_id`, `content_hash`, `blob_store`, `stream_manager`, `tokio_handle`, `runtime`.
- `on_start`: sends `StreamManagerMsg::Accept { stream_id }`.
- `handle(StreamReady)`: takes handle, spawns tokio task:
- Reads manifest from BlobStore via `runtime.new_inbox()` + `poll_inbox` (async Inbox polling).
- Calls `send_blob(&mut send_half, &manifest, |chunk_hash| { ... })` with a callback that reads each chunk on-demand from BlobStore via a fresh Inbox.
- At most one chunk is in memory at a time. Chunks flow directly from BlobStore to stream.
- Actor calls `ctx.stop_self()`.
- `handle(StreamFailed)`: stops self.
#### Modified: `crates/datastore/src/messages.rs`
Added 5 new variants to `DatastoreNodeMsg`:
- `DownloadViaStream { content_hash, source_node, reply_to }` -- triggers a stream download.
- `HandleStreamOffer { stream_id, content_hash, from_node, stream_manager }` -- routes incoming stream offers.
- `StreamDownloadComplete { content_hash, manifest, reply_to }` -- download succeeded; persist metadata.
- `StreamDownloadFailed { content_hash, reason, reply_to }` -- download failed; notify caller.
- `ConfigureStreams { stream_manager, tokio_handle, runtime }` -- late-binding stream support.
Changed from `#[derive(Debug, Clone)]` to `#[derive(Clone)]` with manual `Debug` impl (because `Arc<Runtime>` doesn't implement `Debug`).
#### Modified: `crates/datastore/src/actors/datastore_node.rs`
Added stream support fields and handlers to the coordinator actor.
- **New fields**: `runtime: Option<Arc<Runtime>>`, `tokio_handle: Option<tokio::runtime::Handle>`, `stream_manager: Option<ActorAddress>` -- all initialized to `None`.
- **`handle_configure_streams`**: stores runtime/tokio_handle/stream_manager.
- **`handle_download_via_stream`**: spawns `StreamDownloader`. Returns `TransferFailed` if streams not configured.
- **`handle_stream_offer`**: spawns `StreamServer`.
- **`handle_stream_download_complete`**: creates `ObjectEntry`, sends `MetadataMsg::PutObject` to metadata actor with the original `reply_to` for direct response routing.
- **`handle_stream_download_failed`**: sends `DatastoreResponse::TransferFailed` to `reply_to`.
#### Modified: `crates/datastore/src/actors/mod.rs`
Added module declarations for `stream_downloader`, `stream_listener`, `stream_server`.
#### Modified: `crates/datastore/src/lib.rs`
Added `pub mod blob_transfer`.
#### Modified: `crates/datastore/src/bridge.rs`
- Added `datastore_addr: ActorAddress` field to `DatastoreGroup` (stored during `spawn()`).
- Added `configure_streams(&self, stream_manager, tokio_handle)` method: sends `ConfigureStreams` to DatastoreNode, spawns and registers `StreamListener` under `"StreamListener"`.
#### Modified: `crates/datastore/Cargo.toml`
- Added `swactor-streams = { path = "../streams" }` and `tokio = { version = "1", features = ["sync", "rt", "time"] }` dependencies.
- Added dev-dependencies for testing: `swactor-streams`, `tokio` with `rt-multi-thread`, `macros`, `io-util`.
#### Modified: `crates/swactor-node/src/main.rs`
After StreamManager registration, wires stream support into the datastore:
```rust
if let Some(group) = ds_group {
group.configure_streams(stream_mgr_addr, driver.tokio_handle());
}
```
## Test Coverage
42 tests across all modules:
| Category | Tests | What they verify |
|----------|-------|------------------|
| `types` | 4 | StreamId uniqueness, Debug/Display formatting, StreamConfig defaults |
| `wire` | 8 | Header round-trip (basic + property-based), bad magic/version/truncation rejection, data frame round-trip (basic + property-based), end-of-stripe sentinel |
| `buffer` | 7 | FrameBuf write/read/reset/load, BufferPool checkout/checkin/exhaustion/recycling/sharing |
| `notify` | 4 | Set returns true first time / false on duplicate, clear re-enables, independent flags, read shows all bits |
| `data_plane` | 9 | Single-stripe end-to-end transfer, multi-chunk ordered delivery (20 chunks), 4-stripe round-robin (100 chunks), graceful close, notification coalescing, backpressure detection |
| `messages` | 5 | OneShot take-once semantics, clone sharing, debug format, StreamManagerMsg is Message, StreamNotification is Message |
| `blob_transfer` | 5 | Small blob round-trip (single chunk), multi-chunk round-trip (4MB / 256KB chunks / 16 chunks), corrupted chunk detection (blake3 verification), truncated stream detection, property-based arbitrary blob round-trips |
Property-based tests (via `proptest`) cover:
- Arbitrary stream headers (random IDs, stripe counts 1-16, frame sizes 1KB-1MB, metadata 0-256 bytes)
- Arbitrary data frame payloads (0-256KB)
- Arbitrary blob transfers (random data 1-64KB, chunk sizes 256B-8KB)
## Dependency Footprint
### `swactor-streams` crate
- `swactor` (core actor types, with `serde` feature)
- `swactor-std` (for `CtxMonitoring`, `RuntimeNaming`)
- `shared-types` (ContentHash)
- `distribution` (NodeId, iroh re-exports)
- `crossbeam-queue` (lock-free buffer pool -- already a workspace dep)
- `tokio` (mpsc channels, async I/O traits, io-util)
- `iroh` (QUIC transport, connections, endpoints)
- `blake3`, `serde`, `getrandom`
Dev dependencies: `proptest`, `tokio` (with rt-multi-thread, macros, test-util, io-util).
### `swactor-datastore` crate (Stage 4 additions)
- `swactor-streams` (stream primitives, messages, types)
- `tokio` (sync, rt, time -- for spawning async blob transfer tasks and `poll_inbox`)
Dev dependencies: `swactor-streams`, `tokio` (with rt-multi-thread, macros, io-util).
## Next Steps
### Remaining MVP Work
These items complete the minimum viable stream-based blob transfer:
1. **Two-node integration test** -- full open/accept/data-transfer/close cycle with real iroh endpoints and two `DatastoreGroup` instances. Verifies StreamListener receives offers, StreamServer serves blobs, StreamDownloader receives and persists them. This is the critical end-to-end validation that all the pieces work together over real QUIC.
2. **CtxStreams extension trait** (`crates/streams/src/ctx_ext.rs`) -- convenience methods on `Ctx`: `stream_open()`, `stream_listen()`, `stream_accept()`, `stream_reject()`, `stream_close()`. Looks up `"StreamManager"` via `where_is()` and wraps the message construction. Reduces boilerplate for any actor wanting to use streams.
3. **Resume tokens** -- checkpoint emission every N chunks or N bytes during `send_blob`/`recv_blob`. Stored in `ResumeToken` (already defined in `types.rs`). On reconnect, receiver sends its token in `StreamConfig.metadata` and sender seeks to the right chunk offset.
### Post-MVP Phases
- **Dashboard stream metrics** -- expose active streams, bytes transferred, and transfer rates through the existing dashboard infrastructure.
- **Continuous Streams** -- `ContinuousStream` mode for unbounded data (ML gradient streams, data pipelines). Single bidirectional QUIC stream, variable-sized frames, ring buffer backpressure.
- **Unreliable Datagrams** -- QUIC datagram-based mode for latency-sensitive data (voice/video, game state). Sequence-based dropping, jitter buffer.
- **Priority and QoS** -- per-stream priority, write scheduling across concurrent streams, QUIC stream priority hints.
- **Parallel Unordered Transfer** -- independent per-chunk QUIC streams for workloads where any chunk is consumable independently (distributed ML gradient exchange).