feat(peer)!: cut over to authenticated catalog sharing

Replace address-only trust and pushed peer state with installation identities,
SPKI-pinned QUIC, candidate-only discovery, and bounded responder-owned
protocol-8 pulls. The runtime now owns each network generation and all admitted
work through shutdown.

Add exact bundled content identities, reproducible manifest publishing,
capability-confined downloads, streaming BLAKE3 verification, quarantine and
retry, and crash-recoverable download and install transactions. Ship generated
fixture catalogs and fail closed when production manifests are absent.

The Tauri backend exposes durable sharing policy, redacted identity state, and
attempt-keyed transfer snapshots. Frontend consumption follows in the next
commit. Repository-wide test certificates and protocol-7 paths are removed.

BREAKING CHANGE: peers must use protocol 8 and exact catalog content artifacts;
protocol-7 frames and shared-certificate identities are no longer accepted.

Test Plan:
- `just test` -- passed on the completed stack (708 workspace tests)
- `just clippy` -- passed on the completed stack
- `just build` -- passed with fixture catalogs on the completed stack
- `just catalog-check-production` -- failed closed because the external
  production manifest corpus is absent
- `git diff --cached --check` -- passed
This commit is contained in:
ddidderr committed 2026-08-10 13:59:18 +02:00
1 parent 36c4785775
commit 60fd7ba0c2
128 files changed
+51759 -10784

No files matched your search

+15 -22
View File
@@ -3,10 +3,10 @@
use std::{collections::HashMap, net::SocketAddr, time::Duration};
use lanspread_mdns::{DaemonEvent, LANSPREAD_SERVICE_TYPE, MdnsAdvertiser, MdnsMonitor};
use lanspread_proto::PROTOCOL_VERSION;
use lanspread_proto::{PROTOCOL_VERSION, PeerId};
use tokio_util::sync::CancellationToken;
use crate::{context::PeerCtx, network::select_advertise_ip};
use crate::{context::PeerCtx, network::select_advertise_ip, scoped_blocking::scoped_blocking};
pub(super) async fn start_mdns_advertiser(
ctx: &PeerCtx,
@@ -21,26 +21,29 @@ pub(super) async fn start_mdns_advertiser(
*guard = Some(advertise_addr);
}
let peer_id = ctx.peer_id.as_ref().clone();
let peer_id = ctx.peer_id;
let hostname = gethostname::gethostname().to_string_lossy().into_owned();
let advertised_name = advertised_service_name(&hostname, &peer_id);
let advertised_name = advertised_service_name(&hostname, peer_id);
let monitor_name = advertised_name.clone();
let properties = advertisement_properties(ctx, &hostname, &peer_id).await;
let properties = advertisement_properties(&hostname, peer_id);
let mdns = tokio::task::spawn_blocking(move || {
let mdns = scoped_blocking(move || {
MdnsAdvertiser::new(
LANSPREAD_SERVICE_TYPE,
&advertised_name,
advertise_addr,
Some(properties),
)
})
.await??;
})?;
log::info!("Registered mDNS service with name: {monitor_name}");
Ok(mdns)
}
pub(super) fn close_mdns_advertiser(advertiser: MdnsAdvertiser) -> eyre::Result<()> {
scoped_blocking(move || advertiser.close())
}
pub(super) async fn monitor_mdns_events(monitor: MdnsMonitor, shutdown: CancellationToken) {
loop {
let event = tokio::select! {
@@ -67,7 +70,8 @@ pub(super) async fn monitor_mdns_events(monitor: MdnsMonitor, shutdown: Cancella
}
}
fn advertised_service_name(hostname: &str, peer_id: &str) -> String {
fn advertised_service_name(hostname: &str, peer_id: PeerId) -> String {
let peer_id = peer_id.to_string();
let max_hostname_len = 63usize.saturating_sub(peer_id.len() + 1);
let truncated_hostname = if hostname.len() > max_hostname_len {
hostname.get(..max_hostname_len).unwrap_or(hostname)
@@ -76,27 +80,16 @@ fn advertised_service_name(hostname: &str, peer_id: &str) -> String {
};
if truncated_hostname.is_empty() {
peer_id.to_string()
peer_id
} else {
format!("{truncated_hostname}-{peer_id}")
}
}
async fn advertisement_properties(
ctx: &PeerCtx,
hostname: &str,
peer_id: &str,
) -> HashMap<String, String> {
let (library_rev, library_digest) = {
let library_guard = ctx.local_library.read().await;
(library_guard.revision, library_guard.digest)
};
fn advertisement_properties(hostname: &str, peer_id: PeerId) -> HashMap<String, String> {
let mut properties = HashMap::new();
properties.insert("peer_id".to_string(), peer_id.to_string());
properties.insert("proto_ver".to_string(), PROTOCOL_VERSION.to_string());
properties.insert("library_rev".to_string(), library_rev.to_string());
properties.insert("library_digest".to_string(), library_digest.to_string());
if !hostname.is_empty() {
properties.insert("hostname".to_string(), hostname.to_string());
}
+509 -96
View File
@@ -1,31 +1,175 @@
//! mDNS peer discovery and discovery-time protocol negotiation.
use std::time::Duration;
use std::{
collections::{HashSet, VecDeque},
future::Future,
thread::JoinHandle,
time::Duration,
};
use eyre::WrapErr as _;
use futures::{StreamExt as _, stream::FuturesUnordered};
use lanspread_mdns::{LANSPREAD_SERVICE_TYPE, MdnsBrowser, MdnsService, MdnsServicePoll};
use lanspread_proto::PROTOCOL_VERSION;
use tokio::sync::mpsc::UnboundedSender;
use lanspread_proto::{PROTOCOL_VERSION, PeerEndpoint, PeerId};
use tokio::sync::{
mpsc::{self, UnboundedSender},
oneshot,
};
use tokio_util::sync::CancellationToken;
use crate::{
PeerEvent,
context::Ctx,
context::NetworkServiceCtx,
events,
peer_db::PeerId,
services::handshake::{HandshakeCtx, perform_handshake_with_peer},
services::{
handshake::{HandshakeCtx, ReservedCandidateHandshake},
state_sync::run_state_sync,
},
};
const MAX_ACTIVE_DISCOVERY_CANDIDATES: usize = 64;
const MAX_PENDING_MDNS_SERVICES: usize = 64;
const DISCOVERY_CANDIDATE_COOLDOWN: Duration = Duration::from_secs(5);
#[derive(Default)]
struct RecentCandidates {
entries: VecDeque<(PeerEndpoint, tokio::time::Instant)>,
}
impl RecentCandidates {
fn try_record(&mut self, candidate: PeerEndpoint, now: tokio::time::Instant) -> bool {
self.expire(now);
if self.entries.iter().any(|(endpoint, _)| {
endpoint.peer_id == candidate.peer_id || endpoint.addr == candidate.addr
}) || self.entries.len() >= MAX_ACTIVE_DISCOVERY_CANDIDATES
{
return false;
}
self.entries
.push_back((candidate, now + DISCOVERY_CANDIDATE_COOLDOWN));
true
}
fn expire(&mut self, now: tokio::time::Instant) {
while self
.entries
.front()
.is_some_and(|(_, deadline)| *deadline <= now)
{
self.entries.pop_front();
}
}
#[cfg(test)]
fn len(&self) -> usize {
self.entries.len()
}
}
struct MdnsPeerInfo {
addr: std::net::SocketAddr,
peer_id: Option<PeerId>,
proto_ver: Option<u32>,
library_rev: u64,
library_digest: u64,
}
struct ProtocolNegotiation {
endpoint: PeerEndpoint,
handshake: ReservedCandidateHandshake,
}
struct DiscoveryWorker {
shutdown: CancellationToken,
result_rx: Option<oneshot::Receiver<eyre::Result<()>>>,
thread: Option<JoinHandle<()>>,
}
impl DiscoveryWorker {
fn spawn(
service_type: String,
service_tx: mpsc::Sender<MdnsService>,
shutdown: CancellationToken,
) -> eyre::Result<Self> {
Self::spawn_with(shutdown, move |shutdown| {
run_mdns_browser(&service_type, &service_tx, &shutdown)
})
}
fn spawn_with(
shutdown: CancellationToken,
worker: impl FnOnce(CancellationToken) -> eyre::Result<()> + Send + 'static,
) -> eyre::Result<Self> {
let (result_tx, result_rx) = oneshot::channel();
let worker_shutdown = shutdown.clone();
let thread = std::thread::Builder::new()
.name("lanspread-mdns-browser".to_owned())
.spawn(move || {
let result = worker(worker_shutdown);
let _ = result_tx.send(result);
})
.wrap_err("failed to spawn mDNS discovery worker")?;
Ok(Self {
shutdown,
result_rx: Some(result_rx),
thread: Some(thread),
})
}
async fn wait_result(&mut self) -> eyre::Result<()> {
let result_rx = self
.result_rx
.as_mut()
.ok_or_else(|| eyre::eyre!("mDNS discovery result was already consumed"))?;
result_rx
.await
.map_err(|_| eyre::eyre!("mDNS discovery worker stopped without a result"))?
}
async fn shutdown_and_join(
mut self,
observed_result: Option<eyre::Result<()>>,
) -> eyre::Result<()> {
self.shutdown.cancel();
let result = if let Some(result) = observed_result {
self.result_rx.take();
result
} else {
let result_rx = self
.result_rx
.take()
.ok_or_else(|| eyre::eyre!("mDNS discovery result was already consumed"))?;
result_rx
.await
.map_err(|_| eyre::eyre!("mDNS discovery worker stopped without a result"))?
};
self.join_thread()?;
result
}
fn join_thread(&mut self) -> eyre::Result<()> {
let Some(thread) = self.thread.take() else {
return Ok(());
};
thread
.join()
.map_err(|_| eyre::eyre!("mDNS discovery worker panicked"))
}
}
impl Drop for DiscoveryWorker {
fn drop(&mut self) {
self.shutdown.cancel();
if let Err(err) = self.join_thread() {
log::error!("Failed to join mDNS discovery worker during cleanup: {err}");
}
}
}
/// Runs the peer discovery service using mDNS.
#[allow(clippy::too_many_lines)]
pub async fn run_peer_discovery(
tx_notify_ui: UnboundedSender<PeerEvent>,
ctx: Ctx,
ctx: NetworkServiceCtx,
) -> eyre::Result<()> {
log::info!("Starting peer discovery task");
@@ -34,38 +178,36 @@ pub async fn run_peer_discovery(
}
let service_type = LANSPREAD_SERVICE_TYPE.to_string();
let (service_tx, mut service_rx) = tokio::sync::mpsc::unbounded_channel();
let worker_shutdown = ctx.shutdown.clone();
let service_type_clone = service_type.clone();
let (service_tx, mut service_rx) = tokio::sync::mpsc::channel(MAX_PENDING_MDNS_SERVICES);
let service_shutdown = ctx.shutdown.child_token();
let mut worker = DiscoveryWorker::spawn(service_type, service_tx, service_shutdown.clone())?;
let mut negotiations = FuturesUnordered::new();
let mut active_candidates = HashSet::new();
let mut recent_candidates = RecentCandidates::default();
let mut mismatch_emitted = false;
let mut state_sync = Box::pin(run_state_sync(
ctx.clone(),
tx_notify_ui.clone(),
service_shutdown.clone(),
));
let mut observed_state_sync_result = None;
let worker_handle = ctx
.task_tracker
.spawn_blocking(move || -> eyre::Result<()> {
let browser = MdnsBrowser::new(&service_type_clone)?;
while !worker_shutdown.is_cancelled() {
match browser.next_service_timeout(None, Duration::from_millis(250))? {
MdnsServicePoll::Service(service) => {
if service_tx.send(service).is_err() {
log::debug!("Peer discovery consumer dropped; stopping worker");
break;
}
}
MdnsServicePoll::Timeout => {}
MdnsServicePoll::Closed => {
log::warn!("mDNS browser closed; stopping peer discovery worker");
break;
}
let observed_worker_result = loop {
tokio::select! {
() = ctx.shutdown.cancelled() => break None,
result = worker.wait_result() => break Some(result),
result = &mut state_sync => {
observed_state_sync_result = Some(result);
break None;
}
completed = negotiations.next(), if !negotiations.is_empty() => {
if let Some(endpoint) = completed {
active_candidates.remove(&endpoint);
}
}
Ok(())
});
loop {
tokio::select! {
() = ctx.shutdown.cancelled() => break,
service = service_rx.recv() => {
let Some(service) = service else {
break;
break None;
};
let info = parse_mdns_peer(&service);
@@ -74,30 +216,145 @@ pub async fn run_peer_discovery(
continue;
}
handle_discovered_peer(info, &ctx, &tx_notify_ui).await;
if info.proto_ver != Some(PROTOCOL_VERSION) {
if !mismatch_emitted {
events::send(
&tx_notify_ui,
PeerEvent::IncompatibleProtocolDetected {
observed: info.proto_ver,
expected: PROTOCOL_VERSION,
},
);
mismatch_emitted = true;
}
continue;
}
if let Some(endpoint) = validated_candidate_endpoint(&info) {
if !candidate_is_admissible(
&active_candidates,
&mut recent_candidates,
endpoint,
tokio::time::Instant::now(),
) {
log::warn!(
"Discovery candidate is cooling down or the recent-attempt limit is full; ignoring {}",
endpoint.addr
);
continue;
}
let handshake_ctx = HandshakeCtx::from_network(&ctx, &tx_notify_ui)
.with_cancellation(service_shutdown.clone());
let handshake = match ReservedCandidateHandshake::reserve(handshake_ctx, endpoint).await {
Ok(handshake) => handshake,
Err(error) => {
log::warn!("Failed to reserve discovery candidate {}: {error}", endpoint.addr);
continue;
}
};
active_candidates.insert(endpoint);
negotiations.push(run_protocol_negotiation(ProtocolNegotiation {
endpoint,
handshake,
}));
}
}
}
};
service_shutdown.cancel();
drain_service_children(negotiations).await;
let state_sync_exited_early = observed_state_sync_result.is_some();
let state_sync_result = match observed_state_sync_result {
Some(result) => result,
None => state_sync.await,
};
let worker_result = worker.shutdown_and_join(observed_worker_result).await;
if let Err(error) = state_sync_result {
return Err(error.wrap_err("peer state-sync service failed"));
}
if state_sync_exited_early && !ctx.shutdown.is_cancelled() {
eyre::bail!("peer state-sync service exited unexpectedly");
}
match worker_handle.await {
Ok(Ok(())) if ctx.shutdown.is_cancelled() => Ok(()),
Ok(Ok(())) => {
match worker_result {
Ok(()) if ctx.shutdown.is_cancelled() => Ok(()),
Ok(()) => {
eyre::bail!("mDNS discovery worker exited unexpectedly");
}
Ok(Err(err)) if ctx.shutdown.is_cancelled() => {
Err(err) if ctx.shutdown.is_cancelled() => {
log::debug!("Peer discovery worker stopped during shutdown: {err}");
Ok(())
}
Ok(Err(err)) => Err(err.wrap_err("peer discovery worker failed")),
Err(err) if ctx.shutdown.is_cancelled() => {
log::debug!("Peer discovery worker join ended during shutdown: {err}");
Ok(())
}
Err(err) => Err(eyre::eyre!("peer discovery worker join error: {err}")),
Err(err) => Err(err.wrap_err("peer discovery worker failed")),
}
}
async fn wait_for_local_peer_addr(ctx: &Ctx) -> bool {
fn candidate_conflicts(active: &HashSet<PeerEndpoint>, candidate: PeerEndpoint) -> bool {
active
.iter()
.any(|endpoint| endpoint.peer_id == candidate.peer_id || endpoint.addr == candidate.addr)
}
fn candidate_is_admissible(
active: &HashSet<PeerEndpoint>,
recent: &mut RecentCandidates,
candidate: PeerEndpoint,
now: tokio::time::Instant,
) -> bool {
active.len() < MAX_ACTIVE_DISCOVERY_CANDIDATES
&& !candidate_conflicts(active, candidate)
&& recent.try_record(candidate, now)
}
fn run_mdns_browser(
service_type: &str,
service_tx: &mpsc::Sender<MdnsService>,
shutdown: &CancellationToken,
) -> eyre::Result<()> {
let browser = MdnsBrowser::new(service_type)?;
let browse_result = (|| {
while !shutdown.is_cancelled() {
match browser.next_service_timeout(None, Duration::from_millis(250))? {
MdnsServicePoll::Service(service) => {
match service_tx.try_send(service) {
Ok(()) => {}
Err(mpsc::error::TrySendError::Full(_)) => {
// Repeated mDNS observations are hints only. Coalesce an
// overflow by dropping it rather than letting the native
// browser thread allocate without bound or block shutdown.
log::trace!(
"Coalescing mDNS observation while the bounded discovery queue is full"
);
}
Err(mpsc::error::TrySendError::Closed(_)) => {
log::debug!("Peer discovery consumer dropped; stopping worker");
break;
}
}
}
MdnsServicePoll::Timeout => {}
MdnsServicePoll::Closed => {
log::warn!("mDNS browser closed; stopping peer discovery worker");
break;
}
}
}
Ok(())
})();
let close_result = browser.close();
match (browse_result, close_result) {
(Ok(()), Ok(())) => Ok(()),
(Err(err), Ok(())) | (Ok(()), Err(err)) => Err(err),
(Err(browse_err), Err(close_err)) => Err(eyre::eyre!(
"mDNS browse failed: {browse_err:#}; browser shutdown also failed: {close_err:#}"
)),
}
}
async fn wait_for_local_peer_addr(ctx: &NetworkServiceCtx) -> bool {
loop {
if ctx.local_peer_addr.read().await.is_some() {
return true;
@@ -113,88 +370,244 @@ async fn wait_for_local_peer_addr(ctx: &Ctx) -> bool {
fn parse_mdns_peer(service: &MdnsService) -> MdnsPeerInfo {
MdnsPeerInfo {
addr: service.addr,
peer_id: service.properties.get("peer_id").cloned(),
peer_id: service
.properties
.get("peer_id")
.and_then(|value| value.parse::<PeerId>().ok()),
proto_ver: service
.properties
.get("proto_ver")
.and_then(|value| value.parse::<u32>().ok()),
library_rev: service
.properties
.get("library_rev")
.and_then(|value| value.parse::<u64>().ok())
.unwrap_or(0),
library_digest: service
.properties
.get("library_digest")
.and_then(|value| value.parse::<u64>().ok())
.unwrap_or(0),
}
}
async fn is_self_advertisement(info: &MdnsPeerInfo, ctx: &Ctx) -> bool {
async fn is_self_advertisement(info: &MdnsPeerInfo, ctx: &NetworkServiceCtx) -> bool {
let guard = ctx.local_peer_addr.read().await;
guard.as_ref().is_some_and(|addr| *addr == info.addr)
|| info
.peer_id
.as_ref()
.is_some_and(|peer_id| peer_id == ctx.peer_id.as_ref())
.is_some_and(|peer_id| *peer_id == ctx.peer_id)
}
async fn handle_discovered_peer(
info: MdnsPeerInfo,
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
fn validated_candidate_endpoint(info: &MdnsPeerInfo) -> Option<PeerEndpoint> {
if info.proto_ver != Some(PROTOCOL_VERSION) {
log::debug!(
"Ignoring peer at {} with protocol {:?}; expected {PROTOCOL_VERSION}",
info.addr,
info.proto_ver
);
return;
return None;
}
let Some(peer_id) = info.peer_id.clone() else {
let Some(peer_id) = info.peer_id else {
log::debug!(
"Ignoring current-protocol peer at {} without a peer_id TXT record",
info.addr
);
return;
return None;
};
let upsert = {
let mut db = ctx.peer_game_db.write().await;
let upsert = db.upsert_peer(peer_id.clone(), info.addr);
let features = db.peer_features(&peer_id);
if info.library_rev > 0 || info.library_digest > 0 {
db.update_peer_library(&peer_id, info.library_rev, info.library_digest, features);
}
upsert
Some(PeerEndpoint::new(peer_id, info.addr))
}
async fn run_protocol_negotiation(negotiation: ProtocolNegotiation) -> PeerEndpoint {
let endpoint = negotiation.endpoint;
let result = negotiation.handshake.run().await;
if let Err(err) = result {
log::warn!(
"Failed to negotiate protocol with peer {}: {err}",
endpoint.addr
);
}
endpoint
}
async fn drain_service_children<F>(mut children: FuturesUnordered<F>)
where
F: Future,
{
while children.next().await.is_some() {}
}
#[cfg(test)]
mod tests {
use std::{
collections::HashSet,
net::SocketAddr,
sync::{
Arc,
atomic::{AtomicBool, AtomicUsize, Ordering},
mpsc,
},
time::Duration,
};
if upsert.is_new {
log::info!("Discovered peer at: {}", info.addr);
events::emit_peer_discovered(&ctx.peer_game_db, tx_notify_ui, info.addr).await;
use futures::stream::FuturesUnordered;
use lanspread_proto::{PeerEndpoint, PeerId};
use tokio_util::sync::CancellationToken;
use super::{
DISCOVERY_CANDIDATE_COOLDOWN,
DiscoveryWorker,
MAX_ACTIVE_DISCOVERY_CANDIDATES,
RecentCandidates,
candidate_conflicts,
candidate_is_admissible,
drain_service_children,
};
fn endpoint(seed: u8, port: u16) -> PeerEndpoint {
PeerEndpoint::new(
PeerId::from_bytes([seed; 32]),
SocketAddr::from(([127, 0, 0, 1], port)),
)
}
if upsert.is_new || upsert.addr_changed {
spawn_protocol_negotiation(&info, ctx, tx_notify_ui, peer_id);
#[test]
fn active_candidate_keys_bound_both_claimed_identity_and_address() {
let active_endpoint = endpoint(1, 12001);
let mut active = HashSet::from([active_endpoint]);
assert!(candidate_conflicts(&active, endpoint(1, 12002)));
assert!(candidate_conflicts(&active, endpoint(2, 12001)));
assert!(!candidate_conflicts(&active, endpoint(2, 12002)));
active.remove(&active_endpoint);
assert!(!candidate_conflicts(&active, endpoint(1, 12002)));
assert!(!candidate_conflicts(&active, endpoint(2, 12001)));
}
}
fn spawn_protocol_negotiation(
info: &MdnsPeerInfo,
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
peer_id: PeerId,
) {
let peer_addr = info.addr;
let handshake_ctx = HandshakeCtx::from_ctx(ctx, tx_notify_ui);
#[test]
fn completed_candidate_attempts_are_bounded_and_rate_limited_by_identity_and_address() {
let now = tokio::time::Instant::now();
let first = endpoint(1, 12001);
let mut recent = RecentCandidates::default();
assert!(recent.try_record(first, now));
ctx.task_tracker.spawn(async move {
if let Err(err) = perform_handshake_with_peer(handshake_ctx, peer_addr, Some(peer_id)).await
{
log::warn!("Failed to negotiate protocol with peer {peer_addr}: {err}");
for port in 12002..12102 {
assert!(!recent.try_record(endpoint(1, port), now));
}
});
for seed in 2..=101 {
assert!(!recent.try_record(endpoint(seed, 12001), now));
}
for seed in 2..=u8::try_from(MAX_ACTIVE_DISCOVERY_CANDIDATES).expect("test bound fits u8") {
assert!(recent.try_record(endpoint(seed, 13000 + u16::from(seed)), now));
}
assert_eq!(recent.len(), MAX_ACTIVE_DISCOVERY_CANDIDATES);
assert!(!recent.try_record(endpoint(200, 14000), now));
let after_cooldown = now + DISCOVERY_CANDIDATE_COOLDOWN;
assert!(recent.try_record(endpoint(200, 14000), after_cooldown));
assert_eq!(recent.len(), 1);
}
#[test]
fn expiring_recent_entries_never_bypasses_the_independent_active_cap() {
let now = tokio::time::Instant::now();
let mut active = (0..MAX_ACTIVE_DISCOVERY_CANDIDATES)
.map(|index| {
endpoint(
u8::try_from(index + 1).expect("test index fits u8"),
15000 + u16::try_from(index).expect("test index fits u16"),
)
})
.collect::<HashSet<_>>();
let mut recent = RecentCandidates::default();
for candidate in &active {
assert!(recent.try_record(*candidate, now));
}
let after_cooldown = now + DISCOVERY_CANDIDATE_COOLDOWN;
let next = endpoint(100, 16000);
assert!(!candidate_is_admissible(
&active,
&mut recent,
next,
after_cooldown,
));
assert_eq!(active.len(), MAX_ACTIVE_DISCOVERY_CANDIDATES);
let completed = *active.iter().next().expect("active set should be nonempty");
active.remove(&completed);
assert!(candidate_is_admissible(
&active,
&mut recent,
next,
after_cooldown,
));
}
async fn cancellation_aware_child(
started: tokio::sync::mpsc::UnboundedSender<()>,
shutdown: CancellationToken,
completed: Arc<AtomicUsize>,
) {
started.send(()).expect("start receiver should remain open");
shutdown.cancelled().await;
tokio::task::yield_now().await;
completed.fetch_add(1, Ordering::SeqCst);
}
#[tokio::test]
async fn negotiation_batch_drains_started_children_on_shutdown() {
let shutdown = CancellationToken::new();
let (started_tx, mut started_rx) = tokio::sync::mpsc::unbounded_channel();
let completed = Arc::new(AtomicUsize::new(0));
let children = FuturesUnordered::new();
for _ in 0..2 {
children.push(cancellation_aware_child(
started_tx.clone(),
shutdown.clone(),
completed.clone(),
));
}
let control_shutdown = shutdown.clone();
let control = async move {
for _ in 0..2 {
started_rx
.recv()
.await
.expect("every negotiation should start");
}
control_shutdown.cancel();
};
tokio::time::timeout(Duration::from_secs(1), async {
tokio::join!(drain_service_children(children), control);
})
.await
.expect("shutdown should drain every negotiation");
assert_eq!(completed.load(Ordering::SeqCst), 2);
}
#[test]
fn dropping_discovery_worker_cancels_and_joins_its_thread() {
let shutdown = CancellationToken::new();
let (started_tx, started_rx) = mpsc::sync_channel(0);
let stopped = Arc::new(AtomicBool::new(false));
let worker_stopped = stopped.clone();
let worker = DiscoveryWorker::spawn_with(shutdown, move |shutdown| {
started_tx
.send(())
.expect("test should wait for worker startup");
while !shutdown.is_cancelled() {
std::thread::sleep(Duration::from_millis(1));
}
worker_stopped.store(true, Ordering::SeqCst);
Ok(())
})
.expect("discovery worker should spawn");
started_rx
.recv_timeout(Duration::from_secs(1))
.expect("discovery worker should start");
drop(worker);
assert!(
stopped.load(Ordering::SeqCst),
"worker Drop must not return before its thread stops"
);
}
}
+73 -545
View File
@@ -1,561 +1,89 @@
//! Protocol handshakes and library synchronization between peers.
//! Pinned responder pulls used for discovery and known-peer refreshes.
use std::{net::SocketAddr, sync::Arc};
use lanspread_db::db::GameCatalog;
use lanspread_proto::{Hello, HelloAck, PROTOCOL_VERSION};
use tokio::sync::{RwLock, mpsc::UnboundedSender};
use lanspread_proto::PeerEndpoint;
use crate::{
PeerEvent,
call_to_play::CallToPlayStore,
context::{Ctx, PeerCtx},
events,
identity::default_features,
library::{LocalLibraryState, build_library_snapshot},
network::exchange_hello,
peer_db::{PeerGameDB, PeerId, PeerUpsert},
peer_db::{PeerLivenessSnapshot, PeerNegotiationTicket, RefreshReservation},
services::remote_state::{self, RemoteStateCtx},
};
#[derive(Clone)]
pub(crate) struct HandshakeCtx {
peer_id: Arc<String>,
local_peer_addr: Arc<RwLock<Option<SocketAddr>>>,
local_library: Arc<RwLock<LocalLibraryState>>,
peer_game_db: Arc<RwLock<PeerGameDB>>,
tx_notify_ui: UnboundedSender<PeerEvent>,
catalog: Arc<RwLock<GameCatalog>>,
call_to_play: Arc<RwLock<CallToPlayStore>>,
pub(crate) type HandshakeCtx = RemoteStateCtx;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(crate) enum PeerRefreshOutcome {
Completed,
DeferredByCandidate,
Stale,
}
impl HandshakeCtx {
pub(crate) fn from_ctx(ctx: &Ctx, tx_notify_ui: &UnboundedSender<PeerEvent>) -> Self {
Self {
peer_id: ctx.peer_id.clone(),
local_peer_addr: ctx.local_peer_addr.clone(),
local_library: ctx.local_library.clone(),
peer_game_db: ctx.peer_game_db.clone(),
tx_notify_ui: tx_notify_ui.clone(),
catalog: ctx.catalog.clone(),
call_to_play: ctx.call_to_play.clone(),
}
/// A candidate pull reserved before it enters an async child scope.
///
/// The ticket is an RAII lease. Dropping this value or its running future
/// synchronously clears the candidate's peer/address claims.
pub(crate) struct ReservedCandidateHandshake {
ctx: HandshakeCtx,
endpoint: PeerEndpoint,
ticket: PeerNegotiationTicket,
}
impl ReservedCandidateHandshake {
pub(crate) async fn reserve(ctx: HandshakeCtx, endpoint: PeerEndpoint) -> eyre::Result<Self> {
let ticket = ctx.reserve_candidate(endpoint).await?;
Ok(Self {
ctx,
endpoint,
ticket,
})
}
pub(crate) fn from_peer_ctx(ctx: &PeerCtx) -> Self {
Self {
peer_id: ctx.peer_id.clone(),
local_peer_addr: ctx.local_peer_addr.clone(),
local_library: ctx.local_library.clone(),
peer_game_db: ctx.peer_game_db.clone(),
tx_notify_ui: ctx.tx_notify_ui.clone(),
catalog: ctx.catalog.clone(),
call_to_play: ctx.call_to_play.clone(),
pub(crate) async fn run(self) -> eyre::Result<()> {
let committed =
remote_state::pull_and_commit(&self.ctx, self.endpoint, self.ticket).await?;
if !committed {
log::debug!(
"Discarding stale authenticated candidate result for {} at {}",
self.endpoint.peer_id,
self.endpoint.addr
);
}
Ok(())
}
}
async fn required_listen_addr(
local_peer_addr: &Arc<RwLock<Option<SocketAddr>>>,
) -> eyre::Result<SocketAddr> {
(*local_peer_addr.read().await)
.ok_or_else(|| eyre::eyre!("local peer listener address is not ready"))
}
pub(super) async fn build_hello_ack(ctx: &PeerCtx) -> eyre::Result<HelloAck> {
let listen_addr = required_listen_addr(&ctx.local_peer_addr).await?;
let library = {
let library_guard = ctx.local_library.read().await;
build_library_snapshot(&library_guard)
pub(crate) async fn perform_peer_refresh(
ctx: HandshakeCtx,
snapshot: PeerLivenessSnapshot,
) -> eyre::Result<PeerRefreshOutcome> {
let ticket = match ctx.begin_peer_refresh(snapshot).await? {
RefreshReservation::Reserved(ticket) => ticket,
RefreshReservation::DeferredByCandidate => {
log::debug!(
"Deferring refresh for {} at {} behind a discovery candidate",
snapshot.endpoint.peer_id,
snapshot.endpoint.addr
);
return Ok(PeerRefreshOutcome::DeferredByCandidate);
}
RefreshReservation::Stale => {
log::debug!(
"Discarding refresh for stale peer endpoint {} at {}",
snapshot.endpoint.peer_id,
snapshot.endpoint.addr
);
return Ok(PeerRefreshOutcome::Stale);
}
};
let call_to_play_events = ctx.call_to_play.write().await.snapshot();
Ok(HelloAck {
peer_id: ctx.peer_id.as_ref().clone(),
proto_ver: PROTOCOL_VERSION,
listen_addr,
library,
features: default_features(),
call_to_play_events,
let committed = remote_state::pull_and_commit(&ctx, snapshot.endpoint, ticket).await?;
if !committed {
log::debug!(
"Discarding stale authenticated refresh result for {} at {}",
snapshot.endpoint.peer_id,
snapshot.endpoint.addr
);
}
Ok(if committed {
PeerRefreshOutcome::Completed
} else {
PeerRefreshOutcome::Stale
})
}
async fn build_hello_from_state(ctx: &HandshakeCtx) -> eyre::Result<Hello> {
let listen_addr = required_listen_addr(&ctx.local_peer_addr).await?;
let library = {
let library_guard = ctx.local_library.read().await;
build_library_snapshot(&library_guard)
};
let call_to_play_events = ctx.call_to_play.write().await.snapshot();
Ok(Hello {
peer_id: ctx.peer_id.as_ref().clone(),
proto_ver: PROTOCOL_VERSION,
listen_addr,
library,
features: default_features(),
call_to_play_events,
})
}
pub(crate) async fn perform_handshake_with_peer(
ctx: HandshakeCtx,
peer_addr: SocketAddr,
peer_id_hint: Option<PeerId>,
) -> eyre::Result<()> {
let hello = build_hello_from_state(&ctx).await?;
let ack = exchange_hello(peer_addr, hello).await?;
if ack.proto_ver != PROTOCOL_VERSION {
log::warn!(
"Peer {peer_addr} uses incompatible protocol {} (expected {PROTOCOL_VERSION})",
ack.proto_ver
);
return Ok(());
}
if ack.peer_id == *ctx.peer_id {
log::trace!("Ignoring handshake with self for {peer_addr}");
return Ok(());
}
if let Some(expected) = peer_id_hint.as_ref()
&& expected != &ack.peer_id
{
log::warn!(
"Peer {peer_addr} id mismatch: mDNS advertised {expected}, hello ack returned {}",
ack.peer_id
);
let _ = ctx.peer_game_db.write().await.remove_peer(expected);
}
merge_call_to_play_events(
&ctx.call_to_play,
&ctx.tx_notify_ui,
ack.call_to_play_events,
)
.await;
let record_addr = ack.listen_addr;
let upsert = record_remote_library(
&ctx.peer_game_db,
ack.peer_id.clone(),
record_addr,
ack.features.clone(),
ack.library,
)
.await;
after_peer_library_recorded(&ctx, upsert, record_addr).await;
events::emit_peer_game_list(&ctx.peer_game_db, &ctx.catalog, &ctx.tx_notify_ui).await;
Ok(())
}
pub(super) async fn accept_inbound_hello(
ctx: &PeerCtx,
transport_addr: Option<SocketAddr>,
hello: Hello,
) -> eyre::Result<HelloAck> {
if hello.peer_id == *ctx.peer_id {
log::trace!("Ignoring hello from self");
return build_hello_ack(ctx).await;
}
if hello.proto_ver != PROTOCOL_VERSION {
log::warn!(
"Incompatible protocol from {transport_addr:?}: {}",
hello.proto_ver
);
return build_hello_ack(ctx).await;
}
let addr = hello.listen_addr;
merge_call_to_play_events(
&ctx.call_to_play,
&ctx.tx_notify_ui,
hello.call_to_play_events,
)
.await;
let handshake_ctx = HandshakeCtx::from_peer_ctx(ctx);
let upsert = record_remote_library(
&ctx.peer_game_db,
hello.peer_id.clone(),
addr,
hello.features.clone(),
hello.library,
)
.await;
after_peer_library_recorded(&handshake_ctx, upsert, addr).await;
events::emit_peer_game_list(&ctx.peer_game_db, &ctx.catalog, &ctx.tx_notify_ui).await;
build_hello_ack(ctx).await
}
async fn merge_call_to_play_events(
store: &Arc<RwLock<CallToPlayStore>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
incoming: Vec<lanspread_proto::CallToPlayEvent>,
) {
match store.write().await.merge_batch(incoming) {
Ok(merged) => {
if merged.needs_history() {
log::warn!(
"Call to Play handshake omitted roots for calls: {}",
merged.missing_call_ids.join(", ")
);
}
if !merged.applied.is_empty() {
events::send(tx_notify_ui, PeerEvent::CallToPlayEvents(merged.applied));
}
}
Err(err) => {
log::warn!("Rejecting Call to Play handshake history: {err}");
}
}
}
pub(super) fn spawn_library_resync(
ctx: HandshakeCtx,
peer_addr: SocketAddr,
peer_id_hint: PeerId,
reason: &'static str,
) {
tokio::spawn(async move {
if let Err(err) = perform_handshake_with_peer(ctx, peer_addr, Some(peer_id_hint)).await {
log::warn!("Failed to {reason} library from {peer_addr}: {err}");
}
});
}
async fn record_remote_library(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
peer_id: PeerId,
peer_addr: SocketAddr,
features: Vec<String>,
snapshot: lanspread_proto::LibrarySnapshot,
) -> PeerUpsert {
let mut db = peer_game_db.write().await;
let upsert = db.upsert_peer(peer_id.clone(), peer_addr);
db.apply_library_snapshot(&peer_id, snapshot);
db.update_peer_features(&peer_id, features);
upsert
}
async fn after_peer_library_recorded(
ctx: &HandshakeCtx,
upsert: PeerUpsert,
peer_addr: SocketAddr,
) {
if upsert.is_new {
events::emit_peer_discovered(&ctx.peer_game_db, &ctx.tx_notify_ui, peer_addr).await;
}
}
#[cfg(test)]
mod tests {
use std::{
collections::HashMap,
net::SocketAddr,
path::{Path, PathBuf},
sync::Arc,
};
use lanspread_db::db::GameCatalog;
use lanspread_proto::{
Availability,
CallToPlayAction,
CallToPlayEvent,
GameSummary,
Hello,
LibrarySnapshot,
PROTOCOL_VERSION,
};
use tokio::sync::{RwLock, mpsc};
use tokio_util::{sync::CancellationToken, task::TaskTracker};
use super::{HandshakeCtx, accept_inbound_hello, build_hello_from_state};
use crate::{
PeerEvent,
UnpackFuture,
Unpacker,
context::Ctx,
library::LocalLibraryState,
peer_db::PeerGameDB,
};
struct NoopUnpacker;
impl Unpacker for NoopUnpacker {
fn unpack<'a>(&'a self, _archive: &'a Path, _dest: &'a Path) -> UnpackFuture<'a> {
Box::pin(async { Ok(()) })
}
}
fn addr(ip: [u8; 4], port: u16) -> SocketAddr {
SocketAddr::from((ip, port))
}
fn test_handshake_ctx(local_peer_addr: Option<SocketAddr>) -> HandshakeCtx {
let (tx_notify_ui, _rx_notify_ui) = mpsc::unbounded_channel();
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
HandshakeCtx {
peer_id: Arc::new("local-peer".to_string()),
local_peer_addr: Arc::new(RwLock::new(local_peer_addr)),
local_library: Arc::new(RwLock::new(LocalLibraryState::empty())),
peer_game_db,
tx_notify_ui,
catalog: Arc::new(RwLock::new(GameCatalog::empty())),
call_to_play: Arc::new(RwLock::new(crate::call_to_play::CallToPlayStore::default())),
}
}
fn summary(id: &str) -> GameSummary {
GameSummary {
id: id.to_string(),
name: id.to_string(),
size: 42,
downloaded: true,
installed: true,
eti_version: Some("20250101".to_string()),
manifest_hash: 7,
availability: Availability::Ready,
}
}
fn call_to_play_event() -> CallToPlayEvent {
CallToPlayEvent {
id: "event-1".to_string(),
call_id: "call-1".to_string(),
actor_id: "peer-alice".to_string(),
actor_name: "Alice".to_string(),
at: 8_000_000_000_000,
action: CallToPlayAction::Create {
game_id: "game".to_string(),
max_players: 4,
scheduled_for: None,
deadline: 8_000_000_060_000,
},
}
}
#[tokio::test]
async fn outbound_hello_requires_local_listener_addr() {
let ctx = test_handshake_ctx(None);
let err = build_hello_from_state(&ctx)
.await
.expect_err("hello without listener must fail");
assert_eq!(err.to_string(), "local peer listener address is not ready");
}
#[tokio::test]
async fn outbound_hello_carries_local_listener_addr() {
let advertised = addr([10, 66, 0, 2], 40000);
let ctx = test_handshake_ctx(Some(advertised));
let hello = build_hello_from_state(&ctx)
.await
.expect("listener address is present");
assert_eq!(hello.listen_addr, advertised);
}
#[tokio::test]
async fn outbound_hello_carries_local_library_snapshot() {
let ctx = test_handshake_ctx(Some(addr([10, 66, 0, 2], 40000)));
ctx.local_library
.write()
.await
.update_from_scan(HashMap::from([("game".to_string(), summary("game"))]), 7);
let hello = build_hello_from_state(&ctx)
.await
.expect("listener address is present");
assert_eq!(hello.library.library_rev, 7);
assert_eq!(hello.library.games.len(), 1);
assert_eq!(hello.library.games[0].id, "game");
}
#[tokio::test]
async fn outbound_hello_carries_call_to_play_history() {
let ctx = test_handshake_ctx(Some(addr([10, 66, 0, 2], 40000)));
ctx.call_to_play
.write()
.await
.merge_batch(vec![call_to_play_event()])
.expect("valid event should be merged");
let hello = build_hello_from_state(&ctx)
.await
.expect("listener address is present");
assert_eq!(hello.call_to_play_events, [call_to_play_event()]);
}
#[tokio::test]
async fn inbound_hello_applies_remote_library_snapshot() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let mut catalog = GameCatalog::empty();
catalog.insert("remote-game".to_string(), Some("20250101".to_string()));
let ctx = Ctx::new(
peer_game_db.clone(),
"local-peer".to_string(),
PathBuf::new(),
PathBuf::new(),
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
Arc::new(RwLock::new(catalog)),
Arc::new(RwLock::new(HashMap::new())),
Arc::new(crate::NoopStreamInstallProvider),
);
*ctx.local_peer_addr.write().await = Some(addr([127, 0, 0, 1], 4000));
let (tx_notify_ui, mut rx_notify_ui) = mpsc::unbounded_channel();
let peer_ctx = ctx.to_peer_ctx(tx_notify_ui);
let remote_addr = addr([127, 0, 0, 1], 5000);
let hello = Hello {
peer_id: "remote-peer".to_string(),
proto_ver: PROTOCOL_VERSION,
listen_addr: remote_addr,
library: LibrarySnapshot {
library_rev: 3,
games: vec![summary("remote-game")],
},
features: Vec::new(),
call_to_play_events: Vec::new(),
};
let ack = accept_inbound_hello(&peer_ctx, None, hello)
.await
.expect("current protocol hello should be accepted");
assert_eq!(ack.peer_id, "local-peer");
let snapshots = peer_game_db.read().await.peer_snapshots();
assert_eq!(snapshots.len(), 1);
assert_eq!(snapshots[0].addr, remote_addr);
assert_eq!(snapshots[0].game_count, 1);
assert_eq!(snapshots[0].games[0].id, "remote-game");
assert!(matches!(
rx_notify_ui
.recv()
.await
.expect("peer discovery event should be emitted"),
PeerEvent::PeerDiscovered(addr) if addr == remote_addr
));
assert!(matches!(
rx_notify_ui
.recv()
.await
.expect("peer count event should be emitted"),
PeerEvent::PeerCountUpdated(1)
));
let PeerEvent::ListGames(games) = rx_notify_ui
.recv()
.await
.expect("peer game list should be emitted")
else {
panic!("expected ListGames");
};
assert_eq!(games.len(), 1);
assert_eq!(games[0].id, "remote-game");
assert_eq!(games[0].peer_count, 1);
}
#[tokio::test]
async fn inbound_hello_from_self_is_ignored() {
// Protocol-level self-detection: a hello whose peer_id matches the local
// peer id must be acknowledged but never recorded as a peer. The CLI
// harness short-circuits self-connects with a string compare before any
// network call, so this guard (handshake.rs) is only exercised here.
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let ctx = Ctx::new(
peer_game_db.clone(),
"local-peer".to_string(),
PathBuf::new(),
PathBuf::new(),
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
Arc::new(RwLock::new(GameCatalog::empty())),
Arc::new(RwLock::new(HashMap::new())),
Arc::new(crate::NoopStreamInstallProvider),
);
*ctx.local_peer_addr.write().await = Some(addr([127, 0, 0, 1], 4000));
let (tx_notify_ui, mut rx_notify_ui) = mpsc::unbounded_channel();
let peer_ctx = ctx.to_peer_ctx(tx_notify_ui);
let self_hello = Hello {
peer_id: "local-peer".to_string(),
proto_ver: PROTOCOL_VERSION,
listen_addr: addr([127, 0, 0, 1], 4000),
library: LibrarySnapshot {
library_rev: 9,
games: vec![summary("self-game")],
},
features: Vec::new(),
call_to_play_events: Vec::new(),
};
let ack = accept_inbound_hello(&peer_ctx, None, self_hello)
.await
.expect("self hello should still be acknowledged");
assert_eq!(ack.peer_id, "local-peer");
assert!(
peer_game_db.read().await.peer_snapshots().is_empty(),
"self must never be recorded as a peer"
);
assert!(
rx_notify_ui.try_recv().is_err(),
"self hello must emit no peer discovery events"
);
}
#[tokio::test]
async fn inbound_hello_merges_call_to_play_history_once() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let ctx = Ctx::new(
peer_game_db,
"local-peer".to_string(),
PathBuf::new(),
PathBuf::new(),
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
Arc::new(RwLock::new(GameCatalog::empty())),
Arc::new(RwLock::new(HashMap::new())),
Arc::new(crate::NoopStreamInstallProvider),
);
*ctx.local_peer_addr.write().await = Some(addr([127, 0, 0, 1], 4000));
let (tx_notify_ui, mut rx_notify_ui) = mpsc::unbounded_channel();
let peer_ctx = ctx.to_peer_ctx(tx_notify_ui);
let remote_addr = addr([127, 0, 0, 1], 5000);
let hello = Hello {
peer_id: "remote-peer".to_string(),
proto_ver: PROTOCOL_VERSION,
listen_addr: remote_addr,
library: LibrarySnapshot {
library_rev: 0,
games: Vec::new(),
},
features: Vec::new(),
call_to_play_events: vec![call_to_play_event(), call_to_play_event()],
};
accept_inbound_hello(&peer_ctx, None, hello)
.await
.expect("current protocol hello should be accepted");
assert_eq!(
ctx.call_to_play.write().await.snapshot(),
[call_to_play_event()]
);
assert!(matches!(
rx_notify_ui.recv().await,
Some(PeerEvent::CallToPlayEvents(events)) if events == [call_to_play_event()]
));
}
}
+245 -297
View File
@@ -1,367 +1,315 @@
//! Peer liveness checks and stale-peer cleanup.
//! Pinned liveness checks and generation-conditional peer cleanup.
use std::{collections::HashMap, sync::Arc, time::Duration};
use std::{sync::Arc, time::Duration};
use lanspread_db::db::GameCatalog;
use tokio::sync::{RwLock, mpsc::UnboundedSender};
use tokio_util::{sync::CancellationToken, task::TaskTracker};
use futures::{StreamExt as _, stream};
use tokio::sync::mpsc::UnboundedSender;
use crate::{
PeerEvent,
config::{PEER_PING_IDLE_SECS, PEER_PING_INTERVAL_SECS, peer_stale_timeout},
context::OperationKind,
events,
content_quarantine::ContentQuarantine,
context::{NetworkServiceCtx, OperationKind},
network::ping_peer,
peer_db::{PeerGameDB, PeerId},
peer_db::PeerLivenessSnapshot,
scoped_blocking::scoped_blocking,
services::{HandshakeCtx, remote_state},
};
/// Runs the ping service to check peer liveness.
const MAX_CONCURRENT_PINGS: usize = 8;
/// Runs revision-bearing pinned liveness checks. The idle gate deliberately
/// uses `last_revision_check`; inbound and content traffic only affect
/// `last_seen`, which remains the stale-pruning clock.
pub async fn run_ping_service(
tx_notify_ui: UnboundedSender<PeerEvent>,
peer_game_db: Arc<RwLock<PeerGameDB>>,
catalog: Arc<RwLock<GameCatalog>>,
active_operations: Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: Arc<RwLock<HashMap<String, CancellationToken>>>,
shutdown: CancellationToken,
task_tracker: TaskTracker,
ctx: NetworkServiceCtx,
) -> eyre::Result<()> {
log::info!(
"Starting ping service ({PEER_PING_INTERVAL_SECS}s interval, \
{}s idle threshold, {}s timeout)",
{}s revision-check idle threshold, {}s stale timeout)",
PEER_PING_IDLE_SECS,
peer_stale_timeout().as_secs()
);
let mut interval = tokio::time::interval(Duration::from_secs(PEER_PING_INTERVAL_SECS));
let remote_ctx = HandshakeCtx::from_network(&ctx, &tx_notify_ui);
loop {
tokio::select! {
() = shutdown.cancelled() => return Ok(()),
biased;
() = ctx.shutdown.cancelled() => return Ok(()),
_ = interval.tick() => {}
}
ping_idle_peers(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
&shutdown,
&task_tracker,
)
.await;
let snapshots = ctx
.peer_game_db
.read()
.await
.peer_liveness_snapshot()
.into_iter()
.filter(revision_check_due)
.collect::<Vec<_>>();
let mut checks = stream::iter(snapshots.into_iter().map(|snapshot| {
let ctx = ctx.clone();
let remote_ctx = remote_ctx.clone();
async move { check_peer_liveness(&ctx, &remote_ctx, snapshot).await }
}))
.buffer_unordered(MAX_CONCURRENT_PINGS);
while checks.next().await.is_some() {}
prune_stale_peers(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
)
.await;
if ctx.shutdown.is_cancelled() {
return Ok(());
}
prune_stale_peers(&ctx, &remote_ctx).await?;
}
}
async fn ping_idle_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
shutdown: &CancellationToken,
task_tracker: &TaskTracker,
fn revision_check_due(snapshot: &PeerLivenessSnapshot) -> bool {
snapshot.last_revision_check.elapsed() >= Duration::from_secs(PEER_PING_IDLE_SECS)
}
async fn check_peer_liveness(
ctx: &NetworkServiceCtx,
remote_ctx: &HandshakeCtx,
snapshot: PeerLivenessSnapshot,
) {
let peer_snapshots = { peer_game_db.read().await.peer_liveness_snapshot() };
for (peer_id, peer_addr, last_seen) in peer_snapshots {
if last_seen.elapsed() < Duration::from_secs(PEER_PING_IDLE_SECS) {
continue;
}
let tx_notify_ui = tx_notify_ui.clone();
let peer_game_db = peer_game_db.clone();
let catalog = catalog.clone();
let active_operations = active_operations.clone();
let active_downloads = active_downloads.clone();
let shutdown = shutdown.clone();
task_tracker.spawn(async move {
let ping_result = tokio::select! {
() = shutdown.cancelled() => return,
result = ping_peer(peer_addr) => result,
};
match ping_result {
Ok(true) => {
peer_game_db.write().await.update_last_seen(&peer_id);
}
Ok(false) => {
log::warn!("Peer {peer_addr} failed ping check");
remove_peer_and_refresh(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
peer_id,
"Removed stale peer",
)
.await;
}
Err(err) => {
log::error!("Failed to ping peer {peer_addr}: {err}");
remove_peer_and_refresh(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
peer_id,
"Removed peer due to ping error",
)
.await;
match ping_peer(&ctx.quic, &snapshot.endpoint, &ctx.shutdown).await {
Ok(revisions) => {
match remote_state::observe_pinned_pong(remote_ctx, snapshot, revisions).await {
Ok(remote_state::PongCommit::NeedsPull { .. }) => {
if let Err(error) = ctx
.state_sync
.schedule_pinned_pull(snapshot.endpoint.peer_id, &ctx.shutdown)
.await
{
log::debug!(
"Could not schedule revision refresh for {}: {error:#}",
snapshot.endpoint.peer_id
);
}
}
Ok(
remote_state::PongCommit::Current | remote_state::PongCommit::StaleGeneration,
) => {}
Err(error) => log::error!(
"Failed to apply Pong from {}: {error:#}",
snapshot.endpoint.addr
),
}
});
}
Err(error) => {
log::warn!(
"Pinned ping to {} failed: {error:#}",
snapshot.endpoint.addr
);
// A capacity reset or transient transport failure is not topology
// authority. Leave both clocks unchanged and require repeated
// generation-current failures plus the stale timeout before the
// normal pruning path removes state.
ctx.peer_game_db
.write()
.await
.record_ping_failure_if_generation(snapshot);
}
}
}
async fn prune_stale_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
let stale_peers = {
peer_game_db
.read()
.await
.get_stale_peer_ids(peer_stale_timeout())
};
async fn prune_stale_peers(ctx: &NetworkServiceCtx, remote_ctx: &HandshakeCtx) -> eyre::Result<()> {
let stale = ctx
.peer_game_db
.read()
.await
.stale_peer_liveness_snapshots(peer_stale_timeout());
let mut removed_any = false;
for peer_id in stale_peers {
removed_any |= remove_peer(peer_game_db, tx_notify_ui, peer_id, "Removed stale peer").await;
for snapshot in stale {
removed_any |= remote_state::remove_peer_if_generation(remote_ctx, snapshot).await?;
}
if removed_any {
events::emit_peer_game_list(peer_game_db, catalog, tx_notify_ui).await;
handle_active_downloads_without_peers(
peer_game_db,
active_operations,
active_downloads,
tx_notify_ui,
)
.await;
handle_active_downloads_without_peers(ctx).await;
}
Ok(())
}
async fn remove_peer_and_refresh(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
peer_id: PeerId,
log_label: &str,
) {
if remove_peer(peer_game_db, tx_notify_ui, peer_id, log_label).await {
events::emit_peer_game_list(peer_game_db, catalog, tx_notify_ui).await;
handle_active_downloads_without_peers(
peer_game_db,
active_operations,
active_downloads,
tx_notify_ui,
)
.await;
}
}
async fn remove_peer(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
peer_id: PeerId,
log_label: &str,
) -> bool {
let removed_peer = { peer_game_db.write().await.remove_peer(&peer_id) };
let Some(peer) = removed_peer else {
return false;
};
log::info!("{log_label}: {}", peer.addr);
events::emit_peer_lost(peer_game_db, tx_notify_ui, peer.addr).await;
true
}
async fn handle_active_downloads_without_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
let active_ids = {
active_operations
.read()
.await
.iter()
.filter_map(|(id, kind)| (*kind == OperationKind::Downloading).then_some(id.clone()))
.collect::<Vec<_>>()
};
if active_ids.is_empty() {
return;
}
async fn handle_active_downloads_without_peers(ctx: &NetworkServiceCtx) {
let active_ids = ctx
.active_operations
.read()
.await
.iter()
.filter_map(|(id, kind)| (*kind == OperationKind::Downloading).then_some(id.clone()))
.collect::<Vec<_>>();
for id in active_ids {
if peers_still_have_game(peer_game_db, &id).await {
if eligible_source_remains(ctx, &id).await {
continue;
}
let cancelled = {
// An exclusive guard makes the check-and-cancel transition one-shot even when
// concurrent liveness checks remove the last peers at the same time.
let active_downloads = active_downloads.write().await;
let Some(cancel_token) = active_downloads.get(&id) else {
{
// Signalling is one-shot; the download task retains ownership of
// operation-map cleanup until every transfer worker has drained.
let active_downloads = ctx.active_downloads.read().await;
let Some(download) = active_downloads.get(&id) else {
continue;
};
if cancel_token.is_cancelled() {
false
} else {
cancel_token.cancel();
true
}
};
if !cancelled {
continue;
download.cancel_sources_exhausted();
}
events::send(
tx_notify_ui,
PeerEvent::DownloadGameFilesAllPeersGone { id: id.clone() },
);
}
}
async fn peers_still_have_game(peer_game_db: &Arc<RwLock<PeerGameDB>>, game_id: &str) -> bool {
let guard = peer_game_db.read().await;
!guard.peers_with_game(game_id).is_empty()
async fn eligible_source_remains(ctx: &NetworkServiceCtx, game_id: &str) -> bool {
let catalog = Arc::clone(&ctx.catalog);
let game_id_owned = game_id.to_owned();
let Ok(manifest) = scoped_blocking(move || catalog.manifest(&game_id_owned)) else {
return false;
};
let content_id = manifest.content_id();
let endpoints = ctx
.peer_game_db
.read()
.await
.peer_endpoints_with_content(game_id, content_id);
has_nonquarantined_source(&endpoints, content_id, &ctx.content_quarantine)
}
fn has_nonquarantined_source(
endpoints: &[lanspread_proto::PeerEndpoint],
content_id: lanspread_db::content_manifest::ContentId,
quarantine: &ContentQuarantine,
) -> bool {
endpoints
.iter()
.any(|endpoint| !quarantine.is_quarantined(endpoint, content_id))
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, sync::Arc};
use lanspread_db::content_manifest::ContentId;
use lanspread_proto::{LibrarySnapshot, PeerEndpoint, PeerId, RuntimeSessionId};
use tokio::sync::RwLock;
use tokio_util::sync::CancellationToken;
use super::*;
use crate::peer_db::PeerGameDB;
use super::handle_active_downloads_without_peers;
use crate::{PeerEvent, context::OperationKind, peer_db::PeerGameDB};
#[tokio::test]
async fn all_peers_gone_cancels_once_and_leaves_cleanup_to_download_owner() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let active_operations = Arc::new(RwLock::new(HashMap::from([(
"game".to_string(),
OperationKind::Downloading,
)])));
let cancel = CancellationToken::new();
let active_downloads = Arc::new(RwLock::new(HashMap::from([(
"game".to_string(),
cancel.clone(),
)])));
let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel();
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
)
.await;
assert!(cancel.is_cancelled());
assert_eq!(
active_operations.read().await.get("game"),
Some(&OperationKind::Downloading)
#[tokio::test(start_paused = true)]
async fn dropped_hint_is_revision_checked_within_configured_bound() {
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([1; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12001)),
);
assert!(active_downloads.read().await.contains_key("game"));
let mut interval = tokio::time::interval(Duration::from_secs(PEER_PING_INTERVAL_SECS));
interval.tick().await;
tokio::time::advance(Duration::from_millis(1)).await;
let event = rx.recv().await.expect("peers-gone event should be emitted");
assert!(matches!(
event,
PeerEvent::DownloadGameFilesAllPeersGone { id } if id == "game"
));
let mut db = PeerGameDB::new();
let ticket = db
.begin_candidate_negotiation(endpoint)
.expect("candidate should reserve");
db.commit_authenticated_snapshot(
endpoint,
ticket,
RuntimeSessionId::from_bytes([1; 16]),
Some(LibrarySnapshot {
revision: 0,
games: Vec::new(),
}),
)
.expect("peer should commit")
.expect("ticket should remain current");
let start = tokio::time::Instant::now();
interval.tick().await;
let mut snapshot = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should remain");
snapshot.last_seen = tokio::time::Instant::now();
assert!(
rx.try_recv().is_err(),
"cancellation must not emit a premature active-operation snapshot"
!revision_check_due(&snapshot),
"a tick just before the idle threshold should not ping"
);
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
)
.await;
interval.tick().await;
let mut snapshot = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should remain");
// Simulate repeated inbound/content activity. It may refresh liveness,
// but must not touch the independently captured revision-check clock.
snapshot.last_seen = tokio::time::Instant::now();
assert!(revision_check_due(&snapshot));
assert!(
rx.try_recv().is_err(),
"an already-cancelled download must not emit peers-gone twice"
start.elapsed() <= Duration::from_secs(PEER_PING_IDLE_SECS + PEER_PING_INTERVAL_SECS)
);
}
#[tokio::test]
async fn all_peers_gone_cancels_multiple_downloads_without_releasing_admission() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let first_cancel = CancellationToken::new();
let second_cancel = CancellationToken::new();
let active_operations = Arc::new(RwLock::new(HashMap::from([
("first".to_string(), OperationKind::Downloading),
("second".to_string(), OperationKind::Downloading),
("installing".to_string(), OperationKind::Installing),
])));
let active_downloads = Arc::new(RwLock::new(HashMap::from([
("first".to_string(), first_cancel.clone()),
("second".to_string(), second_cancel.clone()),
])));
let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel();
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
#[tokio::test(start_paused = true)]
async fn one_transient_ping_failure_never_removes_and_success_resets_failure_history() {
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([9; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12009)),
);
let mut db = PeerGameDB::new();
let ticket = db
.begin_candidate_negotiation(endpoint)
.expect("candidate should reserve");
db.commit_authenticated_snapshot(
endpoint,
ticket,
RuntimeSessionId::from_bytes([9; 16]),
Some(LibrarySnapshot {
revision: 0,
games: Vec::new(),
}),
)
.await;
.expect("peer should commit")
.expect("ticket should remain current");
let probe = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should exist");
assert!(first_cancel.is_cancelled());
assert!(second_cancel.is_cancelled());
let operations = active_operations.read().await;
assert_eq!(operations.get("first"), Some(&OperationKind::Downloading));
assert_eq!(operations.get("second"), Some(&OperationKind::Downloading));
assert_eq!(
operations.get("installing"),
Some(&OperationKind::Installing)
);
drop(operations);
let downloads = active_downloads.read().await;
assert!(downloads.contains_key("first"));
assert!(downloads.contains_key("second"));
drop(downloads);
let mut cancelled_ids = Vec::new();
for _ in 0..2 {
let event = rx.recv().await.expect("peers-gone event should be emitted");
let PeerEvent::DownloadGameFilesAllPeersGone { id } = event else {
panic!("expected peers-gone event");
};
cancelled_ids.push(id);
}
cancelled_ids.sort();
assert_eq!(cancelled_ids, vec!["first", "second"]);
tokio::time::advance(peer_stale_timeout() + Duration::from_secs(1)).await;
assert!(db.record_ping_failure_if_generation(probe));
assert!(
rx.try_recv().is_err(),
"multiple cancellations must not emit an active-operation snapshot"
db.stale_peer_liveness_snapshots(peer_stale_timeout())
.is_empty(),
"one transport failure must preserve authenticated state"
);
assert!(matches!(
db.observe_pong_if_generation(
probe,
lanspread_proto::PeerRevisions {
runtime_session_id: RuntimeSessionId::from_bytes([9; 16]),
library_revision: 0,
call_to_play_revision: 0,
},
),
crate::peer_db::PongObservation::Current
| crate::peer_db::PongObservation::RevisionMismatch
));
let refreshed = db
.peer_liveness_for(&endpoint.peer_id)
.expect("successful Pong should preserve peer");
assert_eq!(refreshed.consecutive_ping_failures, 0);
}
#[test]
fn wrong_content_and_quarantined_sources_do_not_keep_download_alive() {
let expected = ContentId::from_bytes([7; 32]);
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([2; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12002)),
);
let quarantine = ContentQuarantine::default();
// Exact-content filtering happens before this seam, so a wrong-content
// peer produces the empty eligible endpoint set.
assert!(!has_nonquarantined_source(&[], expected, &quarantine));
assert!(has_nonquarantined_source(
&[endpoint],
expected,
&quarantine
));
quarantine.record_integrity_failure(&endpoint, expected);
assert!(!has_nonquarantined_source(
&[endpoint],
expected,
&quarantine
));
}
}
@@ -1,28 +1,63 @@
//! Local game directory monitor.
use std::{
collections::HashSet,
path::{Component, Path, PathBuf},
any::Any,
collections::{BTreeMap, BTreeSet, HashSet},
ffi::OsString,
fs,
io,
panic::AssertUnwindSafe,
path::{Path, PathBuf},
sync::Arc,
time::Duration,
time::SystemTime,
};
use notify::{Config, Event, EventKind, RecommendedWatcher, RecursiveMode, Watcher};
use tokio::sync::{RwLock, mpsc::UnboundedSender};
use futures::FutureExt;
use tokio::{
sync::{RwLock, mpsc::UnboundedSender},
task::{JoinError, JoinSet},
time::{Instant, MissedTickBehavior},
};
use crate::{
PeerEvent,
config::LOCAL_GAME_FALLBACK_SCAN_SECS,
config::{LOCAL_GAME_FALLBACK_SCAN_SECS, LOCAL_GAME_POLL_INTERVAL_SECS},
context::Ctx,
game_paths::{is_download_protected_root_name, is_ignored_games_root_name},
handlers::update_and_announce_games,
local_games::{rescan_local_game, scan_local_library},
local_games::{
rescan_local_game_with_recovery_failures,
scan_local_library_with_recovery_failures,
},
scoped_blocking::scoped_blocking,
};
struct WatchState {
watcher: RecommendedWatcher,
#[derive(Debug, Eq, PartialEq)]
struct PollSnapshot {
game_dir: PathBuf,
watched: HashSet<PathBuf>,
games: BTreeMap<String, GameRootSnapshot>,
}
#[derive(Debug, Eq, PartialEq)]
enum GameRootSnapshot {
Readable(BTreeMap<OsString, EntryFingerprint>),
NonDirectory(EntryFingerprint),
Unreadable,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
struct EntryFingerprint {
kind: EntryKind,
len: u64,
modified: Option<SystemTime>,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum EntryKind {
File,
Directory,
LinkOrReparse,
Other,
}
#[derive(Clone, Default)]
@@ -36,202 +71,284 @@ pub async fn run_local_game_monitor(
tx_notify_ui: UnboundedSender<PeerEvent>,
ctx: Ctx,
) -> eyre::Result<()> {
log::info!("Starting notify-based local game directory monitor");
log::info!("Starting polling-based local game directory monitor");
let (watch_tx, mut watch_rx) = tokio::sync::mpsc::unbounded_channel::<notify::Result<Event>>();
let mut watch_state = build_watch_state(&ctx, watch_tx.clone()).await;
let mut snapshot = initial_poll_snapshot(&ctx).await;
let gate = RescanGate::default();
let mut fallback_interval =
tokio::time::interval(Duration::from_secs(LOCAL_GAME_FALLBACK_SCAN_SECS));
let mut rescans = JoinSet::new();
let now = Instant::now();
let poll_period = std::time::Duration::from_secs(LOCAL_GAME_POLL_INTERVAL_SECS);
let fallback_period = std::time::Duration::from_secs(LOCAL_GAME_FALLBACK_SCAN_SECS);
let mut poll_interval = tokio::time::interval_at(now + poll_period, poll_period);
let mut fallback_interval = tokio::time::interval_at(now + fallback_period, fallback_period);
poll_interval.set_missed_tick_behavior(MissedTickBehavior::Skip);
fallback_interval.set_missed_tick_behavior(MissedTickBehavior::Skip);
loop {
tokio::select! {
() = ctx.shutdown.cancelled() => return Ok(()),
_ = fallback_interval.tick() => {
run_fallback_scan(&ctx, &tx_notify_ui).await;
reconcile_watch_state(&ctx, &mut watch_state, watch_tx.clone()).await;
}
Some(event) = watch_rx.recv() => {
handle_watch_event(
&ctx,
&tx_notify_ui,
&gate,
event,
).await;
reconcile_watch_state(&ctx, &mut watch_state, watch_tx.clone()).await;
let loop_outcome = AssertUnwindSafe(async {
loop {
tokio::select! {
biased;
() = ctx.shutdown.cancelled() => break,
result = rescans.join_next(), if !rescans.is_empty() => {
if let Some(result) = result {
log_rescan_join(result);
}
}
_ = poll_interval.tick() => {
poll_local_game_changes(
&ctx,
&tx_notify_ui,
&gate,
&mut rescans,
&mut snapshot,
).await;
}
_ = fallback_interval.tick() => {
run_fallback_scan(&ctx, &tx_notify_ui).await;
}
}
}
}
}
async fn build_watch_state(
ctx: &Ctx,
watch_tx: tokio::sync::mpsc::UnboundedSender<notify::Result<Event>>,
) -> Option<WatchState> {
let game_dir = ctx.game_dir.read().await.clone();
let mut fs_watcher = match RecommendedWatcher::new(
move |result| {
let _ = watch_tx.send(result);
},
Config::default(),
) {
Ok(watcher) => watcher,
Err(err) => {
log::warn!("Filesystem watcher unavailable; falling back to periodic scans: {err}");
return None;
}
};
let watched_paths = match watch_game_roots(&mut fs_watcher, &game_dir).await {
Ok(paths) => paths,
Err(err) => {
log::warn!(
"Failed to initialize filesystem watcher for {}: {err}; falling back to periodic scans",
game_dir.display()
);
return None;
}
};
Some(WatchState {
watcher: fs_watcher,
game_dir,
watched: watched_paths,
Ok(())
})
.catch_unwind()
.await;
finish_monitor_loop(loop_outcome, &mut rescans, &gate).await
}
async fn reconcile_watch_state(
ctx: &Ctx,
watch_state: &mut Option<WatchState>,
watch_tx: tokio::sync::mpsc::UnboundedSender<notify::Result<Event>>,
) {
let current_game_dir = ctx.game_dir.read().await.clone();
if watch_state
.as_ref()
.is_none_or(|state| state.game_dir != current_game_dir)
{
*watch_state = build_watch_state(ctx, watch_tx).await;
return;
}
type MonitorLoopOutcome = std::thread::Result<eyre::Result<()>>;
if let Some(state) = watch_state
&& let Err(err) = reconcile_game_root_watches(state).await
{
log::warn!(
"Failed to reconcile filesystem watches for {}: {err}",
state.game_dir.display()
);
async fn finish_monitor_loop(
loop_outcome: MonitorLoopOutcome,
rescans: &mut JoinSet<()>,
gate: &RescanGate,
) -> eyre::Result<()> {
// Stop admitting polls, then wait for every lexically owned rescan on every
// loop exit, including an unwind. Poll snapshots execute inline and
// therefore cannot outlive this task.
drain_rescans(rescans, gate).await;
match loop_outcome {
Ok(result) => result,
Err(payload) => Err(eyre::eyre!(
"Local game monitor loop panicked: {}",
describe_panic(payload.as_ref())
)),
}
}
async fn watch_game_roots(
watcher: &mut RecommendedWatcher,
game_dir: &Path,
) -> eyre::Result<HashSet<PathBuf>> {
let mut watched_paths = HashSet::new();
watch_path(watcher, game_dir, &mut watched_paths)?;
for root in list_game_roots(game_dir).await? {
watch_path(watcher, &root, &mut watched_paths)?;
}
Ok(watched_paths)
fn describe_panic(payload: &(dyn Any + Send)) -> &str {
payload
.downcast_ref::<&str>()
.copied()
.or_else(|| payload.downcast_ref::<String>().map(String::as_str))
.unwrap_or("non-string panic payload")
}
async fn reconcile_game_root_watches(state: &mut WatchState) -> eyre::Result<()> {
let desired = {
let mut desired = HashSet::from([state.game_dir.clone()]);
desired.extend(list_game_roots(&state.game_dir).await?);
desired
};
fn log_rescan_join(result: Result<(), JoinError>) {
if let Err(error) = result {
log::error!("Local game rescan task failed: {error}");
}
}
let stale_paths = state
.watched
.difference(&desired)
.cloned()
.collect::<Vec<_>>();
for path in stale_paths {
if let Err(err) = state.watcher.unwatch(&path) {
log::debug!("Failed to unwatch {}: {err}", path.display());
async fn drain_rescans(rescans: &mut JoinSet<()>, gate: &RescanGate) {
while let Some(result) = rescans.join_next().await {
log_rescan_join(result);
}
gate.running.write().await.clear();
gate.pending.write().await.clear();
}
async fn initial_poll_snapshot(ctx: &Ctx) -> Option<PollSnapshot> {
match capture_poll_snapshot(ctx).await {
Ok(snapshot) => Some(snapshot),
Err(error) => {
log::warn!("Failed to initialize local game polling snapshot: {error}");
None
}
state.watched.remove(&path);
}
let new_paths = desired
.difference(&state.watched)
.cloned()
.collect::<Vec<_>>();
for path in new_paths {
watch_path(&mut state.watcher, &path, &mut state.watched)?;
}
Ok(())
}
fn watch_path(
watcher: &mut RecommendedWatcher,
path: &Path,
watched_paths: &mut HashSet<PathBuf>,
) -> notify::Result<()> {
watcher.watch(path, RecursiveMode::NonRecursive)?;
watched_paths.insert(path.to_path_buf());
Ok(())
}
async fn list_game_roots(game_dir: &Path) -> eyre::Result<Vec<PathBuf>> {
let mut roots = Vec::new();
let mut entries = match tokio::fs::read_dir(game_dir).await {
Ok(entries) => entries,
Err(err) if err.kind() == std::io::ErrorKind::NotFound => return Ok(roots),
Err(err) => return Err(err.into()),
};
while let Some(entry) = entries.next_entry().await? {
if !entry.file_type().await?.is_dir() {
continue;
}
let Some(name) = entry.file_name().to_str().map(ToOwned::to_owned) else {
continue;
};
if is_ignored_games_root_name(&name) {
continue;
}
roots.push(entry.path());
}
Ok(roots)
}
async fn handle_watch_event(
async fn poll_local_game_changes(
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
gate: &RescanGate,
event: notify::Result<Event>,
rescans: &mut JoinSet<()>,
previous: &mut Option<PollSnapshot>,
) {
let event = match event {
Ok(event) => event,
Err(err) => {
log::warn!("Filesystem watcher event error: {err}");
let current = match capture_poll_snapshot(ctx).await {
Ok(snapshot) => snapshot,
Err(error) => {
log::warn!("Failed to poll local game directory: {error}");
return;
}
};
if matches!(event.kind, EventKind::Access(_)) {
return;
}
let changed_ids = advance_poll_snapshot(previous, current);
queue_changed_games(ctx, tx_notify_ui, gate, rescans, changed_ids).await;
}
async fn capture_poll_snapshot(ctx: &Ctx) -> io::Result<PollSnapshot> {
// Serialize the root read with SetGameDir so one snapshot cannot combine
// entries from two configured roots. The blocking traversal stays scoped
// to this task and is complete before the admission guard is released.
let _admission = ctx.operation_admission.lock().await;
let game_dir = ctx.game_dir.read().await.clone();
let ids = event
.paths
.iter()
.filter_map(|path| game_id_from_event_path(&game_dir, path))
.collect::<HashSet<_>>();
scoped_blocking(|| snapshot_game_directory(&game_dir))
}
for id in ids {
if ctx.active_operations.read().await.contains_key(&id) {
log::debug!("Dropping filesystem event for {id}: operation active");
fn snapshot_game_directory(game_dir: &Path) -> io::Result<PollSnapshot> {
let mut games = BTreeMap::new();
let entries = match fs::read_dir(game_dir) {
Ok(entries) => entries,
Err(error) if error.kind() == io::ErrorKind::NotFound => {
return Ok(PollSnapshot {
game_dir: game_dir.to_path_buf(),
games,
});
}
Err(error) => return Err(error),
};
for entry in entries {
let entry = entry?;
let name = entry.file_name();
let Some(id) = name.to_str() else {
continue;
};
if is_ignored_games_root_name(id) {
continue;
}
queue_rescan(ctx, tx_notify_ui, gate, id).await;
let game_root = match snapshot_game_root(&entry.path()) {
Ok(Some(snapshot)) => snapshot,
Ok(None) => continue,
Err(error) => {
log::debug!(
"Could not snapshot local game root {}: {error}",
entry.path().display()
);
GameRootSnapshot::Unreadable
}
};
games.insert(id.to_owned(), game_root);
}
Ok(PollSnapshot {
game_dir: game_dir.to_path_buf(),
games,
})
}
fn snapshot_game_root(game_root: &Path) -> io::Result<Option<GameRootSnapshot>> {
let root_fingerprint = match fingerprint_entry(game_root) {
Ok(fingerprint) => fingerprint,
Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None),
Err(error) => return Err(error),
};
if root_fingerprint.kind != EntryKind::Directory {
return Ok(Some(GameRootSnapshot::NonDirectory(root_fingerprint)));
}
let entries = match fs::read_dir(game_root) {
Ok(entries) => entries,
Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None),
Err(error) => return Err(error),
};
let mut fingerprints = BTreeMap::new();
for entry in entries {
let entry = entry?;
let name = entry.file_name();
if name.to_str().is_some_and(is_download_protected_root_name) {
continue;
}
fingerprints.insert(name, fingerprint_entry(&entry.path())?);
}
Ok(Some(GameRootSnapshot::Readable(fingerprints)))
}
fn fingerprint_entry(path: &Path) -> io::Result<EntryFingerprint> {
let metadata = fs::symlink_metadata(path)?;
let file_type = metadata.file_type();
let kind = if file_type.is_symlink() || is_windows_reparse_point(&metadata) {
EntryKind::LinkOrReparse
} else if file_type.is_file() {
EntryKind::File
} else if file_type.is_dir() {
EntryKind::Directory
} else {
EntryKind::Other
};
Ok(EntryFingerprint {
kind,
len: metadata.len(),
modified: metadata.modified().ok(),
})
}
#[cfg(target_os = "windows")]
fn is_windows_reparse_point(metadata: &fs::Metadata) -> bool {
use std::os::windows::fs::MetadataExt as _;
const FILE_ATTRIBUTE_REPARSE_POINT: u32 = 0x400;
metadata.file_attributes() & FILE_ATTRIBUTE_REPARSE_POINT != 0
}
#[cfg(not(target_os = "windows"))]
const fn is_windows_reparse_point(_metadata: &fs::Metadata) -> bool {
false
}
fn advance_poll_snapshot(
previous: &mut Option<PollSnapshot>,
current: PollSnapshot,
) -> BTreeSet<String> {
let changed_ids = previous
.as_ref()
.filter(|snapshot| snapshot.game_dir == current.game_dir)
.map_or_else(BTreeSet::new, |snapshot| {
changed_game_ids(snapshot, &current)
});
*previous = Some(current);
changed_ids
}
fn changed_game_ids(previous: &PollSnapshot, current: &PollSnapshot) -> BTreeSet<String> {
previous
.games
.keys()
.chain(current.games.keys())
.filter(|id| previous.games.get(*id) != current.games.get(*id))
.cloned()
.collect()
}
async fn queue_changed_games(
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
gate: &RescanGate,
rescans: &mut JoinSet<()>,
changed_ids: BTreeSet<String>,
) {
let active_operations = ctx.active_operations.read().await;
let ready_ids = changed_ids
.into_iter()
.filter(|id| {
if active_operations.contains_key(id) {
log::debug!("Ignoring polled filesystem change for {id}: operation active");
false
} else {
true
}
})
.collect::<Vec<_>>();
drop(active_operations);
for id in ready_ids {
queue_rescan(ctx, tx_notify_ui, gate, rescans, id).await;
}
}
@@ -239,6 +356,7 @@ async fn queue_rescan(
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
gate: &RescanGate,
rescans: &mut JoinSet<()>,
id: String,
) {
{
@@ -253,7 +371,7 @@ async fn queue_rescan(
let ctx = ctx.clone();
let tx_notify_ui = tx_notify_ui.clone();
let gate = gate.clone();
ctx.task_tracker.clone().spawn(async move {
rescans.spawn(async move {
run_gated_rescan(ctx, tx_notify_ui, gate, id).await;
});
}
@@ -267,13 +385,27 @@ async fn run_gated_rescan(
loop {
gate.pending.write().await.remove(&id);
// SetGameDir holds this barrier through recovery, scan, publication,
// and quarantine settlement. A monitor scan therefore cannot publish
// a pre-recovery projection afterward.
let _admission = ctx.operation_admission.lock().await;
if ctx.active_operations.read().await.contains_key(&id) {
break;
}
let game_dir = ctx.game_dir.read().await.clone();
let catalog = ctx.catalog.read().await.clone();
match rescan_local_game(&game_dir, ctx.state_dir.as_ref(), &catalog, &id).await {
let catalog = ctx.catalog.catalog();
let failed_ids = ctx.recovery_quarantine.failed_ids(&game_dir);
match rescan_local_game_with_recovery_failures(
&game_dir,
ctx.state_dir.as_ref(),
catalog,
&id,
&failed_ids,
)
.await
{
Ok(scan) => update_and_announce_games(&ctx, &tx_notify_ui, scan).await,
Err(err) => log::error!("Failed to rescan local game {id}: {err}"),
}
@@ -287,42 +419,23 @@ async fn run_gated_rescan(
}
async fn run_fallback_scan(ctx: &Ctx, tx_notify_ui: &UnboundedSender<PeerEvent>) {
let _admission = ctx.operation_admission.lock().await;
let game_dir = ctx.game_dir.read().await.clone();
let catalog = ctx.catalog.read().await.clone();
match scan_local_library(&game_dir, ctx.state_dir.as_ref(), &catalog).await {
let catalog = ctx.catalog.catalog();
let failed_ids = ctx.recovery_quarantine.failed_ids(&game_dir);
match scan_local_library_with_recovery_failures(
&game_dir,
ctx.state_dir.as_ref(),
catalog,
&failed_ids,
)
.await
{
Ok(scan) => update_and_announce_games(ctx, tx_notify_ui, scan).await,
Err(err) => log::error!("Failed to scan local games directory: {err}"),
}
}
fn game_id_from_event_path(game_dir: &Path, path: &Path) -> Option<String> {
let relative = path.strip_prefix(game_dir).ok()?;
let mut components = relative.components();
let game_id = component_name(components.next()?)?;
if is_ignored_games_root_name(game_id) {
return None;
}
if let Some(second) = components.next().and_then(component_name)
&& should_ignore_game_child(second)
{
return None;
}
Some(game_id.to_string())
}
fn component_name(component: Component<'_>) -> Option<&str> {
match component {
Component::Normal(name) => name.to_str(),
_ => None,
}
}
fn should_ignore_game_child(name: &str) -> bool {
is_download_protected_root_name(name)
}
#[cfg(test)]
mod tests {
use std::{
@@ -331,11 +444,7 @@ mod tests {
time::Duration,
};
use lanspread_db::db::GameCatalog;
use notify::{
EventKind,
event::{AccessKind, AccessMode},
};
use lanspread_db::content_manifest::CatalogBundle;
use tokio::sync::{RwLock, mpsc};
use tokio_util::{sync::CancellationToken, task::TaskTracker};
@@ -344,14 +453,21 @@ mod tests {
UnpackFuture,
Unpacker,
context::OperationKind,
identity::PeerIdentity,
network_generation::NetworkControl,
peer_db::PeerGameDB,
test_support::TempDir,
test_support::{TempDir, catalog_bundle, empty_catalog_bundle},
};
struct NoopUnpacker;
impl Unpacker for NoopUnpacker {
fn unpack<'a>(&'a self, _archive: &'a Path, _dest: &'a Path) -> UnpackFuture<'a> {
fn unpack<'a>(
&'a self,
_archive: &'a Path,
_dest: &'a Path,
_cancel_token: CancellationToken,
) -> UnpackFuture<'a> {
Box::pin(async { Ok(()) })
}
}
@@ -363,24 +479,37 @@ mod tests {
std::fs::write(path, bytes).expect("file should be written");
}
fn test_ctx(game_dir: PathBuf, catalog: GameCatalog) -> Ctx {
fn test_ctx(game_dir: PathBuf, catalog: Arc<CatalogBundle>) -> Ctx {
let state_dir = game_dir.join(".test-state");
Ctx::new(
let recovery_root = game_dir.clone();
let ctx = Ctx::new(
Arc::new(RwLock::new(PeerGameDB::new())),
"peer".to_string(),
Arc::new(PeerIdentity::generate().expect("test identity should generate")),
game_dir,
state_dir,
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
Arc::new(RwLock::new(catalog)),
catalog,
Arc::new(RwLock::new(std::collections::HashMap::new())),
Arc::new(crate::NoopStreamInstallProvider),
NetworkControl::disabled_for_test(),
)
.expect("test context should initialize");
assert!(
ctx.recovery_quarantine
.settle(&recovery_root, HashSet::new())
);
ctx
}
fn watch_event(path: PathBuf) -> Event {
Event::new(EventKind::Any).add_path(path)
fn snapshot(game_dir: &Path) -> PollSnapshot {
snapshot_game_directory(game_dir).expect("poll snapshot should succeed")
}
async fn injected_monitor_loop_panic() -> eyre::Result<()> {
tokio::task::yield_now().await;
panic!("injected monitor loop panic");
}
async fn recv_local_update(
@@ -397,134 +526,164 @@ mod tests {
}
#[test]
fn event_paths_map_to_top_level_game_id() {
let root = std::path::Path::new("/games");
fn first_snapshot_and_configured_root_change_establish_baselines() {
let first = TempDir::new("lanspread-local-monitor-first-root");
let second = TempDir::new("lanspread-local-monitor-second-root");
write_file(&first.path().join("game/version.ini"), b"20250101");
write_file(&second.path().join("other/version.ini"), b"20250101");
let mut state = None;
assert!(advance_poll_snapshot(&mut state, snapshot(first.path())).is_empty());
assert!(advance_poll_snapshot(&mut state, snapshot(second.path())).is_empty());
assert_eq!(
game_id_from_event_path(root, std::path::Path::new("/games/aoe2/version.ini"))
.as_deref(),
Some("aoe2")
);
assert_eq!(
game_id_from_event_path(root, std::path::Path::new("/games/aoe2/local/save.dat")),
None
);
assert_eq!(
game_id_from_event_path(root, std::path::Path::new("/games/.lanspread/index.json")),
None
state.as_ref().map(|state| state.game_dir.as_path()),
Some(second.path())
);
}
#[test]
fn event_ignore_list_covers_reserved_names() {
for name in [
"local",
".local.installing",
".local.backup",
".version.ini.tmp",
".version.ini.discarded",
".lanspread",
".lanspread.json",
".sync",
".softlan_game_installed",
] {
assert!(should_ignore_game_child(name));
}
assert!(!should_ignore_game_child("version.ini"));
assert!(!should_ignore_game_child("game.eti"));
fn snapshot_diff_detects_game_change_and_disappearance() {
let temp = TempDir::new("lanspread-local-monitor-change");
let version = temp.path().join("game/version.ini");
write_file(&version, b"20250101");
let mut state = Some(snapshot(temp.path()));
write_file(&version, b"202501010");
assert_eq!(
advance_poll_snapshot(&mut state, snapshot(temp.path())),
BTreeSet::from(["game".to_string()])
);
std::fs::remove_dir_all(temp.path().join("game")).expect("game root should be removable");
assert_eq!(
advance_poll_snapshot(&mut state, snapshot(temp.path())),
BTreeSet::from(["game".to_string()]),
"a vanished directory must retain the prior game ID for its removal rescan"
);
}
#[test]
fn snapshot_diff_detects_new_non_directory_game_root_shape() {
let temp = TempDir::new("lanspread-local-monitor-unsafe-root");
let mut state = Some(snapshot(temp.path()));
write_file(&temp.path().join("game"), b"not a directory");
assert_eq!(
advance_poll_snapshot(&mut state, snapshot(temp.path())),
BTreeSet::from(["game".to_string()])
);
}
#[test]
fn snapshot_diff_ignores_protected_game_state_and_library_state() {
let temp = TempDir::new("lanspread-local-monitor-ignore");
write_file(&temp.path().join("game/version.ini"), b"20250101");
let mut state = Some(snapshot(temp.path()));
write_file(&temp.path().join("game/local/save.dat"), b"save");
write_file(
&temp.path().join("game/.local.installing/staged.dat"),
b"staged",
);
write_file(&temp.path().join(".lanspread/library_index.json"), b"index");
assert!(advance_poll_snapshot(&mut state, snapshot(temp.path())).is_empty());
}
#[tokio::test]
async fn watch_event_for_active_game_is_dropped() {
async fn polled_change_for_active_game_is_dropped() {
let temp = TempDir::new("lanspread-local-monitor");
let ctx = test_ctx(
temp.path().to_path_buf(),
GameCatalog::from_ids(["game".to_string()]),
catalog_bundle([("game", "20250101")]),
);
ctx.active_operations
.write()
.await
.insert("game".to_string(), OperationKind::Downloading);
let gate = RescanGate::default();
let rescan_gate = RescanGate::default();
let mut rescans = JoinSet::new();
let (tx, mut rx) = mpsc::unbounded_channel();
handle_watch_event(
queue_changed_games(
&ctx,
&tx,
&gate,
Ok(watch_event(temp.path().join("game").join("version.ini"))),
&rescan_gate,
&mut rescans,
BTreeSet::from(["game".to_string()]),
)
.await;
ctx.task_tracker.close();
ctx.task_tracker.wait().await;
assert!(
tokio::time::timeout(Duration::from_millis(50), rx.recv())
.await
.is_err(),
"active game event should not schedule a UI update"
"an active game change should not schedule a UI update"
);
assert!(gate.running.read().await.is_empty());
assert!(gate.pending.read().await.is_empty());
assert!(rescans.is_empty());
assert!(rescan_gate.running.read().await.is_empty());
assert!(rescan_gate.pending.read().await.is_empty());
}
#[tokio::test]
async fn access_watch_event_is_ignored() {
async fn polling_detects_sideload_and_runs_per_game_rescan() {
let temp = TempDir::new("lanspread-local-monitor");
write_file(&temp.path().join("game").join("version.ini"), b"20250101");
let ctx = test_ctx(
temp.path().to_path_buf(),
GameCatalog::from_ids(["game".to_string()]),
catalog_bundle([("game", "20250101")]),
);
let gate = RescanGate::default();
let rescan_gate = RescanGate::default();
let mut rescans = JoinSet::new();
let (tx, mut rx) = mpsc::unbounded_channel();
handle_watch_event(
&ctx,
&tx,
&gate,
Ok(
Event::new(EventKind::Access(AccessKind::Close(AccessMode::Read)))
.add_path(temp.path().join("game").join("version.ini")),
),
)
.await;
ctx.task_tracker.close();
ctx.task_tracker.wait().await;
assert!(
tokio::time::timeout(Duration::from_millis(50), rx.recv())
let mut state = Some(
capture_poll_snapshot(&ctx)
.await
.is_err(),
"access events should not schedule a UI update"
.expect("initial snapshot should succeed"),
);
assert!(gate.running.read().await.is_empty());
assert!(gate.pending.read().await.is_empty());
write_file(&temp.path().join("game/version.ini"), b"20250101");
poll_local_game_changes(&ctx, &tx, &rescan_gate, &mut rescans, &mut state).await;
drain_rescans(&mut rescans, &rescan_gate).await;
let games = recv_local_update(&mut rx).await;
let game = games
.iter()
.find(|game| game.id == "game")
.expect("sideloaded catalog game should be emitted");
assert!(game.downloaded);
assert!(!game.installed);
assert!(rescans.is_empty());
assert!(rescan_gate.running.read().await.is_empty());
assert!(rescan_gate.pending.read().await.is_empty());
}
#[tokio::test]
async fn burst_watch_events_collapse_to_two_rescans_for_same_game() {
async fn burst_poll_changes_collapse_to_two_rescans_for_same_game() {
let temp = TempDir::new("lanspread-local-monitor");
let game_root = temp.path().join("game");
write_file(&game_root.join("version.ini"), b"20250101");
let ctx = test_ctx(
temp.path().to_path_buf(),
GameCatalog::from_ids(["game".to_string()]),
catalog_bundle([("game", "20250101")]),
);
let gate = RescanGate::default();
let mut rescans = JoinSet::new();
let (tx, mut rx) = mpsc::unbounded_channel();
let library_guard = ctx.local_library.write().await;
queue_rescan(&ctx, &tx, &gate, "game".to_string()).await;
queue_rescan(&ctx, &tx, &gate, &mut rescans, "game".to_string()).await;
tokio::time::sleep(Duration::from_millis(20)).await;
for _ in 0..5 {
queue_rescan(&ctx, &tx, &gate, "game".to_string()).await;
queue_rescan(&ctx, &tx, &gate, &mut rescans, "game".to_string()).await;
}
assert_eq!(gate.pending.read().await.len(), 1);
drop(library_guard);
ctx.task_tracker.close();
ctx.task_tracker.wait().await;
while let Some(result) = rescans.join_next().await {
result.expect("rescan task should finish successfully");
}
let mut update_count = 0;
while let Ok(Some(PeerEvent::LocalLibraryChanged { .. })) =
@@ -538,13 +697,121 @@ mod tests {
);
}
#[tokio::test]
async fn rescan_shutdown_waits_for_blocked_children_to_finish() {
let temp = TempDir::new("lanspread-local-monitor-shutdown");
write_file(&temp.path().join("game/version.ini"), b"20250101");
let ctx = test_ctx(
temp.path().to_path_buf(),
catalog_bundle([("game", "20250101")]),
);
let gate = RescanGate::default();
let mut rescans = JoinSet::new();
let (tx, mut rx) = mpsc::unbounded_channel();
let admission = ctx.operation_admission.lock().await;
queue_rescan(&ctx, &tx, &gate, &mut rescans, "game".to_string()).await;
tokio::task::yield_now().await;
assert_eq!(rescans.len(), 1);
assert!(gate.running.read().await.contains("game"));
let gate_after_shutdown = gate.clone();
let shutdown = tokio::spawn(async move {
drain_rescans(&mut rescans, &gate_after_shutdown).await;
(rescans, gate_after_shutdown)
});
tokio::time::sleep(Duration::from_millis(50)).await;
assert!(
!shutdown.is_finished(),
"shutdown must remain pending while a rescan child owns unfinished work"
);
drop(admission);
let (rescans, gate) = tokio::time::timeout(Duration::from_secs(5), shutdown)
.await
.expect("shutdown should finish after its child")
.expect("shutdown task should not panic");
assert!(rescans.is_empty());
assert!(gate.running.read().await.is_empty());
assert!(gate.pending.read().await.is_empty());
let games = recv_local_update(&mut rx).await;
assert_eq!(games.len(), 1);
assert_eq!(games[0].id, "game");
}
#[tokio::test]
async fn monitor_loop_panic_drains_owned_rescans_before_returning_error() {
let temp = TempDir::new("lanspread-local-monitor-panic");
write_file(&temp.path().join("game/version.ini"), b"20250101");
let ctx = test_ctx(
temp.path().to_path_buf(),
catalog_bundle([("game", "20250101")]),
);
let gate = RescanGate::default();
let mut rescans = JoinSet::new();
let (tx, mut rx) = mpsc::unbounded_channel();
let admission = ctx.operation_admission.lock().await;
queue_rescan(&ctx, &tx, &gate, &mut rescans, "game".to_string()).await;
tokio::task::yield_now().await;
let loop_outcome = AssertUnwindSafe(injected_monitor_loop_panic())
.catch_unwind()
.await;
let gate_after_panic = gate.clone();
let finishing = tokio::spawn(async move {
let result = finish_monitor_loop(loop_outcome, &mut rescans, &gate_after_panic).await;
(result, rescans, gate_after_panic)
});
tokio::time::sleep(Duration::from_millis(50)).await;
assert!(
!finishing.is_finished(),
"a caught loop panic must not abort an unfinished rescan child"
);
drop(admission);
let (result, rescans, gate) = tokio::time::timeout(Duration::from_secs(5), finishing)
.await
.expect("panic epilogue should finish after its child")
.expect("panic epilogue task should not panic");
let error = result.expect_err("caught loop panic should become an error report");
assert!(error.to_string().contains("injected monitor loop panic"));
assert!(rescans.is_empty());
assert!(gate.running.read().await.is_empty());
assert!(gate.pending.read().await.is_empty());
let games = recv_local_update(&mut rx).await;
assert_eq!(games.len(), 1);
assert_eq!(games[0].id, "game");
}
#[tokio::test]
async fn monitor_shutdown_returns_without_background_children() {
let temp = TempDir::new("lanspread-local-monitor-structured-shutdown");
let ctx = test_ctx(temp.path().to_path_buf(), empty_catalog_bundle());
let monitor_ctx = ctx.clone();
let (tx, _rx) = mpsc::unbounded_channel();
let monitor = tokio::spawn(run_local_game_monitor(tx, monitor_ctx));
tokio::task::yield_now().await;
ctx.shutdown.cancel();
tokio::time::timeout(Duration::from_secs(2), monitor)
.await
.expect("monitor shutdown must not wait for an unowned backend")
.expect("monitor task should not panic")
.expect("monitor should stop successfully");
}
#[tokio::test]
async fn fallback_scan_picks_up_sideloaded_catalog_game() {
let temp = TempDir::new("lanspread-local-monitor");
write_file(&temp.path().join("game").join("version.ini"), b"20250101");
let ctx = test_ctx(
temp.path().to_path_buf(),
GameCatalog::from_ids(["game".to_string()]),
catalog_bundle([("game", "20250101")]),
);
let (tx, mut rx) = mpsc::unbounded_channel();
@@ -568,7 +835,7 @@ mod tests {
);
let ctx = test_ctx(
temp.path().to_path_buf(),
GameCatalog::from_ids(["game".to_string()]),
catalog_bundle([("game", "20250101")]),
);
let (tx, mut rx) = mpsc::unbounded_channel();
@@ -582,7 +849,6 @@ mod tests {
);
let library = ctx.local_library.read().await;
assert!(library.games.is_empty());
assert!(library.recent_deltas.is_empty());
assert!(
!temp
.path()
@@ -0,0 +1,510 @@
//! Preparation and atomic commit of responder-owned peer state.
use std::sync::Arc;
use lanspread_proto::{
LibrarySnapshot,
PeerEndpoint,
PeerId,
PeerRevisions,
PeerStateSnapshot,
RuntimeSessionId,
};
use tokio::sync::{RwLock, mpsc::UnboundedSender};
use tokio_util::sync::CancellationToken;
use crate::{
CallToPlayView,
PeerEvent,
call_to_play::{
CallToPlayPublication,
CallToPlayStore,
ObserveRemoteAuthorOutcome,
PreparedCallToPlayPublication,
PreparedRemoteAuthor,
},
context::{Ctx, NetworkServiceCtx, PeerCtx},
events,
library::build_library_snapshot,
network::exchange_hello,
peer_db::{
PeerEndpointGeneration,
PeerGameDB,
PeerLivenessSnapshot,
PeerNegotiationTicket,
PeerUpsert,
PongObservation,
RefreshReservation,
RetiredPeerEndpoint,
},
quic_runtime::QuicConnector,
services::StateSyncHandle,
};
#[derive(Clone)]
pub(crate) struct RemoteStateCtx {
local_peer_id: PeerId,
peer_game_db: Arc<RwLock<PeerGameDB>>,
tx_notify_ui: UnboundedSender<PeerEvent>,
call_to_play: Arc<RwLock<CallToPlayStore>>,
quic: QuicConnector,
cancellation: CancellationToken,
state_sync: StateSyncHandle,
}
impl RemoteStateCtx {
#[must_use]
pub(crate) fn from_network(
ctx: &NetworkServiceCtx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) -> Self {
Self {
local_peer_id: ctx.peer_id,
peer_game_db: ctx.peer_game_db.clone(),
tx_notify_ui: tx_notify_ui.clone(),
call_to_play: ctx.call_to_play.clone(),
quic: ctx.quic.clone(),
cancellation: ctx.shutdown.clone(),
state_sync: ctx.state_sync.clone(),
}
}
#[must_use]
pub(crate) fn with_cancellation(mut self, cancellation: CancellationToken) -> Self {
self.cancellation = cancellation;
self
}
pub(crate) async fn reserve_candidate(
&self,
endpoint: PeerEndpoint,
) -> eyre::Result<PeerNegotiationTicket> {
self.peer_game_db
.write()
.await
.begin_candidate_negotiation(endpoint)
}
pub(crate) async fn begin_peer_refresh(
&self,
snapshot: PeerLivenessSnapshot,
) -> eyre::Result<RefreshReservation> {
self.peer_game_db.write().await.begin_peer_refresh(snapshot)
}
pub(crate) async fn peer_liveness_for(&self, peer_id: PeerId) -> Option<PeerLivenessSnapshot> {
self.peer_game_db.read().await.peer_liveness_for(&peer_id)
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(crate) enum PongCommit {
StaleGeneration,
Current,
NeedsPull { new_session: bool },
}
/// Captures local revision state for a pinned Pong response.
pub(super) async fn local_revisions(ctx: &PeerCtx) -> eyre::Result<PeerRevisions> {
// Keep the local-library -> CTP lock order through publication enqueue.
// A remote DB -> CTP commit therefore cannot publish a newer view before
// this responder publication reaches the ordered UI channel.
let library = ctx.local_library.read().await;
let mut call_to_play = ctx.call_to_play.write().await;
let (call_to_play_revision, publication) = call_to_play.current_responder_state()?;
let library_revision = library.revision;
if let Some(publication) = publication {
enqueue_call_to_play_publication(publication, &ctx.tx_notify_ui, &ctx.state_sync, false);
}
Ok(PeerRevisions {
runtime_session_id: ctx.runtime_session_id,
library_revision,
call_to_play_revision,
})
}
/// Captures both local domains in lock order, then resolves catalog identities
/// from the manifest cache preloaded before network admission.
pub(super) async fn local_snapshot(ctx: &PeerCtx) -> eyre::Result<PeerStateSnapshot> {
let library = ctx.local_library.read().await;
let mut call_to_play_store = ctx.call_to_play.write().await;
let library_publication = library.publication(ctx.catalog.catalog());
let (call_to_play, _call_to_play_revision, publication) =
call_to_play_store.local_responder_snapshot()?;
if let Some(publication) = publication {
enqueue_call_to_play_publication(publication, &ctx.tx_notify_ui, &ctx.state_sync, false);
}
drop(call_to_play_store);
drop(library);
let library = build_library_snapshot(library_publication, &ctx.catalog)?;
Ok(PeerStateSnapshot {
runtime_session_id: ctx.runtime_session_id,
library,
call_to_play,
})
}
/// Performs a pinned full pull and commits its independently validated domains.
pub(crate) async fn pull_and_commit(
ctx: &RemoteStateCtx,
endpoint: PeerEndpoint,
ticket: PeerNegotiationTicket,
) -> eyre::Result<bool> {
if endpoint.peer_id == ctx.local_peer_id {
return Ok(false);
}
let generation = ticket.endpoint_generation();
let snapshot = exchange_hello(&ctx.quic, &endpoint, &ctx.cancellation).await?;
let prepared = PreparedSnapshot::prepare(endpoint.peer_id, generation, snapshot);
commit_prepared(ctx, endpoint, ticket, prepared).await
}
struct PreparedSnapshot {
runtime_session_id: RuntimeSessionId,
library: Option<LibrarySnapshot>,
library_error: Option<lanspread_proto::ControlValidationError>,
call_to_play: PreparedRemoteAuthor,
}
impl PreparedSnapshot {
fn prepare(
author_id: PeerId,
generation: PeerEndpointGeneration,
snapshot: PeerStateSnapshot,
) -> Self {
let PeerStateSnapshot {
runtime_session_id,
library,
call_to_play,
} = snapshot;
let library_error = library.validate().err();
let library = library_error.is_none().then_some(library);
let call_to_play =
PreparedRemoteAuthor::prepare(author_id, generation, runtime_session_id, call_to_play);
Self {
runtime_session_id,
library,
library_error,
call_to_play,
}
}
}
async fn commit_prepared(
ctx: &RemoteStateCtx,
endpoint: PeerEndpoint,
ticket: PeerNegotiationTicket,
prepared: PreparedSnapshot,
) -> eyre::Result<bool> {
if let Some(error) = &prepared.library_error {
log::warn!(
"Rejecting invalid library domain from {}: {error}",
endpoint.peer_id
);
}
if let Some(error) = prepared.call_to_play.validation_error() {
log::warn!(
"Rejecting invalid Call-to-Play domain from {}: {error}",
endpoint.peer_id
);
}
let mut db = ctx.peer_game_db.write().await;
let mut call_to_play = ctx.call_to_play.write().await;
// Resolve the fallible local clock/prune boundary before mutating remote
// state, while retaining the fixed DB-to-CTP lock order through commit.
let publication = call_to_play.prepare_publication()?;
let commit_result = db.commit_authenticated_snapshot(
endpoint,
ticket,
prepared.runtime_session_id,
prepared.library,
);
let upsert = match commit_result {
Ok(Some(upsert)) => upsert,
Ok(None) => {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
return Ok(false);
}
Err(error) => {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
return Err(error);
}
};
let evicted_call_to_play_changed = if let Some(evicted) = upsert.evicted_endpoint {
call_to_play
.remove_remote_author_if_generation(evicted.endpoint.peer_id, evicted.generation)
} else {
false
};
let call_to_play_outcome = call_to_play.observe_prepared_remote(prepared.call_to_play);
log_call_to_play_outcome(endpoint.peer_id, &call_to_play_outcome);
let accepted_call_to_play_revision = call_to_play
.remote_author_state(endpoint.peer_id)
.filter(|state| state.endpoint_generation == upsert.endpoint_generation)
.map(|state| state.revision);
debug_assert!(db.set_call_to_play_revision_if_generation(
endpoint,
upsert.endpoint_generation,
accepted_call_to_play_revision,
));
enqueue_commit_transition(
&db,
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
endpoint,
upsert,
call_to_play_outcome.view_changed() || evicted_call_to_play_changed,
);
Ok(true)
}
fn log_call_to_play_outcome(peer_id: PeerId, outcome: &ObserveRemoteAuthorOutcome) {
match outcome {
ObserveRemoteAuthorOutcome::InvalidCleared(error)
| ObserveRemoteAuthorOutcome::InvalidAbsent(error) => {
log::warn!("Rejected Call-to-Play author {peer_id}: {error}");
}
ObserveRemoteAuthorOutcome::InvalidPreserved { error, .. } => {
log::warn!("Preserving prior Call-to-Play author {peer_id}: {error}");
}
ObserveRemoteAuthorOutcome::AtCapacity => {
log::warn!("Call-to-Play author limit reached; ignoring {peer_id}");
}
ObserveRemoteAuthorOutcome::RejectedLocalIdentity => {
log::warn!("Rejected remote Call-to-Play snapshot for local identity {peer_id}");
}
ObserveRemoteAuthorOutcome::Applied { .. }
| ObserveRemoteAuthorOutcome::Unchanged { .. }
| ObserveRemoteAuthorOutcome::IgnoredStale { .. } => {}
ObserveRemoteAuthorOutcome::EqualRevisionConflict { .. } => {
log::warn!(
"Preserving prior Call-to-Play author {peer_id}: equal revision had different content"
);
}
}
}
#[allow(clippy::too_many_arguments)]
fn enqueue_commit_transition(
db: &PeerGameDB,
call_to_play: &mut CallToPlayStore,
tx_notify_ui: &UnboundedSender<PeerEvent>,
state_sync: &StateSyncHandle,
publication: PreparedCallToPlayPublication,
endpoint: PeerEndpoint,
upsert: PeerUpsert,
call_to_play_changed: bool,
) {
let topology_changed = upsert.is_new
|| upsert.addr_changed
|| upsert.previous_endpoint.is_some()
|| upsert.evicted_endpoint.is_some();
if let Some(previous) = upsert.previous_endpoint {
events::send(tx_notify_ui, PeerEvent::PeerLost(previous));
}
if let Some(evicted) = upsert.evicted_endpoint {
events::send(tx_notify_ui, PeerEvent::PeerLost(evicted.endpoint));
}
if upsert.is_new || upsert.addr_changed {
events::send(tx_notify_ui, PeerEvent::PeerDiscovered(endpoint));
}
if topology_changed {
events::send(
tx_notify_ui,
PeerEvent::PeerCountUpdated(db.peer_endpoints().len()),
);
}
if topology_changed || upsert.library_changed {
events::send(
tx_notify_ui,
PeerEvent::RemoteLibraryView(events::remote_library_view(db)),
);
}
if topology_changed || call_to_play_changed || publication.local_changed() {
let publication = call_to_play.view_from_prepared(publication);
enqueue_call_to_play_publication(publication, tx_notify_ui, state_sync, true);
}
}
fn enqueue_final_views(
db: &PeerGameDB,
call_to_play: &mut CallToPlayStore,
tx_notify_ui: &UnboundedSender<PeerEvent>,
state_sync: &StateSyncHandle,
publication: PreparedCallToPlayPublication,
) {
events::send(
tx_notify_ui,
PeerEvent::PeerCountUpdated(db.peer_endpoints().len()),
);
events::send(
tx_notify_ui,
PeerEvent::RemoteLibraryView(events::remote_library_view(db)),
);
let publication = call_to_play.view_from_prepared(publication);
enqueue_call_to_play_publication(publication, tx_notify_ui, state_sync, true);
}
/// Clears every remote projection after one network generation has fully
/// drained, then queues the authoritative empty replacement views while the
/// database-to-Call-to-Play lock order is still held.
pub(crate) async fn clear_remote_state_and_publish(
ctx: &Ctx,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
let mut db = ctx.peer_game_db.write().await;
let mut call_to_play = ctx.call_to_play.write().await;
let retired = db.clear_remote_peers();
let (publication, preparation_error) = call_to_play.clear_remote_authors_and_project();
for RetiredPeerEndpoint { endpoint, .. } in retired {
events::send(tx_notify_ui, PeerEvent::PeerLost(endpoint));
}
events::send(tx_notify_ui, PeerEvent::PeerCountUpdated(0));
events::send(
tx_notify_ui,
PeerEvent::RemoteLibraryView(events::remote_library_view(&db)),
);
enqueue_call_to_play_publication(publication, tx_notify_ui, &ctx.state_sync, true);
if let Some(error) = preparation_error {
log::warn!(
"Cleared remote Call-to-Play state, but local retention maintenance failed: {error}"
);
}
}
fn enqueue_local_prune_if_changed(
call_to_play: &mut CallToPlayStore,
tx_notify_ui: &UnboundedSender<PeerEvent>,
state_sync: &StateSyncHandle,
publication: PreparedCallToPlayPublication,
) {
if !publication.local_changed() {
return;
}
let publication = call_to_play.view_from_prepared(publication);
enqueue_call_to_play_publication(publication, tx_notify_ui, state_sync, true);
}
fn enqueue_call_to_play_publication(
publication: CallToPlayPublication,
tx_notify_ui: &UnboundedSender<PeerEvent>,
state_sync: &StateSyncHandle,
force_view: bool,
) {
state_sync.publish_call_to_play_revision(publication.local_revision);
if force_view || publication.local_changed {
events::send(
tx_notify_ui,
PeerEvent::CallToPlayView(CallToPlayView::from(publication.view)),
);
}
}
/// Applies a pinned Pong while holding the fixed DB-to-CTP lock order. A new
/// runtime session clears both cached domains before any follow-up pull starts.
pub(crate) async fn observe_pinned_pong(
ctx: &RemoteStateCtx,
snapshot: PeerLivenessSnapshot,
revisions: PeerRevisions,
) -> eyre::Result<PongCommit> {
let mut db = ctx.peer_game_db.write().await;
let mut call_to_play = ctx.call_to_play.write().await;
let publication = call_to_play.prepare_publication()?;
match db.observe_pong_if_generation(snapshot, revisions) {
PongObservation::StaleGeneration => {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
Ok(PongCommit::StaleGeneration)
}
PongObservation::Current => {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
Ok(PongCommit::Current)
}
PongObservation::RevisionMismatch => {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
Ok(PongCommit::NeedsPull { new_session: false })
}
PongObservation::NewSession => {
call_to_play
.remove_remote_author_if_generation(snapshot.endpoint.peer_id, snapshot.generation);
enqueue_final_views(
&db,
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
Ok(PongCommit::NeedsPull { new_session: true })
}
}
}
/// Removes one generation and its authored slice atomically, then enqueues the
/// final replacement views before releasing either state lock.
pub(crate) async fn remove_peer_if_generation(
ctx: &RemoteStateCtx,
snapshot: PeerLivenessSnapshot,
) -> eyre::Result<bool> {
let mut db = ctx.peer_game_db.write().await;
let mut call_to_play = ctx.call_to_play.write().await;
let publication = call_to_play.prepare_publication()?;
let Some(peer) = db.remove_peer_if_generation(snapshot.endpoint, snapshot.generation) else {
drop(db);
enqueue_local_prune_if_changed(
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
return Ok(false);
};
call_to_play.remove_remote_author_if_generation(peer.peer_id, peer.endpoint_generation);
events::send(
&ctx.tx_notify_ui,
PeerEvent::PeerLost(PeerEndpoint::new(peer.peer_id, peer.addr)),
);
enqueue_final_views(
&db,
&mut call_to_play,
&ctx.tx_notify_ui,
&ctx.state_sync,
publication,
);
Ok(true)
}
+623 -62
View File
@@ -1,80 +1,255 @@
//! QUIC server accept loop.
use std::{net::SocketAddr, time::Duration};
use std::{future::Future, net::SocketAddr, panic::AssertUnwindSafe, sync::Arc, time::Duration};
use s2n_quic::{Connection, Server};
use tokio::sync::mpsc::UnboundedSender;
use futures::FutureExt as _;
use s2n_quic::{
Connection,
Server,
application,
provider::endpoint_limits,
stream::BidirectionalStream,
};
use tokio::{
sync::{OwnedSemaphorePermit, Semaphore, oneshot},
task::JoinSet,
};
use tokio_util::sync::CancellationToken;
use crate::{
PeerEvent,
config::{CERT_PEM, KEY_PEM},
context::PeerCtx,
events,
network::{quic_congestion_controller, quic_io, quic_limits},
library::prime_library_manifests,
quic_runtime::{quic_congestion_controller, quic_server_limits, tracked_quic_io},
scoped_blocking::scoped_blocking,
services::{
advertise::{monitor_mdns_events, start_mdns_advertiser},
advertise::{close_mdns_advertiser, monitor_mdns_events, start_mdns_advertiser},
stream::handle_peer_stream,
},
tls,
};
/// Limits unauthenticated handshake memory before QUIC admission completes.
const MAX_INFLIGHT_HANDSHAKES: usize = 64;
/// Limits established connection scopes owned by the application accept loop.
const MAX_ESTABLISHED_CONNECTIONS: usize = 64;
/// Mirrors the transport stream limit and bounds application stream futures.
const MAX_CONTROL_STREAM_TASKS: usize = 32;
/// Server-wide cap acquired before any length-delimited decoder is allocated.
const MAX_GLOBAL_CONTROL_STREAM_TASKS: usize = 64;
/// Long-lived transfers move from the decoder pool to this smaller pool so
/// saturated bulk egress cannot consume every control-plane permit.
const MAX_GLOBAL_BULK_TRANSFER_TASKS: usize = 48;
/// Application idle bound for an established connection with no active
/// request streams. This is independent of transport keepalive traffic.
const CONNECTION_NO_STREAM_IDLE_TIMEOUT: Duration = Duration::from_secs(10);
struct BoundedEndpointLimits {
inner: endpoint_limits::Default,
}
impl endpoint_limits::Limiter for BoundedEndpointLimits {
fn on_connection_attempt(
&mut self,
info: &endpoint_limits::ConnectionAttempt<'_>,
) -> endpoint_limits::Outcome {
if !endpoint_connection_capacity_available(info.connection_count) {
return endpoint_limits::Outcome::close();
}
endpoint_limits::Limiter::on_connection_attempt(&mut self.inner, info)
}
}
const fn endpoint_connection_capacity_available(connection_count: usize) -> bool {
connection_count < MAX_ESTABLISHED_CONNECTIONS
}
fn bounded_endpoint_limits() -> eyre::Result<BoundedEndpointLimits> {
Ok(BoundedEndpointLimits {
inner: endpoint_limits::Default::builder()
.with_inflight_handshake_limit(MAX_INFLIGHT_HANDSHAKES)?
.build()?,
})
}
/// Runs the QUIC server and mDNS advertiser.
pub async fn run_server_component(
addr: SocketAddr,
ctx: PeerCtx,
tx_notify_ui: UnboundedSender<PeerEvent>,
ready: oneshot::Sender<SocketAddr>,
) -> eyre::Result<()> {
let limits = quic_limits()?
.with_max_handshake_duration(Duration::from_secs(3))?
.with_max_idle_timeout(Duration::from_secs(3))?;
// Manifest bodies may be disk-backed on their first access. Resolve only
// the currently publishable local set before opening the public endpoint;
// this bounds retained manifest memory by actual local availability.
let publication = {
let library = ctx.local_library.read().await;
library.publication(ctx.catalog.catalog())
};
let catalog = Arc::clone(&ctx.catalog);
scoped_blocking(move || prime_library_manifests(&publication.game_ids, &catalog))?;
let mut server = Server::builder()
.with_tls((CERT_PEM, KEY_PEM))?
.with_io(quic_io(addr)?)?
.with_limits(limits)?
let (io, endpoint_control) = tracked_quic_io(addr)?;
let server = Server::builder()
.with_tls(tls::server_provider(&ctx.peer_identity)?)?
.with_io(io)?
.with_endpoint_limits(bounded_endpoint_limits()?)?
.with_limits(quic_server_limits()?)?
.with_congestion_controller(quic_congestion_controller())?
.start()?;
let endpoint_task = endpoint_control.take_started()?;
run_body_with_cleanup(
run_server_body(server, ctx, ready),
endpoint_task.shutdown_and_join(),
)
.await
}
async fn run_server_body(
mut server: Server,
ctx: PeerCtx,
ready: oneshot::Sender<SocketAddr>,
) -> eyre::Result<()> {
let server_addr = server.local_addr()?;
log::info!("Peer server listening on {server_addr}");
let mdns_advertiser = start_mdns_advertiser(&ctx, server_addr).await?;
let mdns_monitor = mdns_advertiser.monitor.clone();
let mdns_shutdown = ctx.shutdown.clone();
ctx.task_tracker.spawn(async move {
let server_children_shutdown = ctx.shutdown.child_token();
let mut mdns_tasks = JoinSet::new();
let mdns_shutdown = server_children_shutdown.clone();
mdns_tasks.spawn(async move {
monitor_mdns_events(mdns_monitor, mdns_shutdown).await;
Ok(())
});
let mut connection_tasks = JoinSet::new();
let control_stream_permits = Arc::new(Semaphore::new(MAX_GLOBAL_CONTROL_STREAM_TASKS));
let bulk_transfer_permits = Arc::new(Semaphore::new(MAX_GLOBAL_BULK_TRANSFER_TASKS));
let ready_addr =
(*ctx.local_peer_addr.read().await).unwrap_or_else(|| direct_connect_addr(server_addr));
let _mdns_advertiser = mdns_advertiser;
events::send(
&tx_notify_ui,
PeerEvent::LocalPeerReady {
peer_id: ctx.peer_id.as_ref().clone(),
addr: ready_addr,
},
);
ready
.send(ready_addr)
.map_err(|_| eyre::eyre!("network manager stopped before server readiness"))?;
loop {
let connection = tokio::select! {
() = ctx.shutdown.cancelled() => return Ok(()),
connection = server.accept() => connection,
};
let server_result = match AssertUnwindSafe(async {
loop {
tokio::select! {
biased;
() = ctx.shutdown.cancelled() => break Ok(()),
result = mdns_tasks.join_next(), if !mdns_tasks.is_empty() => {
log_joined_child_result(
"mDNS monitor",
result.expect("non-empty child set"),
);
log::warn!("mDNS monitor ended while the QUIC server is still running");
}
result = connection_tasks.join_next(), if !connection_tasks.is_empty() => {
log_joined_child_result(
"peer connection",
result.expect("non-empty child set"),
);
}
connection = server.accept() => {
let Some(connection) = connection else {
break Err(eyre::eyre!("QUIC server accept loop ended unexpectedly"));
};
let Some(connection) = connection else {
eyre::bail!("QUIC server accept loop ended unexpectedly");
};
if !has_child_capacity(connection_tasks.len(), MAX_ESTABLISHED_CONNECTIONS) {
log::warn!(
"Closing excess peer connection from {} at application limit {}",
connection.remote_addr().map_or_else(
|_| "unknown".to_owned(),
|addr| addr.to_string(),
),
MAX_ESTABLISHED_CONNECTIONS,
);
connection.close(application::Error::UNKNOWN);
continue;
}
let ctx = ctx.clone();
let tx_notify_ui = tx_notify_ui.clone();
let task_tracker = ctx.task_tracker.clone();
task_tracker.spawn(async move {
if let Err(err) = handle_peer_connection(connection, ctx, tx_notify_ui).await {
log::error!("Peer connection error: {err}");
connection_tasks.spawn(handle_peer_connection(
connection,
ctx.clone(),
server_children_shutdown.clone(),
Arc::clone(&control_stream_permits),
Arc::clone(&bulk_transfer_permits),
));
}
}
});
}
})
.catch_unwind()
.await
{
Ok(result) => result,
Err(payload) => Err(eyre::eyre!(
"QUIC server loop panicked: {}",
panic_payload_to_string(payload.as_ref())
)),
};
// Stop future work before dropping the accept owner. Connection children
// cooperatively observe this token and close their stream scopes before
// they return.
server_children_shutdown.cancel();
drop(server);
drain_joined_child_tasks(&mut connection_tasks, "peer connection").await;
drain_joined_child_tasks(&mut mdns_tasks, "mDNS monitor").await;
let mdns_close_result = close_mdns_advertiser(mdns_advertiser);
combine_server_results(server_result, mdns_close_result, Ok(()))
}
async fn run_body_with_cleanup<Body, Cleanup>(body: Body, cleanup: Cleanup) -> eyre::Result<()>
where
Body: Future<Output = eyre::Result<()>>,
Cleanup: Future<Output = eyre::Result<()>>,
{
let body_result = match AssertUnwindSafe(body).catch_unwind().await {
Ok(result) => result,
Err(payload) => Err(eyre::eyre!(
"QUIC server body panicked: {}",
panic_payload_to_string(payload.as_ref())
)),
};
// This is the unconditional endpoint-owner epilogue. Even an unexpected
// panic during setup, serving, or descendant cleanup cannot skip the join.
let endpoint_result = cleanup.await;
combine_server_results(body_result, Ok(()), endpoint_result)
}
fn combine_server_results(
server: eyre::Result<()>,
mdns: eyre::Result<()>,
endpoint: eyre::Result<()>,
) -> eyre::Result<()> {
let mut errors = Vec::new();
if let Err(error) = server {
errors.push(format!("QUIC server failed: {error:#}"));
}
if let Err(error) = mdns {
errors.push(format!("mDNS advertiser shutdown failed: {error:#}"));
}
if let Err(error) = endpoint {
errors.push(format!("QUIC endpoint shutdown failed: {error:#}"));
}
if errors.is_empty() {
Ok(())
} else {
Err(eyre::eyre!(errors.join("; ")))
}
}
fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String {
if let Some(message) = payload.downcast_ref::<&'static str>() {
return (*message).to_string();
}
if let Some(message) = payload.downcast_ref::<String>() {
return message.clone();
}
"unknown panic payload".to_string()
}
fn direct_connect_addr(server_addr: SocketAddr) -> SocketAddr {
@@ -87,31 +262,417 @@ fn direct_connect_addr(server_addr: SocketAddr) -> SocketAddr {
async fn handle_peer_connection(
mut connection: Connection,
ctx: PeerCtx,
tx_notify_ui: UnboundedSender<PeerEvent>,
server_shutdown: CancellationToken,
control_stream_permits: Arc<Semaphore>,
bulk_transfer_permits: Arc<Semaphore>,
) -> eyre::Result<()> {
let remote_addr = connection.remote_addr()?;
log::info!("{remote_addr} peer connected");
events::send(&tx_notify_ui, PeerEvent::PeerConnected(remote_addr));
loop {
let stream = tokio::select! {
() = ctx.shutdown.cancelled() => break,
stream = connection.accept_bidirectional_stream() => stream,
};
let Some(stream) = stream? else {
break;
};
let ctx = ctx.clone();
let task_tracker = ctx.task_tracker.clone();
task_tracker.spawn(async move {
if let Err(err) = handle_peer_stream(stream, ctx, Some(remote_addr)).await {
log::error!("{remote_addr:?} peer stream error: {err}");
let connection_shutdown = server_shutdown.child_token();
let mut stream_tasks = JoinSet::new();
let connection_result = match AssertUnwindSafe(async {
loop {
tokio::select! {
biased;
() = connection_shutdown.cancelled() => break Ok(()),
result = stream_tasks.join_next(), if !stream_tasks.is_empty() => {
log_joined_child_result(
&format!("{remote_addr} peer stream"),
result.expect("non-empty child set"),
);
}
() = tokio::time::sleep(CONNECTION_NO_STREAM_IDLE_TIMEOUT),
if stream_tasks.is_empty() => {
log::debug!(
"Closing idle peer connection from {remote_addr} after {CONNECTION_NO_STREAM_IDLE_TIMEOUT:?}"
);
break Ok(());
}
stream = connection.accept_bidirectional_stream() => {
match stream {
Ok(Some(mut stream)) => {
if !has_child_capacity(stream_tasks.len(), MAX_CONTROL_STREAM_TASKS) {
let _ = stream.stop_sending(application::Error::UNKNOWN);
let _ = stream.reset(application::Error::UNKNOWN);
continue;
}
let Ok(control_permit) = Arc::clone(&control_stream_permits)
.try_acquire_owned()
else {
let _ = stream.stop_sending(application::Error::UNKNOWN);
let _ = stream.reset(application::Error::UNKNOWN);
continue;
};
let stream_ctx = ctx.clone();
let stream_shutdown = connection_shutdown.child_token();
stream_tasks.spawn(handle_admitted_peer_stream(
stream,
stream_ctx,
Some(remote_addr),
stream_shutdown,
control_permit,
Arc::clone(&bulk_transfer_permits),
));
}
Ok(None) => break Ok(()),
Err(error) => break Err(error.into()),
}
}
}
});
}
})
.catch_unwind()
.await
{
Ok(result) => result,
Err(payload) => Err(eyre::eyre!(
"{remote_addr} peer connection scope panicked: {}",
panic_payload_to_string(payload.as_ref())
)),
};
// Cancel the connection-local scope before closing its QUIC owner. Every
// accepted stream and outbound transfer derives from this token, so child
// futures can settle without waiting for process-wide shutdown. Closing
// the connection also wakes any transport operation already in progress.
connection_shutdown.cancel();
connection.close(0u32.into());
drop(connection);
let stream_label = format!("{remote_addr} peer stream");
drain_joined_child_tasks(&mut stream_tasks, &stream_label).await;
log::info!("{remote_addr} peer disconnected");
connection_result
}
async fn handle_admitted_peer_stream(
stream: BidirectionalStream,
ctx: PeerCtx,
remote_addr: Option<SocketAddr>,
stream_shutdown: CancellationToken,
control_permit: OwnedSemaphorePermit,
bulk_transfer_permits: Arc<Semaphore>,
) -> eyre::Result<()> {
handle_peer_stream(
stream,
ctx,
remote_addr,
stream_shutdown,
control_permit,
bulk_transfer_permits,
)
.await
}
const fn has_child_capacity(active: usize, limit: usize) -> bool {
active < limit
}
fn log_child_result(label: &str, result: eyre::Result<()>) {
if let Err(error) = result {
log::error!("{label} error: {error}");
}
}
fn log_joined_child_result(label: &str, result: Result<eyre::Result<()>, tokio::task::JoinError>) {
match result {
Ok(result) => log_child_result(label, result),
Err(error) => log::error!("{label} task failed: {error}"),
}
}
async fn drain_joined_child_tasks(children: &mut JoinSet<eyre::Result<()>>, label: &str) {
while let Some(result) = children.join_next().await {
log_joined_child_result(label, result);
}
}
#[cfg(test)]
mod tests {
use std::{
sync::{
Arc,
atomic::{AtomicUsize, Ordering},
},
time::Duration,
};
use tokio::{
sync::{Semaphore, mpsc},
task::JoinSet,
};
use tokio_util::sync::CancellationToken;
use super::{
CONNECTION_NO_STREAM_IDLE_TIMEOUT,
MAX_CONTROL_STREAM_TASKS,
MAX_ESTABLISHED_CONNECTIONS,
MAX_GLOBAL_BULK_TRANSFER_TASKS,
MAX_GLOBAL_CONTROL_STREAM_TASKS,
MAX_INFLIGHT_HANDSHAKES,
bounded_endpoint_limits,
drain_joined_child_tasks,
endpoint_connection_capacity_available,
has_child_capacity,
run_body_with_cleanup,
};
#[test]
fn unauthenticated_runtime_bounds_are_explicit_and_closed_at_capacity() {
assert_eq!(MAX_INFLIGHT_HANDSHAKES, 64);
assert_eq!(MAX_ESTABLISHED_CONNECTIONS, 64);
assert_eq!(MAX_CONTROL_STREAM_TASKS, 32);
assert_eq!(MAX_GLOBAL_CONTROL_STREAM_TASKS, 64);
assert_eq!(MAX_GLOBAL_BULK_TRANSFER_TASKS, 48);
assert_eq!(
u64::try_from(MAX_CONTROL_STREAM_TASKS).expect("stream task bound fits u64"),
crate::quic_runtime::MAX_OPEN_BIDIRECTIONAL_STREAMS,
);
assert!(has_child_capacity(63, MAX_ESTABLISHED_CONNECTIONS));
assert!(!has_child_capacity(64, MAX_ESTABLISHED_CONNECTIONS));
assert!(!has_child_capacity(32, MAX_CONTROL_STREAM_TASKS));
}
events::send(&tx_notify_ui, PeerEvent::PeerDisconnected(remote_addr));
Ok(())
#[test]
fn endpoint_limiter_rejects_before_the_internal_accept_queue_can_exceed_the_cap() {
bounded_endpoint_limits().expect("endpoint limiter should build");
assert!(endpoint_connection_capacity_available(
MAX_ESTABLISHED_CONNECTIONS - 1
));
assert!(!endpoint_connection_capacity_available(
MAX_ESTABLISHED_CONNECTIONS
));
}
#[test]
fn global_control_permit_saturates_and_releases_across_scopes() {
let permits = Arc::new(Semaphore::new(MAX_GLOBAL_CONTROL_STREAM_TASKS));
let held = (0..MAX_GLOBAL_CONTROL_STREAM_TASKS)
.map(|_| {
Arc::clone(&permits)
.try_acquire_owned()
.expect("configured global permit should be available")
})
.collect::<Vec<_>>();
assert!(Arc::clone(&permits).try_acquire_owned().is_err());
drop(held);
assert!(Arc::clone(&permits).try_acquire_owned().is_ok());
}
#[test]
fn saturated_bulk_pool_preserves_control_plane_permits() {
let control = Arc::new(Semaphore::new(MAX_GLOBAL_CONTROL_STREAM_TASKS));
let bulk = Arc::new(Semaphore::new(MAX_GLOBAL_BULK_TRANSFER_TASKS));
let held_bulk = (0..MAX_GLOBAL_BULK_TRANSFER_TASKS)
.map(|_| {
Arc::clone(&bulk)
.try_acquire_owned()
.expect("configured bulk permit should be available")
})
.collect::<Vec<_>>();
assert!(Arc::clone(&bulk).try_acquire_owned().is_err());
assert!(
Arc::clone(&control).try_acquire_owned().is_ok(),
"bulk saturation must not consume control-plane admission"
);
drop(held_bulk);
}
#[tokio::test]
async fn draining_waits_for_every_childs_natural_cleanup() {
let owner_closed = CancellationToken::new();
let cleanup_release = CancellationToken::new();
let completed = Arc::new(AtomicUsize::new(0));
let (started_tx, mut started_rx) = mpsc::unbounded_channel();
let mut children = JoinSet::new();
for child_id in 0..2 {
let owner_closed = owner_closed.clone();
let cleanup_release = cleanup_release.clone();
let completed = completed.clone();
let started_tx = started_tx.clone();
children.spawn(async move {
started_tx
.send(child_id)
.expect("test observer should remain available");
owner_closed.cancelled().await;
cleanup_release.cancelled().await;
completed.fetch_add(1, Ordering::SeqCst);
Ok(())
});
}
drop(started_tx);
let drain_task = tokio::spawn(async move {
let mut children = children;
drain_joined_child_tasks(&mut children, "synthetic child").await;
});
for _ in 0..2 {
tokio::time::timeout(Duration::from_secs(1), started_rx.recv())
.await
.expect("child should start")
.expect("start channel should remain open");
}
owner_closed.cancel();
tokio::task::yield_now().await;
assert_eq!(completed.load(Ordering::SeqCst), 0);
assert!(!drain_task.is_finished(), "drain must await child cleanup");
cleanup_release.cancel();
tokio::time::timeout(Duration::from_secs(1), drain_task)
.await
.expect("drain should finish after cleanup is released")
.expect("drain task should not panic");
assert_eq!(completed.load(Ordering::SeqCst), 2);
}
#[tokio::test]
async fn one_panicking_child_does_not_skip_a_siblings_cleanup() {
let cleanup_release = CancellationToken::new();
let completed = Arc::new(AtomicUsize::new(0));
let mut children = JoinSet::new();
children.spawn(async {
panic!("injected child panic");
#[allow(unreachable_code)]
Ok(())
});
let child_release = cleanup_release.clone();
let child_completed = completed.clone();
children.spawn(async move {
child_release.cancelled().await;
child_completed.fetch_add(1, Ordering::SeqCst);
Ok(())
});
let drain_task = tokio::spawn(async move {
drain_joined_child_tasks(&mut children, "synthetic joined child").await;
});
tokio::task::yield_now().await;
assert!(!drain_task.is_finished());
cleanup_release.cancel();
tokio::time::timeout(Duration::from_secs(1), drain_task)
.await
.expect("drain should await the surviving child")
.expect("drain task should not panic");
assert_eq!(completed.load(Ordering::SeqCst), 1);
}
#[tokio::test(start_paused = true)]
async fn an_established_connection_without_streams_has_a_finite_application_idle_bound() {
let started = tokio::time::Instant::now();
tokio::time::sleep(CONNECTION_NO_STREAM_IDLE_TIMEOUT).await;
assert_eq!(started.elapsed(), CONNECTION_NO_STREAM_IDLE_TIMEOUT);
}
#[tokio::test]
async fn server_body_panic_still_awaits_owner_cleanup() {
let cleanup_finished = Arc::new(AtomicUsize::new(0));
let cleanup_probe = cleanup_finished.clone();
let result = run_body_with_cleanup(
async {
panic!("injected server body panic");
#[allow(unreachable_code)]
Ok(())
},
async move {
tokio::task::yield_now().await;
cleanup_probe.store(1, Ordering::SeqCst);
Ok(())
},
)
.await;
assert!(result.is_err());
assert_eq!(cleanup_finished.load(Ordering::SeqCst), 1);
}
#[tokio::test]
async fn nested_scopes_publish_completion_from_children_outward() {
let global_shutdown = CancellationToken::new();
let server_shutdown = global_shutdown.child_token();
let stream_cleanup_release = CancellationToken::new();
let (started_tx, mut started_rx) = mpsc::unbounded_channel();
let (order_tx, mut order_rx) = mpsc::unbounded_channel();
let mut connections = JoinSet::new();
let connection_shutdown = server_shutdown.child_token();
let connection_stream_cleanup_release = stream_cleanup_release.clone();
let connection_order_tx = order_tx.clone();
connections.spawn(async move {
let mut streams = JoinSet::new();
for label in ["stream-a", "stream-b"] {
let shutdown = connection_shutdown.child_token();
let cleanup_release = connection_stream_cleanup_release.clone();
let started_tx = started_tx.clone();
let order_tx = connection_order_tx.clone();
streams.spawn(async move {
started_tx
.send(label)
.expect("test observer should remain available");
shutdown.cancelled().await;
cleanup_release.cancelled().await;
order_tx
.send(label)
.expect("order observer should remain available");
Ok(())
});
}
connection_shutdown.cancelled().await;
drain_joined_child_tasks(&mut streams, "synthetic stream").await;
connection_order_tx
.send("peer-disconnected")
.expect("order observer should remain available");
Ok(())
});
let hierarchy_task = tokio::spawn(async move {
server_shutdown.cancel();
drain_joined_child_tasks(&mut connections, "synthetic connection").await;
order_tx
.send("server-return")
.expect("order observer should remain available");
});
for _ in 0..2 {
tokio::time::timeout(Duration::from_secs(1), started_rx.recv())
.await
.expect("stream should start")
.expect("start channel should remain open");
}
assert!(!hierarchy_task.is_finished());
stream_cleanup_release.cancel();
tokio::time::timeout(Duration::from_secs(1), hierarchy_task)
.await
.expect("hierarchy should drain")
.expect("hierarchy task should not panic");
let mut order = Vec::new();
while let Ok(event) = order_rx.try_recv() {
order.push(event);
}
let disconnected = order
.iter()
.position(|event| *event == "peer-disconnected")
.expect("disconnect should be published");
let server_return = order
.iter()
.position(|event| *event == "server-return")
.expect("server return should be published");
assert!(
order[..disconnected]
.iter()
.all(|event| event.starts_with("stream-"))
);
assert_eq!(disconnected, 2);
assert_eq!(server_return, 3);
assert!(
!global_shutdown.is_cancelled(),
"server-local shutdown must not cancel its runtime parent"
);
}
}
@@ -0,0 +1,727 @@
//! Bounded scheduling for responder-owned state pulls and change-hint fanout.
use std::{collections::HashMap, future::Future, pin::Pin, sync::Arc, time::Duration};
use futures::{StreamExt as _, stream::FuturesUnordered};
use lanspread_proto::{ChangeHint, PeerId};
use tokio::sync::{Mutex, mpsc, watch};
use tokio_util::sync::CancellationToken;
use crate::{
PeerEvent,
context::NetworkServiceCtx,
network::{send_call_to_play_changed, send_library_changed},
peer_db::PeerRevisionSnapshot,
services::{HandshakeCtx, PeerRefreshOutcome, perform_peer_refresh},
};
const SYNC_QUEUE_CAPACITY: usize = 64;
const MAX_TRACKED_PEERS: usize = 64;
const MAX_CONCURRENT_PULLS: usize = 8;
const MAX_CONCURRENT_HINT_SENDS: usize = 8;
pub(crate) const PEER_PULL_COALESCE_WINDOW: Duration = Duration::from_secs(5);
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(crate) enum StateDomain {
Library,
CallToPlay,
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
struct LocalRevisions {
library: u64,
call_to_play: u64,
}
#[derive(Clone, Copy, Debug)]
struct HintTrigger {
domain: StateDomain,
hint: ChangeHint,
}
struct StateSyncInbox {
pinned_rx: Mutex<mpsc::Receiver<PeerId>>,
hint_rx: Mutex<mpsc::Receiver<HintTrigger>>,
}
/// Non-blocking ingress for untrusted hints and latest-value local revisions.
///
/// Pinned Pong mismatches use a separate bounded queue and an async admission
/// method so hostile hint saturation cannot displace reconciliation work.
#[derive(Clone)]
pub(crate) struct StateSyncHandle {
local_peer_id: PeerId,
pinned_tx: mpsc::Sender<PeerId>,
hint_tx: mpsc::Sender<HintTrigger>,
local_revisions_tx: watch::Sender<LocalRevisions>,
inbox: Arc<StateSyncInbox>,
}
impl StateSyncHandle {
#[must_use]
pub(crate) fn new(local_peer_id: PeerId) -> Self {
let (pinned_tx, pinned_rx) = mpsc::channel(SYNC_QUEUE_CAPACITY);
let (hint_tx, hint_rx) = mpsc::channel(SYNC_QUEUE_CAPACITY);
let (local_revisions_tx, _) = watch::channel(LocalRevisions {
library: 0,
call_to_play: 0,
});
Self {
local_peer_id,
pinned_tx,
hint_tx,
local_revisions_tx,
inbox: Arc::new(StateSyncInbox {
pinned_rx: Mutex::new(pinned_rx),
hint_rx: Mutex::new(hint_rx),
}),
}
}
/// Enqueues a generation-current pinned Pong mismatch without sharing the
/// untrusted hint queue.
pub(crate) async fn schedule_pinned_pull(
&self,
peer_id: PeerId,
cancellation: &CancellationToken,
) -> eyre::Result<()> {
if peer_id == self.local_peer_id {
return Ok(());
}
tokio::select! {
biased;
() = cancellation.cancelled() => eyre::bail!("state-sync admission was cancelled"),
result = self.pinned_tx.send(peer_id) => {
result.map_err(|_| eyre::eyre!("state-sync pinned queue is closed"))
}
}
}
/// Treats an inbound message only as a lossy invalidation hint.
pub(crate) fn schedule_hint(&self, domain: StateDomain, hint: ChangeHint) {
if hint.claimed_peer_id == self.local_peer_id {
return;
}
match self.hint_tx.try_send(HintTrigger { domain, hint }) {
Ok(()) => {}
Err(mpsc::error::TrySendError::Full(_)) => {
log::trace!("Coalescing remote state hint because the bounded queue is full");
}
Err(mpsc::error::TrySendError::Closed(_)) => {
log::debug!("Ignoring remote state hint because state sync is stopped");
}
}
}
pub(crate) fn publish_library_revision(&self, revision: u64) {
self.local_revisions_tx.send_if_modified(|current| {
if revision <= current.library {
return false;
}
current.library = revision;
true
});
}
pub(crate) fn publish_call_to_play_revision(&self, revision: u64) {
self.local_revisions_tx.send_if_modified(|current| {
if revision <= current.call_to_play {
return false;
}
current.call_to_play = revision;
true
});
}
}
#[derive(Debug)]
struct PeerSlot {
in_flight: bool,
pending: bool,
pinned: bool,
next_allowed: tokio::time::Instant,
}
struct PullCompletion {
peer_id: PeerId,
result: eyre::Result<PeerRefreshOutcome>,
}
type PullFuture = Pin<Box<dyn Future<Output = PullCompletion> + Send>>;
type FanoutFuture = Pin<Box<dyn Future<Output = ()> + Send>>;
/// Runs the bounded pull scheduler and local hint fanout in one lexical scope.
pub(crate) async fn run_state_sync(
ctx: NetworkServiceCtx,
tx_notify_ui: tokio::sync::mpsc::UnboundedSender<PeerEvent>,
cancellation: CancellationToken,
) -> eyre::Result<()> {
let mut pinned_rx = ctx.state_sync.inbox.pinned_rx.lock().await;
let mut hint_rx = ctx.state_sync.inbox.hint_rx.lock().await;
let mut local_revisions_rx = ctx.state_sync.local_revisions_tx.subscribe();
let mut last_fanout = *local_revisions_rx.borrow_and_update();
let mut fanout: Option<FanoutFuture> = None;
let mut fanout_pending = false;
// One actor-owned pinned item preserves backpressure when all tracked
// slots are temporarily non-replaceable. While occupied, the bounded
// pinned channel is deliberately not drained.
let mut blocked_pinned = None;
let mut slots = HashMap::<PeerId, PeerSlot>::new();
let mut pulls = FuturesUnordered::<PullFuture>::new();
let work_cancellation = cancellation.child_token();
let handshake_ctx = HandshakeCtx::from_network(&ctx, &tx_notify_ui)
.with_cancellation(work_cancellation.clone());
loop {
expire_idle_slots(&mut slots);
if let Some(peer_id) = blocked_pinned
&& try_enqueue_peer(&mut slots, peer_id, true)
{
blocked_pinned = None;
}
start_ready_pulls(&mut slots, &mut pulls, &handshake_ctx);
if fanout.is_none() && fanout_pending {
let target = *local_revisions_rx.borrow_and_update();
fanout_pending = false;
if target != last_fanout {
fanout = Some(Box::pin(fanout_local_hints(
ctx.clone(),
last_fanout,
target,
work_cancellation.clone(),
)));
last_fanout = target;
}
}
let deadline = next_slot_deadline(&slots);
tokio::select! {
biased;
() = cancellation.cancelled() => break,
peer_id = pinned_rx.recv(), if blocked_pinned.is_none() => {
let Some(peer_id) = peer_id else { break; };
if !try_enqueue_peer(&mut slots, peer_id, true) {
blocked_pinned = Some(peer_id);
}
}
completed = pulls.next(), if !pulls.is_empty() => {
if let Some(completed) = completed
&& let Some(slot) = slots.get_mut(&completed.peer_id) {
settle_pull_completion(slot, completed.peer_id, completed.result);
}
}
() = async {
if let Some(fanout) = fanout.as_mut() {
fanout.await;
}
}, if fanout.is_some() => {
fanout = None;
if *local_revisions_rx.borrow() != last_fanout {
fanout_pending = true;
}
}
changed = local_revisions_rx.changed() => {
if changed.is_err() {
break;
}
fanout_pending = true;
}
trigger = hint_rx.recv() => {
let Some(trigger) = trigger else { break; };
if hint_requires_pull(&ctx, trigger).await {
let _ = try_enqueue_peer(
&mut slots,
trigger.hint.claimed_peer_id,
false,
);
}
}
() = wait_for_deadline(deadline), if deadline.is_some() => {}
}
}
work_cancellation.cancel();
drain_state_sync_children(&mut pulls, fanout).await;
Ok(())
}
fn settle_pull_completion(
slot: &mut PeerSlot,
peer_id: PeerId,
result: eyre::Result<PeerRefreshOutcome>,
) {
slot.in_flight = false;
match result {
Ok(PeerRefreshOutcome::DeferredByCandidate) => {
// Candidate ownership is temporary authority, not a completed
// refresh. Retain exactly one coalesced retry behind the existing
// rate gate.
slot.pending = true;
slot.pinned = true;
}
Ok(PeerRefreshOutcome::Completed | PeerRefreshOutcome::Stale) => {}
Err(error) => log::warn!("Failed to refresh peer {peer_id}: {error:#}"),
}
}
async fn drain_state_sync_children(
pulls: &mut FuturesUnordered<PullFuture>,
fanout: Option<FanoutFuture>,
) {
while let Some(completed) = pulls.next().await {
if let Err(error) = completed.result {
log::debug!("Peer refresh stopped during state-sync shutdown: {error:#}");
}
}
if let Some(fanout) = fanout {
fanout.await;
}
}
/// Returns whether the trigger was coalesced into a tracked slot. A pinned
/// trigger that returns false must remain actor-owned so bounded-channel
/// backpressure is preserved until a slot becomes replaceable.
fn try_enqueue_peer(slots: &mut HashMap<PeerId, PeerSlot>, peer_id: PeerId, pinned: bool) -> bool {
if let Some(slot) = slots.get_mut(&peer_id) {
slot.pending = true;
slot.pinned |= pinned;
return true;
}
if slots.len() >= MAX_TRACKED_PEERS {
if !pinned {
return false;
}
let replaceable = slots
.iter()
.find_map(|(id, slot)| (!slot.in_flight && !slot.pinned).then_some(*id));
let Some(replaceable) = replaceable else {
return false;
};
slots.remove(&replaceable);
}
slots.insert(
peer_id,
PeerSlot {
in_flight: false,
pending: true,
pinned,
next_allowed: tokio::time::Instant::now(),
},
);
true
}
fn start_ready_pulls(
slots: &mut HashMap<PeerId, PeerSlot>,
pulls: &mut FuturesUnordered<PullFuture>,
handshake_ctx: &HandshakeCtx,
) {
while pulls.len() < MAX_CONCURRENT_PULLS {
let Some(peer_id) = take_ready_peer(slots, tokio::time::Instant::now()) else {
break;
};
let ctx = handshake_ctx.clone();
pulls.push(Box::pin(async move {
let result = perform_refresh_for_peer(ctx, peer_id).await;
PullCompletion { peer_id, result }
}));
}
}
fn take_ready_peer(
slots: &mut HashMap<PeerId, PeerSlot>,
now: tokio::time::Instant,
) -> Option<PeerId> {
let peer_id = slots
.iter()
.filter(|(_, slot)| slot.pending && !slot.in_flight && slot.next_allowed <= now)
.max_by_key(|(_, slot)| slot.pinned)
.map(|(peer_id, _)| *peer_id)?;
let slot = slots
.get_mut(&peer_id)
.expect("selected state-sync slot must still exist");
slot.in_flight = true;
slot.pending = false;
slot.pinned = false;
slot.next_allowed = now + PEER_PULL_COALESCE_WINDOW;
Some(peer_id)
}
async fn perform_refresh_for_peer(
ctx: HandshakeCtx,
peer_id: PeerId,
) -> eyre::Result<PeerRefreshOutcome> {
let snapshot = ctx.peer_liveness_for(peer_id).await;
let Some(snapshot) = snapshot else {
return Ok(PeerRefreshOutcome::Stale);
};
perform_peer_refresh(ctx, snapshot).await
}
async fn hint_requires_pull(ctx: &NetworkServiceCtx, trigger: HintTrigger) -> bool {
let snapshot = ctx
.peer_game_db
.read()
.await
.revision_snapshot(&trigger.hint.claimed_peer_id);
hint_requires_pull_from_snapshot(ctx.peer_id, trigger, snapshot.as_ref())
}
fn hint_requires_pull_from_snapshot(
local_peer_id: PeerId,
trigger: HintTrigger,
snapshot: Option<&PeerRevisionSnapshot>,
) -> bool {
if trigger.hint.claimed_peer_id == local_peer_id {
return false;
}
let Some(snapshot) = snapshot else {
return false;
};
if snapshot.runtime_session_id != trigger.hint.runtime_session_id {
return true;
}
match trigger.domain {
StateDomain::Library => snapshot.library_revision != Some(trigger.hint.revision),
StateDomain::CallToPlay => snapshot.call_to_play_revision != Some(trigger.hint.revision),
}
}
fn expire_idle_slots(slots: &mut HashMap<PeerId, PeerSlot>) {
let now = tokio::time::Instant::now();
slots.retain(|_, slot| slot.in_flight || slot.pending || slot.next_allowed > now);
}
fn next_slot_deadline(slots: &HashMap<PeerId, PeerSlot>) -> Option<tokio::time::Instant> {
let now = tokio::time::Instant::now();
slots
.values()
.filter(|slot| !slot.in_flight && slot.next_allowed > now)
.map(|slot| slot.next_allowed)
.min()
}
async fn wait_for_deadline(deadline: Option<tokio::time::Instant>) {
if let Some(deadline) = deadline {
tokio::time::sleep_until(deadline).await;
}
}
async fn fanout_local_hints(
ctx: NetworkServiceCtx,
previous: LocalRevisions,
target: LocalRevisions,
cancellation: CancellationToken,
) {
let endpoints = ctx.peer_game_db.read().await.peer_endpoints();
let local_peer_id = ctx.peer_id;
let runtime_session_id = ctx.runtime_session_id;
let send_library = target.library != previous.library;
let send_call_to_play = target.call_to_play != previous.call_to_play;
let deliveries = endpoints.into_iter().map(|endpoint| {
let quic = ctx.quic.clone();
let cancellation = cancellation.clone();
async move {
if send_library {
let hint = ChangeHint {
claimed_peer_id: local_peer_id,
runtime_session_id,
revision: target.library,
};
if let Err(error) =
send_library_changed(&quic, &endpoint, hint, &cancellation).await
{
log::debug!(
"Failed to send library hint to {}: {error:#}",
endpoint.addr
);
}
}
if send_call_to_play {
let hint = ChangeHint {
claimed_peer_id: local_peer_id,
runtime_session_id,
revision: target.call_to_play,
};
if let Err(error) =
send_call_to_play_changed(&quic, &endpoint, hint, &cancellation).await
{
log::debug!(
"Failed to send Call-to-Play hint to {}: {error:#}",
endpoint.addr
);
}
}
}
});
drive_bounded_hint_fanout(deliveries).await;
}
async fn drive_bounded_hint_fanout<F>(deliveries: impl IntoIterator<Item = F>)
where
F: Future<Output = ()>,
{
let mut deliveries =
futures::stream::iter(deliveries).buffer_unordered(MAX_CONCURRENT_HINT_SENDS);
while deliveries.next().await.is_some() {}
}
#[cfg(test)]
mod tests {
use std::sync::atomic::{AtomicUsize, Ordering};
use lanspread_proto::RuntimeSessionId;
use super::*;
fn peer(seed: u8) -> PeerId {
PeerId::from_bytes([seed; 32])
}
fn session(seed: u8) -> RuntimeSessionId {
RuntimeSessionId::from_bytes([seed; 16])
}
#[test]
fn repeated_triggers_coalesce_to_one_pending_follow_up() {
let mut slots = HashMap::new();
assert!(try_enqueue_peer(&mut slots, peer(1), false));
let now = tokio::time::Instant::now();
assert_eq!(take_ready_peer(&mut slots, now), Some(peer(1)));
for _ in 0..100 {
assert!(try_enqueue_peer(&mut slots, peer(1), false));
}
assert_eq!(slots.len(), 1);
assert!(slots[&peer(1)].pending);
slots
.get_mut(&peer(1))
.expect("slot should exist")
.in_flight = false;
assert_eq!(
take_ready_peer(
&mut slots,
now + PEER_PULL_COALESCE_WINDOW - Duration::from_millis(1)
),
None
);
assert_eq!(
take_ready_peer(&mut slots, now + PEER_PULL_COALESCE_WINDOW),
Some(peer(1))
);
assert!(!slots[&peer(1)].pending);
slots
.get_mut(&peer(1))
.expect("slot should exist")
.in_flight = false;
assert_eq!(
take_ready_peer(&mut slots, now + PEER_PULL_COALESCE_WINDOW),
None,
"completion or failure must not self-retry"
);
}
#[test]
fn candidate_deferral_rearms_exactly_one_rate_limited_pinned_retry() {
let peer_id = peer(1);
let mut slots = HashMap::new();
assert!(try_enqueue_peer(&mut slots, peer_id, true));
let now = tokio::time::Instant::now();
assert_eq!(take_ready_peer(&mut slots, now), Some(peer_id));
settle_pull_completion(
slots.get_mut(&peer_id).expect("slot should remain"),
peer_id,
Ok(PeerRefreshOutcome::DeferredByCandidate),
);
assert!(slots[&peer_id].pending);
assert!(slots[&peer_id].pinned);
assert_eq!(slots.len(), 1);
assert_eq!(
take_ready_peer(
&mut slots,
now + PEER_PULL_COALESCE_WINDOW - Duration::from_millis(1),
),
None
);
assert_eq!(
take_ready_peer(&mut slots, now + PEER_PULL_COALESCE_WINDOW),
Some(peer_id)
);
settle_pull_completion(
slots.get_mut(&peer_id).expect("slot should remain"),
peer_id,
Ok(PeerRefreshOutcome::Completed),
);
assert!(!slots[&peer_id].pending);
assert!(!slots[&peer_id].pinned);
assert_eq!(
take_ready_peer(&mut slots, now + PEER_PULL_COALESCE_WINDOW),
None
);
}
#[test]
fn pinned_trigger_displaces_only_idle_untrusted_slot_at_capacity() {
let mut slots = HashMap::new();
for seed in 0..u8::try_from(MAX_TRACKED_PEERS).expect("bound fits u8") {
assert!(try_enqueue_peer(&mut slots, peer(seed), false));
}
assert!(try_enqueue_peer(&mut slots, peer(200), true));
assert_eq!(slots.len(), MAX_TRACKED_PEERS);
assert!(slots.contains_key(&peer(200)));
assert!(slots[&peer(200)].pinned);
assert_eq!(
take_ready_peer(&mut slots, tokio::time::Instant::now()),
Some(peer(200)),
"pinned work must start before saturated untrusted work"
);
}
#[test]
fn pinned_trigger_waits_when_every_slot_is_nonreplaceable() {
let mut slots = HashMap::new();
for seed in 0..u8::try_from(MAX_TRACKED_PEERS).expect("bound fits u8") {
assert!(try_enqueue_peer(&mut slots, peer(seed), true));
}
let blocked = peer(200);
assert!(!try_enqueue_peer(&mut slots, blocked, true));
assert_eq!(slots.len(), MAX_TRACKED_PEERS);
assert!(!slots.contains_key(&blocked));
let completed = peer(0);
let slot = slots
.get_mut(&completed)
.expect("tracked pinned slot should exist");
slot.pinned = false;
slot.pending = false;
slot.in_flight = false;
assert!(try_enqueue_peer(&mut slots, blocked, true));
assert_eq!(slots.len(), MAX_TRACKED_PEERS);
assert!(slots.contains_key(&blocked));
assert!(!slots.contains_key(&completed));
}
#[test]
fn quiet_cooldown_slots_have_an_expiry_deadline_before_new_hint_admission() {
let now = tokio::time::Instant::now();
let mut slots = HashMap::new();
for seed in 0..u8::try_from(MAX_TRACKED_PEERS).expect("bound fits u8") {
slots.insert(
peer(seed),
PeerSlot {
in_flight: false,
pending: false,
pinned: false,
next_allowed: now + PEER_PULL_COALESCE_WINDOW,
},
);
}
assert_eq!(
next_slot_deadline(&slots),
Some(now + PEER_PULL_COALESCE_WINDOW)
);
}
#[test]
fn scheduler_starts_at_most_eight_of_sixty_four_ready_peers() {
let mut slots = HashMap::new();
for seed in 0..u8::try_from(MAX_TRACKED_PEERS).expect("bound fits u8") {
assert!(try_enqueue_peer(&mut slots, peer(seed), false));
}
let now = tokio::time::Instant::now();
let started = (0..MAX_CONCURRENT_PULLS)
.filter_map(|_| take_ready_peer(&mut slots, now))
.collect::<Vec<_>>();
assert_eq!(started.len(), MAX_CONCURRENT_PULLS);
assert_eq!(slots.values().filter(|slot| slot.in_flight).count(), 8);
assert_eq!(slots.values().filter(|slot| slot.pending).count(), 56);
assert_eq!(
next_slot_deadline(&slots),
None,
"ready work waiting on the global pull bound must be completion-driven"
);
}
#[test]
fn unknown_and_self_hints_do_not_allocate_slots() {
let local = peer(1);
let mut slots = HashMap::new();
for claimed_peer_id in [local, peer(2)] {
let trigger = HintTrigger {
domain: StateDomain::Library,
hint: ChangeHint {
claimed_peer_id,
runtime_session_id: session(1),
revision: 99,
},
};
if hint_requires_pull_from_snapshot(local, trigger, None) {
let _ = try_enqueue_peer(&mut slots, claimed_peer_id, false);
}
}
assert!(slots.is_empty());
}
#[tokio::test]
async fn local_revision_watch_coalesces_to_latest_value() {
let handle = StateSyncHandle::new(peer(1));
let mut rx = handle.local_revisions_tx.subscribe();
handle.publish_library_revision(1);
handle.publish_library_revision(2);
handle.publish_library_revision(7);
rx.changed().await.expect("watch sender should remain live");
assert_eq!(rx.borrow_and_update().library, 7);
assert!(!rx.has_changed().expect("watch sender should remain live"));
}
#[tokio::test]
async fn hint_fanout_never_exceeds_eight_simultaneous_sends() {
let active = Arc::new(AtomicUsize::new(0));
let maximum = Arc::new(AtomicUsize::new(0));
let deliveries = (0..64).map(|_| {
let active = Arc::clone(&active);
let maximum = Arc::clone(&maximum);
async move {
let current = active.fetch_add(1, Ordering::SeqCst) + 1;
maximum.fetch_max(current, Ordering::SeqCst);
tokio::task::yield_now().await;
active.fetch_sub(1, Ordering::SeqCst);
}
});
drive_bounded_hint_fanout(deliveries).await;
assert!(maximum.load(Ordering::SeqCst) <= MAX_CONCURRENT_HINT_SENDS);
assert_eq!(active.load(Ordering::SeqCst), 0);
}
#[tokio::test]
async fn cancellation_drains_pull_and_fanout_futures() {
let cancellation = CancellationToken::new();
let completed = Arc::new(AtomicUsize::new(0));
let mut pulls = FuturesUnordered::<PullFuture>::new();
for seed in 1..=2 {
let cancellation = cancellation.clone();
let completed = Arc::clone(&completed);
pulls.push(Box::pin(async move {
cancellation.cancelled().await;
completed.fetch_add(1, Ordering::SeqCst);
PullCompletion {
peer_id: peer(seed),
result: Ok(PeerRefreshOutcome::Completed),
}
}));
}
let fanout_cancellation = cancellation.clone();
let fanout_completed = Arc::clone(&completed);
let fanout: FanoutFuture = Box::pin(async move {
fanout_cancellation.cancelled().await;
fanout_completed.fetch_add(1, Ordering::SeqCst);
});
cancellation.cancel();
drain_state_sync_children(&mut pulls, Some(fanout)).await;
assert_eq!(completed.load(Ordering::SeqCst), 3);
}
}
+384 -727
View File
@@ -1,187 +1,334 @@
//! Request dispatch for a single bidirectional QUIC stream.
//! Bounded one-control-frame dispatch for a bidirectional QUIC stream.
use std::net::SocketAddr;
use std::{net::SocketAddr, sync::Arc, time::Duration};
use futures::{SinkExt, StreamExt};
use lanspread_db::db::{Game, GameFileDescription};
use lanspread_proto::{CallToPlayAck, LibraryDelta, Message, Request, Response};
use s2n_quic::stream::{BidirectionalStream, SendStream};
use tokio_util::codec::{FramedRead, FramedWrite, LengthDelimitedCodec};
use futures::{SinkExt as _, StreamExt as _};
use lanspread_proto::{
ControlErrorCode,
ControlMessage,
MAX_CONTROL_FRAME_BYTES,
Request,
Response,
};
use s2n_quic::{
application,
stream::{BidirectionalStream, SendStream},
};
use tokio::sync::{OwnedSemaphorePermit, Semaphore};
use tokio_util::{
codec::{FramedRead, FramedWrite, LengthDelimitedCodec},
sync::CancellationToken,
};
use crate::{
context::PeerCtx,
error::PeerError,
events,
game_paths::is_local_dir_name,
local_games::{get_game_file_descriptions, local_download_matches_catalog},
peer::{send_game_file_chunk, send_game_file_data},
services::handshake::{HandshakeCtx, accept_inbound_hello, spawn_library_resync},
stream_install::{send_game_install_stream, send_stream_install_error},
services::{
remote_state,
state_sync::StateDomain,
transfer::{ChunkDispatch, handle_file_chunk_request, handle_stream_install_request},
},
};
type ResponseWriter = FramedWrite<SendStream, LengthDelimitedCodec>;
/// Handles a bidirectional stream from a peer.
const INBOUND_CONTROL_FRAME_TIMEOUT: Duration = Duration::from_secs(10);
const OUTBOUND_CONTROL_IO_TIMEOUT: Duration = Duration::from_secs(10);
fn control_codec() -> LengthDelimitedCodec {
LengthDelimitedCodec::builder()
.max_frame_length(MAX_CONTROL_FRAME_BYTES)
.new_codec()
}
/// Reads exactly one bounded request frame, requires request-side EOF, sends at
/// most one control response, and then closes the stream. Raw transfer requests
/// consume the response side after the same single control-frame admission.
pub(super) async fn handle_peer_stream(
stream: BidirectionalStream,
ctx: PeerCtx,
remote_addr: Option<SocketAddr>,
stream_shutdown: CancellationToken,
control_permit: OwnedSemaphorePermit,
bulk_transfer_permits: Arc<Semaphore>,
) -> eyre::Result<()> {
let (rx, tx) = stream.split();
let mut framed_rx = FramedRead::new(rx, LengthDelimitedCodec::new());
let mut framed_tx = FramedWrite::new(tx, LengthDelimitedCodec::new());
let mut framed_rx = FramedRead::new(rx, control_codec());
let mut framed_tx = FramedWrite::new(tx, control_codec());
log::trace!("{remote_addr:?} peer stream opened");
loop {
let next_message = tokio::select! {
() = ctx.shutdown.cancelled() => break,
next_message = framed_rx.next() => next_message,
};
match next_message {
Some(Ok(data)) => {
log::trace!(
"{:?} msg: (raw): {}",
remote_addr,
String::from_utf8_lossy(&data)
);
let request = Request::decode(data.freeze());
log::debug!("{remote_addr:?} msg: {request:?}");
note_peer_activity(&ctx, remote_addr).await;
framed_tx = dispatch_request(&ctx, remote_addr, request, framed_tx).await;
}
Some(Err(err)) => {
log::error!("{remote_addr:?} peer stream error: {err}");
break;
}
None => {
log::trace!("{remote_addr:?} peer stream closed");
break;
let first_frame = read_expected_frame(&mut framed_rx, &stream_shutdown).await;
let mut control_permit = Some(control_permit);
let mut _bulk_permit = None;
let mut response_reset = false;
match first_frame {
FrameRead::Frame(data) => {
let trailing = read_expected_eof(&mut framed_rx, &stream_shutdown).await;
if trailing == TrailingRead::Eof {
match Request::decode(data.freeze()) {
Ok(request) => {
log::debug!("{remote_addr:?} msg: {request:?}");
if request_is_bulk(&request) {
let bulk_permit =
Arc::clone(&bulk_transfer_permits).try_acquire_owned();
// Once the single bounded request is decoded, bulk
// work moves to its smaller pool so it cannot hold
// every control-plane permit during long egress.
drop(control_permit.take());
if let Ok(permit) = bulk_permit {
_bulk_permit = Some(permit);
let dispatched =
dispatch_request(&ctx, request, framed_tx, &stream_shutdown)
.await;
framed_tx = dispatched.writer;
response_reset = dispatched.response_reset;
} else {
let mut tx = framed_tx.into_inner();
let _ = tx.reset(application::Error::UNKNOWN);
framed_tx = FramedWrite::new(tx, control_codec());
response_reset = true;
}
} else {
let dispatched =
dispatch_request(&ctx, request, framed_tx, &stream_shutdown).await;
framed_tx = dispatched.writer;
response_reset = dispatched.response_reset;
}
}
Err(error) => {
log::warn!(
"Rejecting invalid control request from {remote_addr:?}: {error}"
);
framed_tx = send_response(
framed_tx,
Response::Error(ControlErrorCode::InvalidRequest),
"invalid-request",
&stream_shutdown,
)
.await;
}
}
} else if trailing != TrailingRead::Cancelled {
log::warn!("Rejecting non-singular control request from {remote_addr:?}");
framed_tx = send_response(
framed_tx,
Response::Error(ControlErrorCode::InvalidRequest),
"invalid-request",
&stream_shutdown,
)
.await;
}
}
FrameRead::Invalid(error) => {
log::warn!("Rejecting malformed control frame from {remote_addr:?}: {error}");
framed_tx = send_response(
framed_tx,
Response::Error(ControlErrorCode::InvalidRequest),
"invalid-request",
&stream_shutdown,
)
.await;
}
FrameRead::Eof => log::trace!("{remote_addr:?} peer stream closed without a request"),
FrameRead::Cancelled => {}
}
close_or_reset_stream(
framed_rx,
framed_tx,
remote_addr,
&stream_shutdown,
response_reset,
)
.await;
Ok(())
}
const fn request_is_bulk(request: &Request) -> bool {
matches!(
request,
Request::GetGameFileChunk { .. } | Request::StreamInstall { .. }
)
}
enum FrameRead {
Frame(bytes::BytesMut),
Invalid(std::io::Error),
Eof,
Cancelled,
}
async fn read_expected_frame(
framed_rx: &mut FramedRead<s2n_quic::stream::ReceiveStream, LengthDelimitedCodec>,
cancellation: &CancellationToken,
) -> FrameRead {
tokio::select! {
biased;
() = cancellation.cancelled() => FrameRead::Cancelled,
() = tokio::time::sleep(INBOUND_CONTROL_FRAME_TIMEOUT) => FrameRead::Invalid(
std::io::Error::new(std::io::ErrorKind::TimedOut, "control request timed out")
),
frame = framed_rx.next() => match frame {
Some(Ok(bytes)) => FrameRead::Frame(bytes),
Some(Err(error)) => FrameRead::Invalid(error),
None => FrameRead::Eof,
}
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum TrailingRead {
Eof,
ExtraFrame,
Invalid,
TimedOut,
Cancelled,
}
async fn read_expected_eof(
framed_rx: &mut FramedRead<s2n_quic::stream::ReceiveStream, LengthDelimitedCodec>,
cancellation: &CancellationToken,
) -> TrailingRead {
tokio::select! {
biased;
() = cancellation.cancelled() => TrailingRead::Cancelled,
() = tokio::time::sleep(INBOUND_CONTROL_FRAME_TIMEOUT) => TrailingRead::TimedOut,
frame = framed_rx.next() => match frame {
Some(Ok(_)) => TrailingRead::ExtraFrame,
Some(Err(_)) => TrailingRead::Invalid,
None => TrailingRead::Eof,
}
}
}
async fn dispatch_request(
ctx: &PeerCtx,
remote_addr: Option<SocketAddr>,
request: Request,
framed_tx: ResponseWriter,
) -> ResponseWriter {
stream_shutdown: &CancellationToken,
) -> DispatchResult {
match request {
Request::Ping => send_response(framed_tx, Response::Pong, "pong").await,
Request::Hello(hello) => match accept_inbound_hello(ctx, remote_addr, hello).await {
Ok(ack) => send_response(framed_tx, Response::HelloAck(ack), "HelloAck").await,
Err(err) => {
log::error!("Failed to accept inbound hello: {err}");
send_response(
framed_tx,
Response::InternalPeerError(err.to_string()),
"HelloAck",
)
Request::Ping => {
match control_io_with_deadline(remote_state::local_revisions(ctx), stream_shutdown)
.await
{
Some(Ok(revisions)) => DispatchResult::close(
send_response(
framed_tx,
Response::Pong(revisions),
"pong",
stream_shutdown,
)
.await,
),
Some(Err(error)) => {
log::error!("Failed to build local revisions: {error:#}");
DispatchResult::close(
send_response(
framed_tx,
Response::Error(ControlErrorCode::Internal),
"pong-error",
stream_shutdown,
)
.await,
)
}
None => reset_response_writer(framed_tx, "pong-computation"),
}
},
Request::ListGames => handle_list_games(ctx, framed_tx).await,
Request::LibraryDelta { peer_id, delta } => {
handle_library_delta(ctx, peer_id, delta).await;
framed_tx
}
Request::CallToPlayEvents {
peer_id,
events: incoming,
} => {
let ack = handle_call_to_play_events(ctx, &peer_id, incoming).await;
send_response(framed_tx, Response::CallToPlayAck(ack), "CallToPlayAck").await
Request::Hello => {
match control_io_with_deadline(remote_state::local_snapshot(ctx), stream_shutdown).await
{
Some(Ok(snapshot)) => DispatchResult::close(
send_response(
framed_tx,
Response::HelloSnapshot(snapshot),
"hello-snapshot",
stream_shutdown,
)
.await,
),
Some(Err(error)) => {
log::error!("Failed to build local peer snapshot: {error:#}");
DispatchResult::close(
send_response(
framed_tx,
Response::Error(ControlErrorCode::Internal),
"hello-error",
stream_shutdown,
)
.await,
)
}
None => reset_response_writer(framed_tx, "hello-computation"),
}
}
Request::LibraryChanged(hint) => {
ctx.state_sync.schedule_hint(StateDomain::Library, hint);
DispatchResult::close(framed_tx)
}
Request::CallToPlayChanged(hint) => {
ctx.state_sync.schedule_hint(StateDomain::CallToPlay, hint);
DispatchResult::close(framed_tx)
}
Request::GetGame { id } => handle_get_game(ctx, id, framed_tx).await,
Request::GetGameFileData(desc) => handle_file_data_request(ctx, desc, framed_tx).await,
Request::GetGameFileChunk {
game_id,
content_id,
relative_path,
offset,
length,
} => {
handle_file_chunk_request(ctx, game_id, relative_path, offset, length, framed_tx).await
}
Request::StreamInstall { game_id } => {
handle_stream_install_request(ctx, game_id, framed_tx).await
}
Request::Goodbye { peer_id } => {
handle_goodbye(ctx, remote_addr, peer_id).await;
framed_tx
}
Request::Invalid(_, _) => {
log::error!("Received invalid request from peer");
framed_tx
}
}
}
async fn handle_call_to_play_events(
ctx: &PeerCtx,
peer_id: &str,
incoming: Vec<lanspread_proto::CallToPlayEvent>,
) -> CallToPlayAck {
let peer_id = peer_id.to_string();
if ctx.peer_game_db.read().await.peer_addr(&peer_id).is_none() {
log::debug!("Requesting a handshake before accepting Call to Play events from {peer_id}");
return CallToPlayAck::NeedHandshake;
}
if incoming.iter().any(|event| event.actor_id != peer_id) {
let reason = format!("event actor does not match envelope peer {peer_id}");
log::warn!("Rejecting Call to Play events: {reason}");
return CallToPlayAck::Rejected { reason };
}
match ctx.call_to_play.write().await.merge_batch(incoming) {
Ok(merged) => {
let ack = if merged.needs_history() {
CallToPlayAck::NeedHistory
} else if !merged.applied.is_empty() {
CallToPlayAck::Applied
} else if merged.obsolete > 0 {
CallToPlayAck::Obsolete
} else if merged.duplicates > 0 {
CallToPlayAck::Duplicate
} else {
CallToPlayAck::Rejected {
reason: "empty Call to Play event batch".to_string(),
}
};
if merged.needs_history() {
log::warn!(
"Ignoring Call to Play actions without history from {peer_id}: {}",
merged.missing_call_ids.join(", ")
);
}
if !merged.applied.is_empty() {
events::send(
&ctx.tx_notify_ui,
crate::PeerEvent::CallToPlayEvents(merged.applied),
);
}
ack
}
Err(err) => {
log::warn!("Rejecting Call to Play events from {peer_id}: {err}");
CallToPlayAck::Rejected {
reason: err.to_string(),
}
}
}
}
async fn note_peer_activity(ctx: &PeerCtx, remote_addr: Option<SocketAddr>) {
if let Some(addr) = remote_addr {
ctx.peer_game_db
.write()
match handle_file_chunk_request(
ctx,
game_id,
content_id,
relative_path,
offset,
length,
framed_tx,
stream_shutdown,
)
.await
.update_last_seen_by_addr(&addr);
{
ChunkDispatch::Finished(writer) => DispatchResult::close(writer),
ChunkDispatch::Reset(writer) => DispatchResult::reset(writer),
}
}
Request::StreamInstall {
game_id,
content_id,
} => DispatchResult::close(
handle_stream_install_request(ctx, game_id, content_id, framed_tx, stream_shutdown)
.await,
),
}
}
fn reset_response_writer(framed_tx: ResponseWriter, label: &str) -> DispatchResult {
let mut tx = framed_tx.into_inner();
if let Err(error) = tx.reset(application::Error::UNKNOWN) {
log::debug!("Failed to reset timed-out {label} response: {error}");
}
DispatchResult::reset(FramedWrite::new(tx, control_codec()))
}
struct DispatchResult {
writer: ResponseWriter,
response_reset: bool,
}
impl DispatchResult {
const fn close(writer: ResponseWriter) -> Self {
Self {
writer,
response_reset: false,
}
}
const fn reset(writer: ResponseWriter) -> Self {
Self {
writer,
response_reset: true,
}
}
}
@@ -189,610 +336,120 @@ async fn send_response(
mut framed_tx: ResponseWriter,
response: Response,
label: &str,
stream_shutdown: &CancellationToken,
) -> ResponseWriter {
if let Err(err) = framed_tx.send(response.encode()).await {
log::error!("Failed to send {label} response: {err}");
let encoded = match response.encode() {
Ok(encoded) => encoded,
Err(error) => {
log::error!("Failed to encode {label} response: {error}");
let mut tx = framed_tx.into_inner();
if let Err(reset_error) = tx.reset(application::Error::UNKNOWN) {
log::debug!("Failed to reset unencodable {label} response: {reset_error}");
}
return FramedWrite::new(tx, control_codec());
}
};
let send_result = control_io_with_deadline(framed_tx.send(encoded), stream_shutdown).await;
let Some(send_result) = send_result else {
let mut tx = framed_tx.into_inner();
let _ = tx.reset(application::Error::UNKNOWN);
return FramedWrite::new(tx, control_codec());
};
if let Err(error) = send_result {
log::debug!("Failed to send {label} response: {error}");
}
framed_tx
}
async fn handle_list_games(ctx: &PeerCtx, framed_tx: ResponseWriter) -> ResponseWriter {
log::info!("Received ListGames request from peer");
let snapshot = {
let db_guard = ctx.local_game_db.read().await;
if let Some(db) = db_guard.as_ref() {
db.all_games().into_iter().cloned().collect::<Vec<Game>>()
} else {
log::info!("Local game database not yet loaded, responding with empty game list");
Vec::new()
}
};
let games = if snapshot.is_empty() {
snapshot
} else {
let active_operations = ctx.active_operations.read().await;
snapshot
.into_iter()
.filter(|game| !active_operations.contains_key(&game.id))
.collect()
};
send_response(framed_tx, Response::ListGames(games), "ListGames").await
}
async fn handle_library_delta(ctx: &PeerCtx, peer_id: String, delta: LibraryDelta) {
let applied = {
let mut db = ctx.peer_game_db.write().await;
db.apply_library_delta(&peer_id, delta)
};
if applied {
events::emit_peer_game_list(&ctx.peer_game_db, &ctx.catalog, &ctx.tx_notify_ui).await;
} else {
let addr = {
let db = ctx.peer_game_db.read().await;
db.peer_addr(&peer_id)
};
let Some(addr) = addr else {
log::debug!("Ignoring library delta from unknown peer {peer_id}");
return;
};
spawn_library_resync(HandshakeCtx::from_peer_ctx(ctx), addr, peer_id, "resync");
async fn close_or_reset_stream(
framed_rx: FramedRead<s2n_quic::stream::ReceiveStream, LengthDelimitedCodec>,
mut framed_tx: ResponseWriter,
remote_addr: Option<SocketAddr>,
cancellation: &CancellationToken,
response_reset: bool,
) {
if cancellation.is_cancelled() {
let mut rx = framed_rx.into_inner();
let _ = rx.stop_sending(application::Error::UNKNOWN);
let mut tx = framed_tx.into_inner();
let _ = tx.reset(application::Error::UNKNOWN);
return;
}
if response_reset {
// The transfer handler already sent RESET_STREAM. A later clean FIN
// would make a rejected zero-byte or truncated raw chunk ambiguous to
// the receiver.
drop(framed_rx);
drop(framed_tx);
return;
}
let close_result = control_io_with_deadline(framed_tx.close(), cancellation).await;
if close_result.is_none() {
let mut rx = framed_rx.into_inner();
let _ = rx.stop_sending(application::Error::UNKNOWN);
let mut tx = framed_tx.into_inner();
let _ = tx.reset(application::Error::UNKNOWN);
return;
}
if let Some(Err(error)) = close_result {
log::debug!("{remote_addr:?} failed to close peer response stream: {error}");
}
}
async fn handle_get_game(ctx: &PeerCtx, id: String, framed_tx: ResponseWriter) -> ResponseWriter {
log::info!("Received GetGame request for {id} from peer");
let response = get_game_response(ctx, id).await;
send_response(framed_tx, response, "GetGame").await
}
async fn get_game_response(ctx: &PeerCtx, id: String) -> Response {
let game_dir = ctx.game_dir.read().await.clone();
if !can_serve_game(ctx, &game_dir, &id).await {
return Response::GameNotFound(id);
async fn control_io_with_deadline<T>(
operation: impl std::future::Future<Output = T>,
cancellation: &CancellationToken,
) -> Option<T> {
tokio::select! {
biased;
() = cancellation.cancelled() => None,
() = tokio::time::sleep(OUTBOUND_CONTROL_IO_TIMEOUT) => None,
result = operation => Some(result),
}
match get_game_file_descriptions(&id, &game_dir).await {
Ok(file_descriptions) => Response::GetGame {
id,
file_descriptions,
},
Err(PeerError::FileSizeDetermination { path, source }) => {
let error_msg = format!("Failed to determine file size for {path}: {source}");
log::error!("File size determination error for game {id}: {error_msg}");
Response::InternalPeerError(error_msg)
}
Err(err) => {
log::error!("Failed to get game file descriptions for {id}: {err}");
Response::GameNotFound(id)
}
}
}
async fn can_serve_game(ctx: &PeerCtx, game_dir: &std::path::Path, game_id: &str) -> bool {
let active_operations = ctx.active_operations.read().await;
let catalog = ctx.catalog.read().await;
local_download_matches_catalog(game_dir, game_id, &active_operations, &catalog).await
}
async fn can_dispatch_file_transfer(
ctx: &PeerCtx,
game_dir: &std::path::Path,
game_id: &str,
relative_path: &str,
) -> bool {
relative_path_belongs_to_game(game_id, relative_path)
&& !path_points_inside_local(game_id, relative_path)
&& can_serve_game(ctx, game_dir, game_id).await
}
fn relative_path_belongs_to_game(game_id: &str, relative_path: &str) -> bool {
let normalised = relative_path.replace('\\', "/");
if normalised.starts_with('/') {
return false;
}
normalised
.split('/')
.find(|part| !part.is_empty())
.is_some_and(|first| first == game_id)
}
fn path_points_inside_local(game_id: &str, relative_path: &str) -> bool {
let normalised = relative_path.replace('\\', "/");
let mut parts = normalised.split('/').filter(|part| !part.is_empty());
match (parts.next(), parts.next()) {
(Some(first), _) if is_local_dir_name(first) => true,
(Some(first), Some(second)) if first == game_id && is_local_dir_name(second) => true,
_ => false,
}
}
use std::sync::atomic::{AtomicU64, Ordering};
static NEXT_TRANSFER_ID: AtomicU64 = AtomicU64::new(1);
struct TransferGuard {
game_id: String,
id: u64,
active_outbound_transfers: crate::context::OutboundTransfers,
tx_notify_ui: tokio::sync::mpsc::UnboundedSender<crate::PeerEvent>,
}
impl TransferGuard {
async fn new(
game_id: String,
active_outbound_transfers: crate::context::OutboundTransfers,
tx_notify_ui: tokio::sync::mpsc::UnboundedSender<crate::PeerEvent>,
shutdown: &tokio_util::sync::CancellationToken,
) -> (Self, tokio_util::sync::CancellationToken) {
let id = NEXT_TRANSFER_ID.fetch_add(1, Ordering::SeqCst);
let token = shutdown.child_token();
{
let mut active = active_outbound_transfers.write().await;
active
.entry(game_id.clone())
.or_default()
.push((id, token.clone()));
}
let _ = tx_notify_ui.send(crate::PeerEvent::OutboundTransferCountChanged);
(
Self {
game_id,
id,
active_outbound_transfers,
tx_notify_ui,
},
token,
)
}
}
impl Drop for TransferGuard {
fn drop(&mut self) {
let game_id = self.game_id.clone();
let id = self.id;
let active_outbound_transfers = self.active_outbound_transfers.clone();
let tx_notify_ui = self.tx_notify_ui.clone();
tokio::spawn(async move {
{
let mut active = active_outbound_transfers.write().await;
if let Some(tokens) = active.get_mut(&game_id) {
tokens.retain(|(tid, _)| *tid != id);
if tokens.is_empty() {
active.remove(&game_id);
}
}
}
let _ = tx_notify_ui.send(crate::PeerEvent::OutboundTransferCountChanged);
});
}
}
async fn handle_file_data_request(
ctx: &PeerCtx,
desc: GameFileDescription,
framed_tx: ResponseWriter,
) -> ResponseWriter {
log::info!(
"Received GetGameFileData request for {} from peer",
desc.relative_path
);
let (guard, cancel_token) = TransferGuard::new(
desc.game_id.clone(),
ctx.active_outbound_transfers.clone(),
ctx.tx_notify_ui.clone(),
&ctx.shutdown,
)
.await;
let mut tx = framed_tx.into_inner();
let game_dir = ctx.game_dir.read().await.clone();
if !can_dispatch_file_transfer(ctx, &game_dir, &desc.game_id, &desc.relative_path).await {
log::info!(
"Declining GetGameFileData for {} because the game is not currently transferable",
desc.relative_path
);
drop(guard);
let _ = tx.close().await;
return FramedWrite::new(tx, LengthDelimitedCodec::new());
}
send_game_file_data(&desc, &mut tx, &game_dir, cancel_token).await;
drop(guard);
FramedWrite::new(tx, LengthDelimitedCodec::new())
}
async fn handle_file_chunk_request(
ctx: &PeerCtx,
game_id: String,
relative_path: String,
offset: u64,
length: u64,
framed_tx: ResponseWriter,
) -> ResponseWriter {
log::info!(
"Received GetGameFileChunk request for {relative_path} (offset {offset}, length {length})"
);
let (guard, cancel_token) = TransferGuard::new(
game_id.clone(),
ctx.active_outbound_transfers.clone(),
ctx.tx_notify_ui.clone(),
&ctx.shutdown,
)
.await;
let mut tx = framed_tx.into_inner();
let game_dir = ctx.game_dir.read().await.clone();
if !can_dispatch_file_transfer(ctx, &game_dir, &game_id, &relative_path).await {
log::info!(
"Declining GetGameFileChunk for {relative_path} because the game is not currently transferable"
);
drop(guard);
let _ = tx.close().await;
return FramedWrite::new(tx, LengthDelimitedCodec::new());
}
send_game_file_chunk(
&game_id,
&relative_path,
offset,
length,
&mut tx,
&game_dir,
cancel_token,
)
.await;
drop(guard);
FramedWrite::new(tx, LengthDelimitedCodec::new())
}
async fn handle_stream_install_request(
ctx: &PeerCtx,
game_id: String,
framed_tx: ResponseWriter,
) -> ResponseWriter {
log::info!("Received StreamInstall request for {game_id} from peer");
let (guard, cancel_token) = TransferGuard::new(
game_id.clone(),
ctx.active_outbound_transfers.clone(),
ctx.tx_notify_ui.clone(),
&ctx.shutdown,
)
.await;
let mut tx = framed_tx.into_inner();
let game_dir = ctx.game_dir.read().await.clone();
if !can_serve_game(ctx, &game_dir, &game_id).await {
log::info!(
"Declining StreamInstall for {game_id} because the game is not currently transferable"
);
tx = send_stream_install_error(tx, format!("game {game_id} is not transferable")).await;
drop(guard);
return FramedWrite::new(tx, LengthDelimitedCodec::new());
}
let game_root = game_dir.join(&game_id);
let (returned_tx, result) = send_game_install_stream(
ctx.stream_install_provider.clone(),
tx,
&game_root,
&game_id,
cancel_token,
)
.await;
if let Err(err) = result {
log::warn!("StreamInstall for {game_id} ended with error: {err}");
}
drop(guard);
FramedWrite::new(returned_tx, LengthDelimitedCodec::new())
}
async fn handle_goodbye(ctx: &PeerCtx, _remote_addr: Option<SocketAddr>, peer_id: String) {
log::info!("Received Goodbye from peer {peer_id}");
let removed = { ctx.peer_game_db.write().await.remove_peer(&peer_id) };
let Some(peer) = removed else { return };
events::emit_peer_lost(&ctx.peer_game_db, &ctx.tx_notify_ui, peer.addr).await;
events::emit_peer_game_list(&ctx.peer_game_db, &ctx.catalog, &ctx.tx_notify_ui).await;
}
#[cfg(test)]
mod tests {
use std::{
path::{Path, PathBuf},
sync::Arc,
};
use lanspread_db::db::GameCatalog;
use lanspread_proto::{CallToPlayAction, CallToPlayEvent};
use tokio::sync::{RwLock, mpsc};
use tokio_util::{sync::CancellationToken, task::TaskTracker};
use lanspread_db::content_manifest::ContentId;
use super::*;
use crate::{
UnpackFuture,
Unpacker,
context::{Ctx, OperationKind},
peer_db::PeerGameDB,
test_support::TempDir,
};
struct NoopUnpacker;
impl Unpacker for NoopUnpacker {
fn unpack<'a>(&'a self, _archive: &'a Path, _dest: &'a Path) -> UnpackFuture<'a> {
Box::pin(async { Ok(()) })
}
}
fn write_file(path: &Path, bytes: &[u8]) {
if let Some(parent) = path.parent() {
std::fs::create_dir_all(parent).expect("parent dir should be created");
}
std::fs::write(path, bytes).expect("file should be written");
}
fn test_ctx(game_dir: PathBuf, catalog: GameCatalog) -> PeerCtx {
let (tx_notify_ui, _rx) = mpsc::unbounded_channel();
let state_dir = game_dir.join(".test-state");
Ctx::new(
Arc::new(RwLock::new(PeerGameDB::new())),
"peer".to_string(),
game_dir,
state_dir,
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
Arc::new(RwLock::new(catalog)),
Arc::new(RwLock::new(std::collections::HashMap::new())),
Arc::new(crate::NoopStreamInstallProvider),
)
.to_peer_ctx(tx_notify_ui)
}
fn call_to_play_event(actor_id: &str, action: CallToPlayAction) -> CallToPlayEvent {
CallToPlayEvent {
id: "event-1".to_string(),
call_id: "call-1".to_string(),
actor_id: actor_id.to_string(),
actor_name: "Alice".to_string(),
at: 8_000_000_000_000,
action,
}
}
fn call_to_play_create(actor_id: &str) -> CallToPlayEvent {
call_to_play_event(
actor_id,
CallToPlayAction::Create {
game_id: "game".to_string(),
max_players: 4,
scheduled_for: None,
deadline: 8_000_000_060_000,
},
)
#[test]
fn every_control_codec_enforces_the_protocol_frame_bound() {
let codec = control_codec();
assert_eq!(codec.max_frame_length(), MAX_CONTROL_FRAME_BYTES);
}
#[test]
fn local_relative_paths_are_never_transferable() {
assert!(path_points_inside_local("game", "game/local/save.dat"));
assert!(path_points_inside_local("game", "local/save.dat"));
assert!(path_points_inside_local("game", "game\\local\\save.dat"));
assert!(!path_points_inside_local("game", "game/version.ini"));
assert!(!path_points_inside_local("game", "game/archive.eti"));
fn trailing_frame_outcomes_are_never_accepted_as_eof() {
for outcome in [
TrailingRead::ExtraFrame,
TrailingRead::Invalid,
TrailingRead::TimedOut,
] {
assert_ne!(outcome, TrailingRead::Eof);
assert_ne!(outcome, TrailingRead::Cancelled);
}
}
#[tokio::test(start_paused = true)]
async fn public_control_computation_and_egress_have_an_absolute_deadline() {
let cancellation = CancellationToken::new();
let start = tokio::time::Instant::now();
assert!(
control_io_with_deadline(std::future::pending::<()>(), &cancellation)
.await
.is_none()
);
assert_eq!(start.elapsed(), OUTBOUND_CONTROL_IO_TIMEOUT);
}
#[test]
fn transferable_paths_must_belong_to_requested_game() {
assert!(relative_path_belongs_to_game("game", "game/version.ini"));
assert!(relative_path_belongs_to_game("game", "game\\archive.eti"));
assert!(!relative_path_belongs_to_game("game", "other/archive.eti"));
assert!(!relative_path_belongs_to_game("game", "archive.eti"));
assert!(!relative_path_belongs_to_game("game", "/game/archive.eti"));
assert!(!relative_path_belongs_to_game(
"game",
"../game/archive.eti"
));
}
#[tokio::test]
async fn known_peer_id_accepts_live_events_without_transport_ip_matching() {
let temp = TempDir::new("lanspread-call-to-play-known-peer");
let ctx = test_ctx(temp.path().to_path_buf(), GameCatalog::empty());
ctx.peer_game_db.write().await.upsert_peer(
"peer-alice".to_string(),
SocketAddr::from(([10, 66, 0, 2], 40000)),
);
let ack =
handle_call_to_play_events(&ctx, "peer-alice", vec![call_to_play_create("peer-alice")])
.await;
assert_eq!(ack, CallToPlayAck::Applied);
assert_eq!(ctx.call_to_play.write().await.snapshot().len(), 1);
}
#[tokio::test]
async fn unknown_peer_and_mismatched_actor_receive_explicit_acks() {
let temp = TempDir::new("lanspread-call-to-play-identity");
let ctx = test_ctx(temp.path().to_path_buf(), GameCatalog::empty());
assert_eq!(
handle_call_to_play_events(
&ctx,
"peer-alice",
vec![call_to_play_create("peer-alice")],
)
.await,
CallToPlayAck::NeedHandshake
);
ctx.peer_game_db.write().await.upsert_peer(
"peer-alice".to_string(),
SocketAddr::from(([10, 66, 0, 2], 40000)),
);
assert!(matches!(
handle_call_to_play_events(
&ctx,
"peer-alice",
vec![call_to_play_create("peer-mallory")],
)
.await,
CallToPlayAck::Rejected { reason }
if reason.contains("does not match envelope peer")
));
}
#[tokio::test]
async fn live_event_ack_reports_missing_history_and_duplicates() {
let temp = TempDir::new("lanspread-call-to-play-outcomes");
let ctx = test_ctx(temp.path().to_path_buf(), GameCatalog::empty());
ctx.peer_game_db.write().await.upsert_peer(
"peer-alice".to_string(),
SocketAddr::from(([10, 66, 0, 2], 40000)),
);
let orphan = call_to_play_event(
"peer-alice",
CallToPlayAction::AddTime {
deadline: 8_000_000_600_000,
},
);
assert_eq!(
handle_call_to_play_events(&ctx, "peer-alice", vec![orphan]).await,
CallToPlayAck::NeedHistory
);
let create = call_to_play_create("peer-alice");
assert_eq!(
handle_call_to_play_events(&ctx, "peer-alice", vec![create.clone()]).await,
CallToPlayAck::Applied
);
assert_eq!(
handle_call_to_play_events(&ctx, "peer-alice", vec![create]).await,
CallToPlayAck::Duplicate
);
}
#[tokio::test]
async fn get_game_response_respects_serve_gates() {
let temp = TempDir::new("lanspread-stream");
write_file(&temp.path().join("ready").join("version.ini"), b"20250101");
write_file(
&temp.path().join("non-catalog").join("version.ini"),
b"20250101",
);
write_file(&temp.path().join("active").join("version.ini"), b"20250101");
write_file(
&temp.path().join("wrong-version").join("version.ini"),
b"20260101",
);
std::fs::create_dir_all(temp.path().join("missing-sentinel"))
.expect("missing sentinel root should be created");
let mut catalog = GameCatalog::empty();
catalog.insert("ready".to_string(), Some("20250101".to_string()));
catalog.insert("active".to_string(), Some("20250101".to_string()));
catalog.insert("missing-sentinel".to_string(), Some("20250101".to_string()));
catalog.insert("wrong-version".to_string(), Some("20250101".to_string()));
let ctx = test_ctx(temp.path().to_path_buf(), catalog);
ctx.active_operations
.write()
.await
.insert("active".to_string(), OperationKind::Downloading);
assert!(matches!(
get_game_response(&ctx, "ready".to_string()).await,
Response::GetGame { id, .. } if id == "ready"
));
assert!(matches!(
get_game_response(&ctx, "non-catalog".to_string()).await,
Response::GameNotFound(id) if id == "non-catalog"
));
assert!(matches!(
get_game_response(&ctx, "active".to_string()).await,
Response::GameNotFound(id) if id == "active"
));
assert!(matches!(
get_game_response(&ctx, "wrong-version".to_string()).await,
Response::GameNotFound(id) if id == "wrong-version"
));
assert!(matches!(
get_game_response(&ctx, "missing-sentinel".to_string()).await,
Response::GameNotFound(id) if id == "missing-sentinel"
));
}
#[tokio::test]
async fn file_transfer_dispatch_respects_serve_gates() {
let temp = TempDir::new("lanspread-stream");
write_file(&temp.path().join("ready").join("version.ini"), b"20250101");
write_file(
&temp.path().join("non-catalog").join("version.ini"),
b"20250101",
);
write_file(&temp.path().join("active").join("version.ini"), b"20250101");
write_file(
&temp.path().join("wrong-version").join("version.ini"),
b"20260101",
);
std::fs::create_dir_all(temp.path().join("missing-sentinel"))
.expect("missing sentinel root should be created");
let mut catalog = GameCatalog::empty();
catalog.insert("ready".to_string(), Some("20250101".to_string()));
catalog.insert("active".to_string(), Some("20250101".to_string()));
catalog.insert("missing-sentinel".to_string(), Some("20250101".to_string()));
catalog.insert("wrong-version".to_string(), Some("20250101".to_string()));
let ctx = test_ctx(temp.path().to_path_buf(), catalog);
ctx.active_operations
.write()
.await
.insert("active".to_string(), OperationKind::Downloading);
assert!(can_dispatch_file_transfer(&ctx, temp.path(), "ready", "ready/version.ini").await);
assert!(
!can_dispatch_file_transfer(&ctx, temp.path(), "ready", "active/version.ini").await
);
assert!(
!can_dispatch_file_transfer(
&ctx,
temp.path(),
"non-catalog",
"non-catalog/version.ini",
)
.await
);
assert!(
!can_dispatch_file_transfer(&ctx, temp.path(), "active", "active/version.ini").await
);
assert!(
!can_dispatch_file_transfer(
&ctx,
temp.path(),
"wrong-version",
"wrong-version/version.ini",
)
.await
);
assert!(
!can_dispatch_file_transfer(
&ctx,
temp.path(),
"missing-sentinel",
"missing-sentinel/archive.eti",
)
.await
);
assert!(
!can_dispatch_file_transfer(&ctx, temp.path(), "ready", "ready/local/save.dat").await
);
fn only_long_lived_payload_requests_move_to_the_reserved_bulk_pool() {
assert!(!request_is_bulk(&Request::Ping));
assert!(request_is_bulk(&Request::StreamInstall {
game_id: "game".to_owned(),
content_id: ContentId::from_bytes([1; 32]),
}));
}
}
@@ -0,0 +1,952 @@
//! Catalog-authorized outbound chunk and streamed-install admission.
use std::{
fs::File,
path::PathBuf,
sync::{
Arc,
atomic::{AtomicU64, Ordering},
},
};
use lanspread_db::{
content_manifest::{
CanonicalCatalogPath,
CatalogContentManifest,
CatalogEntryKind,
CatalogFileEntry,
ContentId,
},
db::Availability,
};
use lanspread_proto::MAX_CONTROL_FRAME_BYTES;
use s2n_quic::{application, stream::SendStream};
use tokio_util::{
codec::{FramedWrite, LengthDelimitedCodec},
sync::CancellationToken,
};
use crate::{
context::PeerCtx,
download::open_catalog_file_for_read,
local_games::version_ini_is_regular_file,
peer::send_game_file_chunk,
scoped_blocking::scoped_blocking,
stream_install::{send_game_install_stream, send_stream_install_error},
};
type ResponseWriter = FramedWrite<SendStream, LengthDelimitedCodec>;
pub(super) enum ChunkDispatch {
Finished(ResponseWriter),
Reset(ResponseWriter),
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum ChunkSendDisposition {
Finished,
Reset,
}
fn chunk_send_disposition(result: &eyre::Result<()>) -> ChunkSendDisposition {
if result.is_ok() {
ChunkSendDisposition::Finished
} else {
ChunkSendDisposition::Reset
}
}
fn control_codec() -> LengthDelimitedCodec {
LengthDelimitedCodec::builder()
.max_frame_length(MAX_CONTROL_FRAME_BYTES)
.new_codec()
}
fn load_expected_catalog_manifest(
ctx: &PeerCtx,
game_id: &str,
) -> Option<Arc<CatalogContentManifest>> {
let catalog = Arc::clone(&ctx.catalog);
let manifest_game_id = game_id.to_owned();
match scoped_blocking(move || catalog.manifest(&manifest_game_id)) {
Ok(manifest) => Some(manifest),
Err(error) => {
log::error!("Failed to load catalog content manifest for {game_id}: {error}");
None
}
}
}
async fn can_serve_game(ctx: &PeerCtx, game_dir: &std::path::Path, game_id: &str) -> bool {
if ctx.recovery_quarantine.is_blocked(game_dir, game_id)
|| ctx.active_operations.read().await.contains_key(game_id)
{
return false;
}
let summary = ctx.local_library.read().await.games.get(game_id).cloned();
let Some(summary) = summary else {
return false;
};
if !summary.downloaded || summary.availability != Availability::Ready {
return false;
}
let catalog = ctx.catalog.catalog();
if !catalog.contains(game_id) {
return false;
}
let expected_version = catalog.expected_version(game_id).map(str::to_owned);
if expected_version
.as_deref()
.is_some_and(|expected| summary.eti_version.as_deref() != Some(expected))
{
return false;
}
let game_root = game_dir.join(game_id);
if !version_ini_is_regular_file(&game_root).await {
return false;
}
let expected_version_for_read = expected_version.clone();
scoped_blocking(move || {
expected_version_for_read.as_deref().is_none_or(|expected| {
lanspread_db::db::read_version_from_ini(&game_root)
.is_ok_and(|version| version.as_deref() == Some(expected))
})
})
}
fn authorize_catalog_file_request<'a>(
manifest: &'a CatalogContentManifest,
game_id: &str,
content_id: ContentId,
relative_path: &CanonicalCatalogPath,
offset: u64,
length: u64,
) -> Option<&'a CatalogFileEntry> {
if manifest.game_id() != game_id || manifest.content_id() != content_id {
return None;
}
// Exact lookup intentionally performs no normalization. Manifest keys have
// already passed the canonical, portable, and reserved-path policy.
let entry = manifest.file_entry(relative_path.as_str())?;
if entry.kind() != CatalogEntryKind::File
|| !catalog_chunk_range_is_exact(entry.size(), manifest.chunk_size(), offset, length)
{
return None;
}
Some(entry)
}
fn catalog_chunk_range_is_exact(file_size: u64, chunk_size: u64, offset: u64, length: u64) -> bool {
if chunk_size == 0 {
return false;
}
if file_size == 0 {
return offset == 0 && length == 0;
}
offset < file_size
&& offset.is_multiple_of(chunk_size)
&& length == std::cmp::min(chunk_size, file_size - offset)
}
static NEXT_TRANSFER_ID: AtomicU64 = AtomicU64::new(1);
struct TransferGuard {
game_id: String,
id: u64,
cancel_token: CancellationToken,
active_outbound_transfers: crate::context::OutboundTransfers,
notifier: crate::context::OutboundTransferNotifier,
armed: bool,
}
impl TransferGuard {
async fn new(
game_id: String,
active_outbound_transfers: crate::context::OutboundTransfers,
notifier: crate::context::OutboundTransferNotifier,
shutdown: &CancellationToken,
) -> (Self, CancellationToken) {
let id = NEXT_TRANSFER_ID.fetch_add(1, Ordering::SeqCst);
let token = shutdown.child_token();
{
let mut active = active_outbound_transfers.write().await;
active
.entry(game_id.clone())
.or_default()
.push((id, token.clone()));
}
notifier.notify();
(
Self {
game_id,
id,
cancel_token: token.clone(),
active_outbound_transfers,
notifier,
armed: true,
},
token,
)
}
/// Removes the registry entry before returning. Dropping this future while
/// it waits retains fail-closed tracking and cancels the transfer.
async fn finish(mut self) {
{
let mut active = self.active_outbound_transfers.write().await;
if let Some(tokens) = active.get_mut(&self.game_id) {
tokens.retain(|(transfer_id, _)| *transfer_id != self.id);
if tokens.is_empty() {
active.remove(&self.game_id);
}
}
}
self.armed = false;
self.notifier.notify();
}
}
impl Drop for TransferGuard {
fn drop(&mut self) {
if !self.armed {
return;
}
log::error!(
"Outbound transfer guard for {} ended unexpectedly; retaining transfer tracking until process restart",
self.game_id
);
self.cancel_token.cancel();
}
}
#[derive(Clone, Copy)]
enum OutboundTransferRequest<'a> {
CatalogChunk {
content_id: ContentId,
relative_path: &'a CanonicalCatalogPath,
offset: u64,
length: u64,
},
StreamInstall {
content_id: ContentId,
},
}
enum AdmittedOutboundPayload {
CatalogFile {
file: File,
},
StreamInstall {
game_dir: PathBuf,
manifest: Arc<CatalogContentManifest>,
},
}
struct AdmittedOutboundTransfer {
guard: TransferGuard,
cancel_token: CancellationToken,
payload: AdmittedOutboundPayload,
}
/// Validates content identity before readiness checks, filesystem opens,
/// transfer registration, or provider work. `SetGameDir` holds the same
/// admission barrier while draining the prior directory epoch.
async fn admit_outbound_transfer(
ctx: &PeerCtx,
game_id: &str,
request: OutboundTransferRequest<'_>,
stream_shutdown: &CancellationToken,
) -> Option<AdmittedOutboundTransfer> {
let admission = ctx.operation_admission.lock().await;
if stream_shutdown.is_cancelled() {
return None;
}
let game_dir = ctx.game_dir.read().await.clone();
let payload = match request {
OutboundTransferRequest::CatalogChunk {
content_id,
relative_path,
offset,
length,
} => {
let manifest = load_expected_catalog_manifest(ctx, game_id)?;
if manifest.content_id() != content_id {
log::warn!(
"Declining catalog chunk for {game_id}: requested content {content_id} does not match the local catalog"
);
return None;
}
if !can_serve_game(ctx, &game_dir, game_id).await {
return None;
}
let authorized_entry = authorize_catalog_file_request(
&manifest,
game_id,
content_id,
relative_path,
offset,
length,
)?;
let file = match open_catalog_file_for_read(
&game_dir,
game_id,
authorized_entry.canonical_path(),
authorized_entry.size(),
) {
Ok(file) => file,
Err(error) => {
log::warn!("Declining catalog file transfer for {relative_path}: {error}");
return None;
}
};
AdmittedOutboundPayload::CatalogFile { file }
}
OutboundTransferRequest::StreamInstall { content_id } => {
let manifest = load_expected_catalog_manifest(ctx, game_id)?;
if manifest.content_id() != content_id {
log::warn!(
"Declining StreamInstall for {game_id}: requested content {content_id} does not match the local catalog"
);
return None;
}
if !can_serve_game(ctx, &game_dir, game_id).await
|| !manifest.supports_streamed_install()
{
return None;
}
AdmittedOutboundPayload::StreamInstall { game_dir, manifest }
}
};
if stream_shutdown.is_cancelled() {
return None;
}
let (guard, cancel_token) = TransferGuard::new(
game_id.to_string(),
ctx.active_outbound_transfers.clone(),
ctx.outbound_transfer_notifier.clone(),
stream_shutdown,
)
.await;
drop(admission);
Some(AdmittedOutboundTransfer {
guard,
cancel_token,
payload,
})
}
#[allow(clippy::too_many_arguments)]
pub(super) async fn handle_file_chunk_request(
ctx: &PeerCtx,
game_id: String,
content_id: ContentId,
relative_path: CanonicalCatalogPath,
offset: u64,
length: u64,
framed_tx: ResponseWriter,
stream_shutdown: &CancellationToken,
) -> ChunkDispatch {
log::info!(
"Received GetGameFileChunk request for {relative_path} (offset {offset}, length {length})"
);
let mut tx = framed_tx.into_inner();
let Some(AdmittedOutboundTransfer {
guard,
cancel_token,
payload: AdmittedOutboundPayload::CatalogFile { file },
}) = admit_outbound_transfer(
ctx,
&game_id,
OutboundTransferRequest::CatalogChunk {
content_id,
relative_path: &relative_path,
offset,
length,
},
stream_shutdown,
)
.await
else {
log::info!("Declining GetGameFileChunk for {relative_path}");
reset_declined_transfer(&mut tx, "GetGameFileChunk");
return ChunkDispatch::Reset(FramedWrite::new(tx, control_codec()));
};
let send_result = send_game_file_chunk(
relative_path.as_str(),
offset,
length,
file,
&mut tx,
cancel_token,
)
.await;
guard.finish().await;
match chunk_send_disposition(&send_result) {
ChunkSendDisposition::Finished => {
ChunkDispatch::Finished(FramedWrite::new(tx, control_codec()))
}
ChunkSendDisposition::Reset => {
// The sender resets on every exceptional exit. Preserve that
// transport failure classification by preventing the dispatcher
// from following it with a clean FIN.
if let Err(error) = send_result {
log::debug!("Chunk send ended with a reset: {error:#}");
}
ChunkDispatch::Reset(FramedWrite::new(tx, control_codec()))
}
}
}
pub(super) async fn handle_stream_install_request(
ctx: &PeerCtx,
game_id: String,
content_id: ContentId,
framed_tx: ResponseWriter,
stream_shutdown: &CancellationToken,
) -> ResponseWriter {
log::info!("Received StreamInstall request for {game_id} from peer");
let mut tx = framed_tx.into_inner();
let Some(AdmittedOutboundTransfer {
guard,
cancel_token,
payload: AdmittedOutboundPayload::StreamInstall { game_dir, manifest },
}) = admit_outbound_transfer(
ctx,
&game_id,
OutboundTransferRequest::StreamInstall { content_id },
stream_shutdown,
)
.await
else {
tx = send_stream_install_error(
tx,
format!("game {game_id} is not transferable"),
&game_id,
stream_shutdown,
)
.await;
return FramedWrite::new(tx, control_codec());
};
let game_root = game_dir.join(&game_id);
let (returned_tx, result) = send_game_install_stream(
ctx.stream_install_provider.clone(),
tx,
&game_root,
&game_id,
manifest,
cancel_token,
)
.await;
if let Err(error) = result {
log::warn!("StreamInstall for {game_id} ended with error: {error}");
}
guard.finish().await;
FramedWrite::new(returned_tx, control_codec())
}
fn reset_declined_transfer(tx: &mut SendStream, label: &str) {
// A clean zero-length FIN is ambiguous with a valid empty catalog chunk,
// and a short clean FIN is an integrity failure at the receiver.
if let Err(error) = tx.reset(application::Error::UNKNOWN) {
log::debug!("Failed to reset declined {label} response: {error}");
}
}
#[cfg(test)]
mod tests {
use std::{
collections::{HashMap, HashSet},
path::Path,
sync::atomic::AtomicUsize,
};
use lanspread_db::content_manifest::{
Blake3Digest,
CATALOG_CHUNK_SIZE,
CatalogBundle,
CatalogContentManifestBody,
CatalogExtractedEntry,
};
use tokio::sync::{RwLock, mpsc};
use tokio_util::task::TaskTracker;
use super::*;
use crate::{
StreamInstallFrameSink,
StreamInstallFuture,
StreamInstallProvider,
UnpackFuture,
Unpacker,
context::{Ctx, OperationKind, PeerCtx},
identity::PeerIdentity,
library::LocalGameSummary,
network_generation::NetworkControl,
peer_db::PeerGameDB,
test_support::TempDir,
};
struct NoopUnpacker;
impl Unpacker for NoopUnpacker {
fn unpack<'a>(
&'a self,
_archive: &'a Path,
_dest: &'a Path,
_cancel_token: CancellationToken,
) -> UnpackFuture<'a> {
Box::pin(async { Ok(()) })
}
}
#[derive(Default)]
struct CountingStreamInstallProvider {
calls: AtomicUsize,
}
impl StreamInstallProvider for CountingStreamInstallProvider {
fn stream_archive<'a>(
&'a self,
_archive: &'a Path,
_frames: StreamInstallFrameSink,
_cancel_token: CancellationToken,
) -> StreamInstallFuture<'a> {
self.calls.fetch_add(1, Ordering::SeqCst);
Box::pin(async { Ok(()) })
}
}
fn manifest_with_streamed_install(
streamed_install_files: Vec<CatalogExtractedEntry>,
) -> CatalogContentManifest {
let bytes = b"payload";
let archive = b"archive";
let version = b"20250101";
CatalogContentManifest::seal(
CatalogContentManifestBody::new(
"game",
"20250101",
vec![
CatalogFileEntry::directory("directory")
.expect("test directory path is canonical"),
CatalogFileEntry::file("empty.bin", 0, Blake3Digest::hash(&[]), Vec::new())
.expect("test empty path is canonical"),
CatalogFileEntry::file(
"game.eti",
u64::try_from(archive.len()).expect("test archive length fits"),
Blake3Digest::hash(archive),
vec![Blake3Digest::hash(archive)],
)
.expect("test archive path is canonical"),
CatalogFileEntry::file(
"payload.bin",
u64::try_from(bytes.len()).expect("test length fits"),
Blake3Digest::hash(bytes),
vec![Blake3Digest::hash(bytes)],
)
.expect("test path is canonical"),
CatalogFileEntry::file(
"version.ini",
u64::try_from(version.len()).expect("test length fits"),
Blake3Digest::hash(version),
vec![Blake3Digest::hash(version)],
)
.expect("test version path is canonical"),
],
streamed_install_files,
)
.expect("test manifest body is valid"),
)
.expect("test manifest seals")
}
fn manifest() -> CatalogContentManifest {
manifest_with_streamed_install(Vec::new())
}
fn streamable_manifest() -> CatalogContentManifest {
manifest_with_streamed_install(vec![
CatalogExtractedEntry::file("installed/payload.bin", 7, Blake3Digest::hash(b"payload"))
.expect("test extracted path is canonical"),
])
}
async fn test_peer_ctx(
root: &Path,
manifest: &CatalogContentManifest,
provider: Arc<CountingStreamInstallProvider>,
) -> (PeerCtx, mpsc::UnboundedReceiver<crate::PeerEvent>) {
let catalog = Arc::new(
CatalogBundle::from_manifests([manifest.clone()])
.expect("test catalog should be complete"),
);
let ctx = Ctx::new(
Arc::new(RwLock::new(PeerGameDB::new())),
Arc::new(PeerIdentity::generate().expect("test identity should generate")),
root.to_path_buf(),
root.join(".state"),
Arc::new(NoopUnpacker),
CancellationToken::new(),
TaskTracker::new(),
catalog,
Arc::new(RwLock::new(HashMap::new())),
provider,
NetworkControl::disabled_for_test(),
)
.expect("test context should initialize");
assert!(ctx.recovery_quarantine.settle(root, HashSet::new()));
ctx.local_library.write().await.games.insert(
"game".to_owned(),
LocalGameSummary {
id: "game".to_owned(),
name: "game".to_owned(),
size: 15,
downloaded: true,
installed: false,
eti_version: Some("20250101".to_owned()),
availability: Availability::Ready,
},
);
let (tx, rx) = mpsc::unbounded_channel();
(ctx.to_peer_ctx(tx, CancellationToken::new()), rx)
}
async fn assert_chunk_rejected(
ctx: &PeerCtx,
content_id: ContentId,
relative_path: &CanonicalCatalogPath,
offset: u64,
length: u64,
) {
assert!(
admit_outbound_transfer(
ctx,
"game",
OutboundTransferRequest::CatalogChunk {
content_id,
relative_path,
offset,
length,
},
&CancellationToken::new(),
)
.await
.is_none()
);
assert!(ctx.active_outbound_transfers.read().await.is_empty());
}
#[test]
fn wrong_identity_path_or_range_never_authorizes_an_entry() {
let manifest = manifest();
let path = CanonicalCatalogPath::new("payload.bin").expect("test path is canonical");
let wrong_path = CanonicalCatalogPath::new("other.bin").expect("test path is canonical");
let content_id = manifest.content_id();
assert!(
authorize_catalog_file_request(
&manifest,
"game",
ContentId::from_bytes([9; 32]),
&path,
0,
7,
)
.is_none()
);
assert!(
authorize_catalog_file_request(&manifest, "wrong-game", content_id, &path, 0, 7,)
.is_none()
);
assert!(
authorize_catalog_file_request(&manifest, "game", content_id, &wrong_path, 0, 7,)
.is_none()
);
assert!(
authorize_catalog_file_request(&manifest, "game", content_id, &path, 1, 6,).is_none()
);
assert!(
authorize_catalog_file_request(&manifest, "game", content_id, &path, 0, 7,).is_some()
);
}
#[test]
fn chunk_boundaries_are_exact_including_empty_files() {
assert!(catalog_chunk_range_is_exact(10, 4, 0, 4));
assert!(catalog_chunk_range_is_exact(10, 4, 4, 4));
assert!(catalog_chunk_range_is_exact(10, 4, 8, 2));
assert!(!catalog_chunk_range_is_exact(10, 4, 1, 4));
assert!(!catalog_chunk_range_is_exact(10, 4, 8, 1));
assert!(catalog_chunk_range_is_exact(0, CATALOG_CHUNK_SIZE, 0, 0));
assert!(!catalog_chunk_range_is_exact(0, CATALOG_CHUNK_SIZE, 0, 1));
}
#[test]
fn admitted_chunk_send_error_preserves_reset_dispatch() {
let error = Err(eyre::eyre!("injected sender I/O failure"));
assert_eq!(chunk_send_disposition(&error), ChunkSendDisposition::Reset);
assert_eq!(
chunk_send_disposition(&Ok(())),
ChunkSendDisposition::Finished
);
}
#[allow(clippy::too_many_lines)]
#[tokio::test]
async fn admission_rejects_every_ineligible_v8_case_before_transfer_registration() {
let temp = TempDir::new("lanspread-transfer-admission");
let game_root = temp.path().join("game");
std::fs::create_dir_all(&game_root).expect("game root should be created");
std::fs::write(game_root.join("version.ini"), b"20250101")
.expect("version sentinel should be written");
std::fs::write(game_root.join("payload.bin"), b"payload")
.expect("payload should be written");
std::fs::write(game_root.join("empty.bin"), b"").expect("empty payload should be written");
std::fs::write(game_root.join("game.eti"), b"archive").expect("archive should be written");
let manifest = manifest();
let content_id = manifest.content_id();
let payload = CanonicalCatalogPath::new("payload.bin").expect("path should be canonical");
let provider = Arc::new(CountingStreamInstallProvider::default());
let (ctx, mut events) = test_peer_ctx(temp.path(), &manifest, Arc::clone(&provider)).await;
assert_chunk_rejected(&ctx, ContentId::from_bytes([9; 32]), &payload, 0, 7).await;
assert!(
admit_outbound_transfer(
&ctx,
"not-in-catalog",
OutboundTransferRequest::CatalogChunk {
content_id,
relative_path: &payload,
offset: 0,
length: 7,
},
&CancellationToken::new(),
)
.await
.is_none()
);
let missing = CanonicalCatalogPath::new("missing.bin").expect("path should be canonical");
assert_chunk_rejected(&ctx, content_id, &missing, 0, 7).await;
let local_path =
CanonicalCatalogPath::new("local/payload.bin").expect("path should be canonical");
assert_chunk_rejected(&ctx, content_id, &local_path, 0, 7).await;
let directory = CanonicalCatalogPath::new("directory").expect("path should be canonical");
assert_chunk_rejected(&ctx, content_id, &directory, 0, 0).await;
assert_chunk_rejected(&ctx, content_id, &payload, 1, 6).await;
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.downloaded = false;
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.downloaded = true;
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.availability = Availability::LocalOnly;
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.availability = Availability::Ready;
ctx.active_operations
.write()
.await
.insert("game".to_owned(), OperationKind::Updating);
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
ctx.active_operations.write().await.remove("game");
ctx.recovery_quarantine.begin(temp.path().to_path_buf());
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
assert!(ctx.recovery_quarantine.settle(temp.path(), HashSet::new()));
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.eti_version = Some("20240101".to_owned());
std::fs::write(game_root.join("version.ini"), b"20240101")
.expect("wrong version should be written");
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
ctx.local_library
.write()
.await
.games
.get_mut("game")
.expect("summary should exist")
.eti_version = Some("20250101".to_owned());
std::fs::write(game_root.join("version.ini"), b"20250101")
.expect("correct version should be restored");
std::fs::remove_file(game_root.join("version.ini")).expect("sentinel should be removable");
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
std::fs::create_dir(game_root.join("version.ini"))
.expect("nonregular sentinel should be created");
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
std::fs::remove_dir(game_root.join("version.ini"))
.expect("nonregular sentinel should be removable");
std::fs::write(game_root.join("version.ini"), b"20250101")
.expect("sentinel should be restored");
std::fs::write(game_root.join("payload.bin"), b"short")
.expect("wrong-sized payload should be written");
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
#[cfg(unix)]
{
use std::os::unix::fs::symlink;
std::fs::remove_file(game_root.join("payload.bin"))
.expect("wrong-sized payload should be removable");
std::fs::write(temp.path().join("outside.bin"), b"payload")
.expect("outside payload should be written");
symlink(
temp.path().join("outside.bin"),
game_root.join("payload.bin"),
)
.expect("payload symlink should be created");
assert_chunk_rejected(&ctx, content_id, &payload, 0, 7).await;
std::fs::remove_file(game_root.join("payload.bin"))
.expect("payload symlink should be removable");
}
assert!(
admit_outbound_transfer(
&ctx,
"game",
OutboundTransferRequest::StreamInstall {
content_id: ContentId::from_bytes([9; 32]),
},
&CancellationToken::new(),
)
.await
.is_none(),
"wrong-content StreamInstall must be rejected"
);
assert!(
admit_outbound_transfer(
&ctx,
"game",
OutboundTransferRequest::StreamInstall { content_id },
&CancellationToken::new(),
)
.await
.is_none(),
"manifest without extracted output must not admit StreamInstall"
);
assert!(ctx.active_outbound_transfers.read().await.is_empty());
assert_eq!(provider.calls.load(Ordering::SeqCst), 0);
assert!(events.try_recv().is_err());
std::fs::write(game_root.join("payload.bin"), b"payload")
.expect("valid payload should be restored");
let admitted = admit_outbound_transfer(
&ctx,
"game",
OutboundTransferRequest::CatalogChunk {
content_id,
relative_path: &payload,
offset: 0,
length: 7,
},
&CancellationToken::new(),
)
.await
.expect("exact catalog request should be admitted");
assert_eq!(ctx.active_outbound_transfers.read().await.len(), 1);
let AdmittedOutboundTransfer { guard, payload, .. } = admitted;
assert!(matches!(
payload,
AdmittedOutboundPayload::CatalogFile { .. }
));
drop(payload);
guard.finish().await;
assert!(ctx.active_outbound_transfers.read().await.is_empty());
assert_eq!(provider.calls.load(Ordering::SeqCst), 0);
}
#[tokio::test]
async fn stream_install_admission_separates_identity_capability_and_valid_payload() {
let temp = TempDir::new("lanspread-stream-install-admission");
let game_root = temp.path().join("game");
std::fs::create_dir_all(&game_root).expect("game root should be created");
std::fs::write(game_root.join("version.ini"), b"20250101")
.expect("version sentinel should be written");
std::fs::write(game_root.join("game.eti"), b"archive").expect("archive should be written");
let manifest = streamable_manifest();
let content_id = manifest.content_id();
let provider = Arc::new(CountingStreamInstallProvider::default());
let (ctx, mut events) = test_peer_ctx(temp.path(), &manifest, Arc::clone(&provider)).await;
assert!(
admit_outbound_transfer(
&ctx,
"game",
OutboundTransferRequest::StreamInstall {
content_id: ContentId::from_bytes([9; 32]),
},
&CancellationToken::new(),
)
.await
.is_none(),
"a stream-capable manifest must still reject the wrong content identity"
);
assert!(ctx.active_outbound_transfers.read().await.is_empty());
assert_eq!(provider.calls.load(Ordering::SeqCst), 0);
assert!(events.try_recv().is_err());
let admitted = admit_outbound_transfer(
&ctx,
"game",
OutboundTransferRequest::StreamInstall { content_id },
&CancellationToken::new(),
)
.await
.expect("exact stream-capable request should cross the provider boundary");
assert_eq!(ctx.active_outbound_transfers.read().await.len(), 1);
let AdmittedOutboundTransfer { guard, payload, .. } = admitted;
let AdmittedOutboundPayload::StreamInstall {
game_dir,
manifest: admitted_manifest,
} = payload
else {
panic!("StreamInstall admission returned a catalog-file payload");
};
assert_eq!(game_dir, temp.path());
assert_eq!(admitted_manifest.content_id(), content_id);
assert!(admitted_manifest.supports_streamed_install());
guard.finish().await;
assert!(ctx.active_outbound_transfers.read().await.is_empty());
assert_eq!(
provider.calls.load(Ordering::SeqCst),
0,
"provider work starts only after admission hands the payload to the sender"
);
}
}