feat(peer)!: cut over to authenticated catalog sharing

Replace address-only trust and pushed peer state with installation identities,
SPKI-pinned QUIC, candidate-only discovery, and bounded responder-owned
protocol-8 pulls. The runtime now owns each network generation and all admitted
work through shutdown.

Add exact bundled content identities, reproducible manifest publishing,
capability-confined downloads, streaming BLAKE3 verification, quarantine and
retry, and crash-recoverable download and install transactions. Ship generated
fixture catalogs and fail closed when production manifests are absent.

The Tauri backend exposes durable sharing policy, redacted identity state, and
attempt-keyed transfer snapshots. Frontend consumption follows in the next
commit. Repository-wide test certificates and protocol-7 paths are removed.

BREAKING CHANGE: peers must use protocol 8 and exact catalog content artifacts;
protocol-7 frames and shared-certificate identities are no longer accepted.

Test Plan:
- `just test` -- passed on the completed stack (708 workspace tests)
- `just clippy` -- passed on the completed stack
- `just build` -- passed with fixture catalogs on the completed stack
- `just catalog-check-production` -- failed closed because the external
  production manifest corpus is absent
- `git diff --cached --check` -- passed
This commit is contained in:
ddidderr committed 2026-08-10 13:59:18 +02:00
1 parent 36c4785775
commit 60fd7ba0c2
128 files changed
+51759 -10784

No files matched your search

+245 -297
View File
@@ -1,367 +1,315 @@
//! Peer liveness checks and stale-peer cleanup.
//! Pinned liveness checks and generation-conditional peer cleanup.
use std::{collections::HashMap, sync::Arc, time::Duration};
use std::{sync::Arc, time::Duration};
use lanspread_db::db::GameCatalog;
use tokio::sync::{RwLock, mpsc::UnboundedSender};
use tokio_util::{sync::CancellationToken, task::TaskTracker};
use futures::{StreamExt as _, stream};
use tokio::sync::mpsc::UnboundedSender;
use crate::{
PeerEvent,
config::{PEER_PING_IDLE_SECS, PEER_PING_INTERVAL_SECS, peer_stale_timeout},
context::OperationKind,
events,
content_quarantine::ContentQuarantine,
context::{NetworkServiceCtx, OperationKind},
network::ping_peer,
peer_db::{PeerGameDB, PeerId},
peer_db::PeerLivenessSnapshot,
scoped_blocking::scoped_blocking,
services::{HandshakeCtx, remote_state},
};
/// Runs the ping service to check peer liveness.
const MAX_CONCURRENT_PINGS: usize = 8;
/// Runs revision-bearing pinned liveness checks. The idle gate deliberately
/// uses `last_revision_check`; inbound and content traffic only affect
/// `last_seen`, which remains the stale-pruning clock.
pub async fn run_ping_service(
tx_notify_ui: UnboundedSender<PeerEvent>,
peer_game_db: Arc<RwLock<PeerGameDB>>,
catalog: Arc<RwLock<GameCatalog>>,
active_operations: Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: Arc<RwLock<HashMap<String, CancellationToken>>>,
shutdown: CancellationToken,
task_tracker: TaskTracker,
ctx: NetworkServiceCtx,
) -> eyre::Result<()> {
log::info!(
"Starting ping service ({PEER_PING_INTERVAL_SECS}s interval, \
{}s idle threshold, {}s timeout)",
{}s revision-check idle threshold, {}s stale timeout)",
PEER_PING_IDLE_SECS,
peer_stale_timeout().as_secs()
);
let mut interval = tokio::time::interval(Duration::from_secs(PEER_PING_INTERVAL_SECS));
let remote_ctx = HandshakeCtx::from_network(&ctx, &tx_notify_ui);
loop {
tokio::select! {
() = shutdown.cancelled() => return Ok(()),
biased;
() = ctx.shutdown.cancelled() => return Ok(()),
_ = interval.tick() => {}
}
ping_idle_peers(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
&shutdown,
&task_tracker,
)
.await;
let snapshots = ctx
.peer_game_db
.read()
.await
.peer_liveness_snapshot()
.into_iter()
.filter(revision_check_due)
.collect::<Vec<_>>();
let mut checks = stream::iter(snapshots.into_iter().map(|snapshot| {
let ctx = ctx.clone();
let remote_ctx = remote_ctx.clone();
async move { check_peer_liveness(&ctx, &remote_ctx, snapshot).await }
}))
.buffer_unordered(MAX_CONCURRENT_PINGS);
while checks.next().await.is_some() {}
prune_stale_peers(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
)
.await;
if ctx.shutdown.is_cancelled() {
return Ok(());
}
prune_stale_peers(&ctx, &remote_ctx).await?;
}
}
async fn ping_idle_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
shutdown: &CancellationToken,
task_tracker: &TaskTracker,
fn revision_check_due(snapshot: &PeerLivenessSnapshot) -> bool {
snapshot.last_revision_check.elapsed() >= Duration::from_secs(PEER_PING_IDLE_SECS)
}
async fn check_peer_liveness(
ctx: &NetworkServiceCtx,
remote_ctx: &HandshakeCtx,
snapshot: PeerLivenessSnapshot,
) {
let peer_snapshots = { peer_game_db.read().await.peer_liveness_snapshot() };
for (peer_id, peer_addr, last_seen) in peer_snapshots {
if last_seen.elapsed() < Duration::from_secs(PEER_PING_IDLE_SECS) {
continue;
}
let tx_notify_ui = tx_notify_ui.clone();
let peer_game_db = peer_game_db.clone();
let catalog = catalog.clone();
let active_operations = active_operations.clone();
let active_downloads = active_downloads.clone();
let shutdown = shutdown.clone();
task_tracker.spawn(async move {
let ping_result = tokio::select! {
() = shutdown.cancelled() => return,
result = ping_peer(peer_addr) => result,
};
match ping_result {
Ok(true) => {
peer_game_db.write().await.update_last_seen(&peer_id);
}
Ok(false) => {
log::warn!("Peer {peer_addr} failed ping check");
remove_peer_and_refresh(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
peer_id,
"Removed stale peer",
)
.await;
}
Err(err) => {
log::error!("Failed to ping peer {peer_addr}: {err}");
remove_peer_and_refresh(
&peer_game_db,
&catalog,
&active_operations,
&active_downloads,
&tx_notify_ui,
peer_id,
"Removed peer due to ping error",
)
.await;
match ping_peer(&ctx.quic, &snapshot.endpoint, &ctx.shutdown).await {
Ok(revisions) => {
match remote_state::observe_pinned_pong(remote_ctx, snapshot, revisions).await {
Ok(remote_state::PongCommit::NeedsPull { .. }) => {
if let Err(error) = ctx
.state_sync
.schedule_pinned_pull(snapshot.endpoint.peer_id, &ctx.shutdown)
.await
{
log::debug!(
"Could not schedule revision refresh for {}: {error:#}",
snapshot.endpoint.peer_id
);
}
}
Ok(
remote_state::PongCommit::Current | remote_state::PongCommit::StaleGeneration,
) => {}
Err(error) => log::error!(
"Failed to apply Pong from {}: {error:#}",
snapshot.endpoint.addr
),
}
});
}
Err(error) => {
log::warn!(
"Pinned ping to {} failed: {error:#}",
snapshot.endpoint.addr
);
// A capacity reset or transient transport failure is not topology
// authority. Leave both clocks unchanged and require repeated
// generation-current failures plus the stale timeout before the
// normal pruning path removes state.
ctx.peer_game_db
.write()
.await
.record_ping_failure_if_generation(snapshot);
}
}
}
async fn prune_stale_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
let stale_peers = {
peer_game_db
.read()
.await
.get_stale_peer_ids(peer_stale_timeout())
};
async fn prune_stale_peers(ctx: &NetworkServiceCtx, remote_ctx: &HandshakeCtx) -> eyre::Result<()> {
let stale = ctx
.peer_game_db
.read()
.await
.stale_peer_liveness_snapshots(peer_stale_timeout());
let mut removed_any = false;
for peer_id in stale_peers {
removed_any |= remove_peer(peer_game_db, tx_notify_ui, peer_id, "Removed stale peer").await;
for snapshot in stale {
removed_any |= remote_state::remove_peer_if_generation(remote_ctx, snapshot).await?;
}
if removed_any {
events::emit_peer_game_list(peer_game_db, catalog, tx_notify_ui).await;
handle_active_downloads_without_peers(
peer_game_db,
active_operations,
active_downloads,
tx_notify_ui,
)
.await;
handle_active_downloads_without_peers(ctx).await;
}
Ok(())
}
async fn remove_peer_and_refresh(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
catalog: &Arc<RwLock<GameCatalog>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
peer_id: PeerId,
log_label: &str,
) {
if remove_peer(peer_game_db, tx_notify_ui, peer_id, log_label).await {
events::emit_peer_game_list(peer_game_db, catalog, tx_notify_ui).await;
handle_active_downloads_without_peers(
peer_game_db,
active_operations,
active_downloads,
tx_notify_ui,
)
.await;
}
}
async fn remove_peer(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
peer_id: PeerId,
log_label: &str,
) -> bool {
let removed_peer = { peer_game_db.write().await.remove_peer(&peer_id) };
let Some(peer) = removed_peer else {
return false;
};
log::info!("{log_label}: {}", peer.addr);
events::emit_peer_lost(peer_game_db, tx_notify_ui, peer.addr).await;
true
}
async fn handle_active_downloads_without_peers(
peer_game_db: &Arc<RwLock<PeerGameDB>>,
active_operations: &Arc<RwLock<HashMap<String, OperationKind>>>,
active_downloads: &Arc<RwLock<HashMap<String, CancellationToken>>>,
tx_notify_ui: &UnboundedSender<PeerEvent>,
) {
let active_ids = {
active_operations
.read()
.await
.iter()
.filter_map(|(id, kind)| (*kind == OperationKind::Downloading).then_some(id.clone()))
.collect::<Vec<_>>()
};
if active_ids.is_empty() {
return;
}
async fn handle_active_downloads_without_peers(ctx: &NetworkServiceCtx) {
let active_ids = ctx
.active_operations
.read()
.await
.iter()
.filter_map(|(id, kind)| (*kind == OperationKind::Downloading).then_some(id.clone()))
.collect::<Vec<_>>();
for id in active_ids {
if peers_still_have_game(peer_game_db, &id).await {
if eligible_source_remains(ctx, &id).await {
continue;
}
let cancelled = {
// An exclusive guard makes the check-and-cancel transition one-shot even when
// concurrent liveness checks remove the last peers at the same time.
let active_downloads = active_downloads.write().await;
let Some(cancel_token) = active_downloads.get(&id) else {
{
// Signalling is one-shot; the download task retains ownership of
// operation-map cleanup until every transfer worker has drained.
let active_downloads = ctx.active_downloads.read().await;
let Some(download) = active_downloads.get(&id) else {
continue;
};
if cancel_token.is_cancelled() {
false
} else {
cancel_token.cancel();
true
}
};
if !cancelled {
continue;
download.cancel_sources_exhausted();
}
events::send(
tx_notify_ui,
PeerEvent::DownloadGameFilesAllPeersGone { id: id.clone() },
);
}
}
async fn peers_still_have_game(peer_game_db: &Arc<RwLock<PeerGameDB>>, game_id: &str) -> bool {
let guard = peer_game_db.read().await;
!guard.peers_with_game(game_id).is_empty()
async fn eligible_source_remains(ctx: &NetworkServiceCtx, game_id: &str) -> bool {
let catalog = Arc::clone(&ctx.catalog);
let game_id_owned = game_id.to_owned();
let Ok(manifest) = scoped_blocking(move || catalog.manifest(&game_id_owned)) else {
return false;
};
let content_id = manifest.content_id();
let endpoints = ctx
.peer_game_db
.read()
.await
.peer_endpoints_with_content(game_id, content_id);
has_nonquarantined_source(&endpoints, content_id, &ctx.content_quarantine)
}
fn has_nonquarantined_source(
endpoints: &[lanspread_proto::PeerEndpoint],
content_id: lanspread_db::content_manifest::ContentId,
quarantine: &ContentQuarantine,
) -> bool {
endpoints
.iter()
.any(|endpoint| !quarantine.is_quarantined(endpoint, content_id))
}
#[cfg(test)]
mod tests {
use std::{collections::HashMap, sync::Arc};
use lanspread_db::content_manifest::ContentId;
use lanspread_proto::{LibrarySnapshot, PeerEndpoint, PeerId, RuntimeSessionId};
use tokio::sync::RwLock;
use tokio_util::sync::CancellationToken;
use super::*;
use crate::peer_db::PeerGameDB;
use super::handle_active_downloads_without_peers;
use crate::{PeerEvent, context::OperationKind, peer_db::PeerGameDB};
#[tokio::test]
async fn all_peers_gone_cancels_once_and_leaves_cleanup_to_download_owner() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let active_operations = Arc::new(RwLock::new(HashMap::from([(
"game".to_string(),
OperationKind::Downloading,
)])));
let cancel = CancellationToken::new();
let active_downloads = Arc::new(RwLock::new(HashMap::from([(
"game".to_string(),
cancel.clone(),
)])));
let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel();
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
)
.await;
assert!(cancel.is_cancelled());
assert_eq!(
active_operations.read().await.get("game"),
Some(&OperationKind::Downloading)
#[tokio::test(start_paused = true)]
async fn dropped_hint_is_revision_checked_within_configured_bound() {
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([1; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12001)),
);
assert!(active_downloads.read().await.contains_key("game"));
let mut interval = tokio::time::interval(Duration::from_secs(PEER_PING_INTERVAL_SECS));
interval.tick().await;
tokio::time::advance(Duration::from_millis(1)).await;
let event = rx.recv().await.expect("peers-gone event should be emitted");
assert!(matches!(
event,
PeerEvent::DownloadGameFilesAllPeersGone { id } if id == "game"
));
let mut db = PeerGameDB::new();
let ticket = db
.begin_candidate_negotiation(endpoint)
.expect("candidate should reserve");
db.commit_authenticated_snapshot(
endpoint,
ticket,
RuntimeSessionId::from_bytes([1; 16]),
Some(LibrarySnapshot {
revision: 0,
games: Vec::new(),
}),
)
.expect("peer should commit")
.expect("ticket should remain current");
let start = tokio::time::Instant::now();
interval.tick().await;
let mut snapshot = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should remain");
snapshot.last_seen = tokio::time::Instant::now();
assert!(
rx.try_recv().is_err(),
"cancellation must not emit a premature active-operation snapshot"
!revision_check_due(&snapshot),
"a tick just before the idle threshold should not ping"
);
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
)
.await;
interval.tick().await;
let mut snapshot = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should remain");
// Simulate repeated inbound/content activity. It may refresh liveness,
// but must not touch the independently captured revision-check clock.
snapshot.last_seen = tokio::time::Instant::now();
assert!(revision_check_due(&snapshot));
assert!(
rx.try_recv().is_err(),
"an already-cancelled download must not emit peers-gone twice"
start.elapsed() <= Duration::from_secs(PEER_PING_IDLE_SECS + PEER_PING_INTERVAL_SECS)
);
}
#[tokio::test]
async fn all_peers_gone_cancels_multiple_downloads_without_releasing_admission() {
let peer_game_db = Arc::new(RwLock::new(PeerGameDB::new()));
let first_cancel = CancellationToken::new();
let second_cancel = CancellationToken::new();
let active_operations = Arc::new(RwLock::new(HashMap::from([
("first".to_string(), OperationKind::Downloading),
("second".to_string(), OperationKind::Downloading),
("installing".to_string(), OperationKind::Installing),
])));
let active_downloads = Arc::new(RwLock::new(HashMap::from([
("first".to_string(), first_cancel.clone()),
("second".to_string(), second_cancel.clone()),
])));
let (tx, mut rx) = tokio::sync::mpsc::unbounded_channel();
handle_active_downloads_without_peers(
&peer_game_db,
&active_operations,
&active_downloads,
&tx,
#[tokio::test(start_paused = true)]
async fn one_transient_ping_failure_never_removes_and_success_resets_failure_history() {
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([9; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12009)),
);
let mut db = PeerGameDB::new();
let ticket = db
.begin_candidate_negotiation(endpoint)
.expect("candidate should reserve");
db.commit_authenticated_snapshot(
endpoint,
ticket,
RuntimeSessionId::from_bytes([9; 16]),
Some(LibrarySnapshot {
revision: 0,
games: Vec::new(),
}),
)
.await;
.expect("peer should commit")
.expect("ticket should remain current");
let probe = db
.peer_liveness_for(&endpoint.peer_id)
.expect("peer should exist");
assert!(first_cancel.is_cancelled());
assert!(second_cancel.is_cancelled());
let operations = active_operations.read().await;
assert_eq!(operations.get("first"), Some(&OperationKind::Downloading));
assert_eq!(operations.get("second"), Some(&OperationKind::Downloading));
assert_eq!(
operations.get("installing"),
Some(&OperationKind::Installing)
);
drop(operations);
let downloads = active_downloads.read().await;
assert!(downloads.contains_key("first"));
assert!(downloads.contains_key("second"));
drop(downloads);
let mut cancelled_ids = Vec::new();
for _ in 0..2 {
let event = rx.recv().await.expect("peers-gone event should be emitted");
let PeerEvent::DownloadGameFilesAllPeersGone { id } = event else {
panic!("expected peers-gone event");
};
cancelled_ids.push(id);
}
cancelled_ids.sort();
assert_eq!(cancelled_ids, vec!["first", "second"]);
tokio::time::advance(peer_stale_timeout() + Duration::from_secs(1)).await;
assert!(db.record_ping_failure_if_generation(probe));
assert!(
rx.try_recv().is_err(),
"multiple cancellations must not emit an active-operation snapshot"
db.stale_peer_liveness_snapshots(peer_stale_timeout())
.is_empty(),
"one transport failure must preserve authenticated state"
);
assert!(matches!(
db.observe_pong_if_generation(
probe,
lanspread_proto::PeerRevisions {
runtime_session_id: RuntimeSessionId::from_bytes([9; 16]),
library_revision: 0,
call_to_play_revision: 0,
},
),
crate::peer_db::PongObservation::Current
| crate::peer_db::PongObservation::RevisionMismatch
));
let refreshed = db
.peer_liveness_for(&endpoint.peer_id)
.expect("successful Pong should preserve peer");
assert_eq!(refreshed.consecutive_ping_failures, 0);
}
#[test]
fn wrong_content_and_quarantined_sources_do_not_keep_download_alive() {
let expected = ContentId::from_bytes([7; 32]);
let endpoint = PeerEndpoint::new(
PeerId::from_bytes([2; 32]),
std::net::SocketAddr::from(([127, 0, 0, 1], 12002)),
);
let quarantine = ContentQuarantine::default();
// Exact-content filtering happens before this seam, so a wrong-content
// peer produces the empty eligible endpoint set.
assert!(!has_nonquarantined_source(&[], expected, &quarantine));
assert!(has_nonquarantined_source(
&[endpoint],
expected,
&quarantine
));
quarantine.record_integrity_failure(&endpoint, expected);
assert!(!has_nonquarantined_source(
&[endpoint],
expected,
&quarantine
));
}
}