//! Ring protocol logic and supporting types. //! //! Mainly maintains a healthy and optimal pool of connections to other peers in the network //! and routes requests to the optimal peers. use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; use std::net::SocketAddr; use std::sync::{Arc, Weak}; use std::time::Duration; // Intentional wall-clock type for suspend/resume detection in // `connection_maintenance`. See `classify_suspend_jump` for why the DST // `TimeSource` abstraction is the *wrong* primitive here (it returns // simulation time, which would never reflect a real OS suspend). This is // the one call site in `crates/core/` that needs a real monotonic wall // clock; renaming sidesteps the crates/core/ DST rule-lint grep while // keeping intent explicit to the reader. use std::time::Instant as WallClockInstant; use tokio::time::Instant; use tracing::Instrument; use either::Either; use freenet_stdlib::prelude::{ContractInstanceId, ContractKey}; use parking_lot::{Mutex, RwLock}; use tokio_util::sync::CancellationToken; pub use hosting::{ AddClientSubscriptionResult, AddSubscriberOutcome, ClientDisconnectResult, HostingReason, HostingReasonStats, SubscribeResult, SubscribedContractSnapshot, }; use crate::message::TransactionType; use crate::operations::connect::op_ctx_task::ClientConnectKind; use crate::topology::TopologyAdjustment; use crate::topology::rate::Rate; use crate::tracing::{NetEventLog, NetEventRegister}; use connection_manager::LatticeProbeMisses; use crate::transport::TransportPublicKey; use crate::util::{Contains, time_source::InstantTimeSrc}; use crate::{ config::{GlobalExecutor, GlobalRng, OPERATION_TTL}, message::Transaction, node::{self, EventLoopNotificationsSender, NodeConfig, OpManager, PeerId}, router::Router, }; // Issue #4350 invariants, pinned at compile time so a future edit to any of the // renewal-timing constants can't silently reopen the discarded-reply bug. The // outer cancel deadline is defined in `renewal_outer_cancel` as exactly // `RENEWAL_TASK_BUDGET + NOTIFICATION_SEND_TIMEOUT + RENEWAL_OUTER_DEADLINE_MARGIN`, // so it clears the driver's worst case (a full-budget slow attempt plus a fully // backpressured `release_pending_op_slot` cleanup) by exactly the margin. These // asserts pin the surrounding relationships that definition depends on. const _: () = { let budget = Ring::RENEWAL_TASK_BUDGET.as_secs(); // The cleanup-headroom margin must be non-zero: the outer cancel // (= budget + cleanup + margin) has to be STRICTLY greater than the // worst-case self-termination time (budget + cleanup), or it could fire // exactly as cleanup completes (issue #4350, Codex review). assert!( Ring::RENEWAL_OUTER_DEADLINE_MARGIN.as_secs() > 0, "RENEWAL_OUTER_DEADLINE_MARGIN must be > 0 so the outer cancel strictly \ exceeds BUDGET + cleanup timeout (issue #4350)" ); // A renewal must be allowed at least one real attempt: the no-start floor // has to sit below the total budget. assert!( Ring::RENEWAL_MIN_ATTEMPT_BUDGET.as_secs() < budget, "RENEWAL_MIN_ATTEMPT_BUDGET must be < RENEWAL_TASK_BUDGET (issue #4350)" ); // No single attempt may exceed the total budget. assert!( Ring::RENEWAL_PER_ATTEMPT_TIMEOUT.as_secs() <= budget, "RENEWAL_PER_ATTEMPT_TIMEOUT must be <= RENEWAL_TASK_BUDGET (issue #4350)" ); // The per-attempt renewal wait stays below the global operation TTL so a // renewal never out-waits a normal client subscribe attempt. assert!( Ring::RENEWAL_PER_ATTEMPT_TIMEOUT.as_secs() < OPERATION_TTL.as_secs(), "RENEWAL_PER_ATTEMPT_TIMEOUT must be < OPERATION_TTL (issue #4350)" ); }; pub(crate) mod broadcast_coverage; mod broken_invariants; mod connection_backoff; mod connection_manager; pub(crate) mod contract_ban_list; pub(crate) mod contract_exec_metrics; pub(crate) mod delta_incompat; /// Shadow-mode detector for repairs that never converge (see the module docs). pub(crate) mod futile_repair; pub(crate) use connection_manager::ConnectionManager; // Nearest-neighbor ring lattice (successor+predecessor base edges). // `nn_lattice_active_for` is the full activation gate — the flag AND the // minimum-degree floor (`NN_LATTICE_MIN_MAX_CONNECTIONS`) — used by the // free-function retention path in `crate::topology`; the acceptance clause and // discovery probe apply the same gate via `ConnectionManager::nn_lattice_active`. // The setter is `pub` so `dev_tool` can flip the stock/fix arm from the // findability validation harness; it is a no-op-in-production, test-only override // (see the module docs). pub(crate) use connection_manager::nn_lattice_active_for; #[cfg(any(test, feature = "testing"))] pub use connection_manager::{set_nn_lattice_enabled, set_nn_lattice_force_active}; mod connection; mod hosting; pub(crate) use broken_invariants::{BrokenInvariant, BrokenInvariantsTracker}; /// Pre-write admission-gate rejection type (#4683, PR 3). Surfaced to the /// executor chokepoints so a write that would overflow the aggregate disk /// budget is refused before any bytes land. pub(crate) use hosting::DiskBudgetExceeded; /// Hosting-BEGIN attribution: WHY this peer started hosting a contract. Every /// production caller of [`Ring::host_contract`] names one. pub(crate) use hosting::HostingCause; /// The pre-A2 flat 1 GiB budget, used as the upgrade-migration sentinel in /// `config::ConfigArgs::build` so an upgraded node re-derives its hosting budget /// instead of keeping the historically-pinned default (#4565). pub(crate) use hosting::LEGACY_FLAT_HOSTING_BUDGET_BYTES; /// Clamp bound re-exported only for the config-default round-trip test. #[cfg(test)] pub(crate) use hosting::MAX_DEFAULT_HOSTING_BUDGET_BYTES; /// The aggregate hosting-disk budget's own floor — also the wasmtime /// compile-cache's configured-budget bound's floor (#5328 review), so this is /// a genuine production dependency now, not test-only. pub(crate) use hosting::MIN_DEFAULT_HOSTING_BUDGET_BYTES; /// Single source of truth for the default hosted-contract-state budget. /// `config::default_max_hosting_storage()` resolves to this so the /// operator-facing default and the in-code fallback can never drift. The /// default is RAM-scaled (capability-relative, A2) rather than a flat constant. pub(crate) use hosting::default_hosting_budget_bytes; /// The aggregate hosting-disk budget's own pure clamp math. A genuine /// production dependency (#5328 review): the wasmtime compile-cache's /// configured-budget bound (`wasm_runtime::runtime::bound_by_configured_disk_budget`) /// projects what the live aggregate budget will resolve to via this SAME /// function, so the compile cache respects an operator-shrunk /// `--max-hosting-disk` rather than only physical disk availability. Also /// used by the wasmtime disk-cache sizing tests to verify headroom against /// the real function rather than a duplicate. pub(crate) use hosting::disk_budget_for_clamped; /// The hosting budget's pure RAM clamp, re-exported (test-only) so the wasmtime /// on-disk compile-cache sizing test can pin that the compile cache never /// exceeds the contract-state budget it accelerates. #[cfg(test)] pub(crate) use hosting::hosting_budget_for_ram; /// Re-export the reconcile controller (pure decision core, its input/action /// types, and the shadow-mode set-membership comparator) so the node layer can /// build inputs and run the shadow compare (keystone step-2, #4642). pub(crate) use hosting::reconcile; pub use hosting::{AccessType, RecordAccessResult}; /// Aggregate disk-budget defaults (#4683). `config` resolves the persisted /// `hosting-disk-pct` / `max-hosting-disk` defaults from these so the operator- /// facing defaults and the in-code sizing math share one source of truth. pub(crate) use hosting::{ DEFAULT_HOSTING_DISK_PCT, DEFAULT_MAX_HOSTING_DISK_BYTES, DEFAULT_RESIDENT_OVERHEAD_MEM_SHARE, }; /// Widths of the two hosted-set demand-signal histograms carried on the router /// snapshot, re-exported so `router` sizes its wire arrays from the same /// definition the bucketing code uses. pub(crate) use hosting::{GENUINE_ACCESS_RECENCY_BUCKETS, READ_COUNT_HIST_BUCKETS}; /// The aggregate disk-usage tracker's mount-availability probe and directory /// walk, re-exported so the wasmtime on-disk compile-cache startup sizing /// (`wasm_runtime::runtime::default_wasmtime_cache_size_bytes_for_dir`, #5014) /// can bound itself by real disk headroom, not just RAM, without duplicating /// the `statvfs` FFI call or the walk. pub(crate) use hosting::{disk_available_bytes, disk_directory_size_bytes}; pub mod interest; mod live_tx; mod location; pub(crate) mod merge_backoff; pub(crate) mod peer_cache; mod peer_connection_backoff; mod peer_key_location; mod placement_migration_metrics; pub(crate) mod resync_rate_limit; pub mod topology_registry; pub(crate) mod update_rate_limit; // GET auto-subscribe (`AUTO_SUBSCRIBE_ON_GET`) was REMOVED in piece E of the // demand-driven hosting redesign (docs/design/demand-driven-hosting.md §9, // .claude/rules/hosting-invariants.md anti-patterns table). Auto-installing a // durable subscription on every GET manufactures demand that no client asked // for — the "GET-auto-subscribe" half of the relay-caching anti-pattern. A GET // may still host on the return path under the demand gauge (evictable, // non-durable; see `cache_contract_locally`), and a client that wants ongoing // freshness sets `subscribe=true` explicitly. do NOT re-add auto-subscribe on // GET — see hosting-invariants (invariants 1 & 2). /// Per-beneficiary weight for a local-client subscription when /// computing the LIVE benefit snapshot each governance reaper tick. /// Strong signal: a real user on this node is currently subscribed. /// Hard to fake without running an actual freenet client locally. /// /// Sourced from the design doc — see "Sybil weighting falls out /// naturally" in `docs/design/contract-hardening.md`. const LOCAL_DEMAND_WEIGHT: f64 = 1.0; /// Per-beneficiary weight for a downstream peer's CURRENT subscription /// when computing the LIVE benefit snapshot. Weaker signal because peer /// identity is attacker-rotatable — without an identity layer, a single /// attacker can spin up many peers and each "subscribes." We weight /// each forwarded subscriber at 0.1 so an attacker would need 10 /// rotating peers to fake the standing demand of one local user. const FORWARDED_DEMAND_WEIGHT: f64 = 0.1; /// Interval between governance reaper ticks. Each tick applies /// decay and runs MAD-based outlier detection across the population. /// One minute balances responsiveness (a sustained-cost contract is /// flagged within a minute or two of crossing the threshold) against /// CPU overhead (MAD over the entire contract set every tick). const GOVERNANCE_TICK_INTERVAL: Duration = Duration::from_secs(60); use connection_backoff::ConnectionBackoff; /// How often connected-peer attributes are written to the routing dataset. const ROUTING_DATASET_PEER_INTERVAL: Duration = Duration::from_secs(60); pub use connection_backoff::ConnectionFailureReason; pub(crate) use peer_connection_backoff::PeerConnectionBackoff; pub use self::live_tx::LiveTransactionTracker; pub use connection::Connection; pub use interest::PeerKey; pub use location::{Distance, Location}; pub use peer_key_location::{KnownPeerKeyLocation, PeerAddr, PeerKeyLocation}; /// Thread safe and friendly data structure to keep track of the local knowledge /// of the state of the ring. /// // Note: For now internally we wrap some of the types internally with locks and/or use // multithreaded maps. In the future if performance requires it some of this can be moved // towards a more lock-free multithreading model if necessary. /// Backoff state for contract-directed CONNECT attempts. struct ContractConnectState { current_backoff: Duration, last_attempt: Instant, } /// Outcome of [`Ring::add_connection_reporting`]. /// /// Exists because `add_connection`'s `bool` answers a different question than /// callers usually want: it reports the readiness-threshold crossing, and is /// `false` both for a rejected connection and for the ordinary successful add /// on an already-ready node. Instrumentation that counts promotions needs /// `added` (issue #4787). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(crate) struct AddConnectionOutcome { /// The ring accepted the connection into the topology. `false` means it /// was rejected — e.g. the `max_connections` cap. pub added: bool, /// Adding this connection crossed the readiness threshold. pub just_became_ready: bool, } /// Per-cause counts of route failure labels (#5657). See /// [`Ring::route_failure_cause_counts`]. #[derive(Default)] struct RouteFailureCauseCounts { /// Ambiguous NotFounds dropped untrained (no proof the contract exists). untrained_not_found: std::sync::atomic::AtomicU64, not_found: std::sync::atomic::AtomicU64, timeout: std::sync::atomic::AtomicU64, send_failure: std::sync::atomic::AtomicU64, /// The subset of `timeout` recorded as the originator of the operation, /// excluding labels recorded while relaying other nodes' operations. originator_timeout: std::sync::atomic::AtomicU64, } /// Distinct peers tracked per snapshot window by [`TimeoutLabelWindow`]. /// Labels for peers beyond it are counted, not attributed. const TIMEOUT_LABEL_WINDOW_MAX_PEERS: usize = 4096; /// Timeout route labels per peer since the last router snapshot (#5657), for /// the chain-blame soak histogram on `RouterSnapshotInfo`. Bounded: at most /// [`TIMEOUT_LABEL_WINDOW_MAX_PEERS`] fixed-size entries, cleared every /// snapshot. #[derive(Default)] struct TimeoutLabelWindow { per_peer: std::collections::HashMap, untracked: u64, } /// Histogram of one [`TimeoutLabelWindow`]: /// `(peers with 1, 2-3, 4-7, 8+ labels, max per peer, untracked labels)`. pub(crate) type TimeoutLabelHistogram = (u64, u64, u64, u64, u64, u64); impl TimeoutLabelWindow { fn record(&mut self, addr: std::net::SocketAddr) { if let Some(count) = self.per_peer.get_mut(&addr) { *count += 1; } else if self.per_peer.len() < TIMEOUT_LABEL_WINDOW_MAX_PEERS { self.per_peer.insert(addr, 1); } else { self.untracked += 1; } } fn take_histogram(&mut self) -> TimeoutLabelHistogram { let (mut b1, mut b2, mut b4, mut b8, mut max) = (0, 0, 0, 0, 0); for count in self.per_peer.values().copied() { match count { 0 => {} 1 => b1 += 1, 2..=3 => b2 += 1, 4..=7 => b4 += 1, _ => b8 += 1, } max = max.max(count); } let untracked = self.untracked; *self = Self::default(); (b1, b2, b4, b8, max, untracked) } } pub(crate) struct Ring { pub max_hops_to_live: usize, pub connection_manager: ConnectionManager, pub router: Arc>, /// Route failure labels fed to the router, by the attempt outcome that /// produced them (#5657). Diagnostics only. route_failure_causes: RouteFailureCauseCounts, /// Timeout labels per peer since the last router snapshot (#5657). timeout_label_window: parking_lot::Mutex, pub live_tx_tracker: LiveTransactionTracker, hosting_manager: hosting::HostingManager, /// Per-contract record of detected CRDT-invariant violations (e.g. a /// non-idempotent `update_state`). Used to gate outbound broadcast /// so a broken contract's state changes don't leave the node. broken_invariants: BrokenInvariantsTracker, /// Per-contract governance scoring + reaper. Routes every /// per-contract resource report through `ingest_cost` so the /// governance state for a contract reflects what the meter /// already records. Mode defaults to `DryRun` per the staged- /// rollout plan. pub(crate) governance: Arc, /// Front-line per-(sender, contract) UPDATE rate limit. Catches /// flood patterns at the receive boundary in milliseconds, before /// the slower MAD-based governance reaper can react. See /// `crate::ring::update_rate_limit` and `docs/design/contract-hardening.md` /// Phase 2. pub(crate) update_rate_limiter: Arc, /// Per-contract merge-failure backoff (#4861). Quarantines "poison" /// contracts whose WASM merges reliably fail/time out: while a contract is /// in its exponential cooldown the UPDATE broadcast drivers skip the merge /// (and its `ResyncRequest` amplification) entirely, so it costs O(1) per /// inbound broadcast instead of O(WASM merge). Cleared the instant a merge /// succeeds. See `crate::ring::merge_backoff`. pub(crate) merge_backoff: Arc, /// SENDER-side memo of contracts whose deltas are known-doomed (the HQk7 /// resync loop). Armed by repeated resync-after-our-delta signals or local /// `Invalid`-class delta-apply failures; while armed, the broadcast path /// skips delta computation and sends full state. TTL-bounded, LRU-capped, /// cleared by a successful delta apply. See `crate::ring::delta_incompat`. pub(crate) delta_incompat: Arc, /// Per-contract cap on how often THIS node emits a `ResyncRequest` (#4861). /// Bounds the full-state resync amplification independently of the merge /// outcome. See `crate::ring::resync_rate_limit`. pub(crate) resync_emit_limiter: Arc, /// Per-`(peer, contract)` cap on how often this node answers a given peer's /// `ResyncRequest` with a full-state `ResyncResponse` (#4861). The /// mixed-version-rollout guard: a not-yet-upgraded peer that still emits /// unlimited `ResyncRequest`s cannot make an upgraded peer full-state-reply /// in a loop. See `crate::ring::resync_rate_limit`. pub(crate) resync_response_limiter: Arc, /// GLOBAL per-contract cap on `ResyncResponse` emission, aggregated across /// ALL requesters (#4861). Per-(peer, contract) limiting alone is /// insufficient — production saw ~45 distinct requester IPs drive ~9,733 /// full-state responses/day for one forked contract. Checked AFTER the /// per-peer limit. See `crate::ring::resync_rate_limit`. pub(crate) resync_response_global_limiter: Arc, /// Correlation map for outstanding `ResyncRequest`s WE emitted (#4864 /// round-8, Codex P1). The `ResyncResponse` apply path runs a full-state WASM /// merge that is deliberately NOT backoff-gated, so without correlation it is /// an unmetered DoS surface — a peer could stream/replay unsolicited responses /// bypassing every emitter-side gate. The receive arm require-and-consumes a /// matching `(contract, source)` entry BEFORE applying; consume-on-first-match /// kills replay. See `crate::ring::resync_rate_limit::OutstandingResyncRequests`. pub(crate) outstanding_resync_requests: Arc, /// Per-contract ban list. Populated by the governance reaper on /// `BanTriggered` / `BanLifted` transitions; consulted at the /// inbound dispatch site to drop wire requests for banned /// contracts. Phase 7 of the contract-hardening plan /// (`docs/design/contract-hardening.md`) and the auto-detection /// half of issue #4274 (operator-CLI blocklist). pub(crate) contract_ban_list: Arc, event_register: Box, op_manager: RwLock>>, /// Whether this peer is a gateway or not. This will affect behavior of the node when acquiring /// and dropping connections. pub(crate) is_gateway: bool, /// Shared connection backoff tracker for all connection failure types. connection_backoff: Arc>, /// Per-contract backoff for contract-directed CONNECT attempts. contract_connect_backoff: Mutex>, /// Injectable time source used by `connection_maintenance`. Using `util::TimeSource` /// (which returns `tokio::time::Instant`) lets tests supply `SharedMockTimeSource` for /// fine-grained control without pausing the entire tokio runtime. pub(crate) time_source: Arc, /// Directory for persisting the peer address cache. When set, the peer cache /// is periodically saved here and loaded on startup for fast reconnection. pub(crate) peer_cache_dir: Option, /// Per-node compiled-WASM module-cache occupancy + eviction telemetry /// (#4440). Constructed once here and threaded as an `Arc`: the /// `RuntimePool` clones it (via `op_manager.ring`) into its labeled module /// caches, which *publish* into it; the `emit_router_snapshot_telemetry` /// task *reads* it. Threading the `Arc` (rather than a process-global) /// keeps the gauges per-node so unit tests stay isolated (#4488). module_cache_metrics: Arc, /// Per-node contract-exec WASM counters: how many `summarize_state` / /// `get_state_delta` invocations this node actually ran, split from the /// cache hits that elided them. Constructed once here and shared via `Arc`: /// the executor increments it through `op_manager.ring`, while /// `emit_router_snapshot_telemetry` reads it. Threading the `Arc` (rather /// than a process-global) keeps the counters per-node so unit tests stay /// isolated (#4488). See [`contract_exec_metrics`] for why an /// undifferentiated span count could not answer the storm question. contract_exec_metrics: Arc, /// Per-node placement-migration activity counters (#4404 follow-up). /// Constructed once here and shared via `Arc`: the migration SEND site /// (`p2p_protoc::migration`) and the two RECEIVE sites (`node::process_message`) /// increment it through `op_manager.ring`, while /// `emit_router_snapshot_telemetry` reads it on the snapshot cadence. Threading /// the `Arc` (rather than a process-global) keeps the counters per-node so unit /// tests stay isolated. Together with the placement-quality gauge it lets us /// observe whether the placement migration is firing and whether it actually /// pulls hosting closer to each contract's key. placement_migration_metrics: Arc, /// Shutdown signal for the long-lived background tasks spawned in /// [`Ring::new`]. Triggered once on node teardown (via /// [`Ring::trigger_shutdown`], fired from `ShutdownTeardown::drop`). /// /// Every long sleep / `interval.tick()` in those loops races this token /// via `tokio::select!` so a shutdown returns promptly instead of waiting /// for the longest outstanding sleep to elapse (up to 5 minutes for /// `interest_heartbeat`). See issue #4278 and /// `.claude/rules/code-style.md` ("Backoff sleeps MUST be interruptible"). /// /// `CancellationToken` is level-triggered: once cancelled, every present /// and future `cancelled()` await returns immediately, so there is no /// missed-wakeup race for a task that is between sleeps when shutdown /// fires. shutdown: CancellationToken, } // /// A data type that represents the fact that a peer has been blacklisted // /// for some action. Has to be coupled with that action // #[derive(Debug)] // struct Blacklisted { // since: Instant, // peer: PeerKey, // } /// Guard that ensures `complete_subscription_request` is called even if the /// subscription task panics. This prevents contracts from being stuck in /// `pending_subscription_requests` forever. pub(crate) struct SubscriptionRecoveryGuard { op_manager: Arc, contract_key: ContractKey, completed: bool, } impl SubscriptionRecoveryGuard { pub(crate) fn new(op_manager: Arc, contract_key: ContractKey) -> Self { Self { op_manager, contract_key, completed: false, } } pub(crate) fn complete(mut self, success: bool) { self.op_manager .ring .complete_subscription_request(&self.contract_key, success); self.completed = true; } } impl Drop for SubscriptionRecoveryGuard { fn drop(&mut self) { if !self.completed { // Task panicked or was cancelled before completion - treat as failure tracing::warn!( contract = %self.contract_key, "Subscription recovery task terminated unexpectedly, marking as failed" ); self.op_manager .ring .complete_subscription_request(&self.contract_key, false); } } } /// State of the `Ring -> OpManager` weak back-reference, as observed by /// the `connection_maintenance` loop. /// /// The startup-vs-shutdown distinction is the whole point: a `None` slot /// (not attached yet) must keep waiting, while a `Some(weak)` that no longer /// upgrades (attached then dropped) must terminate the loop. Conflating the /// two leaves the maintenance task spinning forever on a dead `Weak` — /// the #3308 zombie. enum OpManagerState { /// `attach_op_manager` has not run yet — normal startup window. The /// maintenance loop should keep waiting for attachment. NotAttached, /// The `OpManager` is alive and was successfully upgraded. Live(Arc), /// The back-reference was attached once but the owning `Arc` has since /// been dropped. Nothing re-attaches it, so the loop owner is dead and /// the maintenance task must terminate instead of spinning on a `Weak` /// that can never upgrade again (#3308). Detached, } /// Classify a `RwLock>>` back-reference into the three states /// the maintenance loop cares about. Generic over `T` so it can be /// unit-tested with a stand-in `Arc` instead of a full `OpManager` (#3308). fn classify_op_manager_ref(slot: &RwLock>>) -> OpManagerState { let upgraded = slot.read().as_ref().map(|weak| weak.clone().upgrade()); match upgraded { None => OpManagerState::NotAttached, Some(Some(strong)) => OpManagerState::Live(strong), Some(None) => OpManagerState::Detached, } } /// Pure core of [`Ring::is_subscription_root`]'s neighbor scan: is there NO /// connected neighbor that is both *routable* (a renewal could route to it) and /// strictly closer to the contract than this node? /// /// `my_distance` is this node's ring distance to the contract. Each neighbor is /// `(its location, is it routable)`: /// - A neighbor with no known location can't be compared, so it never makes us /// "not the root" on distance grounds (it is skipped). /// - A non-routable neighbor (transient or not-yet-ready) is skipped, mirroring /// the eligibility `k_closest_potentially_hosting` applies — so the predicate /// agrees with where a renewal would actually route (#4440). /// /// Returns `true` when no routable neighbor is strictly closer (this node is the /// effective terminus), `false` otherwise. Factored out as a pure function so the /// distance/eligibility logic has direct unit coverage without a full `Ring`. fn no_closer_routable_neighbor( my_distance: Distance, contract_location: Location, neighbors: impl Iterator, bool)>, ) -> bool { for (peer_loc, routable) in neighbors { if !routable { continue; } if let Some(peer_loc) = peer_loc { if peer_loc.distance(contract_location) < my_distance { return false; } } } true } /// Pure core of [`Ring::most_keyward_hosting_neighbor`]: from candidate hosting /// neighbors already paired with their ring distance to the contract, pick the /// one STRICTLY closer to the contract than `my_distance` (the most-keyward /// host), breaking ties on equal distance by ascending socket address so the /// choice is deterministic (design §6 point 2 total order). The /// `most_keyward_hosting_neighbor` caller pre-filters candidates through /// `pkl.location()?`, which requires a resolvable socket address, so no /// addressless candidate ever reaches this helper and every item here carries a /// concrete distance and a concrete address for the tiebreak. /// /// Returns `None` when no candidate is strictly closer than us — either the /// candidate set is empty or every candidate is farther-or-equal (so we are the /// terminus / a stranded host; the caller distinguishes those). The strict `<` /// is the acyclicity guarantee: a peer at exactly our distance never becomes our /// upstream. Factored out so the strict-closer + deterministic-tiebreak /// selection has direct unit coverage without a heavyweight async `Ring` /// fixture, mirroring [`no_closer_routable_neighbor`]. /// /// This is the computed-upstream primitive of the demand-driven-hosting redesign /// (#4642 piece D): the eventual replacement for the stored `is_upstream` /// interest flag, which drifts under gossip (#4671). It is introduced here /// behavior-preservingly — computed alongside the stored flag for divergence /// telemetry — ahead of the reconcile-core keystone that computes upstream /// everywhere. fn most_keyward_among( my_distance: Distance, candidates: impl Iterator, ) -> Option { candidates .filter(|(_, dist)| *dist < my_distance) .min_by(|(a, ad), (b, bd)| { ad.cmp(bd) .then_with(|| a.socket_addr().cmp(&b.socket_addr())) }) .map(|(pkl, _)| pkl) } // NOTE (#4642 keystone sub-task 3, "the flip"): the RENEWAL-site shadow helpers // (`renewal_shadow_actual` + `record_renewal_shadow`) were REMOVED when the // renewal site was flipped from record-only shadow to actually DRIVING the // controller's decision. The drive lives in `OpManager::reconcile_wants_renewal` // (built from a fresh at-emission snapshot) and is consulted inline in the // renewal loop; the RENEWAL reconcile counter now measures the flip's // interest-gate suppression rate rather than a shadow divergence. /// Race an `.await` point in a background loop against the Ring shutdown /// token. Returns `true` if shutdown fired (the caller should stop its loop) /// and `false` if the wrapped future completed first (carry on). /// /// This is the single chokepoint that makes every long sleep / `interval.tick()` /// in the [`Ring::new`] background tasks interruptible (issue #4278). Keeping it /// in one place means a new loop only has to call this helper to satisfy the /// `.claude/rules/code-style.md` "backoff sleeps MUST be interruptible" rule. /// /// `CancellationToken::cancelled` is cancellation-safe and level-triggered, so /// wrapping it in `select!` here introduces no missed-wakeup window: if the /// token is already cancelled when this is called, the shutdown arm wins /// immediately. /// /// **`biased;` justification** (per `.claude/rules/code-style.md`): /// - *Why biased:* shutdown must deterministically win when both the timer and /// the token are ready in the same poll, so a teardown is never delayed by an /// extra sleep cycle. /// - *Starvation:* none. The non-shutdown arm is a one-shot timer (`sleep` / /// `interval.tick()`), not a high-throughput channel, and the shutdown arm is /// terminal (the caller breaks its loop on `true`). No hot arm exists for the /// bias to starve, so no per-iteration cap is needed. /// - *Cancellation safety:* both arms are cancellation-safe — dropping the /// loser (a timer future or `cancelled()`) discards no work. async fn sleep_or_shutdown(shutdown: &CancellationToken, wait: F) -> bool where F: std::future::Future, { tokio::select! { biased; _ = shutdown.cancelled() => true, _ = wait => false, } } /// Result of pruning a connection. #[derive(Debug, Default)] pub struct PruneConnectionResult { /// Orphaned transactions that need to be retried or failed. pub orphaned_transactions: Vec, /// True if this prune caused us to drop below the readiness threshold. pub became_unready: bool, } impl Ring { pub const DEFAULT_MIN_CONNECTIONS: usize = 25; pub const DEFAULT_MAX_CONNECTIONS: usize = 200; const DEFAULT_MAX_UPSTREAM_BANDWIDTH: Rate = Rate::new_per_second(1_000_000.0); const DEFAULT_MAX_DOWNSTREAM_BANDWIDTH: Rate = Rate::new_per_second(1_000_000.0); /// Above this number of remaining hops, randomize which node a message which be forwarded to. const DEFAULT_RAND_WALK_ABOVE_HTL: usize = 7; /// Max hops to be performed for certain operations (e.g. propagating connection of a peer in the network). pub const DEFAULT_MAX_HOPS_TO_LIVE: usize = 10; pub fn new( config: &NodeConfig, event_loop_notifier: EventLoopNotificationsSender, event_register: ER, is_gateway: bool, connection_manager: ConnectionManager, task_monitor: &crate::node::background_task_monitor::BackgroundTaskMonitor, ) -> anyhow::Result> { let live_tx_tracker = LiveTransactionTracker::new(); let max_hops_to_live = if let Some(v) = config.max_hops_to_live { v } else { Self::DEFAULT_MAX_HOPS_TO_LIVE }; // Single shutdown signal for every long-lived background task spawned // below, moved into the struct literal at the end. See issue #4278. let shutdown = CancellationToken::new(); // The router starts COLD and learns purely from this session's // observations. There is deliberately no warm start from the on-disk // event log, and the reader that used to attempt one was deleted in the // #4808 follow-up (#4812). Three reasons it is not worth resurrecting: // // 1. It effectively never fired in production. The restore read filtered // records to `record_ts >= NEW_RECORDS_TS`. `NEW_RECORDS_TS` is a // *process-global* `OnceLock` (`tracing::register`), stamped // `SystemTime::now()` at the FIRST event-register construction in the // process — `EventRegister::new`, or `OTEventRegister::new` under // `trace-ot`. A production node builds its register (`node.rs`) before // it can write a single route event and before `OpManager::new` -> // `Ring::new` ran, so the cutoff was always >= that process's start and // every prior-session record was discarded. Dead since #3672 // (2026-03-27). // // The invariant is "effectively never in production", NOT "structurally // unreachable" — do not tighten this claim. Two ways the branch can be // reached: both sides truncate to whole seconds, so a prior-session // record written in the same wall-clock second as the stamp passes; and // an in-process multi-node test that shares one `data_dir` (see // `tests/in_process_restart.rs`) makes the second node's `get_or_init` a // no-op, so it would read the first node's records. Neither happens on a // real node. Corroborated in production: across 8 gateway startups // captured in vega's retained logs, the INFO "Restored routing history // from event log" line and its WARN error arm each appear ZERO times, // while ~141k other INFO lines from `freenet::ring` are captured — i.e. // the read always returned `Ok(empty)`, exactly as the filter predicts. // 2. Half of the state is meaningless across a restart anyway. The // per-peer `peer_adjustments` are keyed to a neighbour set that CONNECT // rebuilds differently on every start. // 3. A partial restore is what caused #4808. Reloading a log that holds // only originator events (relay hops never persist) discards everything // the live router learned by relaying. // // The cost of starting cold is bounded: post-#4809 a peer reaches // `MIN_EVENTS_FOR_PREDICTION` in a few hours at observed event rates. // // The estimators keep themselves batch-accurate: `IsotonicEstimator::add_event` // refits inline once its window has turned over (#4811). There is no periodic // refit task — `refit_router_periodically` was deleted with this change, // because polling was the only thing it did. // Interval for topology snapshot registration (1 second in test mode) // Registers subscription topology with the global registry for validation #[cfg(any(test, feature = "testing"))] const TOPOLOGY_SNAPSHOT_INTERVAL: Duration = Duration::from_secs(1); // Just initialize with a fake location, this will be later updated when the peer has an actual location assigned. let peer_cache_dir = if is_gateway { // Gateways don't need peer cache — peers connect to them. None } else { Some(config.config.data_dir()) }; let time_source: Arc = Arc::new(InstantTimeSrc::new()); // Production always passes `None`, which resolves to // `GovernanceConfig::default()`. The override exists only so // simulation tests can inject compressed timescales + lower // `min_samples` to drive the rate-limit → MAD → evict → ban // chain within a paused-time sim. See issue #4301. let governance_config = config .governance_config_override .clone() .unwrap_or_default(); let governance = Arc::new(crate::contract::governance::GovernanceManager::new( governance_config, time_source.clone(), )); // Read before `connection_manager` is moved into the literal. // The UPDATE limiter's per-sender budget is keyed by the // immediate upstream hop, so its map is sized from this node's // OWN connection cap rather than a hardcoded default. let max_connections = connection_manager.max_connections; // Built here, after the time source and the connection cap it depends // on: the hierarchical routing estimator's horizons run on the ring's // `InstantTimeSrc`, which reads tokio's clock (so they advance under a // paused tokio runtime, but do NOT follow a hosting-only time override), // and its peer tables are sized from this node's own `max_connections`. let router = Arc::new(RwLock::new( Router::new(&[]) .with_time_source(time_source.clone()) .with_max_connections(max_connections), )); crate::node::network_status::set_router(router.clone()); let ring = Ring { max_hops_to_live, router, route_failure_causes: RouteFailureCauseCounts::default(), timeout_label_window: parking_lot::Mutex::new(TimeoutLabelWindow::default()), connection_manager, // Production passes the Ring's default `Arc` // (wall clock). Simulation tests can inject a controllable clock via // `NodeConfig::hosting_time_source_override` so hosting-cache TTL / // eviction is deterministic (#4642 piece A). Same `Arc` clone as the // rest of the Ring when no override is set. hosting_manager: hosting::HostingManager::with_time_source( config.config.max_hosting_storage, config .hosting_time_source_override .clone() .unwrap_or_else(|| time_source.clone()), ), broken_invariants: BrokenInvariantsTracker::new(time_source.clone()), governance, update_rate_limiter: Arc::new(update_rate_limit::UpdateRateLimiter::new( // Production passes `None` and gets `time_source` (the real // clock). The override exists so a test can decide the // limiter's verdict instead of racing MIN_UPDATE_INTERVAL — // see `NodeConfig::update_rate_limit_time_source_override`. config .update_rate_limit_time_source_override .clone() .unwrap_or_else(|| time_source.clone()), max_connections, )), merge_backoff: Arc::new(merge_backoff::MergeBackoff::new(time_source.clone())), delta_incompat: Arc::new(delta_incompat::DeltaIncompat::new(time_source.clone())), resync_emit_limiter: Arc::new(resync_rate_limit::new_emit_limiter(time_source.clone())), resync_response_limiter: Arc::new(resync_rate_limit::new_response_limiter( time_source.clone(), )), resync_response_global_limiter: Arc::new( resync_rate_limit::new_response_global_limiter(time_source.clone()), ), outstanding_resync_requests: Arc::new( resync_rate_limit::new_outstanding_resync_requests(time_source.clone()), ), contract_ban_list: Arc::new(contract_ban_list::ContractBanList::new( time_source.clone(), )), live_tx_tracker: live_tx_tracker.clone(), event_register: Box::new(event_register), op_manager: RwLock::new(None), is_gateway, connection_backoff: Arc::new(Mutex::new(ConnectionBackoff::new())), contract_connect_backoff: Mutex::new(HashMap::new()), time_source, peer_cache_dir, // One sink per node, shared with the module caches via the `Arc` // (the `RuntimePool` reaches it through `op_manager.ring`). See // the field docs and #4488. module_cache_metrics: Arc::new(crate::wasm_runtime::ModuleCacheMetrics::new()), contract_exec_metrics: Arc::new(contract_exec_metrics::ContractExecMetrics::default()), // One placement-migration counter sink per node, shared with the // migration send/receive sites via the `Arc` (reached through // `op_manager.ring`). See the field docs. placement_migration_metrics: Arc::new( placement_migration_metrics::PlacementMigrationMetrics::default(), ), shutdown, }; if let Some(loc) = config.location { if config.own_addr.is_none() && is_gateway { return Err(anyhow::anyhow!("own_addr is required for gateways")); } ring.connection_manager.update_location(Some(loc)); } let ring = Arc::new(ring); // Conformance focus selects over the contracts this peer HOSTS (RFC #5320), // so the ring - which owns the hosting cache - is what can answer that. A // `Weak` deliberately: this closure is stored in a process-global that // outlives the node, and holding a strong `Arc` there would keep an entire // ring, its background tasks' handles and its caches alive past teardown. A // dead ring reads as "no candidates", which surfaces as `focused=0` rather // than as a stale set. Registration is a no-op unless capture is enabled. { let weak = Arc::downgrade(&ring); crate::conformance::capture::set_hosted_contracts_source(Box::new(move || { weak.upgrade() .map(|ring| { ring.hosting_manager .hosting_contract_keys() .into_iter() .map(|key| *key.id()) .collect() }) .unwrap_or_default() })); } let current_span = tracing::Span::current(); let span = if current_span.is_none() { tracing::info_span!("connection_maintenance") } else { tracing::info_span!(parent: current_span, "connection_maintenance") }; task_monitor.register( "connection_maintenance", GlobalExecutor::spawn({ let fut = ring .clone() .connection_maintenance(event_loop_notifier, live_tx_tracker) .instrument(span); async move { if let Err(e) = fut.await { tracing::error!(error = %e, "connection_maintenance exited with error"); } } }), ); // Spawn periodic subscription state telemetry task task_monitor.register( "emit_subscription_state_telemetry", GlobalExecutor::spawn(Self::emit_subscription_state_telemetry( ring.clone(), Self::SUBSCRIPTION_STATE_INTERVAL, )), ); // Spawn periodic subscription recovery task to fix "orphaned hosters" // (peers that have contracts cached but aren't in the subscription tree) task_monitor.register( "recover_orphaned_subscriptions", GlobalExecutor::spawn(Self::recover_orphaned_subscriptions( ring.clone(), Self::SUBSCRIPTION_RECOVERY_INTERVAL, )), ); // Spawn periodic GET subscription cache sweep task // Cleans up expired GET-triggered subscriptions to maintain bounded memory task_monitor.register( "sweep_get_subscription_cache", GlobalExecutor::spawn(Self::sweep_get_subscription_cache( ring.clone(), Self::GET_SUBSCRIPTION_SWEEP_INTERVAL, )), ); // Spawn periodic topology snapshot registration task (test mode only) // This allows SimNetwork to validate subscription topology during tests #[cfg(any(test, feature = "testing"))] task_monitor.register( "register_topology_snapshots", GlobalExecutor::spawn(Self::register_topology_snapshots_periodically( ring.clone(), TOPOLOGY_SNAPSHOT_INTERVAL, )), ); // Spawn periodic router model snapshot telemetry (every 5 minutes) task_monitor.register( "emit_router_snapshot_telemetry", GlobalExecutor::spawn(Self::emit_router_snapshot_telemetry( ring.clone(), Duration::from_secs(60 * 5), )), ); // Peer-attribute snapshots for the opt-in routing dataset (#4485). Spawned // only when an operator enabled recording, so a default node — and every // simulation — runs no extra task and consumes no extra timer. if let Some(dataset) = crate::router::dataset::global() { task_monitor.register( "record_routing_dataset_peers", GlobalExecutor::spawn(Self::record_routing_dataset_peers( ring.clone(), dataset, ROUTING_DATASET_PEER_INTERVAL, )), ); } // Spawn periodic contract-directed CONNECT task. // When a peer is a "subscription root" (closest to contract among neighbors), // it sends CONNECTs toward the contract's ring location to merge disconnected // subscription subtrees. const CONTRACT_CONNECT_INTERVAL: Duration = Duration::from_secs(30); task_monitor.register( "contract_directed_connects", GlobalExecutor::spawn(Self::contract_directed_connects( ring.clone(), CONTRACT_CONNECT_INTERVAL, )), ); // Spawn periodic interest heartbeat task. // Sends full Interests { hashes } to each connected peer to keep // interest entries alive and prevent the death spiral where expired // entries block broadcast delivery. task_monitor.register( "interest_heartbeat", GlobalExecutor::spawn(Self::interest_heartbeat(ring.clone())), ); // NOTE: the advertisement-layer anti-entropy re-request (#4642 spec step 1, // "Fix 1") is NOT a separate task — it is piggybacked on the // `interest_heartbeat` loop above (see the `HostingStateRequest` send // there). A separate task with its own `GlobalRng` initial-delay draw // perturbed the shared per-thread seeded RNG stream that the deterministic // simulation harness depends on; riding the existing heartbeat loop adds no // RNG consumer and no second timer, so the sim stream is unchanged. // Spawn periodic governance reaper tick. // Computes per-contract state from accumulated cost/benefit // samples and logs `ReaperDecision`s. Mode defaults to `Off` // (see `GovernanceConfig` in contract/governance.rs): the MAD // outlier detector is dormant and being replaced by demand-driven // eviction (#4296, #4642), so this task is a no-op on default // nodes and only ever logs (never evicts) even when an operator // explicitly flips it to DryRun/Enforce. The dashboard reads the // governance snapshot independently of the tick; the tick is only // the engine that updates state. task_monitor.register( "governance_reaper", GlobalExecutor::spawn(Self::governance_reaper_loop(ring.clone())), ); Ok(ring) } pub fn attach_op_manager(&self, op_manager: &Arc) { self.op_manager.write().replace(Arc::downgrade(op_manager)); // The hosting cache charges each hosted contract the neighbour-summary // bytes the interest manager holds for it (#5647). Holding the // `InterestManager` Arc directly creates no cycle: it references // neither the ring nor the hosting manager. let interest = op_manager.interest_manager.clone(); interest.set_neighbour_summary_budget(self.neighbour_summary_budget()); self.hosting_manager .set_interest_bytes_provider(Arc::new(move |key| interest.resident_bytes_for(key))); } /// Shared per-node module-cache telemetry sink (#4440 / #4488). The /// `RuntimePool` clones this into its labeled module caches (which publish /// occupancy/eviction into it) while the snapshot task reads it; threading /// this `Arc` is what replaced the old `MODULE_CACHE_METRICS` process-global. pub(crate) fn module_cache_metrics(&self) -> Arc { self.module_cache_metrics.clone() } /// Shared per-node contract-exec WASM counters. Returns a BORROW, not an /// `Arc` clone: the increment sites sit on the contract-handling loop's hot /// path (tens of calls/sec per node), where an `Arc` refcount bump would be /// a needless atomic RMW on top of the counter's own. The snapshot reader /// borrows the same way on its 5-minute cadence. #[inline] pub(crate) fn contract_exec_metrics(&self) -> &contract_exec_metrics::ContractExecMetrics { &self.contract_exec_metrics } /// Shared per-node placement-migration counter sink (#4404 follow-up). The /// migration send/receive sites clone this to increment `sent` / `received` /// / `acted`; the snapshot task reads it on the `router_snapshot` cadence. pub(crate) fn placement_migration_metrics( &self, ) -> Arc { self.placement_migration_metrics.clone() } /// Signal the long-lived background tasks (spawned in [`Ring::new`]) to /// shut down. Idempotent and thread-safe — calling it more than once, or /// from multiple threads, is a no-op after the first call. /// /// Fired from `ShutdownTeardown::drop` on every `run_node` exit path so a /// graceful (or error-triggered) shutdown doesn't wait for the longest /// outstanding sleep in those loops to elapse. See issue #4278. pub(crate) fn trigger_shutdown(&self) { self.shutdown.cancel(); } /// A clone of the background-task shutdown token. Long-lived loops race /// their sleeps against [`CancellationToken::cancelled`] on this token via /// `tokio::select!`. pub(crate) fn shutdown_token(&self) -> CancellationToken { self.shutdown.clone() } pub(crate) fn upgrade_op_manager(&self) -> Option> { self.op_manager .read() .as_ref() .and_then(|weak| weak.clone().upgrade()) } /// Classify the current state of the `OpManager` back-reference. /// /// The maintenance loop must distinguish "not attached yet" (normal /// startup window, before [`Ring::attach_op_manager`] runs) from /// "attached then dropped" (the node has been torn down and the /// `Arc` is gone). The first case must keep waiting; the /// second must terminate the loop so it doesn't become a zombie /// spinning forever on a `Weak` that can never upgrade again (#3308). /// /// This is the sole production instantiation of the generic /// [`classify_op_manager_ref`] (with `T = OpManager`); the unit tests /// instantiate it with a stand-in `Arc` to exercise the same logic. fn op_manager_state(&self) -> OpManagerState { classify_op_manager_ref(&self.op_manager) } pub fn is_gateway(&self) -> bool { self.is_gateway } pub fn open_connections(&self) -> usize { self.connection_manager.connection_count() } /// Record a connection failure to the backoff tracker. pub fn record_connection_failure(&self, target: Location, reason: ConnectionFailureReason) { let mut backoff = self.connection_backoff.lock(); backoff.record_failure_with_reason(target, reason); } /// Record a short, non-escalating backoff for `target` (#4362). /// /// Used when a capacity `Rejected` arrives while the node is still below /// `min_connections`: we want to throttle the retry cadence briefly so it /// stops hammering a saturated gateway neighborhood every fast-tick, /// WITHOUT stamping the escalating 30s→600s backoff that would trap the /// under-connected node. See `ConnectionBackoff::record_short_reject_backoff`. pub fn record_connection_short_reject_backoff(&self, target: Location) { let mut backoff = self.connection_backoff.lock(); backoff.record_short_reject_backoff(target); } /// Record a successful connection to clear backoff. pub fn record_connection_success(&self, target: Location) { let mut backoff = self.connection_backoff.lock(); backoff.record_success(target); } /// Check if a target is currently in backoff. pub fn is_in_connection_backoff(&self, target: Location) -> bool { self.connection_backoff.lock().is_in_backoff(target) } /// Periodic cleanup of expired backoff entries. pub fn cleanup_connection_backoff(&self) { self.connection_backoff.lock().cleanup_expired(); } /// Reset all connection backoff state. Used during isolation recovery /// when the node has had zero ring connections for an extended period. pub fn reset_all_connection_backoff(&self) { self.connection_backoff.lock().clear(); } // ==================== Contract-Directed CONNECT ==================== /// Initial backoff before retrying a contract-directed CONNECT. const INITIAL_CONTRACT_CONNECT_BACKOFF: Duration = Duration::from_secs(30); /// Maximum backoff cap for contract-directed CONNECTs (24 hours). const MAX_CONTRACT_CONNECT_BACKOFF: Duration = Duration::from_secs(24 * 60 * 60); /// Maximum contract-directed CONNECTs per cycle. const MAX_CONTRACT_CONNECTS_PER_CYCLE: usize = 2; /// If this peer is the body-holding subscription root for the contract /// identified by `instance_id`, returns the resolved [`ContractKey`]; /// otherwise returns `None`. /// /// "Body-holding subscription root" means: this peer hosts the contract (has /// the body) AND no connected neighbor is closer to the contract's ring /// location than this peer — the "body-holding terminus" from the /// placement-migration design. Such a peer has no peer closer than itself to /// subscribe to, so a renewal toward the contract would dead-end and retry /// forever — the #4440 renewal storm. The renewal driver (which holds only /// the instance id) uses this to short-circuit (proposal 1); it returns the /// key so the caller can refresh the local lease without a second lookup. /// /// Resolves the hosted [`ContractKey`] by matching `instance_id` against the /// hosting set (a node hosts at most one contract per instance id, so the /// match is exact), then delegates to [`Self::is_subscription_root`], whose /// definition already requires `is_hosting_contract` (= has body) and /// closest-connected, so the two never disagree. Returns `None` when the /// contract is not hosted (no body → not a body-holding terminus). pub(crate) fn body_holding_subscription_root_key( &self, instance_id: &ContractInstanceId, ) -> Option { // `hosting_contract_keys()` clones the hosting set and we linear-scan for // the matching instance id. This runs at most once per renewal task // (`SUBSCRIPTION_RECOVERY_INTERVAL` = 30 s, bounded by // `MAX_RECOVERY_ATTEMPTS_PER_INTERVAL` per cycle), so the clone+scan is // not on any hot path. A reverse instance-id → key index on the hosting // cache would avoid the clone if this ever becomes per-message. let key = self .hosting_manager .hosting_contract_keys() .into_iter() .find(|k| k.id() == instance_id)?; self.is_subscription_root(&key).then_some(key) } /// Record that a renewal short-circuited because this node is the /// body-holding subscription root for the contract (#4440 proposal 1). pub(crate) fn record_renewal_terminus_satisfied(&self) { self.placement_migration_metrics .record_renewal_terminus_satisfied(); } /// Returns true if this peer is the closest to the contract among its connected neighbors /// (i.e., it's a subscription root for this contract). /// /// `pub(crate)` so the reconcile input-builder (keystone step-2, #4642) can /// populate `ReconcileInputs::is_verified_root`. Note the `is_hosting_contract` /// precondition below: a peer that does not yet hold the body is not reported /// as root (the builder inherits that binary hosting semantics). pub(crate) fn is_subscription_root(&self, contract_key: &ContractKey) -> bool { if !self.is_hosting_contract(contract_key) { return false; } let contract_location = Location::from(contract_key); let my_location = match self.connection_manager.own_location().location() { Some(loc) => loc, None => return false, }; let my_distance = my_location.distance(contract_location); let connections = self.connection_manager.get_connections_by_location(); let neighbors = connections.iter().flat_map(|(_loc, conns)| { conns.iter().map(|conn| { // A peer counts as a "closer routable neighbor" (so I'm NOT the // root) only if a renewal could actually route to it. Match the // ACTUAL eligibility `k_closest_potentially_hosting` applies, and // mind its asymmetry between the two filters: // // * Transient peers (short-TTL CONNECT-coordination slots) are // excluded UNCONDITIONALLY by `k_closest_potentially_hosting` // (no fallback — see its `is_transient(addr)` / // `skipped_transient` branch). So a transient closer neighbor // is NOT a real route target — exclude it here too. Without // this, a node whose only closer neighbor is transient would // fail to recognise itself as the terminus, wire-renew, // dead-end (the transient is excluded by `k_closest`), and // storm (#4440). // // * Not-yet-ready peers are excluded by `k_closest` ONLY when // ready candidates exist; if there are none it FALLS BACK to // the not-ready peers (its `not_ready_fallback` path). So a // not-ready closer neighbor CAN still be a route target. Treat // it as routable // here (conservative): excluding it would wrongly classify the // node as the root in early-startup / low-degree topologies and // suppress a renewal that would in fact route to that closer // peer, delaying upstream subscription propagation. The cost of // treating a not-ready peer as routable is at most one extra // wire renewal (re-evaluated next cycle once it is ready), // which is strictly safer than suppressing a needed renewal. // // Addressless peers are treated as routable (conservative): they // bypass the addr-keyed filters in `k_closest` too, so a renewal // could still route to one. let routable = match conn.location.socket_addr() { Some(addr) => !self.connection_manager.is_transient(addr), None => true, }; (conn.location.location(), routable) }) }); no_closer_routable_neighbor(my_distance, contract_location, neighbors) } /// Compute the CURRENT upstream for a contract this peer hosts: among /// `hosting_neighbors` (connected neighbors that have advertised hosting the /// contract, from `NeighborHostingManager::neighbors_with_contract_id`), the /// one CLOSEST to the contract's key that is STRICTLY closer to the key than /// this peer. Returns `None` when no such neighbor exists — either because /// this peer is the most-keyward host it can see (a terminus, design §5d-a) /// or because closer neighbors exist but none host the contract (a stranded /// host that must re-root, §5d-b). No current caller distinguishes those two /// `None` cases — this step is observation-only; a later keystone step will /// (via `is_subscription_root`). /// /// This is the *computed upstream* of the demand-driven-hosting design (§4, /// #4642 piece D): derived on demand from live neighbor-hosting /// advertisements + ring distance, never a stored formation flag. "Strictly /// closer to the key" makes the upstream relation a strict descent toward the /// key, so it is **acyclic by construction** (§4 point 2); the deterministic /// peer-address tiebreak makes equidistant hosts a total order (§6 point 2). /// /// It is introduced here alongside — not in place of — the stored /// `is_upstream` interest flag: the reconcile-core keystone will compute /// upstream everywhere and delete the stored flag, but this first step only /// computes it in parallel to measure how often the two disagree (the /// `hosting_upstream_computed_vs_stored_divergence` telemetry recorded at /// `OpManager::send_unsubscribe_upstream`), producing field evidence for the /// flag's drift (#4671). No decision consults this method yet. pub(crate) fn most_keyward_hosting_neighbor( &self, instance_id: &ContractInstanceId, hosting_neighbors: &[TransportPublicKey], ) -> Option { let contract_location = Location::from(instance_id); let my_location = self.connection_manager.own_location().location()?; let my_distance = my_location.distance(contract_location); // Resolve each advertised pub key to a live connection with a known // location, then delegate the strict-closer + deterministic-tiebreak // selection to `most_keyward_among` (unit-tested in `most_keyward_tests`). let candidates = hosting_neighbors .iter() .filter_map(|pk| self.connection_manager.get_peer_by_pub_key(pk)) .filter_map(|pkl| { let dist = pkl.location()?.distance(contract_location); Some((pkl, dist)) }); most_keyward_among(my_distance, candidates) } /// Check if a contract-directed CONNECT is currently in backoff. fn is_in_contract_connect_backoff(&self, contract_key: &ContractKey) -> bool { let backoff = self.contract_connect_backoff.lock(); if let Some(state) = backoff.get(contract_key) { state.last_attempt.elapsed() < state.current_backoff } else { false } } /// Record a contract-directed CONNECT attempt. Doubles the backoff (up to cap). fn record_contract_connect_attempt(&self, contract_key: &ContractKey) { let mut backoff = self.contract_connect_backoff.lock(); let now = self.time_source.now(); let state = backoff .entry(*contract_key) .or_insert_with(|| ContractConnectState { current_backoff: Self::INITIAL_CONTRACT_CONNECT_BACKOFF, last_attempt: now, }); state.last_attempt = now; state.current_backoff = (state.current_backoff * 2).min(Self::MAX_CONTRACT_CONNECT_BACKOFF); } /// Periodic task: when this peer is a subscription root for a contract, /// initiate a CONNECT toward the contract's ring location to merge /// disconnected subscription subtrees. async fn contract_directed_connects(ring: Arc, interval: Duration) { let shutdown = ring.shutdown_token(); // Random initial delay to prevent thundering herd. let initial_delay = Duration::from_secs(GlobalRng::random_range(30u64..=60u64)); if sleep_or_shutdown(&shutdown, tokio::time::sleep(initial_delay)).await { return; } let mut tick_interval = tokio::time::interval(interval); tick_interval.tick().await; // skip first immediate tick loop { if sleep_or_shutdown(&shutdown, async { tick_interval.tick().await; }) .await { return; } // Skip if we have too few connections to be meaningful. let conn_count = ring.connection_manager.connection_count(); if conn_count < 2 { continue; } let contracts = ring.hosting_contract_keys(); if contracts.is_empty() { continue; } let Some(op_manager) = ring.upgrade_op_manager() else { continue; }; let mut connects_this_cycle = 0; for contract_key in &contracts { if connects_this_cycle >= Self::MAX_CONTRACT_CONNECTS_PER_CYCLE { break; } if !ring.is_subscription_root(contract_key) { // No longer root — a closer connection now exists. // If we had backoff state, we previously sent a CONNECT as root. // Now that a closer peer is connected, expire the subscription // so it re-routes through the closer peer on the next renewal cycle. let had_backoff = ring .contract_connect_backoff .lock() .remove(contract_key) .is_some(); if had_backoff { ring.force_subscription_renewal(contract_key); tracing::info!( contract = %contract_key, "No longer subscription root after contract-directed CONNECT; \ expired subscription to re-route through closer peer" ); } continue; } let contract_location = Location::from(contract_key); let my_location = ring.connection_manager.own_location().location(); let my_distance = my_location.map(|l| l.distance(contract_location).as_f64()); if ring.is_in_contract_connect_backoff(contract_key) { continue; } // Emit telemetry for the root detection. crate::tracing::telemetry::send_standalone_event( "subscription_root_detected", serde_json::json!({ "contract": contract_key.to_string(), "contract_location": contract_location.as_f64(), "neighbor_count": conn_count, "my_distance": my_distance, }), ); ring.record_contract_connect_attempt(contract_key); let backoff_secs = { let b = ring.contract_connect_backoff.lock(); b.get(contract_key) .map(|s| s.current_backoff.as_secs()) .unwrap_or(0) }; tracing::info!( contract = %contract_key, %contract_location, backoff_secs, "Initiating contract-directed CONNECT as subscription root" ); crate::tracing::telemetry::send_standalone_event( "contract_directed_connect", serde_json::json!({ "contract": contract_key.to_string(), "contract_location": contract_location.as_f64(), "my_distance": my_distance, "backoff_secs": backoff_secs, }), ); let skip_list = HashSet::new(); match ring .acquire_new( contract_location, &skip_list, &op_manager.to_event_listener, &ring.live_tx_tracker, &op_manager, ClientConnectKind::Standard, ) .await { Ok(Some(tx)) => { tracing::debug!( %tx, contract = %contract_key, "Contract-directed CONNECT initiated" ); connects_this_cycle += 1; } Ok(None) => { tracing::debug!( contract = %contract_key, "Contract-directed CONNECT: no routing target found" ); } Err(e) => { tracing::warn!( contract = %contract_key, error = %e, "Contract-directed CONNECT failed" ); } } } } } /// Periodic heartbeat: send `Interests { hashes }` to each connected peer. /// /// This prevents the death spiral where interest entries expire because no /// broadcasts are flowing, which in turn prevents broadcasts from ever /// flowing again. By periodically re-sending the full interest set, we /// Periodic governance reaper loop. Ticks every /// `GOVERNANCE_TICK_INTERVAL` and: /// /// 1. Calls `governance_tick(interval)` which snapshots live /// benefit, applies cost decay, and runs the MAD detector /// over the per-contract distribution. /// 2. Logs each `ReaperDecision` at INFO level with the from→to /// state and reason. In `DryRun` mode (the default) this is /// the only effect; in `Enforce` mode the caller would also /// emit `EvictContract` events here. Enforce-mode wiring /// lands in a subsequent commit. /// 3. Logs the network-norms summary (median, MAD, threshold, /// sample size) at DEBUG so a busy node doesn't spam INFO. /// /// **Cancellation safety:** the loop's only `.await` points are the /// shutdown-raced `tokio::time::sleep` / `interval.tick()` (via /// [`sleep_or_shutdown`]). No channel sends, no long-running operations — /// the design-doc rule that "the dashboard reflects the back-end" is /// respected here too: the reaper writes governance state; readers consume /// it asynchronously. On shutdown the loop returns immediately rather than /// finishing the current sleep (#4278). async fn governance_reaper_loop(ring: Arc) { // Initial delay so a fleet-wide restart doesn't have every node // ticking in lockstep. Derived from the node's own ring location // (already random + node-distinct) rather than from `GlobalRng`, // because consuming from `GlobalRng` at startup shifts the // deterministic seed state and breaks simulation tests whose // routing/topology decisions are RNG-driven (the // `test_get_routing_coverage_low_htl` regression was caught by // CI on PR #4270). 30-90 second offset; the tick itself runs // every minute below. let own_loc = ring .connection_manager .own_location() .location() .map(|l| l.as_f64()) .unwrap_or(0.5); let initial_delay_secs = 30 + ((own_loc * 60.0) as u64); let initial_delay = Duration::from_secs(initial_delay_secs); let shutdown = ring.shutdown_token(); if sleep_or_shutdown(&shutdown, tokio::time::sleep(initial_delay)).await { return; } let mut interval = tokio::time::interval(GOVERNANCE_TICK_INTERVAL); interval.tick().await; // Skip first immediate tick. let mut last_tick = ring.time_source.now(); loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } let now = ring.time_source.now(); let elapsed = now.saturating_duration_since(last_tick); last_tick = now; let result = ring.governance_tick(elapsed); // Cycle the UPDATE rate limiter's idle-entry cleanup on // the same cadence. Keeps the `(sender, contract)` map // bounded; CLEANUP_AGE is 5 minutes so this drops entries // for pairs that haven't tried an UPDATE since then. ring.update_rate_limiter.cleanup(); // Sweep idle merge-failure backoff entries (#4861) on the same // cadence. Drops entries for contracts past their cooldown that // haven't failed a merge recently; a poison contract still in // cooldown is preserved. ring.merge_backoff.cleanup_expired(); // Sweep stale delta-send attributions and idle unarmed entries in // the delta-incompatibility memo (HQk7 resync loop); an armed // (unexpired) memo is preserved. ring.delta_incompat.cleanup(); // Sweep recovered/idle resync rate-limiter buckets (#4861) so the // per-contract emit and per-(peer, contract) responder maps stay // bounded. ring.resync_emit_limiter.cleanup(); ring.resync_response_limiter.cleanup(); ring.resync_response_global_limiter.cleanup(); // Reap outstanding-ResyncRequest correlation entries past their TTL // (#4864 round-8) so a burst of requests whose responses never arrive // cannot pin the map past OUTSTANDING_RESYNC_TTL. ring.outstanding_resync_requests.cleanup(); // Phase 7 ban-list maintenance. Defense-in-depth — the // BanLifted decisions below explicitly unban, but if any // entry slipped through the reaper (e.g. mode flipped off // mid-window), cleanup catches it on the next tick. ring.contract_ban_list.cleanup(); // Sweep expired broken-invariant flags on the same cadence so a // suppressed contract (especially a false-positive) recovers // after BROKEN_INVARIANT_TTL instead of being bricked forever. ring.broken_invariants.cleanup(); // Network-norms summary at DEBUG — useful when debugging // calibration but too noisy for INFO on a healthy node. tracing::debug!( median = ?result.median_log_ratio, mad = ?result.mad, threshold = ?result.threshold, sample_size = result.sample_size, capacity_ceiling_binding = result.capacity_ceiling_binding, skip_reason = ?result.skip_reason, "governance reaper tick", ); // Phase 7 enforcement wiring: BanTriggered → ban list; // BanLifted → unban. Done AFTER the cleanup() so a // freshly-added entry (with expiry = now + ban_ttl) is // not immediately swept by the same tick's cleanup. The // two calls are sequential in this task — no concurrent // race — the ordering is purely to defend the // not-quite-expired-yet invariant. Self::apply_ban_decisions( &ring.contract_ban_list, &result.decisions, ring.time_source.now() + ring.governance.ban_ttl(), ); // Decisions at INFO — these are the dashboard-relevant // events. Empty in healthy state, surfaces every // transition during incidents. for decision in result.decisions { tracing::info!( contract = %decision.key, from = ?decision.from, to = ?decision.to, reason = ?decision.reason, actionable = decision.actionable, "governance state transition", ); } } } /// Translate a batch of governance decisions into ban-list /// mutations. Pulled out of the reaper loop so the wiring is /// directly testable: a missing or reversed branch here would /// silently break Phase 7 enforcement even with the /// `GovernanceManager` emitting correct transitions. /// /// **`actionable` filter:** non-actionable decisions (DryRun /// mode) are skipped. Today the `GovernanceManager` only emits /// `BanTriggered` in Enforce mode so this is a defense-in-depth /// guard — but a future "shadow" mode that wants to surface /// would-have-banned transitions in the dashboard must not /// silently begin enforcing them through this wiring. pub(crate) fn apply_ban_decisions( ban_list: &contract_ban_list::ContractBanList, decisions: &[crate::contract::governance::ReaperDecision], ban_expiry: tokio::time::Instant, ) { use crate::contract::governance::TransitionReason; for decision in decisions { if !decision.actionable { continue; } #[allow(clippy::wildcard_enum_match_arm)] match decision.reason { TransitionReason::BanTriggered => { ban_list.ban( decision.key, ban_expiry, contract_ban_list::BanReason::AutoMad, ); } TransitionReason::BanLifted => { ban_list.unban(&decision.key); } _ => {} } } } /// keep entries alive on the remote side. /// /// Sends are spread evenly across the interval to avoid bursts. async fn interest_heartbeat(ring: Arc) { use crate::ring::interest::INTEREST_HEARTBEAT_INTERVAL; let shutdown = ring.shutdown_token(); // Random initial delay to prevent synchronized heartbeats across peers let initial_delay = Duration::from_secs(GlobalRng::random_range(15u64..=45u64)); if sleep_or_shutdown(&shutdown, tokio::time::sleep(initial_delay)).await { return; } let mut interval = tokio::time::interval(INTEREST_HEARTBEAT_INTERVAL); interval.tick().await; // Skip first immediate tick loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } let Some(op_manager) = ring.upgrade_op_manager() else { continue; }; // Get our current interest hashes. NOTE: unlike the pre-Fix-1 loop, // an empty `hashes` does NOT skip the cycle. The InterestSync interest // send is gated on `hashes` per-peer below, but the piggybacked // advertisement-layer re-request (Fix 1) must run for ANY connected // node: every node consults its `neighbor_contracts` view for terminal // GET/SUBSCRIBE advertisement lookups (invariant 5) and Source-1 UPDATE // fan-out, even a pure relay that hosts nothing and has no interests, // so that view must be reconciled on the anti-entropy cadence — else a // dropped retraction leaves a phantom advertised host unhealed // indefinitely. The only gate is having connected peers (below). let hashes = op_manager.interest_manager.get_all_interest_hashes(); // Get all connected peer addresses (deduplicated) let connections = ring.connection_manager.get_connections_by_location(); let peer_addrs: Vec = { let mut seen = HashSet::new(); connections .values() .flat_map(|conns| conns.iter()) .filter_map(|conn| conn.location.socket_addr()) .filter(|addr| seen.insert(*addr)) .collect() }; if peer_addrs.is_empty() { continue; } let num_peers = peer_addrs.len(); // num_peers >= 1 guaranteed by the is_empty() check above let spread_delay = INTEREST_HEARTBEAT_INTERVAL / num_peers as u32; tracing::debug!( num_peers, num_hashes = hashes.len(), "Interest heartbeat: sending Interests to peers" ); let sender = op_manager.to_event_listener.notifications_sender(); let mut peers_sent = 0usize; for (i, peer_addr) in peer_addrs.into_iter().enumerate() { // InterestSync STATE heartbeat: only when we actually have // interest hashes to advertise. if !hashes.is_empty() { let message = crate::message::InterestMessage::Interests { hashes: hashes.clone(), }; if let Err(e) = sender .send(either::Either::Right( crate::message::NodeEvent::SendInterestMessage { target: peer_addr, message, }, )) .await { // Channel send failure means the receiver is dropped (node // shutting down). No point sending to remaining peers. tracing::debug!( peer = %peer_addr, error = %e, "Interest heartbeat: failed to queue message" ); break; } } // Advertisement-layer anti-entropy (#4642 spec step 1, "Fix 1"), // PIGGYBACKED on the InterestSync heartbeat's cycle / peer / cadence. // Re-request this neighbor's full hosted-contract set; the // `HostingStateResponse` handler REPLACES our view of it, so a // dropped on-connect exchange or a dropped per-eviction retraction // self-heals within one heartbeat interval (both directions, since // every node runs this). ALWAYS sent when the cycle runs — // decoupled from the interest send above, so an advertising-but-not- // interested node (e.g. just-restarted with disk-loaded hosts) still // reconciles. Carried in THIS loop rather than a separate background // task on purpose: an independent task's random initial-delay draw // from the shared `GlobalRng` shifts every seeded simulation's RNG // stream (perturbing routing/topology) and a second per-node timer // adds cross-run determinism surface — reusing this loop keeps the // sim RNG stream byte-identical. It is WASM-free / state-free (only // contract IDs move), so it cannot re-arm the #4440/#4473 summarize // storm. // Best-effort and NON-BLOCKING: `try_send`, never `send().await`. // This is the anti-entropy backstop that heals next cycle, so a // request dropped under backpressure is the DESIGNED behavior — and // it rides the cap-2048 event-loop notification channel (the // #4145/#4231/#4442 wedge channel), which a blocking send inside a // background loop must never stall on (channel-safety rule). Mirrors // the sibling retraction path (`announce_contract_unhosted` uses // `try_notify_node_event`). No `break` on error: shutdown is caught // by the `sleep_or_shutdown` on the spread delay below (and by the // interest `send().await` above), so a dropped/closed send here just // moves on. if let Err(e) = sender.try_send(either::Either::Right( crate::message::NodeEvent::SendNetMessage { target: peer_addr, msg: Box::new(crate::message::NetMessage::V1( crate::message::NetMessageV1::NeighborHosting { message: crate::message::NeighborHostingMessage::HostingStateRequest, }, )), }, )) { tracing::debug!( peer = %peer_addr, error = %e, "Interest heartbeat: hosting re-request dropped (best-effort; heals next cycle)" ); } peers_sent += 1; // Spread sends evenly across the interval (skip delay after last). // `spread_delay` can be many seconds when peer count is low, so // race it against shutdown too (#4278). if i + 1 < num_peers && sleep_or_shutdown(&shutdown, tokio::time::sleep(spread_delay)).await { return; } } // #4956: tag with the local peer so this can be JOINED per-peer // against `resource_utilization` / `outbound_message_mix`. Without // it the event reports fleet-wide distributions only, which is why // the InterestSync cost hypothesis could not be confirmed by // correlation and had to stay an estimate. crate::tracing::telemetry::send_standalone_event_with_peer_id( "interest_heartbeat_cycle", &op_manager .ring .connection_manager .own_location() .to_string(), serde_json::json!({ "peers_sent": peers_sent, "interest_hashes": hashes.len(), }), ); } } /// Register events with the event system. /// This is used by operations to emit failure and other events. pub async fn register_events<'a>( &self, events: either::Either< crate::tracing::NetEventLog<'a>, Vec>, >, ) { self.event_register.register_events(events).await; } /// Periodically record the attributes of every connected peer into the /// routing dataset, keyed like its route events so the two join offline. async fn record_routing_dataset_peers( ring: Arc, dataset: &'static crate::router::dataset::RoutingDataset, interval_duration: Duration, ) { let shutdown = ring.shutdown_token(); let mut interval = tokio::time::interval(interval_duration); loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { break; } // Skip, never leave the loop: this task is registered with the // background task monitor, and any monitored task exiting ends the // node. A recorder that reached its byte cap must not take a // gateway down with it. if !dataset.is_recording() { continue; } let peers = ring.routing_dataset_peer_attributes(); // The dataset's own clock, shared with route events so the two join. dataset.record_peers(crate::router::dataset::now_ms(), peers); } } /// Snapshot connected-peer attributes. Each lock is taken on its own and /// released before the next, so this cannot participate in a lock-order /// inversion; assembly happens afterwards with no lock held. fn routing_dataset_peer_attributes(&self) -> Vec { use crate::router::dataset::{PeerSnapshotInputs, peer_attributes}; let connections: Vec<(PeerKeyLocation, f64)> = self .connection_manager .get_connections_by_location() .into_values() .flatten() .map(|connection| { let connected_s = connection.duration_ms() as f64 / 1000.0; (connection.location, connected_s) }) .collect(); let addrs: Vec = connections .iter() .filter_map(|(peer, _)| peer.socket_addr()) .collect(); let gateways: Option> = self.upgrade_op_manager().map(|op_manager| { op_manager .configured_gateways .iter() .map(|gateway| gateway.pub_key().clone()) .collect() }); let versions: HashMap = addrs .iter() .filter_map(|addr| { self.connection_manager .remote_version(*addr) .map(|version| (*addr, version)) }) .collect(); let health: HashMap = { let tracker = self.connection_manager.peer_health.lock(); addrs .iter() .filter_map(|addr| tracker.counts(addr).map(|counts| (*addr, counts))) .collect() }; let transfer: HashMap = crate::transport::metrics::TRANSPORT_METRICS .per_peer_snapshot() .into_iter() .map(|(addr, sent, received)| (addr, (sent, received))) .collect(); peer_attributes(&PeerSnapshotInputs { connections: &connections, gateways: gateways.as_deref(), versions: &versions, health: &health, transfer: &transfer, }) } /// Periodically emit a router model snapshot as an EventKind::RouterSnapshot event. /// /// This captures the isotonic regression curves and model state, including the /// connect forward estimator if available via OpManager. async fn emit_router_snapshot_telemetry(ring: Arc, interval_duration: Duration) { let shutdown = ring.shutdown_token(); let mut interval = tokio::time::interval(interval_duration); // Skip the first immediate tick interval.tick().await; // Previous lifetime broadcast-stream-failure count, so each snapshot can // emit the per-window delta directly (#4440) without a stateful // collector. Loop-local because this task is the sole reader; the // monotonic totals are still emitted for collector-side differencing. let mut prev_broadcast_stream_failures_total: u64 = crate::node::BROADCAST_STREAM_METRICS .snapshot() .streaming_failures_total; // Same shape for the contract-exec WASM counters: the monotonic totals // are emitted for collector-side differencing, and these loop-local // previous values turn the four headline arms into per-window deltas so // a SINGLE snapshot says whether this node's summarize/delta load was // cache hits or real WASM work. Seeded from the current values so the // first emitted window covers only elapsed-since-start work. let mut prev_exec = ring.contract_exec_metrics.snapshot(); // The diagnostic block is substantially wider than the ordinary // snapshot gauges. Its counters are lifetime-monotonic, so collecting // locally on every event but exporting one in six snapshots preserves // counter evidence while bounding collector/storage load. Start at five // so a newly started node reports after its first five-minute snapshot. let mut network_efficiency_tick = 5u8; loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } let mut snapshot = ring.router.read().snapshot(); // Try to include connect forward estimator data. // Drop the read guard immediately after `snapshot()` so unrelated // consumers don't queue behind us for the four field assignments // (clippy: `significant_drop_tightening`). if let Some(op_manager) = ring.upgrade_op_manager() { let (curve, data_range, events, adjustments) = op_manager.connect_forward_estimator.read().snapshot(); snapshot.connect_forward_curve = Some(curve); snapshot.connect_forward_data_range = Some(data_range); snapshot.connect_forward_events = Some(events); snapshot.connect_forward_peer_adjustments = Some(adjustments); } // Node-health gauges (#4440): make fd-exhaustion headroom observable // on the existing snapshot cadence. Best-effort — `None` where the // platform can't report it (see `read_fd_usage`). let (open_fds, fd_soft_limit) = read_fd_usage(); snapshot.open_fds = open_fds; snapshot.fd_soft_limit = fd_soft_limit; // Chain-blame soak histogram (#5657). Hand-mirrored into // `event_kind_to_json` like every field here (pinned by // `router_snapshot_json_includes_timeout_label_histogram`). let (peers_1, peers_2_3, peers_4_7, peers_8_plus, max_per_peer, untracked) = ring.take_timeout_label_histogram(); snapshot.timeout_label_peers_1 = Some(peers_1); snapshot.timeout_label_peers_2_3 = Some(peers_2_3); snapshot.timeout_label_peers_4_7 = Some(peers_4_7); snapshot.timeout_label_peers_8_plus = Some(peers_8_plus); snapshot.timeout_label_max_per_peer = Some(max_per_peer); snapshot.timeout_labels_untracked = Some(untracked); // Nearest-neighbor ring-lattice completeness + probe health (#4760), // mirrored from the home-page ring-stats provider (see // `node/p2p_impl.rs`) onto the central-telemetry snapshot cadence so // the lattice fix's network-wide impact — fraction of peers holding // both immediate-neighbor edges, median edge distances, probe success // rate — is a one-line query over central telemetry as 0.2.96 rolls // out (#4642). Same calls, same `None`/`0` semantics as the home page: // a distance is `None` when that side is unheld or own location is // unknown; the probe counters are `0` before any probe fires. let cm = &ring.connection_manager; let lattice_successor = cm.nearest_lattice_neighbor_dist(true); let lattice_predecessor = cm.nearest_lattice_neighbor_dist(false); let (lattice_probes_issued, lattice_probe_improvements) = cm.lattice_probe_stats(); snapshot.lattice_has_successor = Some(lattice_successor.is_some()); snapshot.lattice_has_predecessor = Some(lattice_predecessor.is_some()); snapshot.lattice_successor_distance = lattice_successor; snapshot.lattice_predecessor_distance = lattice_predecessor; snapshot.lattice_probes_issued = Some(lattice_probes_issued); snapshot.lattice_probe_improvements = Some(lattice_probe_improvements); snapshot.lattice_probe_misses = Some(cm.lattice_probe_miss_total()); // Version-gate refusal counters (#5156): why // `supports_hash_first_summaries` / `supports_summary_first_put` // fell back to the full-bytes path, split into the two causes // with opposite implications (unknown remote version, which never // self-heals, vs a known pre-floor version, which self-heals as // the fleet upgrades). Read directly from the gate's own counters // — see `ConnectionManager::version_gate_refusal_stats`. let version_gate_refusals = cm.version_gate_refusal_stats(); snapshot.hash_first_summaries_declined_unknown_version = Some(version_gate_refusals.hash_first_declined_unknown_version); snapshot.hash_first_summaries_declined_pre_floor = Some(version_gate_refusals.hash_first_declined_pre_floor); snapshot.summary_first_put_declined_unknown_version = Some(version_gate_refusals.summary_first_put_declined_unknown_version); snapshot.summary_first_put_declined_pre_floor = Some(version_gate_refusals.summary_first_put_declined_pre_floor); // Compiled-WASM module-cache occupancy + eviction gauges (#4440), // read from the per-node `Arc` the caches publish into (they live // behind the contract-handler channel, unreachable from here; the // `RuntimePool` shares this `Arc` via `op_manager.ring`). // // Force a fresh recompute of the interest split (cold-evictable / // interested bytes + would-reclassify) FIRST, so this snapshot is // fresh even on a cache that's been idle since its last mutation — // the throttled get/insert/remove refresh is unbounded on a quiet // node (#4441/#4534 shadow-staleness fix). No-op before the runtime // pool is built. Cheap O(entries), once per snapshot. ring.module_cache_metrics.refresh_interest_shadow_now(); let mc = ring.module_cache_metrics.snapshot(); snapshot.contract_module_cache_entries = Some(mc.contract_entries); snapshot.contract_module_cache_total_bytes = Some(mc.contract_total_bytes); snapshot.contract_module_cache_budget_bytes = Some(mc.contract_budget_bytes); snapshot.contract_module_cache_evictions_total = Some(mc.contract_evictions_total); snapshot.delegate_module_cache_entries = Some(mc.delegate_entries); snapshot.delegate_module_cache_total_bytes = Some(mc.delegate_total_bytes); snapshot.delegate_module_cache_budget_bytes = Some(mc.delegate_budget_bytes); snapshot.delegate_module_cache_evictions_total = Some(mc.delegate_evictions_total); // Capability-relative hosting-budget gauges (#4642 A2): the // RAM-scaled budget, its current occupancy (occupancy/utilization // ratio = current / budget, headroom = 1 - that; derived by the // collector), the hosted-contract count, and the monotonic // budget-triggered eviction counter. These let us observe in // production whether the RAM-scaled budget keeps a small box off the // #4565 OOM path. Per-node aggregate scalars only. let hosting = ring.hosting_manager.hosting_cache_stats(); snapshot.hosting_budget_bytes = Some(hosting.budget_bytes); snapshot.hosting_current_bytes = Some(hosting.current_bytes); snapshot.hosting_contract_count = Some(hosting.contract_count); snapshot.hosting_budget_evictions_total = Some(hosting.budget_evictions_total); // Demand-ordered eviction gauge (#4642 A3): the // #4338 miscalibration signal (evictions of repeatedly-read // contracts). Per-node aggregate scalar on the same cadence. snapshot.hosting_evictions_of_recently_read_total = Some(hosting.evictions_of_recently_read_total); // PUT-durability falsifier (#4642): unread-seed evictions (a PUT // seed evicted before its first reader) and their running age sum, // on the same periodic cadence. Differenced against // `hosting_budget_evictions_total` (the total-eviction denominator) // this measures whether the "PUT is not read-demand" decision is // discarding fresh content before it can be found. snapshot.hosting_evicted_unread_total = Some(hosting.evicted_unread_total); snapshot.hosting_evicted_unread_age_secs_sum = Some(hosting.evicted_unread_age_secs_sum); // OOM-valve falsifier (#4642 subscriber-primary rework): evictions // that shed a subscribed contract under genuine RAM overflow. Stays 0 // until the Overflow trigger is wired (mechanism-only this release). snapshot.hosting_oom_valve_evictions_total = Some(hosting.oom_valve_evictions_total); // Subscribed-eviction falsifier (#4642 subscriber-primary rework): the // count of evictions that shed a SUBSCRIBED contract (the riskiest new // behavior). Unlike the OOM-valve counter this can go nonzero as soon // as the rework ships (AtCapacity now sheds subscribed as a last // resort). Same periodic cadence. snapshot.hosting_subscribed_evictions_total = Some(hosting.subscribed_evictions_total); // Cost-pressure eviction falsifier (cost-aware eviction, #4861): // zero-demand contracts shed because their attributed update-work // (CPU / broadcast fan-out) dominated the node's total. Nonzero = // the trigger is firing; runaway = floors/share miscalibrated. snapshot.hosting_cost_evictions_total = Some(hosting.cost_evictions_total); // Resident-overhead pressure axis (#5325). These were computed and // rendered on the node's own dashboard from the day the axis landed, // but never mirrored here, so the collector could not see the SECOND // eviction pressure at all: a node shedding purely under slot // pressure reported a low state-byte occupancy and nothing else. Same // hand-mirror footgun as the gauges above — pinned by // `hosting_cache_stats_fields_are_all_mirrored`, which fails when a // `HostingCacheStats` field has no reader in this block. snapshot.hosting_resident_overhead_budget_bytes = Some(hosting.resident_overhead_budget_bytes); snapshot.hosting_resident_overhead_bytes = Some(hosting.resident_overhead_bytes); snapshot.hosting_resident_overhead_evictions_total = Some(hosting.resident_overhead_evictions_total); snapshot.hosting_resident_overhead_evicted_charged_bytes_total = Some(hosting.resident_overhead_evicted_charged_bytes_total); // All neighbour-record bytes, hosted or not (#5647): compared with // the hosted part of `hosting_resident_overhead_bytes`, the excess // is what #5782's reconciliation has not yet freed. snapshot.interest_resident_bytes_total = ring .upgrade_op_manager() .map(|op| op.interest_manager.total_resident_bytes()); // Neighbour-summary bound trims and node-wide bytes (#5781). if let Some(op) = ring.upgrade_op_manager() { snapshot.interest_neighbour_summary_bytes = Some(op.interest_manager.neighbour_summary_bytes()); let (trims, bytes) = op.interest_manager.summary_bound_trim_totals(); snapshot.interest_summary_bound_trims_total = Some(trims); snapshot.interest_summary_bound_trimmed_bytes_total = Some(bytes); } // Local notification-delivery outcomes (#4681). PER-NODE counters // (see HostingManager), read once per snapshot — no per-event // stream. Read from the manager, not the stats snapshot, for the // same reason `hosting_local_hits_total` does below. let (notif_full, notif_closed, notif_none) = ring.hosting_manager.notification_delivery_counts(); snapshot.notifications_dropped_channel_full = Some(notif_full); snapshot.notifications_dropped_channel_closed = Some(notif_closed); snapshot.notifications_no_local_subscriber = Some(notif_none); // Phantom-hosting falsifier (SUBSCRIBE-retirement step 10 §1d): // current count of contracts in-use via a downstream subscriber with // NO state on disk (contract_in_use && !contract_state_present). After // the register-after-state fix this should read 0; a nonzero value // means a hop registered demand it cannot serve (#4404/#4612 phantom). // redb-scoped by construction (contract_state_present is // conservative-true elsewhere). Same hand-mirror footgun as the gauges // above — invisible to the collector unless added to event_kind_to_json // too (pinned by router_snapshot_json_includes_phantom_in_use_gauge). snapshot.phantom_in_use_contracts = Some(ring.hosting_manager.phantom_in_use_count()); // Local-client GET hit-rate (#4642 A3): served-locally vs routed to // the network. Driven by the real serve/forward decision in the // client GET handler, so it is read from the manager counters (not // cache membership). snapshot.hosting_local_hits_total = Some(ring.hosting_manager.local_get_serves()); snapshot.hosting_local_misses_total = Some(ring.hosting_manager.local_get_forwards()); // Aggregate on-disk usage gauges (#4683): state (delta-tracked) + // WASM blobs + wasmtime compile cache (both du-measured) + their sum. // `None` until the tracker is configured and seeded (early startup), // in which case the fields stay unset. Observational only in this PR. let disk_usage = ring.hosting_manager.disk_usage_stats(); if let Some(disk) = disk_usage { snapshot.hosting_disk_state_bytes = Some(disk.state_bytes); snapshot.hosting_disk_wasm_bytes = Some(disk.wasm_bytes); snapshot.hosting_disk_compile_cache_bytes = Some(disk.compile_cache_bytes); snapshot.hosting_disk_total_bytes = Some(disk.total_bytes); } // Terminal advertisement-consult counters (hosting redesign piece C, // #4646). Exported here per #4658 so the production findability // baseline — the dead-end decomposition (off-path-advertised-host // dead-ends that C closes: attempts→hits→resolved_found, vs the // no-host-near-key residual `still_not_found` that needs piece D) — // is legible in central telemetry ahead of the piece-E / 0.2.92 // decision, instead of only in the internal singleton and sim-only // metrics. Read from the per-node network_status singleton where the // consult sites record them; `None` only before the singleton is // initialized (i.e. always populated in production snapshots). if let Some((attempts, hits, resolved_found, still_not_found)) = crate::node::network_status::terminal_consult_counts() { snapshot.terminal_consult_attempts = Some(attempts); snapshot.terminal_consult_hits = Some(hits); snapshot.terminal_consult_resolved_found = Some(resolved_found); snapshot.terminal_consult_still_not_found = Some(still_not_found); } // Eviction-retraction emission counters (#5059). Same hand-mirror // footgun as every counter here: a new `RouterSnapshotInfo` field is // invisible to the collector unless added BOTH here AND in // `event_kind_to_json` (pinned by // `router_snapshot_json_includes_hosting_retraction_counters`). if let Some((emitted, dropped)) = crate::node::network_status::hosting_retraction_counts() { snapshot.hosting_retractions_emitted = Some(emitted); snapshot.hosting_retractions_dropped = Some(dropped); } // Streamed-transfer abort counters (Group B): isolate the // large-contract failure class on the existing snapshot cadence. // Same hand-mirror footgun as the counters above — a new // `RouterSnapshotInfo` field is invisible to the collector unless // added here AND in `event_kind_to_json` (pinned by // `router_snapshot_json_includes_stream_abort_counters`). Read from // the per-node network_status singleton where the abort sites record // them; `None` only before the singleton is initialized. if let Some(a) = crate::node::network_status::stream_abort_counts() { snapshot.stream_recv_aborts_inactivity_total = Some(a.recv_inactivity); snapshot.stream_recv_aborts_cancelled_total = Some(a.recv_cancelled); snapshot.stream_recv_aborts_claim_timeout_total = Some(a.recv_claim_timeout); snapshot.stream_recv_aborts_deserialize_total = Some(a.recv_deserialize); snapshot.stream_send_aborts_cwnd_total = Some(a.send_cwnd); snapshot.stream_recv_abort_frac_0 = Some(a.frac_0); snapshot.stream_recv_abort_frac_1 = Some(a.frac_1); snapshot.stream_recv_abort_frac_lt50 = Some(a.frac_lt50); snapshot.stream_recv_abort_frac_50_90 = Some(a.frac_50_90); snapshot.stream_recv_abort_frac_ge90 = Some(a.frac_ge90); } // Routing/hosting attribution (Group C): current connection gauges // from `connection_manager` plus relayed-op counters and the // gateway-connection gauge from the network_status singleton. Same // hand-mirror footgun — pinned by // `router_snapshot_json_includes_routing_attribution_counters`. snapshot.ring_connections = Some(cm.connection_count() as u64); snapshot.transient_connections = Some(cm.transient_count() as u64); snapshot.connections_to_gateways = crate::node::network_status::connections_to_gateways(); if let Some((gets, puts, subscribes, updates)) = crate::node::network_status::relayed_op_counts() { snapshot.relayed_gets_total = Some(gets); snapshot.relayed_puts_total = Some(puts); snapshot.relayed_subscribes_total = Some(subscribes); snapshot.relayed_updates_total = Some(updates); } // Connect-event emission counters (firehose-retirement precursor). // Additive aggregate for a future net-negative retirement of the // per-event `connect_connected` / `connect_rejected` firehose. if let Some((accepts, rejects)) = crate::node::network_status::connect_emit_counts() { snapshot.connect_accepts_emitted = Some(accepts); snapshot.connect_rejects_emitted = Some(rejects); } // Bootstrap-acceptance-churn counters (#4787): gateway-side // transient registration/expiry/promotion totals plus // joiner-side time-to-min-connections and startup retry count. // Instrumentation only, no behavior change — see // `network_status::BootstrapChurnStats`. if let Some(b) = crate::node::network_status::bootstrap_churn_counts() { snapshot.bootstrap_transient_registered = Some(b.transient_registered); snapshot.bootstrap_transient_expired = Some(b.transient_expired); snapshot.bootstrap_promoted_to_ring = Some(b.promoted_to_ring); snapshot.bootstrap_time_to_min_connections_secs = b.time_to_min_connections.map(|d| d.as_secs_f64()); // `Some(false)` (never bootstrapped) is a different fact from // `None` (this build/collector doesn't report it) — see #4787 // finding 3. snapshot.bootstrap_completed = Some(b.time_to_min_connections.is_some()); snapshot.bootstrap_startup_rounds_connect_issued_gateway = Some(b.startup_rounds_connect_issued_gateway); snapshot.bootstrap_startup_rounds_connect_issued_routed = Some(b.startup_rounds_connect_issued_routed); snapshot.bootstrap_startup_rounds_backoff_blocked = Some(b.startup_rounds_backoff_blocked); snapshot.bootstrap_startup_rounds_no_target = Some(b.startup_rounds_no_target); } // Computed-upstream vs. stored-`is_upstream`-flag divergence counters // (#4642 piece D / #4671). Recorded at `send_unsubscribe_upstream` // where the stored flag is still consulted; exported here so the // stored flag's real-world drift rate is legible in central telemetry // ahead of the reconcile-core keystone that deletes the flag. Read // from the per-node network_status singleton; `None` only before the // singleton is initialized (i.e. always populated in production). if let Some((comparisons, divergences)) = crate::node::network_status::upstream_divergence_counts() { snapshot.upstream_computed_vs_stored_comparisons = Some(comparisons); snapshot.upstream_computed_vs_stored_divergences = Some(divergences); } // Reconcile-controller SHADOW comparison counters, split PER SITE // (keystone step-2, #4642). Recorded at the on-`main` hosting decision // sites (collapse via `send_unsubscribe_upstream`, renewal via // `contracts_needing_renewal`) where the pure `reconcile` controller // is run in parallel and compared BY SET MEMBERSHIP to the current // behavior; exported here so the controller's real-world divergence // from today's scattered decisions is legible in central telemetry // BEFORE any decision is flipped to it. Split per site because the // flip is site-by-site and each must be separately gate-able. Same // hand-mirror footgun as the counters above — a new // `RouterSnapshotInfo` field is invisible to the collector unless // added here AND in `event_kind_to_json` (pinned by // `router_snapshot_json_includes_reconcile_shadow_counters`). Read // from the per-node network_status singleton; `None` only before it is // initialized. if let Some(shadow) = crate::node::network_status::reconcile_shadow_counts() { // Every assignment reads uniformly from `shadow..` // (NO aliasing), so the mirror-seam pin // `reconcile_shadow_export_maps_each_field_to_its_own_site` can // verify each snapshot field is fed from its OWN site — a // field-swap here would silently emit the wrong per-site value. // // MAINTENANCE sites (collapse, renewal): full per-action export // (all 9 counters). snapshot.reconcile_shadow_collapse_comparisons = Some(shadow.collapse.comparisons); snapshot.reconcile_shadow_collapse_divergences = Some(shadow.collapse.divergences); snapshot.reconcile_shadow_collapse_subscribe_diffs = Some(shadow.collapse.subscribe_diffs); snapshot.reconcile_shadow_collapse_renew_diffs = Some(shadow.collapse.renew_diffs); snapshot.reconcile_shadow_collapse_unsubscribe_diffs = Some(shadow.collapse.unsubscribe_diffs); snapshot.reconcile_shadow_collapse_collapse_diffs = Some(shadow.collapse.collapse_diffs); snapshot.reconcile_shadow_collapse_announce_diffs = Some(shadow.collapse.announce_diffs); snapshot.reconcile_shadow_collapse_retract_diffs = Some(shadow.collapse.retract_diffs); snapshot.reconcile_shadow_collapse_reroot_search_diffs = Some(shadow.collapse.reroot_search_diffs); snapshot.reconcile_shadow_renewal_comparisons = Some(shadow.renewal.comparisons); snapshot.reconcile_shadow_renewal_divergences = Some(shadow.renewal.divergences); snapshot.reconcile_shadow_renewal_subscribe_diffs = Some(shadow.renewal.subscribe_diffs); snapshot.reconcile_shadow_renewal_renew_diffs = Some(shadow.renewal.renew_diffs); snapshot.reconcile_shadow_renewal_unsubscribe_diffs = Some(shadow.renewal.unsubscribe_diffs); snapshot.reconcile_shadow_renewal_collapse_diffs = Some(shadow.renewal.collapse_diffs); snapshot.reconcile_shadow_renewal_announce_diffs = Some(shadow.renewal.announce_diffs); snapshot.reconcile_shadow_renewal_retract_diffs = Some(shadow.renewal.retract_diffs); snapshot.reconcile_shadow_renewal_reroot_search_diffs = Some(shadow.renewal.reroot_search_diffs); // EDGE sites: focused on one class each, so comparisons + // divergences fully capture the signal. snapshot.reconcile_shadow_inbound_unsubscribe_comparisons = Some(shadow.inbound_unsubscribe.comparisons); snapshot.reconcile_shadow_inbound_unsubscribe_divergences = Some(shadow.inbound_unsubscribe.divergences); snapshot.reconcile_shadow_connection_drop_comparisons = Some(shadow.connection_drop.comparisons); snapshot.reconcile_shadow_connection_drop_divergences = Some(shadow.connection_drop.divergences); snapshot.reconcile_shadow_host_formation_comparisons = Some(shadow.host_formation.comparisons); snapshot.reconcile_shadow_host_formation_divergences = Some(shadow.host_formation.divergences); } // Keep the hosting manager's copy of our own ring location current so // the proximity-prior demand estimate (#4642 A3) can turn a contract // key into a distance. Best-effort: only push a known location. if let Some(own_loc) = ring.connection_manager.own_location().location() { ring.hosting_manager.set_own_location(own_loc); } // Interest-weighted (two-tier) module-cache SHADOW gauges // (#4441/#4534): always-on, independent of the // FREENET_MODULE_CACHE_INTEREST_TIERED feature flag. They quantify // what the two-tier policy WOULD reclaim/reclassify; the // migration-admission counter now records actual admissions the // interested-occupancy gate (#4534) RECOVERS versus the old raw gate. snapshot.contract_module_cache_cold_evictable_bytes = Some(mc.contract_cold_evictable_bytes); snapshot.contract_module_cache_interested_bytes = Some(mc.contract_interested_bytes); snapshot.contract_module_cache_evictions_would_reclassify_total = Some(mc.contract_evictions_would_reclassify_total); snapshot.migration_admission_recovered_total = Some(mc.migration_admission_recovered_total); // UPDATE-broadcast stream-assembly failure gauge (#4440): the exact // signal that flagged the v0.2.73 incident. The broadcast queue // publishes monotonic totals into the process-global; emit the totals // (for collector-side differencing) plus the per-window failure delta // sampled here. let bs = crate::node::BROADCAST_STREAM_METRICS.snapshot(); let broadcast_stream_failures_delta = window_delta( bs.streaming_failures_total, &mut prev_broadcast_stream_failures_total, ); snapshot.broadcast_stream_attempts_total = Some(bs.streaming_attempts_total); snapshot.broadcast_stream_failures_total = Some(bs.streaming_failures_total); snapshot.broadcast_stream_failures_last_snapshot = Some(broadcast_stream_failures_delta); // Contract-exec WASM counters: the split that makes a summarize rate // interpretable. A high `fast_hits` rate with a low `wasm_calls` rate // is a warm cache doing cheap work; the two converging means the // change-detector has stopped covering the load and the storm class // (#4473 / #4610 / #5040 / #5238) has re-armed. // // EVERY arm is windowed, not a chosen headline subset. Emitting some // arms as a 5-minute delta and others as a lifetime total, under // parallel names on one log line, invites reading them as comparable // magnitudes — and the arm that would be understated that way is // `delta_wasm_uncached`, the per-local-subscriber fan-out delta that // has no cache in front of it at all and can dominate on a // client-facing node. An overstated saving is worse than a missing // one, because it terminates the investigation. let ce = ring.contract_exec_metrics.snapshot(); let ce_d = ce.window_deltas(&mut prev_exec); // Exhaustive destructure with no `..` rest pattern, on BOTH the // lifetime snapshot and the window deltas. A 9th counter arm then // fails to COMPILE here until it is exported, which is strictly // stronger than the source-scrape pin below: a scrape that hardcodes // today's eight names passes unchanged when a ninth is added, which // is exactly how a counter ends up recorded but never exported. let contract_exec_metrics::ContractExecSnapshot { summarize_fast_hits: _, summarize_reload_hits: _, summarize_wasm_calls: _, summarize_wasm_uncached: _, delta_fast_hits: _, delta_reload_hits: _, delta_wasm_calls: _, delta_wasm_uncached: _, } = ce; let contract_exec_metrics::ContractExecSnapshot { summarize_fast_hits: _, summarize_reload_hits: _, summarize_wasm_calls: _, summarize_wasm_uncached: _, delta_fast_hits: _, delta_reload_hits: _, delta_wasm_calls: _, delta_wasm_uncached: _, } = ce_d; snapshot.contract_exec_summarize_fast_hits_total = Some(ce.summarize_fast_hits); snapshot.contract_exec_summarize_reload_hits_total = Some(ce.summarize_reload_hits); snapshot.contract_exec_summarize_wasm_calls_total = Some(ce.summarize_wasm_calls); snapshot.contract_exec_summarize_wasm_uncached_total = Some(ce.summarize_wasm_uncached); snapshot.contract_exec_delta_fast_hits_total = Some(ce.delta_fast_hits); snapshot.contract_exec_delta_reload_hits_total = Some(ce.delta_reload_hits); snapshot.contract_exec_delta_wasm_calls_total = Some(ce.delta_wasm_calls); snapshot.contract_exec_delta_wasm_uncached_total = Some(ce.delta_wasm_uncached); snapshot.contract_exec_summarize_fast_hits_last_snapshot = Some(ce_d.summarize_fast_hits); snapshot.contract_exec_summarize_reload_hits_last_snapshot = Some(ce_d.summarize_reload_hits); snapshot.contract_exec_summarize_wasm_calls_last_snapshot = Some(ce_d.summarize_wasm_calls); snapshot.contract_exec_summarize_wasm_uncached_last_snapshot = Some(ce_d.summarize_wasm_uncached); snapshot.contract_exec_delta_fast_hits_last_snapshot = Some(ce_d.delta_fast_hits); snapshot.contract_exec_delta_reload_hits_last_snapshot = Some(ce_d.delta_reload_hits); snapshot.contract_exec_delta_wasm_calls_last_snapshot = Some(ce_d.delta_wasm_calls); snapshot.contract_exec_delta_wasm_uncached_last_snapshot = Some(ce_d.delta_wasm_uncached); // Summary/delta fast-path cache occupancy: WHY the arms above miss // when they do (byte budget binding vs. count target vs. churn). let fpc = ring.contract_exec_metrics.fast_path_cache_snapshot(); snapshot.contract_summary_cache_entries = Some(fpc.summary.entries); snapshot.contract_summary_cache_bytes = Some(fpc.summary.bytes); snapshot.contract_summary_cache_budget_bytes = Some(fpc.summary.budget_bytes); snapshot.contract_summary_cache_count_cap = Some(fpc.summary.count_cap); snapshot.contract_summary_cache_count_cap_evictions_total = Some(fpc.summary.count_cap_evictions_total); snapshot.contract_summary_cache_byte_budget_evictions_total = Some(fpc.summary.byte_budget_evictions_total); snapshot.contract_delta_cache_entries = Some(fpc.delta.entries); snapshot.contract_delta_cache_bytes = Some(fpc.delta.bytes); snapshot.contract_delta_cache_budget_bytes = Some(fpc.delta.budget_bytes); snapshot.contract_delta_cache_count_cap = Some(fpc.delta.count_cap); snapshot.contract_delta_cache_count_cap_evictions_total = Some(fpc.delta.count_cap_evictions_total); snapshot.contract_delta_cache_byte_budget_evictions_total = Some(fpc.delta.byte_budget_evictions_total); // Placement-quality gauge (#4404 follow-up): host-to-hosted-key // ring-distance distribution. If the SubscribeHint placement // migration is working, hosting drifts toward each contract's key, // so this distribution tightens over time. Cheap — a few thousand // distance computations at most, every 5 minutes. Skip gracefully if // the node has no ring location yet (emit nothing) or hosts nothing // (emit count=0, distances absent). No lock is held across an await: // `hosting_contract_keys()` returns an owned `Vec` and the distance // math is pure. let own_location = ring .connection_manager .own_location() .location() .map(|loc| loc.as_f64()); if let Some(node_loc) = own_location { let hosted_keys = ring.hosting_contract_keys(); let contract_locations: Vec = hosted_keys .iter() .map(|key| Location::from(key).as_f64()) .collect(); if let Some(stats) = placement_migration_metrics::placement_quality(node_loc, &contract_locations) { // `stats.count` equals `contract_locations.len()`; read it here // (rather than the Vec len) so the field stays live in // non-test builds. snapshot.hosted_contracts_count = Some(stats.count); snapshot.hosted_key_distance_median = Some(stats.median); snapshot.hosted_key_distance_p90 = Some(stats.p90); snapshot.hosted_key_distance_min = Some(stats.min); snapshot.hosted_key_distance_mean = Some(stats.mean); snapshot.hosted_key_distance_frac_within_0_1 = Some(stats.frac_within_0_1); } else { // Node hosts nothing: count is 0, distance gauges absent. snapshot.hosted_contracts_count = Some(0); } } // Placement-migration activity counters (#4404 follow-up): is the // migration firing, and at what rate? Monotonic lifetime totals read // from the per-node counter sink the send/receive sites publish into. let pm = ring.placement_migration_metrics.snapshot(); snapshot.subscribe_hint_sent = Some(pm.sent); snapshot.subscribe_hint_received = Some(pm.received); snapshot.subscribe_hint_acted = Some(pm.acted); snapshot.renewal_terminus_satisfied = Some(pm.renewal_terminus_satisfied); // Per-gate refusal + directed-subscribe outcome breakdown (#4534 // diagnostics). snapshot.subscribe_hint_refused_version = Some(pm.received_refused_version_floor); snapshot.subscribe_hint_refused_already_hosting = Some(pm.received_refused_already_hosting); snapshot.subscribe_hint_refused_holder = Some(pm.received_refused_holder_mismatch); snapshot.subscribe_hint_refused_cache = Some(pm.received_refused_cache_admission); snapshot.subscribe_hint_acted_succeeded = Some(pm.acted_succeeded); snapshot.subscribe_hint_acted_failed = Some(pm.acted_failed); // Large-state decision evidence (#5090). The fixed-cardinality // block rides the existing snapshot event once every six ticks // (30 minutes), with no new event stream, peer identifiers, or // contract-key map. Lifetime-monotonic counters recover both local // snapshot drops and the deliberately skipped five-minute ticks. network_efficiency_tick = (network_efficiency_tick + 1) % 6; if network_efficiency_tick == 0 && let (Some(op_manager), Some(disk), Some(telemetry)) = ( ring.upgrade_op_manager(), disk_usage, crate::tracing::telemetry::telemetry_local_metrics_snapshot(), ) { let lifecycle = op_manager.interest_manager.interest_lifecycle_snapshot(); // SHADOW MODE: futile-repair evidence (#nc-merge). Rides the // same fixed-cardinality block for the same reason the rest of // it does — `network_efficiency_v1` is the only family that // bypasses both the node rate limiter and the collector's 5% // sampler, so a rare-but-decisive counter is not sampled away. let futile_repair = op_manager.interest_manager.futile_repair_snapshot(); let receiver = op_manager.payload_mix.receiver_apply_stats(); let queue = crate::node::BROADCAST_QUEUE_EFFICIENCY_METRICS.snapshot(); let state_rejections = crate::contract::state_size_rejection_snapshot(); let rate_to_u64 = |rate: f64| { if !rate.is_finite() || rate <= 0.0 { 0 } else if rate >= u64::MAX as f64 { u64::MAX } else { rate.round() as u64 } }; let cost_axes = ring.hosting_cost_pressure_axes(); let eligibility = ring.hosting_manager.cost_eligibility_stats(&cost_axes); let cost = std::array::from_fn(|index| { cost_axes.get(index).map_or([0; 5], |axis| { let max_any = axis.rates.values().copied().fold(0.0, f64::max); let max_eligible = eligibility.max_eligible_rate[index] .iter() .copied() .fold(0.0, f64::max); [ rate_to_u64(axis.total_rate), rate_to_u64(axis.floor), rate_to_u64(max_any), rate_to_u64(max_eligible), axis.rates.len() as u64, ] }) }); let cost_bucket_any = std::array::from_fn(|axis| { std::array::from_fn(|bucket| { rate_to_u64(eligibility.max_attributed_rate[axis][bucket]) }) }); let cost_bucket_eligible = std::array::from_fn(|axis| { std::array::from_fn(|bucket| { rate_to_u64(eligibility.max_eligible_rate[axis][bucket]) }) }); let tel = [ telemetry.enqueue_attempts_total, telemetry.enqueue_succeeded_total, telemetry.enqueue_dropped_full_total, telemetry.enqueue_dropped_closed_total, telemetry.rate_limit_admitted_operational_total, telemetry.rate_limit_admitted_shadow_total, telemetry.dropped_aggregate_limit_operational_total, telemetry.dropped_aggregate_limit_shadow_total, telemetry.dropped_shadow_limit_total, telemetry.dropped_backoff_buffer_full_total, telemetry.batch_send_attempts_total, telemetry.batch_send_successes_total, telemetry.batch_send_failures_total, telemetry.batch_events_sent_total, telemetry.batch_retry_truncated_events_total, ]; let shadow = std::array::from_fn(|index| { let stream = telemetry.known_shadow_rollups[index]; [ stream.enqueue_attempts_total, stream.enqueue_dropped_full_total, stream.enqueue_dropped_closed_total, stream.rate_limit_admitted_total, stream.dropped_aggregate_limit_total, stream.dropped_shadow_limit_total, stream.dropped_backoff_buffer_full_total, stream.retry_truncated_total, stream.sent_total, ] }); snapshot.network_efficiency_v1 = Some(crate::router::NetworkEfficiencyV1 { v: 1, ms_s: lifecycle.delivered_sends, ms_b: lifecycle.delivered_bytes, ms_age: lifecycle.first_send_age, ms_size: lifecycle.delivered_size_hist, ms_unt_age: lifecycle.untracked_prior_removal_age, reg_ow_k: lifecycle.registration_overwrite_known, reg_ow_m: lifecycle.registration_overwrite_missing, reg_new_k: lifecycle.registration_new_known, reg_new_m: lifecycle.registration_new_missing, reg_cap: lifecycle.registration_cap_rejected, removed: lifecycle.removals, current: lifecycle.current_summary_state, recreated: lifecycle.recreated_after_removal, populated: lifecycle.population, corr_ovf: [lifecycle.history_overflow, lifecycle.active_overflow], queue: [ queue.capacity_evictions, queue.dedup_replacements, queue.enqueues_while_pair_active, queue.large_head_blocking_incidents, queue.large_head_blocked_millis, queue.small_entry_millis_blocked, queue.queued_large_actual_small, queue.queued_small_actual_large, queue.queued_large_actual_small_bytes, queue.queued_small_actual_large_bytes, queue.scheduled_small, queue.scheduled_large, queue.scheduled_small_state_bytes, queue.scheduled_large_state_bytes, queue.active_tracking_overflow, ], recv_n: receiver.counts, recv_tn: receiver.terminal_counts, recv_tb: receiver.terminal_bytes, state_n: disk.state_size_bucket_counts, state_b: disk.state_size_bucket_bytes, state: [ disk.state_count, disk.state_max_bytes, disk.state_over_limit_count, disk.state_over_limit_bytes, disk.state_limit_bytes, state_rejections.pre_wasm_count, state_rejections.pre_wasm_max_bytes, state_rejections.post_merge_count, state_rejections.post_merge_max_bytes, ], evict_n: eligibility.counts, evict_b: eligibility.bytes, vict_n: hosting.eviction_victim_counts, vict_b: hosting.eviction_victim_bytes, cost, cost_ba: cost_bucket_any, cost_be: cost_bucket_eligible, tel, shadow, eff: telemetry.network_efficiency_delivery, // Hosting observability (#4642): WHY this peer began hosting // each contract, and the two demand-signal gauges the // eviction ranking is built on. `hosting` is the same // `hosting_cache_stats()` read that feeds `vict_n`/`vict_b` // above, so all four describe one consistent hosted set. host_begin: hosting.hosting_begins, host_reads: hosting.read_count_hist, host_recency: hosting.genuine_access_recency, futile: futile_repair.to_row(), futile_ladder: futile_repair.ladder, }); } tracing::info!( failure_events = snapshot.failure_events, success_events = snapshot.success_events, prediction_active = snapshot.prediction_active, consider_n_closest_peers = snapshot.consider_n_closest_peers, open_fds = ?snapshot.open_fds, fd_soft_limit = ?snapshot.fd_soft_limit, contract_module_cache_entries = mc.contract_entries, contract_module_cache_total_bytes = mc.contract_total_bytes, contract_module_cache_evictions_total = mc.contract_evictions_total, broadcast_stream_attempts_total = bs.streaming_attempts_total, broadcast_stream_failures_total = bs.streaming_failures_total, broadcast_stream_failures_last_snapshot = broadcast_stream_failures_delta, // The per-window arms are logged (not the lifetime totals) so a // local operator reading `journalctl` gets the answer from ONE // line, and ALL EIGHT are logged in the same unit — mixing // 5-minute deltas with lifetime totals under parallel names // would invite reading them as comparable magnitudes. `info!` // survives `release_max_level_info`; `debug!` would not. contract_exec_summarize_fast_hits_last_snapshot = ce_d.summarize_fast_hits, contract_exec_summarize_reload_hits_last_snapshot = ce_d.summarize_reload_hits, contract_exec_summarize_wasm_calls_last_snapshot = ce_d.summarize_wasm_calls, contract_exec_summarize_wasm_uncached_last_snapshot = ce_d.summarize_wasm_uncached, contract_exec_delta_fast_hits_last_snapshot = ce_d.delta_fast_hits, contract_exec_delta_reload_hits_last_snapshot = ce_d.delta_reload_hits, contract_exec_delta_wasm_calls_last_snapshot = ce_d.delta_wasm_calls, contract_exec_delta_wasm_uncached_last_snapshot = ce_d.delta_wasm_uncached, contract_summary_cache_entries = fpc.summary.entries, contract_summary_cache_bytes = fpc.summary.bytes, contract_summary_cache_budget_bytes = fpc.summary.budget_bytes, contract_summary_cache_count_cap = fpc.summary.count_cap, contract_summary_cache_count_cap_evictions_total = fpc.summary.count_cap_evictions_total, contract_summary_cache_byte_budget_evictions_total = fpc.summary.byte_budget_evictions_total, contract_delta_cache_entries = fpc.delta.entries, contract_delta_cache_bytes = fpc.delta.bytes, contract_delta_cache_budget_bytes = fpc.delta.budget_bytes, contract_delta_cache_count_cap = fpc.delta.count_cap, contract_delta_cache_count_cap_evictions_total = fpc.delta.count_cap_evictions_total, contract_delta_cache_byte_budget_evictions_total = fpc.delta.byte_budget_evictions_total, hosted_contracts_count = ?snapshot.hosted_contracts_count, hosted_key_distance_median = ?snapshot.hosted_key_distance_median, hosted_key_distance_p90 = ?snapshot.hosted_key_distance_p90, hosted_key_distance_frac_within_0_1 = ?snapshot.hosted_key_distance_frac_within_0_1, subscribe_hint_sent = pm.sent, subscribe_hint_received = pm.received, subscribe_hint_acted = pm.acted, renewal_terminus_satisfied = pm.renewal_terminus_satisfied, terminal_consult_attempts = ?snapshot.terminal_consult_attempts, terminal_consult_hits = ?snapshot.terminal_consult_hits, terminal_consult_resolved_found = ?snapshot.terminal_consult_resolved_found, terminal_consult_still_not_found = ?snapshot.terminal_consult_still_not_found, upstream_computed_vs_stored_comparisons = ?snapshot.upstream_computed_vs_stored_comparisons, upstream_computed_vs_stored_divergences = ?snapshot.upstream_computed_vs_stored_divergences, "router_snapshot" ); if let Some(event) = NetEventLog::router_snapshot(&ring, snapshot) { ring.event_register .register_events(Either::Left(event)) .await; } } } /// Periodically emit subscription_state telemetry events for all active subscriptions. /// /// This enables the telemetry dashboard to reconstruct historical subscription trees /// and show accurate subscription state at any point in time. async fn emit_subscription_state_telemetry(ring: Arc, interval_duration: Duration) { let shutdown = ring.shutdown_token(); let mut interval = tokio::time::interval(interval_duration); // Skip the first immediate tick interval.tick().await; loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } // Get subscription states from the new lease-based model let subscription_states = ring.get_subscription_states(); if subscription_states.is_empty() { continue; } tracing::debug!( subscription_count = subscription_states.len(), "Emitting periodic subscription state telemetry" ); // Log subscription states (simplified - no upstream/downstream in new model) for (key, has_client, is_active, _expires_at) in subscription_states { tracing::trace!( %key, has_client_subscription = has_client, is_active_subscription = is_active, "Subscription state" ); } } } /// Maximum renewal tasks spawned per tick. /// /// At 30s intervals, this yields up to 20 renewals/minute. Since /// `contracts_needing_renewal()` returns contracts expiring within the /// 2-minute renewal window, this effectively handles ~40 concurrent /// subscriptions before renewals can't keep up. The mid-cycle channel /// capacity check (RENEWAL_STOP_CAPACITY_FRACTION) provides backpressure /// if the network can't absorb this many. const MAX_RECOVERY_ATTEMPTS_PER_INTERVAL: usize = 10; /// Maximum phantom repair fetches (#4612) spawned per recovery tick. /// /// Bounds the producer (the #4440/#4610 lesson: never an unbounded /// producer feeding the shared serial paths): at 30s intervals this is at /// most 8 sub-op GETs/minute network-wide fetch pressure from phantom /// repair, regardless of how many phantoms the relay path accumulated. /// Per-contract pacing/giving-up lives in /// `HostingManager::reconcile_phantom_in_use` (cooldown + max attempts). const MAX_PHANTOM_REPAIRS_PER_INTERVAL: usize = 4; /// Skip renewal cycle when channel remaining capacity falls below this /// fraction of max (i.e. channel is more than 50% full). const RENEWAL_DEFER_CAPACITY_FRACTION: usize = 2; // channel_max / 2 /// Stop spawning mid-cycle when remaining capacity falls below this /// fraction of max (i.e. channel is more than 75% full). const RENEWAL_STOP_CAPACITY_FRACTION: usize = 4; // channel_max / 4 /// Interval for periodic subscription state telemetry snapshots. pub(crate) const SUBSCRIPTION_STATE_INTERVAL: Duration = Duration::from_secs(60); /// Interval for periodic subscription recovery attempts. /// /// This recovers "orphaned hosters" - peers that have contracts in cache /// but failed to establish subscription (no upstream in subscription tree). pub(crate) const SUBSCRIPTION_RECOVERY_INTERVAL: Duration = Duration::from_secs(30); /// Total wall-clock budget the renewal driver gives **itself** across all /// of a renewal's network attempts. /// /// Issue #4350: `recover_orphaned_subscriptions` wraps each renewal task in /// an outer cancel deadline ([`Self::renewal_outer_cancel`]), but the /// renewal driver's per-attempt network wait used the global /// [`OPERATION_TTL`] (60 s) while the old outer deadline was only 25 s. A /// peer answering between 25 s and 60 s was killed by the outer cancel /// *mid-await* — the in-flight reply then landed on the renewal task's /// already-dropped capacity-1 receiver (the dominant source of /// `try_forward_driver_reply` closed-receiver drops; log-level fixed in /// PR #4351). The driver now clamps **each** attempt's timeout to the budget /// remaining until this deadline, so no attempt — first or retry — can still /// be awaiting when the outer cancel fires. /// /// Derived from the config interval (most of one recovery cycle) rather than /// independently hardcoded. 20 s. pub(crate) const RENEWAL_TASK_BUDGET: Duration = Duration::from_secs(Self::SUBSCRIPTION_RECOVERY_INTERVAL.as_secs() - 10); /// Maximum single-attempt network wait on the renewal path. Each attempt is /// actually capped at `min(this, remaining task budget)` (see /// [`Self::RENEWAL_TASK_BUDGET`]); this is the ceiling for the first /// attempt when the full budget is available. Set equal to the total budget /// so a single slow peer may consume the whole budget in one attempt; /// multi-peer renewals naturally get shorter per-attempt waits as the /// remaining budget shrinks. pub(crate) const RENEWAL_PER_ATTEMPT_TIMEOUT: Duration = Self::RENEWAL_TASK_BUDGET; /// Slack between a renewal task's worst-case self-termination time /// (`RENEWAL_TASK_BUDGET` + a fully backpressured `release_pending_op_slot` /// cleanup) and the hard outer cancel deadline. Gives the driver room to /// finish its own cleanup and telemetry before the outer cancel could fire. /// 5 s. pub(crate) const RENEWAL_OUTER_DEADLINE_MARGIN: Duration = Duration::from_secs(5); /// Hard outer cancel deadline wrapping each spawned renewal task in /// `recover_orphaned_subscriptions`. Pure backstop for a *genuinely wedged* /// driver: in normal operation the driver self-terminates within /// [`Self::RENEWAL_TASK_BUDGET`] (slow peer) and then runs cleanup, so this /// deadline never fires. /// /// Issue #4350 (Codex review): it must clear the driver's worst case — /// `RENEWAL_TASK_BUDGET` (a full slow attempt) **plus** a fully /// backpressured `release_pending_op_slot`, which awaits up to /// [`OpManager::NOTIFICATION_SEND_TIMEOUT`] (30 s) — or the outer cancel /// could interrupt cleanup, leaving the per-attempt waiter in /// `pending_op_results` (the very closed-receiver drop this change removes). /// Deadline = `BUDGET (20 s) + NOTIFICATION_SEND_TIMEOUT (30 s) + /// RENEWAL_OUTER_DEADLINE_MARGIN (5 s) = 55 s`. /// /// Outliving the 30 s recovery interval is safe: `mark_subscription_pending` /// (released via `SubscriptionRecoveryGuard` on completion **or** cancel) /// blocks a second concurrent renewal for the same contract, and the /// per-cycle spawn count is bounded by `MAX_RECOVERY_ATTEMPTS_PER_INTERVAL` /// and the channel-capacity gates — so a long-running task can't accumulate. pub(crate) fn renewal_outer_cancel() -> Duration { Self::RENEWAL_TASK_BUDGET + crate::node::OpManager::NOTIFICATION_SEND_TIMEOUT + Self::RENEWAL_OUTER_DEADLINE_MARGIN } /// Floor below which the renewal driver will **not** start another attempt: /// a remaining budget this small can't complete a useful network round-trip /// before the task budget (and then the outer cancel) elapses, so the /// driver stops cleanly and lets the next 30 s recovery cycle retry rather /// than starting an attempt doomed to be cut short. 2 s. pub(crate) const RENEWAL_MIN_ATTEMPT_BUDGET: Duration = Duration::from_secs(2); /// Interval for periodic GET subscription cache sweep. pub(crate) const GET_SUBSCRIPTION_SWEEP_INTERVAL: Duration = Duration::from_secs(60); /// Periodically attempt to recover "orphaned hosters" - contracts we're hosting /// but don't have an upstream subscription for. /// /// This can happen when: /// - The initial subscription after GET/PUT failed (network issues, timeout) /// - Our upstream peer disconnected and we haven't found a new one /// - A race condition left us hosting without subscription /// /// The task respects existing backoff mechanisms to avoid subscription spam. /// /// **Connection gating (#3676):** This task skips renewal cycles entirely when /// the node has zero ring connections. Without this gate, disconnected peers /// generate a subscribe retry storm — thousands of subscribe requests per cycle /// that all fail immediately because there's no one to send them to. Telemetry /// showed 3 peers with 0 connections generating 96% of all subscribe traffic. async fn recover_orphaned_subscriptions(ring: Arc, interval_duration: Duration) { let shutdown = ring.shutdown_token(); // Wait indefinitely for the first ring connection before starting // subscription recovery. The per-cycle connection check below is the // real gate; this just avoids running the loop body with no peers. // // The 500ms poll is itself <1s, but the *loop* can park here forever on // a node that never connects — so race shutdown here too, otherwise a // never-connected node would hang teardown indefinitely (#4278). let mut wait_logged = false; loop { if sleep_or_shutdown(&shutdown, tokio::time::sleep(Duration::from_millis(500))).await { return; } if ring.open_connections() > 0 { tracing::info!( hosted_contracts = ring.hosting_contract_keys().len(), "Ring connection established, starting subscription recovery" ); break; } // Log periodically so operators can diagnose stuck nodes. if !wait_logged { wait_logged = true; tracing::info!( hosted_contracts = ring.hosting_contract_keys().len(), "Waiting for ring connection before starting subscription recovery" ); } } // Small jitter (2-5s) after first connection to let the ring stabilize // slightly before flooding with subscribe requests. let jitter = Duration::from_secs(GlobalRng::random_range(2u64..=5u64)); if sleep_or_shutdown(&shutdown, tokio::time::sleep(jitter)).await { return; } let mut interval = tokio::time::interval(interval_duration); // Skip the first immediate tick — we run the first pass immediately // below (no tick wait) so client subscriptions get prompt renewal. interval.tick().await; let mut first_pass = true; loop { if first_pass { first_pass = false; } else if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } // Always run expiry sweeps, even when disconnected. Stale // subscriptions and downstream subscribers must be cleaned up // to keep interest manager counts accurate. Only the renewal // spawning (below) is gated on having connections. // // First, expire any stale subscriptions let expired = ring.expire_stale_subscriptions(); if !expired.is_empty() { tracing::debug!( expired_count = expired.len(), "Expired {} stale subscriptions", expired.len() ); } // Expire stale downstream subscribers and decrement interest manager let ds_expired = ring.expire_stale_downstream_subscribers(); if !ds_expired.is_empty() { tracing::debug!( expired_count = ds_expired.len(), "Expired stale downstream subscribers" ); if let Some(op_manager) = ring.upgrade_op_manager() { for (contract, expired_count) in &ds_expired { // Decrement interest manager for each expired peer for _ in 0..*expired_count { op_manager .interest_manager .remove_downstream_subscriber(contract); } // Send Unsubscribe upstream if no remaining interest. // FLIP (keystone P6, #4642): the collapse decision is driven // by the reconcile controller's strict-farther interest gate // (`reconcile_wants_collapse` = `!contract_in_use`), replacing // the legacy ANY-downstream `should_unsubscribe_upstream`. The // teardown still targets the STORED upstream (narrow flip). if op_manager.reconcile_wants_collapse( contract, crate::node::network_status::ReconcileShadowSite::Collapse, ) { let op_mgr = op_manager.clone(); let contract = *contract; GlobalExecutor::spawn(async move { op_mgr.send_unsubscribe_upstream(&contract).await; }); } } } } // Gate: skip renewal spawning if we have no ring connections (#3676). // Subscribe requests require connected peers to route through. // Without this, disconnected peers flood the notification channel // with doomed subscribe requests every 30 seconds. if ring.open_connections() == 0 { tracing::debug!("Skipping subscription renewal: no ring connections"); continue; } // Phantom in-use reconciliation (#4612): enforce `in-use ⇒ // has-state`. The transit-relay SUBSCRIBE path registers // downstream subscribers without ever fetching the contract's // state, leaving this node advertised as a "host" it cannot be // (GET dead-ends #4404; summarize storm #4610 before the #4611 // gate). Repair each phantom with a bounded one-shot sub-op GET; // after MAX_PHANTOM_REPAIR_ATTEMPTS failures drop the phantom // registration instead of keeping it forever. Runs after the // connection gate: the repair fetch needs peers to route through, // and a disconnected node must not burn its bounded attempts on // fetches that are doomed locally. let repairs = ring.reconcile_phantom_in_use(Self::MAX_PHANTOM_REPAIRS_PER_INTERVAL); if !repairs.is_empty() && let Some(op_manager) = ring.upgrade_op_manager() { for repair in repairs { // The #4770 CYCLING Drop stays neutralized (step 10 §1c); // `Drop` here is the review-Fix-C ABSOLUTE-AGE-bounded drop — // it fires only after PHANTOM_ABSOLUTE_MAX_AGE and is // non-cycling (a re-registration goes through // finalize_host_subscribe / fetch-first). match repair { crate::ring::hosting::PhantomRepair::Fetch(key) => { tracing::info!( contract = %key, "phantom in-use contract (no stored state): \ starting one-shot repair fetch" ); // Fire-and-forget: the side effect (contract + // state cached locally) is what restores the // invariant; the next 30s pass observes the store. let _ = crate::operations::get::op_ctx_task::start_sub_op_get( &op_manager, *key.id(), true, ); } crate::ring::hosting::PhantomRepair::Drop(key) => { let removed = ring.drop_phantom_downstream(&key); tracing::warn!( contract = %key, removed_subscribers = removed, "phantom in-use contract unrepairable past absolute-age bound: \ dropping stale downstream registration (review Fix C — the \ downstream kept renewing but state stayed unfetchable)" ); // Decrement interest symmetrically, then collapse // upstream if no interest remains. for _ in 0..removed { op_manager .interest_manager .remove_downstream_subscriber(&key); } if op_manager.reconcile_wants_collapse( &key, crate::node::network_status::ReconcileShadowSite::Collapse, ) { let op_mgr = op_manager.clone(); GlobalExecutor::spawn(async move { op_mgr.send_unsubscribe_upstream(&key).await; }); } } } } } // Get contracts that need subscription renewal (have client subscriptions) let mut contracts_needing_renewal = ring.contracts_needing_renewal(); if contracts_needing_renewal.is_empty() { tracing::debug!( hosted = ring.hosting_contract_keys().len(), "No contracts needing subscription renewal" ); continue; } tracing::info!( needing_renewal = contracts_needing_renewal.len(), hosted = ring.hosting_contract_keys().len(), "Starting subscription renewal cycle" ); // Shuffle to prevent starvation: without this, the same failing contracts // (first N in iteration order) would always be tried first, blocking later // contracts from ever being attempted when they hit the batch limit. GlobalRng::shuffle(&mut contracts_needing_renewal); // Get op_manager to spawn subscription requests let Some(op_manager) = ring.upgrade_op_manager() else { tracing::debug!("OpManager not available for subscription renewal"); continue; }; // Backpressure: reduce batch size when the notification channel is // congested, but never skip entirely. Renewals are critical-path — // skipping a full cycle when the channel is busy lets subscriptions // expire, which causes cascading failures as the subscription tree // thins out and remaining renewals take longer paths. let sender = op_manager.to_event_listener.notifications_sender(); let channel_remaining = sender.capacity(); let channel_max = sender.max_capacity(); let batch_limit = if channel_remaining < channel_max / Self::RENEWAL_DEFER_CAPACITY_FRACTION { // Channel >50% full: allow a reduced batch (quarter of normal) // so critical renewals still get through. Always attempt at least 1. let reduced = (Self::MAX_RECOVERY_ATTEMPTS_PER_INTERVAL / 4).max(1); tracing::warn!( channel_remaining, channel_max, batch_limit = reduced, contracts = contracts_needing_renewal.len(), "Notification channel >50% full, reducing renewal batch size" ); reduced } else { Self::MAX_RECOVERY_ATTEMPTS_PER_INTERVAL }; let mut attempted = 0; let mut skipped = 0; // Per-tick budget on the number of interest-gate EVALUATIONS // (`OpManager::reconcile_wants_renewal`) we run. Each evaluation builds // a FRESH `ReconcileInputs` snapshot: a synchronous redb // `get_state_size` read plus an `is_subscription_root` neighbor-map // scan. The `attempted >= batch_limit` break below counts only SPAWNED // renewals; a gate-SUPPRESSED candidate does `continue` WITHOUT // advancing `attempted`, so on a high-hosting peer (every peer, now // that every-hop placement is live) a single ~30s tick could otherwise // build hundreds-to-thousands of these snapshots when the gate // suppresses most candidates. Counting every gate evaluation against // this budget restores the pre-flip `shadow_budget` bound: at most // `batch_limit` snapshots per tick regardless of how many are // suppressed. (Codex P2, #4725.) let mut evaluated = 0; // Reconcile-controller FLIP (keystone sub-task 3, #4642), RENEWAL // site. The controller no longer records a shadow divergence here; it // DRIVES: for each candidate the loop calls // `OpManager::reconcile_wants_renewal`, which builds a FRESH // `ReconcileInputs` snapshot AT EMISSION time and renews only while the // controller wants the contract's place in the mesh maintained. Each // snapshot is a redb state-store read plus an `is_subscription_root` // neighbor scan, so the per-tick cost is bounded by the `evaluated` // budget below (at most `batch_limit` gate evaluations per tick), // mirroring the old shadow sample budget. The `attempted >= // batch_limit` break bounds only the SPAWNED renewals and must NOT be // relied on to bound the evaluations, because a gate-suppressed // candidate does not advance `attempted`. for contract in contracts_needing_renewal { // Limit concurrent renewal attempts to avoid overwhelming the network if attempted >= batch_limit { tracing::debug!( limit = batch_limit, "Reached max renewal attempts for this interval, remaining will be tried next cycle" ); break; } // Stop early if the channel is filling from our own spawns. let remaining_now = sender.capacity(); if remaining_now < channel_max / Self::RENEWAL_STOP_CAPACITY_FRACTION { tracing::warn!( channel_remaining = remaining_now, attempted, "Notification channel >75% full during renewal spawning, stopping early" ); break; } // Phase 7 egress gate (#4373). Don't renew a subscription // for a contract we have banned: a renewal re-registers // interest via the same outbound-SUBSCRIBE machinery as a // client-initiated request, but unlike the four // `start_client_*` originator entry points this scheduler // doesn't pass through `reject_if_contract_banned`. Without // this check the node keeps emitting outbound SUBSCRIBE // renewals for a banned-but-still-subscribed contract on // every maintenance cycle until the ban TTL lifts. Checked // before `can_request_subscription` so a banned contract is // skipped regardless of its spam-backoff state. if ring.contract_ban_list.is_banned(contract.id()) { tracing::debug!( %contract, phase = "subscription_renewal_banned_skip", "skipping subscription renewal for banned contract" ); skipped += 1; continue; } // Check spam prevention (respects exponential backoff and pending checks) if !ring.can_request_subscription(&contract) { skipped += 1; continue; } // Per-tick evaluation budget (Codex P2, #4725): stop building // expensive `ReconcileInputs` snapshots once we've run // `batch_limit` interest-gate evaluations this tick, whether they // led to a spawn or a suppression. This is the bound that keeps a // high-hosting peer from building hundreds-to-thousands of // snapshots per tick when the gate suppresses most candidates (the // `attempted >= batch_limit` break above does NOT cover this, // because a suppressed candidate never advances `attempted`). // Remaining candidates are retried next cycle. if evaluated >= batch_limit { tracing::debug!( limit = batch_limit, attempted, skipped, "Reached max renewal-gate evaluations for this interval, \ remaining will be tried next cycle" ); break; } evaluated += 1; // FLIP (#4642 keystone): interest-gated renewal DRIVEN by the // reconcile controller's interest gate (design §5a // `contract_in_use`). Build a FRESH snapshot at emission time and // renew iff a local client, a STRICTLY-farther downstream // subscriber (piece D), OR recent local GET/PUT access (invariant // 3: reads/PUTs are permanent demand — a read-only / PUT-only // contract stays renewed without a subscription) depends on this // peer hosting. When it goes false — no longer in use — SKIP: the lease // lapses and the chain collapses inward (non-renewal is the collapse // primitive; the #3763 storm fix). The gate narrows the // ANY-downstream candidate set that `contracts_needing_renewal` // built to the strict gate; the suppression is recorded on the // RENEWAL reconcile counter (post-flip that counter measures the // flip's effect — the "renewal tracks active demand, not cache size" // ship-gate falsifier — not a shadow divergence). if !op_manager.reconcile_wants_renewal(&contract) { tracing::debug!( %contract, "renewal skipped: reconcile interest gate says not in use \ (strict-farther gate) — letting the lease lapse so the \ chain collapses inward" ); crate::node::network_status::record_reconcile_shadow_comparison( crate::node::network_status::ReconcileShadowSite::Renewal, crate::ring::reconcile::ReconcileActionDivergence { renew: true, ..Default::default() }, ); skipped += 1; continue; } // Renewal proceeds: record a non-divergent renewal-gate evaluation // so the counter's denominator (`comparisons`) tracks total gate // decisions and the suppression ratio stays legible. crate::node::network_status::record_reconcile_shadow_comparison( crate::node::network_status::ReconcileShadowSite::Renewal, crate::ring::reconcile::ReconcileActionDivergence::default(), ); // Mark as pending and spawn subscription request if ring.mark_subscription_pending(contract) { attempted += 1; // Re-root and renewal share ONE spawn path. Per design §5b // "re-rooting is just the renewal stream following the computed // upstream", the connection-drop PROMPT re-root (#4642 piece F, // `OpManager::spawn_prompt_reroots`) calls this SAME helper, so // the storm-safety scaffolding (jitter, recovery guard, // outer-cancel deadline, per-contract pending dedup) has a single // source of truth and cannot drift between the two callers. Self::spawn_renewal_subscribe_task( op_manager.clone(), contract, shutdown.clone(), ); } // `mark_subscription_pending` returning false (a renewal for this // contract is already in flight) needs no separate handling: the // controller drive + counter above already ran for this contract // this tick, and the in-flight guard is the per-contract dedup. } if attempted > 0 || skipped > 0 { tracing::info!( attempted, skipped_rate_limited = skipped, "Subscription renewal cycle complete" ); } // Simulation-test observability (no-op in production — gated on a // current network name, like the wire-attempt/terminus counters): // record this cycle's spawn count so a migration-at-scale test can // assert the per-node renewal rate cap holds (`attempted` is bounded // by `MAX_RECOVERY_ATTEMPTS_PER_INTERVAL` above). Only when this // cycle actually spawned renewals — a zero batch never raises the // per-node max, so skip it to avoid creating empty registry entries. // See #4601. if attempted > 0 { if let Some(addr) = ring.connection_manager.get_own_addr() { crate::ring::topology_registry::record_renewal_cycle_batch( addr, attempted as u64, ); } } } } /// Spawn ONE renewal / re-root subscribe task for `contract_key` and return /// immediately. The single storm-safe spawn path shared by the periodic /// renewal loop ([`recover_orphaned_subscriptions`]) and the event-driven /// connection-drop PROMPT re-root ([`OpManager::spawn_prompt_reroots`], #4642 /// piece F). /// /// # Why one helper, not two /// /// Per the demand-driven-hosting design §5b, "re-rooting is not a separate /// operation — it is just the renewal stream following the computed upstream." /// A prompt re-root and a scheduled renewal are the SAME wire action /// (`run_renewal_subscribe`, which routes a single SUBSCRIBE toward the key to /// the current computed upstream); the only difference is the trigger (a drop /// event vs. a lease-age tick). Keeping ONE spawn path means the storm-safety /// scaffolding lives in one place and cannot drift between callers (the /// "manually-mirrored side effect" bug class, `bug-prevention-patterns.md`). /// /// # Make-before-break /// /// This issues a fresh SUBSCRIBE; it does NOT tear down any existing lease /// first. The old upstream registration lapses on its own once this peer stops /// renewing toward it (design §5b / §6), and the peer keeps serving its local /// copy throughout (invariant 1 serve-DURING). So a re-root never drops the /// subscription before the new root is acquired. /// /// # Per-contract rate cap (caller-owned) /// /// The caller MUST have already claimed the per-contract slot via /// `mark_subscription_pending` (which blocks a second concurrent /// renewal/re-root for the same contract until this task's /// [`SubscriptionRecoveryGuard`] releases it) — that mark, plus /// `can_request_subscription`'s exponential backoff, is the per-contract wire /// rate cap. This helper only owns the jitter spread + outer-cancel deadline. pub(crate) fn spawn_renewal_subscribe_task( op_manager: Arc, contract_key: ContractKey, shutdown: CancellationToken, ) { // Spread tasks across the interval to avoid thundering-herd bursts. On a // mass disconnect this jitter is what turns O(contracts) simultaneous // re-subscribes into a spread-out trickle — the storm-safety knob for the // piece-F prompt re-root caller as much as for the renewal loop. let jitter_ms = GlobalRng::random_range(0u64..=15_000); GlobalExecutor::spawn(async move { // Guard ensures complete_subscription_request is called even on panic. // Created BEFORE the jitter sleep so an early shutdown return below // still clears the `mark_subscription_pending` flag set above — the // guard's Drop marks the request failed (#4278). let guard = SubscriptionRecoveryGuard::new(op_manager.clone(), contract_key); // The jitter can be up to 15s; bail (dropping the guard, which // completes the pending request as failed) before doing any // renewal work if the node is shutting down (#4278). if sleep_or_shutdown( &shutdown, tokio::time::sleep(Duration::from_millis(jitter_ms)), ) .await { return; } let instance_id = *contract_key.id(); // Renewal driver: same machinery as // client-initiated SUBSCRIBE, with delivery // returned to this task instead of via // `result_router_tx`. `is_renewal=true` so // the responder skips sending state. // // Outer renewal cancel deadline: a pure backstop for a // *genuinely wedged* driver. In normal operation the // renewal driver self-terminates within // `RENEWAL_TASK_BUDGET` (slow peer) and then runs its // cleanup, so this deadline never fires. // // Issue #4350: the driver clamps each attempt's wait to // the budget remaining until its own task deadline // (`RENEWAL_TASK_BUDGET`, 20 s), so no attempt — first // or retry — is still awaiting when this outer cancel // fires; and the deadline is sized // (`renewal_outer_cancel`, 55 s) to clear the driver's // worst case (full-budget attempt + a fully // backpressured `release_pending_op_slot` cleanup, up to // `NOTIFICATION_SEND_TIMEOUT`). A task outliving the 30 s // recovery interval is safe: `mark_subscription_pending` // (released by `SubscriptionRecoveryGuard` on completion // or cancel) blocks a second concurrent renewal for the // same contract. let renewal_tx = crate::message::Transaction::new::(); let renewal_deadline = Self::renewal_outer_cancel(); let outcome_enum = match tokio::time::timeout( renewal_deadline, crate::operations::subscribe::run_renewal_subscribe( op_manager.clone(), instance_id, renewal_tx, ), ) .await { Ok(outcome) => outcome, Err(_) => crate::operations::subscribe::RenewalOutcome::Failed { reason: format!( "renewal task exceeded {}s cycle deadline", renewal_deadline.as_secs() ), }, }; let (outcome, error_msg) = match outcome_enum { crate::operations::subscribe::RenewalOutcome::Success => { tracing::info!( %contract_key, "Subscription renewal succeeded" ); guard.complete(true); ("success", None) } crate::operations::subscribe::RenewalOutcome::ChannelCongestion => { // Channel congestion is a local resource issue, not a // protocol failure. Don't penalize with backoff — just // clear the pending mark so the contract is eligible on // the next cycle. tracing::warn!( %contract_key, "Subscription renewal skipped (channel full), will retry next cycle" ); guard.complete(true); ("dropped_channel_full", None) } crate::operations::subscribe::RenewalOutcome::Failed { reason } => { tracing::debug!( %contract_key, error = %reason, "Subscription renewal failed (will retry with backoff)" ); guard.complete(false); ("failed", Some(reason)) } }; crate::tracing::telemetry::send_standalone_event( "subscription_renewal_outcome", serde_json::json!({ "contract": contract_key.to_string(), "outcome": outcome, "error": error_msg, }), ); }); } /// Background task to sweep expired entries from the GET subscription cache. /// /// When contracts are evicted (past max entries and beyond TTL), this task /// cleans up the local subscription state. The upstream peer will eventually /// prune us when updates fail to deliver. async fn sweep_get_subscription_cache(ring: Arc, interval_duration: Duration) { let shutdown = ring.shutdown_token(); // Add random initial delay to prevent synchronized sweeps across peers let initial_delay = Duration::from_secs(GlobalRng::random_range(10u64..=30u64)); if sleep_or_shutdown(&shutdown, tokio::time::sleep(initial_delay)).await { return; } let mut interval = tokio::time::interval(interval_duration); interval.tick().await; // Skip first immediate tick loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } // Resident-overhead budget (#5325, #5647): the share of the node's // memory limit hosted contracts may hold in RAM. Recomputed every // tick so a cgroup limit changed at runtime is picked up, and // BEFORE the sweep below, so the neighbour-summary budget it // installs and the eviction it runs both use the current value. // Falls back to 1 GiB in the rare case the RAM read itself fails. let total_ram = crate::ring::hosting::total_ram_or_fallback( crate::wasm_runtime::read_total_ram_bytes(), ); ring.hosting_manager .recompute_resident_overhead_budget(total_ram); // Sweep expired entries from GET subscription cache let crate::ring::hosting::HostingSweepResult { expired, evicted_in_use_teardown, } = ring.sweep_expired_get_subscriptions(); // Do NOT early-continue when `expired` is empty: pending // reclamation retries below must run every cycle, otherwise // entries queued by the in-use / queue-full skip points stay // leaked indefinitely whenever the cache is under budget // (the common case). See Codex r10 P2. if !expired.is_empty() { tracing::debug!( expired_count = expired.len(), "GET subscription cache sweep found expired entries" ); } // Reclaim on-disk storage for the expired contracts so the hosting // budget is a real disk bound. The sweep task only holds an // `Arc`; reach the `OpManager` the same way // `recover_orphaned_subscriptions` does, via the weak back-reference. // `reclaim_evicted_contract` re-checks the subscription gate per // key before emitting the eviction event. let op_manager = ring.upgrade_op_manager(); if op_manager.is_none() { // The weak back-reference is dropped only during node shutdown; // surface the skipped reclamation so an unexpected `None` (and // the resulting on-disk leak for this cycle) is observable. tracing::debug!( expired_count = expired.len(), "OpManager unavailable during GET subscription sweep — \ on-disk reclamation skipped for the expired contracts this cycle" ); } // Sync the `InterestManager` for any still-in-use victim the sweep // shed as a last resort: `teardown_evicted_in_use_contract` cleared // the hosting maps but the `InterestManager` lives on `OpManager`, so // ghost `interested_peers` / `peer_contracts` / `local_client_count` // entries survive unless we replay the same removals here (PR #4734 // Fix 1). No-op in the common zero-subscriber sweep. Run BEFORE the // `unregister_local_hosting` loop below so that call observes the // now-zeroed subscriber counts and reports full interest loss (→ // retraction), mirroring the GET/PUT host-formation paths. if let Some(op_manager) = &op_manager { for teardown in &evicted_in_use_teardown { op_manager.interest_manager.remove_evicted_in_use( &teardown.key, &teardown.downstream_peers, teardown.local_client_count, ); } } // Clean up local subscription state for each expired contract. // Note: under the subscriber-primary ordering (#4642, invariant 3) a // contract with subscribers is ordered LAST but NOT hard-pinned, so // `sweep_expired_hosting()` CAN shed a still-in-use contract as a last // resort — and when it does, it has already torn down that contract's // subscription state (downstream + client subscriptions + upstream // lease) so `contract_in_use` is false before we reclaim it. The // `ring.unsubscribe(&key)` below is therefore a belt-and-suspenders // no-op for those (its lease is already gone) and, for a // zero-subscriber eviction, drops any lingering upstream lease. // The `expected_generation` snapshot is captured atomically with // the eviction decision in `HostingCache::record_access` / // `sweep_expired`; it is re-checked at deletion time by // `RuntimePool::remove_contract` to close the re-host race. // // Also retract the local hosting advertisement for each evicted // contract, mirroring the GET (`get/op_ctx_task.rs`) and PUT // (`put/op_ctx_task.rs`) host-formation paths (PR #4734 Fix 1): // clear `LocalInterest.hosting` via `unregister_local_hosting` and // collect the contracts that thereby lose ALL interest so the // neighbor advertisement can be retracted below. Without this an // evicted contract keeps `hosting = true` and its advertisement // lingers on neighbors — a stale-interest leak, now widened because // subscribed contracts are sweep-evictable under invariant 3. let mut removed_contracts = Vec::new(); for (key, expected_generation) in expired { ring.unsubscribe(&key); tracing::info!( %key, "Cleaned up expired hosting subscription from local state" ); if let Some(op_manager) = &op_manager { // A GET/PUT may have re-hosted it since the eviction decision // (#5780); unregistering then would leave a hosted contract // outside anti-entropy (until a restart; see #5784). if !ring.is_hosting_contract(&key) && op_manager.interest_manager.unregister_local_hosting(&key) { removed_contracts.push(key); } crate::operations::reclaim_evicted_contract( op_manager, key, expected_generation, ); } } // Retract neighbor advertisements for contracts that lost all local // interest above. The sweep never ADDS interest, so `added` is always // empty here (unlike the GET/PUT paths). `broadcast_change_interests` // is a no-op when `removed_contracts` is empty (the common case). if let Some(op_manager) = &op_manager { crate::operations::broadcast_change_interests( op_manager, Vec::new(), removed_contracts, ) .await; } // Interest-record reconciliation (#5780): drop neighbour records for // contracts this node has neither hosted nor used for // `RECONCILE_MIN_UNUSED_AGE`. Dropped records reach neighbours // through the next interest heartbeat, which is a full replace. A // stale local-hosting flag cleared here has its co-host // advertisement retracted, as an eviction would; neighbours are told // the interest ended only if it did (a delegate or local client can // keep it). Advertisements are retracted every pass for every // contract past the wait that is unhosted, unused and holds no // lease of this node's own, including advertised contracts with no // records left, so it does not matter how a lease or the records // ended. A contract whose lease is live is left alone until the // lease, which is not demand, lapses unrenewed. if let Some(op_manager) = &op_manager { let outcome = op_manager.interest_manager.reconcile_with_hosting( &op_manager.neighbor_hosting.advertised_contract_keys(), |key| ring.is_hosting_contract(key), |key| ring.contract_in_use(key), |key| ring.is_subscribed(key), ); for key in &outcome.advertisements_to_retract { crate::operations::retract_advertisement_for_evicted_contract(op_manager, key); } if !outcome.hosting_flags_cleared.is_empty() || outcome.contracts_dropped > 0 { tracing::info!( hosting_flags_cleared = outcome.hosting_flags_cleared.len(), contracts_dropped = outcome.contracts_dropped, records_dropped = outcome.records_dropped, "interest records reconciled with the hosted set" ); } // Skip any contract that regained local interest since the pass, // so this removal cannot follow a concurrent re-host's addition. let interest_lost: Vec = outcome .interest_lost .into_iter() .filter(|key| !op_manager.interest_manager.has_local_interest(key)) .collect(); if !interest_lost.is_empty() { crate::operations::broadcast_change_interests( op_manager, Vec::new(), interest_lost, ) .await; } } // Retry pending reclamations queued by the two skip points // (fair-queue rejection of `EvictContract` and the // `contract_in_use` skip in `RuntimePool::remove_contract`). // The snapshot iterates without holding the DashMap shard // guards. Each retry routes through // `reclaim_evicted_contract`, which re-checks // `contract_in_use` — entries that are still in use stay in // the queue (no event emitted) and will be retried next // cycle. Successful reclamations clear their pending entry // in `RuntimePool::remove_contract`. if let Some(op_manager) = &op_manager { let pending = ring.pending_reclamation_snapshot(); if !pending.is_empty() { tracing::debug!( pending_count = pending.len(), "Retrying pending reclamations from previous skipped \ `EvictContract` events" ); for (key, expected_generation) in pending { crate::operations::reclaim_evicted_contract( op_manager, key, expected_generation, ); } } } // Disk-usage accounting maintenance (#4683). Runs OUTSIDE any // hosting-cache write lock (the du-walks would stall cache readers). // First tick lazily seeds the tracker from the true on-disk state // total; every tick re-walks the `du`-measured WASM-blob and // compile-cache totals so telemetry stays fresh. Observational only // in this PR — no admission/eviction decision reads these yet. // // The seed + refresh are synchronous, blocking disk I/O: a recursive // `std::fs` walk of `contracts_dir` + the wasmtime cache dir on every // tick, plus a one-time sync redb `load_all_hosting_metadata` on the // seeding tick. `contracts_dir` is unbounded and non-self-pruning, so // on a node hosting many contracts the walk can run for a while with // no `.await` point. Push it onto a blocking thread so it can never // stall other tasks on the async reactor — the same discipline // `secrets_store/sweep.rs` uses for its "disk walk + ReDb reads". let ring_for_disk = ring.clone(); if let Err(err) = tokio::task::spawn_blocking(move || { #[cfg(feature = "redb")] ring_for_disk.hosting_manager.seed_disk_tracker_if_absent(); ring_for_disk.hosting_manager.refresh_disk_usage(); }) .await { // The walk is observational only; a panic here must not wedge the // sweep loop, but it must not be swallowed silently either. tracing::warn!(%err, "disk-usage accounting maintenance task failed"); } // Eviction floor (#4683, PR 2): recompute // `effective = min(ram_budget, disk_budget)` and install it as the // hosting-cache budget so the next `evict_over_budget` sheds state // down to the tighter of the two resource floors. The free-space read // is the determinism seam — production reads the data-dir mount; a // failed read falls back to `u64::MAX`, which makes the disk budget // clamp to its cap (degrade to "cap only", never to zero). The // du-walks above already ran OUTSIDE any cache lock (in the blocking // task); only the O(1) `set_budget_bytes` inside // `recompute_effective_budget` takes the cache write lock. let available = ring .hosting_manager .disk_available_bytes() .unwrap_or(u64::MAX); ring.hosting_manager.recompute_effective_budget(available); } } /// Periodically register topology snapshots for simulation testing. /// /// This task only runs when `CURRENT_NETWORK_NAME` is set (i.e., during SimNetwork tests). /// It allows SimNetwork to validate subscription topology by querying the global registry. #[cfg(any(test, feature = "testing"))] async fn register_topology_snapshots_periodically( ring: Arc, interval_duration: Duration, ) { use topology_registry::{get_current_network_name, register_topology_snapshot}; tracing::info!("Topology snapshot registration task started"); let shutdown = ring.shutdown_token(); // Add small initial delay to let network stabilize (use short delay in tests) tokio::time::sleep(Duration::from_millis(100)).await; let mut interval = tokio::time::interval(interval_duration); interval.tick().await; // Skip first immediate tick loop { if sleep_or_shutdown(&shutdown, async { interval.tick().await; }) .await { return; } // Only register if we're in a simulation context let Some(network_name) = get_current_network_name() else { tracing::debug!("Topology snapshot: no network name set, skipping"); continue; }; let Some(peer_addr) = ring.connection_manager.get_own_addr() else { tracing::debug!("Topology snapshot: no peer address yet, skipping"); continue; }; // Use get_stored_location() for consistency with set_upstream distance check. // This ensures topology validation uses the same location as the tie-breaker. let location = ring .connection_manager .get_stored_location() .map(|l| l.as_f64()) .unwrap_or(0.0); let mut snapshot = ring .hosting_manager .generate_topology_snapshot(peer_addr, location); // Stamp the live connection count so consumers can use it as a // join/peer_ready progress signal (snapshot presence alone only // means the bind address is set — see `TopologySnapshot::connection_count`). snapshot.connection_count = ring.connection_manager.connection_count(); snapshot.orphan_interest_contracts = ring.orphan_interest_contract_count(); snapshot.stale_advertisements = ring.stale_advertisement_count(); snapshot.reconcile_contracts_dropped = ring .upgrade_op_manager() .map(|op| op.interest_manager.reconcile_contracts_dropped_total()); let contract_count = snapshot.contracts.len(); register_topology_snapshot(&network_name, snapshot); tracing::info!( %peer_addr, location, network = %network_name, contract_count, "Registered topology snapshot" ); } } /// Record an access to a contract in the hosting cache. /// /// This adds or refreshes the contract in the unified hosting cache. /// ALL contracts in the hosting cache get subscription renewal. /// /// Returns a `RecordAccessResult` containing: /// - `is_new`: Whether this contract was newly added (vs. refreshed existing) /// - `evicted`: Contracts that were evicted to make room /// /// `cause` attributes WHY this peer is (possibly) starting to host: the /// caller is the only code that knows whether this is its own client's /// request, someone else's request in transit, or a sub-op fetch. It feeds /// telemetry only — it changes nothing about what is hosted or evicted. pub fn host_contract( &self, key: ContractKey, size_bytes: u64, access_type: AccessType, cause: HostingCause, ) -> RecordAccessResult { self.hosting_manager .record_contract_access(key, size_bytes, access_type, cause) } /// Record a GET access to a contract in the hosting cache. /// /// Returns a `RecordAccessResult` indicating whether this was a new addition /// and which contracts were evicted (if any). pub fn record_get_access( &self, key: ContractKey, size_bytes: u64, cause: HostingCause, ) -> RecordAccessResult { self.host_contract(key, size_bytes, AccessType::Get, cause) } /// Record that a local-client GET was answered from local hosted state /// (a hit) — see [`HostingManager::record_local_get_serve`]. (#4642 A3) pub fn record_get_served_locally(&self) { self.hosting_manager.record_local_get_serve(); } /// Record that a local-client GET was routed to the network (a forward/miss) /// — see [`HostingManager::record_local_get_forward`]. (#4642 A3) pub fn record_get_forwarded(&self) { self.hosting_manager.record_local_get_forward(); } /// Record a local `UpdateNotification` dropped by a FULL subscriber channel /// (#4681) — see [`HostingManager::record_notification_dropped_channel_full`]. pub(crate) fn record_notification_dropped_channel_full(&self) { self.hosting_manager .record_notification_dropped_channel_full(); } /// Record a local `UpdateNotification` dropped by a CLOSED subscriber /// channel (#4681). pub(crate) fn record_notification_dropped_channel_closed(&self) { self.hosting_manager .record_notification_dropped_channel_closed(); } /// Record a committed update that found NO local subscriber (#4681/#5040). pub(crate) fn record_notification_no_local_subscriber(&self) { self.hosting_manager .record_notification_no_local_subscriber(); } /// Number of client GETs this node answered from local hosted state (A3 /// serve-DURING hit counter). Read accessor for the counter incremented by /// [`Self::record_get_served_locally`]; used by the serve-DURING sim /// falsifier to assert a demandless-copy GET was served locally, not routed. #[cfg(any(test, feature = "testing"))] pub fn local_get_serves(&self) -> u64 { self.hosting_manager.local_get_serves() } /// Number of client GETs this node routed to the network (A3 forward/miss /// counter). Read accessor for the counter incremented by /// [`Self::record_get_forwarded`]; the serve-DURING falsifier asserts this /// stays `0` for a demandless-copy GET (the node never went dark). #[cfg(any(test, feature = "testing"))] pub fn local_get_forwards(&self) -> u64 { self.hosting_manager.local_get_forwards() } /// Whether this node is hosting this contract (has it in cache). #[inline] pub fn is_hosting_contract(&self, key: &ContractKey) -> bool { self.hosting_manager.is_hosting_contract(key) } /// The composed #4610 summarize/broadcast gate (see /// [`HostingManager::should_summarize_or_broadcast`]): /// `(is_hosting_contract || contract_in_use) && contract_state_present`. /// Skips phantom (interested-but-stateless) contracts while still serving /// evicted-but-in-use ones whose state remains on disk. #[inline] pub fn should_summarize_or_broadcast(&self, key: &ContractKey) -> bool { self.hosting_manager.should_summarize_or_broadcast(key) } /// Set the storage reference for hosting metadata persistence. /// /// Must be called after executor creation. This enables automatic /// cleanup of persisted metadata when contracts are evicted. pub fn set_hosting_storage(&self, storage: crate::contract::storages::Storage) { self.hosting_manager.set_storage(storage); } /// Install the aggregate disk-usage tracker's paths (#4683). Called once at /// startup with the node's mode-resolved contracts dir and the relocated /// wasmtime compile-cache dir. The tracker is seeded lazily on the first /// sweep tick. pub fn set_hosting_disk_paths( &self, contracts_dir: std::path::PathBuf, wasmtime_cache_dir: std::path::PathBuf, hosting_disk_pct: f64, max_hosting_disk: u64, ) { self.hosting_manager .configure_disk_tracker(contracts_dir, wasmtime_cache_dir); // #4683 eviction-floor sizing knobs (mirrors the disk paths install; the // config is only reachable here). The 60s sweep's recompute reads them. self.hosting_manager .configure_disk_budget(hosting_disk_pct, max_hosting_disk); } /// Install the operator-configurable share of the node's memory limit /// that hosted contracts may hold in RAM (`--hosting-mem-share`, #5333, /// #5647). Called once at startup; the 60s sweep's recompute reads it. pub fn configure_resident_overhead_mem_share(&self, mem_share: f64) { self.hosting_manager .configure_resident_overhead_mem_share(mem_share); } /// Drop the ring's clones of the redb `Storage` handle (hosting metadata + /// broken-invariants persistence). Called on node shutdown so the redb /// `Database` Arc count can fall to zero and release the on-disk file lock /// — otherwise an in-process restart against the same data dir deadlocks on /// the still-held lock. See issue #4401. pub(crate) fn clear_redb_storage(&self) { self.hosting_manager.clear_storage(); self.broken_invariants.clear_storage(); } /// Load hosting cache from persisted storage. /// /// Call this during startup after storage is available to restore /// the hosting cache from the previous run. Also migrates legacy contracts /// that have state but no hosting metadata. /// /// # Arguments /// * `storage` - The storage backend /// * `code_hash_lookup` - Function to look up CodeHash from ContractInstanceId. /// Uses ContractStore which has the id->code_hash mapping. #[cfg(feature = "redb")] pub fn load_hosting_cache( &self, storage: &crate::contract::storages::Storage, code_hash_lookup: F, ) -> Result where F: Fn( &freenet_stdlib::prelude::ContractInstanceId, ) -> Option, { self.hosting_manager .load_from_storage(storage, code_hash_lookup) } /// Load hosting cache from persisted storage (sqlite version). /// /// Also migrates legacy contracts that have state but no hosting metadata. #[cfg(all(feature = "sqlite", not(feature = "redb")))] pub async fn load_hosting_cache( &self, storage: &crate::contract::storages::Storage, code_hash_lookup: F, ) -> Result where F: Fn( &freenet_stdlib::prelude::ContractInstanceId, ) -> Option, { self.hosting_manager .load_from_storage(storage, code_hash_lookup) .await } pub fn record_request( &self, recipient: PeerKeyLocation, target: Location, request_type: TransactionType, ) { self.connection_manager .topology_manager .write() .record_request(recipient, target, request_type); } /// Add a connection to the ring topology. /// /// Returns `true` if this connection caused us to cross the readiness threshold /// (i.e., we just became ready to accept non-CONNECT operations). /// Returns `false` if the connection was rejected (e.g., capacity cap) or we /// were already ready. /// /// NOTE the two meanings of `false` collapsed here: callers that need to /// know whether the connection was actually *added* — as opposed to /// whether readiness was just crossed — must use /// [`Ring::add_connection_reporting`]. Reading this `bool` as "added" is /// wrong in both directions: it is `false` for the overwhelmingly common /// successful add (we were already ready), and it is only ever `true` for /// the single add that crosses the threshold. pub async fn add_connection(&self, loc: Location, peer: PeerId, was_reserved: bool) -> bool { self.add_connection_reporting(loc, peer, was_reserved) .await .just_became_ready } /// [`Ring::add_connection`], reporting whether the ring actually accepted /// the connection as well as whether readiness was crossed. pub async fn add_connection_reporting( &self, loc: Location, peer: PeerId, was_reserved: bool, ) -> AddConnectionOutcome { tracing::info!( peer = %peer, peer_location = %loc, this = ?self.connection_manager.get_own_addr(), was_reserved = %was_reserved, "Adding connection to peer" ); let min_ready = self.connection_manager.min_ready_connections; let was_ready = min_ready == 0 || self.connection_manager.connection_count() >= min_ready; let addr = peer.socket_addr(); let pub_key = peer.pub_key().clone(); let added = self .connection_manager .add_connection(loc, addr, pub_key, was_reserved); if !added { tracing::warn!( peer = %peer, peer_location = %loc, "Ring rejected connection - not updating caches or logging connection event" ); return AddConnectionOutcome { added: false, just_became_ready: false, }; } if let Some(own_loc) = self.connection_manager.own_location().location() { crate::node::network_status::set_own_location(own_loc.as_f64()); } // ConnectEvent::Connected telemetry is emitted by the CONNECT state // machine with proper transaction context; not duplicated here (#3578). self.refresh_density_request_cache(); let is_ready = self.connection_manager.is_self_ready(); AddConnectionOutcome { added: true, // Only report readiness if we just crossed the threshold. just_became_ready: !was_ready && is_ready, } } pub fn update_connection_identity(&self, old_peer: &PeerId, new_peer: PeerId) { if self.connection_manager.update_peer_identity( old_peer.socket_addr(), new_peer.socket_addr(), new_peer.pub_key().clone(), ) { self.refresh_density_request_cache(); } } fn refresh_density_request_cache(&self) { let cbl = self.connection_manager.get_connections_by_location(); let topology_manager = &mut self.connection_manager.topology_manager.write(); let _refreshed = topology_manager.refresh_cache(&cbl); } /// Returns a filtered iterator for peers that are not connected to this node already. pub fn is_not_connected<'a>( &self, peers: impl Iterator, ) -> impl Iterator + Send { let mut filtered = Vec::new(); for peer in peers { if let Some(addr) = peer.socket_addr() { if !self.connection_manager.has_connection_or_pending(addr) { filtered.push(peer); } } else { // If address is unknown, include the peer filtered.push(peer); } } filtered.into_iter() } /// Return the most optimal peer for hosting a given contract. /// /// This function only considers connected peers, not the node itself. /// `log_as` says whether this selection is a routing decision for the /// routing dataset's candidate log. #[inline] pub fn closest_potentially_hosting( &self, log_as: crate::router::dataset::DecisionLog, contract_key: &ContractKey, skip_list: impl Contains, ) -> Option { let routes = matches!(log_as, crate::router::dataset::DecisionLog::Joinable(_)); let log = self.candidate_log_for(log_as); // The router read lock is held across candidate gathering and // selection, as before candidate logging existed; it is released // before the dataset write and the debug trace. let router = self.router.read(); let target = Location::from(contract_key); self.connection_manager .routing_with(target, None, skip_list, move |candidates| { let (selected, decision, capture) = router.select_k_best_peers_capturing( candidates.iter(), target, 1, log.as_ref().is_some_and(|(log, _)| log.capture), routes, ); drop(router); if let Some((log, op)) = log { log.record( op, target, &selected, matches!( decision.strategy, crate::router::RoutingStrategy::DistanceBased ), capture, ); } let peer = selected.into_iter().next().cloned(); tracing::debug!( target_location = %target.as_f64(), strategy = ?decision.strategy, num_candidates = decision.candidates.len(), total_routing_events = decision.total_routing_events, selected = peer.is_some(), "routing_decision" ); peer }) .flatten() } /// The candidate log for a selection, and its op, when it is a routing /// decision and logging is on. `Unlogged` consults nothing. fn candidate_log_for( &self, log_as: crate::router::dataset::DecisionLog, ) -> Option<( crate::router::dataset::CandidateLog<'static>, crate::node::network_status::OpType, )> { match log_as { crate::router::dataset::DecisionLog::Joinable(op) => { crate::router::dataset::candidate_log(|| self.time_source.now()) .map(|log| (log, op)) } crate::router::dataset::DecisionLog::Unlogged => None, } } /// Get k best peers for hosting a contract, ranked by routing predictions. /// Accepts either &ContractKey or &ContractInstanceId (both implement From<&T> for Location). pub fn k_closest_potentially_hosting( &self, log_as: crate::router::dataset::DecisionLog, contract_id: &K, skip_list: impl Contains + Clone, k: usize, ) -> Vec where for<'a> Location: From<&'a K>, { // Router read-lock is only needed for the final `select_*` call; // building the candidate list does not need it. Acquire it late to // keep the critical section short (clippy: `significant_drop_tightening`). let target_location = Location::from(contract_id); let mut seen = HashSet::new(); let mut candidates: Vec = Vec::new(); let mut not_ready_fallback: Vec = Vec::new(); let mut skipped_not_ready: usize = 0; let mut skipped_transient: usize = 0; let connections = self.connection_manager.get_connections_by_location(); // Sort keys for deterministic iteration order (HashMap iteration is non-deterministic) // This ensures the `seen.insert()` check behaves consistently across runs let mut sorted_keys: Vec<_> = connections.keys().collect(); sorted_keys.sort(); for loc in sorted_keys { let conns = connections.get(loc).expect("key exists"); // Sort connections for deterministic iteration order let mut sorted_conns: Vec<_> = conns.iter().collect(); sorted_conns.sort_by_key(|c| c.location.clone()); for conn in sorted_conns { if let Some(addr) = conn.location.socket_addr() { if skip_list.has_element(addr) || !seen.insert(addr) { continue; } // Skip transient peers — these are short-TTL connections used for // CONNECT coordination, not stable routing targets. PUT/UPDATE // already exclude them via `ConnectionManager::routing_candidates` // (see connection_manager.rs:1578); previously GET/SUBSCRIBE // could route through them and waste hops on a forwarder that // was about to be dropped. Issue #4222 / #3570. if self.connection_manager.is_transient(addr) { tracing::debug!( %addr, target_location = %target_location.as_f64(), "k_closest: skipping transient peer" ); skipped_transient += 1; continue; } // Skip peers that haven't advertised readiness, but collect them // as fallback candidates in case all peers fail the readiness check. if !self.connection_manager.is_peer_ready(addr) { tracing::debug!( %addr, target_location = %target_location.as_f64(), "k_closest: skipping peer not yet ready" ); not_ready_fallback.push(conn.location.clone()); skipped_not_ready += 1; continue; } } else { // Addressless candidates bypass every filter above (skip // list, dedup, transient, readiness) because all of them // key on the socket addr. If the router then selects one, // the GET advance helpers can't use it as a wire target // and the hop dies. Keep the inclusion (the candidate may // gain an addr by send time) but make it visible (#4361). tracing::debug!( peer = ?conn.location, target_location = %target_location.as_f64(), "k_closest: including addressless candidate (bypasses addr-keyed filters)" ); } candidates.push(conn.location.clone()); } } // If all connected peers failed the readiness check, fall back to using them anyway. // This prevents GET/SUBSCRIBE operations from failing with EmptyRing when the node // is connected but peers haven't yet sent ReadyState messages (e.g., early after // connecting, or in network topologies where the min_ready_connections threshold is // never satisfied). A warn-level log is emitted so operators know gating was bypassed. // Note: ConnectionManager::routing_candidates has the same fallback for PUT/UPDATE. if candidates.is_empty() && !not_ready_fallback.is_empty() { tracing::warn!( count = not_ready_fallback.len(), target_location = %target_location.as_f64(), "k_closest: no ready peers available, falling back to not-yet-ready peers to avoid EmptyRing" ); candidates = not_ready_fallback; } // If the entire connection set was filtered out and transients carried // weight in that filtering, warn the operator. There is no transient // fallback by design — transient peers are short-TTL coordination slots // and routing through them just wastes hops — but a sustained // empty-candidate state means the local node has no viable routing // options for this hop, which is operator-visible information that // would otherwise only surface as an `EmptyRing` higher up. if candidates.is_empty() && skipped_transient > 0 { tracing::warn!( skipped_transient, target_location = %target_location.as_f64(), "k_closest: no viable peers — all eligible connections were transient \ (no fallback by design; routing will fail this hop)" ); } // Sort candidates for deterministic input to select_k_best_peers candidates.sort(); // Note: We intentionally do NOT fall back to known_locations here. // known_locations may contain peers we're not currently connected to, // and attempting to route to them would require establishing a new connection // which may fail (especially in NAT scenarios without coordination). // It's better to return fewer candidates than unreachable ones. let routes = matches!(log_as, crate::router::dataset::DecisionLog::Joinable(_)); let log = self.candidate_log_for(log_as); let (selected, decision, capture) = self.router.read().select_k_best_peers_capturing( candidates.iter(), target_location, k, log.as_ref().is_some_and(|(log, _)| log.capture), routes, ); // `selected` and `capture` borrow from `candidates`, not from the // router guard, so the read lock is released at the end of the // statement above, before the dataset write. if let Some((log, op)) = log { log.record( op, target_location, &selected, matches!( decision.strategy, crate::router::RoutingStrategy::DistanceBased ), capture, ); } tracing::debug!( target_location = %target_location.as_f64(), strategy = ?decision.strategy, num_candidates = decision.candidates.len(), total_routing_events = decision.total_routing_events, selected_count = selected.len(), "routing_decision" ); tracing::debug!( target_location = %target_location.as_f64(), candidates_found = selected.len(), skipped_not_ready, skipped_transient, "k_closest_potentially_hosting result" ); selected.into_iter().cloned().collect() } pub fn routing_finished(&self, event: crate::router::RouteEvent) { self.report_route_outcome_to_health(&event); self.router.write().add_event(event); } /// Everything [`Self::routing_finished`] does except feed the router: the /// topology manager's outbound-request accounting and the `peer_health` /// success or failure. `routing_finished` is exactly this plus /// `Router::add_event`, so the two cannot drift. /// /// #5657 labels the ROUTER against the hop an attempt was actually /// forwarded to. Peer health and topology keep their pre-#5657 inputs /// unchanged (the originator's `current_target`, the same events, the same /// conditions): changing them would change health-based eviction and the /// request-density model, which #5657 does not set out to do. pub(crate) fn report_route_outcome_to_health(&self, event: &crate::router::RouteEvent) { self.connection_manager .topology_manager .write() .report_outbound_request(event.peer.clone(), event.contract_location); // Update peer health tracking based on routing outcome. if let Some(addr) = event.peer.socket_addr() { let mut health = self.connection_manager.peer_health.lock(); match &event.outcome { crate::router::RouteOutcome::Success { .. } | crate::router::RouteOutcome::SuccessUntimed => { health.record_success(addr); } crate::router::RouteOutcome::Failure => { health.record_failure(addr); } } } } /// Feed a route event to the routing model ONLY, bypassing the /// `peer_health` and `topology_manager` side effects of /// [`Self::routing_finished`]. /// /// Two reasons, both load-bearing: /// /// 1. **Peer health is not routing success.** `PeerHealthTracker` evicts a /// connection at a 90 % failure rate over 10 events, or after 10 minutes /// with one failure and no success. A `NotFound` (the peer does not have /// this contract) or an end-to-end timeout (anywhere down the chain) says /// nothing about whether the connection to the first hop is healthy, and /// feeding them in would evict healthy peers, most exposed a gateway that /// fields many requests for absent contracts. /// 2. **Determinism.** `PeerHealthTracker` stamps `std::time::Instant::now()` /// (a pre-existing TimeSource rule violation). An earlier iteration of /// the relay hooks routed relay events through `routing_finished` and /// broke the strict-determinism tests (`test_strict_determinism_*`, /// `test_direct_runner_determinism`, `test_thundering_herd_connect_storm`). /// /// `source` tags the event in the opt-in routing dataset (#5648): `Relay` /// for an outcome a relay hop observed about its downstream peer, /// `Originator` for one observed by the node that started the operation. /// The model treats both identically. pub(crate) fn record_route_event_router_only( &self, event: crate::router::RouteEvent, source: crate::router::dataset::RouteSource, ) { use crate::router::dataset::RouteSource; let mut router = self.router.write(); match source { RouteSource::Originator => router.add_event(event), RouteSource::Relay => router.add_relay_event(event), } } /// Record a routing FAILURE label for one attempt. Router only; see /// [`Self::record_route_event_router_only`]. The sole production sink of /// [`crate::operations::route_attempt::RouteAttemptRecorder`]; `cause` is /// counted in [`Self::route_failure_cause_counts`]. pub(crate) fn record_route_failure( &self, event: crate::router::RouteEvent, cause: crate::operations::route_attempt::AttemptFailure, source: crate::router::dataset::RouteSource, ) { use crate::operations::route_attempt::AttemptFailure; debug_assert!( matches!(event.outcome, crate::router::RouteOutcome::Failure), "record_route_failure called with a non-failure outcome" ); let counter = match cause { AttemptFailure::NotFound => &self.route_failure_causes.not_found, AttemptFailure::Timeout => &self.route_failure_causes.timeout, AttemptFailure::SendFailure => &self.route_failure_causes.send_failure, }; counter.fetch_add(1, std::sync::atomic::Ordering::Relaxed); if cause == AttemptFailure::Timeout && matches!(source, crate::router::dataset::RouteSource::Originator) { self.route_failure_causes .originator_timeout .fetch_add(1, std::sync::atomic::Ordering::Relaxed); } if cause == AttemptFailure::Timeout { if let Some(addr) = event.peer.socket_addr() { self.timeout_label_window.lock().record(addr); } } self.record_route_event_router_only(event, source); } /// Count ambiguous `NotFound`s an operation dropped untrained (#5657). /// Never trained on; lets a test prove NotFounds actually occurred. pub(crate) fn record_untrained_not_founds(&self, count: u64) { self.route_failure_causes .untrained_not_found .fetch_add(count, std::sync::atomic::Ordering::Relaxed); } /// Ambiguous `NotFound`s dropped untrained on this node (#5657). #[cfg_attr(not(any(test, feature = "testing")), allow(dead_code))] pub(crate) fn untrained_not_found_count(&self) -> u64 { self.route_failure_causes .untrained_not_found .load(std::sync::atomic::Ordering::Relaxed) } /// Drain the per-peer timeout-label window into its histogram (#5657). /// Called once per router snapshot, which is its only consumer: the /// window is a drain, so no second reader (a local dashboard) can share /// it without stealing labels from the telemetry export. pub(crate) fn take_timeout_label_histogram(&self) -> TimeoutLabelHistogram { self.timeout_label_window.lock().take_histogram() } /// These count only labels that went through the recorder. They do NOT /// reconcile with `Router::outcome_totals().failures`: PUT relay's /// downstream forwarding still labels its own failures through /// `record_relay_route_event` (not yet migrated), and `LabelMode::Legacy` /// restores `routing_finished` failure events that bypass them too. /// /// `(not_found, timeout, send_failure)` failure labels this node has fed /// its router, by cause (#5657). #[cfg_attr(not(any(test, feature = "testing")), allow(dead_code))] pub(crate) fn route_failure_cause_counts(&self) -> (u64, u64, u64) { use std::sync::atomic::Ordering::Relaxed; ( self.route_failure_causes.not_found.load(Relaxed), self.route_failure_causes.timeout.load(Relaxed), self.route_failure_causes.send_failure.load(Relaxed), ) } /// Timeout failure labels this node recorded as the originator of an /// operation: the subset of [`Self::route_failure_cause_counts`]'s timeouts /// that excludes labels recorded while relaying (#5660). #[cfg_attr(not(any(test, feature = "testing")), allow(dead_code))] pub(crate) fn originator_route_timeout_count(&self) -> u64 { self.route_failure_causes .originator_timeout .load(std::sync::atomic::Ordering::Relaxed) } // ==================== Subscription Management (Lease-Based) ==================== /// Subscribe to a contract with a lease. /// /// Creates a new subscription or renews an existing one. The subscription /// will expire after `SUBSCRIPTION_LEASE_DURATION` unless renewed. pub fn subscribe(&self, contract: ContractKey) -> SubscribeResult { self.hosting_manager.subscribe(contract) } /// Unsubscribe from a contract. /// /// Removes the active subscription. The contract may still be hosted /// (in the hosting cache) until evicted by LRU. pub fn unsubscribe(&self, contract: &ContractKey) { self.hosting_manager.unsubscribe(contract) } /// Check if we have an active (non-expired) subscription to a contract. pub fn is_subscribed(&self, contract: &ContractKey) -> bool { self.hosting_manager.is_subscribed(contract) } /// Whether the FULL contract state (code + params + state) is present on /// disk for `contract` — the cheap on-disk state-store existence check, NOT /// the in-memory hosting cache. Hosting is binary: a `true` from a healthy /// store means this peer holds the whole contract. /// /// CAVEAT — conservative assume-present fallback: this returns `true` (not /// `false`) when the state store is transiently unavailable — a redb read /// error, an unset storage handle before startup, or a non-redb build. So a /// `true` is "present, OR presence currently unknowable", never a hard /// guarantee. In the reconcile shadow (keystone step-2, #4642) that can /// slightly INFLATE the renewal-site `announce` divergence during a transient /// store outage (the controller would `Announce` a host we can't confirm has /// the body); it is a startup/outage transient, documented so the counter is /// read correctly rather than mistaken for a real anomaly. pub(crate) fn contract_state_present(&self, contract: &ContractKey) -> bool { self.hosting_manager.contract_state_present(contract) } /// Async existence probe (redb sync fast-path + real SQLite EXISTS probe) — /// see [`hosting::HostingManager::contract_state_present_async`]. Used by the /// ResyncResponse responder so the pre-limiter existence gate is backend- /// agnostic (#4864 round-5 P1). pub(crate) async fn contract_state_present_async(&self, contract: &ContractKey) -> bool { self.hosting_manager .contract_state_present_async(contract) .await } /// Whether at least one LOCAL WebSocket client is currently subscribed to /// `contract` (real local demand, distinct from downstream peer /// subscribers). Used by the reconcile input-builder (keystone step-2, /// #4642) for `ReconcileInputs::has_local_client`. pub(crate) fn has_client_subscriptions(&self, contract: &ContractKey) -> bool { self.hosting_manager.has_client_subscriptions(contract.id()) } /// Lease-valid downstream subscriber peer keys for `contract` (see /// [`hosting::HostingManager::downstream_subscriber_peers`]). Used by the /// reconcile input-builder (keystone step-2, #4642) to apply the piece-D /// strictly-farther filter. pub(crate) fn downstream_subscriber_peers(&self, contract: &ContractKey) -> Vec { self.hosting_manager.downstream_subscriber_peers(contract) } /// Get all contracts with active subscriptions. pub fn get_subscribed_contracts(&self) -> Vec { self.hosting_manager.get_subscribed_contracts() } /// Bounded, sort-free lookup of currently-subscribed `ContractKey`s whose /// instance-id is in `wanted` (see /// [`hosting::HostingManager::subscribed_keys_in`]). Used by the reconcile /// connection-drop shadow (keystone step-2, #4642) to avoid the O(S log S) /// full-set scan of [`Self::get_subscribed_contracts`] per disconnect. pub(crate) fn subscribed_keys_in( &self, wanted: &std::collections::HashSet, scan_cap: usize, max_matches: usize, ) -> Vec { self.hosting_manager .subscribed_keys_in(wanted, scan_cap, max_matches) } /// Force-expire a contract's subscription so it gets renewed through the /// current best route on the next recovery cycle. fn force_subscription_renewal(&self, contract: &ContractKey) { self.hosting_manager.force_subscription_renewal(contract); } /// Expire stale subscriptions and return the contracts that were expired. /// /// Should be called periodically by a background task. pub fn expire_stale_subscriptions(&self) -> Vec { self.hosting_manager.expire_stale_subscriptions() } // ==================== Downstream Subscriber Tracking ==================== pub fn add_downstream_subscriber( &self, contract: &ContractKey, peer: PeerKey, ) -> crate::ring::hosting::AddSubscriberOutcome { // No governance demand is ingested here anymore. Benefit is a // LIVE SNAPSHOT read fresh each reaper tick from the hosting // manager's standing subscriber count (see // `Ring::governance_tick` and // `HostingManager::downstream_subscriber_count`), not an // accumulator fed by subscribe events. That makes the // Sybil-on-renewal concern moot: renewals merely extend a lease // the snapshot already counts, so they cannot inflate benefit — // the count reflects the CURRENT lease-valid subscriber set, no // matter how often each peer renews. // // Downstream demand is still weighted low (FORWARDED_DEMAND_WEIGHT // = 0.1) when the snapshot is computed, because peer identities // are attacker-rotatable: without an identity layer a single // attacker can spin up many peers that each "subscribe", so each // forwarded subscriber is worth one tenth of a real local client. // The NewAdd/Renewal distinction is load-bearing for the caller // (#4952 follow-through): the demand-counter increment in // `register_downstream_subscriber` must key on HOSTING-map newness, // not on `interested_peers` newness — summary-upserted entries make // the latter unreliable (a delivery-seeded co-host that later // genuinely subscribes is not "new" in the interest map but IS new // demand). self.hosting_manager .add_downstream_subscriber(contract, peer) } #[allow(dead_code)] // Only used in tests pub fn renew_downstream_subscriber(&self, contract: &ContractKey, peer: &PeerKey) -> bool { self.hosting_manager .renew_downstream_subscriber(contract, peer) } pub fn remove_downstream_subscriber(&self, contract: &ContractKey, peer: &PeerKey) -> bool { self.hosting_manager .remove_downstream_subscriber(contract, peer) } pub fn has_downstream_subscribers(&self, contract: &ContractKey) -> bool { self.hosting_manager.has_downstream_subscribers(contract) } /// Whether something still depends on this node hosting `contract` — a /// live local client subscription or a downstream peer subscriber. /// /// Used to gate hosting-cache eviction reclamation: a contract that is in /// use must not have its on-disk state/code deleted. See /// `HostingManager::contract_in_use` for why an active upstream network /// subscription alone does NOT make the contract in-use (the renewal /// machinery would refresh the lease unboundedly). pub(crate) fn contract_in_use(&self, contract: &ContractKey) -> bool { self.hosting_manager.contract_in_use(contract) } /// Instance ids of every contract [`Self::contract_in_use`] holds for. See /// `HostingManager::in_use_contract_ids`. pub(crate) fn in_use_contract_ids(&self) -> Vec { self.hosting_manager.in_use_contract_ids() } /// Single helper for every state-write chokepoint. Does the three /// things a chokepoint MUST do, in order: /// /// 1. Bump the per-contract write generation (closes the /// `EvictContract` re-host race — see /// `HostingManager::state_generation`). /// 2. Refresh the hosting-cache snapshot of that generation so /// already-hosted contracts don't leak on eviction after this /// write — see `HostingCache::refresh_entry_generation`. /// 3. Report `state_size` bytes against this contract on the /// `StateBytesWritten` axis of the topology meter, feeding the /// governance scoring layer (see `crate::governance`). /// /// Every state-write chokepoint in the executor MUST go through /// this helper, NOT call the three primitives by hand. The /// "manually-mirrored side effects after a task-per-tx migration" /// pattern in `.claude/rules/bug-prevention-patterns.md` lists this /// exact failure mode: one site drops the report and governance /// silently undercounts that path for months before anyone notices. /// Pre-write admission gate for a state write (#4683, PR 3). Call this /// BEFORE the `state_store.{store,update}` at every chokepoint: it rejects a /// write that would push aggregate on-disk usage past the disk budget, so no /// bytes ever land for a rejected write (no rollback needed for the write /// itself). Read-only — the `+delta` is applied by /// [`Self::commit_state_write`] only on the post-write success path, so a /// rejected or later-failed write never mutates the tracker. /// /// A no-op admit until the disk tracker is seeded (early startup), so it /// never spuriously blocks a write before the aggregate is meaningful. /// /// On rejection the chokepoint converts the error to a non-fatal /// `ExecutorError::request(StdContractError::Put/Update)` → `new_value: Err`, /// which for PUT rides `PutMsg::Error` to the client and the network. pub(crate) fn admit_state_write( &self, contract: &ContractKey, new_size: usize, ) -> Result<(), DiskBudgetExceeded> { self.hosting_manager .admit_state_write(contract, new_size as u64) } /// Pre-write admission gate for a state **UPDATE** to an already-hosted /// contract (#4683). Growth-only: a shrinking or size-holding UPDATE /// (`new_size <= old`) is admitted unconditionally, even when the aggregate /// is over budget, because an UPDATE mutates an already-counted footprint and /// rejecting it would stall CRDT convergence without freeing any bytes (and a /// relayed UPDATE rejection is silently dropped — no one would learn of the /// stall). Only genuine growth is subjected to the aggregate bound. Use this /// at UPDATE / re-PUT-merge chokepoints; PUT of a NEW contract keeps the hard /// [`Self::admit_state_write`] gate. pub(crate) fn admit_state_update( &self, contract: &ContractKey, new_size: usize, ) -> Result<(), DiskBudgetExceeded> { self.hosting_manager .admit_state_update(contract, new_size as u64) } /// Pre-write admission gate for a newly-stored (deduped) WASM code blob /// (#4683, PR 3). Call this BEFORE `runtime.store_contract` for a blob that /// is not already on disk. See [`Self::admit_state_write`] for the /// deferred-delta / rollback discipline. pub(crate) fn admit_wasm_write(&self, blob_len: usize) -> Result<(), DiskBudgetExceeded> { self.hosting_manager.admit_wasm_write(blob_len as u64) } /// Charge a newly-stored (deduped) WASM code blob to the disk tracker on the /// post-store success path (#4683). Call this AFTER a successful /// `runtime.store_contract` for a blob that was not already on disk, so the /// aggregate reflects it immediately (burst protection + so the state gate on /// the same PUT sees it). See [`super::hosting::HostingManager::record_wasm_write`]. pub(crate) fn record_wasm_write(&self, blob_len: usize) { self.hosting_manager.record_wasm_write(blob_len as u64); } /// Subtract a WASM code blob's contribution from the disk tracker on contract /// removal (#4683). Mirror of [`Self::record_wasm_write`]. pub(crate) fn record_wasm_removed(&self, blob_len: usize) { self.hosting_manager.record_wasm_removed(blob_len as u64); } /// Test-only passthroughs to the `HostingManager` disk-budget seams (#4683). /// `hosting_manager` is a private field, so an executor-path test holding an /// `Arc` cannot reach the seeding/budget helpers directly; these /// forward to them so a wired `Executor<_>` can be driven into a disk-budget /// rejection without waiting for the 60s recompute. No production behavior. #[cfg(test)] pub(crate) fn seed_disk_tracker_for_test(&self, rows: I) where I: IntoIterator, { self.hosting_manager.seed_disk_tracker_for_test(rows); } #[cfg(test)] pub(crate) fn configure_disk_budget_for_test(&self, disk_pct: f64, max_hosting_disk: u64) { self.hosting_manager .configure_disk_budget(disk_pct, max_hosting_disk); } #[cfg(test)] pub(crate) fn recompute_effective_budget_for_test(&self, available: u64) -> Option { self.hosting_manager.recompute_effective_budget(available) } #[cfg(test)] pub(crate) fn disk_budget_bytes_for_test(&self) -> u64 { self.hosting_manager.disk_budget_bytes() } #[cfg(test)] pub(crate) fn disk_usage_stats_for_test(&self) -> Option { self.hosting_manager.disk_usage_stats() } pub(crate) fn commit_state_write(&self, contract: &ContractKey, state_size: usize) { let new_gen = self.hosting_manager.bump_state_generation(contract); self.hosting_manager .refresh_cache_generation(contract, new_gen); // Disk-usage accounting (#4683): maintain the aggregate on-disk state // total by signed delta (`new − previous_for_key`). Observational only // in this PR — no admission/eviction decision reads it yet. No-op until // the tracker is seeded. self.hosting_manager .record_state_write(contract, state_size as u64); self.report_contract_resource_usage( *contract.id(), crate::topology::meter::ResourceType::StateBytesWritten, state_size as f64, ); } /// Subtract a reclaimed contract's state bytes from the aggregate disk-usage /// tracker (#4683). Called from the executor's reclaim path on successful /// state deletion. Observational only in this PR; no-op until the tracker is /// seeded. pub(crate) fn record_state_removed(&self, contract: &ContractKey) { self.hosting_manager.record_state_removed(contract); } /// Read the current state-write generation for `contract` (0 if never written). pub(crate) fn state_generation(&self, contract: &ContractKey) -> u64 { self.hosting_manager.state_generation(contract) } /// Forget the state-write generation entry for `contract` after a /// successful disk reclamation. Keeps the generation map bounded. pub(crate) fn forget_state_generation(&self, contract: &ContractKey) { self.hosting_manager.forget_state_generation(contract) } /// Add `contract` to the pending-reclamation retry queue with the /// captured `expected_generation`. Called from the two skip points /// that drop an `EvictContract` event before it can complete (fair /// queue rejection, `contract_in_use` skip in `RuntimePool::remove_contract`). /// See `HostingManager::pending_reclamation_add` for the queue's /// invariants and how the periodic sweep retries entries. pub(crate) fn pending_reclamation_add(&self, contract: ContractKey, expected_generation: u64) { self.hosting_manager .pending_reclamation_add(contract, expected_generation) } /// Remove `contract` from the pending-reclamation retry queue after /// a successful disk reclamation. See /// `HostingManager::pending_reclamation_remove`. pub(crate) fn pending_reclamation_remove(&self, contract: &ContractKey) { self.hosting_manager.pending_reclamation_remove(contract) } /// Snapshot the pending-reclamation queue for the periodic sweep /// to iterate without holding any DashMap shard guard. See /// `HostingManager::pending_reclamation_snapshot`. pub(crate) fn pending_reclamation_snapshot(&self) -> Vec<(ContractKey, u64)> { self.hosting_manager.pending_reclamation_snapshot() } pub fn expire_stale_downstream_subscribers(&self) -> Vec<(ContractKey, usize)> { self.hosting_manager.expire_stale_downstream_subscribers() } /// Reconcile downstream-driven in-use contracts against the state store /// (#4612): emit repair fetches for phantoms (in-use, stateless) and /// drops for unrepairable ones. See /// `HostingManager::reconcile_phantom_in_use`. pub(crate) fn reconcile_phantom_in_use( &self, max_fetches: usize, ) -> Vec { self.hosting_manager.reconcile_phantom_in_use(max_fetches) } /// Drop the downstream registration of an absolute-age-exhausted phantom /// contract (review Fix C), returning the number of removed subscriber /// entries. See `HostingManager::drop_phantom_downstream`. pub(crate) fn drop_phantom_downstream(&self, key: &ContractKey) -> usize { self.hosting_manager.drop_phantom_downstream(key) } pub fn should_unsubscribe_upstream(&self, contract: &ContractKey) -> bool { self.hosting_manager.should_unsubscribe_upstream(contract) } /// Check if this node is actively receiving updates for a contract. /// /// Returns true only when we have an active network subscription or local /// client subscriptions. The hosting LRU cache alone is not sufficient, /// since cached state may be stale after subscription expiry. pub fn is_receiving_updates(&self, contract: &ContractKey) -> bool { self.hosting_manager.is_receiving_updates(contract) } /// Get contracts that need subscription renewal. /// /// Returns contracts where: /// - We have an active subscription that will expire soon, OR /// - We have client subscriptions but no active network subscription, OR /// - We have hosted contracts without active subscriptions (THE FIX) pub fn contracts_needing_renewal(&self) -> Vec { self.hosting_manager.contracts_needing_renewal() } // ==================== Client Subscription Management ==================== /// Register a client subscription for a contract (WebSocket client subscribed). /// /// Returns information about the operation for telemetry. pub fn add_client_subscription( &self, instance_id: &ContractInstanceId, client_id: crate::client_events::ClientId, ) -> AddClientSubscriptionResult { // No governance demand is ingested here anymore. A local client // subscription is the strong demand signal (full // LOCAL_DEMAND_WEIGHT = 1.0), but benefit is now a LIVE SNAPSHOT // read fresh each reaper tick from // `HostingManager::local_client_count` (see // `Ring::governance_tick`), not an accumulator fed here. The // snapshot counts the currently-subscribed client set, so an // idempotent re-subscribe cannot inflate benefit and there is no // need to gate on `is_new_for_client` for scoring purposes. self.hosting_manager .add_client_subscription(instance_id, client_id) } /// Remove a client from all its subscriptions (used when client disconnects). /// /// Returns a [`ClientDisconnectResult`] with: /// - `affected_contracts`: all contracts where the client was subscribed (for cleanup) pub fn remove_client_from_all_subscriptions( &self, client_id: crate::client_events::ClientId, ) -> ClientDisconnectResult { self.hosting_manager .remove_client_from_all_subscriptions(client_id) } /// Get all hosted contract keys from the hosting cache. pub fn hosting_contract_keys(&self) -> Vec { self.hosting_manager.hosting_contract_keys() } /// Get the cached state size in bytes for a hosted contract. pub fn hosting_contract_size(&self, key: &ContractKey) -> u64 { self.hosting_manager.hosting_contract_size(key) } /// Get the number of contracts in the hosting cache. /// This is the actual count of contracts this node is caching/hosting. pub fn hosting_contracts_count(&self) -> usize { self.hosting_manager.hosting_contracts_count() } /// The same hosted set as [`Self::hosting_contracts_count`], partitioned by /// WHY each contract is held, with state bytes per bucket. Backs the /// `freenet.node.contracts.hosted{,.bytes}` OTel gauges. See /// [`HostingReason`]. pub fn hosted_by_reason(&self) -> HostingReasonStats { self.hosting_manager.hosted_by_reason() } /// Number of active network subscription leases this node currently holds. /// /// Together with [`hosting_contracts_count`](Self::hosting_contracts_count) /// and [`active_demand_count`](Self::active_demand_count), this is the /// per-node measurement a simulation test asserts on to detect a #3763-style /// subscription storm: a healthy node's subscription count tracks active /// demand, NOT cache size. /// /// Test/sim-only accessor (read by `ControlledSimulationResult`). #[cfg(any(test, feature = "testing"))] pub fn active_subscription_count(&self) -> usize { self.hosting_manager.active_subscription_count() } /// Number of contracts this node has *real demand* for — a local client /// subscription or a downstream subscriber — EXCLUDING cache-only hosting. /// /// Reads the live `InterestManager` (on the `OpManager`) via the Ring's /// back-reference. Returns `None` if the `OpManager` is not attached yet or /// has been torn down (the normal startup / shutdown windows) — distinct /// from `Some(0)` ("attached, genuinely no demand"). This distinction is /// load-bearing for the #3763 no-storm invariant: a no-storm assertion must /// NOT treat a `None` (unmeasurable, detached node) as "demand == 0" and /// falsely conclude there was no storm. See /// [`InterestManager::active_demand_count`](crate::ring::interest::InterestManager::active_demand_count). /// /// Test/sim-only accessor (read by `ControlledSimulationResult`). #[cfg(any(test, feature = "testing"))] pub fn active_demand_count(&self) -> Option { self.upgrade_op_manager() .map(|op_manager| op_manager.interest_manager.active_demand_count()) } /// Number of contracts this node keeps neighbour records for although it /// neither hosts nor uses them and has no local interest in them (#5780). /// Such records keep the contract indexed and advertised in the interest /// heartbeat, so neighbours keep refreshing them; reconciliation removes /// them. `None` if the `OpManager` is not attached (unmeasurable). #[cfg(any(test, feature = "testing"))] pub fn orphan_interest_contract_count(&self) -> Option { self.upgrade_op_manager().map(|op_manager| { op_manager .interest_manager .contracts_with_peer_records() .into_iter() .filter(|key| { !self.is_hosting_contract(key) && !self.contract_in_use(key) && !op_manager.interest_manager.has_local_interest(key) }) .count() }) } /// Number of contracts this node still advertises to co-hosts although it /// neither hosts nor uses them and holds no live lease toward them (#5782). /// Such an advertisement keeps co-hosts sending updates for a copy this /// node no longer holds. A live lease is excluded because the retraction /// deliberately waits for it to lapse. `None` if the `OpManager` is not /// attached (unmeasurable). #[cfg(any(test, feature = "testing"))] pub fn stale_advertisement_count(&self) -> Option { self.upgrade_op_manager().map(|op_manager| { let mut held: HashSet = self .hosting_contract_keys() .iter() .chain(self.get_subscribed_contracts().iter()) .map(|key| *key.id()) .collect(); held.extend(self.hosting_manager.in_use_contract_ids()); op_manager .neighbor_hosting .advertised_contract_keys() .iter() .filter(|key| !held.contains(key.id())) .count() }) } /// Number of *upstream* peers this node has recorded for `contract` — i.e. /// peers it subscribed THROUGH (its parent in the subscription tree). Reads /// the live `InterestManager`'s `is_upstream` edges. /// /// This projects the subscription-tree upstream edge that the topology /// snapshot does NOT carry (it hardcodes `upstream: None`), giving a sim test /// a way to assert a KNOWN `subscriber → upstream` edge — e.g. verify the /// edge exists, crash the upstream, then observe the edge re-form (#4642 /// piece F). NOTE: an upstream edge is only recorded when the subscription /// resolved at a real connected peer (not at the originator/root), so a test /// must place the contract key so the subscribe routes through an /// intermediate hop. Returns `None` if the `OpManager` is not attached /// (unmeasurable) — distinct from `Some(0)` ("attached, no upstream edge"). #[cfg(any(test, feature = "testing"))] pub fn upstream_interest_count(&self, contract: &ContractKey) -> Option { self.upgrade_op_manager().map(|op_manager| { op_manager .interest_manager .get_interested_peers(contract) .into_iter() .filter(|(_, interest)| interest.is_upstream) .count() }) } /// Get subscription state for all contracts (for telemetry). /// /// Returns: (contract, has_client_subscription, is_active_subscription, expires_at) pub fn get_subscription_states(&self) -> Vec<(ContractKey, bool, bool, Option)> { self.hosting_manager.get_subscription_states() } /// Snapshot of every active subscription for the local-peer dashboard. /// Reads directly from the canonical lease map. pub fn dashboard_subscription_snapshot(&self) -> Vec { self.hosting_manager.dashboard_subscription_snapshot() } /// Snapshot of per-contract governance state for the local-peer /// dashboard. Reads directly from the canonical `GovernanceManager` /// state — no mirror, no cache, no derived recomputation. If a /// state appears here it's because the manager computed it from /// real meter samples and real subscription events. pub fn dashboard_governance_snapshot(&self) -> crate::node::network_status::GovernanceSnapshot { use crate::contract::governance as gov; use crate::node::network_status as ns; let now = self.time_source.now(); let mode = match self.governance.mode() { gov::GovernanceMode::Off => ns::GovernanceModeSnapshot::Off, gov::GovernanceMode::DryRun => ns::GovernanceModeSnapshot::DryRun, gov::GovernanceMode::Enforce => ns::GovernanceModeSnapshot::Enforce, }; let map_state = |s: gov::GovernanceState| match s { gov::GovernanceState::Normal => ns::GovernanceStateSnapshot::Normal, gov::GovernanceState::Borderline => ns::GovernanceStateSnapshot::Borderline, gov::GovernanceState::WouldEvict => ns::GovernanceStateSnapshot::WouldEvict, gov::GovernanceState::Evicted => ns::GovernanceStateSnapshot::Evicted, gov::GovernanceState::Banned => ns::GovernanceStateSnapshot::Banned, }; let map_reason = |r: gov::TransitionReason| match r { gov::TransitionReason::FirstSeen => ns::GovernanceTransitionReasonSnapshot::FirstSeen, gov::TransitionReason::BorderlineEntered => { ns::GovernanceTransitionReasonSnapshot::BorderlineEntered } gov::TransitionReason::ThresholdCrossed => { ns::GovernanceTransitionReasonSnapshot::ThresholdCrossed } gov::TransitionReason::Evicted => ns::GovernanceTransitionReasonSnapshot::Evicted, gov::TransitionReason::BanTriggered => { ns::GovernanceTransitionReasonSnapshot::BanTriggered } gov::TransitionReason::Recovered => ns::GovernanceTransitionReasonSnapshot::Recovered, gov::TransitionReason::BanLifted => ns::GovernanceTransitionReasonSnapshot::BanLifted, }; // Only iterate flagged contracts for the dashboard mirror — // the renderer hides Normal anyway. Avoids cloning thousands of // entries per refresh on a busy node. See `iter_flagged_scores`. let contracts: Vec = self .governance .iter_flagged_scores() .into_iter() .map(|(id, score)| { let instance_id = id.to_string(); let instance_id_short = if instance_id.chars().count() > 12 { let trunc: String = instance_id.chars().take(12).collect(); format!("{trunc}...") } else { instance_id.clone() }; let history = score .history .iter() .map(|t| ns::GovernanceTransitionEntry { secs_ago: now.saturating_duration_since(t.at).as_secs(), from: map_state(t.from), to: map_state(t.to), reason: map_reason(t.reason), }) .collect(); ns::ContractGovernanceEntry { instance_id, instance_id_short, state: map_state(score.state), cost_used: score.cost_used, benefit_score: score.benefit_score, log_ratio: score.log_ratio(self.governance.benefit_floor()), age_secs: now.saturating_duration_since(score.first_seen).as_secs(), last_transition_secs_ago: now .saturating_duration_since(score.last_transition) .as_secs(), history, } }) .collect(); let norms = match self.governance.latest_norms() { Some(n) => ns::NetworkNorms { median_log_ratio: n.median_log_ratio, mad: n.mad, threshold: n.threshold, sample_size: n.sample_size, capacity_ceiling_binding: n.capacity_ceiling_binding, skip_reason: n.skip_reason.map(|r| match r { crate::governance::SkipReason::InsufficientSamples => { ns::GovernanceSkipReasonSnapshot::InsufficientSamples } crate::governance::SkipReason::MadCollapsed => { ns::GovernanceSkipReasonSnapshot::MadCollapsed } crate::governance::SkipReason::NoExtractableRatios => { ns::GovernanceSkipReasonSnapshot::NoExtractableRatios } }), }, None => ns::NetworkNorms::default(), }; // state_by_id: map ContractInstanceId → state for the // Subscribed Contracts table's Gov column cross-reference. // Walking iter_flagged_scores() once for the `contracts` // list above gave us the flagged set; for unflagged // contracts the absence from this map means "Normal". let state_by_id: std::collections::HashMap = contracts .iter() .map(|c| (c.instance_id.clone(), c.state)) .collect(); let observed_count = self.governance.len(); let min_samples = self.governance.outlier_min_samples(); let last_tick_at = self.governance.latest_norms().map(|n| n.at); ns::GovernanceSnapshot { mode, contracts, observed_count, min_samples, norms, last_tick_at, state_by_id, } } /// Snapshot of the demand-driven hosting state for the local-peer /// dashboard (#4642). Reads the canonical hosting cache — the /// capability-relative budget plus the per-contract rows. No mirror, no /// cache: the aggregate gauges and per-contract rows come straight from /// the `HostingManager`, so the panel can't drift the way a mirrored /// counter would. /// /// The demoted telemetry-only estimator (`keep_score` / /// `predicted_demand`) is deliberately NOT carried on these rows: eviction /// does not read it, and rendering it implied a ranking it never governed. /// Real eviction ordering is subscriber-primary (`victim_order`); /// `recency_seq` is the one ranking input available here, and is what the /// cache actually sorts these rows by. /// /// Per-contract rows are returned in EVICTION order (next victim first). /// The renderer bounds how many it displays; the full count is /// `contract_count`. pub fn dashboard_hosting_snapshot(&self) -> crate::node::network_status::HostingSnapshot { use crate::node::network_status as ns; let stats = self.hosting_manager.hosting_cache_stats(); // Two-phase: `dashboard_hosting_scores()` returns owned rows with the // `hosting_cache` read lock already dropped, so folding in // `is_eviction_eligible` (which reads only the subscription maps via // `contract_in_use`) holds no cache lock — no re-lock deadlock. let contracts: Vec = self .hosting_manager .dashboard_hosting_scores() .into_iter() .map(|row| { let eviction_eligible = self.hosting_manager.is_eviction_eligible(&row); let key_full = row.key.to_string(); let key_short = if key_full.chars().count() > 12 { let trunc: String = key_full.chars().take(12).collect(); format!("{trunc}...") } else { key_full.clone() }; ns::HostedContractEntry { key_full, key_short, size_bytes: row.size_bytes, read_count: row.read_count, recency_seq: row.recency_seq, eviction_eligible, } }) .collect(); // Aggregate on-disk usage (#4683) + the disk budget the admission gate // checks against (#4702), for the same "measuring…" pre-seed gate the // `hosting_disk_*` telemetry gauges use (see the population above at // the `RouterSnapshot` disk-usage block): `None` until the tracker is // seeded, so an unseeded tracker is distinguishable from genuine zero // usage. let disk = self.hosting_manager.disk_usage_stats(); let disk_state_bytes = disk.map(|d| d.state_bytes); let disk_wasm_bytes = disk.map(|d| d.wasm_bytes); let disk_compile_cache_bytes = disk.map(|d| d.compile_cache_bytes); let disk_total_bytes = disk.map(|d| d.total_bytes); // `disk_budget_bytes` starts at `u64::MAX` until the first 60s // recompute installs a real value (see `HostingManager:: // recompute_effective_budget`); surface that as `None` rather than an // astronomical byte count. let raw_disk_budget = self.hosting_manager.disk_budget_bytes(); let disk_budget_bytes = (raw_disk_budget != u64::MAX).then_some(raw_disk_budget); ns::HostingSnapshot { budget_bytes: stats.budget_bytes, used_bytes: stats.current_bytes, contract_count: stats.contract_count, budget_evictions_total: stats.budget_evictions_total, evictions_of_recently_read_total: stats.evictions_of_recently_read_total, contracts, disk_state_bytes, disk_wasm_bytes, disk_compile_cache_bytes, disk_total_bytes, disk_budget_bytes, resident_overhead_budget_bytes: stats.resident_overhead_budget_bytes, resident_overhead_bytes: stats.resident_overhead_bytes, resident_overhead_evictions_total: stats.resident_overhead_evictions_total, } } /// Snapshot of the contract ban list for the local-peer dashboard /// (#4302). Reads directly from the canonical `contract_ban_list` — /// no mirror, no cache. The count, capacity-rejection counter, and /// per-entry list (key + reason + time remaining) all come from the /// list's own accessors, so the panel can't drift the way a mirrored /// counter would. pub fn dashboard_ban_list_snapshot(&self) -> crate::node::network_status::BanListSnapshot { use crate::node::network_status as ns; use crate::ring::contract_ban_list::BanReason; let map_reason = |r: BanReason| match r { BanReason::AutoMad => ns::BanReasonSnapshot::AutoMad, BanReason::Operator => ns::BanReasonSnapshot::Operator, }; let mut entries: Vec = self .contract_ban_list .snapshot() .into_iter() .map(|e| ns::BanListEntry { instance_id: e.contract.to_string(), reason: map_reason(e.reason), expires_in_secs: e.remaining.as_secs(), }) .collect(); // Stable display order: soonest-to-lift first, then by id so the // dashboard doesn't reshuffle rows between refreshes (DashMap // iteration order is unspecified). entries.sort_by(|a, b| { a.expires_in_secs .cmp(&b.expires_in_secs) .then_with(|| a.instance_id.cmp(&b.instance_id)) }); // Count LIVE entries — derive from the filtered snapshot, not // `ContractBanList::len()`. `len()` includes entries that have // expired but not yet been swept by `cleanup()`/`unban()`; // `snapshot()` filters those out with the same `now < expires_at` // predicate as `is_banned`. Using `len()` here would let the // count tile read "1 contract banned" while the entry list (and // the wire boundary) show zero — an inconsistency in the // expiry-before-sweep window. (Codex review on #4464.) ns::BanListSnapshot { count: entries.len(), capacity_rejected_total: self.contract_ban_list.capacity_rejected_total(), entries, } } /// Record that a state update was observed for `contract`. /// No-op if the contract is not currently subscribed. pub fn record_contract_update(&self, contract: &ContractKey) { self.hosting_manager.record_contract_update(contract) } /// True if `contract` has been flagged as violating a CRDT invariant /// (e.g. non-idempotent `update_state`). Callers on the broadcast path /// must skip emission when this returns true to prevent the broken /// contract from generating a propagation storm. pub fn is_contract_broken(&self, contract: &ContractKey) -> bool { self.broken_invariants.is_broken(contract.id()) } /// Mark `contract` as broken with `kind`. Idempotent. See /// [`broken_invariants`] module docs. pub(crate) fn record_broken_invariant(&self, contract: ContractKey, kind: BrokenInvariant) { self.broken_invariants.record(*contract.id(), kind); } /// Claim a slot for the deterministic identical-input idempotency probe /// (`Executor::probe_identical_input_idempotency`). Bounded to one /// probe per contract per `IDENTITY_PROBE_COOLDOWN`. pub(crate) fn try_claim_identity_probe(&self, contract: &ContractKey) -> bool { self.broken_invariants .try_claim_identity_probe(contract.id()) } /// Wire persistent storage for the broken-invariants tracker. Called /// once at executor wiring time. pub(crate) fn set_broken_invariants_storage( &self, storage: crate::contract::storages::Storage, ) { self.broken_invariants.set_storage(storage); } /// Report a per-contract resource sample to the topology meter. /// /// Lightweight non-blocking write: takes the topology manager's /// write lock briefly to insert into the running-average store. /// Safe to call from executor commit paths — does NOT use channels /// (cf. `.claude/rules/channel-safety.md`); the May 2026 deadlock /// pattern (`#4145`) is not reachable through this surface. /// /// Used by `Executor::commit_state_update` to attribute state-write /// bytes to a contract, and (in follow-up work) by the WASM call /// wrappers for CPU/fuel and the broadcast dispatcher for fanout /// cost. Per-contract attribution lets the shared governance /// outlier-detection module (`crate::governance`) score contracts /// against the network's observed cost-per-benefit distribution. pub(crate) fn report_contract_resource_usage( &self, contract_id: freenet_stdlib::prelude::ContractInstanceId, resource: crate::topology::meter::ResourceType, amount: f64, ) { self.report_contract_resource_usage_batch(contract_id, &[(resource, amount)]); } /// Report several cost axes for ONE contract under a SINGLE topology /// write-lock acquisition. /// /// The broadcast send path (#4903 review MEDIUM) reports the send-time /// WASM CPU AND the actual fan-out payload bytes for each per-peer send; /// batching them into one lock (instead of two separate /// `report_contract_resource_usage` calls) halves the per-send topology /// write-lock traffic on a busy gateway's fan-out. /// /// Uses the injectable `TimeSource` (rather than `Instant::now()` /// directly) so deterministic simulation tests can drive this path /// (`.claude/rules/code-style.md`: "Need current time? → USE: TimeSource"). /// `now` is read INSIDE the write lock (#4903 review L2) so concurrent /// reporters insert in lock-acquisition order and cannot trip the /// `RunningAverage::insert_with_time` monotonicity `debug_assert!`. pub(crate) fn report_contract_resource_usage_batch( &self, contract_id: freenet_stdlib::prelude::ContractInstanceId, samples: &[(crate::topology::meter::ResourceType, f64)], ) { if samples.is_empty() { return; } let source = crate::topology::meter::AttributionSource::Contract(contract_id); { let mut topo = self.connection_manager.topology_manager.write(); // Read `now` under the lock (see rustdoc — L2 monotonicity). let now = self.time_source.now(); for (resource, amount) in samples { topo.report_resource_usage(&source, *resource, *amount, now); } } // Also feed the governance manager — the meter stores rates for // telemetry/the dashboard's bandwidth view, while the governance // manager aggregates cost into the cost/benefit ratio that drives its // ban state. But governance ingests ONLY the axes in // `governance_ingests_axis` (currently just `StateBytesWritten`) — the // three #4861 cost-eviction axes are deliberately NOT fed to it (see // that fn's rustdoc for the #4296 rationale). The meter above still // gets EVERY axis, so cost-eviction is unaffected. // // Per-resource weight: identity (1.0) for now. Future tuning could // weight axes differently, but until we have live data to calibrate // against, unit weights match what the design doc proposed. for (resource, amount) in samples { if !governance_ingests_axis(*resource) { continue; } let weight = resource_weight(*resource); self.governance.ingest_cost(contract_id, *amount * weight); } } // Note: there is intentionally no `ingest_contract_demand` method // here, and demand is no longer pushed on subscribe events at all. // Benefit is a LIVE SNAPSHOT pulled each reaper tick by // `governance_tick` from the hosting manager's standing subscriber // counts (`local_client_count` + `downstream_subscriber_count`). // A push entry point would re-introduce the accumulate-on-event // model this redesign removed (#4296). /// Run one governance reaper tick. Caller (the periodic /// `governance_reaper_loop` task) is expected to feed /// `tick_interval` = wall-time since the previous tick for cost /// decay. Returns the `ReaperTickResult` for the caller to act on /// (emit `EvictContract` events for actionable decisions, log /// dry-run decisions, surface stats on the dashboard). /// /// Before ticking, this builds the LIVE benefit snapshot for every /// contract the governance manager is tracking: for each tracked /// contract id, `LOCAL_DEMAND_WEIGHT × current local-client count + /// FORWARDED_DEMAND_WEIGHT × current (non-expired) downstream /// subscriber count`, read fresh from the hosting manager. This is /// the cost-per-current-beneficiary denominator — a popular /// contract keeps a high benefit because its beneficiaries are /// counted live, not decayed from old subscribe events. pub(crate) fn governance_tick( &self, tick_interval: Duration, ) -> crate::contract::governance::ReaperTickResult { // Single-pass live benefit snapshot: one iteration over // `client_subscriptions` and one over `downstream_subscribers` // (see `HostingManager::beneficiary_counts`), then filter to the // contracts governance is tracking. This replaces the previous // shape that deep-cloned every `ContractScore` (incl. its // history Vec) via `iter_scores()` and then re-scanned the whole // `downstream_subscribers` map per contract — O(N×M) per 60s // tick. `tracked_ids()` returns keys only (no score clone). let tracked = self.governance.tracked_ids(); let mut benefits = self .hosting_manager .beneficiary_counts(LOCAL_DEMAND_WEIGHT, FORWARDED_DEMAND_WEIGHT); // Keep only tracked contracts (a contract with beneficiaries but // no ingested cost has no score and must not enter the map — // preserves the prior per-contract behavior). benefits.retain(|id, _| tracked.contains(id)); self.governance.tick(tick_interval, &benefits) } /// Test-only accessor: current local-client beneficiary count for a /// contract, read from the hosting manager. Used by the governance /// sim e2e tests to assert the live benefit snapshot reflects real /// subscriptions. #[cfg(all(test, feature = "simulation_tests"))] pub(crate) fn hosting_manager_local_client_count( &self, instance_id: &ContractInstanceId, ) -> usize { self.hosting_manager.local_client_count(instance_id) } /// Test-only accessor: current (non-expired) downstream subscriber /// count for a contract, read from the hosting manager. #[cfg(all(test, feature = "simulation_tests"))] pub(crate) fn hosting_manager_downstream_subscriber_count( &self, instance_id: &ContractInstanceId, ) -> usize { self.hosting_manager .downstream_subscriber_count(instance_id) } } /// Per-resource weight for cost aggregation. Defaults to 1.0 for all /// dimensions so total cost = simple sum of meter samples. Hooks here /// for future tuning if any one dimension proves to dominate the /// signal-to-noise ratio. fn resource_weight(resource: crate::topology::meter::ResourceType) -> f64 { use crate::topology::meter::ResourceType::*; match resource { InboundBandwidthBytes | OutboundBandwidthBytes => 1.0, ExecCpuMicros | ExecFuelUnits => 1.0, StateBytesWritten | BroadcastFanoutCost | BroadcastMessagesSent => 1.0, } } /// Whether a cost axis feeds the GOVERNANCE cost/benefit ban decision. /// /// This is deliberately an inclusion allowlist (only `StateBytesWritten` /// today), NOT "every axis" or "everything except X" — and it is exhaustive so /// a new [`ResourceType`](crate::topology::meter::ResourceType) variant must be /// consciously classified here. /// /// WHY governance must NOT ingest the #4861 cost-eviction axes /// (`ExecCpuMicros`, `BroadcastFanoutCost`, `BroadcastMessagesSent`): those are /// per-send / per-update signals that scale with a contract's POPULARITY (a /// heavily-subscribed contract fans out more, so it accrues more send CPU and /// fan-out cost). Governance is a benefit-thresholded BAN system that — unlike /// the cost-eviction sweep — has NO zero-demand candidacy protection (invariant /// 3), so feeding it a popularity-scaled cost reproduces the #4296 "popular /// contract banned for being popular" false positive BY CONSTRUCTION: the /// contract crosses the ban threshold before its live-beneficiary count (the /// denominator that would spare it) can register, and a banned contract then /// drops inbound Subscribe, so its benefit can never catch up. The /// `popular_contract_with_subscribers_not_evicted` sim test is exactly this /// scenario. Storms are the cost-eviction sweep's job (it HAS the zero-demand /// protection); governance is Off in production, so excluding these axes is a /// pure de-risking that restores the cost basis governance was calibrated /// against on `main` (`StateBytesWritten`, the sole production feed via /// `commit_state_write`). /// /// A future governance-On effort that wants to weigh CPU/fan-out MUST first add /// a benefit-aware normalization (e.g. cost-PER-beneficiary), NOT re-add these /// raw axes here — doing so silently re-opens #4296. See #4296 / #4861. fn governance_ingests_axis(resource: crate::topology::meter::ResourceType) -> bool { use crate::topology::meter::ResourceType::*; match resource { // The axis governance was calibrated against on `main`. StateBytesWritten => true, // #4861 cost-eviction axes: popularity-scaled → meter/sweep only. ExecCpuMicros | BroadcastFanoutCost | BroadcastMessagesSent => false, // Peer-attributed bandwidth / fuel: never fed governance (it keys on // per-contract cost), keep them out. InboundBandwidthBytes | OutboundBandwidthBytes | ExecFuelUnits => false, } } /// Bridge the transport layer's per-peer wire-byte counters into the /// topology meter so `TopologyManager::adjust_topology` can make load-aware /// connection decisions in production (#3453). /// /// `TRANSPORT_METRICS.per_peer_snapshot()` exposes *cumulative* sent/received /// byte counts per peer socket address. The topology meter, by contrast, /// accumulates each reported sample into a sliding-window sum and divides by /// the elapsed wall-time (`now − oldest_sample`) to produce a rate. We diff /// the current snapshot against the previous tick's counts (`prev`) to recover /// the bytes transferred *during the interval*, and feed that delta as one /// sample per direction: /// - received bytes → `InboundBandwidthBytes` /// - sent bytes → `OutboundBandwidthBytes` /// /// `interval_start` MUST be the time of the *previous* maintenance tick, not /// the current one. The meter divides the windowed byte sum by /// `query_time − oldest_sample_time` with a 1-second floor /// (`RunningAverage::get_rate_at_time`). `adjust_topology` queries the meter at /// the current tick time immediately after this call, so timestamping the /// interval's whole byte delta at the interval START makes the very first /// sample divide by the real interval length (e.g. ~60s) instead of flooring /// to 1s. Timestamping at the interval END would treat a full interval's bytes /// as if they were transferred in a single second, overstating the rate by up /// to the tick interval (~60×) and triggering spurious connection removals /// under ordinary traffic (#3453 review). /// /// Only deltas with a resolvable `PeerKeyLocation` are reported: an address /// that doesn't (yet) map to a ring peer — e.g. a transient handshake /// connection — is skipped rather than attributed to a phantom source, so we /// don't pollute the meter with samples that can't be matched against the /// node's actual neighbors. /// /// Three things are bounded as peers churn (#3453 review): /// - `prev` is pruned of peers absent from `snapshot`, so it stays bounded /// by the transport metrics table size (`MAX_TRACKED_PEERS`). /// - The topology meter's per-source bandwidth meters AND /// `source_creation_times` are pruned to the set of peers resolved this /// tick via `retain_peer_sources`, so neither grows without bound (and /// `adjust_topology`, which iterates `source_creation_times` every tick, /// stays O(live peers)). /// /// This takes the topology manager's write lock once per reported delta plus /// once for the retain. It performs no channel sends and no `.await`, so it is /// safe to call inline on the `connection_maintenance` tick /// (cf. `.claude/rules/channel-safety.md`). fn feed_peer_bandwidth_to_meter( connection_manager: &ConnectionManager, snapshot: &[(SocketAddr, u64, u64)], prev: &mut HashMap, interval_start: Instant, ) { use crate::topology::meter::{AttributionSource, ResourceType}; // Peers resolved to a ring `PeerKeyLocation` this tick. Used to bound the // meter / source_creation_times to the live connection set below. let mut live_peers: std::collections::HashSet = std::collections::HashSet::new(); for &(addr, cum_sent, cum_recv) in snapshot { // Cumulative counters only ever increase, but use saturating // subtraction so a counter reset (e.g. the per-peer slot was evicted // and re-created since the last tick) can never underflow into a // huge bogus delta. let (prev_sent, prev_recv) = prev.get(&addr).copied().unwrap_or((0, 0)); let sent_delta = cum_sent.saturating_sub(prev_sent); let recv_delta = cum_recv.saturating_sub(prev_recv); // Always record the latest cumulative counts so the next tick diffs // against them, even when this tick produced no usable delta. prev.insert(addr, (cum_sent, cum_recv)); let Some(peer) = connection_manager.get_peer_by_addr(addr) else { // No ring peer for this address (transient/handshake connection, // or just-torn-down). Skip rather than attribute to a phantom. continue; }; // Track every resolvable peer (even with a zero delta this tick) as // live so retain below doesn't evict a currently-connected, currently // idle peer's accumulated samples. live_peers.insert(peer.clone()); if sent_delta == 0 && recv_delta == 0 { continue; } let source = AttributionSource::Peer(peer); let mut topo = connection_manager.topology_manager.write(); if recv_delta > 0 { topo.report_resource_usage( &source, ResourceType::InboundBandwidthBytes, recv_delta as f64, interval_start, ); } if sent_delta > 0 { topo.report_resource_usage( &source, ResourceType::OutboundBandwidthBytes, sent_delta as f64, interval_start, ); } } // Bound the meter / source_creation_times to the live peer set so they // don't accumulate departed peers forever (#3453 review). Non-Peer // (Contract/Delegate) sources are retained by `retain_peer_sources`. connection_manager .topology_manager .write() .retain_peer_sources(&live_peers); // Drop prior-tick entries for peers that vanished from the snapshot so // `prev` stays bounded as peers churn (#3453). After the loop above, // every snapshot address is present in `prev`, so `prev.len() > // snapshot.len()` holds exactly when some prior peer is no longer in the // snapshot — the only case where a prune is needed. if prev.len() > snapshot.len() { let live: std::collections::HashSet = snapshot.iter().map(|&(addr, _, _)| addr).collect(); prev.retain(|addr, _| live.contains(addr)); } } impl Ring { // ==================== Subscription Retry Spam Prevention ==================== /// Check if a subscription request can be made for a contract. /// Returns false if request is already pending or in backoff period. pub fn can_request_subscription(&self, contract: &ContractKey) -> bool { self.hosting_manager.can_request_subscription(contract) } /// Mark a subscription request as in-flight. /// Returns false if already pending. pub fn mark_subscription_pending(&self, contract: ContractKey) -> bool { self.hosting_manager.mark_subscription_pending(contract) } /// Mark a subscription request as completed. /// If success is false, applies exponential backoff. pub fn complete_subscription_request(&self, contract: &ContractKey, success: bool) { self.hosting_manager .complete_subscription_request(contract, success) } // ==================== Hosting Cache Management ==================== /// Touch a contract in the hosting cache (refresh TTL without adding). /// /// Called when a user GET serves a hosted contract from local cache. pub fn touch_hosting(&self, key: &ContractKey) { self.hosting_manager.touch_hosting(key) } /// Mark a contract as accessed by a local client (HTTP/WebSocket). pub fn mark_local_client_access(&self, key: &ContractKey) { self.hosting_manager.mark_local_client_access(key) } // NOTE: `has_local_client_access` (the plain, non-recency variant) had its // only production caller removed by serve-DURING (#4642 R3 piece C): the // originator GET gate now serves on `interest_manager.has_local_interest` // (a fresh in-mesh copy) rather than on whether the LOCAL user had touched // the contract. The flag is still maintained (`mark_local_client_access`) // and read via the recency variant below; the plain read accessor survives // only on `HostingManager` for its unit tests. /// Whether a local client GET/PUT touched this contract within the renewal age /// gate (`SUBSCRIPTION_LEASE_DURATION`) — the read/PUT demand signal the /// reconcile input-builder feeds into `contract_in_use` (invariant 3). pub fn has_recent_local_client_access(&self, key: &ContractKey) -> bool { self.hosting_manager.has_recent_local_client_access(key) } /// The node-wide budget for distinct neighbour-summary bytes (#5781): a /// fixed quarter of the hosting resident budget, enforced when a summary /// is written and trimmed to at each sweep. It does not shrink with the /// rest of the resident charge: at capacity, neighbour summaries (at most /// this quarter) count as hosting cost and can trigger the ordinary /// demand-ordered eviction of the lowest-ranked contracts, rather than /// being wiped to make room. fn neighbour_summary_budget(&self) -> u64 { self.hosting_manager.resident_overhead_budget_bytes() / crate::ring::interest::NEIGHBOUR_SUMMARY_BUDGET_DIVISOR } /// Sweep for expired entries in the hosting cache. /// /// Returns a [`HostingSweepResult`]: the `(ContractKey, write_generation)` /// reclaim pairs plus, for any still-in-use victim shed as a last resort, /// the subscription state torn down (so the caller can sync the /// `InterestManager`). Under subscriber-primary ordering a contract with /// subscribers is evicted LAST (shed only when nothing with fewer /// subscribers is eligible). The generation snapshot is carried through /// `EvictContract` so the deletion-time guard can detect a re-host race. pub fn sweep_expired_hosting(&self) -> crate::ring::hosting::HostingSweepResult { // Neighbour-summary bounds (#5781), BEFORE the hosting sweep re-reads // the interest bytes it charges: each hosted contract's distinct // summaries are capped relative to our own summary (independent of // who sent them), and no single neighbour may make this node hold more // than 1/PEER_SUMMARY_SHARE_DIVISOR of the hosting budget in summaries // only it sent. Neighbours therefore cannot inflate the resident axis // past those bounds to get other contracts evicted. Running it here, // rather than on the interest manager's own timer, means the cache // never charges bytes beyond them. if let Some(op_manager) = self.upgrade_op_manager() { // The node-wide neighbour-summary budget follows the resident // budget, which the sweep task recomputes each tick. op_manager .interest_manager .set_neighbour_summary_budget(self.neighbour_summary_budget()); let share = self.hosting_manager.resident_overhead_budget_bytes() / crate::ring::interest::PEER_SUMMARY_SHARE_DIVISOR; op_manager .interest_manager .enforce_summary_bounds(share, |key| self.is_hosting_contract(key)); } // Cost-aware eviction (#4861): feed the sweep the node's attributed // update-work cost so a zero-subscriber contract dominating CPU / // broadcast capacity is shed even while UNDER the byte budget (the // byte-gated sweep alone never fires for a tiny-state storm contract). let cost_axes = self.hosting_cost_pressure_axes(); self.hosting_manager .sweep_expired_hosting_with_cost(&cost_axes) } /// Build the per-axis attributed-cost snapshots for the hosting sweep's /// cost-pressure trigger (cost-aware eviction, #4861): per-contract /// `ExecCpuMicros`, `BroadcastFanoutCost` (bytes), and /// `BroadcastMessagesSent` (per-send count — the load-bearing storm /// signal) rates plus their node totals, read from the topology meter. /// Sparse samples are amortized over at least /// [`hosting::COST_RATE_MIN_WINDOW`] (a lone burst can't masquerade as a /// sustained storm), a saturated sample buffer reads its true rate, and /// candidacy is pre-filtered to sustained sources — see /// `Meter::contract_cost_rates`. The floors, share threshold, and axis /// assembly live beside the decision in `ring/hosting/cache.rs` /// (`build_cost_axes`, shared with the storm-frequency integration test). /// /// Lock discipline: takes the topology read lock briefly and drops it /// before the caller takes the hosting-cache write lock — the two are /// never held together. fn hosting_cost_pressure_axes(&self) -> Vec { use crate::topology::meter::ResourceType; let now = self.time_source.now(); // Read all three cost axes in ONE pass over the attribution meters // (#4903 review perf) rather than three separate scans under the read // lock. Order here must match the `build_cost_axes` argument order. let axes = [ ResourceType::ExecCpuMicros, ResourceType::BroadcastFanoutCost, ResourceType::BroadcastMessagesSent, ]; let topo = self.connection_manager.topology_manager.read(); let mut rates = topo.contract_cost_rates_multi(&axes, now, hosting::COST_RATE_MIN_WINDOW); drop(topo); let messages = rates.pop().expect("three axis results"); let fanout_bytes = rates.pop().expect("three axis results"); let cpu = rates.pop().expect("three axis results"); hosting::build_cost_axes(cpu, fanout_bytes, messages) } // ==================== Legacy GET Auto-Subscription (delegating to hosting cache) ==================== /// Sweep for expired entries (delegated to hosting cache). /// /// Returns a [`HostingSweepResult`] (see [`Self::sweep_expired_hosting`]). pub fn sweep_expired_get_subscriptions(&self) -> crate::ring::hosting::HostingSweepResult { // Delegate to hosting cache self.sweep_expired_hosting() } // ==================== Connection Pruning ==================== /// Prune a peer connection. /// /// Returns orphaned transactions that need to be retried or failed. /// In the new lease-based subscription model, subscriptions are not tied to specific /// peers, so no subscription pruning is needed when a peer disconnects. pub async fn prune_connection(&self, peer: PeerId) -> PruneConnectionResult { use crate::tracing::DisconnectReason; tracing::debug!(%peer, "Removing connection"); crate::node::network_status::record_peer_disconnected(peer.socket_addr()); let orphaned_transactions = self .live_tx_tracker .prune_transactions_from_peer(peer.socket_addr()); if !orphaned_transactions.is_empty() { tracing::debug!( %peer, orphaned_count = orphaned_transactions.len(), "Connection pruned with orphaned transactions" ); } let min_ready = self.connection_manager.min_ready_connections; let was_ready = self.connection_manager.is_self_ready(); // Capture connection duration before pruning let connection_duration_ms = self .connection_manager .get_connection_duration_ms(peer.socket_addr()); // This case would be when a connection is being open, so peer location hasn't been recorded yet let Some(_loc) = self .connection_manager .prune_alive_connection(peer.socket_addr()) else { return PruneConnectionResult { orphaned_transactions, became_unready: false, }; }; if let Some(event) = NetEventLog::disconnected_with_context( self, &peer, DisconnectReason::Pruned, connection_duration_ms, None, // bytes_sent not tracked yet None, // bytes_received not tracked yet ) { self.event_register .register_events(Either::Left(event)) .await; } let is_ready = self.connection_manager.is_self_ready(); let became_unready = min_ready > 0 && was_ready && !is_ready; PruneConnectionResult { orphaned_transactions, became_unready, } } async fn connection_maintenance( self: Arc, notifier: EventLoopNotificationsSender, live_tx_tracker: LiveTransactionTracker, ) -> anyhow::Result<()> { let is_gateway = self.is_gateway; let shutdown = self.shutdown_token(); tracing::info!(is_gateway, "Connection maintenance task starting"); #[cfg(not(test))] const CHECK_TICK_DURATION: Duration = Duration::from_secs(60); #[cfg(test)] const CHECK_TICK_DURATION: Duration = Duration::from_secs(2); // Faster tick when below min_connections, so initial mesh formation // doesn't bottleneck on the 60-second steady-state interval. #[cfg(not(test))] const FAST_CHECK_TICK_DURATION: Duration = Duration::from_secs(5); #[cfg(test)] const FAST_CHECK_TICK_DURATION: Duration = Duration::from_secs(1); const REGENERATE_DENSITY_MAP_INTERVAL: Duration = Duration::from_secs(60); /// Base number of concurrent connection acquisition attempts (steady-state). const BASE_CONCURRENT_CONNECTIONS: usize = 3; let mut check_interval = tokio::time::interval(CHECK_TICK_DURATION); check_interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); let mut refresh_density_map = tokio::time::interval(REGENERATE_DENSITY_MAP_INTERVAL); refresh_density_map.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); // if the peer is just starting wait a bit before // we even attempt acquiring more connections if sleep_or_shutdown(&shutdown, tokio::time::sleep(Duration::from_secs(2))).await { // Shutdown before we ever did any work — park (see #4292 note below). std::future::pending::<()>().await; } let mut pending_conn_adds = BTreeSet::new(); let mut last_backoff_cleanup = self.time_source.now(); let mut last_health_check = self.time_source.now(); let mut last_peer_cache_save = self.time_source.now(); const HEALTH_CHECK_INTERVAL: Duration = Duration::from_secs(300); // How often to snapshot the peer cache to disk. const PEER_CACHE_SAVE_INTERVAL: Duration = Duration::from_secs(30); const BACKOFF_CLEANUP_INTERVAL: Duration = Duration::from_secs(60); /// Duration of zero ring connections before escalating recovery. /// Uses a shorter threshold initially (before first successful connection) /// so that cold-start failures after OS restart recover faster (#3737). const ISOLATION_ESCALATION_THRESHOLD: Duration = Duration::from_secs(120); const INITIAL_ISOLATION_ESCALATION_THRESHOLD: Duration = Duration::from_secs(30); /// Max time to hold a deferred swap drop before abandoning it. const DEFERRED_SWAP_DROP_TTL: Duration = Duration::from_secs(120); // --- Nearest-neighbor lattice discovery (route-to-self probe) --- // // Mechanism 2: to fill each peer's empty successor+predecessor lattice // slots, periodically route a self-targeted CONNECT toward own_location // (reusing acquire_new + the existing CONNECT path — no wire change). // Every self-initiated CONNECT already pre-populates the visited bloom // with all currently-connected peers (`start_client_connect`), so the // probe routes PAST everything already held and terminates at the nearest // UNCONNECTED peer to own_location; successive probes fill outward on BOTH // ring sides as each newly-connected peer is excluded from the next probe. // The terminus installs the edge via the per-side clause in // `should_accept`. Runs CONCURRENTLY with long-link targeting; long links // are never gated on lattice completion (they supply the weak connectivity // that lets route-to-self converge from a cold start). // // DISCOVERY RUNS UNTIL THE LATTICE IS TIGHT, THEN SLEEPS UNTIL IT CHANGES // (#5814). The probe keeps firing while it makes progress, including when // both sides are filled, so a filled-but-LOOSE edge (holding a farther // neighbor while the exact nearest is an unconnected peer) tightens toward // this peer's TRUE nearest ring neighbor (#4760 review N1: relying on the // passive per-side acceptance clause alone left ~half the last-mile edges // loose). The interval backs off exponentially toward tau_max while // nothing changes, and any change to a per-side nearest distance (a fill, // a tighten, a lost side, or a widening when the nearest drops and a // farther peer remains) resets it to tau0 and probes promptly. // // A probe whose acceptor is NOT a lattice edge (`is_per_side_nearest`) is // a MISS: evidence that the acceptor's side is tight, since the probe // aims at the nearest unconnected peer. Once both sides have missed since // the last change, discovery SLEEPS: it wakes at once on any per-side // distance change, and otherwise re-checks a few times (2h, 4h, 8h; or // 10 min doubling, up to four times, if a closer peer was found but // could not be connected) before sleeping until the lattice changes. It used to // keep re-probing every tau_max forever, and every result it found was // kept (below max_connections the far end usually accepts, and nothing // prunes below max at low bandwidth), so a converged peer gained a // non-lattice link per probe and degree climbed with uptime (live median // 36 -> ~100 over 20h). A miss is evidence, not proof (the walk can stop // short of the true nearest: a failed hole punch, a near-terminus relay, // a recently-failed or rejected peer), which is why it re-checks; the // re-checks are finite because each one keeps a link. // Results are never dropped after connecting: the transport has no close // message, so a dropped link would sit dead on the far end until its // idle timeout. See `LatticeProbeScheduler`. /// How often to probe a gateway for version discovery (#3677). #[cfg(not(test))] const GATEWAY_VERSION_PROBE_INTERVAL: Duration = Duration::from_secs(4 * 3600); #[cfg(test)] const GATEWAY_VERSION_PROBE_INTERVAL: Duration = Duration::from_secs(10); const GATEWAY_PROBE_JITTER_FACTOR: f64 = 0.2; // Deferred swap drops: (addr, queued_at) using time_source for // deterministic simulation support. let mut deferred_swap_drops: Vec<(SocketAddr, tokio::time::Instant)> = Vec::new(); // Nearest-neighbor lattice discovery state (mechanism 2), seeded to fire // on the first eligible tick. Production timing in // `lattice_probe_timing`; shorter under test. #[cfg(not(test))] let probe_timing = lattice_probe_timing::production(); #[cfg(test)] let probe_timing = LatticeProbeTiming { probe: crate::util::backoff::ExponentialBackoff::new( Duration::from_secs(1), Duration::from_secs(8), ), recheck: crate::util::backoff::ExponentialBackoff::new( Duration::from_secs(60), Duration::from_secs(240), ), rechecks: lattice_probe_timing::RECHECKS, retry: crate::util::backoff::ExponentialBackoff::new( Duration::from_secs(20), Duration::from_secs(160), ), retries: lattice_probe_timing::RETRIES, }; let mut lattice_probe = LatticeProbeScheduler::new( self.time_source.now(), self.connection_manager.lattice_probe_misses(), probe_timing, ); let mut zero_connections_since: Option = None; // Track whether we've ever had ring connections. Before the first // successful connection, use a shorter isolation escalation threshold // for faster recovery from cold-start failures (#3737). let mut ever_had_connections = false; // Prior-tick cumulative per-peer wire-byte counts, keyed by the // transport socket address. Each maintenance tick we diff the current // `TRANSPORT_METRICS.per_peer_snapshot()` against this to derive the // bytes transferred *during the tick*, then feed that delta into the // topology meter so `adjust_topology` can make load-aware decisions // (#3453). Bounded: entries for peers absent from the latest snapshot // are pruned each tick, so this never outgrows the transport metrics // table (capped at `MAX_TRACKED_PEERS`). // // Seed with the current cumulative counts so the first maintenance // tick attributes only the bytes transferred *during* that first // interval, not the entire history accumulated before the loop // started (which would over-count any pre-maintenance bootstrap // traffic into a single inflated first sample). let mut prev_peer_bandwidth: HashMap = crate::transport::metrics::TRANSPORT_METRICS .per_peer_snapshot() .into_iter() .map(|(addr, sent, recv)| (addr, (sent, recv))) .collect(); // Time of the previous bandwidth feed (the start of the interval whose // byte delta the next feed reports). Seeded with "now" so the first // feed's delta is divided by the real first-interval length rather than // flooring to the meter's 1s minimum (#3453 review — see // `feed_peer_bandwidth_to_meter`). let mut prev_bandwidth_tick = self.time_source.now(); // Gateway version probe: random initial delay to prevent thundering herd. // The loop guard (`!is_gateway`) ensures gateways never probe themselves. let initial_probe_delay_secs = GlobalRng::random_u64() % GATEWAY_VERSION_PROBE_INTERVAL.as_secs(); let mut next_gateway_probe = self.time_source.now() + Duration::from_secs(initial_probe_delay_secs); // Adaptive fast-tick backoff: increase the fast-tick interval when // connection count stops growing, to avoid hammering the network // with CONNECTs indefinitely (#3578). let mut last_conn_count: usize = 0; let mut no_progress_ticks: u32 = 0; // After this many consecutive no-progress ticks, start doubling // the fast-tick interval. const FAST_TICK_BACKOFF_THRESHOLD: u32 = 6; // 30s at 5s/tick // Maximum fast-tick multiplier: caps backoff at the normal tick rate. // Derived from tick ratio so invariant holds in both prod and test cfg. const MAX_FAST_TICK_MULTIPLIER: u32 = (CHECK_TICK_DURATION.as_secs() / FAST_CHECK_TICK_DURATION.as_secs()) as u32; // Suspend/resume detection. // // We compare two clocks that differ only under OS suspend: // // * `boot_time::Instant` → CLOCK_BOOTTIME on Linux. Advances while // the machine is suspended. // * `std::time::Instant` → CLOCK_MONOTONIC on Linux. Does *not* // advance while the machine is suspended. // // The delta between them across a loop iteration is the amount of // time the machine was suspended. Using just `boot_elapsed` alone // also counts scheduler stalls and virtual-time jumps in simulation // tests (`tokio::time::start_paused(true)`), which previously caused // false positives: the Apr 2026 nightly logs showed // `boot_elapsed_secs=139` on CI runners under load, which tripped the // 30s test-only threshold and fired `DropAllConnections` mid-test, // wiping in-flight GET ops and tripping the `debug_assert!` in // `GetMsg::Request` handling. Comparing against a monotonic baseline // eliminates that: a heavy-CPU stall advances *both* clocks equally // (delta ≈ 0) and doesn't trip the detector, while a real suspend // advances only boot time (delta ≈ suspend duration) and does. let mut last_boot_time = boot_time::Instant::now(); let mut last_mono_time = WallClockInstant::now(); // 2x the check tick is plenty of headroom to tell a real suspend // (minutes) from the sub-tick jitter that a healthy monotonic clock // can still exhibit relative to CLOCK_BOOTTIME. const SUSPEND_DETECTION_THRESHOLD: Duration = CHECK_TICK_DURATION.saturating_mul(2); let mut this_peer = None; 'maintenance: loop { // Stop promptly on node teardown. The OpManager-dropped check below // (`OpManagerState::Detached`) is the other exit, but the shutdown // token fires from `ShutdownTeardown::drop` *before* the OpManager // Arc is dropped, so check it here too so we don't keep running a // full maintenance pass during teardown (#4278). if shutdown.is_cancelled() { tracing::info!( is_gateway, "Shutdown signalled; connection maintenance ending" ); break 'maintenance; } // Update clock tracking at the top of every iteration (including // early-continue paths) so elapsed time doesn't accumulate during // startup. The four clock operations below MUST stay back-to-back // with no intervening `.await` — any suspension point between // them lets the two clocks drift relative to one another for // reasons unrelated to suspend/resume, which would poison the // delta in `classify_suspend_jump` below. Keep them as a block. let boot_elapsed = last_boot_time.elapsed(); let mono_elapsed = last_mono_time.elapsed(); last_boot_time = boot_time::Instant::now(); last_mono_time = WallClockInstant::now(); let suspend_jump = classify_suspend_jump(boot_elapsed, mono_elapsed); // Diagnostic: a small monotonic-ahead skew is normal non-atomic // read jitter; a large one would indicate a virtualization TSC // anomaly or a monotonic clock going backwards. Surface it so // nobody has to rediscover the clock-ordering assumption the // hard way. if let Some(skew) = mono_elapsed.checked_sub(boot_elapsed) && skew > Duration::from_millis(100) { tracing::warn!( mono_ahead_ms = skew.as_millis() as u64, boot_elapsed_ms = boot_elapsed.as_millis() as u64, mono_elapsed_ms = mono_elapsed.as_millis() as u64, "connection_maintenance: monotonic clock is significantly \ ahead of boot clock — possible virtualization TSC anomaly \ or monotonic clock regression; suspend detection is \ saturated to zero for this iteration" ); } let op_manager = match self.op_manager_state() { OpManagerState::Live(op_manager) => op_manager, OpManagerState::NotAttached => { // Still in the startup window before attach_op_manager; // wait for the owner to wire itself up. tokio::time::sleep(Duration::from_millis(100)).await; continue; } OpManagerState::Detached => { // The OpManager was attached and then dropped (node // shutdown). Stop the maintenance work instead of spinning // forever on a Weak that can never upgrade again (#3308). // We do NOT `return Ok(())` here: this task is registered // with `BackgroundTaskMonitor`, whose `wait_for_any_exit` // treats any clean return as a fatal "task exited // unexpectedly". Per the #4292 convention, a monitored // long-lived task whose work has ended must hand a // non-completing future to the monitor — so we break out // of the loop and park below. tracing::info!( is_gateway, "OpManager dropped; connection maintenance loop ending (parking)" ); break 'maintenance; } }; let Some(this_addr) = &this_peer else { let Some(addr) = self.connection_manager.get_own_addr() else { tokio::time::sleep(Duration::from_secs(1)).await; continue; }; this_peer = Some(addr); continue; }; // avoid connecting to the same peer multiple times let mut skip_list = HashSet::new(); skip_list.insert(*this_addr); // Resets both connection (location-based) and gateway (address-based) // backoff state, and clears stale pending reservations for gateways. // Used during isolation recovery to ensure all gateways are retryable // when the node has zero ring connections (#3319). // Wakes any task sleeping on gateway backoff // (`initial_join_procedure`) so it can retry immediately. let reset_all_backoff = || { self.reset_all_connection_backoff(); op_manager.gateway_backoff.lock().clear(); op_manager.gateway_backoff_cleared.notify_waiters(); // Also clear stale pending reservations for gateways — without this, // gateways appear "connected/pending" via has_connection_or_pending() // even after backoff is reset, blocking retry attempts (#3319). let gateway_addrs: Vec<_> = op_manager .configured_gateways .iter() .filter_map(|gw| gw.socket_addr()) .collect(); self.connection_manager .clear_pending_reservations_for(&gateway_addrs); }; // Suspend/resume detection: if boot time advanced much more than // monotonic time, the machine was suspended. Both `suspend_jump` // and the underlying clock reads happen at the top of the loop so // they include early-continue time. if suspend_jump > SUSPEND_DETECTION_THRESHOLD { tracing::warn!( boot_elapsed_secs = boot_elapsed.as_secs(), mono_elapsed_secs = mono_elapsed.as_secs(), suspend_jump_secs = suspend_jump.as_secs(), "Detected suspend/resume (boot-time jump) — dropping all connections and clearing state" ); reset_all_backoff(); // Clear recently-failed addresses since they may be reachable again. self.connection_manager.cleanup_all_failed_addrs(); // Drop all connections (including transient gateway connections). // After suspend, transport sockets are dead but connection entries // persist as zombies — keepalive tasks exit on socket error but // don't trigger connection cleanup. The bootstrap loop then sends // CONNECT messages into dead sockets that never reach the gateway. notifier .notifications_sender .send(Either::Right(crate::message::NodeEvent::DropAllConnections)) .await .map_err(|error| { tracing::debug!(?error, "Failed to send DropAllConnections"); error })?; zero_connections_since = None; } // Periodic cleanup of expired backoff entries if last_backoff_cleanup.elapsed() > BACKOFF_CLEANUP_INTERVAL { self.cleanup_connection_backoff(); last_backoff_cleanup = self.time_source.now(); } // Clean up stale pending reservations to prevent permanent isolation // when CONNECT operations fail to complete cleanly. let stale_removed = self.connection_manager.cleanup_stale_reservations(); if stale_removed > 0 { tracing::warn!( stale_removed, "Cleaned up stale reservations and orphaned location entries" ); } // Capture a single `now` for all TTL/cleanup checks in this tick so that // they all see the same moment rather than drifting across calls. Using // `self.time_source` (rather than `Instant::now()` directly) allows tests // to supply a `SharedMockTimeSource` without pausing the whole tokio runtime. let tick_now = self.time_source.now(); // Expire old NAT traversal failure entries self.connection_manager.cleanup_stale_failed_addrs(); // Expire acceptor reliability entries for peers whose TTL has elapsed self.connection_manager .cleanup_expired_acceptor_stats(tick_now); // Clean up expired transient connections let expired_transients = self.connection_manager.cleanup_expired_transients(); if expired_transients > 0 { tracing::debug!( expired_transients, "Cleaned up expired transient connections" ); } // Periodic peer health check: evict peers with sustained routing failures. if last_health_check.elapsed() > HEALTH_CHECK_INTERVAL { last_health_check = self.time_source.now(); let current_ring = self.connection_manager.connection_count(); let unhealthy = self .connection_manager .peer_health .lock() .unhealthy_peers(self.connection_manager.min_connections, current_ring); for addr in unhealthy { tracing::warn!( peer = %addr, "Evicting unhealthy peer (sustained routing failures)" ); if let Err(e) = notifier .notifications_sender .send(Either::Right(crate::message::NodeEvent::DropConnection( addr, ))) .await { tracing::debug!(error = ?e, "Failed to send DropConnection for unhealthy peer"); } } } // Periodically save peer cache for fast reconnection after restart. if last_peer_cache_save.elapsed() > PEER_CACHE_SAVE_INTERVAL { last_peer_cache_save = self.time_source.now(); if let Some(ref dir) = self.peer_cache_dir { let cache = peer_cache::PeerCache::snapshot_from( &self.connection_manager, self.time_source.as_ref(), ); if !cache.peers.is_empty() { if let Err(e) = cache.save(dir) { tracing::warn!(error = %e, "Failed to save peer cache"); } } } } // Isolation recovery: when we have zero ring connections for too long, // reset all backoff state so we can retry aggressively (#2928). let current_conn_count = self.connection_manager.connection_count(); // Expose to update check task for version mismatch decisions (#3204). crate::transport::set_open_connection_count(current_conn_count); if current_conn_count == 0 { // Use shorter threshold before first successful connection so // cold-start failures (e.g., after OS restart) recover in ~30s // instead of ~120s. Once the node has connected at least once, // use the steady-state threshold. See #3737. let threshold = if ever_had_connections { ISOLATION_ESCALATION_THRESHOLD } else { INITIAL_ISOLATION_ESCALATION_THRESHOLD }; if let Some(since) = zero_connections_since { if since.elapsed() > threshold { tracing::warn!( is_gateway, isolated_for_secs = since.elapsed().as_secs(), threshold_secs = threshold.as_secs(), ever_connected = ever_had_connections, "Node isolated with zero ring connections — resetting all backoff state" ); reset_all_backoff(); zero_connections_since = Some(self.time_source.now()); } } else { zero_connections_since = Some(self.time_source.now()); tracing::warn!( is_gateway, "Zero ring connections detected — starting isolation timer" ); } } else if zero_connections_since.take().is_some() { ever_had_connections = true; tracing::info!( connections = current_conn_count, "Recovered from zero-connection state" ); } // Periodic gateway version probe: initiate a CONNECT to a gateway so the // transport handshake exchanges version info, even when the ring is full (#3677). // // The combined predicate (`!is_gateway && !configured_gateways.is_empty() && // now >= next_probe`) is extracted into `should_probe_gateway` so each branch // is unit-testable. Without the empty-gateways guard merged into the same // condition, the index/modulo on the next line would panic. if should_probe_gateway( is_gateway, !op_manager.configured_gateways.is_empty(), self.time_source.now(), next_gateway_probe, ) { let gw_index = GlobalRng::random_u64() as usize % op_manager.configured_gateways.len(); let gateway = &op_manager.configured_gateways[gw_index]; if let Err(e) = crate::operations::connect::gateway_version_probe(gateway, &op_manager).await { tracing::debug!( error = %e, gateway = %gateway, "Gateway version probe failed, will retry next cycle" ); } let base_secs = GATEWAY_VERSION_PROBE_INTERVAL.as_secs() as f64; let jitter_range = base_secs * GATEWAY_PROBE_JITTER_FACTOR; let uniform_01 = (GlobalRng::random_u64() as f64) / (u64::MAX as f64); let jittered_secs = base_secs + jitter_range * (2.0 * uniform_01 - 1.0); next_gateway_probe = self.time_source.now() + Duration::from_secs_f64(jittered_secs.max(1.0)); } // Gateway bootstrap fallback: at zero connections, acquire_new always // fails (no routing candidates). Connect to gateways directly (#3219). // // Note: initial_join_procedure may also be attempting gateway connections // concurrently. is_not_connected provides best-effort dedup via pending // reservations (should_accept creates one on each join_ring_request), so // the second caller typically sees the gateway as "pending" and skips it. // Occasional duplicate CONNECTs are harmless — the connect state machine // handles them gracefully. if current_conn_count == 0 && !op_manager.configured_gateways.is_empty() { let eligible: Vec<_> = { let backoff = op_manager.gateway_backoff.lock(); self.is_not_connected(op_manager.configured_gateways.iter()) .filter(|gw| { gw.socket_addr() .map(|addr| !backoff.is_in_backoff(addr)) .unwrap_or(false) // skip gateways without addresses }) .cloned() .collect() }; if eligible.is_empty() { tracing::debug!( total_gateways = op_manager.configured_gateways.len(), "Zero connections — all gateways connected/pending or in backoff" ); } else { let attempt_count = eligible.len().min(BASE_CONCURRENT_CONNECTIONS); tracing::info!( eligible = eligible.len(), attempting = attempt_count, "Zero connections — attempting gateway bootstrap" ); for gw in eligible.iter().take(BASE_CONCURRENT_CONNECTIONS) { match crate::operations::connect::join_ring_request(gw, &op_manager, None) .await { Ok(()) => tracing::debug!(gateway = %gw, "Gateway bootstrap initiated"), Err(e) => { tracing::warn!(gateway = %gw, error = %e, "Gateway bootstrap failed") } } } } } // Scale concurrent connection limit based on deficit to min_connections. // During bootstrap (far below min_connections), allow more parallel attempts // to avoid stalling when slots fill with slow/timing-out transactions. let max_concurrent = calculate_max_concurrent_connections( current_conn_count, self.connection_manager.min_connections, ); // Drain pending connections, initiating multiple attempts per tick // (up to max_concurrent) for faster mesh formation. Counts only this // node's own in-flight acquisitions, NOT CONNECTs it is relaying for // others (#4348). let mut active_count = live_tx_tracker.active_acquisition_transaction_count(); // Under-min nodes bypass per-target backoff to escape the straggler // trap (#4348); see `should_respect_location_backoff`. let respect_backoff = should_respect_location_backoff( current_conn_count, self.connection_manager.min_connections, ); // Nearest-neighbor lattice discovery (mechanism 2): inject a // route-to-self probe target (own_location), gated by // `LatticeProbeScheduler`, to fill or tighten each peer's // successor/predecessor lattice slots. Queued into pending_conn_adds // just before it is drained below, so the probe launches this tick // and its results are usually in by the next tick's scheduler // decision (late ones are fenced by the generation tag). No // wire change: the probe is a plain CONNECT toward own_location whose // bloom already excludes held peers, so it aims at the nearest // UNCONNECTED peer, and the terminus installs the edge via the // per-side clause in should_accept. The CONNECT driver classifies // each acceptor (`record_lattice_probe_result`); a non-lattice one is // the scheduler's evidence that a side is tight (see the discovery // comment above). if self.connection_manager.nn_lattice_active() { if let Some(me) = self.connection_manager.get_stored_location() { let outcome = lattice_probe.tick_for( &self.connection_manager, self.time_source.now(), // Jitter against synchronized probe bursts across peers // that bootstrapped together. || { crate::config::GlobalRng::random_range( lattice_probe_timing::JITTER_LOW ..=lattice_probe_timing::JITTER_HIGH, ) }, ); if outcome.improved { self.connection_manager.record_lattice_probe_improvement(); } if let Some(interval) = outcome.fired { self.connection_manager.record_lattice_probe_issued(); pending_conn_adds.insert(me); tracing::debug!( generation = lattice_probe.generation(), interval_secs = interval.as_secs(), "lattice discovery: queued route-to-self probe" ); } } } // The route-to-self lattice probe is queued as own location (see the // discovery block above); tag it, with the scheduler's current // generation, so the CONNECT driver reports a result that is not a // lattice edge as a probe miss for that generation (#5814). Only a // bootstrap target (fewer than 5 connections) can share own location, // and a miss reported for it is harmless. let lattice_probe_target = self .connection_manager .get_stored_location() .filter(|_| self.connection_manager.nn_lattice_active()); while let Some(ideal_location) = pending_conn_adds.pop_first() { if respect_backoff && self.is_in_connection_backoff(ideal_location) { tracing::debug!( target_location = %ideal_location, "Skipping connection attempt - target in backoff" ); // Intentionally not re-queued: adjust_topology will re-request // this location on the next cycle if still below min_connections. continue; } if active_count >= max_concurrent { tracing::debug!( active_connections = active_count, max_concurrent, target_location = %ideal_location, "At max concurrent connections, re-queuing location" ); pending_conn_adds.insert(ideal_location); break; } tracing::debug!( active_connections = active_count, max_concurrent, target_location = %ideal_location, "Attempting to acquire new connection" ); let tx = self .acquire_new( ideal_location, &skip_list, ¬ifier, &live_tx_tracker, &op_manager, if lattice_probe_target == Some(ideal_location) { ClientConnectKind::LatticeProbe { generation: lattice_probe.generation(), } } else { ClientConnectKind::Standard }, ) .await .map_err(|error| { tracing::error!( ?error, "FATAL: Connection maintenance task failed - shutting down" ); error })?; if tx.is_none() { let conns = self.connection_manager.connection_count(); tracing::debug!( connections = conns, target_location = %ideal_location, "acquire_new returned None - likely no peers to query through" ); // Don't record a backoff against the target location here. // acquire_new returning None means we have insufficient routing // candidates locally — the target location itself is fine. // Backing off the target would block future attempts when we // gain more connections and could actually route to it. // adjust_topology will re-request this location on the next tick. } else { active_count += 1; tracing::info!( active_connections = active_count, "Successfully initiated connection acquisition" ); } } let current_connections = self.connection_manager.connection_count(); let pending_connection_targets = pending_conn_adds.len(); let peers = self.connection_manager.get_connections_by_location(); let connections_considered: usize = peers.values().map(|c| c.len()).sum(); let mut neighbor_locations: BTreeMap<_, Vec<_>> = peers .iter() .map(|(loc, conns)| { let conns: Vec<_> = conns .iter() .filter(|conn| { conn.location .socket_addr() .map(|addr| !live_tx_tracker.has_live_connection(addr)) .unwrap_or(true) }) .cloned() .collect(); (*loc, conns) }) .filter(|(_, conns)| !conns.is_empty()) .collect(); if neighbor_locations.is_empty() && connections_considered > 0 { tracing::debug!( current_connections, connections_considered, live_tx_peers = live_tx_tracker.len(), "Neighbor filtering removed all candidates; using all connections" ); neighbor_locations = peers .iter() .map(|(loc, conns)| (*loc, conns.clone())) .filter(|(_, conns)| !conns.is_empty()) .collect(); } if current_connections > self.connection_manager.max_connections { // When over capacity, consider all connections for removal regardless of live_tx filter. neighbor_locations = peers.clone(); } tracing::debug!( current_connections, candidates = peers.len(), live_tx_peers = live_tx_tracker.len(), "Evaluating topology maintenance" ); // Feed the per-peer bandwidth measured since the previous tick into // the topology meter BEFORE adjust_topology reads it. Without this // the meter sees no peer-attributed bandwidth samples in production, // so `calculate_usage_proportion` always reports ~0% usage and // `adjust_topology` perpetually takes the "add connections" branch — // the load-aware add/hold/remove logic never engages (#3453). let bandwidth_tick_now = self.time_source.now(); feed_peer_bandwidth_to_meter( &self.connection_manager, &crate::transport::metrics::TRANSPORT_METRICS.per_peer_snapshot(), &mut prev_peer_bandwidth, // Timestamp this interval's byte delta at the PREVIOUS tick // (interval start) so the meter divides by the real interval // length, not the 1s floor. See feed_peer_bandwidth_to_meter. prev_bandwidth_tick, ); prev_bandwidth_tick = bandwidth_tick_now; let adjustment = self .connection_manager .topology_manager .write() .adjust_topology( &neighbor_locations, &self.connection_manager.own_location().location(), self.time_source.now(), current_connections, ); tracing::debug!( adjustment = ?adjustment, current_connections, is_gateway, pending_adds = pending_connection_targets, "Topology adjustment result" ); match adjustment { TopologyAdjustment::AddConnections(target_locs) => { let allowed = calculate_allowed_connection_additions( current_connections, pending_connection_targets, self.connection_manager.min_connections, self.connection_manager.max_connections, target_locs.len(), ); if allowed == 0 { tracing::debug!( requested = target_locs.len(), current_connections, pending = pending_connection_targets, min_connections = self.connection_manager.min_connections, max_connections = self.connection_manager.max_connections, "Skipping queuing new connection targets – backlog already satisfies capacity constraints" ); } else { let total_pending_after = pending_connection_targets + allowed; tracing::debug!( requested = target_locs.len(), allowed, total_pending_after, "Queuing additional connection targets" ); pending_conn_adds.extend(target_locs.into_iter().take(allowed)); } } TopologyAdjustment::RemoveConnections(should_disconnect_peers) => { for peer in should_disconnect_peers { if let Some(addr) = peer.socket_addr() { notifier .notifications_sender .send(Either::Right(crate::message::NodeEvent::DropConnection( addr, ))) .await .map_err(|error| { tracing::debug!( error = ?error, "Shutting down connection maintenance task" ); error })?; } } } TopologyAdjustment::SwapConnection { remove, add_location, } => { // Connect-first swap: defer the drop until the replacement // connects (connection_count > min_connections), preventing // undershoot that would block future swaps. Deferred drops // expire after DEFERRED_SWAP_DROP_TTL if replacement fails. if let Some(addr) = remove.socket_addr() { tracing::info!( remove_peer = %remove, add_target = %add_location, "Topology swap: queuing replacement connection (drop deferred)" ); pending_conn_adds.insert(add_location); // Deduplicate: don't queue the same peer twice if // consecutive swaps select it before the first executes. if !deferred_swap_drops.iter().any(|(a, _)| *a == addr) { deferred_swap_drops.push((addr, tick_now)); } } else { tracing::warn!( remove_peer = %remove, "Topology swap skipped: peer has no socket address" ); } } TopologyAdjustment::NoChange => {} } // Execute deferred swap drops: only drop as many peers as we // have headroom above min_connections to avoid undershooting. // Expire stale entries whose replacement never connected. { let before_len = deferred_swap_drops.len(); deferred_swap_drops.retain(|(_, queued_at)| { tick_now.saturating_duration_since(*queued_at) < DEFERRED_SWAP_DROP_TTL }); let expired = before_len - deferred_swap_drops.len(); if expired > 0 { tracing::debug!( expired, "Deferred swap drops expired (replacement never connected)" ); } if !deferred_swap_drops.is_empty() { let fresh_count = self.connection_manager.connection_count(); let min_conn = self.connection_manager.min_connections; let n_to_drop = deferred_swap_drops_to_execute( fresh_count, min_conn, deferred_swap_drops.len(), ); for (addr, _) in deferred_swap_drops.drain(..n_to_drop) { tracing::info!( peer = %addr, connections = fresh_count, "Executing deferred swap drop (replacement connected)" ); notifier .notifications_sender .send(Either::Right(crate::message::NodeEvent::DropConnection( addr, ))) .await .map_err(|error| { tracing::debug!( ?error, "Shutting down connection maintenance task" ); error })?; } } } let needs_fast_tick = current_connections < self.connection_manager.min_connections; if needs_fast_tick { // Adaptive backoff: reset on any connection count change // (gain OR loss), otherwise slow down. A loss means topology // changed and we should re-enter aggressive mode. if current_connections != last_conn_count { no_progress_ticks = 0; } else { no_progress_ticks = no_progress_ticks.saturating_add(1); } last_conn_count = current_connections; let multiplier = if no_progress_ticks <= FAST_TICK_BACKOFF_THRESHOLD { 1u32 } else { let excess = no_progress_ticks - FAST_TICK_BACKOFF_THRESHOLD; 2u32.saturating_pow(excess).min(MAX_FAST_TICK_MULTIPLIER) }; // Apply ±20% jitter to prevent synchronized CONNECT bursts // across peers that bootstrapped simultaneously. let jitter: f64 = crate::config::GlobalRng::random_range(0.8..=1.2); let adaptive_duration = FAST_CHECK_TICK_DURATION.mul_f64(multiplier as f64 * jitter); if multiplier > 1 { tracing::debug!( current_connections, min_connections = self.connection_manager.min_connections, no_progress_ticks, tick_interval_secs = adaptive_duration.as_secs(), "Fast-tick backed off due to no connection progress" ); } // Uses sleep() instead of the check_interval so we don't need a // second Interval object. We reset check_interval on transition // back to steady-state to avoid an immediate burst tick. // The shutdown arm makes the (up to multi-second) wait // interruptible (#4278); the loop's top-of-iteration // `is_cancelled()` check then breaks out. crate::deterministic_select! { _ = refresh_density_map.tick() => { self.refresh_density_request_cache(); }, _ = tokio::time::sleep(adaptive_duration) => {}, _ = shutdown.cancelled() => {}, } } else { // Reached min_connections: reset backoff for next time. no_progress_ticks = 0; last_conn_count = current_connections; // Reset the interval on transition from fast to normal tick so // accumulated missed ticks don't cause an immediate burst. check_interval.reset(); // Shutdown arm: interrupt the up-to-60s steady-state tick (#4278). crate::deterministic_select! { _ = refresh_density_map.tick() => { self.refresh_density_request_cache(); }, _ = check_interval.tick() => {}, _ = shutdown.cancelled() => {}, } } } // Intentional teardown parking (#4292): the loop only breaks on node // shutdown — either the OpManager has been dropped (the `Detached` // arm) or the shutdown token fired (#4278) — so there is no more work // to do. This task is registered with `BackgroundTaskMonitor`; parking // on a never-resolving future — rather than returning `Ok(())` — keeps // `wait_for_any_exit` from misreading orderly teardown as a fatal // background-task exit. The unreachable `Ok(())` after it keeps the // `anyhow::Result<()>` signature. std::future::pending::<()>().await; Ok(()) } #[tracing::instrument(level = "debug", skip(self, notifier, live_tx_tracker, op_manager), fields(peer = %self.connection_manager.pub_key))] async fn acquire_new( &self, ideal_location: Location, skip_list: &HashSet, notifier: &EventLoopNotificationsSender, live_tx_tracker: &LiveTransactionTracker, op_manager: &Arc, kind: ClientConnectKind, ) -> anyhow::Result> { let current_connections = self.connection_manager.connection_count(); let is_gateway = self.is_gateway; tracing::debug!( current_connections, is_gateway, target_location = %ideal_location, "acquire_new: attempting to find peer to query" ); let query_target = { // The router read-lock is only needed for the single // `select_k_best_peers_with_telemetry` call; routing_candidates // takes its own connection-manager locks. Acquire late to keep // the critical section short (clippy: `significant_drop_tightening`). let num_connections = self.connection_manager.num_connections(); tracing::debug!( target_location = %ideal_location, num_connections, skip_list_size = skip_list.len(), self_addr = ?self.connection_manager.get_own_addr(), "Looking for peer to route through" ); // CONNECT operations bypass readiness gating — peers need to route // through ANY ring connection to acquire new connections, even if those // connections haven't advertised readiness yet. Without this, peers get // stuck below the readiness threshold: they can't initiate CONNECTs // because routing() filters out all their "not ready" connections, // but they can't become ready without more connections. let candidates = self.connection_manager.routing_candidates( ideal_location, None, skip_list, false, // bypass readiness — this is a CONNECT for connection acquisition ); let selected = if !candidates.is_empty() { let (selected, _) = self.router.read().select_k_best_peers_with_telemetry( candidates.iter(), ideal_location, 1, ); selected.into_iter().next().cloned() } else { None }; if let Some(target) = selected { tracing::debug!( query_target = %target, target_location = %ideal_location, "connection_maintenance selected routing target" ); target } else { tracing::warn!( current_connections, is_gateway, target_location = %ideal_location, "acquire_new: no routing candidates found - cannot find peer to query" ); return Ok(None); } }; let joiner = self.connection_manager.own_location(); tracing::debug!( this_peer = %joiner, query_target_peer = %query_target, target_location = %ideal_location, "Sending connect request via connection_maintenance" ); // Driver computes ttl / target_connections / exclude_addrs internally // (see start_client_connect at op_ctx_task.rs). acquire_new only // needs to allocate the joiner tx, register it with // live_tx_tracker, and spawn the driver task. let _ = notifier; let tx = Transaction::new::(); let gateway_addr = match query_target.socket_addr() { Some(addr) => addr, None => { tracing::warn!( target_location = %ideal_location, "acquire_new: selected query target has no socket address, skipping spawn" ); return Ok(None); } }; // Register tx with the live transaction tracker BEFORE spawning the // driver. Otherwise the driver's first Response could land on the // bypass before the registration completes. (The acquisition-throttle // registration is done inside `start_client_connect`, shared by every // self-initiated CONNECT — ring acquisition, gateway join, and probe.) live_tx_tracker.add_transaction(gateway_addr, tx); let op_manager_spawn = op_manager.clone(); let gateway = query_target; let joiner_for_driver = joiner; GlobalExecutor::spawn(async move { if let Err(err) = crate::operations::connect::op_ctx_task::start_client_connect( tx, gateway, gateway_addr, &op_manager_spawn, joiner_for_driver, ideal_location, None, kind, ) .await { tracing::debug!( %tx, %err, "acquire_new: CONNECT driver completed with error" ); } }); tracing::debug!(tx = %tx, "Connect request sent"); Ok(Some(tx)) } /// Register a topology snapshot for this peer with the global registry. /// /// This should be called periodically during simulation tests to enable /// topology validation. The snapshot captures the current subscription /// state for all contracts. #[cfg(any(test, feature = "testing"))] #[allow(dead_code)] // Used by SimNetwork tests pub fn register_topology_snapshot(&self, network_name: &str) { let Some(peer_addr) = self.connection_manager.get_own_addr() else { return; }; // Use get_stored_location() for consistency with set_upstream distance check. let location = self .connection_manager .get_stored_location() .map(|l| l.as_f64()) .unwrap_or(0.0); let mut snapshot = self .hosting_manager .generate_topology_snapshot(peer_addr, location); snapshot.connection_count = self.connection_manager.connection_count(); snapshot.orphan_interest_contracts = self.orphan_interest_contract_count(); snapshot.stale_advertisements = self.stale_advertisement_count(); snapshot.reconcile_contracts_dropped = self .upgrade_op_manager() .map(|op| op.interest_manager.reconcile_contracts_dropped_total()); topology_registry::register_topology_snapshot(network_name, snapshot); } /// Get a topology snapshot for this peer without registering it. #[cfg(any(test, feature = "testing"))] #[allow(dead_code)] // Used by SimNetwork tests pub fn get_topology_snapshot(&self) -> Option { let peer_addr = self.connection_manager.get_own_addr()?; // Use get_stored_location() for consistency with set_upstream distance check. let location = self .connection_manager .get_stored_location() .map(|l| l.as_f64()) .unwrap_or(0.0); Some( self.hosting_manager .generate_topology_snapshot(peer_addr, location), ) } } /// Calculate the maximum number of concurrent connection acquisition attempts. /// /// During bootstrap (below `min_connections`), scales up from the base to allow /// more parallel attempts, preventing stalls when slots fill with slow transactions. /// Once at or above `min_connections`, returns the base value. fn calculate_max_concurrent_connections( current_connections: usize, min_connections: usize, ) -> usize { /// Base concurrent connection slots (steady-state). const BASE: usize = 3; /// How many missing connections map to one additional concurrent slot. const CONNECTIONS_PER_EXTRA_SLOT: usize = 3; if current_connections >= min_connections { return BASE; } let deficit = min_connections - current_connections; // Cap at half of min_connections, floored at BASE. let bootstrap_cap = (min_connections / 2).max(BASE); (BASE + deficit / CONNECTIONS_PER_EXTRA_SLOT).min(bootstrap_cap) } /// Whether `connection_maintenance` should honor per-target-location connection /// backoff for a node with `current_connections` open connections. /// /// Backoff is honored only once the node has reached `min_connections`. A node /// still below min MUST keep probing even recently-rejected ring regions: its /// under-connection is a local capacity problem, not the target's fault (cf. the /// ring.md "Backoff Target Must Match Failure Cause" rule), and the 30s→600s /// location backoff stamped on a `Rejected` would otherwise trap a poorly /// positioned node permanently below min — the straggler tail in #4348. This /// mirrors the zero-connection re-bootstrap escape, extended to the whole /// under-min regime; at/above min, steady-state backoff is unchanged. /// /// Storm safety: bypassing location backoff below min does NOT let a stuck node /// hammer indefinitely. The maintenance loop's adaptive fast-tick backoff still /// stretches the retry cadence toward the steady ~60s `CHECK_TICK` after /// consecutive no-progress ticks (connection count not changing), per-tick /// attempts remain bounded by [`calculate_max_concurrent_connections`], and the /// separate gateway re-bootstrap backoff (`gateway_backoff`) is untouched. fn should_respect_location_backoff(current_connections: usize, min_connections: usize) -> bool { current_connections >= min_connections } fn calculate_allowed_connection_additions( current_connections: usize, pending_connections: usize, min_connections: usize, max_connections: usize, requested: usize, ) -> usize { if requested == 0 { return 0; } let effective_connections = current_connections.saturating_add(pending_connections); if effective_connections >= max_connections { return 0; } let mut available_capacity = max_connections - effective_connections; if current_connections < min_connections { let deficit_to_min = min_connections.saturating_sub(effective_connections); available_capacity = available_capacity.min(deficit_to_min); } available_capacity.min(requested) } /// Compute how many deferred swap-drops can be executed this tick. /// /// A deferred drop is safe to execute only when a replacement peer has actually /// connected. The guard: `current_connections > min_connections + pending_drops` /// — meaning we have at least one extra connection above what's needed to cover /// all pending drops plus the minimum. Each drop we commit to reduces the /// effective count by one, so we re-evaluate per element. /// /// Drops are sent as async events; `connection_count()` won't reflect them until /// the events are processed. We track the decrement locally instead. fn deferred_swap_drops_to_execute( current_connections: usize, min_connections: usize, pending_drops: usize, ) -> usize { let mut effective_count = current_connections; let mut n_to_drop = 0usize; for _ in 0..pending_drops { let remaining_pending = pending_drops - n_to_drop; if effective_count > min_connections.saturating_add(remaining_pending) { n_to_drop += 1; effective_count = effective_count.saturating_sub(1); } else { break; } } n_to_drop } /// Amount of wall time the machine was suspended across a maintenance loop /// iteration, derived from the two clocks the caller samples at the top of /// each loop pass. /// /// ## Clock pairing /// /// On Linux/Android/openBSD/L4Re, `boot_time::Instant` uses `CLOCK_BOOTTIME` /// (advances during suspend) and `std::time::Instant` uses `CLOCK_MONOTONIC` /// (does not advance during suspend). The delta across a loop iteration is /// the amount of time the machine was suspended: /// /// * Real suspend → boot advances by the suspend duration, monotonic /// stays flat, delta ≈ suspend duration. /// * Scheduler stall / heavy CPU work → both advance equally, delta ≈ 0. /// * Virtual-time jumps under `tokio::time::start_paused(true)` → neither /// wall clock is touched by tokio's virtual clock; both still advance /// by whatever real wall time elapsed during the iteration, delta ≈ 0. /// /// Using just `boot_elapsed` alone against a fixed threshold conflates all /// three cases, which caused spurious `DropAllConnections` under CI load /// (the 2026-04-14 nightly logs showed `boot_elapsed_secs=139` tripping the /// old 30s test-only threshold mid-test). /// /// ## Non-Linux platforms /// /// On macOS / FreeBSD / Emscripten the `boot_time` crate resolves to the /// same clock Rust's `std::time::Instant` uses (`mach_continuous_time` on /// Darwin; `CLOCK_MONOTONIC` fallback on FreeBSD and Emscripten per the /// boot_time-0.1.3 source). Both sides of the subtraction therefore advance /// together and the delta stays near zero — the detector becomes a safe /// no-op on those platforms rather than a false-positive source. Freenet /// production gateways and CI runners are Linux, where the detector is /// load-bearing; the no-op elsewhere is an intentional safe-degradation. /// /// ## Monotonic-ahead diagnostic /// /// `saturating_sub` intentionally clamps to zero if `mono_elapsed > /// boot_elapsed`. Small (O(ns)..O(µs)) negative deltas are normal — the /// two clocks are read back-to-back, not atomically, so scheduling jitter /// between the two reads shows up as skew. A *large* negative delta would /// indicate something pathological (virtualization TSC jump after live /// migration, or a monotonic clock going backwards); the caller logs a /// diagnostic warning in that case instead of silently swallowing the /// signal. #[inline] fn classify_suspend_jump(boot_elapsed: Duration, mono_elapsed: Duration) -> Duration { boot_elapsed.saturating_sub(mono_elapsed) } /// Compute the per-snapshot window delta of a monotonic counter and advance the /// running previous-total in one step (#4440). /// /// `current_total` is this snapshot's lifetime count; `prev_total` is the count /// at the previous snapshot. Returns `current_total - prev_total` (saturating, /// so a counter reset can never underflow to a huge bogus rate) and stores /// `current_total` into `*prev_total` for the next call. Extracted from the /// snapshot loop and unit-tested because a future reorder of the /// compute-then-advance steps would silently zero the broadcast-failure incident /// signal with no test failure otherwise. fn window_delta(current_total: u64, prev_total: &mut u64) -> u64 { let delta = current_total.saturating_sub(*prev_total); *prev_total = current_total; delta } /// Best-effort snapshot of this process's open file-descriptor count and the /// `RLIMIT_NOFILE` soft limit, for the node-health telemetry gauges (#4440). /// /// fd exhaustion (open fds reaching the soft limit → `EMFILE`) drove the /// v0.2.73 gateway crash-loop and was invisible to central telemetry at the /// time. Emitting these on the existing `router_snapshot` cadence makes the /// headroom observable to operators. /// /// Returns `(open_fds, fd_soft_limit)`. Either element is `None` on a platform /// where it can't be read cheaply: the open-fd count is Linux-only (via /// `/proc/self/fd`); the soft limit is any-unix (via `getrlimit`). fn read_fd_usage() -> (Option, Option) { (read_open_fd_count(), read_fd_soft_limit()) } /// Count this process's open file descriptors by enumerating `/proc/self/fd`. /// /// Returns `None` if the process is already at its fd limit — `read_dir` itself /// needs a descriptor and fails with `EMFILE` at the cliff. The companion /// [`read_fd_soft_limit`] still reports in that case, and the 5-minute cadence /// captures the *approach* to the limit, which is the actionable signal. #[cfg(target_os = "linux")] fn read_open_fd_count() -> Option { // `read_dir` itself holds one descriptor open for the duration of the // iteration, so it is included in the entry count; subtract it to report // the steady-state number of descriptors the process actually holds. let count = std::fs::read_dir("/proc/self/fd").ok()?.count() as u64; Some(count.saturating_sub(1)) } #[cfg(not(target_os = "linux"))] fn read_open_fd_count() -> Option { None } /// Read the current `RLIMIT_NOFILE` soft limit (the ceiling that triggers /// `EMFILE`). Mirrors the `getrlimit` read in `bin/freenet.rs::raise_fd_limit`. #[cfg(unix)] fn read_fd_soft_limit() -> Option { let mut limits = libc::rlimit { rlim_cur: 0, rlim_max: 0, }; // SAFETY: `getrlimit` with a valid resource id and an exclusively-borrowed, // initialized out-param is sound; the return code is checked before the // struct is read. if unsafe { libc::getrlimit(libc::RLIMIT_NOFILE, &mut limits) } != 0 { return None; } // `rlim_t` is `u64` on linux-gnu (CI target) but varies across unix targets; // normalize to `u64`. The cast is redundant on linux-gnu, hence the scoped // allow rather than a bare `as u64` that trips clippy. #[allow(clippy::unnecessary_cast)] Some(limits.rlim_cur as u64) } #[cfg(not(unix))] fn read_fd_soft_limit() -> Option { None } #[cfg(test)] mod fd_usage_tests { //! Validates the node-health fd gauges (#4440): the open-fd count and the //! `RLIMIT_NOFILE` soft limit must report sane values on Linux so the //! fd-exhaustion headroom signal that was missing during the v0.2.73 //! incident is trustworthy. #[cfg(target_os = "linux")] #[test] fn read_fd_usage_reports_sane_values_on_linux() { let (open, soft) = super::read_fd_usage(); let open = open.expect("linux reports an open-fd count via /proc/self/fd"); let soft = soft.expect("unix reports the RLIMIT_NOFILE soft limit"); // A running process always holds at least stdin/stdout/stderr. assert!(open > 0, "expected a positive open-fd count, got {open}"); // The kernel enforces open-fds <= soft limit, so this is invariant, not // a heuristic — it pins that we read the two values the right way round. assert!( soft >= open, "open fds ({open}) must not exceed the soft limit ({soft})" ); } } #[cfg(test)] mod window_delta_tests { //! Pins the per-snapshot window-delta of the monotonic broadcast-failure //! counter (#4440): it must return `current - prev` AND advance `prev` to //! `current`, so a reorder of those two steps (which would silently zero //! the incident signal) fails CI. use super::window_delta; #[test] fn broadcast_failure_window_delta_computes_and_advances_prev() { let mut prev: u64 = 10; // First window: 13 - 10 = 3, and prev advances to 13. assert_eq!(window_delta(13, &mut prev), 3, "delta over the window"); assert_eq!(prev, 13, "prev advanced to current"); // No new failures: 13 - 13 = 0, prev stays 13. assert_eq!(window_delta(13, &mut prev), 0, "no failures => zero delta"); assert_eq!(prev, 13); // Next window: 20 - 13 = 7, prev advances to 20. assert_eq!(window_delta(20, &mut prev), 7, "delta resumes from prev"); assert_eq!(prev, 20, "prev advanced to current again"); // Counter reset / restart must saturate to 0, never underflow. assert_eq!( window_delta(5, &mut prev), 0, "reset saturates, no underflow" ); assert_eq!(prev, 5, "prev tracks the reset value"); } } #[cfg(test)] mod resource_meter_bridge_tests { //! Regression tests for #3453: the topology resource meter was never fed //! per-peer bandwidth data in production, so `adjust_topology` always saw //! ~0% usage and perpetually took the "add connections" branch. These //! tests drive `feed_peer_bandwidth_to_meter` — the production bridge from //! `TRANSPORT_METRICS.per_peer_snapshot()` into the meter — and assert the //! meter actually receives the samples, exercising the load-aware //! remove/hold/add decision logic that was previously dead in production. // `super::*` brings in ConnectionManager, TopologyAdjustment, Location, // SocketAddr, HashMap, BTreeMap, Duration, and tokio's Instant. use super::*; use crate::topology::meter::{AttributionSource, ResourceType}; use crate::transport::TransportKeypair; /// Mirror of `topology::constants::SOURCE_RAMP_UP_DURATION` (5 min). That /// constant is `pub(super)` to the topology module, so we restate it here. /// Samples reported older than this window use their measured attributed /// rate rather than the P50 ramp-up extrapolation. const SOURCE_RAMP_UP_DURATION: Duration = Duration::from_secs(5 * 60); /// Add `n` ring connections to `cm`, returning their socket addresses in /// the order added. Each gets a distinct loopback address and location. fn add_connections(cm: &ConnectionManager, n: usize) -> Vec { let mut addrs = Vec::with_capacity(n); for i in 0..n { let addr: SocketAddr = format!("127.0.0.1:{}", 9000 + i).parse().unwrap(); // Spread locations across the ring so each peer is a distinct // neighbor location (adjust_topology keys neighbors by location). let loc = Location::new((i as f64 + 0.5) / n as f64); let keypair = TransportKeypair::new(); assert!(cm.add_connection(loc, addr, keypair.public().clone(), false)); addrs.push(addr); } addrs } /// Drive the bridge once per simulated second, placing the first sample at /// `first_sample` and one sample per second thereafter for `ticks` seconds. /// `per_sec_recv[i]` is the bytes-received delta attributed to `addrs[i]` /// each second; cumulative counters are simulated by accumulating the /// per-second deltas (mirrors production, where each maintenance tick feeds /// one delta per peer). Returns the final `prev` map for inspection. /// /// Callers that want the meter to report a *measured* rate (rather than the /// ramp-up P50 estimate) must arrange two things, both governed by the /// sample timestamps relative to the eventual `adjust_topology` query time /// `q`: /// 1. The first sample (the source's creation time) must be older than /// `SOURCE_RAMP_UP_DURATION` before `q`, or `extrapolated_usage` /// treats the source as ramping up and substitutes the network P50. /// 2. The samples must extend up to just before `q`, because the meter's /// rate is `sum(retained samples) / (q − oldest_retained_sample)`. /// A window that ends far in the past inflates the denominator and /// dilutes the rate toward zero. The meter retains only the most /// recent `RUNNING_AVERAGE_WINDOW` (100) samples. fn drive_bridge( cm: &ConnectionManager, addrs: &[SocketAddr], per_sec_recv: &[u64], first_sample: Instant, ticks: u64, ) -> HashMap { let mut prev: HashMap = HashMap::new(); let mut cumulative: Vec = vec![0; addrs.len()]; for t in 0..ticks { for (i, c) in cumulative.iter_mut().enumerate() { *c += per_sec_recv[i]; } let snapshot: Vec<(SocketAddr, u64, u64)> = addrs .iter() .zip(cumulative.iter()) .map(|(&addr, &recv)| (addr, 0u64, recv)) .collect(); let at = first_sample + Duration::from_secs(t); feed_peer_bandwidth_to_meter(cm, &snapshot, &mut prev, at); } prev } /// The core regression: with the bridge feeding high per-peer bandwidth, /// `adjust_topology` reaches the resource-overload "remove" branch. Before /// the fix nothing fed the meter, so this path was unreachable in /// production and the node would instead always try to ADD connections. #[test_log::test] fn bridge_feeds_meter_so_overloaded_node_removes() { // test_default: min_connections=4, max_connections=10, // max_upstream/downstream = 1_000_000 bytes/sec. let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 6); // `extrapolated_usage` SUMS the per-source rates across all peers, so // 200_000 B/s inbound per peer × 6 peers = 1_200_000 B/s = 1.2× the // 1_000_000 downstream limit → usage proportion 1.2 > 0.9 → remove. let per_sec_recv = vec![200_000u64; 6]; // Choose the query time `now`, then lay samples from // (now − RAMP_UP − 30s) up to (now − 1s): the first sample is older // than the ramp-up window (so the source reports its measured rate), // and the retained 100-sample tail ends ~1s before `now` (so the rate // isn't diluted by a stale denominator). See `drive_bridge` docs. let now = Instant::now(); let first_sample = now - SOURCE_RAMP_UP_DURATION - Duration::from_secs(30); let ticks = SOURCE_RAMP_UP_DURATION.as_secs() + 29; // last sample ≈ now − 1s drive_bridge(&cm, &addrs, &per_sec_recv, first_sample, ticks); let mut neighbor_locations = BTreeMap::new(); let conns = cm.get_connections_by_location(); for (loc, c) in &conns { neighbor_locations.insert(*loc, c.clone()); } let adjustment = cm.topology_manager .write() .adjust_topology(&neighbor_locations, &None, now, 6); assert!( matches!(adjustment, TopologyAdjustment::RemoveConnections(_)), "overloaded node fed via the bridge must remove a connection, got {adjustment:?}" ); } /// Demonstrates the bug: with NO bridge call (the production state before /// this fix), the meter is empty and `adjust_topology` on a node above /// min_connections takes the low-usage "add" branch. This is the symptom /// #3453 describes; the test above proves the bridge cures it. #[test_log::test] fn empty_meter_never_removes() { let cm = ConnectionManager::test_default(); let _addrs = add_connections(&cm, 6); let mut neighbor_locations = BTreeMap::new(); let conns = cm.get_connections_by_location(); for (loc, c) in &conns { neighbor_locations.insert(*loc, c.clone()); } let adjustment = cm.topology_manager.write().adjust_topology( &neighbor_locations, &None, Instant::now(), 6, ); assert!( !matches!(adjustment, TopologyAdjustment::RemoveConnections(_)), "with an unfed meter the remove branch must be unreachable (the #3453 bug), got {adjustment:?}" ); } /// The bridge must attribute samples to the right `ResourceType` and only /// for peers that resolve to a ring `PeerKeyLocation`. #[test_log::test] fn bridge_records_inbound_and_outbound_for_known_peer() { let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 1); let peer = cm .get_peer_by_addr(addrs[0]) .expect("peer is a ring connection"); let at = Instant::now(); let mut prev = HashMap::new(); // First snapshot establishes the baseline; cumulative counts start here. feed_peer_bandwidth_to_meter(&cm, &[(addrs[0], 1_000, 2_000)], &mut prev, at); let source = AttributionSource::Peer(peer); // `attributed_usage_rate` takes `&mut self` (it can refresh the // meter's cached estimate), so a write guard is required. let mut topo = cm.topology_manager.write(); // First tick: delta == cumulative (prev was 0), so both directions get // a sample equal to the cumulative count. let outbound = topo.attributed_usage_rate( &source, &ResourceType::OutboundBandwidthBytes, at + Duration::from_secs(1), ); let inbound = topo.attributed_usage_rate( &source, &ResourceType::InboundBandwidthBytes, at + Duration::from_secs(1), ); drop(topo); assert_eq!(outbound.expect("outbound recorded").per_second(), 1_000.0); assert_eq!(inbound.expect("inbound recorded").per_second(), 2_000.0); } /// An address with no ring connection (e.g. a transient handshake) must be /// skipped, not attributed to a phantom source. `prev` is still updated so /// the next tick diffs correctly if the peer later resolves. #[test_log::test] fn bridge_skips_unresolvable_address() { let cm = ConnectionManager::test_default(); // Six real peers above min_connections so the only thing that could // push usage into the remove branch is bandwidth attribution. let _addrs = add_connections(&cm, 6); // An address that is NOT a ring connection. let unknown: SocketAddr = "127.0.0.1:55555".parse().unwrap(); // Feed enormous traffic for the unknown address across a full window. let now = Instant::now(); let window_end = now - SOURCE_RAMP_UP_DURATION - Duration::from_secs(30); let mut prev = HashMap::new(); let mut cum = 0u64; for t in 0..120u64 { cum += 10_000_000; // far above any limit, IF it were attributed feed_peer_bandwidth_to_meter( &cm, &[(unknown, 0, cum)], &mut prev, window_end - Duration::from_secs(120 - t), ); } let mut neighbor_locations = BTreeMap::new(); for (loc, c) in &cm.get_connections_by_location() { neighbor_locations.insert(*loc, c.clone()); } let adjustment = cm.topology_manager .write() .adjust_topology(&neighbor_locations, &None, now, 6); // The unknown address contributed no meter sample, so usage stays ~0 // and the node must NOT remove — proving the skip. assert!( !matches!(adjustment, TopologyAdjustment::RemoveConnections(_)), "traffic for an unresolvable address must not be attributed, got {adjustment:?}" ); // But prev is still updated so a future tick computes the delta from here. assert_eq!(prev.get(&unknown).copied(), Some((0, cum))); } /// A counter that goes backwards (per-peer slot evicted then re-created, /// resetting cumulative counts) must not underflow into a giant bogus /// delta. saturating_sub clamps it to zero. #[test_log::test] fn bridge_handles_counter_reset_without_underflow() { let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 1); let peer = cm.get_peer_by_addr(addrs[0]).unwrap(); let source = AttributionSource::Peer(peer); let t0 = Instant::now(); let mut prev = HashMap::new(); // Establish a high cumulative count. feed_peer_bandwidth_to_meter(&cm, &[(addrs[0], 0, 1_000_000)], &mut prev, t0); // Counter reset: cumulative drops to a small value. Delta must be 0, // not u64::MAX - something. let t1 = t0 + Duration::from_secs(1); feed_peer_bandwidth_to_meter(&cm, &[(addrs[0], 0, 10)], &mut prev, t1); let rate = cm .topology_manager .write() .attributed_usage_rate( &source, &ResourceType::InboundBandwidthBytes, t1 + Duration::from_secs(1), ) .unwrap() .per_second(); // Only the first tick's 1_000_000 sample is present; the reset tick // contributed a 0 delta (no sample). Sum=1_000_000 over a ~2s window. assert!( rate > 0.0 && rate.is_finite(), "counter reset must clamp to a finite, non-underflowed rate, got {rate}" ); } /// `prev` must stay bounded as peers churn: entries for peers absent from /// the latest snapshot are pruned so it never outgrows the transport /// metrics table. #[test_log::test] fn bridge_prunes_departed_peers_from_prev() { let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 3); let at = Instant::now(); let mut prev = HashMap::new(); let full: Vec<_> = addrs.iter().map(|&a| (a, 1u64, 1u64)).collect(); feed_peer_bandwidth_to_meter(&cm, &full, &mut prev, at); assert_eq!(prev.len(), 3); // Next tick: only one peer remains in the snapshot. The other two must // be pruned from prev. feed_peer_bandwidth_to_meter( &cm, &[(addrs[0], 2, 2)], &mut prev, at + Duration::from_secs(1), ); assert_eq!(prev.len(), 1, "departed peers must be pruned from prev"); assert!(prev.contains_key(&addrs[0])); } /// #3453 review (P1): the per-interval byte delta must be timestamped at /// the interval START so the meter divides it by the real interval length, /// not the 1-second floor in `RunningAverage::get_rate_at_time`. A single /// tick reporting one interval's worth of bytes must yield /// `bytes / interval`, NOT `bytes / 1s` (which would overstate the rate by /// the interval length and trigger spurious removals). #[test_log::test] fn bridge_timestamps_delta_at_interval_start_no_inflation() { let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 1); let peer = cm.get_peer_by_addr(addrs[0]).unwrap(); let source = AttributionSource::Peer(peer); // 60_000 bytes transferred over a 60s interval = 1000 B/s. Production // passes the PREVIOUS tick time as `interval_start`. let interval = Duration::from_secs(60); let interval_start = Instant::now(); let query_time = interval_start + interval; let mut prev = HashMap::new(); feed_peer_bandwidth_to_meter(&cm, &[(addrs[0], 0, 60_000)], &mut prev, interval_start); let rate = cm .topology_manager .write() .attributed_usage_rate(&source, &ResourceType::InboundBandwidthBytes, query_time) .expect("inbound recorded") .per_second(); // Correct: 60_000 / 60s = 1000 B/s. The interval-END bug would give // 60_000 / 1s = 60_000 B/s (60× inflation). assert!( (rate - 1000.0).abs() < 1.0, "interval delta must be divided by the real interval, got {rate} B/s (expected ~1000)" ); assert!( rate < 2000.0, "rate must not be inflated toward the 1s-floor (delta/1s = 60000), got {rate}" ); } /// #3453 review (P2): peers that leave the live set must be pruned from the /// topology meter (and `source_creation_times`), not accumulate forever. /// After a peer departs, a subsequent bridge tick that no longer resolves /// it must drop its meter samples. #[test_log::test] fn bridge_prunes_departed_peer_from_meter() { let cm = ConnectionManager::test_default(); let addrs = add_connections(&cm, 2); let peer0 = cm.get_peer_by_addr(addrs[0]).unwrap(); let peer1 = cm.get_peer_by_addr(addrs[1]).unwrap(); let src0 = AttributionSource::Peer(peer0); let src1 = AttributionSource::Peer(peer1); let t0 = Instant::now(); let mut prev = HashMap::new(); // Tick 1: both peers transfer bytes — both get meter entries. feed_peer_bandwidth_to_meter( &cm, &[(addrs[0], 0, 10_000), (addrs[1], 0, 10_000)], &mut prev, t0, ); let q = t0 + Duration::from_secs(60); { let mut topo = cm.topology_manager.write(); assert!( topo.attributed_usage_rate(&src0, &ResourceType::InboundBandwidthBytes, q) .is_some(), "peer0 should have a meter entry after tick 1" ); assert!( topo.attributed_usage_rate(&src1, &ResourceType::InboundBandwidthBytes, q) .is_some(), "peer1 should have a meter entry after tick 1" ); } // peer1 disconnects from the ring; only peer0 remains a resolvable // connection. The transport snapshot drops the departed peer too. cm.prune_alive_connection(addrs[1]); assert!( cm.get_peer_by_addr(addrs[1]).is_none(), "peer1 must no longer resolve after prune" ); // Tick 2: only peer0 present in the snapshot. peer1 is no longer in the // live set, so its meter samples must be pruned. feed_peer_bandwidth_to_meter(&cm, &[(addrs[0], 0, 20_000)], &mut prev, t0); { let mut topo = cm.topology_manager.write(); assert!( topo.attributed_usage_rate(&src0, &ResourceType::InboundBandwidthBytes, q) .is_some(), "live peer0 must keep its meter entry" ); assert!( topo.attributed_usage_rate(&src1, &ResourceType::InboundBandwidthBytes, q) .is_none(), "departed peer1 must be pruned from the meter" ); } } } #[cfg(test)] mod k_closest_source_tests { //! Source-scrape pin tests for `Ring::k_closest_potentially_hosting`. //! //! Constructing a real `Ring` for behavioral tests requires significant //! scaffolding (NodeConfig, EventLoopNotificationsSender, NetEventRegister, //! BackgroundTaskMonitor) that does not currently exist in the test suite. //! These tests instead scrape the production source to ensure load-bearing //! filter invariants stay wired in — a future refactor that silently drops //! the transient filter (issue #4222) or the readiness fallback should fail //! CI rather than silently regress production routing behavior. //! //! Behavioral coverage exists transitively via the simulation_integration //! tests; a dedicated integration test for the transient filter is filed //! as a follow-up to the Ring test-scaffolding work. fn production_source() -> &'static str { const FULL: &str = include_str!("ring.rs"); // ring.rs has inline `#[cfg(test)]` annotations on individual const // declarations inside `connection_maintenance` (lines ~1894–1940), so // a plain `find("#[cfg(test)]")` would cut off the file mid-impl and // miss the function we want to scrape. Anchor on the first *top-level* // test module declaration instead. let cutoff = FULL .find("\n#[cfg(test)]\nmod ") .expect("ring.rs must have a top-level #[cfg(test)] mod section"); &FULL[..cutoff] } fn extract_fn_body<'a>(source: &'a str, signature_prefix: &str) -> &'a str { let start = source .find(signature_prefix) .unwrap_or_else(|| panic!("could not find {signature_prefix}")); let brace = source[start..].find('{').expect("fn sig must have body"); let body_start = start + brace + 1; let bytes = source.as_bytes(); let mut depth: i32 = 1; let mut i = body_start; while i < bytes.len() { match bytes[i] { b'{' => depth += 1, b'}' => { depth -= 1; if depth == 0 { return &source[body_start..i]; } } _ => {} } i += 1; } panic!("unbalanced braces while extracting {signature_prefix}"); } /// Issue #4222: `k_closest_potentially_hosting` must skip transient peers /// so GET/SUBSCRIBE doesn't route through about-to-be-dropped connections. /// The PUT/UPDATE path already filters them via `routing_candidates`; this /// pin makes sure the GET path stays in sync. #[test] fn k_closest_potentially_hosting_filters_transient_peers() { let src = production_source(); let body = extract_fn_body(src, "pub fn k_closest_potentially_hosting("); assert!( body.contains("is_transient(addr)"), "k_closest_potentially_hosting must call is_transient(addr) on each \ candidate connection. If the filter was deliberately removed, also \ update .claude/rules/ring.md (which documents the filter rule) and \ this test. Issue #4222 / #3570." ); assert!( body.contains("skipped_transient"), "k_closest_potentially_hosting must surface a skipped_transient \ counter so operators can see when the filter is removing peers." ); } /// The existing readiness fallback must remain — without it, a node whose /// peers haven't yet sent ReadyState would fail every GET with EmptyRing. /// Pinning it here so a future refactor of the filter chain can't silently /// drop the fallback. #[test] fn k_closest_potentially_hosting_preserves_not_ready_fallback() { let src = production_source(); let body = extract_fn_body(src, "pub fn k_closest_potentially_hosting("); assert!( body.contains("not_ready_fallback"), "k_closest_potentially_hosting must keep the not-yet-ready peer \ fallback so cold-start nodes don't fail every GET with EmptyRing." ); } /// #4440: `is_subscription_root`'s routability mapping must stay in sync with /// `k_closest_potentially_hosting`'s eligibility asymmetry, because the /// renewal short-circuit only suppresses a wire renewal when no routable /// neighbor is closer. The two filters differ: /// * transient peers are excluded UNCONDITIONALLY by k_closest → the root /// check must call `is_transient(addr)` and treat transient as /// non-routable, else a node whose only closer neighbor is transient /// fails to recognise itself as the terminus and storms. /// * not-ready peers are k_closest's *fallback* when no ready candidate /// exists → the root check must NOT exclude them (treat as routable), /// else it wrongly classifies a node as root in cold-start / low-degree /// topologies and suppresses a renewal that would in fact route to the /// closer (not-ready) peer. /// /// This pin fails the build if the mapping silently drifts from that intent. #[test] fn is_subscription_root_routability_matches_k_closest_eligibility() { let src = production_source(); let body = extract_fn_body( src, "fn is_subscription_root(&self, contract_key: &ContractKey) -> bool {", ); assert!( body.contains("is_transient(addr)"), "is_subscription_root must call is_transient(addr) when deciding whether a \ closer neighbor is routable — k_closest excludes transient peers \ unconditionally, so a transient closer neighbor must NOT keep this node \ from being the terminus (#4440)." ); // The not-ready filter (`is_peer_ready`) must NOT appear in the routability // mapping: k_closest falls back to not-ready peers, so they remain valid // route targets and the root check must treat them as routable. Anchoring on // the absence of `is_peer_ready` guards against a future edit that // "symmetrises" the two filters and reintroduces the false-positive-root bug. assert!( !body.contains("is_peer_ready"), "is_subscription_root must NOT exclude not-ready peers (do not call \ is_peer_ready in the routability mapping): k_closest falls back to \ not-ready peers, so a not-ready closer neighbor is still a valid route \ target and must keep this node from short-circuiting its renewal (#4440)." ); } /// #5780: the periodic hosting sweep must run the interest-record /// reconciliation, or evicted contracts keep their neighbours' records and /// stay advertised. Requires the call on a code line (not a comment), so a /// commented-out call fails this pin. #[test] fn sweep_reconciles_interest_records_with_the_hosted_set() { let src = production_source(); let body = extract_fn_body( src, "async fn sweep_get_subscription_cache(ring: Arc, interval_duration: Duration) {", ); // Code only, whitespace removed: layout-proof, and a commented-out // line cannot satisfy it. let code: String = body .lines() .filter(|line| !line.trim_start().starts_with("//")) .flat_map(|line| line.chars()) .filter(|c| !c.is_whitespace()) .collect(); for needle in [ // the call, with the hosting facts in the right order concat!( "interest_manager.reconcile_with_hosting(", "&op_manager.neighbor_hosting.advertised_contract_keys(),", "|key|ring.is_hosting_contract(key),|key|ring.contract_in_use(key),", "|key|ring.is_subscribed(key),)" ), // every aged, unhosted, unused, lease-free contract has any // standing advertisement retracted, every pass concat!( "forkeyin&outcome.advertisements_to_retract{crate::operations::", "retract_advertisement_for_evicted_contract(op_manager,key);}" ), // neighbours are told only when interest actually ended, and still // has not come back concat!( ".filter(|key|!op_manager.interest_manager.has_local_interest(key))", ".collect();if!interest_lost.is_empty(){" ), concat!( "crate::operations::broadcast_change_interests(op_manager,", "Vec::new(),interest_lost,)" ), // the post-eviction unregister skips a re-hosted contract concat!( "if!ring.is_hosting_contract(&key)&&op_manager", ".interest_manager.unregister_local_hosting(&key)" ), ] { assert!( code.contains(needle), "sweep_get_subscription_cache must contain `{needle}` (#5780)" ); } // The retraction runs before the interest broadcast for the same pass. let retract = code .find("forkeyin&outcome.advertisements_to_retract{") .expect("retraction loop"); let broadcast = code .find("Vec::new(),interest_lost,)") .expect("interest broadcast"); assert!( retract < broadcast, "retract before broadcasting lost interest" ); } /// #5647: the periodic hosting sweep must re-read the memory limit and /// recompute the resident budget every tick, or a cgroup limit changed at /// runtime is never picked up and the budget stays at its startup value. /// Code only, whitespace removed, so a reflow cannot break it and a /// commented-out call cannot satisfy it. #[test] fn sweep_recomputes_the_resident_budget_from_the_memory_limit() { let body = extract_fn_body( production_source(), "async fn sweep_get_subscription_cache(ring: Arc, interval_duration: Duration) {", ); // Only the loop body runs every tick; a recompute before the loop // would run once at startup. let (_, loop_body) = body .split_once("loop {") .expect("sweep_get_subscription_cache must have its tick loop"); let code: String = loop_body .lines() .filter(|line| !line.trim_start().starts_with("//")) .flat_map(|line| line.chars()) .filter(|c| !c.is_whitespace()) .collect(); let needle = concat!( "lettotal_ram=crate::ring::hosting::total_ram_or_fallback(", "crate::wasm_runtime::read_total_ram_bytes(),);", "ring.hosting_manager.recompute_resident_overhead_budget(total_ram);" ); let recompute_at = code.find(needle).unwrap_or_else(|| { panic!( "sweep_get_subscription_cache must read the memory limit and recompute \ the resident budget each tick (#5647)" ) }); let sweep_at = code .find("ring.sweep_expired_get_subscriptions()") .expect("the tick must run the hosting sweep"); assert!( recompute_at < sweep_at, "the budget must be recomputed before the sweep uses it" ); } /// PR #4734 Fix 1: the periodic hosting sweep must retract the local hosting /// advertisement for every contract it evicts, exactly like the GET/PUT /// host-formation paths (`cache_contract_locally` / the PUT relay store). An /// evicted contract that keeps `LocalInterest.hosting = true` and does not /// retract its neighbor advertisement is a stale-interest leak — now widened /// because subscribed contracts are sweep-evictable under invariant 3. /// /// The GET/PUT pins (`cache_contract_locally_syncs_interest_on_subscribed_eviction`) /// only cover `remove_evicted_in_use`; this pins the sweep's matching /// `unregister_local_hosting` + `broadcast_change_interests` so a future refactor /// can't silently drop the retraction from the maintenance path. #[test] fn sweep_retracts_hosting_interest_on_eviction() { let src = production_source(); let body = extract_fn_body( src, "async fn sweep_get_subscription_cache(ring: Arc, interval_duration: Duration) {", ); // Teardown of a still-in-use victim (subscribed contract shed as a last // resort) must sync the InterestManager, matching GET/PUT. assert!( body.contains("remove_evicted_in_use"), "sweep must call remove_evicted_in_use to sync the InterestManager for \ a subscribed contract it sheds (mirrors GET/PUT)." ); // Every evicted contract must clear its local hosting flag … assert!( body.contains("unregister_local_hosting"), "sweep must call unregister_local_hosting for each evicted contract so \ the evicted contract does not keep LocalInterest.hosting = true \ (PR #4734 Fix 1 — mirrors GET/PUT host-formation paths)." ); // … and retract the neighbor advertisement for those that lost all interest. assert!( body.contains("broadcast_change_interests"), "sweep must call broadcast_change_interests to retract neighbor \ advertisements for evicted contracts (PR #4734 Fix 1)." ); // Ordering: the remove_evicted_in_use CALL must run BEFORE the // unregister_local_hosting CALL so the latter observes zeroed subscriber // counts and reports full interest loss (→ retraction), matching the // GET/PUT comment discipline. Anchor on the call expressions (`.method(`), // not the bare identifiers, so a prose mention in a leading comment (e.g. // "Run BEFORE the `unregister_local_hosting` loop") can't fool the check. let teardown_pos = body .find("interest_manager.remove_evicted_in_use(") .expect("remove_evicted_in_use call present"); let unregister_pos = body .find("interest_manager.unregister_local_hosting(") .expect("unregister_local_hosting call present"); assert!( teardown_pos < unregister_pos, "remove_evicted_in_use must precede unregister_local_hosting in the sweep \ so interest loss is reported against the already-zeroed subscriber counts." ); } /// Source-scrape pin (keystone sub-task 3 "the flip", #4642): the RENEWAL /// site is FLIPPED — it no longer records a shadow, it DRIVES. The renewal /// loop must gate its real renewal spawn on the reconcile controller via /// `OpManager::reconcile_wants_renewal`, and the record-only shadow helper /// (`record_renewal_shadow`) must be gone. A regression that re-introduces the /// record-only helper, or drops the drive gate, trips this guard. #[test] fn renewal_site_is_driven_not_shadowed() { let src = production_source(); // The record-only renewal shadow helper is removed by the flip. assert!( !src.contains("fn record_renewal_shadow("), "record_renewal_shadow must be REMOVED — the renewal site now drives, \ it does not record a shadow" ); assert!( !src.contains("fn renewal_shadow_actual("), "renewal_shadow_actual must be REMOVED — the renewal site now drives" ); // The renewal loop drives the controller's interest gate before spawning. let loop_body = extract_fn_body(src, "async fn recover_orphaned_subscriptions("); assert!( loop_body.contains("reconcile_wants_renewal"), "the renewal loop must gate its spawn on the reconcile controller \ (OpManager::reconcile_wants_renewal) — the flip" ); assert!( loop_body.contains("mark_subscription_pending"), "the renewal loop still drives the real renewal via mark_subscription_pending" ); // The gate must be consulted BEFORE the pending-mark/spawn (interest-gated // renewal), not after the work is already scheduled. let gate = loop_body .find("reconcile_wants_renewal") .expect("gate present"); let spawn = loop_body .find("mark_subscription_pending(contract)") .expect("spawn present"); assert!( gate < spawn, "the reconcile interest gate must run BEFORE mark_subscription_pending \ so a torn-down contract is never renewed" ); } /// Single-source-of-truth pin (#4642 piece F): the periodic renewal loop and /// the event-driven connection-drop PROMPT re-root spawn through the SAME /// helper (`spawn_renewal_subscribe_task`), so the storm-safety scaffolding /// (jitter, recovery guard, outer-cancel deadline) cannot drift between the two /// callers — the "manually-mirrored side effect" bug class /// (`bug-prevention-patterns.md`). The re-root caller (`spawn_prompt_reroots`) /// lives in `op_state_manager.rs` and is pinned from that side by /// `connection_drop_re_root_is_driven`; this end pins the renewal loop. #[test] fn renewal_and_reroot_share_one_spawn_path() { let src = production_source(); // The shared helper exists and owns the actual renewal wire action. assert!( src.contains("fn spawn_renewal_subscribe_task("), "the shared renewal/re-root spawn helper must exist" ); assert!( src.contains("run_renewal_subscribe("), "the shared spawn helper must drive run_renewal_subscribe" ); // The renewal loop delegates to the shared helper rather than inlining the // spawn (which is what let it drift from the re-root path before extraction). let loop_body = extract_fn_body(src, "async fn recover_orphaned_subscriptions("); assert!( loop_body.contains("spawn_renewal_subscribe_task"), "the renewal loop must delegate its spawn to spawn_renewal_subscribe_task, \ not inline it — otherwise the re-root path silently rots against it" ); // ...and does NOT re-inline the run_renewal_subscribe call directly. assert!( !loop_body.contains("run_renewal_subscribe("), "the renewal loop must NOT inline run_renewal_subscribe — it belongs to \ the shared helper so both callers stay in lockstep" ); } /// Regression pin for the per-tick renewal-gate EVALUATION budget (Codex P2, /// #4725). The interest gate `OpManager::reconcile_wants_renewal` builds a /// fresh `ReconcileInputs` snapshot (a redb `get_state_size` read + an /// `is_subscription_root` neighbor scan) on EVERY call. The pre-fix loop /// bounded only SPAWNED renewals via `attempted >= batch_limit`, but a /// gate-suppressed candidate `continue`s WITHOUT advancing `attempted`, so on /// a high-hosting peer (every peer, now that every-hop placement is live) one /// ~30s maintenance tick could evaluate the gate for the whole renewal set — /// hundreds-to-thousands of snapshots instead of the intended ~`batch_limit`. /// This pins that a DISTINCT evaluation budget (`evaluated`) breaks the loop /// BEFORE the expensive gate runs, and is advanced on every gate evaluation /// (not only on spawn, which is what `attempted` counts). A refactor that /// drops the budget or reverts to an `attempted`-only bound fails CI here. #[test] fn renewal_gate_evaluations_bounded_before_gate() { let src = production_source(); let body = extract_fn_body(src, "async fn recover_orphaned_subscriptions("); let budget_break = body.find("if evaluated >= batch_limit").expect( "the renewal loop must bound the number of interest-gate evaluations \ per tick with `if evaluated >= batch_limit { break }` (Codex P2, \ #4725) — otherwise a gate-suppressed candidate never advances \ `attempted` and the loop builds an unbounded number of \ ReconcileInputs snapshots per tick", ); let gate = body .find("reconcile_wants_renewal(&contract)") .expect("renewal loop must consult the reconcile interest gate"); assert!( budget_break < gate, "the evaluation-budget break (offset {budget_break}) must run BEFORE \ the expensive reconcile_wants_renewal gate (offset {gate}) so the \ per-snapshot cost is bounded by batch_limit" ); // The budget must advance on every gate evaluation (between the break and // the gate), which is what distinguishes it from the `attempted` spawn cap. let increment = body.find("evaluated += 1").expect( "the evaluation budget `evaluated` must be incremented per gate \ evaluation, not only when a renewal is spawned", ); assert!( budget_break < increment && increment < gate, "`evaluated += 1` (offset {increment}) must sit between the budget \ break (offset {budget_break}) and the gate (offset {gate}) so every \ gate evaluation is counted toward the per-tick bound" ); } /// Behavioral regression for the per-tick evaluation budget (Codex P2, /// #4725). Models the renewal loop's exact two-counter structure and asserts /// the number of expensive interest-gate evaluations stays bounded by /// `batch_limit` regardless of how many candidates the gate suppresses. The /// pre-fix `attempted`-only bound (encoded in `simulate_attempted_only`) /// evaluated the WHOLE candidate set when the gate suppressed everything — /// this is the unbounded-per-heartbeat-read cost the fix removes. #[test] fn renewal_gate_evaluation_count_is_bounded_when_suppressed() { // Faithful model of the loop's per-tick counting: `attempted` is the // spawn cap (advanced only on a spawn) and `evaluated` is the evaluation // budget (advanced on every gate call). Returns (evaluations, spawns). fn simulate( candidates: usize, batch_limit: usize, gate: impl Fn(usize) -> bool, ) -> (usize, usize) { let mut attempted = 0usize; let mut evaluated = 0usize; for i in 0..candidates { if attempted >= batch_limit { break; // spawn cap } if evaluated >= batch_limit { break; // evaluation budget (the fix) } evaluated += 1; if !gate(i) { continue; // suppressed: does NOT advance `attempted` } attempted += 1; } (evaluated, attempted) } // The pre-fix model: only the `attempted` spawn cap bounds the loop. fn simulate_attempted_only( candidates: usize, batch_limit: usize, gate: impl Fn(usize) -> bool, ) -> usize { let mut attempted = 0usize; let mut evaluated = 0usize; for i in 0..candidates { if attempted >= batch_limit { break; } evaluated += 1; if !gate(i) { continue; } attempted += 1; } evaluated } let batch_limit = 10usize; let many = 5_000usize; // Pathological case: the gate suppresses every candidate. Evaluations // must still be capped at `batch_limit`, and nothing spawns. let (evaluated, attempted) = simulate(many, batch_limit, |_| false); assert!( evaluated <= batch_limit, "gate evaluations ({evaluated}) must be bounded by batch_limit \ ({batch_limit}) even when every candidate is suppressed" ); assert_eq!( attempted, 0, "no renewals spawn when every candidate is suppressed" ); // Demonstrates the bug the budget fixes: the old attempted-only bound // evaluated the WHOLE set when the gate suppressed everything. let unbounded = simulate_attempted_only(many, batch_limit, |_| false); assert_eq!( unbounded, many, "sanity: without the evaluation budget an all-suppressing gate builds \ a snapshot for every candidate — the unbounded cost the fix removes" ); // Happy path: the gate wants every candidate. The evaluation budget must // NOT reduce throughput below the spawn cap. let (evaluated, attempted) = simulate(many, batch_limit, |_| true); assert_eq!( attempted, batch_limit, "the spawn cap is still reached in the happy path" ); assert!( evaluated <= batch_limit, "evaluations never exceed batch_limit" ); } /// Hardening pin (keystone step-2 completion, #4642): the /// network_status → router_snapshot MIRROR SEAM. The export block hand-copies /// each per-site counter into a matching `RouterSnapshotInfo` field; a field /// swap (e.g. feeding `reconcile_shadow_collapse_comparisons` from /// `shadow.renewal`) would silently emit the wrong per-site value to the /// collector with no compile error. This asserts every one of the 24 export /// assignments reads from `shadow..` for its OWN site, so a swap /// fails CI here. Whitespace-normalized so rustfmt line-wrapping is irrelevant. #[test] fn reconcile_shadow_export_maps_each_field_to_its_own_site() { let src = production_source(); let block = extract_fn_body( src, "if let Some(shadow) = crate::node::network_status::reconcile_shadow_counts()", ); // Normalize all runs of whitespace to single spaces. let norm = block.split_whitespace().collect::>().join(" "); let full: &[&str] = &[ "comparisons", "divergences", "subscribe_diffs", "renew_diffs", "unsubscribe_diffs", "collapse_diffs", "announce_diffs", "retract_diffs", "reroot_search_diffs", ]; let edge: &[&str] = &["comparisons", "divergences"]; let sites: [(&str, &[&str]); 5] = [ ("collapse", full), ("renewal", full), ("inbound_unsubscribe", edge), ("connection_drop", edge), ("host_formation", edge), ]; let mut checked = 0usize; for (site, fields) in sites { for field in fields { let expected = format!( "snapshot.reconcile_shadow_{site}_{field} = Some(shadow.{site}.{field});" ); assert!( norm.contains(&expected), "mirror-seam: export must contain `{expected}` — a field-swap here \ silently emits the wrong per-site value to the collector" ); checked += 1; } } assert_eq!(checked, 24, "expected exactly 24 export assignments"); } /// The routing-dataset peer task is registered with the background task /// monitor, and ANY monitored task exiting ends the node /// (`p2p_impl.rs`, `wait_for_any_exit`). So its loop may leave only on /// shutdown: a recorder that stops — at its byte cap, or on a write error — /// must not take the gateway down. An earlier revision `break`ed there. #[test] fn routing_dataset_peer_task_exits_only_on_shutdown() { let src = production_source(); let body = extract_fn_body(src, "async fn record_routing_dataset_peers("); let breaks = body.matches("break").count(); let returns = body.matches("return").count(); assert_eq!( (breaks, returns), (1, 0), "record_routing_dataset_peers must leave its loop only on shutdown; \ any other exit ends the node" ); let (before_break, _) = body.split_once("break").unwrap(); assert!( before_break.contains("sleep_or_shutdown"), "the single break must be the shutdown one" ); } /// Same mirror seam, for the contract-exec WASM counters. The export block /// hand-copies each `ContractExecSnapshot` field into its `RouterSnapshotInfo` /// twin, so a swap — feeding `..._wasm_calls_total` from `fast_hits`, say — /// compiles cleanly and emits a plausible number that is measuring the /// opposite thing. /// /// That failure mode is not hypothetical here: mistaking cache hits for WASM /// work is the exact blindness these counters exist to remove, and an /// overstated saving is worse than a missing one because it terminates the /// investigation. Assert every assignment reads its own field. /// Whitespace-normalized so rustfmt line-wrapping is irrelevant. #[test] fn contract_exec_export_maps_each_field_to_its_own_counter() { let src = production_source(); let block = extract_fn_body(src, "async fn emit_router_snapshot_telemetry("); let norm = block.split_whitespace().collect::>().join(" "); let fields = [ "summarize_fast_hits", "summarize_reload_hits", "summarize_wasm_calls", "summarize_wasm_uncached", "delta_fast_hits", "delta_reload_hits", "delta_wasm_calls", "delta_wasm_uncached", ]; for field in fields { // Every arm is exported in BOTH units. A lifetime total sitting on // a line of otherwise-parallel `_last_snapshot` names reads as a // comparable magnitude and understates nothing visibly — which is // why the pin demands both rather than either. for expected in [ format!("snapshot.contract_exec_{field}_total = Some(ce.{field});"), format!("snapshot.contract_exec_{field}_last_snapshot = Some(ce_d.{field});"), ] { assert!( norm.contains(&expected), "mirror-seam: export must contain `{expected}` — a field swap here \ silently reports one arm's count under another arm's name" ); } } // The delta computation itself is NOT scraped: it is // `ContractExecSnapshot::window_deltas`, one function whose field // correspondence is structural and unit-tested by // `each_field_differences_its_own_twin`. What must be pinned here is // that the emitter uses it rather than re-deriving eight deltas by hand // at the call site, which is where a cross-wiring hides. assert!( norm.contains("let ce_d = ce.window_deltas(&mut prev_exec);"), "the emitter must delegate to ContractExecSnapshot::window_deltas, not \ hand-difference each arm at the call site" ); assert!( !norm.contains("window_delta(ce."), "no hand-written per-arm window_delta call may remain — that is the \ shape in which one arm gets differenced against another's previous value" ); } } #[cfg(test)] mod renewal_ban_gate_source_tests { //! Source-scrape pin test for the subscription-renewal ban gate (#4373). //! //! `recover_orphaned_subscriptions` spawns a `run_renewal_subscribe` //! driver for every contract in `contracts_needing_renewal()`. That //! driver emits an outbound SUBSCRIBE using the same machinery as a //! client-initiated request, but — unlike the four `start_client_*` //! originator entry points — the renewal scheduler does NOT route //! through `operations::reject_if_contract_banned`. So before #4373 a //! contract that was banned while the node still held an active //! subscription kept emitting outbound SUBSCRIBE renewals on every //! maintenance cycle until the ban TTL lifted. //! //! The fix adds an `is_banned` gate in the renewal loop. As the //! `k_closest_source_tests` module above documents, building a real //! `Ring` for a behavioral test requires scaffolding that does not yet //! exist in this suite, so — mirroring the `*_dispatch_gates_banned_contracts` //! pins in `contract_ban_list.rs` — this test scrapes the production //! source to ensure the gate stays wired in and keeps running BEFORE //! the renewal is spawned. A refactor that drops or reorders the gate //! would fail CI rather than silently re-open the egress leak. fn production_source() -> &'static str { const FULL: &str = include_str!("ring.rs"); let cutoff = FULL .find("\n#[cfg(test)]\nmod ") .expect("ring.rs must have a top-level #[cfg(test)] mod section"); &FULL[..cutoff] } fn extract_fn_body<'a>(source: &'a str, signature_prefix: &str) -> &'a str { let start = source .find(signature_prefix) .unwrap_or_else(|| panic!("could not find {signature_prefix}")); let brace = source[start..].find('{').expect("fn sig must have body"); let body_start = start + brace + 1; let bytes = source.as_bytes(); let mut depth: i32 = 1; let mut i = body_start; while i < bytes.len() { match bytes[i] { b'{' => depth += 1, b'}' => { depth -= 1; if depth == 0 { return &source[body_start..i]; } } _ => {} } i += 1; } panic!("unbalanced braces while extracting {signature_prefix}"); } #[test] fn renewal_loop_gates_banned_contracts_before_spawning() { let src = production_source(); let body = extract_fn_body( src, "async fn recover_orphaned_subscriptions(ring: Arc", ); let gate_pos = body.find("contract_ban_list.is_banned").expect( "the subscription-renewal loop must gate on \ `contract_ban_list.is_banned` so a banned-but-still-subscribed \ contract stops emitting outbound SUBSCRIBE renewals (#4373). If \ this gate was removed, the egress leak is back.", ); // The gate must precede the spam-prevention check: a banned contract // is skipped regardless of its `can_request_subscription` backoff // state, so order matters. let can_request_pos = body .find("can_request_subscription(&contract)") .expect("renewal loop must still consult can_request_subscription"); assert!( gate_pos < can_request_pos, "ban gate (offset {gate_pos}) must run BEFORE the \ can_request_subscription spam check (offset {can_request_pos}) so \ a banned contract is always skipped" ); // The gate must precede the renewal spawn. Both `mark_subscription_pending` // (which flips per-contract pending state) and `run_renewal_subscribe` // (the outbound-egress driver) must only be reached for non-banned // contracts. let mark_pending_pos = body .find("mark_subscription_pending(contract)") .expect("renewal loop must still call mark_subscription_pending"); assert!( gate_pos < mark_pending_pos, "ban gate (offset {gate_pos}) must run BEFORE \ mark_subscription_pending (offset {mark_pending_pos})" ); // The outbound-egress driver (`run_renewal_subscribe`) now lives in the // shared `spawn_renewal_subscribe_task` helper (#4642 piece F extraction); // the loop reaches it via that call, which must only run for non-banned // contracts. let spawn_pos = body .find("spawn_renewal_subscribe_task(") .expect("renewal loop must still spawn via spawn_renewal_subscribe_task"); assert!( gate_pos < spawn_pos, "ban gate (offset {gate_pos}) must run BEFORE the \ spawn_renewal_subscribe_task outbound-egress spawn (offset {spawn_pos})" ); } } #[cfg(test)] mod renewal_ban_gate_behavior_tests { //! Behavioral coverage for the predicate the renewal-loop ban gate //! relies on (#4373): a contract on the `ContractBanList` reports //! `is_banned == true` (so the loop's `continue` fires), and an //! un-banned contract reports `false` (so renewal proceeds). This //! exercises the exact `contract_ban_list.is_banned(contract.id())` //! call the loop makes, including the `ContractKey::id()` projection, //! against the real ban-list type and a controllable time source. use super::contract_ban_list::{BanReason, ContractBanList}; // `TimeSource` is needed in scope for the `ts.now()` trait method. use crate::util::time_source::{SharedMockTimeSource, TimeSource}; use freenet_stdlib::prelude::{CodeHash, ContractInstanceId, ContractKey}; use std::sync::Arc; use std::time::Duration; fn mk_key(byte: u8) -> ContractKey { // A full `ContractKey` (not a bare instance id) so the test // exercises the same `contract.id()` projection the renewal loop's // `is_banned(contract.id())` gate performs. ContractKey::from_id_and_code( ContractInstanceId::new([byte; 32]), CodeHash::new([byte; 32]), ) } #[test] fn banned_contract_is_skipped_for_renewal_while_unbanned_proceeds() { let ts = SharedMockTimeSource::new(); let ban_list = ContractBanList::new(Arc::new(ts.clone())); let banned = mk_key(1); let healthy = mk_key(2); // Ban one contract; leave the other alone. ban_list.ban( *banned.id(), ts.now() + Duration::from_secs(60), BanReason::AutoMad, ); // The renewal loop's gate is `if is_banned(contract.id()) { continue }`. // Banned -> skipped; un-banned -> proceeds to renewal. assert!( ban_list.is_banned(banned.id()), "a banned contract must be skipped for subscription renewal (#4373)" ); assert!( !ban_list.is_banned(healthy.id()), "a non-banned contract must still be eligible for renewal" ); // Once the ban TTL lifts, the formerly-banned contract becomes // eligible for renewal again — the gate is time-bounded, matching // the issue's "self-resolves when the ban TTL expires" note. ts.advance_time(Duration::from_secs(61)); assert!( !ban_list.is_banned(banned.id()), "after the ban TTL expires the contract is eligible for renewal again" ); } } #[cfg(test)] mod suspend_jump_tests { use super::classify_suspend_jump; use std::time::Duration; /// Scheduler stall: both clocks advance equally (the loop task was /// starved, but the machine wasn't suspended). The detector must NOT /// treat this as a suspend event — that was the root cause of the /// 2026-04-14 nightly `test_get_reliability_with_latency` false /// positive at `boot_elapsed_secs=139`. #[test] fn scheduler_stall_is_not_suspend() { let boot = Duration::from_secs(139); let mono = Duration::from_secs(139); assert_eq!(classify_suspend_jump(boot, mono), Duration::ZERO); } /// Real OS suspend: CLOCK_BOOTTIME advances by the suspend duration, /// CLOCK_MONOTONIC stays essentially flat. The detector must surface /// the full suspend duration so the caller can compare it against the /// threshold and trigger the DropAllConnections recovery path. #[test] fn real_suspend_is_detected() { let boot = Duration::from_secs(3600); let mono = Duration::from_millis(50); assert_eq!( classify_suspend_jump(boot, mono), Duration::from_secs(3600) - Duration::from_millis(50) ); } /// Healthy tick: both clocks advance by roughly the tick interval and /// the delta is near zero. #[test] fn healthy_tick_is_not_suspend() { let boot = Duration::from_millis(2050); let mono = Duration::from_millis(2048); assert_eq!(classify_suspend_jump(boot, mono), Duration::from_millis(2)); } /// Monotonic exceeding boot (can happen from scheduling jitter between /// the two non-atomic clock reads, or from virtualization TSC anomalies) /// must saturate to zero rather than underflow. A large skew here is /// logged separately by `connection_maintenance` as a diagnostic. #[test] fn monotonic_ahead_of_boot_saturates_to_zero() { let boot = Duration::from_millis(100); let mono = Duration::from_millis(101); assert_eq!(classify_suspend_jump(boot, mono), Duration::ZERO); } /// Threshold comparison: the value the live code compares against the /// detection threshold must be the delta, not `boot_elapsed` alone. /// A 139s scheduler stall at a 30s threshold would trip the old /// detector; with the delta it does not. #[test] fn threshold_comparison_rejects_scheduler_stall() { let threshold = Duration::from_secs(30); let boot = Duration::from_secs(139); let mono = Duration::from_secs(139); assert!(classify_suspend_jump(boot, mono) <= threshold); } /// Threshold comparison: a real suspend of two full check ticks must /// exceed the 2x-tick detection threshold. #[test] fn threshold_comparison_accepts_real_suspend() { let threshold = Duration::from_secs(4); let boot = Duration::from_secs(300); let mono = Duration::from_millis(3); assert!(classify_suspend_jump(boot, mono) > threshold); } } /// Predicate controlling when `connection_maintenance` fires a gateway version probe. /// /// Extracted as a free function so each branch (gateway role, empty configuration, /// timing) can be exercised in unit tests without standing up an `OpManager`. The /// caller relies on `has_configured_gateways` to gate the modulo-indexing into /// `op_manager.configured_gateways` — flipping that flag at the call site (rather /// than inside this function) keeps the borrow of the gateway slice local to the /// caller. #[inline] fn should_probe_gateway( is_gateway: bool, has_configured_gateways: bool, now: Instant, next_probe: Instant, ) -> bool { !is_gateway && has_configured_gateways && now >= next_probe } #[cfg(test)] mod gateway_version_probe_predicate_tests { use super::should_probe_gateway; use std::time::Duration; use tokio::time::Instant; #[test] fn fires_on_non_gateway_with_configured_gateways_at_due_time() { let now = Instant::now(); let next_probe = now - Duration::from_secs(1); assert!(should_probe_gateway(false, true, now, next_probe)); } #[test] fn skipped_when_running_as_gateway() { // Gateways must never probe themselves — they are the destination, not // the originator. This is the protection that keeps the gateway loop // from generating phantom CONNECTs. let now = Instant::now(); let next_probe = now - Duration::from_secs(1); assert!(!should_probe_gateway(true, true, now, next_probe)); } #[test] fn skipped_when_no_gateways_are_configured() { // The empty-gateways case is the boundary that previously sat in an // inner `if` and was unreachable from any test. Merging it into the // predicate makes the modulo-indexing in the caller structurally safe. let now = Instant::now(); let next_probe = now - Duration::from_secs(1); assert!(!should_probe_gateway(false, false, now, next_probe)); } #[test] fn skipped_before_due_time() { let now = Instant::now(); let next_probe = now + Duration::from_secs(60); assert!(!should_probe_gateway(false, true, now, next_probe)); } #[test] fn fires_at_exact_due_time_boundary() { // `>=` semantics: a now equal to next_probe should fire, not skip. let now = Instant::now(); assert!(should_probe_gateway(false, true, now, now)); } #[test] fn gateway_with_no_configured_gateways_is_still_skipped() { let now = Instant::now(); assert!(!should_probe_gateway(true, false, now, now)); } } #[cfg(test)] mod max_concurrent_connections_tests { use super::calculate_max_concurrent_connections; #[test] fn at_min_connections_returns_base() { assert_eq!(calculate_max_concurrent_connections(25, 25), 3); } #[test] fn above_min_connections_returns_base() { assert_eq!(calculate_max_concurrent_connections(30, 25), 3); } #[test] fn large_deficit_scales_up() { // deficit=15, 3 + 15/3 = 8, cap = 25/2 = 12 → 8 assert_eq!(calculate_max_concurrent_connections(10, 25), 8); } #[test] fn full_deficit_capped_at_half_min() { // deficit=25, 3 + 25/3 = 11, cap = 25/2 = 12 → 11 assert_eq!(calculate_max_concurrent_connections(0, 25), 11); } #[test] fn small_deficit_adds_nothing() { // deficit=2, 3 + 2/3 = 3, cap = 25/2 = 12 → 3 assert_eq!(calculate_max_concurrent_connections(23, 25), 3); } #[test] fn very_small_min_connections() { // min_conns=1, deficit=1, 3 + 0 = 3, cap = max(0, 3) = 3 → 3 assert_eq!(calculate_max_concurrent_connections(0, 1), 3); // min_conns=2, deficit=2, 3 + 0 = 3, cap = max(1, 3) = 3 → 3 assert_eq!(calculate_max_concurrent_connections(0, 2), 3); } #[test] fn high_min_connections_scales_cap() { // deficit=50, 3 + 50/3 = 19, cap = 50/2 = 25 → 19 assert_eq!(calculate_max_concurrent_connections(0, 50), 19); } } #[cfg(test)] mod respect_location_backoff_tests { use super::should_respect_location_backoff; /// Regression guard for the under-min backoff escape (#4348): a node below /// min_connections must bypass per-target location backoff so it cannot be /// trapped below min by a stamped `Rejected` backoff; at/above min, backoff /// is respected as before. Pins the inequality direction and boundary. #[test] fn below_min_bypasses_backoff() { assert!(!should_respect_location_backoff(0, 10)); assert!(!should_respect_location_backoff(7, 10)); assert!(!should_respect_location_backoff(9, 10)); } #[test] fn at_or_above_min_respects_backoff() { assert!(should_respect_location_backoff(10, 10)); // exactly min assert!(should_respect_location_backoff(11, 10)); assert!(should_respect_location_backoff(20, 10)); } } #[cfg(test)] mod pending_additions_tests { use super::calculate_allowed_connection_additions; #[test] fn respects_minimum_when_backlog_exists() { let allowed = calculate_allowed_connection_additions(1, 24, 25, 200, 24); assert_eq!(allowed, 0, "Backlog should satisfy minimum deficit"); } #[test] fn permits_requests_until_minimum_is_met() { let allowed = calculate_allowed_connection_additions(1, 0, 25, 200, 24); assert_eq!(allowed, 24); } #[test] fn caps_additions_at_available_capacity() { let allowed = calculate_allowed_connection_additions(190, 5, 25, 200, 10); assert_eq!(allowed, 5); } #[test] fn respects_requested_when_capacity_allows() { let allowed = calculate_allowed_connection_additions(50, 0, 25, 200, 3); assert_eq!(allowed, 3); } } #[cfg(test)] mod op_manager_state_tests { use super::{OpManagerState, classify_op_manager_ref}; use parking_lot::RwLock; use std::sync::{Arc, Weak}; // Stand-in for `OpManager`: `classify_op_manager_ref` is generic over the // pointee, so we exercise the exact production logic without building a // full node. The three slot states map 1:1 onto the maintenance loop's // startup / running / shutdown decisions (#3308). #[test] fn empty_slot_is_not_attached() { // Mirrors the startup window before `attach_op_manager` runs. let slot: RwLock>> = RwLock::new(None); assert!(matches!( classify_op_manager_ref(&slot), OpManagerState::NotAttached )); } #[test] fn live_weak_upgrades_to_live() { let owner = Arc::new(7u32); let slot: RwLock>> = RwLock::new(Some(Arc::downgrade(&owner))); match classify_op_manager_ref(&slot) { OpManagerState::Live(got) => assert_eq!(*got, 7), OpManagerState::NotAttached | OpManagerState::Detached => { panic!("expected Live while the owner Arc is alive") } } } #[test] fn dropped_owner_is_detached_not_not_attached() { // The #3308 regression: the slot was attached (`Some(weak)`) but the // owning Arc has been dropped. This MUST classify as `Detached` // (terminate the loop), NOT `NotAttached` (which spins forever). let owner = Arc::new(7u32); let weak = Arc::downgrade(&owner); let slot: RwLock>> = RwLock::new(Some(weak)); drop(owner); assert!( matches!(classify_op_manager_ref(&slot), OpManagerState::Detached), "a dropped owner must be Detached so connection_maintenance exits \ instead of zombie-spinning (#3308)" ); } /// End-to-end wiring guard for the maintenance loop's `Detached` arm. /// /// The pure classifier tests above only pin the `slot -> state` mapping. /// They would all stay green if a refactor swapped the loop's `Detached` /// arm back to `continue` (the original #3308 bug) or to a bare /// `return Ok(())` (the #4292 monitor-convention violation). This test /// pins the *action*: a maintenance-shaped task that polls /// `classify_op_manager_ref`, breaks on `Detached`, and then parks must, /// once its owner `Arc` is dropped, hand a *non-completing* future to a /// real `BackgroundTaskMonitor` — `wait_for_any_exit` must NOT fire. /// /// Mirrors `reference_ping::spawn_with_unbindable_target_does_not_exit`. #[tokio::test(start_paused = true)] async fn detached_maintenance_task_parks_without_tripping_monitor() { use crate::node::background_task_monitor::BackgroundTaskMonitor; use std::time::Duration; // Shared back-reference slot, exactly like `Ring::op_manager`. let owner = Arc::new(0u32); let slot: Arc>>> = Arc::new(RwLock::new(Some(Arc::downgrade(&owner)))); let task_slot = Arc::clone(&slot); let handle = tokio::spawn(async move { // Reproduces the maintenance loop's arm-to-action wiring without // standing up a full Ring: poll the back-reference, act per state. loop { match classify_op_manager_ref(&task_slot) { // Owner still alive: keep running (the Live work is a // no-op here; the point is that it does NOT exit). OpManagerState::Live(_) => { tokio::time::sleep(Duration::from_millis(50)).await; } // Startup window: keep waiting. OpManagerState::NotAttached => { tokio::time::sleep(Duration::from_millis(50)).await; } // Owner dropped: stop the loop, then park (#3308 / #4292). OpManagerState::Detached => break, } } // Teardown parking: must NOT return cleanly, or the monitor fires. std::future::pending::<()>().await; }); let monitor = BackgroundTaskMonitor::new(); monitor.register("connection_maintenance_sim", handle); // Let the task observe `Live` at least once and start polling. tokio::time::advance(Duration::from_millis(100)).await; tokio::task::yield_now().await; // Drop the owner: the weak now fails to upgrade -> `Detached`. drop(owner); // Give the task time to observe `Detached`, break, and park. tokio::time::advance(Duration::from_millis(200)).await; tokio::task::yield_now().await; // The monitor must NOT fire: orderly teardown parks instead of // returning. A `continue` (zombie spin) or a `return Ok(())` // (monitor-convention violation) would both fail this assertion — // the latter by resolving `wait_for_any_exit`. let exit = monitor.wait_for_any_exit(); tokio::pin!(exit); let still_parked = tokio::time::timeout(Duration::from_millis(100), &mut exit) .await .is_err(); assert!( still_parked, "Detached maintenance task must park, not exit: a clean return \ would trip BackgroundTaskMonitor::wait_for_any_exit and crash \ the node (#3308 / #4292)" ); } } #[cfg(test)] mod deferred_swap_drop_tests { use super::deferred_swap_drops_to_execute; /// 3-node ring: current=2, min=1, pending=1. /// The original bug: headroom = 2-1 = 1 > 0, so the old code fired the drop, /// kicking the only other peer and leaving the node isolated. /// Fixed: guard requires current > min + pending (2 > 1+1 = 2) → false. #[test] fn three_node_ring_no_drop_without_replacement() { assert_eq!(deferred_swap_drops_to_execute(2, 1, 1), 0); } /// Replacement connected in a 3-node ring: current=3, min=1, pending=1. /// 3 > 1+1=2 → true. One drop allowed. #[test] fn three_node_ring_drop_when_replacement_connected() { assert_eq!(deferred_swap_drops_to_execute(3, 1, 1), 1); } /// At min_connections with no pending drops: current=10, min=10, pending=0. /// No pending drops means nothing to execute. #[test] fn no_pending_drops_returns_zero() { assert_eq!(deferred_swap_drops_to_execute(10, 10, 0), 0); } /// Large network, no replacements yet: current=12, min=10, pending=3. /// 12 > 10+3=13 → false. None dropped. #[test] fn large_network_no_drop_without_replacement() { assert_eq!(deferred_swap_drops_to_execute(12, 10, 3), 0); } /// Large network, exactly enough for one replacement: current=13, min=10, pending=3. /// i=0: remaining=3, 13 > 13 → false. Still none. #[test] fn large_network_boundary_no_drop() { assert_eq!(deferred_swap_drops_to_execute(13, 10, 3), 0); } /// Large network, one replacement connected: current=14, min=10, pending=3. /// i=0: remaining=3, 14 > 13 → true (effective=13, n=1) /// i=1: remaining=2, 13 > 12 → true (effective=12, n=2) /// i=2: remaining=1, 12 > 11 → true (effective=11, n=3) /// All 3 dropped; after drops effective=11 ≥ min=10. #[test] fn large_network_all_replacements_connected() { assert_eq!(deferred_swap_drops_to_execute(14, 10, 3), 3); } /// current exactly at min with one pending: current=10, min=10, pending=1. /// 10 > 10+1=11 → false. No drop. #[test] fn at_min_with_pending_no_drop() { assert_eq!(deferred_swap_drops_to_execute(10, 10, 1), 0); } /// current = min + 1 with one pending: current=11, min=10, pending=1. /// 11 > 11 → false. No drop (1 extra but still need the pending slot covered). #[test] fn one_above_min_with_pending_no_drop() { assert_eq!(deferred_swap_drops_to_execute(11, 10, 1), 0); } /// current = min + 2 with one pending: current=12, min=10, pending=1. /// 12 > 11 → true. One drop allowed. #[test] fn two_above_min_with_one_pending_drops_one() { assert_eq!(deferred_swap_drops_to_execute(12, 10, 1), 1); } /// Overflow-safe: current=0, min=0, pending=0. #[test] fn all_zero_returns_zero() { assert_eq!(deferred_swap_drops_to_execute(0, 0, 0), 0); } /// current < min (below minimum, e.g. still bootstrapping): no drops. #[test] fn below_min_connections_no_drop() { assert_eq!(deferred_swap_drops_to_execute(5, 10, 2), 0); } } /// Production intervals of the route-to-self lattice probe (#4760, #5814), /// see [`LatticeProbeScheduler`]. Also used by the topology model test, so it /// exercises the real values. pub(crate) mod lattice_probe_timing { use super::LatticeProbeTiming; use crate::util::backoff::ExponentialBackoff; use std::time::Duration; /// Awake probe backoff: first interval. pub(crate) const TAU0: Duration = Duration::from_secs(5); /// Awake probe backoff: cap. pub(crate) const TAU_MAX: Duration = Duration::from_secs(300); /// First re-check after sleeping on a tight lattice. It outlasts the /// recently-failed-address exclusion (`FAILED_ADDR_MAX_TTL`, 1h). pub(crate) const RECHECK_MIN: Duration = Duration::from_secs(2 * 3600); /// Re-checks per lattice state (2h, 4h, 8h). Each re-check of a tight /// lattice keeps a non-lattice link or two, so they are finite: after the /// last one, only a lattice change wakes discovery. pub(crate) const RECHECKS: u32 = 3; /// First re-check after a generation in which a closer peer was found but /// could not be connected. It outlasts the first recently-failed-address /// exclusion (`FAILED_ADDR_BASE_TTL`, 5 min). pub(crate) const RETRY_MIN: Duration = Duration::from_secs(600); /// Retries for such a peer (10, 20, 40, 80 min), not reset by lattice /// changes, only by a sleep that runs its course without a failed hit: a /// closer peer that is never reachable costs at most this many retries /// between such clean sleeps, then the re-check ladder takes over. pub(crate) const RETRIES: u32 = 4; /// Bounds of the jitter multiplier on every interval, against synchronized /// bursts across peers that bootstrapped together. pub(crate) const JITTER_LOW: f64 = 0.8; pub(crate) const JITTER_HIGH: f64 = 1.2; /// The production timing. pub(crate) fn production() -> LatticeProbeTiming { let ladder = |min: Duration, rungs: u32| { ExponentialBackoff::new(min, min * 2u32.pow(rungs.saturating_sub(1))) }; LatticeProbeTiming { probe: ExponentialBackoff::new(TAU0, TAU_MAX), recheck: ladder(RECHECK_MIN, RECHECKS), rechecks: RECHECKS, retry: ladder(RETRY_MIN, RETRIES), retries: RETRIES, } } } /// Source-grep pin test: lock down the set of bare `Instant::now()` call /// sites in this file so a future change can't silently reintroduce a /// wall-clock time read on the connection-maintenance path. /// /// All production time reads in `ring.rs` go through `self.time_source.now()` /// (the injectable `TimeSource`) so simulation/governance tests can drive the /// connection-maintenance loop under `tokio::time::start_paused(true)` (or a /// `SharedMockTimeSource`) deterministically. See #4277 and the #4260 fix that /// migrated the first such call site (`report_contract_resource_usage`). /// /// Two categories are deliberately exempt and are NOT bare `Instant::now()`: /// * `boot_time::Instant::now()` / `WallClockInstant::now()` — the OS /// suspend/resume detector in `connection_maintenance`, which by design /// needs real wall-clock time (see `.claude/rules/code-style.md`). /// These carry a type prefix, so they don't match a *bare* `Instant::now()`. /// /// The only remaining bare `Instant::now()` call sites live in two /// `#[cfg(test)]` modules: /// * `gateway_version_probe_predicate_tests` (6 sites) — constructs inputs /// for the pure `should_probe_gateway` predicate (no production time read). /// * `resource_meter_bridge_tests` (8 sites) — drives the #3453 bandwidth /// bridge directly with synthetic timestamps; `feed_peer_bandwidth_to_meter` /// itself takes the time as a parameter, and the production caller in /// `connection_maintenance` supplies `self.time_source.now()`. /// /// If you add a new production time read, route it through /// `self.time_source.now()` rather than bumping this count. /// Per-side nearest-neighbor lattice distances observed at a maintenance tick: /// the unsigned ring distance to the nearest connected successor / predecessor, /// or `None` when that side has no connected neighbor. Used by /// [`lattice_probe_progress`] to detect a fill OR a tighten between ticks. #[derive(Debug, Clone, Copy)] pub(crate) struct LatticeSides { pub succ: Option, pub pred: Option, } /// Classification of the lattice change between two consecutive maintenance /// ticks, for the discovery backoff. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(crate) struct LatticeProbeProgress { /// A side was newly FILLED, or an already-held side's nearest neighbor got /// strictly CLOSER (a successful tighten). Resets the probe backoff to tau0. pub improved: bool, /// A side was LOST (left empty) or WIDENED (its nearest dropped and a farther /// neighbor remains). Either way a lattice edge is gone, so discovery wakes, /// resets the backoff to tau0 and re-probes promptly. pub regressed: bool, } /// Classify the lattice change between the previous tick's per-side nearest /// distances (`prev`, `None` on the first observation) and this tick's (`curr`). /// An IMPROVEMENT (a side newly filled, or an already-held side whose nearest /// got strictly closer) keeps the probe aggressive; a REGRESSION (a side lost, /// or its nearest got farther) re-probes promptly; a plateau (neither) lets the /// backoff decay toward tau_max. A filled side is not a stop condition, so the /// probe keeps tightening a loose edge toward the peer's TRUE nearest ring /// neighbor; what stops it is a probe result that is not a lattice edge (see /// [`LatticeProbeScheduler`]). pub(crate) fn lattice_probe_progress( prev: Option, curr: LatticeSides, ) -> LatticeProbeProgress { let Some(prev) = prev else { // First observation: any held side is a fill from the cold-start baseline. return LatticeProbeProgress { improved: curr.succ.is_some() || curr.pred.is_some(), regressed: false, }; }; // A side improved if it went empty -> held (a fill) or held -> strictly closer // (a tighten), and regressed if it went held -> empty or held -> strictly // farther. Distances are derived from fixed peer locations, so a side's // nearest only changes when the connection set changes; a strict comparison // is a real change, not float jitter. let side_improved = |p: Option, c: Option| match (p, c) { (None, Some(_)) => true, (Some(pd), Some(cd)) => cd < pd, _ => false, }; let side_regressed = |p: Option, c: Option| match (p, c) { (Some(_), None) => true, (Some(pd), Some(cd)) => cd > pd, _ => false, }; LatticeProbeProgress { improved: side_improved(prev.succ, curr.succ) || side_improved(prev.pred, curr.pred), regressed: side_regressed(prev.succ, curr.succ) || side_regressed(prev.pred, curr.pred), } } /// When the route-to-self lattice probe fires (mechanism 2, #4760, #5814). /// /// While AWAKE the probe fires on an exponential backoff (tau0 doubling to /// tau_max). Any change to a per-side nearest distance /// ([`lattice_probe_progress`]: fill, tighten, loss or widening) starts a new /// GENERATION, resets the backoff to tau0 and probes promptly, so a loose or /// broken lattice is worked on at once. /// /// A probe MISS on one side (an acceptor there that is not a lattice edge, /// recorded by the CONNECT driver through /// `ConnectionManager::record_lattice_probe_result` with the generation current /// when the probe was launched) is evidence that side is tight: the probe aims /// at the nearest unconnected peer, and it found nothing closer than the held /// nearest. A miss is evidence for the other side only if it lies farther out /// than that side's held nearest; otherwise the nearest unconnected peer may /// simply be on the tight side, so discovery keeps probing, outward past each /// kept miss, until BOTH sides have missed in the current generation. Misses /// from earlier generations (probes in flight across a change) do not count. /// /// It then SLEEPS. A miss is evidence, not proof (the walk can stop short of /// the true nearest: a failed hole punch, a near-terminus relay accepting /// first, a recently-failed peer, or a closer peer that rejected the request), /// so a sleep re-checks (a new generation that needs fresh misses on both /// sides). Each re-check of a tight lattice keeps a link or two, so they are /// FINITE ([`lattice_probe_timing`]): a few per lattice state on the re-check /// ladder, after which only a lattice change wakes discovery. If a closer peer /// WAS found in the generation but could not be connected (a failed hole /// punch, or our own pre-flight refusal at the cap: a failed hit), the misses /// around it may just be the walk going past it, so the sleep uses the shorter /// retry ladder instead. Its count survives lattice changes (a closer peer that /// is never reachable is re-found after every change) and resets only when a /// sleep runs its full course without a failed hit; once it is used up, such /// sleeps fall back to the re-check ladder. A failed hit reported after a /// clean sleep began moves it onto the retry ladder, if that is sooner. /// /// Before #5814 there was no sleep: a converged peer probed every tau_max and /// kept every non-lattice result, so degree grew with uptime. pub(crate) struct LatticeProbeScheduler { timing: LatticeProbeTiming, next_at: Instant, backoff_attempt: u32, last_sides: Option, generation: u64, sleep: Option, recheck_attempt: u32, retry_attempt: u32, } /// Intervals and ladder lengths of [`LatticeProbeScheduler`]. pub(crate) struct LatticeProbeTiming { /// Awake probe backoff. pub probe: crate::util::backoff::ExponentialBackoff, /// Re-check ladder after a clean sleep, and its number of rungs. pub recheck: crate::util::backoff::ExponentialBackoff, pub rechecks: u32, /// Retry ladder after a failed hit, and its number of rungs. pub retry: crate::util::backoff::ExponentialBackoff, pub retries: u32, } /// A sleep of [`LatticeProbeScheduler`]: until when (`None`: until the /// lattice changes), and whether a failed hit has already been considered for /// it (it is on the retry ladder, or a late one did not shorten it). #[derive(Debug, Clone, Copy)] struct LatticeProbeSleep { until: Option, saw_failed_hit: bool, } /// What one [`LatticeProbeScheduler::tick`] decided. #[derive(Debug, Clone, Copy, PartialEq)] pub(crate) struct LatticeProbeTick { /// A side filled or tightened since the previous tick (telemetry). pub improved: bool, /// `Some(interval until the next probe)` if a probe should be queued now. pub fired: Option, } impl LatticeProbeScheduler { /// Awake, due to fire on the first tick at or after `now`. `misses` is the /// peer's recorded probe evidence so far: the first generation starts above /// it, so recorded evidence from an earlier scheduler never counts. pub(crate) fn new( now: Instant, misses: LatticeProbeMisses, timing: LatticeProbeTiming, ) -> Self { Self { timing, next_at: now, backoff_attempt: 0, last_sides: None, generation: misses.succ.max(misses.pred).max(misses.failed_hit) + 1, sleep: None, recheck_attempt: 0, retry_attempt: 0, } } /// The current generation, to tag a probe with when it is issued. pub(crate) fn generation(&self) -> u64 { self.generation } /// Start a new generation and probe promptly. fn restart(&mut self, now: Instant) { self.generation += 1; self.sleep = None; self.backoff_attempt = 0; self.next_at = now; } /// The next rung of the retry ladder, if any is left. fn next_retry(&self, now: Instant, jitter: &mut impl FnMut() -> f64) -> Option { (self.retry_attempt < self.timing.retries).then(|| { now + self .timing .retry .delay(self.retry_attempt) .mul_f64(jitter()) }) } /// Go to sleep on a lattice that both sides' misses show tight. fn fall_asleep( &mut self, now: Instant, failed_hit: bool, jitter: &mut impl FnMut() -> f64, ) -> LatticeProbeSleep { if failed_hit { if let Some(until) = self.next_retry(now, jitter) { self.retry_attempt += 1; return LatticeProbeSleep { until: Some(until), saw_failed_hit: true, }; } // Retries used up: fall back to the ordinary re-checks, which also // guard the other side against a false miss. } let until = self.next_recheck(now, jitter); if until.is_some() { self.recheck_attempt += 1; } LatticeProbeSleep { until, saw_failed_hit: failed_hit, } } /// The next rung of the re-check ladder, if any is left. fn next_recheck(&self, now: Instant, jitter: &mut impl FnMut() -> f64) -> Option { (self.recheck_attempt < self.timing.rechecks).then(|| { now + self .timing .recheck .delay(self.recheck_attempt) .mul_f64(jitter()) }) } /// One maintenance tick for the peer behind `cm`: reads its per-side /// nearest distances and probe misses and calls [`Self::tick`]. pub(crate) fn tick_for( &mut self, cm: &ConnectionManager, now: Instant, jitter: impl FnMut() -> f64, ) -> LatticeProbeTick { let sides = LatticeSides { succ: cm.nearest_lattice_neighbor_dist(true), pred: cm.nearest_lattice_neighbor_dist(false), }; self.tick(now, sides, cm.lattice_probe_misses(), jitter) } /// One maintenance tick. `sides` is this tick's per-side nearest distances, /// `misses` the latest generations with probe evidence, `jitter` draws the /// multiplier applied to a newly scheduled interval (e.g. 0.8..=1.2); it is /// only called when an interval is scheduled or considered. pub(crate) fn tick( &mut self, now: Instant, sides: LatticeSides, misses: LatticeProbeMisses, mut jitter: impl FnMut() -> f64, ) -> LatticeProbeTick { let progress = lattice_probe_progress(self.last_sides, sides); self.last_sides = Some(sides); if progress.improved || progress.regressed { // The lattice changed: work on it promptly, there may be an even // closer neighbor to find (or a lost edge to replace). A new // lattice state gets its re-checks back; the retry count does not // reset, since an unreachable closer peer is re-found after any // change. self.recheck_attempt = 0; self.restart(now); } let tight = |g: u64| misses.succ >= g && misses.pred >= g; if tight(self.generation) { let failed_hit = misses.failed_hit >= self.generation; match self.sleep { Some(LatticeProbeSleep { until: Some(at), saw_failed_hit, }) if now >= at => { // A sleep that ended with nothing in the way: whatever // blocked us is gone, so a later failed hit starts its // retry ladder afresh. (Reset when the sleep ENDS, not when // it begins: a failed hit can still arrive during it.) if !saw_failed_hit { self.retry_attempt = 0; } self.restart(now); } None => self.sleep = Some(self.fall_asleep(now, failed_hit, &mut jitter)), // A failed hit reported after a clean sleep began, considered // once: move onto the retry ladder if that is sooner. The clean // rung it replaces was never slept, so it is given back. Some(sleep) if failed_hit && !sleep.saw_failed_hit => { let mut sleep = LatticeProbeSleep { saw_failed_hit: true, ..sleep }; if let Some(at) = self.next_retry(now, &mut jitter) { if sleep.until.is_none_or(|until| at < until) { if sleep.until.is_some() { self.recheck_attempt = self.recheck_attempt.saturating_sub(1); } self.retry_attempt += 1; sleep = LatticeProbeSleep { until: Some(at), saw_failed_hit: true, }; } } self.sleep = Some(sleep); } Some(_) => {} } } let asleep = tight(self.generation); let fired = (!asleep && now >= self.next_at).then(|| { let interval = self.timing.probe.delay(self.backoff_attempt); self.next_at = now + interval.mul_f64(jitter()); self.backoff_attempt = self.backoff_attempt.saturating_add(1); interval }); LatticeProbeTick { improved: progress.improved, fired, } } } #[cfg(test)] mod lattice_probe_state_machine_tests { use super::{ LatticeProbeMisses, LatticeProbeScheduler, LatticeProbeTiming, LatticeSides, lattice_probe_progress, }; use crate::util::backoff::ExponentialBackoff; use std::time::Duration; use tokio::time::Instant; /// Filled sides are not a stop condition. A tick where both sides are /// filled and UNCHANGED is a plateau (neither improved nor regressed): the /// backoff grows, and only probe misses (see the scheduler tests below) put /// discovery to sleep. #[test] fn both_sides_filled_and_unchanged_is_a_plateau_not_a_stop() { let both = LatticeSides { succ: Some(0.05), pred: Some(0.05), }; let p = lattice_probe_progress(Some(both), both); assert!( !p.improved && !p.regressed, "a steady filled state is a plateau, not a stop" ); } /// A newly-filled side (empty -> held) is an improvement. #[test] fn fill_is_an_improvement() { let prev = LatticeSides { succ: None, pred: Some(0.1), }; let curr = LatticeSides { succ: Some(0.2), pred: Some(0.1), }; let p = lattice_probe_progress(Some(prev), curr); assert!(p.improved && !p.regressed); } /// A TIGHTEN — an already-held side whose nearest gets strictly closer — is an /// improvement. This is exactly the signal the fill-only probe ignored. #[test] fn tighten_of_a_filled_side_is_an_improvement() { let prev = LatticeSides { succ: Some(0.20), pred: Some(0.10), }; let curr = LatticeSides { succ: Some(0.08), pred: Some(0.10), }; let p = lattice_probe_progress(Some(prev), curr); assert!( p.improved, "a strictly-closer nearest on a filled side is an improvement" ); assert!(!p.regressed); } /// A lost side (held -> empty) is a regression, and so is a widened side (a /// farther nearest: the closest dropped and a farther one remains). At /// production degree a side almost never empties, so widening is how a lost /// lattice edge shows up, and it must wake sleeping discovery (#5814). #[test] fn lost_or_widened_side_is_a_regression() { let prev = LatticeSides { succ: Some(0.05), pred: Some(0.05), }; let lost = LatticeSides { succ: None, pred: Some(0.05), }; let p = lattice_probe_progress(Some(prev), lost); assert!(p.regressed && !p.improved); let widened = LatticeSides { succ: Some(0.30), pred: Some(0.05), }; let p2 = lattice_probe_progress(Some(prev), widened); assert!( p2.regressed && !p2.improved, "a farther nearest on a still-held side is a regression" ); } /// First observation: any held side counts as a fill from the cold baseline; /// nothing held is a plateau. #[test] fn first_observation_classification() { let held = LatticeSides { succ: Some(0.3), pred: None, }; let p = lattice_probe_progress(None, held); assert!(p.improved && !p.regressed); let empty = LatticeSides { succ: None, pred: None, }; let p0 = lattice_probe_progress(None, empty); assert!(!p0.improved && !p0.regressed); } /// The probe backoff uses the shared `ExponentialBackoff` and DECAYS toward /// tau_max on a plateau: the interval is always finite, monotonic, and /// floored at tau_max. #[test] fn backoff_decays_to_tau_max() { let tau0 = Duration::from_secs(5); let tau_max = Duration::from_secs(300); let backoff = ExponentialBackoff::new(tau0, tau_max); assert_eq!(backoff.delay(0), tau0, "attempt 0 probes at tau0"); let mut attempt: u32 = 0; let mut last = Duration::ZERO; for _ in 0..20 { let d = backoff.delay(attempt); assert!(d >= last, "interval is monotonic non-decreasing"); assert!(d <= tau_max, "interval floors at tau_max, never longer"); last = d; attempt = attempt.saturating_add(1); } assert_eq!( last, tau_max, "the backoff saturates at (floors at) tau_max" ); // An improvement or a lost edge resets the attempt to 0 -> aggressive tau0. assert_eq!(backoff.delay(0), tau0); } const BOTH: LatticeSides = LatticeSides { succ: Some(0.01), pred: Some(0.01), }; const HOUR: u64 = 3600; /// Test ladders: re-checks 1h, 2h, 4h; retries 10, 20, 40, 80 min. fn timing() -> LatticeProbeTiming { LatticeProbeTiming { probe: ExponentialBackoff::new(Duration::from_secs(5), Duration::from_secs(300)), recheck: ExponentialBackoff::new( Duration::from_secs(HOUR), Duration::from_secs(4 * HOUR), ), rechecks: 3, retry: ExponentialBackoff::new(Duration::from_secs(600), Duration::from_secs(4800)), retries: 4, } } fn scheduler(start: Instant) -> LatticeProbeScheduler { LatticeProbeScheduler::new(start, LatticeProbeMisses::default(), timing()) } fn misses(succ: u64, pred: u64) -> LatticeProbeMisses { LatticeProbeMisses { succ, pred, failed_hit: 0, } } fn failed(g: u64) -> LatticeProbeMisses { LatticeProbeMisses { succ: g, pred: g, failed_hit: g, } } /// Ticks once a second over `(from, from + secs]` with unchanged sides; /// `misses_for(generation)` gives the evidence the driver would have /// recorded by then. Returns the seconds at which the probe fired. fn run( s: &mut LatticeProbeScheduler, start: Instant, from: u64, secs: u64, sides: LatticeSides, misses_for: impl Fn(u64) -> LatticeProbeMisses, ) -> Vec { (from + 1..=from + secs) .filter(|t| { let m = misses_for(s.generation()); s.tick(start + Duration::from_secs(*t), sides, m, || 1.0) .fired .is_some() }) .collect() } fn within(t: u64, at: u64) -> bool { (at..at + 5).contains(&t) } /// Converged discovery sleeps (#5814): once BOTH sides have missed in the /// current generation, the probe stops firing except for a finite ladder of /// re-checks (each needing fresh misses), instead of re-firing every /// tau_max and keeping a non-lattice link each time. #[test] fn misses_on_both_sides_sleep_with_finitely_many_rechecks() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); assert!(run(&mut s, start, 0, 600, BOTH, |_| misses(0, 0)).len() >= 4); let fired = run(&mut s, start, 600, 48 * HOUR, BOTH, |g| misses(g, g)); assert_eq!( fired.len(), 3, "re-checks after 1h, 2h, 4h, then none: {fired:?}" ); assert!(within(fired[0], 601 + HOUR), "{fired:?}"); assert!(within(fired[1], fired[0] + 1 + 2 * HOUR), "{fired:?}"); assert!(within(fired[2], fired[1] + 1 + 4 * HOUR), "{fired:?}"); } /// A re-check starts a new generation, so the misses that put discovery to /// sleep no longer count and it probes until both sides miss again. #[test] fn recheck_needs_fresh_misses() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); let slept_in = s.generation(); run(&mut s, start, 60, 60, BOTH, |g| misses(g, g)); assert_eq!(s.generation(), slept_in); // Only the old generation's misses: after the re-check it stays awake. let fired = run(&mut s, start, 120, 3 * HOUR, BOTH, move |_| { misses(slept_in, slept_in) }); assert!(fired.len() >= 20, "awake after the re-check: {fired:?}"); } /// One side's miss is evidence only for that side: the probe aims at the /// nearest unconnected peer, which may simply be on the tight side, so the /// other side must still be probed. Misses on the two sides may arrive on /// different ticks. #[test] fn a_miss_on_one_side_keeps_probing() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); assert!(run(&mut s, start, 60, HOUR, BOTH, |g| misses(g, 0)).len() >= 10); assert!(run(&mut s, start, 60 + HOUR, HOUR / 2, BOTH, |g| misses(g, g)).is_empty()); } /// Misses recorded for an earlier generation (a probe issued before a /// lattice change) are not evidence about the lattice after the change. #[test] fn misses_from_an_earlier_generation_do_not_count() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); let before = s.generation(); let widened = LatticeSides { succ: Some(0.02), pred: Some(0.01), }; run(&mut s, start, 60, 1, widened, |_| misses(0, 0)); assert!(s.generation() > before); assert!( run(&mut s, start, 61, 600, widened, move |_| misses( before, before )) .len() >= 4, "stale misses must not put the new generation to sleep" ); } /// A closer peer found but not connected in this generation means the /// misses around it may be the walk going past it: re-check on the retry /// ladder (10 min), which climbs and is finite. #[test] fn failed_hits_use_a_finite_retry_ladder() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); let fired = run(&mut s, start, 60, 12 * HOUR, BOTH, failed); assert_eq!( fired.len(), 4 + 3, "retries after 10, 20, 40, 80 min, then the re-checks: {fired:?}" ); assert!(within(fired[0], 61 + 600), "{fired:?}"); assert!(within(fired[1], fired[0] + 1 + 1200), "{fired:?}"); assert!(within(fired[2], fired[1] + 1 + 2400), "{fired:?}"); assert!(within(fired[3], fired[2] + 1 + 4800), "{fired:?}"); } /// A lattice change gives the re-check ladder back its rungs but keeps the /// retry count (an unreachable closer peer is re-found after any change); a /// generation that sleeps clean resets the retry count. #[test] fn a_change_resets_rechecks_but_not_retries() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Two clean re-checks (1h, 2h)... let fired = run(&mut s, start, 60, 3 * HOUR + 10, BOTH, |g| misses(g, g)); assert_eq!(fired.len(), 2, "{fired:?}"); // ...then a change: the next clean sleep is back on the first rung. let widened = LatticeSides { succ: Some(0.02), pred: Some(0.01), }; let t = 3 * HOUR + 70; run(&mut s, start, t, 1, widened, |_| misses(0, 0)); let fired = run(&mut s, start, t + 1, 2 * HOUR, widened, |g| misses(g, g)); assert!(within(fired[0], t + 2 + HOUR), "{fired:?}"); // Two failed-hit retries (10, 20 min), then asleep on the third rung // (40 min) when the change comes... let t = fired[0]; let fired = run(&mut s, start, t, 1900, widened, failed); assert_eq!(fired.len(), 2, "{fired:?}"); // ...so the next failed hit gets the fourth rung (80 min), not 10 min. let t = t + 1900; run(&mut s, start, t, 1, BOTH, |_| misses(0, 0)); let fired = run(&mut s, start, t + 1, 2 * HOUR, BOTH, failed); assert!(within(fired[0], t + 2 + 4800), "{fired:?}"); } /// A failed hit reported after a clean sleep began shortens it to the /// retry ladder; the clean rung it replaced is given back, and the retry /// rung is used up. #[test] fn late_failed_hit_shortens_sleep() { for later_failed in [false, true] { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Asleep on the 1h re-check... assert!(run(&mut s, start, 60, 300, BOTH, |g| misses(g, g)).is_empty()); // ...then the failed hit from an in-flight probe of this generation. let sleeping = s.generation(); let fired = run(&mut s, start, 360, 3 * HOUR, BOTH, move |g| { if g == sleeping || later_failed { failed(g) } else { misses(g, g) } }); assert!(within(fired[0], 361 + 600), "{fired:?}"); if later_failed { // The retry rung was used: the next failed hit sleeps 20 min. assert!(within(fired[1], fired[0] + 1 + 1200), "{fired:?}"); } else { // The clean rung was given back: the next clean sleep is 1h. assert!(within(fired[1], fired[0] + 1 + HOUR), "{fired:?}"); } } } /// A failed hit reported late in a clean sleep, when the retry rung would /// end after the sleep does, changes nothing: no later re-check, no rung /// used up. #[test] fn late_failed_hit_never_extends_sleep() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Asleep on the 1h re-check from 61 (due at 3661)... assert!(run(&mut s, start, 60, 3240, BOTH, |g| misses(g, g)).is_empty()); // ...a failed hit appears at 3301, when 10 more minutes would end // after 3661: the re-check stays at 3661. let fired = run(&mut s, start, 3300, 1200, BOTH, failed); assert!(within(fired[0], 3661), "{fired:?}"); // The retry ladder was not used: the failed hit in the new generation // sleeps 10 min, not 20. assert!(within(fired[1], fired[0] + 1 + 600), "{fired:?}"); } /// A failed hit in every generation, always reported late (after a clean /// sleep began, as when a near-terminus relay's miss arrives before the /// closer peer's hole punch fails), still uses up the retry ladder: a /// closer peer that is never reachable costs at most `retries` retries, /// then the re-check ladder, then nothing until the lattice changes. #[test] fn unreachable_closer_peer_costs_finitely_many_wakes() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Each generation sleeps clean on its first tick, and the failed hit // for it shows up 30 s later. let mut slept_at: std::collections::HashMap = Default::default(); let mut fired = Vec::new(); for t in 61..61 + 48 * HOUR { let g = s.generation(); let first = *slept_at.entry(g).or_insert(t); let m = if t >= first + 30 { failed(g) } else { misses(g, g) }; if s.tick(start + Duration::from_secs(t), BOTH, m, || 1.0) .fired .is_some() { fired.push(t); } } assert!( fired.len() == 4 + 3, "the retries and then the re-checks, nothing more: {fired:?}" ); } /// Once retries are used up, a failed hit falls back to the re-check /// ladder rather than sleeping until a change, so the other side keeps /// its guard against a false miss. #[test] fn exhausted_retries_fall_back_to_rechecks() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); let fired = run(&mut s, start, 60, 48 * HOUR, BOTH, failed); assert_eq!(fired.len(), 4 + 3, "4 retries then 3 re-checks: {fired:?}"); let t = fired[3] + 1; assert!(within(fired[4], t + HOUR), "{fired:?}"); } /// A sleep that runs its course without a failed hit resets the retry /// count: the next failed hit starts at 10 min again. #[test] fn clean_sleep_resets_retries() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Two retries (10, 20 min), then the third sleep starts... let fired = run(&mut s, start, 60, 1900, BOTH, failed); assert_eq!(fired.len(), 2, "{fired:?}"); // ...a clean generation sleeps its full hour... let t = 60 + 1900 + 2400 + 10; run(&mut s, start, 1960, t - 1960, BOTH, |g| misses(g, g)); let fired = run(&mut s, start, t, 2 * HOUR, BOTH, |g| misses(g, g)); // ...and the failed hit after that is back on the first rung. let t2 = fired[0]; let fired = run(&mut s, start, t2, HOUR, BOTH, failed); assert!(within(fired[0], t2 + 1 + 600), "{fired:?}"); } /// Awake, the probe backs off 5 s doubling to 300 s; a change while the /// next probe is far off fires at once. #[test] fn awake_cadence_and_prompt_wake() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); let fired = run(&mut s, start, 0, 1300, BOTH, |_| misses(0, 0)); let gaps: Vec = fired.windows(2).map(|w| w[1] - w[0]).collect(); assert_eq!(&gaps[..7], &[5, 10, 20, 40, 80, 160, 300], "{fired:?}"); let widened = LatticeSides { succ: Some(0.02), pred: Some(0.01), }; let tick = s.tick( start + Duration::from_secs(1301), widened, misses(0, 0), || 1.0, ); assert!(tick.fired.is_some()); } /// The jitter scales the awake backoff and the re-check sleeps. #[test] fn jitter_scales_awake_probes_and_rechecks() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); let mut fired = Vec::new(); for t in 1..=3 * HOUR { let g = s.generation(); let m = if t < 100 { misses(0, 0) } else { misses(g, g) }; if s.tick(start + Duration::from_secs(t), BOTH, m, || 1.2) .fired .is_some() { fired.push(t); } } let gaps: Vec = fired.windows(2).map(|w| w[1] - w[0]).collect(); assert_eq!(&gaps[..3], &[6, 12, 24], "{fired:?}"); // Asleep from 100 on the 1h rung, scaled by 1.2. let wake = fired.iter().copied().find(|t| *t > 100).unwrap(); assert!(within(wake, 100 + 4320), "{fired:?}"); } /// A late failed hit is considered once per sleep: if the retry rung it /// draws would end after the clean sleep, a later, luckier draw does not /// get to shorten the sleep after all. #[test] fn late_failed_hit_is_considered_once() { let start = tokio::time::Instant::now(); let mut s = scheduler(start); let draw = std::cell::Cell::new(1.0); let mut fired = Vec::new(); for t in 1..=2 * HOUR { let g = s.generation(); // Clean sleep from 61 (due 3661); the failed hit appears at 3000, // first with a draw of 1.2 (3720, not sooner), then 0.8 (it would // be 3481 if it were considered again). let m = if t <= 60 { misses(0, 0) } else if t < 3000 { misses(g, g) } else { failed(g) }; draw.set(match t { 3000 => 1.2, 3001.. => 0.8, _ => 1.0, }); if s.tick(start + Duration::from_secs(t), BOTH, m, || draw.get()) .fired .is_some() { fired.push(t); } } let wake = fired.iter().copied().find(|t| *t > 61).unwrap(); assert!(within(wake, 3661), "{fired:?}"); } /// A scheduler created on a peer with earlier evidence starts above it, so /// that evidence cannot put it to sleep; each of the three records alone /// sets the floor. #[test] fn new_scheduler_ignores_earlier_evidence() { let start = tokio::time::Instant::now(); for (succ, pred, failed_hit) in [(20, 1, 1), (1, 20, 1), (1, 1, 20)] { let earlier = LatticeProbeMisses { succ, pred, failed_hit, }; let mut s = LatticeProbeScheduler::new(start, earlier, timing()); assert_eq!(s.generation(), 21); assert!(run(&mut s, start, 0, 600, BOTH, move |_| earlier).len() >= 4); } } /// An asleep scheduler wakes immediately on any lattice change: a widened /// side (lattice edge lost, farther neighbor remains), a lost side, or a /// tighten. This includes a scheduler whose re-checks are used up. #[test] fn any_lattice_change_wakes_discovery() { let changes = [ LatticeSides { succ: Some(0.02), pred: Some(0.01), }, LatticeSides { succ: None, pred: Some(0.01), }, LatticeSides { succ: Some(0.005), pred: Some(0.01), }, ]; for changed in changes { let start = tokio::time::Instant::now(); let mut s = scheduler(start); run(&mut s, start, 0, 60, BOTH, |_| misses(0, 0)); // Past all three re-checks: asleep until something changes. assert_eq!( run(&mut s, start, 60, 12 * HOUR, BOTH, |g| misses(g, g)).len(), 3 ); assert!( run(&mut s, start, 60 + 12 * HOUR, 24 * HOUR, BOTH, |g| misses( g, g )) .is_empty() ); let gen_asleep = s.generation(); let woke = s.tick( start + Duration::from_secs(37 * HOUR), changed, misses(gen_asleep, gen_asleep), || 1.0, ); assert_eq!( woke.fired, Some(Duration::from_secs(5)), "{changed:?} must wake discovery at tau0" ); } } /// The production ladders are finite and their first rungs outlast the /// failed-address exclusions (checked against the exclusion values in /// `connection_manager`), even at the low end of the jitter. #[test] fn production_ladders() { use super::lattice_probe_timing as t; let p = t::production(); assert_eq!(p.probe.delay(0), Duration::from_secs(5)); assert_eq!(p.probe.delay(20), Duration::from_secs(300)); assert_eq!(p.recheck.delay(0), Duration::from_secs(2 * HOUR)); assert_eq!(p.retry.delay(0), Duration::from_secs(600)); assert_eq!(p.recheck.delay(0), t::RECHECK_MIN); assert_eq!(p.recheck.delay(t::RECHECKS - 1), t::RECHECK_MIN * 4); assert_eq!(p.retry.delay(0), t::RETRY_MIN); assert_eq!(p.retry.delay(t::RETRIES - 1), t::RETRY_MIN * 8); assert_eq!((p.rechecks, p.retries), (t::RECHECKS, t::RETRIES)); assert_eq!((t::JITTER_LOW, t::JITTER_HIGH), (0.8, 1.2)); } } #[cfg(test)] mod instant_now_pin_test { /// Number of bare `Instant::now()` call sites expected in this file, all /// in test code. Production code must use `self.time_source.now()`. /// 6 in `gateway_version_probe_predicate_tests` + 8 in /// `resource_meter_bridge_tests` (#3453). const EXPECTED_BARE_INSTANT_NOW: usize = 14; #[test] fn no_unexpected_bare_instant_now_call_sites() { let src = include_str!("ring.rs"); // Build the needle from fragments so this test's own source line does // not itself contain a verbatim bare call-site token that it would count. let needle = format!("Instant{}now()", "::"); let mut count = 0usize; for line in src.lines() { // Strip a `//` line comment so doc/comment mentions (including this // test's own docs) don't count. let code = match line.split_once("//") { Some((before, _)) => before, None => line, }; // Count bare occurrences — i.e. not preceded by an identifier or // path separator (excludes `boot_time::Instant::now()` and // `WallClockInstant::now()`). let bytes = code.as_bytes(); let mut idx = 0; while let Some(pos) = code[idx..].find(&needle) { let abs = idx + pos; let prev_is_pathy = abs .checked_sub(1) .map(|p| { let c = bytes[p]; c == b':' || c == b'_' || c.is_ascii_alphanumeric() }) .unwrap_or(false); if !prev_is_pathy { count += 1; } idx = abs + needle.len(); } } assert_eq!( count, EXPECTED_BARE_INSTANT_NOW, "Unexpected number of bare wall-clock time-read call sites in ring.rs. \ Production code must read time via `self.time_source.now()` so the \ connection-maintenance loop stays deterministic under start_paused \ (#4277). If you intentionally added or removed a TEST-only call site, \ update EXPECTED_BARE_INSTANT_NOW; if this fired on PRODUCTION code, \ migrate it to `self.time_source.now()` instead of bumping the count." ); } } /// Wiring pin for the shadow-mode futile-repair rows on /// `network_efficiency_v1` (`crate::ring::futile_repair`). /// /// The detector's state machine is unit-tested in its own module and its /// handler wiring in `node.rs`. What NEITHER can see is the assignment in /// `router_snapshot_telemetry` that connects the two: drop it, or feed a row /// from the wrong source, and every one of those tests stays green while the /// fleet publishes zeros — for a release whose entire purpose is to establish /// the detector's real frequency in production. That is the #4009/#4010 /// manually-mirrored-telemetry footgun, and the snapshot loop is a 30-minute /// background cadence inside a fully-built `Ring`, which is why it is pinned by /// source scrape rather than executed. #[cfg(test)] mod futile_repair_wiring_pin { /// Production source only: the needles below appear in this module too, and /// a pin that matches its own source is a pin that can never fail. fn production_source() -> &'static str { const FULL: &str = include_str!("ring.rs"); let cutoff = FULL .find("\n#[cfg(test)]\nmod ") .expect("ring.rs must have a top-level #[cfg(test)] mod section"); &FULL[..cutoff] } #[test] fn network_efficiency_block_is_fed_from_the_futile_repair_snapshot() { let src = production_source(); assert!( src.contains( "let futile_repair = op_manager.interest_manager.futile_repair_snapshot();" ), "the futile-repair snapshot is no longer taken in the \ router_snapshot block — the detector then collects its counters \ and exports nothing, while every test for it still passes" ); for (field, source_expr) in [ ("futile", "futile_repair.to_row()"), ("futile_ladder", "futile_repair.ladder"), ] { let needle = format!("{field}: {source_expr},"); assert!( src.contains(&needle), "`NetworkEfficiencyV1.{field}` must be populated as `{needle}` \ in the router_snapshot block. Dropping it, or feeding it from \ anything else, publishes an empty or wrong series with no \ failing test — #4009/#4010." ); } } } /// Pin tests for the periodic hosting-advertisement re-request (#4642 spec step /// 1, "Fix 1"): the reliability backstop that heals a dropped on-connect /// advertisement exchange or a dropped per-eviction retraction. It is PIGGYBACKED /// on the `interest_heartbeat` loop (NOT a separate task) so it adds no /// `GlobalRng` draw / second timer that would perturb the deterministic sim /// harness. These source-scrape pins lock that: the re-request rides /// `interest_heartbeat`, and it carries only contract IDs (no state), so it can't /// re-arm the #4440/#4473 summarize storm. #[cfg(test)] mod hosting_heartbeat_pin_test { fn extract_interest_heartbeat_body() -> &'static str { let src = include_str!("ring.rs"); let head = ["async fn ", "interest_heartbeat("].concat(); let start = src .find(&head) .expect("`async fn interest_heartbeat(` must exist in ring.rs"); let body_open = src[start..].find('{').map(|off| start + off).unwrap(); let mut depth: i32 = 0; let mut end = body_open; for (i, ch) in src[body_open..].char_indices() { match ch { '{' => depth += 1, '}' => { depth -= 1; if depth == 0 { end = body_open + i + 1; break; } } _ => {} } } &src[start..end] } #[test] fn hosting_re_request_is_piggybacked_on_interest_heartbeat() { let body = extract_interest_heartbeat_body(); assert!( body.contains("HostingStateRequest"), "the advertisement-layer re-request (Fix 1) must be PIGGYBACKED on \ `interest_heartbeat` — it must send a `HostingStateRequest` in that \ loop. Do NOT move it to a separate task: a task with its own \ `GlobalRng` initial-delay draw perturbs the shared seeded RNG stream \ the deterministic sim harness depends on." ); } #[test] fn hosting_re_request_carries_no_state() { // Advertisement-layer ONLY: the re-request must NOT reach into the STATE // anti-entropy (no WASM summarize / state fetch), or it would risk // re-arming the #4440/#4473 summarize storm the ring.md gate guards against. let body = extract_interest_heartbeat_body(); assert!( !body.contains("get_contract_summary") && !body.contains("summarize_state"), "the piggybacked hosting re-request must carry only contract IDs — \ never fetch / summarize STATE — so it cannot re-arm the summarize storm." ); } #[test] fn no_separate_hosting_heartbeat_task() { let src = include_str!("ring.rs"); // Runtime-compose so this test does not match itself. let needle = ["async fn ", "hosting_heartbeat("].concat(); assert!( !src.contains(&needle), "there must be NO separate `hosting_heartbeat` task — its startup \ `GlobalRng` draw perturbed the deterministic sim harness. The \ re-request is piggybacked on `interest_heartbeat` instead." ); } } /// Direct unit coverage for the pure core of `is_subscription_root` / /// `body_holding_subscription_root_key` (#4440). Building a full `Ring` /// fixture is heavyweight (async `Ring::new` spawns background tasks), so the /// distance + routability decision is factored into the pure /// `no_closer_routable_neighbor` helper and tested here. The hosting-gate and /// instance-id resolution are covered end-to-end by the simulation test /// `test_subscription_root_renewal_does_not_storm`. #[cfg(test)] mod subscription_root_predicate_tests { use super::no_closer_routable_neighbor; use crate::ring::location::Location; // Contract at 0.50; "me" hosting it at distance 0.05 (I'm at 0.55). const CONTRACT: f64 = 0.50; fn contract_loc() -> Location { Location::new(CONTRACT) } fn my_distance() -> super::Distance { Location::new(0.55).distance(contract_loc()) } #[test] fn no_neighbors_means_terminus() { // Hosting, closest by default (nobody else) → terminus. assert!(no_closer_routable_neighbor( my_distance(), contract_loc(), std::iter::empty(), )); } #[test] fn routable_closer_neighbor_means_not_terminus() { // A ready, non-transient peer AT the key (distance 0) is strictly closer // → I am NOT the terminus. let neighbors = [(Some(Location::new(CONTRACT)), true)]; assert!(!no_closer_routable_neighbor( my_distance(), contract_loc(), neighbors.into_iter(), )); } #[test] fn farther_routable_neighbor_keeps_terminus() { // A routable peer that is FARTHER from the key than me (at 0.20, // distance 0.30 > my 0.05) doesn't unseat me. let neighbors = [(Some(Location::new(0.20)), true)]; assert!(no_closer_routable_neighbor( my_distance(), contract_loc(), neighbors.into_iter(), )); } #[test] fn closer_but_non_routable_neighbor_keeps_terminus() { // The ONLY closer neighbor is non-routable (a transient peer, which // `k_closest` excludes unconditionally). A renewal could not route to // it, so I am still the effective terminus — the #4440 false-negative // guard. (Not-ready peers are mapped to routable=true by // `is_subscription_root` because `k_closest` falls back to them, so they // do NOT reach this `routable=false` case — see the mapping there.) let neighbors = [(Some(Location::new(CONTRACT)), false)]; assert!(no_closer_routable_neighbor( my_distance(), contract_loc(), neighbors.into_iter(), )); } #[test] fn closer_routable_among_non_routable_means_not_terminus() { // Mixed: one closer non-routable peer (ignored) AND one closer routable // peer (decisive) → NOT the terminus. let neighbors = [ (Some(Location::new(0.51)), false), // closer but non-routable → skip (Some(Location::new(0.49)), true), // closer AND routable → unseats ]; assert!(!no_closer_routable_neighbor( my_distance(), contract_loc(), neighbors.into_iter(), )); } #[test] fn locationless_neighbor_does_not_unseat() { // A routable neighbor with no known location can't be compared on // distance, so it never makes us "not the root" on its own. let neighbors = [(None, true)]; assert!(no_closer_routable_neighbor( my_distance(), contract_loc(), neighbors.into_iter(), )); } } /// Direct unit coverage for the pure core of `Ring::most_keyward_hosting_neighbor` /// (piece D, computed-upstream selection). Building a full `Ring` fixture is /// heavyweight (async `Ring::new` spawns background tasks) and a peer's ring /// location is derived from its (masked) socket address, so distances aren't /// freely settable through a real `ConnectionManager`. The strict-closer + /// deterministic-tiebreak selection is therefore factored into the pure /// `most_keyward_among` helper and tested here with distances supplied directly; /// the pub-key resolution / own-location glue is a thin wrapper covered /// end-to-end by the computed-upstream simulation proofs (later keystone steps). #[cfg(test)] mod most_keyward_tests { use super::most_keyward_among; use crate::ring::PeerKeyLocation; use crate::ring::location::{Distance, Location}; use crate::transport::TransportKeypair; use std::net::SocketAddr; // Contract at 0.50; "me" hosting it at distance 0.10 (I'm at 0.60). const CONTRACT: f64 = 0.50; fn my_distance() -> Distance { Location::new(0.60).distance(Location::new(CONTRACT)) } /// A candidate peer at `addr` whose distance-to-contract is `dist`. The /// distance is supplied directly (the helper takes it pre-computed), so it is /// decoupled from the address — the address only drives the tiebreak. fn candidate(addr: &str, dist: f64) -> (PeerKeyLocation, Distance) { let addr: SocketAddr = addr.parse().unwrap(); let pk = TransportKeypair::new().public().clone(); (PeerKeyLocation::new(pk, addr), Distance::new(dist)) } #[test] fn picks_closest_strictly_closer_neighbor() { // Two neighbors, both strictly closer than my 0.10: one at 0.05, one at // 0.02. The most-keyward (smallest distance) wins. let near = candidate("127.0.0.1:9001", 0.02); let mid = candidate("127.0.0.1:9002", 0.05); let want = near.0.clone(); let picked = most_keyward_among(my_distance(), [mid, near].into_iter()); assert_eq!( picked, Some(want), "the neighbor closest to the key (0.02) must be chosen over the farther-but-still-closer one (0.05)" ); } #[test] fn excludes_farther_or_equal_neighbors() { // This test PINS the strict `<` boundary (the acyclicity guarantee) // against a regression to `<=`, so the "equal" candidate's distance must // be BIT-IDENTICAL to `my_distance()`. The literal 0.10 is NOT: the ring // distance `Location::new(0.60).distance(Location::new(0.50))` is // `0.09999999999999998`, whereas `Distance::new(0.10)` stores // `0.10000000000000001`, so a candidate built from `0.10` is actually // strictly GREATER and is excluded as "farther" under BOTH `<` and `<=` // — pinning nothing. Feeding `my_distance()` back in (via `as_f64`, which // `Distance::new` stores verbatim for values <= 0.5) makes the candidate // exactly equal, so `<` excludes it (None) while a regression to `<=` // would include and pick it (Some): the two now disagree and this test // fails under `<=`. let equal_dist = my_distance().as_f64(); // A strictly-closer neighbor (0.02) coexists with one at EXACTLY my // distance and one FARTHER (0.15). Only the strictly-closer one is // eligible, so it is chosen — the equal and farther peers are excluded. let closer = candidate("127.0.0.1:9001", 0.02); let equal = candidate("127.0.0.1:9002", equal_dist); let farther = candidate("127.0.0.1:9003", 0.15); let want = closer.0.clone(); let picked = most_keyward_among(my_distance(), [equal, farther, closer].into_iter()); assert_eq!( picked, Some(want), "only the strictly-closer neighbor is eligible" ); // With ONLY an equal-distance neighbor present, the strict `<` excludes // it → None (a peer at our exact distance never becomes our upstream, the // acyclicity guarantee). This is the assertion that DISTINGUISHES `<` // from `<=`: under the shipped `<` the equal candidate yields None, while // a regression to `<=` would yield `Some(equal)` and fail here. let equal_only = candidate("127.0.0.1:9002", equal_dist); assert_eq!( most_keyward_among(my_distance(), std::iter::once(equal_only)), None, "a neighbor at exactly our distance is farther-or-equal and must be excluded (pins `<`, not `<=`)" ); } #[test] fn none_when_no_strictly_closer_neighbor() { // Empty candidate set → None. assert_eq!( most_keyward_among(my_distance(), std::iter::empty()), None, "no candidates → no computed upstream" ); // All candidates farther than us → None (we are the terminus / stranded). let far_a = candidate("127.0.0.1:9001", 0.15); let far_b = candidate("127.0.0.1:9002", 0.30); assert_eq!( most_keyward_among(my_distance(), [far_a, far_b].into_iter()), None, "every candidate farther than us → no strictly-closer upstream" ); } #[test] fn deterministic_tiebreak_on_equal_distance() { // Two neighbors equidistant from the key (both 0.03, both strictly closer // than 0.10) but with different socket addresses. The tiebreak picks the // smaller socket address, and the result is independent of iteration // order. let low = candidate("127.0.0.1:9001", 0.03); let high = candidate("127.0.0.1:9002", 0.03); let want = low.0.clone(); let picked_fwd = most_keyward_among(my_distance(), [low.clone(), high.clone()].into_iter()); let picked_rev = most_keyward_among(my_distance(), [high, low].into_iter()); assert_eq!( picked_fwd, Some(want.clone()), "equidistant tie must resolve to the smaller socket address" ); assert_eq!( picked_rev, picked_fwd, "the tiebreak must be deterministic regardless of iteration order" ); } } #[cfg(test)] mod sleep_or_shutdown_tests { use super::sleep_or_shutdown; use std::time::Duration; use tokio_util::sync::CancellationToken; /// The wrapped future completes before shutdown → returns `false` (carry on). #[tokio::test(start_paused = true)] async fn returns_false_when_wait_completes_first() { let shutdown = CancellationToken::new(); let cancelled = sleep_or_shutdown(&shutdown, tokio::time::sleep(Duration::from_millis(10))).await; assert!( !cancelled, "an uncancelled token must let the wrapped sleep complete" ); } /// Shutdown fires mid-wait → returns `true` (stop the loop) and does NOT /// block for the full sleep. Regression for the #4278 failure mode: a /// 5-minute sleep that ignores shutdown. #[tokio::test(start_paused = true)] async fn returns_true_promptly_when_cancelled_mid_wait() { let shutdown = CancellationToken::new(); let token = shutdown.clone(); // Cancel after a short virtual delay, well before the long sleep ends. tokio::spawn(async move { tokio::time::sleep(Duration::from_millis(10)).await; token.cancel(); }); let start = tokio::time::Instant::now(); let cancelled = sleep_or_shutdown(&shutdown, tokio::time::sleep(Duration::from_secs(300))).await; let elapsed = start.elapsed(); assert!(cancelled, "a cancelled token must short-circuit the sleep"); assert!( elapsed < Duration::from_secs(300), "shutdown must not wait out the full 5-minute sleep (waited {elapsed:?})" ); } /// Already-cancelled token → returns `true` immediately, even with a long /// wait. `CancellationToken` is level-triggered, so a task that reaches the /// helper after shutdown already fired still bails (no missed-wakeup race). #[tokio::test(start_paused = true)] async fn returns_true_when_already_cancelled() { let shutdown = CancellationToken::new(); shutdown.cancel(); let cancelled = sleep_or_shutdown(&shutdown, tokio::time::sleep(Duration::from_secs(300))).await; assert!( cancelled, "an already-cancelled token must short-circuit immediately" ); } } /// Wiring pin for the hosting-observability rows on `network_efficiency_v1`. /// /// The counters themselves are unit-tested in `ring::hosting` (attribution) and /// `tracing::telemetry` (serialization). What NEITHER can see is the assignment /// in `router_snapshot_telemetry` that connects them: drop it, or feed a row /// from the wrong source, and every one of those tests stays green while the /// fleet publishes zeros or the wrong series. That is the #4009/#4010 /// manually-mirrored-telemetry footgun, and the snapshot loop is a 30-minute /// background cadence inside a fully-built `Ring`, which is why it is pinned by /// source scrape rather than executed. #[cfg(test)] mod hosting_observability_wiring_pin { /// Production source only: the needles below appear in this module too, and /// a pin that matches its own source is a pin that can never fail. fn production_source() -> &'static str { const FULL: &str = include_str!("ring.rs"); let cutoff = FULL .find("\n#[cfg(test)]\nmod ") .expect("ring.rs must have a top-level #[cfg(test)] mod section"); &FULL[..cutoff] } #[test] fn network_efficiency_block_is_fed_from_the_hosting_cache_stats() { let src = production_source(); for (field, source_expr) in [ ("host_begin", "hosting.hosting_begins"), ("host_reads", "hosting.read_count_hist"), ("host_recency", "hosting.genuine_access_recency"), ] { let needle = format!("{field}: {source_expr},"); assert!( src.contains(&needle), "`NetworkEfficiencyV1.{field}` must be populated as `{needle}` in \ the router_snapshot block. Without this assignment the hosting \ counters are collected and never exported, and every unit test \ for them still passes — the exact failure mode of #4009/#4010." ); } } } /// End-to-end seam test for cost-aware eviction (#4861 / #4903 review): /// drives the message axis through the REAL production reporter /// (`Ring::report_contract_resource_usage` → topology meter) and the REAL /// `Ring::sweep_expired_hosting()` (→ `hosting_cost_pressure_axes` → /// `HostingManager::sweep_expired_hosting_with_cost`) — the exact /// reporter → meter → axes → sweep chain the manager-level storm test /// (`storm_frequency_profile_crosses_cost_trigger_through_real_meter`) /// bypasses by assembling its axes by hand from a standalone `Meter`. #[cfg(test)] mod cost_pressure_seam_tests { use std::time::Duration; fn seam_key(seed: u8) -> freenet_stdlib::prelude::ContractKey { freenet_stdlib::prelude::ContractKey::from_id_and_code( freenet_stdlib::prelude::ContractInstanceId::new([seed; 32]), freenet_stdlib::prelude::CodeHash::new([seed.wrapping_add(1); 32]), ) } /// A real `OpManager` whose ring has been attached the production way /// (`Ring::attach_op_manager`), for the #5647/#5781 wiring tests. async fn attached_op_manager(id: &str) -> std::sync::Arc { let config_args = crate::config::ConfigArgs { id: Some(id.to_string()), mode: Some(crate::contract::OperationMode::Local), ..Default::default() }; let node_config = crate::node::NodeConfig::new(config_args.build().await.expect("build Config")) .await .expect("build NodeConfig"); let (_notification_rx, notification_tx) = crate::node::event_loop_notification_channel(); let (ops_ch_channel, _ch_channel, _wait_for_event) = crate::contract::contract_handler_channel(); let connection_manager = crate::ring::ConnectionManager::new(&node_config); let (result_router_tx, _result_router_rx) = tokio::sync::mpsc::channel(100); let task_monitor = crate::node::background_task_monitor::BackgroundTaskMonitor::new(); let op_manager = std::sync::Arc::new( crate::node::OpManager::new( notification_tx, ops_ch_channel, &node_config, crate::tracing::DynamicRegister::new(vec![]), connection_manager, result_router_tx, &task_monitor, ) .expect("build OpManager"), ); op_manager.ring.attach_op_manager(&op_manager); op_manager } fn wiring_key(seed: u32) -> freenet_stdlib::prelude::ContractKey { use freenet_stdlib::prelude::{CodeHash, ContractInstanceId, ContractKey}; let mut id = [7u8; 32]; id[..4].copy_from_slice(&seed.to_le_bytes()); ContractKey::from_id_and_code(ContractInstanceId::new(id), CodeHash::new([8u8; 32])) } /// #5647: `attach_op_manager` must install the interest-bytes provider, so /// the hosting cache charges each hosted contract the neighbour-summary /// bytes the REAL interest manager holds for it. Goes through the /// production sweep entry (`Ring::sweep_expired_hosting`). Without the /// provider the cache counts only its fixed per-entry bytes and the /// resident axis would silently stop seeing summaries, with every other /// test still green. #[tokio::test] async fn attach_op_manager_wires_interest_bytes_into_the_hosting_cache() { use freenet_stdlib::prelude::StateSummary; let op_manager = attached_op_manager("interest-bytes-wiring-5647").await; let key = wiring_key(0); op_manager.ring.hosting_manager.record_contract_access( key, 10, crate::ring::hosting::AccessType::Get, crate::ring::hosting::HostingCause::Other, ); op_manager.interest_manager.register_local_hosting(&key); let neighbour = crate::ring::PeerKey::from(crate::transport::TransportKeypair::new().public().clone()); assert!(op_manager.interest_manager.upsert_peer_summary( &key, &neighbour, StateSummary::from(vec![0u8; 4096]), )); let _ = op_manager.ring.sweep_expired_hosting(); let held = op_manager.interest_manager.resident_bytes_for(&key); assert!(held > 4096); assert_eq!( op_manager .ring .hosting_manager .hosting_cache_stats() .resident_overhead_bytes, crate::ring::hosting::HOSTED_ENTRY_BYTES + held, "the hosting cache must charge the summary bytes the interest manager holds" ); } fn wiring_peer() -> crate::ring::PeerKey { crate::ring::PeerKey::from(crate::transport::TransportKeypair::new().public().clone()) } fn host_for_wiring( op_manager: &crate::node::OpManager, key: freenet_stdlib::prelude::ContractKey, ) { op_manager.ring.hosting_manager.record_contract_access( key, 10, crate::ring::hosting::AccessType::Get, crate::ring::hosting::HostingCause::Other, ); op_manager.interest_manager.register_local_hosting(&key); } /// #5781 review blocker, at the 64 MiB floor budget: identities acting /// together, sending identical and distinct oversized summaries for 600 /// hosted contracts, cannot push the resident axis over budget, so they /// cannot get any contract evicted. 600 contracts at the 128 KiB /// per-contract cap would be 75 MiB; the node-wide budget (a quarter of /// the resident budget, enforced at write time) is what keeps the total /// under 64 MiB, and the counter never exceeds it. #[tokio::test] async fn colluding_peers_flooding_summaries_cannot_push_hosting_over_budget() { use freenet_stdlib::prelude::StateSummary; const MIB: u64 = 1024 * 1024; let op_manager = attached_op_manager("summary-flood-5781").await; let hosting = &op_manager.ring.hosting_manager; // A 512 MiB limit at the default share: the 64 MiB floor budget. hosting.configure_resident_overhead_mem_share(0.125); let budget = hosting.recompute_resident_overhead_budget(512 * MIB); assert_eq!(budget, 64 * MIB); let node_cap = budget / crate::ring::interest::NEIGHBOUR_SUMMARY_BUDGET_DIVISOR; // What the sweep installs; installed here so the flood below is // judged against the floor budget from the first write. op_manager .interest_manager .set_neighbour_summary_budget(node_cap); let im = &op_manager.interest_manager; let colluders: Vec<_> = (0..8).map(|_| wiring_peer()).collect(); let hosted = 600u32; let cap = crate::ring::interest::FALLBACK_CONTRACT_SUMMARY_CAP as usize; for i in 0..hosted { let key = wiring_key(i); host_for_wiring(&op_manager, key); for (n, peer) in colluders.iter().enumerate() { // 1 MiB each, identical across identities: over the cap. im.upsert_peer_summary(&key, peer, StateSummary::from(vec![0u8; MIB as usize])); assert!(im.neighbour_summary_bytes() <= node_cap); // Distinct per identity, each under the cap but together over it. im.upsert_peer_summary(&key, peer, StateSummary::from(vec![n as u8 + 1; cap / 2])); assert!(im.neighbour_summary_bytes() <= node_cap); // Identical and exactly at the cap: stored once if it fits. im.upsert_peer_summary(&key, peer, StateSummary::from(vec![0xAA; cap])); assert!(im.neighbour_summary_bytes() <= node_cap); } } assert!( im.neighbour_summary_bytes() > 0, "the flood was partly admitted" ); let mut with_summaries = 0; for i in 0..hosted { let held = im.distinct_summary_bytes_for(&wiring_key(i)); assert!(held <= cap as u64, "contract {i} holds {held} bytes"); if held > 0 { with_summaries += 1; } } assert!(with_summaries > 0 && with_summaries < hosted); let _ = op_manager.ring.sweep_expired_hosting(); let stats = hosting.hosting_cache_stats(); assert!( stats.resident_overhead_bytes <= budget, "the colluders pushed the resident axis to {} bytes against a {budget} budget", stats.resident_overhead_bytes ); assert_eq!( stats.contract_count, u64::from(hosted), "nothing was evicted" ); assert_eq!(stats.resident_overhead_evictions_total, 0); } /// #5781: `Ring::attach_op_manager` installs the node-wide summary budget /// at once (a real value, never 0 or the unset `u64::MAX`), and every /// sweep re-installs it from the current resident budget. #[tokio::test] async fn neighbour_summary_budget_is_installed_at_attach_and_each_sweep() { const MIB: u64 = 1024 * 1024; let op_manager = attached_op_manager("summary-budget-install-5781").await; let hosting = &op_manager.ring.hosting_manager; let im = &op_manager.interest_manager; let at_attach = im.neighbour_summary_budget(); assert_eq!( at_attach, hosting.resident_overhead_budget_bytes() / crate::ring::interest::NEIGHBOUR_SUMMARY_BUDGET_DIVISOR ); assert!(at_attach > 0 && at_attach < u64::MAX); hosting.configure_resident_overhead_mem_share(0.125); assert_eq!( hosting.recompute_resident_overhead_budget(512 * MIB), 64 * MIB ); let _ = op_manager.ring.sweep_expired_hosting(); assert_eq!(im.neighbour_summary_budget(), 16 * MIB); } /// #5781: on a node at capacity the summary budget stays a quarter of the /// resident budget and the sweep keeps the neighbours' summaries; it does /// not wipe them to clear the breach. Relieving the breach is left to the /// ordinary demand-ordered eviction once it has lasted the sustained /// window (covered by the hosting-cache eviction tests; this sweep runs /// before that window, so nothing is evicted yet). #[tokio::test] async fn a_node_at_capacity_keeps_its_neighbour_summaries() { use freenet_stdlib::prelude::StateSummary; const MIB: u64 = 1024 * 1024; let op_manager = attached_op_manager("summary-at-capacity-5781").await; let hosting = &op_manager.ring.hosting_manager; let im = &op_manager.interest_manager; hosting.configure_resident_overhead_mem_share(0.125); let budget = hosting.recompute_resident_overhead_budget(512 * MIB); assert_eq!(budget, 64 * MIB); // Entries alone exceed the budget. let hosted = (budget / crate::ring::hosting::HOSTED_ENTRY_BYTES) as u32 + 100; for i in 0..hosted { hosting.record_contract_access( wiring_key(i), 10, crate::ring::hosting::AccessType::Get, crate::ring::hosting::HostingCause::Other, ); } for i in 0..50u32 { let key = wiring_key(i); im.register_local_hosting(&key); assert!(im.upsert_peer_summary( &key, &wiring_peer(), StateSummary::from(vec![i as u8; 10_000]) )); } let held = im.neighbour_summary_bytes(); assert_eq!(held, 50 * 10_000); let _ = op_manager.ring.sweep_expired_hosting(); assert_eq!( im.neighbour_summary_budget(), budget / crate::ring::interest::NEIGHBOUR_SUMMARY_BUDGET_DIVISOR, "the summary budget does not collapse at capacity" ); assert_eq!(im.neighbour_summary_bytes(), held, "no summary was trimmed"); assert_eq!(im.summary_bound_trim_totals(), (0, 0)); let stats = hosting.hosting_cache_stats(); assert!( stats.resident_overhead_bytes > budget, "the node is over budget" ); assert_eq!(stats.resident_overhead_evictions_total, 0); } /// #5781 relative cap, end to end through the production sweep: a /// neighbour's 100,000-byte summary fits the 128 KiB cap while our own /// summary is unknown. Once a delivery records our own 1,000-byte summary /// the cap is 69,536 bytes, and the next `Ring::sweep_expired_hosting` /// drops the oversized summary before the hosting cache charges it. #[tokio::test] async fn relative_summary_cap_is_enforced_before_hosting_charges() { use crate::ring::interest::SummaryPopulationSource; use freenet_stdlib::prelude::StateSummary; let op_manager = attached_op_manager("summary-relative-5781").await; let key = wiring_key(0); host_for_wiring(&op_manager, key); let neighbour = wiring_peer(); let us_to = wiring_peer(); let im = &op_manager.interest_manager; assert!(im.upsert_peer_summary(&key, &neighbour, StateSummary::from(vec![1u8; 100_000]))); im.upsert_peer_summary_from( &key, &us_to, StateSummary::from(vec![2u8; 1_000]), SummaryPopulationSource::Delivery, ); assert_eq!(im.distinct_summary_bytes_for(&key), 101_000); let _ = op_manager.ring.sweep_expired_hosting(); assert!( im.get_peer_summary(&key, &neighbour).is_none(), "the neighbour's summary is over 4 x ours + 64 KiB" ); assert_eq!(im.distinct_summary_bytes_for(&key), 1_000); assert_eq!( op_manager .ring .hosting_manager .hosting_cache_stats() .resident_overhead_bytes, crate::ring::hosting::HOSTED_ENTRY_BYTES + 2 * crate::ring::interest::PEER_INTEREST_ENTRY_BYTES + 1_000 ); } /// `Ring::add_connection`'s `bool` reports the READINESS-threshold /// crossing, not acceptance — it is `false` both when the ring rejects the /// connection and (far more often) when the connection is added while the /// node is already ready. #4787's promotion counter has to know which /// happened, so `add_connection_reporting` splits the two. This pins that /// split, because gating the counter on the plain `bool` — the obvious /// reading of a function returning `false` on rejection — would silently /// count almost no promotions at all. #[tokio::test] async fn add_connection_reporting_separates_added_from_readiness() { let config_args = crate::config::ConfigArgs { id: Some("add-conn-outcome-4787".to_string()), mode: Some(crate::contract::OperationMode::Local), ..Default::default() }; let node_config = crate::node::NodeConfig::new(config_args.build().await.expect("build Config")) .await .expect("build NodeConfig"); let (_notification_rx, notification_tx) = crate::node::event_loop_notification_channel(); let (ops_ch_channel, _ch_channel, _wait_for_event) = crate::contract::contract_handler_channel(); let connection_manager = crate::ring::ConnectionManager::new(&node_config); let (result_router_tx, _result_router_rx) = tokio::sync::mpsc::channel(100); let task_monitor = crate::node::background_task_monitor::BackgroundTaskMonitor::new(); let op_manager = std::sync::Arc::new( crate::node::OpManager::new( notification_tx, ops_ch_channel, &node_config, crate::tracing::DynamicRegister::new(vec![]), connection_manager, result_router_tx, &task_monitor, ) .expect("build OpManager"), ); op_manager.ring.attach_op_manager(&op_manager); op_manager .ring .connection_manager .set_own_addr_local_for_test("127.0.0.1:14101".parse().unwrap()); let ring = &op_manager.ring; let max = ring.connection_manager.max_connections; // Beyond max + LATTICE_OVERMAX_SLACK the ceiling is hard, so a margin // past that guarantees we observe a rejection. let attempts = max + super::connection_manager::LATTICE_OVERMAX_SLACK + 16; let mut added = 0usize; let mut ready_crossings = 0usize; let mut added_without_readiness = 0usize; let mut first_rejection: Option = None; for i in 0..attempts { let kp = crate::transport::TransportKeypair::new(); let addr: std::net::SocketAddr = format!("127.0.0.2:{}", 20000 + i as u16) .parse() .expect("addr"); let loc = super::Location::new((i as f64 + 0.5) / attempts as f64); let outcome = ring .add_connection_reporting(loc, super::PeerId::new(kp.public().clone(), addr), false) .await; if outcome.added { added += 1; if outcome.just_became_ready { ready_crossings += 1; } else { added_without_readiness += 1; } } else { first_rejection = Some(i); break; } } let rejected_at = first_rejection.expect( "the ring must eventually reject an add at the connection ceiling — \ without a rejection this test cannot show `added` is meaningful", ); assert!( rejected_at >= max, "rejection came at {rejected_at}, before max_connections={max}" ); assert!( added_without_readiness > 0, "the overwhelming majority of accepted adds report \ just_became_ready=false; if this is 0 the test proves nothing" ); assert!( ready_crossings <= 1, "readiness can be crossed at most once, got {ready_crossings}" ); assert!( added >= max, "expected at least {max} accepted adds, got {added}" ); } /// Runs under `start_paused` so `tokio::time::Instant` — the clock behind /// BOTH the Ring's production `InstantTimeSrc` (meter timestamps, sweep /// reads, hosting recency) and every background interval — is virtual and /// advanced deterministically with `tokio::time::advance`. The Ring's /// periodic background tasks (including the 60s hosting sweep, which runs /// this same seam) fire during the advances; the assertions are therefore /// on the FINAL hosting state, which is identical whether a background /// sweep or the explicit call below sheds the storm contract. #[tokio::test(start_paused = true)] async fn zero_demand_storm_is_shed_through_real_reporter_and_sweep() { use crate::topology::meter::ResourceType; // --- a real OpManager (and thus a real Ring with its production // InstantTimeSrc), mirroring the fixture in node.rs // (`resync_request_for_bogus_keys_does_not_consume_limiter_slots`). --- let config_args = crate::config::ConfigArgs { id: Some("cost-seam-4903".to_string()), mode: Some(crate::contract::OperationMode::Local), ..Default::default() }; let node_config = crate::node::NodeConfig::new(config_args.build().await.expect("build Config")) .await .expect("build NodeConfig"); let (_notification_rx, notification_tx) = crate::node::event_loop_notification_channel(); let (ops_ch_channel, _ch_channel, _wait_for_event) = crate::contract::contract_handler_channel(); let connection_manager = crate::ring::ConnectionManager::new(&node_config); let (result_router_tx, _result_router_rx) = tokio::sync::mpsc::channel(100); let task_monitor = crate::node::background_task_monitor::BackgroundTaskMonitor::new(); let op_manager = std::sync::Arc::new( crate::node::OpManager::new( notification_tx, ops_ch_channel, &node_config, crate::tracing::DynamicRegister::new(vec![]), connection_manager, result_router_tx, &task_monitor, ) .expect("build OpManager"), ); op_manager.ring.attach_op_manager(&op_manager); op_manager .ring .connection_manager .set_own_addr_local_for_test("127.0.0.1:14100".parse().unwrap()); let ring = &op_manager.ring; let junk = seam_key(1); let subscribed = seam_key(2); let read_hot = seam_key(3); // Host all three through the production entry point (PUT seeds). let _ = ring.host_contract( junk, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); let _ = ring.host_contract( subscribed, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); let _ = ring.host_contract( read_hot, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); assert!(ring.is_hosting_contract(&junk), "precondition: hosted"); // Age the PUT-seed recency stamps past the cost window, so only the // demand signals added BELOW protect anything. tokio::time::advance(super::hosting::COST_RATE_MIN_WINDOW + Duration::from_secs(1)).await; // Demand: `subscribed` gains a downstream subscriber (lease outlives // this test's remaining ~3 virtual minutes); `read_hot` is genuinely // GET-read now (within the cost window of the final sweep). assert!(!matches!( ring.add_downstream_subscriber( &subscribed, crate::ring::interest::PeerKey(crate::transport::TransportPublicKey::from_bytes( [9u8; 32] )), ), crate::ring::hosting::AddSubscriberOutcome::Rejected )); let _ = ring.record_get_access(read_hot, 121, crate::ring::HostingCause::Other); // The FX2j storm profile, reported for ALL THREE contracts through // the production reporter: one 58-target fan-out dispatch every 1.6s // for ~3 minutes (~36 per-peer sends/s sustained). Equal rates prove // protection comes from demand, not from a smaller cost share. for _ in 0..110u32 { for key in [&junk, &subscribed, &read_hot] { ring.report_contract_resource_usage( *key.id(), ResourceType::BroadcastMessagesSent, 58.0, ); } tokio::time::advance(Duration::from_millis(1600)).await; } // The REAL sweep (the same call the 60s background sweep task makes). let _ = ring.sweep_expired_hosting(); assert!( !ring.is_hosting_contract(&junk), "the zero-demand storm contract must be shed through the real \ reporter → meter → hosting_cost_pressure_axes → sweep seam" ); assert!( ring.is_hosting_contract(&subscribed), "a subscribed contract storming at the SAME rate must survive \ (candidacy, not ranking, protects it)" ); assert!( ring.is_hosting_contract(&read_hot), "a recently-GET-read contract storming at the SAME rate must \ survive (reads are demand — invariant 3)" ); } /// Everything a seam test needs kept alive for its whole run: the OpManager /// plus the channel ends and task monitor whose drop would tear the Ring's /// background tasks down mid-test. struct SeamFixture { op_manager: std::sync::Arc, node_events: tokio::sync::mpsc::Receiver< either::Either, >, /// The remaining channel ends and the task monitor. Never read — held /// only so dropping them does not tear the Ring's background tasks down /// mid-test. Boxed opaquely because several of these types are private to /// their own modules. _keep_alive: Box, } /// A real `OpManager` over a real `Ring` (production `InstantTimeSrc`, real /// background tasks), with the node-event receiver handed back so a test can /// assert on EMITTED events rather than on internal bookkeeping. `id` /// isolates the on-disk state in its own temp dir. async fn seam_fixture(id: &str) -> SeamFixture { let config_args = crate::config::ConfigArgs { id: Some(id.to_string()), mode: Some(crate::contract::OperationMode::Local), ..Default::default() }; let node_config = crate::node::NodeConfig::new(config_args.build().await.expect("build Config")) .await .expect("build NodeConfig"); let (notification_rx, notification_tx) = crate::node::event_loop_notification_channel(); let (ops_ch_channel, _ch_channel, _wait_for_event) = crate::contract::contract_handler_channel(); let connection_manager = crate::ring::ConnectionManager::new(&node_config); let (result_router_tx, result_router_rx) = tokio::sync::mpsc::channel(100); let task_monitor = crate::node::background_task_monitor::BackgroundTaskMonitor::new(); let op_manager = std::sync::Arc::new( crate::node::OpManager::new( notification_tx, ops_ch_channel, &node_config, crate::tracing::DynamicRegister::new(vec![]), connection_manager, result_router_tx, &task_monitor, ) .expect("build OpManager"), ); op_manager.ring.attach_op_manager(&op_manager); SeamFixture { op_manager, node_events: notification_rx.notifications_receiver, _keep_alive: Box::new(( notification_rx.op_execution_receiver, _ch_channel, _wait_for_event, result_router_rx, task_monitor, )), } } /// Every contract id retracted by a `HostingAnnounce` queued on the node-event /// channel so far. This is the wire-visible effect co-hosts act on: it is what /// makes them drop us from their fan-out target set. fn drain_retracted_ids( fixture: &mut SeamFixture, ) -> Vec { let mut retracted = Vec::new(); while let Ok(event) = fixture.node_events.try_recv() { if let either::Either::Right(crate::message::NodeEvent::BroadcastHostingUpdate { message: crate::message::NeighborHostingMessage::HostingAnnounce { removed, .. }, }) = event { retracted.extend(removed); } } retracted } /// Run the maintenance loop's post-sweep step for one sweep result, in the /// same order as `Ring::sweep_get_subscription_cache`: drop the upstream lease /// for each expired key, THEN reclaim it. The order is load-bearing now that /// the retraction treats a live lease as "still a host" — a helper that /// skipped the unsubscribe would suppress the very retraction under test and /// pass for the wrong reason. fn reclaim_swept( op_manager: &crate::node::OpManager, swept: crate::ring::hosting::HostingSweepResult, ) { for (key, expected_generation) in swept.expired { op_manager.ring.unsubscribe(&key); crate::operations::reclaim_evicted_contract(op_manager, key, expected_generation); } } /// #5059: a cost-pressure eviction must RETRACT the evicted contract's co-host /// advertisement, not just drop its hosting-cache entry. /// /// Broadcast fan-out targets are resolved from `neighbor_hosting` (advertised /// co-hosts) with no interest check, so an evicted-but-still-advertised /// contract keeps being sent updates and keeps applying them. That burns the /// CPU the eviction was supposed to reclaim, AND it bumps the state generation /// on every apply — so the deferred `EvictContract` bails at /// `RuntimePool::remove_contract`'s newer-generation guard and the retraction /// wired behind that guard never runs. Field evidence in #5040: the storm /// contract's per-update warn ran at ~10-11/s continuously through three /// separate cost evictions of it. /// /// The assertion is on the EMITTED EFFECT — a `BroadcastHostingUpdate` /// carrying `HostingAnnounce { removed: [junk] }` — rather than on /// `my_contracts`, because the wire message is what makes co-hosts stop /// sending. /// /// Nothing in this fixture consumes the `EvictContract` event, which is the /// point: it stands in for the field case where reclamation never reaches its /// own retraction. Before this fix the test sees no retraction at all. /// /// `start_paused` for the same reason as the sibling seam test above: the /// virtual clock drives the meter, the sweep, and the background tasks /// together. A background hosting sweep may shed `junk` before the explicit /// sweep below; either way the retraction has to reach the channel, which is /// why the assertion scans everything queued rather than a single event. #[tokio::test(start_paused = true)] async fn cost_evicted_contract_emits_a_co_host_advertisement_retraction() { use crate::topology::meter::ResourceType; let mut fixture = seam_fixture("cost-retraction-5059").await; let op_manager = fixture.op_manager.clone(); op_manager .ring .connection_manager .set_own_addr_local_for_test("127.0.0.1:14101".parse().unwrap()); let ring = &op_manager.ring; let junk = seam_key(1); let subscribed = seam_key(2); let _ = ring.host_contract( junk, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); let _ = ring.host_contract( subscribed, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); // Both advertised to co-hosts — the state this fix is about. Set directly // (not via `announce_contract_hosted`) so the only `HostingAnnounce` the // channel can carry is a retraction. op_manager.neighbor_hosting.on_contract_hosted(&junk); op_manager.neighbor_hosting.on_contract_hosted(&subscribed); assert!( op_manager.neighbor_hosting.is_hosted_locally(&junk), "precondition: the storm contract is advertised to co-hosts" ); tokio::time::advance(super::hosting::COST_RATE_MIN_WINDOW + Duration::from_secs(1)).await; assert!(!matches!( ring.add_downstream_subscriber( &subscribed, crate::ring::interest::PeerKey(crate::transport::TransportPublicKey::from_bytes( [9u8; 32] )), ), crate::ring::hosting::AddSubscriberOutcome::Rejected )); // The FX2j storm profile through the production reporter, identical for // both contracts so demand — not cost share — decides who is shed. for _ in 0..110u32 { for key in [&junk, &subscribed] { ring.report_contract_resource_usage( *key.id(), ResourceType::BroadcastMessagesSent, 58.0, ); } tokio::time::advance(Duration::from_millis(1600)).await; } // The background hosting sweep may have shed `junk` (and reclaimed it) // during the advances above, in which case this explicit sweep finds // nothing left to evict and the retraction is already queued. Either way // it has to be on the channel, which is why the drain below covers the // whole run rather than only the window around this call. Nothing here // ever announces hosting, so every `HostingAnnounce` on the channel is a // retraction. reclaim_swept(&op_manager, ring.sweep_expired_hosting()); assert!( !ring.is_hosting_contract(&junk), "precondition for the assertions below: the storm contract was shed" ); let retracted = drain_retracted_ids(&mut fixture); assert!( retracted.contains(junk.id()), "the cost-evicted contract must emit a co-host advertisement retraction \ — without it the fan-out never stops and the eviction reclaims nothing \ (#5059). Retractions seen: {retracted:?}" ); assert!( !retracted.contains(subscribed.id()), "a contract that was never evicted must not be retracted" ); assert!( !op_manager.neighbor_hosting.is_hosted_locally(&junk), "the evicted contract must also leave the locally-advertised set, or the \ ~5-min full-set re-request re-asserts the phantom advertisement" ); assert!( op_manager.neighbor_hosting.is_hosted_locally(&subscribed), "the surviving subscribed contract keeps its advertisement" ); } /// The eviction retraction must NOT fire for a contract that is hosted again, /// wanted again, or back in the update mesh by the time it runs — the three /// conditions that genuinely mean "still a host". Retracting any of them would /// leave a fresh, in-mesh contract unadvertised, which the ~5-min full-set /// re-request cannot heal because that exchange replays the same (wrong) /// `my_contracts`. /// /// All three go through the real `reclaim_evicted_contract` funnel, but they /// are stopped at different points: re-hosted and leased by the retraction /// helper's own guard, in-use by `reclaim_evicted_contract`'s pre-existing /// `contract_in_use` early return (the helper re-checks that one too, for the /// window between that return and the removal — covered directly by /// `on_contract_unhosted_unless_rehosted_retracts_only_when_still_unhosted`). #[tokio::test(start_paused = true)] async fn eviction_retraction_skips_a_rehosted_or_in_use_contract() { let mut fixture = seam_fixture("cost-retraction-guards-5059").await; let op_manager = fixture.op_manager.clone(); op_manager .ring .connection_manager .set_own_addr_local_for_test("127.0.0.1:14102".parse().unwrap()); let ring = &op_manager.ring; // (1) Re-hosted: present in the hosting cache when the reclaim runs. let rehosted = seam_key(4); let _ = ring.host_contract( rehosted, 121, crate::ring::AccessType::Put, crate::ring::HostingCause::Other, ); op_manager.neighbor_hosting.on_contract_hosted(&rehosted); // (2) Back in use: absent from the hosting cache, but a downstream // subscriber re-registered interest. let in_use = seam_key(5); op_manager.neighbor_hosting.on_contract_hosted(&in_use); assert!(!matches!( ring.add_downstream_subscriber( &in_use, crate::ring::interest::PeerKey(crate::transport::TransportPublicKey::from_bytes( [7u8; 32] )), ), crate::ring::hosting::AddSubscriberOutcome::Rejected )); assert!( !ring.is_hosting_contract(&in_use) && ring.contract_in_use(&in_use), "precondition: in-use but not in the hosting cache" ); // (3) Re-subscribed: no cache entry and no subscriber of its own, but a // live upstream lease, so it IS in the update mesh. This is the SUBSCRIBE // path's shape — `finalize_originator_subscribe` installs the lease, then // announces on the body being present ON DISK, and the ring-level client // subscription lands later from `client_events`. Without the lease check // a pending-reclamation retry in that window retracts the announce right // back out, leaving the node subscribed, holding the body, unadvertised. let resubscribed = seam_key(6); op_manager .neighbor_hosting .on_contract_hosted(&resubscribed); ring.subscribe(resubscribed); assert!( !ring.is_hosting_contract(&resubscribed) && !ring.contract_in_use(&resubscribed) && ring.is_subscribed(&resubscribed), "precondition: leased only — not cached, no subscriber of our own" ); let _ = drain_retracted_ids(&mut fixture); for key in [rehosted, in_use, resubscribed] { let generation = ring.state_generation(&key); crate::operations::reclaim_evicted_contract(&op_manager, key, generation); } let retracted = drain_retracted_ids(&mut fixture); assert!( retracted.is_empty(), "none of a re-hosted, in-use, or leased contract may be retracted; \ got {retracted:?}" ); assert!( op_manager.neighbor_hosting.is_hosted_locally(&rehosted), "a contract re-hosted since the eviction decision keeps its advertisement" ); assert!( op_manager.neighbor_hosting.is_hosted_locally(&in_use), "a contract back in use since the eviction decision keeps its advertisement" ); assert!( op_manager.neighbor_hosting.is_hosted_locally(&resubscribed), "a contract holding a live upstream lease keeps its advertisement — it \ is in the update mesh, so it is a host under invariant 1" ); } /// Drift guard (#4861 Codex round-3): the meter's per-axis cost floors /// ([`crate::topology::meter::ResourceType::cost_pressure_floor`]) and its /// sustained window ([`crate::topology::meter::COST_SUSTAINED_WINDOW`]) are /// MIRRORS — the authoritative constants live in `ring::hosting::cache`, /// which is not nameable from `topology::meter` (that module is private to /// `hosting`, itself private to `ring`), so the meter cannot import them. /// This test reads the authoritative values through the `ring`-visible /// surface — `build_cost_axes` (which binds each axis to its cache.rs floor /// constant) and the re-exported `COST_RATE_MIN_WINDOW` — and asserts the /// meter's mirrors match, so the insert-time above-floor detection can /// never key on a floor/window that differs from the eviction decision's. #[test] fn meter_cost_floors_mirror_cache_source_of_truth() { use crate::topology::meter::{COST_SUSTAINED_WINDOW, ResourceType}; fn empty_read() -> ( f64, std::collections::HashMap, ) { (0.0, std::collections::HashMap::new()) } // build_cost_axes argument/return order: cpu, fanout_bytes, messages. let axes = super::hosting::build_cost_axes(empty_read(), empty_read(), empty_read()); assert_eq!( axes[0].floor, ResourceType::ExecCpuMicros .cost_pressure_floor() .expect("ExecCpuMicros is a cost axis"), "CPU floor mirror drifted from cache.rs source of truth", ); assert_eq!( axes[1].floor, ResourceType::BroadcastFanoutCost .cost_pressure_floor() .expect("BroadcastFanoutCost is a cost axis"), "fan-out byte floor mirror drifted from cache.rs source of truth", ); assert_eq!( axes[2].floor, ResourceType::BroadcastMessagesSent .cost_pressure_floor() .expect("BroadcastMessagesSent is a cost axis"), "message floor mirror drifted from cache.rs source of truth", ); assert_eq!( super::hosting::COST_RATE_MIN_WINDOW, COST_SUSTAINED_WINDOW, "sustained-window mirror drifted from cache.rs COST_RATE_MIN_WINDOW", ); } /// Governance cost ingestion accepts ONLY `StateBytesWritten` and REJECTS /// the three #4861 cost-eviction axes (and the peer/fuel axes). Pinning /// this prevents silently re-feeding governance a popularity-scaled cost /// axis, which reproduces the #4296 "popular contract banned for being /// popular" false positive (see `governance_ingests_axis` rustdoc and the /// `popular_contract_with_subscribers_not_evicted` sim test). Exhaustive so /// a new ResourceType variant must be classified with the same guard. #[test] fn governance_ingests_only_state_bytes_written() { use crate::topology::meter::ResourceType; assert!( super::governance_ingests_axis(ResourceType::StateBytesWritten), "governance MUST ingest StateBytesWritten (its main-calibrated basis)" ); for axis in [ ResourceType::ExecCpuMicros, ResourceType::BroadcastFanoutCost, ResourceType::BroadcastMessagesSent, ] { assert!( !super::governance_ingests_axis(axis), "governance MUST NOT ingest the #4861 cost-eviction axis {axis:?} \ (popularity-scaled → resurrects #4296); it feeds the meter/sweep only" ); } for axis in [ ResourceType::InboundBandwidthBytes, ResourceType::OutboundBandwidthBytes, ResourceType::ExecFuelUnits, ] { assert!( !super::governance_ingests_axis(axis), "peer/fuel axis {axis:?} is not per-contract cost; governance must not ingest it" ); } } /// Source-scrape pin: the governance feed in `report_contract_resource_usage_batch` /// MUST stay gated by `governance_ingests_axis`, so a future edit can't drop /// the gate and re-flood governance with the #4861 cost-eviction axes. #[test] fn governance_feed_is_gated_by_axis_allowlist() { let src = include_str!("ring.rs"); let start = src .find("fn report_contract_resource_usage_batch(") .expect("report_contract_resource_usage_batch not found"); let body_end = src[start..] .find("\n // Note: there is intentionally no `ingest_contract_demand`") .expect("end of report_contract_resource_usage_batch not found") + start; let body = &src[start..body_end]; let ingest = body .find(".governance.ingest_cost(") .expect("governance.ingest_cost call missing from the batch feed"); let gate = body .find("if !governance_ingests_axis(") .expect("governance feed MUST be gated by governance_ingests_axis (#4296 guard)"); assert!( gate < ingest, "the governance_ingests_axis gate MUST precede (guard) the \ ingest_cost call, else the #4861 axes re-flood governance and \ resurrect #4296" ); } } #[derive(thiserror::Error, Debug)] pub(crate) enum RingError { #[error(transparent)] ConnError(#[from] Box), /// Retained for completeness; client_events maps it to a stable /// `ClientError::EmptyRing`. Currently unconstructed. #[error("No ring connections found")] #[allow(dead_code)] EmptyRing, #[error("Ran out of, or haven't found any, hosting peers for contract {0}")] NoHostingPeers(ContractInstanceId), #[error("Peer has not joined the network yet (no ring location established)")] PeerNotJoined, } #[cfg(test)] mod timeout_label_window_tests { use super::{TIMEOUT_LABEL_WINDOW_MAX_PEERS, TimeoutLabelWindow}; use std::net::SocketAddr; fn addr(i: usize) -> SocketAddr { SocketAddr::from(([10, (i >> 16) as u8, (i >> 8) as u8, i as u8], 4000)) } #[test] fn histogram_buckets_labels_per_peer_and_resets() { let mut window = TimeoutLabelWindow::default(); // peer 0: 1 label, peer 1: 3, peer 2: 5, peer 3: 9 for (peer, labels) in [(0, 1), (1, 3), (2, 5), (3, 9)] { for _ in 0..labels { window.record(addr(peer)); } } assert_eq!(window.take_histogram(), (1, 1, 1, 1, 9, 0)); assert_eq!( window.take_histogram(), (0, 0, 0, 0, 0, 0), "each snapshot window starts empty" ); } #[test] fn window_is_bounded_and_counts_overflow() { let mut window = TimeoutLabelWindow::default(); for i in 0..TIMEOUT_LABEL_WINDOW_MAX_PEERS + 10 { window.record(addr(i)); } // A tracked peer keeps counting past the cap. window.record(addr(0)); assert_eq!(window.per_peer.len(), TIMEOUT_LABEL_WINDOW_MAX_PEERS); let (b1, b2, _, _, max, untracked) = window.take_histogram(); assert_eq!(untracked, 10); assert_eq!(b1 + b2, TIMEOUT_LABEL_WINDOW_MAX_PEERS as u64); assert_eq!(max, 2); } } #[cfg(test)] mod hosting_stats_mirror_source_tests { //! Source-scrape pin: every `HostingCacheStats` field must be mirrored into //! `RouterSnapshotInfo` by `emit_router_snapshot_telemetry`, INTO THE FIELD //! THAT MATCHES IT. //! //! The telemetry path is hand-mirrored twice over (`HostingCacheStats` -> //! `RouterSnapshotInfo` -> the OTLP JSON body), and nothing in the type //! system connects the hops. The resident-overhead pressure axis (#5325) //! was computed, rendered on the node's own dashboard, and dropped on the //! floor at THIS step for its whole life: the collector could not see the //! second eviction pressure at all, so a node evicting purely under slot //! pressure looked idle in fleet telemetry. //! //! The pin asserts the WHOLE assignment, not merely that the field is read //! somewhere in the function — the same shape as //! `contract_exec_export_maps_each_field_to_its_own_counter`, and for the //! same reason its doc gives: a swap "compiles cleanly and emits a plausible //! number that is measuring the opposite thing". Here a swapped pair would //! report every node as over its resident-overhead budget. A presence-only //! check is also satisfied by `let _ = hosting.x;`, by an assignment into the //! WRONG destination (which additionally clobbers a live gauge), and by a //! field name left behind in a comment. //! //! It checks the FIRST hop only. The second hop (`RouterSnapshotInfo` -> //! JSON) is guarded by the per-gauge pins in `tracing::telemetry`, including //! `router_snapshot_json_includes_resident_overhead_gauges`; it has no //! structural pin of its own, which is a known gap, not an oversight. /// Fields whose destination is a key in the `NetworkEfficiencyV1` struct /// literal rather than a `snapshot.hosting_*` assignment, with that key. /// These are the histogram arms; their names are deliberately abbreviated at /// the destination, so the mechanical `hosting_` rule does not apply /// and the expected statement has to be spelled out. const STRUCT_LITERAL_DESTINATIONS: &[(&str, &str)] = &[ ("eviction_victim_counts", "vict_n"), ("eviction_victim_bytes", "vict_b"), ("hosting_begins", "host_begin"), ("read_count_hist", "host_reads"), ("genuine_access_recency", "host_recency"), ]; /// Fields deliberately NOT mirrored to telemetry. Each entry needs a reason: /// adding one is a decision to make a stat invisible to the collector, which /// is exactly what this pin exists to stop happening by accident. /// /// Empty today, so `not_mirrored_exclusions_carry_a_reason` below cannot /// currently fail. That is deliberate — it arms the escape hatch before /// anyone uses it — but it means the test name overstates present coverage; /// do not read it as evidence that anything was checked. const DELIBERATELY_NOT_MIRRORED: &[(&str, &str)] = &[]; /// Every field on `HostingCacheStats`. An exact count, not a floor: the /// scrape below can only under-count (a declaration shape it cannot parse), /// and a floor set below the true count lets exactly that go unnoticed. /// Adding a field means bumping this deliberately AND mirroring the field. // 19 since #5647: `contract_slot_budget` was removed (no per-contract // constant to divide by), the estimated field was renamed // `resident_overhead_bytes`, and `resident_overhead_evicted_charged_bytes_total` // was added. const EXPECTED_HOSTING_CACHE_STATS_FIELDS: usize = 19; fn production_source() -> &'static str { const FULL: &str = include_str!("ring.rs"); // ring.rs has inline `#[cfg(test)]` annotations on individual `use` // declarations near the top of the file, so a plain // `find("#[cfg(test)]")` would cut the file off before the function we // want to scrape. Anchor on the first *top-level* test module instead. // Keep this comment: without it the next editor "simplifies" the anchor // and the scrape silently starts returning the wrong region. let cutoff = FULL .find("\n#[cfg(test)]\nmod ") .expect("ring.rs must have a top-level #[cfg(test)] mod section"); &FULL[..cutoff] } /// Body of the item starting at `signature_prefix`, brace-balanced. Bounding /// to the item is load-bearing: an unbounded search over an 8000-line file /// would match this module's own assertion strings and pass vacuously. (The /// `production_source` cutoff makes that structurally impossible here too, /// since this module sits past it — belt and braces.) fn item_body<'a>(source: &'a str, signature_prefix: &str) -> &'a str { let start = source .find(signature_prefix) .unwrap_or_else(|| panic!("could not find {signature_prefix}")); let brace = source[start..].find('{').expect("item must have a body"); let body_start = start + brace + 1; let bytes = source.as_bytes(); let mut depth: i32 = 1; let mut i = body_start; while i < bytes.len() { match bytes[i] { b'{' => depth += 1, b'}' => { depth -= 1; if depth == 0 { return &source[body_start..i]; } } _ => {} } i += 1; } panic!("unbalanced braces while extracting {signature_prefix}"); } /// `body` with `//` comments removed and whitespace collapsed. /// /// Both halves are load-bearing. Comments must go because `str::contains` /// over raw source cannot tell a statement from `// snapshot.hosting_x = /// Some(hosting.x);`, so commenting a mirror out — what someone actually /// does while debugging, which is precisely when the pin is the only thing /// still watching — would otherwise keep it green. Whitespace must collapse /// because rustfmt wraps these assignments across two lines. /// /// Stripping `//` inside a string literal would be wrong, but only in the /// strict direction: it can make the pin fail spuriously, never pass /// spuriously. There are no such literals in the scraped function today. fn strip_comments_and_normalize(body: &str) -> String { let uncommented: Vec<&str> = body .lines() .map(|line| match line.find("//") { Some(at) => &line[..at], None => line, }) .collect(); uncommented .join(" ") .split_whitespace() .collect::>() .join(" ") } /// Field names declared on `HostingCacheStats`, in declaration order. /// /// Fails LOUDLY on any line it does not recognise rather than skipping it. /// A scraper that silently drops an unparseable declaration fails OPEN: the /// field is never checked for a mirror, which is the exact omission this /// module exists to catch. `pub(crate)` is the shape that bit an earlier /// draft — `HostingCacheStats` is itself `pub(crate)`, so a contributor /// writing `pub(crate)` on a field is entirely natural. fn hosting_cache_stats_fields() -> Vec { const CACHE_SRC: &str = include_str!("ring/hosting/cache.rs"); let body = item_body(CACHE_SRC, "pub(crate) struct HostingCacheStats {"); let mut fields = Vec::new(); for raw in body.lines() { let mut line = raw.trim(); if line.is_empty() || line.starts_with("///") || line.starts_with("//") { continue; } // An inline attribute (`#[doc(hidden)] pub x: u64,`) must not hide a // field: strip the attribute and keep reading the same line. while let Some(rest) = line.strip_prefix("#[") { match rest.find(']') { Some(close) => line = rest[close + 1..].trim(), None => break, } } if line.is_empty() { continue; } let decl = line .strip_prefix("pub(crate) ") .or_else(|| line.strip_prefix("pub ")) .unwrap_or_else(|| { panic!( "unrecognised line in HostingCacheStats: {raw:?}. Every line in \ that struct must be blank, a comment, an attribute, or a \ `pub`/`pub(crate)` field declaration. If a new shape is \ legitimate, teach this scraper about it — do NOT let it skip \ the line, because an unscraped field is never checked for a \ mirror, which is the omission this module exists to catch." ) }); let (name, _) = decl.split_once(':').unwrap_or_else(|| { panic!("HostingCacheStats field declaration has no `:`: {raw:?}") }); fields.push(name.trim().to_string()); } assert_eq!( fields.len(), EXPECTED_HOSTING_CACHE_STATS_FIELDS, "scraped {} HostingCacheStats fields, expected {}: {fields:?}. If you added \ or removed a field, bump EXPECTED_HOSTING_CACHE_STATS_FIELDS deliberately \ (and mirror the new field). If you did not, the scrape has gone wrong and \ this pin is measuring less than it claims — fix the scrape, do not relax \ the count.", fields.len(), EXPECTED_HOSTING_CACHE_STATS_FIELDS, ); fields } /// The exact statement that must appear for `field`. fn expected_mirror(field: &str) -> String { match STRUCT_LITERAL_DESTINATIONS .iter() .find(|(name, _)| *name == field) { Some((_, key)) => format!("{key}: hosting.{field},"), None => format!("snapshot.hosting_{field} = Some(hosting.{field});"), } } #[test] fn hosting_cache_stats_fields_are_all_mirrored() { let body = item_body( production_source(), "async fn emit_router_snapshot_telemetry(", ); let norm = strip_comments_and_normalize(body); // The mirror reads every stat off one binding. Anchoring on it gives a // directed failure if the read moves, instead of 19 confusing ones. assert!( norm.contains("let hosting = ring.hosting_manager.hosting_cache_stats();"), "emit_router_snapshot_telemetry must bind the hosting stats as \ `hosting`; if that binding was renamed, update this pin's expected \ statements too — they all read through it." ); let mut missing = Vec::new(); for field in hosting_cache_stats_fields() { if DELIBERATELY_NOT_MIRRORED .iter() .any(|(name, _)| *name == field) { continue; } let expected = expected_mirror(&field); if !norm.contains(&expected) { missing.push(expected); } } assert!( missing.is_empty(), "HostingCacheStats fields not mirrored into RouterSnapshotInfo by \ emit_router_snapshot_telemetry. Expected statements not found: \ {missing:#?}\n\nEvery stat must reach `RouterSnapshotInfo` (and from \ there the OTLP body) into the field that matches it, or be listed in \ DELIBERATELY_NOT_MIRRORED with a reason. A stat that exists but is not \ exported is invisible to fleet telemetry — see #5325, where the \ resident-overhead eviction axis was dropped exactly here. A stat \ exported under the WRONG name is worse: it reads as a plausible number \ measuring the opposite thing." ); } /// The expected-statement table must actually describe the destinations, not /// merely be self-consistent: every `STRUCT_LITERAL_DESTINATIONS` entry names /// a real `HostingCacheStats` field. A stale entry here would silently excuse /// a field from the mechanical rule without anyone noticing. #[test] fn struct_literal_destinations_name_real_fields() { let fields = hosting_cache_stats_fields(); for (field, key) in STRUCT_LITERAL_DESTINATIONS { assert!( fields.iter().any(|f| f == field), "STRUCT_LITERAL_DESTINATIONS names {field:?} (destination key \ {key:?}), which is not a HostingCacheStats field. Remove the stale \ entry — while it is here, a field by that name is exempt from the \ mechanical `snapshot.hosting_` rule for no reason." ); } } /// The exclusion list is an escape hatch, so make using it deliberate: an /// entry with no reason is the shape that turns this pin ornamental. Note /// this cannot fail while the list is empty — see the const's doc. #[test] fn not_mirrored_exclusions_carry_a_reason() { for (name, reason) in DELIBERATELY_NOT_MIRRORED { assert!( reason.len() > 20, "DELIBERATELY_NOT_MIRRORED entry {name:?} needs a real reason, \ got {reason:?}" ); } } /// The two properties the scrape helpers must have, checked directly rather /// than inferred from the pin passing: a commented-out mirror must not count, /// and rustfmt's line wrapping must not stop a mirror counting. #[test] fn normalization_drops_comments_and_rejoins_wrapped_statements() { let commented = " // snapshot.hosting_x = Some(hosting.x);\n"; assert!( !strip_comments_and_normalize(commented) .contains("snapshot.hosting_x = Some(hosting.x);"), "a commented-out mirror must not satisfy the pin" ); let wrapped = " snapshot.hosting_x =\n Some(hosting.x);\n"; assert!( strip_comments_and_normalize(wrapped).contains("snapshot.hosting_x = Some(hosting.x);"), "a mirror rustfmt wrapped across two lines must still satisfy the pin" ); let trailing = " snapshot.hosting_x = Some(hosting.x); // why\n"; assert!( strip_comments_and_normalize(trailing) .contains("snapshot.hosting_x = Some(hosting.x);"), "a trailing comment must not hide the statement in front of it" ); } /// Asserting the whole statement is what makes a transposition visible; a /// presence-only check (`body.contains("hosting.")`) would not see it, /// and a swapped pair here would report every node as over its /// resident-overhead budget. Checked directly so the property is evidenced /// rather than asserted in a doc comment. #[test] fn expected_mirror_pins_the_destination_not_just_the_read() { let e = expected_mirror("resident_overhead_budget_bytes"); assert_eq!( e, "snapshot.hosting_resident_overhead_budget_bytes = \ Some(hosting.resident_overhead_budget_bytes);" ); // A transposed assignment does NOT contain the expected statement. let transposed = "snapshot.hosting_resident_overhead_budget_bytes = \ Some(hosting.resident_overhead_bytes);"; assert!(!transposed.contains(&e)); // Neither does a bare read, nor a read into the wrong destination. assert!(!"let _ = hosting.resident_overhead_budget_bytes;".contains(&e)); assert!( !"snapshot.hosting_current_bytes = Some(hosting.resident_overhead_budget_bytes);" .contains(&e) ); // The histogram arms keep their abbreviated destination keys. assert_eq!( expected_mirror("read_count_hist"), "host_reads: hosting.read_count_hist," ); } } /// The ring's selection functions are where candidate logging is wired to the /// op that routed; the router- and writer-level tests cannot see a wrong op, a /// dropped `record` call or a probe that logs. #[cfg(test)] pub(crate) mod candidate_log_wiring_tests { use std::net::SocketAddr; use std::sync::Arc; use freenet_stdlib::prelude::{CodeHash, ContractInstanceId, ContractKey}; use crate::node::network_status::OpType; use crate::operations::route_attempt::driver_test_support::op_manager_with_peers; use crate::ring::{Location, PeerKeyLocation}; use crate::router::dataset::{self, DecisionLog, RoutingDataset, UncapturedReason}; use crate::router::{RouteEvent, RouteOutcome}; pub(crate) fn recorder() -> (Arc, tempfile::TempDir, std::path::PathBuf) { let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("routing.jsonl"); let recorder = RoutingDataset::open(&path, dataset::DEFAULT_MAX_BYTES).unwrap(); (Arc::new(recorder), dir, path) } /// Every line up to and including a sentinel written last, so an extra /// trailing line cannot be missed by reading too early. pub(crate) fn lines_through_sentinel( recorder: &RoutingDataset, path: &std::path::Path, ) -> Vec { const SENTINEL_T_MS: u64 = 424_242; recorder.record_peers(SENTINEL_T_MS, Vec::new()); dataset::lines_eventually(path, |lines| { lines .iter() .any(|line| line["kind"] == "peers" && line["t_ms"] == SENTINEL_T_MS) }) } /// Enough routing history that decisions are prediction-based. fn warm_router(ring: &super::Ring, peers: &[PeerKeyLocation], contract: Location) { let mut router = ring.router.write(); for i in 0..120 { router.add_event(RouteEvent { peer: peers[i % peers.len()].clone(), contract_location: contract, outcome: if i % 7 == 0 { RouteOutcome::Failure } else { RouteOutcome::SuccessUntimed }, op_type: Some(OpType::Get), }); } } fn selected_at(line: &serde_json::Value, position: usize) -> String { line["candidates"] .as_array() .unwrap() .iter() .find(|c| c["selected_position"] == position) .unwrap_or_else(|| panic!("no candidate at position {position}: {line}"))["peer"] .as_str() .unwrap() .to_string() } fn contract_key() -> ContractKey { ContractKey::from_id_and_code(ContractInstanceId::new([7u8; 32]), CodeHash::new([0u8; 32])) } /// The peer page's per-peer eligible/chosen counts take only real routing /// decisions, from both ring entry points; probes and pre-selections /// (`DecisionLog::Unlogged`) leave them untouched. #[tokio::test] async fn only_routing_decisions_count_toward_per_peer_selection() { let (op_manager, _rx, peers, _guards) = op_manager_with_peers("selection-counts", 5).await; let ring = &op_manager.ring; let key = contract_key(); warm_router(ring, &peers, Location::from(&key)); let none: Vec = Vec::new(); let totals = || { let router = ring.router.read(); peers .iter() .fold((0u64, 0.0f64), |(eligible, chosen), peer| { let selection = router.peer_snapshot(peer).selection.unwrap_or_default(); (eligible + selection.eligible, chosen + selection.chosen) }) }; ring.k_closest_potentially_hosting(DecisionLog::Unlogged, key.id(), none.as_slice(), 2); ring.closest_potentially_hosting(DecisionLog::Unlogged, &key, none.as_slice()); assert_eq!( totals(), (0, 0.0), "probes and pre-selections are not counted" ); ring.closest_potentially_hosting(DecisionLog::Joinable(OpType::Put), &key, none.as_slice()) .expect("a peer is selected"); let (single, chosen) = totals(); assert!( single >= 1 && chosen == 1.0, "the single-peer entry point counts" ); ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Get), key.id(), none.as_slice(), 2, ); let (both, chosen) = totals(); assert!(both > single, "the k-closest entry point counts"); assert_eq!( chosen, 2.0, "one first choice per decision, even with k = 2" ); } #[tokio::test] async fn ring_selections_log_only_routing_decisions_with_their_op() { let (op_manager, _rx, peers, _guards) = op_manager_with_peers("candidate-wiring", 5).await; let ring = &op_manager.ring; let key = contract_key(); let contract = Location::from(&key); warm_router(ring, &peers, contract); let (recorder, _dir, path) = recorder(); let none: Vec = Vec::new(); let all: Vec = peers.iter().filter_map(|p| p.socket_addr()).collect(); let (subscribe, put, get) = { let _log = dataset::force_candidate_log(recorder.clone(), 1.0); // Probes and pre-selections log nothing. assert_eq!( ring.k_closest_potentially_hosting( DecisionLog::Unlogged, key.id(), none.as_slice(), 2 ) .len(), 2 ); assert!( ring.closest_potentially_hosting(DecisionLog::Unlogged, &key, none.as_slice()) .is_some() ); // A selection of nobody routes nowhere and logs nothing. assert!( ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Get), key.id(), all.as_slice(), 1 ) .is_empty() ); let subscribe = ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Subscribe), key.id(), none.as_slice(), 2, ); let put = ring .closest_potentially_hosting( DecisionLog::Joinable(OpType::Put), &key, none.as_slice(), ) .unwrap(); let get = ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Get), key.id(), none.as_slice(), 1, ); (subscribe, put, get) }; let again = { // Below rate 1 the test override never draws a capture: this GET // decision is uncaptured. Tied costs are shuffled, so it may pick a // different peer than the capture above; every peer is made a live // GET capture first, so whichever it picks is closed. for peer in &peers { recorder.record_decision(live_capture(peer, OpType::Get, contract)); } let _log = dataset::force_candidate_log(recorder.clone(), 0.5); let again = ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Get), key.id(), none.as_slice(), 1, ); assert_eq!(again.len(), 1); dataset::record_bypass( OpType::Subscribe, contract, &subscribe[0], UncapturedReason::DirectedFirstHop, ); again }; let lines = lines_through_sentinel(&recorder, &path); let kinds: Vec<&str> = lines.iter().map(|l| l["kind"].as_str().unwrap()).collect(); assert_eq!( kinds, [ "start", "decision", "decision", "decision", // The live GET captures of every peer. "decision", "decision", "decision", "decision", "decision", "decision_uncaptured", "decision_uncaptured", "peers" ], "{lines:?}" ); let (sub_line, put_line, get_line) = (&lines[1], &lines[2], &lines[3]); assert_eq!(sub_line["op"], "SUBSCRIBE"); assert_eq!(sub_line["contract_location"], contract.as_f64()); assert_eq!(sub_line["k"], 2); assert_eq!(sub_line["candidates_considered"], peers.len()); assert_eq!(selected_at(sub_line, 0), dataset::peer_hash(&subscribe[0])); assert_eq!(selected_at(sub_line, 1), dataset::peer_hash(&subscribe[1])); assert_eq!(put_line["op"], "PUT"); assert_eq!(put_line["k"], 1); assert_eq!(selected_at(put_line, 0), dataset::peer_hash(&put)); assert_eq!(get_line["op"], "GET"); assert_eq!(selected_at(get_line, 0), dataset::peer_hash(&get[0])); let (superseded, bypass) = (&lines[9], &lines[10]); assert_eq!(superseded["op"], "GET"); assert_eq!(superseded["reason"], "sampled_out"); assert_eq!( superseded["selected"], serde_json::json!([dataset::peer_hash(&again[0])]) ); assert_eq!(bypass["op"], "SUBSCRIBE"); assert_eq!(bypass["reason"], "directed_first_hop"); assert_eq!( bypass["selected"], serde_json::json!([dataset::peer_hash(&subscribe[0])]) ); } /// A router with too little history ranks by distance; the ring must say /// so rather than blame sampling. #[tokio::test] async fn a_cold_router_decision_is_logged_as_distance_based() { let (op_manager, _rx, _peers, _guards) = op_manager_with_peers("candidate-cold", 5).await; let ring = &op_manager.ring; let key = contract_key(); let contract = Location::from(&key); let (recorder, _dir, path) = recorder(); let none: Vec = Vec::new(); let _log = dataset::force_candidate_log(recorder.clone(), 1.0); let chosen = ring .closest_potentially_hosting(DecisionLog::Unlogged, &key, none.as_slice()) .unwrap(); // Make the selection a live capture, so the uncaptured line is written. recorder.record_decision(live_capture(&chosen, OpType::Put, contract)); let again = ring .closest_potentially_hosting(DecisionLog::Joinable(OpType::Put), &key, none.as_slice()) .unwrap(); assert_eq!(again, chosen); let lines = lines_through_sentinel(&recorder, &path); let kinds: Vec<&str> = lines.iter().map(|l| l["kind"].as_str().unwrap()).collect(); assert_eq!( kinds, ["start", "decision", "decision_uncaptured", "peers"], "{lines:?}" ); assert_eq!(lines[2]["reason"], "distance_based"); assert_eq!(lines[2]["op"], "PUT"); } /// Pacing must run on the ring's injected clock: under a paused runtime /// only that clock moves, so a capture refused as paced is allowed again /// once virtual time passes. #[tokio::test(flavor = "current_thread", start_paused = true)] async fn ring_pacing_reads_the_injected_clock() { let (op_manager, _rx, peers, _guards) = op_manager_with_peers("candidate-paced", 5).await; let ring = &op_manager.ring; let key = contract_key(); warm_router(ring, &peers, Location::from(&key)); let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("routing.jsonl"); // A 40 KB budget released over an hour: the burst is 2.5 KB, less // than one captured decision. let recorder = Arc::new( RoutingDataset::open_with_decisions( &path, dataset::DEFAULT_MAX_BYTES, 40_000, 3_600_000, ) .unwrap(), ); let _log = dataset::force_candidate_log(recorder.clone(), 1.0); let none: Vec = Vec::new(); // Selecting every peer makes each call the same selection whatever // order tied costs are shuffled into, so the second call's selections // are live and its pacing is written as a line. let get = || { ring.k_closest_potentially_hosting( DecisionLog::Joinable(OpType::Get), key.id(), none.as_slice(), peers.len(), ) }; let kinds = |lines: &[serde_json::Value]| -> Vec { lines .iter() .map(|l| l["kind"].as_str().unwrap().to_string()) .collect() }; // Each call's line is awaited before the next call, so the lines pin // which call was captured and which was paced. assert_eq!(get().len(), peers.len()); dataset::lines_eventually(&path, |lines| kinds(lines) == ["start", "decision"]); assert_eq!(get().len(), peers.len()); dataset::lines_eventually(&path, |lines| { kinds(lines) == ["start", "decision", "decision_uncaptured"] }); tokio::time::advance(std::time::Duration::from_secs(3600)).await; assert_eq!(get().len(), peers.len()); let lines = lines_through_sentinel(&recorder, &path); assert_eq!( kinds(&lines), [ "start", "decision", "decision_uncaptured", "decision", "peers" ], "{lines:?}" ); assert_eq!(lines[2]["reason"], "paced", "the second call is paced"); } /// A captured decision for `peer` alone, as a warm router would write it. pub(crate) fn live_capture( peer: &PeerKeyLocation, op: OpType, contract: Location, ) -> dataset::DecisionRecord { dataset::DecisionCapture { contract_location: contract, acting_model: dataset::RoutingModel::Legacy, prediction_fallback: false, k: 1, candidates_available: 1, prior_failure_events: 0, candidates: vec![dataset::CapturedCandidate { peer, legacy: None, hierarchical: None, hierarchical_stages: dataset::HierarchicalStages::default(), selected_position: Some(0), }], } .into_record(op, 1) } }