//! TRACES: FR-CAT-9 | FR-NC-12 //! Whether the remote is reachable, and what the app does while it is not. //! //! # Why this is a state machine rather than a boolean //! //! "Are we online?" cannot be answered by asking the operating system. A //! laptop with a live wifi association and no route, a captive portal that //! answers every request with a login page, a Nextcloud instance that is down //! while the internet is fine — all three report a working network and none of //! them can serve an image. The only evidence that counts is whether *this //! backend* answered, so reachability is inferred from the traffic the app was //! already making rather than probed for separately. //! //! That inversion is what keeps the cost at zero. Every remote call already //! returns a `Result`; [`Reachability::observe`] turns those results into the //! state, so a library that is browsing happily never issues a probe at all. //! A probe happens only when something failed and the app wants to know //! whether it has come back (ARCH §9.0). //! //! # Why leaving offline is harder than entering it //! //! One failed request is enough to go offline: the user is *already* //! experiencing the failure, and the honest thing is to say so immediately. //! But a single success is not enough to declare recovery, because the failure //! mode that matters — a flapping connection — produces exactly that. So //! recovery requires a deliberate probe, and the app backs off between //! attempts rather than retrying in a tight loop against a server that is //! plainly down. use std::time::{Duration, Instant}; use crate::error::RemoteError; /// How long to wait before the first reconnection probe. /// /// Short enough that a brief drop — a laptop changing access points, a phone /// moving between cells — recovers before the user has finished noticing it. const FIRST_BACKOFF: Duration = Duration::from_secs(5); /// The longest gap between probes. /// /// Capped rather than growing without bound: a user who left the app open /// overnight on a dead connection should reconnect within a minute of the /// server returning, not hours later because the backoff had doubled its way /// into the distance. const MAX_BACKOFF: Duration = Duration::from_secs(60); /// What the app currently believes about the remote. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Connectivity { /// The remote answered the last time anything asked. Online, /// The remote could not be reached. The library runs from local data. Offline, } impl Connectivity { pub fn is_online(self) -> bool { matches!(self, Connectivity::Online) } pub fn is_offline(self) -> bool { matches!(self, Connectivity::Offline) } } /// Tracks reachability from observed request outcomes. /// /// Cheap to construct and `Clone`-free by design — one lives beside the /// session and every worker reports into it. #[derive(Debug)] pub struct Reachability { state: Connectivity, /// When the app went offline. Shown to the user, because "offline" without /// "since when" leaves them unable to tell a momentary drop from a /// connection that died an hour ago. since: Option, /// How long to wait before the next probe, doubling per failure. backoff: Duration, /// When the next probe becomes worthwhile. next_probe: Option, /// Why we think we are offline, for the banner. The underlying transport /// message, which is usually specific enough to be actionable ("dns error", /// "connection refused"). reason: Option, } impl Default for Reachability { fn default() -> Self { Self::new() } } impl Reachability { /// Start optimistic. /// /// Assuming online until proven otherwise is deliberate: the alternative /// is a probe on every launch, which makes startup wait on the network for /// a library that may be entirely cached. The first real request settles /// it either way, and settles it with evidence. pub fn new() -> Self { Self { state: Connectivity::Online, since: None, backoff: FIRST_BACKOFF, next_probe: None, reason: None, } } pub fn state(&self) -> Connectivity { self.state } pub fn is_offline(&self) -> bool { self.state.is_offline() } /// Why the app believes it is offline, if it does. pub fn reason(&self) -> Option<&str> { self.reason.as_deref() } /// How long the app has been offline. pub fn offline_for(&self, now: Instant) -> Option { self.since.map(|t| now.saturating_duration_since(t)) } /// Record the outcome of a remote call. /// /// Returns `true` if the connectivity state *changed*, so the caller can /// repaint a banner or kick off a rescan without diffing the state itself. /// /// Takes the result by reference so callers can report an outcome they are /// still going to use — this observes, it never consumes. pub fn observe(&mut self, outcome: &Result, now: Instant) -> bool { match outcome { Ok(_) => self.mark_reachable(now), Err(e) if e.indicates_offline() => self.mark_unreachable(e.to_string(), now), // The server answered. Whatever went wrong is not connectivity, so // it must not move this state — a 404 on one file says nothing // about the other 17,000. Err(_) => false, } } /// Record that the remote answered. pub fn mark_reachable(&mut self, _now: Instant) -> bool { let changed = self.state.is_offline(); self.state = Connectivity::Online; self.since = None; self.backoff = FIRST_BACKOFF; self.next_probe = None; self.reason = None; changed } /// Record that the remote could not be reached. /// /// Repeated calls while already offline extend the backoff rather than /// resetting it, so a library with twelve workers all failing at once /// does not schedule twelve immediate probes. pub fn mark_unreachable(&mut self, reason: String, now: Instant) -> bool { let changed = self.state.is_online(); if changed { self.state = Connectivity::Offline; self.since = Some(now); self.backoff = FIRST_BACKOFF; } else { self.backoff = (self.backoff * 2).min(MAX_BACKOFF); } self.next_probe = Some(now + self.backoff); self.reason = Some(reason); changed } /// Whether enough time has passed to be worth trying the remote again. /// /// Always false while online — there is nothing to probe for. pub fn should_probe(&self, now: Instant) -> bool { self.state.is_offline() && self.next_probe.is_some_and(|t| now >= t) } /// How long until the next probe is due, for a countdown in the banner. pub fn until_probe(&self, now: Instant) -> Option { self.next_probe.map(|t| t.saturating_duration_since(now)) } } #[cfg(test)] mod tests { use super::*; fn network() -> RemoteError { RemoteError::Network("connection refused".into()) } #[test] fn starts_online_without_probing() { // Launch must not wait on the network: a fully cached library opens // with no request at all, and an optimistic start is what allows that. let r = Reachability::new(); assert!(r.state().is_online()); assert!(!r.should_probe(Instant::now())); } #[test] fn one_network_failure_goes_offline() { let mut r = Reachability::new(); let now = Instant::now(); let changed = r.observe::<()>(&Err(network()), now); assert!(changed, "the first failure is a state change"); assert!(r.is_offline()); assert_eq!(r.reason(), Some("network error: connection refused")); } #[test] fn a_server_error_is_not_offline() { // The distinction the whole mode rests on: the server answered, so it // is reachable. Going offline here would blank the grid over one // forbidden file. let mut r = Reachability::new(); let now = Instant::now(); for e in [ RemoteError::PermissionDenied, RemoteError::NotFound("a.CR2".into()), RemoteError::AuthFailed, RemoteError::Server { status: 500, detail: String::new(), }, RemoteError::Server { status: 423, detail: String::new(), }, ] { let mut probe = Reachability::new(); assert!(!probe.observe::<()>(&Err(e), now)); assert!(probe.state().is_online()); } assert!(!r.observe::<()>(&Err(RemoteError::PermissionDenied), now)); assert!(r.state().is_online()); } #[test] fn success_brings_it_back() { let mut r = Reachability::new(); let now = Instant::now(); r.observe::<()>(&Err(network()), now); assert!(r.is_offline()); let changed = r.observe(&Ok(()), now); assert!(changed, "recovery is a state change"); assert!(r.state().is_online()); assert_eq!(r.reason(), None); } #[test] fn repeated_failures_back_off_rather_than_reset() { // Twelve sweep lanes failing together must not schedule twelve // immediate probes against a server that is plainly down. let mut r = Reachability::new(); let now = Instant::now(); assert!(r.observe::<()>(&Err(network()), now)); let first = r.until_probe(now).unwrap(); assert!(!r.observe::<()>(&Err(network()), now), "already offline"); let second = r.until_probe(now).unwrap(); assert!( second > first, "backoff must grow: {second:?} should exceed {first:?}" ); } #[test] fn backoff_is_capped() { let mut r = Reachability::new(); let now = Instant::now(); for _ in 0..20 { r.observe::<()>(&Err(network()), now); } assert!( r.until_probe(now).unwrap() <= MAX_BACKOFF, "an app left overnight must still reconnect promptly" ); } #[test] fn probing_waits_for_the_backoff() { let mut r = Reachability::new(); let now = Instant::now(); r.observe::<()>(&Err(network()), now); assert!(!r.should_probe(now), "not immediately"); assert!(r.should_probe(now + FIRST_BACKOFF)); } #[test] fn recovery_resets_the_backoff() { // Otherwise a connection that flaps all day arrives at the maximum // backoff and stays there, so the next real drop takes a minute to // notice recovery. let mut r = Reachability::new(); let now = Instant::now(); for _ in 0..5 { r.observe::<()>(&Err(network()), now); } r.observe(&Ok(()), now); r.observe::<()>(&Err(network()), now); assert_eq!(r.until_probe(now), Some(FIRST_BACKOFF)); } #[test] fn offline_duration_is_measured_from_the_first_failure() { // Not from the most recent one: a connection that has been down for an // hour must not report five seconds because a worker retried. let mut r = Reachability::new(); let start = Instant::now(); r.observe::<()>(&Err(network()), start); let later = start + Duration::from_secs(600); r.observe::<()>(&Err(network()), later); let elapsed = r .offline_for(later) .expect("offline since the first failure"); assert!( elapsed >= Duration::from_secs(600), "measured from the first failure, got {elapsed:?}" ); } }