From d2af6a398141aef99616c042780ff8beea9feecf Mon Sep 17 00:00:00 2001 From: Duncan Tourolle Date: Fri, 28 Aug 2026 23:08:16 +0200 Subject: [PATCH] Time out on a stalled transfer, not on a slow one MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The HTTP client had a 60-second *total* request timeout. That is not a hang detector; it is a floor on link speed. A face shard runs to 25 MB, so it demanded a sustained 425 KB/s or the transfer failed — and having failed it was retried on the next pass and failed again, for ever. A tablet on ordinary wifi could therefore never finish taking in a library's faces, and nothing said why: each attempt looked like a network blip rather than an arithmetic impossibility. The catalog snapshot is 36 MB and has the same problem. `read_timeout` fires when no bytes arrive for the period, which is the condition actually worth failing on. A slow transfer that is still moving now finishes, however long it takes; a connection that has genuinely died is still caught in a minute. The connect timeout is unchanged. Co-Authored-By: Claude Opus 5 (1M context) --- core/dr-sync-nextcloud/src/lib.rs | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/core/dr-sync-nextcloud/src/lib.rs b/core/dr-sync-nextcloud/src/lib.rs index 8db855e..7cda5e9 100644 --- a/core/dr-sync-nextcloud/src/lib.rs +++ b/core/dr-sync-nextcloud/src/lib.rs @@ -637,11 +637,23 @@ pub fn http_client(user_agent: &str) -> Result { // webpki-roots set is still installed alongside. .tls_certs_only(extra_roots()) // A request that hangs forever is indistinguishable from a worker that - // died, and cost a long time to tell apart once. Connect and total - // timeouts turn that into an error the UI can show. Generous enough for - // a slow phone on mobile data; the login poll has its own deadline. + // died, and cost a long time to tell apart once. These turn that into + // an error the UI can show. .connect_timeout(std::time::Duration::from_secs(15)) - .timeout(std::time::Duration::from_secs(60)) + // **Inactivity, not duration.** This was a 60-second *total* timeout, + // which is not a hang detector at all — it is a floor on link speed. + // A face shard runs to 25 MB, so it demanded a sustained 425 KB/s or + // the transfer failed; and having failed it was retried on the next + // pass, and failed again, for ever. A tablet on ordinary wifi could + // therefore never finish adopting a library's faces, and nothing said + // why: each attempt looked like a network blip rather than an + // arithmetic impossibility. + // + // `read_timeout` fires when *no bytes arrive* for the given period, + // which is the condition actually worth failing on. A slow transfer + // that is still moving now finishes, however long it takes, while a + // connection that has genuinely died is still caught in a minute. + .read_timeout(std::time::Duration::from_secs(60)) .build() .map_err(|e| RemoteError::Network(e.to_string())) }