From 1496a481993591f11af480bda568179bb9056ba6 Mon Sep 17 00:00:00 2001 From: DebaucheryLibrarian Date: Mon, 14 Sep 2026 20:08:55 +0200 Subject: [PATCH] Improved timeout handling. --- src/fetch/item.js | 35 +++++++++++++++-------------------- 1 file changed, 15 insertions(+), 20 deletions(-) diff --git a/src/fetch/item.js b/src/fetch/item.js index 65631b4..f8ff183 100644 --- a/src/fetch/item.js +++ b/src/fetch/item.js @@ -10,21 +10,6 @@ const { agentFor } = require('../proxy'); const logger = require('../logger')(__filename); const limiterFor = require('../limiter'); -// bhttp has no timeout at all by default - if the server (or, more likely, a proxy tunnel -// that connects fine but then never relays anything) just stops responding mid-request without -// erroring, the returned promise never settles. With the limiter's concurrency capped, that one -// stuck request blocks every other download queued behind it, indefinitely. Race it against a -// hard wall-clock ceiling so a stall fails (and retries, below) instead of hanging forever. -function withTimeout(promise, ms, url) { - let timer; - - const timedOut = new Promise((resolve, reject) => { - timer = setTimeout(() => reject(new Error(`Timed out after ${ms}ms fetching '${url}'`)), ms); - }); - - return Promise.race([promise, timedOut]).finally(() => clearTimeout(timer)); -} - async function fetchItem(url, attempt, context) { async function retry(error) { logger.warn(`Failed to fetch '${url}', ${attempt < config.fetch.retries ? 'retrying' : 'giving up'}: ${error.message} (${context.post ? context.post.permalink : 'no post'})`); @@ -42,11 +27,21 @@ async function fetchItem(url, attempt, context) { // whichever of proxied/direct is NOT what it'd normally get - see resolveProxyUrl() const flip = config.proxy.retryFlipped && attempt > 0; - const res = await limiterFor(url).schedule(async () => withTimeout( - bhttp.get(url, { headers: context.headers, agent: agentFor(url, { flip }) }), - config.fetch.timeout, - url, - )); + const res = await limiterFor(url).schedule(() => bhttp.get(url, { + headers: context.headers, + agent: agentFor(url, { flip }), + // bhttp has no timeout at all by default - a server (or, more likely, a proxy + // tunnel) that connects fine but then just stops responding would otherwise hang + // forever. This isn't just a wall-clock ceiling on the *promise* (an external + // Promise.race achieves that, but leaves the real connection dangling) - bhttp + // calls req.abort() itself when this fires, actually releasing the socket. Without + // that, every stalled attempt leaks one connection onto the shared agent for that + // host; enough of those piling up (e.g. after several retries against a host that's + // currently black-holing everything, as seen with i.redd.it) exhausts it, and every + // *future* request to that host silently queues forever waiting for a socket that's + // never coming back - a freeze with no further log lines, not just a slow retry. + responseTimeout: config.fetch.timeout, + })); if (res.statusCode !== 200) { throw new Error(`Response not OK for ${url} (${res.statusCode})`);