Librarian: survive transient Crossref failures instead of losing the search
CI / compile (pull_request) Successful in 12s
CI / unit (pull_request) Successful in 29s
CI / integration (pull_request) Successful in 31s
build / build (push) Successful in 33s
CI / compile (push) Successful in 15s
CI / unit (push) Successful in 30s
CI / integration (push) Successful in 34s
CI / compile (pull_request) Successful in 12s
CI / unit (pull_request) Successful in 29s
CI / integration (pull_request) Successful in 31s
build / build (push) Successful in 33s
CI / compile (push) Successful in 15s
CI / unit (push) Successful in 30s
CI / integration (push) Successful in 34s
Field report: one httpx ReadTimeout inside habanero surfaced as 'Search <uuid> crashed', and the worker's crash handler then FORGOT the request - so an expensive search vanished and the user was told it was eaten, because a public API blinked once. Two defences: * Every habanero call goes through _crossref_call, which retries with linear backoff (CONJURER_CROSSREF_ATTEMPTS, default 4; backoff CONJURER_CROSSREF_BACKOFF, 5s). habanero wraps httpx errors in a plain RuntimeError so we can't filter narrowly - retries are simply bounded and the last error is re-raised. They now also run via asyncio.to_thread, so a slow Crossref no longer blocks the worker's event loop. * A crashed search is no longer dropped on the first failure: the attempt count is persisted with the request and the search is requeued (keeping any checkpoint, so a crashed DB scan resumes rather than restarts) until CONJURER_SEARCH_MAX_ATTEMPTS (default 3). It stays 'queued' for the bot's watchdog while retrying, and only after the cap is it forgotten. Also: scrape_bot's 'Got blocked' is routine sci-hub behaviour (it backs off an hour and carries on) - log it as WARNING, not ERROR, so it stops looking like a fault when scanning for real problems. Tests: retry-then-succeed, bounded re-raise, no retry on success, the persisted attempt counter, and forget-on-give-up. Suite: 58 unit + 70 integration green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit was merged in pull request #22.
This commit is contained in:
@@ -232,6 +232,37 @@ def search_heartbeat(app_logger) -> None:
|
||||
app_logger.warning("Heartbeat tick failed: %s", exc)
|
||||
|
||||
|
||||
# Crossref is a public service that times out / rate-limits under load. A single
|
||||
# transient ReadTimeout used to blow up the whole (expensive) search, so every
|
||||
# habanero call is retried with backoff, and a search that still fails is retried
|
||||
# as a whole a few times before being given up on.
|
||||
CROSSREF_ATTEMPTS = int(_env("CONJURER_CROSSREF_ATTEMPTS", "4"))
|
||||
CROSSREF_BACKOFF = float(_env("CONJURER_CROSSREF_BACKOFF", "5"))
|
||||
SEARCH_MAX_ATTEMPTS = int(_env("CONJURER_SEARCH_MAX_ATTEMPTS", "3"))
|
||||
|
||||
|
||||
def _crossref_call(app_logger, what, func, *args, **kwargs):
|
||||
"""Run one habanero call, retrying transient failures with linear backoff.
|
||||
|
||||
habanero wraps httpx errors (ReadTimeout, connection resets, 5xx) in a plain
|
||||
RuntimeError, so we cannot filter narrowly - we retry a BOUNDED number of
|
||||
times on any failure and re-raise the last error if none succeed. Blocking
|
||||
on purpose: callers invoke it via asyncio.to_thread, which also keeps the
|
||||
worker's event loop free while Crossref is slow."""
|
||||
last_exc = None
|
||||
for attempt in range(1, max(1, CROSSREF_ATTEMPTS) + 1):
|
||||
try:
|
||||
return func(*args, **kwargs)
|
||||
except Exception as exc: # pylint: disable=broad-exception-caught
|
||||
last_exc = exc
|
||||
app_logger.warning(
|
||||
"Crossref %s failed (attempt %d/%d): %s", what, attempt, CROSSREF_ATTEMPTS, exc
|
||||
)
|
||||
if attempt < CROSSREF_ATTEMPTS:
|
||||
time.sleep(CROSSREF_BACKOFF * attempt)
|
||||
raise last_exc
|
||||
|
||||
|
||||
def _forget_search(uuid) -> None:
|
||||
"""A search is fully done (or abandoned): drop its persisted request and any
|
||||
checkpoint so it is never replayed or resumed again."""
|
||||
@@ -470,14 +501,20 @@ class Librarian(object):
|
||||
|
||||
if not deep_search:
|
||||
query_limit = MAX_CR_RESULTS if MAX_CR_RESULTS < 1000 else 1000
|
||||
cr_result = self.cr.works(query=query, limit=query_limit)
|
||||
cr_result = await asyncio.to_thread(
|
||||
_crossref_call, self.app.logger, "works",
|
||||
self.cr.works, query=query, limit=query_limit,
|
||||
)
|
||||
self.search_result_from_cr.update(cr_result)
|
||||
self.total = cr_result["message"]["total-results"]
|
||||
self.fetched += len(cr_result["message"]["items"])
|
||||
self.app.logger.info(self.total)
|
||||
self.app.logger.info(self.fetched)
|
||||
while self.total > self.fetched and self.limit > self.fetched:
|
||||
tmp_result = self.cr.works(query=query, limit=query_limit, offset=self.fetched)
|
||||
tmp_result = await asyncio.to_thread(
|
||||
_crossref_call, self.app.logger, "works(offset)",
|
||||
self.cr.works, query=query, limit=query_limit, offset=self.fetched,
|
||||
)
|
||||
cr_result["message"]["items"].extend(tmp_result["message"]["items"])
|
||||
self.total = tmp_result["message"]["total-results"]
|
||||
self.fetched = len(cr_result["message"]["items"])
|
||||
@@ -486,7 +523,10 @@ class Librarian(object):
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
else:
|
||||
cr_result = self.cr.works(query=query, cursor_max=15000, cursor='*', progress_bar = True)
|
||||
cr_result = await asyncio.to_thread(
|
||||
_crossref_call, self.app.logger, "works(deep cursor)",
|
||||
self.cr.works, query=query, cursor_max=15000, cursor='*', progress_bar=True,
|
||||
)
|
||||
result = cr_result[0]
|
||||
for item in cr_result[1:]:
|
||||
result["message"]["items"].extend(item["message"]["items"])
|
||||
@@ -766,6 +806,7 @@ class BackgroundTaskSearch(threading.Thread):
|
||||
# behind us, and active_queries so the bot's watchdog can tell a
|
||||
# finished-and-gone query from one still in flight.
|
||||
worker_busy.set()
|
||||
requeued = False # set when a crash schedules another attempt
|
||||
with _active_lock:
|
||||
active_queries[str(librarian.uuid)] = "processing"
|
||||
try:
|
||||
@@ -827,16 +868,48 @@ class BackgroundTaskSearch(threading.Thread):
|
||||
# forget its request + checkpoint (never replay/resume it again).
|
||||
_forget_search(uuid)
|
||||
except Exception as exc: # pylint: disable=broad-exception-caught
|
||||
# A crashing search must not kill the worker thread (which would
|
||||
# freeze the whole queue). Log and give up on it - forget the
|
||||
# request/checkpoint so it isn't retried forever as a poison pill;
|
||||
# the bot's watchdog tells the user it vanished.
|
||||
self.app.logger.exception("Search %s crashed: %s", librarian.uuid, exc)
|
||||
_forget_search(str(librarian.uuid))
|
||||
# A crashing search must not kill the worker thread (that would
|
||||
# freeze the whole queue). It also must not silently vanish just
|
||||
# because Crossref timed out once: retry the whole search a
|
||||
# bounded number of times (attempt count persisted with the
|
||||
# request, so it can't loop forever), keeping any checkpoint so a
|
||||
# crashed DB scan resumes rather than restarts. Only after
|
||||
# SEARCH_MAX_ATTEMPTS do we give up and let the bot's watchdog
|
||||
# tell the user it vanished.
|
||||
uuid = str(librarian.uuid)
|
||||
self.app.logger.exception("Search %s crashed: %s", uuid, exc)
|
||||
stored = _requests.get(uuid) or {}
|
||||
attempts = int(stored.get("attempts", 0)) + 1
|
||||
if attempts < max(1, SEARCH_MAX_ATTEMPTS):
|
||||
stored.update({
|
||||
"query": librarian.query,
|
||||
"deep_search": librarian.deep_search,
|
||||
"callback": librarian.callback,
|
||||
"attempts": attempts,
|
||||
})
|
||||
_requests.put(uuid, stored)
|
||||
librarian_queue.put(
|
||||
Librarian(self.app, librarian.query, uuid,
|
||||
librarian.deep_search, librarian.callback)
|
||||
)
|
||||
requeued = True
|
||||
self.app.logger.warning(
|
||||
"Search %s requeued after crash (attempt %d/%d)",
|
||||
uuid, attempts, SEARCH_MAX_ATTEMPTS,
|
||||
)
|
||||
else:
|
||||
self.app.logger.error(
|
||||
"Search %s failed %d times - giving up", uuid, attempts
|
||||
)
|
||||
_forget_search(uuid)
|
||||
finally:
|
||||
worker_busy.clear()
|
||||
with _active_lock:
|
||||
active_queries.pop(str(librarian.uuid), None)
|
||||
if requeued:
|
||||
# Still known to the bot's watchdog - it's going round again.
|
||||
active_queries[str(librarian.uuid)] = "queued"
|
||||
else:
|
||||
active_queries.pop(str(librarian.uuid), None)
|
||||
await asyncio.sleep(1)
|
||||
SHUTDOWN_DONE.set()
|
||||
self.app.logger.info("Search worker stopped cleanly")
|
||||
|
||||
@@ -106,8 +106,10 @@ def check_if_exists_brute_force(logger):
|
||||
):
|
||||
pass
|
||||
if blocked:
|
||||
logger.info(item)
|
||||
logger.error("Got blocked. Fuck.")
|
||||
# Expected, routine sci-hub behaviour (we back off an hour and carry
|
||||
# on) - WARNING, not ERROR, so it stops masquerading as a fault when
|
||||
# you're scanning the log for real problems.
|
||||
logger.warning("Got blocked. Fuck. Backing off an hour: %s", item[0])
|
||||
time.sleep(60 * 60)
|
||||
# trunk-ignore(bandit/B311)
|
||||
rand = random.randint(1, 60)
|
||||
|
||||
Reference in New Issue
Block a user