ac16b77f56
CI / compile (pull_request) Successful in 10s
CI / unit (pull_request) Successful in 28s
CI / integration (pull_request) Successful in 27s
build / build (push) Successful in 42s
CI / compile (push) Successful in 10s
CI / unit (push) Successful in 26s
CI / integration (push) Successful in 25s
A restart of the librarian used to throw away an in-flight search (and any
searches still queued). Now search state survives a restart:
* Resumable DB scan (search_bot): each producer records a tell()-cookie
watermark per chunk file as it goes (safe because search_for_doi drains
the work queue before returning), and can seek back to it. search_for_doi
now takes stop_event + resume and returns (result_list, positions,
interrupted).
* Persisted requests: /query writes the accepted request to a disk queue
before enqueuing; replay_requests re-enqueues unfinished ones on startup.
So even a search still waiting in the queue survives a restart.
* Checkpoints: when a graceful shutdown interrupts a scan, the librarian
writes {dois, found-so-far, per-file offsets}. On restart answer_query
loads it, skips the (already done) Crossref+refine, and continues the
scan from the saved offsets with the found DOIs pre-marked - no line is
read twice and none is missed. A finished or crashed search forgets its
request+checkpoint (no poison-pill replay).
* Graceful shutdown: SIGTERM/SIGINT set a shutdown event; the running scan
checkpoints and the worker stops. The main thread then exits within a
BOUNDED window (CONJURER_LIBRARIAN_GRACEFUL_TIMEOUT, default 45s) so the
pod can never become an un-killable zombie. Needs terminationGracePeriod
>= that in the deploy (separate PR).
Tests: search_bot resume correctness (seek past scanned, don't miss/re-scan;
stop_event -> interrupted) and librarian state mechanics (request replay,
forget, checkpoint round-trip, poison-pill drop). Suite: 58 unit + 49
integration green.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
217 lines
9.3 KiB
Python
217 lines
9.3 KiB
Python
"""Unit tests for the librarian DOI search - specifically that it always
|
|
terminates.
|
|
|
|
The producer/consumer search used to hang whenever the number of chunk files it
|
|
was told to read (MAXTHREADS) did not exactly match the files on disk: too few
|
|
and it silently skipped trailing chunks, too many and a producer pointed at a
|
|
missing file crashed before emitting its sentinel, starving the consumers'
|
|
termination count forever. These tests pin down the fix: chunk files are
|
|
auto-discovered, the sentinel threshold equals the number of producers actually
|
|
started, and every producer emits its sentinel even on error.
|
|
"""
|
|
import logging
|
|
import threading
|
|
|
|
import search_bot
|
|
|
|
_LOG = logging.getLogger("test-search-bot")
|
|
_LOG.addHandler(logging.NullHandler())
|
|
|
|
|
|
def _write_chunks(directory, count, target=None, target_index=None):
|
|
"""Create <n>_chunk.txt files; optionally drop `target` into one of them."""
|
|
for n in range(count):
|
|
lines = [f"10.0000/decoy-{n}-a\n", f"10.0000/decoy-{n}-b\n"]
|
|
if target is not None and n == target_index:
|
|
lines.append(target + "\n")
|
|
(directory / f"{n}_chunk.txt").write_text("".join(lines), encoding="utf-8")
|
|
|
|
|
|
def _run_full(dois, timeout=20, stop_event=None, resume=None):
|
|
"""Run search_for_doi in a thread; return the whole result box.
|
|
|
|
search_for_doi now returns (result_list, positions, interrupted); the box
|
|
exposes all three (plus 'finished' and 'live') for the resume tests.
|
|
"""
|
|
box = {}
|
|
live = []
|
|
|
|
def _run():
|
|
result_list, positions, interrupted = search_bot.search_for_doi(
|
|
dois, live, _LOG, stop_event=stop_event, resume=resume
|
|
)
|
|
box.update(result=result_list, positions=positions, interrupted=interrupted, live=live)
|
|
|
|
worker = threading.Thread(target=_run, daemon=True)
|
|
worker.start()
|
|
worker.join(timeout)
|
|
box["finished"] = not worker.is_alive()
|
|
return box
|
|
|
|
|
|
def _run_bounded(dois, timeout=20, stop_event=None, resume=None):
|
|
"""Back-compat wrapper: return (finished_in_time, result_list)."""
|
|
box = _run_full(dois, timeout, stop_event, resume)
|
|
return box.get("finished"), box.get("result")
|
|
|
|
|
|
def test_finds_doi_in_trailing_chunk(tmp_path, monkeypatch):
|
|
# Target lives in the LAST chunk - the one the old MAXTHREADS=N-too-low would
|
|
# never have read. Auto-discovery must read every chunk present.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
target = "10.1234/target.in.trailing.chunk"
|
|
_write_chunks(tmp_path, count=6, target=target, target_index=5)
|
|
|
|
finished, result = _run_bounded([(target, "DATA"), ("10.9999/absent", "DATA")])
|
|
|
|
assert finished, "search hung instead of terminating"
|
|
hit = [r for r in result if r["DOI"] == target and r["exists"]]
|
|
assert hit, "DOI in the trailing chunk was not found"
|
|
|
|
|
|
def test_terminates_when_a_chunk_is_unreadable(tmp_path, monkeypatch):
|
|
# A chunk that exists at discovery time but cannot be opened (here: it is a
|
|
# directory) makes its producer raise. The finally-sentinel must still fire
|
|
# so the consumers' count completes and the search does not deadlock.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
_write_chunks(tmp_path, count=3)
|
|
(tmp_path / "9_chunk.txt").mkdir() # discovered as a chunk, un-openable
|
|
|
|
finished, _ = _run_bounded([("10.0000/decoy-0-a", "DATA")])
|
|
|
|
assert finished, "an unreadable chunk deadlocked the search"
|
|
|
|
|
|
def test_no_chunks_returns_immediately(tmp_path, monkeypatch):
|
|
# Empty database dir: return an (all-not-found) result at once, never hang.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
|
|
finished, result = _run_bounded([("10.0/x", "DATA")], timeout=10)
|
|
|
|
assert finished
|
|
assert result == [{"DOI": "10.0/x", "exists": False, "data": "DATA"}]
|
|
|
|
|
|
def test_survives_invalid_utf8_byte_and_still_finds_later_doi(tmp_path, monkeypatch):
|
|
# A chunk with a stray non-UTF-8 byte (0x96, the one from the field report)
|
|
# must not crash the producer or abort the file mid-read: DOIs AFTER the bad
|
|
# byte still have to be found.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
target = "10.1234/after.the.bad.byte"
|
|
(tmp_path / "0_chunk.txt").write_bytes(
|
|
b"10.0000/before\n" + b"\x96 broken \x96 line \x96\n" + target.encode() + b"\n"
|
|
)
|
|
|
|
finished, result = _run_bounded([(target, "DATA")])
|
|
|
|
assert finished, "an invalid UTF-8 byte hung or crashed the search"
|
|
hit = [r for r in result if r["DOI"] == target and r["exists"]]
|
|
assert hit, "DOI after the bad byte was not found - the file was aborted mid-read"
|
|
|
|
|
|
def test_doi_match_is_exact_not_substring(tmp_path, monkeypatch):
|
|
# A DB line "10.1/12" must NOT satisfy a search for "10.1/1" (the old
|
|
# `doi in line` substring test did). The exact DOI must still be found.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
(tmp_path / "0_chunk.txt").write_text(
|
|
"10.1/12\n10.1/1\n10.2/999\n", encoding="utf-8"
|
|
)
|
|
|
|
finished, result = _run_bounded([("10.1/1", "DATA"), ("10.9/absent", "DATA")])
|
|
|
|
assert finished
|
|
by_doi = {r["DOI"]: r["exists"] for r in result}
|
|
assert by_doi["10.1/1"] is True # exact line present -> found
|
|
assert by_doi["10.9/absent"] is False
|
|
|
|
|
|
def test_doi_match_handles_line_with_trailing_metadata(tmp_path, monkeypatch):
|
|
# Lines of the form "<DOI>\t<metadata>" still match on the first token.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
(tmp_path / "0_chunk.txt").write_text("10.5/abc\tsome title here\n", encoding="utf-8")
|
|
|
|
finished, result = _run_bounded([("10.5/abc", "DATA")])
|
|
|
|
assert finished
|
|
assert result[0]["exists"] is True
|
|
|
|
|
|
def test_discover_chunk_files_sorted_numerically(tmp_path, monkeypatch):
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
for n in (0, 2, 10, 1):
|
|
(tmp_path / f"{n}_chunk.txt").write_text("x\n", encoding="utf-8")
|
|
(tmp_path / "notes.txt").write_text("ignore me\n", encoding="utf-8")
|
|
|
|
found = search_bot.discover_chunk_files(_LOG)
|
|
|
|
# Numeric order (10 after 2, not lexicographic), and non-chunk files ignored.
|
|
assert found == ["0_chunk.txt", "1_chunk.txt", "2_chunk.txt", "10_chunk.txt"]
|
|
|
|
|
|
def _offset_after(path, marker):
|
|
"""Byte-cookie (tell) just past the line equal to `marker` in `path`."""
|
|
with open(path, "r", encoding="utf-8") as handle:
|
|
while True:
|
|
line = handle.readline()
|
|
if not line:
|
|
raise AssertionError(f"marker {marker!r} not found")
|
|
if line.strip() == marker:
|
|
return handle.tell()
|
|
|
|
|
|
def test_resume_seeks_past_scanned_part_and_continues(tmp_path, monkeypatch):
|
|
# Chunk: early | first-half decoy | MIDDLE | late. Resume from just past
|
|
# MIDDLE with 'early' pre-found. The scan must: keep 'early' (pre-marked),
|
|
# find 'late' (after the resume point), and NOT find the first-half decoy
|
|
# (proving it seeked past it instead of re-reading from the top).
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
path = tmp_path / "0_chunk.txt"
|
|
path.write_text(
|
|
"10.1/early\n10.1/only-first-half\nMIDDLE\n10.1/late\n", encoding="utf-8"
|
|
)
|
|
offset = _offset_after(str(path), "MIDDLE")
|
|
|
|
resume = {"found": ["10.1/early"], "positions": {"0_chunk.txt": offset}}
|
|
box = _run_full(
|
|
[("10.1/early", "D"), ("10.1/late", "D"), ("10.1/only-first-half", "D")],
|
|
resume=resume,
|
|
)
|
|
|
|
assert box["finished"]
|
|
by_doi = {r["DOI"]: r["exists"] for r in box["result"]}
|
|
assert by_doi["10.1/early"] is True # carried over from the checkpoint
|
|
assert by_doi["10.1/late"] is True # found after the resume offset
|
|
assert by_doi["10.1/only-first-half"] is False # skipped - not re-scanned
|
|
|
|
|
|
def test_stop_event_interrupts_and_reports_positions(tmp_path, monkeypatch):
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
_write_chunks(tmp_path, count=2)
|
|
stop = __import__("threading").Event()
|
|
stop.set() # already asked to stop before it starts
|
|
|
|
box = _run_full([("10.0000/decoy-0-a", "D")], stop_event=stop)
|
|
|
|
assert box["finished"], "an already-set stop must not hang the search"
|
|
assert box["interrupted"] is True
|
|
assert isinstance(box["positions"], dict)
|
|
|
|
|
|
def test_bounded_queue_does_not_deadlock_on_early_termination(tmp_path, monkeypatch):
|
|
# The OOM fix bounds the work queue. That means a producer can block on a
|
|
# FULL queue - and if the consumers have already finished (all DOIs found)
|
|
# it must notice the TERM sentinel instead of hanging forever. Tiny queue +
|
|
# target on the first line + thousands of trailing decoys the producer still
|
|
# holds is exactly that situation.
|
|
monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/")
|
|
monkeypatch.setattr(search_bot, "WORK_Q_SIZE", 3) # force the producer to block
|
|
target = "10.1234/found.on.line.one"
|
|
lines = [target + "\n"] + [f"10.0000/decoy-{i}\n" for i in range(5000)]
|
|
(tmp_path / "0_chunk.txt").write_text("".join(lines), encoding="utf-8")
|
|
|
|
finished, result = _run_bounded([(target, "DATA")], timeout=20)
|
|
|
|
assert finished, "a full bounded queue deadlocked the producer on early termination"
|
|
hit = [r for r in result if r["DOI"] == target and r["exists"]]
|
|
assert hit, "the target on the first line should have been found"
|