"""Unit tests for the librarian DOI search - specifically that it always terminates. The producer/consumer search used to hang whenever the number of chunk files it was told to read (MAXTHREADS) did not exactly match the files on disk: too few and it silently skipped trailing chunks, too many and a producer pointed at a missing file crashed before emitting its sentinel, starving the consumers' termination count forever. These tests pin down the fix: chunk files are auto-discovered, the sentinel threshold equals the number of producers actually started, and every producer emits its sentinel even on error. """ import logging import threading import search_bot _LOG = logging.getLogger("test-search-bot") _LOG.addHandler(logging.NullHandler()) def _write_chunks(directory, count, target=None, target_index=None): """Create _chunk.txt files; optionally drop `target` into one of them.""" for n in range(count): lines = [f"10.0000/decoy-{n}-a\n", f"10.0000/decoy-{n}-b\n"] if target is not None and n == target_index: lines.append(target + "\n") (directory / f"{n}_chunk.txt").write_text("".join(lines), encoding="utf-8") def _run_full(dois, timeout=20, stop_event=None, resume=None): """Run search_for_doi in a thread; return the whole result box. search_for_doi now returns (result_list, positions, interrupted); the box exposes all three (plus 'finished' and 'live') for the resume tests. """ box = {} live = [] def _run(): result_list, positions, interrupted = search_bot.search_for_doi( dois, live, _LOG, stop_event=stop_event, resume=resume ) box.update(result=result_list, positions=positions, interrupted=interrupted, live=live) worker = threading.Thread(target=_run, daemon=True) worker.start() worker.join(timeout) box["finished"] = not worker.is_alive() return box def _run_bounded(dois, timeout=20, stop_event=None, resume=None): """Back-compat wrapper: return (finished_in_time, result_list).""" box = _run_full(dois, timeout, stop_event, resume) return box.get("finished"), box.get("result") def test_finds_doi_in_trailing_chunk(tmp_path, monkeypatch): # Target lives in the LAST chunk - the one the old MAXTHREADS=N-too-low would # never have read. Auto-discovery must read every chunk present. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") target = "10.1234/target.in.trailing.chunk" _write_chunks(tmp_path, count=6, target=target, target_index=5) finished, result = _run_bounded([(target, "DATA"), ("10.9999/absent", "DATA")]) assert finished, "search hung instead of terminating" hit = [r for r in result if r["DOI"] == target and r["exists"]] assert hit, "DOI in the trailing chunk was not found" def test_terminates_when_a_chunk_is_unreadable(tmp_path, monkeypatch): # A chunk that exists at discovery time but cannot be opened (here: it is a # directory) makes its producer raise. The finally-sentinel must still fire # so the consumers' count completes and the search does not deadlock. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") _write_chunks(tmp_path, count=3) (tmp_path / "9_chunk.txt").mkdir() # discovered as a chunk, un-openable finished, _ = _run_bounded([("10.0000/decoy-0-a", "DATA")]) assert finished, "an unreadable chunk deadlocked the search" def test_no_chunks_returns_immediately(tmp_path, monkeypatch): # Empty database dir: return an (all-not-found) result at once, never hang. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") finished, result = _run_bounded([("10.0/x", "DATA")], timeout=10) assert finished assert result == [{"DOI": "10.0/x", "exists": False, "data": "DATA"}] def test_survives_invalid_utf8_byte_and_still_finds_later_doi(tmp_path, monkeypatch): # A chunk with a stray non-UTF-8 byte (0x96, the one from the field report) # must not crash the producer or abort the file mid-read: DOIs AFTER the bad # byte still have to be found. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") target = "10.1234/after.the.bad.byte" (tmp_path / "0_chunk.txt").write_bytes( b"10.0000/before\n" + b"\x96 broken \x96 line \x96\n" + target.encode() + b"\n" ) finished, result = _run_bounded([(target, "DATA")]) assert finished, "an invalid UTF-8 byte hung or crashed the search" hit = [r for r in result if r["DOI"] == target and r["exists"]] assert hit, "DOI after the bad byte was not found - the file was aborted mid-read" def test_doi_match_is_exact_not_substring(tmp_path, monkeypatch): # A DB line "10.1/12" must NOT satisfy a search for "10.1/1" (the old # `doi in line` substring test did). The exact DOI must still be found. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") (tmp_path / "0_chunk.txt").write_text( "10.1/12\n10.1/1\n10.2/999\n", encoding="utf-8" ) finished, result = _run_bounded([("10.1/1", "DATA"), ("10.9/absent", "DATA")]) assert finished by_doi = {r["DOI"]: r["exists"] for r in result} assert by_doi["10.1/1"] is True # exact line present -> found assert by_doi["10.9/absent"] is False def test_doi_match_handles_line_with_trailing_metadata(tmp_path, monkeypatch): # Lines of the form "\t" still match on the first token. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") (tmp_path / "0_chunk.txt").write_text("10.5/abc\tsome title here\n", encoding="utf-8") finished, result = _run_bounded([("10.5/abc", "DATA")]) assert finished assert result[0]["exists"] is True def test_discover_chunk_files_sorted_numerically(tmp_path, monkeypatch): monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") for n in (0, 2, 10, 1): (tmp_path / f"{n}_chunk.txt").write_text("x\n", encoding="utf-8") (tmp_path / "notes.txt").write_text("ignore me\n", encoding="utf-8") found = search_bot.discover_chunk_files(_LOG) # Numeric order (10 after 2, not lexicographic), and non-chunk files ignored. assert found == ["0_chunk.txt", "1_chunk.txt", "2_chunk.txt", "10_chunk.txt"] def _offset_after(path, marker): """Byte-cookie (tell) just past the line equal to `marker` in `path`.""" with open(path, "r", encoding="utf-8") as handle: while True: line = handle.readline() if not line: raise AssertionError(f"marker {marker!r} not found") if line.strip() == marker: return handle.tell() def test_resume_seeks_past_scanned_part_and_continues(tmp_path, monkeypatch): # Chunk: early | first-half decoy | MIDDLE | late. Resume from just past # MIDDLE with 'early' pre-found. The scan must: keep 'early' (pre-marked), # find 'late' (after the resume point), and NOT find the first-half decoy # (proving it seeked past it instead of re-reading from the top). monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") path = tmp_path / "0_chunk.txt" path.write_text( "10.1/early\n10.1/only-first-half\nMIDDLE\n10.1/late\n", encoding="utf-8" ) offset = _offset_after(str(path), "MIDDLE") resume = {"found": ["10.1/early"], "positions": {"0_chunk.txt": offset}} box = _run_full( [("10.1/early", "D"), ("10.1/late", "D"), ("10.1/only-first-half", "D")], resume=resume, ) assert box["finished"] by_doi = {r["DOI"]: r["exists"] for r in box["result"]} assert by_doi["10.1/early"] is True # carried over from the checkpoint assert by_doi["10.1/late"] is True # found after the resume offset assert by_doi["10.1/only-first-half"] is False # skipped - not re-scanned def test_stop_event_interrupts_and_reports_positions(tmp_path, monkeypatch): monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") _write_chunks(tmp_path, count=2) stop = __import__("threading").Event() stop.set() # already asked to stop before it starts box = _run_full([("10.0000/decoy-0-a", "D")], stop_event=stop) assert box["finished"], "an already-set stop must not hang the search" assert box["interrupted"] is True assert isinstance(box["positions"], dict) def test_bounded_queue_does_not_deadlock_on_early_termination(tmp_path, monkeypatch): # The OOM fix bounds the work queue. That means a producer can block on a # FULL queue - and if the consumers have already finished (all DOIs found) # it must notice the TERM sentinel instead of hanging forever. Tiny queue + # target on the first line + thousands of trailing decoys the producer still # holds is exactly that situation. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") monkeypatch.setattr(search_bot, "WORK_Q_SIZE", 3) # force the producer to block target = "10.1234/found.on.line.one" lines = [target + "\n"] + [f"10.0000/decoy-{i}\n" for i in range(5000)] (tmp_path / "0_chunk.txt").write_text("".join(lines), encoding="utf-8") finished, result = _run_bounded([(target, "DATA")], timeout=20) assert finished, "a full bounded queue deadlocked the producer on early termination" hit = [r for r in result if r["DOI"] == target and r["exists"]] assert hit, "the target on the first line should have been found"