"""Integration: the 'still searching' heartbeat and its cheap progress estimate. The estimate must stay free: producers already record a byte offset per chunk file and the total is stat()'d once, so a reading is just a sum over ~40 ints. """ import logging import sys import types import pytest if "habanero" not in sys.modules: _habanero = types.ModuleType("habanero") _habanero.Crossref = object sys.modules["habanero"] = _habanero import conjurer_librarian as lib # noqa: E402 import search_bot # noqa: E402 _LOG = logging.getLogger("test-heartbeat") _LOG.addHandler(logging.NullHandler()) def test_progress_summary_percentages(): progress = {"positions": {"0_chunk.txt": 250, "1_chunk.txt": 250}, "total_bytes": 1000} done, total, percent = lib._progress_summary(progress) assert (done, total) == (500, 1000) assert percent == pytest.approx(50.0) def test_progress_summary_unknown_total_is_zero_percent(): done, total, percent = lib._progress_summary({"positions": {"a": 10}}) assert (done, total, percent) == (10, 0, 0.0) def test_progress_summary_handles_empty_and_none(): assert lib._progress_summary(None) == (0, 0, 0.0) assert lib._progress_summary({}) == (0, 0, 0.0) def test_progress_summary_is_clamped_to_100(): # A partially-buffered tail can push the summed offsets past the total. _done, _total, percent = lib._progress_summary( {"positions": {"a": 1500}, "total_bytes": 1000} ) assert percent == pytest.approx(100.0) def test_current_search_registration_round_trip(): progress = {"positions": {"a": 5}, "total_bytes": 10} live = [{"DOI": "10.1/x"}] lib._set_current_search("uuid-1", "kwas foliowy", progress, live) with lib._current_lock: snapshot = dict(lib._current_search) assert snapshot["uuid"] == "uuid-1" assert snapshot["query"] == "kwas foliowy" assert lib._progress_summary(snapshot["progress"])[2] == pytest.approx(50.0) lib._clear_current_search() with lib._current_lock: assert not lib._current_search def _write_two_chunks(tmp_path): (tmp_path / "0_chunk.txt").write_text("10.1/a\n10.1/b\n", encoding="utf-8") (tmp_path / "1_chunk.txt").write_text("10.1/c\n", encoding="utf-8") return sum( (tmp_path / name).stat().st_size for name in ("0_chunk.txt", "1_chunk.txt") ) def test_search_fills_progress_and_reaches_full_coverage(tmp_path, monkeypatch): # Coverage must be measured on a search that CANNOT stop early. Once every # queried DOI is found the consumer signals TERM and the producers stop # mid-file, so a search for a DOI that exists reaches an arbitrary offset - # asserting 100% there is a race (it failed roughly one run in two). # An absent DOI forces the whole database to be read. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") expected_total = _write_two_chunks(tmp_path) progress = {} search_bot.search_for_doi([("10.9/absent", "DATA")], [], _LOG, progress=progress) assert progress["total_bytes"] == expected_total assert progress["chunk_files"] == 2 done, total, percent = lib._progress_summary(progress) assert total == expected_total assert done == expected_total # nothing stopped it: whole DB scanned assert percent == pytest.approx(100.0) def test_progress_is_populated_for_a_search_that_finds_its_target(tmp_path, monkeypatch): # The early-termination case: the target is found, so coverage is whatever # the producers reached. Assert what IS deterministic - the total is known, # progress is bounded and sane, and the hit is reported. monkeypatch.setattr(search_bot, "DATABASE_PATH", str(tmp_path) + "/") expected_total = _write_two_chunks(tmp_path) progress = {} result, _positions, _interrupted = search_bot.search_for_doi( [("10.1/c", "DATA")], [], _LOG, progress=progress ) assert progress["total_bytes"] == expected_total done, total, percent = lib._progress_summary(progress) assert total == expected_total assert 0 <= done <= total # bounded, never nonsense assert 0.0 <= percent <= 100.0 assert [r for r in result if r["DOI"] == "10.1/c" and r["exists"]]