AI: personal assistants without the dead API, and keep Ollama warm
CI / compile (pull_request) Successful in 5s
CI / unit (pull_request) Successful in 22s
CI / integration (pull_request) Successful in 26s
build / build (push) Successful in 12s
CI / compile (push) Successful in 5s
CI / unit (push) Successful in 22s
CI / integration (push) Successful in 26s
CI / compile (pull_request) Successful in 5s
CI / unit (pull_request) Successful in 22s
CI / integration (pull_request) Successful in 26s
build / build (push) Successful in 12s
CI / compile (push) Successful in 5s
CI / unit (push) Successful in 22s
CI / integration (push) Successful in 26s
Two things the field report asked for. 1) PERSONAL ASSISTANTS (replacing the sunset OpenAI Assistants API) The old implementation gave three capabilities. Two are reimplemented here, the third was confirmed unused and is deliberately not replaced: * per-user persona - it already lived in system_gpt_settings.json; it was only ever being shipped to OpenAI. It is now the system prompt. * per-user conversation thread - OpenAI held this server-side. It now lives in assistant_memory.json, keyed by discord user id, trimmed to the most recent turns (CONJURER_ASSISTANT_MEMORY_TURNS) and written atomically so a torn write cannot lose someone's history. Deliberately a plain trim, not the AI summarisation used for the bar's shared memory: these are private DMs and must not end up in a public "legend". * file_search - not replaced. Confirmed not in use. The conversation goes through handle_response with request_type="NONE" and an explicit message list, which keeps it out of the bar's shared memory. The big win: create_chat_assistant hardcoded model="gpt-4o", so assistants were locked to OpenAI. They now run on whatever $gadaj_teraz selects - Claude and Ollama included. create_chat_assistant / chat_with_assistant are gone, and with them the last call to beta.threads in the startup path - so the cog cannot be killed by that API again. (add_files_to_vector_store / delete_files_from_vector_store still reference beta.assistants but are dead code - nothing calls them - so they cannot crash anything; left alone rather than widening this change.) 2) KEEPING A SELF-HOSTED MODEL WARM Loading is the slow part - the GPU is shared with other users - so we preload via Ollama's documented mechanism: /api/generate with a model, a keep_alive and NO prompt. It loads the model and generates nothing. * on switching to ollama, $gadaj_teraz fires a preload in the BACKGROUND (not awaited: loading can take minutes and the command must answer at once), so the wait lands on the operator rather than the first user; * a warm loop re-asserts keep_alive every CONJURER_OLLAMA_WARM_MINUTES. Both are hard-guarded on the ACTIVE provider being ollama. Warming a metered API would burn tokens and money for nothing, so that guard is pinned by a test asserting the preload is never called for gpt/claude, and another asserting the preload body carries no prompt (a prompt would make every warm-up generate). Tests: 82 unit + 71 integration green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit was merged in pull request #28.
This commit is contained in:
@@ -394,3 +394,143 @@ def test_persist_survives_an_unreadable_settings_file(tmp_path, monkeypatch):
|
||||
broken.write_text("{ not json", encoding="utf-8")
|
||||
monkeypatch.setattr(ai_functions, "SYSTEM_GPT_SETTINGS", str(broken))
|
||||
ai_functions._persist_active_ai_config("gpt", model_for="gpt") # must not raise
|
||||
|
||||
|
||||
# ----------------------------------------------- keep-warm (Ollama ONLY) ----
|
||||
# The money guard: preloading a self-hosted model is free, but firing the same
|
||||
# thing at a metered API would burn tokens for nothing. These pin that it can
|
||||
# only ever happen for Ollama.
|
||||
|
||||
|
||||
def test_warm_active_model_is_a_noop_for_paid_providers(monkeypatch):
|
||||
called = []
|
||||
monkeypatch.setattr(
|
||||
ai_functions, "_ollama_preload", lambda *a, **k: called.append(a) or True
|
||||
)
|
||||
for paid in ("gpt", "claude"):
|
||||
_reset_active(paid)
|
||||
try:
|
||||
assert asyncio.run(ai_functions.warm_active_model()) is False
|
||||
finally:
|
||||
_reset_active("gpt")
|
||||
assert called == [], "a paid backend must never be preloaded"
|
||||
|
||||
|
||||
def test_warm_active_model_preloads_when_ollama_is_active(monkeypatch):
|
||||
monkeypatch.setitem(
|
||||
ai_functions.AI_CONFIGS,
|
||||
"ollama",
|
||||
{"provider": "ollama", "latest_model": "qwen2.5:7b", "cheap_model": "c"},
|
||||
)
|
||||
seen = {}
|
||||
monkeypatch.setattr(
|
||||
ai_functions, "_ollama_preload", lambda model, *a, **k: seen.update(model=model) or True
|
||||
)
|
||||
_reset_active("ollama")
|
||||
try:
|
||||
assert asyncio.run(ai_functions.warm_active_model()) is True
|
||||
finally:
|
||||
_reset_active("gpt")
|
||||
assert seen["model"] == "qwen2.5:7b"
|
||||
|
||||
|
||||
def test_active_provider_reports_the_switch():
|
||||
_reset_active("gpt")
|
||||
assert ai_functions.active_provider() == "openai"
|
||||
_reset_active("claude")
|
||||
try:
|
||||
assert ai_functions.active_provider() == "anthropic"
|
||||
finally:
|
||||
_reset_active("gpt")
|
||||
|
||||
|
||||
def test_preload_sends_no_prompt_so_it_generates_nothing(monkeypatch):
|
||||
# Ollama's documented preload: a model and keep_alive, and NO prompt. If a
|
||||
# prompt ever crept in, every warm-up would silently generate tokens.
|
||||
sent = {}
|
||||
|
||||
class _Resp:
|
||||
status_code = 200
|
||||
|
||||
monkeypatch.setattr(ai_functions, "OLLAMA_URL", "http://ollama:11434")
|
||||
monkeypatch.setattr(
|
||||
ai_functions.requests, "post",
|
||||
lambda url, json=None, timeout=None: sent.update(url=url, body=json) or _Resp(),
|
||||
)
|
||||
assert ai_functions._ollama_preload("qwen2.5:7b") is True
|
||||
assert sent["url"].endswith("/api/generate")
|
||||
assert sent["body"]["model"] == "qwen2.5:7b"
|
||||
assert "keep_alive" in sent["body"]
|
||||
assert "prompt" not in sent["body"], "a preload must not generate"
|
||||
|
||||
|
||||
def test_preload_without_endpoint_is_a_noop(monkeypatch):
|
||||
monkeypatch.setattr(ai_functions, "OLLAMA_URL", "")
|
||||
assert ai_functions._ollama_preload("x") is False
|
||||
|
||||
|
||||
# ------------------------------------------- personal assistants (per user) --
|
||||
# Replaces the sunset OpenAI Assistants API. The two properties that matter:
|
||||
# each user's DM history is ISOLATED (private DMs must not leak into another
|
||||
# user's context or the bar's shared memory), and it stays BOUNDED.
|
||||
|
||||
|
||||
def _fresh_assistant_memory(tmp_path, monkeypatch, turns=40):
|
||||
monkeypatch.setattr(
|
||||
ai_functions, "ASSISTANT_MEMORY_FILE", str(tmp_path / "assistant_memory.json")
|
||||
)
|
||||
monkeypatch.setattr(ai_functions, "ASSISTANT_MEMORY_TURNS", turns)
|
||||
monkeypatch.setattr(ai_functions, "_ASSISTANT_MEMORY", None)
|
||||
|
||||
|
||||
def test_assistant_history_is_isolated_per_user(tmp_path, monkeypatch):
|
||||
_fresh_assistant_memory(tmp_path, monkeypatch)
|
||||
ai_functions.remember_assistant_turn(111, "sekret Anny", "ok Anna")
|
||||
ai_functions.remember_assistant_turn(222, "sekret Bartka", "ok Bartek")
|
||||
|
||||
anna = ai_functions.assistant_history(111)
|
||||
bartek = ai_functions.assistant_history(222)
|
||||
assert [m["content"] for m in anna] == ["sekret Anny", "ok Anna"]
|
||||
assert [m["content"] for m in bartek] == ["sekret Bartka", "ok Bartek"]
|
||||
assert "sekret Anny" not in str(bartek) # no cross-user bleed
|
||||
|
||||
|
||||
def test_assistant_history_is_trimmed_to_the_bound(tmp_path, monkeypatch):
|
||||
_fresh_assistant_memory(tmp_path, monkeypatch, turns=4)
|
||||
for i in range(10):
|
||||
ai_functions.remember_assistant_turn(1, f"u{i}", f"a{i}")
|
||||
history = ai_functions.assistant_history(1)
|
||||
assert len(history) == 4 # bounded
|
||||
assert history[-1]["content"] == "a9" # newest kept
|
||||
assert all("u0" != m["content"] for m in history) # oldest dropped
|
||||
|
||||
|
||||
def test_assistant_history_survives_a_restart(tmp_path, monkeypatch):
|
||||
_fresh_assistant_memory(tmp_path, monkeypatch)
|
||||
ai_functions.remember_assistant_turn(7, "pamietaj", "pamietam")
|
||||
# Simulate a restart: drop the in-memory cache, re-read from disk.
|
||||
monkeypatch.setattr(ai_functions, "_ASSISTANT_MEMORY", None)
|
||||
assert [m["content"] for m in ai_functions.assistant_history(7)] == [
|
||||
"pamietaj",
|
||||
"pamietam",
|
||||
]
|
||||
|
||||
|
||||
def test_assistant_messages_carry_persona_history_and_new_turn(tmp_path, monkeypatch):
|
||||
_fresh_assistant_memory(tmp_path, monkeypatch)
|
||||
ai_functions.remember_assistant_turn(5, "wczoraj", "odpowiedz")
|
||||
msgs = ai_functions.build_assistant_messages(
|
||||
5, "Towarzysz Młotek", "Mówisz po polsku.", "dzisiaj"
|
||||
)
|
||||
assert msgs[0]["role"] == "system"
|
||||
assert "Towarzysz Młotek" in msgs[0]["content"]
|
||||
assert "Mówisz po polsku." in msgs[0]["content"]
|
||||
assert [m["content"] for m in msgs[1:]] == ["wczoraj", "odpowiedz", "dzisiaj"]
|
||||
|
||||
|
||||
def test_corrupt_assistant_memory_starts_empty_instead_of_crashing(tmp_path, monkeypatch):
|
||||
path = tmp_path / "assistant_memory.json"
|
||||
path.write_text("{ not json", encoding="utf-8")
|
||||
monkeypatch.setattr(ai_functions, "ASSISTANT_MEMORY_FILE", str(path))
|
||||
monkeypatch.setattr(ai_functions, "_ASSISTANT_MEMORY", None)
|
||||
assert ai_functions.assistant_history(1) == []
|
||||
|
||||
Reference in New Issue
Block a user