diff --git a/conjurer/bot.yaml b/conjurer/bot.yaml index 98246e6..4eee66f 100644 --- a/conjurer/bot.yaml +++ b/conjurer/bot.yaml @@ -31,6 +31,19 @@ spec: # Where the librarian sends THIS bot's results/pongs back to (its own # NodePort). Lets one librarian serve both bots - see deploy-bot.yaml. - { name: CONJURER_SELF_CALLBACK, value: "http://192.168.1.73:32442" } + # Self-hosted models (Ollama). No API key - the endpoint IS the + # configuration, and the backend stays unselectable while unset. + # Pick a model at runtime with: $gadaj_teraz ollama + # ($modele_ai lists what the server actually has pulled). + - { name: CONJURER_OLLAMA_URL, value: "http://192.168.1.72:11434" } + # The server currently has exactly one model pulled (verified via + # /v1/models): gemma4:e2b. Without this the built-in default + # (llama3.1:8b) would be requested and every reply would fail. + - { name: CONJURER_OLLAMA_MODEL, value: "gemma4:e2b" } + # Self-hosted generation is far slower than a hosted API, especially + # the first request after the model is evicted from VRAM. Applies to + # every backend, so keep it only as high as you actually need. + - { name: CONJURER_AI_TIMEOUT_SECONDS, value: "240" } volumeMounts: - { name: data, mountPath: /data } - { name: netrc, mountPath: /secrets, readOnly: true } diff --git a/conjurer/deploy-bot.yaml b/conjurer/deploy-bot.yaml index a604a50..ec63bb1 100644 --- a/conjurer/deploy-bot.yaml +++ b/conjurer/deploy-bot.yaml @@ -51,6 +51,19 @@ spec: # query and answers results/pongs HERE - so it serves this bot AND the # test bot from one instance, no CONJURER_MAIN_BOT repointing needed. - { name: CONJURER_SELF_CALLBACK, value: "http://192.168.1.73:32443" } + # Self-hosted models (Ollama). No API key - the endpoint IS the + # configuration, and the backend stays unselectable while unset. + # Pick a model at runtime with: $gadaj_teraz ollama + # ($modele_ai lists what the server actually has pulled). + - { name: CONJURER_OLLAMA_URL, value: "http://192.168.1.72:11434" } + # The server currently has exactly one model pulled (verified via + # /v1/models): gemma4:e2b. Without this the built-in default + # (llama3.1:8b) would be requested and every reply would fail. + - { name: CONJURER_OLLAMA_MODEL, value: "gemma4:e2b" } + # Self-hosted generation is far slower than a hosted API, especially + # the first request after the model is evicted from VRAM. Applies to + # every backend, so keep it only as high as you actually need. + - { name: CONJURER_AI_TIMEOUT_SECONDS, value: "240" } volumeMounts: - { name: data, mountPath: /data } - { name: netrc, mountPath: /secrets, readOnly: true }