Files
conjurer/bot.py
T
gitea ed8b271b4e
CI / compile (pull_request) Successful in 9s
CI / unit (pull_request) Successful in 18s
CI / integration (pull_request) Successful in 18s
Gate librarian cog on a full ping round-trip, not a bare GET
The librarian health check was a plain GET to '/', which only proved
Flask was listening - not that the service could actually take a query,
run it through its internal queue+worker, and answer back. So the cog
could load against a librarian whose worker was wedged or that couldn't
reach the bot on the return leg.

Replace it with a ping that travels the SAME path a real search does, on
both sides:
  bot: QueryControl -> OUT_COMM_Q -> scan_queue -> awaiting_q
  librarian: POST /ping -> librarian_queue -> worker pulls it off
             (no Crossref/DOI search) -> pongs back with the same uuid
  bot: /conjurer -> incoming_q -> scan_incoming matches uuid, wakes waiter
The cog enables only when that whole loop closes within 3s. This also
proves the librarian->bot return path, which a GET never did.

Safety: uuid is random per ping; the wait and POST are both bounded so
startup can't stall; a pong that finds no waiter is dropped (never
orphaned into IN_COMM_Q, which would make the cog post a bogus 'no
results' message); and a ping whose pong never returns is swept out of
awaiting_q after PING_TTL_SECONDS so nothing leaks. All awaiting_q writes
stay within scan_queue (append) and scan_incoming (remove) - no locks,
no cross-thread mutation.

Integration tests cover: OK round-trip, timeout when accepted-but-no-pong,
unreachable, non-200, orphan-pong-dropped, and that real results still
reach IN_COMM_Q. Suite: 24 integration + 41 unit green.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-08-01 15:00:04 +02:00

378 lines
14 KiB
Python
Executable File

# This Python file uses the following encoding: utf-8
# trunk-ignore-all(bandit/B311)
# pylint: disable=line-too-long
# pylint: disable=too-many-lines
"""
Module of a python bot named Conjurer - used to work on BDSM discord servers.
Startup is defensive by design:
* every cog is loaded independently - one broken/missing dependency disables
that cog only, never the whole bot,
* cogs that need a sibling service (musician / librarian) are only enabled
after a positive health check; a watchdog keeps re-checking and enables them
the moment the service comes alive,
* fatal misconfiguration (missing Discord token) exits loudly on stderr with a
clear message instead of dying silently into a log file.
"""
import asyncio
import logging
import os
import sys
# *=========================================== Standard Library Imports
import random
import threading
from logging import handlers
# *==============Imported libraries
import discord
import requests
from discord.ext import commands
from communication_subroutine import comm_subroutine, librarian_ping
from constants import (
ENCODING,
FILE_SERVICE_ADDRESS,
GET_MP3,
LIBRARIAN_PING,
LIBRARIAN_SERVICE_ADDRESS,
LOGFILE,
RADIO_SERVICE_ADDRESS,
TOKEN,
service_headers,
)
# Round-trip health check budget for the librarian ping (bot -> librarian
# internal queue -> pong back). Deliberately short so startup never stalls.
LIBRARIAN_PING_TIMEOUT = 3.0
logger = logging.getLogger("discord")
logger.setLevel(logging.DEBUG)
formatter = logging.Formatter("%(asctime)s - %(name)s - %(levelname)s - %(message)s")
# File log (rotated). constants ensures the directory exists; this is belt and
# braces for exotic overrides.
os.makedirs(os.path.dirname(LOGFILE) or ".", exist_ok=True)
handler = handlers.RotatingFileHandler(
filename=LOGFILE,
encoding=ENCODING,
mode="a",
maxBytes=6 * 1024 * 1024,
backupCount=6,
)
handler.setFormatter(formatter)
logger.addHandler(handler)
# Console log so `docker logs` / journalctl actually show what happened.
console_handler = logging.StreamHandler()
console_handler.setLevel(logging.INFO)
console_handler.setFormatter(formatter)
logger.addHandler(console_handler)
# Some dependency calls logging.basicConfig(), adding a root handler; without
# this the 'discord' logger's records get printed twice (our format + root's).
logger.propagate = False
# *=========================================== Initializations
intents = discord.Intents.default()
intents.message_content = True
intents.typing = True
intents.presences = True
intents.members = True
intents.messages = True
intents.voice_states = True
intents.moderation = True
# on_member_ban - wyswietl na glownym kanale pieczatke "Niech spierdala"
# on_member_unban - "mam wyjebane"
random.seed()
client = commands.Bot(intents=intents, command_prefix="$")
# *=========================================== Extension groups
# Core cogs depend on nothing but the bot itself (broken ones are skipped
# individually). Service groups are gated on a health check of the service
# they talk to and enabled later by the watchdog when the service appears.
CORE_EXTENSIONS = [
"administration_commands",
"ai_commands",
"other_commands",
"bar_commands",
"lore_commands",
"oracle_commands",
"latex_commands",
"voice_recognition_commands",
"conanjurer_commands",
]
SERVICE_EXTENSION_GROUPS = {
# musician (file service): Discord music download/search, file shares
"musician": {
"health_url": f"{FILE_SERVICE_ADDRESS}{GET_MP3}",
"extensions": ["music_commands", "file_search_commands"],
},
# betoniarka (radio operator colocated with Liquidsoap): radio playlists
"radio": {
"health_url": f"{RADIO_SERVICE_ADDRESS}/ping",
"extensions": ["radio_commands"],
},
# librarian: DOI / Crossref search. NOTE: this one is NOT a plain GET - it
# is gated on a full ping round-trip (see _load_service_groups); the URL
# here is only the label used in the "unreachable" log line.
"librarian": {
"health_url": f"{LIBRARIAN_SERVICE_ADDRESS}{LIBRARIAN_PING}",
"extensions": ["librarian_commands"],
},
}
SERVICE_RECHECK_SECONDS = 300
def _service_health(url: str):
"""Return None when the service answers HTTP at all (any status counts),
otherwise the connection error explaining WHY it's unreachable (refused vs
timeout vs DNS - the difference points straight at the cause)."""
try:
requests.get(url, timeout=3)
return None
except requests.exceptions.RequestException as exc:
return exc
async def _load_extension_safe(name: str) -> bool:
"""Load one extension; log and continue instead of killing startup."""
if name in client.extensions:
return True
try:
await client.load_extension(name)
logger.info("Extension loaded: %s", name)
return True
except Exception: # pylint: disable=broad-exception-caught
logger.exception("Extension FAILED (disabled, bot continues): %s", name)
return False
async def _load_service_groups() -> bool:
"""Health-check each service group and load its cogs when alive.
Returns True when anything new was loaded (caller may want to re-sync).
"""
loaded_any = False
for service, group in SERVICE_EXTENSION_GROUPS.items():
missing = [e for e in group["extensions"] if e not in client.extensions]
if not missing:
continue
if service == "librarian":
# A plain GET only proves Flask is up. The librarian is only useful
# once its internal queue + worker are flowing, so prove exactly that
# with a ping that must complete the full round-trip (see
# communication_subroutine.librarian_ping).
alive = await asyncio.to_thread(
librarian_ping,
LIBRARIAN_SERVICE_ADDRESS,
LIBRARIAN_PING,
service_headers(),
LIBRARIAN_PING_TIMEOUT,
)
if not alive:
logger.warning(
"Service 'librarian' ping round-trip failed (%s) - cogs stay disabled: %s",
group["health_url"],
", ".join(missing),
)
continue
else:
err = await asyncio.to_thread(_service_health, group["health_url"])
if err is not None:
logger.warning(
"Service '%s' unreachable (%s) [%s: %s] - cogs stay disabled: %s",
service,
group["health_url"],
type(err).__name__,
err,
", ".join(missing),
)
continue
logger.info("Service '%s' is alive - enabling: %s", service, ", ".join(missing))
for extension in missing:
if await _load_extension_safe(extension):
loaded_any = True
return loaded_any
async def _sync_tree() -> None:
try:
await client.tree.sync()
except Exception: # pylint: disable=broad-exception-caught
logger.exception("Slash-command tree sync failed (commands may lag)")
async def _service_watchdog() -> None:
"""Periodically retry offline services and enable their cogs when up."""
while not client.is_closed():
await asyncio.sleep(SERVICE_RECHECK_SECONDS)
try:
if await _load_service_groups():
await _sync_tree()
except Exception: # pylint: disable=broad-exception-caught
logger.exception("Service watchdog tick failed")
_STARTUP_DONE = False
# *=========================================== Define Events
@client.event
async def on_ready():
"""Metoda wywoływana przy połączeniu do serwera."""
global _STARTUP_DONE # pylint: disable=global-statement
logger.info("%s has connected to Discord!", client.user)
if _STARTUP_DONE:
logger.info("Reconnected - extensions already loaded")
return
_STARTUP_DONE = True
logger.info("Reactor: online")
for extension in CORE_EXTENSIONS:
await _load_extension_safe(extension)
# Log the ACTUALLY-resolved service addresses. When one shows the built-in
# default (192.168.1.15:5000) it means the matching CONJURER_* env var never
# reached the process - the single most common cause of "service unreachable"
# confusion. Printing them makes env-vs-default obvious at a glance.
logger.info(
"Resolved service addresses -> musician(file): %s | librarian: %s | radio: %s",
FILE_SERVICE_ADDRESS,
LIBRARIAN_SERVICE_ADDRESS,
RADIO_SERVICE_ADDRESS,
)
await _load_service_groups()
logger.info("Sensors: online")
logger.info(client.cogs)
await _sync_tree()
for com in client.commands:
logger.info("Command %s is awejleble", com.qualified_name)
asyncio.create_task(_service_watchdog())
logger.info("Logged in as ----> %s", client.user)
logger.info("ID:%s ", client.user.id)
logger.info("All systems: operational")
# *=========================================== Runtime orchestration
# The legacy bootstrap used two bare threads (client.run + comm_subroutine) and
# joined them, which made a clean shutdown impossible. We now drive everything
# from a single asyncio loop: the Flask comm layer still runs in its own
# threads (via asyncio.to_thread) but is steered through a shared stop_event so
# the bot can stop both halves cooperatively.
async def _run_comm_subroutine(stop_event: threading.Event) -> None:
"""Run the blocking comm subroutine in a worker thread.
A crash here takes down the internal HTTP endpoints only - the Discord
side keeps running, so log loudly and swallow.
"""
try:
await asyncio.to_thread(comm_subroutine, stop_event)
except Exception: # pylint: disable=broad-exception-caught
logger.exception(
"Comm layer crashed - internal HTTP endpoints are down "
"(musician/librarian callbacks will not arrive); bot continues"
)
async def _run_bot(token: str, shutdown_event: asyncio.Event) -> None:
"""Start the Discord client; log WHY it died before flagging shutdown.
Exceptions were previously swallowed by ``gather(return_exceptions=True)``
which made a failed login look like a clean exit (silent crash-loop in
docker). Now every death is diagnosed on the console first.
"""
try:
# NOTE: log_handler is a Client.run()-only kwarg (run() configures
# logging, then calls start()); start() takes just the token. We set
# up our own handlers above, so nothing is lost.
await client.start(token)
except discord.LoginFailure:
logger.critical(
"FATAL: Discord REJECTED the token (Improper token). Fix the "
"'discord' entry in the netrc mounted at CONJURER_NETRC_FILE or "
"the DISCORD_TOKEN env var. If the token leaked/reset, generate a "
"new one in the Discord Developer Portal -> Bot -> Reset Token."
)
raise
except discord.PrivilegedIntentsRequired:
logger.critical(
"FATAL: this bot application does not have the Privileged Gateway "
"Intents enabled. Open Discord Developer Portal -> your app -> "
"Bot -> enable 'Presence', 'Server Members' and 'Message Content' "
"Intents, then restart."
)
raise
except Exception: # pylint: disable=broad-exception-caught
logger.exception("FATAL: Discord client crashed at startup/runtime")
raise
finally:
shutdown_event.set()
async def main() -> int:
"""Run both halves; return a process exit code (0 = clean shutdown)."""
if TOKEN:
token_source = (
"env DISCORD_TOKEN" if os.getenv("DISCORD_TOKEN") else "netrc file"
)
logger.info(
"Discord token: loaded from %s (length %d)", token_source, len(TOKEN)
)
logger.info("Starting discord bot")
shutdown_event = asyncio.Event()
comm_stop_event = threading.Event()
comm_task = asyncio.create_task(_run_comm_subroutine(comm_stop_event))
bot_task = asyncio.create_task(_run_bot(TOKEN, shutdown_event))
exit_code = 0
try:
await shutdown_event.wait()
except (KeyboardInterrupt, asyncio.CancelledError):
logger.info("Shutdown signal received")
comm_stop_event.set()
await client.close()
finally:
comm_stop_event.set()
if not client.is_closed():
await client.close()
results = await asyncio.gather(bot_task, comm_task, return_exceptions=True)
for result in results:
if isinstance(result, BaseException) and not isinstance(
result, asyncio.CancelledError
):
# Already logged with full traceback inside the task; repeat
# the one-liner so it is the LAST thing in `docker logs`.
logger.critical("Task died: %r", result)
exit_code = 1
return exit_code
# *================================== Run
if __name__ == "__main__":
if not TOKEN:
# Loud, unmissable and in `docker logs`: this is THE most common cause
# of a silent container crash-loop.
MSG = (
"FATAL: Discord token missing.\n"
"Provide it via the DISCORD_TOKEN environment variable or a netrc "
"file (machine 'discord') at the path in CONJURER_NETRC_FILE.\n"
"Docker: check that your secrets mount exists, e.g.\n"
" -v /srv/conjurer/secrets/.netrc:/secrets/.netrc:ro\n"
" CONJURER_NETRC_FILE=/secrets/.netrc"
)
logger.critical(MSG)
sys.exit(MSG)
sys.exit(asyncio.run(main()))