From 47ef30800fc72e658e2832e8e753f6e8404341b0 Mon Sep 17 00:00:00 2001 From: zovos Date: Wed, 2 Sep 2026 11:32:46 +0000 Subject: [PATCH] fix: fall back to secondary LLM when the primary call raises MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Primary model errors (HTTP 4xx/5xx, connection failures) raised out of _call_llm uncaught, skipping the fallback branch entirely — fallback only ran when the primary returned an empty/"can't answer" string. Co-Authored-By: Claude Sonnet 5 --- AI/talk_handler.py | 26 +++++++++++++++----------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/AI/talk_handler.py b/AI/talk_handler.py index 1a25a3b..c12daee 100644 --- a/AI/talk_handler.py +++ b/AI/talk_handler.py @@ -614,17 +614,21 @@ async def _call_llm( ) -> str: disable_thinking = FORCE_DISABLE_THINKING if disable_thinking is None else disable_thinking - result = await _call_single_model( - messages, - api_url=LLAMA_API_URL, - api_key=LLAMA_API_KEY, - model=LLAMA_MODEL, - max_tokens=max_tokens, - temperature=temperature, - top_p=top_p, - disable_thinking=disable_thinking, - reasoning_budget=reasoning_budget, - ) + try: + result = await _call_single_model( + messages, + api_url=LLAMA_API_URL, + api_key=LLAMA_API_KEY, + model=LLAMA_MODEL, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + disable_thinking=disable_thinking, + reasoning_budget=reasoning_budget, + ) + except Exception: + logger.exception("Primary model call failed") + result = "" # Если основная модель не смогла ответить — пробуем fallback if LLAMA_FALLBACK_API_URL and (not result or _CANT_ANSWER_RE.search(result)):