diff --git a/AI/talk_handler.py b/AI/talk_handler.py index 1a25a3b..c12daee 100644 --- a/AI/talk_handler.py +++ b/AI/talk_handler.py @@ -614,17 +614,21 @@ async def _call_llm( ) -> str: disable_thinking = FORCE_DISABLE_THINKING if disable_thinking is None else disable_thinking - result = await _call_single_model( - messages, - api_url=LLAMA_API_URL, - api_key=LLAMA_API_KEY, - model=LLAMA_MODEL, - max_tokens=max_tokens, - temperature=temperature, - top_p=top_p, - disable_thinking=disable_thinking, - reasoning_budget=reasoning_budget, - ) + try: + result = await _call_single_model( + messages, + api_url=LLAMA_API_URL, + api_key=LLAMA_API_KEY, + model=LLAMA_MODEL, + max_tokens=max_tokens, + temperature=temperature, + top_p=top_p, + disable_thinking=disable_thinking, + reasoning_budget=reasoning_budget, + ) + except Exception: + logger.exception("Primary model call failed") + result = "" # Если основная модель не смогла ответить — пробуем fallback if LLAMA_FALLBACK_API_URL and (not result or _CANT_ANSWER_RE.search(result)):