fix: fall back to secondary LLM when the primary call raises
Primary model errors (HTTP 4xx/5xx, connection failures) raised out of _call_llm uncaught, skipping the fallback branch entirely — fallback only ran when the primary returned an empty/"can't answer" string. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
parent
0cfa67e85f
commit
47ef30800f
1 changed files with 15 additions and 11 deletions
|
|
@ -614,17 +614,21 @@ async def _call_llm(
|
||||||
) -> str:
|
) -> str:
|
||||||
disable_thinking = FORCE_DISABLE_THINKING if disable_thinking is None else disable_thinking
|
disable_thinking = FORCE_DISABLE_THINKING if disable_thinking is None else disable_thinking
|
||||||
|
|
||||||
result = await _call_single_model(
|
try:
|
||||||
messages,
|
result = await _call_single_model(
|
||||||
api_url=LLAMA_API_URL,
|
messages,
|
||||||
api_key=LLAMA_API_KEY,
|
api_url=LLAMA_API_URL,
|
||||||
model=LLAMA_MODEL,
|
api_key=LLAMA_API_KEY,
|
||||||
max_tokens=max_tokens,
|
model=LLAMA_MODEL,
|
||||||
temperature=temperature,
|
max_tokens=max_tokens,
|
||||||
top_p=top_p,
|
temperature=temperature,
|
||||||
disable_thinking=disable_thinking,
|
top_p=top_p,
|
||||||
reasoning_budget=reasoning_budget,
|
disable_thinking=disable_thinking,
|
||||||
)
|
reasoning_budget=reasoning_budget,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logger.exception("Primary model call failed")
|
||||||
|
result = ""
|
||||||
|
|
||||||
# Если основная модель не смогла ответить — пробуем fallback
|
# Если основная модель не смогла ответить — пробуем fallback
|
||||||
if LLAMA_FALLBACK_API_URL and (not result or _CANT_ANSWER_RE.search(result)):
|
if LLAMA_FALLBACK_API_URL and (not result or _CANT_ANSWER_RE.search(result)):
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue