fix(server): идемпотентность upload, WAL, буфер ELM, таймаут LLM

- api/db.py: WAL mode, busy_timeout, request_id UNIQUE, close(), контекстный менеджер
- api/routes.py: проверка request_id при upload, /ping-llm кэш 60с, /chat через roles
- api/config.py: lru_cache на load()
- obd/protocol.py: reset_input_buffer перед _write(), условный READY в send()
- brain/client.py: модель 120b, таймаут из параметра, LLMError класс, обработка 429/5xx
This commit is contained in:
Repinoid
2026-05-31 16:56:21 +03:00
parent eab73cd76a
commit 6893ae33f5
5 changed files with 177 additions and 100 deletions
+49 -21
View File
@@ -14,39 +14,67 @@ brain/client.py — LLM-клиент для диагностики авто.
qwen3-6-27b-fp8 — быстрая (но CoT leak bug)
"""
import logging
import requests
logger = logging.getLogger("brain.client")
DEFAULT_BASE = "https://api.aillm.ru/v1"
DEFAULT_MODEL = "gpt-oss-20b"
DEFAULT_MODEL = "gpt-oss-120b"
DEFAULT_TIMEOUT = 180
class LLMError(Exception):
"""Ошибка LLM API с безопасным для клиента сообщением."""
pass
class Diagnoser:
"""Отправляет данные в DeepSeek и возвращает диагноз."""
"""LLM-клиент для OpenAI-совместимого API."""
def __init__(self, api_key: str, model: str = DEFAULT_MODEL, base_url: str = DEFAULT_BASE):
def __init__(self, api_key: str, model: str = DEFAULT_MODEL,
base_url: str = DEFAULT_BASE, timeout: int = DEFAULT_TIMEOUT):
self.api_key = api_key
self.model = model
self.base_url = base_url.rstrip("/")
self.timeout = timeout
def ask(self, messages: list[dict]) -> str:
"""Отправляет сообщения в DeepSeek, возвращает текст ответа."""
resp = requests.post(
f"{self.base_url}/chat/completions",
headers={
"Authorization": f"Bearer {self.api_key}",
"Content-Type": "application/json",
},
json={
"model": self.model,
"messages": messages,
"temperature": 0.3, # пониже — меньше фантазий
"max_tokens": 4096,
},
timeout=120, # api.aillm.ru бывает медленным
)
resp.raise_for_status()
data = resp.json()
return data["choices"][0]["message"]["content"]
"""Отправляет сообщения в LLM API, возвращает текст ответа."""
try:
resp = requests.post(
f"{self.base_url}/chat/completions",
headers={
"Authorization": f"Bearer {self.api_key}",
"Content-Type": "application/json",
},
json={
"model": self.model,
"messages": messages,
"temperature": 0.3,
"max_tokens": 4096,
},
timeout=self.timeout,
)
resp.raise_for_status()
data = resp.json()
return data["choices"][0]["message"]["content"]
except requests.Timeout:
logger.warning(f"LLM timeout after {self.timeout}s")
raise LLMError("LLM не ответил вовремя. Попробуйте позже.")
except requests.HTTPError as e:
status = e.response.status_code if e.response is not None else 0
logger.warning(f"LLM HTTP {status}: {e}")
if status == 429:
raise LLMError("Слишком много запросов. Подождите минуту.")
elif 500 <= status < 600:
raise LLMError("LLM временно недоступен. Попробуйте позже.")
else:
raise LLMError("Ошибка LLM. Попробуйте позже.")
except Exception as e:
logger.error(f"LLM unexpected: {e}")
raise LLMError("LLM временно недоступен.")
def diagnose(
self,