Complete DDS training workflow and delivery package
This commit is contained in:
parent
68dd83c7c2
commit
4c4b91064f
229 changed files with 11969 additions and 1024 deletions
|
|
@ -9,6 +9,7 @@
|
|||
"""
|
||||
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
|
|
@ -20,6 +21,11 @@ from app.domain.events import Mood
|
|||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
NUMBER_WORDS = {
|
||||
"один", "одна", "одно", "двое", "два", "две", "трое", "три", "четверо", "четыре",
|
||||
"пять", "шесть", "семь", "восемь", "девять", "десять",
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class CallerLine:
|
||||
|
|
@ -152,6 +158,11 @@ class LlmCaller:
|
|||
persona.on_repeat()
|
||||
mood = persona.remember()
|
||||
|
||||
# Непонятная реплика не передаёт модели право импровизировать фактами.
|
||||
# Отбор фактов и вопросная карта — только слот-автомат, не LLM.
|
||||
if not (turn.revealed or turn.refined or turn.repeated):
|
||||
return await self._fallback.reply(turn, persona, slots)
|
||||
|
||||
facts = {fact.id: fact.value for fact in slots.revealed_facts()}
|
||||
say_now = [
|
||||
facts[fact_id]
|
||||
|
|
@ -161,7 +172,7 @@ class LlmCaller:
|
|||
repeated = [facts[fact_id] for fact_id in turn.repeated if fact_id in facts]
|
||||
|
||||
system = _prompt("caller.md").format(
|
||||
scenario=slots.scenario.title,
|
||||
scenario="учебное происшествие",
|
||||
mood=MOOD_WORDS.get(mood, mood.value),
|
||||
directive=_directive_line(persona),
|
||||
revealed="\n".join(f"- {value}" for value in facts.values()) or "- пока ничего",
|
||||
|
|
@ -173,13 +184,15 @@ class LlmCaller:
|
|||
repeated="; ".join(repeated), repeats=persona.repeats
|
||||
)
|
||||
|
||||
self._history.append({"role": "user", "content": turn.text})
|
||||
current_message = {"role": "user", "content": turn.text}
|
||||
try:
|
||||
text = await self._client.complete(
|
||||
LlmRequest(
|
||||
messages=[{"role": "system", "content": system}, *self._history[-6:]],
|
||||
messages=[{"role": "system", "content": system},
|
||||
*self._history[-6:], current_message],
|
||||
model=self._model,
|
||||
temperature=self._temperature,
|
||||
max_tokens=160,
|
||||
)
|
||||
)
|
||||
except LlmUnavailable as exc:
|
||||
|
|
@ -187,7 +200,13 @@ class LlmCaller:
|
|||
log.warning("звонящий на заготовках: %s", exc)
|
||||
return await self._fallback.reply(turn, persona, slots)
|
||||
|
||||
self._history.append({"role": "assistant", "content": text})
|
||||
if not _allowed_reply(text, facts, say_now + repeated, slots):
|
||||
self.fallbacks += 1
|
||||
log.warning("ответ модели нарушил протокол раскрытия фактов — использована заготовка")
|
||||
return await self._fallback.reply(turn, persona, slots)
|
||||
# Отклонённый ответ и провокационный вопрос не должны загрязнять
|
||||
# последующий контекст. Запоминаем только проверенную пару ходов.
|
||||
self._history.extend((current_message, {"role": "assistant", "content": text}))
|
||||
return CallerLine(text=text, mood=mood)
|
||||
|
||||
async def aclose(self) -> None:
|
||||
|
|
@ -196,6 +215,48 @@ class LlmCaller:
|
|||
await self._client.aclose()
|
||||
|
||||
|
||||
def _allowed_reply(text: str, allowed: dict[str, str], required_now: list[str], slots: SlotMachine) -> bool:
|
||||
"""Консервативная граница для текста модели; протокол 112 всё равно в коде.
|
||||
|
||||
Невозможно доказать истинность произвольной русской фразы регулярками,
|
||||
поэтому сомнительный ответ заменяется детерминированной репликой.
|
||||
"""
|
||||
if not text or len(text) > 300 or "\n" in text or "<think>" in text.lower():
|
||||
return False
|
||||
normalized = text.casefold().replace("ё", "е")
|
||||
# Модель не вправе назвать числовой адрес или телефон, которого нет в
|
||||
# раскрытых фактах, даже если оператор предположил его в своей реплике.
|
||||
allowed_digits = set(re.findall(r"\d+", " ".join(allowed.values())))
|
||||
if any(number not in allowed_digits for number in re.findall(r"\d+", normalized)):
|
||||
return False
|
||||
allowed_words = set(re.findall(r"[а-яё]+", " ".join(allowed.values()).casefold().replace("ё", "е")))
|
||||
spoken_words = set(re.findall(r"[а-яё]+", normalized))
|
||||
if (spoken_words & NUMBER_WORDS) - allowed_words:
|
||||
return False
|
||||
allowed_stems = {word[:4] for word in allowed_words if len(word) >= 4}
|
||||
spoken_stems = {word[:4] for word in spoken_words if len(word) >= 4}
|
||||
for fact in slots.scenario.facts:
|
||||
if fact.id in allowed:
|
||||
continue
|
||||
for value in (fact.value, fact.refined):
|
||||
if not value:
|
||||
continue
|
||||
if len(value) >= 7 and value.casefold().replace("ё", "е") in normalized:
|
||||
return False
|
||||
# Полная фраза — не единственный способ слить скрытый факт:
|
||||
# «муж курил» раскрывает причину, даже без «на балконе».
|
||||
hidden_stems = {word[:4] for word in re.findall(
|
||||
r"[а-яё]{4,}", value.casefold().replace("ё", "е")
|
||||
)} - allowed_stems
|
||||
if hidden_stems & spoken_stems:
|
||||
return False
|
||||
# Новый обязательный факт нельзя опустить ради красивой реплики.
|
||||
for value in required_now:
|
||||
if value.casefold().replace("ё", "е") not in normalized:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
MOOD_WORDS = {
|
||||
Mood.PANIC: "паника, ты кричишь",
|
||||
Mood.AGGRESSIVE: "злость, ты срываешься на оператора",
|
||||
|
|
@ -210,6 +271,8 @@ def _directive_line(persona: PersonaState) -> str:
|
|||
|
||||
if persona.directive in SOFT:
|
||||
return f"ПРЕПОДАВАТЕЛЬ ВЕДЁТ СИТУАЦИЮ: {SOFT[persona.directive].lower()}."
|
||||
if persona.directive:
|
||||
return f"ПРЕПОДАВАТЕЛЬ ПРОСИТ ИЗМЕНИТЬ ПОДАЧУ: {persona.directive}. Факты не меняй."
|
||||
return ""
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ class DirectiveResult:
|
|||
say: str | None = None
|
||||
#: Оборвать связь: звук на полуслове, дальше обратный дозвон.
|
||||
drop_line: bool = False
|
||||
#: Директива требует сети (свободный текст без LLM).
|
||||
#: Директива требует локально работающую модель (без неё — только кнопки).
|
||||
needs_network: bool = False
|
||||
|
||||
|
||||
|
|
@ -72,14 +72,15 @@ def apply(state, directive: str) -> DirectiveResult:
|
|||
state.slots.invalidate(fact.id)
|
||||
return DirectiveResult(applied=True, say=HARD_LINES[directive])
|
||||
|
||||
# Свободный текст уходит в контекст персоны — но подставить его в реплику
|
||||
# может только LLM. Офлайн-дерево предгенерировано, произвольную фразу
|
||||
# взять неоткуда (docs/arch/CONTRACT.md).
|
||||
# Свободный текст разрешён лишь в режиме локальной модели. Он управляет
|
||||
# интонацией, но не раскрывает факты и не меняет правила 112.
|
||||
from app.dialog.caller import LlmCaller
|
||||
|
||||
if not isinstance(state.caller, LlmCaller) or len(directive) > 500:
|
||||
return DirectiveResult(applied=False, needs_network=True)
|
||||
if state.persona is not None:
|
||||
state.persona.directive = None
|
||||
if state.directives is not None:
|
||||
state.directives.append(directive)
|
||||
return DirectiveResult(applied=False, needs_network=True)
|
||||
state.persona.directive = directive
|
||||
return DirectiveResult(applied=True)
|
||||
|
||||
|
||||
def mood_of(state) -> Mood:
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Кто играет звонящего: LLM, если есть ключ, иначе заготовки.
|
||||
"""Кто играет звонящего: локальная LLM, таблица или заготовки.
|
||||
|
||||
Провайдер и модель меняются значением в конфиге, а не кодом. Заготовки —
|
||||
не запасной костыль, а рабочий режим: занятие идёт и без сети.
|
||||
|
|
@ -9,30 +9,45 @@ import logging
|
|||
from app.config import get_settings
|
||||
from app.dialog.caller import Caller, LlmCaller, TemplateCaller
|
||||
from app.dialog.llm import LlmClient
|
||||
from app.dialog.llm import is_loopback_url
|
||||
from app.dialog.tree import TreeCaller, has_table
|
||||
from app.db.base import get_sessionmaker
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def build_caller(scenario_id: str | None = None, sessionmaker=None) -> Caller:
|
||||
def build_caller(
|
||||
scenario_id: str | None = None,
|
||||
sessionmaker=None,
|
||||
*,
|
||||
use_pregenerated: bool = False,
|
||||
) -> Caller:
|
||||
settings = get_settings()
|
||||
|
||||
# Офлайн и «нет ключа» — это один и тот же путь: предгенерированная таблица,
|
||||
# а не локальная модель. Формулировки в ней от облачной модели, а задержка
|
||||
# нулевая (docs/arch/STACK.md).
|
||||
if (settings.offline or not settings.llm_api_key) and scenario_id and has_table(scenario_id):
|
||||
local = settings.llm_provider == "local"
|
||||
mode = settings.dialogue_model_mode
|
||||
if mode not in {"dialogue", "russian_control"}:
|
||||
raise ValueError(f"неизвестный DIALOGUE_MODEL_MODE: {mode}")
|
||||
model = settings.llm_model_control if mode == "russian_control" else settings.llm_model_caller
|
||||
base_url = settings.llm_control_base_url if mode == "russian_control" else settings.llm_base_url
|
||||
model_allowed = settings.llm_provider != "disabled" and bool(model and base_url and (
|
||||
is_loopback_url(base_url, allow_docker_host=settings.allow_docker_host_models)
|
||||
if settings.offline or local else settings.llm_api_key
|
||||
))
|
||||
|
||||
# Локальный loopback не считается внешней сетью. Если сервер модели не
|
||||
# отвечает, LlmCaller откатится на проверенные заготовки этой же реплики.
|
||||
if model_allowed:
|
||||
client = LlmClient(sessionmaker=sessionmaker or _safe_sessionmaker(), base_url=base_url)
|
||||
log.info("звонящий: режим %s, модель %s, адрес %s", mode, model, base_url)
|
||||
return LlmCaller(client, model=model)
|
||||
|
||||
if use_pregenerated and scenario_id and has_table(scenario_id):
|
||||
log.info("звонящий по предгенерированной таблице сценария %s", scenario_id)
|
||||
return TreeCaller(scenario_id)
|
||||
|
||||
if not settings.llm_api_key or settings.offline:
|
||||
reason = "офлайн-режим" if settings.offline else "нет ключа LLM"
|
||||
log.info("звонящий отвечает заготовками: %s (таблицы нет — make pregen)", reason)
|
||||
return TemplateCaller()
|
||||
|
||||
client = LlmClient(sessionmaker=sessionmaker or _safe_sessionmaker())
|
||||
log.info("звонящий на модели %s", settings.llm_model_caller)
|
||||
return LlmCaller(client, model=settings.llm_model_caller)
|
||||
log.info("звонящий отвечает фактами сценария: модель отключена или предгенерация не разрешена")
|
||||
return TemplateCaller()
|
||||
|
||||
|
||||
def _safe_sessionmaker():
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Клиент облачной LLM за интерфейсом: провайдер меняется значением в конфиге.
|
||||
"""Клиент совместимого API для локальных или внешних моделей.
|
||||
|
||||
Кэш ответов по хешу контекста лежит в Postgres, а не в Redis: база уже поднята,
|
||||
лишняя движущаяся часть на стенде не нужна (docs/arch/STACK.md). Кэш работает
|
||||
|
|
@ -9,7 +9,10 @@
|
|||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from ipaddress import ip_address
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import httpx
|
||||
from sqlalchemy import select
|
||||
|
|
@ -26,6 +29,48 @@ class LlmUnavailable(RuntimeError):
|
|||
занятие продолжается — молчащий звонящий хуже шаблонной фразы."""
|
||||
|
||||
|
||||
def is_loopback_url(value: str, *, allow_docker_host: bool = False) -> bool:
|
||||
"""В офлайн-режиме модели разрешены лишь на той же машине.
|
||||
|
||||
Не доверяем доменам или hosts-записям: они могут указывать наружу.
|
||||
"""
|
||||
try:
|
||||
url = urlsplit(value)
|
||||
host = url.hostname or ""
|
||||
try:
|
||||
local_host = ip_address(host).is_loopback
|
||||
except ValueError:
|
||||
local_host = allow_docker_host and host == "host.docker.internal"
|
||||
return (url.scheme == "http" and local_host and url.port is not None
|
||||
and not url.username and not url.password)
|
||||
except (ValueError, TypeError):
|
||||
return False
|
||||
|
||||
|
||||
def _spoken_content(raw: str, *, strip_reasoning: bool = False) -> str:
|
||||
"""Убрать только пустой служебный хвост Qwen3, не рассуждения модели.
|
||||
|
||||
llama.cpp с выключенным thinking иногда возвращает в `content` один
|
||||
закрывающий `</think>` перед самой репликой. Внутренний текст размышлений
|
||||
мы намеренно не вырезаем: если он есть, ответ небезопасен и идёт fallback.
|
||||
"""
|
||||
text = raw.strip()
|
||||
if strip_reasoning:
|
||||
closing = list(re.finditer(r"</think>\s*", text, re.IGNORECASE))
|
||||
if closing:
|
||||
text = text[closing[-1].end():].strip()
|
||||
elif text.startswith("<|"):
|
||||
start = text.find("{")
|
||||
if start >= 0:
|
||||
text = text[start:].strip()
|
||||
text = re.sub(r"^(?:</think>\s*)+", "", text, flags=re.IGNORECASE).strip()
|
||||
if re.search(r"</?think\b", text, re.IGNORECASE) or "<|" in text:
|
||||
raise LlmUnavailable("ответ содержит служебные токены модели")
|
||||
if not text:
|
||||
raise LlmUnavailable("пустой ответ модели: весь бюджет токенов ушёл в рассуждение")
|
||||
return text
|
||||
|
||||
|
||||
@dataclass
|
||||
class LlmRequest:
|
||||
messages: list[dict]
|
||||
|
|
@ -34,10 +79,15 @@ class LlmRequest:
|
|||
# С запасом на рассуждающие модели: Qwen3 тратит на размышление сотни токенов
|
||||
# и при малом бюджете возвращает пустой ответ с finish_reason="length".
|
||||
max_tokens: int = 400
|
||||
response_format: dict | None = None
|
||||
# Только для внутренних структурированных задач. В репликах звонящего
|
||||
# рассуждение всегда отвергается, чтобы оно не попало в эфир.
|
||||
strip_reasoning: bool = False
|
||||
|
||||
def cache_key(self) -> str:
|
||||
payload = json.dumps(
|
||||
{"m": self.model, "t": self.temperature, "msgs": self.messages},
|
||||
{"m": self.model, "t": self.temperature, "msgs": self.messages,
|
||||
"format": self.response_format, "strip_reasoning": self.strip_reasoning},
|
||||
ensure_ascii=False,
|
||||
sort_keys=True,
|
||||
)
|
||||
|
|
@ -50,36 +100,43 @@ class LlmClient:
|
|||
*,
|
||||
sessionmaker: async_sessionmaker | None = None,
|
||||
transport: httpx.AsyncBaseTransport | None = None,
|
||||
base_url: str | None = None,
|
||||
# Ответ дольше этого бессмысленен: бюджет хода — 1.5 с, а звонящий
|
||||
# с заготовками ответит сразу.
|
||||
timeout: float = 8.0,
|
||||
) -> None:
|
||||
settings = get_settings()
|
||||
self._base_url = settings.llm_base_url.rstrip("/")
|
||||
self._base_url = (base_url or settings.llm_base_url).rstrip("/")
|
||||
self._key = settings.llm_api_key
|
||||
self._local_only = settings.offline or settings.llm_provider == "local"
|
||||
self._allow_docker_host = settings.allow_docker_host_models
|
||||
self._sessionmaker = sessionmaker
|
||||
self._client = httpx.AsyncClient(timeout=timeout, transport=transport)
|
||||
self._client = httpx.AsyncClient(timeout=timeout, transport=transport, trust_env=False)
|
||||
|
||||
@property
|
||||
def configured(self) -> bool:
|
||||
if self._local_only:
|
||||
return is_loopback_url(
|
||||
self._base_url, allow_docker_host=self._allow_docker_host
|
||||
)
|
||||
return bool(self._key and self._base_url)
|
||||
|
||||
async def complete(self, request: LlmRequest, *, use_cache: bool = True) -> str:
|
||||
"""Ответ модели. Кэш по хешу контекста: та же реплика на том же месте
|
||||
занятия звучит одинаково у каждой группы."""
|
||||
if not self.configured:
|
||||
raise LlmUnavailable("не задан ключ или адрес провайдера")
|
||||
raise LlmUnavailable("локальный адрес модели недопустим или провайдер не настроен")
|
||||
|
||||
key = request.cache_key()
|
||||
if use_cache:
|
||||
cached = await self._from_cache(key)
|
||||
if cached is not None:
|
||||
return cached
|
||||
return _spoken_content(cached, strip_reasoning=request.strip_reasoning)
|
||||
|
||||
try:
|
||||
response = await self._client.post(
|
||||
f"{self._base_url}/chat/completions",
|
||||
headers={"Authorization": f"Bearer {self._key}"},
|
||||
headers={"Authorization": f"Bearer {self._key}"} if self._key else {},
|
||||
json={
|
||||
"model": request.model,
|
||||
"messages": request.messages,
|
||||
|
|
@ -88,6 +145,8 @@ class LlmClient:
|
|||
# Рассуждение в ответе не нужно: оно только раздувает трафик.
|
||||
# Провайдеры, которые про это поле не знают, его игнорируют.
|
||||
"reasoning": {"exclude": True},
|
||||
**({"response_format": request.response_format}
|
||||
if request.response_format is not None else {}),
|
||||
},
|
||||
)
|
||||
except httpx.HTTPError as exc:
|
||||
|
|
@ -97,13 +156,14 @@ class LlmClient:
|
|||
# Тело ошибки в лог, ключ в заголовке — не логируется.
|
||||
raise LlmUnavailable(f"HTTP {response.status_code}: {response.text[:200]}")
|
||||
|
||||
message = response.json()["choices"][0]["message"]
|
||||
text = (message.get("content") or "").strip()
|
||||
if not text:
|
||||
# У рассуждающих моделей при нехватке бюджета весь ответ уходит
|
||||
# в размышление, а content приходит пустым. Для занятия это отказ:
|
||||
# звонящий откатится на заготовку, а не промолчит.
|
||||
raise LlmUnavailable("пустой ответ модели: весь бюджет токенов ушёл в рассуждение")
|
||||
try:
|
||||
message = response.json()["choices"][0]["message"]
|
||||
content = message.get("content")
|
||||
if content is not None and not isinstance(content, str):
|
||||
raise TypeError("content не строка")
|
||||
text = _spoken_content(content or "", strip_reasoning=request.strip_reasoning)
|
||||
except (ValueError, KeyError, IndexError, TypeError, AttributeError) as exc:
|
||||
raise LlmUnavailable("некорректный ответ локальной модели") from exc
|
||||
if use_cache and text:
|
||||
await self._to_cache(key, request, text)
|
||||
return text
|
||||
|
|
|
|||
|
|
@ -19,3 +19,5 @@
|
|||
4. Если состояние — паника или крик: обрывки, повторы, незаконченные фразы.
|
||||
5. Не задавай оператору вопросов о ходе разговора и не подсказывай ему, что спросить.
|
||||
6. Отвечай только репликой, без пояснений и без кавычек.
|
||||
7. Каждый факт из раздела «ЧТО НУЖНО СКАЗАТЬ ЭТОЙ РЕПЛИКОЙ» произнеси
|
||||
полностью и дословно. Одного «да», «нет» или намёка недостаточно.
|
||||
|
|
|
|||
Loading…
Reference in a new issue