lct-12: детерминированная оценка по ГОСТ, таксономия, радар
Каждая метрика — факт против норматива со ссылкой: «94 с при ≤ 75 с, ГОСТ Р 22.7.03-2021», а не балл. По недобытому факту — отдельная отметка E1 с эталонным вопросом: в разборе нужен конкретный вопрос, не процент. Радар — проекция тех же метрик без пересчёта весов, коммуникация без судьи не рисуется нулём. Таймер, не остановленный событием, — провал, а не зачёт: время недоказуемо, операция не завершена. Метрика, которую нечем посчитать (полнота опроса без модели эмбеддингов), видна как «не посчитано» — молча выброшенная выглядела бы пройденной. Найдено противоречие: таймеры опроса (75 с) и оповещения ДДС (60 с) стартовали на ответе и останавливались передачей в ДДС — один отрезок, два лимита. Опрос за законные 70 с давал E3, а на экране курсанта краснел таймер посреди нормального разговора. События «опрос закончен» в контракте нет, поэтому dds_notify снят с учёта, вопрос записан в CONTRACT.md. KIO проверяет присваивание: card.dds = "03" клало в карточку сырую строку вместо кода ДДС, и падала уже оценка, далеко от места ошибки.
This commit is contained in:
parent
7c97310938
commit
f01f2dab76
9 changed files with 489 additions and 7 deletions
|
|
@ -261,7 +261,7 @@ class SelfAssessmentSubmit(BaseModel):
|
||||||
|
|
||||||
|
|
||||||
class DdsDispatch(BaseModel):
|
class DdsDispatch(BaseModel):
|
||||||
"""Передача в ДДС. Замораживает карточку снимком и останавливает `dds_notify`."""
|
"""Передача в ДДС. Замораживает карточку снимком и останавливает опрос (`interview`)."""
|
||||||
|
|
||||||
type: Literal["dds.dispatch"] = "dds.dispatch"
|
type: Literal["dds.dispatch"] = "dds.dispatch"
|
||||||
service: DDSCode
|
service: DDSCode
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ from enum import StrEnum
|
||||||
from typing import Any
|
from typing import Any
|
||||||
from uuid import UUID, uuid4
|
from uuid import UUID, uuid4
|
||||||
|
|
||||||
from pydantic import BaseModel, Field
|
from pydantic import BaseModel, ConfigDict, Field
|
||||||
|
|
||||||
from app.domain.classifiers import DDSCode, IncidentType
|
from app.domain.classifiers import DDSCode, IncidentType
|
||||||
|
|
||||||
|
|
@ -73,7 +73,13 @@ class UtilityDetails(BaseModel):
|
||||||
|
|
||||||
class KIO(BaseModel):
|
class KIO(BaseModel):
|
||||||
"""Полная карточка. Наблюдателям уходит целиком (`kio.state`),
|
"""Полная карточка. Наблюдателям уходит целиком (`kio.state`),
|
||||||
курсанту — дельтой (`kio.patch`)."""
|
курсанту — дельтой (`kio.patch`).
|
||||||
|
|
||||||
|
Присваивание проверяется: без этого `card.dds = "03"` кладёт в карточку
|
||||||
|
сырую строку вместо кода ДДС, и падает уже оценка, далеко от места ошибки.
|
||||||
|
"""
|
||||||
|
|
||||||
|
model_config = ConfigDict(validate_assignment=True)
|
||||||
|
|
||||||
# Служебное — заполняется системой
|
# Служебное — заполняется системой
|
||||||
card_id: UUID = Field(default_factory=uuid4)
|
card_id: UUID = Field(default_factory=uuid4)
|
||||||
|
|
|
||||||
0
backend/app/scoring/__init__.py
Normal file
0
backend/app/scoring/__init__.py
Normal file
34
backend/app/scoring/competency.py
Normal file
34
backend/app/scoring/competency.py
Normal file
|
|
@ -0,0 +1,34 @@
|
||||||
|
"""Радар шести компетенций.
|
||||||
|
|
||||||
|
Чистая проекция уже посчитанных метрик, без пересчёта весов: радар и оценка
|
||||||
|
всегда согласованы, и расхождение между ними невозможно по построению
|
||||||
|
(docs/product/METHODOLOGY.md#компетентная-модель).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from app.domain.events import CompetencyScore, Metric
|
||||||
|
from app.domain.taxonomy import Competency
|
||||||
|
from app.scoring.taxonomy import METRIC_MAP
|
||||||
|
|
||||||
|
|
||||||
|
def radar(metrics: list[Metric]) -> list[CompetencyScore]:
|
||||||
|
"""Доля пройденного веса по каждой компетенции, 0–1.
|
||||||
|
|
||||||
|
Компетенция без метрик не рисуется нулём: «коммуникация» без судьи —
|
||||||
|
это «не оценивалось», а не «провалено».
|
||||||
|
"""
|
||||||
|
total: dict[Competency, float] = {}
|
||||||
|
passed: dict[Competency, float] = {}
|
||||||
|
for metric in metrics:
|
||||||
|
mapping = METRIC_MAP.get(metric.key)
|
||||||
|
if mapping is None:
|
||||||
|
continue
|
||||||
|
competency = mapping[1]
|
||||||
|
total[competency] = total.get(competency, 0.0) + metric.weight
|
||||||
|
if metric.passed:
|
||||||
|
passed[competency] = passed.get(competency, 0.0) + metric.weight
|
||||||
|
|
||||||
|
return [
|
||||||
|
CompetencyScore(competency=competency.value, value=round(passed.get(competency, 0.0) / weight, 3))
|
||||||
|
for competency in Competency
|
||||||
|
if (weight := total.get(competency))
|
||||||
|
]
|
||||||
231
backend/app/scoring/gost.py
Normal file
231
backend/app/scoring/gost.py
Normal file
|
|
@ -0,0 +1,231 @@
|
||||||
|
"""Детерминированный слой оценки — 60% веса, считается кодом.
|
||||||
|
|
||||||
|
Воспроизводится стопроцентно: один и тот же ход занятия даёт один и тот же
|
||||||
|
результат. Каждая метрика — «факт против норматива со ссылкой», а не балл:
|
||||||
|
«опрос 94 с при нормативе 75 с (ГОСТ Р 22.7.03-2021)» можно предъявить
|
||||||
|
и проверить руками (docs/product/DEBRIEF.md).
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
|
||||||
|
from app.domain.events import CallEndReason, Metric
|
||||||
|
from app.domain.kio import KIO, missing_fields
|
||||||
|
from app.domain.taxonomy import ERRORS, Competency, Finding, FindingSource
|
||||||
|
from app.domain.timers import GOST_REF, NORMATIVES, TimerCode
|
||||||
|
from app.scenarios.schema import Scenario
|
||||||
|
from app.scoring.taxonomy import METRIC_MAP
|
||||||
|
from app.session.timers import SessionTimers
|
||||||
|
|
||||||
|
SOURCE_BY_CODE = {
|
||||||
|
"E1": FindingSource.SLOTS,
|
||||||
|
"E2": FindingSource.GROUND_TRUTH,
|
||||||
|
"E3": FindingSource.TIMERS,
|
||||||
|
"E5": FindingSource.KIO,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class GostResult:
|
||||||
|
metrics: list[Metric] = field(default_factory=list)
|
||||||
|
findings: list[Finding] = field(default_factory=list)
|
||||||
|
#: Метрики, которые посчитать было нечем. Не штрафуют, но видны в отчёте:
|
||||||
|
#: молча выброшенная метрика выглядит как пройденная.
|
||||||
|
unavailable: list[str] = field(default_factory=list)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def score(self) -> float:
|
||||||
|
"""Доля пройденного веса, 0–100."""
|
||||||
|
total = sum(metric.weight for metric in self.metrics)
|
||||||
|
if not total:
|
||||||
|
return 0.0
|
||||||
|
passed = sum(metric.weight for metric in self.metrics if metric.passed)
|
||||||
|
return round(100 * passed / total, 1)
|
||||||
|
|
||||||
|
|
||||||
|
def _seconds(ms: int) -> str:
|
||||||
|
return f"{round(ms / 1000)} с"
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_address(text: str | None) -> set[str]:
|
||||||
|
"""Слова адреса без служебных: «ул. Ленина д. 14» и «улица Ленина, 14»
|
||||||
|
должны совпасть, иначе курсанта штрафуют за сокращение."""
|
||||||
|
if not text:
|
||||||
|
return set()
|
||||||
|
noise = {"улица", "ул", "дом", "д", "проспект", "пр", "переулок", "пер", "г", "город", "москва"}
|
||||||
|
words = re.findall(r"[\w-]+", text.lower().replace("ё", "е"))
|
||||||
|
return {word for word in words if word not in noise}
|
||||||
|
|
||||||
|
|
||||||
|
class _Builder:
|
||||||
|
def __init__(self) -> None:
|
||||||
|
self.result = GostResult()
|
||||||
|
|
||||||
|
def add(self, key: str, title: str, fact: str, norm: str, passed: bool, ref: str | None = None,
|
||||||
|
finding: str | None = None) -> None:
|
||||||
|
self.result.metrics.append(
|
||||||
|
Metric(key=key, title=title, fact=fact, norm=norm, ref=ref, passed=passed)
|
||||||
|
)
|
||||||
|
if passed:
|
||||||
|
return
|
||||||
|
code, competency = METRIC_MAP[key]
|
||||||
|
self.result.findings.append(
|
||||||
|
Finding(
|
||||||
|
code=code,
|
||||||
|
source=SOURCE_BY_CODE[code.value],
|
||||||
|
summary=finding or f"{ERRORS[code].title}: {title.lower()}",
|
||||||
|
fact=fact,
|
||||||
|
norm=norm,
|
||||||
|
ref=ref,
|
||||||
|
competency=competency,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def timer(self, key: str, code: TimerCode, timers: SessionTimers, limit_ms: int,
|
||||||
|
not_stopped: str) -> None:
|
||||||
|
normative = NORMATIVES[code]
|
||||||
|
norm = f"≤ {_seconds(limit_ms)}"
|
||||||
|
measured = timers.measured_ms(code)
|
||||||
|
if measured is None:
|
||||||
|
# Таймер не остановлен событием — время недоказуемо, и это провал:
|
||||||
|
# норматив не выполнен, пока операция не завершена.
|
||||||
|
self.add(key, normative.title, not_stopped, norm, passed=False, ref=GOST_REF)
|
||||||
|
return
|
||||||
|
self.add(
|
||||||
|
key,
|
||||||
|
normative.title,
|
||||||
|
f"{_seconds(measured)}",
|
||||||
|
norm,
|
||||||
|
passed=measured <= limit_ms,
|
||||||
|
ref=GOST_REF,
|
||||||
|
finding=f"{normative.title}: {_seconds(measured)} при нормативе {_seconds(limit_ms)}",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def evaluate(
|
||||||
|
*,
|
||||||
|
scenario: Scenario,
|
||||||
|
kio: KIO,
|
||||||
|
timers: SessionTimers,
|
||||||
|
revealed_facts: list[str] | None,
|
||||||
|
end_reason: CallEndReason | None = None,
|
||||||
|
) -> GostResult:
|
||||||
|
"""Посчитать детерминированный слой по завершённому занятию.
|
||||||
|
|
||||||
|
`revealed_facts` — из слот-автомата. None означает, что автомата не было
|
||||||
|
(нет модели эмбеддингов): полнота опроса тогда не считается и не штрафует.
|
||||||
|
"""
|
||||||
|
build = _Builder()
|
||||||
|
truth = scenario.ground_truth
|
||||||
|
|
||||||
|
# ── нормативы времени, E3 ──
|
||||||
|
build.timer("answer_time", TimerCode.ANSWER, timers, timers.limits[TimerCode.ANSWER],
|
||||||
|
"вызов не принят")
|
||||||
|
build.timer("interview_time", TimerCode.INTERVIEW, timers, timers.limits[TimerCode.INTERVIEW],
|
||||||
|
"опрос не завершён передачей в ДДС — время не зафиксировано")
|
||||||
|
build.result.unavailable.append(
|
||||||
|
"dds_notify_time: нет события конца опроса, от которого отсчитывать 60 с "
|
||||||
|
"(см. docs/arch/CONTRACT.md, коды таймеров)"
|
||||||
|
)
|
||||||
|
|
||||||
|
if end_reason is CallEndReason.DROPPED:
|
||||||
|
callback = timers.timers.get(TimerCode.CALLBACK)
|
||||||
|
attempts = callback.attempt if callback and callback.started_at is not None else 0
|
||||||
|
limit = NORMATIVES[TimerCode.CALLBACK]
|
||||||
|
build.add(
|
||||||
|
"callback", limit.title,
|
||||||
|
f"попыток дозвона: {attempts}" if attempts else "обратного дозвона не было",
|
||||||
|
f"не более {limit.attempts} попыток по {_seconds(limit.limit_ms)}",
|
||||||
|
passed=0 < attempts <= limit.attempts,
|
||||||
|
ref=GOST_REF,
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── полнота опроса, E1 ──
|
||||||
|
if revealed_facts is None:
|
||||||
|
build.result.unavailable.append(
|
||||||
|
"checklist_completeness: нет слот-автомата (не скачана модель эмбеддингов)"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
required = truth.required_facts
|
||||||
|
got = [fact_id for fact_id in required if fact_id in revealed_facts]
|
||||||
|
build.result.metrics.append(
|
||||||
|
Metric(
|
||||||
|
key="checklist_completeness",
|
||||||
|
title="Полнота опроса",
|
||||||
|
fact=f"добыто {len(got)} из {len(required)} обязательных фактов",
|
||||||
|
norm="все обязательные факты",
|
||||||
|
ref="чек-лист сценария",
|
||||||
|
passed=len(got) == len(required),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
# По отметке на каждый недобытый факт: в разборе нужен конкретный
|
||||||
|
# пропущенный вопрос, а не процент.
|
||||||
|
questions = {item.fact: item.question for item in scenario.checklist if item.fact}
|
||||||
|
for fact_id in required:
|
||||||
|
if fact_id in revealed_facts:
|
||||||
|
continue
|
||||||
|
question = questions.get(fact_id)
|
||||||
|
build.result.findings.append(
|
||||||
|
Finding(
|
||||||
|
code=METRIC_MAP["checklist_completeness"][0],
|
||||||
|
source=FindingSource.SLOTS,
|
||||||
|
summary=f"Не добыт обязательный факт {fact_id}",
|
||||||
|
fact="вопрос не прозвучал",
|
||||||
|
norm=f"эталонный вопрос: «{question}»" if question else "обязательный факт сценария",
|
||||||
|
ref="чек-лист сценария",
|
||||||
|
competency=Competency.INTERVIEW,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── классификация и маршрутизация, E2 ──
|
||||||
|
if truth.incident_type is not None:
|
||||||
|
actual = kio.incident_type.value if kio.incident_type else "не указан"
|
||||||
|
build.add(
|
||||||
|
"incident_type", "Тип происшествия",
|
||||||
|
actual, truth.incident_type.value,
|
||||||
|
passed=kio.incident_type == truth.incident_type,
|
||||||
|
ref="классификатор происшествий",
|
||||||
|
finding=f"Тип происшествия {actual}, верный — {truth.incident_type.value}",
|
||||||
|
)
|
||||||
|
if truth.dds is not None:
|
||||||
|
actual = kio.dds.value if kio.dds else "не выбрана"
|
||||||
|
build.add(
|
||||||
|
"dds_choice", "Выбор ДДС",
|
||||||
|
actual, truth.dds.value,
|
||||||
|
passed=kio.dds == truth.dds,
|
||||||
|
ref="классификатор ДДС",
|
||||||
|
finding=f"Карточка ушла в ДДС {actual}, верная — {truth.dds.value}",
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── карточка, E5 ──
|
||||||
|
if truth.address:
|
||||||
|
written = kio.address or " ".join(filter(None, [kio.street, kio.building]))
|
||||||
|
expected = _normalize_address(truth.address)
|
||||||
|
build.add(
|
||||||
|
"address", "Адрес",
|
||||||
|
written or "не заполнен", truth.address,
|
||||||
|
passed=bool(expected) and expected <= _normalize_address(written),
|
||||||
|
ref="ground_truth сценария",
|
||||||
|
finding=f"Адрес в карточке «{written or 'пусто'}», верный — «{truth.address}»",
|
||||||
|
)
|
||||||
|
if truth.victims is not None:
|
||||||
|
actual = "не указано" if kio.victims_count is None else str(kio.victims_count)
|
||||||
|
build.add(
|
||||||
|
"victims_count", "Число пострадавших",
|
||||||
|
actual, str(truth.victims),
|
||||||
|
passed=kio.victims_count == truth.victims,
|
||||||
|
ref="ground_truth сценария",
|
||||||
|
finding=f"Пострадавших в карточке {actual}, верно — {truth.victims}",
|
||||||
|
)
|
||||||
|
if scenario.required_fields:
|
||||||
|
empty = missing_fields(kio, scenario.required_fields)
|
||||||
|
build.add(
|
||||||
|
"required_fields", "Обязательные поля КИО",
|
||||||
|
"все заполнены" if not empty else f"пусто: {', '.join(empty)}",
|
||||||
|
f"заполнены: {', '.join(scenario.required_fields)}",
|
||||||
|
passed=not empty,
|
||||||
|
ref="ГОСТ Р 22.7.03-2021, структура КИО",
|
||||||
|
finding=f"Не заполнены обязательные поля: {', '.join(empty)}" if empty else None,
|
||||||
|
)
|
||||||
|
|
||||||
|
return build.result
|
||||||
25
backend/app/scoring/taxonomy.py
Normal file
25
backend/app/scoring/taxonomy.py
Normal file
|
|
@ -0,0 +1,25 @@
|
||||||
|
"""Какая метрика каким кодом ошибки и какой компетенцией размечается.
|
||||||
|
|
||||||
|
Это методика, а не вычисление: таблица читается глазами и сверяется
|
||||||
|
с docs/product/METHODOLOGY.md. Считает метрики gost.py, сюда он только смотрит.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from app.domain.taxonomy import Competency, ErrorCode
|
||||||
|
|
||||||
|
#: Метрика → (код ошибки при провале, компетенция радара).
|
||||||
|
METRIC_MAP: dict[str, tuple[ErrorCode, Competency]] = {
|
||||||
|
"answer_time": (ErrorCode.E3, Competency.INTAKE),
|
||||||
|
"callback": (ErrorCode.E3, Competency.INTAKE),
|
||||||
|
"checklist_completeness": (ErrorCode.E1, Competency.INTERVIEW),
|
||||||
|
"interview_time": (ErrorCode.E3, Competency.NORMS),
|
||||||
|
"incident_type": (ErrorCode.E2, Competency.ROUTING),
|
||||||
|
"dds_choice": (ErrorCode.E2, Competency.ROUTING),
|
||||||
|
"address": (ErrorCode.E5, Competency.CARD),
|
||||||
|
"victims_count": (ErrorCode.E5, Competency.CARD),
|
||||||
|
"required_fields": (ErrorCode.E5, Competency.CARD),
|
||||||
|
}
|
||||||
|
|
||||||
|
#: Вес детерминированного слоя в итоговой оценке. Остальное — LLM-судья
|
||||||
|
#: на мягкие критерии (E4), и не больше (docs/arch/BACKEND.md).
|
||||||
|
DETERMINISTIC_WEIGHT = 0.6
|
||||||
|
JUDGE_WEIGHT = 0.4
|
||||||
|
|
@ -14,9 +14,16 @@ from dataclasses import dataclass, field
|
||||||
from app.domain.timers import NORMATIVES, TimerCode, TimerSnapshot, state_for
|
from app.domain.timers import NORMATIVES, TimerCode, TimerSnapshot, state_for
|
||||||
|
|
||||||
#: Какое событие какой таймер запускает.
|
#: Какое событие какой таймер запускает.
|
||||||
|
#:
|
||||||
|
#: `dds_notify` (≤ 60 с) не запускается ничем, и это сознательно. Стартуй он
|
||||||
|
#: на ответе, как опрос, оба таймера мерили бы один отрезок с разными лимитами:
|
||||||
|
#: курсант, опросивший за законные 70 секунд, получал бы E3 «ДДС не оповещена
|
||||||
|
#: за 60 с», а на экране краснел бы таймер посреди нормального разговора.
|
||||||
|
#: Норматив, судя по порядку операций, отсчитывается от конца опроса — а события
|
||||||
|
#: «опрос закончен» в контракте нет. Вопрос к людям: docs/arch/CONTRACT.md.
|
||||||
STARTS: dict[str, tuple[TimerCode, ...]] = {
|
STARTS: dict[str, tuple[TimerCode, ...]] = {
|
||||||
"call.incoming": (TimerCode.ANSWER,),
|
"call.incoming": (TimerCode.ANSWER,),
|
||||||
"call.answer": (TimerCode.INTERVIEW, TimerCode.DDS_NOTIFY),
|
"call.answer": (TimerCode.INTERVIEW,),
|
||||||
"dds.dispatch": (TimerCode.DDS_ACK, TimerCode.CLOSE),
|
"dds.dispatch": (TimerCode.DDS_ACK, TimerCode.CLOSE),
|
||||||
"card.received": (TimerCode.ZONE_CHECK,),
|
"card.received": (TimerCode.ZONE_CHECK,),
|
||||||
"callback.dial": (TimerCode.CALLBACK,),
|
"callback.dial": (TimerCode.CALLBACK,),
|
||||||
|
|
@ -25,7 +32,7 @@ STARTS: dict[str, tuple[TimerCode, ...]] = {
|
||||||
#: Какое событие какой таймер останавливает.
|
#: Какое событие какой таймер останавливает.
|
||||||
STOPS: dict[str, tuple[TimerCode, ...]] = {
|
STOPS: dict[str, tuple[TimerCode, ...]] = {
|
||||||
"call.answer": (TimerCode.ANSWER,),
|
"call.answer": (TimerCode.ANSWER,),
|
||||||
"dds.dispatch": (TimerCode.INTERVIEW, TimerCode.DDS_NOTIFY),
|
"dds.dispatch": (TimerCode.INTERVIEW,),
|
||||||
"card.ack": (TimerCode.DDS_ACK,),
|
"card.ack": (TimerCode.DDS_ACK,),
|
||||||
"zone.decision": (TimerCode.ZONE_CHECK,),
|
"zone.decision": (TimerCode.ZONE_CHECK,),
|
||||||
"crew.arrived": (TimerCode.CLOSE,),
|
"crew.arrived": (TimerCode.CLOSE,),
|
||||||
|
|
|
||||||
179
backend/tests/test_scoring.py
Normal file
179
backend/tests/test_scoring.py
Normal file
|
|
@ -0,0 +1,179 @@
|
||||||
|
"""Детерминированная оценка по ГОСТ.
|
||||||
|
|
||||||
|
Проверяются пункты приёмки lct-12: воспроизводимость, ни одной отметки без
|
||||||
|
обоснования, время опроса, полнота опроса, неверная ДДС.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.domain.events import CallEndReason
|
||||||
|
from app.domain.kio import KIO
|
||||||
|
from app.domain.taxonomy import Competency, ErrorCode
|
||||||
|
from app.scenarios.loader import load_file
|
||||||
|
from app.scoring.competency import radar
|
||||||
|
from app.scoring.gost import evaluate
|
||||||
|
from app.session.timers import SessionTimers
|
||||||
|
|
||||||
|
LIBRARY = Path(__file__).resolve().parents[2] / "scenarios"
|
||||||
|
ALL_FACTS = ["f_address", "f_what_burns", "f_people", "f_smoke", "f_gas"]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture(scope="module")
|
||||||
|
def scenario():
|
||||||
|
return load_file(LIBRARY / "fire-apartment-l2.yaml", LIBRARY)
|
||||||
|
|
||||||
|
|
||||||
|
def timeline(answer_s: float = 5, interview_s: float = 60) -> SessionTimers:
|
||||||
|
"""Таймеры с явными моментами событий — без часов, чтобы результат не плыл."""
|
||||||
|
timers = SessionTimers()
|
||||||
|
timers.on_event("call.incoming", now=0.0)
|
||||||
|
timers.on_event("call.answer", now=answer_s)
|
||||||
|
timers.on_event("dds.dispatch", now=answer_s + interview_s)
|
||||||
|
return timers
|
||||||
|
|
||||||
|
|
||||||
|
def good_card() -> KIO:
|
||||||
|
return KIO(
|
||||||
|
address="ул. Ленина, д. 14, кв. 47",
|
||||||
|
floor="5",
|
||||||
|
incident_type="fire",
|
||||||
|
dds="01",
|
||||||
|
victims_count=2,
|
||||||
|
description="горит балкон",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_perfect_call_passes_everything(scenario):
|
||||||
|
result = evaluate(scenario=scenario, kio=good_card(), timers=timeline(), revealed_facts=ALL_FACTS)
|
||||||
|
failed = [metric.key for metric in result.metrics if not metric.passed]
|
||||||
|
assert failed == [], f"провалено: {failed}"
|
||||||
|
assert result.findings == []
|
||||||
|
assert result.score == 100.0
|
||||||
|
|
||||||
|
|
||||||
|
def test_same_call_gives_identical_json(scenario):
|
||||||
|
"""Оценку можно предъявить и проверить только если она повторяется."""
|
||||||
|
def run():
|
||||||
|
result = evaluate(
|
||||||
|
scenario=scenario, kio=KIO(dds="03"), timers=timeline(interview_s=94),
|
||||||
|
revealed_facts=["f_address"],
|
||||||
|
)
|
||||||
|
return json.dumps(
|
||||||
|
{
|
||||||
|
"metrics": [m.model_dump() for m in result.metrics],
|
||||||
|
"findings": [f.model_dump(mode="json") for f in result.findings],
|
||||||
|
"score": result.score,
|
||||||
|
},
|
||||||
|
ensure_ascii=False,
|
||||||
|
sort_keys=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert run() == run()
|
||||||
|
|
||||||
|
|
||||||
|
def test_interview_over_75_seconds_is_e3_with_the_numbers(scenario):
|
||||||
|
result = evaluate(scenario=scenario, kio=good_card(), timers=timeline(interview_s=94), revealed_facts=ALL_FACTS)
|
||||||
|
metric = next(m for m in result.metrics if m.key == "interview_time")
|
||||||
|
assert not metric.passed
|
||||||
|
assert (metric.fact, metric.norm) == ("94 с", "≤ 75 с")
|
||||||
|
assert metric.ref == "ГОСТ Р 22.7.03-2021"
|
||||||
|
|
||||||
|
finding = next(f for f in result.findings if f.code is ErrorCode.E3)
|
||||||
|
assert "94 с" in finding.summary and "75 с" in finding.summary
|
||||||
|
|
||||||
|
|
||||||
|
def test_interview_never_stopped_is_a_failure_not_a_pass(scenario):
|
||||||
|
"""Таймер не остановлен событием — время недоказуемо. Молча засчитать
|
||||||
|
такую метрику значит поставить зачёт за непереданную карточку."""
|
||||||
|
timers = SessionTimers()
|
||||||
|
timers.on_event("call.incoming", now=0.0)
|
||||||
|
timers.on_event("call.answer", now=3.0)
|
||||||
|
result = evaluate(scenario=scenario, kio=good_card(), timers=timers, revealed_facts=ALL_FACTS)
|
||||||
|
metric = next(m for m in result.metrics if m.key == "interview_time")
|
||||||
|
assert not metric.passed and "не зафиксировано" in metric.fact
|
||||||
|
|
||||||
|
|
||||||
|
def test_each_missing_fact_is_its_own_e1_with_the_reference_question(scenario):
|
||||||
|
"""В разборе нужен конкретный пропущенный вопрос, а не процент."""
|
||||||
|
result = evaluate(
|
||||||
|
scenario=scenario, kio=good_card(), timers=timeline(),
|
||||||
|
revealed_facts=["f_address", "f_what_burns", "f_people"],
|
||||||
|
)
|
||||||
|
e1 = [f for f in result.findings if f.code is ErrorCode.E1]
|
||||||
|
assert len(e1) == 2
|
||||||
|
assert all("эталонный вопрос" in f.norm for f in e1)
|
||||||
|
|
||||||
|
completeness = next(m for m in result.metrics if m.key == "checklist_completeness")
|
||||||
|
assert completeness.fact == "добыто 3 из 5 обязательных фактов"
|
||||||
|
|
||||||
|
|
||||||
|
def test_wrong_dds_is_e2(scenario):
|
||||||
|
card = good_card()
|
||||||
|
card.dds = "03"
|
||||||
|
result = evaluate(scenario=scenario, kio=card, timers=timeline(), revealed_facts=ALL_FACTS)
|
||||||
|
finding = next(f for f in result.findings if f.code is ErrorCode.E2)
|
||||||
|
assert "03" in finding.summary and "01" in finding.summary
|
||||||
|
|
||||||
|
|
||||||
|
def test_address_abbreviations_are_not_punished(scenario):
|
||||||
|
for written in ("улица Ленина, 14", "ул. Ленина д. 14 кв. 47", "Ленина 14"):
|
||||||
|
card = good_card()
|
||||||
|
card.address = written
|
||||||
|
result = evaluate(scenario=scenario, kio=card, timers=timeline(), revealed_facts=ALL_FACTS)
|
||||||
|
assert next(m for m in result.metrics if m.key == "address").passed, written
|
||||||
|
|
||||||
|
|
||||||
|
def test_wrong_building_is_caught(scenario):
|
||||||
|
card = good_card()
|
||||||
|
card.address = "улица Ленина, 41"
|
||||||
|
result = evaluate(scenario=scenario, kio=card, timers=timeline(), revealed_facts=ALL_FACTS)
|
||||||
|
assert not next(m for m in result.metrics if m.key == "address").passed
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_finding_without_justification(scenario):
|
||||||
|
"""Ни одной отметки без кода и обоснования — ни в одном интерфейсе."""
|
||||||
|
result = evaluate(
|
||||||
|
scenario=scenario, kio=KIO(), timers=SessionTimers(), revealed_facts=[],
|
||||||
|
end_reason=CallEndReason.DROPPED,
|
||||||
|
)
|
||||||
|
assert result.findings, "пустая карточка должна дать отметки"
|
||||||
|
for finding in result.findings:
|
||||||
|
assert finding.code and finding.summary and finding.fact and finding.norm, finding
|
||||||
|
|
||||||
|
|
||||||
|
def test_without_slot_machine_completeness_is_not_silently_passed(scenario):
|
||||||
|
result = evaluate(scenario=scenario, kio=good_card(), timers=timeline(), revealed_facts=None)
|
||||||
|
assert "checklist_completeness" not in [m.key for m in result.metrics]
|
||||||
|
assert any("checklist_completeness" in note for note in result.unavailable)
|
||||||
|
|
||||||
|
|
||||||
|
def test_dropped_call_without_callback_fails(scenario):
|
||||||
|
result = evaluate(
|
||||||
|
scenario=scenario, kio=good_card(), timers=timeline(), revealed_facts=ALL_FACTS,
|
||||||
|
end_reason=CallEndReason.DROPPED,
|
||||||
|
)
|
||||||
|
callback = next(m for m in result.metrics if m.key == "callback")
|
||||||
|
assert not callback.passed and callback.fact == "обратного дозвона не было"
|
||||||
|
|
||||||
|
|
||||||
|
def test_radar_is_a_projection_of_the_same_metrics(scenario):
|
||||||
|
card = good_card()
|
||||||
|
card.dds = "03"
|
||||||
|
result = evaluate(scenario=scenario, kio=card, timers=timeline(interview_s=94), revealed_facts=ALL_FACTS)
|
||||||
|
values = {score.competency: score.value for score in radar(result.metrics)}
|
||||||
|
|
||||||
|
assert values["routing"] == 0.5, "из двух метрик маршрутизации провалена одна"
|
||||||
|
assert values["card"] == 1.0
|
||||||
|
assert Competency.COMMUNICATION.value not in values, "без судьи коммуникация не оценивалась, а не провалена"
|
||||||
|
|
||||||
|
|
||||||
|
def test_legal_interview_is_not_an_e3(scenario):
|
||||||
|
"""70 с опроса укладываются в норматив 75 с. Раньше таймер оповещения ДДС
|
||||||
|
стартовал на ответе, мерил тот же отрезок с лимитом 60 с и выставлял E3
|
||||||
|
за законный опрос. Пока нет события конца опроса, он не считается."""
|
||||||
|
result = evaluate(scenario=scenario, kio=good_card(), timers=timeline(interview_s=70), revealed_facts=ALL_FACTS)
|
||||||
|
assert [f for f in result.findings if f.code is ErrorCode.E3] == []
|
||||||
|
assert any(note.startswith("dds_notify_time") for note in result.unavailable)
|
||||||
|
|
@ -105,7 +105,7 @@ export interface CrewDispatched {
|
||||||
/** Дежурно-диспетчерская служба, в которую уходит карточка. */
|
/** Дежурно-диспетчерская служба, в которую уходит карточка. */
|
||||||
export type DDSCode = "01" | "02" | "03" | "04" | "gkh";
|
export type DDSCode = "01" | "02" | "03" | "04" | "gkh";
|
||||||
|
|
||||||
/** Передача в ДДС. Замораживает карточку снимком и останавливает `dds_notify`. */
|
/** Передача в ДДС. Замораживает карточку снимком и останавливает опрос (`interview`). */
|
||||||
export interface DdsDispatch {
|
export interface DdsDispatch {
|
||||||
type: "dds.dispatch";
|
type: "dds.dispatch";
|
||||||
service: DDSCode;
|
service: DDSCode;
|
||||||
|
|
@ -201,7 +201,7 @@ export interface InstructorNoteShown {
|
||||||
author: string;
|
author: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Полная карточка. Наблюдателям уходит целиком (`kio.state`), курсанту — дельтой (`kio.patch`). */
|
/** Полная карточка. Наблюдателям уходит целиком (`kio.state`), курсанту — дельтой (`kio.patch`). Присваивание проверяется: без этого `card.dds = "03"` кладёт в карточку сырую строку вместо кода ДДС, и падает уже оценка, далеко от места ошибки. */
|
||||||
export interface KIO {
|
export interface KIO {
|
||||||
card_id?: string;
|
card_id?: string;
|
||||||
registered_at?: string | null;
|
registered_at?: string | null;
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue