lct-16 и половина lct-19: разбор, отчёт, внешний монитор, эталон
Эталонный диалог собирается кодом из фактов и чек-листа: написанный руками, он разошёлся бы с фактами при первой же правке сценария, и курсанта оштрафовали бы за правильный ответ. Отчёт: метрики фактом против норматива со ссылкой, отметка E1 на каждый недобытый факт с эталонным вопросом, расхождение самооценки — что курсант заметил сам, чего не заметил, что отметил зря. Не заметил — самое ценное для разбора. Внешний монитор — не отдельное приложение, а другой режим отрисовки тех же событий: крупный таймер опроса, ход разговора, карточка, после оценки разбор на весь экран. Коррекция преподавателем сохраняет автооценку рядом: видно, что скорректировано и кем. Найдено: все метрики весили одинаково, и курсант, не задавший ни одного вопроса, но заполнивший карточку руками, получал 87 из 100. Предварительные веса (полнота опроса — 4) дают 74; окончательные утверждает методист, вопрос записан в DEBRIEF.md.
This commit is contained in:
parent
25143645d3
commit
fadfa3e479
20 changed files with 781 additions and 9 deletions
|
|
@ -14,7 +14,7 @@ from app.domain.kio import KIO, missing_fields
|
|||
from app.domain.taxonomy import ERRORS, Competency, Finding, FindingSource
|
||||
from app.domain.timers import GOST_REF, NORMATIVES, TimerCode
|
||||
from app.scenarios.schema import Scenario
|
||||
from app.scoring.taxonomy import METRIC_MAP
|
||||
from app.scoring.taxonomy import METRIC_MAP, METRIC_WEIGHTS
|
||||
from app.session.timers import SessionTimers
|
||||
|
||||
SOURCE_BY_CODE = {
|
||||
|
|
@ -64,7 +64,8 @@ class _Builder:
|
|||
def add(self, key: str, title: str, fact: str, norm: str, passed: bool, ref: str | None = None,
|
||||
finding: str | None = None) -> None:
|
||||
self.result.metrics.append(
|
||||
Metric(key=key, title=title, fact=fact, norm=norm, ref=ref, passed=passed)
|
||||
Metric(key=key, title=title, fact=fact, norm=norm, ref=ref, passed=passed,
|
||||
weight=METRIC_WEIGHTS.get(key, 1.0))
|
||||
)
|
||||
if passed:
|
||||
return
|
||||
|
|
@ -156,6 +157,7 @@ def evaluate(
|
|||
norm="все обязательные факты",
|
||||
ref="чек-лист сценария",
|
||||
passed=len(got) == len(required),
|
||||
weight=METRIC_WEIGHTS["checklist_completeness"],
|
||||
)
|
||||
)
|
||||
# По отметке на каждый недобытый факт: в разборе нужен конкретный
|
||||
|
|
|
|||
70
backend/app/scoring/reference.py
Normal file
70
backend/app/scoring/reference.py
Normal file
|
|
@ -0,0 +1,70 @@
|
|||
"""Эталонный диалог: как должен был пройти опрос.
|
||||
|
||||
Собирается **кодом** из чек-листа и фактов сценария. Писать его руками нельзя:
|
||||
методист поправит факт и забудет эталон, и курсант получит штраф за правильный
|
||||
ответ (docs/spec/SCENARIO-FORMAT.md).
|
||||
|
||||
Работает в двух местах: в разборе — «вот какой вопрос стоял на этом шаге»
|
||||
(docs/product/DEBRIEF.md), и в эталон-плеере на внешнем мониторе, когда
|
||||
появится предгенерация озвучки (lct-19, требует ключа LLM).
|
||||
"""
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from app.dialog.caller import REVEAL
|
||||
from app.dialog.persona import BASE_MOOD
|
||||
from app.domain.events import Mood
|
||||
from app.scenarios.schema import Scenario
|
||||
|
||||
|
||||
class ReferenceStep(BaseModel):
|
||||
"""Шаг эталонного опроса: вопрос оператора и ответ звонящего на него."""
|
||||
|
||||
checklist_id: str
|
||||
question: str
|
||||
fact_id: str | None = None
|
||||
answer: str | None = None
|
||||
required: bool = False
|
||||
hidden: bool = False
|
||||
|
||||
|
||||
class ReferenceDialog(BaseModel):
|
||||
scenario_id: str
|
||||
first_line: str
|
||||
steps: list[ReferenceStep]
|
||||
|
||||
|
||||
def build(scenario: Scenario) -> ReferenceDialog:
|
||||
facts = {fact.id: fact for fact in scenario.facts}
|
||||
required = set(scenario.ground_truth.required_facts)
|
||||
mood = BASE_MOOD.get(scenario.persona.base, Mood.CALM)
|
||||
|
||||
steps: list[ReferenceStep] = []
|
||||
for item in scenario.checklist:
|
||||
if not item.question:
|
||||
continue
|
||||
fact = facts.get(item.fact) if item.fact else None
|
||||
steps.append(
|
||||
ReferenceStep(
|
||||
checklist_id=item.id,
|
||||
question=item.question,
|
||||
fact_id=fact.id if fact else None,
|
||||
# Ответ звонящего — тем же шаблоном, каким он отвечает в живом
|
||||
# звонке: эталон должен звучать так же, а не литературно.
|
||||
answer=REVEAL[mood].format(fact=fact.value) if fact else None,
|
||||
required=bool(fact and fact.id in required),
|
||||
hidden=bool(fact and fact.hidden),
|
||||
)
|
||||
)
|
||||
return ReferenceDialog(scenario_id=scenario.id, first_line=scenario.first_line, steps=steps)
|
||||
|
||||
|
||||
def missed_steps(scenario: Scenario, revealed: list[str] | None) -> list[ReferenceStep]:
|
||||
"""Шаги, которые курсант не отработал. `None` — слот-автомата не было."""
|
||||
if revealed is None:
|
||||
return []
|
||||
return [
|
||||
step
|
||||
for step in build(scenario).steps
|
||||
if step.fact_id and step.fact_id not in revealed and step.required
|
||||
]
|
||||
93
backend/app/scoring/report.py
Normal file
93
backend/app/scoring/report.py
Normal file
|
|
@ -0,0 +1,93 @@
|
|||
"""Сборка отчёта сессии — единицы истории.
|
||||
|
||||
Из отчётов складываются профиль курсанта, дельта попыток и аналитика группы.
|
||||
Ни одной отметки без обоснования: у каждой есть код, факт и норматив
|
||||
(docs/product/DEBRIEF.md).
|
||||
"""
|
||||
|
||||
from uuid import UUID
|
||||
|
||||
from app.domain.events import (
|
||||
CompetencyScore,
|
||||
HintShown,
|
||||
HintUsage,
|
||||
InstructorNoteShown,
|
||||
Metric,
|
||||
SelfAssessment,
|
||||
SelfAssessmentDiff,
|
||||
SessionReport,
|
||||
)
|
||||
from app.domain.taxonomy import Finding
|
||||
from app.scenarios.schema import Scenario
|
||||
from app.scoring.reference import build as build_reference
|
||||
|
||||
|
||||
def assess_difference(
|
||||
scenario: Scenario, revealed: list[str] | None, missed_claimed: list[str]
|
||||
) -> SelfAssessmentDiff:
|
||||
"""Сверить, что курсант считает пропущенным, с тем, что он пропустил на деле."""
|
||||
reference = build_reference(scenario)
|
||||
required = {step.checklist_id: step.fact_id for step in reference.steps if step.required}
|
||||
really_missed = {
|
||||
checklist_id
|
||||
for checklist_id, fact_id in required.items()
|
||||
if revealed is not None and fact_id not in revealed
|
||||
}
|
||||
claimed = set(missed_claimed)
|
||||
return SelfAssessmentDiff(
|
||||
noticed=sorted(claimed & really_missed),
|
||||
unnoticed=sorted(really_missed - claimed),
|
||||
overcautious=sorted(claimed - really_missed),
|
||||
)
|
||||
|
||||
|
||||
def build(session_id: UUID, state, scenario: Scenario) -> SessionReport:
|
||||
score = state.score or {}
|
||||
revealed = [fact.id for fact in state.slots.revealed_facts()] if state.slots else None
|
||||
reference = build_reference(scenario)
|
||||
|
||||
self_assessment = None
|
||||
difference = None
|
||||
if state.self_assessment is not None:
|
||||
self_assessment = SelfAssessment(
|
||||
missed=state.self_assessment["missed"],
|
||||
comment=state.self_assessment.get("comment", ""),
|
||||
submitted_at=state.ended_at or state.transcript[-1].at,
|
||||
)
|
||||
difference = assess_difference(scenario, revealed, self_assessment.missed)
|
||||
|
||||
questions = {step.checklist_id: step.question for step in reference.steps}
|
||||
missed_checklist = [
|
||||
step.checklist_id
|
||||
for step in reference.steps
|
||||
if step.required and revealed is not None and step.fact_id not in revealed
|
||||
]
|
||||
|
||||
return SessionReport(
|
||||
session_id=session_id,
|
||||
scenario_id=scenario.id,
|
||||
mode=state.mode,
|
||||
attempt=state.attempt,
|
||||
transcript=list(state.transcript),
|
||||
findings=[Finding.model_validate(item) for item in score.get("findings", [])],
|
||||
metrics=[Metric.model_validate(item) for item in score.get("metrics", [])],
|
||||
competencies=[CompetencyScore.model_validate(item) for item in score.get("competencies", [])],
|
||||
# Эталонные вопросы по шагам: по каждому пропущенному пункту видно,
|
||||
# какой вопрос был правильным.
|
||||
reference_questions=[
|
||||
HintShown(checklist_id=step.checklist_id, question=step.question)
|
||||
for step in reference.steps
|
||||
],
|
||||
missed_checklist=missed_checklist,
|
||||
hints_used=[
|
||||
HintUsage(checklist_id=checklist_id, question=questions.get(checklist_id, ""), at=at)
|
||||
for checklist_id, at in state.hints_log
|
||||
],
|
||||
self_assessment=self_assessment,
|
||||
self_assessment_diff=difference,
|
||||
notes=[InstructorNoteShown.model_validate(note) for note in state.notes],
|
||||
score_auto=score.get("score_auto", 0.0),
|
||||
score_final=score.get("score_final", score.get("score_auto", 0.0)),
|
||||
overridden_by=score.get("overridden_by"),
|
||||
override_comment=score.get("override_comment"),
|
||||
)
|
||||
|
|
@ -19,6 +19,25 @@ METRIC_MAP: dict[str, tuple[ErrorCode, Competency]] = {
|
|||
"required_fields": (ErrorCode.E5, Competency.CARD),
|
||||
}
|
||||
|
||||
#: Вес метрики в детерминированной оценке.
|
||||
#:
|
||||
#: **Предварительные значения, требуют утверждения методистом.** Без весов все
|
||||
#: метрики равны, и курсант, не задавший ни одного вопроса, но заполнивший
|
||||
#: карточку руками, получает 87 из 100: «ответ за секунду» стоит столько же,
|
||||
#: сколько «добыл все обязательные факты». Опрос — то, ради чего существует
|
||||
#: тренажёр, поэтому он весит больше всего.
|
||||
METRIC_WEIGHTS: dict[str, float] = {
|
||||
"checklist_completeness": 4.0,
|
||||
"incident_type": 2.0,
|
||||
"dds_choice": 2.0,
|
||||
"address": 2.0,
|
||||
"required_fields": 2.0,
|
||||
"interview_time": 1.5,
|
||||
"victims_count": 1.0,
|
||||
"answer_time": 1.0,
|
||||
"callback": 1.0,
|
||||
}
|
||||
|
||||
#: Вес детерминированного слоя в итоговой оценке. Остальное — LLM-судья
|
||||
#: на мягкие критерии (E4), и не больше (docs/arch/BACKEND.md).
|
||||
DETERMINISTIC_WEIGHT = 0.6
|
||||
|
|
|
|||
Loading…
Reference in a new issue