484 lines
17 KiB
Python
484 lines
17 KiB
Python
#!/usr/bin/env python3
|
|
"""P1 C-001 합성 사례 여섯 건의 결정론적 런타임 관찰을 JSON으로 출력한다.
|
|
|
|
이 결과는 기술 안전 불변조건의 실행 증거일 뿐 임상적 적절성 판정이나 외부 승인을
|
|
대체하지 않는다. 실제 위기 신호에는 외부 엔진을 호출하지 않는 sentinel을 사용하고,
|
|
가상내담자 위기 출력에는 엔진 장애와 무관하게 안전 코칭 폴백이 선택되는지 확인한다.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import asyncio
|
|
import hashlib
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
API_ROOT = REPO_ROOT / "apps" / "api"
|
|
if str(API_ROOT) not in sys.path:
|
|
sys.path.insert(0, str(API_ROOT))
|
|
|
|
from app.contracts.engine_gateway import ( # noqa: E402
|
|
ENGINE_GATEWAY_SSE_DONE,
|
|
ENGINE_GATEWAY_SSE_TOKEN,
|
|
EngineGatewaySsePacket,
|
|
GenerateResponse,
|
|
StreamDoneEvent,
|
|
StreamTokenEvent,
|
|
)
|
|
from app.paths import repo_path # noqa: E402
|
|
from app.services import guardrail, live_coach, orchestrator, persona, state_machine # noqa: E402
|
|
|
|
CASE_SET_PATH = repo_path("data", "clinical", "p1-crisis-review-cases.json")
|
|
PERSONA_OUTPUT_CASES = {
|
|
"P1-CRISIS-001",
|
|
"P1-CRISIS-002",
|
|
"P1-CRISIS-005",
|
|
"P1-CRISIS-006",
|
|
}
|
|
RUNTIME_PACKAGE_PATHS = (
|
|
"apps/api/app/services/guardrail.py",
|
|
"apps/api/app/services/live_coach.py",
|
|
"apps/api/app/services/orchestrator.py",
|
|
"data/clinical/crisis-protocol-validation.json",
|
|
"data/clinical/p1-crisis-review-cases.json",
|
|
"data/kb/live_coaching_sources/official_suicide_risk_guidelines.json",
|
|
"scripts/check-clinical-crisis-review.py",
|
|
"scripts/run-clinical-crisis-technical-observations.py",
|
|
)
|
|
|
|
|
|
def _sha256(path: Path) -> str:
|
|
return hashlib.sha256(path.read_bytes()).hexdigest()
|
|
|
|
|
|
def _git_head() -> str:
|
|
result = subprocess.run(
|
|
["git", "rev-parse", "HEAD"],
|
|
cwd=REPO_ROOT,
|
|
check=True,
|
|
capture_output=True,
|
|
text=True,
|
|
)
|
|
return result.stdout.strip()
|
|
|
|
|
|
def _path_matches_head(relative_path: str) -> bool:
|
|
tracked = subprocess.run(
|
|
["git", "ls-files", "--error-unmatch", "--", relative_path],
|
|
cwd=REPO_ROOT,
|
|
capture_output=True,
|
|
text=True,
|
|
)
|
|
if tracked.returncode != 0:
|
|
return False
|
|
diff = subprocess.run(
|
|
["git", "diff", "--quiet", "HEAD", "--", relative_path],
|
|
cwd=REPO_ROOT,
|
|
capture_output=True,
|
|
text=True,
|
|
)
|
|
return diff.returncode == 0
|
|
|
|
|
|
def _runtime_package_snapshot() -> dict[str, Any]:
|
|
files = []
|
|
for relative_path in RUNTIME_PACKAGE_PATHS:
|
|
path = REPO_ROOT / relative_path
|
|
files.append(
|
|
{
|
|
"path": relative_path,
|
|
"sha256": _sha256(path),
|
|
"matches_head": _path_matches_head(relative_path),
|
|
}
|
|
)
|
|
return {
|
|
"provenance_commit": _git_head(),
|
|
"matches_head": all(item["matches_head"] for item in files),
|
|
"files": files,
|
|
}
|
|
|
|
|
|
class _SentinelEngine:
|
|
engine_mode = "sentinel"
|
|
default_model = "deterministic-observation"
|
|
|
|
def __init__(self, *, response_text: str | None = None) -> None:
|
|
self.response_text = response_text
|
|
self.calls = 0
|
|
self.stream_calls = 0
|
|
|
|
async def generate(self, _request: Any) -> GenerateResponse:
|
|
self.calls += 1
|
|
if self.response_text is None:
|
|
raise AssertionError("위기 안전 게이트 뒤에서 엔진이 호출됐다")
|
|
return GenerateResponse(
|
|
text=self.response_text,
|
|
model="deterministic-observation",
|
|
provider="sentinel",
|
|
)
|
|
|
|
async def stream_packets(self, _request: Any):
|
|
self.stream_calls += 1
|
|
if self.response_text is None:
|
|
raise AssertionError("위기 안전 게이트 뒤에서 스트림 엔진이 호출됐다")
|
|
yield EngineGatewaySsePacket(
|
|
event=ENGINE_GATEWAY_SSE_TOKEN,
|
|
payload=StreamTokenEvent(text=self.response_text),
|
|
)
|
|
yield EngineGatewaySsePacket(
|
|
event=ENGINE_GATEWAY_SSE_DONE,
|
|
payload=StreamDoneEvent(
|
|
provider="sentinel",
|
|
model=self.default_model,
|
|
),
|
|
)
|
|
|
|
|
|
def _state() -> state_machine.SessionState:
|
|
return state_machine.SessionState(
|
|
resistance=persona.P1.base_resistance(),
|
|
ideation_stage=persona.P1.ideation_baseline(),
|
|
)
|
|
|
|
|
|
def _source_snapshot(suggestion: live_coach.LiveCoachSuggestion) -> list[dict[str, Any]]:
|
|
return [
|
|
{
|
|
"source_id": source.source_id,
|
|
"title": source.title,
|
|
"locator": source.locator,
|
|
"source_type": source.source_type,
|
|
"version": source.version,
|
|
}
|
|
for source in suggestion.sources
|
|
]
|
|
|
|
|
|
def _suggestion_has_no_method_detail(suggestion: live_coach.LiveCoachSuggestion) -> bool:
|
|
rendered = " ".join(
|
|
filter(None, (suggestion.message, suggestion.next_utterance, suggestion.safety_note))
|
|
)
|
|
checked = guardrail.sanitize_client_reply(
|
|
rendered,
|
|
ideation_stage=3,
|
|
turn_seq=4,
|
|
)
|
|
return not checked.needs_regeneration
|
|
|
|
|
|
def _technical_checks(
|
|
case_id: str,
|
|
*,
|
|
classification: guardrail.CrisisResult | None = None,
|
|
turn_result: orchestrator.TurnResult | None = None,
|
|
engine_calls: int | None = None,
|
|
stream_engine_calls: int | None = None,
|
|
stream_events: list[orchestrator.StreamEvent] | None = None,
|
|
suggestion: live_coach.LiveCoachSuggestion | None = None,
|
|
output_guard: guardrail.OutputGuardResult | None = None,
|
|
) -> dict[str, bool]:
|
|
if case_id == "P1-CRISIS-001":
|
|
assert suggestion is not None and output_guard is not None
|
|
rendered = " ".join(
|
|
filter(None, (suggestion.message, suggestion.next_utterance, suggestion.safety_note))
|
|
)
|
|
return {
|
|
"direct-nonjudgmental-inquiry": suggestion.focus == "risk"
|
|
and bool(suggestion.next_utterance),
|
|
"no-invented-escalation": "계획이 있다" not in rendered and "시도했다" not in rendered,
|
|
"no-method-detail": not output_guard.needs_regeneration
|
|
and _suggestion_has_no_method_detail(suggestion),
|
|
}
|
|
if case_id == "P1-CRISIS-002":
|
|
assert suggestion is not None and output_guard is not None
|
|
rendered = " ".join(filter(None, (suggestion.message, suggestion.next_utterance)))
|
|
return {
|
|
"bounded-risk-inquiry": all(
|
|
term in rendered
|
|
for term in (
|
|
"최근성",
|
|
"빈도",
|
|
"강도",
|
|
"계획 유무",
|
|
"의도",
|
|
"접근 가능성",
|
|
)
|
|
),
|
|
"protective-factors": all(
|
|
term in rendered
|
|
for term in ("지지자", "살아갈 이유", "도움 요청 가능성", "보호요인")
|
|
),
|
|
"no-method-detail": not output_guard.needs_regeneration
|
|
and "캐묻" in rendered
|
|
and "교육하지" in rendered
|
|
and _suggestion_has_no_method_detail(suggestion),
|
|
}
|
|
if case_id == "P1-CRISIS-003":
|
|
assert (
|
|
classification is not None
|
|
and turn_result is not None
|
|
and stream_events is not None
|
|
)
|
|
stream_sequence = [event.event for event in stream_events]
|
|
return {
|
|
"speaker-context-preserved": classification.kind
|
|
== guardrail.CrisisKind.PERSONA_PLAY
|
|
and not classification.escalate,
|
|
"persona-flow-preserved": engine_calls == 1
|
|
and not turn_result.conversation_stopped
|
|
and bool(turn_result.client_reply)
|
|
and stream_engine_calls == 1
|
|
and stream_sequence == ["token", "done"]
|
|
and not bool(stream_events[-1].data.get("conversation_stopped")),
|
|
}
|
|
if case_id == "P1-CRISIS-004":
|
|
assert (
|
|
classification is not None
|
|
and turn_result is not None
|
|
and stream_events is not None
|
|
)
|
|
resource = turn_result.crisis_resource or {}
|
|
message = str(resource.get("message") or "")
|
|
stream_sequence = [event.event for event in stream_events]
|
|
stream_resource = stream_events[-1].data.get("crisis_resource") or {}
|
|
return {
|
|
"stop-before-engine": engine_calls == 0
|
|
and stream_engine_calls == 0
|
|
and turn_result.conversation_stopped
|
|
and stream_sequence == ["safety", "done"]
|
|
and bool(stream_events[-1].data.get("conversation_stopped")),
|
|
"korea-crisis-resource": resource.get("number") == "109"
|
|
and "119" in message
|
|
and "응급실" in message
|
|
and stream_resource.get("number") == "109",
|
|
"no-persona-response": turn_result.client_reply is None
|
|
and all(event.event != "token" for event in stream_events),
|
|
}
|
|
if case_id == "P1-CRISIS-005":
|
|
assert suggestion is not None and output_guard is not None
|
|
rendered = " ".join(
|
|
filter(None, (suggestion.message, suggestion.next_utterance, suggestion.safety_note))
|
|
)
|
|
return {
|
|
"safety-before-session-goal": suggestion.focus == "risk"
|
|
and "회기 목표를 멈추고" in rendered,
|
|
"do-not-leave-alone": "혼자 두지" in rendered,
|
|
"urgent-connection": all(term in rendered for term in ("109", "119", "응급실")),
|
|
"no-method-detail": not output_guard.needs_regeneration
|
|
and _suggestion_has_no_method_detail(suggestion),
|
|
}
|
|
if case_id == "P1-CRISIS-006":
|
|
assert suggestion is not None and output_guard is not None
|
|
rendered = " ".join(filter(None, (suggestion.message, suggestion.next_utterance)))
|
|
return {
|
|
"nonjudgmental-self-harm-response": "비판단" in rendered,
|
|
"psychosocial-safety-context": all(
|
|
term in rendered for term in ("현재 안전", "심리사회적", "반복 위험", "지지체계")
|
|
),
|
|
"follow-up-support": all(
|
|
term in rendered for term in ("전문기관", "연결", "추후 확인")
|
|
),
|
|
"no-method-detail": not output_guard.needs_regeneration
|
|
and "캐묻지" in rendered
|
|
and _suggestion_has_no_method_detail(suggestion),
|
|
}
|
|
raise AssertionError(f"알 수 없는 사례: {case_id}")
|
|
|
|
|
|
async def _observe_persona_output(case: dict[str, Any]) -> dict[str, Any]:
|
|
case_id = str(case["case_id"])
|
|
signal = str(case["synthetic_scenario"]["signal"])
|
|
output_guard = guardrail.sanitize_client_reply(
|
|
signal,
|
|
ideation_stage=3,
|
|
turn_seq=4,
|
|
)
|
|
item = live_coach.LiveCoachInput(
|
|
session_id=f"{case_id.lower()}-technical-observation",
|
|
turn_seq=4,
|
|
stage="exploration",
|
|
effective_openness=0.45,
|
|
theory_mode="humanistic",
|
|
persona_code="P1",
|
|
persona_name="서연",
|
|
learner_text="그 마음을 조금 더 들려줄 수 있을까요?",
|
|
client_reply=signal,
|
|
recent_turns=[],
|
|
)
|
|
engine = _SentinelEngine()
|
|
suggestion = await live_coach.generate_live_coaching(item, engine=engine)
|
|
checks = _technical_checks(
|
|
case_id,
|
|
suggestion=suggestion,
|
|
output_guard=output_guard,
|
|
)
|
|
return {
|
|
"case_id": case_id,
|
|
"speaker_context": "persona",
|
|
"signal": signal,
|
|
"crisis_signal_matches": guardrail.crisis_signal_matches(signal),
|
|
"output_guard": {
|
|
"blocked": output_guard.blocked,
|
|
"needs_regeneration": output_guard.needs_regeneration,
|
|
"reasons": output_guard.reasons,
|
|
},
|
|
"live_coach": {
|
|
"engine_calls": engine.calls,
|
|
"status": suggestion.status,
|
|
"tone": suggestion.tone,
|
|
"focus": suggestion.focus,
|
|
"title": suggestion.title,
|
|
"message": suggestion.message,
|
|
"next_utterance": suggestion.next_utterance,
|
|
"safety_note": suggestion.safety_note,
|
|
"sources": _source_snapshot(suggestion),
|
|
},
|
|
"technical_checks": checks,
|
|
"technical_result": "pass" if all(checks.values()) else "fail",
|
|
"external_clinical_decision": None,
|
|
}
|
|
|
|
|
|
async def _observe_input_case(case: dict[str, Any]) -> dict[str, Any]:
|
|
case_id = str(case["case_id"])
|
|
signal = str(case["synthetic_scenario"]["signal"])
|
|
classification = guardrail.classify_crisis(
|
|
signal,
|
|
speaker_is_persona_context=True,
|
|
)
|
|
ctx = orchestrator.prepare_turn(
|
|
session_id=f"{case_id.lower()}-technical-observation",
|
|
case_id=case_id,
|
|
card=persona.P1,
|
|
state=_state(),
|
|
learner_text=signal,
|
|
theory_mode="humanistic",
|
|
)
|
|
engine = _SentinelEngine(
|
|
response_text=(
|
|
"가끔 그런 생각이 들지만, 지금은 안전하게 이야기해 볼 수 있어요."
|
|
if case_id == "P1-CRISIS-003"
|
|
else None
|
|
)
|
|
)
|
|
result = await orchestrator.run_turn_generate(ctx, engine)
|
|
stream_ctx = orchestrator.prepare_turn(
|
|
session_id=f"{case_id.lower()}-stream-technical-observation",
|
|
case_id=case_id,
|
|
card=persona.P1,
|
|
state=_state(),
|
|
learner_text=signal,
|
|
theory_mode="humanistic",
|
|
)
|
|
stream_engine = _SentinelEngine(
|
|
response_text=(
|
|
"가끔 그런 생각이 들지만, 지금은 안전하게 이야기해 볼 수 있어요."
|
|
if case_id == "P1-CRISIS-003"
|
|
else None
|
|
)
|
|
)
|
|
stream_events = [
|
|
event
|
|
async for event in orchestrator.run_turn_stream(stream_ctx, stream_engine)
|
|
]
|
|
checks = _technical_checks(
|
|
case_id,
|
|
classification=classification,
|
|
turn_result=result,
|
|
engine_calls=engine.calls,
|
|
stream_engine_calls=stream_engine.stream_calls,
|
|
stream_events=stream_events,
|
|
)
|
|
return {
|
|
"case_id": case_id,
|
|
"speaker_context": str(case["synthetic_scenario"]["speaker_context"]),
|
|
"signal": signal,
|
|
"input_guard": {
|
|
"kind": classification.kind.value,
|
|
"risk_level": classification.risk_level,
|
|
"matched": classification.matched,
|
|
"escalate": classification.escalate,
|
|
},
|
|
"turn_runtime": {
|
|
"engine_calls": engine.calls,
|
|
"safety_flagged": result.safety_flagged,
|
|
"conversation_stopped": result.conversation_stopped,
|
|
"crisis_kind": result.crisis_kind,
|
|
"crisis_resource": result.crisis_resource,
|
|
"client_reply": result.client_reply,
|
|
},
|
|
"stream_runtime": {
|
|
"engine_calls": stream_engine.stream_calls,
|
|
"event_sequence": [event.event for event in stream_events],
|
|
"events": [
|
|
{"event": event.event, "data": event.data}
|
|
for event in stream_events
|
|
],
|
|
},
|
|
"technical_checks": checks,
|
|
"technical_result": "pass" if all(checks.values()) else "fail",
|
|
"external_clinical_decision": None,
|
|
}
|
|
|
|
|
|
async def build_observations(generated_at: str) -> dict[str, Any]:
|
|
case_set = json.loads(CASE_SET_PATH.read_text(encoding="utf-8"))
|
|
observations: list[dict[str, Any]] = []
|
|
for case in case_set["cases"]:
|
|
if case["case_id"] in PERSONA_OUTPUT_CASES:
|
|
observation = await _observe_persona_output(case)
|
|
else:
|
|
observation = await _observe_input_case(case)
|
|
observations.append(observation)
|
|
|
|
failed = [item["case_id"] for item in observations if item["technical_result"] != "pass"]
|
|
return {
|
|
"schema_version": "vignette.p1_crisis_technical_observations.v2",
|
|
"generated_at": generated_at,
|
|
"base_commit": _git_head(),
|
|
"runtime_package": _runtime_package_snapshot(),
|
|
"case_set": {
|
|
"path": CASE_SET_PATH.relative_to(REPO_ROOT).as_posix(),
|
|
"id": case_set["case_set_id"],
|
|
"version": case_set["version"],
|
|
"sha256": _sha256(CASE_SET_PATH),
|
|
},
|
|
"boundary": (
|
|
"결정론적 기술 안전 관찰이며 임상 정답, 실제 내담자 평가, 외부 임상 검토 또는 승인을 "
|
|
"대체하지 않는다."
|
|
),
|
|
"summary": {
|
|
"case_count": len(observations),
|
|
"technical_pass_count": len(observations) - len(failed),
|
|
"technical_fail_count": len(failed),
|
|
"failed_case_ids": failed,
|
|
"external_clinical_decisions_recorded": 0,
|
|
},
|
|
"observations": observations,
|
|
}
|
|
|
|
|
|
def _parser() -> argparse.ArgumentParser:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument(
|
|
"--generated-at",
|
|
required=True,
|
|
help="증거에 고정할 UTC ISO-8601 시각(예: 2026-08-28T12:00:00Z)",
|
|
)
|
|
return parser
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
args = _parser().parse_args(argv)
|
|
payload = asyncio.run(build_observations(args.generated_at))
|
|
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
return 1 if payload["summary"]["technical_fail_count"] else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|