Jev 내담자 평가·표현 v2와 속마음 공개

상담자 발화 판정(A)·감정(B)·표현(C) 20문항 질문 세트, 감쇠 없는 이번 턴 반응과 비대칭 기분 전이, 개방도 게이트로 생성 지시를 만들고 ccd.coping_strategy 전달 누락을 고친다.

trace v2와 고정 문구 속마음 요약(migration 24, AI 경로 차단 RLS)을 같은 트랜잭션에 저장하고 피드백 정책이 켜진 경우에만 done·TurnResponse·음성 reply·리뷰로 노출한다. 회기 화면 속마음 보기 토글, 리뷰 접힘 블록, 관리자 감정 관측 v2 표시를 추가한다.
This commit is contained in:
Yun Chan 2026-09-30 13:23:09 +09:00
parent 36cb847e38
commit 29c406d89f
39 changed files with 5726 additions and 343 deletions

View file

@ -1,4 +1,4 @@
"""Jev HTTP 어댑터의 단위 계약."""
"""Jev HTTP 어댑터(질문 세트 v2)의 단위 계약."""
from __future__ import annotations
@ -14,36 +14,96 @@ from pydantic import SecretStr
from .services import jev_client as jev_module
from .services.jev_client import (
CHOICE_QUESTION_IDS,
EMOTION_DIMENSIONS,
MAX_SORE_SPOTS,
NOUL_QUESTION_IDS,
OPENROUTER_JEV_ENDPOINT,
SORE_SPOT_QUESTION_ID,
TYPESAFE_JEV_ENDPOINT,
JevClient,
JevError,
)
def _answer(score: float = 2.0) -> dict[str, object]:
def _state(
*,
first_turn: bool = False,
sore_spots: tuple[str, ...] = ("성급한 조언", "능력 평가"),
forbidden: tuple[str, ...] = (),
) -> dict[str, object]:
recent_turns: list[dict[str, str]] = []
if not first_turn:
recent_turns.append({"speaker": "client", "text": "저도 몰라서 온 건 아니에요."})
return {
"counselor_utterance": "일단 긍정적으로 생각하고 운동부터 해보면 어떨까요?",
"recent_turns": recent_turns,
"client_profile": {
"presenting": "해결책보다 이해받길 바란다.",
"history": "노력 부족이라는 말을 반복해서 들었다.",
"core_belief": "실수하면 가치가 없다.",
"coping_strategy": "설명하거나 날카롭게 항의한다.",
"big5": {"O": 0.48, "C": 0.82, "E": 0.43, "A": 0.46, "N": 0.69},
"sore_spots": list(sore_spots),
"forbidden": list(forbidden),
"speech_style": {"register": "존댓말"},
},
"pinned_facts": ["유능하지 못하다는 평가에 민감하다."],
"recall_summary": "문제를 설명할 때마다 노력 부족이라는 말을 들었다.",
"relationship": {"stage": "탐색", "openness": "guarded", "resistance": "high"},
"previous_feelings": {dimension: "moderate" for dimension in EMOTION_DIMENSIONS},
}
def _score_answer(score: float = 2.0) -> dict[str, object]:
return {
"type": "score",
"score": score,
"confidence": 0.8,
"legend": {str(index): f"level {index}" for index in range(5)},
"probabilities": {
"0": 0.0,
"1": 0.1,
"2": 0.8,
"3": 0.1,
"4": 0.0,
},
"probabilities": {"0": 0.0, "1": 0.1, "2": 0.8, "3": 0.1, "4": 0.0},
}
def _response(model: str = "typesafe/jev-1.13-20260917") -> dict[str, object]:
return {
def _noul_answer(probability: float = 0.7) -> dict[str, object]:
# docs.typesafe.ai/primitives/noul(2026-09-29 확인): 응답은 {"type":"noul","noul":p}이고
# confidence 필드는 없다("There is no separate confidence field for Noul answers").
return {"type": "noul", "noul": probability}
def _choice_answer(criteria: dict[str, object]) -> dict[str, object]:
codes = list(criteria)
probabilities = {code: (0.7 if index == 0 else 0.3 / (len(codes) - 1)) for index, code in enumerate(codes)}
return {"type": "choice", "choice": codes[0], "probabilities": probabilities, "confidence": 0.6}
def _answers_for(questions: dict[str, dict[str, object]]) -> dict[str, object]:
answers: dict[str, object] = {}
for question_id, question in questions.items():
if question["type"] == "score":
answers[question_id] = _score_answer()
elif question["type"] == "noul":
answers[question_id] = _noul_answer()
else:
answers[question_id] = _choice_answer(question["criteria"]) # type: ignore[arg-type]
return answers
def _response(
state: dict[str, object],
*,
model: str = "typesafe/jev-1.13-20260917",
questions: dict[str, dict[str, object]] | None = None,
) -> tuple[dict[str, object], dict[str, dict[str, object]]]:
built_questions = questions if questions is not None else JevClient(
provider="openrouter", api_key="k", model="m"
)._questions(state)
payload = {
"model": model,
"answers": {dimension: _answer() for dimension in EMOTION_DIMENSIONS},
"answers": _answers_for(built_questions),
"usage": {"input_tokens": 120, "output_tokens": 45, "cost": 0.000019992},
}
return payload, built_questions
class JevClientTest(unittest.IsolatedAsyncioTestCase):
@ -62,7 +122,10 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
self.addAsyncCleanup(client.shutdown)
return client
async def test_appraise_posts_one_typed_request_and_normalizes_scores(self) -> None:
async def test_appraise_posts_full_question_set_and_normalizes_scores(self) -> None:
state = _state()
payload, questions = _response(state)
async def handler(request: httpx.Request) -> httpx.Response:
self.calls += 1
self.assertEqual("POST", request.method)
@ -70,18 +133,31 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
self.assertEqual("Bearer test-key", request.headers["Authorization"])
body = json.loads(request.content)
self.assertEqual("~typesafe/jev-latest", body["model"])
self.assertEqual(set(EMOTION_DIMENSIONS), set(body["questions"]))
for dimension, question in body["questions"].items():
self.assertEqual(set(questions), set(body["questions"]))
for dimension in EMOTION_DIMENSIONS:
question = body["questions"][dimension]
self.assertEqual("score", question["type"])
self.assertEqual(5, len(question["criteria"]))
self.assertIn(dimension, question["instructions"])
self.assertIn("counselor_utterance", question["instructions"])
self.assertIn("pinned facts", question["instructions"])
self.assertLessEqual(len(question["instructions"].split()), 50)
return httpx.Response(200, json=_response())
self.assertIn("pinned_facts", question["instructions"])
for question_id in NOUL_QUESTION_IDS:
if question_id not in body["questions"]:
continue
question = body["questions"][question_id]
self.assertEqual("noul", question["type"])
self.assertEqual({"true", "false"}, set(question["criteria"]))
for question_id in CHOICE_QUESTION_IDS:
question = body["questions"][question_id]
self.assertEqual("choice", question["type"])
self.assertGreaterEqual(len(question["criteria"]), 2)
sore_spot_question = body["questions"][SORE_SPOT_QUESTION_ID]
self.assertEqual("choice", sore_spot_question["type"])
self.assertEqual({"none", "spot_1", "spot_2"}, set(sore_spot_question["criteria"]))
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise({"turn": "I hear you."})
result = await client.appraise(state)
self.assertEqual(1, self.calls)
self.assertEqual("typesafe/jev-1.13-20260917", result.model)
@ -91,15 +167,90 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
self.assertEqual(45, result.output_tokens)
self.assertEqual(0.5, result.emotions["anxiety"].score)
self.assertEqual(0.8, result.emotions["trust"].confidence)
self.assertEqual((0.0, 0.1, 0.8, 0.1, 0.0), result.emotions["anxiety"].probabilities)
self.assertEqual(2, result.sore_spot_count)
self.assertEqual(set(NOUL_QUESTION_IDS), set(result.noul_judgments))
self.assertEqual(
(0.0, 0.1, 0.8, 0.1, 0.0),
result.emotions["anxiety"].probabilities,
set(CHOICE_QUESTION_IDS) | {SORE_SPOT_QUESTION_ID},
set(result.choice_judgments),
)
self.assertEqual(0.7, result.noul_judgments["a_judged"].probability)
# 공식 noul 응답에는 confidence 필드가 없다 — 응답에 없으면 None으로 보존한다.
self.assertIsNone(result.noul_judgments["a_judged"].confidence)
self.assertGreaterEqual(result.latency_ms, 0)
async def test_noul_confidence_is_preserved_when_the_response_includes_it(self) -> None:
state = _state()
payload, questions = _response(state)
payload["answers"]["a_judged"] = {"type": "noul", "noul": 0.9, "confidence": 0.55}
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise(state)
self.assertEqual(0.9, result.noul_judgments["a_judged"].probability)
self.assertEqual(0.55, result.noul_judgments["a_judged"].confidence)
async def test_first_turn_excludes_a_understood(self) -> None:
state = _state(first_turn=True)
payload, questions = _response(state)
async def handler(request: httpx.Request) -> httpx.Response:
body = json.loads(request.content)
self.assertNotIn("a_understood", body["questions"])
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise(state)
self.assertNotIn("a_understood", questions)
self.assertNotIn("a_understood", result.noul_judgments)
async def test_empty_sore_spots_excludes_a_sore_spot(self) -> None:
state = _state(sore_spots=(), forbidden=())
payload, questions = _response(state)
async def handler(request: httpx.Request) -> httpx.Response:
body = json.loads(request.content)
self.assertNotIn(SORE_SPOT_QUESTION_ID, body["questions"])
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise(state)
self.assertNotIn(SORE_SPOT_QUESTION_ID, questions)
self.assertNotIn(SORE_SPOT_QUESTION_ID, result.choice_judgments)
self.assertEqual(0, result.sore_spot_count)
async def test_sore_spots_and_forbidden_are_combined_and_capped_at_twelve(self) -> None:
state = _state(
sore_spots=tuple(f"민감{i}" for i in range(8)),
forbidden=tuple(f"금기{i}" for i in range(8)),
)
payload, questions = _response(state)
async def handler(request: httpx.Request) -> httpx.Response:
body = json.loads(request.content)
criteria = body["questions"][SORE_SPOT_QUESTION_ID]["criteria"]
self.assertEqual(13, len(criteria)) # none + 최대 12개
self.assertEqual(
{"none", *(f"spot_{index}" for index in range(1, MAX_SORE_SPOTS + 1))},
set(criteria),
)
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise(state)
self.assertEqual(MAX_SORE_SPOTS, result.sore_spot_count)
self.assertEqual(13, len(questions[SORE_SPOT_QUESTION_ID]["criteria"])) # type: ignore[arg-type]
async def test_startup_does_not_issue_a_request(self) -> None:
async def handler(request: httpx.Request) -> httpx.Response:
self.calls += 1
payload, _ = _response(_state())
return httpx.Response(500)
client = await self._client(handler)
@ -109,7 +260,8 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
async def test_empty_key_fails_without_external_call(self) -> None:
async def handler(request: httpx.Request) -> httpx.Response:
self.calls += 1
return httpx.Response(200, json=_response())
payload, _ = _response(_state())
return httpx.Response(200, json=payload)
client = JevClient(
api_key="",
@ -141,24 +293,26 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
client = await self._client(handler)
with self.assertRaisesRegex(JevError, code):
await client.appraise({})
await client.appraise(_state())
self.assertEqual(1, self.calls)
async def test_timeout_and_transport_failures_are_typed(self) -> None:
payload, _ = _response(_state())
async def delayed(request: httpx.Request) -> httpx.Response:
await asyncio.sleep(1)
return httpx.Response(200, json=_response())
return httpx.Response(200, json=payload)
client = await self._client(delayed)
with self.assertRaisesRegex(JevError, "timeout"):
await client.appraise({})
await client.appraise(_state())
async def unavailable(request: httpx.Request) -> httpx.Response:
raise httpx.ConnectError("network unavailable", request=request)
client = await self._client(unavailable)
with self.assertRaisesRegex(JevError, "transport"):
await client.appraise({})
await client.appraise(_state())
async def test_cancellation_propagates(self) -> None:
async def cancelled(request: httpx.Request) -> httpx.Response:
@ -166,86 +320,146 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
client = await self._client(cancelled)
with self.assertRaises(asyncio.CancelledError):
await client.appraise({})
await client.appraise(_state())
async def test_rejects_malformed_score_responses(self) -> None:
async def test_rejects_malformed_responses(self) -> None:
state = _state()
base_payload, questions = _response(state)
invalid_payloads: list[dict[str, object]] = []
missing_dimension = _response()
del missing_dimension["answers"]["trust"]
invalid_payloads.append(missing_dimension)
missing_question = copy.deepcopy(base_payload)
del missing_question["answers"]["trust"]
invalid_payloads.append(missing_question)
non_finite = _response()
non_finite["answers"]["anxiety"]["score"] = float("nan")
invalid_payloads.append(non_finite)
extra_question = copy.deepcopy(base_payload)
extra_question["answers"]["unexpected_question"] = _noul_answer()
invalid_payloads.append(extra_question)
out_of_range = _response()
out_of_range["answers"]["anxiety"]["score"] = 4.1
invalid_payloads.append(out_of_range)
non_finite_score = copy.deepcopy(base_payload)
non_finite_score["answers"]["anxiety"]["score"] = float("nan")
invalid_payloads.append(non_finite_score)
invalid_probabilities = _response()
invalid_probabilities["answers"]["anxiety"]["probabilities"]["2"] = 0.7
invalid_payloads.append(invalid_probabilities)
out_of_range_score = copy.deepcopy(base_payload)
out_of_range_score["answers"]["anxiety"]["score"] = 4.1
invalid_payloads.append(out_of_range_score)
invalid_high_probabilities = _response()
invalid_high_probabilities["answers"]["anxiety"]["probabilities"]["2"] = 0.9
invalid_payloads.append(invalid_high_probabilities)
wrong_type = copy.deepcopy(base_payload)
wrong_type["answers"]["a_judged"]["type"] = "score"
invalid_payloads.append(wrong_type)
missing_legend = _response()
del missing_legend["answers"]["anxiety"]["legend"]["4"]
invalid_payloads.append(missing_legend)
noul_out_of_range = copy.deepcopy(base_payload)
noul_out_of_range["answers"]["a_judged"]["noul"] = 1.5
invalid_payloads.append(noul_out_of_range)
bad_usage = _response()
noul_missing_field = copy.deepcopy(base_payload)
del noul_missing_field["answers"]["a_judged"]["noul"]
invalid_payloads.append(noul_missing_field)
noul_wrong_field_name = copy.deepcopy(base_payload)
del noul_wrong_field_name["answers"]["a_judged"]["noul"]
noul_wrong_field_name["answers"]["a_judged"]["probability"] = 0.7
invalid_payloads.append(noul_wrong_field_name)
noul_non_finite = copy.deepcopy(base_payload)
noul_non_finite["answers"]["a_judged"]["noul"] = float("nan")
invalid_payloads.append(noul_non_finite)
noul_boolean = copy.deepcopy(base_payload)
noul_boolean["answers"]["a_judged"]["noul"] = True
invalid_payloads.append(noul_boolean)
choice_unknown_code = copy.deepcopy(base_payload)
choice_unknown_code["answers"]["a_coping"]["choice"] = "not_a_code"
invalid_payloads.append(choice_unknown_code)
choice_missing_option = copy.deepcopy(base_payload)
del choice_missing_option["answers"]["a_coping"]["probabilities"]["overwhelming"]
invalid_payloads.append(choice_missing_option)
choice_bad_sum = copy.deepcopy(base_payload)
choice_bad_sum["answers"]["a_coping"]["probabilities"] = {
"nothing_asked": 0.5,
"manageable": 0.5,
"stretch": 0.5,
"overwhelming": 0.5,
}
invalid_payloads.append(choice_bad_sum)
bad_usage = copy.deepcopy(base_payload)
bad_usage["usage"]["input_tokens"] = -1
invalid_payloads.append(bad_usage)
bad_cost = _response()
bad_cost["usage"]["cost"] = -0.01
invalid_payloads.append(bad_cost)
for payload in invalid_payloads:
with self.subTest(payload=payload):
async def handler(request: httpx.Request, payload: dict[str, object] = payload) -> httpx.Response:
return httpx.Response(
200,
content=json.dumps(copy.deepcopy(payload), allow_nan=True),
content=json.dumps(payload, allow_nan=True),
headers={"Content-Type": "application/json"},
)
client = await self._client(handler)
with self.assertRaisesRegex(JevError, "malformed_response"):
await client.appraise({})
await client.appraise(state)
async def test_accepts_two_decimal_probability_sum_rounding(self) -> None:
for probability, expected_sum in ((0.79, 0.99), (0.81, 1.01)):
with self.subTest(expected_sum=expected_sum):
payload = _response()
payload["answers"]["anxiety"]["probabilities"]["2"] = probability
async def test_choice_probability_tolerance_scales_with_option_count(self) -> None:
state = _state()
payload, questions = _response(state)
option_count = len(questions["c_display"]["criteria"]) # type: ignore[arg-type]
codes = list(questions["c_display"]["criteria"]) # type: ignore[arg-type]
# 허용오차 경계 바로 안쪽: N * 0.005 만큼 반올림된 합.
drift = option_count * 0.005 - 0.0005
probabilities = {code: 1.0 / option_count for code in codes}
probabilities[codes[0]] += drift
payload["answers"]["c_display"] = {
"type": "choice",
"choice": codes[0],
"probabilities": probabilities,
"confidence": 0.6,
}
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise({})
self.assertEqual(0.5, result.emotions["anxiety"].score)
self.assertEqual(
(0.0, 0.1, probability, 0.1, 0.0),
result.emotions["anxiety"].probabilities,
)
client = await self._client(handler)
result = await client.appraise(state)
self.assertEqual(codes[0], result.choice_judgments["c_display"].choice)
async def test_choice_probability_order_follows_criteria_order(self) -> None:
state = _state()
payload, questions = _response(state)
codes = list(questions["c_display"]["criteria"]) # type: ignore[arg-type]
reordered = {code: (0.7 if code == codes[-1] else 0.1) for code in codes}
payload["answers"]["c_display"] = {
"type": "choice",
"choice": codes[-1],
"probabilities": reordered,
"confidence": 0.6,
}
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise(state)
self.assertEqual(tuple(codes), tuple(result.choice_judgments["c_display"].probabilities))
async def test_rejects_a_response_from_a_different_model(self) -> None:
payload = _response()
payload["model"] = "jev-unknown"
state = _state()
payload, _ = _response(state, model="jev-unknown")
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
client = await self._client(handler)
with self.assertRaisesRegex(JevError, "model_mismatch"):
await client.appraise({})
await client.appraise(state)
async def test_explicit_typesafe_alias_records_the_resolved_version(self) -> None:
payload = _response("jev-1.13.0")
state = _state()
payload, _ = _response(state, model="jev-1.13.0")
async def handler(request: httpx.Request) -> httpx.Response:
self.assertEqual("jev-latest", json.loads(request.content)["model"])
@ -260,14 +474,17 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
)
await client.startup()
self.addAsyncCleanup(client.shutdown)
result = await client.appraise({})
result = await client.appraise(state)
self.assertEqual("jev-1.13.0", result.model)
async def test_exact_openrouter_model_slug_is_preserved(self) -> None:
state = _state()
payload, _ = _response(state, model="typesafe/jev-1.13")
async def handler(request: httpx.Request) -> httpx.Response:
self.assertEqual("typesafe/jev-1.13", json.loads(request.content)["model"])
return httpx.Response(200, json=_response("typesafe/jev-1.13"))
return httpx.Response(200, json=payload)
client = JevClient(
provider="openrouter",
@ -278,12 +495,13 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
)
await client.startup()
self.addAsyncCleanup(client.shutdown)
result = await client.appraise({})
result = await client.appraise(state)
self.assertEqual("typesafe/jev-1.13", result.model)
async def test_typesafe_uses_only_its_explicit_route_and_key(self) -> None:
payload = _response("jev-1.13.0")
state = _state()
payload, _ = _response(state, model="jev-1.13.0")
del payload["usage"]["cost"]
async def handler(request: httpx.Request) -> httpx.Response:
@ -300,7 +518,7 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
)
await client.startup()
self.addAsyncCleanup(client.shutdown)
result = await client.appraise({})
result = await client.appraise(state)
self.assertEqual("typesafe", result.provider)
self.assertIsNone(result.cost_usd)
@ -312,6 +530,7 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
jev_model="~typesafe/jev-latest",
jev_timeout_seconds=0.05,
)
state = _state()
seen_headers: list[str] = []
async def handler(request: httpx.Request) -> httpx.Response:
@ -321,7 +540,8 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
if str(request.url) == OPENROUTER_JEV_ENDPOINT
else "jev-1.13.0"
)
return httpx.Response(200, json=_response(response_model))
payload, _ = _response(state, model=response_model)
return httpx.Response(200, json=payload)
with patch.object(jev_module, "settings", configured):
openrouter = JevClient(
@ -338,26 +558,27 @@ class JevClientTest(unittest.IsolatedAsyncioTestCase):
await typesafe.startup()
self.addAsyncCleanup(openrouter.shutdown)
self.addAsyncCleanup(typesafe.shutdown)
await openrouter.appraise({})
await typesafe.appraise({})
await openrouter.appraise(state)
await typesafe.appraise(state)
self.assertEqual(["Bearer openrouter-key", "Bearer typesafe-key"], seen_headers)
async def test_openrouter_allows_optional_score_metadata(self) -> None:
payload = _response()
async def test_openrouter_allows_optional_score_and_noul_metadata(self) -> None:
state = _state()
payload, _ = _response(state)
for answer in payload["answers"].values():
del answer["confidence"]
del answer["legend"]
del answer["probabilities"]
answer.pop("confidence", None)
answer.pop("legend", None)
async def handler(request: httpx.Request) -> httpx.Response:
return httpx.Response(200, json=payload)
client = await self._client(handler)
result = await client.appraise({})
result = await client.appraise(state)
self.assertIsNone(result.emotions["anxiety"].confidence)
self.assertIsNone(result.emotions["anxiety"].probabilities)
self.assertIsNone(result.noul_judgments["a_judged"].confidence)
self.assertIsNone(result.choice_judgments["a_coping"].confidence)
if __name__ == "__main__":