vignette/scripts/smoke-deliberate-practice-api.py
Yun Chan 16e791e044 G0~G8 성과·동맹 측정 OS 작업 일괄 고정
8월 7일까지 워킹트리에만 남아 있던 미커밋 작업을 커밋한다. 여러 사본
폴더(worktree·clone)에 흩어져 있던 중간 스냅샷을 정리하기 전에 원본을
git 이력으로 고정하는 것이 목적이다.

- contracts/routes/services: measurement, outcome_trajectory, rupture_repair,
  deliberate_practice, calibration_transfer, supervision_research,
  multimodal_alliance, continuous_improvement 계열 신규 모듈과 테스트
- infra/db/init: 07~16 마이그레이션(측정 기반~calibration transfer 실행)
- apps/web: 세션 리뷰 카드·관리 화면·E2E 스펙 추가
- docs/ops: G0~G8 라이브 통합·배포·롤백 증거 문서와 evidence JSON/PNG
- scripts: smoke·ledger·릴리스 에이전트·NAS 프리뷰 운영 스크립트

engine.public 로그 .bak과 apps/web/test-results 산출물은 커밋에서 제외했다.
2026-08-08 01:30:53 +09:00

725 lines
28 KiB
Python

"""Exercise the G4 deliberate-practice lifecycle over live HTTP and PostgreSQL."""
from __future__ import annotations
import argparse
import asyncio
import copy
import json
import os
import secrets
import time
import urllib.error
import urllib.request
from dataclasses import dataclass
from http.cookiejar import CookieJar
from pathlib import Path
from typing import Any
from uuid import uuid4
COHORT_ID = "e2e-hanshin"
API_ENV = Path(__file__).resolve().parents[1] / "apps" / "api" / ".env"
class SmokeError(RuntimeError):
pass
@dataclass(frozen=True)
class ApiResponse:
status: int
body: Any
class ApiClient:
def __init__(self, base_url: str, timeout: float) -> None:
self.base_url = base_url.rstrip("/")
self.timeout = timeout
self._opener = urllib.request.build_opener(
urllib.request.HTTPCookieProcessor(CookieJar())
)
def request(
self,
method: str,
path: str,
payload: dict[str, Any] | None = None,
*,
expected: set[int] | None = None,
headers: dict[str, str] | None = None,
) -> ApiResponse:
data = None
request_headers = {"Accept": "application/json", **(headers or {})}
if payload is not None:
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
request_headers["Content-Type"] = "application/json"
request = urllib.request.Request(
f"{self.base_url}{path}",
data=data,
headers=request_headers,
method=method,
)
try:
with self._opener.open(request, timeout=self.timeout) as response:
raw = response.read().decode("utf-8")
result = ApiResponse(response.status, json.loads(raw) if raw else {})
except urllib.error.HTTPError as exc:
raw = exc.read().decode("utf-8", errors="replace")
try:
body = json.loads(raw) if raw else {}
except json.JSONDecodeError:
body = {"detail": raw[:500]}
result = ApiResponse(exc.code, body)
except urllib.error.URLError as exc:
raise SmokeError(
f"{method} {path} transport failed: {type(exc.reason).__name__}"
) from exc
if result.status not in (expected or {200}):
detail = result.body.get("detail") if isinstance(result.body, dict) else None
raise SmokeError(
f"{method} {path} returned HTTP {result.status}; detail={detail!r}"
)
return result
def _sign_in(
client: ApiClient,
*,
suffix: str,
identity: str,
role: str,
cohort_ids: list[str],
) -> str:
client.request(
"POST",
"/auth/dev-login",
{
"email": f"dev.e2e.practice.{identity}.{suffix}@hs.ac.kr",
"role": role,
"display_name": f"Practice {identity.title()}",
"cohort_ids": cohort_ids,
},
)
client.request(
"POST",
"/users/me/onboarding",
{
"legal_name": f"Practice {identity.title()}",
"affiliation": "한신대학교",
"department": "상담심리학과",
"grade_level": "통합검증",
"phone": "010-0000-0000",
"contact_address": "경기도 오산시 한신대학교",
"nickname": f"Practice {identity.title()}",
"self_introduction": "G4 숙의연습 API 검증 fixture입니다.",
"avatar_url": "",
"terms_accepted": True,
"privacy_accepted": True,
},
)
me = client.request("GET", "/auth/me")
user_id = str(me.body.get("user_id") or "")
if not user_id:
raise SmokeError(f"dev-login omitted user_id for {identity}")
return user_id
def _load_api_env() -> None:
if not API_ENV.exists():
return
for raw_line in API_ENV.read_text(encoding="utf-8").splitlines():
line = raw_line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
key, value = line.split("=", 1)
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
def _choose_distinct_personas(client: ApiClient) -> tuple[str, str]:
response = client.request("GET", "/personas")
usable = [
item
for item in response.body
if isinstance(item, dict)
and item.get("source") == "database"
and not item.get("degraded")
and item.get("code")
]
codes = list(dict.fromkeys(str(item["code"]) for item in usable))
if len(codes) < 2:
raise SmokeError(
"persona catalog requires two distinct non-degraded database personas"
)
source = "P1" if "P1" in codes else codes[0]
runtime = next(code for code in codes if code != source)
return source, runtime
def _wait_for_session_review(
client: ApiClient,
session_id: str,
*,
timeout: float,
interval: float,
) -> dict[str, Any]:
if timeout <= 0 or interval <= 0:
raise SmokeError("review poll timeout and interval must be positive")
deadline = time.monotonic() + timeout
poll_count = 0
while True:
poll_count += 1
detail = client.request("GET", f"/sessions/{session_id}")
if detail.body.get("review_ready") is True:
review = client.request("GET", f"/sessions/{session_id}/review")
if review.body.get("reviewReady") is not True:
raise SmokeError(
"session detail was review-ready but review payload was not ready"
)
return {"poll_count": poll_count, "review": review.body}
now = time.monotonic()
if now >= deadline:
raise SmokeError(
f"session_end evaluator did not become ready within {timeout:.1f}s "
f"for session {session_id}"
)
time.sleep(min(interval, deadline - now))
def _durable_turn_ids(review: dict[str, Any]) -> list[str]:
ids = [str(item["turn_id"]) for item in review.get("turns", []) if item.get("turn_id")]
if len(ids) < 2:
raise SmokeError("session review did not expose both durable turn UUIDs")
return ids[:2]
def _assert_no_aggregate_score(value: Any, path: str = "response") -> None:
if isinstance(value, dict):
forbidden = {"total", "total_score", "overall_score", "xp"} & set(value)
if forbidden:
raise SmokeError(f"{path} exposed aggregate/reward keys: {forbidden}")
for key, child in value.items():
_assert_no_aggregate_score(child, f"{path}.{key}")
elif isinstance(value, list):
for index, child in enumerate(value):
_assert_no_aggregate_score(child, f"{path}[{index}]")
def _runtime_attempt_read_proof(
read_model: dict[str, Any],
*,
session_id: str,
durable_turn_ids: list[str],
) -> dict[str, Any]:
episode = next(
(
item
for item in read_model.get("episodes", [])
if str(item.get("session_id")) == session_id
),
None,
)
if episode is None:
raise SmokeError("learner read model omitted the runtime practice episode")
attempts = episode.get("attempts") or []
if not attempts:
raise SmokeError("runtime practice episode omitted durable attempts")
durable = set(durable_turn_ids)
evidence: set[str] = set()
model_run_ids: list[str] = []
attempt_record_ids: list[str] = []
for attempt in attempts:
if attempt.get("scenario_novelty") != "unseen_transfer":
raise SmokeError("different-persona runtime session was not classified as unseen")
if attempt.get("learner_claimed_success") is not False:
raise SmokeError("runtime attempt trusted a learner success claim")
attempt_evidence = {str(item) for item in attempt.get("evidence_turn_ids") or []}
if not attempt_evidence or not attempt_evidence.issubset(durable):
raise SmokeError("runtime attempt evidence is not bound to its durable turns")
evidence.update(attempt_evidence)
observation = (attempt.get("attempt_payload") or {}).get("observation") or {}
criterion = observation.get("criterion") or {}
if (
criterion.get("source_kind") != "model_inferred"
or criterion.get("perspective") != "independent_observer"
or not criterion.get("model_run_id")
):
raise SmokeError("runtime criterion omitted independent model provenance")
model_run_ids.append(str(criterion["model_run_id"]))
attempt_record_ids.append(str(attempt["attempt_record_id"]))
if evidence != durable:
raise SmokeError("runtime episode did not preserve every durable turn UUID")
return {
"episode_submission_id": str(episode["episode_submission_id"]),
"attempt_record_ids": attempt_record_ids,
"model_run_ids": model_run_ids,
"durable_evidence_turn_ids": sorted(evidence),
"scenario_novelty": "unseen_transfer",
}
def _json_object(value: Any) -> dict[str, Any]:
if isinstance(value, dict):
return value
if isinstance(value, str):
parsed = json.loads(value)
if isinstance(parsed, dict):
return parsed
raise SmokeError("Postgres model-run metadata was not a JSON object")
async def _fetch_runtime_db_proof(
dsn: str,
*,
learner_id: str,
prescription_id: str,
session_id: str,
durable_turn_ids: list[str],
model_run_ids: list[str],
) -> dict[str, Any]:
import asyncpg
conn = await asyncpg.connect(dsn)
try:
async with conn.transaction():
await conn.execute("SELECT set_config('app.ai_context', 'true', true)")
await conn.execute("SELECT set_config('app.current_ai_view', 'evaluator', true)")
await conn.execute("SELECT set_config('app.current_role', 'admin', true)")
await conn.execute("SELECT set_config('app.current_uid', $1, true)", learner_id)
await conn.execute("SELECT set_config('app.current_cohort', $1, true)", COHORT_ID)
turn_rows = await conn.fetch(
"""
SELECT id::text, speaker, seq, audio_ref, silence_ms, speech_rate
FROM app.turns
WHERE session_id = $1::uuid
ORDER BY seq
""",
session_id,
)
model_rows = await conn.fetch(
"""
SELECT model_run_id::text, session_id::text, turn_id::text,
agent_role, provider, model, prompt_bundle_id,
prompt_bundle_version, structured_schema_version,
input_evidence_hash, status, metadata
FROM audit.model_run
WHERE session_id = $1::uuid
AND provider = 'vignette-runtime'
AND model = 'practice-runtime-observer'
ORDER BY model_run_id
""",
session_id,
)
finally:
await conn.close()
durable = set(durable_turn_ids)
actual_turns = {str(row["id"]) for row in turn_rows}
if not durable.issubset(actual_turns):
raise SmokeError("Postgres omitted runtime attempt durable turn UUIDs")
if {str(row["speaker"]) for row in turn_rows} != {"counselor", "client"}:
raise SmokeError("runtime session did not persist counselor/client turn roles")
if any(
row["audio_ref"] is not None
or row["silence_ms"] is not None
or row["speech_rate"] is not None
for row in turn_rows
):
raise SmokeError("text-only runtime smoke unexpectedly persisted voice features")
expected_runs = set(model_run_ids)
actual_runs = {str(row["model_run_id"]) for row in model_rows}
if actual_runs != expected_runs:
raise SmokeError("Postgres observer model runs differ from API attempt provenance")
counselor_turns = {
str(row["id"]) for row in turn_rows if str(row["speaker"]) == "counselor"
}
for row in model_rows:
metadata = _json_object(row["metadata"])
if (
row["agent_role"] != "evaluator"
or row["provider"] != "vignette-runtime"
or row["model"] != "practice-runtime-observer"
or row["prompt_bundle_id"] != "practice-runtime-observer"
or row["prompt_bundle_version"] != "practice-runtime-observer-v1"
or row["structured_schema_version"]
!= "vignette.practice-runtime-observation.v1"
or row["status"] != "ready"
or not row["input_evidence_hash"]
or str(row["turn_id"]) not in counselor_turns
or metadata.get("prescription_id") != prescription_id
or metadata.get("practice_session_id") != session_id
):
raise SmokeError("Postgres observer model-run provenance is incomplete")
return {
"durable_turn_count": len(turn_rows),
"durable_turn_ids_match_api": True,
"observer_model_run_count": len(model_rows),
"observer_model_run_ids_match_api": True,
"observer_schema": "vignette.practice-runtime-observation.v1",
"voice_feature_row_count": 0,
}
def _load_live_case(path: Path, turn_ids: list[str]) -> dict[str, Any]:
pack = json.loads(path.read_text(encoding="utf-8"))
case = copy.deepcopy(pack["cases"][0])
card = case["coaching_cards"][0]
old_card_ref = card["evidence_refs"][0]["ref_id"]
card["evidence_refs"][0]["ref_id"] = turn_ids[0]
for target in card["targets"]:
activity = target["activity"]
if activity.get("pause_at_evidence_ref") == old_card_ref:
activity["pause_at_evidence_ref"] = turn_ids[0]
attempt = case["episodes"][0]["attempts"][0]
attempt["evidence_refs"][0]["ref_id"] = turn_ids[0]
attempt["evidence_refs"][1]["ref_id"] = turn_ids[1]
return case
def run(args: argparse.Namespace) -> dict[str, Any]:
_load_api_env()
dsn = args.database_url or os.environ.get("DATABASE_URL")
if not dsn:
raise SmokeError("DATABASE_URL is required via --database-url or apps/api/.env")
if len(args.internal_token) < 32:
raise SmokeError("--internal-token must contain at least 32 characters")
health = ApiClient(args.api_base_url, args.request_timeout).request("GET", "/health")
if not health.body.get("db") or not health.body.get("engine"):
raise SmokeError("API health is not DB+engine ready")
suffix = f"{int(time.time())}.{secrets.token_hex(3)}"
learner = ApiClient(args.api_base_url, args.request_timeout)
teacher = ApiClient(args.api_base_url, args.request_timeout)
other_learner = ApiClient(args.api_base_url, args.request_timeout)
other_teacher = ApiClient(args.api_base_url, args.request_timeout)
learner_id = _sign_in(
learner,
suffix=suffix,
identity="learner",
role="learner",
cohort_ids=[COHORT_ID],
)
_sign_in(
teacher,
suffix=suffix,
identity="teacher",
role="teacher",
cohort_ids=[COHORT_ID],
)
_sign_in(
other_learner,
suffix=suffix,
identity="other-learner",
role="learner",
cohort_ids=[COHORT_ID],
)
_sign_in(
other_teacher,
suffix=suffix,
identity="other-teacher",
role="teacher",
cohort_ids=["e2e-other-cohort"],
)
source_persona, runtime_persona = _choose_distinct_personas(learner)
started = learner.request(
"POST",
"/sessions",
{
"persona_code": source_persona,
"theory_mode": "humanistic",
"goal_stages": ["라포", "탐색"],
},
expected={201},
)
session_id = str(started.body["session_id"])
learner.request(
"POST",
f"/sessions/{session_id}/turn",
{"text": "지금 느끼는 막막함이 얼마나 큰지 제가 제대로 이해했는지 확인해도 될까요?"},
)
learner.request("POST", f"/sessions/{session_id}/end")
source_review = _wait_for_session_review(
learner,
session_id,
timeout=args.review_poll_timeout,
interval=args.review_poll_interval,
)
turn_ids = _durable_turn_ids(source_review["review"])
live_case = _load_live_case(Path(args.benchmark_path), turn_ids)
internal = ApiClient(args.api_base_url, args.request_timeout)
token_header = {"X-Vignette-Practice-Token": args.internal_token}
prescription_submission_id = str(uuid4())
prescription = {
"submission_id": prescription_submission_id,
"coaching_cards": live_case["coaching_cards"],
"competency_graph": live_case["graph"],
"evidence_turn_ids": turn_ids,
}
prescription_path = f"/internal/sessions/{session_id}/practice/prescriptions"
created = internal.request(
"POST", prescription_path, prescription, expected={201}, headers=token_header
)
retried = internal.request(
"POST", prescription_path, prescription, expected={201}, headers=token_header
)
if created.body.get("prescription_ids") != retried.body.get(
"prescription_ids"
) or retried.body.get("idempotent_replay") is not True:
raise SmokeError("same prescription submission retry was not stable")
changed_prescription = copy.deepcopy(prescription)
changed_prescription["coaching_cards"][0]["uncertainty"] = 0.21
internal.request(
"POST",
prescription_path,
changed_prescription,
expected={409},
headers=token_header,
)
prescription_id = str(created.body["next_prescription_id"])
episode = copy.deepcopy(live_case["episodes"][0])
if episode["prescription_id"] != prescription_id:
raise SmokeError("live curriculum selected a prescription other than the benchmark expectation")
attempt_submission_id = str(uuid4())
attempt_body = {"submission_id": attempt_submission_id, "episode": episode}
attempt_path = f"/practice/{prescription_id}/attempts"
attempted = learner.request("POST", attempt_path, attempt_body, expected={201})
attempted_retry = learner.request("POST", attempt_path, attempt_body, expected={201})
if attempted.body.get("decision_id") != attempted_retry.body.get(
"decision_id"
) or attempted_retry.body.get("idempotent_replay") is not True:
raise SmokeError("same learner attempt retry was not stable")
changed_attempt = copy.deepcopy(attempt_body)
changed_attempt["episode"]["attempts"][0]["uncertainty"] = 0.21
learner.request("POST", attempt_path, changed_attempt, expected={409})
if attempted.body.get("progress") != "practicing" or attempted.body.get(
"mastery_allowed"
) is not False:
raise SmokeError("reward-claim attempt bypassed the transfer mastery gate")
second_suffix = secrets.token_hex(4)
second_episode = copy.deepcopy(episode)
second_episode["episode_id"] = f"oas-g4-episode-live-second-{second_suffix}"
second_episode["attempts"][0]["attempt_id"] = (
f"oas-g4-attempt-live-second-{second_suffix}"
)
learner.request(
"POST",
attempt_path,
{"submission_id": str(uuid4()), "episode": second_episode},
expected={201},
)
runtime_started = learner.request(
"POST",
"/sessions",
{
"persona_code": runtime_persona,
"theory_mode": "humanistic",
"goal_stages": ["라포", "탐색"],
},
expected={201},
)
runtime_session_id = str(runtime_started.body["session_id"])
learner.request(
"POST",
f"/sessions/{runtime_session_id}/turn",
{
"text": (
"그 막막함이 하루를 시작하기 어렵게 만드는 것 같아요. "
"제가 이해한 게 맞을까요?"
)
},
)
learner.request("POST", f"/sessions/{runtime_session_id}/end")
runtime_review = _wait_for_session_review(
learner,
runtime_session_id,
timeout=args.review_poll_timeout,
interval=args.review_poll_interval,
)
runtime_turn_ids = _durable_turn_ids(runtime_review["review"])
runtime_attempt_path = (
f"/practice/{prescription_id}/attempts/from-session/{runtime_session_id}"
)
runtime_attempted = learner.request(
"POST", runtime_attempt_path, expected={201}
)
runtime_attempted_retry = learner.request(
"POST", runtime_attempt_path, expected={201}
)
stable_runtime_fields = ("submission_id", "snapshot_id", "decision_id")
if (
any(
runtime_attempted.body.get(field)
!= runtime_attempted_retry.body.get(field)
for field in stable_runtime_fields
)
or runtime_attempted.body.get("idempotent_replay") is not False
or runtime_attempted_retry.body.get("idempotent_replay") is not True
):
raise SmokeError("same runtime session observation retry was not stable")
learner_read = learner.request("GET", "/practice/learners/me")
teacher_read = teacher.request("GET", f"/practice/learners/{learner_id}")
other_learner.request("GET", f"/practice/learners/{learner_id}", expected={403})
other_teacher.request("GET", f"/practice/learners/{learner_id}", expected={404})
episodes = learner_read.body.get("episodes") or []
target_episode = next(
(item for item in episodes if item.get("episode_key") == episode["episode_id"]),
None,
)
attempts = (target_episode or {}).get("attempts") or []
if not attempts:
raise SmokeError("learner read model omitted the durable attempt")
attempt_record_id = str(attempts[0]["attempt_record_id"])
second_target_episode = next(
(
item
for item in episodes
if item.get("episode_key") == second_episode["episode_id"]
),
None,
)
second_target_attempts = (second_target_episode or {}).get("attempts") or []
if not second_target_attempts:
raise SmokeError("live practice fixture did not persist two distinct attempt ledgers")
second_attempt_record_id = str(second_target_attempts[0]["attempt_record_id"])
if second_attempt_record_id == attempt_record_id:
raise SmokeError("manual practice attempts collapsed into one ledger row")
runtime_read_proof = _runtime_attempt_read_proof(
learner_read.body,
session_id=runtime_session_id,
durable_turn_ids=runtime_turn_ids,
)
runtime_db_proof = asyncio.run(
_fetch_runtime_db_proof(
dsn,
learner_id=learner_id,
prescription_id=prescription_id,
session_id=runtime_session_id,
durable_turn_ids=runtime_turn_ids,
model_run_ids=runtime_read_proof["model_run_ids"],
)
)
correction_submission_id = str(uuid4())
correction = {
"submission_id": correction_submission_id,
"corrected_outcome": "needs_retry",
"correction_reason": "자기 성공 주장과 실제 내담자 반응을 분리해 다시 관찰함",
"evidence_turn_ids": turn_ids,
"counterevidence": ["unseen_transfer_not_verified"],
}
correction_path = f"/practice/attempts/{attempt_record_id}/correction"
corrected = teacher.request("PATCH", correction_path, correction, expected={201})
corrected_retry = teacher.request("PATCH", correction_path, correction, expected={201})
if corrected.body.get("correction_id") != corrected_retry.body.get(
"correction_id"
) or corrected_retry.body.get("idempotent_replay") is not True:
raise SmokeError("same teacher correction retry was not stable")
teacher.request(
"PATCH",
correction_path,
dict(correction, correction_reason="변경된 교정 내용은 충돌이어야 함"),
expected={409},
)
final_read = teacher.request("GET", f"/practice/learners/{learner_id}")
final_attempt = next(
(
item
for final_episode in final_read.body.get("episodes") or []
for item in final_episode.get("attempts") or []
if str(item.get("attempt_record_id")) == attempt_record_id
),
None,
)
final_corrections = (final_attempt or {}).get("corrections") or []
if not final_corrections or final_corrections[-1].get("correction_id") != corrected.body.get(
"correction_id"
):
raise SmokeError("teacher correction did not appear in the append-only read model")
for payload in (learner_read.body, teacher_read.body, final_read.body):
_assert_no_aggregate_score(payload)
if payload.get("clinical_claim_allowed") is not False:
raise SmokeError("practice read omitted the non-clinical boundary")
return {
"ok": True,
"api_base_url": args.api_base_url,
"fixture_policy": "retained unique dev:e2e identities; no fixture deletion",
"session_id": session_id,
"runtime_session_id": runtime_session_id,
"learner_id": learner_id,
"prescription_id": prescription_id,
"attempt_record_id": attempt_record_id,
"second_attempt_record_id": second_attempt_record_id,
"correction_id": str(corrected.body["correction_id"]),
"proof": {
"durable_turn_uuid_count": len(turn_ids),
"source_review_poll_count": source_review["poll_count"],
"same_prescription_retry_stable": True,
"changed_prescription_retry_rejected": True,
"same_attempt_retry_stable": True,
"changed_attempt_retry_rejected": True,
"two_distinct_attempt_ledgers": True,
"reward_claim_did_not_master": True,
"unseen_transfer_gate_preserved": True,
"runtime_session_started_after_prescription": True,
"runtime_persona_differs_from_source": runtime_persona != source_persona,
"runtime_session_end_review_ready": True,
"runtime_review_poll_count": runtime_review["poll_count"],
"runtime_attempt_retry_stable": True,
"runtime_unseen_transfer_classified": True,
"runtime_learner_claim_ignored": True,
"runtime_read_model": runtime_read_proof,
"runtime_postgres": runtime_db_proof,
"physical_mic_or_voice_api_calls": 0,
"same_correction_retry_stable": True,
"changed_correction_retry_rejected": True,
"teacher_correction_append_only": True,
"other_learner_rejected": True,
"cross_cohort_teacher_rejected": True,
"no_aggregate_or_xp": True,
"clinical_claim_allowed": False,
},
}
def main() -> None:
_load_api_env()
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--api-base-url", default="http://127.0.0.1:8009")
parser.add_argument(
"--internal-token",
default=os.environ.get("VIGNETTE_PRACTICE_INTERNAL_TOKEN", ""),
)
parser.add_argument("--database-url", default="")
parser.add_argument("--request-timeout", type=float, default=180.0)
parser.add_argument("--review-poll-timeout", type=float, default=180.0)
parser.add_argument("--review-poll-interval", type=float, default=1.0)
parser.add_argument(
"--benchmark-path",
default="apps/api/app/data/deliberate_practice_benchmark_g4.v1.json",
)
parser.add_argument("--out", default="")
args = parser.parse_args()
result = run(args)
text = json.dumps(result, ensure_ascii=False, indent=2)
if args.out:
path = Path(args.out)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text + "\n", encoding="utf-8")
print(text)
if __name__ == "__main__":
main()