vignette/apps/api/app/services/usage_report.py
2026-08-29 23:59:17 +09:00

223 lines
8.7 KiB
Python

"""AI usage cost verification reports.
The admin API already owns collection. This module turns that existing usage
shape into a deterministic model/provider cost report for ops evidence without
adding any enforcement policy.
"""
from __future__ import annotations
from typing import Any, Mapping
REPORT_SCHEMA = "vignette.ai_usage_model_cost_report.v1"
def _number(value: Any, default: float = 0.0) -> float:
try:
return float(value)
except (TypeError, ValueError):
return default
def _integer(value: Any, default: int = 0) -> int:
try:
return int(value)
except (TypeError, ValueError):
return default
def _ratio(part: float, whole: float) -> float:
if whole <= 0:
return 0.0
return round(part / whole, 6)
def _cost_per_1k_tokens(cost_usd: float, tokens: int) -> float | None:
if tokens <= 0:
return None
return round(cost_usd / tokens * 1000.0, 6)
def _cost_per_turn(cost_usd: float, turns: int) -> float | None:
if turns <= 0:
return None
return round(cost_usd / turns, 6)
def build_model_cost_report(usage: Mapping[str, Any]) -> dict[str, Any]:
"""Build an ops report from an AdminUsageResponse-like mapping."""
total_cost = round(_number(usage.get("cost_usd")), 6)
recorded_cost = round(_number(usage.get("recorded_cost_usd"), total_cost), 6)
estimated_cost = round(_number(usage.get("estimated_cost_usd")), 6)
total_turns = _integer(usage.get("total_turns"))
metered_turns = _integer(usage.get("metered_turns"))
tokens_in = _integer(usage.get("tokens_in"))
tokens_out = _integer(usage.get("tokens_out"))
total_tokens = tokens_in + tokens_out
token_metered_turns = _integer(
usage.get("token_metered_turns"),
metered_turns if total_tokens > 0 else 0,
)
token_unmetered_turns = _integer(
usage.get("token_unmetered_turns"),
max(0, metered_turns - token_metered_turns),
)
by_provider = list(usage.get("by_provider") or [])
budget = dict(usage.get("budget") or {})
evaluator_cache = dict(usage.get("evaluator_cache") or {})
total_cost_basis = str(usage.get("cost_basis") or "provider_reported")
exact_cost_bases = {"provider_estimate", "provider_reported", "reference_rate"}
models: list[dict[str, Any]] = []
for item in by_provider:
row = dict(item or {})
turns = _integer(row.get("turns"))
row_tokens_in = _integer(row.get("tokens_in"))
row_tokens_out = _integer(row.get("tokens_out"))
row_tokens = row_tokens_in + row_tokens_out
row_token_metered_turns = _integer(
row.get("token_metered_turns"),
turns if row_tokens > 0 else 0,
)
row_token_unmetered_turns = _integer(
row.get("token_unmetered_turns"),
max(0, turns - row_token_metered_turns),
)
row_cost = round(_number(row.get("cost_usd")), 6)
row_recorded_cost = round(
_number(row.get("recorded_cost_usd"), row_cost), 6
)
row_estimated_cost = round(_number(row.get("estimated_cost_usd")), 6)
row_cost_basis = str(row.get("cost_basis") or "provider_reported")
row_cost_complete = row_cost_basis not in {
"partial",
"partial_upper_bound",
"unavailable",
}
models.append(
{
"provider": str(row.get("provider") or "unknown"),
"model": str(row.get("model") or "unknown"),
"turns": turns,
"token_metered_turns": row_token_metered_turns,
"token_unmetered_turns": row_token_unmetered_turns,
"tokens_in": row_tokens_in,
"tokens_out": row_tokens_out,
"tokens_total": row_tokens,
"cost_usd": row_cost,
"recorded_cost_usd": row_recorded_cost,
"estimated_cost_usd": row_estimated_cost,
"cost_basis": row_cost_basis,
"rate_label": row.get("rate_label"),
"rate_source_url": row.get("rate_source_url"),
"cost_share": (
_ratio(row_cost, total_cost)
if total_cost_basis in exact_cost_bases
and row_cost_basis in exact_cost_bases
else None
),
"token_share": _ratio(float(row_tokens), float(total_tokens)),
"cost_per_turn_usd": (
_cost_per_turn(row_cost, turns) if row_cost_complete else None
),
"cost_per_1k_tokens_usd": (
_cost_per_1k_tokens(row_cost, row_tokens)
if row_cost_complete
else None
),
}
)
models.sort(key=lambda item: (-float(item["cost_usd"]), item["provider"], item["model"]))
warnings: list[str] = []
if total_turns > 0 and metered_turns < total_turns:
warnings.append("partial_metering")
if token_unmetered_turns > 0:
warnings.append("partial_token_metering")
if total_cost == 0 and metered_turns > 0:
warnings.append("zero_cost_metered_usage")
if estimated_cost > 0:
warnings.append("reference_rate_cost")
if any(item["cost_basis"] == "partial" for item in models):
warnings.append("partial_model_cost")
if any(item["cost_basis"] == "partial_upper_bound" for item in models):
warnings.append("partial_upper_bound_model_cost")
if any(item["cost_basis"] == "reference_upper_bound" for item in models):
warnings.append("reference_upper_bound_cost")
if any(
item["cost_basis"] in {"partial", "partial_upper_bound", "unavailable"}
for item in models
):
warnings.append("unavailable_model_cost")
if str(budget.get("status") or "") in {"warn", "exceeded", "indeterminate"}:
warnings.append(f"budget_{budget.get('status')}")
cache_hit_rate = _number(evaluator_cache.get("hit_rate"))
if bool(evaluator_cache.get("enabled")) and _integer(evaluator_cache.get("requests")) > 0:
if cache_hit_rate < 0.25:
warnings.append("low_evaluator_cache_hit_rate")
if (
models
and isinstance(models[0]["cost_share"], (float, int))
and models[0]["cost_share"] >= 0.8
):
warnings.append("dominant_model_cost")
return {
"schema": REPORT_SCHEMA,
"source": str(usage.get("source") or "unknown"),
"durable": bool(usage.get("durable")),
"window_days": _integer(usage.get("window_days")),
"summary": {
"total_turns": total_turns,
"metered_turns": metered_turns,
"metered_coverage": _ratio(float(metered_turns), float(total_turns)),
"token_metered_turns": token_metered_turns,
"token_unmetered_turns": token_unmetered_turns,
"token_metered_coverage": _ratio(
float(token_metered_turns),
float(metered_turns),
),
"tokens_in": tokens_in,
"tokens_out": tokens_out,
"tokens_total": total_tokens,
"cost_usd": total_cost,
"recorded_cost_usd": recorded_cost,
"estimated_cost_usd": estimated_cost,
"cost_basis": total_cost_basis,
"cost_per_turn_usd": (
_cost_per_turn(total_cost, metered_turns)
if total_cost_basis not in {"partial", "partial_upper_bound", "unavailable"}
else None
),
"cost_per_1k_tokens_usd": (
_cost_per_1k_tokens(total_cost, total_tokens)
if total_cost_basis not in {"partial", "partial_upper_bound", "unavailable"}
else None
),
},
"budget": {
"status": str(budget.get("status") or "disabled"),
"limit_usd": round(_number(budget.get("limit_usd")), 6),
"used_ratio": round(_number(budget.get("used_ratio")), 6),
"remaining_usd": (
round(_number(budget.get("remaining_usd")), 6)
if budget.get("remaining_usd") is not None
else None
),
"cost_basis": str(budget.get("cost_basis") or total_cost_basis),
},
"evaluator_cache": {
"enabled": bool(evaluator_cache.get("enabled")),
"requests": _integer(evaluator_cache.get("requests")),
"hits": _integer(evaluator_cache.get("hits")),
"misses": _integer(evaluator_cache.get("misses")),
"hit_rate": round(cache_hit_rate, 6),
},
"models": models,
"top_cost_model": models[0] if models else None,
"warnings": warnings,
}
__all__ = ["REPORT_SCHEMA", "build_model_cost_report"]