Broker usage: add per-model token/spend breakdown + total_tokens + email
This commit is contained in:
1 parent
60d1bf2328
commit
0453cee019
1 file changed
+52
-11
+52
-11
@@ -1183,14 +1183,19 @@ async def broker_revoke(name: str, request: Request):
|
||||
|
||||
|
||||
async def broker_get_usage(email: str) -> dict:
|
||||
"""Return the member's spend vs. balance, from their team.
|
||||
"""Return the member's spend vs. balance, plus per-model and token detail.
|
||||
|
||||
LiteLLM's team object exposes `spend` and `max_budget` directly, so we read
|
||||
the member's team and return {balance, spend, remaining}. This is the
|
||||
read-only primitive the member dashboard uses (no master key on the client).
|
||||
Reads the member's team (spend + max_budget) and aggregates their recent
|
||||
spend logs by model for token counts. This is the read-only primitive the
|
||||
member dashboard uses (no master key on the client).
|
||||
"""
|
||||
team_id = await litellm_get_or_create_team(email, get_member_balance(email))
|
||||
|
||||
balance = get_member_balance(email)
|
||||
spend = 0.0
|
||||
|
||||
async with httpx.AsyncClient() as client:
|
||||
# 1. Team spend + budget
|
||||
r = await client.get(
|
||||
f"{LITELLM_BASE}/team/list",
|
||||
headers={"Authorization": f"Bearer {LITELLM_MASTER_KEY}"},
|
||||
@@ -1198,14 +1203,50 @@ async def broker_get_usage(email: str) -> dict:
|
||||
r.raise_for_status()
|
||||
for t in r.json():
|
||||
if t.get("team_id") == team_id:
|
||||
balance = float(t.get("max_budget") or get_member_balance(email))
|
||||
balance = float(t.get("max_budget") or balance)
|
||||
spend = float(t.get("spend") or 0.0)
|
||||
return {
|
||||
"balance": balance,
|
||||
"spend": spend,
|
||||
"remaining": max(balance - spend, 0.0),
|
||||
}
|
||||
return {"balance": get_member_balance(email), "spend": 0.0, "remaining": get_member_balance(email)}
|
||||
break
|
||||
|
||||
# 2. Per-model token/spend breakdown from spend logs (filtered by team).
|
||||
# /spend/logs does not reliably filter by team_id server-side, so we
|
||||
# filter client-side on the returned rows.
|
||||
models = {}
|
||||
total_tokens = 0
|
||||
try:
|
||||
r2 = await client.get(
|
||||
f"{LITELLM_BASE}/spend/logs",
|
||||
headers={"Authorization": f"Bearer {LITELLM_MASTER_KEY}"},
|
||||
)
|
||||
r2.raise_for_status()
|
||||
rows = r2.json()
|
||||
if isinstance(rows, dict):
|
||||
rows = rows.get("data", rows.get("logs", []))
|
||||
for row in rows if isinstance(rows, list) else []:
|
||||
if row.get("team_id") != team_id:
|
||||
continue
|
||||
model = row.get("model") or "unknown"
|
||||
m = models.setdefault(model, {"spend": 0.0, "tokens": 0, "calls": 0})
|
||||
m["spend"] += float(row.get("spend") or 0.0)
|
||||
m["tokens"] += int(row.get("total_tokens") or 0)
|
||||
m["calls"] += 1
|
||||
total_tokens += int(row.get("total_tokens") or 0)
|
||||
except Exception as e: # spend logs are best-effort; don't fail the whole view
|
||||
logger.warning("spend/logs fetch failed: %s", e)
|
||||
|
||||
# Sort models by spend descending (which models are draining the most).
|
||||
model_list = [
|
||||
{"model": name, **stats}
|
||||
for name, stats in sorted(models.items(), key=lambda kv: kv[1]["spend"], reverse=True)
|
||||
]
|
||||
|
||||
return {
|
||||
"email": email,
|
||||
"balance": balance,
|
||||
"spend": spend,
|
||||
"remaining": max(balance - spend, 0.0),
|
||||
"total_tokens": total_tokens,
|
||||
"models": model_list,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/broker/usage")
|
||||
|
||||
Reference in new issue
Block a user