feat(proxy): T-X1 预算四档渐进干预(采纳 ai-model-router budgets 设计)

- billing.BillingMixin 新增 budget_mode(student, est, ts):整数基点判定
  normal(<80%) / optimize(>=80%) / cheap(>=95%) / block(>100%);
  cap<=0 不限额恒 normal;学生缺失 normal(扣费与拒绝权威仍在 try_hold)
- routes._run_chat:optimize/cheap 档自动降档池条目(_downgrade_entry,
  TIERS 排名制降档、杜绝反向升档、跳过停用与 mock),重算 model/est;
  非正常档位经 X-Budget-Mode 响应头如实上报(缓存命中路径同样携带)
- 黄金用例锁死边界:7999/8000/9499/9500/10000/10001bp 九组参数化断言,
  全整数运算无浮点漂移;跨日重置与不限额口径与 try_hold 一致

pytest 443 passed(基线 425 + 18)
This commit is contained in:
tzt
2026-09-18 22:28:50 +08:00
parent ebb3cbb41d
commit aa7cb0704c
4 changed files with 237 additions and 8 deletions
+39
View File
@@ -21,6 +21,10 @@ _SQL_HOLD = """UPDATE students SET
spent_today_milli = ?, spent_date = ? spent_today_milli = ?, spent_date = ?
WHERE id = ? AND balance_milli >= ?""" WHERE id = ? AND balance_milli >= ?"""
# 预算四档阈值(T-X1,基点 bp:1% = 100bp;整数比较,杜绝浮点边界漂移)
BUDGET_OPTIMIZE_BP = 8000 # >= 80% 建议降一档
BUDGET_CHEAP_BP = 9500 # >= 95% 强制最低档
_SQL_HOLD_MARK = """INSERT INTO usage_ledger _SQL_HOLD_MARK = """INSERT INTO usage_ledger
(request_id, ts, key_id, model, bucket, charged_milli, status) (request_id, ts, key_id, model, bucket, charged_milli, status)
VALUES (?, ?, ?, ?, ?, ?, 'holding')""" VALUES (?, ?, ?, ?, ?, ?, 'holding')"""
@@ -132,6 +136,41 @@ class BillingMixin:
" WHERE request_id = ?", (status, request_id)) " WHERE request_id = ?", (status, request_id))
return True return True
# ---------- 预算四档(T-X1 ----------
def budget_mode(self, student_id: int, est_milli: int, ts: float) -> str:
"""预算档位判定:normal / optimize / cheap / block。
以「当日已 spent + 本次预估」占日上限的比例判定(基点 bp,全整数运算,
无浮点边界漂移;黄金用例锁死):
< 8000bp80% -> normal(现行行为)
>= 8000bp80% -> optimize(调用方应降一档模型)
>= 9500bp95% -> cheap(调用方应强制最低档)
> 10000bp100% -> blocktry_hold 的硬拒绝语义兜底)
daily_cap_milli <= 0 视为不限额 -> 恒 normal;学生不存在 -> normal
(try_hold 才是扣费与拒绝的唯一权威,本方法只做档位建议)。
"""
today = _today(ts)
with self._lock, self._connect() as conn:
stu = conn.execute(
"SELECT daily_cap_milli, spent_today_milli, spent_date"
" FROM students WHERE id = ?", (student_id,)).fetchone()
if stu is None:
return "normal"
cap = int(stu["daily_cap_milli"] or 0)
if cap <= 0:
return "normal"
spent = int(stu["spent_today_milli"] or 0) \
if stu["spent_date"] == today else 0
projected = spent + max(0, int(est_milli))
projected_bp = projected * 10000 // cap
if projected_bp > 10000:
return "block"
if projected_bp >= BUDGET_CHEAP_BP:
return "cheap"
if projected_bp >= BUDGET_OPTIMIZE_BP:
return "optimize"
return "normal"
# ---------- 流水查询 ---------- # ---------- 流水查询 ----------
def get_usage(self, request_id: str) -> Optional[Dict[str, Any]]: def get_usage(self, request_id: str) -> Optional[Dict[str, Any]]:
with self._lock, self._connect() as conn: with self._lock, self._connect() as conn:
+63 -8
View File
@@ -267,6 +267,41 @@ def _resolve_entry(pool, model: str, cfg: ProxyConfig) -> Optional[Dict[str, Any
return None return None
def _budget_headers(budget_mode: str) -> Dict[str, str]:
"""预算档位响应头(normal 不发,避免噪音)。"""
return {"X-Budget-Mode": budget_mode} if budget_mode and budget_mode != "normal" else {}
def _downgrade_entry(pool, entry: Dict[str, Any], mode: str) -> Optional[Dict[str, Any]]:
"""预算降档(T-X1):在池内找恰好低一档/最低档的启用条目。
档位序 TIERS = (local, budget, premium)
- optimize:降一档(premium->budgetbudget->local);
- cheap:直落最低档 local
- 已在 local / 找不到该档启用条目 -> None(保持原条目,档位语义照常上报)。
"""
from gateway.model_pool import TIERS
tier = str(entry.get("tier") or "")
if tier not in TIERS:
return None
rank = TIERS.index(tier)
if mode == "cheap":
target = 0
elif mode == "optimize":
target = rank - 1
else:
return None
if target < 0:
return None
want = TIERS[target]
for e in pool.list().get("entries", []):
if (e.get("enabled") and e.get("id") != entry.get("id")
and e.get("tier") == want
and e.get("backend") not in ("mock",) and e.get("base_url")):
return e
return None
_semcache_instances: Dict[str, Any] = {} _semcache_instances: Dict[str, Any] = {}
@@ -312,6 +347,21 @@ async def _run_chat(body: dict, headers: Dict[str, str], ctx: Dict[str, Any],
if isinstance(body.get("stream_options"), dict) else False if isinstance(body.get("stream_options"), dict) else False
is_stream = bool(body.get("stream")) is_stream = bool(body.get("stream"))
# ---- 预算四档(T-X1):接近日上限渐进降档;>100% 仍由 try_hold 硬拒 ----
budget_mode = "normal"
try:
est_probe = _estimate_hold_milli(body, cfg, model)
budget_mode = await asyncio.to_thread(
ledger.budget_mode, ctx["student_id"], est_probe, ts)
except Exception:
budget_mode = "normal"
if budget_mode in ("optimize", "cheap"):
down = _downgrade_entry(pool, entry, budget_mode)
if down is not None:
entry = down
model = str(entry.get("model") or model)
body = {**body, "model": model}
# ---- 缓存分支(T-P6,§7 时序):仅缓存准入(stop+单轮)查询 ---- # ---- 缓存分支(T-P6,§7 时序):仅缓存准入(stop+单轮)查询 ----
cacheable = False cacheable = False
cache = None cache = None
@@ -350,7 +400,8 @@ async def _run_chat(body: dict, headers: Dict[str, str], ctx: Dict[str, Any],
replay(), media_type="text/event-stream", replay(), media_type="text/event-stream",
headers={"Cache-Control": "no-cache", headers={"Cache-Control": "no-cache",
"X-Cache": "HIT", "X-Cache": "HIT",
"X-Request-Id": request_id}) "X-Request-Id": request_id,
**_budget_headers(budget_mode)})
return JSONResponse({ return JSONResponse({
"id": f"chatcmpl-{request_id}", "object": "chat.completion", "id": f"chatcmpl-{request_id}", "object": "chat.completion",
"created": int(time.time()), "model": model, "created": int(time.time()), "model": model,
@@ -360,7 +411,8 @@ async def _run_chat(body: dict, headers: Dict[str, str], ctx: Dict[str, Any],
"finish_reason": "stop"}], "finish_reason": "stop"}],
"usage": {"prompt_tokens": 0, "completion_tokens": 0, "usage": {"prompt_tokens": 0, "completion_tokens": 0,
"total_tokens": 0}, "total_tokens": 0},
}, headers={"X-Cache": "HIT", "X-Request-Id": request_id}) }, headers={"X-Cache": "HIT", "X-Request-Id": request_id,
**_budget_headers(budget_mode)})
raise BalanceError("余额或当日额度不足") raise BalanceError("余额或当日额度不足")
except BalanceError: except BalanceError:
raise raise
@@ -379,11 +431,12 @@ async def _run_chat(body: dict, headers: Dict[str, str], ctx: Dict[str, Any],
if is_stream: if is_stream:
return await _stream_response(body, entry, sink, headers, client_wants_usage, return await _stream_response(body, entry, sink, headers, client_wants_usage,
request_id, ctx, cfg, ledger, model, est, t0, request_id, ctx, cfg, ledger, model, est, t0,
cache=cache, cacheable=cacheable) cache=cache, cacheable=cacheable,
budget_mode=budget_mode)
return await _json_response(body, entry, sink, request_id, ctx, cfg, return await _json_response(body, entry, sink, request_id, ctx, cfg,
ledger, model, est, t0, ledger, model, est, t0,
cache=cache, cacheable=cacheable, cache=cache, cacheable=cacheable,
headers=headers) headers=headers, budget_mode=budget_mode)
except UpstreamAborted as e: except UpstreamAborted as e:
# 流中失败:按已收 usage 结算(无 usage 按字符估算),不缓存(D-P4) # 流中失败:按已收 usage 结算(无 usage 按字符估算),不缓存(D-P4)
usage = sink.get("usage") or _estimate_usage_from_sink(sink) usage = sink.get("usage") or _estimate_usage_from_sink(sink)
@@ -412,7 +465,7 @@ def _estimate_usage_from_sink(sink: Dict[str, Any]) -> Dict[str, int]:
async def _stream_response(body, entry, sink, headers, client_wants_usage, async def _stream_response(body, entry, sink, headers, client_wants_usage,
request_id, ctx, cfg, ledger, model, est, t0, request_id, ctx, cfg, ledger, model, est, t0,
cache=None, cacheable=False): cache=None, cacheable=False, budget_mode: str = "normal"):
usage = {"in_miss": 0, "in_hit": 0, "out": 0} usage = {"in_miss": 0, "in_hit": 0, "out": 0}
async def gen(): async def gen():
@@ -459,12 +512,13 @@ async def _stream_response(body, entry, sink, headers, client_wants_usage,
return StreamingResponse(gen(), media_type="text/event-stream", return StreamingResponse(gen(), media_type="text/event-stream",
headers={"Cache-Control": "no-cache", headers={"Cache-Control": "no-cache",
"X-Request-Id": request_id}) "X-Request-Id": request_id,
**_budget_headers(budget_mode)})
async def _json_response(body, entry, sink, request_id, ctx, cfg, ledger, async def _json_response(body, entry, sink, request_id, ctx, cfg, ledger,
model, est, t0, cache=None, cacheable=False, model, est, t0, cache=None, cacheable=False,
headers=None): headers=None, budget_mode: str = "normal"):
parts = [] parts = []
async for raw_bytes in upstream_stream(body, entry, sink, [entry]): async for raw_bytes in upstream_stream(body, entry, sink, [entry]):
line = raw_bytes.decode("utf-8").strip() line = raw_bytes.decode("utf-8").strip()
@@ -521,4 +575,5 @@ async def _json_response(body, entry, sink, request_id, ctx, cfg, ledger,
"completion_tokens": usage.get("out", 0), "completion_tokens": usage.get("out", 0),
"total_tokens": usage.get("in_miss", 0) + usage.get("in_hit", 0) "total_tokens": usage.get("in_miss", 0) + usage.get("in_hit", 0)
+ usage.get("out", 0)}, + usage.get("out", 0)},
}, headers={"X-Request-Id": request_id}) }, headers={"X-Request-Id": request_id,
**_budget_headers(budget_mode)})
+134
View File
@@ -0,0 +1,134 @@
"""预算四档渐进干预测试(T-X1):budget_mode 黄金边界 + 降档选择器。
黄金用例锁死(整数基点 bp1% = 100bp):
projected < 8000bp -> normal
8000bp <= projected -> optimize
9500bp <= projected -> cheap
projected > 10000bp -> blocktry_hold 硬拒兜底)
"""
import pytest
from gateway.model_pool import PoolStore
from gateway.proxy.billing import BUDGET_CHEAP_BP, BUDGET_OPTIMIZE_BP
from gateway.proxy.ledger import Ledger
from gateway.proxy.ledgerutil import _today
from gateway.proxy.routes import _budget_headers, _downgrade_entry
pytest.importorskip("fastapi")
_TS = 1789874000.0 # 固定时间戳(黄金用例确定性)
def _make_ledger(tmp_path, daily_cap_yuan: float = 10.0) -> Ledger:
led = Ledger.init_db(tmp_path / "budget.sqlite3")
led.upsert_student("张三", "软件2201", balance_yuan=100.0,
daily_cap_yuan=daily_cap_yuan)
return led
def _set_spent(led: Ledger, student_id: int, spent_milli: int,
spent_date: str = "") -> None:
"""直接写当日已花(白盒:budget_mode 的输入口径与 try_hold 一致)。"""
with led._lock, led._connect() as conn:
conn.execute(
"UPDATE students SET spent_today_milli = ?, spent_date = ? WHERE id = ?",
(spent_milli, spent_date, student_id))
def test_threshold_constants_golden():
"""阈值常量黄金锁定:80% / 95%"""
assert BUDGET_OPTIMIZE_BP == 8000
assert BUDGET_CHEAP_BP == 9500
@pytest.mark.parametrize("spent,est,expected", [
(0, 7999, "normal"), # 7999bp
(0, 8000, "optimize"), # 恰好 8000bp(边界含头)
(0, 9499, "optimize"),
(0, 9500, "cheap"), # 恰好 9500bp(边界含头)
(0, 10000, "cheap"), # 恰好达上限 = cheap(不 block
(0, 10001, "block"), # 严格超出才 block
(5000, 3000, "optimize"), # 8000bpspent 与 est 合并计算
(5000, 4500, "cheap"), # 9500bp
(5000, 5001, "block"), # 10001bp
])
def test_budget_mode_golden_boundaries(tmp_path, spent, est, expected):
led = _make_ledger(tmp_path, daily_cap_yuan=10.0) # cap = 10000 毫元
_set_spent(led, 1, spent, spent_date=_today(_TS)) # 记账日 = 判定日(当日口径)
assert led.budget_mode(1, est, ts=_TS) == expected
def test_budget_mode_zero_cap_means_unlimited(tmp_path):
led = _make_ledger(tmp_path, daily_cap_yuan=0.0)
sid = 1
_set_spent(led, sid, 999999, spent_date=_today(_TS))
assert led.budget_mode(sid, 10**9, ts=_TS) == "normal"
def test_budget_mode_spent_resets_by_date(tmp_path):
"""spent_date 非今日 -> 当日已花按 0 计(与 try_hold 口径一致)。"""
led = _make_ledger(tmp_path, daily_cap_yuan=10.0)
_set_spent(led, 1, 9999, spent_date="2000-01-01") # 非今日
assert led.budget_mode(1, 1, ts=1789874000.0) == "normal"
def test_budget_mode_missing_student_normal(tmp_path):
led = Ledger.init_db(tmp_path / "b.sqlite3")
assert led.budget_mode(999, 100, ts=1789874000.0) == "normal"
def _make_pool() -> PoolStore:
import gateway.model_pool as mp
store = PoolStore() # 不落盘(path=None 仅内存)
mp.reset_pool()
store.upsert({"id": "p1", "name": "旗舰", "tier": "premium", "backend": "openai",
"base_url": "https://api.example.com", "model": "big-x",
"enabled": True})
store.upsert({"id": "b1", "name": "实惠", "tier": "budget", "backend": "openai",
"base_url": "https://api.example.com/v2", "model": "mid-y",
"enabled": True})
store.upsert({"id": "l1", "name": "本地", "tier": "local",
"backend": "llama_server", "base_url": "http://127.0.0.1:8901",
"model": "qwen-local", "enabled": True})
return store
def test_downgrade_optimize_one_tier():
pool = _make_pool()
premium = pool.find_by_model("big-x")
down = _downgrade_entry(pool, premium, "optimize")
assert down is not None and down["tier"] == "budget"
def test_downgrade_cheap_to_lowest():
pool = _make_pool()
premium = pool.find_by_model("big-x")
down = _downgrade_entry(pool, premium, "cheap")
assert down is not None and down["tier"] == "local"
def test_downgrade_stops_at_local():
"""已在最低档:cheap/optimize 均不再降(杜绝反向升档)。"""
pool = _make_pool()
local = pool.find_by_model("qwen-local")
assert _downgrade_entry(pool, local, "cheap") is None
assert _downgrade_entry(pool, local, "optimize") is None
def test_downgrade_skips_disabled_and_mock():
pool = _make_pool()
pool.upsert({"id": "b2", "name": "停用", "tier": "budget", "backend": "openai",
"base_url": "https://api.example.com/v3", "model": "mid-z",
"enabled": False})
pool.upsert({"id": "mk", "name": "", "tier": "budget", "backend": "mock",
"model": "mock", "enabled": True})
premium = pool.find_by_model("big-x")
down = _downgrade_entry(pool, premium, "optimize")
assert down is not None and down["id"] == "b1" # 跳过停用与 mock
def test_budget_headers_only_when_abnormal():
assert _budget_headers("normal") == {}
assert _budget_headers("") == {}
assert _budget_headers("optimize") == {"X-Budget-Mode": "optimize"}
assert _budget_headers("cheap") == {"X-Budget-Mode": "cheap"}
+1
View File
@@ -165,3 +165,4 @@ P0 完成后的能力:干净的后端抽象 + 可量化的评测 + 可追溯
| T-G7 | 审计+前端:ReviewQueue 抽样 + tier 指标卡 + 客户端来源显示/一键升级 | ✅ 完成 | T-G7 | | T-G7 | 审计+前端:ReviewQueue 抽样 + tier 指标卡 + 客户端来源显示/一键升级 | ✅ 完成 | T-G7 |
| T-G8 | 实验:E-G1/E-G3 报告;(可选)LoraRemote + E-G2 线性 vs LoRA | ✅ 完成 | T-G8 | | T-G8 | 实验:E-G1/E-G3 报告;(可选)LoraRemote + E-G2 线性 vs LoRA | ✅ 完成 | T-G8 |
| OPT-1 | 分支推进:语义缓存 L2 查找 3.39x(免并集计分+预筛)+ 安全加固(15 高危清零:SSRF/路径穿越/假凭据) | ✅ 完成 | ad3bf41 | | OPT-1 | 分支推进:语义缓存 L2 查找 3.39x(免并集计分+预筛)+ 安全加固(15 高危清零:SSRF/路径穿越/假凭据) | ✅ 完成 | ad3bf41 |
| T-X1 | 预算四档渐进干预(外部采纳 ai-model-router):budget_mode 整数基点判定(80/95/100%+ optimize/cheap 自动降档 + X-Budget-Mode 上报;黄金用例锁边界 | ✅ 完成 | T-X1 |