Files
projectAIpopular/tests/test_cache.py
T
tzt b2fa8c3c81 feat(v2): 架构与算法优化——语义缓存 2.37x、拓扑排序 O(V+E)、分类器确定性决胜
算法:
- RouterCache:语义条目写入时预计算向量范数、语义查找单遍完成(消除命中后二次 O(N) 查找)、
  相似度=1.0 提前终止;微基准(3000 条目×200 查询):3986ms -> 1685ms,2.37x
- TaskGraph.topo_order:O(V²logV) 重排序/成员扫描 -> 邻接表+deque 的 O(V+E) Kahn,
  输出顺序契约不变(初始就绪层按插入序、循环依赖按插入序兜底、未知依赖忽略)
- RuleClassifier:同分决胜按领域名字典序(与规则表排列无关),次高分 O(n) 扫描

工程卫生:
- .mimosa/(扫描器工作目录)加入 .gitignore 并移出索引
- test_review 抽样测试改用内联确定性 LCG,消除 2 个低危(不安全随机数)

测试:新增 11 项(topo 契约 6 + 缓存回归 3 + 分类器 2)
pytest 230 passed(基线 219 全绿 + 11)
2026-09-18 08:35:36 +08:00

79 lines
2.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
from router_system.cache import RouterCache
def test_semantic_lookup_after_many_entries():
"""多条目下语义命中正确(范数预计算 + 单遍扫描的回归)。"""
c = RouterCache(similarity_threshold=0.5)
for i in range(50):
c.put(f"完全不相关的查询主题编号{i}关于烹饪的意见", {"response": f"r{i}"})
c.put("用 Python 实现快速排序函数", {"response": "code-answer"})
level, got = c.get("用 Python 实现快速排序的函数写法") # 相似但不完全相同
assert level in ("semantic", "exact")
assert got["response"] == "code-answer"
def test_promotion_clears_semantic_state():
"""提升为精确缓存后,语义列表与范数索引无残留。"""
c = RouterCache(promote_frequency=2)
c.put("查询甲", {"response": "a"})
first = c.get("查询甲") # 相似度=1.0 计 exacthits 达阈值即提升
assert first is not None and first[0] == "exact"
second = c.get("查询甲")
assert second is not None and second[0] == "exact"
assert c.stats()["exact_size"] == 1
assert c.stats()["semantic_size"] == 0
assert len(c._sem_norms) == 0
def test_semantic_eviction_clears_norms():
"""语义缓存满员淘汰最旧条目时,向量与范数索引同步清理。"""
c = RouterCache(max_semantic=2)
c.put("查询一", {"response": "1"})
c.put("查询二", {"response": "2"})
c.put("查询三", {"response": "3"}) # 淘汰查询一
assert len(c._semantic) == 2
assert len(c._sem_vecs) == 2
assert len(c._sem_norms) == 2
assert c.get("查询一") is None
def test_exact_hit():
c = RouterCache()
result = {"response": "hello", "domain": "general"}
assert c.get("query") is None
c.put("query", result)
level, got = c.get("query")
assert level == "exact"
assert got["response"] == "hello"
def test_semantic_hit():
c = RouterCache(semantic_enabled=True, similarity_threshold=0.5)
c.put("用python写一个快速排序", {"response": "code", "domain": "code"})
# 相似改写查询命中 L2 语义缓存
hit = c.get("用python写一个快速排序算法")
assert hit is not None
assert hit[0] == "semantic"
def test_promote_to_exact():
c = RouterCache(promote_frequency=3)
result = {"response": "x", "domain": "general"}
c.put("query", result)
# 语义命中 3 次后提升为精确缓存
for _ in range(3):
hit = c.get("query")
assert hit is not None
assert c.stats()["exact_size"] == 1
def test_stats():
c = RouterCache()
c.put("q", {"response": "r"})
c.get("q")
c.get("q")
c.get("miss")
s = c.stats()
assert s["exact_hits"] == 2
assert s["misses"] == 1