code/04-rag/rerank.py
55 lines · 2.2 KBCode and program output are shown exactly as they ran, so comments and printed output are in Chinese.
"""在混合检索的基础上加一步重排:先粗选出 20 块,再用重排模型逐块打分,重新排序。
在 AI-Course 目录下运行:python code/04-rag/rerank.py
需要先运行过一次 hybrid.py,让改写结果缓存在 rewrites.json 里。
"""
import json
import sys
import time
from pathlib import Path
from sentence_transformers import CrossEncoder
sys.path.insert(0, str(Path(__file__).parent))
from chunking import load_docs, split_by_heading_capped
from hybrid import BM25, rewrite, rrf
from vector_search import VectorIndex, evaluate
HERE = Path(__file__).parent
chunks = [(file, c) for file, text in load_docs().items() for c in split_by_heading_capped(text)]
questions = [json.loads(line) for line in (HERE / "eval_qa.jsonl").read_text().splitlines()]
bm25 = BM25(chunks)
vec = VectorIndex("intfloat/multilingual-e5-small")
vec.build(chunks)
# 重排模型:同时读问题和文档块,直接判断两者有多相关。比向量检索准,但也慢得多
reranker = CrossEncoder("BAAI/bge-reranker-base", max_length=512)
def candidates(question, k=20):
# 和 hybrid.py 一样:两路各取 20 个,融合后取前 k 个
q = rewrite(question)
return rrf([vec.search(q, 20), bm25.search(q, 20)], k)
def reranked(question, k=5, use_rewrite=False):
pool = candidates(question, 20)
query = rewrite(question) if use_rewrite else question
scores = reranker.predict([(query, text) for _, _, text in pool])
order = sorted(range(len(pool)), key=lambda i: -scores[i])
return [(float(scores[i]), pool[i][1], pool[i][2]) for i in order[:k]]
start = time.time()
methods = {
"混合 RRF(改写后)": lambda q, k: candidates(q, k),
"混合 + 重排(用原中文问题)": lambda q, k: reranked(q, k, use_rewrite=False),
"混合 + 重排(用改写后的问题)": lambda q, k: reranked(q, k, use_rewrite=True),
}
for name, search in methods.items():
t = time.time()
h1, h3, h5, mrr, misses = evaluate(search, questions)
print(f"{name:18s} 第 1 名 {h1:4.0%} 前 3 名 {h3:4.0%} 前 5 名 {h5:4.0%} MRR {mrr:.3f} (20 题用时 {time.time() - t:.1f} 秒)")
for qa, (_, file, text) in misses:
print(f" 没找到:{qa['question']}(应在 {qa['file']})→ 第 1 名 {file}")