code/04-rag/rerank.py

55 lines · 2.2 KB

Code and program output are shown exactly as they ran, so comments and printed output are in Chinese.

"""在混合检索的基础上加一步重排:先粗选出 20 块,再用重排模型逐块打分,重新排序。

在 AI-Course 目录下运行:python code/04-rag/rerank.py
需要先运行过一次 hybrid.py,让改写结果缓存在 rewrites.json 里。
"""
import json
import sys
import time
from pathlib import Path

from sentence_transformers import CrossEncoder

sys.path.insert(0, str(Path(__file__).parent))
from chunking import load_docs, split_by_heading_capped
from hybrid import BM25, rewrite, rrf
from vector_search import VectorIndex, evaluate

HERE = Path(__file__).parent
chunks = [(file, c) for file, text in load_docs().items() for c in split_by_heading_capped(text)]
questions = [json.loads(line) for line in (HERE / "eval_qa.jsonl").read_text().splitlines()]

bm25 = BM25(chunks)
vec = VectorIndex("intfloat/multilingual-e5-small")
vec.build(chunks)
# 重排模型:同时读问题和文档块,直接判断两者有多相关。比向量检索准,但也慢得多
reranker = CrossEncoder("BAAI/bge-reranker-base", max_length=512)


def candidates(question, k=20):
    # 和 hybrid.py 一样:两路各取 20 个,融合后取前 k 个
    q = rewrite(question)
    return rrf([vec.search(q, 20), bm25.search(q, 20)], k)


def reranked(question, k=5, use_rewrite=False):
    pool = candidates(question, 20)
    query = rewrite(question) if use_rewrite else question
    scores = reranker.predict([(query, text) for _, _, text in pool])
    order = sorted(range(len(pool)), key=lambda i: -scores[i])
    return [(float(scores[i]), pool[i][1], pool[i][2]) for i in order[:k]]


start = time.time()
methods = {
    "混合 RRF(改写后)": lambda q, k: candidates(q, k),
    "混合 + 重排(用原中文问题)": lambda q, k: reranked(q, k, use_rewrite=False),
    "混合 + 重排(用改写后的问题)": lambda q, k: reranked(q, k, use_rewrite=True),
}
for name, search in methods.items():
    t = time.time()
    h1, h3, h5, mrr, misses = evaluate(search, questions)
    print(f"{name:18s} 第 1 名 {h1:4.0%}  前 3 名 {h3:4.0%}  前 5 名 {h5:4.0%}  MRR {mrr:.3f}  (20 题用时 {time.time() - t:.1f} 秒)")
    for qa, (_, file, text) in misses:
        print(f"    没找到:{qa['question']}(应在 {qa['file']})→ 第 1 名 {file}")