File size: 2,924 Bytes
2834b30
 
 
b3f9415
2834b30
b3f9415
 
2834b30
 
 
 
 
 
 
 
b3f9415
 
 
 
 
2834b30
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
import os
import json
import bm25s
import yaml
from pathlib import Path
from langchain_core.messages import SystemMessage


def load_config(path="config.yaml"):
    with open(path, "r") as f:
        return yaml.safe_load(f)


def load_prompt(prompt_location: str) -> SystemMessage:
    """Load system prompt from YAML file."""
    with open(prompt_location) as f:
        try:
            prompt = yaml.safe_load(f)["prompt"]
            return SystemMessage(content=prompt)
        except yaml.YAMLError as exc:
            print(exc)
            return SystemMessage(content="You are a helpful assistant.")



def init_bm25_index(corpus_file = "data/metadata.jsonl"):
    """BM25 Index Initialization (Local Corpus)"""
    try:
        if not os.path.exists(corpus_file):
            print(f"Warning: {corpus_file} not found. BM25 will use empty index.")
            return None, [], []
            
        search_texts = []  # question-only — used for BM25 indexing
        corpus_texts = []  # Q+A+Steps — returned for context injection
        corpus_ids = []
        with open(corpus_file, "r") as f:
            for line in f:
                item = json.loads(line)
                question = item.get('Question', '')
                answer = item.get('Final answer', '')
                steps = item.get('Annotator Metadata', {}).get('Steps', '')
                search_texts.append(question)
                parts = [f"Question: {question}"]
                if answer:
                    parts.append(f"Final Answer: {answer}")
                if steps:
                    parts.append(f"Solution Steps: {steps}")
                corpus_texts.append("\n".join(parts))
                corpus_ids.append(item.get('task_id', ''))

        corpus_tokens = bm25s.tokenize(search_texts, stopwords="en", stemmer=None)
        
        retriever_bm25 = bm25s.BM25()
        retriever_bm25.index(corpus_tokens)
        
        print(f"BM25 Index initialized with {len(corpus_texts)} documents.")
        return retriever_bm25, corpus_texts, corpus_ids
    except Exception as e:
        print(f"Error initializing BM25: {e}")
        return None, [], []
    

def reciprocal_rank_fusion(results: list[list[dict]], k=60) -> list[tuple[dict, float]]:
    """
    Fuse multiple ranked lists using Reciprocal Rank Fusion (RRF).
    """
    fused_scores = {}
    
    for rank_list in results:
        for rank, doc in enumerate(rank_list):
            doc_id = doc["metadata"]["task_id"]
            doc_content = doc["content"]
            if doc_id not in fused_scores:
                fused_scores[doc_id] = {"id": doc_id, "content": doc_content, "score": 0.0}
            fused_scores[doc_id]["score"] += 1.0 / (k + rank + 1)
            
    sorted_results = sorted(fused_scores.values(), key=lambda x: x["score"], reverse=True)
    return [(item["id"], item["content"], item["score"]) for item in sorted_results]