Spaces:
Sleeping
Sleeping
File size: 5,583 Bytes
0793785 f2b81d9 0793785 0bb1cb9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 | import pandas as pd
# import anthropic
import openai
import os
from dotenv import load_dotenv
import random
import yaml
from datetime import datetime
from concurrent.futures import ThreadPoolExecutor
load_dotenv()
def get_response(model, messages, timeout=6000):
"""Get response from specified model via OpenRouter."""
try:
client = openai.OpenAI(
base_url="https://openrouter.ai/api/v1",
api_key=os.getenv("OPENROUTER_API_KEY")
)
response = client.chat.completions.create(
model=model,
messages=messages,
timeout=timeout,
# max_tokens=1024
)
return response.choices[0].message.content
except openai.APITimeoutError:
return f"Error: Request timed out after {timeout} seconds"
except Exception as e:
return f"Error: {str(e)}"
def get_randomly_selected_models():
# randomly select two models
config_path = os.path.join(os.path.dirname(__file__), "config.yaml")
with open(config_path, 'r') as file:
data = yaml.safe_load(file)
selected_models = random.sample(data["models"], 2)
return selected_models
def get_prompt_columns():
"""Return mapping of vignette types to prompt columns."""
return {
"news": {
"spec_col": "Please specify the vignette further by choosing a recent news event or historical topic you might want to ask an LLM about. Please write your specification (i.e. recent news event or historical topic) below.",
"prompt_col": "What prompt would you enter to get information about [QID11-ChoiceTextEntryValue]? Please write it below as if you were typing it directly into the LLM."
},
"essay": {
"spec_col": "Please specify the vignette further by choosing an essay topic. Please write your specification (i.e. essay topic) below.",
"prompt_col": 'What prompt would you enter to edit or critique "[QID17-ChoiceTextEntryValue]"? Please write it below as if you were typing it directly into the LLM.'
},
"conflict": {
"spec_col": "Please specify the vignette further by selecting a type of person (e.g., family member, friend, roommate, partner, etc.) and a specific kind of conflict you could imagine having with them. Please describe both in the text box below.",
"prompt_col": "What prompt would you enter in this situation? Please write it below as if you were typing it directly into the LLM."
}
}
def load_survey_data(csv_file):
"""Load survey data from CSV, skipping metadata rows."""
# csv_file = "LLM Leaderboard: Survey 1_March 2, 2026_16.00.csv"
# Skip first row which are import IDs, row 0 is the column headers we need
df = pd.read_csv(csv_file, skiprows=1)
# Filter out any metadata rows (where Prolific ID contains "ImportId")
prolific_col = "What is your Prolific ID?"
df = df[~df[prolific_col].astype(str).str.contains("ImportId", case=False, na=False)]
df = df.reset_index(drop=True)
return df
def extract_response_data(row):
"""Extract prolific_id, prompts, and specifications from survey row."""
prolific_id = row.get("What is your Prolific ID?", "unknown")
# Clean up prolific_id to use as folder name
if pd.isna(prolific_id) or not str(prolific_id).strip():
prolific_id = row.get("Response ID", "unknown_response")
prolific_id = str(prolific_id).strip()
prompt_cols = get_prompt_columns()
responses_data = []
for vignette_type, cols in prompt_cols.items():
spec = row.get(cols["spec_col"], "")
prompt = row.get(cols["prompt_col"], "")
# Only include if both specification and prompt exist
if pd.notna(spec) and pd.notna(prompt) and str(spec).strip() and str(prompt).strip():
responses_data.append({
"type": vignette_type,
"specification": str(spec).strip(),
"prompt": str(prompt).strip()
})
return prolific_id, responses_data
def main(csv_file):
df = load_survey_data(csv_file)
llm_responses = {
"prolific_id": [],
"prompt": [],
"model_a": [],
"model_b": [],
"response_a": [],
"response_b": []
}
for idx, row in df.iterrows():
prolific_id, responses_data = extract_response_data(row)
prompt = responses_data[0]["prompt"]
selected_models = get_randomly_selected_models()
with ThreadPoolExecutor(max_workers=2) as executor:
future_a = executor.submit(get_response, selected_models[0], [{"role": "user", "content": prompt}])
future_b = executor.submit(get_response, selected_models[1], [{"role": "user", "content": prompt}])
response_a = future_a.result()
response_b = future_b.result()
# add everything to the dictionary
llm_responses["prolific_id"].append(prolific_id)
llm_responses["prompt"].append(prompt)
llm_responses["model_a"].append(selected_models[0])
llm_responses["model_b"].append(selected_models[1])
llm_responses["response_a"].append(response_a)
llm_responses["response_b"].append(response_b)
llm_responses_df = pd.DataFrame(llm_responses)
filename = datetime.now().strftime("%Y%m%d_%H%M%S") + "_llm_responses.csv"
llm_responses_df.to_csv(filename)
if __name__ == "__main__":
main("LLM Leaderboard: Survey 1_March 16, 2026_11.57.csv") |