import pandas as pd # import anthropic import openai import os from dotenv import load_dotenv import random import yaml from datetime import datetime from concurrent.futures import ThreadPoolExecutor load_dotenv() def get_response(model, messages, timeout=6000): """Get response from specified model via OpenRouter.""" try: client = openai.OpenAI( base_url="https://openrouter.ai/api/v1", api_key=os.getenv("OPENROUTER_API_KEY") ) response = client.chat.completions.create( model=model, messages=messages, timeout=timeout, # max_tokens=1024 ) return response.choices[0].message.content except openai.APITimeoutError: return f"Error: Request timed out after {timeout} seconds" except Exception as e: return f"Error: {str(e)}" def get_randomly_selected_models(): # randomly select two models config_path = os.path.join(os.path.dirname(__file__), "config.yaml") with open(config_path, 'r') as file: data = yaml.safe_load(file) selected_models = random.sample(data["models"], 2) return selected_models def get_prompt_columns(): """Return mapping of vignette types to prompt columns.""" return { "news": { "spec_col": "Please specify the vignette further by choosing a recent news event or historical topic you might want to ask an LLM about. Please write your specification (i.e. recent news event or historical topic) below.", "prompt_col": "What prompt would you enter to get information about [QID11-ChoiceTextEntryValue]? Please write it below as if you were typing it directly into the LLM." }, "essay": { "spec_col": "Please specify the vignette further by choosing an essay topic. Please write your specification (i.e. essay topic) below.", "prompt_col": 'What prompt would you enter to edit or critique "[QID17-ChoiceTextEntryValue]"? Please write it below as if you were typing it directly into the LLM.' }, "conflict": { "spec_col": "Please specify the vignette further by selecting a type of person (e.g., family member, friend, roommate, partner, etc.) and a specific kind of conflict you could imagine having with them. Please describe both in the text box below.", "prompt_col": "What prompt would you enter in this situation? Please write it below as if you were typing it directly into the LLM." } } def load_survey_data(csv_file): """Load survey data from CSV, skipping metadata rows.""" # csv_file = "LLM Leaderboard: Survey 1_March 2, 2026_16.00.csv" # Skip first row which are import IDs, row 0 is the column headers we need df = pd.read_csv(csv_file, skiprows=1) # Filter out any metadata rows (where Prolific ID contains "ImportId") prolific_col = "What is your Prolific ID?" df = df[~df[prolific_col].astype(str).str.contains("ImportId", case=False, na=False)] df = df.reset_index(drop=True) return df def extract_response_data(row): """Extract prolific_id, prompts, and specifications from survey row.""" prolific_id = row.get("What is your Prolific ID?", "unknown") # Clean up prolific_id to use as folder name if pd.isna(prolific_id) or not str(prolific_id).strip(): prolific_id = row.get("Response ID", "unknown_response") prolific_id = str(prolific_id).strip() prompt_cols = get_prompt_columns() responses_data = [] for vignette_type, cols in prompt_cols.items(): spec = row.get(cols["spec_col"], "") prompt = row.get(cols["prompt_col"], "") # Only include if both specification and prompt exist if pd.notna(spec) and pd.notna(prompt) and str(spec).strip() and str(prompt).strip(): responses_data.append({ "type": vignette_type, "specification": str(spec).strip(), "prompt": str(prompt).strip() }) return prolific_id, responses_data def main(csv_file): df = load_survey_data(csv_file) llm_responses = { "prolific_id": [], "prompt": [], "model_a": [], "model_b": [], "response_a": [], "response_b": [] } for idx, row in df.iterrows(): prolific_id, responses_data = extract_response_data(row) prompt = responses_data[0]["prompt"] selected_models = get_randomly_selected_models() with ThreadPoolExecutor(max_workers=2) as executor: future_a = executor.submit(get_response, selected_models[0], [{"role": "user", "content": prompt}]) future_b = executor.submit(get_response, selected_models[1], [{"role": "user", "content": prompt}]) response_a = future_a.result() response_b = future_b.result() # add everything to the dictionary llm_responses["prolific_id"].append(prolific_id) llm_responses["prompt"].append(prompt) llm_responses["model_a"].append(selected_models[0]) llm_responses["model_b"].append(selected_models[1]) llm_responses["response_a"].append(response_a) llm_responses["response_b"].append(response_b) llm_responses_df = pd.DataFrame(llm_responses) filename = datetime.now().strftime("%Y%m%d_%H%M%S") + "_llm_responses.csv" llm_responses_df.to_csv(filename) if __name__ == "__main__": main("LLM Leaderboard: Survey 1_March 16, 2026_11.57.csv")