Spaces:
Sleeping
Sleeping
| import pandas as pd | |
| # import anthropic | |
| import openai | |
| import os | |
| from dotenv import load_dotenv | |
| import random | |
| import yaml | |
| from datetime import datetime | |
| from concurrent.futures import ThreadPoolExecutor | |
| load_dotenv() | |
| def get_response(model, messages, timeout=6000): | |
| """Get response from specified model via OpenRouter.""" | |
| try: | |
| client = openai.OpenAI( | |
| base_url="https://openrouter.ai/api/v1", | |
| api_key=os.getenv("OPENROUTER_API_KEY") | |
| ) | |
| response = client.chat.completions.create( | |
| model=model, | |
| messages=messages, | |
| timeout=timeout, | |
| # max_tokens=1024 | |
| ) | |
| return response.choices[0].message.content | |
| except openai.APITimeoutError: | |
| return f"Error: Request timed out after {timeout} seconds" | |
| except Exception as e: | |
| return f"Error: {str(e)}" | |
| def get_randomly_selected_models(): | |
| # randomly select two models | |
| config_path = os.path.join(os.path.dirname(__file__), "config.yaml") | |
| with open(config_path, 'r') as file: | |
| data = yaml.safe_load(file) | |
| selected_models = random.sample(data["models"], 2) | |
| return selected_models | |
| def get_prompt_columns(): | |
| """Return mapping of vignette types to prompt columns.""" | |
| return { | |
| "news": { | |
| "spec_col": "Please specify the vignette further by choosing a recent news event or historical topic you might want to ask an LLM about. Please write your specification (i.e. recent news event or historical topic) below.", | |
| "prompt_col": "What prompt would you enter to get information about [QID11-ChoiceTextEntryValue]? Please write it below as if you were typing it directly into the LLM." | |
| }, | |
| "essay": { | |
| "spec_col": "Please specify the vignette further by choosing an essay topic. Please write your specification (i.e. essay topic) below.", | |
| "prompt_col": 'What prompt would you enter to edit or critique "[QID17-ChoiceTextEntryValue]"? Please write it below as if you were typing it directly into the LLM.' | |
| }, | |
| "conflict": { | |
| "spec_col": "Please specify the vignette further by selecting a type of person (e.g., family member, friend, roommate, partner, etc.) and a specific kind of conflict you could imagine having with them. Please describe both in the text box below.", | |
| "prompt_col": "What prompt would you enter in this situation? Please write it below as if you were typing it directly into the LLM." | |
| } | |
| } | |
| def load_survey_data(csv_file): | |
| """Load survey data from CSV, skipping metadata rows.""" | |
| # csv_file = "LLM Leaderboard: Survey 1_March 2, 2026_16.00.csv" | |
| # Skip first row which are import IDs, row 0 is the column headers we need | |
| df = pd.read_csv(csv_file, skiprows=1) | |
| # Filter out any metadata rows (where Prolific ID contains "ImportId") | |
| prolific_col = "What is your Prolific ID?" | |
| df = df[~df[prolific_col].astype(str).str.contains("ImportId", case=False, na=False)] | |
| df = df.reset_index(drop=True) | |
| return df | |
| def extract_response_data(row): | |
| """Extract prolific_id, prompts, and specifications from survey row.""" | |
| prolific_id = row.get("What is your Prolific ID?", "unknown") | |
| # Clean up prolific_id to use as folder name | |
| if pd.isna(prolific_id) or not str(prolific_id).strip(): | |
| prolific_id = row.get("Response ID", "unknown_response") | |
| prolific_id = str(prolific_id).strip() | |
| prompt_cols = get_prompt_columns() | |
| responses_data = [] | |
| for vignette_type, cols in prompt_cols.items(): | |
| spec = row.get(cols["spec_col"], "") | |
| prompt = row.get(cols["prompt_col"], "") | |
| # Only include if both specification and prompt exist | |
| if pd.notna(spec) and pd.notna(prompt) and str(spec).strip() and str(prompt).strip(): | |
| responses_data.append({ | |
| "type": vignette_type, | |
| "specification": str(spec).strip(), | |
| "prompt": str(prompt).strip() | |
| }) | |
| return prolific_id, responses_data | |
| def main(csv_file): | |
| df = load_survey_data(csv_file) | |
| llm_responses = { | |
| "prolific_id": [], | |
| "prompt": [], | |
| "model_a": [], | |
| "model_b": [], | |
| "response_a": [], | |
| "response_b": [] | |
| } | |
| for idx, row in df.iterrows(): | |
| prolific_id, responses_data = extract_response_data(row) | |
| prompt = responses_data[0]["prompt"] | |
| selected_models = get_randomly_selected_models() | |
| with ThreadPoolExecutor(max_workers=2) as executor: | |
| future_a = executor.submit(get_response, selected_models[0], [{"role": "user", "content": prompt}]) | |
| future_b = executor.submit(get_response, selected_models[1], [{"role": "user", "content": prompt}]) | |
| response_a = future_a.result() | |
| response_b = future_b.result() | |
| # add everything to the dictionary | |
| llm_responses["prolific_id"].append(prolific_id) | |
| llm_responses["prompt"].append(prompt) | |
| llm_responses["model_a"].append(selected_models[0]) | |
| llm_responses["model_b"].append(selected_models[1]) | |
| llm_responses["response_a"].append(response_a) | |
| llm_responses["response_b"].append(response_b) | |
| llm_responses_df = pd.DataFrame(llm_responses) | |
| filename = datetime.now().strftime("%Y%m%d_%H%M%S") + "_llm_responses.csv" | |
| llm_responses_df.to_csv(filename) | |
| if __name__ == "__main__": | |
| main("LLM Leaderboard: Survey 1_March 16, 2026_11.57.csv") |