study-interface / process_qualtrics_data.py
Rachel Kim
debugging new pilot
f2b81d9
Raw
History Blame Contribute Delete
5.58 kB
import pandas as pd
# import anthropic
import openai
import os
from dotenv import load_dotenv
import random
import yaml
from datetime import datetime
from concurrent.futures import ThreadPoolExecutor
load_dotenv()
def get_response(model, messages, timeout=6000):
"""Get response from specified model via OpenRouter."""
try:
client = openai.OpenAI(
base_url="https://openrouter.ai/api/v1",
api_key=os.getenv("OPENROUTER_API_KEY")
)
response = client.chat.completions.create(
model=model,
messages=messages,
timeout=timeout,
# max_tokens=1024
)
return response.choices[0].message.content
except openai.APITimeoutError:
return f"Error: Request timed out after {timeout} seconds"
except Exception as e:
return f"Error: {str(e)}"
def get_randomly_selected_models():
# randomly select two models
config_path = os.path.join(os.path.dirname(__file__), "config.yaml")
with open(config_path, 'r') as file:
data = yaml.safe_load(file)
selected_models = random.sample(data["models"], 2)
return selected_models
def get_prompt_columns():
"""Return mapping of vignette types to prompt columns."""
return {
"news": {
"spec_col": "Please specify the vignette further by choosing a recent news event or historical topic you might want to ask an LLM about. Please write your specification (i.e. recent news event or historical topic) below.",
"prompt_col": "What prompt would you enter to get information about [QID11-ChoiceTextEntryValue]? Please write it below as if you were typing it directly into the LLM."
},
"essay": {
"spec_col": "Please specify the vignette further by choosing an essay topic. Please write your specification (i.e. essay topic) below.",
"prompt_col": 'What prompt would you enter to edit or critique "[QID17-ChoiceTextEntryValue]"? Please write it below as if you were typing it directly into the LLM.'
},
"conflict": {
"spec_col": "Please specify the vignette further by selecting a type of person (e.g., family member, friend, roommate, partner, etc.) and a specific kind of conflict you could imagine having with them. Please describe both in the text box below.",
"prompt_col": "What prompt would you enter in this situation? Please write it below as if you were typing it directly into the LLM."
}
}
def load_survey_data(csv_file):
"""Load survey data from CSV, skipping metadata rows."""
# csv_file = "LLM Leaderboard: Survey 1_March 2, 2026_16.00.csv"
# Skip first row which are import IDs, row 0 is the column headers we need
df = pd.read_csv(csv_file, skiprows=1)
# Filter out any metadata rows (where Prolific ID contains "ImportId")
prolific_col = "What is your Prolific ID?"
df = df[~df[prolific_col].astype(str).str.contains("ImportId", case=False, na=False)]
df = df.reset_index(drop=True)
return df
def extract_response_data(row):
"""Extract prolific_id, prompts, and specifications from survey row."""
prolific_id = row.get("What is your Prolific ID?", "unknown")
# Clean up prolific_id to use as folder name
if pd.isna(prolific_id) or not str(prolific_id).strip():
prolific_id = row.get("Response ID", "unknown_response")
prolific_id = str(prolific_id).strip()
prompt_cols = get_prompt_columns()
responses_data = []
for vignette_type, cols in prompt_cols.items():
spec = row.get(cols["spec_col"], "")
prompt = row.get(cols["prompt_col"], "")
# Only include if both specification and prompt exist
if pd.notna(spec) and pd.notna(prompt) and str(spec).strip() and str(prompt).strip():
responses_data.append({
"type": vignette_type,
"specification": str(spec).strip(),
"prompt": str(prompt).strip()
})
return prolific_id, responses_data
def main(csv_file):
df = load_survey_data(csv_file)
llm_responses = {
"prolific_id": [],
"prompt": [],
"model_a": [],
"model_b": [],
"response_a": [],
"response_b": []
}
for idx, row in df.iterrows():
prolific_id, responses_data = extract_response_data(row)
prompt = responses_data[0]["prompt"]
selected_models = get_randomly_selected_models()
with ThreadPoolExecutor(max_workers=2) as executor:
future_a = executor.submit(get_response, selected_models[0], [{"role": "user", "content": prompt}])
future_b = executor.submit(get_response, selected_models[1], [{"role": "user", "content": prompt}])
response_a = future_a.result()
response_b = future_b.result()
# add everything to the dictionary
llm_responses["prolific_id"].append(prolific_id)
llm_responses["prompt"].append(prompt)
llm_responses["model_a"].append(selected_models[0])
llm_responses["model_b"].append(selected_models[1])
llm_responses["response_a"].append(response_a)
llm_responses["response_b"].append(response_b)
llm_responses_df = pd.DataFrame(llm_responses)
filename = datetime.now().strftime("%Y%m%d_%H%M%S") + "_llm_responses.csv"
llm_responses_df.to_csv(filename)
if __name__ == "__main__":
main("LLM Leaderboard: Survey 1_March 16, 2026_11.57.csv")