laura.wagner commited on
Commit
eb286ec
·
1 Parent(s): b5829da

add llm consensus analysis

Browse files
jupyter_notebooks/GEMMA_3.ipynb CHANGED
The diff for this file is too large to render. See raw diff
 
jupyter_notebooks/MISTRAL.ipynb CHANGED
The diff for this file is too large to render. See raw diff
 
jupyter_notebooks/QWEN.ipynb CHANGED
The diff for this file is too large to render. See raw diff
 
jupyter_notebooks/Section_2-3-4_compare-models.ipynb CHANGED
@@ -1,16 +1,25 @@
1
  {
2
  "cells": [
 
 
 
 
 
 
 
 
 
3
  {
4
  "cell_type": "code",
5
- "execution_count": 24,
6
  "id": "6cbaef9d-3058-4a59-a8ee-32fcc2062ed6",
7
  "metadata": {
8
  "execution": {
9
- "iopub.execute_input": "2025-12-07T15:53:19.934286Z",
10
- "iopub.status.busy": "2025-12-07T15:53:19.934073Z",
11
- "iopub.status.idle": "2025-12-07T15:53:37.185333Z",
12
- "shell.execute_reply": "2025-12-07T15:53:37.184777Z",
13
- "shell.execute_reply.started": "2025-12-07T15:53:19.934269Z"
14
  }
15
  },
16
  "outputs": [
@@ -18,71 +27,41 @@
18
  "name": "stdout",
19
  "output_type": "stream",
20
  "text": [
21
- "Processing: gemma_local_annotated_POI.csv\n"
22
- ]
23
- },
24
- {
25
- "name": "stderr",
26
- "output_type": "stream",
27
- "text": [
28
- "/tmp/ipykernel_1771575/3158489896.py:277: DtypeWarning: Columns (52,53,54,55,56) have mixed types. Specify dtype option on import or set low_memory=False.\n",
29
- " df = pd.read_csv(input_path)\n"
30
- ]
31
- },
32
- {
33
- "name": "stdout",
34
- "output_type": "stream",
35
- "text": [
36
  " - Found and standardized 'country' column\n",
37
  " - Updated 'name' column for 4 fictional entries\n",
38
  " - Updated 'real_name' column for 4 fictional entries\n",
39
  " - Updated 'full_name' column for 4 fictional entries\n",
40
- " - Changes: 8458 standardized, 4 fictional→Unknown\n",
41
- " - Unique countries after standardization: ['Afghanistan', 'Albania', 'Algeria', 'American Samoa', 'Angola', 'Argentina', 'Armenia', 'Australia', 'Austria', 'Azerbaijan', 'Bahamas', 'Bangladesh', 'Barbados', 'Belarus', 'Belgium', 'Benin', 'Bolivia', 'Brazil', 'Bulgaria', 'Cambodia', 'Canada', 'Central African Republic', 'Chile', 'China', 'Colombia', 'Costa Rica', 'Croatia', 'Cuba', 'Cyprus', 'Czechia', 'Denmark', 'Dominican Republic', 'Egypt', 'El Salvador', 'Estonia', 'Ethiopia', 'Finland', 'France', 'French Polynesia', 'Georgia', 'Germany', 'Ghana', 'Greece', 'Guatemala', 'Guyana', 'Hong Kong', 'Hungary', 'Iceland', 'India', 'Indonesia', 'Iran', 'Iraq', 'Ireland', 'Israel', 'Italy', 'Jamaica', 'Japan', 'Jordan', 'Kazakhstan', 'Kenya', 'Kosovo', 'Kuwait', 'Kyrgyzstan', 'Latvia', 'Lebanon', 'Lithuania', 'Macau', 'Malaysia', 'Malta', 'Mexico', 'Moldova', 'Monaco', 'Mongolia', 'Morocco', 'Myanmar', 'Namibia', 'Nepal', 'Netherlands', 'New Zealand', 'Nicaragua', 'Nigeria', 'North Korea', 'North Macedonia', 'Norway', 'Pakistan', 'Palestine', 'Paraguay', 'Peru', 'Philippines', 'Poland', 'Portugal', 'Puerto Rico', 'Republic of the Congo', 'Romania', 'Russia', 'Samoa', 'Saudi Arabia', 'Senegal', 'Serbia', 'Singapore', 'Slovakia', 'Slovenia', 'Somalia', 'South Africa', 'South Korea', 'Spain', 'Sri Lanka', 'Sudan', 'Sweden', 'Switzerland', 'Syria', 'Taiwan', 'Tanzania', 'Thailand', 'Trinidad and Tobago', 'Tunisia', 'Türkiye', 'UK', 'USA', 'Uganda', 'Ukraine', 'Unknown', 'Uruguay', 'Uzbekistan', 'Venezuela', 'Vietnam', 'Zambia', 'Zimbabwe']\n",
42
  " - Saved to: gemma_standardized_country.csv\n",
43
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/gemma_standardized_country.csv\n",
44
  "\n",
45
- "Processing: mistral_local_annotated_POI.csv\n"
46
- ]
47
- },
48
- {
49
- "name": "stderr",
50
- "output_type": "stream",
51
- "text": [
52
- "/tmp/ipykernel_1771575/3158489896.py:247: DtypeWarning: Columns (52,53,54,55,56) have mixed types. Specify dtype option on import or set low_memory=False.\n",
53
- " df = pd.read_csv(input_path)\n"
54
- ]
55
- },
56
- {
57
- "name": "stdout",
58
- "output_type": "stream",
59
- "text": [
60
- " - Found and standardized 'country' column\n",
61
  " - Updated 'name' column for fictional entries\n",
62
  " - Saved to: mistral_standardized_country.csv\n",
63
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/mistral_standardized_country.csv\n",
64
  "\n",
65
- "Processing: qwen_local_annotated_POI.csv\n"
66
- ]
67
- },
68
- {
69
- "name": "stderr",
70
- "output_type": "stream",
71
- "text": [
72
- "/tmp/ipykernel_1771575/3158489896.py:277: DtypeWarning: Columns (53,54,55,56,57) have mixed types. Specify dtype option on import or set low_memory=False.\n",
73
- " df = pd.read_csv(input_path)\n"
74
- ]
75
- },
76
- {
77
- "name": "stdout",
78
- "output_type": "stream",
79
- "text": [
80
  " - Found and standardized 'country' column\n",
81
- " - Updated 'name' column for 13 fictional entries\n",
82
- " - Updated 'real_name' column for 13 fictional entries\n",
83
- " - Updated 'full_name' column for 13 fictional entries\n",
84
- " - Changes: 30873 standardized, 13 fictional→Unknown\n",
85
- " - Unique countries after standardization: ['Afghanistan', 'Albania', 'Algeria', 'American Samoa', 'Argentina', 'Armenia', 'Australia', 'Austria', 'Azerbaijan', 'Bangladesh', 'Barbados', 'Belarus', 'Belgium', 'Benin', 'Bolivia', 'Bosnia and Herzegovina', 'Brazil', 'Bulgaria', 'Canada', 'Central African Republic', 'Chile', 'China', 'Colombia', 'Costa Rica', 'Croatia', 'Cuba', 'Czechia', 'Denmark', 'Discworld', 'Dominican Republic', 'Ecuador', 'Egypt', 'El Salvador', 'Estonia', 'Finland', 'France', 'Georgia', 'Germany', 'Ghana', 'Greece', 'Guatemala', 'Haiti', 'Hong Kong', 'Hungary', 'Iceland', 'India', 'Indonesia', 'Iran', 'Iraq', 'Ireland', 'Israel', 'Italy', 'Jamaica', 'Japan', 'Jordan', 'Kazakhstan', 'Kenya', 'Kuwait', 'Kyrgyzstan', 'Laos', 'Latvia', 'Lebanon', 'Lithuania', 'Malaysia', 'Malta', 'Mexico', 'Moldova', 'Monaco', 'Morocco', 'Myanmar', 'Nepal', 'Netherlands', 'New Zealand', 'Nicaragua', 'Nigeria', 'North Korea', 'North Macedonia', 'Norway', 'Pakistan', 'Palestine', 'Papua New Guinea', 'Paraguay', 'Peru', 'Philippines', 'Poland', 'Portugal', 'Puerto Rico', 'Romania', 'Rome', 'Russia', 'Samoa', 'Saudi Arabia', 'Senegal', 'Serbia', 'Singapore', 'Slovakia', 'Slovenia', 'South Africa', 'South Korea', 'South Sudan', 'Spain', 'Sri Lanka', 'Sudan', 'Sweden', 'Switzerland', 'Syria', 'Taiwan', 'Thailand', 'Tunisia', 'Türkiye', 'UK', 'USA', 'Uganda', 'Ukraine', 'United Arab Emirates', 'Unknown', 'Uruguay', 'Uzbekistan', 'Vatican City', 'Venezuela', 'Vietnam', 'Zimbabwe']\n",
86
  " - Saved to: qwen_standardized_country.csv\n",
87
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/qwen_standardized_country.csv\n",
88
  "\n",
@@ -93,44 +72,44 @@
93
  "GEMMA:\n",
94
  " Total rows: 50861\n",
95
  " Top 10 countries:\n",
96
- " - USA: 18879\n",
97
- " - Unknown: 5639\n",
98
- " - Japan: 3911\n",
99
- " - UK: 2792\n",
100
- " - South Korea: 2058\n",
101
- " - China: 1721\n",
102
- " - Canada: 1118\n",
103
- " - India: 1081\n",
104
- " - Russia: 973\n",
105
- " - France: 782\n",
106
  "\n",
107
  "MISTRAL:\n",
108
  " Total rows: 50861\n",
109
  " Top 10 countries:\n",
110
- " - Unknown: 10787\n",
111
- " - USA: 6622\n",
112
- " - Japan: 2498\n",
113
- " - South Korea: 1650\n",
114
- " - UK: 1609\n",
115
- " - China: 836\n",
116
- " - Canada: 542\n",
117
- " - India: 405\n",
118
- " - Russia: 358\n",
119
- " - Australia: 336\n",
120
  "\n",
121
  "QWEN:\n",
122
  " Total rows: 50861\n",
123
  " Top 10 countries:\n",
124
- " - USA: 19841\n",
125
- " - Japan: 3924\n",
126
- " - UK: 2787\n",
127
- " - South Korea: 2178\n",
128
- " - Unknown: 2093\n",
129
- " - China: 2046\n",
130
- " - India: 932\n",
131
- " - Russia: 840\n",
132
- " - Canada: 730\n",
133
- " - France: 656\n",
134
  "\n",
135
  "============================================================\n",
136
  "Processing complete!\n",
@@ -139,16 +118,25 @@
139
  " - gemma_standardized_country.csv\n",
140
  " - mistral_standardized_country.csv\n",
141
  " - qwen_standardized_country.csv\n",
 
 
 
 
 
 
 
 
 
 
 
142
  "============================================================\n"
143
  ]
144
  }
145
  ],
146
  "source": [
147
- "# Comprehensive Country Name Standardization with Fictional Place Detection\n",
148
- "# Saves files as modelname_standardized_country.csv\n",
149
- "\n",
150
  "import pandas as pd\n",
151
  "from pathlib import Path\n",
 
152
  "\n",
153
  "# Define fictional places, companies, and invalid entries\n",
154
  "FICTIONAL_PLACES = {\n",
@@ -295,6 +283,53 @@
295
  " 'trinidad': 'Trinidad and Tobago',\n",
296
  "}\n",
297
  "\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
298
  "def is_fictional_or_invalid(country_str):\n",
299
  " \"\"\"\n",
300
  " Check if a country string is fictional, a company, or historically invalid.\n",
@@ -327,9 +362,13 @@
327
  " \n",
328
  " return False\n",
329
  "\n",
330
- "def standardize_country(country_value):\n",
331
  " \"\"\"\n",
332
  " Standardize a single country name based on the mapping.\n",
 
 
 
 
333
  " \"\"\"\n",
334
  " if pd.isna(country_value):\n",
335
  " return country_value\n",
@@ -337,9 +376,16 @@
337
  " # Convert to string and strip whitespace\n",
338
  " country_str = str(country_value).strip()\n",
339
  " \n",
 
 
 
 
 
 
 
340
  " # Return if empty or already 'Unknown'\n",
341
  " if not country_str or country_str == 'Unknown':\n",
342
- " return country_str\n",
343
  " \n",
344
  " # Check if it's fictional or invalid\n",
345
  " if is_fictional_or_invalid(country_str):\n",
@@ -381,51 +427,82 @@
381
  " input_path = Path(input_file)\n",
382
  " output_path = Path(output_file)\n",
383
  " \n",
 
 
 
384
  " print(f\"Processing: {input_path.name}\")\n",
 
 
385
  " \n",
386
  " # Track changes\n",
387
- " changes_made = {'standardized': 0, 'fictional_to_unknown': 0}\n",
388
  " \n",
389
  " # For mistral.csv which might have no header, we need special handling\n",
390
- " if 'mistral' in input_path.name.lower():\n",
391
  " # Try to read normally first\n",
392
  " try:\n",
393
  " df = pd.read_csv(input_path)\n",
394
  " # Check if 'country' column exists\n",
395
  " if 'country' in df.columns:\n",
396
  " original_values = df['country'].copy()\n",
397
- " df['country'] = df['country'].apply(standardize_country)\n",
 
 
 
 
398
  " \n",
399
  " # Count changes\n",
400
  " changes_made['standardized'] = (original_values != df['country']).sum()\n",
401
  " changes_made['fictional_to_unknown'] = ((df['country'] == 'Unknown') & (original_values != 'Unknown')).sum()\n",
402
  " \n",
403
- " print(f\" - Found and standardized 'country' column\")\n",
 
404
  " \n",
405
  " # Also update 'name' column for fictional entries\n",
406
  " if 'name' in df.columns:\n",
407
  " fictional_mask = df['country'] == 'Unknown'\n",
408
  " df.loc[fictional_mask & (original_values != 'Unknown'), 'name'] = 'Unknown'\n",
409
  " print(f\" - Updated 'name' column for fictional entries\")\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
410
  " else:\n",
411
  " # If no country column, assume last column\n",
412
  " last_col = df.columns[-1]\n",
413
- " df[last_col] = df[last_col].apply(standardize_country)\n",
414
  " print(f\" - Standardized column '{last_col}' (assumed to be country)\")\n",
415
  " except:\n",
416
  " # If normal reading fails, try without header\n",
417
  " df = pd.read_csv(input_path, header=None)\n",
418
  " last_col = df.columns[-1]\n",
419
- " df[last_col] = df[last_col].apply(standardize_country)\n",
420
  " print(f\" - Standardized column {last_col} (assumed to be country, no header)\")\n",
421
  " else:\n",
422
- " # Normal CSV with header\n",
423
  " df = pd.read_csv(input_path)\n",
424
  " \n",
425
  " # Check if 'country' column exists\n",
426
  " if 'country' in df.columns:\n",
427
  " original_values = df['country'].copy()\n",
428
- " df['country'] = df['country'].apply(standardize_country)\n",
429
  " \n",
430
  " # Count changes\n",
431
  " changes_made['standardized'] = (original_values != df['country']).sum()\n",
@@ -471,6 +548,17 @@
471
  "# Create a results dictionary to store processed dataframes\n",
472
  "results = {}\n",
473
  "\n",
 
 
 
 
 
 
 
 
 
 
 
474
  "for input_path in input_files:\n",
475
  " # Convert to Path object if it isn't already\n",
476
  " input_path = Path(input_path)\n",
@@ -518,6 +606,15 @@
518
  "print(\"\\nOutput files created:\")\n",
519
  "for model_name in results.keys():\n",
520
  " print(f\" - {model_name}_standardized_country.csv\")\n",
 
 
 
 
 
 
 
 
 
521
  "print(\"=\" * 60)"
522
  ]
523
  },
@@ -531,15 +628,15 @@
531
  },
532
  {
533
  "cell_type": "code",
534
- "execution_count": 25,
535
  "id": "7f2fe3fb-5863-45ec-9df3-9fa64751936b",
536
  "metadata": {
537
  "execution": {
538
- "iopub.execute_input": "2025-12-07T15:54:25.968287Z",
539
- "iopub.status.busy": "2025-12-07T15:54:25.968073Z",
540
- "iopub.status.idle": "2025-12-07T15:54:32.371854Z",
541
- "shell.execute_reply": "2025-12-07T15:54:32.371349Z",
542
- "shell.execute_reply.started": "2025-12-07T15:54:25.968269Z"
543
  }
544
  },
545
  "outputs": [
@@ -547,81 +644,25 @@
547
  "name": "stdout",
548
  "output_type": "stream",
549
  "text": [
550
- "Reading base file from gemma: gemma_standardized_country.csv\n"
551
- ]
552
- },
553
- {
554
- "name": "stderr",
555
- "output_type": "stream",
556
- "text": [
557
- "/tmp/ipykernel_1771575/428075483.py:48: DtypeWarning: Columns (52,53,54,55,56) have mixed types. Specify dtype option on import or set low_memory=False.\n",
558
- " base_df = pd.read_csv(first_path)\n"
559
- ]
560
- },
561
- {
562
- "name": "stdout",
563
- "output_type": "stream",
564
- "text": [
565
  " - Shape: (50861, 58)\n",
566
  " - Columns: ['id', 'name', 'type', 'baseModel', 'downloadCount', 'nsfwLevel', 'modelVersions', 'publishedAt', 'usernameHash', 'downloadUrl']...\n",
567
  "\n",
568
- "Processing gemma: gemma_standardized_country.csv\n"
569
- ]
570
- },
571
- {
572
- "name": "stderr",
573
- "output_type": "stream",
574
- "text": [
575
- "/tmp/ipykernel_1771575/428075483.py:79: DtypeWarning: Columns (52,53,54,55,56) have mixed types. Specify dtype option on import or set low_memory=False.\n",
576
- " df = pd.read_csv(file_path)\n"
577
- ]
578
- },
579
- {
580
- "name": "stdout",
581
- "output_type": "stream",
582
- "text": [
583
  " - Added gemma_full_name\n",
584
  " - Added gemma_aliases\n",
585
  " - Added gemma_gender\n",
586
  " - Added gemma_profession_llm\n",
587
  " - Added gemma_country\n",
588
  "\n",
589
- "Processing mistral: mistral_standardized_country.csv\n"
590
- ]
591
- },
592
- {
593
- "name": "stderr",
594
- "output_type": "stream",
595
- "text": [
596
- "/tmp/ipykernel_1771575/428075483.py:79: DtypeWarning: Columns (52,53,54,55,56) have mixed types. Specify dtype option on import or set low_memory=False.\n",
597
- " df = pd.read_csv(file_path)\n"
598
- ]
599
- },
600
- {
601
- "name": "stdout",
602
- "output_type": "stream",
603
- "text": [
604
  " - Added mistral_full_name\n",
605
  " - Added mistral_aliases\n",
606
  " - Added mistral_gender\n",
607
  " - Added mistral_profession_llm\n",
608
  " - Added mistral_country\n",
609
  "\n",
610
- "Processing qwen: qwen_standardized_country.csv\n"
611
- ]
612
- },
613
- {
614
- "name": "stderr",
615
- "output_type": "stream",
616
- "text": [
617
- "/tmp/ipykernel_1771575/428075483.py:79: DtypeWarning: Columns (53,54,55,56,57) have mixed types. Specify dtype option on import or set low_memory=False.\n",
618
- " df = pd.read_csv(file_path)\n"
619
- ]
620
- },
621
- {
622
- "name": "stdout",
623
- "output_type": "stream",
624
- "text": [
625
  " - Added qwen_full_name\n",
626
  " - Added qwen_aliases\n",
627
  " - Added qwen_gender\n",
@@ -645,12 +686,12 @@
645
  "\n",
646
  "FULL_NAME Comparison:\n",
647
  "----------------------------------------\n",
648
- " name gemma mistral qwen\n",
649
- "0 IU Lee Ji-eun Lee Ji-eun (Lee, Ji-eun) Lee Ji-eun\n",
650
- "1 Super Pose Book Vol.1 - ControlNet Unknown Unknown Super Pose Book\n",
651
- "2 Liyuu LoRA Li Yuchan Liyuu (Full legal name unknown) Li Yuxiao\n",
652
- "3 Irene Irene Kim Irene (Full legal name unknown) Irene Kim\n",
653
- "4 AESPA Karina Yoo Jimin Karina (Kim Jung-yeon) Kim Karina\n",
654
  "\n",
655
  "COUNTRY Comparison:\n",
656
  "----------------------------------------\n",
@@ -676,29 +717,29 @@
676
  "\n",
677
  "GEMMA Country Distribution:\n",
678
  "gemma_country\n",
679
- "USA 18879\n",
680
- "Unknown 5639\n",
681
- "Japan 3911\n",
682
- "UK 2792\n",
683
- "South Korea 2058\n",
684
  "Name: count, dtype: int64\n",
685
  "\n",
686
  "MISTRAL Country Distribution:\n",
687
  "mistral_country\n",
688
- "Unknown 10787\n",
689
- "USA 6622\n",
690
- "Japan 2498\n",
691
- "South Korea 1650\n",
692
- "UK 1609\n",
693
  "Name: count, dtype: int64\n",
694
  "\n",
695
  "QWEN Country Distribution:\n",
696
  "qwen_country\n",
697
- "USA 19841\n",
698
- "Japan 3924\n",
699
- "UK 2787\n",
700
- "South Korea 2178\n",
701
- "Unknown 2093\n",
702
  "Name: count, dtype: int64\n",
703
  "\n",
704
  "============================================================\n",
@@ -980,6 +1021,494 @@
980
  " print(f\" - {path}\")"
981
  ]
982
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
983
  {
984
  "cell_type": "markdown",
985
  "id": "e0f8f331-a6f3-49e8-9734-ee7ffafebabc",
@@ -1011,15 +1540,15 @@
1011
  },
1012
  {
1013
  "cell_type": "code",
1014
- "execution_count": 26,
1015
  "id": "fe38cf23-771d-4888-a5ed-f2b8beb17017",
1016
  "metadata": {
1017
  "execution": {
1018
- "iopub.execute_input": "2025-12-07T15:54:36.393391Z",
1019
- "iopub.status.busy": "2025-12-07T15:54:36.393240Z",
1020
- "iopub.status.idle": "2025-12-07T15:54:49.470031Z",
1021
- "shell.execute_reply": "2025-12-07T15:54:49.469484Z",
1022
- "shell.execute_reply.started": "2025-12-07T15:54:36.393376Z"
1023
  }
1024
  },
1025
  "outputs": [
@@ -1027,21 +1556,7 @@
1027
  "name": "stdout",
1028
  "output_type": "stream",
1029
  "text": [
1030
- "Loading combined data from: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/combined_llm_annotations.csv\n"
1031
- ]
1032
- },
1033
- {
1034
- "name": "stderr",
1035
- "output_type": "stream",
1036
- "text": [
1037
- "/tmp/ipykernel_1771575/114048965.py:455: DtypeWarning: Columns (52,53,54,55,56,57,58,59,60,61,62,63,64,65,66) have mixed types. Specify dtype option on import or set low_memory=False.\n",
1038
- " df_combined = pd.read_csv(combined_file)\n"
1039
- ]
1040
- },
1041
- {
1042
- "name": "stdout",
1043
- "output_type": "stream",
1044
- "text": [
1045
  "Loaded 50,861 rows with 67 columns\n",
1046
  "Analyzing agreement between models: gemma, qwen, mistral\n",
1047
  "Total rows to analyze: 50861\n",
@@ -1051,21 +1566,21 @@
1051
  "============================================================\n",
1052
  "\n",
1053
  "Total rows analyzed: 50,861\n",
1054
- "Rows with at least 2 models MEANINGFULLY agreeing (valid): 33,517 (65.9%)\n",
1055
- "Rows with all 3 models MEANINGFULLY agreeing: 14,179 (27.9%)\n",
1056
- "Rows with all-unknown values for any field: 6,489 (12.8%)\n",
1057
  "\n",
1058
  "Field-wise ANY agreement (including unknown matches):\n",
1059
- " - Country: 37,685 rows (74.1%)\n",
1060
- " - Gender: 42,147 rows (82.9%)\n",
1061
- " - Profession: 42,356 rows (83.3%)\n",
1062
- " - Name: 36,859 rows (72.5%)\n",
1063
  "\n",
1064
  "Field-wise MEANINGFUL agreement (excluding unknown matches):\n",
1065
- " - Country: 34,120 rows (67.1%)\n",
1066
- " - Gender: 41,753 rows (82.1%)\n",
1067
- " - Profession: 42,356 rows (83.3%)\n",
1068
- " - Name: 36,859 rows (72.5%)\n",
1069
  "\n",
1070
  "Examples of successful name matches with variations:\n",
1071
  " Row 0:\n",
@@ -1089,69 +1604,118 @@
1089
  "Number of MEANINGFUL agreeing pairs per field (out of 3 possible pairs):\n",
1090
  "\n",
1091
  "Country:\n",
1092
- " - 0 meaningful agreeing pairs: 16,741 rows (32.9%)\n",
1093
- " - 1 meaningful agreeing pairs: 18,697 rows (36.8%)\n",
1094
- " - 3 meaningful agreeing pairs: 15,423 rows (30.3%)\n",
1095
  "\n",
1096
  "Gender:\n",
1097
- " - 0 meaningful agreeing pairs: 9,108 rows (17.9%)\n",
1098
- " - 1 meaningful agreeing pairs: 14,756 rows (29.0%)\n",
1099
- " - 3 meaningful agreeing pairs: 26,997 rows (53.1%)\n",
1100
  "\n",
1101
  "Profession:\n",
1102
- " - 0 meaningful agreeing pairs: 8,505 rows (16.7%)\n",
1103
- " - 1 meaningful agreeing pairs: 14,318 rows (28.2%)\n",
1104
- " - 2 meaningful agreeing pairs: 1,538 rows (3.0%)\n",
1105
- " - 3 meaningful agreeing pairs: 26,500 rows (52.1%)\n",
1106
  "\n",
1107
  "Name:\n",
1108
- " - 0 meaningful agreeing pairs: 14,002 rows (27.5%)\n",
1109
- " - 1 meaningful agreeing pairs: 14,494 rows (28.5%)\n",
1110
- " - 2 meaningful agreeing pairs: 500 rows (1.0%)\n",
1111
- " - 3 meaningful agreeing pairs: 21,865 rows (43.0%)\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1112
  "\n",
1113
  "Saved analyzed data to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement.csv\n",
1114
- "Saved valid rows (33,517) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_valid.csv\n",
1115
- "Saved MEANINGFUL consensus rows (14,179) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_consensus.csv\n",
1116
- "Saved all-unknown rows (6,489) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_all_unknown.csv\n",
1117
  "\n",
1118
  "============================================================\n",
1119
  "SAMPLE OF MEANINGFUL CONSENSUS ROWS (first 2 rows)\n",
1120
  "============================================================\n",
1121
  "\n",
1122
  "Consensus Row 0:\n",
1123
- " gemma:\n",
1124
- " Name: Lee Ji-eun\n",
1125
- " Country: South Korea\n",
1126
- " Gender: Female\n",
1127
- " Profession: singer/musician, actor, tv personality\n",
1128
- " qwen:\n",
1129
- " Name: Lee Ji-eun\n",
1130
- " Country: South Korea\n",
1131
- " Gender: Female\n",
1132
- " Profession: singer/musician, model, public figure\n",
1133
- " mistral:\n",
1134
- " Name: Lee Ji-eun (Lee, Ji-eun)\n",
1135
- " Country: South Korea\n",
1136
- " Gender: Female\n",
1137
- " Profession: singer/musician, tv personality, actress\n",
 
 
 
 
 
1138
  "\n",
1139
  "Consensus Row 4:\n",
1140
- " gemma:\n",
1141
- " Name: Yoo Jimin\n",
1142
- " Country: South Korea\n",
1143
- " Gender: Female\n",
1144
- " Profession: singer/musician, tv personality, online personality\n",
1145
- " qwen:\n",
1146
- " Name: Kim Karina\n",
1147
- " Country: South Korea\n",
1148
- " Gender: Female\n",
1149
- " Profession: singer/musician, model, online personality\n",
1150
- " mistral:\n",
1151
- " Name: Karina (Kim Jung-yeon)\n",
1152
- " Country: South Korea\n",
1153
- " Gender: Female\n",
1154
- " Profession: singer/musician, tv personality, public figure\n"
 
 
 
 
 
1155
  ]
1156
  }
1157
  ],
@@ -1160,6 +1724,8 @@
1160
  "import numpy as np\n",
1161
  "import re\n",
1162
  "from typing import List, Set, Tuple\n",
 
 
1163
  "\n",
1164
  "def is_unknown_value(value) -> bool:\n",
1165
  " \"\"\"Check if a value is considered 'unknown' or empty.\"\"\"\n",
@@ -1443,6 +2009,117 @@
1443
  " \n",
1444
  " return results\n",
1445
  "\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1446
  "def analyze_model_agreement(df: pd.DataFrame, models: List[str]) -> pd.DataFrame:\n",
1447
  " \"\"\"\n",
1448
  " Analyze agreement between models and add agreement columns.\n",
@@ -1602,55 +2279,69 @@
1602
  " print(f\"Saved all-unknown rows ({len(unknown_df):,}) to: {unknown_path}\")\n",
1603
  "\n",
1604
  "# ============================================================\n",
1605
- "# RUN THE ANALYSIS\n",
1606
  "# ============================================================\n",
1607
  "\n",
1608
- "# Load the combined data\n",
1609
- "combined_file = current_dir.parent / \"data/CSV/combined_llm_annotations.csv\"\n",
1610
- "print(f\"Loading combined data from: {combined_file}\")\n",
1611
- "\n",
1612
- "if combined_file.exists():\n",
1613
- " df_combined = pd.read_csv(combined_file)\n",
1614
- " print(f\"Loaded {len(df_combined):,} rows with {len(df_combined.columns)} columns\")\n",
1615
- " \n",
1616
- " # Define the models\n",
1617
- " models = ['gemma', 'qwen', 'mistral']\n",
1618
- " \n",
1619
- " # Run the agreement analysis\n",
1620
- " df_analyzed = analyze_model_agreement(df_combined, models)\n",
1621
- " \n",
1622
- " # Save the results\n",
1623
- " output_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement.csv\"\n",
1624
- " save_analysis_results(df_analyzed, output_file)\n",
1625
- " \n",
1626
- " # Show sample of consensus rows\n",
1627
- " print(\"\\n\" + \"=\"*60)\n",
1628
- " print(\"SAMPLE OF MEANINGFUL CONSENSUS ROWS (first 2 rows)\")\n",
1629
- " print(\"=\"*60)\n",
1630
  " \n",
1631
- " consensus_rows = df_analyzed[df_analyzed['all_models_agree']].head(2)\n",
 
 
1632
  " \n",
1633
- " if len(consensus_rows) > 0:\n",
1634
- " for idx, row in consensus_rows.iterrows():\n",
1635
- " print(f\"\\nConsensus Row {idx}:\")\n",
1636
- " for model in models:\n",
1637
- " name_col = f'{model}_full_name'\n",
1638
- " country_col = f'{model}_country'\n",
1639
- " gender_col = f'{model}_gender'\n",
1640
- " prof_col = f'{model}_profession_llm'\n",
1641
- " \n",
1642
- " if name_col in row:\n",
1643
- " print(f\" {model}:\")\n",
1644
- " print(f\" Name: {row[name_col]}\")\n",
1645
- " print(f\" Country: {row[country_col]}\")\n",
1646
- " print(f\" Gender: {row[gender_col]}\")\n",
1647
- " print(f\" Profession: {row[prof_col]}\")\n"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1648
  ]
1649
  },
1650
  {
1651
  "cell_type": "code",
1652
  "execution_count": null,
1653
- "id": "5df51c5d-9a40-4742-9916-cc4863770d56",
1654
  "metadata": {},
1655
  "outputs": [],
1656
  "source": []
 
1
  {
2
  "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "id": "b222ae9b-d94a-4892-9b7f-bf6a201ede47",
6
+ "metadata": {},
7
+ "source": [
8
+ "## Comprehensive Country Name Standardization with Fictional Place Detection\n",
9
+ "### Saves files as modelname_standardized_country.csv"
10
+ ]
11
+ },
12
  {
13
  "cell_type": "code",
14
+ "execution_count": 12,
15
  "id": "6cbaef9d-3058-4a59-a8ee-32fcc2062ed6",
16
  "metadata": {
17
  "execution": {
18
+ "iopub.execute_input": "2025-12-08T12:56:23.897161Z",
19
+ "iopub.status.busy": "2025-12-08T12:56:23.896726Z",
20
+ "iopub.status.idle": "2025-12-08T12:56:34.193068Z",
21
+ "shell.execute_reply": "2025-12-08T12:56:34.191636Z",
22
+ "shell.execute_reply.started": "2025-12-08T12:56:23.897125Z"
23
  }
24
  },
25
  "outputs": [
 
27
  "name": "stdout",
28
  "output_type": "stream",
29
  "text": [
30
+ "============================================================\n",
31
+ "COUNTRY STANDARDIZATION WITH SPECIAL MISTRAL HANDLING\n",
32
+ "============================================================\n",
33
+ "MISTRAL TRANSFORMATIONS:\n",
34
+ " - Remove brackets () from 'country' column\n",
35
+ " - Remove brackets () from 'full_name' column\n",
36
+ " - Convert 'actress' → 'actor' in profession\n",
37
+ " - Convert 'cosplayer' → 'online personality' in profession\n",
38
+ "\n",
39
+ "GEMMA & QWEN: No special transformations\n",
40
+ "============================================================\n",
41
+ "\n",
42
+ "Processing: gemma_local_annotated_POI.csv\n",
 
 
43
  " - Found and standardized 'country' column\n",
44
  " - Updated 'name' column for 4 fictional entries\n",
45
  " - Updated 'real_name' column for 4 fictional entries\n",
46
  " - Updated 'full_name' column for 4 fictional entries\n",
47
+ " - Changes: 6536 standardized, 4 fictional→Unknown\n",
48
+ " - Unique countries after standardization: ['Afghanistan', 'Albania', 'Algeria', 'American Samoa', 'Angola', 'Argentina', 'Armenia', 'Australia', 'Austria', 'Azerbaijan', 'Bahamas', 'Bangladesh', 'Barbados', 'Belarus', 'Belgium', 'Benin', 'Bolivia', 'Brazil', 'Bulgaria', 'Cambodia', 'Cameroon', 'Canada', 'Central African Republic', 'Chile', 'China', 'Colombia', 'Costa Rica', 'Croatia', 'Cuba', 'Cyprus', 'Czechia', 'Denmark', 'Dominican Republic', 'Egypt', 'El Salvador', 'Estonia', 'Ethiopia', 'Finland', 'France', 'French Polynesia', 'Georgia', 'Germany', 'Ghana', 'Greece', 'Guatemala', 'Guyana', 'Hong Kong', 'Hungary', 'Iceland', 'India', 'Indonesia', 'Iran', 'Iraq', 'Ireland', 'Israel', 'Italy', 'Jamaica', 'Japan', 'Jordan', 'Kazakhstan', 'Kenya', 'Kosovo', 'Kuwait', 'Kyrgyzstan', 'Latvia', 'Lebanon', 'Libya', 'Lithuania', 'Macau', 'Malaysia', 'Malta', 'Mexico', 'Moldova', 'Monaco', 'Mongolia', 'Morocco', 'Myanmar', 'Namibia', 'Nepal', 'Netherlands', 'New Zealand', 'Nicaragua', 'Nigeria', 'North Korea', 'North Macedonia', 'Norway', 'Pakistan', 'Palestine', 'Paraguay', 'Peru', 'Philippines', 'Poland', 'Portugal', 'Puerto Rico', 'Republic of the Congo', 'Romania', 'Russia', 'Samoa', 'Saudi Arabia', 'Senegal', 'Serbia', 'Singapore', 'Slovakia', 'Slovenia', 'Somalia', 'South Africa', 'South Korea', 'Spain', 'Sri Lanka', 'Sudan', 'Sweden', 'Switzerland', 'Syria', 'Taiwan', 'Tanzania', 'Thailand', 'Tonga', 'Trinidad and Tobago', 'Tunisia', 'Türkiye', 'UK', 'USA', 'Uganda', 'Ukraine', 'Unknown', 'Uruguay', 'Uzbekistan', 'Venezuela', 'Vietnam', 'Zambia', 'Zimbabwe']\n",
49
  " - Saved to: gemma_standardized_country.csv\n",
50
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/gemma_standardized_country.csv\n",
51
  "\n",
52
+ "Processing: mistral_local_annotated_POI.csv\n",
53
+ " ⚠️ MISTRAL DATA: Will remove bracketed content ()\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
54
  " - Updated 'name' column for fictional entries\n",
55
  " - Saved to: mistral_standardized_country.csv\n",
56
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/mistral_standardized_country.csv\n",
57
  "\n",
58
+ "Processing: qwen_local_annotated_POI.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
59
  " - Found and standardized 'country' column\n",
60
+ " - Updated 'name' column for 14 fictional entries\n",
61
+ " - Updated 'real_name' column for 14 fictional entries\n",
62
+ " - Updated 'full_name' column for 14 fictional entries\n",
63
+ " - Changes: 26750 standardized, 14 fictional→Unknown\n",
64
+ " - Unique countries after standardization: ['Afghanistan', 'Albania', 'Algeria', 'American Samoa', 'Argentina', 'Armenia', 'Australia', 'Austria', 'Azerbaijan', 'Babylonian Empire', 'Bangladesh', 'Barbados', 'Belarus', 'Belgium', 'Benin', 'Bolivia', 'Bosnia and Herzegovina', 'Brazil', 'Bulgaria', 'Cameroon', 'Canada', 'Central African Republic', 'Chile', 'China', 'Colombia', 'Costa Rica', 'Croatia', 'Cuba', 'Cyprus', 'Czechia', 'Denmark', 'Discworld', 'Dominican Republic', 'Ecuador', 'Egypt', 'El Salvador', 'Estonia', 'Europe', 'Finland', 'France', 'French Polynesia', 'Georgia', 'Germany', 'Ghana', 'Greece', 'Guatemala', 'Haiti', 'Hong Kong', 'Hungary', 'Iceland', 'India', 'Indonesia', 'Iran', 'Iraq', 'Ireland', 'Israel', 'Italy', 'Jamaica', 'Japan', 'Jordan', 'Kazakhstan', 'Kenya', 'Kuwait', 'Kyrgyzstan', 'Laos', 'Latvia', 'Lebanon', 'Libya', 'Lithuania', 'Malaysia', 'Mali', 'Malta', 'Mexico', 'Moldova', 'Monaco', 'Mongolia', 'Morocco', 'Myanmar', 'Nepal', 'Netherlands', 'New Zealand', 'Nicaragua', 'Nigeria', 'North Korea', 'North Macedonia', 'Norway', 'Oman', 'Pakistan', 'Palestine', 'Papua New Guinea', 'Paraguay', 'Peru', 'Philippines', 'Poland', 'Portugal', 'Puerto Rico', 'Romania', 'Rome', 'Russia', 'Samoa', 'Saudi Arabia', 'Senegal', 'Serbia', 'Singapore', 'Slovakia', 'Slovenia', 'South Africa', 'South Korea', 'South Sudan', 'Spain', 'Sri Lanka', 'Sudan', 'Sweden', 'Switzerland', 'Syria', 'Taiwan', 'Thailand', 'Tunisia', 'Türkiye', 'UK', 'USA', 'Uganda', 'Ukraine', 'United Arab Emirates', 'Unknown', 'Uruguay', 'Uzbekistan', 'Vatican City', 'Venezuela', 'Vietnam', 'Worldwide', 'Zimbabwe']\n",
65
  " - Saved to: qwen_standardized_country.csv\n",
66
  " - Full path: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/qwen_standardized_country.csv\n",
67
  "\n",
 
72
  "GEMMA:\n",
73
  " Total rows: 50861\n",
74
  " Top 10 countries:\n",
75
+ " - USA: 19378\n",
76
+ " - Unknown: 6320\n",
77
+ " - Japan: 4042\n",
78
+ " - UK: 2869\n",
79
+ " - South Korea: 2088\n",
80
+ " - China: 1765\n",
81
+ " - India: 1163\n",
82
+ " - Canada: 1150\n",
83
+ " - Russia: 1003\n",
84
+ " - France: 803\n",
85
  "\n",
86
  "MISTRAL:\n",
87
  " Total rows: 50861\n",
88
  " Top 10 countries:\n",
89
+ " - Unknown: 21033\n",
90
+ " - USA: 10564\n",
91
+ " - Japan: 3068\n",
92
+ " - UK: 2627\n",
93
+ " - South Korea: 2097\n",
94
+ " - China: 1409\n",
95
+ " - India: 1161\n",
96
+ " - Canada: 854\n",
97
+ " - France: 662\n",
98
+ " - Russia: 620\n",
99
  "\n",
100
  "QWEN:\n",
101
  " Total rows: 50861\n",
102
  " Top 10 countries:\n",
103
+ " - USA: 23017\n",
104
+ " - Japan: 4326\n",
105
+ " - UK: 3174\n",
106
+ " - Unknown: 2672\n",
107
+ " - China: 2403\n",
108
+ " - South Korea: 2333\n",
109
+ " - India: 1341\n",
110
+ " - Russia: 1036\n",
111
+ " - France: 851\n",
112
+ " - Canada: 828\n",
113
  "\n",
114
  "============================================================\n",
115
  "Processing complete!\n",
 
118
  " - gemma_standardized_country.csv\n",
119
  " - mistral_standardized_country.csv\n",
120
  " - qwen_standardized_country.csv\n",
121
+ "\n",
122
+ "============================================================\n",
123
+ "MISTRAL-SPECIFIC TRANSFORMATIONS:\n",
124
+ " ✓ Country: Brackets () removed\n",
125
+ " ✓ Full_name: Brackets () removed\n",
126
+ " ✓ Profession: 'actress' → 'actor'\n",
127
+ " ✓ Profession: 'cosplayer' → 'online personality'\n",
128
+ "\n",
129
+ "GEMMA & QWEN:\n",
130
+ " ✓ No bracket removal\n",
131
+ " ✓ No profession transformations\n",
132
  "============================================================\n"
133
  ]
134
  }
135
  ],
136
  "source": [
 
 
 
137
  "import pandas as pd\n",
138
  "from pathlib import Path\n",
139
+ "import re\n",
140
  "\n",
141
  "# Define fictional places, companies, and invalid entries\n",
142
  "FICTIONAL_PLACES = {\n",
 
283
  " 'trinidad': 'Trinidad and Tobago',\n",
284
  "}\n",
285
  "\n",
286
+ "def remove_bracketed_content(text):\n",
287
+ " \"\"\"\n",
288
+ " Remove everything in brackets (parentheses) from text.\n",
289
+ " Example: \"Japan (Asian country)\" -> \"Japan\"\n",
290
+ " \"\"\"\n",
291
+ " if pd.isna(text):\n",
292
+ " return text\n",
293
+ " \n",
294
+ " text_str = str(text).strip()\n",
295
+ " \n",
296
+ " # Remove content in parentheses including the parentheses\n",
297
+ " cleaned = re.sub(r'\\([^)]*\\)', '', text_str)\n",
298
+ " \n",
299
+ " # Strip any extra whitespace left after removal\n",
300
+ " return cleaned.strip()\n",
301
+ "\n",
302
+ "def standardize_profession(profession_value, is_mistral=False):\n",
303
+ " \"\"\"\n",
304
+ " Standardize profession values.\n",
305
+ " For Mistral data: actress -> actor, cosplayer -> online personality\n",
306
+ " \n",
307
+ " Args:\n",
308
+ " profession_value: The profession value to standardize\n",
309
+ " is_mistral: If True, applies Mistral-specific transformations\n",
310
+ " \"\"\"\n",
311
+ " if pd.isna(profession_value):\n",
312
+ " return profession_value\n",
313
+ " \n",
314
+ " profession_str = str(profession_value).strip()\n",
315
+ " \n",
316
+ " if not profession_str or profession_str == 'Unknown':\n",
317
+ " return profession_str\n",
318
+ " \n",
319
+ " # For Mistral data only, apply specific transformations\n",
320
+ " if is_mistral:\n",
321
+ " profession_lower = profession_str.lower()\n",
322
+ " \n",
323
+ " # actress -> actor\n",
324
+ " if profession_lower == 'actress':\n",
325
+ " return 'actor'\n",
326
+ " \n",
327
+ " # cosplayer -> online personality\n",
328
+ " if profession_lower == 'cosplayer':\n",
329
+ " return 'online personality'\n",
330
+ " \n",
331
+ " return profession_str\n",
332
+ "\n",
333
  "def is_fictional_or_invalid(country_str):\n",
334
  " \"\"\"\n",
335
  " Check if a country string is fictional, a company, or historically invalid.\n",
 
362
  " \n",
363
  " return False\n",
364
  "\n",
365
+ "def standardize_country(country_value, is_mistral=False):\n",
366
  " \"\"\"\n",
367
  " Standardize a single country name based on the mapping.\n",
368
+ " \n",
369
+ " Args:\n",
370
+ " country_value: The country value to standardize\n",
371
+ " is_mistral: If True, removes bracketed content first (MISTRAL ONLY)\n",
372
  " \"\"\"\n",
373
  " if pd.isna(country_value):\n",
374
  " return country_value\n",
 
376
  " # Convert to string and strip whitespace\n",
377
  " country_str = str(country_value).strip()\n",
378
  " \n",
379
+ " # For mistral data ONLY, remove bracketed content first\n",
380
+ " if is_mistral:\n",
381
+ " original_str = country_str\n",
382
+ " country_str = remove_bracketed_content(country_str)\n",
383
+ " # if original_str != country_str:\n",
384
+ " # print(f\" [Mistral] Removed brackets: '{original_str}' -> '{country_str}'\")\n",
385
+ " \n",
386
  " # Return if empty or already 'Unknown'\n",
387
  " if not country_str or country_str == 'Unknown':\n",
388
+ " return country_str if country_str else 'Unknown'\n",
389
  " \n",
390
  " # Check if it's fictional or invalid\n",
391
  " if is_fictional_or_invalid(country_str):\n",
 
427
  " input_path = Path(input_file)\n",
428
  " output_path = Path(output_file)\n",
429
  " \n",
430
+ " # Determine if this is mistral data\n",
431
+ " is_mistral = 'mistral' in input_path.name.lower()\n",
432
+ " \n",
433
  " print(f\"Processing: {input_path.name}\")\n",
434
+ " if is_mistral:\n",
435
+ " print(f\" ⚠️ MISTRAL DATA: Will remove bracketed content ()\")\n",
436
  " \n",
437
  " # Track changes\n",
438
+ " changes_made = {'standardized': 0, 'fictional_to_unknown': 0, 'brackets_removed': 0}\n",
439
  " \n",
440
  " # For mistral.csv which might have no header, we need special handling\n",
441
+ " if is_mistral:\n",
442
  " # Try to read normally first\n",
443
  " try:\n",
444
  " df = pd.read_csv(input_path)\n",
445
  " # Check if 'country' column exists\n",
446
  " if 'country' in df.columns:\n",
447
  " original_values = df['country'].copy()\n",
448
+ " \n",
449
+ " # Count brackets before removal\n",
450
+ " changes_made['brackets_removed'] = original_values.astype(str).str.contains(r'\\(').sum()\n",
451
+ " \n",
452
+ " df['country'] = df['country'].apply(lambda x: standardize_country(x, is_mistral=True))\n",
453
  " \n",
454
  " # Count changes\n",
455
  " changes_made['standardized'] = (original_values != df['country']).sum()\n",
456
  " changes_made['fictional_to_unknown'] = ((df['country'] == 'Unknown') & (original_values != 'Unknown')).sum()\n",
457
  " \n",
458
+ " #print(f\" - Found and standardized 'country' column\")\n",
459
+ " #print(f\" - Removed brackets from {changes_made['brackets_removed']} country entries\")\n",
460
  " \n",
461
  " # Also update 'name' column for fictional entries\n",
462
  " if 'name' in df.columns:\n",
463
  " fictional_mask = df['country'] == 'Unknown'\n",
464
  " df.loc[fictional_mask & (original_values != 'Unknown'), 'name'] = 'Unknown'\n",
465
  " print(f\" - Updated 'name' column for fictional entries\")\n",
466
+ " \n",
467
+ " # MISTRAL SPECIFIC: Remove brackets from full_name column\n",
468
+ " if 'full_name' in df.columns:\n",
469
+ " original_full_names = df['full_name'].copy()\n",
470
+ " brackets_in_names = original_full_names.astype(str).str.contains(r'\\(').sum()\n",
471
+ " df['full_name'] = df['full_name'].apply(remove_bracketed_content)\n",
472
+ " #print(f\" - Removed brackets from {brackets_in_names} full_name entries\")\n",
473
+ " \n",
474
+ " # MISTRAL SPECIFIC: Standardize profession_llm column\n",
475
+ " if 'profession_llm' in df.columns:\n",
476
+ " original_professions = df['profession_llm'].copy()\n",
477
+ " df['profession_llm'] = df['profession_llm'].apply(lambda x: standardize_profession(x, is_mistral=True))\n",
478
+ " \n",
479
+ " actress_count = (original_professions.astype(str).str.lower() == 'actress').sum()\n",
480
+ " cosplayer_count = (original_professions.astype(str).str.lower() == 'cosplayer').sum()\n",
481
+ " \n",
482
+ " if actress_count > 0:\n",
483
+ " print(f\" - Converted {actress_count} 'actress' → 'actor'\")\n",
484
+ " if cosplayer_count > 0:\n",
485
+ " print(f\" - Converted {cosplayer_count} 'cosplayer' → 'online personality'\")\n",
486
+ " \n",
487
  " else:\n",
488
  " # If no country column, assume last column\n",
489
  " last_col = df.columns[-1]\n",
490
+ " df[last_col] = df[last_col].apply(lambda x: standardize_country(x, is_mistral=True))\n",
491
  " print(f\" - Standardized column '{last_col}' (assumed to be country)\")\n",
492
  " except:\n",
493
  " # If normal reading fails, try without header\n",
494
  " df = pd.read_csv(input_path, header=None)\n",
495
  " last_col = df.columns[-1]\n",
496
+ " df[last_col] = df[last_col].apply(lambda x: standardize_country(x, is_mistral=True))\n",
497
  " print(f\" - Standardized column {last_col} (assumed to be country, no header)\")\n",
498
  " else:\n",
499
+ " # Normal CSV with header (GEMMA and QWEN - NO bracket removal)\n",
500
  " df = pd.read_csv(input_path)\n",
501
  " \n",
502
  " # Check if 'country' column exists\n",
503
  " if 'country' in df.columns:\n",
504
  " original_values = df['country'].copy()\n",
505
+ " df['country'] = df['country'].apply(lambda x: standardize_country(x, is_mistral=False))\n",
506
  " \n",
507
  " # Count changes\n",
508
  " changes_made['standardized'] = (original_values != df['country']).sum()\n",
 
548
  "# Create a results dictionary to store processed dataframes\n",
549
  "results = {}\n",
550
  "\n",
551
+ "print(\"=\"*60)\n",
552
+ "print(\"COUNTRY STANDARDIZATION WITH SPECIAL MISTRAL HANDLING\")\n",
553
+ "print(\"=\"*60)\n",
554
+ "print(\"MISTRAL TRANSFORMATIONS:\")\n",
555
+ "print(\" - Remove brackets () from 'country' column\")\n",
556
+ "print(\" - Remove brackets () from 'full_name' column\")\n",
557
+ "print(\" - Convert 'actress' → 'actor' in profession\")\n",
558
+ "print(\" - Convert 'cosplayer' → 'online personality' in profession\")\n",
559
+ "print(\"\\nGEMMA & QWEN: No special transformations\")\n",
560
+ "print(\"=\"*60 + \"\\n\")\n",
561
+ "\n",
562
  "for input_path in input_files:\n",
563
  " # Convert to Path object if it isn't already\n",
564
  " input_path = Path(input_path)\n",
 
606
  "print(\"\\nOutput files created:\")\n",
607
  "for model_name in results.keys():\n",
608
  " print(f\" - {model_name}_standardized_country.csv\")\n",
609
+ "print(\"\\n\" + \"=\" * 60)\n",
610
+ "print(\"MISTRAL-SPECIFIC TRANSFORMATIONS:\")\n",
611
+ "print(\" ✓ Country: Brackets () removed\")\n",
612
+ "print(\" ✓ Full_name: Brackets () removed\")\n",
613
+ "print(\" ✓ Profession: 'actress' → 'actor'\")\n",
614
+ "print(\" ✓ Profession: 'cosplayer' → 'online personality'\")\n",
615
+ "print(\"\\nGEMMA & QWEN:\")\n",
616
+ "print(\" ✓ No bracket removal\")\n",
617
+ "print(\" ✓ No profession transformations\")\n",
618
  "print(\"=\" * 60)"
619
  ]
620
  },
 
628
  },
629
  {
630
  "cell_type": "code",
631
+ "execution_count": 13,
632
  "id": "7f2fe3fb-5863-45ec-9df3-9fa64751936b",
633
  "metadata": {
634
  "execution": {
635
+ "iopub.execute_input": "2025-12-08T12:56:43.814045Z",
636
+ "iopub.status.busy": "2025-12-08T12:56:43.813613Z",
637
+ "iopub.status.idle": "2025-12-08T12:56:50.203009Z",
638
+ "shell.execute_reply": "2025-12-08T12:56:50.201693Z",
639
+ "shell.execute_reply.started": "2025-12-08T12:56:43.814012Z"
640
  }
641
  },
642
  "outputs": [
 
644
  "name": "stdout",
645
  "output_type": "stream",
646
  "text": [
647
+ "Reading base file from gemma: gemma_standardized_country.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
648
  " - Shape: (50861, 58)\n",
649
  " - Columns: ['id', 'name', 'type', 'baseModel', 'downloadCount', 'nsfwLevel', 'modelVersions', 'publishedAt', 'usernameHash', 'downloadUrl']...\n",
650
  "\n",
651
+ "Processing gemma: gemma_standardized_country.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
652
  " - Added gemma_full_name\n",
653
  " - Added gemma_aliases\n",
654
  " - Added gemma_gender\n",
655
  " - Added gemma_profession_llm\n",
656
  " - Added gemma_country\n",
657
  "\n",
658
+ "Processing mistral: mistral_standardized_country.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
659
  " - Added mistral_full_name\n",
660
  " - Added mistral_aliases\n",
661
  " - Added mistral_gender\n",
662
  " - Added mistral_profession_llm\n",
663
  " - Added mistral_country\n",
664
  "\n",
665
+ "Processing qwen: qwen_standardized_country.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
666
  " - Added qwen_full_name\n",
667
  " - Added qwen_aliases\n",
668
  " - Added qwen_gender\n",
 
686
  "\n",
687
  "FULL_NAME Comparison:\n",
688
  "----------------------------------------\n",
689
+ " name gemma mistral qwen\n",
690
+ "0 IU Lee Ji-eun Lee Ji-eun Lee Ji-eun\n",
691
+ "1 Super Pose Book Vol.1 - ControlNet Unknown Unknown Super Pose Book\n",
692
+ "2 Liyuu LoRA Li Yuchan Liyuu Li Yuxiao\n",
693
+ "3 Irene Irene Kim Irene Irene Kim\n",
694
+ "4 AESPA Karina Yoo Jimin Karina Kim Karina\n",
695
  "\n",
696
  "COUNTRY Comparison:\n",
697
  "----------------------------------------\n",
 
717
  "\n",
718
  "GEMMA Country Distribution:\n",
719
  "gemma_country\n",
720
+ "USA 19378\n",
721
+ "Unknown 6320\n",
722
+ "Japan 4042\n",
723
+ "UK 2869\n",
724
+ "South Korea 2088\n",
725
  "Name: count, dtype: int64\n",
726
  "\n",
727
  "MISTRAL Country Distribution:\n",
728
  "mistral_country\n",
729
+ "Unknown 21033\n",
730
+ "USA 10564\n",
731
+ "Japan 3068\n",
732
+ "UK 2627\n",
733
+ "South Korea 2097\n",
734
  "Name: count, dtype: int64\n",
735
  "\n",
736
  "QWEN Country Distribution:\n",
737
  "qwen_country\n",
738
+ "USA 23017\n",
739
+ "Japan 4326\n",
740
+ "UK 3174\n",
741
+ "Unknown 2672\n",
742
+ "China 2403\n",
743
  "Name: count, dtype: int64\n",
744
  "\n",
745
  "============================================================\n",
 
1021
  " print(f\" - {path}\")"
1022
  ]
1023
  },
1024
+ {
1025
+ "cell_type": "markdown",
1026
+ "id": "ae787ec8-7403-492d-8552-026f95d9e5b9",
1027
+ "metadata": {},
1028
+ "source": [
1029
+ "### Strict version"
1030
+ ]
1031
+ },
1032
+ {
1033
+ "cell_type": "code",
1034
+ "execution_count": 15,
1035
+ "id": "ca58ae0f-a3e1-44c9-9645-412997f1777d",
1036
+ "metadata": {
1037
+ "execution": {
1038
+ "iopub.execute_input": "2025-12-08T13:00:58.348391Z",
1039
+ "iopub.status.busy": "2025-12-08T13:00:58.347965Z",
1040
+ "iopub.status.idle": "2025-12-08T13:01:02.440993Z",
1041
+ "shell.execute_reply": "2025-12-08T13:01:02.439833Z",
1042
+ "shell.execute_reply.started": "2025-12-08T13:00:58.348357Z"
1043
+ }
1044
+ },
1045
+ "outputs": [
1046
+ {
1047
+ "name": "stdout",
1048
+ "output_type": "stream",
1049
+ "text": [
1050
+ "============================================================\n",
1051
+ "STRICT CONSENSUS FILTER\n",
1052
+ "============================================================\n",
1053
+ "Reading: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/combined_llm_annotations.csv\n",
1054
+ "Input shape: (50861, 67)\n",
1055
+ "Models to check: gemma, mistral, qwen\n",
1056
+ "\n",
1057
+ "Processing rows...\n",
1058
+ " Processed 50000/50861 rows...\n",
1059
+ " Processed 50861 rows. \n",
1060
+ "\n",
1061
+ "============================================================\n",
1062
+ "FILTERING RESULTS\n",
1063
+ "============================================================\n",
1064
+ "Total input rows: 50,861\n",
1065
+ "Rows passing all criteria: 4,181 (8.2%)\n",
1066
+ "\n",
1067
+ "Failure reasons (rows can fail multiple):\n",
1068
+ " - Country disagreement: 25,292 (49.7%)\n",
1069
+ " - Gender disagreement: 6,783 (13.3%)\n",
1070
+ " - Profession disagreement: 43,698 (85.9%)\n",
1071
+ " - Contains 'Unknown': 46,680 (91.8%)\n",
1072
+ "\n",
1073
+ "============================================================\n",
1074
+ "CONSENSUS DISTRIBUTIONS\n",
1075
+ "============================================================\n",
1076
+ "\n",
1077
+ "Country (top 10):\n",
1078
+ "consensus_country\n",
1079
+ "usa 2186\n",
1080
+ "uk 405\n",
1081
+ "japan 337\n",
1082
+ "south korea 237\n",
1083
+ "india 146\n",
1084
+ "china 101\n",
1085
+ "canada 79\n",
1086
+ "russia 74\n",
1087
+ "brazil 73\n",
1088
+ "france 69\n",
1089
+ "Name: count, dtype: int64\n",
1090
+ "\n",
1091
+ "Gender:\n",
1092
+ "consensus_gender\n",
1093
+ "female 3622\n",
1094
+ "male 559\n",
1095
+ "Name: count, dtype: int64\n",
1096
+ "\n",
1097
+ "Profession (top 10):\n",
1098
+ "consensus_profession\n",
1099
+ "actor, tv personality, public figure 1154\n",
1100
+ "actor, tv personality, online personality 633\n",
1101
+ "singer/musician, model, online personality 341\n",
1102
+ "actor, singer/musician, tv personality 322\n",
1103
+ "actor, tv personality, model 267\n",
1104
+ "actor, model, online personality 161\n",
1105
+ "actor, model, tv personality 125\n",
1106
+ "model, adult performer, online personality 80\n",
1107
+ "adult performer, model, online personality 75\n",
1108
+ "singer/musician, tv personality, public figure 71\n",
1109
+ "Name: count, dtype: int64\n",
1110
+ "\n",
1111
+ "Profession agreement level:\n",
1112
+ "profession_agreement_count\n",
1113
+ "2 4096\n",
1114
+ "3 85\n",
1115
+ "Name: count, dtype: int64\n",
1116
+ "\n",
1117
+ "============================================================\n",
1118
+ "✓ Strict consensus file saved to: strict_consensus.csv\n",
1119
+ " Total rows: 4,181\n",
1120
+ " Total columns: 71\n",
1121
+ "============================================================\n",
1122
+ "\n",
1123
+ "============================================================\n",
1124
+ "SAMPLE COMPARISONS (first 5 rows)\n",
1125
+ "============================================================\n",
1126
+ "\n",
1127
+ "--- Row 1: Liu Yifei ---\n",
1128
+ "Country:\n",
1129
+ " Consensus: china\n",
1130
+ " gemma: China\n",
1131
+ " mistral: China\n",
1132
+ " qwen: China\n",
1133
+ "Gender:\n",
1134
+ " Consensus: female\n",
1135
+ " gemma: Female\n",
1136
+ " mistral: Female\n",
1137
+ " qwen: Female\n",
1138
+ "Profession (agreement: 2/3):\n",
1139
+ " Consensus: actor, model, singer/musician\n",
1140
+ " gemma: actor, model, singer/musician\n",
1141
+ " mistral: actor, tv personality, model\n",
1142
+ " qwen: actor, model, singer/musician\n",
1143
+ "\n",
1144
+ "--- Row 2: Emma Watson (JG) ---\n",
1145
+ "Country:\n",
1146
+ " Consensus: uk\n",
1147
+ " gemma: UK\n",
1148
+ " mistral: UK\n",
1149
+ " qwen: UK\n",
1150
+ "Gender:\n",
1151
+ " Consensus: female\n",
1152
+ " gemma: Female\n",
1153
+ " mistral: Female\n",
1154
+ " qwen: Female\n",
1155
+ "Profession (agreement: 2/3):\n",
1156
+ " Consensus: actor, public figure, model\n",
1157
+ " gemma: actor, public figure, model\n",
1158
+ " mistral: actor, tv personality, public figure\n",
1159
+ " qwen: actor, public figure, model\n",
1160
+ "\n",
1161
+ "--- Row 3: Gal Gadot「LoRa」 ---\n",
1162
+ "Country:\n",
1163
+ " Consensus: israel\n",
1164
+ " gemma: Israel\n",
1165
+ " mistral: Israel\n",
1166
+ " qwen: Israel\n",
1167
+ "Gender:\n",
1168
+ " Consensus: female\n",
1169
+ " gemma: Female\n",
1170
+ " mistral: Female\n",
1171
+ " qwen: Female\n",
1172
+ "Profession (agreement: 2/3):\n",
1173
+ " Consensus: actor, model, tv personality\n",
1174
+ " gemma: actor, model, tv personality\n",
1175
+ " mistral: actor, model, tv personality\n",
1176
+ " qwen: actor, model, public figure\n",
1177
+ "\n",
1178
+ "--- Row 4: Game of Thrones Cast ---\n",
1179
+ "Country:\n",
1180
+ " Consensus: uk\n",
1181
+ " gemma: UK\n",
1182
+ " mistral: UK\n",
1183
+ " qwen: UK\n",
1184
+ "Gender:\n",
1185
+ " Consensus: female\n",
1186
+ " gemma: Female\n",
1187
+ " mistral: Female\n",
1188
+ " qwen: Female\n",
1189
+ "Profession (agreement: 2/3):\n",
1190
+ " Consensus: actor, tv personality, online personality\n",
1191
+ " gemma: actor, tv personality, online personality\n",
1192
+ " mistral: actor, tv personality, online personality\n",
1193
+ " qwen: actor, model\n",
1194
+ "\n",
1195
+ "--- Row 5: Karina Makina Lora ---\n",
1196
+ "Country:\n",
1197
+ " Consensus: south korea\n",
1198
+ " gemma: South Korea\n",
1199
+ " mistral: South Korea\n",
1200
+ " qwen: South Korea\n",
1201
+ "Gender:\n",
1202
+ " Consensus: female\n",
1203
+ " gemma: Female\n",
1204
+ " mistral: Female\n",
1205
+ " qwen: Female\n",
1206
+ "Profession (agreement: 2/3):\n",
1207
+ " Consensus: singer/musician, model, online personality\n",
1208
+ " gemma: singer/musician, model, online personality\n",
1209
+ " mistral: singer/musician, tv personality, kpop idol\n",
1210
+ " qwen: singer/musician, model, online personality\n",
1211
+ "\n",
1212
+ "============================================================\n",
1213
+ "COMPLETE!\n",
1214
+ "============================================================\n"
1215
+ ]
1216
+ }
1217
+ ],
1218
+ "source": [
1219
+ "# Script to create strict consensus CSV\n",
1220
+ "# Only keeps rows where:\n",
1221
+ "# - All 3 models agree on country\n",
1222
+ "# - All 3 models agree on gender\n",
1223
+ "# - At least 2 models agree on profession\n",
1224
+ "\n",
1225
+ "import pandas as pd\n",
1226
+ "from pathlib import Path\n",
1227
+ "from collections import Counter\n",
1228
+ "\n",
1229
+ "def normalize_value(value):\n",
1230
+ " \"\"\"Normalize values for comparison (handle NaN, whitespace, case)\"\"\"\n",
1231
+ " if pd.isna(value):\n",
1232
+ " return None\n",
1233
+ " return str(value).strip().lower()\n",
1234
+ "\n",
1235
+ "def is_unknown_value(value):\n",
1236
+ " \"\"\"Check if a value represents 'unknown' or similar non-informative values\"\"\"\n",
1237
+ " if value is None:\n",
1238
+ " return True\n",
1239
+ " \n",
1240
+ " value_str = str(value).strip().lower()\n",
1241
+ " \n",
1242
+ " # List of patterns that indicate unknown/missing data\n",
1243
+ " unknown_patterns = [\n",
1244
+ " 'unknown',\n",
1245
+ " 'n/a',\n",
1246
+ " 'na',\n",
1247
+ " 'none',\n",
1248
+ " 'not specified',\n",
1249
+ " 'not available',\n",
1250
+ " 'unclear',\n",
1251
+ " 'uncertain',\n",
1252
+ " '',\n",
1253
+ " 'null'\n",
1254
+ " ]\n",
1255
+ " \n",
1256
+ " return value_str in unknown_patterns\n",
1257
+ "\n",
1258
+ "def get_consensus_value(values, required_agreement=2):\n",
1259
+ " \"\"\"\n",
1260
+ " Get consensus value if at least 'required_agreement' models agree.\n",
1261
+ " Returns (consensus_value, count_agreeing, all_values_dict)\n",
1262
+ " \"\"\"\n",
1263
+ " # Normalize values\n",
1264
+ " normalized = [normalize_value(v) for v in values]\n",
1265
+ " \n",
1266
+ " # Count occurrences (excluding None)\n",
1267
+ " valid_values = [v for v in normalized if v is not None]\n",
1268
+ " \n",
1269
+ " if not valid_values:\n",
1270
+ " return None, 0, {}\n",
1271
+ " \n",
1272
+ " value_counts = Counter(valid_values)\n",
1273
+ " most_common_value, count = value_counts.most_common(1)[0]\n",
1274
+ " \n",
1275
+ " # Create a dict of all values for inspection\n",
1276
+ " all_values = {f\"model_{i+1}\": values[i] for i in range(len(values))}\n",
1277
+ " \n",
1278
+ " if count >= required_agreement:\n",
1279
+ " return most_common_value, count, all_values\n",
1280
+ " else:\n",
1281
+ " return None, count, all_values\n",
1282
+ "\n",
1283
+ "def create_strict_consensus(input_file, output_file, models=['gemma', 'mistral', 'qwen']):\n",
1284
+ " \"\"\"\n",
1285
+ " Create a strict consensus CSV with only rows where:\n",
1286
+ " - All 3 models agree on country\n",
1287
+ " - All 3 models agree on gender\n",
1288
+ " - At least 2 models agree on profession\n",
1289
+ " \"\"\"\n",
1290
+ " print(\"=\"*60)\n",
1291
+ " print(\"STRICT CONSENSUS FILTER\")\n",
1292
+ " print(\"=\"*60)\n",
1293
+ " print(f\"Reading: {input_file}\")\n",
1294
+ " \n",
1295
+ " # Read the combined file\n",
1296
+ " df = pd.read_csv(input_file)\n",
1297
+ " \n",
1298
+ " print(f\"Input shape: {df.shape}\")\n",
1299
+ " print(f\"Models to check: {', '.join(models)}\")\n",
1300
+ " \n",
1301
+ " # Initialize lists to store results\n",
1302
+ " rows_to_keep = []\n",
1303
+ " consensus_data = []\n",
1304
+ " \n",
1305
+ " # Stats tracking\n",
1306
+ " stats = {\n",
1307
+ " 'total_rows': len(df),\n",
1308
+ " 'country_fail': 0,\n",
1309
+ " 'gender_fail': 0,\n",
1310
+ " 'profession_fail': 0,\n",
1311
+ " 'unknown_values': 0,\n",
1312
+ " 'all_pass': 0\n",
1313
+ " }\n",
1314
+ " \n",
1315
+ " print(\"\\nProcessing rows...\")\n",
1316
+ " \n",
1317
+ " for idx, row in df.iterrows():\n",
1318
+ " if idx % 1000 == 0:\n",
1319
+ " print(f\" Processed {idx}/{len(df)} rows...\", end='\\r')\n",
1320
+ " \n",
1321
+ " # Get values for each field from all models\n",
1322
+ " countries = [row[f'{model}_country'] for model in models]\n",
1323
+ " genders = [row[f'{model}_gender'] for model in models]\n",
1324
+ " professions = [row[f'{model}_profession_llm'] for model in models]\n",
1325
+ " \n",
1326
+ " # Check country consensus (all 3 must agree)\n",
1327
+ " country_consensus, country_count, country_vals = get_consensus_value(countries, required_agreement=3)\n",
1328
+ " \n",
1329
+ " # Check gender consensus (all 3 must agree)\n",
1330
+ " gender_consensus, gender_count, gender_vals = get_consensus_value(genders, required_agreement=3)\n",
1331
+ " \n",
1332
+ " # Check profession consensus (at least 2 must agree)\n",
1333
+ " profession_consensus, profession_count, profession_vals = get_consensus_value(professions, required_agreement=2)\n",
1334
+ " \n",
1335
+ " # Determine if row passes all criteria\n",
1336
+ " country_pass = country_count == 3\n",
1337
+ " gender_pass = gender_count == 3\n",
1338
+ " profession_pass = profession_count >= 2\n",
1339
+ " \n",
1340
+ " # Check if any consensus value is \"unknown\" or similar\n",
1341
+ " has_unknown = (\n",
1342
+ " is_unknown_value(country_consensus) or \n",
1343
+ " is_unknown_value(gender_consensus) or \n",
1344
+ " is_unknown_value(profession_consensus)\n",
1345
+ " )\n",
1346
+ " \n",
1347
+ " # Track failures\n",
1348
+ " if not country_pass:\n",
1349
+ " stats['country_fail'] += 1\n",
1350
+ " if not gender_pass:\n",
1351
+ " stats['gender_fail'] += 1\n",
1352
+ " if not profession_pass:\n",
1353
+ " stats['profession_fail'] += 1\n",
1354
+ " if has_unknown:\n",
1355
+ " stats['unknown_values'] += 1\n",
1356
+ " \n",
1357
+ " # Only keep row if all criteria pass AND no unknown values\n",
1358
+ " if country_pass and gender_pass and profession_pass and not has_unknown:\n",
1359
+ " rows_to_keep.append(idx)\n",
1360
+ " stats['all_pass'] += 1\n",
1361
+ " \n",
1362
+ " consensus_data.append({\n",
1363
+ " 'consensus_country': country_consensus,\n",
1364
+ " 'consensus_gender': gender_consensus,\n",
1365
+ " 'consensus_profession': profession_consensus,\n",
1366
+ " 'profession_agreement_count': profession_count\n",
1367
+ " })\n",
1368
+ " \n",
1369
+ " print(f\"\\n Processed {len(df)} rows. \")\n",
1370
+ " \n",
1371
+ " # Create consensus dataframe\n",
1372
+ " if rows_to_keep:\n",
1373
+ " # Get the original data for kept rows\n",
1374
+ " result_df = df.iloc[rows_to_keep].copy().reset_index(drop=True)\n",
1375
+ " \n",
1376
+ " # Add consensus columns at the beginning (after common columns)\n",
1377
+ " consensus_df = pd.DataFrame(consensus_data)\n",
1378
+ " \n",
1379
+ " # Find where to insert consensus columns (after 'name' if it exists, otherwise at start)\n",
1380
+ " if 'name' in result_df.columns:\n",
1381
+ " name_idx = result_df.columns.get_loc('name') + 1\n",
1382
+ " else:\n",
1383
+ " name_idx = 0\n",
1384
+ " \n",
1385
+ " # Insert consensus columns\n",
1386
+ " for i, col in enumerate(consensus_df.columns):\n",
1387
+ " result_df.insert(name_idx + i, col, consensus_df[col])\n",
1388
+ " \n",
1389
+ " # Save the result\n",
1390
+ " result_df.to_csv(output_file, index=False)\n",
1391
+ " \n",
1392
+ " print(\"\\n\" + \"=\"*60)\n",
1393
+ " print(\"FILTERING RESULTS\")\n",
1394
+ " print(\"=\"*60)\n",
1395
+ " print(f\"Total input rows: {stats['total_rows']:,}\")\n",
1396
+ " print(f\"Rows passing all criteria: {stats['all_pass']:,} ({stats['all_pass']/stats['total_rows']*100:.1f}%)\")\n",
1397
+ " print(f\"\\nFailure reasons (rows can fail multiple):\")\n",
1398
+ " print(f\" - Country disagreement: {stats['country_fail']:,} ({stats['country_fail']/stats['total_rows']*100:.1f}%)\")\n",
1399
+ " print(f\" - Gender disagreement: {stats['gender_fail']:,} ({stats['gender_fail']/stats['total_rows']*100:.1f}%)\")\n",
1400
+ " print(f\" - Profession disagreement: {stats['profession_fail']:,} ({stats['profession_fail']/stats['total_rows']*100:.1f}%)\")\n",
1401
+ " print(f\" - Contains 'Unknown': {stats['unknown_values']:,} ({stats['unknown_values']/stats['total_rows']*100:.1f}%)\")\n",
1402
+ " \n",
1403
+ " print(\"\\n\" + \"=\"*60)\n",
1404
+ " print(\"CONSENSUS DISTRIBUTIONS\")\n",
1405
+ " print(\"=\"*60)\n",
1406
+ " \n",
1407
+ " print(\"\\nCountry (top 10):\")\n",
1408
+ " print(result_df['consensus_country'].value_counts().head(10))\n",
1409
+ " \n",
1410
+ " print(\"\\nGender:\")\n",
1411
+ " print(result_df['consensus_gender'].value_counts())\n",
1412
+ " \n",
1413
+ " print(\"\\nProfession (top 10):\")\n",
1414
+ " print(result_df['consensus_profession'].value_counts().head(10))\n",
1415
+ " \n",
1416
+ " print(\"\\nProfession agreement level:\")\n",
1417
+ " print(result_df['profession_agreement_count'].value_counts().sort_index())\n",
1418
+ " \n",
1419
+ " print(\"\\n\" + \"=\"*60)\n",
1420
+ " print(f\"✓ Strict consensus file saved to: {output_file.name}\")\n",
1421
+ " print(f\" Total rows: {len(result_df):,}\")\n",
1422
+ " print(f\" Total columns: {len(result_df.columns)}\")\n",
1423
+ " print(\"=\"*60)\n",
1424
+ " \n",
1425
+ " return result_df\n",
1426
+ " else:\n",
1427
+ " print(\"\\n⚠ WARNING: No rows passed all criteria!\")\n",
1428
+ " print(\"Creating empty file with proper columns...\")\n",
1429
+ " \n",
1430
+ " # Create empty dataframe with proper structure\n",
1431
+ " result_df = df.iloc[:0].copy()\n",
1432
+ " consensus_cols = ['consensus_country', 'consensus_gender', 'consensus_profession', 'profession_agreement_count']\n",
1433
+ " for col in consensus_cols:\n",
1434
+ " result_df.insert(0, col, [])\n",
1435
+ " \n",
1436
+ " result_df.to_csv(output_file, index=False)\n",
1437
+ " return result_df\n",
1438
+ "\n",
1439
+ "def show_sample_comparisons(df, models, n_samples=5):\n",
1440
+ " \"\"\"Show sample comparisons between models for quality check\"\"\"\n",
1441
+ " print(\"\\n\" + \"=\"*60)\n",
1442
+ " print(f\"SAMPLE COMPARISONS (first {n_samples} rows)\")\n",
1443
+ " print(\"=\"*60)\n",
1444
+ " \n",
1445
+ " if len(df) == 0:\n",
1446
+ " print(\"No data to display\")\n",
1447
+ " return\n",
1448
+ " \n",
1449
+ " sample_df = df.head(n_samples)\n",
1450
+ " \n",
1451
+ " for idx, row in sample_df.iterrows():\n",
1452
+ " print(f\"\\n--- Row {idx + 1}: {row.get('name', 'N/A')} ---\")\n",
1453
+ " \n",
1454
+ " # Country comparison\n",
1455
+ " print(\"Country:\")\n",
1456
+ " print(f\" Consensus: {row['consensus_country']}\")\n",
1457
+ " for model in models:\n",
1458
+ " col = f'{model}_country'\n",
1459
+ " if col in row:\n",
1460
+ " print(f\" {model}: {row[col]}\")\n",
1461
+ " \n",
1462
+ " # Gender comparison\n",
1463
+ " print(\"Gender:\")\n",
1464
+ " print(f\" Consensus: {row['consensus_gender']}\")\n",
1465
+ " for model in models:\n",
1466
+ " col = f'{model}_gender'\n",
1467
+ " if col in row:\n",
1468
+ " print(f\" {model}: {row[col]}\")\n",
1469
+ " \n",
1470
+ " # Profession comparison\n",
1471
+ " print(f\"Profession (agreement: {row['profession_agreement_count']}/3):\")\n",
1472
+ " print(f\" Consensus: {row['consensus_profession']}\")\n",
1473
+ " for model in models:\n",
1474
+ " col = f'{model}_profession_llm'\n",
1475
+ " if col in row:\n",
1476
+ " print(f\" {model}: {row[col]}\")\n",
1477
+ "\n",
1478
+ "# ============================================================\n",
1479
+ "# MAIN EXECUTION\n",
1480
+ "# ============================================================\n",
1481
+ "\n",
1482
+ "if __name__ == \"__main__\":\n",
1483
+ " # Get current directory\n",
1484
+ " current_dir = Path.cwd()\n",
1485
+ " \n",
1486
+ " # Define input file (the combined file from previous script)\n",
1487
+ " input_file = current_dir.parent / \"data/CSV/combined_llm_annotations.csv\"\n",
1488
+ " \n",
1489
+ " # Define output file\n",
1490
+ " output_file = current_dir.parent / \"data/CSV/strict_consensus.csv\"\n",
1491
+ " \n",
1492
+ " # Check if input file exists\n",
1493
+ " if not input_file.exists():\n",
1494
+ " print(f\"Error: Input file not found: {input_file}\")\n",
1495
+ " print(\"Please run the combine script first to create combined_llm_annotations.csv\")\n",
1496
+ " else:\n",
1497
+ " # Models to check\n",
1498
+ " models = ['gemma', 'mistral', 'qwen']\n",
1499
+ " \n",
1500
+ " # Create strict consensus file\n",
1501
+ " result_df = create_strict_consensus(input_file, output_file, models)\n",
1502
+ " \n",
1503
+ " # Show sample comparisons\n",
1504
+ " if len(result_df) > 0:\n",
1505
+ " show_sample_comparisons(result_df, models, n_samples=5)\n",
1506
+ " \n",
1507
+ " print(\"\\n\" + \"=\"*60)\n",
1508
+ " print(\"COMPLETE!\")\n",
1509
+ " print(\"=\"*60)"
1510
+ ]
1511
+ },
1512
  {
1513
  "cell_type": "markdown",
1514
  "id": "e0f8f331-a6f3-49e8-9734-ee7ffafebabc",
 
1540
  },
1541
  {
1542
  "cell_type": "code",
1543
+ "execution_count": 4,
1544
  "id": "fe38cf23-771d-4888-a5ed-f2b8beb17017",
1545
  "metadata": {
1546
  "execution": {
1547
+ "iopub.execute_input": "2025-12-08T10:58:11.949915Z",
1548
+ "iopub.status.busy": "2025-12-08T10:58:11.949461Z",
1549
+ "iopub.status.idle": "2025-12-08T10:58:29.470022Z",
1550
+ "shell.execute_reply": "2025-12-08T10:58:29.468836Z",
1551
+ "shell.execute_reply.started": "2025-12-08T10:58:11.949880Z"
1552
  }
1553
  },
1554
  "outputs": [
 
1556
  "name": "stdout",
1557
  "output_type": "stream",
1558
  "text": [
1559
+ "Loading combined data from: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/combined_llm_annotations.csv\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1560
  "Loaded 50,861 rows with 67 columns\n",
1561
  "Analyzing agreement between models: gemma, qwen, mistral\n",
1562
  "Total rows to analyze: 50861\n",
 
1566
  "============================================================\n",
1567
  "\n",
1568
  "Total rows analyzed: 50,861\n",
1569
+ "Rows with at least 2 models MEANINGFULLY agreeing (valid): 39,434 (77.5%)\n",
1570
+ "Rows with all 3 models MEANINGFULLY agreeing: 21,459 (42.2%)\n",
1571
+ "Rows with all-unknown values for any field: 2,010 (4.0%)\n",
1572
  "\n",
1573
  "Field-wise ANY agreement (including unknown matches):\n",
1574
+ " - Country: 46,508 rows (91.4%)\n",
1575
+ " - Gender: 49,884 rows (98.1%)\n",
1576
+ " - Profession: 50,509 rows (99.3%)\n",
1577
+ " - Name: 44,504 rows (87.5%)\n",
1578
  "\n",
1579
  "Field-wise MEANINGFUL agreement (excluding unknown matches):\n",
1580
+ " - Country: 39,979 rows (78.6%)\n",
1581
+ " - Gender: 49,197 rows (96.7%)\n",
1582
+ " - Profession: 50,509 rows (99.3%)\n",
1583
+ " - Name: 44,504 rows (87.5%)\n",
1584
  "\n",
1585
  "Examples of successful name matches with variations:\n",
1586
  " Row 0:\n",
 
1604
  "Number of MEANINGFUL agreeing pairs per field (out of 3 possible pairs):\n",
1605
  "\n",
1606
  "Country:\n",
1607
+ " - 0 meaningful agreeing pairs: 10,882 rows (21.4%)\n",
1608
+ " - 1 meaningful agreeing pairs: 16,159 rows (31.8%)\n",
1609
+ " - 3 meaningful agreeing pairs: 23,820 rows (46.8%)\n",
1610
  "\n",
1611
  "Gender:\n",
1612
+ " - 0 meaningful agreeing pairs: 1,664 rows (3.3%)\n",
1613
+ " - 1 meaningful agreeing pairs: 5,190 rows (10.2%)\n",
1614
+ " - 3 meaningful agreeing pairs: 44,007 rows (86.5%)\n",
1615
  "\n",
1616
  "Profession:\n",
1617
+ " - 0 meaningful agreeing pairs: 352 rows (0.7%)\n",
1618
+ " - 1 meaningful agreeing pairs: 3,916 rows (7.7%)\n",
1619
+ " - 2 meaningful agreeing pairs: 3,239 rows (6.4%)\n",
1620
+ " - 3 meaningful agreeing pairs: 43,354 rows (85.2%)\n",
1621
  "\n",
1622
  "Name:\n",
1623
+ " - 0 meaningful agreeing pairs: 6,357 rows (12.5%)\n",
1624
+ " - 1 meaningful agreeing pairs: 7,484 rows (14.7%)\n",
1625
+ " - 2 meaningful agreeing pairs: 774 rows (1.5%)\n",
1626
+ " - 3 meaningful agreeing pairs: 36,246 rows (71.3%)\n",
1627
+ "\n",
1628
+ "Adding consensus columns...\n",
1629
+ "\n",
1630
+ "Consensus column statistics:\n",
1631
+ " - Consensus country: 48,970 rows (96.3%)\n",
1632
+ " - Consensus gender: 50,775 rows (99.8%)\n",
1633
+ " - Consensus primary profession: 50,849 rows (100.0%)\n",
1634
+ "\n",
1635
+ "Examples of primary profession consensus (first 5 rows with consensus):\n",
1636
+ "\n",
1637
+ " Row 0:\n",
1638
+ " gemma: singer/musician (from singer/musician, actor, tv personality)\n",
1639
+ " qwen: singer/musician (from singer/musician, model, public figure)\n",
1640
+ " mistral: singer/musician (from singer/musician, tv personality, actress)\n",
1641
+ " → Consensus: singer/musician\n",
1642
+ "\n",
1643
+ " Row 1:\n",
1644
+ " gemma: model (from model, adult performer, online personality)\n",
1645
+ " qwen: model (from model, online personality, artist)\n",
1646
+ " mistral: model (from model, online personality, actress)\n",
1647
+ " → Consensus: model\n",
1648
+ "\n",
1649
+ " Row 2:\n",
1650
+ " gemma: singer/musician (from singer/musician, actor, model)\n",
1651
+ " qwen: model (from model, online personality)\n",
1652
+ " mistral: model (from model, online personality, artist)\n",
1653
+ " → Consensus: model\n",
1654
+ "\n",
1655
+ " Row 3:\n",
1656
+ " gemma: model (from model, online personality, actor)\n",
1657
+ " qwen: singer/musician (from singer/musician, model, public figure)\n",
1658
+ " mistral: actor (from actor, model, tv personality)\n",
1659
+ " → Consensus: model\n",
1660
+ "\n",
1661
+ " Row 4:\n",
1662
+ " gemma: singer/musician (from singer/musician, tv personality, online personality)\n",
1663
+ " qwen: singer/musician (from singer/musician, model, online personality)\n",
1664
+ " mistral: singer/musician (from singer/musician, tv personality, public figure)\n",
1665
+ " → Consensus: singer/musician\n",
1666
  "\n",
1667
  "Saved analyzed data to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement.csv\n",
1668
+ "Saved valid rows (39,434) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_valid.csv\n",
1669
+ "Saved MEANINGFUL consensus rows (21,459) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_consensus.csv\n",
1670
+ "Saved all-unknown rows (2,010) to: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/analyzed_llm_agreement_all_unknown.csv\n",
1671
  "\n",
1672
  "============================================================\n",
1673
  "SAMPLE OF MEANINGFUL CONSENSUS ROWS (first 2 rows)\n",
1674
  "============================================================\n",
1675
  "\n",
1676
  "Consensus Row 0:\n",
1677
+ " Consensus Country: South Korea\n",
1678
+ " Consensus Gender: Female\n",
1679
+ " Consensus Primary Profession: singer/musician\n",
1680
+ "\n",
1681
+ " Individual Model Outputs:\n",
1682
+ " gemma:\n",
1683
+ " Name: Lee Ji-eun\n",
1684
+ " Country: South Korea\n",
1685
+ " Gender: Female\n",
1686
+ " Profession: singer/musician, actor, tv personality\n",
1687
+ " qwen:\n",
1688
+ " Name: Lee Ji-eun\n",
1689
+ " Country: South Korea\n",
1690
+ " Gender: Female\n",
1691
+ " Profession: singer/musician, model, public figure\n",
1692
+ " mistral:\n",
1693
+ " Name: Lee Ji-eun (Lee, Ji-eun)\n",
1694
+ " Country: South Korea\n",
1695
+ " Gender: Female\n",
1696
+ " Profession: singer/musician, tv personality, actress\n",
1697
  "\n",
1698
  "Consensus Row 4:\n",
1699
+ " Consensus Country: South Korea\n",
1700
+ " Consensus Gender: Female\n",
1701
+ " Consensus Primary Profession: singer/musician\n",
1702
+ "\n",
1703
+ " Individual Model Outputs:\n",
1704
+ " gemma:\n",
1705
+ " Name: Yoo Jimin\n",
1706
+ " Country: South Korea\n",
1707
+ " Gender: Female\n",
1708
+ " Profession: singer/musician, tv personality, online personality\n",
1709
+ " qwen:\n",
1710
+ " Name: Kim Karina\n",
1711
+ " Country: South Korea\n",
1712
+ " Gender: Female\n",
1713
+ " Profession: singer/musician, model, online personality\n",
1714
+ " mistral:\n",
1715
+ " Name: Karina (Kim Jung-yeon)\n",
1716
+ " Country: South Korea\n",
1717
+ " Gender: Female\n",
1718
+ " Profession: singer/musician, tv personality, public figure\n"
1719
  ]
1720
  }
1721
  ],
 
1724
  "import numpy as np\n",
1725
  "import re\n",
1726
  "from typing import List, Set, Tuple\n",
1727
+ "from collections import Counter\n",
1728
+ "from pathlib import Path\n",
1729
  "\n",
1730
  "def is_unknown_value(value) -> bool:\n",
1731
  " \"\"\"Check if a value is considered 'unknown' or empty.\"\"\"\n",
 
2009
  " \n",
2010
  " return results\n",
2011
  "\n",
2012
+ "def extract_consensus_values(row: pd.Series, models: List[str]) -> Tuple[str, str, str]:\n",
2013
+ " \"\"\"\n",
2014
+ " Extract consensus values for country, gender, and primary profession.\n",
2015
+ " \n",
2016
+ " Returns:\n",
2017
+ " Tuple of (consensus_country, consensus_gender, consensus_primary_profession)\n",
2018
+ " \"\"\"\n",
2019
+ " # Extract country and gender (straightforward - they're the same in consensus rows)\n",
2020
+ " countries = [row.get(f'{model}_country', None) for model in models]\n",
2021
+ " genders = [row.get(f'{model}_gender', None) for model in models]\n",
2022
+ " \n",
2023
+ " # Get first non-null, non-unknown country and gender\n",
2024
+ " consensus_country = None\n",
2025
+ " for country in countries:\n",
2026
+ " if pd.notna(country) and not is_unknown_value(country):\n",
2027
+ " consensus_country = country\n",
2028
+ " break\n",
2029
+ " \n",
2030
+ " consensus_gender = None\n",
2031
+ " for gender in genders:\n",
2032
+ " if pd.notna(gender) and not is_unknown_value(gender):\n",
2033
+ " consensus_gender = gender\n",
2034
+ " break\n",
2035
+ " \n",
2036
+ " # Extract primary profession (first profession from each model)\n",
2037
+ " primary_professions = []\n",
2038
+ " all_professions = [] # Track all professions for frequency counting\n",
2039
+ " \n",
2040
+ " for model in models:\n",
2041
+ " prof_col = f'{model}_profession_llm'\n",
2042
+ " prof_str = row.get(prof_col, None)\n",
2043
+ " \n",
2044
+ " if pd.notna(prof_str) and not is_unknown_value(prof_str):\n",
2045
+ " # Split by comma and get professions\n",
2046
+ " professions = [p.strip().lower() for p in str(prof_str).split(',') if p.strip()]\n",
2047
+ " \n",
2048
+ " if professions:\n",
2049
+ " # First profession is the primary one\n",
2050
+ " primary_professions.append(professions[0])\n",
2051
+ " # Track all professions for frequency counting\n",
2052
+ " all_professions.extend(professions)\n",
2053
+ " \n",
2054
+ " # Determine consensus primary profession\n",
2055
+ " consensus_primary_profession = None\n",
2056
+ " \n",
2057
+ " if len(primary_professions) >= 2:\n",
2058
+ " # Check if 2+ models agree on the same primary profession\n",
2059
+ " primary_counter = Counter(primary_professions)\n",
2060
+ " most_common_primary = primary_counter.most_common(1)[0]\n",
2061
+ " \n",
2062
+ " # If 2+ models agree on the primary profession, use it\n",
2063
+ " if most_common_primary[1] >= 2:\n",
2064
+ " consensus_primary_profession = most_common_primary[0]\n",
2065
+ " else:\n",
2066
+ " # All different - use the profession that appears most across ALL professions\n",
2067
+ " all_counter = Counter(all_professions)\n",
2068
+ " if all_counter:\n",
2069
+ " consensus_primary_profession = all_counter.most_common(1)[0][0]\n",
2070
+ " \n",
2071
+ " return consensus_country, consensus_gender, consensus_primary_profession\n",
2072
+ "\n",
2073
+ "def add_consensus_columns(df_analyzed: pd.DataFrame, models: List[str]) -> pd.DataFrame:\n",
2074
+ " \"\"\"\n",
2075
+ " Add consensus columns for country, gender, and primary profession.\n",
2076
+ " \"\"\"\n",
2077
+ " print(\"\\nAdding consensus columns...\")\n",
2078
+ " \n",
2079
+ " consensus_data = []\n",
2080
+ " for idx, row in df_analyzed.iterrows():\n",
2081
+ " country, gender, profession = extract_consensus_values(row, models)\n",
2082
+ " consensus_data.append({\n",
2083
+ " 'consensus_country': country,\n",
2084
+ " 'consensus_gender': gender,\n",
2085
+ " 'consensus_primary_profession': profession\n",
2086
+ " })\n",
2087
+ " \n",
2088
+ " consensus_df = pd.DataFrame(consensus_data)\n",
2089
+ " \n",
2090
+ " # Add columns to the analyzed dataframe\n",
2091
+ " df_with_consensus = df_analyzed.copy()\n",
2092
+ " df_with_consensus['consensus_country'] = consensus_df['consensus_country']\n",
2093
+ " df_with_consensus['consensus_gender'] = consensus_df['consensus_gender']\n",
2094
+ " df_with_consensus['consensus_primary_profession'] = consensus_df['consensus_primary_profession']\n",
2095
+ " \n",
2096
+ " # Print statistics about consensus values\n",
2097
+ " total_rows = len(df_with_consensus)\n",
2098
+ " \n",
2099
+ " # Count non-null consensus values\n",
2100
+ " country_count = df_with_consensus['consensus_country'].notna().sum()\n",
2101
+ " gender_count = df_with_consensus['consensus_gender'].notna().sum()\n",
2102
+ " profession_count = df_with_consensus['consensus_primary_profession'].notna().sum()\n",
2103
+ " \n",
2104
+ " print(f\"\\nConsensus column statistics:\")\n",
2105
+ " print(f\" - Consensus country: {country_count:,} rows ({country_count/total_rows*100:.1f}%)\")\n",
2106
+ " print(f\" - Consensus gender: {gender_count:,} rows ({gender_count/total_rows*100:.1f}%)\")\n",
2107
+ " print(f\" - Consensus primary profession: {profession_count:,} rows ({profession_count/total_rows*100:.1f}%)\")\n",
2108
+ " \n",
2109
+ " # Show examples of profession consensus\n",
2110
+ " print(\"\\nExamples of primary profession consensus (first 5 rows with consensus):\")\n",
2111
+ " sample_rows = df_with_consensus[df_with_consensus['consensus_primary_profession'].notna()].head(5)\n",
2112
+ " for idx, row in sample_rows.iterrows():\n",
2113
+ " print(f\"\\n Row {idx}:\")\n",
2114
+ " for model in models:\n",
2115
+ " prof_col = f'{model}_profession_llm'\n",
2116
+ " if prof_col in row and pd.notna(row[prof_col]):\n",
2117
+ " profs = [p.strip() for p in str(row[prof_col]).split(',')]\n",
2118
+ " print(f\" {model}: {profs[0] if profs else 'N/A'} (from {row[prof_col]})\")\n",
2119
+ " print(f\" → Consensus: {row['consensus_primary_profession']}\")\n",
2120
+ " \n",
2121
+ " return df_with_consensus\n",
2122
+ "\n",
2123
  "def analyze_model_agreement(df: pd.DataFrame, models: List[str]) -> pd.DataFrame:\n",
2124
  " \"\"\"\n",
2125
  " Analyze agreement between models and add agreement columns.\n",
 
2279
  " print(f\"Saved all-unknown rows ({len(unknown_df):,}) to: {unknown_path}\")\n",
2280
  "\n",
2281
  "# ============================================================\n",
2282
+ "# EXAMPLE USAGE\n",
2283
  "# ============================================================\n",
2284
  "\n",
2285
+ "if __name__ == \"__main__\":\n",
2286
+ " # Set up paths (adjust to your directory structure)\n",
2287
+ " current_dir = path.cwd()\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2288
  " \n",
2289
+ " # Load the combined data\n",
2290
+ " combined_file = current_dir.parent / \"data/CSV/combined_llm_annotations.csv\"\n",
2291
+ " print(f\"Loading combined data from: {combined_file}\")\n",
2292
  " \n",
2293
+ " if combined_file.exists():\n",
2294
+ " df_combined = pd.read_csv(combined_file)\n",
2295
+ " print(f\"Loaded {len(df_combined):,} rows with {len(df_combined.columns)} columns\")\n",
2296
+ " \n",
2297
+ " # Define the models\n",
2298
+ " models = ['gemma', 'qwen', 'mistral']\n",
2299
+ " \n",
2300
+ " # Run the agreement analysis\n",
2301
+ " df_analyzed = analyze_model_agreement(df_combined, models)\n",
2302
+ " \n",
2303
+ " # Add consensus columns\n",
2304
+ " df_with_consensus = add_consensus_columns(df_analyzed, models)\n",
2305
+ " \n",
2306
+ " # Save the results\n",
2307
+ " output_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement.csv\"\n",
2308
+ " save_analysis_results(df_with_consensus, output_file)\n",
2309
+ " \n",
2310
+ " # Show sample of consensus rows\n",
2311
+ " print(\"\\n\" + \"=\"*60)\n",
2312
+ " print(\"SAMPLE OF MEANINGFUL CONSENSUS ROWS (first 2 rows)\")\n",
2313
+ " print(\"=\"*60)\n",
2314
+ " \n",
2315
+ " consensus_rows = df_with_consensus[df_with_consensus['all_models_agree']].head(2)\n",
2316
+ " \n",
2317
+ " if len(consensus_rows) > 0:\n",
2318
+ " for idx, row in consensus_rows.iterrows():\n",
2319
+ " print(f\"\\nConsensus Row {idx}:\")\n",
2320
+ " print(f\" Consensus Country: {row['consensus_country']}\")\n",
2321
+ " print(f\" Consensus Gender: {row['consensus_gender']}\")\n",
2322
+ " print(f\" Consensus Primary Profession: {row['consensus_primary_profession']}\")\n",
2323
+ " print(f\"\\n Individual Model Outputs:\")\n",
2324
+ " for model in models:\n",
2325
+ " name_col = f'{model}_full_name'\n",
2326
+ " country_col = f'{model}_country'\n",
2327
+ " gender_col = f'{model}_gender'\n",
2328
+ " prof_col = f'{model}_profession_llm'\n",
2329
+ " \n",
2330
+ " if name_col in row:\n",
2331
+ " print(f\" {model}:\")\n",
2332
+ " print(f\" Name: {row[name_col]}\")\n",
2333
+ " print(f\" Country: {row[country_col]}\")\n",
2334
+ " print(f\" Gender: {row[gender_col]}\")\n",
2335
+ " print(f\" Profession: {row[prof_col]}\")\n",
2336
+ " else:\n",
2337
+ " print(f\"Error: Could not find combined data file at {combined_file}\")\n",
2338
+ " print(\"Please adjust the path in the script to point to your data file.\")"
2339
  ]
2340
  },
2341
  {
2342
  "cell_type": "code",
2343
  "execution_count": null,
2344
+ "id": "70403f7c-6ea9-4f21-9704-aec0c37a591b",
2345
  "metadata": {},
2346
  "outputs": [],
2347
  "source": []
misc/query_indicies/gemma_local_query_index.txt CHANGED
@@ -1 +1 @@
1
- 49840
 
1
+ 50861
misc/query_indicies/mistral_local_query_index.txt CHANGED
@@ -1 +1 @@
1
- 30550
 
1
+ 50861
misc/query_indicies/qwen_local_query_index.txt CHANGED
@@ -1 +1 @@
1
- 44060
 
1
+ 50861