Laura Wagner commited on
Commit
5fabb43
·
1 Parent(s): d317593

change visualization to bar chart

Browse files
jupyter_notebooks/Section_2-3-4_Figure_8_Step_1_LLM_annotation.ipynb CHANGED
@@ -10,14 +10,496 @@
10
  },
11
  {
12
  "cell_type": "markdown",
 
13
  "metadata": {},
14
  "source": [
15
- "### Unified Model Loading & Inference\nAbstracted interface for loading and querying Mistral, Gemma, and Qwen models."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  ]
17
  },
18
  {
19
  "cell_type": "code",
 
 
20
  "metadata": {},
 
21
  "source": [
22
  "import pandas as pd\n",
23
  "import json\n",
@@ -86,13 +568,22 @@
86
  " \"sports professional\",\n",
87
  " \"tv personality\"\n",
88
  "]\n"
89
- ],
90
- "execution_count": null,
91
- "outputs": []
 
 
 
 
 
 
92
  },
93
  {
94
  "cell_type": "code",
 
 
95
  "metadata": {},
 
96
  "source": [
97
  "def load_model(model_type='mistral'):\n",
98
  " \"\"\"\n",
@@ -156,13 +647,22 @@
156
  " print(f\"VRAM used: {vram_gb:.2f} GB\\n\")\n",
157
  " \n",
158
  " return model, tokenizer, config\n"
159
- ],
160
- "execution_count": null,
161
- "outputs": []
 
 
 
 
 
 
162
  },
163
  {
164
  "cell_type": "code",
 
 
165
  "metadata": {},
 
166
  "source": [
167
  "@contextmanager\n",
168
  "def timeout(duration):\n",
@@ -239,13 +739,22 @@
239
  " except Exception as e:\n",
240
  " print(f\"[ERROR] Generation failed: {e}\")\n",
241
  " return None\n"
242
- ],
243
- "execution_count": null,
244
- "outputs": []
 
 
 
 
 
 
245
  },
246
  {
247
  "cell_type": "code",
 
 
248
  "metadata": {},
 
249
  "source": [
250
  "def create_prompt(row):\n",
251
  " \"\"\"Create annotation prompt from row data.\"\"\"\n",
@@ -298,13 +807,22 @@
298
  "3. Female\n",
299
  "4. singer/musician, public figure\n",
300
  "5. United States\"\"\"\n"
301
- ],
302
- "execution_count": null,
303
- "outputs": []
 
 
 
 
 
 
304
  },
305
  {
306
  "cell_type": "code",
 
 
307
  "metadata": {},
 
308
  "source": [
309
  "def parse_response(response):\n",
310
  " \"\"\"Parse model response into structured fields.\"\"\"\n",
@@ -345,13 +863,22 @@
345
  " fields['country'] = country_raw\n",
346
  " \n",
347
  " return fields\n"
348
- ],
349
- "execution_count": null,
350
- "outputs": []
 
 
 
 
 
 
351
  },
352
  {
353
  "cell_type": "code",
 
 
354
  "metadata": {},
 
355
  "source": [
356
  "def annotate_dataset(model_type='mistral', test_mode=False, test_size=100, max_rows=50862, save_interval=10):\n",
357
  " \"\"\"\n",
@@ -454,20 +981,23 @@
454
  " df.to_csv(output_file, index=False)\n",
455
  " index_file.write_text(str(current_index))\n",
456
  " print(f\"✓ Finished annotation with {model_type}\")\n"
457
- ],
458
- "execution_count": null,
459
- "outputs": []
460
  },
461
  {
462
  "cell_type": "markdown",
 
463
  "metadata": {},
464
  "source": [
465
- "### Usage Examples\nRun annotation with your chosen model."
 
466
  ]
467
  },
468
  {
469
  "cell_type": "code",
 
 
470
  "metadata": {},
 
471
  "source": [
472
  "# Example 1: Annotate with Mistral (13.5 GB VRAM)\n",
473
  "# annotate_dataset(model_type='mistral', test_mode=False)\n",
@@ -480,9 +1010,7 @@
480
  "\n",
481
  "# Test mode (first 100 rows)\n",
482
  "# annotate_dataset(model_type='mistral', test_mode=True, test_size=100)\n"
483
- ],
484
- "execution_count": null,
485
- "outputs": []
486
  }
487
  ],
488
  "metadata": {
@@ -506,4 +1034,4 @@
506
  },
507
  "nbformat": 4,
508
  "nbformat_minor": 5
509
- }
 
10
  },
11
  {
12
  "cell_type": "markdown",
13
+ "id": "e4407358",
14
  "metadata": {},
15
  "source": [
16
+ "### Unified Model Loading & Inference\n",
17
+ "Code for querying Mistral, Gemma, and Qwen models."
18
+ ]
19
+ },
20
+ {
21
+ "cell_type": "markdown",
22
+ "id": "1a1b9d0e",
23
+ "metadata": {},
24
+ "source": [
25
+ "## CLEANING & PREPROCESSING"
26
+ ]
27
+ },
28
+ {
29
+ "cell_type": "markdown",
30
+ "id": "3df42c46",
31
+ "metadata": {},
32
+ "source": [
33
+ "#### Named Entity Recognitition (NER) using SpaCy "
34
+ ]
35
+ },
36
+ {
37
+ "cell_type": "code",
38
+ "execution_count": 3,
39
+ "id": "a287eef4",
40
+ "metadata": {},
41
+ "outputs": [
42
+ {
43
+ "name": "stdout",
44
+ "output_type": "stream",
45
+ "text": [
46
+ "✅ spaCy model loaded: en_core_web_sm\n",
47
+ "Loaded 50861 rows\n",
48
+ "\n",
49
+ "🔄 Processing names with spaCy NER...\n",
50
+ "\n",
51
+ "📊 Name cleaning examples (with spaCy NER):\n",
52
+ "====================================================================================================\n",
53
+ "Original Name | Cleaned Name \n",
54
+ "====================================================================================================\n",
55
+ "Super Pose Book Vol.1 - ControlNet | Super Pose Book \n",
56
+ "Liyuu LoRA | Liyuu \n",
57
+ "HashimotoKanna/ 橋本環奈 _JP_Actress | HashimotoKanna \n",
58
+ "Emma Watson (JG) | Emma Watson \n",
59
+ "Gal Gadot「LoRa」 | Gal Gadot \n",
60
+ "Scarlett Johansson「LoRa」 | Scarlett Johansson \n",
61
+ "Gakki | Aragaki Yui | 新垣結衣 | Gakki \n",
62
+ "Actress Satomi_石原○○ | Actress Satomi \n",
63
+ "Game of Thrones Cast | Game Thrones Cast \n",
64
+ "Natalie Portman「LoRa」 | Natalie Portman \n",
65
+ "Emma Watson LoRA | Emma Watson \n",
66
+ "Karina Makina Lora | Karina Makina \n",
67
+ "WRAV YUA_三xx亜 | WRAV YUA \n",
68
+ "Dilraba Dilmurat 迪丽热巴 | Dilraba Dilmurat \n",
69
+ "MIMI,大幂幂 | MIMI \n",
70
+ "Chinese Idol - YangMi杨幂 | Chinese Idol YangMi杨幂 \n",
71
+ "Xiaorouseeu / 小柔SeeU - Chinese cosplayer and influencer | Xiaorouseeu \n",
72
+ "Jennifer Connelly (80s/90s) | Jennifer Connelly \n",
73
+ "====================================================================================================\n",
74
+ "\n",
75
+ "🧪 Leetspeak translation examples:\n",
76
+ " 4kira LoRA -> akira\n",
77
+ " 3mma Watson v2 -> Watson\n",
78
+ " 1rene LORA -> irene\n",
79
+ " L3vi Ackerman -> Levi Ackerman\n",
80
+ "\n",
81
+ "📈 Statistics:\n",
82
+ " Total rows: 50861\n",
83
+ " Non-empty names: 50858\n",
84
+ " Empty names: 3\n",
85
+ "\n",
86
+ "🎯 Sample spaCy NER results:\n",
87
+ " 1. IU\n",
88
+ " 2. Super Pose Book\n",
89
+ " 3. Liyuu\n",
90
+ " 4. Irene\n",
91
+ " 5. AESPA Karina\n",
92
+ " 6. Saika Kawakita\n",
93
+ " 7. Liu Yifei\n",
94
+ " 8. HashimotoKanna\n",
95
+ " 9. Emma Watson\n",
96
+ " 10. Gal Gadot\n",
97
+ "\n",
98
+ "✅ Cleaned 50861 names using spaCy NER\n",
99
+ "💾 Saved to /home/lauhp/000_PHD/000_010_PUBLICATION/CODE/pm-paper/data/CSV/model_adapter/real_person_adapter_step_01_NER.csv\n"
100
+ ]
101
+ }
102
+ ],
103
+ "source": [
104
+ "import pandas as pd\n",
105
+ "import re\n",
106
+ "from pathlib import Path\n",
107
+ "import emoji\n",
108
+ "import spacy\n",
109
+ "\n",
110
+ "# Load spaCy model\n",
111
+ "# You may need to download it first: python -m spacy download en_core_web_sm\n",
112
+ "try:\n",
113
+ " nlp = spacy.load(\"en_core_web_sm\")\n",
114
+ " print(\"✅ spaCy model loaded: en_core_web_sm\")\n",
115
+ "except OSError:\n",
116
+ " print(\"❌ spaCy model not found. Downloading...\")\n",
117
+ " import subprocess\n",
118
+ " subprocess.run([\"python\", \"-m\", \"spacy\", \"download\", \"en_core_web_sm\"])\n",
119
+ " nlp = spacy.load(\"en_core_web_sm\")\n",
120
+ " print(\"✅ spaCy model downloaded and loaded\")\n",
121
+ "\n",
122
+ "# Set up paths\n",
123
+ "current_dir = Path.cwd()\n",
124
+ "#input_file = current_dir.parent / \"data/CSV/real_person_adapters.csv\"\n",
125
+ "input_file = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter.csv\"\n",
126
+ "\n",
127
+ "# Load dataset\n",
128
+ "df = pd.read_csv(input_file)\n",
129
+ "print(f\"Loaded {len(df)} rows\")\n",
130
+ "\n",
131
+ "def translate_leetspeak(text: str) -> str:\n",
132
+ " \"\"\"\n",
133
+ " Translate common leetspeak patterns to normal letters.\n",
134
+ " Examples: 4kira -> Akira, 3mma -> Emma, 1rene -> Irene\n",
135
+ " \"\"\"\n",
136
+ " if not text:\n",
137
+ " return text\n",
138
+ " \n",
139
+ " # Common leetspeak mappings (order matters!)\n",
140
+ " leetspeak_map = {\n",
141
+ " '4': 'a',\n",
142
+ " '3': 'e', \n",
143
+ " '1': 'i',\n",
144
+ " '0': 'o',\n",
145
+ " '7': 't',\n",
146
+ " '5': 's',\n",
147
+ " '8': 'b',\n",
148
+ " '9': 'g',\n",
149
+ " '@': 'a',\n",
150
+ " '$': 's',\n",
151
+ " '!': 'i',\n",
152
+ " }\n",
153
+ " \n",
154
+ " result = text\n",
155
+ " # Apply mappings at word boundaries or start of string\n",
156
+ " for leet, normal in leetspeak_map.items():\n",
157
+ " # Replace at start of word\n",
158
+ " result = re.sub(rf'\\b{re.escape(leet)}', normal, result, flags=re.IGNORECASE)\n",
159
+ " # Replace standalone numbers that look like letters in context\n",
160
+ " result = re.sub(rf'(?<=[a-z]){re.escape(leet)}(?=[a-z])', normal, result, flags=re.IGNORECASE)\n",
161
+ " \n",
162
+ " return result\n",
163
+ "\n",
164
+ "def preprocess_for_ner(name: str) -> str:\n",
165
+ " \"\"\"\n",
166
+ " Preprocess the name before spaCy NER.\n",
167
+ " Remove noise but keep the actual name parts.\n",
168
+ " \"\"\"\n",
169
+ " if pd.isna(name):\n",
170
+ " return \"\"\n",
171
+ " \n",
172
+ " name = str(name)\n",
173
+ " \n",
174
+ " # FIRST: Translate leetspeak\n",
175
+ " name = translate_leetspeak(name)\n",
176
+ " \n",
177
+ " # Remove emoji\n",
178
+ " name = emoji.replace_emoji(name, replace=' ')\n",
179
+ " \n",
180
+ " # Remove version indicators (v1, v2, v1.0, etc.)\n",
181
+ " name = re.sub(r'\\s*[vV]\\d+(\\.\\d+)?\\s*', ' ', name)\n",
182
+ " \n",
183
+ " # Remove LoRA-related terms (case insensitive)\n",
184
+ " lora_terms = ['lora', 'loha', 'lycoris', 'controlnet', 'textual inversion', \n",
185
+ " 'embedding', 'ti', 'checkpoint', 'model', 'adapter', 'pony', 'sdxl', 'flux', 'illustrious', 'sd14', 'sd14', 'sd2', 'sd3', 'diffusion', 'stable', 'hunyuan']\n",
186
+ " for term in lora_terms:\n",
187
+ " name = re.sub(rf'\\b{term}\\b', '', name, flags=re.IGNORECASE)\n",
188
+ " \n",
189
+ " # Remove content in parentheses or brackets (often metadata)\n",
190
+ " name = re.sub(r'\\([^)]*\\)', '', name)\n",
191
+ " name = re.sub(r'\\[[^\\]]*\\]', '', name)\n",
192
+ " \n",
193
+ " # Remove special characters like 「」\n",
194
+ " name = re.sub(r'[「」『』【】〈〉《》]', '', name)\n",
195
+ " \n",
196
+ " # Handle pipe - keep first part\n",
197
+ " if '|' in name:\n",
198
+ " name = name.split('|')[0]\n",
199
+ " \n",
200
+ " # Handle forward slash - keep first part\n",
201
+ " if '/' in name:\n",
202
+ " name = name.split('/')[0]\n",
203
+ " \n",
204
+ " # Replace underscores with spaces\n",
205
+ " name = name.replace('_', ' ')\n",
206
+ " \n",
207
+ " # Remove multiple spaces\n",
208
+ " name = re.sub(r'\\s+', ' ', name)\n",
209
+ " \n",
210
+ " # Strip\n",
211
+ " name = name.strip()\n",
212
+ " \n",
213
+ " return name\n",
214
+ "\n",
215
+ "def extract_person_name(text: str) -> str:\n",
216
+ " \"\"\"\n",
217
+ " Use spaCy NER to extract person names from text.\n",
218
+ " Falls back to cleaned text if no PERSON entity found.\n",
219
+ " \"\"\"\n",
220
+ " if not text:\n",
221
+ " return \"\"\n",
222
+ " \n",
223
+ " # Run spaCy NER\n",
224
+ " doc = nlp(text)\n",
225
+ " \n",
226
+ " # Extract PERSON entities\n",
227
+ " person_entities = [ent.text for ent in doc.ents if ent.label_ == \"PERSON\"]\n",
228
+ " \n",
229
+ " if person_entities:\n",
230
+ " # Return the first (usually longest/best) person name\n",
231
+ " return person_entities[0].strip()\n",
232
+ " \n",
233
+ " # If no PERSON entity found, try to extract capitalized words (likely names)\n",
234
+ " # This helps with names spaCy might miss\n",
235
+ " words = text.split()\n",
236
+ " capitalized_words = [w for w in words if w and w[0].isupper() and len(w) > 1]\n",
237
+ " \n",
238
+ " if capitalized_words:\n",
239
+ " # Join first few capitalized words (likely the name)\n",
240
+ " return ' '.join(capitalized_words[:3]).strip()\n",
241
+ " \n",
242
+ " # Last resort: return cleaned text\n",
243
+ " return text.strip()\n",
244
+ "\n",
245
+ "def clean_name_with_spacy(name: str) -> str:\n",
246
+ " \"\"\"\n",
247
+ " Complete name cleaning pipeline with spaCy NER.\n",
248
+ " \n",
249
+ " Pipeline:\n",
250
+ " 1. Translate leetspeak (4→a, 3→e, 1→i, etc.)\n",
251
+ " 2. Remove noise (emoji, version tags, LoRA terms)\n",
252
+ " 3. Use spaCy to extract PERSON entities\n",
253
+ " 4. Fallback to capitalized words or cleaned text\n",
254
+ " \"\"\"\n",
255
+ " # Step 1 & 2: Preprocess (leetspeak + noise removal)\n",
256
+ " preprocessed = preprocess_for_ner(name)\n",
257
+ " \n",
258
+ " if not preprocessed:\n",
259
+ " return \"\"\n",
260
+ " \n",
261
+ " # Step 3: Extract person name using spaCy NER\n",
262
+ " person_name = extract_person_name(preprocessed)\n",
263
+ " \n",
264
+ " return person_name\n",
265
+ "\n",
266
+ "# Apply name cleaning with spaCy\n",
267
+ "print(\"\\n🔄 Processing names with spaCy NER...\")\n",
268
+ "df['real_name'] = df['name'].apply(clean_name_with_spacy)\n",
269
+ "\n",
270
+ "# Show examples with detailed comparison\n",
271
+ "print(\"\\n📊 Name cleaning examples (with spaCy NER):\")\n",
272
+ "print(\"=\" * 100)\n",
273
+ "print(f\"{'Original Name':<50} | {'Cleaned Name':<30}\")\n",
274
+ "print(\"=\" * 100)\n",
275
+ "\n",
276
+ "examples = df[['name', 'real_name']].head(30)\n",
277
+ "shown = 0\n",
278
+ "for idx, row in examples.iterrows():\n",
279
+ " if row['name'] != row['real_name'] and shown < 20:\n",
280
+ " print(f\"{row['name']:<50} | {row['real_name']:<30}\")\n",
281
+ " shown += 1\n",
282
+ "\n",
283
+ "print(\"=\" * 100)\n",
284
+ "\n",
285
+ "# Show specific test cases\n",
286
+ "print(\"\\n🧪 Leetspeak translation examples:\")\n",
287
+ "test_names = ['4kira LoRA', '3mma Watson v2', '1rene LORA', 'L3vi Ackerman']\n",
288
+ "for test in test_names:\n",
289
+ " result = clean_name_with_spacy(test)\n",
290
+ " print(f\" {test:<30} -> {result}\")\n",
291
+ "\n",
292
+ "# Statistics\n",
293
+ "print(f\"\\n📈 Statistics:\")\n",
294
+ "print(f\" Total rows: {len(df)}\")\n",
295
+ "print(f\" Non-empty names: {(df['real_name'] != '').sum()}\")\n",
296
+ "print(f\" Empty names: {(df['real_name'] == '').sum()}\")\n",
297
+ "\n",
298
+ "# Show some examples of what spaCy identified\n",
299
+ "print(\"\\n🎯 Sample spaCy NER results:\")\n",
300
+ "sample_names = df['real_name'].head(20).tolist()\n",
301
+ "for i, name in enumerate(sample_names[:10], 1):\n",
302
+ " if name:\n",
303
+ " print(f\" {i}. {name}\")\n",
304
+ "\n",
305
+ "print(f\"\\n✅ Cleaned {len(df)} names using spaCy NER\")\n",
306
+ "\n",
307
+ "# Save intermediate result\n",
308
+ "output_step1 = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter_step_01_NER.csv\"\n",
309
+ "df.to_csv(output_step1, index=False)\n",
310
+ "print(f\"💾 Saved to {output_step1}\")\n"
311
+ ]
312
+ },
313
+ {
314
+ "cell_type": "markdown",
315
+ "id": "64687c72",
316
+ "metadata": {},
317
+ "source": [
318
+ "#### STEP 02: Nationality tag to Country hint\n",
319
+ "here tags related to nationality gets converted to the country equivalent."
320
+ ]
321
+ },
322
+ {
323
+ "cell_type": "code",
324
+ "execution_count": null,
325
+ "id": "d6eaef5b",
326
+ "metadata": {},
327
+ "outputs": [],
328
+ "source": [
329
+ "import pandas as pd\n",
330
+ "from pathlib import Path\n",
331
+ "\n",
332
+ "# Set up paths\n",
333
+ "current_dir = Path.cwd()\n",
334
+ "countries_file = current_dir.parent / \"misc/lists/countries.csv\"\n",
335
+ "professions_file = current_dir.parent / \"misc/lists/professions.csv\"\n",
336
+ "input_file = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter_step_01_NER.csv\"\n",
337
+ "output_file = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter_step_02_NER.csv\"\n",
338
+ "\n",
339
+ "# Load datasets\n",
340
+ "poi_df = pd.read_csv(input_file)\n",
341
+ "countries_df = pd.read_csv(countries_file)\n",
342
+ "professions_df = pd.read_csv(professions_file)\n",
343
+ "\n",
344
+ "# Define uninhabited or non-relevant territories to exclude\n",
345
+ "excluded_territories = {\n",
346
+ " 'isle of man', 'bouvet island', 'heard island and mcdonald islands',\n",
347
+ " 'french southern territories', 'south georgia and the south sandwich islands',\n",
348
+ " 'svalbard and jan mayen', 'british indian ocean territory', 'antarctica',\n",
349
+ " 'christmas island', 'cocos (keeling) islands', 'norfolk island',\n",
350
+ " 'pitcairn', 'tokelau', 'united states minor outlying islands',\n",
351
+ " 'wallis and futuna', 'western sahara'\n",
352
+ "}\n",
353
+ "\n",
354
+ "# Step 1: Combine tags into one lowercase list\n",
355
+ "def combine_tags(row):\n",
356
+ " return [str(row[f\"tag_{i}\"]).strip().lower() for i in range(1, 8) if pd.notna(row.get(f\"tag_{i}\"))]\n",
357
+ "\n",
358
+ "poi_df[\"tags\"] = poi_df.apply(combine_tags, axis=1)\n",
359
+ "\n",
360
+ "# Step 2: Build tag → (country, nationality) mapping with PRIORITIES\n",
361
+ "tag_to_country_nationality = {}\n",
362
+ "# We'll use a priority score: direct country name = 3, nationality = 2, word parts = 1\n",
363
+ "\n",
364
+ "for _, row in countries_df.iterrows():\n",
365
+ " country = str(row[\"en_short_name\"]).strip()\n",
366
+ " nationality = str(row[\"nationality\"]).strip()\n",
367
+ " \n",
368
+ " # Skip excluded territories\n",
369
+ " if country.lower() in excluded_territories:\n",
370
+ " continue\n",
371
+ "\n",
372
+ " country_lc = country.lower()\n",
373
+ " nationality_lc = nationality.lower()\n",
374
+ "\n",
375
+ " # Store as (country, nationality, priority)\n",
376
+ " # Exact country name match = highest priority\n",
377
+ " if country_lc not in tag_to_country_nationality:\n",
378
+ " tag_to_country_nationality[country_lc] = (country, \"\", 3)\n",
379
+ " \n",
380
+ " # Exact nationality match = medium priority \n",
381
+ " if nationality_lc not in tag_to_country_nationality:\n",
382
+ " tag_to_country_nationality[nationality_lc] = (\"\", nationality, 2)\n",
383
+ " \n",
384
+ " # No-space versions\n",
385
+ " country_no_space = country_lc.replace(\" \", \"\")\n",
386
+ " nationality_no_space = nationality_lc.replace(\" \", \"\")\n",
387
+ " \n",
388
+ " if country_no_space not in tag_to_country_nationality:\n",
389
+ " tag_to_country_nationality[country_no_space] = (country, \"\", 3)\n",
390
+ " if nationality_no_space not in tag_to_country_nationality:\n",
391
+ " tag_to_country_nationality[nationality_no_space] = (\"\", nationality, 2)\n",
392
+ "\n",
393
+ " # Word parts = lowest priority (only for longer words to avoid false matches)\n",
394
+ " for part in country_lc.split():\n",
395
+ " if len(part) > 4: # Only words longer than 4 chars\n",
396
+ " if part not in tag_to_country_nationality:\n",
397
+ " tag_to_country_nationality[part] = (country, \"\", 1)\n",
398
+ " for part in nationality_lc.split():\n",
399
+ " if len(part) > 4:\n",
400
+ " if part not in tag_to_country_nationality:\n",
401
+ " tag_to_country_nationality[part] = (\"\", nationality, 1)\n",
402
+ "\n",
403
+ "print(f\"Built country/nationality mapping with {len(tag_to_country_nationality)} entries\")\n",
404
+ "\n",
405
+ "# Step 3: Infer likely_country and likely_nationality by checking ALL tags\n",
406
+ "def infer_country_and_nationality(tags):\n",
407
+ " \"\"\"\n",
408
+ " Check ALL tags and return the best match based on priority.\n",
409
+ " Priority: exact country name > nationality > word parts\n",
410
+ " \"\"\"\n",
411
+ " best_match = None\n",
412
+ " best_priority = 0\n",
413
+ " \n",
414
+ " for tag in tags:\n",
415
+ " # Try cleaned version (no spaces)\n",
416
+ " cleaned = tag.replace(\" \", \"\").lower()\n",
417
+ " \n",
418
+ " # Check cleaned version\n",
419
+ " if cleaned in tag_to_country_nationality:\n",
420
+ " country, nationality, priority = tag_to_country_nationality[cleaned]\n",
421
+ " if priority > best_priority and country and country.lower() not in excluded_territories:\n",
422
+ " best_match = (country, nationality)\n",
423
+ " best_priority = priority\n",
424
+ " \n",
425
+ " # Also check original tag\n",
426
+ " if tag in tag_to_country_nationality:\n",
427
+ " country, nationality, priority = tag_to_country_nationality[tag]\n",
428
+ " if priority > best_priority and country and country.lower() not in excluded_territories:\n",
429
+ " best_match = (country, nationality)\n",
430
+ " best_priority = priority\n",
431
+ " \n",
432
+ " if best_match:\n",
433
+ " return pd.Series(best_match)\n",
434
+ " return pd.Series([\"\", \"\"])\n",
435
+ "\n",
436
+ "poi_df[[\"likely_country\", \"likely_nationality\"]] = poi_df[\"tags\"].apply(infer_country_and_nationality)\n",
437
+ "\n",
438
+ "# Step 4: Build tag → profession mapping\n",
439
+ "profession_alias_map = {}\n",
440
+ "\n",
441
+ "for _, row in professions_df.iterrows():\n",
442
+ " canonical = str(row['profession']).strip().lower()\n",
443
+ " profession_alias_map[canonical] = canonical\n",
444
+ " for alias_col in ['alias_1', 'alias_2', 'alias_3']:\n",
445
+ " alias = row.get(alias_col)\n",
446
+ " if pd.notna(alias):\n",
447
+ " profession_alias_map[str(alias).strip().lower()] = canonical\n",
448
+ "\n",
449
+ "# Step 5: Infer likely profession from tags\n",
450
+ "def infer_profession_from_tags(tags):\n",
451
+ " matched = []\n",
452
+ " for tag in tags:\n",
453
+ " cleaned = tag.strip().lower()\n",
454
+ " if cleaned in profession_alias_map:\n",
455
+ " matched.append(profession_alias_map[cleaned])\n",
456
+ "\n",
457
+ " if not matched:\n",
458
+ " return \"\"\n",
459
+ " if \"celebrity\" in matched and len(set(matched)) > 1:\n",
460
+ " # Drop 'celebrity' if other professions are present\n",
461
+ " matched = [m for m in matched if m != \"celebrity\"]\n",
462
+ "\n",
463
+ " return matched[0] # Return the first specific match\n",
464
+ "\n",
465
+ "\n",
466
+ "poi_df[\"likely_profession\"] = poi_df[\"tags\"].apply(infer_profession_from_tags)\n",
467
+ "\n",
468
+ "# Step 6: Save enriched dataset\n",
469
+ "poi_df.to_csv(output_file, index=False)\n",
470
+ "\n",
471
+ "# Preview results\n",
472
+ "print(f\"\\nProcessed {len(poi_df)} rows\")\n",
473
+ "print(f\"Rows with country: {(poi_df['likely_country'] != '').sum()}\")\n",
474
+ "print(f\"Rows with nationality: {(poi_df['likely_nationality'] != '').sum()}\")\n",
475
+ "print(f\"Rows with profession: {(poi_df['likely_profession'] != '').sum()}\")\n",
476
+ "\n",
477
+ "print(f\"\\nTop 10 countries:\")\n",
478
+ "print(poi_df[poi_df['likely_country'] != '']['likely_country'].value_counts().head(10))\n"
479
+ ]
480
+ },
481
+ {
482
+ "cell_type": "markdown",
483
+ "id": "4a4a58b3",
484
+ "metadata": {},
485
+ "source": [
486
+ "## LLM ANNOTATION"
487
+ ]
488
+ },
489
+ {
490
+ "cell_type": "markdown",
491
+ "id": "b298844d",
492
+ "metadata": {},
493
+ "source": [
494
+ "#### Model Configurations"
495
  ]
496
  },
497
  {
498
  "cell_type": "code",
499
+ "execution_count": null,
500
+ "id": "39f3d65e",
501
  "metadata": {},
502
+ "outputs": [],
503
  "source": [
504
  "import pandas as pd\n",
505
  "import json\n",
 
568
  " \"sports professional\",\n",
569
  " \"tv personality\"\n",
570
  "]\n"
571
+ ]
572
+ },
573
+ {
574
+ "cell_type": "markdown",
575
+ "id": "c215b38c",
576
+ "metadata": {},
577
+ "source": [
578
+ "#### Load Model Function"
579
+ ]
580
  },
581
  {
582
  "cell_type": "code",
583
+ "execution_count": null,
584
+ "id": "cfb5b13e",
585
  "metadata": {},
586
+ "outputs": [],
587
  "source": [
588
  "def load_model(model_type='mistral'):\n",
589
  " \"\"\"\n",
 
647
  " print(f\"VRAM used: {vram_gb:.2f} GB\\n\")\n",
648
  " \n",
649
  " return model, tokenizer, config\n"
650
+ ]
651
+ },
652
+ {
653
+ "cell_type": "markdown",
654
+ "id": "11b2221a",
655
+ "metadata": {},
656
+ "source": [
657
+ "#### Inference Code"
658
+ ]
659
  },
660
  {
661
  "cell_type": "code",
662
+ "execution_count": null,
663
+ "id": "229f96bd",
664
  "metadata": {},
665
+ "outputs": [],
666
  "source": [
667
  "@contextmanager\n",
668
  "def timeout(duration):\n",
 
739
  " except Exception as e:\n",
740
  " print(f\"[ERROR] Generation failed: {e}\")\n",
741
  " return None\n"
742
+ ]
743
+ },
744
+ {
745
+ "cell_type": "markdown",
746
+ "id": "88f005f8",
747
+ "metadata": {},
748
+ "source": [
749
+ "#### Prompt creation"
750
+ ]
751
  },
752
  {
753
  "cell_type": "code",
754
+ "execution_count": null,
755
+ "id": "dfe05463",
756
  "metadata": {},
757
+ "outputs": [],
758
  "source": [
759
  "def create_prompt(row):\n",
760
  " \"\"\"Create annotation prompt from row data.\"\"\"\n",
 
807
  "3. Female\n",
808
  "4. singer/musician, public figure\n",
809
  "5. United States\"\"\"\n"
810
+ ]
811
+ },
812
+ {
813
+ "cell_type": "markdown",
814
+ "id": "854fa668",
815
+ "metadata": {},
816
+ "source": [
817
+ "#### Response parsing code"
818
+ ]
819
  },
820
  {
821
  "cell_type": "code",
822
+ "execution_count": null,
823
+ "id": "1a4be2ee",
824
  "metadata": {},
825
+ "outputs": [],
826
  "source": [
827
  "def parse_response(response):\n",
828
  " \"\"\"Parse model response into structured fields.\"\"\"\n",
 
863
  " fields['country'] = country_raw\n",
864
  " \n",
865
  " return fields\n"
866
+ ]
867
+ },
868
+ {
869
+ "cell_type": "markdown",
870
+ "id": "7e2f7a86",
871
+ "metadata": {},
872
+ "source": [
873
+ "#### CSV annotation"
874
+ ]
875
  },
876
  {
877
  "cell_type": "code",
878
+ "execution_count": null,
879
+ "id": "5f3dd5d6",
880
  "metadata": {},
881
+ "outputs": [],
882
  "source": [
883
  "def annotate_dataset(model_type='mistral', test_mode=False, test_size=100, max_rows=50862, save_interval=10):\n",
884
  " \"\"\"\n",
 
981
  " df.to_csv(output_file, index=False)\n",
982
  " index_file.write_text(str(current_index))\n",
983
  " print(f\"✓ Finished annotation with {model_type}\")\n"
984
+ ]
 
 
985
  },
986
  {
987
  "cell_type": "markdown",
988
+ "id": "55da2f4c",
989
  "metadata": {},
990
  "source": [
991
+ "### Usage Examples\n",
992
+ "Run annotation with your chosen model."
993
  ]
994
  },
995
  {
996
  "cell_type": "code",
997
+ "execution_count": null,
998
+ "id": "351ea40c",
999
  "metadata": {},
1000
+ "outputs": [],
1001
  "source": [
1002
  "# Example 1: Annotate with Mistral (13.5 GB VRAM)\n",
1003
  "# annotate_dataset(model_type='mistral', test_mode=False)\n",
 
1010
  "\n",
1011
  "# Test mode (first 100 rows)\n",
1012
  "# annotate_dataset(model_type='mistral', test_mode=True, test_size=100)\n"
1013
+ ]
 
 
1014
  }
1015
  ],
1016
  "metadata": {
 
1034
  },
1035
  "nbformat": 4,
1036
  "nbformat_minor": 5
1037
+ }
jupyter_notebooks/Section_2-3-4_Figure_8_Step_2_response_comparison_and_consensus_extraction.ipynb CHANGED
@@ -1031,7 +1031,7 @@
1031
  },
1032
  {
1033
  "cell_type": "code",
1034
- "execution_count": 15,
1035
  "id": "ca58ae0f-a3e1-44c9-9645-412997f1777d",
1036
  "metadata": {
1037
  "execution": {
@@ -1047,171 +1047,66 @@
1047
  "name": "stdout",
1048
  "output_type": "stream",
1049
  "text": [
1050
- "============================================================\n",
1051
- "STRICT CONSENSUS FILTER\n",
1052
- "============================================================\n",
1053
- "Reading: /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/combined_llm_annotations.csv\n",
 
1054
  "Input shape: (50861, 67)\n",
1055
- "Models to check: gemma, mistral, qwen\n",
1056
  "\n",
1057
  "Processing rows...\n",
1058
  " Processed 50000/50861 rows...\n",
1059
  " Processed 50861 rows. \n",
1060
  "\n",
1061
- "============================================================\n",
1062
- "FILTERING RESULTS\n",
1063
- "============================================================\n",
1064
  "Total input rows: 50,861\n",
1065
- "Rows passing all criteria: 4,181 (8.2%)\n",
1066
- "\n",
1067
- "Failure reasons (rows can fail multiple):\n",
1068
- " - Country disagreement: 25,292 (49.7%)\n",
1069
- " - Gender disagreement: 6,783 (13.3%)\n",
1070
- " - Profession disagreement: 43,698 (85.9%)\n",
1071
- " - Contains 'Unknown': 46,680 (91.8%)\n",
1072
- "\n",
1073
- "============================================================\n",
1074
- "CONSENSUS DISTRIBUTIONS\n",
1075
- "============================================================\n",
1076
  "\n",
1077
- "Country (top 10):\n",
1078
- "consensus_country\n",
1079
- "usa 2186\n",
1080
- "uk 405\n",
1081
- "japan 337\n",
1082
- "south korea 237\n",
1083
- "india 146\n",
1084
- "china 101\n",
1085
- "canada 79\n",
1086
- "russia 74\n",
1087
- "brazil 73\n",
1088
- "france 69\n",
1089
- "Name: count, dtype: int64\n",
1090
  "\n",
1091
- "Gender:\n",
1092
- "consensus_gender\n",
1093
- "female 3622\n",
1094
- "male 559\n",
1095
- "Name: count, dtype: int64\n",
1096
  "\n",
1097
- "Profession (top 10):\n",
1098
  "consensus_profession\n",
1099
- "actor, tv personality, public figure 1154\n",
1100
- "actor, tv personality, online personality 633\n",
1101
- "singer/musician, model, online personality 341\n",
1102
- "actor, singer/musician, tv personality 322\n",
1103
- "actor, tv personality, model 267\n",
1104
- "actor, model, online personality 161\n",
1105
- "actor, model, tv personality 125\n",
1106
- "model, adult performer, online personality 80\n",
1107
- "adult performer, model, online personality 75\n",
1108
- "singer/musician, tv personality, public figure 71\n",
1109
- "Name: count, dtype: int64\n",
1110
- "\n",
1111
- "Profession agreement level:\n",
1112
- "profession_agreement_count\n",
1113
- "2 4096\n",
1114
- "3 85\n",
 
 
 
 
1115
  "Name: count, dtype: int64\n",
1116
  "\n",
1117
- "============================================================\n",
1118
- "✓ Strict consensus file saved to: strict_consensus.csv\n",
1119
- " Total rows: 4,181\n",
1120
- " Total columns: 71\n",
1121
- "============================================================\n",
1122
- "\n",
1123
- "============================================================\n",
1124
- "SAMPLE COMPARISONS (first 5 rows)\n",
1125
- "============================================================\n",
1126
  "\n",
1127
- "--- Row 1: Liu Yifei ---\n",
1128
- "Country:\n",
1129
- " Consensus: china\n",
1130
- " gemma: China\n",
1131
- " mistral: China\n",
1132
- " qwen: China\n",
1133
- "Gender:\n",
1134
- " Consensus: female\n",
1135
- " gemma: Female\n",
1136
- " mistral: Female\n",
1137
- " qwen: Female\n",
1138
- "Profession (agreement: 2/3):\n",
1139
- " Consensus: actor, model, singer/musician\n",
1140
- " gemma: actor, model, singer/musician\n",
1141
- " mistral: actor, tv personality, model\n",
1142
- " qwen: actor, model, singer/musician\n",
1143
- "\n",
1144
- "--- Row 2: Emma Watson (JG) ---\n",
1145
- "Country:\n",
1146
- " Consensus: uk\n",
1147
- " gemma: UK\n",
1148
- " mistral: UK\n",
1149
- " qwen: UK\n",
1150
- "Gender:\n",
1151
- " Consensus: female\n",
1152
- " gemma: Female\n",
1153
- " mistral: Female\n",
1154
- " qwen: Female\n",
1155
- "Profession (agreement: 2/3):\n",
1156
- " Consensus: actor, public figure, model\n",
1157
- " gemma: actor, public figure, model\n",
1158
- " mistral: actor, tv personality, public figure\n",
1159
- " qwen: actor, public figure, model\n",
1160
- "\n",
1161
- "--- Row 3: Gal Gadot「LoRa」 ---\n",
1162
- "Country:\n",
1163
- " Consensus: israel\n",
1164
- " gemma: Israel\n",
1165
- " mistral: Israel\n",
1166
- " qwen: Israel\n",
1167
- "Gender:\n",
1168
- " Consensus: female\n",
1169
- " gemma: Female\n",
1170
- " mistral: Female\n",
1171
- " qwen: Female\n",
1172
- "Profession (agreement: 2/3):\n",
1173
- " Consensus: actor, model, tv personality\n",
1174
- " gemma: actor, model, tv personality\n",
1175
- " mistral: actor, model, tv personality\n",
1176
- " qwen: actor, model, public figure\n",
1177
- "\n",
1178
- "--- Row 4: Game of Thrones Cast ---\n",
1179
- "Country:\n",
1180
- " Consensus: uk\n",
1181
- " gemma: UK\n",
1182
- " mistral: UK\n",
1183
- " qwen: UK\n",
1184
- "Gender:\n",
1185
- " Consensus: female\n",
1186
- " gemma: Female\n",
1187
- " mistral: Female\n",
1188
- " qwen: Female\n",
1189
- "Profession (agreement: 2/3):\n",
1190
- " Consensus: actor, tv personality, online personality\n",
1191
- " gemma: actor, tv personality, online personality\n",
1192
- " mistral: actor, tv personality, online personality\n",
1193
- " qwen: actor, model\n",
1194
- "\n",
1195
- "--- Row 5: Karina Makina Lora ---\n",
1196
- "Country:\n",
1197
- " Consensus: south korea\n",
1198
- " gemma: South Korea\n",
1199
- " mistral: South Korea\n",
1200
- " qwen: South Korea\n",
1201
- "Gender:\n",
1202
- " Consensus: female\n",
1203
- " gemma: Female\n",
1204
- " mistral: Female\n",
1205
- " qwen: Female\n",
1206
- "Profession (agreement: 2/3):\n",
1207
- " Consensus: singer/musician, model, online personality\n",
1208
- " gemma: singer/musician, model, online personality\n",
1209
- " mistral: singer/musician, tv personality, kpop idol\n",
1210
- " qwen: singer/musician, model, online personality\n",
1211
  "\n",
1212
- "============================================================\n",
1213
- "COMPLETE!\n",
1214
- "============================================================\n"
1215
  ]
1216
  }
1217
  ],
@@ -2338,6 +2233,602 @@
2338
  " print(\"Please adjust the path in the script to point to your data file.\")"
2339
  ]
2340
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2341
  {
2342
  "cell_type": "code",
2343
  "execution_count": null,
@@ -2349,7 +2840,7 @@
2349
  ],
2350
  "metadata": {
2351
  "kernelspec": {
2352
- "display_name": "Python 3 (ipykernel)",
2353
  "language": "python",
2354
  "name": "python3"
2355
  },
@@ -2363,7 +2854,7 @@
2363
  "name": "python",
2364
  "nbconvert_exporter": "python",
2365
  "pygments_lexer": "ipython3",
2366
- "version": "3.12.10"
2367
  }
2368
  },
2369
  "nbformat": 4,
 
1031
  },
1032
  {
1033
  "cell_type": "code",
1034
+ "execution_count": null,
1035
  "id": "ca58ae0f-a3e1-44c9-9645-412997f1777d",
1036
  "metadata": {
1037
  "execution": {
 
1047
  "name": "stdout",
1048
  "output_type": "stream",
1049
  "text": [
1050
+ "================================================================================\n",
1051
+ "IMPROVED CONSENSUS CREATION\n",
1052
+ "================================================================================\n",
1053
+ "Method: hybrid\n",
1054
+ "Reading: /home/lauhp/000_PHD/000_010_PUBLICATION/CODE/pm-paper/data/CSV/combined_llm_annotations.csv\n",
1055
  "Input shape: (50861, 67)\n",
1056
+ "Models: gemma, mistral, qwen\n",
1057
  "\n",
1058
  "Processing rows...\n",
1059
  " Processed 50000/50861 rows...\n",
1060
  " Processed 50861 rows. \n",
1061
  "\n",
1062
+ "================================================================================\n",
1063
+ "RESULTS\n",
1064
+ "================================================================================\n",
1065
  "Total input rows: 50,861\n",
1066
+ "Rows passing all criteria: 23,084 (45.4%)\n",
 
 
 
 
 
 
 
 
 
 
1067
  "\n",
1068
+ "Consensus method usage:\n",
1069
+ " - weighted: 43,445 (188.2%)\n",
1070
+ " - adult_special: 5,622 (24.4%)\n",
1071
+ " - adult_any_position: 1,784 (7.7%)\n",
1072
+ " - no_consensus: 10 (0.0%)\n",
 
 
 
 
 
 
 
 
1073
  "\n",
1074
+ "================================================================================\n",
1075
+ "PROFESSION DISTRIBUTION\n",
1076
+ "================================================================================\n",
 
 
1077
  "\n",
1078
+ "Top 20 professions:\n",
1079
  "consensus_profession\n",
1080
+ "actor 11046\n",
1081
+ "singer/musician 3939\n",
1082
+ "model 3017\n",
1083
+ "online personality 1651\n",
1084
+ "adult performer 1206\n",
1085
+ "public figure 986\n",
1086
+ "sports professional 458\n",
1087
+ "voice actor/asmr 376\n",
1088
+ "tv personality 373\n",
1089
+ "wrestler 10\n",
1090
+ "comedian 8\n",
1091
+ "cheerleader 2\n",
1092
+ "actress 2\n",
1093
+ "dancer 2\n",
1094
+ "architect 1\n",
1095
+ "entrepreneur 1\n",
1096
+ "basketball player 1\n",
1097
+ "podcaster 1\n",
1098
+ "artist 1\n",
1099
+ "gymnast 1\n",
1100
  "Name: count, dtype: int64\n",
1101
  "\n",
1102
+ "🎯 Adult performer variants: 1206 (5.22%)\n",
 
 
 
 
 
 
 
 
1103
  "\n",
1104
+ "================================================================================\n",
1105
+ "✓ Improved consensus saved to: improved_consensus.csv\n",
1106
+ " Total rows: 23,084\n",
1107
+ "================================================================================\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1108
  "\n",
1109
+ "✅ Complete!\n"
 
 
1110
  ]
1111
  }
1112
  ],
 
2233
  " print(\"Please adjust the path in the script to point to your data file.\")"
2234
  ]
2235
  },
2236
+ {
2237
+ "cell_type": "markdown",
2238
+ "id": "2d761d7f",
2239
+ "metadata": {},
2240
+ "source": [
2241
+ "# Improved Consensus Script with Position-Aware Logic\n",
2242
+ "# Specifically addresses the \"adult performer\" underrepresentation problem"
2243
+ ]
2244
+ },
2245
+ {
2246
+ "cell_type": "code",
2247
+ "execution_count": 2,
2248
+ "id": "2aac8386",
2249
+ "metadata": {},
2250
+ "outputs": [
2251
+ {
2252
+ "name": "stdout",
2253
+ "output_type": "stream",
2254
+ "text": [
2255
+ "================================================================================\n",
2256
+ "IMPROVED CONSENSUS CREATION\n",
2257
+ "================================================================================\n",
2258
+ "Method: hybrid\n",
2259
+ "Reading: /home/lauhp/000_PHD/000_010_PUBLICATION/CODE/pm-paper/data/CSV/combined_llm_annotations.csv\n",
2260
+ "Input shape: (50861, 67)\n",
2261
+ "Models: gemma, mistral, qwen\n",
2262
+ "\n",
2263
+ "Processing rows...\n",
2264
+ " Processed 50000/50861 rows...\n",
2265
+ " Processed 50861 rows. \n",
2266
+ "\n",
2267
+ "================================================================================\n",
2268
+ "RESULTS\n",
2269
+ "================================================================================\n",
2270
+ "Total input rows: 50,861\n",
2271
+ "Rows passing all criteria: 23,084 (45.4%)\n",
2272
+ "\n",
2273
+ "Consensus method usage:\n",
2274
+ " - weighted: 43,445 (188.2%)\n",
2275
+ " - adult_special: 5,622 (24.4%)\n",
2276
+ " - adult_any_position: 1,784 (7.7%)\n",
2277
+ " - no_consensus: 10 (0.0%)\n",
2278
+ "\n",
2279
+ "================================================================================\n",
2280
+ "PROFESSION DISTRIBUTION\n",
2281
+ "================================================================================\n",
2282
+ "\n",
2283
+ "Top 20 professions:\n",
2284
+ "consensus_profession\n",
2285
+ "actor 11046\n",
2286
+ "singer/musician 3939\n",
2287
+ "model 3017\n",
2288
+ "online personality 1651\n",
2289
+ "adult performer 1206\n",
2290
+ "public figure 986\n",
2291
+ "sports professional 458\n",
2292
+ "voice actor/asmr 376\n",
2293
+ "tv personality 373\n",
2294
+ "wrestler 10\n",
2295
+ "comedian 8\n",
2296
+ "cheerleader 2\n",
2297
+ "actress 2\n",
2298
+ "dancer 2\n",
2299
+ "architect 1\n",
2300
+ "entrepreneur 1\n",
2301
+ "basketball player 1\n",
2302
+ "podcaster 1\n",
2303
+ "artist 1\n",
2304
+ "gymnast 1\n",
2305
+ "Name: count, dtype: int64\n",
2306
+ "\n",
2307
+ "🎯 Adult performer variants: 1206 (5.22%)\n",
2308
+ "\n",
2309
+ "================================================================================\n",
2310
+ "✓ Improved consensus saved to: improved_consensus.csv\n",
2311
+ " Total rows: 23,084\n",
2312
+ "================================================================================\n",
2313
+ "\n",
2314
+ "✅ Complete!\n"
2315
+ ]
2316
+ }
2317
+ ],
2318
+ "source": [
2319
+ "\n",
2320
+ "\n",
2321
+ "import pandas as pd\n",
2322
+ "from pathlib import Path\n",
2323
+ "from collections import Counter\n",
2324
+ "import re\n",
2325
+ "\n",
2326
+ "# ============================================================\n",
2327
+ "# CONFIGURATION\n",
2328
+ "# ============================================================\n",
2329
+ "\n",
2330
+ "# Semantic grouping for professions\n",
2331
+ "PROFESSION_GROUPS = {\n",
2332
+ " 'adult_entertainment': [\n",
2333
+ " 'adult performer', 'adult film', 'pornstar', 'av actress', \n",
2334
+ " 'av idol', 'jav idol', 'adult model', 'adult entertainer'\n",
2335
+ " ],\n",
2336
+ " 'mainstream_model': [\n",
2337
+ " 'model', 'fashion model', 'instagram model', 'supermodel',\n",
2338
+ " 'runway model', 'commercial model'\n",
2339
+ " ],\n",
2340
+ " 'actor': [\n",
2341
+ " 'actor', 'actress', 'film actor', 'tv actor', 'television actor'\n",
2342
+ " ],\n",
2343
+ " 'musician': [\n",
2344
+ " 'singer', 'musician', 'singer/musician', 'music artist', 'vocalist'\n",
2345
+ " ],\n",
2346
+ " 'online_personality': [\n",
2347
+ " 'online personality', 'influencer', 'content creator', \n",
2348
+ " 'youtuber', 'streamer', 'social media personality'\n",
2349
+ " ],\n",
2350
+ " 'tv_personality': [\n",
2351
+ " 'tv personality', 'television personality', 'tv host', 'presenter'\n",
2352
+ " ]\n",
2353
+ "}\n",
2354
+ "\n",
2355
+ "# Position weights (1st mention = most important)\n",
2356
+ "POSITION_WEIGHTS = {\n",
2357
+ " 0: 3.0, # First position\n",
2358
+ " 1: 2.0, # Second position\n",
2359
+ " 2: 1.0 # Third position\n",
2360
+ "}\n",
2361
+ "\n",
2362
+ "# Model reliability weights (based on analysis)\n",
2363
+ "MODEL_WEIGHTS = {\n",
2364
+ " 'gemma': 1.0,\n",
2365
+ " 'mistral': 0.85, # Slightly lower due to 76% detection rate vs 92%\n",
2366
+ " 'qwen': 1.0\n",
2367
+ "}\n",
2368
+ "\n",
2369
+ "# ============================================================\n",
2370
+ "# UTILITY FUNCTIONS\n",
2371
+ "# ============================================================\n",
2372
+ "\n",
2373
+ "def normalize_value(value):\n",
2374
+ " \"\"\"Normalize values for comparison (handle NaN, whitespace, case)\"\"\"\n",
2375
+ " if pd.isna(value):\n",
2376
+ " return None\n",
2377
+ " return str(value).strip().lower()\n",
2378
+ "\n",
2379
+ "def is_unknown_value(value):\n",
2380
+ " \"\"\"Check if a value represents 'unknown' or similar non-informative values\"\"\"\n",
2381
+ " if value is None:\n",
2382
+ " return True\n",
2383
+ " \n",
2384
+ " value_str = str(value).strip().lower()\n",
2385
+ " \n",
2386
+ " unknown_patterns = [\n",
2387
+ " 'unknown', 'n/a', 'na', 'none', 'not specified', \n",
2388
+ " 'not available', 'unclear', 'uncertain', '', 'null'\n",
2389
+ " ]\n",
2390
+ " \n",
2391
+ " return value_str in unknown_patterns\n",
2392
+ "\n",
2393
+ "def parse_profession_list(profession_str):\n",
2394
+ " \"\"\"Parse comma-separated profession list into normalized list\"\"\"\n",
2395
+ " if pd.isna(profession_str):\n",
2396
+ " return []\n",
2397
+ " \n",
2398
+ " professions = [p.strip().lower() for p in str(profession_str).split(',')]\n",
2399
+ " return [p for p in professions if p and not is_unknown_value(p)]\n",
2400
+ "\n",
2401
+ "def find_profession_group(profession, groups=PROFESSION_GROUPS):\n",
2402
+ " \"\"\"Find which semantic group a profession belongs to\"\"\"\n",
2403
+ " profession_lower = profession.lower()\n",
2404
+ " \n",
2405
+ " for group_name, terms in groups.items():\n",
2406
+ " if profession_lower in terms:\n",
2407
+ " return group_name\n",
2408
+ " # Partial match for compound terms\n",
2409
+ " if any(term in profession_lower for term in terms):\n",
2410
+ " return group_name\n",
2411
+ " \n",
2412
+ " return profession_lower # Return as-is if no group found\n",
2413
+ "\n",
2414
+ "# ============================================================\n",
2415
+ "# CONSENSUS ALGORITHMS\n",
2416
+ "# ============================================================\n",
2417
+ "\n",
2418
+ "def get_position_weighted_consensus(profession_lists, model_names, \n",
2419
+ " weights=POSITION_WEIGHTS, \n",
2420
+ " model_weights=MODEL_WEIGHTS):\n",
2421
+ " \"\"\"\n",
2422
+ " Get consensus using position-based weighting.\n",
2423
+ " \n",
2424
+ " Professions mentioned first get more weight than those mentioned second or third.\n",
2425
+ " Different models can have different reliability weights.\n",
2426
+ " \n",
2427
+ " Returns: (consensus_profession, weighted_score, breakdown_dict)\n",
2428
+ " \"\"\"\n",
2429
+ " scores = {}\n",
2430
+ " breakdown = {}\n",
2431
+ " \n",
2432
+ " for model, prof_list in zip(model_names, profession_lists):\n",
2433
+ " professions = parse_profession_list(prof_list)\n",
2434
+ " model_weight = model_weights.get(model, 1.0)\n",
2435
+ " \n",
2436
+ " for position, profession in enumerate(professions[:3]): # Only top 3\n",
2437
+ " pos_weight = weights.get(position, 0)\n",
2438
+ " score = pos_weight * model_weight\n",
2439
+ " \n",
2440
+ " scores[profession] = scores.get(profession, 0) + score\n",
2441
+ " \n",
2442
+ " if profession not in breakdown:\n",
2443
+ " breakdown[profession] = []\n",
2444
+ " breakdown[profession].append({\n",
2445
+ " 'model': model,\n",
2446
+ " 'position': position + 1,\n",
2447
+ " 'weight': score\n",
2448
+ " })\n",
2449
+ " \n",
2450
+ " if not scores:\n",
2451
+ " return None, 0, {}\n",
2452
+ " \n",
2453
+ " best_profession = max(scores.items(), key=lambda x: x[1])\n",
2454
+ " return best_profession[0], best_profession[1], breakdown\n",
2455
+ "\n",
2456
+ "def get_semantic_consensus(profession_lists, model_names, required_agreement=2):\n",
2457
+ " \"\"\"\n",
2458
+ " Get consensus by grouping semantically similar professions.\n",
2459
+ " \n",
2460
+ " First votes on profession categories (e.g., \"adult entertainment\"),\n",
2461
+ " then picks the most common specific term within the winning category.\n",
2462
+ " \n",
2463
+ " Returns: (consensus_profession, agreement_count, category)\n",
2464
+ " \"\"\"\n",
2465
+ " category_votes = {}\n",
2466
+ " profession_within_category = {}\n",
2467
+ " \n",
2468
+ " for model, prof_list in zip(model_names, profession_lists):\n",
2469
+ " professions = parse_profession_list(prof_list)\n",
2470
+ " \n",
2471
+ " if not professions:\n",
2472
+ " continue\n",
2473
+ " \n",
2474
+ " # Use first profession from each model for category voting\n",
2475
+ " first_prof = professions[0]\n",
2476
+ " category = find_profession_group(first_prof)\n",
2477
+ " \n",
2478
+ " category_votes[category] = category_votes.get(category, 0) + 1\n",
2479
+ " \n",
2480
+ " if category not in profession_within_category:\n",
2481
+ " profession_within_category[category] = []\n",
2482
+ " profession_within_category[category].append(first_prof)\n",
2483
+ " \n",
2484
+ " if not category_votes:\n",
2485
+ " return None, 0, None\n",
2486
+ " \n",
2487
+ " # Find winning category\n",
2488
+ " winning_category, count = max(category_votes.items(), key=lambda x: x[1])\n",
2489
+ " \n",
2490
+ " if count < required_agreement:\n",
2491
+ " return None, count, winning_category\n",
2492
+ " \n",
2493
+ " # Pick most common specific term within winning category\n",
2494
+ " specific_terms = profession_within_category[winning_category]\n",
2495
+ " most_common_term = Counter(specific_terms).most_common(1)[0][0]\n",
2496
+ " \n",
2497
+ " return most_common_term, count, winning_category\n",
2498
+ "\n",
2499
+ "def get_any_position_consensus(profession_lists, target_profession, \n",
2500
+ " required_agreement=2):\n",
2501
+ " \"\"\"\n",
2502
+ " Check if target profession appears ANYWHERE in the lists.\n",
2503
+ " \n",
2504
+ " Useful for professions that are consistently mentioned but not always first.\n",
2505
+ " \n",
2506
+ " Returns: (found, agreement_count, positions_found)\n",
2507
+ " \"\"\"\n",
2508
+ " count = 0\n",
2509
+ " positions = []\n",
2510
+ " \n",
2511
+ " for prof_list in profession_lists:\n",
2512
+ " professions = parse_profession_list(prof_list)\n",
2513
+ " \n",
2514
+ " for i, prof in enumerate(professions):\n",
2515
+ " if target_profession.lower() in prof.lower():\n",
2516
+ " count += 1\n",
2517
+ " positions.append(i + 1)\n",
2518
+ " break\n",
2519
+ " \n",
2520
+ " return count >= required_agreement, count, positions\n",
2521
+ "\n",
2522
+ "def get_hybrid_consensus(profession_lists, model_names):\n",
2523
+ " \"\"\"\n",
2524
+ " Hybrid consensus strategy that tries multiple approaches.\n",
2525
+ " \n",
2526
+ " Strategy priority:\n",
2527
+ " 1. Check for \"adult performer\" anywhere in lists (special case)\n",
2528
+ " 2. Position-weighted consensus\n",
2529
+ " 3. Semantic consensus\n",
2530
+ " 4. Fallback to most common first profession\n",
2531
+ " \n",
2532
+ " Returns: (consensus_profession, method_used, confidence_score)\n",
2533
+ " \"\"\"\n",
2534
+ " # Special case: Check for adult entertainment professions\n",
2535
+ " adult_terms = PROFESSION_GROUPS['adult_entertainment']\n",
2536
+ " for term in adult_terms:\n",
2537
+ " found, count, positions = get_any_position_consensus(profession_lists, term, required_agreement=2)\n",
2538
+ " if found:\n",
2539
+ " # Use weighted consensus to pick the exact term\n",
2540
+ " weighted_prof, score, _ = get_position_weighted_consensus(profession_lists, model_names)\n",
2541
+ " \n",
2542
+ " # Check if the weighted winner is an adult entertainment term\n",
2543
+ " if any(term in weighted_prof for term in adult_terms):\n",
2544
+ " return weighted_prof, 'adult_special', score\n",
2545
+ " \n",
2546
+ " # If weighted winner isn't adult term, but 2+ models mentioned it, use it\n",
2547
+ " return term, 'adult_any_position', count * 2.0\n",
2548
+ " \n",
2549
+ " # Try position-weighted consensus\n",
2550
+ " weighted_prof, score, breakdown = get_position_weighted_consensus(profession_lists, model_names)\n",
2551
+ " \n",
2552
+ " if score >= 3.0: # Reasonable threshold (e.g., 2 models first position)\n",
2553
+ " return weighted_prof, 'weighted', score\n",
2554
+ " \n",
2555
+ " # Try semantic consensus\n",
2556
+ " semantic_prof, count, category = get_semantic_consensus(profession_lists, model_names)\n",
2557
+ " \n",
2558
+ " if count >= 2:\n",
2559
+ " return semantic_prof, 'semantic', count * 1.5\n",
2560
+ " \n",
2561
+ " # Fallback: most common first profession\n",
2562
+ " first_professions = []\n",
2563
+ " for prof_list in profession_lists:\n",
2564
+ " professions = parse_profession_list(prof_list)\n",
2565
+ " if professions:\n",
2566
+ " first_professions.append(professions[0])\n",
2567
+ " \n",
2568
+ " if first_professions:\n",
2569
+ " most_common = Counter(first_professions).most_common(1)[0]\n",
2570
+ " if most_common[1] >= 2:\n",
2571
+ " return most_common[0], 'simple_majority', most_common[1]\n",
2572
+ " \n",
2573
+ " return None, 'no_consensus', 0\n",
2574
+ "\n",
2575
+ "# ============================================================\n",
2576
+ "# ORIGINAL CONSENSUS (for comparison)\n",
2577
+ "# ============================================================\n",
2578
+ "\n",
2579
+ "def get_consensus_value_original(values, required_agreement=2):\n",
2580
+ " \"\"\"Original consensus method - compares entire strings\"\"\"\n",
2581
+ " normalized = [normalize_value(v) for v in values]\n",
2582
+ " valid_values = [v for v in normalized if v is not None]\n",
2583
+ " \n",
2584
+ " if not valid_values:\n",
2585
+ " return None, 0\n",
2586
+ " \n",
2587
+ " value_counts = Counter(valid_values)\n",
2588
+ " most_common_value, count = value_counts.most_common(1)[0]\n",
2589
+ " \n",
2590
+ " if count >= required_agreement:\n",
2591
+ " return most_common_value, count\n",
2592
+ " else:\n",
2593
+ " return None, count\n",
2594
+ "\n",
2595
+ "# ============================================================\n",
2596
+ "# MAIN CONSENSUS CREATION\n",
2597
+ "# ============================================================\n",
2598
+ "\n",
2599
+ "def create_improved_consensus(input_file, output_file, \n",
2600
+ " models=['gemma', 'mistral', 'qwen'],\n",
2601
+ " consensus_method='hybrid'):\n",
2602
+ " \"\"\"\n",
2603
+ " Create improved consensus CSV with better profession detection.\n",
2604
+ " \n",
2605
+ " Parameters:\n",
2606
+ " - input_file: Path to combined_llm_annotations.csv\n",
2607
+ " - output_file: Path to save improved consensus\n",
2608
+ " - models: List of model names\n",
2609
+ " - consensus_method: 'hybrid', 'weighted', 'semantic', or 'original'\n",
2610
+ " \"\"\"\n",
2611
+ " print(\"=\"*80)\n",
2612
+ " print(\"IMPROVED CONSENSUS CREATION\")\n",
2613
+ " print(\"=\"*80)\n",
2614
+ " print(f\"Method: {consensus_method}\")\n",
2615
+ " print(f\"Reading: {input_file}\")\n",
2616
+ " \n",
2617
+ " df = pd.read_csv(input_file)\n",
2618
+ " \n",
2619
+ " print(f\"Input shape: {df.shape}\")\n",
2620
+ " print(f\"Models: {', '.join(models)}\")\n",
2621
+ " \n",
2622
+ " consensus_data = []\n",
2623
+ " stats = {\n",
2624
+ " 'total_rows': len(df),\n",
2625
+ " 'country_fail': 0,\n",
2626
+ " 'gender_fail': 0,\n",
2627
+ " 'profession_fail': 0,\n",
2628
+ " 'unknown_values': 0,\n",
2629
+ " 'all_pass': 0,\n",
2630
+ " 'method_counts': {}\n",
2631
+ " }\n",
2632
+ " \n",
2633
+ " print(\"\\nProcessing rows...\")\n",
2634
+ " \n",
2635
+ " for idx, row in df.iterrows():\n",
2636
+ " if idx % 1000 == 0:\n",
2637
+ " print(f\" Processed {idx}/{len(df)} rows...\", end='\\r')\n",
2638
+ " \n",
2639
+ " # Get values for each field\n",
2640
+ " countries = [row[f'{model}_country'] for model in models]\n",
2641
+ " genders = [row[f'{model}_gender'] for model in models]\n",
2642
+ " professions = [row[f'{model}_profession_llm'] for model in models]\n",
2643
+ " \n",
2644
+ " # Country consensus (strict: all 3 must agree)\n",
2645
+ " country_consensus, country_count = get_consensus_value_original(countries, required_agreement=3)\n",
2646
+ " \n",
2647
+ " # Gender consensus (strict: all 3 must agree)\n",
2648
+ " gender_consensus, gender_count = get_consensus_value_original(genders, required_agreement=3)\n",
2649
+ " \n",
2650
+ " # Profession consensus (IMPROVED)\n",
2651
+ " if consensus_method == 'hybrid':\n",
2652
+ " profession_consensus, method, prof_score = get_hybrid_consensus(professions, models)\n",
2653
+ " prof_count = int(prof_score / 1.5) # Rough conversion to count\n",
2654
+ " elif consensus_method == 'weighted':\n",
2655
+ " profession_consensus, prof_score, _ = get_position_weighted_consensus(professions, models)\n",
2656
+ " prof_count = int(prof_score / 2)\n",
2657
+ " method = 'weighted'\n",
2658
+ " elif consensus_method == 'semantic':\n",
2659
+ " profession_consensus, prof_count, _ = get_semantic_consensus(professions, models)\n",
2660
+ " method = 'semantic'\n",
2661
+ " prof_score = prof_count\n",
2662
+ " else: # original\n",
2663
+ " # Get first profession from each model\n",
2664
+ " first_profs = [parse_profession_list(p)[0] if parse_profession_list(p) else None \n",
2665
+ " for p in professions]\n",
2666
+ " profession_consensus, prof_count = get_consensus_value_original(first_profs, required_agreement=2)\n",
2667
+ " method = 'original'\n",
2668
+ " prof_score = prof_count\n",
2669
+ " \n",
2670
+ " # Track method usage\n",
2671
+ " stats['method_counts'][method] = stats['method_counts'].get(method, 0) + 1\n",
2672
+ " \n",
2673
+ " # Determine if row passes\n",
2674
+ " country_pass = country_count == 3\n",
2675
+ " gender_pass = gender_count == 3\n",
2676
+ " profession_pass = profession_consensus is not None\n",
2677
+ " \n",
2678
+ " has_unknown = (\n",
2679
+ " is_unknown_value(country_consensus) or \n",
2680
+ " is_unknown_value(gender_consensus) or \n",
2681
+ " is_unknown_value(profession_consensus)\n",
2682
+ " )\n",
2683
+ " \n",
2684
+ " if not country_pass:\n",
2685
+ " stats['country_fail'] += 1\n",
2686
+ " if not gender_pass:\n",
2687
+ " stats['gender_fail'] += 1\n",
2688
+ " if not profession_pass:\n",
2689
+ " stats['profession_fail'] += 1\n",
2690
+ " if has_unknown:\n",
2691
+ " stats['unknown_values'] += 1\n",
2692
+ " \n",
2693
+ " if country_pass and gender_pass and profession_pass and not has_unknown:\n",
2694
+ " stats['all_pass'] += 1\n",
2695
+ " \n",
2696
+ " consensus_data.append({\n",
2697
+ " 'row_index': idx,\n",
2698
+ " 'consensus_country': country_consensus,\n",
2699
+ " 'consensus_gender': gender_consensus,\n",
2700
+ " 'consensus_profession': profession_consensus,\n",
2701
+ " 'profession_method': method,\n",
2702
+ " 'profession_confidence': prof_score\n",
2703
+ " })\n",
2704
+ " \n",
2705
+ " print(f\"\\n Processed {len(df)} rows. \")\n",
2706
+ " \n",
2707
+ " # Create result dataframe\n",
2708
+ " if consensus_data:\n",
2709
+ " result_df = df.iloc[[c['row_index'] for c in consensus_data]].copy().reset_index(drop=True)\n",
2710
+ " \n",
2711
+ " # Add consensus columns\n",
2712
+ " for key in ['consensus_country', 'consensus_gender', 'consensus_profession', \n",
2713
+ " 'profession_method', 'profession_confidence']:\n",
2714
+ " result_df[key] = [c[key] for c in consensus_data]\n",
2715
+ " \n",
2716
+ " # Reorder columns (consensus columns first)\n",
2717
+ " consensus_cols = ['consensus_country', 'consensus_gender', 'consensus_profession',\n",
2718
+ " 'profession_method', 'profession_confidence']\n",
2719
+ " other_cols = [c for c in result_df.columns if c not in consensus_cols]\n",
2720
+ " result_df = result_df[consensus_cols + other_cols]\n",
2721
+ " \n",
2722
+ " # Save\n",
2723
+ " result_df.to_csv(output_file, index=False)\n",
2724
+ " \n",
2725
+ " print(\"\\n\" + \"=\"*80)\n",
2726
+ " print(\"RESULTS\")\n",
2727
+ " print(\"=\"*80)\n",
2728
+ " print(f\"Total input rows: {stats['total_rows']:,}\")\n",
2729
+ " print(f\"Rows passing all criteria: {stats['all_pass']:,} ({stats['all_pass']/stats['total_rows']*100:.1f}%)\")\n",
2730
+ " \n",
2731
+ " print(f\"\\nConsensus method usage:\")\n",
2732
+ " for method, count in sorted(stats['method_counts'].items(), key=lambda x: -x[1]):\n",
2733
+ " print(f\" - {method}: {count:,} ({count/stats['all_pass']*100:.1f}%)\")\n",
2734
+ " \n",
2735
+ " print(\"\\n\" + \"=\"*80)\n",
2736
+ " print(\"PROFESSION DISTRIBUTION\")\n",
2737
+ " print(\"=\"*80)\n",
2738
+ " print(\"\\nTop 20 professions:\")\n",
2739
+ " print(result_df['consensus_profession'].value_counts().head(20))\n",
2740
+ " \n",
2741
+ " # Specifically check adult performer\n",
2742
+ " adult_count = result_df['consensus_profession'].apply(\n",
2743
+ " lambda x: 'adult' in str(x).lower() if pd.notna(x) else False\n",
2744
+ " ).sum()\n",
2745
+ " print(f\"\\n🎯 Adult performer variants: {adult_count} ({adult_count/len(result_df)*100:.2f}%)\")\n",
2746
+ " \n",
2747
+ " print(\"\\n\" + \"=\"*80)\n",
2748
+ " print(f\"✓ Improved consensus saved to: {output_file.name}\")\n",
2749
+ " print(f\" Total rows: {len(result_df):,}\")\n",
2750
+ " print(\"=\"*80)\n",
2751
+ " \n",
2752
+ " return result_df\n",
2753
+ " else:\n",
2754
+ " print(\"\\n⚠ WARNING: No rows passed all criteria!\")\n",
2755
+ " return pd.DataFrame()\n",
2756
+ "\n",
2757
+ "# ============================================================\n",
2758
+ "# COMPARISON FUNCTION\n",
2759
+ "# ============================================================\n",
2760
+ "\n",
2761
+ "def compare_consensus_methods(input_file, models=['gemma', 'mistral', 'qwen']):\n",
2762
+ " \"\"\"Compare different consensus methods side by side\"\"\"\n",
2763
+ " \n",
2764
+ " print(\"=\"*80)\n",
2765
+ " print(\"CONSENSUS METHOD COMPARISON\")\n",
2766
+ " print(\"=\"*80)\n",
2767
+ " \n",
2768
+ " methods = ['original', 'weighted', 'semantic', 'hybrid']\n",
2769
+ " results = {}\n",
2770
+ " \n",
2771
+ " for method in methods:\n",
2772
+ " print(f\"\\n--- Testing {method} method ---\")\n",
2773
+ " \n",
2774
+ " output_file = Path(input_file).parent / f\"consensus_{method}.csv\"\n",
2775
+ " result_df = create_improved_consensus(input_file, output_file, models, method)\n",
2776
+ " \n",
2777
+ " if len(result_df) > 0:\n",
2778
+ " adult_count = result_df['consensus_profession'].apply(\n",
2779
+ " lambda x: 'adult' in str(x).lower() if pd.notna(x) else False\n",
2780
+ " ).sum()\n",
2781
+ " \n",
2782
+ " results[method] = {\n",
2783
+ " 'total_rows': len(result_df),\n",
2784
+ " 'adult_performer_count': adult_count,\n",
2785
+ " 'adult_performer_pct': adult_count / len(result_df) * 100\n",
2786
+ " }\n",
2787
+ " \n",
2788
+ " print(\"\\n\" + \"=\"*80)\n",
2789
+ " print(\"COMPARISON SUMMARY\")\n",
2790
+ " print(\"=\"*80)\n",
2791
+ " \n",
2792
+ " print(f\"\\n{'Method':<15} {'Total Rows':<12} {'Adult Performer':<16} {'% Adult':<10}\")\n",
2793
+ " print(\"-\" * 65)\n",
2794
+ " \n",
2795
+ " for method, stats in results.items():\n",
2796
+ " print(f\"{method:<15} {stats['total_rows']:<12,} {stats['adult_performer_count']:<16,} {stats['adult_performer_pct']:<10.2f}%\")\n",
2797
+ " \n",
2798
+ " if 'original' in results and 'hybrid' in results:\n",
2799
+ " improvement = results['hybrid']['adult_performer_count'] - results['original']['adult_performer_count']\n",
2800
+ " pct_improvement = improvement / results['original']['adult_performer_count'] * 100\n",
2801
+ " \n",
2802
+ " print(f\"\\n✨ Hybrid method improvement over original:\")\n",
2803
+ " print(f\" +{improvement} adult performer cases (+{pct_improvement:.1f}%)\")\n",
2804
+ "\n",
2805
+ "# ============================================================\n",
2806
+ "# MAIN EXECUTION\n",
2807
+ "# ============================================================\n",
2808
+ "\n",
2809
+ "if __name__ == \"__main__\":\n",
2810
+ " current_dir = Path.cwd()\n",
2811
+ " \n",
2812
+ " input_file = current_dir.parent / \"data/CSV/combined_llm_annotations.csv\"\n",
2813
+ " output_file = current_dir.parent / \"data/CSV/improved_consensus.csv\"\n",
2814
+ " \n",
2815
+ " if not input_file.exists():\n",
2816
+ " print(f\"Error: Input file not found: {input_file}\")\n",
2817
+ " else:\n",
2818
+ " # Run comparison (comment out if you just want hybrid)\n",
2819
+ " # compare_consensus_methods(input_file)\n",
2820
+ " \n",
2821
+ " # Or run single method (hybrid recommended)\n",
2822
+ " result_df = create_improved_consensus(\n",
2823
+ " input_file, \n",
2824
+ " output_file, \n",
2825
+ " models=['gemma', 'mistral', 'qwen'],\n",
2826
+ " consensus_method='hybrid'\n",
2827
+ " )\n",
2828
+ " \n",
2829
+ " print(\"\\n✅ Complete!\")"
2830
+ ]
2831
+ },
2832
  {
2833
  "cell_type": "code",
2834
  "execution_count": null,
 
2840
  ],
2841
  "metadata": {
2842
  "kernelspec": {
2843
+ "display_name": "latm",
2844
  "language": "python",
2845
  "name": "python3"
2846
  },
 
2854
  "name": "python",
2855
  "nbconvert_exporter": "python",
2856
  "pygments_lexer": "ipython3",
2857
+ "version": "3.10.15"
2858
  }
2859
  },
2860
  "nbformat": 4,
jupyter_notebooks/Section_2-3-4__Figure_8a_sunburst_gender.ipynb CHANGED
@@ -9,14 +9,14 @@
9
  },
10
  {
11
  "cell_type": "code",
12
- "execution_count": 8,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
- "8a.json\n"
20
  ]
21
  }
22
  ],
@@ -29,9 +29,10 @@
29
  "current_dir = Path.cwd()\n",
30
  "sunburst_json = current_dir.parent / \"public/json/8a.json\"\n",
31
  "\n",
 
 
 
32
  "\n",
33
- "aggregated_poi = current_dir.parent / \"data/CSV/Deepseek_annotated_POI_aggregated.csv\"\n",
34
- "df = pd.read_csv(aggregated_poi)\n",
35
  "# ---- Normalize Gender (group Non-binary and Unknown into 'Other') ----\n",
36
  "def normalize_gender(g):\n",
37
  " g = str(g).strip().lower()\n",
@@ -42,29 +43,30 @@
42
  " else:\n",
43
  " return \"Other\"\n",
44
  "\n",
45
- "df['gender_normalized'] = df['gender'].apply(normalize_gender)\n",
46
  "\n",
47
  "# ---- Step 1: Limit to top 10 countries ----\n",
48
- "df['country_cleaned'] = df['country'].apply(lambda x: x if x not in ['Unknown', '', None] else 'Other')\n",
49
  "top_countries = df['country_cleaned'].value_counts().nlargest(12).index.tolist()\n",
50
  "df['country_limited'] = df['country_cleaned'].apply(lambda x: x if x in top_countries else 'Other')\n",
51
  "\n",
52
  "# ---- Step 2: Normalize and limit professions ----\n",
53
  "valid_categories = [\n",
54
- " \"Actor\", \"Adult Performer\", \"Singer, Musician\", \"Model\",\n",
55
- " \"Online Personality\", \"TV Personality\", \"Voice Actor\",\"Public Figure\", \"Sports Professional\"\n",
56
  "]\n",
57
  "\n",
58
  "def remap_profession(profession):\n",
59
- " if profession == 'Unknown' or profession not in valid_categories:\n",
 
60
  " return 'Other'\n",
61
- " elif profession == 'Fictional Character':\n",
62
- " return 'Actor'\n",
63
- " elif profession == 'Voice actor':\n",
64
- " return 'Voice Actor'\n",
65
- " return profession\n",
66
  "\n",
67
- "df['profession_limited'] = df['mapped_profession'].apply(remap_profession)\n",
68
  "\n",
69
  "# ---- Step 3: Group by gender and profession ----\n",
70
  "sunburst_data = df.groupby(['gender_normalized', 'profession_limited']).size().reset_index(name='count')\n",
@@ -92,7 +94,7 @@
92
  "with open(sunburst_json, \"w\", encoding='utf-8') as f:\n",
93
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
94
  "\n",
95
- "print(\"8a.json\")\n"
96
  ]
97
  },
98
  {
 
9
  },
10
  {
11
  "cell_type": "code",
12
+ "execution_count": 1,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
+ "✓ Saved 8a.json\n"
20
  ]
21
  }
22
  ],
 
29
  "current_dir = Path.cwd()\n",
30
  "sunburst_json = current_dir.parent / \"public/json/8a.json\"\n",
31
  "\n",
32
+ "# Load consensus CSV\n",
33
+ "consensus_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement_consensus.csv\"\n",
34
+ "df = pd.read_csv(consensus_file)\n",
35
  "\n",
 
 
36
  "# ---- Normalize Gender (group Non-binary and Unknown into 'Other') ----\n",
37
  "def normalize_gender(g):\n",
38
  " g = str(g).strip().lower()\n",
 
43
  " else:\n",
44
  " return \"Other\"\n",
45
  "\n",
46
+ "df['gender_normalized'] = df['consensus_gender'].apply(normalize_gender)\n",
47
  "\n",
48
  "# ---- Step 1: Limit to top 10 countries ----\n",
49
+ "df['country_cleaned'] = df['consensus_country'].apply(lambda x: x if x not in ['Unknown', '', None] else 'Other')\n",
50
  "top_countries = df['country_cleaned'].value_counts().nlargest(12).index.tolist()\n",
51
  "df['country_limited'] = df['country_cleaned'].apply(lambda x: x if x in top_countries else 'Other')\n",
52
  "\n",
53
  "# ---- Step 2: Normalize and limit professions ----\n",
54
  "valid_categories = [\n",
55
+ " \"actor\", \"adult performer\", \"singer/musician\", \"model\",\n",
56
+ " \"online personality\", \"tv personality\", \"voice actor/asmr\", \"public figure\", \"sports professional\"\n",
57
  "]\n",
58
  "\n",
59
  "def remap_profession(profession):\n",
60
+ " profession_lower = str(profession).strip().lower()\n",
61
+ " if profession_lower == 'unknown' or profession_lower not in valid_categories:\n",
62
  " return 'Other'\n",
63
+ " elif profession_lower == 'fictional character':\n",
64
+ " return 'actor'\n",
65
+ " elif profession_lower in ['voice actor', 'voice actor/asmr']:\n",
66
+ " return 'voice actor/ASMR'\n",
67
+ " return profession_lower\n",
68
  "\n",
69
+ "df['profession_limited'] = df['consensus_primary_profession'].apply(remap_profession)\n",
70
  "\n",
71
  "# ---- Step 3: Group by gender and profession ----\n",
72
  "sunburst_data = df.groupby(['gender_normalized', 'profession_limited']).size().reset_index(name='count')\n",
 
94
  "with open(sunburst_json, \"w\", encoding='utf-8') as f:\n",
95
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
96
  "\n",
97
+ "print(\"✓ Saved 8a.json\")\n"
98
  ]
99
  },
100
  {
jupyter_notebooks/Section_2-3-4__Figure_8b_sunburst_profession.ipynb CHANGED
@@ -9,14 +9,14 @@
9
  },
10
  {
11
  "cell_type": "code",
12
- "execution_count": 5,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
- "✅ Sunburst data saved to sunburst_data.json\n"
20
  ]
21
  }
22
  ],
@@ -30,60 +30,273 @@
30
  "\n",
31
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
32
  "\n",
 
 
 
33
  "\n",
34
- "aggregated_poi = current_dir.parent / \"data/CSV/Deepseek_annotated_POI_aggregated.csv\"\n",
 
 
35
  "\n",
36
- "df = pd.read_csv(aggregated_poi)\n",
37
- "df['country_cleaned'] = df['country'].apply(lambda x: x if x not in ['Unknown', '', None] else 'Other')\n",
 
 
38
  "\n",
39
- "# Now get top countries excluding what was forced into 'Other'\n",
40
- "top_countries = df['country_cleaned'].value_counts().nlargest(15).index.tolist()\n",
 
 
 
 
 
 
 
41
  "\n",
42
- "# Final limited country column\n",
43
- "df['country_limited'] = df['country_cleaned'].apply(lambda x: x if x in top_countries else 'Other')\n",
44
  "\n",
45
- "# ---- Step 2: Limit to top 7 professions and combine Unknown and Sports Professional with Other ----\n",
46
- "top_professions = df['mapped_profession'].value_counts().nlargest(7).index.tolist()\n",
47
  "\n",
48
- "# Explicitly remove 'Unknown' and 'Sports Professional' even if they are in the top 7\n",
49
- "top_professions = [p for p in top_professions if p not in ['Unknown', 'Sports Professional']]\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
50
  "\n",
51
- "# Normalize 'Unknown' and '\n",
52
- "# Sports Professional' into 'Other'\n",
53
- "# Normalize 'Unknown' and 'Sports Professional' into 'Other'\n",
54
- "df['profession_limited'] = df['mapped_profession'].apply(\n",
55
- " lambda x: 'Other' if x in ['Unknown', 'Sports Professional'] or x not in top_professions else x\n",
 
56
  ")\n",
57
  "\n",
 
 
 
 
58
  "\n",
 
 
 
59
  "\n",
60
- "# ---- Step 3: Group by the limited country and profession ----\n",
61
- "sunburst_data = df.groupby(['country_limited', 'profession_limited']).size().reset_index(name='count')\n",
 
 
 
62
  "\n",
63
- "# ---- Step 4: Create a nested structure for D3.js ----\n",
64
  "sunburst_dict = {\"name\": \"root\", \"children\": []}\n",
65
  "country_map = defaultdict(list)\n",
66
  "\n",
67
  "for _, row in sunburst_data.iterrows():\n",
68
- " country = row['country_limited']\n",
69
- " profession = row['profession_limited']\n",
70
- " count = int(row['count'])\n",
71
- " country_map[country].append({\"name\": profession, \"value\": count})\n",
 
 
 
 
 
 
72
  "\n",
73
- "# For each country, sort the profession list so that \"Other\" appears at the end\n",
74
- "for country, professions in country_map.items():\n",
75
- " professions_sorted = sorted(professions, key=lambda d: (d[\"name\"] == \"Other\", d[\"name\"]))\n",
76
- " country_map[country] = professions_sorted\n",
 
77
  "\n",
78
- "for country, professions in country_map.items():\n",
79
- " sunburst_dict[\"children\"].append({\"name\": country, \"children\": professions})\n",
80
  "\n",
81
- "# ---- Step 5: Save to a JSON file ----\n",
82
- "with open(sunburst_path, \"w\", encoding='utf-8') as f:\n",
 
 
 
 
 
 
 
83
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
84
  "\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
85
  "\n",
86
- "print(\"✅ Sunburst data saved to sunburst_data.json\")\n"
87
  ]
88
  },
89
  {
@@ -101,7 +314,7 @@
101
  ],
102
  "metadata": {
103
  "kernelspec": {
104
- "display_name": "Python 3 (ipykernel)",
105
  "language": "python",
106
  "name": "python3"
107
  },
@@ -115,7 +328,7 @@
115
  "name": "python",
116
  "nbconvert_exporter": "python",
117
  "pygments_lexer": "ipython3",
118
- "version": "3.12.10"
119
  }
120
  },
121
  "nbformat": 4,
 
9
  },
10
  {
11
  "cell_type": "code",
12
+ "execution_count": 13,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
+ "✓ Saved sunburst_countries_A.json\n"
20
  ]
21
  }
22
  ],
 
30
  "\n",
31
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
32
  "\n",
33
+ "# Load consensus CSV\n",
34
+ "consensus_file = current_dir.parent / \"data/CSV/improved_consensus.csv\"\n",
35
+ "df = pd.read_csv(consensus_file)\n",
36
  "\n",
37
+ "# ============================================================\n",
38
+ "# 1. NORMALIZE COUNTRIES\n",
39
+ "# ============================================================\n",
40
  "\n",
41
+ "def normalize_country(x: str):\n",
42
+ " if not isinstance(x, str) or x.strip() == \"\" or x.lower() == \"unknown\":\n",
43
+ " return \"Other\"\n",
44
+ " x = x.strip()\n",
45
  "\n",
46
+ " # Ensure matching with your HTML manualOrder\n",
47
+ " replacements = {\n",
48
+ " \"USA\": \"United States\",\n",
49
+ " \"US\": \"United States\",\n",
50
+ " \"U.S.\": \"United States\",\n",
51
+ " \"UK\": \"United Kingdom\",\n",
52
+ " \"U.K.\": \"United Kingdom\"\n",
53
+ " }\n",
54
+ " return replacements.get(x, x)\n",
55
  "\n",
56
+ "df[\"country_clean\"] = df[\"consensus_country\"].apply(normalize_country)\n",
 
57
  "\n",
58
+ "# Limit to the top 15 countries, merge the rest into \"Other\"\n",
59
+ "top_countries = df[\"country_clean\"].value_counts().nlargest(25).index.tolist()\n",
60
  "\n",
61
+ "df[\"country_limited\"] = df[\"country_clean\"].apply(\n",
62
+ " lambda x: x if x in top_countries else \"Other\"\n",
63
+ ")\n",
64
+ "\n",
65
+ "# ============================================================\n",
66
+ "# 2. NORMALIZE PROFESSIONS\n",
67
+ "# ============================================================\n",
68
+ "\n",
69
+ "def normalize_profession(x: str):\n",
70
+ " if not isinstance(x, str) or x.strip() == \"\" or x.lower() == \"unknown\":\n",
71
+ " return \"Other\"\n",
72
+ " x = x.strip().lower()\n",
73
+ " \n",
74
+ " mapping = {\n",
75
+ " \"actor\": \"Actor\",\n",
76
+ " \"model\": \"Model\",\n",
77
+ " \"adult performer\": \"Adult Performer\",\n",
78
+ " \"singer/musician\": \"Singer, Musician\",\n",
79
+ " \"online personality\": \"Online Personality\",\n",
80
+ " \"sports professional\": \"Sports Professional\",\n",
81
+ " \"voice actor/asmr\": \"Voice Actor\", # ← fixed key\n",
82
+ " \"public figure\": \"Public Figure\", # ← now its own category\n",
83
+ " \"tv personality\": \"Other\",\n",
84
+ " }\n",
85
+ " return mapping.get(x, \"Other\")\n",
86
+ "\n",
87
+ "df[\"profession_clean\"] = df[\"consensus_profession\"].apply(normalize_profession)\n",
88
  "\n",
89
+ "# Only keep the top 7 professions (excluding Other)\n",
90
+ "top_prof = (\n",
91
+ " df[df[\"profession_clean\"] != \"Other\"][\"profession_clean\"]\n",
92
+ " .value_counts()\n",
93
+ " .nlargest(7)\n",
94
+ " .index.tolist()\n",
95
  ")\n",
96
  "\n",
97
+ "# Re-limit professions, everything else → Other\n",
98
+ "df[\"profession_limited\"] = df[\"profession_clean\"].apply(\n",
99
+ " lambda x: x if x in top_prof else \"Other\"\n",
100
+ ")\n",
101
  "\n",
102
+ "# ============================================================\n",
103
+ "# 3. GROUP INTO SUNBURST STRUCTURE\n",
104
+ "# ============================================================\n",
105
  "\n",
106
+ "sunburst_data = (\n",
107
+ " df.groupby([\"country_limited\", \"profession_limited\"])\n",
108
+ " .size()\n",
109
+ " .reset_index(name=\"count\")\n",
110
+ ")\n",
111
  "\n",
 
112
  "sunburst_dict = {\"name\": \"root\", \"children\": []}\n",
113
  "country_map = defaultdict(list)\n",
114
  "\n",
115
  "for _, row in sunburst_data.iterrows():\n",
116
+ " c = row[\"country_limited\"]\n",
117
+ " p = row[\"profession_limited\"]\n",
118
+ " v = int(row[\"count\"])\n",
119
+ "\n",
120
+ " country_map[c].append({\"name\": p, \"value\": v})\n",
121
+ "\n",
122
+ "# Sort professions inside each country so \"Other\" is last\n",
123
+ "for c, profs in country_map.items():\n",
124
+ " profs_sorted = sorted(profs, key=lambda d: (d[\"name\"] == \"Other\", d[\"name\"]))\n",
125
+ " country_map[c] = profs_sorted\n",
126
  "\n",
127
+ "# Calculate total datapoints per country and sort\n",
128
+ "country_totals = []\n",
129
+ "for c, profs in country_map.items():\n",
130
+ " total = sum(p[\"value\"] for p in profs)\n",
131
+ " country_totals.append((c, total, profs))\n",
132
  "\n",
133
+ "# Sort by total (descending), but put \"Other\" last\n",
134
+ "country_totals.sort(key=lambda x: (x[0] == \"Other\", -x[1]))\n",
135
  "\n",
136
+ "# Build final JSON with sorted countries\n",
137
+ "for c, total, profs in country_totals:\n",
138
+ " sunburst_dict[\"children\"].append({\"name\": c, \"children\": profs})\n",
139
+ "\n",
140
+ "# ============================================================\n",
141
+ "# 4. SAVE JSON\n",
142
+ "# ============================================================\n",
143
+ "\n",
144
+ "with open(sunburst_path, \"w\", encoding=\"utf-8\") as f:\n",
145
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
146
  "\n",
147
+ "print(\"✓ Saved sunburst_countries_A.json\")"
148
+ ]
149
+ },
150
+ {
151
+ "cell_type": "markdown",
152
+ "metadata": {},
153
+ "source": [
154
+ "# version that only considers data up until dec 31st 2024"
155
+ ]
156
+ },
157
+ {
158
+ "cell_type": "code",
159
+ "execution_count": 14,
160
+ "metadata": {},
161
+ "outputs": [
162
+ {
163
+ "name": "stdout",
164
+ "output_type": "stream",
165
+ "text": [
166
+ "✓ Filtered to 17356 records published on or before December 31, 2024\n",
167
+ "✓ Saved sunburst_countries_A.json (2024 data only)\n"
168
+ ]
169
+ }
170
+ ],
171
+ "source": [
172
+ "import pandas as pd\n",
173
+ "from collections import defaultdict\n",
174
+ "import json\n",
175
+ "from pathlib import Path\n",
176
+ "\n",
177
+ "current_dir = Path.cwd()\n",
178
+ "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
179
+ "\n",
180
+ "# Load consensus CSV\n",
181
+ "consensus_file = current_dir.parent / \"data/CSV/improved_consensus.csv\"\n",
182
+ "df = pd.read_csv(consensus_file)\n",
183
+ "\n",
184
+ "# ============================================================\n",
185
+ "# FILTER DATA UP TO DECEMBER 31, 2024\n",
186
+ "# ============================================================\n",
187
+ "# Convert publishedAt to datetime\n",
188
+ "df[\"publishedAt\"] = pd.to_datetime(df[\"publishedAt\"], errors=\"coerce\", utc=True)\n",
189
+ "\n",
190
+ "# Filter to only include data up to December 31, 2024\n",
191
+ "# Make cutoff_date timezone-aware (UTC) to match publishedAt\n",
192
+ "cutoff_date = pd.Timestamp(\"2024-12-31 23:59:59\", tz=\"UTC\")\n",
193
+ "df = df[df[\"publishedAt\"] <= cutoff_date]\n",
194
+ "\n",
195
+ "print(f\"✓ Filtered to {len(df)} records published on or before December 31, 2024\")\n",
196
+ "\n",
197
+ "# ============================================================\n",
198
+ "# 1. NORMALIZE COUNTRIES\n",
199
+ "# ============================================================\n",
200
+ "def normalize_country(x: str):\n",
201
+ " if not isinstance(x, str) or x.strip() == \"\" or x.lower() == \"unknown\":\n",
202
+ " return \"Other\"\n",
203
+ " x = x.strip()\n",
204
+ " # Ensure matching with your HTML manualOrder\n",
205
+ " replacements = {\n",
206
+ " \"USA\": \"United States\",\n",
207
+ " \"US\": \"United States\",\n",
208
+ " \"U.S.\": \"United States\",\n",
209
+ " \"UK\": \"United Kingdom\",\n",
210
+ " \"U.K.\": \"United Kingdom\"\n",
211
+ " }\n",
212
+ " return replacements.get(x, x)\n",
213
+ "\n",
214
+ "df[\"country_clean\"] = df[\"consensus_country\"].apply(normalize_country)\n",
215
+ "\n",
216
+ "# Limit to the top 15 countries, merge the rest into \"Other\"\n",
217
+ "top_countries = df[\"country_clean\"].value_counts().nlargest(25).index.tolist()\n",
218
+ "df[\"country_limited\"] = df[\"country_clean\"].apply(\n",
219
+ " lambda x: x if x in top_countries else \"Other\"\n",
220
+ ")\n",
221
+ "\n",
222
+ "# ============================================================\n",
223
+ "# 2. NORMALIZE PROFESSIONS\n",
224
+ "# ============================================================\n",
225
+ "def normalize_profession(x: str):\n",
226
+ " if not isinstance(x, str) or x.strip() == \"\" or x.lower() == \"unknown\":\n",
227
+ " return \"Other\"\n",
228
+ " x = x.strip().lower()\n",
229
+ " mapping = {\n",
230
+ " \"actor\": \"Actor\",\n",
231
+ " \"model\": \"Model\",\n",
232
+ " \"adult performer\": \"Adult Performer\",\n",
233
+ " \"singer/musician\": \"Singer, Musician\",\n",
234
+ " \"online personality\": \"Online Personality\",\n",
235
+ " \"sports professional\": \"Sports Professional\",\n",
236
+ " \"voice actor/asmr\": \"Voice Actor\", # ← fixed key\n",
237
+ " \"public figure\": \"Public Figure\", # ← now its own category\n",
238
+ " \"tv personality\": \"Other\",\n",
239
+ " }\n",
240
+ " return mapping.get(x, \"Other\")\n",
241
+ "\n",
242
+ "df[\"profession_clean\"] = df[\"consensus_profession\"].apply(normalize_profession)\n",
243
+ "\n",
244
+ "# Only keep the top 7 professions (excluding Other)\n",
245
+ "top_prof = (\n",
246
+ " df[df[\"profession_clean\"] != \"Other\"][\"profession_clean\"]\n",
247
+ " .value_counts()\n",
248
+ " .nlargest(7)\n",
249
+ " .index.tolist()\n",
250
+ ")\n",
251
+ "\n",
252
+ "# Re-limit professions, everything else → Other\n",
253
+ "df[\"profession_limited\"] = df[\"profession_clean\"].apply(\n",
254
+ " lambda x: x if x in top_prof else \"Other\"\n",
255
+ ")\n",
256
+ "\n",
257
+ "# ============================================================\n",
258
+ "# 3. GROUP INTO SUNBURST STRUCTURE\n",
259
+ "# ============================================================\n",
260
+ "sunburst_data = (\n",
261
+ " df.groupby([\"country_limited\", \"profession_limited\"])\n",
262
+ " .size()\n",
263
+ " .reset_index(name=\"count\")\n",
264
+ ")\n",
265
+ "\n",
266
+ "sunburst_dict = {\"name\": \"root\", \"children\": []}\n",
267
+ "country_map = defaultdict(list)\n",
268
+ "\n",
269
+ "for _, row in sunburst_data.iterrows():\n",
270
+ " c = row[\"country_limited\"]\n",
271
+ " p = row[\"profession_limited\"]\n",
272
+ " v = int(row[\"count\"])\n",
273
+ " country_map[c].append({\"name\": p, \"value\": v})\n",
274
+ "\n",
275
+ "# Sort professions inside each country so \"Other\" is last\n",
276
+ "for c, profs in country_map.items():\n",
277
+ " profs_sorted = sorted(profs, key=lambda d: (d[\"name\"] == \"Other\", d[\"name\"]))\n",
278
+ " country_map[c] = profs_sorted\n",
279
+ "\n",
280
+ "# Calculate total datapoints per country and sort\n",
281
+ "country_totals = []\n",
282
+ "for c, profs in country_map.items():\n",
283
+ " total = sum(p[\"value\"] for p in profs)\n",
284
+ " country_totals.append((c, total, profs))\n",
285
+ "\n",
286
+ "# Sort by total (descending), but put \"Other\" last\n",
287
+ "country_totals.sort(key=lambda x: (x[0] == \"Other\", -x[1]))\n",
288
+ "\n",
289
+ "# Build final JSON with sorted countries\n",
290
+ "for c, total, profs in country_totals:\n",
291
+ " sunburst_dict[\"children\"].append({\"name\": c, \"children\": profs})\n",
292
+ "\n",
293
+ "# ============================================================\n",
294
+ "# 4. SAVE JSON\n",
295
+ "# ============================================================\n",
296
+ "with open(sunburst_path, \"w\", encoding=\"utf-8\") as f:\n",
297
+ " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
298
  "\n",
299
+ "print(\"✓ Saved sunburst_countries_A.json (2024 data only)\")"
300
  ]
301
  },
302
  {
 
314
  ],
315
  "metadata": {
316
  "kernelspec": {
317
+ "display_name": "latm",
318
  "language": "python",
319
  "name": "python3"
320
  },
 
328
  "name": "python",
329
  "nbconvert_exporter": "python",
330
  "pygments_lexer": "ipython3",
331
+ "version": "3.10.15"
332
  }
333
  },
334
  "nbformat": 4,
public/Figure_8a_barchart.html ADDED
@@ -0,0 +1,244 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <!DOCTYPE html>
2
+ <meta charset="utf-8">
3
+ <style>
4
+ body {
5
+ font-family: "Segoe UI", sans-serif;
6
+ margin: 20px;
7
+ }
8
+
9
+ svg {
10
+ font: 12px sans-serif;
11
+ }
12
+
13
+ .bar {
14
+ cursor: pointer;
15
+ }
16
+
17
+ .bar:hover {
18
+ opacity: 0.8;
19
+ }
20
+
21
+ .axis text {
22
+ font-size: 11px;
23
+ }
24
+
25
+ .legend {
26
+ font-size: 12px;
27
+ }
28
+
29
+ .legend rect {
30
+ stroke-width: 1;
31
+ stroke: #000;
32
+ }
33
+
34
+ h2 {
35
+ text-align: center;
36
+ }
37
+
38
+ #downloadBtn {
39
+ display: block;
40
+ margin: 0 auto 20px;
41
+ }
42
+ </style>
43
+ <body>
44
+ <h2>Bar Chart: Gender → Profession</h2>
45
+ <button id="downloadBtn">Download SVG</button>
46
+ <svg id="chart"></svg>
47
+ <script src="https://d3js.org/d3.v6.min.js"></script>
48
+ <script>
49
+ const colorMap = {
50
+ "Adult Performer": "#8A2BE2",
51
+ "Model": "#DC143C",
52
+ "Actor": "#FF7F50",
53
+ "Performer": "magenta",
54
+ "Singer, Musician": "wheat",
55
+ "TV Personality": "#708090",
56
+ "Sports Professional": "gold",
57
+ "Public Figure": "maroon",
58
+ "Voice Actor": "lightgreen",
59
+ "Online Personality": "#4682B4",
60
+ "Other": "#ccc"
61
+ };
62
+
63
+ const professionOrder = [
64
+ "Adult Performer", "Actor", "Singer, Musician", "Model", "Online Personality", "Public Figure", "Sports Professional", "Voice Actor", "TV Personality", "Other"
65
+ ];
66
+
67
+ const genderOrder = ["Female", "Male", "Other"];
68
+
69
+ d3.json("json/sunburst_gender_profession.json").then(data => {
70
+ // Transform hierarchical data into flat array
71
+ const flatData = [];
72
+
73
+ data.children.forEach(gender => {
74
+ if (gender.children) {
75
+ gender.children.forEach(profession => {
76
+ flatData.push({
77
+ gender: gender.name,
78
+ profession: profession.name,
79
+ value: profession.value
80
+ });
81
+ });
82
+ }
83
+ });
84
+
85
+ // Group by gender and calculate totals
86
+ const genderData = d3.rollup(
87
+ flatData,
88
+ v => ({
89
+ total: d3.sum(v, d => d.value),
90
+ professions: v
91
+ }),
92
+ d => d.gender
93
+ );
94
+
95
+ // Convert to array and sort by gender order
96
+ const genders = Array.from(genderData.keys()).sort((a, b) => {
97
+ const aIndex = genderOrder.indexOf(a);
98
+ const bIndex = genderOrder.indexOf(b);
99
+ return (aIndex === -1 ? Infinity : aIndex) - (bIndex === -1 ? Infinity : bIndex);
100
+ });
101
+
102
+ // Set up dimensions - responsive to window size
103
+ const margin = {top: 40, right: 200, bottom: 150, left: 80};
104
+ const width = Math.max(800, window.innerWidth - 100) - margin.left - margin.right;
105
+ const height = 800 - margin.top - margin.bottom;
106
+
107
+ const svg = d3.select("#chart")
108
+ .attr("width", width + margin.left + margin.right)
109
+ .attr("height", height + margin.top + margin.bottom)
110
+ .append("g")
111
+ .attr("transform", `translate(${margin.left},${margin.top})`);
112
+
113
+ // Create stacked data
114
+ const stack = d3.stack()
115
+ .keys(professionOrder)
116
+ .value((d, key) => {
117
+ const prof = d[1].professions.find(p => p.profession === key);
118
+ return prof ? prof.value : 0;
119
+ });
120
+
121
+ const series = stack(Array.from(genderData));
122
+
123
+ // Scales
124
+ const x = d3.scaleBand()
125
+ .domain(genders)
126
+ .range([0, width])
127
+ .padding(0.3);
128
+
129
+ const y = d3.scaleLinear()
130
+ .domain([0, d3.max(Array.from(genderData.values()), d => d.total)])
131
+ .nice()
132
+ .range([height, 0]);
133
+
134
+ // X axis
135
+ svg.append("g")
136
+ .attr("class", "axis")
137
+ .attr("transform", `translate(0,${height})`)
138
+ .call(d3.axisBottom(x))
139
+ .selectAll("text")
140
+ .attr("transform", "rotate(-90)")
141
+ .style("text-anchor", "end")
142
+ .style("font-size", "11px")
143
+ .style("font-weight", "bold")
144
+ .attr("dx", "-0.5em")
145
+ .attr("dy", "-0.5em");
146
+
147
+ // Y axis
148
+ svg.append("g")
149
+ .attr("class", "axis")
150
+ .call(d3.axisLeft(y));
151
+
152
+ // Y axis label
153
+ svg.append("text")
154
+ .attr("transform", "rotate(-90)")
155
+ .attr("y", 0 - margin.left + 20)
156
+ .attr("x", 0 - (height / 2))
157
+ .attr("dy", "1em")
158
+ .style("text-anchor", "middle")
159
+ .style("font-size", "14px")
160
+ .style("font-weight", "bold")
161
+ .text("Count");
162
+
163
+ // Draw bars
164
+ svg.append("g")
165
+ .selectAll("g")
166
+ .data(series)
167
+ .join("g")
168
+ .attr("fill", d => colorMap[d.key])
169
+ .selectAll("rect")
170
+ .data(d => d)
171
+ .join("rect")
172
+ .attr("class", "bar")
173
+ .attr("x", d => x(d.data[0]))
174
+ .attr("y", d => y(d[1]))
175
+ .attr("height", d => y(d[0]) - y(d[1]))
176
+ .attr("width", x.bandwidth())
177
+ .append("title")
178
+ .text(d => {
179
+ const profession = series.find(s => s.find(item => item === d))?.key;
180
+ return `${d.data[0]} - ${profession}: ${d[1] - d[0]}`;
181
+ });
182
+
183
+ // Count total values per profession
184
+ const professionCounts = {};
185
+ flatData.forEach(d => {
186
+ professionCounts[d.profession] = (professionCounts[d.profession] || 0) + d.value;
187
+ });
188
+
189
+ // Legend
190
+ const legend = svg.append("g")
191
+ .attr("class", "legend")
192
+ .attr("transform", `translate(${width + 20}, 0)`);
193
+
194
+ professionOrder.forEach((prof, i) => {
195
+ const row = legend.append("g")
196
+ .attr("transform", `translate(0,${i * 22})`);
197
+
198
+ row.append("rect")
199
+ .attr("width", 18)
200
+ .attr("height", 18)
201
+ .attr("fill", colorMap[prof]);
202
+
203
+ row.append("text")
204
+ .attr("x", 24)
205
+ .attr("y", 9)
206
+ .attr("dy", "0.35em")
207
+ .style("font-size", "12px")
208
+ .text(`${prof} (${professionCounts[prof] || 0})`);
209
+ });
210
+
211
+ // Download button functionality
212
+ document.getElementById("downloadBtn").addEventListener("click", () => {
213
+ const svgNode = document.querySelector("#chart");
214
+ const clonedSvg = svgNode.cloneNode(true);
215
+
216
+ clonedSvg.setAttribute("xmlns", "http://www.w3.org/2000/svg");
217
+
218
+ // Inline styles
219
+ const allElements = clonedSvg.querySelectorAll("*");
220
+ allElements.forEach(el => {
221
+ const style = window.getComputedStyle(el);
222
+ el.setAttribute("style", `
223
+ font: ${style.font};
224
+ fill: ${style.fill};
225
+ stroke: ${style.stroke};
226
+ stroke-width: ${style.strokeWidth};
227
+ `);
228
+ });
229
+
230
+ const svgData = new XMLSerializer().serializeToString(clonedSvg);
231
+ const svgBlob = new Blob([svgData], { type: "image/svg+xml;charset=utf-8" });
232
+ const url = URL.createObjectURL(svgBlob);
233
+ const a = document.createElement("a");
234
+ a.href = url;
235
+ a.download = "bar_chart_gender.svg";
236
+ document.body.appendChild(a);
237
+ a.click();
238
+ document.body.removeChild(a);
239
+ URL.revokeObjectURL(url);
240
+ });
241
+ });
242
+ </script>
243
+ </body>
244
+ </html>
public/Figure_8b_barchart.html CHANGED
@@ -46,11 +46,14 @@
46
  <svg id="chart"></svg>
47
  <script src="https://d3js.org/d3.v6.min.js"></script>
48
  <script>
 
 
 
49
  const colorMap = {
50
  "Adult Performer": "#8A2BE2",
51
  "Model": "#DC143C",
52
  "Actor": "#FF7F50",
53
- "Performer": "magenta",
54
  "Singer, Musician": "wheat",
55
  "Sports Professional": "gold",
56
  "Voice Actor": "lightgreen",
@@ -58,10 +61,25 @@ const colorMap = {
58
  "Other": "#ccc"
59
  };
60
 
 
 
61
  const professionOrder = [
62
- "Actor", "Adult Performer", "Singer, Musician", "Model", "Online Personality", "Sports Professional", "Voice Actor", "Other"
 
 
 
 
 
 
 
 
63
  ];
64
 
 
 
 
 
 
65
  const manualOrder = [
66
  "United States",
67
  "Japan",
@@ -92,27 +110,29 @@ const manualOrder = [
92
  "Philippines",
93
  "Hong Kong",
94
  "Macau",
95
- "British Virgin Islands",
96
  "Other"
97
  ];
98
 
 
 
 
99
  d3.json("json/sunburst_countries_A.json").then(data => {
100
- // Transform hierarchical data into flat array
101
  const flatData = [];
102
 
103
  data.children.forEach(country => {
104
  if (country.children) {
105
- country.children.forEach(profession => {
106
  flatData.push({
107
  country: country.name,
108
- profession: profession.name,
109
- value: profession.value
110
  });
111
  });
112
  }
113
  });
114
 
115
- // Group by country and calculate totals
116
  const countryData = d3.rollup(
117
  flatData,
118
  v => ({
@@ -122,14 +142,16 @@ d3.json("json/sunburst_countries_A.json").then(data => {
122
  d => d.country
123
  );
124
 
125
- // Convert to array and sort by manual order
126
  const countries = Array.from(countryData.keys()).sort((a, b) => {
127
- const aIndex = manualOrder.indexOf(a);
128
- const bIndex = manualOrder.indexOf(b);
129
- return (aIndex === -1 ? Infinity : aIndex) - (bIndex === -1 ? Infinity : bIndex);
130
  });
131
 
132
- // Set up dimensions - responsive to window size
 
 
133
  const margin = {top: 40, right: 200, bottom: 150, left: 80};
134
  const width = Math.max(800, window.innerWidth - 100) - margin.left - margin.right;
135
  const height = 800 - margin.top - margin.bottom;
@@ -140,7 +162,9 @@ d3.json("json/sunburst_countries_A.json").then(data => {
140
  .append("g")
141
  .attr("transform", `translate(${margin.left},${margin.top})`);
142
 
143
- // Create stacked data
 
 
144
  const stack = d3.stack()
145
  .keys(professionOrder)
146
  .value((d, key) => {
@@ -150,11 +174,13 @@ d3.json("json/sunburst_countries_A.json").then(data => {
150
 
151
  const series = stack(Array.from(countryData));
152
 
153
- // Scales - only show top 25 countries
154
- const top25Countries = countries.slice(0, 25);
 
 
155
 
156
  const x = d3.scaleBand()
157
- .domain(top25Countries)
158
  .range([0, width])
159
  .padding(0.3);
160
 
@@ -163,36 +189,29 @@ d3.json("json/sunburst_countries_A.json").then(data => {
163
  .nice()
164
  .range([height, 0]);
165
 
166
- // X axis
167
  svg.append("g")
168
- .attr("class", "axis")
169
  .attr("transform", `translate(0,${height})`)
170
  .call(d3.axisBottom(x))
171
  .selectAll("text")
172
  .attr("transform", "rotate(-90)")
173
  .style("text-anchor", "end")
174
- .style("font-size", "11px")
175
  .style("font-weight", "bold")
176
  .attr("dx", "-0.5em")
177
  .attr("dy", "-0.5em");
178
 
179
- // Y axis
180
  svg.append("g")
181
- .attr("class", "axis")
182
  .call(d3.axisLeft(y));
183
 
184
- // Y axis label
185
  svg.append("text")
186
  .attr("transform", "rotate(-90)")
187
  .attr("y", 0 - margin.left + 20)
188
- .attr("x", 0 - (height / 2))
189
- .attr("dy", "1em")
190
- .style("text-anchor", "middle")
191
- .style("font-size", "14px")
192
  .style("font-weight", "bold")
193
  .text("Count");
194
 
195
- // Draw bars
 
 
196
  svg.append("g")
197
  .selectAll("g")
198
  .data(series)
@@ -208,24 +227,23 @@ d3.json("json/sunburst_countries_A.json").then(data => {
208
  .attr("width", x.bandwidth())
209
  .append("title")
210
  .text(d => {
211
- const profession = series.find(s => s.find(item => item === d))?.key;
212
- return `${d.data[0]} - ${profession}: ${d[1] - d[0]}`;
213
  });
214
 
215
- // Count total values per profession
 
 
216
  const professionCounts = {};
217
  flatData.forEach(d => {
218
  professionCounts[d.profession] = (professionCounts[d.profession] || 0) + d.value;
219
  });
220
 
221
- // Legend
222
  const legend = svg.append("g")
223
- .attr("class", "legend")
224
  .attr("transform", `translate(${width + 20}, 0)`);
225
 
226
  professionOrder.forEach((prof, i) => {
227
- const row = legend.append("g")
228
- .attr("transform", `translate(0,${i * 22})`);
229
 
230
  row.append("rect")
231
  .attr("width", 18)
@@ -236,41 +254,37 @@ d3.json("json/sunburst_countries_A.json").then(data => {
236
  .attr("x", 24)
237
  .attr("y", 9)
238
  .attr("dy", "0.35em")
239
- .style("font-size", "12px")
240
  .text(`${prof} (${professionCounts[prof] || 0})`);
241
  });
242
 
243
- // Download button functionality
 
 
244
  document.getElementById("downloadBtn").addEventListener("click", () => {
245
  const svgNode = document.querySelector("#chart");
246
- const clonedSvg = svgNode.cloneNode(true);
247
 
248
- clonedSvg.setAttribute("xmlns", "http://www.w3.org/2000/svg");
249
 
250
- // Inline styles
251
- const allElements = clonedSvg.querySelectorAll("*");
252
- allElements.forEach(el => {
253
  const style = window.getComputedStyle(el);
254
- el.setAttribute("style", `
255
- font: ${style.font};
256
- fill: ${style.fill};
257
- stroke: ${style.stroke};
258
- stroke-width: ${style.strokeWidth};
259
- `);
260
  });
261
 
262
- const svgData = new XMLSerializer().serializeToString(clonedSvg);
263
- const svgBlob = new Blob([svgData], { type: "image/svg+xml;charset=utf-8" });
264
- const url = URL.createObjectURL(svgBlob);
265
  const a = document.createElement("a");
 
266
  a.href = url;
267
  a.download = "bar_chart.svg";
268
- document.body.appendChild(a);
269
  a.click();
270
- document.body.removeChild(a);
271
  URL.revokeObjectURL(url);
272
  });
 
273
  });
274
  </script>
 
275
  </body>
276
  </html>
 
46
  <svg id="chart"></svg>
47
  <script src="https://d3js.org/d3.v6.min.js"></script>
48
  <script>
49
+ /* --------------------------------------------------------
50
+ 1. PROFESSION COLORS + ORDER
51
+ -------------------------------------------------------- */
52
  const colorMap = {
53
  "Adult Performer": "#8A2BE2",
54
  "Model": "#DC143C",
55
  "Actor": "#FF7F50",
56
+ "Public Figure": "#20B2AA", // ← add this (or pick your color)
57
  "Singer, Musician": "wheat",
58
  "Sports Professional": "gold",
59
  "Voice Actor": "lightgreen",
 
61
  "Other": "#ccc"
62
  };
63
 
64
+ // IMPORTANT → this now matches exactly the categories
65
+ // that your updated Python script produces.
66
  const professionOrder = [
67
+ "Actor",
68
+ "Adult Performer",
69
+ "Singer, Musician",
70
+ "Model",
71
+ "Online Personality",
72
+ "Sports Professional",
73
+ "Voice Actor",
74
+ "Public Figure",
75
+ "Other"
76
  ];
77
 
78
+ //#DC143C
79
+
80
+ /* --------------------------------------------------------
81
+ 2. COUNTRY ORDER (unchanged)
82
+ -------------------------------------------------------- */
83
  const manualOrder = [
84
  "United States",
85
  "Japan",
 
110
  "Philippines",
111
  "Hong Kong",
112
  "Macau",
 
113
  "Other"
114
  ];
115
 
116
+ /* --------------------------------------------------------
117
+ 3. LOAD JSON + TRANSFORM IT
118
+ -------------------------------------------------------- */
119
  d3.json("json/sunburst_countries_A.json").then(data => {
120
+
121
  const flatData = [];
122
 
123
  data.children.forEach(country => {
124
  if (country.children) {
125
+ country.children.forEach(prof => {
126
  flatData.push({
127
  country: country.name,
128
+ profession: prof.name,
129
+ value: prof.value
130
  });
131
  });
132
  }
133
  });
134
 
135
+ // Group per country
136
  const countryData = d3.rollup(
137
  flatData,
138
  v => ({
 
142
  d => d.country
143
  );
144
 
145
+ // Sort according to manual ordering
146
  const countries = Array.from(countryData.keys()).sort((a, b) => {
147
+ const ai = manualOrder.indexOf(a);
148
+ const bi = manualOrder.indexOf(b);
149
+ return (ai === -1 ? Infinity : ai) - (bi === -1 ? Infinity : bi);
150
  });
151
 
152
+ /* --------------------------------------------------------
153
+ 4. SVG SETUP
154
+ -------------------------------------------------------- */
155
  const margin = {top: 40, right: 200, bottom: 150, left: 80};
156
  const width = Math.max(800, window.innerWidth - 100) - margin.left - margin.right;
157
  const height = 800 - margin.top - margin.bottom;
 
162
  .append("g")
163
  .attr("transform", `translate(${margin.left},${margin.top})`);
164
 
165
+ /* --------------------------------------------------------
166
+ 5. STACKED DATA
167
+ -------------------------------------------------------- */
168
  const stack = d3.stack()
169
  .keys(professionOrder)
170
  .value((d, key) => {
 
174
 
175
  const series = stack(Array.from(countryData));
176
 
177
+ /* --------------------------------------------------------
178
+ 6. AXES
179
+ -------------------------------------------------------- */
180
+ const topCountries = countries.slice(0, 25);
181
 
182
  const x = d3.scaleBand()
183
+ .domain(topCountries)
184
  .range([0, width])
185
  .padding(0.3);
186
 
 
189
  .nice()
190
  .range([height, 0]);
191
 
 
192
  svg.append("g")
 
193
  .attr("transform", `translate(0,${height})`)
194
  .call(d3.axisBottom(x))
195
  .selectAll("text")
196
  .attr("transform", "rotate(-90)")
197
  .style("text-anchor", "end")
 
198
  .style("font-weight", "bold")
199
  .attr("dx", "-0.5em")
200
  .attr("dy", "-0.5em");
201
 
 
202
  svg.append("g")
 
203
  .call(d3.axisLeft(y));
204
 
 
205
  svg.append("text")
206
  .attr("transform", "rotate(-90)")
207
  .attr("y", 0 - margin.left + 20)
208
+ .attr("x", 0 - height / 2)
 
 
 
209
  .style("font-weight", "bold")
210
  .text("Count");
211
 
212
+ /* --------------------------------------------------------
213
+ 7. DRAW BARS
214
+ -------------------------------------------------------- */
215
  svg.append("g")
216
  .selectAll("g")
217
  .data(series)
 
227
  .attr("width", x.bandwidth())
228
  .append("title")
229
  .text(d => {
230
+ const profKey = series.find(s => s.includes(d))?.key;
231
+ return `${d.data[0]} – ${profKey}: ${d[1] - d[0]}`;
232
  });
233
 
234
+ /* --------------------------------------------------------
235
+ 8. LEGEND
236
+ -------------------------------------------------------- */
237
  const professionCounts = {};
238
  flatData.forEach(d => {
239
  professionCounts[d.profession] = (professionCounts[d.profession] || 0) + d.value;
240
  });
241
 
 
242
  const legend = svg.append("g")
 
243
  .attr("transform", `translate(${width + 20}, 0)`);
244
 
245
  professionOrder.forEach((prof, i) => {
246
+ const row = legend.append("g").attr("transform", `translate(0,${i * 22})`);
 
247
 
248
  row.append("rect")
249
  .attr("width", 18)
 
254
  .attr("x", 24)
255
  .attr("y", 9)
256
  .attr("dy", "0.35em")
 
257
  .text(`${prof} (${professionCounts[prof] || 0})`);
258
  });
259
 
260
+ /* --------------------------------------------------------
261
+ 9. DOWNLOAD SVG BUTTON
262
+ -------------------------------------------------------- */
263
  document.getElementById("downloadBtn").addEventListener("click", () => {
264
  const svgNode = document.querySelector("#chart");
265
+ const clone = svgNode.cloneNode(true);
266
 
267
+ clone.setAttribute("xmlns", "http://www.w3.org/2000/svg");
268
 
269
+ const all = clone.querySelectorAll("*");
270
+ all.forEach(el => {
 
271
  const style = window.getComputedStyle(el);
272
+ el.setAttribute("style", `font:${style.font}; fill:${style.fill}; stroke:${style.stroke};`);
 
 
 
 
 
273
  });
274
 
275
+ const svgData = new XMLSerializer().serializeToString(clone);
276
+ const blob = new Blob([svgData], { type: "image/svg+xml;charset=utf-8" });
277
+ const url = URL.createObjectURL(blob);
278
  const a = document.createElement("a");
279
+
280
  a.href = url;
281
  a.download = "bar_chart.svg";
 
282
  a.click();
 
283
  URL.revokeObjectURL(url);
284
  });
285
+
286
  });
287
  </script>
288
+
289
  </body>
290
  </html>
public/Figure_8b_sunburst.html CHANGED
@@ -37,26 +37,23 @@ const width = 800;
37
  const radius = width / 3;
38
 
39
  const colorMap = {
40
- "Adult Performer": "#8A2BE2",
41
- "Model": "#DC143C",
42
- "Actor": "#FF7F50",
43
- "Performer": "magenta",
44
- "Singer, Musician": "wheat",
45
- //"Fictional Character": "#708090",
46
- "Sports Professional": "gold",
47
- "Voice Actor": "lightgreen",
48
- "Online Personality": "#4682B4",
49
  "Other": "#ccc"
50
  };
51
 
52
  const professionOrder = [
53
- "Actor", "Adult Performer", "Singer, Musician", "Model", "Online Personality", "Sports Professional", "Voice Actor", "Other"
54
  ];
55
 
56
  const manualOrder = [
57
- "United States",
58
  "Japan",
59
- "United Kingdom",
60
  "South Korea",
61
  "Russia",
62
  "India",
@@ -69,21 +66,6 @@ const manualOrder = [
69
  "Australia",
70
  "Spain",
71
  "Ukraine",
72
- "Turkey",
73
- "Netherlands",
74
- "Indonesia",
75
- "Poland",
76
- "Czech Republic",
77
- "Argentina",
78
- "Sweden",
79
- "Mexico",
80
- "Taiwan",
81
- "Thailand",
82
- "Ireland",
83
- "Philippines",
84
- "Hong Kong",
85
- "Macau",
86
- "British Virgin Islands",
87
  "Other"
88
  ];
89
 
 
37
  const radius = width / 3;
38
 
39
  const colorMap = {
40
+ "actor": "#FF7F50",
41
+ "adult performer": "#8A2BE2",
42
+ "singer/musician": "#FFD700",
43
+ "model": "#DC143C",
44
+ "online personality": "#4682B4",
45
+ "public figure": "#9370DB",
 
 
 
46
  "Other": "#ccc"
47
  };
48
 
49
  const professionOrder = [
50
+ "actor", "adult performer", "singer/musician", "model", "online personality", "public figure", "Other"
51
  ];
52
 
53
  const manualOrder = [
54
+ "USA",
55
  "Japan",
56
+ "UK",
57
  "South Korea",
58
  "Russia",
59
  "India",
 
66
  "Australia",
67
  "Spain",
68
  "Ukraine",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
69
  "Other"
70
  ];
71
 
public/json/8a.json CHANGED
@@ -5,28 +5,44 @@
5
  "name": "Female",
6
  "children": [
7
  {
8
- "name": "Actor",
9
- "value": 8
10
  },
11
  {
12
- "name": "Adult Performer",
13
- "value": 2
14
  },
15
  {
16
- "name": "Online Personality",
17
- "value": 4
18
  },
19
  {
20
- "name": "Singer, Musician",
21
- "value": 2
22
  },
23
  {
24
- "name": "Sports Professional",
25
- "value": 1
26
  },
27
  {
28
- "name": "Voice Actor",
29
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
30
  }
31
  ]
32
  },
@@ -34,8 +50,44 @@
34
  "name": "Male",
35
  "children": [
36
  {
37
- "name": "Voice Actor",
38
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
39
  }
40
  ]
41
  },
@@ -43,8 +95,8 @@
43
  "name": "Other",
44
  "children": [
45
  {
46
- "name": "Other",
47
- "value": 2
48
  }
49
  ]
50
  }
 
5
  "name": "Female",
6
  "children": [
7
  {
8
+ "name": "actor",
9
+ "value": 9397
10
  },
11
  {
12
+ "name": "adult performer",
13
+ "value": 718
14
  },
15
  {
16
+ "name": "model",
17
+ "value": 2715
18
  },
19
  {
20
+ "name": "online personality",
21
+ "value": 1348
22
  },
23
  {
24
+ "name": "public figure",
25
+ "value": 406
26
  },
27
  {
28
+ "name": "singer/musician",
29
+ "value": 3324
30
+ },
31
+ {
32
+ "name": "sports professional",
33
+ "value": 166
34
+ },
35
+ {
36
+ "name": "tv personality",
37
+ "value": 247
38
+ },
39
+ {
40
+ "name": "voice actor/ASMR",
41
+ "value": 343
42
+ },
43
+ {
44
+ "name": "Other",
45
+ "value": 15
46
  }
47
  ]
48
  },
 
50
  "name": "Male",
51
  "children": [
52
  {
53
+ "name": "actor",
54
+ "value": 1373
55
+ },
56
+ {
57
+ "name": "adult performer",
58
+ "value": 12
59
+ },
60
+ {
61
+ "name": "model",
62
+ "value": 43
63
+ },
64
+ {
65
+ "name": "online personality",
66
+ "value": 144
67
+ },
68
+ {
69
+ "name": "public figure",
70
+ "value": 522
71
+ },
72
+ {
73
+ "name": "singer/musician",
74
+ "value": 341
75
+ },
76
+ {
77
+ "name": "sports professional",
78
+ "value": 221
79
+ },
80
+ {
81
+ "name": "tv personality",
82
+ "value": 45
83
+ },
84
+ {
85
+ "name": "voice actor/ASMR",
86
+ "value": 4
87
+ },
88
+ {
89
+ "name": "Other",
90
+ "value": 15
91
  }
92
  ]
93
  },
 
95
  "name": "Other",
96
  "children": [
97
  {
98
+ "name": "actor",
99
+ "value": 1
100
  }
101
  ]
102
  }
public/json/sunburst_countries_A.json CHANGED
@@ -2,126 +2,856 @@
2
  "name": "root",
3
  "children": [
4
  {
5
- "name": "Australia",
6
  "children": [
7
  {
8
  "name": "Actor",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9
  "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
10
  }
11
  ]
12
  },
13
  {
14
- "name": "China",
15
  "children": [
16
  {
17
  "name": "Actor",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
18
  "value": 1
 
 
 
 
19
  }
20
  ]
21
  },
22
  {
23
- "name": "Colombia",
24
  "children": [
25
  {
26
  "name": "Actor",
27
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
28
  }
29
  ]
30
  },
31
  {
32
- "name": "Czechia",
33
  "children": [
34
  {
35
- "name": "Voice Actor",
36
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
37
  }
38
  ]
39
  },
40
  {
41
- "name": "Denmark",
42
  "children": [
43
  {
44
  "name": "Actor",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
45
  "value": 1
46
  }
47
  ]
48
  },
49
  {
50
- "name": "Germany",
51
  "children": [
 
 
 
 
 
 
 
 
 
 
 
 
52
  {
53
  "name": "Online Personality",
54
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
55
  }
56
  ]
57
  },
58
  {
59
- "name": "India",
60
  "children": [
61
  {
62
  "name": "Actor",
63
- "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
64
  }
65
  ]
66
  },
67
  {
68
- "name": "Italy",
69
  "children": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
70
  {
71
  "name": "Other",
72
- "value": 1
73
  }
74
  ]
75
  },
76
  {
77
- "name": "Japan",
78
  "children": [
79
  {
80
- "name": "Voice Actor",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
81
  "value": 1
82
  }
83
  ]
84
  },
85
  {
86
- "name": "Other",
87
  "children": [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
88
  {
89
  "name": "Singer, Musician",
90
- "value": 1
91
  },
92
  {
93
  "name": "Other",
94
- "value": 2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
95
  }
96
  ]
97
  },
98
  {
99
- "name": "Poland",
100
  "children": [
101
  {
102
  "name": "Actor",
 
 
 
 
103
  "value": 1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
104
  }
105
  ]
106
  },
107
  {
108
- "name": "US",
109
  "children": [
110
  {
111
  "name": "Actor",
112
- "value": 2
113
  },
114
  {
115
  "name": "Adult Performer",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
116
  "value": 2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
117
  },
118
  {
119
  "name": "Online Personality",
 
 
 
 
 
 
 
 
 
 
 
 
120
  "value": 3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
121
  },
122
  {
123
  "name": "Singer, Musician",
124
- "value": 1
 
 
 
 
125
  }
126
  ]
127
  }
 
2
  "name": "root",
3
  "children": [
4
  {
5
+ "name": "usa",
6
  "children": [
7
  {
8
  "name": "Actor",
9
+ "value": 3912
10
+ },
11
+ {
12
+ "name": "Adult Performer",
13
+ "value": 262
14
+ },
15
+ {
16
+ "name": "Model",
17
+ "value": 356
18
+ },
19
+ {
20
+ "name": "Online Personality",
21
+ "value": 255
22
+ },
23
+ {
24
+ "name": "Public Figure",
25
+ "value": 225
26
+ },
27
+ {
28
+ "name": "Singer, Musician",
29
+ "value": 733
30
+ },
31
+ {
32
+ "name": "Voice Actor",
33
+ "value": 5
34
+ },
35
+ {
36
+ "name": "Other",
37
+ "value": 231
38
+ }
39
+ ]
40
+ },
41
+ {
42
+ "name": "japan",
43
+ "children": [
44
+ {
45
+ "name": "Actor",
46
+ "value": 357
47
+ },
48
+ {
49
+ "name": "Adult Performer",
50
+ "value": 392
51
+ },
52
+ {
53
+ "name": "Model",
54
+ "value": 616
55
+ },
56
+ {
57
+ "name": "Online Personality",
58
+ "value": 136
59
+ },
60
+ {
61
+ "name": "Public Figure",
62
+ "value": 30
63
+ },
64
+ {
65
+ "name": "Singer, Musician",
66
+ "value": 503
67
+ },
68
+ {
69
+ "name": "Voice Actor",
70
+ "value": 360
71
+ },
72
+ {
73
+ "name": "Other",
74
+ "value": 31
75
+ }
76
+ ]
77
+ },
78
+ {
79
+ "name": "south korea",
80
+ "children": [
81
+ {
82
+ "name": "Actor",
83
+ "value": 182
84
+ },
85
+ {
86
+ "name": "Model",
87
+ "value": 119
88
+ },
89
+ {
90
+ "name": "Online Personality",
91
+ "value": 45
92
+ },
93
+ {
94
+ "name": "Public Figure",
95
+ "value": 2
96
+ },
97
+ {
98
+ "name": "Singer, Musician",
99
+ "value": 1093
100
+ },
101
+ {
102
+ "name": "Other",
103
+ "value": 9
104
+ }
105
+ ]
106
+ },
107
+ {
108
+ "name": "uk",
109
+ "children": [
110
+ {
111
+ "name": "Actor",
112
+ "value": 1012
113
+ },
114
+ {
115
+ "name": "Adult Performer",
116
+ "value": 12
117
+ },
118
+ {
119
+ "name": "Model",
120
+ "value": 111
121
+ },
122
+ {
123
+ "name": "Online Personality",
124
+ "value": 26
125
+ },
126
+ {
127
+ "name": "Public Figure",
128
+ "value": 46
129
+ },
130
+ {
131
+ "name": "Singer, Musician",
132
+ "value": 152
133
+ },
134
+ {
135
+ "name": "Voice Actor",
136
+ "value": 2
137
+ },
138
+ {
139
+ "name": "Other",
140
+ "value": 67
141
+ }
142
+ ]
143
+ },
144
+ {
145
+ "name": "china",
146
+ "children": [
147
+ {
148
+ "name": "Actor",
149
+ "value": 229
150
+ },
151
+ {
152
+ "name": "Adult Performer",
153
+ "value": 34
154
+ },
155
+ {
156
+ "name": "Model",
157
+ "value": 257
158
+ },
159
+ {
160
+ "name": "Online Personality",
161
+ "value": 220
162
+ },
163
+ {
164
+ "name": "Public Figure",
165
+ "value": 13
166
+ },
167
+ {
168
+ "name": "Singer, Musician",
169
+ "value": 63
170
+ },
171
+ {
172
+ "name": "Other",
173
+ "value": 6
174
+ }
175
+ ]
176
+ },
177
+ {
178
+ "name": "india",
179
+ "children": [
180
+ {
181
+ "name": "Actor",
182
+ "value": 691
183
+ },
184
+ {
185
+ "name": "Adult Performer",
186
+ "value": 4
187
+ },
188
+ {
189
+ "name": "Model",
190
+ "value": 30
191
+ },
192
+ {
193
+ "name": "Online Personality",
194
+ "value": 14
195
+ },
196
+ {
197
+ "name": "Public Figure",
198
+ "value": 7
199
+ },
200
+ {
201
+ "name": "Singer, Musician",
202
+ "value": 6
203
+ },
204
+ {
205
+ "name": "Other",
206
+ "value": 6
207
+ }
208
+ ]
209
+ },
210
+ {
211
+ "name": "france",
212
+ "children": [
213
+ {
214
+ "name": "Actor",
215
+ "value": 204
216
+ },
217
+ {
218
+ "name": "Adult Performer",
219
+ "value": 27
220
+ },
221
+ {
222
+ "name": "Model",
223
+ "value": 40
224
+ },
225
+ {
226
+ "name": "Online Personality",
227
+ "value": 23
228
+ },
229
+ {
230
+ "name": "Public Figure",
231
+ "value": 60
232
+ },
233
+ {
234
+ "name": "Singer, Musician",
235
+ "value": 38
236
+ },
237
+ {
238
+ "name": "Other",
239
+ "value": 28
240
+ }
241
+ ]
242
+ },
243
+ {
244
+ "name": "russia",
245
+ "children": [
246
+ {
247
+ "name": "Actor",
248
+ "value": 36
249
+ },
250
+ {
251
+ "name": "Adult Performer",
252
+ "value": 57
253
+ },
254
+ {
255
+ "name": "Model",
256
+ "value": 118
257
+ },
258
+ {
259
+ "name": "Online Personality",
260
+ "value": 89
261
+ },
262
+ {
263
+ "name": "Public Figure",
264
+ "value": 46
265
+ },
266
+ {
267
+ "name": "Singer, Musician",
268
+ "value": 24
269
+ },
270
+ {
271
+ "name": "Other",
272
+ "value": 17
273
+ }
274
+ ]
275
+ },
276
+ {
277
+ "name": "canada",
278
+ "children": [
279
+ {
280
+ "name": "Actor",
281
+ "value": 246
282
+ },
283
+ {
284
+ "name": "Adult Performer",
285
+ "value": 4
286
+ },
287
+ {
288
+ "name": "Model",
289
+ "value": 19
290
+ },
291
+ {
292
+ "name": "Online Personality",
293
+ "value": 11
294
+ },
295
+ {
296
+ "name": "Public Figure",
297
+ "value": 18
298
+ },
299
+ {
300
+ "name": "Singer, Musician",
301
+ "value": 66
302
+ },
303
+ {
304
+ "name": "Voice Actor",
305
  "value": 1
306
+ },
307
+ {
308
+ "name": "Other",
309
+ "value": 11
310
+ }
311
+ ]
312
+ },
313
+ {
314
+ "name": "brazil",
315
+ "children": [
316
+ {
317
+ "name": "Actor",
318
+ "value": 90
319
+ },
320
+ {
321
+ "name": "Adult Performer",
322
+ "value": 21
323
+ },
324
+ {
325
+ "name": "Model",
326
+ "value": 95
327
+ },
328
+ {
329
+ "name": "Online Personality",
330
+ "value": 34
331
+ },
332
+ {
333
+ "name": "Public Figure",
334
+ "value": 15
335
+ },
336
+ {
337
+ "name": "Singer, Musician",
338
+ "value": 39
339
+ },
340
+ {
341
+ "name": "Other",
342
+ "value": 27
343
+ }
344
+ ]
345
+ },
346
+ {
347
+ "name": "australia",
348
+ "children": [
349
+ {
350
+ "name": "Actor",
351
+ "value": 172
352
+ },
353
+ {
354
+ "name": "Adult Performer",
355
+ "value": 6
356
+ },
357
+ {
358
+ "name": "Model",
359
+ "value": 37
360
+ },
361
+ {
362
+ "name": "Online Personality",
363
+ "value": 16
364
+ },
365
+ {
366
+ "name": "Public Figure",
367
+ "value": 3
368
+ },
369
+ {
370
+ "name": "Singer, Musician",
371
+ "value": 25
372
+ },
373
+ {
374
+ "name": "Other",
375
+ "value": 7
376
+ }
377
+ ]
378
+ },
379
+ {
380
+ "name": "germany",
381
+ "children": [
382
+ {
383
+ "name": "Actor",
384
+ "value": 51
385
+ },
386
+ {
387
+ "name": "Adult Performer",
388
+ "value": 7
389
+ },
390
+ {
391
+ "name": "Model",
392
+ "value": 52
393
+ },
394
+ {
395
+ "name": "Online Personality",
396
+ "value": 22
397
+ },
398
+ {
399
+ "name": "Public Figure",
400
+ "value": 47
401
+ },
402
+ {
403
+ "name": "Singer, Musician",
404
+ "value": 27
405
+ },
406
+ {
407
+ "name": "Other",
408
+ "value": 22
409
  }
410
  ]
411
  },
412
  {
413
+ "name": "italy",
414
  "children": [
415
  {
416
  "name": "Actor",
417
+ "value": 74
418
+ },
419
+ {
420
+ "name": "Adult Performer",
421
+ "value": 7
422
+ },
423
+ {
424
+ "name": "Model",
425
+ "value": 37
426
+ },
427
+ {
428
+ "name": "Online Personality",
429
+ "value": 29
430
+ },
431
+ {
432
+ "name": "Public Figure",
433
+ "value": 20
434
+ },
435
+ {
436
+ "name": "Singer, Musician",
437
+ "value": 28
438
+ },
439
+ {
440
+ "name": "Voice Actor",
441
  "value": 1
442
+ },
443
+ {
444
+ "name": "Other",
445
+ "value": 26
446
  }
447
  ]
448
  },
449
  {
450
+ "name": "ukraine",
451
  "children": [
452
  {
453
  "name": "Actor",
454
+ "value": 11
455
+ },
456
+ {
457
+ "name": "Adult Performer",
458
+ "value": 49
459
+ },
460
+ {
461
+ "name": "Model",
462
+ "value": 44
463
+ },
464
+ {
465
+ "name": "Online Personality",
466
+ "value": 48
467
+ },
468
+ {
469
+ "name": "Public Figure",
470
+ "value": 13
471
+ },
472
+ {
473
+ "name": "Singer, Musician",
474
+ "value": 6
475
+ },
476
+ {
477
+ "name": "Other",
478
+ "value": 4
479
  }
480
  ]
481
  },
482
  {
483
+ "name": "türkiye",
484
  "children": [
485
  {
486
+ "name": "Actor",
487
+ "value": 82
488
+ },
489
+ {
490
+ "name": "Model",
491
+ "value": 32
492
+ },
493
+ {
494
+ "name": "Online Personality",
495
+ "value": 14
496
+ },
497
+ {
498
+ "name": "Public Figure",
499
+ "value": 15
500
+ },
501
+ {
502
+ "name": "Singer, Musician",
503
+ "value": 10
504
+ },
505
+ {
506
+ "name": "Other",
507
+ "value": 9
508
  }
509
  ]
510
  },
511
  {
512
+ "name": "thailand",
513
  "children": [
514
  {
515
  "name": "Actor",
516
+ "value": 25
517
+ },
518
+ {
519
+ "name": "Adult Performer",
520
+ "value": 12
521
+ },
522
+ {
523
+ "name": "Model",
524
+ "value": 19
525
+ },
526
+ {
527
+ "name": "Online Personality",
528
+ "value": 43
529
+ },
530
+ {
531
+ "name": "Singer, Musician",
532
+ "value": 41
533
+ },
534
+ {
535
+ "name": "Other",
536
  "value": 1
537
  }
538
  ]
539
  },
540
  {
541
+ "name": "spain",
542
  "children": [
543
+ {
544
+ "name": "Actor",
545
+ "value": 60
546
+ },
547
+ {
548
+ "name": "Adult Performer",
549
+ "value": 6
550
+ },
551
+ {
552
+ "name": "Model",
553
+ "value": 11
554
+ },
555
  {
556
  "name": "Online Personality",
557
+ "value": 17
558
+ },
559
+ {
560
+ "name": "Public Figure",
561
+ "value": 11
562
+ },
563
+ {
564
+ "name": "Singer, Musician",
565
+ "value": 12
566
+ },
567
+ {
568
+ "name": "Other",
569
+ "value": 22
570
  }
571
  ]
572
  },
573
  {
574
+ "name": "netherlands",
575
  "children": [
576
  {
577
  "name": "Actor",
578
+ "value": 27
579
+ },
580
+ {
581
+ "name": "Adult Performer",
582
+ "value": 2
583
+ },
584
+ {
585
+ "name": "Model",
586
+ "value": 24
587
+ },
588
+ {
589
+ "name": "Online Personality",
590
+ "value": 17
591
+ },
592
+ {
593
+ "name": "Public Figure",
594
+ "value": 34
595
+ },
596
+ {
597
+ "name": "Singer, Musician",
598
+ "value": 10
599
+ },
600
+ {
601
+ "name": "Other",
602
+ "value": 13
603
  }
604
  ]
605
  },
606
  {
607
+ "name": "ireland",
608
  "children": [
609
+ {
610
+ "name": "Actor",
611
+ "value": 61
612
+ },
613
+ {
614
+ "name": "Online Personality",
615
+ "value": 2
616
+ },
617
+ {
618
+ "name": "Public Figure",
619
+ "value": 28
620
+ },
621
+ {
622
+ "name": "Singer, Musician",
623
+ "value": 10
624
+ },
625
  {
626
  "name": "Other",
627
+ "value": 10
628
  }
629
  ]
630
  },
631
  {
632
+ "name": "indonesia",
633
  "children": [
634
  {
635
+ "name": "Actor",
636
+ "value": 14
637
+ },
638
+ {
639
+ "name": "Adult Performer",
640
+ "value": 2
641
+ },
642
+ {
643
+ "name": "Model",
644
+ "value": 26
645
+ },
646
+ {
647
+ "name": "Online Personality",
648
+ "value": 30
649
+ },
650
+ {
651
+ "name": "Public Figure",
652
+ "value": 4
653
+ },
654
+ {
655
+ "name": "Singer, Musician",
656
+ "value": 28
657
+ },
658
+ {
659
+ "name": "Other",
660
  "value": 1
661
  }
662
  ]
663
  },
664
  {
665
+ "name": "sweden",
666
  "children": [
667
+ {
668
+ "name": "Actor",
669
+ "value": 28
670
+ },
671
+ {
672
+ "name": "Model",
673
+ "value": 15
674
+ },
675
+ {
676
+ "name": "Online Personality",
677
+ "value": 9
678
+ },
679
+ {
680
+ "name": "Public Figure",
681
+ "value": 21
682
+ },
683
  {
684
  "name": "Singer, Musician",
685
+ "value": 23
686
  },
687
  {
688
  "name": "Other",
689
+ "value": 9
690
+ }
691
+ ]
692
+ },
693
+ {
694
+ "name": "poland",
695
+ "children": [
696
+ {
697
+ "name": "Actor",
698
+ "value": 25
699
+ },
700
+ {
701
+ "name": "Adult Performer",
702
+ "value": 6
703
+ },
704
+ {
705
+ "name": "Model",
706
+ "value": 30
707
+ },
708
+ {
709
+ "name": "Online Personality",
710
+ "value": 14
711
+ },
712
+ {
713
+ "name": "Public Figure",
714
+ "value": 9
715
+ },
716
+ {
717
+ "name": "Singer, Musician",
718
+ "value": 8
719
+ },
720
+ {
721
+ "name": "Other",
722
+ "value": 10
723
  }
724
  ]
725
  },
726
  {
727
+ "name": "argentina",
728
  "children": [
729
  {
730
  "name": "Actor",
731
+ "value": 12
732
+ },
733
+ {
734
+ "name": "Adult Performer",
735
  "value": 1
736
+ },
737
+ {
738
+ "name": "Model",
739
+ "value": 5
740
+ },
741
+ {
742
+ "name": "Online Personality",
743
+ "value": 3
744
+ },
745
+ {
746
+ "name": "Public Figure",
747
+ "value": 23
748
+ },
749
+ {
750
+ "name": "Singer, Musician",
751
+ "value": 30
752
+ },
753
+ {
754
+ "name": "Other",
755
+ "value": 15
756
  }
757
  ]
758
  },
759
  {
760
+ "name": "taiwan",
761
  "children": [
762
  {
763
  "name": "Actor",
764
+ "value": 14
765
  },
766
  {
767
  "name": "Adult Performer",
768
+ "value": 3
769
+ },
770
+ {
771
+ "name": "Model",
772
+ "value": 18
773
+ },
774
+ {
775
+ "name": "Online Personality",
776
+ "value": 31
777
+ },
778
+ {
779
+ "name": "Public Figure",
780
+ "value": 2
781
+ },
782
+ {
783
+ "name": "Singer, Musician",
784
+ "value": 15
785
+ },
786
+ {
787
+ "name": "Other",
788
  "value": 2
789
+ }
790
+ ]
791
+ },
792
+ {
793
+ "name": "mexico",
794
+ "children": [
795
+ {
796
+ "name": "Actor",
797
+ "value": 57
798
+ },
799
+ {
800
+ "name": "Adult Performer",
801
+ "value": 1
802
+ },
803
+ {
804
+ "name": "Model",
805
+ "value": 9
806
  },
807
  {
808
  "name": "Online Personality",
809
+ "value": 2
810
+ },
811
+ {
812
+ "name": "Public Figure",
813
+ "value": 2
814
+ },
815
+ {
816
+ "name": "Singer, Musician",
817
+ "value": 7
818
+ },
819
+ {
820
+ "name": "Other",
821
  "value": 3
822
+ }
823
+ ]
824
+ },
825
+ {
826
+ "name": "Other",
827
+ "children": [
828
+ {
829
+ "name": "Actor",
830
+ "value": 255
831
+ },
832
+ {
833
+ "name": "Adult Performer",
834
+ "value": 39
835
+ },
836
+ {
837
+ "name": "Model",
838
+ "value": 204
839
+ },
840
+ {
841
+ "name": "Online Personality",
842
+ "value": 95
843
+ },
844
+ {
845
+ "name": "Public Figure",
846
+ "value": 87
847
  },
848
  {
849
  "name": "Singer, Musician",
850
+ "value": 110
851
+ },
852
+ {
853
+ "name": "Other",
854
+ "value": 62
855
  }
856
  ]
857
  }