laura.wagner commited on
Commit
d162b65
Β·
1 Parent(s): 10d474d

added bloomz query notebook

Browse files
jupyter_notebooks/.ipynb_checkpoints/Section_2-3-4_Bloomz_query-checkpoint.ipynb ADDED
@@ -0,0 +1,370 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": null,
6
+ "id": "3b87c378-241e-41ab-be6e-84222594f22f",
7
+ "metadata": {},
8
+ "outputs": [],
9
+ "source": [
10
+ "import pandas as pd\n",
11
+ "import json\n",
12
+ "import time\n",
13
+ "import re\n",
14
+ "from pathlib import Path\n",
15
+ "from tqdm import tqdm\n",
16
+ "import torch\n",
17
+ "from transformers import AutoModelForCausalLM, AutoTokenizer\n",
18
+ "\n",
19
+ "# Import is used for pd.notna() and pd.isna() checks\n",
20
+ "\n",
21
+ "current_dir = Path.cwd()\n",
22
+ "input_file = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter_step_02_NER.csv\"\n",
23
+ "\n",
24
+ "# === CONFIGURATION ===\n",
25
+ "TEST_MODE = True\n",
26
+ "TEST_SIZE = 10\n",
27
+ "MAX_ROWS = 20000\n",
28
+ "SAVE_INTERVAL = 10\n",
29
+ "\n",
30
+ "# Model settings - BLOOMZ (BigScience - European consortium)\n",
31
+ "MODEL_NAME = \"bigscience/bloomz-7b1\" # Largest instruction-tuned BLOOM model\n",
32
+ "CACHE_DIR = current_dir.parent / \"data/models\"\n",
33
+ "CACHE_DIR.mkdir(parents=True, exist_ok=True)\n",
34
+ "\n",
35
+ "PROFESSION_CATEGORIES = [\n",
36
+ " \"actor\", \"adult performer\", \"singer/musician\", \"model\",\n",
37
+ " \"online personality\", \"public figure\", \"voice actor/ASMR\",\n",
38
+ " \"sports professional\", \"tv personality\"\n",
39
+ "]\n",
40
+ "\n",
41
+ "# === LOAD MODEL ===\n",
42
+ "print(f\"Loading model: {MODEL_NAME}\")\n",
43
+ "print(f\"Cache directory: {CACHE_DIR}\")\n",
44
+ "print(f\"This may take a while on first run (~14GB download)...\\n\")\n",
45
+ "\n",
46
+ "# Check GPU availability\n",
47
+ "device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n",
48
+ "print(f\"Device: {device}\")\n",
49
+ "\n",
50
+ "if device == \"cpu\":\n",
51
+ " print(\"⚠️ WARNING: No GPU detected! Inference will be VERY slow.\")\n",
52
+ " print(\" Consider using a GPU or reducing model size.\")\n",
53
+ "\n",
54
+ "# Load tokenizer\n",
55
+ "print(\"Loading tokenizer...\")\n",
56
+ "try:\n",
57
+ " tokenizer = AutoTokenizer.from_pretrained(\n",
58
+ " MODEL_NAME,\n",
59
+ " cache_dir=str(CACHE_DIR)\n",
60
+ " )\n",
61
+ " print(\"βœ… Tokenizer loaded\")\n",
62
+ "except Exception as e:\n",
63
+ " print(f\"❌ Error loading tokenizer: {e}\")\n",
64
+ " raise\n",
65
+ "\n",
66
+ "# Ensure pad token is set\n",
67
+ "if tokenizer.pad_token is None:\n",
68
+ " tokenizer.pad_token = tokenizer.eos_token\n",
69
+ " print(f\"Set pad_token to eos_token: {tokenizer.eos_token}\")\n",
70
+ "\n",
71
+ "# Load model with optimizations\n",
72
+ "print(\"Loading model (this may take several minutes)...\")\n",
73
+ "try:\n",
74
+ " model = AutoModelForCausalLM.from_pretrained(\n",
75
+ " MODEL_NAME,\n",
76
+ " cache_dir=str(CACHE_DIR),\n",
77
+ " torch_dtype=torch.bfloat16, # Use BF16 for efficiency\n",
78
+ " device_map=\"auto\", # Automatically distribute across GPUs\n",
79
+ " low_cpu_mem_usage=True # Optimize memory usage\n",
80
+ " )\n",
81
+ " model.eval() # Set to evaluation mode\n",
82
+ " print(\"βœ… Model loaded\")\n",
83
+ "except Exception as e:\n",
84
+ " print(f\"❌ Error loading model: {e}\")\n",
85
+ " raise\n",
86
+ "\n",
87
+ "# Check VRAM usage\n",
88
+ "if torch.cuda.is_available():\n",
89
+ " vram_gb = torch.cuda.max_memory_allocated() / 1024**3\n",
90
+ " print(f\"VRAM used: {vram_gb:.2f} GB\\n\")\n",
91
+ "\n",
92
+ "# === LOAD DATA ===\n",
93
+ "df = pd.read_csv(input_file)\n",
94
+ "print(f\"Loaded {len(df)} rows\")\n",
95
+ "\n",
96
+ "if TEST_MODE:\n",
97
+ " print(f\"Running in TEST MODE with {TEST_SIZE} samples\")\n",
98
+ " df = df.head(TEST_SIZE).copy()\n",
99
+ "elif MAX_ROWS:\n",
100
+ " df = df.head(MAX_ROWS).copy()\n",
101
+ "\n",
102
+ "# === CREATE PROMPT (Exact DeepSeek style) ===\n",
103
+ "def create_prompt(row):\n",
104
+ " \"\"\"Create prompt.\"\"\"\n",
105
+ " name = row.get('real_name', row.get('name', ''))\n",
106
+ " if pd.isna(name):\n",
107
+ " name = row.get('name', '')\n",
108
+ " \n",
109
+ " # Gather hints exactly like DeepSeek version\n",
110
+ " hints = []\n",
111
+ " if pd.notna(row.get('likely_profession')):\n",
112
+ " hints.append(str(row['likely_profession']))\n",
113
+ " if pd.notna(row.get('likely_nationality')):\n",
114
+ " hints.append(str(row['likely_nationality']))\n",
115
+ " if pd.notna(row.get('likely_country')):\n",
116
+ " hints.append(str(row['likely_country']))\n",
117
+ " \n",
118
+ " # Add tags if we don't have enough hints\n",
119
+ " if len(hints) < 3:\n",
120
+ " for i in range(1, 8):\n",
121
+ " tag_col = f'tag_{i}'\n",
122
+ " if tag_col in row and pd.notna(row[tag_col]):\n",
123
+ " tag_val = str(row[tag_col])\n",
124
+ " if tag_val not in hints:\n",
125
+ " hints.append(tag_val)\n",
126
+ " if len(hints) >= 5:\n",
127
+ " break\n",
128
+ " \n",
129
+ " hint_text = \", \".join(hints[:5]) if hints else \"none\"\n",
130
+ " \n",
131
+ " return f\"\"\"Given '{name}' ({hint_text}), provide:\n",
132
+ "1. Full legal name (Western order if non-latin script)\n",
133
+ "2. Any stage names/aliases (comma separated)\n",
134
+ "3. Gender (Male/Female/Other/Unknown)\n",
135
+ "4. Top 3 most likely professions from ONLY these categories:\n",
136
+ " - actor\n",
137
+ " - adult performer\n",
138
+ " - singer/musician\n",
139
+ " - model\n",
140
+ " - online personality (includes streamers, cosplayers, influencers)\n",
141
+ " - public figure (includes politicians, activists, journalists, authors)\n",
142
+ " - voice actor/ASMR\n",
143
+ " - sports professional\n",
144
+ " - tv personality (includes hosts, presenters, reality TV)\n",
145
+ "\n",
146
+ "5. Primary country associated\n",
147
+ "\n",
148
+ "IMPORTANT:\n",
149
+ "- Choose professions ONLY from the 9 categories above\n",
150
+ "- Provide up to 3 professions, comma-separated, ordered by relevance\n",
151
+ "- Be SPECIFIC: choose the most accurate category for each role\n",
152
+ "- \"online personality\" includes: streamers, cosplayers, YouTubers, influencers, content creators\n",
153
+ "- Use 'Unknown' when uncertain or for fictional characters/places\n",
154
+ "- For multi-role people, list all relevant categories (e.g., \"actor, singer/musician, online personality\")\n",
155
+ "\n",
156
+ "Respond with exactly 5 numbered lines.\"\"\"\n",
157
+ "\n",
158
+ "df['prompt'] = df.apply(create_prompt, axis=1)\n",
159
+ "\n",
160
+ "# === QUERY BLOOMZ LOCAL ===\n",
161
+ "def query_bloomz_local(prompt: str) -> str:\n",
162
+ " \"\"\"Query BLOOMZ-7B1 locally via transformers, return raw response string.\"\"\"\n",
163
+ " try:\n",
164
+ " # BLOOMZ works better with instruction-response format\n",
165
+ " full_prompt = f\"\"\"Instruction: Extract key data on a person based on the name and hints.\n",
166
+ "You must respond with exactly 5 numbered lines in this format:\n",
167
+ "1. Full legal name\n",
168
+ "2. Stage names/aliases \n",
169
+ "3. Gender\n",
170
+ "4. Professions (comma-separated, choose ONLY from: actor, adult performer, singer/musician, model, online personality, public figure, voice actor/ASMR, sports professional, tv personality)\n",
171
+ "5. Country\n",
172
+ "\n",
173
+ "{prompt}\n",
174
+ "\n",
175
+ "Response:\"\"\"\n",
176
+ " \n",
177
+ " inputs = tokenizer(\n",
178
+ " full_prompt, \n",
179
+ " return_tensors=\"pt\", \n",
180
+ " truncation=True,\n",
181
+ " max_length=2048\n",
182
+ " ).to(device)\n",
183
+ " \n",
184
+ " # Generate with adjusted parameters for BLOOMZ\n",
185
+ " with torch.no_grad():\n",
186
+ " outputs = model.generate(\n",
187
+ " **inputs,\n",
188
+ " max_new_tokens=256,\n",
189
+ " temperature=0.3, # Increased for more variability\n",
190
+ " do_sample=True,\n",
191
+ " top_p=0.9,\n",
192
+ " top_k=40,\n",
193
+ " repetition_penalty=1.1,\n",
194
+ " pad_token_id=tokenizer.eos_token_id, # Use EOS as pad token\n",
195
+ " eos_token_id=tokenizer.eos_token_id,\n",
196
+ " early_stopping=True\n",
197
+ " )\n",
198
+ " \n",
199
+ " # Decode the entire output to see what's happening\n",
200
+ " full_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n",
201
+ " \n",
202
+ " # Extract only the generated part (after the prompt)\n",
203
+ " generated_text = full_output[len(tokenizer.decode(inputs['input_ids'][0], skip_special_tokens=True)):]\n",
204
+ " \n",
205
+ " # Debug output\n",
206
+ " if not hasattr(query_bloomz_local, 'debug_count'):\n",
207
+ " query_bloomz_local.debug_count = 0\n",
208
+ " \n",
209
+ " if query_bloomz_local.debug_count < 3:\n",
210
+ " print(f\"\\nπŸ“ BLOOMZ Debug #{query_bloomz_local.debug_count + 1}:\")\n",
211
+ " print(f\"Prompt: {full_prompt[:200]}...\")\n",
212
+ " print(f\"Full output: {full_output[:500]}...\")\n",
213
+ " print(f\"Generated text: {generated_text}\")\n",
214
+ " print(f\"{'='*60}\\n\")\n",
215
+ " query_bloomz_local.debug_count += 1\n",
216
+ " \n",
217
+ " return generated_text.strip()\n",
218
+ " \n",
219
+ " except Exception as e:\n",
220
+ " print(f\"Error querying BLOOMZ: {e}\")\n",
221
+ " return None\n",
222
+ "\n",
223
+ "# === PARSE RESPONSE (Exact DeepSeek format) ===\n",
224
+ "def parse_response(response):\n",
225
+ " \"\"\"Parse numbered response into structured fields.\"\"\"\n",
226
+ " if not response:\n",
227
+ " return {\n",
228
+ " 'full_name': 'Unknown',\n",
229
+ " 'aliases': 'Unknown',\n",
230
+ " 'gender': 'Unknown',\n",
231
+ " 'profession_llm': 'Unknown',\n",
232
+ " 'country': 'Unknown'\n",
233
+ " }\n",
234
+ " \n",
235
+ " # Split into lines and clean\n",
236
+ " lines = [line.strip() for line in response.split('\\n') if line.strip()]\n",
237
+ " \n",
238
+ " # Initialize with Unknown values\n",
239
+ " fields = {\n",
240
+ " 'full_name': 'Unknown',\n",
241
+ " 'aliases': 'Unknown',\n",
242
+ " 'gender': 'Unknown',\n",
243
+ " 'profession_llm': 'Unknown',\n",
244
+ " 'country': 'Unknown'\n",
245
+ " }\n",
246
+ " \n",
247
+ " # Extract information from each numbered line\n",
248
+ " for line in lines:\n",
249
+ " if line.startswith('1.') or line.startswith('1)'):\n",
250
+ " fields['full_name'] = line[2:].strip()\n",
251
+ " elif line.startswith('2.') or line.startswith('2)'):\n",
252
+ " fields['aliases'] = line[2:].strip()\n",
253
+ " elif line.startswith('3.') or line.startswith('3)'):\n",
254
+ " fields['gender'] = line[2:].strip()\n",
255
+ " elif line.startswith('4.') or line.startswith('4)'):\n",
256
+ " fields['profession_llm'] = line[2:].strip()\n",
257
+ " elif line.startswith('5.') or line.startswith('5)'):\n",
258
+ " fields['country'] = line[2:].strip()\n",
259
+ " \n",
260
+ " return fields\n",
261
+ "\n",
262
+ "# === PROCESS ===\n",
263
+ "output_file = current_dir.parent / f\"data/CSV/bloomz_annotated_POI{'_test' if TEST_MODE else ''}.csv\"\n",
264
+ "index_file = current_dir.parent / \"misc/bloomz_query_index.txt\"\n",
265
+ "\n",
266
+ "current_index = 0\n",
267
+ "if index_file.exists():\n",
268
+ " with open(index_file) as f:\n",
269
+ " current_index = int(f.read().strip())\n",
270
+ " print(f\"Resuming from index {current_index}\")\n",
271
+ "\n",
272
+ "# Initialize columns (same as DeepSeek)\n",
273
+ "for col in ['full_name', 'gender', 'profession_llm', 'country', 'aliases']:\n",
274
+ " if col not in df.columns:\n",
275
+ " df[col] = 'Unknown'\n",
276
+ "\n",
277
+ "# Create prompts for all rows (same as DeepSeek)\n",
278
+ "print(\"Creating prompts...\")\n",
279
+ "df['prompt'] = df.apply(create_prompt, axis=1)\n",
280
+ "\n",
281
+ "print(f\"\\nAnnotating with BLOOMZ-7B1 LOCAL - rows {current_index} to {len(df)}...\")\n",
282
+ "print(f\"Model: {MODEL_NAME}\")\n",
283
+ "print(f\"This may take a while...\\n\")\n",
284
+ "\n",
285
+ "try:\n",
286
+ " start_time = time.time()\n",
287
+ " \n",
288
+ " for i in tqdm(range(current_index, len(df)), desc=\"Annotating\"):\n",
289
+ " row = df.iloc[i]\n",
290
+ " \n",
291
+ " # Query BLOOMZ (equivalent to DeepSeek query)\n",
292
+ " response = query_bloomz_local(row['prompt'])\n",
293
+ " parsed_data = parse_response(response)\n",
294
+ " \n",
295
+ " # Update dataframe\n",
296
+ " for key, value in parsed_data.items():\n",
297
+ " df.at[i, key] = value\n",
298
+ " \n",
299
+ " current_index = i + 1\n",
300
+ " \n",
301
+ " # Save progress at intervals\n",
302
+ " if (i + 1) % SAVE_INTERVAL == 0 or (i + 1) == len(df):\n",
303
+ " df.to_csv(output_file, index=False)\n",
304
+ " with open(index_file, 'w') as f:\n",
305
+ " f.write(str(current_index))\n",
306
+ " print(f\"βœ… Progress saved after {i+1} rows\")\n",
307
+ " \n",
308
+ " # Optional: Add small delay to prevent overheating (not needed for rate limiting like DeepSeek)\n",
309
+ " # time.sleep(0.1)\n",
310
+ " \n",
311
+ " elapsed_total = time.time() - start_time\n",
312
+ " print(f\"\\nβœ… Done! Final results saved to {output_file}\")\n",
313
+ " \n",
314
+ " # Summary statistics (same as DeepSeek)\n",
315
+ " print(\"\\n=== Summary Statistics ===\")\n",
316
+ " print(f\"Total processed: {len(df)}\")\n",
317
+ " print(f\"\\nGender distribution:\")\n",
318
+ " print(df['gender'].value_counts())\n",
319
+ " print(f\"\\nTop 10 profession combinations:\")\n",
320
+ " print(df['profession_llm'].value_counts().head(10))\n",
321
+ " print(f\"\\nTop 10 countries:\")\n",
322
+ " print(df['country'].value_counts().head(10))\n",
323
+ " \n",
324
+ " # Sample results\n",
325
+ " print(\"\\n=== Sample Results ===\")\n",
326
+ " display_cols = ['real_name', 'full_name', 'gender', 'profession_llm', 'country']\n",
327
+ " available_cols = [col for col in display_cols if col in df.columns]\n",
328
+ " print(df[available_cols].head(10).to_string(index=False))\n",
329
+ " \n",
330
+ " # Additional info for local model\n",
331
+ " print(f\"\\nTotal time: {elapsed_total/60:.1f} minutes\")\n",
332
+ " print(f\"Average speed: {len(df)/(elapsed_total/3600):.1f} samples/hour\")\n",
333
+ " if torch.cuda.is_available():\n",
334
+ " print(f\"Final VRAM usage: {torch.cuda.max_memory_allocated() / 1024**3:.2f} GB\")\n",
335
+ "\n",
336
+ "except Exception as e:\n",
337
+ " print(f\"⚠️ Error encountered: {e}\")\n",
338
+ " print(f\"⚠️ Last processed index: {current_index}\")\n",
339
+ " \n",
340
+ " # Save progress before exiting\n",
341
+ " df.to_csv(output_file, index=False)\n",
342
+ " with open(index_file, 'w') as f:\n",
343
+ " f.write(str(current_index))\n",
344
+ " \n",
345
+ " print(f\"⚠️ Progress saved up to row {current_index}\")"
346
+ ]
347
+ }
348
+ ],
349
+ "metadata": {
350
+ "kernelspec": {
351
+ "display_name": "pm-paper",
352
+ "language": "python",
353
+ "name": "pm-paper"
354
+ },
355
+ "language_info": {
356
+ "codemirror_mode": {
357
+ "name": "ipython",
358
+ "version": 3
359
+ },
360
+ "file_extension": ".py",
361
+ "mimetype": "text/x-python",
362
+ "name": "python",
363
+ "nbconvert_exporter": "python",
364
+ "pygments_lexer": "ipython3",
365
+ "version": "3.11.13"
366
+ }
367
+ },
368
+ "nbformat": 4,
369
+ "nbformat_minor": 5
370
+ }
jupyter_notebooks/Section_2-3-4_Bloomz_query.ipynb ADDED
@@ -0,0 +1,370 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "code",
5
+ "execution_count": null,
6
+ "id": "3b87c378-241e-41ab-be6e-84222594f22f",
7
+ "metadata": {},
8
+ "outputs": [],
9
+ "source": [
10
+ "import pandas as pd\n",
11
+ "import json\n",
12
+ "import time\n",
13
+ "import re\n",
14
+ "from pathlib import Path\n",
15
+ "from tqdm import tqdm\n",
16
+ "import torch\n",
17
+ "from transformers import AutoModelForCausalLM, AutoTokenizer\n",
18
+ "\n",
19
+ "# Import is used for pd.notna() and pd.isna() checks\n",
20
+ "\n",
21
+ "current_dir = Path.cwd()\n",
22
+ "input_file = current_dir.parent / \"data/CSV/model_adapter/real_person_adapter_step_02_NER.csv\"\n",
23
+ "\n",
24
+ "# === CONFIGURATION ===\n",
25
+ "TEST_MODE = True\n",
26
+ "TEST_SIZE = 10\n",
27
+ "MAX_ROWS = 20000\n",
28
+ "SAVE_INTERVAL = 10\n",
29
+ "\n",
30
+ "# Model settings - BLOOMZ (BigScience - European consortium)\n",
31
+ "MODEL_NAME = \"bigscience/bloomz-7b1\" # Largest instruction-tuned BLOOM model\n",
32
+ "CACHE_DIR = current_dir.parent / \"data/models\"\n",
33
+ "CACHE_DIR.mkdir(parents=True, exist_ok=True)\n",
34
+ "\n",
35
+ "PROFESSION_CATEGORIES = [\n",
36
+ " \"actor\", \"adult performer\", \"singer/musician\", \"model\",\n",
37
+ " \"online personality\", \"public figure\", \"voice actor/ASMR\",\n",
38
+ " \"sports professional\", \"tv personality\"\n",
39
+ "]\n",
40
+ "\n",
41
+ "# === LOAD MODEL ===\n",
42
+ "print(f\"Loading model: {MODEL_NAME}\")\n",
43
+ "print(f\"Cache directory: {CACHE_DIR}\")\n",
44
+ "print(f\"This may take a while on first run (~14GB download)...\\n\")\n",
45
+ "\n",
46
+ "# Check GPU availability\n",
47
+ "device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n",
48
+ "print(f\"Device: {device}\")\n",
49
+ "\n",
50
+ "if device == \"cpu\":\n",
51
+ " print(\"⚠️ WARNING: No GPU detected! Inference will be VERY slow.\")\n",
52
+ " print(\" Consider using a GPU or reducing model size.\")\n",
53
+ "\n",
54
+ "# Load tokenizer\n",
55
+ "print(\"Loading tokenizer...\")\n",
56
+ "try:\n",
57
+ " tokenizer = AutoTokenizer.from_pretrained(\n",
58
+ " MODEL_NAME,\n",
59
+ " cache_dir=str(CACHE_DIR)\n",
60
+ " )\n",
61
+ " print(\"βœ… Tokenizer loaded\")\n",
62
+ "except Exception as e:\n",
63
+ " print(f\"❌ Error loading tokenizer: {e}\")\n",
64
+ " raise\n",
65
+ "\n",
66
+ "# Ensure pad token is set\n",
67
+ "if tokenizer.pad_token is None:\n",
68
+ " tokenizer.pad_token = tokenizer.eos_token\n",
69
+ " print(f\"Set pad_token to eos_token: {tokenizer.eos_token}\")\n",
70
+ "\n",
71
+ "# Load model with optimizations\n",
72
+ "print(\"Loading model (this may take several minutes)...\")\n",
73
+ "try:\n",
74
+ " model = AutoModelForCausalLM.from_pretrained(\n",
75
+ " MODEL_NAME,\n",
76
+ " cache_dir=str(CACHE_DIR),\n",
77
+ " torch_dtype=torch.bfloat16, # Use BF16 for efficiency\n",
78
+ " device_map=\"auto\", # Automatically distribute across GPUs\n",
79
+ " low_cpu_mem_usage=True # Optimize memory usage\n",
80
+ " )\n",
81
+ " model.eval() # Set to evaluation mode\n",
82
+ " print(\"βœ… Model loaded\")\n",
83
+ "except Exception as e:\n",
84
+ " print(f\"❌ Error loading model: {e}\")\n",
85
+ " raise\n",
86
+ "\n",
87
+ "# Check VRAM usage\n",
88
+ "if torch.cuda.is_available():\n",
89
+ " vram_gb = torch.cuda.max_memory_allocated() / 1024**3\n",
90
+ " print(f\"VRAM used: {vram_gb:.2f} GB\\n\")\n",
91
+ "\n",
92
+ "# === LOAD DATA ===\n",
93
+ "df = pd.read_csv(input_file)\n",
94
+ "print(f\"Loaded {len(df)} rows\")\n",
95
+ "\n",
96
+ "if TEST_MODE:\n",
97
+ " print(f\"Running in TEST MODE with {TEST_SIZE} samples\")\n",
98
+ " df = df.head(TEST_SIZE).copy()\n",
99
+ "elif MAX_ROWS:\n",
100
+ " df = df.head(MAX_ROWS).copy()\n",
101
+ "\n",
102
+ "# === CREATE PROMPT (Exact DeepSeek style) ===\n",
103
+ "def create_prompt(row):\n",
104
+ " \"\"\"Create prompt.\"\"\"\n",
105
+ " name = row.get('real_name', row.get('name', ''))\n",
106
+ " if pd.isna(name):\n",
107
+ " name = row.get('name', '')\n",
108
+ " \n",
109
+ " # Gather hints exactly like DeepSeek version\n",
110
+ " hints = []\n",
111
+ " if pd.notna(row.get('likely_profession')):\n",
112
+ " hints.append(str(row['likely_profession']))\n",
113
+ " if pd.notna(row.get('likely_nationality')):\n",
114
+ " hints.append(str(row['likely_nationality']))\n",
115
+ " if pd.notna(row.get('likely_country')):\n",
116
+ " hints.append(str(row['likely_country']))\n",
117
+ " \n",
118
+ " # Add tags if we don't have enough hints\n",
119
+ " if len(hints) < 3:\n",
120
+ " for i in range(1, 8):\n",
121
+ " tag_col = f'tag_{i}'\n",
122
+ " if tag_col in row and pd.notna(row[tag_col]):\n",
123
+ " tag_val = str(row[tag_col])\n",
124
+ " if tag_val not in hints:\n",
125
+ " hints.append(tag_val)\n",
126
+ " if len(hints) >= 5:\n",
127
+ " break\n",
128
+ " \n",
129
+ " hint_text = \", \".join(hints[:5]) if hints else \"none\"\n",
130
+ " \n",
131
+ " return f\"\"\"Given '{name}' ({hint_text}), provide:\n",
132
+ "1. Full legal name (Western order if non-latin script)\n",
133
+ "2. Any stage names/aliases (comma separated)\n",
134
+ "3. Gender (Male/Female/Other/Unknown)\n",
135
+ "4. Top 3 most likely professions from ONLY these categories:\n",
136
+ " - actor\n",
137
+ " - adult performer\n",
138
+ " - singer/musician\n",
139
+ " - model\n",
140
+ " - online personality (includes streamers, cosplayers, influencers)\n",
141
+ " - public figure (includes politicians, activists, journalists, authors)\n",
142
+ " - voice actor/ASMR\n",
143
+ " - sports professional\n",
144
+ " - tv personality (includes hosts, presenters, reality TV)\n",
145
+ "\n",
146
+ "5. Primary country associated\n",
147
+ "\n",
148
+ "IMPORTANT:\n",
149
+ "- Choose professions ONLY from the 9 categories above\n",
150
+ "- Provide up to 3 professions, comma-separated, ordered by relevance\n",
151
+ "- Be SPECIFIC: choose the most accurate category for each role\n",
152
+ "- \"online personality\" includes: streamers, cosplayers, YouTubers, influencers, content creators\n",
153
+ "- Use 'Unknown' when uncertain or for fictional characters/places\n",
154
+ "- For multi-role people, list all relevant categories (e.g., \"actor, singer/musician, online personality\")\n",
155
+ "\n",
156
+ "Respond with exactly 5 numbered lines.\"\"\"\n",
157
+ "\n",
158
+ "df['prompt'] = df.apply(create_prompt, axis=1)\n",
159
+ "\n",
160
+ "# === QUERY BLOOMZ LOCAL ===\n",
161
+ "def query_bloomz_local(prompt: str) -> str:\n",
162
+ " \"\"\"Query BLOOMZ-7B1 locally via transformers, return raw response string.\"\"\"\n",
163
+ " try:\n",
164
+ " # BLOOMZ works better with instruction-response format\n",
165
+ " full_prompt = f\"\"\"Instruction: Extract key data on a person based on the name and hints.\n",
166
+ "You must respond with exactly 5 numbered lines in this format:\n",
167
+ "1. Full legal name\n",
168
+ "2. Stage names/aliases \n",
169
+ "3. Gender\n",
170
+ "4. Professions (comma-separated, choose ONLY from: actor, adult performer, singer/musician, model, online personality, public figure, voice actor/ASMR, sports professional, tv personality)\n",
171
+ "5. Country\n",
172
+ "\n",
173
+ "{prompt}\n",
174
+ "\n",
175
+ "Response:\"\"\"\n",
176
+ " \n",
177
+ " inputs = tokenizer(\n",
178
+ " full_prompt, \n",
179
+ " return_tensors=\"pt\", \n",
180
+ " truncation=True,\n",
181
+ " max_length=2048\n",
182
+ " ).to(device)\n",
183
+ " \n",
184
+ " # Generate with adjusted parameters for BLOOMZ\n",
185
+ " with torch.no_grad():\n",
186
+ " outputs = model.generate(\n",
187
+ " **inputs,\n",
188
+ " max_new_tokens=256,\n",
189
+ " temperature=0.3, # Increased for more variability\n",
190
+ " do_sample=True,\n",
191
+ " top_p=0.9,\n",
192
+ " top_k=40,\n",
193
+ " repetition_penalty=1.1,\n",
194
+ " pad_token_id=tokenizer.eos_token_id, # Use EOS as pad token\n",
195
+ " eos_token_id=tokenizer.eos_token_id,\n",
196
+ " early_stopping=True\n",
197
+ " )\n",
198
+ " \n",
199
+ " # Decode the entire output to see what's happening\n",
200
+ " full_output = tokenizer.decode(outputs[0], skip_special_tokens=True)\n",
201
+ " \n",
202
+ " # Extract only the generated part (after the prompt)\n",
203
+ " generated_text = full_output[len(tokenizer.decode(inputs['input_ids'][0], skip_special_tokens=True)):]\n",
204
+ " \n",
205
+ " # Debug output\n",
206
+ " if not hasattr(query_bloomz_local, 'debug_count'):\n",
207
+ " query_bloomz_local.debug_count = 0\n",
208
+ " \n",
209
+ " if query_bloomz_local.debug_count < 3:\n",
210
+ " print(f\"\\nπŸ“ BLOOMZ Debug #{query_bloomz_local.debug_count + 1}:\")\n",
211
+ " print(f\"Prompt: {full_prompt[:200]}...\")\n",
212
+ " print(f\"Full output: {full_output[:500]}...\")\n",
213
+ " print(f\"Generated text: {generated_text}\")\n",
214
+ " print(f\"{'='*60}\\n\")\n",
215
+ " query_bloomz_local.debug_count += 1\n",
216
+ " \n",
217
+ " return generated_text.strip()\n",
218
+ " \n",
219
+ " except Exception as e:\n",
220
+ " print(f\"Error querying BLOOMZ: {e}\")\n",
221
+ " return None\n",
222
+ "\n",
223
+ "# === PARSE RESPONSE (Exact DeepSeek format) ===\n",
224
+ "def parse_response(response):\n",
225
+ " \"\"\"Parse numbered response into structured fields.\"\"\"\n",
226
+ " if not response:\n",
227
+ " return {\n",
228
+ " 'full_name': 'Unknown',\n",
229
+ " 'aliases': 'Unknown',\n",
230
+ " 'gender': 'Unknown',\n",
231
+ " 'profession_llm': 'Unknown',\n",
232
+ " 'country': 'Unknown'\n",
233
+ " }\n",
234
+ " \n",
235
+ " # Split into lines and clean\n",
236
+ " lines = [line.strip() for line in response.split('\\n') if line.strip()]\n",
237
+ " \n",
238
+ " # Initialize with Unknown values\n",
239
+ " fields = {\n",
240
+ " 'full_name': 'Unknown',\n",
241
+ " 'aliases': 'Unknown',\n",
242
+ " 'gender': 'Unknown',\n",
243
+ " 'profession_llm': 'Unknown',\n",
244
+ " 'country': 'Unknown'\n",
245
+ " }\n",
246
+ " \n",
247
+ " # Extract information from each numbered line\n",
248
+ " for line in lines:\n",
249
+ " if line.startswith('1.') or line.startswith('1)'):\n",
250
+ " fields['full_name'] = line[2:].strip()\n",
251
+ " elif line.startswith('2.') or line.startswith('2)'):\n",
252
+ " fields['aliases'] = line[2:].strip()\n",
253
+ " elif line.startswith('3.') or line.startswith('3)'):\n",
254
+ " fields['gender'] = line[2:].strip()\n",
255
+ " elif line.startswith('4.') or line.startswith('4)'):\n",
256
+ " fields['profession_llm'] = line[2:].strip()\n",
257
+ " elif line.startswith('5.') or line.startswith('5)'):\n",
258
+ " fields['country'] = line[2:].strip()\n",
259
+ " \n",
260
+ " return fields\n",
261
+ "\n",
262
+ "# === PROCESS ===\n",
263
+ "output_file = current_dir.parent / f\"data/CSV/bloomz_annotated_POI{'_test' if TEST_MODE else ''}.csv\"\n",
264
+ "index_file = current_dir.parent / \"misc/bloomz_query_index.txt\"\n",
265
+ "\n",
266
+ "current_index = 0\n",
267
+ "if index_file.exists():\n",
268
+ " with open(index_file) as f:\n",
269
+ " current_index = int(f.read().strip())\n",
270
+ " print(f\"Resuming from index {current_index}\")\n",
271
+ "\n",
272
+ "# Initialize columns (same as DeepSeek)\n",
273
+ "for col in ['full_name', 'gender', 'profession_llm', 'country', 'aliases']:\n",
274
+ " if col not in df.columns:\n",
275
+ " df[col] = 'Unknown'\n",
276
+ "\n",
277
+ "# Create prompts for all rows (same as DeepSeek)\n",
278
+ "print(\"Creating prompts...\")\n",
279
+ "df['prompt'] = df.apply(create_prompt, axis=1)\n",
280
+ "\n",
281
+ "print(f\"\\nAnnotating with BLOOMZ-7B1 LOCAL - rows {current_index} to {len(df)}...\")\n",
282
+ "print(f\"Model: {MODEL_NAME}\")\n",
283
+ "print(f\"This may take a while...\\n\")\n",
284
+ "\n",
285
+ "try:\n",
286
+ " start_time = time.time()\n",
287
+ " \n",
288
+ " for i in tqdm(range(current_index, len(df)), desc=\"Annotating\"):\n",
289
+ " row = df.iloc[i]\n",
290
+ " \n",
291
+ " # Query BLOOMZ (equivalent to DeepSeek query)\n",
292
+ " response = query_bloomz_local(row['prompt'])\n",
293
+ " parsed_data = parse_response(response)\n",
294
+ " \n",
295
+ " # Update dataframe\n",
296
+ " for key, value in parsed_data.items():\n",
297
+ " df.at[i, key] = value\n",
298
+ " \n",
299
+ " current_index = i + 1\n",
300
+ " \n",
301
+ " # Save progress at intervals\n",
302
+ " if (i + 1) % SAVE_INTERVAL == 0 or (i + 1) == len(df):\n",
303
+ " df.to_csv(output_file, index=False)\n",
304
+ " with open(index_file, 'w') as f:\n",
305
+ " f.write(str(current_index))\n",
306
+ " print(f\"βœ… Progress saved after {i+1} rows\")\n",
307
+ " \n",
308
+ " # Optional: Add small delay to prevent overheating (not needed for rate limiting like DeepSeek)\n",
309
+ " # time.sleep(0.1)\n",
310
+ " \n",
311
+ " elapsed_total = time.time() - start_time\n",
312
+ " print(f\"\\nβœ… Done! Final results saved to {output_file}\")\n",
313
+ " \n",
314
+ " # Summary statistics (same as DeepSeek)\n",
315
+ " print(\"\\n=== Summary Statistics ===\")\n",
316
+ " print(f\"Total processed: {len(df)}\")\n",
317
+ " print(f\"\\nGender distribution:\")\n",
318
+ " print(df['gender'].value_counts())\n",
319
+ " print(f\"\\nTop 10 profession combinations:\")\n",
320
+ " print(df['profession_llm'].value_counts().head(10))\n",
321
+ " print(f\"\\nTop 10 countries:\")\n",
322
+ " print(df['country'].value_counts().head(10))\n",
323
+ " \n",
324
+ " # Sample results\n",
325
+ " print(\"\\n=== Sample Results ===\")\n",
326
+ " display_cols = ['real_name', 'full_name', 'gender', 'profession_llm', 'country']\n",
327
+ " available_cols = [col for col in display_cols if col in df.columns]\n",
328
+ " print(df[available_cols].head(10).to_string(index=False))\n",
329
+ " \n",
330
+ " # Additional info for local model\n",
331
+ " print(f\"\\nTotal time: {elapsed_total/60:.1f} minutes\")\n",
332
+ " print(f\"Average speed: {len(df)/(elapsed_total/3600):.1f} samples/hour\")\n",
333
+ " if torch.cuda.is_available():\n",
334
+ " print(f\"Final VRAM usage: {torch.cuda.max_memory_allocated() / 1024**3:.2f} GB\")\n",
335
+ "\n",
336
+ "except Exception as e:\n",
337
+ " print(f\"⚠️ Error encountered: {e}\")\n",
338
+ " print(f\"⚠️ Last processed index: {current_index}\")\n",
339
+ " \n",
340
+ " # Save progress before exiting\n",
341
+ " df.to_csv(output_file, index=False)\n",
342
+ " with open(index_file, 'w') as f:\n",
343
+ " f.write(str(current_index))\n",
344
+ " \n",
345
+ " print(f\"⚠️ Progress saved up to row {current_index}\")"
346
+ ]
347
+ }
348
+ ],
349
+ "metadata": {
350
+ "kernelspec": {
351
+ "display_name": "pm-paper",
352
+ "language": "python",
353
+ "name": "pm-paper"
354
+ },
355
+ "language_info": {
356
+ "codemirror_mode": {
357
+ "name": "ipython",
358
+ "version": 3
359
+ },
360
+ "file_extension": ".py",
361
+ "mimetype": "text/x-python",
362
+ "name": "python",
363
+ "nbconvert_exporter": "python",
364
+ "pygments_lexer": "ipython3",
365
+ "version": "3.11.13"
366
+ }
367
+ },
368
+ "nbformat": 4,
369
+ "nbformat_minor": 5
370
+ }
jupyter_notebooks/Section_2-3-4_Figure_8_deepfake_adapters.ipynb CHANGED
@@ -1829,15 +1829,26 @@
1829
  },
1830
  {
1831
  "cell_type": "code",
1832
- "execution_count": null,
1833
  "id": "72c468dc-1cbe-41d4-b71b-17c40ea4f9a1",
1834
  "metadata": {
1835
  "execution": {
1836
- "iopub.execute_input": "2025-11-21T19:50:12.690047Z",
1837
- "iopub.status.busy": "2025-11-21T19:50:12.689854Z"
 
 
 
1838
  }
1839
  },
1840
  "outputs": [
 
 
 
 
 
 
 
 
1841
  {
1842
  "name": "stdout",
1843
  "output_type": "stream",
@@ -1849,9 +1860,22 @@
1849
  "Device: cuda\n",
1850
  "Loading tokenizer...\n",
1851
  "βœ… Tokenizer loaded\n",
1852
- "Loading model (this may take several minutes)...\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
1853
  "βœ… Model loaded\n",
1854
- "VRAM used: 28.59 GB\n",
1855
  "\n",
1856
  "Loaded 50861 rows\n",
1857
  "Running in TEST MODE with 10 samples\n",
@@ -1867,7 +1891,151 @@
1867
  "name": "stderr",
1868
  "output_type": "stream",
1869
  "text": [
1870
- "Annotating: 0%| | 0/10 [00:00<?, ?it/s]The following generation flags are not valid and may be ignored: ['early_stopping']. Set `TRANSFORMERS_VERBOSITY=info` for more details.\n"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1871
  ]
1872
  }
1873
  ],
 
1829
  },
1830
  {
1831
  "cell_type": "code",
1832
+ "execution_count": 1,
1833
  "id": "72c468dc-1cbe-41d4-b71b-17c40ea4f9a1",
1834
  "metadata": {
1835
  "execution": {
1836
+ "iopub.execute_input": "2025-11-28T09:54:50.978895Z",
1837
+ "iopub.status.busy": "2025-11-28T09:54:50.978724Z",
1838
+ "iopub.status.idle": "2025-11-28T10:02:38.826002Z",
1839
+ "shell.execute_reply": "2025-11-28T10:02:38.825201Z",
1840
+ "shell.execute_reply.started": "2025-11-28T09:54:50.978881Z"
1841
  }
1842
  },
1843
  "outputs": [
1844
+ {
1845
+ "name": "stderr",
1846
+ "output_type": "stream",
1847
+ "text": [
1848
+ "/shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/.venv/lib/python3.11/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html\n",
1849
+ " from .autonotebook import tqdm as notebook_tqdm\n"
1850
+ ]
1851
+ },
1852
  {
1853
  "name": "stdout",
1854
  "output_type": "stream",
 
1860
  "Device: cuda\n",
1861
  "Loading tokenizer...\n",
1862
  "βœ… Tokenizer loaded\n",
1863
+ "Loading model (this may take several minutes)...\n"
1864
+ ]
1865
+ },
1866
+ {
1867
+ "name": "stderr",
1868
+ "output_type": "stream",
1869
+ "text": [
1870
+ "`torch_dtype` is deprecated! Use `dtype` instead!\n"
1871
+ ]
1872
+ },
1873
+ {
1874
+ "name": "stdout",
1875
+ "output_type": "stream",
1876
+ "text": [
1877
  "βœ… Model loaded\n",
1878
+ "VRAM used: 15.08 GB\n",
1879
  "\n",
1880
  "Loaded 50861 rows\n",
1881
  "Running in TEST MODE with 10 samples\n",
 
1891
  "name": "stderr",
1892
  "output_type": "stream",
1893
  "text": [
1894
+ "Annotating: 0%| | 0/10 [00:00<?, ?it/s]The following generation flags are not valid and may be ignored: ['early_stopping']. Set `TRANSFORMERS_VERBOSITY=info` for more details.\n",
1895
+ "Annotating: 20%|β–ˆβ–ˆ | 2/10 [00:03<00:13, 1.66s/it]"
1896
+ ]
1897
+ },
1898
+ {
1899
+ "name": "stdout",
1900
+ "output_type": "stream",
1901
+ "text": [
1902
+ "\n",
1903
+ "πŸ“ BLOOMZ Debug #1:\n",
1904
+ "Prompt: Instruction: Extract key data on a person based on the name and hints.\n",
1905
+ "You must respond with exactly 5 numbered lines in this format:\n",
1906
+ "1. Full legal name\n",
1907
+ "2. Stage names/aliases \n",
1908
+ "3. Gender\n",
1909
+ "4. Professio...\n",
1910
+ "Full output: Instruction: Extract key data on a person based on the name and hints.\n",
1911
+ "You must respond with exactly 5 numbered lines in this format:\n",
1912
+ "1. Full legal name\n",
1913
+ "2. Stage names/aliases \n",
1914
+ "3. Gender\n",
1915
+ "4. Professions (comma-separated, choose ONLY from: actor, adult performer, singer/musician, model, online personality, public figure, voice actor/ASMR, sports professional, tv personality)\n",
1916
+ "5. Country\n",
1917
+ "\n",
1918
+ "Given 'IU' (celebrity, girl, photorealistic, female, asian), provide:\n",
1919
+ "1. Full legal name (Western order if non-...\n",
1920
+ "Generated text: \n",
1921
+ " IU, celebrity, girl, photorealistic, female, asian\n",
1922
+ "============================================================\n",
1923
+ "\n",
1924
+ "\n",
1925
+ "πŸ“ BLOOMZ Debug #2:\n",
1926
+ "Prompt: Instruction: Extract key data on a person based on the name and hints.\n",
1927
+ "You must respond with exactly 5 numbered lines in this format:\n",
1928
+ "1. Full legal name\n",
1929
+ "2. Stage names/aliases \n",
1930
+ "3. Gender\n",
1931
+ "4. Professio...\n",
1932
+ "Full output: Instruction: Extract key data on a person based on the name and hints.\n",
1933
+ "You must respond with exactly 5 numbered lines in this format:\n",
1934
+ "1. Full legal name\n",
1935
+ "2. Stage names/aliases \n",
1936
+ "3. Gender\n",
1937
+ "4. Professions (comma-separated, choose ONLY from: actor, adult performer, singer/musician, model, online personality, public figure, voice actor/ASMR, sports professional, tv personality)\n",
1938
+ "5. Country\n",
1939
+ "\n",
1940
+ "Given 'Super Pose Book' (sexy, female, general purpose, poses, art style), provide:\n",
1941
+ "1. Full legal name (Western...\n",
1942
+ "Generated text: \n",
1943
+ "Alice Smith\n",
1944
+ "============================================================\n",
1945
+ "\n"
1946
+ ]
1947
+ },
1948
+ {
1949
+ "name": "stderr",
1950
+ "output_type": "stream",
1951
+ "text": [
1952
+ "Annotating: 30%|β–ˆβ–ˆβ–ˆ | 3/10 [00:04<00:08, 1.19s/it]"
1953
+ ]
1954
+ },
1955
+ {
1956
+ "name": "stdout",
1957
+ "output_type": "stream",
1958
+ "text": [
1959
+ "\n",
1960
+ "πŸ“ BLOOMZ Debug #3:\n",
1961
+ "Prompt: Instruction: Extract key data on a person based on the name and hints.\n",
1962
+ "You must respond with exactly 5 numbered lines in this format:\n",
1963
+ "1. Full legal name\n",
1964
+ "2. Stage names/aliases \n",
1965
+ "3. Gender\n",
1966
+ "4. Professio...\n",
1967
+ "Full output: Instruction: Extract key data on a person based on the name and hints.\n",
1968
+ "You must respond with exactly 5 numbered lines in this format:\n",
1969
+ "1. Full legal name\n",
1970
+ "2. Stage names/aliases \n",
1971
+ "3. Gender\n",
1972
+ "4. Professions (comma-separated, choose ONLY from: actor, adult performer, singer/musician, model, online personality, public figure, voice actor/ASMR, sports professional, tv personality)\n",
1973
+ "5. Country\n",
1974
+ "\n",
1975
+ "Given 'Liyuu' (girl, photorealistic, portraits, real person), provide:\n",
1976
+ "1. Full legal name (Western order if non...\n",
1977
+ "Generated text: \n",
1978
+ "Liyuu is a girl who has been a model, singer/musician, and actor. She is also known as \"Lilith\"\n",
1979
+ "============================================================\n",
1980
+ "\n"
1981
+ ]
1982
+ },
1983
+ {
1984
+ "name": "stderr",
1985
+ "output_type": "stream",
1986
+ "text": [
1987
+ "Annotating: 100%|β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ| 10/10 [00:06<00:00, 1.64it/s]"
1988
+ ]
1989
+ },
1990
+ {
1991
+ "name": "stdout",
1992
+ "output_type": "stream",
1993
+ "text": [
1994
+ "βœ… Progress saved after 10 rows\n",
1995
+ "\n",
1996
+ "βœ… Done! Final results saved to /shares/weddigen.ki.uzh/laura_wagner/phase_01/pm-paper/data/CSV/bloomz_annotated_POI_test.csv\n",
1997
+ "\n",
1998
+ "=== Summary Statistics ===\n",
1999
+ "Total processed: 10\n",
2000
+ "\n",
2001
+ "Gender distribution:\n",
2002
+ "gender\n",
2003
+ "Unknown 10\n",
2004
+ "Name: count, dtype: int64\n",
2005
+ "\n",
2006
+ "Top 10 profession combinations:\n",
2007
+ "profession_llm\n",
2008
+ "Unknown 10\n",
2009
+ "Name: count, dtype: int64\n",
2010
+ "\n",
2011
+ "Top 10 countries:\n",
2012
+ "country\n",
2013
+ "Unknown 10\n",
2014
+ "Name: count, dtype: int64\n",
2015
+ "\n",
2016
+ "=== Sample Results ===\n",
2017
+ " real_name full_name gender profession_llm country\n",
2018
+ " IU Unknown Unknown Unknown Unknown\n",
2019
+ "Super Pose Book Unknown Unknown Unknown Unknown\n",
2020
+ " Liyuu Unknown Unknown Unknown Unknown\n",
2021
+ " Irene Unknown Unknown Unknown Unknown\n",
2022
+ " AESPA Karina Unknown Unknown Unknown Unknown\n",
2023
+ " Saika Kawakita Unknown Unknown Unknown Unknown\n",
2024
+ " Liu Yifei Unknown Unknown Unknown Unknown\n",
2025
+ " HashimotoKanna Unknown Unknown Unknown Unknown\n",
2026
+ " Emma Watson Unknown Unknown Unknown Unknown\n",
2027
+ " Gal Gadot Unknown Unknown Unknown Unknown\n",
2028
+ "\n",
2029
+ "Total time: 0.1 minutes\n",
2030
+ "Average speed: 5801.5 samples/hour\n",
2031
+ "Final VRAM usage: 15.08 GB\n"
2032
+ ]
2033
+ },
2034
+ {
2035
+ "name": "stderr",
2036
+ "output_type": "stream",
2037
+ "text": [
2038
+ "\n"
2039
  ]
2040
  }
2041
  ],
misc/deepseek_query_index.txt DELETED
@@ -1 +0,0 @@
1
- 10
 
 
misc/gemma_local_query_index.txt DELETED
@@ -1 +0,0 @@
1
- 10
 
 
misc/mistral_local_query_index.txt DELETED
@@ -1 +0,0 @@
1
- 10
 
 
misc/qwen_local_query_index.txt DELETED
@@ -1 +0,0 @@
1
- 10
 
 
misc/qwen_query_index.txt DELETED
@@ -1 +0,0 @@
1
- 10