laura.wagner commited on
Commit
f8cb255
·
1 Parent(s): 7103328

added tag hieararchy

Browse files
jupyter_notebooks/Section_1_Figure_1_image_grid.ipynb CHANGED
@@ -214,7 +214,7 @@
214
  },
215
  {
216
  "cell_type": "code",
217
- "execution_count": 5,
218
  "id": "88a6ca8e-51ee-4d27-ba0c-5f026d5750df",
219
  "metadata": {
220
  "execution": {
@@ -225,14 +225,23 @@
225
  "shell.execute_reply.started": "2025-02-08T21:58:52.986878Z"
226
  }
227
  },
228
- "outputs": [],
 
 
 
 
 
 
 
 
 
229
  "source": [
230
- "#download_and_save_data(input_dir, images) "
231
  ]
232
  },
233
  {
234
  "cell_type": "code",
235
- "execution_count": 6,
236
  "id": "f247091b-7b45-4cd4-a6ca-3d4dea414e19",
237
  "metadata": {
238
  "execution": {
@@ -253,6 +262,13 @@
253
  }
254
  ],
255
  "source": [
 
 
 
 
 
 
 
256
  "directory = current_dir.parent / 'data/sorted/images/'\n",
257
  "#directory = '/home/lauwag/shares/laura_wagner/Civitai_page_analysis/Civitai_dataset/dataset/chronological/full/prompts-images/2024/2024-05/2024-05-31/'\n",
258
  "output = current_dir.parent / 'plots/grid10x120.png'\n",
@@ -331,19 +347,31 @@
331
  "\n",
332
  "file_types = ('png', 'jpg', 'jpeg') # Define acceptable image file types\n",
333
  "\n",
334
- "images = []\n",
 
 
 
 
335
  "for root, dirs, files in os.walk(directory):\n",
336
  " for file in files:\n",
337
  " if file.lower().endswith(file_types):\n",
338
  " image_path = os.path.join(root, file)\n",
339
  " json_path = image_path.rsplit('.', 1)[0] + '.json'\n",
340
  " if os.path.exists(json_path):\n",
341
- " img = process_image(image_path, json_path, cell_size)\n",
342
- " images.append(img)\n",
343
- " if len(images) == grid_size[0] * grid_size[1]:\n",
344
- " break\n",
345
- " if len(images) == grid_size[0] * grid_size[1]:\n",
346
- " break\n",
 
 
 
 
 
 
 
 
347
  "\n",
348
  "# Create the grid image\n",
349
  "grid_img = Image.new('RGB', (grid_size[1] * cell_size, grid_size[0] * cell_size))\n",
 
214
  },
215
  {
216
  "cell_type": "code",
217
+ "execution_count": 6,
218
  "id": "88a6ca8e-51ee-4d27-ba0c-5f026d5750df",
219
  "metadata": {
220
  "execution": {
 
225
  "shell.execute_reply.started": "2025-02-08T21:58:52.986878Z"
226
  }
227
  },
228
+ "outputs": [
229
+ {
230
+ "name": "stdout",
231
+ "output_type": "stream",
232
+ "text": [
233
+ "Scanning directory: /home/lauhp/000_PHD/000_010_PUBLICATION/CODE/pm-paper/data/sorted/image_metadata\n",
234
+ "No JSON files found in the directory.\n"
235
+ ]
236
+ }
237
+ ],
238
  "source": [
239
+ "download_and_save_data(input_dir, images) "
240
  ]
241
  },
242
  {
243
  "cell_type": "code",
244
+ "execution_count": null,
245
  "id": "f247091b-7b45-4cd4-a6ca-3d4dea414e19",
246
  "metadata": {
247
  "execution": {
 
262
  }
263
  ],
264
  "source": [
265
+ "import random\n",
266
+ "\n",
267
+ "# For creating the figure grid\n",
268
+ "images = []\n",
269
+ "all_valid_images = []\n",
270
+ "\n",
271
+ "\n",
272
  "directory = current_dir.parent / 'data/sorted/images/'\n",
273
  "#directory = '/home/lauwag/shares/laura_wagner/Civitai_page_analysis/Civitai_dataset/dataset/chronological/full/prompts-images/2024/2024-05/2024-05-31/'\n",
274
  "output = current_dir.parent / 'plots/grid10x120.png'\n",
 
347
  "\n",
348
  "file_types = ('png', 'jpg', 'jpeg') # Define acceptable image file types\n",
349
  "\n",
350
+ "\n",
351
+ "\n",
352
+ "\n",
353
+ "\n",
354
+ "# First, collect all valid image paths\n",
355
  "for root, dirs, files in os.walk(directory):\n",
356
  " for file in files:\n",
357
  " if file.lower().endswith(file_types):\n",
358
  " image_path = os.path.join(root, file)\n",
359
  " json_path = image_path.rsplit('.', 1)[0] + '.json'\n",
360
  " if os.path.exists(json_path):\n",
361
+ " all_valid_images.append((image_path, json_path))\n",
362
+ "\n",
363
+ "# Randomly sample from valid images\n",
364
+ "num_needed = grid_size[0] * grid_size[1]\n",
365
+ "if len(all_valid_images) >= num_needed:\n",
366
+ " random.seed(42) # For reproducibility\n",
367
+ " sampled_images = random.sample(all_valid_images, num_needed)\n",
368
+ " \n",
369
+ " for image_path, json_path in sampled_images:\n",
370
+ " img = process_image(image_path, json_path, cell_size)\n",
371
+ " images.append(img)\n",
372
+ "else:\n",
373
+ " print(f\"Warning: Only {len(all_valid_images)} valid images found, need {num_needed}\")\n",
374
+ "\n",
375
  "\n",
376
  "# Create the grid image\n",
377
  "grid_img = Image.new('RGB', (grid_size[1] * cell_size, grid_size[0] * cell_size))\n",
jupyter_notebooks/Section_2-3-4_Figure_8_Step_1_LLM_annotation.ipynb CHANGED
@@ -1919,9 +1919,9 @@
1919
  ],
1920
  "metadata": {
1921
  "kernelspec": {
1922
- "display_name": "pm-paper",
1923
  "language": "python",
1924
- "name": "pm-paper"
1925
  },
1926
  "language_info": {
1927
  "codemirror_mode": {
@@ -1933,7 +1933,7 @@
1933
  "name": "python",
1934
  "nbconvert_exporter": "python",
1935
  "pygments_lexer": "ipython3",
1936
- "version": "3.11.13"
1937
  }
1938
  },
1939
  "nbformat": 4,
 
1919
  ],
1920
  "metadata": {
1921
  "kernelspec": {
1922
+ "display_name": "latm",
1923
  "language": "python",
1924
+ "name": "python3"
1925
  },
1926
  "language_info": {
1927
  "codemirror_mode": {
 
1933
  "name": "python",
1934
  "nbconvert_exporter": "python",
1935
  "pygments_lexer": "ipython3",
1936
+ "version": "3.10.15"
1937
  }
1938
  },
1939
  "nbformat": 4,
jupyter_notebooks/Section_2-3-4_Figure_8_Step_2_response_comparison_and_consensus_extraction.ipynb CHANGED
@@ -2831,9 +2831,258 @@
2831
  },
2832
  {
2833
  "cell_type": "code",
2834
- "execution_count": null,
2835
  "id": "70403f7c-6ea9-4f21-9704-aec0c37a591b",
2836
  "metadata": {},
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2837
  "outputs": [],
2838
  "source": []
2839
  }
 
2831
  },
2832
  {
2833
  "cell_type": "code",
2834
+ "execution_count": 5,
2835
  "id": "70403f7c-6ea9-4f21-9704-aec0c37a591b",
2836
  "metadata": {},
2837
+ "outputs": [
2838
+ {
2839
+ "name": "stdout",
2840
+ "output_type": "stream",
2841
+ "text": [
2842
+ "Original rows: 21400\n",
2843
+ "Rows with name consensus: 16464\n",
2844
+ "Unique persons after aggregation: 8242\n",
2845
+ "Saved to: /home/lauhp/000_PHD/000_010_PUBLICATION/CODE/pm-paper/data/CSV/final_aggregated.csv\n"
2846
+ ]
2847
+ },
2848
+ {
2849
+ "data": {
2850
+ "text/html": [
2851
+ "<div>\n",
2852
+ "<style scoped>\n",
2853
+ " .dataframe tbody tr th:only-of-type {\n",
2854
+ " vertical-align: middle;\n",
2855
+ " }\n",
2856
+ "\n",
2857
+ " .dataframe tbody tr th {\n",
2858
+ " vertical-align: top;\n",
2859
+ " }\n",
2860
+ "\n",
2861
+ " .dataframe thead th {\n",
2862
+ " text-align: right;\n",
2863
+ " }\n",
2864
+ "</style>\n",
2865
+ "<table border=\"1\" class=\"dataframe\">\n",
2866
+ " <thead>\n",
2867
+ " <tr style=\"text-align: right;\">\n",
2868
+ " <th></th>\n",
2869
+ " <th>name</th>\n",
2870
+ " <th>model_ids</th>\n",
2871
+ " <th>number_of_models</th>\n",
2872
+ " <th>consensus_profession</th>\n",
2873
+ " <th>consensus_gender</th>\n",
2874
+ " <th>consensus_country</th>\n",
2875
+ " </tr>\n",
2876
+ " </thead>\n",
2877
+ " <tbody>\n",
2878
+ " <tr>\n",
2879
+ " <th>0</th>\n",
2880
+ " <td>AJ Applegate</td>\n",
2881
+ " <td>[1335260]</td>\n",
2882
+ " <td>1</td>\n",
2883
+ " <td>adult performer</td>\n",
2884
+ " <td>Female</td>\n",
2885
+ " <td>USA</td>\n",
2886
+ " </tr>\n",
2887
+ " <tr>\n",
2888
+ " <th>1</th>\n",
2889
+ " <td>AJ Langer</td>\n",
2890
+ " <td>[1216782]</td>\n",
2891
+ " <td>1</td>\n",
2892
+ " <td>actor</td>\n",
2893
+ " <td>Female</td>\n",
2894
+ " <td>USA</td>\n",
2895
+ " </tr>\n",
2896
+ " <tr>\n",
2897
+ " <th>2</th>\n",
2898
+ " <td>Aaliyah Dana Haughton</td>\n",
2899
+ " <td>[221782, 230769, 1191059, 1564721]</td>\n",
2900
+ " <td>4</td>\n",
2901
+ " <td>singer/musician</td>\n",
2902
+ " <td>Female</td>\n",
2903
+ " <td>USA</td>\n",
2904
+ " </tr>\n",
2905
+ " <tr>\n",
2906
+ " <th>3</th>\n",
2907
+ " <td>Aaliyah Love</td>\n",
2908
+ " <td>[305961]</td>\n",
2909
+ " <td>1</td>\n",
2910
+ " <td>adult performer</td>\n",
2911
+ " <td>Female</td>\n",
2912
+ " <td>USA</td>\n",
2913
+ " </tr>\n",
2914
+ " <tr>\n",
2915
+ " <th>4</th>\n",
2916
+ " <td>Aaron Boone</td>\n",
2917
+ " <td>[437968]</td>\n",
2918
+ " <td>1</td>\n",
2919
+ " <td>sports professional</td>\n",
2920
+ " <td>Male</td>\n",
2921
+ " <td>USA</td>\n",
2922
+ " </tr>\n",
2923
+ " <tr>\n",
2924
+ " <th>...</th>\n",
2925
+ " <td>...</td>\n",
2926
+ " <td>...</td>\n",
2927
+ " <td>...</td>\n",
2928
+ " <td>...</td>\n",
2929
+ " <td>...</td>\n",
2930
+ " <td>...</td>\n",
2931
+ " </tr>\n",
2932
+ " <tr>\n",
2933
+ " <th>8237</th>\n",
2934
+ " <td>Özge Gürel</td>\n",
2935
+ " <td>[144371]</td>\n",
2936
+ " <td>1</td>\n",
2937
+ " <td>actor</td>\n",
2938
+ " <td>Female</td>\n",
2939
+ " <td>Türkiye</td>\n",
2940
+ " </tr>\n",
2941
+ " <tr>\n",
2942
+ " <th>8238</th>\n",
2943
+ " <td>Özge Özacar</td>\n",
2944
+ " <td>[975857]</td>\n",
2945
+ " <td>1</td>\n",
2946
+ " <td>actor</td>\n",
2947
+ " <td>Female</td>\n",
2948
+ " <td>Türkiye</td>\n",
2949
+ " </tr>\n",
2950
+ " <tr>\n",
2951
+ " <th>8239</th>\n",
2952
+ " <td>Özge Özberk</td>\n",
2953
+ " <td>[199114]</td>\n",
2954
+ " <td>1</td>\n",
2955
+ " <td>actor</td>\n",
2956
+ " <td>Female</td>\n",
2957
+ " <td>Türkiye</td>\n",
2958
+ " </tr>\n",
2959
+ " <tr>\n",
2960
+ " <th>8240</th>\n",
2961
+ " <td>Şener Şen</td>\n",
2962
+ " <td>[125489]</td>\n",
2963
+ " <td>1</td>\n",
2964
+ " <td>actor</td>\n",
2965
+ " <td>Male</td>\n",
2966
+ " <td>Türkiye</td>\n",
2967
+ " </tr>\n",
2968
+ " <tr>\n",
2969
+ " <th>8241</th>\n",
2970
+ " <td>Şifanur Gül</td>\n",
2971
+ " <td>[133589]</td>\n",
2972
+ " <td>1</td>\n",
2973
+ " <td>model</td>\n",
2974
+ " <td>Female</td>\n",
2975
+ " <td>Türkiye</td>\n",
2976
+ " </tr>\n",
2977
+ " </tbody>\n",
2978
+ "</table>\n",
2979
+ "<p>8242 rows × 6 columns</p>\n",
2980
+ "</div>"
2981
+ ],
2982
+ "text/plain": [
2983
+ " name model_ids \\\n",
2984
+ "0 AJ Applegate [1335260] \n",
2985
+ "1 AJ Langer [1216782] \n",
2986
+ "2 Aaliyah Dana Haughton [221782, 230769, 1191059, 1564721] \n",
2987
+ "3 Aaliyah Love [305961] \n",
2988
+ "4 Aaron Boone [437968] \n",
2989
+ "... ... ... \n",
2990
+ "8237 Özge Gürel [144371] \n",
2991
+ "8238 Özge Özacar [975857] \n",
2992
+ "8239 Özge Özberk [199114] \n",
2993
+ "8240 Şener Şen [125489] \n",
2994
+ "8241 Şifanur Gül [133589] \n",
2995
+ "\n",
2996
+ " number_of_models consensus_profession consensus_gender consensus_country \n",
2997
+ "0 1 adult performer Female USA \n",
2998
+ "1 1 actor Female USA \n",
2999
+ "2 4 singer/musician Female USA \n",
3000
+ "3 1 adult performer Female USA \n",
3001
+ "4 1 sports professional Male USA \n",
3002
+ "... ... ... ... ... \n",
3003
+ "8237 1 actor Female Türkiye \n",
3004
+ "8238 1 actor Female Türkiye \n",
3005
+ "8239 1 actor Female Türkiye \n",
3006
+ "8240 1 actor Male Türkiye \n",
3007
+ "8241 1 model Female Türkiye \n",
3008
+ "\n",
3009
+ "[8242 rows x 6 columns]"
3010
+ ]
3011
+ },
3012
+ "execution_count": 5,
3013
+ "metadata": {},
3014
+ "output_type": "execute_result"
3015
+ }
3016
+ ],
3017
+ "source": [
3018
+ "import pandas as pd\n",
3019
+ "from pathlib import Path\n",
3020
+ "from collections import Counter\n",
3021
+ "\n",
3022
+ "current_dir = Path.cwd()\n",
3023
+ "input_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement_consensus_va.csv\"\n",
3024
+ "output_file = current_dir.parent / \"data/CSV/final_aggregated.csv\"\n",
3025
+ "\n",
3026
+ "# Load the data\n",
3027
+ "df = pd.read_csv(input_file)\n",
3028
+ "\n",
3029
+ "# Function to get consensus name from the three LLM columns (at least 2 must agree)\n",
3030
+ "def get_consensus_name(row):\n",
3031
+ " names = [row['gemma_full_name'], row['mistral_full_name'], row['qwen_full_name']]\n",
3032
+ " # Filter out NaN/None values\n",
3033
+ " names = [n for n in names if pd.notna(n)]\n",
3034
+ " \n",
3035
+ " if len(names) == 0:\n",
3036
+ " return None\n",
3037
+ " \n",
3038
+ " # Count occurrences of each name\n",
3039
+ " name_counts = Counter(names)\n",
3040
+ " \n",
3041
+ " # Find names with at least 2 occurrences\n",
3042
+ " for name, count in name_counts.most_common():\n",
3043
+ " if count >= 2:\n",
3044
+ " return name\n",
3045
+ " \n",
3046
+ " # If no consensus (all 3 different), return None or first non-null\n",
3047
+ " return None\n",
3048
+ "\n",
3049
+ "# Create consensus_name column\n",
3050
+ "df['consensus_name'] = df.apply(get_consensus_name, axis=1)\n",
3051
+ "\n",
3052
+ "# Filter out rows without consensus name\n",
3053
+ "df_with_consensus = df[df['consensus_name'].notna()].copy()\n",
3054
+ "\n",
3055
+ "# Aggregate by consensus_name\n",
3056
+ "aggregated = df_with_consensus.groupby('consensus_name').agg(\n",
3057
+ " model_ids=('id', lambda x: list(x)),\n",
3058
+ " number_of_models=('id', 'count'),\n",
3059
+ " consensus_profession=('consensus_primary_profession', 'first'),\n",
3060
+ " consensus_gender=('consensus_gender', 'first'),\n",
3061
+ " consensus_country=('consensus_country', 'first')\n",
3062
+ ").reset_index()\n",
3063
+ "\n",
3064
+ "# Rename consensus_name to name for clarity\n",
3065
+ "aggregated = aggregated.rename(columns={'consensus_name': 'name'})\n",
3066
+ "\n",
3067
+ "# Reorder columns\n",
3068
+ "aggregated = aggregated[['name', 'model_ids', 'number_of_models', 'consensus_profession', 'consensus_gender', 'consensus_country']]\n",
3069
+ "\n",
3070
+ "# Save to CSV\n",
3071
+ "aggregated.to_csv(output_file, index=False)\n",
3072
+ "\n",
3073
+ "print(f\"Original rows: {len(df)}\")\n",
3074
+ "print(f\"Rows with name consensus: {len(df_with_consensus)}\")\n",
3075
+ "print(f\"Unique persons after aggregation: {len(aggregated)}\")\n",
3076
+ "print(f\"Saved to: {output_file}\")\n",
3077
+ "\n",
3078
+ "aggregated"
3079
+ ]
3080
+ },
3081
+ {
3082
+ "cell_type": "code",
3083
+ "execution_count": null,
3084
+ "id": "d1cd2440",
3085
+ "metadata": {},
3086
  "outputs": [],
3087
  "source": []
3088
  }
jupyter_notebooks/Section_2-3-4__Figure_8a_sunburst_gender.ipynb CHANGED
@@ -9,14 +9,15 @@
9
  },
10
  {
11
  "cell_type": "code",
12
- "execution_count": 1,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
- "✓ Saved 8a.json\n"
 
20
  ]
21
  }
22
  ],
@@ -24,77 +25,118 @@
24
  "import pandas as pd\n",
25
  "from collections import defaultdict\n",
26
  "import json\n",
27
- "from pathlib import Path \n",
28
  "\n",
29
  "current_dir = Path.cwd()\n",
30
- "sunburst_json = current_dir.parent / \"public/json/8a.json\"\n",
31
  "\n",
32
  "# Load consensus CSV\n",
33
- "consensus_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement_consensus.csv\"\n",
34
  "df = pd.read_csv(consensus_file)\n",
35
  "\n",
36
- "# ---- Normalize Gender (group Non-binary and Unknown into 'Other') ----\n",
 
 
 
 
37
  "def normalize_gender(g):\n",
38
- " g = str(g).strip().lower()\n",
39
- " if g in [\"female\", \"woman\", \"female (group)\", \"female (transgender)\", \"female (virtual persona)\", \"female (group members)\"]:\n",
 
 
 
 
40
  " return \"Female\"\n",
41
  " elif g in [\"male\", \"male (android)\", \"male (character)\"]:\n",
42
  " return \"Male\"\n",
43
  " else:\n",
44
  " return \"Other\"\n",
45
  "\n",
46
- "df['gender_normalized'] = df['consensus_gender'].apply(normalize_gender)\n",
47
  "\n",
48
- "# ---- Step 1: Limit to top 10 countries ----\n",
49
- "df['country_cleaned'] = df['consensus_country'].apply(lambda x: x if x not in ['Unknown', '', None] else 'Other')\n",
50
- "top_countries = df['country_cleaned'].value_counts().nlargest(12).index.tolist()\n",
51
- "df['country_limited'] = df['country_cleaned'].apply(lambda x: x if x in top_countries else 'Other')\n",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
52
  "\n",
53
- "# ---- Step 2: Normalize and limit professions ----\n",
54
- "valid_categories = [\n",
55
- " \"actor\", \"adult performer\", \"singer/musician\", \"model\",\n",
56
- " \"online personality\", \"tv personality\", \"voice actor/asmr\", \"public figure\", \"sports professional\"\n",
57
- "]\n",
58
  "\n",
59
- "def remap_profession(profession):\n",
60
- " profession_lower = str(profession).strip().lower()\n",
61
- " if profession_lower == 'unknown' or profession_lower not in valid_categories:\n",
62
- " return 'Other'\n",
63
- " elif profession_lower == 'fictional character':\n",
64
- " return 'actor'\n",
65
- " elif profession_lower in ['voice actor', 'voice actor/asmr']:\n",
66
- " return 'voice actor/ASMR'\n",
67
- " return profession_lower\n",
 
 
 
68
  "\n",
69
- "df['profession_limited'] = df['consensus_primary_profession'].apply(remap_profession)\n",
 
 
 
70
  "\n",
71
- "# ---- Step 3: Group by gender and profession ----\n",
72
- "sunburst_data = df.groupby(['gender_normalized', 'profession_limited']).size().reset_index(name='count')\n",
 
 
 
 
 
 
73
  "\n",
74
- "# ---- Step 4: Create nested structure for D3.js ----\n",
75
  "sunburst_dict = {\"name\": \"root\", \"children\": []}\n",
76
  "gender_map = defaultdict(list)\n",
77
  "\n",
78
  "for _, row in sunburst_data.iterrows():\n",
79
- " gender = row['gender_normalized']\n",
80
- " profession = row['profession_limited']\n",
81
- " count = int(row['count'])\n",
82
- " gender_map[gender].append({\"name\": profession, \"value\": count})\n",
83
- "\n",
84
- "# Sort 'Other' professions to appear last\n",
85
- "for gender, professions in gender_map.items():\n",
86
- " professions_sorted = sorted(professions, key=lambda d: (d[\"name\"] == \"Other\", d[\"name\"]))\n",
87
- " gender_map[gender] = professions_sorted\n",
88
- "\n",
89
- "# Build the final nested structure\n",
90
- "for gender, professions in gender_map.items():\n",
91
- " sunburst_dict[\"children\"].append({\"name\": gender, \"children\": professions})\n",
92
- "\n",
93
- "# ---- Step 5: Save to a JSON file ----\n",
94
- "with open(sunburst_json, \"w\", encoding='utf-8') as f:\n",
 
 
 
 
 
 
 
 
 
 
 
95
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
96
  "\n",
97
- "print(\"✓ Saved 8a.json\")\n"
98
  ]
99
  },
100
  {
 
9
  },
10
  {
11
  "cell_type": "code",
12
+ "execution_count": 3,
13
  "metadata": {},
14
  "outputs": [
15
  {
16
  "name": "stdout",
17
  "output_type": "stream",
18
  "text": [
19
+ "✓ Loaded 8242 records\n",
20
+ "✓ Saved sunburst_gender_A.json\n"
21
  ]
22
  }
23
  ],
 
25
  "import pandas as pd\n",
26
  "from collections import defaultdict\n",
27
  "import json\n",
28
+ "from pathlib import Path\n",
29
  "\n",
30
  "current_dir = Path.cwd()\n",
31
+ "sunburst_path = current_dir.parent / \"public/json/sunburst_gender_A.json\"\n",
32
  "\n",
33
  "# Load consensus CSV\n",
34
+ "consensus_file = current_dir.parent / \"data/CSV/final_aggregated.csv\"\n",
35
  "df = pd.read_csv(consensus_file)\n",
36
  "\n",
37
+ "print(f\"✓ Loaded {len(df)} records\")\n",
38
+ "\n",
39
+ "# ============================================================\n",
40
+ "# 1. NORMALIZE GENDER\n",
41
+ "# ============================================================\n",
42
  "def normalize_gender(g):\n",
43
+ " if not isinstance(g, str):\n",
44
+ " g = str(g)\n",
45
+ " g = g.strip().lower()\n",
46
+ " \n",
47
+ " if g in [\"female\", \"woman\", \"female (group)\", \"female (transgender)\", \n",
48
+ " \"female (virtual persona)\", \"female (group members)\"]:\n",
49
  " return \"Female\"\n",
50
  " elif g in [\"male\", \"male (android)\", \"male (character)\"]:\n",
51
  " return \"Male\"\n",
52
  " else:\n",
53
  " return \"Other\"\n",
54
  "\n",
55
+ "df[\"gender_normalized\"] = df[\"consensus_gender\"].apply(normalize_gender)\n",
56
  "\n",
57
+ "# ============================================================\n",
58
+ "# 2. NORMALIZE PROFESSIONS\n",
59
+ "# ============================================================\n",
60
+ "def normalize_profession(x: str):\n",
61
+ " if not isinstance(x, str) or x.strip() == \"\" or x.lower() == \"unknown\":\n",
62
+ " return \"Other\"\n",
63
+ " x = x.strip().lower()\n",
64
+ " mapping = {\n",
65
+ " \"actor\": \"Actor\",\n",
66
+ " \"model\": \"Model\",\n",
67
+ " \"adult performer\": \"Adult Performer\",\n",
68
+ " \"singer/musician\": \"Singer, Musician\",\n",
69
+ " \"online personality\": \"Online Personality\",\n",
70
+ " \"sports professional\": \"Sports Professional\",\n",
71
+ " \"voice actor/asmr\": \"Voice Actor\",\n",
72
+ " \"public figure\": \"Public Figure\",\n",
73
+ " \"tv personality\": \"Other\",\n",
74
+ " }\n",
75
+ " return mapping.get(x, \"Other\")\n",
76
  "\n",
77
+ "df[\"profession_clean\"] = df[\"consensus_profession\"].apply(normalize_profession)\n",
 
 
 
 
78
  "\n",
79
+ "# Define top professions to keep\n",
80
+ "top_prof = [\n",
81
+ " \"Actor\",\n",
82
+ " \"Model\", \n",
83
+ " \"Adult Performer\",\n",
84
+ " \"Singer, Musician\",\n",
85
+ " \"Online Personality\",\n",
86
+ " \"Sports Professional\",\n",
87
+ " \"Voice Actor\",\n",
88
+ " \"Public Figure\",\n",
89
+ " \"Other\"\n",
90
+ "]\n",
91
  "\n",
92
+ "# Re-limit professions, everything else → Other\n",
93
+ "df[\"profession_limited\"] = df[\"profession_clean\"].apply(\n",
94
+ " lambda x: x if x in top_prof else \"Other\"\n",
95
+ ")\n",
96
  "\n",
97
+ "# ============================================================\n",
98
+ "# 3. GROUP INTO SUNBURST STRUCTURE\n",
99
+ "# ============================================================\n",
100
+ "sunburst_data = (\n",
101
+ " df.groupby([\"gender_normalized\", \"profession_limited\"])\n",
102
+ " .size()\n",
103
+ " .reset_index(name=\"count\")\n",
104
+ ")\n",
105
  "\n",
 
106
  "sunburst_dict = {\"name\": \"root\", \"children\": []}\n",
107
  "gender_map = defaultdict(list)\n",
108
  "\n",
109
  "for _, row in sunburst_data.iterrows():\n",
110
+ " g = row[\"gender_normalized\"]\n",
111
+ " p = row[\"profession_limited\"]\n",
112
+ " v = int(row[\"count\"])\n",
113
+ " gender_map[g].append({\"name\": p, \"value\": v})\n",
114
+ "\n",
115
+ "# Sort professions inside each gender so \"Other\" is last\n",
116
+ "for g, profs in gender_map.items():\n",
117
+ " profs_sorted = sorted(profs, key=lambda d: (d[\"name\"] == \"Other\", d[\"name\"]))\n",
118
+ " gender_map[g] = profs_sorted\n",
119
+ "\n",
120
+ "# Calculate total datapoints per gender and sort\n",
121
+ "gender_totals = []\n",
122
+ "for g, profs in gender_map.items():\n",
123
+ " total = sum(p[\"value\"] for p in profs)\n",
124
+ " gender_totals.append((g, total, profs))\n",
125
+ "\n",
126
+ "# Sort by total (descending), but put \"Other\" last\n",
127
+ "gender_totals.sort(key=lambda x: (x[0] == \"Other\", -x[1]))\n",
128
+ "\n",
129
+ "# Build final JSON with sorted genders\n",
130
+ "for g, total, profs in gender_totals:\n",
131
+ " sunburst_dict[\"children\"].append({\"name\": g, \"children\": profs})\n",
132
+ "\n",
133
+ "# ============================================================\n",
134
+ "# 4. SAVE JSON\n",
135
+ "# ============================================================\n",
136
+ "with open(sunburst_path, \"w\", encoding=\"utf-8\") as f:\n",
137
  " json.dump(sunburst_dict, f, ensure_ascii=False, indent=2)\n",
138
  "\n",
139
+ "print(\"✓ Saved sunburst_gender_A.json\")"
140
  ]
141
  },
142
  {
jupyter_notebooks/Section_2-3-4__Figure_8b_sunburst_profession.ipynb CHANGED
@@ -9,7 +9,7 @@
9
  },
10
  {
11
  "cell_type": "code",
12
- "execution_count": 9,
13
  "metadata": {},
14
  "outputs": [
15
  {
@@ -31,7 +31,7 @@
31
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
32
  "\n",
33
  "# Load consensus CSV\n",
34
- "consensus_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement_consensus_va.csv\"\n",
35
  "df = pd.read_csv(consensus_file)\n",
36
  "\n",
37
  "# ============================================================\n",
@@ -84,7 +84,7 @@
84
  " }\n",
85
  " return mapping.get(x, \"Other\")\n",
86
  "\n",
87
- "df[\"profession_clean\"] = df[\"consensus_primary_profession\"].apply(normalize_profession)\n",
88
  "\n",
89
  "\n",
90
  "df[\"profession_limited\"] = df[\"profession_clean\"]\n",
@@ -147,15 +147,28 @@
147
  },
148
  {
149
  "cell_type": "code",
150
- "execution_count": 10,
151
  "metadata": {},
152
  "outputs": [
153
  {
154
- "name": "stdout",
155
- "output_type": "stream",
156
- "text": [
157
- "✓ Filtered to 16059 records published on or before December 31, 2024\n",
158
- "✓ Saved sunburst_countries_A.json (2024 data only)\n"
 
 
 
 
 
 
 
 
 
 
 
 
 
159
  ]
160
  }
161
  ],
@@ -169,7 +182,7 @@
169
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
170
  "\n",
171
  "# Load consensus CSV\n",
172
- "consensus_file = current_dir.parent / \"data/CSV/analyzed_llm_agreement_consensus_va.csv\"\n",
173
  "df = pd.read_csv(consensus_file)\n",
174
  "\n",
175
  "# ============================================================\n",
 
9
  },
10
  {
11
  "cell_type": "code",
12
+ "execution_count": 4,
13
  "metadata": {},
14
  "outputs": [
15
  {
 
31
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
32
  "\n",
33
  "# Load consensus CSV\n",
34
+ "consensus_file = current_dir.parent / \"data/CSV/final_aggregated.csv\"\n",
35
  "df = pd.read_csv(consensus_file)\n",
36
  "\n",
37
  "# ============================================================\n",
 
84
  " }\n",
85
  " return mapping.get(x, \"Other\")\n",
86
  "\n",
87
+ "df[\"profession_clean\"] = df[\"consensus_profession\"].apply(normalize_profession)\n",
88
  "\n",
89
  "\n",
90
  "df[\"profession_limited\"] = df[\"profession_clean\"]\n",
 
147
  },
148
  {
149
  "cell_type": "code",
150
+ "execution_count": 2,
151
  "metadata": {},
152
  "outputs": [
153
  {
154
+ "ename": "KeyError",
155
+ "evalue": "'publishedAt'",
156
+ "output_type": "error",
157
+ "traceback": [
158
+ "\u001b[0;31m---------------------------------------------------------------------------\u001b[0m",
159
+ "\u001b[0;31mKeyError\u001b[0m Traceback (most recent call last)",
160
+ "File \u001b[0;32m~/anaconda3/envs/latm/lib/python3.10/site-packages/pandas/core/indexes/base.py:3805\u001b[0m, in \u001b[0;36mIndex.get_loc\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m 3804\u001b[0m \u001b[38;5;28;01mtry\u001b[39;00m:\n\u001b[0;32m-> 3805\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43m_engine\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mget_loc\u001b[49m\u001b[43m(\u001b[49m\u001b[43mcasted_key\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m 3806\u001b[0m \u001b[38;5;28;01mexcept\u001b[39;00m \u001b[38;5;167;01mKeyError\u001b[39;00m \u001b[38;5;28;01mas\u001b[39;00m err:\n",
161
+ "File \u001b[0;32mindex.pyx:167\u001b[0m, in \u001b[0;36mpandas._libs.index.IndexEngine.get_loc\u001b[0;34m()\u001b[0m\n",
162
+ "File \u001b[0;32mindex.pyx:196\u001b[0m, in \u001b[0;36mpandas._libs.index.IndexEngine.get_loc\u001b[0;34m()\u001b[0m\n",
163
+ "File \u001b[0;32mpandas/_libs/hashtable_class_helper.pxi:7081\u001b[0m, in \u001b[0;36mpandas._libs.hashtable.PyObjectHashTable.get_item\u001b[0;34m()\u001b[0m\n",
164
+ "File \u001b[0;32mpandas/_libs/hashtable_class_helper.pxi:7089\u001b[0m, in \u001b[0;36mpandas._libs.hashtable.PyObjectHashTable.get_item\u001b[0;34m()\u001b[0m\n",
165
+ "\u001b[0;31mKeyError\u001b[0m: 'publishedAt'",
166
+ "\nThe above exception was the direct cause of the following exception:\n",
167
+ "\u001b[0;31mKeyError\u001b[0m Traceback (most recent call last)",
168
+ "Cell \u001b[0;32mIn[2], line 17\u001b[0m\n\u001b[1;32m 11\u001b[0m df \u001b[38;5;241m=\u001b[39m pd\u001b[38;5;241m.\u001b[39mread_csv(consensus_file)\n\u001b[1;32m 13\u001b[0m \u001b[38;5;66;03m# ============================================================\u001b[39;00m\n\u001b[1;32m 14\u001b[0m \u001b[38;5;66;03m# FILTER DATA UP TO DECEMBER 31, 2024\u001b[39;00m\n\u001b[1;32m 15\u001b[0m \u001b[38;5;66;03m# ============================================================\u001b[39;00m\n\u001b[1;32m 16\u001b[0m \u001b[38;5;66;03m# Convert publishedAt to datetime\u001b[39;00m\n\u001b[0;32m---> 17\u001b[0m df[\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mpublishedAt\u001b[39m\u001b[38;5;124m\"\u001b[39m] \u001b[38;5;241m=\u001b[39m pd\u001b[38;5;241m.\u001b[39mto_datetime(\u001b[43mdf\u001b[49m\u001b[43m[\u001b[49m\u001b[38;5;124;43m\"\u001b[39;49m\u001b[38;5;124;43mpublishedAt\u001b[39;49m\u001b[38;5;124;43m\"\u001b[39;49m\u001b[43m]\u001b[49m, errors\u001b[38;5;241m=\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mcoerce\u001b[39m\u001b[38;5;124m\"\u001b[39m, utc\u001b[38;5;241m=\u001b[39m\u001b[38;5;28;01mTrue\u001b[39;00m)\n\u001b[1;32m 19\u001b[0m \u001b[38;5;66;03m# Filter to only include data up to December 31, 2024\u001b[39;00m\n\u001b[1;32m 20\u001b[0m \u001b[38;5;66;03m# Make cutoff_date timezone-aware (UTC) to match publishedAt\u001b[39;00m\n\u001b[1;32m 21\u001b[0m cutoff_date \u001b[38;5;241m=\u001b[39m pd\u001b[38;5;241m.\u001b[39mTimestamp(\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124m2024-12-31 23:59:59\u001b[39m\u001b[38;5;124m\"\u001b[39m, tz\u001b[38;5;241m=\u001b[39m\u001b[38;5;124m\"\u001b[39m\u001b[38;5;124mUTC\u001b[39m\u001b[38;5;124m\"\u001b[39m)\n",
169
+ "File \u001b[0;32m~/anaconda3/envs/latm/lib/python3.10/site-packages/pandas/core/frame.py:4102\u001b[0m, in \u001b[0;36mDataFrame.__getitem__\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m 4100\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39mcolumns\u001b[38;5;241m.\u001b[39mnlevels \u001b[38;5;241m>\u001b[39m \u001b[38;5;241m1\u001b[39m:\n\u001b[1;32m 4101\u001b[0m \u001b[38;5;28;01mreturn\u001b[39;00m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_getitem_multilevel(key)\n\u001b[0;32m-> 4102\u001b[0m indexer \u001b[38;5;241m=\u001b[39m \u001b[38;5;28;43mself\u001b[39;49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mcolumns\u001b[49m\u001b[38;5;241;43m.\u001b[39;49m\u001b[43mget_loc\u001b[49m\u001b[43m(\u001b[49m\u001b[43mkey\u001b[49m\u001b[43m)\u001b[49m\n\u001b[1;32m 4103\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m is_integer(indexer):\n\u001b[1;32m 4104\u001b[0m indexer \u001b[38;5;241m=\u001b[39m [indexer]\n",
170
+ "File \u001b[0;32m~/anaconda3/envs/latm/lib/python3.10/site-packages/pandas/core/indexes/base.py:3812\u001b[0m, in \u001b[0;36mIndex.get_loc\u001b[0;34m(self, key)\u001b[0m\n\u001b[1;32m 3807\u001b[0m \u001b[38;5;28;01mif\u001b[39;00m \u001b[38;5;28misinstance\u001b[39m(casted_key, \u001b[38;5;28mslice\u001b[39m) \u001b[38;5;129;01mor\u001b[39;00m (\n\u001b[1;32m 3808\u001b[0m \u001b[38;5;28misinstance\u001b[39m(casted_key, abc\u001b[38;5;241m.\u001b[39mIterable)\n\u001b[1;32m 3809\u001b[0m \u001b[38;5;129;01mand\u001b[39;00m \u001b[38;5;28many\u001b[39m(\u001b[38;5;28misinstance\u001b[39m(x, \u001b[38;5;28mslice\u001b[39m) \u001b[38;5;28;01mfor\u001b[39;00m x \u001b[38;5;129;01min\u001b[39;00m casted_key)\n\u001b[1;32m 3810\u001b[0m ):\n\u001b[1;32m 3811\u001b[0m \u001b[38;5;28;01mraise\u001b[39;00m InvalidIndexError(key)\n\u001b[0;32m-> 3812\u001b[0m \u001b[38;5;28;01mraise\u001b[39;00m \u001b[38;5;167;01mKeyError\u001b[39;00m(key) \u001b[38;5;28;01mfrom\u001b[39;00m \u001b[38;5;21;01merr\u001b[39;00m\n\u001b[1;32m 3813\u001b[0m \u001b[38;5;28;01mexcept\u001b[39;00m \u001b[38;5;167;01mTypeError\u001b[39;00m:\n\u001b[1;32m 3814\u001b[0m \u001b[38;5;66;03m# If we have a listlike key, _check_indexing_error will raise\u001b[39;00m\n\u001b[1;32m 3815\u001b[0m \u001b[38;5;66;03m# InvalidIndexError. Otherwise we fall through and re-raise\u001b[39;00m\n\u001b[1;32m 3816\u001b[0m \u001b[38;5;66;03m# the TypeError.\u001b[39;00m\n\u001b[1;32m 3817\u001b[0m \u001b[38;5;28mself\u001b[39m\u001b[38;5;241m.\u001b[39m_check_indexing_error(key)\n",
171
+ "\u001b[0;31mKeyError\u001b[0m: 'publishedAt'"
172
  ]
173
  }
174
  ],
 
182
  "sunburst_path = current_dir.parent / \"public/json/sunburst_countries_A.json\"\n",
183
  "\n",
184
  "# Load consensus CSV\n",
185
+ "consensus_file = current_dir.parent / \"data/CSV/final_aggregated.csv\"\n",
186
  "df = pd.read_csv(consensus_file)\n",
187
  "\n",
188
  "# ============================================================\n",
public/Figure_13_interactive.html ADDED
@@ -0,0 +1,403 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ <!DOCTYPE html>
2
+ <html lang="en" class="light-mode">
3
+ <head>
4
+ <meta charset="UTF-8">
5
+ <title>Danbooru Tag Taxonomy</title>
6
+ <script src="https://d3js.org/d3.v7.min.js"></script>
7
+ <link rel="stylesheet" href="styles.css">
8
+ <style>
9
+ /* Tree-specific styles (not in shared styles.css) */
10
+ html, body { overflow: auto; }
11
+ svg { position: static; margin-top: 0; z-index: auto; }
12
+ .tree-svg {
13
+ width: 100%;
14
+ overflow: visible;
15
+ display: block;
16
+ }
17
+ .link {
18
+ fill: none;
19
+ stroke: #ccc;
20
+ stroke-width: 1.5px;
21
+ }
22
+ .node circle {
23
+ stroke: #555;
24
+ stroke-width: 1.5px;
25
+ cursor: pointer;
26
+ }
27
+ .node.leaf circle {
28
+ cursor: default;
29
+ }
30
+ .node text {
31
+ font-size: 12px;
32
+ fill: #333;
33
+ pointer-events: none;
34
+ }
35
+ .node.leaf text {
36
+ font-size: 10px;
37
+ fill: #666;
38
+ }
39
+ .node:hover circle {
40
+ stroke: #000;
41
+ stroke-width: 2.5px;
42
+ }
43
+ .tree-controls {
44
+ margin: 0.5em 0 1em 0;
45
+ }
46
+ .tree-controls button {
47
+ margin-right: 6px;
48
+ padding: 6px 14px;
49
+ cursor: pointer;
50
+ background-color: rgba(138, 43, 226, 0.7);
51
+ color: white;
52
+ border: none;
53
+ border-radius: 4px;
54
+ transition: background-color 0.3s;
55
+ }
56
+ .tree-controls button:hover {
57
+ background-color: rgba(138, 43, 226, 0.9);
58
+ }
59
+ </style>
60
+ </head>
61
+ <body>
62
+ <!-- Header Overlay -->
63
+ <div class="header-overlay">
64
+ <div class="visualization-container">
65
+ <h1>Danbooru Tag Taxonomy</h1>
66
+ <h2 class="caption">Interactive hierarchical tree of Danbooru's tag classification system (~39,000 tags across 200 categories)</h2>
67
+
68
+ <div class="nav-buttons"></div>
69
+ <button class="nav-button" onclick="location.href='index.html'">Home</button>
70
+ <button class="nav-button" onclick="location.href='figure_15.html'">Next: Textual Training Data</button>
71
+
72
+ <div class="tree-controls">
73
+ <button id="expandAllBtn">Expand All Categories</button>
74
+ <button id="collapseAllBtn">Collapse All</button>
75
+ <button id="downloadBtn">Download SVG</button>
76
+ </div>
77
+ </div>
78
+ </div>
79
+
80
+ <svg class="tree-svg"></svg>
81
+
82
+ <script>
83
+ // Default to light mode
84
+ document.documentElement.classList.add('light-mode');
85
+
86
+ d3.json("json/danbooru_flat_full.json").then(function(data) {
87
+
88
+ // ---- CONFIG ----
89
+ const margin = { top: 20, right: 250, bottom: 20, left: 80 };
90
+ const nodeWidth = 220;
91
+ const duration = 400;
92
+
93
+ const svg = d3.select("svg.tree-svg");
94
+ const g = svg.append("g")
95
+ .attr("transform", `translate(${margin.left},${margin.top})`);
96
+
97
+ const linkGroup = g.append("g").attr("class", "links");
98
+ const nodeGroup = g.append("g").attr("class", "nodes");
99
+
100
+ // ---- STEP 0: Distinct color per top-level category ----
101
+ const categoryColors = {
102
+ "attire_and_body_accessories": "#e6194b",
103
+ "body": "#f58231",
104
+ "characters": "#3cb44b",
105
+ "copyrights_artists_projects_and_media": "#4363d8",
106
+ "creatures": "#42d4f4",
107
+ "drawing_software": "#911eb4",
108
+ "games": "#f032e6",
109
+ "metatags": "#bfef45",
110
+ "more": "#9A6324",
111
+ "objects": "#e6beff",
112
+ "plant": "#006400",
113
+ "real_world": "#ffd8b1",
114
+ "sex": "#ff69b4",
115
+ "visual_characteristics": "#000075",
116
+ "subject": "#ffe119",
117
+ "uncategorized": "#a9a9a9",
118
+ "actions_and_expressions": "#800000",
119
+ "objects_and_backgrounds": "#469990",
120
+ };
121
+
122
+ function assignColors(node, inheritedColor) {
123
+ if (inheritedColor) node.color = inheritedColor;
124
+ if (node.children) node.children.forEach(child => assignColors(child, node.color));
125
+ }
126
+ if (data.children) {
127
+ data.children.forEach(child => {
128
+ child.color = categoryColors[child.name] || "#888";
129
+ assignColors(child, child.color);
130
+ });
131
+ }
132
+
133
+ // ---- Pre-process: sort + convert tags to leaf nodes ----
134
+ function preprocess(node) {
135
+ if (node.children && node.children.length > 0) {
136
+ node.children.forEach(preprocess);
137
+ node.children.sort((a, b) => (b.tag_count || 0) - (a.tag_count || 0));
138
+ }
139
+ else if (node.tags && node.tags.length > 0) {
140
+ node.children = node.tags
141
+ .filter(t => t !== "...")
142
+ .map(tag => ({
143
+ name: tag,
144
+ _isTag: true,
145
+ color: node.color,
146
+ tag_count: 0,
147
+ category_count: 0
148
+ }));
149
+ node.children.sort((a, b) => a.name.localeCompare(b.name));
150
+ }
151
+ return node;
152
+ }
153
+ preprocess(data);
154
+
155
+ function getColor(d) {
156
+ return d.data.color || "#888";
157
+ }
158
+
159
+ // ---- Build hierarchy ----
160
+ const root = d3.hierarchy(data, d => d.children);
161
+ root.x0 = 0;
162
+ root.y0 = 0;
163
+
164
+ let idCounter = 0;
165
+ root.each(d => { d.id = ++idCounter; });
166
+
167
+ // ---- Size scale for tag_count → radius ----
168
+ const allTagCounts = root.descendants()
169
+ .filter(d => d.depth > 0 && !d.data._isTag)
170
+ .map(d => d.data.tag_count || 0);
171
+ const sizeScale = d3.scaleSqrt()
172
+ .domain([0, d3.max(allTagCounts)])
173
+ .range([4, 18]);
174
+
175
+ function nodeRadius(d) {
176
+ if (d.data._isTag) return 2.5;
177
+ if (d.depth === 0) return 8;
178
+ return sizeScale(d.data.tag_count || 0);
179
+ }
180
+
181
+ function collapse(d) {
182
+ if (d.children) {
183
+ d._children = d.children;
184
+ d._children.forEach(collapse);
185
+ d.children = null;
186
+ }
187
+ }
188
+
189
+ function expandCategories(d) {
190
+ if (d._children) {
191
+ if (!d._children[0].data._isTag) {
192
+ d.children = d._children;
193
+ d._children = null;
194
+ }
195
+ }
196
+ if (d.children) d.children.forEach(expandCategories);
197
+ }
198
+
199
+ function expandAll(d) {
200
+ if (d._children) {
201
+ d.children = d._children;
202
+ d._children = null;
203
+ }
204
+ if (d.children) d.children.forEach(expandAll);
205
+ }
206
+
207
+ // Collapse everything below root initially
208
+ if (root.children) {
209
+ root.children.forEach(collapse);
210
+ }
211
+
212
+ function computeTreeLayout() {
213
+ const leaves = root.descendants().filter(d =>
214
+ !d.children || d.children.length === 0
215
+ );
216
+ const visibleCount = leaves.length;
217
+ const rowHeight = 18;
218
+ const treeHeight = Math.max(300, visibleCount * rowHeight);
219
+
220
+ const treeLayout = d3.tree()
221
+ .size([treeHeight, 0])
222
+ .separation((a, b) => {
223
+ if (a.data._isTag && b.data._isTag) return 0.6;
224
+ return a.parent === b.parent ? 1 : 1.3;
225
+ });
226
+
227
+ treeLayout(root);
228
+
229
+ root.each(d => { d.y = d.depth * nodeWidth; });
230
+
231
+ const maxDepth = d3.max(root.descendants(), d => d.depth) || 0;
232
+ const totalWidth = margin.left + margin.right + (maxDepth + 1) * nodeWidth;
233
+ const totalHeight = margin.top + margin.bottom + treeHeight;
234
+ svg.attr("width", totalWidth)
235
+ .attr("height", totalHeight)
236
+ .attr("viewBox", `0 0 ${totalWidth} ${totalHeight}`);
237
+ }
238
+
239
+ // ---- MAIN UPDATE FUNCTION ----
240
+ function update(source) {
241
+ computeTreeLayout();
242
+
243
+ const nodes = root.descendants();
244
+ const links = root.links();
245
+
246
+ // Links
247
+ const link = linkGroup.selectAll("path.link")
248
+ .data(links, d => d.target.id);
249
+
250
+ const linkEnter = link.enter()
251
+ .append("path")
252
+ .attr("class", "link")
253
+ .attr("d", () => {
254
+ const o = { x: source.x0, y: source.y0 };
255
+ return diagonal({ source: o, target: o });
256
+ });
257
+
258
+ linkEnter.merge(link).transition().duration(duration)
259
+ .attr("d", d => diagonal(d))
260
+ .attr("stroke", d => d.target.data.color || "#ccc")
261
+ .attr("stroke-opacity", 0.5);
262
+
263
+ link.exit().transition().duration(duration)
264
+ .attr("d", () => {
265
+ const o = { x: source.x, y: source.y };
266
+ return diagonal({ source: o, target: o });
267
+ })
268
+ .remove();
269
+
270
+ // Nodes
271
+ const node = nodeGroup.selectAll("g.node")
272
+ .data(nodes, d => d.id);
273
+
274
+ const nodeEnter = node.enter()
275
+ .append("g")
276
+ .attr("class", d => d.data._isTag ? "node leaf" : "node")
277
+ .attr("transform", () => `translate(${source.y0},${source.x0})`)
278
+ .on("click", (event, d) => {
279
+ if (d.data._isTag) return;
280
+ if (d.children) {
281
+ d._children = d.children;
282
+ d.children = null;
283
+ } else if (d._children) {
284
+ d.children = d._children;
285
+ d._children = null;
286
+ }
287
+ update(d);
288
+ });
289
+
290
+ nodeEnter.append("circle").attr("r", 1e-6);
291
+
292
+ nodeEnter.append("text")
293
+ .attr("dy", "0.32em")
294
+ .attr("x", d => nodeRadius(d) + 4)
295
+ .text(d => {
296
+ if (d.data._isTag) return d.data.name;
297
+ const count = d.data.tag_count || 0;
298
+ return count > 0 ? `${d.data.name} (${count})` : d.data.name;
299
+ });
300
+
301
+ nodeEnter.append("title");
302
+
303
+ const nodeUpdate = nodeEnter.merge(node);
304
+
305
+ nodeUpdate.transition().duration(duration)
306
+ .attr("transform", d => `translate(${d.y},${d.x})`);
307
+
308
+ nodeUpdate.select("circle")
309
+ .attr("r", d => nodeRadius(d))
310
+ .style("fill", d => {
311
+ if (d.data._isTag) return getColor(d);
312
+ return (d._children && d._children.length) ? getColor(d) : "#fff";
313
+ })
314
+ .style("stroke", d => getColor(d));
315
+
316
+ nodeUpdate.select("text")
317
+ .style("fill-opacity", 1)
318
+ .style("font-weight", d => {
319
+ if (d.data._isTag) return "normal";
320
+ return d.depth <= 1 ? "bold" : "normal";
321
+ })
322
+ .style("font-size", d => d.data._isTag ? "10px" : "12px")
323
+ .style("fill", d => d.data._isTag ? "#666" : "#333");
324
+
325
+ nodeUpdate.select("title")
326
+ .text(d => {
327
+ if (d.data._isTag) return d.data.name;
328
+ const parts = [d.data.name];
329
+ if (d.data.tag_count) parts.push(`Tags: ${d.data.tag_count}`);
330
+ if (d.data.category_count) parts.push(`Sub-categories: ${d.data.category_count}`);
331
+ const hidden = d._children ? d._children.length : 0;
332
+ if (hidden) {
333
+ const type = d._children[0].data._isTag ? "tags" : "children";
334
+ parts.push(`Click to expand (${hidden} ${type})`);
335
+ } else if (d.children && d.children.length) {
336
+ parts.push("Click to collapse");
337
+ }
338
+ return parts.join("\n");
339
+ });
340
+
341
+ const nodeExit = node.exit().transition().duration(duration)
342
+ .attr("transform", () => `translate(${source.y},${source.x})`)
343
+ .remove();
344
+
345
+ nodeExit.select("circle").attr("r", 1e-6);
346
+ nodeExit.select("text").style("fill-opacity", 1e-6);
347
+
348
+ nodes.forEach(d => { d.x0 = d.x; d.y0 = d.y; });
349
+ }
350
+
351
+ function diagonal(d) {
352
+ return `M${d.source.y},${d.source.x}
353
+ C${(d.source.y + d.target.y) / 2},${d.source.x}
354
+ ${(d.source.y + d.target.y) / 2},${d.target.x}
355
+ ${d.target.y},${d.target.x}`;
356
+ }
357
+
358
+ // Initial render
359
+ update(root);
360
+
361
+ // Expand All (categories only)
362
+ document.getElementById("expandAllBtn").addEventListener("click", () => {
363
+ expandCategories(root);
364
+ update(root);
365
+ });
366
+
367
+ // Collapse All
368
+ document.getElementById("collapseAllBtn").addEventListener("click", () => {
369
+ if (root.children) root.children.forEach(collapse);
370
+ update(root);
371
+ });
372
+
373
+ // Download SVG
374
+ document.getElementById("downloadBtn").addEventListener("click", () => {
375
+ const svgNode = document.querySelector("svg");
376
+ const clonedSvg = svgNode.cloneNode(true);
377
+ clonedSvg.setAttribute("xmlns", "http://www.w3.org/2000/svg");
378
+
379
+ const styleEl = document.createElement("style");
380
+ styleEl.textContent = `
381
+ .link { fill: none; stroke: #ccc; stroke-width: 1.5px; }
382
+ .node circle { stroke-width: 1.5px; }
383
+ .node text { font-size: 12px; fill: #333; font-family: sans-serif; }
384
+ .node.leaf text { font-size: 10px; fill: #666; }
385
+ `;
386
+ clonedSvg.insertBefore(styleEl, clonedSvg.firstChild);
387
+
388
+ const svgData = new XMLSerializer().serializeToString(clonedSvg);
389
+ const svgBlob = new Blob([svgData], { type: "image/svg+xml;charset=utf-8" });
390
+ const url = URL.createObjectURL(svgBlob);
391
+ const a = document.createElement("a");
392
+ a.href = url;
393
+ a.download = "danbooru_tree_interactive.svg";
394
+ document.body.appendChild(a);
395
+ a.click();
396
+ document.body.removeChild(a);
397
+ URL.revokeObjectURL(url);
398
+ });
399
+
400
+ });
401
+ </script>
402
+ </body>
403
+ </html>
public/Figure_8a_barchart.html CHANGED
@@ -135,16 +135,30 @@ d3.json("json/sunburst_gender_A.json").then(data => {
135
  .attr("transform", `translate(${margin.left},${margin.top})`);
136
 
137
  /* --------------------------------------------------------
138
- 5. STACKED DATA
139
  -------------------------------------------------------- */
140
- const stack = d3.stack()
141
- .keys(professionOrder)
142
- .value((d, key) => {
143
- const prof = d[1].professions.find(p => p.profession === key);
144
- return prof ? prof.value : 0;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
145
  });
146
-
147
- const series = stack(Array.from(genderData));
148
 
149
  /* --------------------------------------------------------
150
  6. AXES
@@ -178,29 +192,22 @@ d3.json("json/sunburst_gender_A.json").then(data => {
178
  .text("Count");
179
 
180
  /* --------------------------------------------------------
181
- 7. DRAW BARS
182
  -------------------------------------------------------- */
183
- svg.append("g")
184
- .selectAll("g")
185
- .data(series)
186
- .join("g")
187
- .attr("fill", d => colorMap[d.key])
188
- .selectAll("rect")
189
- .data(d => d)
190
  .join("rect")
191
  .attr("class", "bar")
192
- .attr("x", d => x(d.data[0]))
193
- .attr("y", d => y(d[1]))
194
- .attr("height", d => y(d[0]) - y(d[1]))
195
  .attr("width", x.bandwidth())
 
196
  .append("title")
197
- .text(d => {
198
- const profKey = series.find(s => s.includes(d))?.key;
199
- return `${d.data[0]} – ${profKey}: ${d[1] - d[0]}`;
200
- });
201
 
202
  /* --------------------------------------------------------
203
- 8. LEGEND (ordered by occurrence)
204
  -------------------------------------------------------- */
205
  const legend = svg.append("g")
206
  .attr("transform", `translate(${width + 20}, 0)`);
 
135
  .attr("transform", `translate(${margin.left},${margin.top})`);
136
 
137
  /* --------------------------------------------------------
138
+ 5. PREPARE DATA WITH SORTED SEGMENTS PER GENDER
139
  -------------------------------------------------------- */
140
+ // Build stacked data manually with segments sorted by size per gender
141
+ const stackedData = [];
142
+
143
+ genders.forEach(gender => {
144
+ const genderInfo = genderData.get(gender);
145
+ const professions = genderInfo.professions;
146
+
147
+ // Sort professions by value descending (largest at bottom)
148
+ const sortedProfs = [...professions].sort((a, b) => b.value - a.value);
149
+
150
+ let y0 = 0;
151
+ sortedProfs.forEach(prof => {
152
+ stackedData.push({
153
+ gender: gender,
154
+ profession: prof.profession,
155
+ value: prof.value,
156
+ y0: y0,
157
+ y1: y0 + prof.value
158
+ });
159
+ y0 += prof.value;
160
  });
161
+ });
 
162
 
163
  /* --------------------------------------------------------
164
  6. AXES
 
192
  .text("Count");
193
 
194
  /* --------------------------------------------------------
195
+ 7. DRAW BARS (using manually stacked data)
196
  -------------------------------------------------------- */
197
+ svg.selectAll(".bar")
198
+ .data(stackedData)
 
 
 
 
 
199
  .join("rect")
200
  .attr("class", "bar")
201
+ .attr("x", d => x(d.gender))
202
+ .attr("y", d => y(d.y1))
203
+ .attr("height", d => y(d.y0) - y(d.y1))
204
  .attr("width", x.bandwidth())
205
+ .attr("fill", d => colorMap[d.profession])
206
  .append("title")
207
+ .text(d => `${d.gender} – ${d.profession}: ${d.value}`);
 
 
 
208
 
209
  /* --------------------------------------------------------
210
+ 8. LEGEND (ordered by occurrence - unchanged)
211
  -------------------------------------------------------- */
212
  const legend = svg.append("g")
213
  .attr("transform", `translate(${width + 20}, 0)`);
public/Figure_8b_barchart.html CHANGED
@@ -62,7 +62,7 @@ const colorMap = {
62
  };
63
 
64
  /* --------------------------------------------------------
65
- 2. LOAD JSON + TRANSFORM IT
66
  -------------------------------------------------------- */
67
  d3.json("json/sunburst_countries_A.json").then(data => {
68
 
@@ -81,25 +81,25 @@ d3.json("json/sunburst_countries_A.json").then(data => {
81
  });
82
 
83
  /* --------------------------------------------------------
84
- 3. CALCULATE PROFESSION ORDER (Other last)
85
  -------------------------------------------------------- */
86
  const professionCounts = {};
87
  flatData.forEach(d => {
88
  professionCounts[d.profession] = (professionCounts[d.profession] || 0) + d.value;
89
  });
90
 
 
91
  const professionOrder = Object.entries(professionCounts)
92
  .filter(entry => entry[0] !== "Other")
93
  .sort((a, b) => b[1] - a[1])
94
  .map(entry => entry[0]);
95
-
 
96
  if (professionCounts["Other"]) {
97
  professionOrder.push("Other");
98
  }
99
 
100
- /* --------------------------------------------------------
101
- 4. GROUP DATA BY COUNTRY
102
- -------------------------------------------------------- */
103
  const countryData = d3.rollup(
104
  flatData,
105
  v => ({
@@ -110,22 +110,14 @@ d3.json("json/sunburst_countries_A.json").then(data => {
110
  );
111
 
112
  /* --------------------------------------------------------
113
- 5. SORT COUNTRIES BY TOTAL — but put "Other" last
114
  -------------------------------------------------------- */
115
- let countries = Array.from(countryData.entries())
116
  .sort((a, b) => b[1].total - a[1].total)
117
  .map(d => d[0]);
118
 
119
- // Move country named "Other" to the final position
120
- countries = countries.filter(c => c !== "Other");
121
- if (countryData.has("Other")) {
122
- countries.push("Other");
123
- }
124
-
125
- const topCountries = countries.slice(0, 25);
126
-
127
  /* --------------------------------------------------------
128
- 6. SVG SETUP
129
  -------------------------------------------------------- */
130
  const margin = {top: 40, right: 200, bottom: 150, left: 80};
131
  const width = Math.max(800, window.innerWidth - 100) - margin.left - margin.right;
@@ -138,19 +130,35 @@ d3.json("json/sunburst_countries_A.json").then(data => {
138
  .attr("transform", `translate(${margin.left},${margin.top})`);
139
 
140
  /* --------------------------------------------------------
141
- 7. STACK LAYOUT
142
  -------------------------------------------------------- */
143
- const stack = d3.stack()
144
- .keys(professionOrder)
145
- .value((d, key) => {
146
- const prof = d[1].professions.find(p => p.profession === key);
147
- return prof ? prof.value : 0;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
148
  });
149
-
150
- const series = stack(Array.from(countryData));
151
 
152
  /* --------------------------------------------------------
153
- 8. AXES
154
  -------------------------------------------------------- */
155
  const x = d3.scaleBand()
156
  .domain(topCountries)
@@ -183,29 +191,22 @@ d3.json("json/sunburst_countries_A.json").then(data => {
183
  .text("Count");
184
 
185
  /* --------------------------------------------------------
186
- 9. DRAW STACKED BARS
187
  -------------------------------------------------------- */
188
- svg.append("g")
189
- .selectAll("g")
190
- .data(series)
191
- .join("g")
192
- .attr("fill", d => colorMap[d.key])
193
- .selectAll("rect")
194
- .data(d => d)
195
  .join("rect")
196
  .attr("class", "bar")
197
- .attr("x", d => x(d.data[0]))
198
- .attr("y", d => y(d[1]))
199
- .attr("height", d => y(d[0]) - y(d[1]))
200
  .attr("width", x.bandwidth())
 
201
  .append("title")
202
- .text(d => {
203
- const profKey = series.find(s => s.includes(d))?.key;
204
- return `${d.data[0]} – ${profKey}: ${d[1] - d[0]}`;
205
- });
206
 
207
  /* --------------------------------------------------------
208
- 10. LEGEND
209
  -------------------------------------------------------- */
210
  const legend = svg.append("g")
211
  .attr("transform", `translate(${width + 20}, 0)`);
@@ -215,18 +216,18 @@ d3.json("json/sunburst_countries_A.json").then(data => {
215
 
216
  row.append("rect")
217
  .attr("width", 18)
218
- .attr("height": 18)
219
- .attr("fill": colorMap[prof]);
220
 
221
  row.append("text")
222
- .attr("x": 24)
223
- .attr("y": 9)
224
- .attr("dy": "0.35em")
225
  .text(`${prof} (${professionCounts[prof] || 0})`);
226
  });
227
 
228
  /* --------------------------------------------------------
229
- 11. DOWNLOAD SVG BUTTON
230
  -------------------------------------------------------- */
231
  document.getElementById("downloadBtn").addEventListener("click", () => {
232
  const svgNode = document.querySelector("#chart");
@@ -254,6 +255,5 @@ d3.json("json/sunburst_countries_A.json").then(data => {
254
  });
255
  </script>
256
 
257
-
258
  </body>
259
  </html>
 
62
  };
63
 
64
  /* --------------------------------------------------------
65
+ 3. LOAD JSON + TRANSFORM IT
66
  -------------------------------------------------------- */
67
  d3.json("json/sunburst_countries_A.json").then(data => {
68
 
 
81
  });
82
 
83
  /* --------------------------------------------------------
84
+ 3a. CALCULATE PROFESSION ORDER BY TOTAL OCCURRENCE
85
  -------------------------------------------------------- */
86
  const professionCounts = {};
87
  flatData.forEach(d => {
88
  professionCounts[d.profession] = (professionCounts[d.profession] || 0) + d.value;
89
  });
90
 
91
+ // Sort professions by count (descending), but always put "Other" last
92
  const professionOrder = Object.entries(professionCounts)
93
  .filter(entry => entry[0] !== "Other")
94
  .sort((a, b) => b[1] - a[1])
95
  .map(entry => entry[0]);
96
+
97
+ // Add "Other" at the end if it exists
98
  if (professionCounts["Other"]) {
99
  professionOrder.push("Other");
100
  }
101
 
102
+ // Group per country
 
 
103
  const countryData = d3.rollup(
104
  flatData,
105
  v => ({
 
110
  );
111
 
112
  /* --------------------------------------------------------
113
+ NEW: Sort countries by total occurrences (descending)
114
  -------------------------------------------------------- */
115
+ const countries = Array.from(countryData.entries())
116
  .sort((a, b) => b[1].total - a[1].total)
117
  .map(d => d[0]);
118
 
 
 
 
 
 
 
 
 
119
  /* --------------------------------------------------------
120
+ 4. SVG SETUP
121
  -------------------------------------------------------- */
122
  const margin = {top: 40, right: 200, bottom: 150, left: 80};
123
  const width = Math.max(800, window.innerWidth - 100) - margin.left - margin.right;
 
130
  .attr("transform", `translate(${margin.left},${margin.top})`);
131
 
132
  /* --------------------------------------------------------
133
+ 5. PREPARE DATA WITH SORTED SEGMENTS PER COUNTRY
134
  -------------------------------------------------------- */
135
+ const topCountries = countries.slice(0, 25);
136
+
137
+ // Build stacked data manually with segments sorted by size per country
138
+ const stackedData = [];
139
+
140
+ topCountries.forEach(country => {
141
+ const countryInfo = countryData.get(country);
142
+ const professions = countryInfo.professions;
143
+
144
+ // Sort professions by value descending (largest at bottom)
145
+ const sortedProfs = [...professions].sort((a, b) => b.value - a.value);
146
+
147
+ let y0 = 0;
148
+ sortedProfs.forEach(prof => {
149
+ stackedData.push({
150
+ country: country,
151
+ profession: prof.profession,
152
+ value: prof.value,
153
+ y0: y0,
154
+ y1: y0 + prof.value
155
+ });
156
+ y0 += prof.value;
157
  });
158
+ });
 
159
 
160
  /* --------------------------------------------------------
161
+ 6. AXES
162
  -------------------------------------------------------- */
163
  const x = d3.scaleBand()
164
  .domain(topCountries)
 
191
  .text("Count");
192
 
193
  /* --------------------------------------------------------
194
+ 7. DRAW BARS (using manually stacked data)
195
  -------------------------------------------------------- */
196
+ svg.selectAll(".bar")
197
+ .data(stackedData)
 
 
 
 
 
198
  .join("rect")
199
  .attr("class", "bar")
200
+ .attr("x", d => x(d.country))
201
+ .attr("y", d => y(d.y1))
202
+ .attr("height", d => y(d.y0) - y(d.y1))
203
  .attr("width", x.bandwidth())
204
+ .attr("fill", d => colorMap[d.profession])
205
  .append("title")
206
+ .text(d => `${d.country} – ${d.profession}: ${d.value}`);
 
 
 
207
 
208
  /* --------------------------------------------------------
209
+ 8. LEGEND (unchanged - uses original professionOrder)
210
  -------------------------------------------------------- */
211
  const legend = svg.append("g")
212
  .attr("transform", `translate(${width + 20}, 0)`);
 
216
 
217
  row.append("rect")
218
  .attr("width", 18)
219
+ .attr("height", 18)
220
+ .attr("fill", colorMap[prof]);
221
 
222
  row.append("text")
223
+ .attr("x", 24)
224
+ .attr("y", 9)
225
+ .attr("dy", "0.35em")
226
  .text(`${prof} (${professionCounts[prof] || 0})`);
227
  });
228
 
229
  /* --------------------------------------------------------
230
+ 9. DOWNLOAD SVG BUTTON
231
  -------------------------------------------------------- */
232
  document.getElementById("downloadBtn").addEventListener("click", () => {
233
  const svgNode = document.querySelector("#chart");
 
255
  });
256
  </script>
257
 
 
258
  </body>
259
  </html>
public/figure_15.html CHANGED
@@ -1,5 +1,5 @@
1
  <!DOCTYPE html>
2
- <html lang="en">
3
  <head>
4
  <meta charset="utf-8">
5
  <script src="https://d3js.org/d3.v7.min.js"></script>
 
1
  <!DOCTYPE html>
2
+ <html lang="en" class="light-mode">
3
  <head>
4
  <meta charset="utf-8">
5
  <script src="https://d3js.org/d3.v7.min.js"></script>
public/figure_5.html CHANGED
@@ -1,5 +1,5 @@
1
  <!DOCTYPE html>
2
- <html lang="en">
3
  <head>
4
  <meta charset="utf-8">
5
  <script src="https://d3js.org/d3.v7.min.js"></script>
@@ -401,7 +401,7 @@
401
  const toggle = document.getElementById('modeToggle');
402
  const root = document.documentElement;
403
 
404
- root.classList.remove('light-mode');
405
 
406
  toggle.addEventListener('change', () => {
407
  root.classList.toggle('light-mode');
 
1
  <!DOCTYPE html>
2
+ <html lang="en" class="light-mode">
3
  <head>
4
  <meta charset="utf-8">
5
  <script src="https://d3js.org/d3.v7.min.js"></script>
 
401
  const toggle = document.getElementById('modeToggle');
402
  const root = document.documentElement;
403
 
404
+ root.classList.add('light-mode');
405
 
406
  toggle.addEventListener('change', () => {
407
  root.classList.toggle('light-mode');
public/img/danbooru_tree_interactive.svg ADDED
public/index.html CHANGED
@@ -22,11 +22,6 @@
22
  </p>
23
  </div>
24
 
25
- <div class="theme-toggle" style="text-align: center; margin-bottom: 2rem;">
26
- <label>
27
- <input type="checkbox" id="modeToggle"> Toggle Light Mode
28
- </label>
29
- </div>
30
  <div class="card-grid">
31
  <a class="card-link" href="figure_15.html">
32
  <div class="preview-card">
@@ -35,7 +30,7 @@
35
  <div class="card-footer">View Graph →</div>
36
  </div>
37
  </a>
38
-
39
  <a class="card-link" href="figure_5.html">
40
  <div class="preview-card">
41
  <img src="img/japan.svg" alt="Preview 3">
@@ -43,19 +38,20 @@
43
  <div class="card-footer">Explore Tags →</div>
44
  </div>
45
  </a>
 
 
 
 
 
 
 
 
 
46
  </div>
47
-
48
- </div>
49
  <script>
50
- const toggle = document.getElementById('modeToggle');
51
- const root = document.documentElement;
52
-
53
- // Default to dark mode
54
- root.classList.remove('light-mode');
55
-
56
- toggle.addEventListener('change', () => {
57
- root.classList.toggle('light-mode');
58
- });
59
  </script>
60
 
61
  </body>
 
22
  </p>
23
  </div>
24
 
 
 
 
 
 
25
  <div class="card-grid">
26
  <a class="card-link" href="figure_15.html">
27
  <div class="preview-card">
 
30
  <div class="card-footer">View Graph →</div>
31
  </div>
32
  </a>
33
+
34
  <a class="card-link" href="figure_5.html">
35
  <div class="preview-card">
36
  <img src="img/japan.svg" alt="Preview 3">
 
38
  <div class="card-footer">Explore Tags →</div>
39
  </div>
40
  </a>
41
+
42
+ <a class="card-link" href="Figure_13_interactive.html">
43
+ <div class="preview-card">
44
+ <img src="img/danbooru_tree_interactive.svg" alt="Preview 3">
45
+ <!-- <div style="width:100%;height:100%;display:flex;align-items:center;justify-content:center;background:#f0f0f0;color:#888;font-size:3rem;">🌳</div> -->
46
+ <div class="card-title">Danbooru Tag Taxonomy</div>
47
+ <div class="card-footer">Explore Tree →</div>
48
+ </div>
49
+ </a>
50
  </div>
51
+
 
52
  <script>
53
+ // Default to light mode
54
+ document.documentElement.classList.add('light-mode');
 
 
 
 
 
 
 
55
  </script>
56
 
57
  </body>
public/json/danbooru_flat_full.json ADDED
The diff for this file is too large to render. See raw diff
 
public/styles.css CHANGED
@@ -159,7 +159,7 @@ input[type=range] {
159
  display: grid;
160
  grid-template-columns: repeat(auto-fit, minmax(220px, 1fr));
161
  gap: 2rem;
162
- max-width: 700px;
163
  margin: 2rem auto;
164
  padding: 0 1rem;
165
  justify-content: center;
 
159
  display: grid;
160
  grid-template-columns: repeat(auto-fit, minmax(220px, 1fr));
161
  gap: 2rem;
162
+ max-width: 900px;
163
  margin: 2rem auto;
164
  padding: 0 1rem;
165
  justify-content: center;