Laura Wagner commited on
Commit
47cec89
·
1 Parent(s): e092e94

added model metadata scraping ipynb

Browse files
jupyter_notebooks/0_Scraping_image_metadata.ipynb CHANGED
@@ -79,7 +79,7 @@
79
  },
80
  {
81
  "cell_type": "code",
82
- "execution_count": 3,
83
  "id": "f8decb63-43f5-4731-823d-94632eee7618",
84
  "metadata": {
85
  "execution": {
@@ -108,7 +108,7 @@
108
  },
109
  {
110
  "cell_type": "code",
111
- "execution_count": 4,
112
  "id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
113
  "metadata": {
114
  "execution": {
@@ -140,7 +140,7 @@
140
  "outputs": [],
141
  "source": [
142
  "# Define the input timestamp in ISO 8601 format\n",
143
- "input_timestamp = \"2025-02-01T00:00:00.000Z\" # point in time from when you want to obtain metadata (you can copy the timestamp from the last *.json batch obtained to get the data of longer timespans)\n",
144
  "\n",
145
  "# Function to convert an ISO 8601 date string to a Unix timestamp in milliseconds with a 2-hour offset \n",
146
  "def iso_to_timestamp(iso_str):\n",
@@ -304,7 +304,7 @@
304
  },
305
  {
306
  "cell_type": "code",
307
- "execution_count": 6,
308
  "id": "7c89cb68-983b-47df-a028-e02d7ca0829d",
309
  "metadata": {
310
  "execution": {
@@ -317,7 +317,7 @@
317
  },
318
  "outputs": [],
319
  "source": [
320
- "#get_image_metadata()"
321
  ]
322
  },
323
  {
@@ -1323,7 +1323,7 @@
1323
  ],
1324
  "metadata": {
1325
  "kernelspec": {
1326
- "display_name": "Python 3 (ipykernel)",
1327
  "language": "python",
1328
  "name": "python3"
1329
  },
@@ -1337,7 +1337,7 @@
1337
  "name": "python",
1338
  "nbconvert_exporter": "python",
1339
  "pygments_lexer": "ipython3",
1340
- "version": "3.11.9"
1341
  }
1342
  },
1343
  "nbformat": 4,
 
79
  },
80
  {
81
  "cell_type": "code",
82
+ "execution_count": 1,
83
  "id": "f8decb63-43f5-4731-823d-94632eee7618",
84
  "metadata": {
85
  "execution": {
 
108
  },
109
  {
110
  "cell_type": "code",
111
+ "execution_count": 2,
112
  "id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
113
  "metadata": {
114
  "execution": {
 
140
  "outputs": [],
141
  "source": [
142
  "# Define the input timestamp in ISO 8601 format\n",
143
+ "input_timestamp = \"2025-03-24T12:59:03.335Z\" # point in time from when you want to obtain metadata (you can copy the timestamp from the last *.json batch obtained to get the data of longer timespans)\n",
144
  "\n",
145
  "# Function to convert an ISO 8601 date string to a Unix timestamp in milliseconds with a 2-hour offset \n",
146
  "def iso_to_timestamp(iso_str):\n",
 
304
  },
305
  {
306
  "cell_type": "code",
307
+ "execution_count": null,
308
  "id": "7c89cb68-983b-47df-a028-e02d7ca0829d",
309
  "metadata": {
310
  "execution": {
 
317
  },
318
  "outputs": [],
319
  "source": [
320
+ "get_image_metadata()"
321
  ]
322
  },
323
  {
 
1323
  ],
1324
  "metadata": {
1325
  "kernelspec": {
1326
+ "display_name": "latm",
1327
  "language": "python",
1328
  "name": "python3"
1329
  },
 
1337
  "name": "python",
1338
  "nbconvert_exporter": "python",
1339
  "pygments_lexer": "ipython3",
1340
+ "version": "3.10.15"
1341
  }
1342
  },
1343
  "nbformat": 4,
jupyter_notebooks/0_Scraping_model_metadata.ipynb ADDED
@@ -0,0 +1,635 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cells": [
3
+ {
4
+ "cell_type": "markdown",
5
+ "id": "1111ea95-d385-49b9-a4d9-ef886ace5c7a",
6
+ "metadata": {
7
+ "execution": {
8
+ "iopub.execute_input": "2025-02-06T11:24:25.566747Z",
9
+ "iopub.status.busy": "2025-02-06T11:24:25.566066Z",
10
+ "iopub.status.idle": "2025-02-06T11:24:25.571748Z",
11
+ "shell.execute_reply": "2025-02-06T11:24:25.571305Z",
12
+ "shell.execute_reply.started": "2025-02-06T11:24:25.566705Z"
13
+ }
14
+ },
15
+ "source": [
16
+ "# 0 Scraping Metadata and Dataset consolidation\n"
17
+ ]
18
+ },
19
+ {
20
+ "cell_type": "markdown",
21
+ "id": "6632505a-e7ca-4463-9ffc-e36fad42235f",
22
+ "metadata": {},
23
+ "source": [
24
+ "## IMAGES\n",
25
+ "---"
26
+ ]
27
+ },
28
+ {
29
+ "cell_type": "markdown",
30
+ "id": "e3388bac-bb71-40bc-a693-9ac7a2d5f32c",
31
+ "metadata": {
32
+ "execution": {
33
+ "iopub.execute_input": "2025-02-06T10:08:22.229784Z",
34
+ "iopub.status.busy": "2025-02-06T10:08:22.229287Z",
35
+ "iopub.status.idle": "2025-02-06T10:08:22.232210Z",
36
+ "shell.execute_reply": "2025-02-06T10:08:22.231793Z",
37
+ "shell.execute_reply.started": "2025-02-06T10:08:22.229766Z"
38
+ }
39
+ },
40
+ "source": [
41
+ "### Step 1: Image metadata scraping, sorting and CSV consolidation"
42
+ ]
43
+ },
44
+ {
45
+ "cell_type": "code",
46
+ "execution_count": 1,
47
+ "id": "f8decb63-43f5-4731-823d-94632eee7618",
48
+ "metadata": {
49
+ "execution": {
50
+ "iopub.execute_input": "2025-02-08T19:37:51.115763Z",
51
+ "iopub.status.busy": "2025-02-08T19:37:51.114573Z",
52
+ "iopub.status.idle": "2025-02-08T19:37:51.170027Z",
53
+ "shell.execute_reply": "2025-02-08T19:37:51.169401Z",
54
+ "shell.execute_reply.started": "2025-02-08T19:37:51.115738Z"
55
+ }
56
+ },
57
+ "outputs": [],
58
+ "source": [
59
+ "import os\n",
60
+ "import json\n",
61
+ "import csv\n",
62
+ "import requests\n",
63
+ "from datetime import datetime\n",
64
+ "import time\n",
65
+ "from pathlib import Path\n",
66
+ "import hashlib\n",
67
+ "import pandas as pd\n",
68
+ "import sys\n",
69
+ "from datetime import datetime, timedelta\n",
70
+ "import shutil"
71
+ ]
72
+ },
73
+ {
74
+ "cell_type": "code",
75
+ "execution_count": 2,
76
+ "id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
77
+ "metadata": {
78
+ "execution": {
79
+ "iopub.execute_input": "2025-02-08T19:37:51.809027Z",
80
+ "iopub.status.busy": "2025-02-08T19:37:51.808835Z",
81
+ "iopub.status.idle": "2025-02-08T19:37:51.812922Z",
82
+ "shell.execute_reply": "2025-02-08T19:37:51.812429Z",
83
+ "shell.execute_reply.started": "2025-02-08T19:37:51.809009Z"
84
+ }
85
+ },
86
+ "outputs": [],
87
+ "source": [
88
+ "current_dir = Path.cwd()"
89
+ ]
90
+ },
91
+ {
92
+ "cell_type": "markdown",
93
+ "id": "11647bb7-5ce9-414a-8486-5bdce8d9cfea",
94
+ "metadata": {
95
+ "execution": {
96
+ "iopub.execute_input": "2025-02-06T12:49:43.126762Z",
97
+ "iopub.status.busy": "2025-02-06T12:49:43.125797Z",
98
+ "iopub.status.idle": "2025-02-06T12:49:43.129759Z",
99
+ "shell.execute_reply": "2025-02-06T12:49:43.129176Z",
100
+ "shell.execute_reply.started": "2025-02-06T12:49:43.126736Z"
101
+ }
102
+ },
103
+ "source": [
104
+ "## MODELS"
105
+ ]
106
+ },
107
+ {
108
+ "cell_type": "markdown",
109
+ "id": "5c9cb5e3-7cea-4574-9319-f3cd89354b1f",
110
+ "metadata": {
111
+ "execution": {
112
+ "iopub.execute_input": "2025-02-06T13:23:38.017078Z",
113
+ "iopub.status.busy": "2025-02-06T13:23:38.016639Z",
114
+ "iopub.status.idle": "2025-02-06T13:23:38.019993Z",
115
+ "shell.execute_reply": "2025-02-06T13:23:38.019549Z",
116
+ "shell.execute_reply.started": "2025-02-06T13:23:38.017053Z"
117
+ }
118
+ },
119
+ "source": [
120
+ "### Step 1: Scrape model metadata"
121
+ ]
122
+ },
123
+ {
124
+ "cell_type": "markdown",
125
+ "id": "83e12b9f-5ae1-407d-9754-5979d837f787",
126
+ "metadata": {
127
+ "execution": {
128
+ "iopub.execute_input": "2025-02-06T14:06:04.857874Z",
129
+ "iopub.status.busy": "2025-02-06T14:06:04.857500Z",
130
+ "iopub.status.idle": "2025-02-06T14:06:04.860438Z",
131
+ "shell.execute_reply": "2025-02-06T14:06:04.860030Z",
132
+ "shell.execute_reply.started": "2025-02-06T14:06:04.857856Z"
133
+ }
134
+ },
135
+ "source": [
136
+ "#### the resulting files will appear in data/raw/model_metadata as *.json"
137
+ ]
138
+ },
139
+ {
140
+ "cell_type": "code",
141
+ "execution_count": 12,
142
+ "id": "5db9c00e",
143
+ "metadata": {},
144
+ "outputs": [],
145
+ "source": [
146
+ "key_karussell = current_dir.parent / 'misc/credentials/civitai_api_keys.txt'\n",
147
+ "directory_path = current_dir.parent / 'data/raw/model_metadata/'"
148
+ ]
149
+ },
150
+ {
151
+ "cell_type": "code",
152
+ "execution_count": 13,
153
+ "id": "41ee14e0-fb78-4f91-aba1-13faf05af7d8",
154
+ "metadata": {
155
+ "execution": {
156
+ "iopub.execute_input": "2025-02-08T19:37:56.687832Z",
157
+ "iopub.status.busy": "2025-02-08T19:37:56.687251Z",
158
+ "iopub.status.idle": "2025-02-08T19:37:56.696572Z",
159
+ "shell.execute_reply": "2025-02-08T19:37:56.696059Z",
160
+ "shell.execute_reply.started": "2025-02-08T19:37:56.687809Z"
161
+ }
162
+ },
163
+ "outputs": [],
164
+ "source": [
165
+ "import datetime\n",
166
+ "\n",
167
+ "def load_api_keys():\n",
168
+ " \"\"\"Load API keys from a text file, one per line.\"\"\"\n",
169
+ " if not os.path.exists(key_karussell):\n",
170
+ " raise FileNotFoundError(f\"API key file '{API_KEYS_FILE}' not found!\")\n",
171
+ " \n",
172
+ " with open(key_karussell, 'r') as file:\n",
173
+ " keys = [line.strip() for line in file if line.strip()]\n",
174
+ " \n",
175
+ " if not keys:\n",
176
+ " raise ValueError(\"No API keys found in the file!\")\n",
177
+ " \n",
178
+ " return keys\n",
179
+ "\n",
180
+ "def get_model_metadata():\n",
181
+ " base_url = \"https://civitai.com/api/v1/models\"\n",
182
+ " params = {\"sort\": \"Newest\", \"nsfw\": True}\n",
183
+ "\n",
184
+ " # Load API keys\n",
185
+ " api_keys = load_api_keys()\n",
186
+ " key_index = 0 # Start with the first key\n",
187
+ "\n",
188
+ " page_counter = 0\n",
189
+ " max_pages = 300000000 # Adjust as needed\n",
190
+ " os.makedirs(directory_path, exist_ok=True)\n",
191
+ "\n",
192
+ " while True:\n",
193
+ " if page_counter >= max_pages:\n",
194
+ " print(f\"Reached the limit of {max_pages} pages.\")\n",
195
+ " break\n",
196
+ "\n",
197
+ " headers = {\n",
198
+ " \"Accept\": \"application/json\",\n",
199
+ " \"Authorization\": f\"Bearer {api_keys[key_index]}\"\n",
200
+ " }\n",
201
+ "\n",
202
+ " response = requests.get(base_url, headers=headers, params=params)\n",
203
+ "\n",
204
+ " if response.status_code == 200:\n",
205
+ " data = response.json()\n",
206
+ " page_counter += 1\n",
207
+ "\n",
208
+ " # Add timestamp\n",
209
+ " formatted_timestamp = datetime.datetime.now().strftime(\"data obtained on the %d.%m.%Y at %H:%M CEST\")\n",
210
+ "\n",
211
+ " data['timestamp'] = formatted_timestamp\n",
212
+ "\n",
213
+ " # Save data to file\n",
214
+ " file_path = os.path.join(directory_path, f'newest_models_{page_counter}.json')\n",
215
+ " with open(file_path, 'w', encoding='utf-8') as file:\n",
216
+ " json.dump(data, file, indent=4)\n",
217
+ "\n",
218
+ " # Check for nextCursor\n",
219
+ " next_cursor = data.get('metadata', {}).get('nextCursor')\n",
220
+ " if not next_cursor:\n",
221
+ " print(\"No more data available.\")\n",
222
+ " break\n",
223
+ " else:\n",
224
+ " params['cursor'] = next_cursor\n",
225
+ " \n",
226
+ " elif response.status_code in (401, 403): # Unauthorized or Forbidden\n",
227
+ " print(f\"API Key {key_index + 1} failed with status {response.status_code}. Trying next key...\")\n",
228
+ " key_index += 1\n",
229
+ "\n",
230
+ " if key_index >= len(api_keys):\n",
231
+ " print(\"All API keys failed. Exiting.\")\n",
232
+ " break # Stop if all keys fail\n",
233
+ " \n",
234
+ " else:\n",
235
+ " print(f\"Failed to fetch data: HTTP {response.status_code}\")\n",
236
+ " break # Stop on other errors\n"
237
+ ]
238
+ },
239
+ {
240
+ "cell_type": "markdown",
241
+ "id": "8f43ce23-5986-4b67-be2d-453831af9a6e",
242
+ "metadata": {},
243
+ "source": [
244
+ "uncomment this to get model metadata"
245
+ ]
246
+ },
247
+ {
248
+ "cell_type": "code",
249
+ "execution_count": null,
250
+ "id": "3f95b4ba-5742-4268-b2e8-de9145faf495",
251
+ "metadata": {
252
+ "execution": {
253
+ "iopub.execute_input": "2025-02-08T19:37:58.117386Z",
254
+ "iopub.status.busy": "2025-02-08T19:37:58.117158Z",
255
+ "iopub.status.idle": "2025-02-08T19:37:58.121162Z",
256
+ "shell.execute_reply": "2025-02-08T19:37:58.120580Z",
257
+ "shell.execute_reply.started": "2025-02-08T19:37:58.117369Z"
258
+ }
259
+ },
260
+ "outputs": [],
261
+ "source": [
262
+ "get_model_metadata()"
263
+ ]
264
+ },
265
+ {
266
+ "cell_type": "markdown",
267
+ "id": "1ed5f44e-612e-4b6d-8974-904fb3e058d6",
268
+ "metadata": {
269
+ "execution": {
270
+ "iopub.execute_input": "2025-02-06T12:52:40.162173Z",
271
+ "iopub.status.busy": "2025-02-06T12:52:40.159989Z",
272
+ "iopub.status.idle": "2025-02-06T12:52:40.170634Z",
273
+ "shell.execute_reply": "2025-02-06T12:52:40.169945Z",
274
+ "shell.execute_reply.started": "2025-02-06T12:52:40.162124Z"
275
+ }
276
+ },
277
+ "source": [
278
+ "### Step 2 Consolidate Model-dataset CSV"
279
+ ]
280
+ },
281
+ {
282
+ "cell_type": "code",
283
+ "execution_count": 8,
284
+ "id": "464d82f5-c24e-4b53-9b50-682fa4bf3430",
285
+ "metadata": {
286
+ "execution": {
287
+ "iopub.execute_input": "2025-02-08T19:37:59.052245Z",
288
+ "iopub.status.busy": "2025-02-08T19:37:59.051645Z",
289
+ "iopub.status.idle": "2025-02-08T19:37:59.056017Z",
290
+ "shell.execute_reply": "2025-02-08T19:37:59.055598Z",
291
+ "shell.execute_reply.started": "2025-02-08T19:37:59.052224Z"
292
+ }
293
+ },
294
+ "outputs": [],
295
+ "source": [
296
+ "## path thingy\n",
297
+ "try: #scripts\n",
298
+ " current_dir = Path(__file__).resolve().parent\n",
299
+ "except NameError:\n",
300
+ " # jupyter\n",
301
+ " current_dir = Path.cwd()"
302
+ ]
303
+ },
304
+ {
305
+ "cell_type": "code",
306
+ "execution_count": null,
307
+ "id": "14f3db4d-ef93-4629-8a27-971adb49248b",
308
+ "metadata": {
309
+ "execution": {
310
+ "iopub.execute_input": "2025-02-08T19:37:59.443457Z",
311
+ "iopub.status.busy": "2025-02-08T19:37:59.443236Z",
312
+ "iopub.status.idle": "2025-02-08T19:37:59.890103Z",
313
+ "shell.execute_reply": "2025-02-08T19:37:59.889571Z",
314
+ "shell.execute_reply.started": "2025-02-08T19:37:59.443439Z"
315
+ }
316
+ },
317
+ "outputs": [
318
+ {
319
+ "name": "stdout",
320
+ "output_type": "stream",
321
+ "text": [
322
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_2.json\n",
323
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_1.json\n",
324
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_4.json\n",
325
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_6.json\n",
326
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_7.json\n",
327
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_5.json\n",
328
+ "Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_3.json\n"
329
+ ]
330
+ }
331
+ ],
332
+ "source": [
333
+ "import os\n",
334
+ "import json\n",
335
+ "import pandas as pd\n",
336
+ "import hashlib\n",
337
+ "from pathlib import Path\n",
338
+ "from datetime import datetime, timezone\n",
339
+ "\n",
340
+ "\n",
341
+ "\n",
342
+ "\n",
343
+ "def hash_username(username):\n",
344
+ " return hashlib.sha256(username.encode('utf-8')).hexdigest()[:16]\n",
345
+ "\n",
346
+ "def parse_date(date_str):\n",
347
+ " try:\n",
348
+ " return datetime.fromisoformat(date_str.replace('Z', '+00:00'))\n",
349
+ " except Exception:\n",
350
+ " return datetime.min.replace(tzinfo=timezone.utc) # Make it timezone-aware\n",
351
+ "\n",
352
+ "def get_latest_model_version(model_versions):\n",
353
+ " return max(model_versions, key=lambda mv: parse_date(mv.get('publishedAt', '')))\n",
354
+ "\n",
355
+ "def process_directory_recursively(root_dir):\n",
356
+ " root_path = Path(root_dir)\n",
357
+ " seen = {} # id -> (publishedAt, record)\n",
358
+ " data_records = []\n",
359
+ "\n",
360
+ " for json_file in root_path.rglob('*.json'):\n",
361
+ " if not json_file.is_file():\n",
362
+ " continue\n",
363
+ "\n",
364
+ " #print(f\"Processing file: {json_file}\")\n",
365
+ " try:\n",
366
+ " with open(json_file, 'r', encoding='utf-8') as f:\n",
367
+ " data = json.load(f)\n",
368
+ " except Exception as e:\n",
369
+ " print(f\"Failed to load {json_file}: {e}\")\n",
370
+ " continue\n",
371
+ "\n",
372
+ " items = data.get('items') or data.get('data') or []\n",
373
+ " for item in items:\n",
374
+ " if not isinstance(item, dict):\n",
375
+ " continue\n",
376
+ "\n",
377
+ " model_id = item.get('id')\n",
378
+ " model_versions = item.get('modelVersions', [])\n",
379
+ " if not model_versions:\n",
380
+ " continue\n",
381
+ "\n",
382
+ " latest_version = get_latest_model_version(model_versions)\n",
383
+ " published_at = latest_version.get('publishedAt', '')\n",
384
+ " current_dt = parse_date(published_at)\n",
385
+ "\n",
386
+ " if model_id in seen and current_dt <= seen[model_id][0]:\n",
387
+ " continue\n",
388
+ " seen[model_id] = (current_dt, item)\n",
389
+ "\n",
390
+ " for model_id, (_, item) in seen.items():\n",
391
+ " model_versions = item.get('modelVersions', [])\n",
392
+ " latest_version = get_latest_model_version(model_versions)\n",
393
+ " version_ids = [mv.get('id', '') for mv in model_versions[:20]]\n",
394
+ "\n",
395
+ " files = latest_version.get('files', [])\n",
396
+ " auto_hashes = files[0].get('hashes', {}) if files else {}\n",
397
+ " images = latest_version.get('images', [])\n",
398
+ " first_image_url = images[0]['url'] if images else ''\n",
399
+ " latest_image_url = images[-1]['url'] if images else ''\n",
400
+ "\n",
401
+ " username = item.get('creator', {}).get('username', '')\n",
402
+ " record = {\n",
403
+ " 'id': item.get('id', ''),\n",
404
+ " 'name': item.get('name', ''),\n",
405
+ " 'type': item.get('type', ''),\n",
406
+ " 'baseModel': latest_version.get('baseModel', ''),\n",
407
+ " 'downloadCount': item.get('stats', {}).get('downloadCount', 0),\n",
408
+ " 'nsfwLevel': item.get('nsfwLevel', 0),\n",
409
+ " 'modelVersions': len(model_versions),\n",
410
+ " 'publishedAt': latest_version.get('publishedAt', ''),\n",
411
+ " 'usernameHash': hash_username(username) if username else '',\n",
412
+ " 'downloadUrl': latest_version.get('downloadUrl', ''),\n",
413
+ " 'firstImageUrl': first_image_url,\n",
414
+ " 'latestImageUrl': latest_image_url,\n",
415
+ " 'poi': item.get('poi', False),\n",
416
+ " 'AutoV1': auto_hashes.get('AutoV1', ''),\n",
417
+ " 'AutoV2': auto_hashes.get('AutoV2', ''),\n",
418
+ " 'AutoV3': auto_hashes.get('AutoV3', ''),\n",
419
+ " 'SHA256': auto_hashes.get('SHA256', ''),\n",
420
+ " 'CRC32': auto_hashes.get('CRC32', ''),\n",
421
+ " 'BLAKE3': auto_hashes.get('BLAKE3', ''),\n",
422
+ " 'previewImage': latest_image_url\n",
423
+ " }\n",
424
+ "\n",
425
+ " for i in range(20):\n",
426
+ " record[f'version_id_{i+1}'] = version_ids[i] if i < len(version_ids) else ''\n",
427
+ "\n",
428
+ " tags = item.get('tags', [])\n",
429
+ " for i in range(7):\n",
430
+ " record[f'tag_{i+1}'] = tags[i] if i < len(tags) else ''\n",
431
+ "\n",
432
+ " data_records.append(record)\n",
433
+ "\n",
434
+ " return pd.DataFrame(data_records)\n",
435
+ "\n",
436
+ "# Usage Example\n",
437
+ "root_directory = current_dir.parent / 'data/raw/model_metadata/'\n",
438
+ "df = process_directory_recursively(root_directory)\n",
439
+ "df_sorted = df.sort_values(by='downloadCount', ascending=False)\n",
440
+ "\n",
441
+ "# Optionally save\n",
442
+ "# df_sorted.to_csv('combined_metadata.csv', index=False)\n"
443
+ ]
444
+ },
445
+ {
446
+ "cell_type": "markdown",
447
+ "id": "2d6f822f-754f-4feb-9ed6-6f868421c454",
448
+ "metadata": {},
449
+ "source": [
450
+ "### Save model-data to CSV"
451
+ ]
452
+ },
453
+ {
454
+ "cell_type": "code",
455
+ "execution_count": null,
456
+ "id": "f861cf7b-5eb3-46ad-ab2d-9e2fdf2169b4",
457
+ "metadata": {
458
+ "execution": {
459
+ "iopub.execute_input": "2025-02-08T19:38:02.047421Z",
460
+ "iopub.status.busy": "2025-02-08T19:38:02.046761Z",
461
+ "iopub.status.idle": "2025-02-08T19:38:02.193381Z",
462
+ "shell.execute_reply": "2025-02-08T19:38:02.192886Z",
463
+ "shell.execute_reply.started": "2025-02-08T19:38:02.047397Z"
464
+ }
465
+ },
466
+ "outputs": [
467
+ {
468
+ "name": "stdout",
469
+ "output_type": "stream",
470
+ "text": [
471
+ "Data has been saved to /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/CSV/Civiverse-Models.csv\n"
472
+ ]
473
+ }
474
+ ],
475
+ "source": [
476
+ "output_csv = current_dir.parent / 'data/CSV/Civiverse-Models_2025.csv'\n",
477
+ "output_csv.parent.mkdir(parents=True, exist_ok=True)\n",
478
+ "df_sorted.to_csv(output_csv, index=False)\n",
479
+ "print(f\"Data has been saved to {output_csv}\")"
480
+ ]
481
+ },
482
+ {
483
+ "cell_type": "markdown",
484
+ "id": "9b948933-2a2f-42b0-9083-2d81586ae3f2",
485
+ "metadata": {
486
+ "execution": {
487
+ "iopub.execute_input": "2025-02-06T13:42:32.447860Z",
488
+ "iopub.status.busy": "2025-02-06T13:42:32.446832Z",
489
+ "iopub.status.idle": "2025-02-06T13:42:32.453827Z",
490
+ "shell.execute_reply": "2025-02-06T13:42:32.453271Z",
491
+ "shell.execute_reply.started": "2025-02-06T13:42:32.447821Z"
492
+ }
493
+ },
494
+ "source": [
495
+ "### Step 3 Create Subsets: Checkpoint only, POI True, POI False"
496
+ ]
497
+ },
498
+ {
499
+ "cell_type": "code",
500
+ "execution_count": 11,
501
+ "id": "ae894e26-4984-40e6-80a6-54fb6b61c873",
502
+ "metadata": {
503
+ "execution": {
504
+ "iopub.execute_input": "2025-02-08T19:38:03.089160Z",
505
+ "iopub.status.busy": "2025-02-08T19:38:03.088490Z",
506
+ "iopub.status.idle": "2025-02-08T19:38:03.093067Z",
507
+ "shell.execute_reply": "2025-02-08T19:38:03.092634Z",
508
+ "shell.execute_reply.started": "2025-02-08T19:38:03.089138Z"
509
+ }
510
+ },
511
+ "outputs": [],
512
+ "source": [
513
+ "file_path = current_dir.parent / 'data/CSV/Civiverse-Models.csv' # Update this with your actual file path\n",
514
+ "(current_dir.parent / 'data/CSV/model_subsets').mkdir(parents=True, exist_ok=True)\n"
515
+ ]
516
+ },
517
+ {
518
+ "cell_type": "code",
519
+ "execution_count": 13,
520
+ "id": "88438f88-c723-423e-a4e5-e07f59096b72",
521
+ "metadata": {
522
+ "execution": {
523
+ "iopub.execute_input": "2025-02-08T19:41:39.279495Z",
524
+ "iopub.status.busy": "2025-02-08T19:41:39.279039Z",
525
+ "iopub.status.idle": "2025-02-08T19:41:39.674445Z",
526
+ "shell.execute_reply": "2025-02-08T19:41:39.673981Z",
527
+ "shell.execute_reply.started": "2025-02-08T19:41:39.279476Z"
528
+ }
529
+ },
530
+ "outputs": [
531
+ {
532
+ "name": "stdout",
533
+ "output_type": "stream",
534
+ "text": [
535
+ "Files saved successfully!\n"
536
+ ]
537
+ }
538
+ ],
539
+ "source": [
540
+ "import pandas as pd\n",
541
+ "\n",
542
+ "# Load the dataset\n",
543
+ "\n",
544
+ "data = pd.read_csv(file_path)\n",
545
+ "\n",
546
+ "# Version 1: Only 'poi' true models\n",
547
+ "poi_true_models = data[data['poi'] == True]\n",
548
+ "\n",
549
+ "# Version 2: Only types lora, dora, locon, textual inversion\n",
550
+ "specific_types = ['LORA', 'DORA', 'LOCON', 'textualInversion']\n",
551
+ "adapters = data[data['type'].isin(specific_types)]\n",
552
+ "\n",
553
+ "# Version 3: Only type checkpoint\n",
554
+ "checkpoint_models = data[data['type'] == 'Checkpoint']\n",
555
+ "\n",
556
+ "# Version 4: All models apart from 'poi' true\n",
557
+ "non_poi_models = data[data['poi'] != True]\n",
558
+ "\n",
559
+ "# Version 5: All models apart from 'poi' true and with nsfwLevel below 13\n",
560
+ "non_poi_low_nsfw_models = data[(data['poi'] != True) & (data['nsfwLevel'] < 13)]\n",
561
+ "\n",
562
+ "# Save the versions as separate CSV files\n",
563
+ "poi_true_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_adapters_poi_true.csv', index=False)\n",
564
+ "adapters.to_csv(current_dir.parent / 'data/CSV/adapters.csv', index=False)\n",
565
+ "checkpoint_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_checkpoint_only.csv', index=False)\n",
566
+ "non_poi_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_adapters_poi_false.csv', index=False)\n",
567
+ "\n",
568
+ "print(\"Files saved successfully!\")\n"
569
+ ]
570
+ },
571
+ {
572
+ "cell_type": "code",
573
+ "execution_count": null,
574
+ "id": "49e35088-8b2e-4189-83e1-3098d55dcad2",
575
+ "metadata": {},
576
+ "outputs": [],
577
+ "source": [
578
+ "import pandas as pd\n",
579
+ "import os\n",
580
+ "\n",
581
+ "# Load the dataset\n",
582
+ "df = pd.read_csv('data/all_models_with_tags.csv')\n",
583
+ "\n",
584
+ "# Filter for rows where poi is True\n",
585
+ "filtered_df = df[df['poi'] == True]\n",
586
+ "os.makedirs('data/model_subsets', exist_ok=True)\n",
587
+ "\n",
588
+ "# Save the filtered DataFrame to a new CSV file\n",
589
+ "filtered_df.to_csv('data/model_subsets/all_models_poi.csv', index=False)\n"
590
+ ]
591
+ },
592
+ {
593
+ "cell_type": "code",
594
+ "execution_count": null,
595
+ "id": "06c15f2c",
596
+ "metadata": {},
597
+ "outputs": [],
598
+ "source": [
599
+ "import pandas as pd\n",
600
+ "import os\n",
601
+ "\n",
602
+ "# Load the dataset\n",
603
+ "df = pd.read_csv('data/all_models_with_tags.csv')\n",
604
+ "\n",
605
+ "# Filter for rows where poi is True\n",
606
+ "filtered_df = df[df['poi'] == False]\n",
607
+ "os.makedirs('data/model_subsets', exist_ok=True)\n",
608
+ "\n",
609
+ "# Save the filtered DataFrame to a new CSV file\n",
610
+ "filtered_df.to_csv('data/model_subsets/all_models_poi_false.csv', index=False)\n"
611
+ ]
612
+ }
613
+ ],
614
+ "metadata": {
615
+ "kernelspec": {
616
+ "display_name": "latm",
617
+ "language": "python",
618
+ "name": "python3"
619
+ },
620
+ "language_info": {
621
+ "codemirror_mode": {
622
+ "name": "ipython",
623
+ "version": 3
624
+ },
625
+ "file_extension": ".py",
626
+ "mimetype": "text/x-python",
627
+ "name": "python",
628
+ "nbconvert_exporter": "python",
629
+ "pygments_lexer": "ipython3",
630
+ "version": "3.10.15"
631
+ }
632
+ },
633
+ "nbformat": 4,
634
+ "nbformat_minor": 5
635
+ }
jupyter_notebooks/Section_1_Figure_1_image_grid.ipynb CHANGED
@@ -367,7 +367,7 @@
367
  ],
368
  "metadata": {
369
  "kernelspec": {
370
- "display_name": "Python 3 (ipykernel)",
371
  "language": "python",
372
  "name": "python3"
373
  },
@@ -381,7 +381,7 @@
381
  "name": "python",
382
  "nbconvert_exporter": "python",
383
  "pygments_lexer": "ipython3",
384
- "version": "3.11.9"
385
  }
386
  },
387
  "nbformat": 4,
 
367
  ],
368
  "metadata": {
369
  "kernelspec": {
370
+ "display_name": "latm",
371
  "language": "python",
372
  "name": "python3"
373
  },
 
381
  "name": "python",
382
  "nbconvert_exporter": "python",
383
  "pygments_lexer": "ipython3",
384
+ "version": "3.10.15"
385
  }
386
  },
387
  "nbformat": 4,
jupyter_notebooks/Section_3-4_extract_LoRA_metadata.ipynb CHANGED
@@ -13,7 +13,7 @@
13
  }
14
  },
15
  "source": [
16
- "# Section 6.5: LoRA metadata"
17
  ]
18
  },
19
  {
 
13
  }
14
  },
15
  "source": [
16
+ "# Section 3-4: LoRA metadata"
17
  ]
18
  },
19
  {
jupyter_notebooks/{SuppM_Figure_13_Danbooru_taxonomy.ipynb → SuppM_Figure_12_Danbooru_categories.ipynb} RENAMED
File without changes