Laura Wagner commited on
Commit ·
47cec89
1
Parent(s): e092e94
added model metadata scraping ipynb
Browse files- jupyter_notebooks/0_Scraping_image_metadata.ipynb +7 -7
- jupyter_notebooks/0_Scraping_model_metadata.ipynb +635 -0
- jupyter_notebooks/Section_1_Figure_1_image_grid.ipynb +2 -2
- jupyter_notebooks/Section_3-4_extract_LoRA_metadata.ipynb +1 -1
- jupyter_notebooks/{SuppM_Figure_13_Danbooru_taxonomy.ipynb → SuppM_Figure_12_Danbooru_categories.ipynb} +0 -0
jupyter_notebooks/0_Scraping_image_metadata.ipynb
CHANGED
|
@@ -79,7 +79,7 @@
|
|
| 79 |
},
|
| 80 |
{
|
| 81 |
"cell_type": "code",
|
| 82 |
-
"execution_count":
|
| 83 |
"id": "f8decb63-43f5-4731-823d-94632eee7618",
|
| 84 |
"metadata": {
|
| 85 |
"execution": {
|
|
@@ -108,7 +108,7 @@
|
|
| 108 |
},
|
| 109 |
{
|
| 110 |
"cell_type": "code",
|
| 111 |
-
"execution_count":
|
| 112 |
"id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
|
| 113 |
"metadata": {
|
| 114 |
"execution": {
|
|
@@ -140,7 +140,7 @@
|
|
| 140 |
"outputs": [],
|
| 141 |
"source": [
|
| 142 |
"# Define the input timestamp in ISO 8601 format\n",
|
| 143 |
-
"input_timestamp = \"2025-
|
| 144 |
"\n",
|
| 145 |
"# Function to convert an ISO 8601 date string to a Unix timestamp in milliseconds with a 2-hour offset \n",
|
| 146 |
"def iso_to_timestamp(iso_str):\n",
|
|
@@ -304,7 +304,7 @@
|
|
| 304 |
},
|
| 305 |
{
|
| 306 |
"cell_type": "code",
|
| 307 |
-
"execution_count":
|
| 308 |
"id": "7c89cb68-983b-47df-a028-e02d7ca0829d",
|
| 309 |
"metadata": {
|
| 310 |
"execution": {
|
|
@@ -317,7 +317,7 @@
|
|
| 317 |
},
|
| 318 |
"outputs": [],
|
| 319 |
"source": [
|
| 320 |
-
"
|
| 321 |
]
|
| 322 |
},
|
| 323 |
{
|
|
@@ -1323,7 +1323,7 @@
|
|
| 1323 |
],
|
| 1324 |
"metadata": {
|
| 1325 |
"kernelspec": {
|
| 1326 |
-
"display_name": "
|
| 1327 |
"language": "python",
|
| 1328 |
"name": "python3"
|
| 1329 |
},
|
|
@@ -1337,7 +1337,7 @@
|
|
| 1337 |
"name": "python",
|
| 1338 |
"nbconvert_exporter": "python",
|
| 1339 |
"pygments_lexer": "ipython3",
|
| 1340 |
-
"version": "3.
|
| 1341 |
}
|
| 1342 |
},
|
| 1343 |
"nbformat": 4,
|
|
|
|
| 79 |
},
|
| 80 |
{
|
| 81 |
"cell_type": "code",
|
| 82 |
+
"execution_count": 1,
|
| 83 |
"id": "f8decb63-43f5-4731-823d-94632eee7618",
|
| 84 |
"metadata": {
|
| 85 |
"execution": {
|
|
|
|
| 108 |
},
|
| 109 |
{
|
| 110 |
"cell_type": "code",
|
| 111 |
+
"execution_count": 2,
|
| 112 |
"id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
|
| 113 |
"metadata": {
|
| 114 |
"execution": {
|
|
|
|
| 140 |
"outputs": [],
|
| 141 |
"source": [
|
| 142 |
"# Define the input timestamp in ISO 8601 format\n",
|
| 143 |
+
"input_timestamp = \"2025-03-24T12:59:03.335Z\" # point in time from when you want to obtain metadata (you can copy the timestamp from the last *.json batch obtained to get the data of longer timespans)\n",
|
| 144 |
"\n",
|
| 145 |
"# Function to convert an ISO 8601 date string to a Unix timestamp in milliseconds with a 2-hour offset \n",
|
| 146 |
"def iso_to_timestamp(iso_str):\n",
|
|
|
|
| 304 |
},
|
| 305 |
{
|
| 306 |
"cell_type": "code",
|
| 307 |
+
"execution_count": null,
|
| 308 |
"id": "7c89cb68-983b-47df-a028-e02d7ca0829d",
|
| 309 |
"metadata": {
|
| 310 |
"execution": {
|
|
|
|
| 317 |
},
|
| 318 |
"outputs": [],
|
| 319 |
"source": [
|
| 320 |
+
"get_image_metadata()"
|
| 321 |
]
|
| 322 |
},
|
| 323 |
{
|
|
|
|
| 1323 |
],
|
| 1324 |
"metadata": {
|
| 1325 |
"kernelspec": {
|
| 1326 |
+
"display_name": "latm",
|
| 1327 |
"language": "python",
|
| 1328 |
"name": "python3"
|
| 1329 |
},
|
|
|
|
| 1337 |
"name": "python",
|
| 1338 |
"nbconvert_exporter": "python",
|
| 1339 |
"pygments_lexer": "ipython3",
|
| 1340 |
+
"version": "3.10.15"
|
| 1341 |
}
|
| 1342 |
},
|
| 1343 |
"nbformat": 4,
|
jupyter_notebooks/0_Scraping_model_metadata.ipynb
ADDED
|
@@ -0,0 +1,635 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cells": [
|
| 3 |
+
{
|
| 4 |
+
"cell_type": "markdown",
|
| 5 |
+
"id": "1111ea95-d385-49b9-a4d9-ef886ace5c7a",
|
| 6 |
+
"metadata": {
|
| 7 |
+
"execution": {
|
| 8 |
+
"iopub.execute_input": "2025-02-06T11:24:25.566747Z",
|
| 9 |
+
"iopub.status.busy": "2025-02-06T11:24:25.566066Z",
|
| 10 |
+
"iopub.status.idle": "2025-02-06T11:24:25.571748Z",
|
| 11 |
+
"shell.execute_reply": "2025-02-06T11:24:25.571305Z",
|
| 12 |
+
"shell.execute_reply.started": "2025-02-06T11:24:25.566705Z"
|
| 13 |
+
}
|
| 14 |
+
},
|
| 15 |
+
"source": [
|
| 16 |
+
"# 0 Scraping Metadata and Dataset consolidation\n"
|
| 17 |
+
]
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"cell_type": "markdown",
|
| 21 |
+
"id": "6632505a-e7ca-4463-9ffc-e36fad42235f",
|
| 22 |
+
"metadata": {},
|
| 23 |
+
"source": [
|
| 24 |
+
"## IMAGES\n",
|
| 25 |
+
"---"
|
| 26 |
+
]
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"cell_type": "markdown",
|
| 30 |
+
"id": "e3388bac-bb71-40bc-a693-9ac7a2d5f32c",
|
| 31 |
+
"metadata": {
|
| 32 |
+
"execution": {
|
| 33 |
+
"iopub.execute_input": "2025-02-06T10:08:22.229784Z",
|
| 34 |
+
"iopub.status.busy": "2025-02-06T10:08:22.229287Z",
|
| 35 |
+
"iopub.status.idle": "2025-02-06T10:08:22.232210Z",
|
| 36 |
+
"shell.execute_reply": "2025-02-06T10:08:22.231793Z",
|
| 37 |
+
"shell.execute_reply.started": "2025-02-06T10:08:22.229766Z"
|
| 38 |
+
}
|
| 39 |
+
},
|
| 40 |
+
"source": [
|
| 41 |
+
"### Step 1: Image metadata scraping, sorting and CSV consolidation"
|
| 42 |
+
]
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"cell_type": "code",
|
| 46 |
+
"execution_count": 1,
|
| 47 |
+
"id": "f8decb63-43f5-4731-823d-94632eee7618",
|
| 48 |
+
"metadata": {
|
| 49 |
+
"execution": {
|
| 50 |
+
"iopub.execute_input": "2025-02-08T19:37:51.115763Z",
|
| 51 |
+
"iopub.status.busy": "2025-02-08T19:37:51.114573Z",
|
| 52 |
+
"iopub.status.idle": "2025-02-08T19:37:51.170027Z",
|
| 53 |
+
"shell.execute_reply": "2025-02-08T19:37:51.169401Z",
|
| 54 |
+
"shell.execute_reply.started": "2025-02-08T19:37:51.115738Z"
|
| 55 |
+
}
|
| 56 |
+
},
|
| 57 |
+
"outputs": [],
|
| 58 |
+
"source": [
|
| 59 |
+
"import os\n",
|
| 60 |
+
"import json\n",
|
| 61 |
+
"import csv\n",
|
| 62 |
+
"import requests\n",
|
| 63 |
+
"from datetime import datetime\n",
|
| 64 |
+
"import time\n",
|
| 65 |
+
"from pathlib import Path\n",
|
| 66 |
+
"import hashlib\n",
|
| 67 |
+
"import pandas as pd\n",
|
| 68 |
+
"import sys\n",
|
| 69 |
+
"from datetime import datetime, timedelta\n",
|
| 70 |
+
"import shutil"
|
| 71 |
+
]
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"cell_type": "code",
|
| 75 |
+
"execution_count": 2,
|
| 76 |
+
"id": "4b2426c3-96a0-468e-b6dc-78dea9c3e92b",
|
| 77 |
+
"metadata": {
|
| 78 |
+
"execution": {
|
| 79 |
+
"iopub.execute_input": "2025-02-08T19:37:51.809027Z",
|
| 80 |
+
"iopub.status.busy": "2025-02-08T19:37:51.808835Z",
|
| 81 |
+
"iopub.status.idle": "2025-02-08T19:37:51.812922Z",
|
| 82 |
+
"shell.execute_reply": "2025-02-08T19:37:51.812429Z",
|
| 83 |
+
"shell.execute_reply.started": "2025-02-08T19:37:51.809009Z"
|
| 84 |
+
}
|
| 85 |
+
},
|
| 86 |
+
"outputs": [],
|
| 87 |
+
"source": [
|
| 88 |
+
"current_dir = Path.cwd()"
|
| 89 |
+
]
|
| 90 |
+
},
|
| 91 |
+
{
|
| 92 |
+
"cell_type": "markdown",
|
| 93 |
+
"id": "11647bb7-5ce9-414a-8486-5bdce8d9cfea",
|
| 94 |
+
"metadata": {
|
| 95 |
+
"execution": {
|
| 96 |
+
"iopub.execute_input": "2025-02-06T12:49:43.126762Z",
|
| 97 |
+
"iopub.status.busy": "2025-02-06T12:49:43.125797Z",
|
| 98 |
+
"iopub.status.idle": "2025-02-06T12:49:43.129759Z",
|
| 99 |
+
"shell.execute_reply": "2025-02-06T12:49:43.129176Z",
|
| 100 |
+
"shell.execute_reply.started": "2025-02-06T12:49:43.126736Z"
|
| 101 |
+
}
|
| 102 |
+
},
|
| 103 |
+
"source": [
|
| 104 |
+
"## MODELS"
|
| 105 |
+
]
|
| 106 |
+
},
|
| 107 |
+
{
|
| 108 |
+
"cell_type": "markdown",
|
| 109 |
+
"id": "5c9cb5e3-7cea-4574-9319-f3cd89354b1f",
|
| 110 |
+
"metadata": {
|
| 111 |
+
"execution": {
|
| 112 |
+
"iopub.execute_input": "2025-02-06T13:23:38.017078Z",
|
| 113 |
+
"iopub.status.busy": "2025-02-06T13:23:38.016639Z",
|
| 114 |
+
"iopub.status.idle": "2025-02-06T13:23:38.019993Z",
|
| 115 |
+
"shell.execute_reply": "2025-02-06T13:23:38.019549Z",
|
| 116 |
+
"shell.execute_reply.started": "2025-02-06T13:23:38.017053Z"
|
| 117 |
+
}
|
| 118 |
+
},
|
| 119 |
+
"source": [
|
| 120 |
+
"### Step 1: Scrape model metadata"
|
| 121 |
+
]
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"cell_type": "markdown",
|
| 125 |
+
"id": "83e12b9f-5ae1-407d-9754-5979d837f787",
|
| 126 |
+
"metadata": {
|
| 127 |
+
"execution": {
|
| 128 |
+
"iopub.execute_input": "2025-02-06T14:06:04.857874Z",
|
| 129 |
+
"iopub.status.busy": "2025-02-06T14:06:04.857500Z",
|
| 130 |
+
"iopub.status.idle": "2025-02-06T14:06:04.860438Z",
|
| 131 |
+
"shell.execute_reply": "2025-02-06T14:06:04.860030Z",
|
| 132 |
+
"shell.execute_reply.started": "2025-02-06T14:06:04.857856Z"
|
| 133 |
+
}
|
| 134 |
+
},
|
| 135 |
+
"source": [
|
| 136 |
+
"#### the resulting files will appear in data/raw/model_metadata as *.json"
|
| 137 |
+
]
|
| 138 |
+
},
|
| 139 |
+
{
|
| 140 |
+
"cell_type": "code",
|
| 141 |
+
"execution_count": 12,
|
| 142 |
+
"id": "5db9c00e",
|
| 143 |
+
"metadata": {},
|
| 144 |
+
"outputs": [],
|
| 145 |
+
"source": [
|
| 146 |
+
"key_karussell = current_dir.parent / 'misc/credentials/civitai_api_keys.txt'\n",
|
| 147 |
+
"directory_path = current_dir.parent / 'data/raw/model_metadata/'"
|
| 148 |
+
]
|
| 149 |
+
},
|
| 150 |
+
{
|
| 151 |
+
"cell_type": "code",
|
| 152 |
+
"execution_count": 13,
|
| 153 |
+
"id": "41ee14e0-fb78-4f91-aba1-13faf05af7d8",
|
| 154 |
+
"metadata": {
|
| 155 |
+
"execution": {
|
| 156 |
+
"iopub.execute_input": "2025-02-08T19:37:56.687832Z",
|
| 157 |
+
"iopub.status.busy": "2025-02-08T19:37:56.687251Z",
|
| 158 |
+
"iopub.status.idle": "2025-02-08T19:37:56.696572Z",
|
| 159 |
+
"shell.execute_reply": "2025-02-08T19:37:56.696059Z",
|
| 160 |
+
"shell.execute_reply.started": "2025-02-08T19:37:56.687809Z"
|
| 161 |
+
}
|
| 162 |
+
},
|
| 163 |
+
"outputs": [],
|
| 164 |
+
"source": [
|
| 165 |
+
"import datetime\n",
|
| 166 |
+
"\n",
|
| 167 |
+
"def load_api_keys():\n",
|
| 168 |
+
" \"\"\"Load API keys from a text file, one per line.\"\"\"\n",
|
| 169 |
+
" if not os.path.exists(key_karussell):\n",
|
| 170 |
+
" raise FileNotFoundError(f\"API key file '{API_KEYS_FILE}' not found!\")\n",
|
| 171 |
+
" \n",
|
| 172 |
+
" with open(key_karussell, 'r') as file:\n",
|
| 173 |
+
" keys = [line.strip() for line in file if line.strip()]\n",
|
| 174 |
+
" \n",
|
| 175 |
+
" if not keys:\n",
|
| 176 |
+
" raise ValueError(\"No API keys found in the file!\")\n",
|
| 177 |
+
" \n",
|
| 178 |
+
" return keys\n",
|
| 179 |
+
"\n",
|
| 180 |
+
"def get_model_metadata():\n",
|
| 181 |
+
" base_url = \"https://civitai.com/api/v1/models\"\n",
|
| 182 |
+
" params = {\"sort\": \"Newest\", \"nsfw\": True}\n",
|
| 183 |
+
"\n",
|
| 184 |
+
" # Load API keys\n",
|
| 185 |
+
" api_keys = load_api_keys()\n",
|
| 186 |
+
" key_index = 0 # Start with the first key\n",
|
| 187 |
+
"\n",
|
| 188 |
+
" page_counter = 0\n",
|
| 189 |
+
" max_pages = 300000000 # Adjust as needed\n",
|
| 190 |
+
" os.makedirs(directory_path, exist_ok=True)\n",
|
| 191 |
+
"\n",
|
| 192 |
+
" while True:\n",
|
| 193 |
+
" if page_counter >= max_pages:\n",
|
| 194 |
+
" print(f\"Reached the limit of {max_pages} pages.\")\n",
|
| 195 |
+
" break\n",
|
| 196 |
+
"\n",
|
| 197 |
+
" headers = {\n",
|
| 198 |
+
" \"Accept\": \"application/json\",\n",
|
| 199 |
+
" \"Authorization\": f\"Bearer {api_keys[key_index]}\"\n",
|
| 200 |
+
" }\n",
|
| 201 |
+
"\n",
|
| 202 |
+
" response = requests.get(base_url, headers=headers, params=params)\n",
|
| 203 |
+
"\n",
|
| 204 |
+
" if response.status_code == 200:\n",
|
| 205 |
+
" data = response.json()\n",
|
| 206 |
+
" page_counter += 1\n",
|
| 207 |
+
"\n",
|
| 208 |
+
" # Add timestamp\n",
|
| 209 |
+
" formatted_timestamp = datetime.datetime.now().strftime(\"data obtained on the %d.%m.%Y at %H:%M CEST\")\n",
|
| 210 |
+
"\n",
|
| 211 |
+
" data['timestamp'] = formatted_timestamp\n",
|
| 212 |
+
"\n",
|
| 213 |
+
" # Save data to file\n",
|
| 214 |
+
" file_path = os.path.join(directory_path, f'newest_models_{page_counter}.json')\n",
|
| 215 |
+
" with open(file_path, 'w', encoding='utf-8') as file:\n",
|
| 216 |
+
" json.dump(data, file, indent=4)\n",
|
| 217 |
+
"\n",
|
| 218 |
+
" # Check for nextCursor\n",
|
| 219 |
+
" next_cursor = data.get('metadata', {}).get('nextCursor')\n",
|
| 220 |
+
" if not next_cursor:\n",
|
| 221 |
+
" print(\"No more data available.\")\n",
|
| 222 |
+
" break\n",
|
| 223 |
+
" else:\n",
|
| 224 |
+
" params['cursor'] = next_cursor\n",
|
| 225 |
+
" \n",
|
| 226 |
+
" elif response.status_code in (401, 403): # Unauthorized or Forbidden\n",
|
| 227 |
+
" print(f\"API Key {key_index + 1} failed with status {response.status_code}. Trying next key...\")\n",
|
| 228 |
+
" key_index += 1\n",
|
| 229 |
+
"\n",
|
| 230 |
+
" if key_index >= len(api_keys):\n",
|
| 231 |
+
" print(\"All API keys failed. Exiting.\")\n",
|
| 232 |
+
" break # Stop if all keys fail\n",
|
| 233 |
+
" \n",
|
| 234 |
+
" else:\n",
|
| 235 |
+
" print(f\"Failed to fetch data: HTTP {response.status_code}\")\n",
|
| 236 |
+
" break # Stop on other errors\n"
|
| 237 |
+
]
|
| 238 |
+
},
|
| 239 |
+
{
|
| 240 |
+
"cell_type": "markdown",
|
| 241 |
+
"id": "8f43ce23-5986-4b67-be2d-453831af9a6e",
|
| 242 |
+
"metadata": {},
|
| 243 |
+
"source": [
|
| 244 |
+
"uncomment this to get model metadata"
|
| 245 |
+
]
|
| 246 |
+
},
|
| 247 |
+
{
|
| 248 |
+
"cell_type": "code",
|
| 249 |
+
"execution_count": null,
|
| 250 |
+
"id": "3f95b4ba-5742-4268-b2e8-de9145faf495",
|
| 251 |
+
"metadata": {
|
| 252 |
+
"execution": {
|
| 253 |
+
"iopub.execute_input": "2025-02-08T19:37:58.117386Z",
|
| 254 |
+
"iopub.status.busy": "2025-02-08T19:37:58.117158Z",
|
| 255 |
+
"iopub.status.idle": "2025-02-08T19:37:58.121162Z",
|
| 256 |
+
"shell.execute_reply": "2025-02-08T19:37:58.120580Z",
|
| 257 |
+
"shell.execute_reply.started": "2025-02-08T19:37:58.117369Z"
|
| 258 |
+
}
|
| 259 |
+
},
|
| 260 |
+
"outputs": [],
|
| 261 |
+
"source": [
|
| 262 |
+
"get_model_metadata()"
|
| 263 |
+
]
|
| 264 |
+
},
|
| 265 |
+
{
|
| 266 |
+
"cell_type": "markdown",
|
| 267 |
+
"id": "1ed5f44e-612e-4b6d-8974-904fb3e058d6",
|
| 268 |
+
"metadata": {
|
| 269 |
+
"execution": {
|
| 270 |
+
"iopub.execute_input": "2025-02-06T12:52:40.162173Z",
|
| 271 |
+
"iopub.status.busy": "2025-02-06T12:52:40.159989Z",
|
| 272 |
+
"iopub.status.idle": "2025-02-06T12:52:40.170634Z",
|
| 273 |
+
"shell.execute_reply": "2025-02-06T12:52:40.169945Z",
|
| 274 |
+
"shell.execute_reply.started": "2025-02-06T12:52:40.162124Z"
|
| 275 |
+
}
|
| 276 |
+
},
|
| 277 |
+
"source": [
|
| 278 |
+
"### Step 2 Consolidate Model-dataset CSV"
|
| 279 |
+
]
|
| 280 |
+
},
|
| 281 |
+
{
|
| 282 |
+
"cell_type": "code",
|
| 283 |
+
"execution_count": 8,
|
| 284 |
+
"id": "464d82f5-c24e-4b53-9b50-682fa4bf3430",
|
| 285 |
+
"metadata": {
|
| 286 |
+
"execution": {
|
| 287 |
+
"iopub.execute_input": "2025-02-08T19:37:59.052245Z",
|
| 288 |
+
"iopub.status.busy": "2025-02-08T19:37:59.051645Z",
|
| 289 |
+
"iopub.status.idle": "2025-02-08T19:37:59.056017Z",
|
| 290 |
+
"shell.execute_reply": "2025-02-08T19:37:59.055598Z",
|
| 291 |
+
"shell.execute_reply.started": "2025-02-08T19:37:59.052224Z"
|
| 292 |
+
}
|
| 293 |
+
},
|
| 294 |
+
"outputs": [],
|
| 295 |
+
"source": [
|
| 296 |
+
"## path thingy\n",
|
| 297 |
+
"try: #scripts\n",
|
| 298 |
+
" current_dir = Path(__file__).resolve().parent\n",
|
| 299 |
+
"except NameError:\n",
|
| 300 |
+
" # jupyter\n",
|
| 301 |
+
" current_dir = Path.cwd()"
|
| 302 |
+
]
|
| 303 |
+
},
|
| 304 |
+
{
|
| 305 |
+
"cell_type": "code",
|
| 306 |
+
"execution_count": null,
|
| 307 |
+
"id": "14f3db4d-ef93-4629-8a27-971adb49248b",
|
| 308 |
+
"metadata": {
|
| 309 |
+
"execution": {
|
| 310 |
+
"iopub.execute_input": "2025-02-08T19:37:59.443457Z",
|
| 311 |
+
"iopub.status.busy": "2025-02-08T19:37:59.443236Z",
|
| 312 |
+
"iopub.status.idle": "2025-02-08T19:37:59.890103Z",
|
| 313 |
+
"shell.execute_reply": "2025-02-08T19:37:59.889571Z",
|
| 314 |
+
"shell.execute_reply.started": "2025-02-08T19:37:59.443439Z"
|
| 315 |
+
}
|
| 316 |
+
},
|
| 317 |
+
"outputs": [
|
| 318 |
+
{
|
| 319 |
+
"name": "stdout",
|
| 320 |
+
"output_type": "stream",
|
| 321 |
+
"text": [
|
| 322 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_2.json\n",
|
| 323 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_1.json\n",
|
| 324 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_4.json\n",
|
| 325 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_6.json\n",
|
| 326 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_7.json\n",
|
| 327 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_5.json\n",
|
| 328 |
+
"Processing file: /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/raw/model_metadata/newest_models_3.json\n"
|
| 329 |
+
]
|
| 330 |
+
}
|
| 331 |
+
],
|
| 332 |
+
"source": [
|
| 333 |
+
"import os\n",
|
| 334 |
+
"import json\n",
|
| 335 |
+
"import pandas as pd\n",
|
| 336 |
+
"import hashlib\n",
|
| 337 |
+
"from pathlib import Path\n",
|
| 338 |
+
"from datetime import datetime, timezone\n",
|
| 339 |
+
"\n",
|
| 340 |
+
"\n",
|
| 341 |
+
"\n",
|
| 342 |
+
"\n",
|
| 343 |
+
"def hash_username(username):\n",
|
| 344 |
+
" return hashlib.sha256(username.encode('utf-8')).hexdigest()[:16]\n",
|
| 345 |
+
"\n",
|
| 346 |
+
"def parse_date(date_str):\n",
|
| 347 |
+
" try:\n",
|
| 348 |
+
" return datetime.fromisoformat(date_str.replace('Z', '+00:00'))\n",
|
| 349 |
+
" except Exception:\n",
|
| 350 |
+
" return datetime.min.replace(tzinfo=timezone.utc) # Make it timezone-aware\n",
|
| 351 |
+
"\n",
|
| 352 |
+
"def get_latest_model_version(model_versions):\n",
|
| 353 |
+
" return max(model_versions, key=lambda mv: parse_date(mv.get('publishedAt', '')))\n",
|
| 354 |
+
"\n",
|
| 355 |
+
"def process_directory_recursively(root_dir):\n",
|
| 356 |
+
" root_path = Path(root_dir)\n",
|
| 357 |
+
" seen = {} # id -> (publishedAt, record)\n",
|
| 358 |
+
" data_records = []\n",
|
| 359 |
+
"\n",
|
| 360 |
+
" for json_file in root_path.rglob('*.json'):\n",
|
| 361 |
+
" if not json_file.is_file():\n",
|
| 362 |
+
" continue\n",
|
| 363 |
+
"\n",
|
| 364 |
+
" #print(f\"Processing file: {json_file}\")\n",
|
| 365 |
+
" try:\n",
|
| 366 |
+
" with open(json_file, 'r', encoding='utf-8') as f:\n",
|
| 367 |
+
" data = json.load(f)\n",
|
| 368 |
+
" except Exception as e:\n",
|
| 369 |
+
" print(f\"Failed to load {json_file}: {e}\")\n",
|
| 370 |
+
" continue\n",
|
| 371 |
+
"\n",
|
| 372 |
+
" items = data.get('items') or data.get('data') or []\n",
|
| 373 |
+
" for item in items:\n",
|
| 374 |
+
" if not isinstance(item, dict):\n",
|
| 375 |
+
" continue\n",
|
| 376 |
+
"\n",
|
| 377 |
+
" model_id = item.get('id')\n",
|
| 378 |
+
" model_versions = item.get('modelVersions', [])\n",
|
| 379 |
+
" if not model_versions:\n",
|
| 380 |
+
" continue\n",
|
| 381 |
+
"\n",
|
| 382 |
+
" latest_version = get_latest_model_version(model_versions)\n",
|
| 383 |
+
" published_at = latest_version.get('publishedAt', '')\n",
|
| 384 |
+
" current_dt = parse_date(published_at)\n",
|
| 385 |
+
"\n",
|
| 386 |
+
" if model_id in seen and current_dt <= seen[model_id][0]:\n",
|
| 387 |
+
" continue\n",
|
| 388 |
+
" seen[model_id] = (current_dt, item)\n",
|
| 389 |
+
"\n",
|
| 390 |
+
" for model_id, (_, item) in seen.items():\n",
|
| 391 |
+
" model_versions = item.get('modelVersions', [])\n",
|
| 392 |
+
" latest_version = get_latest_model_version(model_versions)\n",
|
| 393 |
+
" version_ids = [mv.get('id', '') for mv in model_versions[:20]]\n",
|
| 394 |
+
"\n",
|
| 395 |
+
" files = latest_version.get('files', [])\n",
|
| 396 |
+
" auto_hashes = files[0].get('hashes', {}) if files else {}\n",
|
| 397 |
+
" images = latest_version.get('images', [])\n",
|
| 398 |
+
" first_image_url = images[0]['url'] if images else ''\n",
|
| 399 |
+
" latest_image_url = images[-1]['url'] if images else ''\n",
|
| 400 |
+
"\n",
|
| 401 |
+
" username = item.get('creator', {}).get('username', '')\n",
|
| 402 |
+
" record = {\n",
|
| 403 |
+
" 'id': item.get('id', ''),\n",
|
| 404 |
+
" 'name': item.get('name', ''),\n",
|
| 405 |
+
" 'type': item.get('type', ''),\n",
|
| 406 |
+
" 'baseModel': latest_version.get('baseModel', ''),\n",
|
| 407 |
+
" 'downloadCount': item.get('stats', {}).get('downloadCount', 0),\n",
|
| 408 |
+
" 'nsfwLevel': item.get('nsfwLevel', 0),\n",
|
| 409 |
+
" 'modelVersions': len(model_versions),\n",
|
| 410 |
+
" 'publishedAt': latest_version.get('publishedAt', ''),\n",
|
| 411 |
+
" 'usernameHash': hash_username(username) if username else '',\n",
|
| 412 |
+
" 'downloadUrl': latest_version.get('downloadUrl', ''),\n",
|
| 413 |
+
" 'firstImageUrl': first_image_url,\n",
|
| 414 |
+
" 'latestImageUrl': latest_image_url,\n",
|
| 415 |
+
" 'poi': item.get('poi', False),\n",
|
| 416 |
+
" 'AutoV1': auto_hashes.get('AutoV1', ''),\n",
|
| 417 |
+
" 'AutoV2': auto_hashes.get('AutoV2', ''),\n",
|
| 418 |
+
" 'AutoV3': auto_hashes.get('AutoV3', ''),\n",
|
| 419 |
+
" 'SHA256': auto_hashes.get('SHA256', ''),\n",
|
| 420 |
+
" 'CRC32': auto_hashes.get('CRC32', ''),\n",
|
| 421 |
+
" 'BLAKE3': auto_hashes.get('BLAKE3', ''),\n",
|
| 422 |
+
" 'previewImage': latest_image_url\n",
|
| 423 |
+
" }\n",
|
| 424 |
+
"\n",
|
| 425 |
+
" for i in range(20):\n",
|
| 426 |
+
" record[f'version_id_{i+1}'] = version_ids[i] if i < len(version_ids) else ''\n",
|
| 427 |
+
"\n",
|
| 428 |
+
" tags = item.get('tags', [])\n",
|
| 429 |
+
" for i in range(7):\n",
|
| 430 |
+
" record[f'tag_{i+1}'] = tags[i] if i < len(tags) else ''\n",
|
| 431 |
+
"\n",
|
| 432 |
+
" data_records.append(record)\n",
|
| 433 |
+
"\n",
|
| 434 |
+
" return pd.DataFrame(data_records)\n",
|
| 435 |
+
"\n",
|
| 436 |
+
"# Usage Example\n",
|
| 437 |
+
"root_directory = current_dir.parent / 'data/raw/model_metadata/'\n",
|
| 438 |
+
"df = process_directory_recursively(root_directory)\n",
|
| 439 |
+
"df_sorted = df.sort_values(by='downloadCount', ascending=False)\n",
|
| 440 |
+
"\n",
|
| 441 |
+
"# Optionally save\n",
|
| 442 |
+
"# df_sorted.to_csv('combined_metadata.csv', index=False)\n"
|
| 443 |
+
]
|
| 444 |
+
},
|
| 445 |
+
{
|
| 446 |
+
"cell_type": "markdown",
|
| 447 |
+
"id": "2d6f822f-754f-4feb-9ed6-6f868421c454",
|
| 448 |
+
"metadata": {},
|
| 449 |
+
"source": [
|
| 450 |
+
"### Save model-data to CSV"
|
| 451 |
+
]
|
| 452 |
+
},
|
| 453 |
+
{
|
| 454 |
+
"cell_type": "code",
|
| 455 |
+
"execution_count": null,
|
| 456 |
+
"id": "f861cf7b-5eb3-46ad-ab2d-9e2fdf2169b4",
|
| 457 |
+
"metadata": {
|
| 458 |
+
"execution": {
|
| 459 |
+
"iopub.execute_input": "2025-02-08T19:38:02.047421Z",
|
| 460 |
+
"iopub.status.busy": "2025-02-08T19:38:02.046761Z",
|
| 461 |
+
"iopub.status.idle": "2025-02-08T19:38:02.193381Z",
|
| 462 |
+
"shell.execute_reply": "2025-02-08T19:38:02.192886Z",
|
| 463 |
+
"shell.execute_reply.started": "2025-02-08T19:38:02.047397Z"
|
| 464 |
+
}
|
| 465 |
+
},
|
| 466 |
+
"outputs": [
|
| 467 |
+
{
|
| 468 |
+
"name": "stdout",
|
| 469 |
+
"output_type": "stream",
|
| 470 |
+
"text": [
|
| 471 |
+
"Data has been saved to /shares/weddigen.ki.uzh/laura_wagner/Civitai_page_analysis/Civitai_visualizations/data/CSV/Civiverse-Models.csv\n"
|
| 472 |
+
]
|
| 473 |
+
}
|
| 474 |
+
],
|
| 475 |
+
"source": [
|
| 476 |
+
"output_csv = current_dir.parent / 'data/CSV/Civiverse-Models_2025.csv'\n",
|
| 477 |
+
"output_csv.parent.mkdir(parents=True, exist_ok=True)\n",
|
| 478 |
+
"df_sorted.to_csv(output_csv, index=False)\n",
|
| 479 |
+
"print(f\"Data has been saved to {output_csv}\")"
|
| 480 |
+
]
|
| 481 |
+
},
|
| 482 |
+
{
|
| 483 |
+
"cell_type": "markdown",
|
| 484 |
+
"id": "9b948933-2a2f-42b0-9083-2d81586ae3f2",
|
| 485 |
+
"metadata": {
|
| 486 |
+
"execution": {
|
| 487 |
+
"iopub.execute_input": "2025-02-06T13:42:32.447860Z",
|
| 488 |
+
"iopub.status.busy": "2025-02-06T13:42:32.446832Z",
|
| 489 |
+
"iopub.status.idle": "2025-02-06T13:42:32.453827Z",
|
| 490 |
+
"shell.execute_reply": "2025-02-06T13:42:32.453271Z",
|
| 491 |
+
"shell.execute_reply.started": "2025-02-06T13:42:32.447821Z"
|
| 492 |
+
}
|
| 493 |
+
},
|
| 494 |
+
"source": [
|
| 495 |
+
"### Step 3 Create Subsets: Checkpoint only, POI True, POI False"
|
| 496 |
+
]
|
| 497 |
+
},
|
| 498 |
+
{
|
| 499 |
+
"cell_type": "code",
|
| 500 |
+
"execution_count": 11,
|
| 501 |
+
"id": "ae894e26-4984-40e6-80a6-54fb6b61c873",
|
| 502 |
+
"metadata": {
|
| 503 |
+
"execution": {
|
| 504 |
+
"iopub.execute_input": "2025-02-08T19:38:03.089160Z",
|
| 505 |
+
"iopub.status.busy": "2025-02-08T19:38:03.088490Z",
|
| 506 |
+
"iopub.status.idle": "2025-02-08T19:38:03.093067Z",
|
| 507 |
+
"shell.execute_reply": "2025-02-08T19:38:03.092634Z",
|
| 508 |
+
"shell.execute_reply.started": "2025-02-08T19:38:03.089138Z"
|
| 509 |
+
}
|
| 510 |
+
},
|
| 511 |
+
"outputs": [],
|
| 512 |
+
"source": [
|
| 513 |
+
"file_path = current_dir.parent / 'data/CSV/Civiverse-Models.csv' # Update this with your actual file path\n",
|
| 514 |
+
"(current_dir.parent / 'data/CSV/model_subsets').mkdir(parents=True, exist_ok=True)\n"
|
| 515 |
+
]
|
| 516 |
+
},
|
| 517 |
+
{
|
| 518 |
+
"cell_type": "code",
|
| 519 |
+
"execution_count": 13,
|
| 520 |
+
"id": "88438f88-c723-423e-a4e5-e07f59096b72",
|
| 521 |
+
"metadata": {
|
| 522 |
+
"execution": {
|
| 523 |
+
"iopub.execute_input": "2025-02-08T19:41:39.279495Z",
|
| 524 |
+
"iopub.status.busy": "2025-02-08T19:41:39.279039Z",
|
| 525 |
+
"iopub.status.idle": "2025-02-08T19:41:39.674445Z",
|
| 526 |
+
"shell.execute_reply": "2025-02-08T19:41:39.673981Z",
|
| 527 |
+
"shell.execute_reply.started": "2025-02-08T19:41:39.279476Z"
|
| 528 |
+
}
|
| 529 |
+
},
|
| 530 |
+
"outputs": [
|
| 531 |
+
{
|
| 532 |
+
"name": "stdout",
|
| 533 |
+
"output_type": "stream",
|
| 534 |
+
"text": [
|
| 535 |
+
"Files saved successfully!\n"
|
| 536 |
+
]
|
| 537 |
+
}
|
| 538 |
+
],
|
| 539 |
+
"source": [
|
| 540 |
+
"import pandas as pd\n",
|
| 541 |
+
"\n",
|
| 542 |
+
"# Load the dataset\n",
|
| 543 |
+
"\n",
|
| 544 |
+
"data = pd.read_csv(file_path)\n",
|
| 545 |
+
"\n",
|
| 546 |
+
"# Version 1: Only 'poi' true models\n",
|
| 547 |
+
"poi_true_models = data[data['poi'] == True]\n",
|
| 548 |
+
"\n",
|
| 549 |
+
"# Version 2: Only types lora, dora, locon, textual inversion\n",
|
| 550 |
+
"specific_types = ['LORA', 'DORA', 'LOCON', 'textualInversion']\n",
|
| 551 |
+
"adapters = data[data['type'].isin(specific_types)]\n",
|
| 552 |
+
"\n",
|
| 553 |
+
"# Version 3: Only type checkpoint\n",
|
| 554 |
+
"checkpoint_models = data[data['type'] == 'Checkpoint']\n",
|
| 555 |
+
"\n",
|
| 556 |
+
"# Version 4: All models apart from 'poi' true\n",
|
| 557 |
+
"non_poi_models = data[data['poi'] != True]\n",
|
| 558 |
+
"\n",
|
| 559 |
+
"# Version 5: All models apart from 'poi' true and with nsfwLevel below 13\n",
|
| 560 |
+
"non_poi_low_nsfw_models = data[(data['poi'] != True) & (data['nsfwLevel'] < 13)]\n",
|
| 561 |
+
"\n",
|
| 562 |
+
"# Save the versions as separate CSV files\n",
|
| 563 |
+
"poi_true_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_adapters_poi_true.csv', index=False)\n",
|
| 564 |
+
"adapters.to_csv(current_dir.parent / 'data/CSV/adapters.csv', index=False)\n",
|
| 565 |
+
"checkpoint_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_checkpoint_only.csv', index=False)\n",
|
| 566 |
+
"non_poi_models.to_csv(current_dir.parent / 'data/CSV/model_subsets/Civiverse_adapters_poi_false.csv', index=False)\n",
|
| 567 |
+
"\n",
|
| 568 |
+
"print(\"Files saved successfully!\")\n"
|
| 569 |
+
]
|
| 570 |
+
},
|
| 571 |
+
{
|
| 572 |
+
"cell_type": "code",
|
| 573 |
+
"execution_count": null,
|
| 574 |
+
"id": "49e35088-8b2e-4189-83e1-3098d55dcad2",
|
| 575 |
+
"metadata": {},
|
| 576 |
+
"outputs": [],
|
| 577 |
+
"source": [
|
| 578 |
+
"import pandas as pd\n",
|
| 579 |
+
"import os\n",
|
| 580 |
+
"\n",
|
| 581 |
+
"# Load the dataset\n",
|
| 582 |
+
"df = pd.read_csv('data/all_models_with_tags.csv')\n",
|
| 583 |
+
"\n",
|
| 584 |
+
"# Filter for rows where poi is True\n",
|
| 585 |
+
"filtered_df = df[df['poi'] == True]\n",
|
| 586 |
+
"os.makedirs('data/model_subsets', exist_ok=True)\n",
|
| 587 |
+
"\n",
|
| 588 |
+
"# Save the filtered DataFrame to a new CSV file\n",
|
| 589 |
+
"filtered_df.to_csv('data/model_subsets/all_models_poi.csv', index=False)\n"
|
| 590 |
+
]
|
| 591 |
+
},
|
| 592 |
+
{
|
| 593 |
+
"cell_type": "code",
|
| 594 |
+
"execution_count": null,
|
| 595 |
+
"id": "06c15f2c",
|
| 596 |
+
"metadata": {},
|
| 597 |
+
"outputs": [],
|
| 598 |
+
"source": [
|
| 599 |
+
"import pandas as pd\n",
|
| 600 |
+
"import os\n",
|
| 601 |
+
"\n",
|
| 602 |
+
"# Load the dataset\n",
|
| 603 |
+
"df = pd.read_csv('data/all_models_with_tags.csv')\n",
|
| 604 |
+
"\n",
|
| 605 |
+
"# Filter for rows where poi is True\n",
|
| 606 |
+
"filtered_df = df[df['poi'] == False]\n",
|
| 607 |
+
"os.makedirs('data/model_subsets', exist_ok=True)\n",
|
| 608 |
+
"\n",
|
| 609 |
+
"# Save the filtered DataFrame to a new CSV file\n",
|
| 610 |
+
"filtered_df.to_csv('data/model_subsets/all_models_poi_false.csv', index=False)\n"
|
| 611 |
+
]
|
| 612 |
+
}
|
| 613 |
+
],
|
| 614 |
+
"metadata": {
|
| 615 |
+
"kernelspec": {
|
| 616 |
+
"display_name": "latm",
|
| 617 |
+
"language": "python",
|
| 618 |
+
"name": "python3"
|
| 619 |
+
},
|
| 620 |
+
"language_info": {
|
| 621 |
+
"codemirror_mode": {
|
| 622 |
+
"name": "ipython",
|
| 623 |
+
"version": 3
|
| 624 |
+
},
|
| 625 |
+
"file_extension": ".py",
|
| 626 |
+
"mimetype": "text/x-python",
|
| 627 |
+
"name": "python",
|
| 628 |
+
"nbconvert_exporter": "python",
|
| 629 |
+
"pygments_lexer": "ipython3",
|
| 630 |
+
"version": "3.10.15"
|
| 631 |
+
}
|
| 632 |
+
},
|
| 633 |
+
"nbformat": 4,
|
| 634 |
+
"nbformat_minor": 5
|
| 635 |
+
}
|
jupyter_notebooks/Section_1_Figure_1_image_grid.ipynb
CHANGED
|
@@ -367,7 +367,7 @@
|
|
| 367 |
],
|
| 368 |
"metadata": {
|
| 369 |
"kernelspec": {
|
| 370 |
-
"display_name": "
|
| 371 |
"language": "python",
|
| 372 |
"name": "python3"
|
| 373 |
},
|
|
@@ -381,7 +381,7 @@
|
|
| 381 |
"name": "python",
|
| 382 |
"nbconvert_exporter": "python",
|
| 383 |
"pygments_lexer": "ipython3",
|
| 384 |
-
"version": "3.
|
| 385 |
}
|
| 386 |
},
|
| 387 |
"nbformat": 4,
|
|
|
|
| 367 |
],
|
| 368 |
"metadata": {
|
| 369 |
"kernelspec": {
|
| 370 |
+
"display_name": "latm",
|
| 371 |
"language": "python",
|
| 372 |
"name": "python3"
|
| 373 |
},
|
|
|
|
| 381 |
"name": "python",
|
| 382 |
"nbconvert_exporter": "python",
|
| 383 |
"pygments_lexer": "ipython3",
|
| 384 |
+
"version": "3.10.15"
|
| 385 |
}
|
| 386 |
},
|
| 387 |
"nbformat": 4,
|
jupyter_notebooks/Section_3-4_extract_LoRA_metadata.ipynb
CHANGED
|
@@ -13,7 +13,7 @@
|
|
| 13 |
}
|
| 14 |
},
|
| 15 |
"source": [
|
| 16 |
-
"# Section
|
| 17 |
]
|
| 18 |
},
|
| 19 |
{
|
|
|
|
| 13 |
}
|
| 14 |
},
|
| 15 |
"source": [
|
| 16 |
+
"# Section 3-4: LoRA metadata"
|
| 17 |
]
|
| 18 |
},
|
| 19 |
{
|
jupyter_notebooks/{SuppM_Figure_13_Danbooru_taxonomy.ipynb → SuppM_Figure_12_Danbooru_categories.ipynb}
RENAMED
|
File without changes
|