Ganyu
#1
by hsrgi - opened
README.md
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from datasets import load_dataset
|
| 2 |
+
import soundfile as sf
|
| 3 |
+
import os
|
| 4 |
+
|
| 5 |
+
# Load the dataset
|
| 6 |
+
dataset = load_dataset('simon3000/genshin-voice', split='train', streaming=True)
|
| 7 |
+
|
| 8 |
+
# Filter the dataset for Chinese voices of Ganyu with transcriptions
|
| 9 |
+
chinese_ganyu = dataset.filter(lambda voice: voice['language'] == 'Chinese' and voice['speaker'] == 'Ganyu' and voice['transcription'] != '')
|
| 10 |
+
|
| 11 |
+
# Create a folder to store the audio and transcription files
|
| 12 |
+
ganyu_folder = 'ganyu'
|
| 13 |
+
os.makedirs(ganyu_folder, exist_ok=True)
|
| 14 |
+
|
| 15 |
+
# Process each voice in the filtered dataset
|
| 16 |
+
for i, voice in enumerate(chinese_ganyu):
|
| 17 |
+
audio_path = os.path.join(ganyu_folder, f'{i}_audio.wav') # Path to save the audio file
|
| 18 |
+
transcription_path = os.path.join(ganyu_folder, f'{i}_transcription.txt') # Path to save the transcription file
|
| 19 |
+
|
| 20 |
+
# Save the audio file
|
| 21 |
+
sf.write(audio_path, voice['audio']['array'], voice['audio']['sampling_rate'])
|
| 22 |
+
|
| 23 |
+
# Save the transcription file
|
| 24 |
+
with open(transcription_path, 'w') as transcription_file:
|
| 25 |
+
transcription_file.write(voice['transcription'])
|
| 26 |
+
|
| 27 |
+
print(f'{i} done') # Print the progress
|