Files changed (1) hide show
  1. README.md +27 -0
README.md ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from datasets import load_dataset
2
+ import soundfile as sf
3
+ import os
4
+
5
+ # Load the dataset
6
+ dataset = load_dataset('simon3000/genshin-voice', split='train', streaming=True)
7
+
8
+ # Filter the dataset for Chinese voices of Ganyu with transcriptions
9
+ chinese_ganyu = dataset.filter(lambda voice: voice['language'] == 'Chinese' and voice['speaker'] == 'Ganyu' and voice['transcription'] != '')
10
+
11
+ # Create a folder to store the audio and transcription files
12
+ ganyu_folder = 'ganyu'
13
+ os.makedirs(ganyu_folder, exist_ok=True)
14
+
15
+ # Process each voice in the filtered dataset
16
+ for i, voice in enumerate(chinese_ganyu):
17
+ audio_path = os.path.join(ganyu_folder, f'{i}_audio.wav') # Path to save the audio file
18
+ transcription_path = os.path.join(ganyu_folder, f'{i}_transcription.txt') # Path to save the transcription file
19
+
20
+ # Save the audio file
21
+ sf.write(audio_path, voice['audio']['array'], voice['audio']['sampling_rate'])
22
+
23
+ # Save the transcription file
24
+ with open(transcription_path, 'w') as transcription_file:
25
+ transcription_file.write(voice['transcription'])
26
+
27
+ print(f'{i} done') # Print the progress