File size: 5,171 Bytes
dd9b42b
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
'''
このコードはPythonで学ぶ音源分離(第5章)のサンプルコード5-2を改変したものです.
https://raw.githubusercontent.com/masahitotogami/python_source_separation/master/section5/sample_code_c5_2.py
'''
import os
import requests
import tarfile
import wave as wave
import pyroomacoustics as pa
import numpy as np
import random


def download_and_extract_cmu_arctic():
    folder_name = "cmu_arctic_concat15"
    tar_filename = "cmu_arctic_concat15.tar.gz"
    url = "https://zenodo.org/records/3066489/files/cmu_arctic_concat15.tar.gz?download=1"

    # 既にフォルダがある場合は処理しない
    if os.path.exists(folder_name):
        print(f"📂 フォルダ '{folder_name}' は既に存在しています。ダウンロードしません。")
        return

    print(f"⬇️ '{folder_name}' が見つかりません。ダウンロード開始...")

    # ===== ダウンロード =====
    response = requests.get(url, stream=True)
    response.raise_for_status()

    with open(tar_filename, "wb") as f:
        for chunk in response.iter_content(chunk_size=8192):
            f.write(chunk)

    print(f"📦 ダウンロード完了: {tar_filename}")

    # ===== 展開 =====
    print(f"📂 展開中...")
    with tarfile.open(tar_filename, "r:gz") as tar:
        tar.extractall()

    print("🎉 展開完了!")

    # ===== ダウンロードファイル削除 =====
    os.remove(tar_filename)
    print(f"🧹 掃除完了: {tar_filename} を削除しました")


def main(audio_output="mixed_audio.wav"):
    # CMU Arcticデータセットをダウンロード・展開
    download_and_extract_cmu_arctic()

    # 乱数の種を初期化
    np.random.seed(0)

    # ===== ① 使用する音源ファイル =====
    base_dir = "cmu_arctic_concat15/"
    files = os.listdir(base_dir)
    wav_files = [
        base_dir + f for f in files
        if not f.startswith(".") and f.endswith(".wav")
    ]
    clean_wave_files = random.sample(wav_files, 2)

    # 音源数
    n_sources = len(clean_wave_files)

    # 長さを調べる
    n_samples = 0
    n_samples_min = 1e10
    for clean_wave_file in clean_wave_files:
        wav = wave.open(clean_wave_file)
        if n_samples < wav.getnframes():
            n_samples = wav.getnframes()
        n_samples_min = min(n_samples_min, wav.getnframes())
        wav.close()

    clean_data = np.zeros([n_sources, n_samples])

    # ファイルを読み込む
    s = 0
    for clean_wave_file in clean_wave_files:
        wav = wave.open(clean_wave_file)
        data = wav.readframes(wav.getnframes())
        data = np.frombuffer(data, dtype=np.int16)
        data = data / np.iinfo(np.int16).max  # -1〜1に正規化
        clean_data[s, :wav.getnframes()] = data
        wav.close()
        s = s + 1

    # ===== シミュレーションのパラメータ =====

    sample_rate = 16000   # サンプリング周波数
    SNR = 50.             # 音声と雑音の比率 [dB]
    room_dim = np.r_[10.0, 10.0, 10.0]  # 部屋の大きさ

    # マイクロホンアレイを置く部屋の場所
    mic_array_loc = room_dim / 2 + np.random.randn(3) * 0.1

    # マイクロホンアレイのマイク配置(2ch固定)
    mic_alignments = np.array(
        [
            [-0.01, 0.0, 0.0],
            [0.01, 0.0, 0.0],
        ]
    )

    n_channels = mic_alignments.shape[0]
    R = mic_alignments.T + mic_array_loc[:, None]

    # 部屋を生成する
    room = pa.ShoeBox(room_dim, fs=sample_rate, max_order=0)

    # マイクロホンアレイの情報を設定
    room.add_microphone_array(pa.MicrophoneArray(R, fs=room.fs))

    # 音源の場所
    doas = np.array(
        [[np.pi/2., 0],
        [np.pi/2., np.pi/2.]]
    )
    distance = 1.
    source_locations = np.zeros((3, doas.shape[0]), dtype=doas.dtype)
    source_locations[0, :] = np.cos(doas[:, 1]) * np.sin(doas[:, 0])
    source_locations[1, :] = np.sin(doas[:, 1]) * np.sin(doas[:, 0])
    source_locations[2, :] = np.cos(doas[:, 0])
    source_locations *= distance
    source_locations += mic_array_loc[:, None]

    # 各音源を部屋に追加
    for s in range(n_sources):
        clean_data[s] /= np.std(clean_data[s])
        room.add_source(source_locations[:, s], signal=clean_data[s])

    # ===== シミュレーション実行 =====
    room.simulate(snr=SNR)

    # ===== ステレオ1つのファイルとして保存 =====
    multi_conv_data = room.mic_array.signals  # (channels, samples)

    # int16スケール調整
    data_scaled_L = (multi_conv_data[0] * np.iinfo(np.int16).max / 10).astype(np.int16)
    data_scaled_R = (multi_conv_data[1] * np.iinfo(np.int16).max / 10).astype(np.int16)

    # ステレオ結合
    stereo_data = np.stack([data_scaled_L, data_scaled_R], axis=1)[:n_samples_min]

    # ===== 保存 =====
    out = wave.open(audio_output, "w")
    out.setnchannels(2)
    out.setsampwidth(2)      # int16
    out.setframerate(sample_rate)
    out.writeframes(stereo_data.tobytes())
    out.close()

    print(f"🎧 {audio_output} を生成しました(2chステレオ)")

if __name__ == "__main__":
    main()