batoon commited on
Commit
6c2ab00
·
verified ·
1 Parent(s): cce9569

Automatically fit sampled bass register with GM fallback

Browse files
Files changed (4) hide show
  1. README.md +5 -3
  2. app.py +1 -1
  3. pipeline.py +35 -9
  4. test_sampler.py +30 -4
README.md CHANGED
@@ -20,9 +20,11 @@ Upload a short hum or record one in the browser. One request generates seven
20
  source-timed comparisons. SheetSage2 extracts notes and beats; FluidSynth
21
  uses the full [FluidR3 GM bank](https://www.fluidsynth.org/wiki/GettingStarted/)
22
  for the selected instrument and drum kit, plus sampled electric
23
- guitar and fingered bass in clean and processed versions. The bass is shifted
24
- down two octaves into the actual sample range, with note starts and lengths
25
- preserved. A continuous-F0 synth
 
 
26
  follows voice pitch and amplitude. The sampler uses full-size FreePats electric
27
  guitar and fingered bass SoundFonts, with FFmpeg compression, filtering,
28
  oversampled soft clipping, and a short guitar room echo in processed versions.
 
20
  source-timed comparisons. SheetSage2 extracts notes and beats; FluidSynth
21
  uses the full [FluidR3 GM bank](https://www.fluidsynth.org/wiki/GettingStarted/)
22
  for the selected instrument and drum kit, plus sampled electric
23
+ guitar and fingered bass in clean and processed versions. The bass octave is
24
+ chosen automatically to fit all notes into the pinned sample's measured MIDI
25
+ range 26..45. If no single octave fits the whole phrase, the route uses the
26
+ full GM Fingered Bass bank instead of dropping notes. Note starts and lengths
27
+ are preserved. A continuous-F0 synth
28
  follows voice pitch and amplitude. The sampler uses full-size FreePats electric
29
  guitar and fingered bass SoundFonts, with FFmpeg compression, filtering,
30
  oversampled soft clipping, and a short guitar room echo in processed versions.
app.py CHANGED
@@ -72,7 +72,7 @@ with gr.Blocks(title="Hum · напев в инструмент", css=CSS, theme
72
  go = gr.Button("Создать все варианты", variant="primary", size="lg")
73
  gr.HTML("""<div class='note'><b>Текущая цель:</b> один инструмент и
74
  приблизительно исходное время. Электрогитара и бас используют
75
- записанные сэмплы; бас звучит на две октавы ниже. Обработанные версии
76
  добавляют усиление и лёгкий эффект, но не новые ноты. Барабаны
77
  привязаны к долям SheetSage2. YuE2 может изменить
78
  длительность, добавить человеческие звуки или другие инструменты;
 
72
  go = gr.Button("Создать все варианты", variant="primary", size="lg")
73
  gr.HTML("""<div class='note'><b>Текущая цель:</b> один инструмент и
74
  приблизительно исходное время. Электрогитара и бас используют
75
+ записанные сэмплы; регистр баса выбирается автоматически. Обработанные версии
76
  добавляют усиление и лёгкий эффект, но не новые ноты. Барабаны
77
  привязаны к долям SheetSage2. YuE2 может изменить
78
  длительность, добавить человеческие звуки или другие инструменты;
pipeline.py CHANGED
@@ -20,9 +20,9 @@ YUE_PYTHON = Path(os.environ.get("HUM_YUE_PYTHON", "/opt/yue/bin/python"))
20
  SOUNDFONT_DIR = Path(os.environ.get("HUM_SOUNDFONTS", "/opt/soundfonts"))
21
  GM_SOUNDFONT = Path(os.environ.get("HUM_GM_SOUNDFONT",
22
  "/usr/share/sounds/sf2/FluidR3_GM.sf2"))
23
- # FreePats Finger Bass is silent over much of the voice-minus-one-octave
24
- # range; two octaves put the seen 54..68 melody into its audible 30..44 range.
25
- BASS_TRANSPOSE = -24
26
  SAMPLER_VARIANTS = (
27
  "instrument", "instrument_drums", "contour",
28
  "guitar_clean", "guitar_amp", "bass_clean", "bass_amp",
@@ -134,6 +134,28 @@ def render_instrument(midi_path: Path, folder: Path, instrument: str,
134
  duration, log)
135
 
136
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
137
  def apply_effect(source: Path, output: Path, kind: str, duration: float,
138
  log: Path) -> Path:
139
  # Both chains use the same sampled notes and retain the source duration.
@@ -336,6 +358,8 @@ def generate_all(source: str, instrument: str, styles: dict[str, str], seed: int
336
  yue_timing = {}
337
  drum_count = 0
338
  drums_midi = None
 
 
339
 
340
  def deliver(name: str, raw: Path, *, normalize: bool = True,
341
  max_gain: float = 4.) -> None:
@@ -351,7 +375,8 @@ def generate_all(source: str, instrument: str, styles: dict[str, str], seed: int
351
  "yue_requested": include_yue, "yue_timing": dict(yue_timing),
352
  "errors": dict(errors), "complete": complete,
353
  "source_timing_preserved": list(SAMPLER_VARIANTS),
354
- "bass_transpose_semitones": BASS_TRANSPOSE,
 
355
  "yue_vocal_absence_guaranteed": False,
356
  "elapsed_seconds": round(time.monotonic() - began, 2)}
357
  return (*(outputs[name] for name in VARIANTS), str(melody_midi),
@@ -364,6 +389,7 @@ def generate_all(source: str, instrument: str, styles: dict[str, str], seed: int
364
  if progress:
365
  progress(.12, desc="SheetSage2 извлекает мелодию и сетку долей")
366
  midi, melody_abc, full_abc, beat_lab = transcribe(pcm24, folder, log)
 
367
  if progress:
368
  progress(.30, desc="Синтезирую инструмент и барабаны")
369
  melody_raw, melody_midi = render_instrument(midi, folder, instrument, duration, log)
@@ -388,13 +414,13 @@ def generate_all(source: str, instrument: str, styles: dict[str, str], seed: int
388
  yield snapshot()
389
  if progress:
390
  progress(.48, desc="Сэмплирую электрогитару и бас")
391
- for name, font_name, transpose, effect in (
392
- ("guitar_clean", "electric-clean.sf2", 0, "guitar"),
393
- ("bass_clean", "finger-bass.sf2", BASS_TRANSPOSE, "bass"),
394
  ):
395
  try:
396
- raw, _ = render_preset(midi, folder, name, 0, duration, log,
397
- soundfont=SOUNDFONT_DIR / font_name,
398
  transpose=transpose)
399
  deliver(name, raw, max_gain=20.)
400
  deliver(name.replace("clean", "amp"),
 
20
  SOUNDFONT_DIR = Path(os.environ.get("HUM_SOUNDFONTS", "/opt/soundfonts"))
21
  GM_SOUNDFONT = Path(os.environ.get("HUM_GM_SOUNDFONT",
22
  "/usr/share/sounds/sf2/FluidR3_GM.sf2"))
23
+ # A pinned MIDI sweep found audible notes 26..45 in Finger Bass YR. Route
24
+ # whole phrases by octaves so no individual note silently exceeds that range.
25
+ BASS_SAMPLE_LOW, BASS_SAMPLE_HIGH = 26, 45
26
  SAMPLER_VARIANTS = (
27
  "instrument", "instrument_drums", "contour",
28
  "guitar_clean", "guitar_amp", "bass_clean", "bass_amp",
 
134
  duration, log)
135
 
136
 
137
+ def choose_bass_route(midi_path: Path) -> tuple[int, Path, int, str]:
138
+ """Choose one octave for the whole phrase; use GM bass if samples cannot fit."""
139
+ import pretty_midi
140
+
141
+ pitches = [note.pitch for track in pretty_midi.PrettyMIDI(str(midi_path)).instruments
142
+ for note in track.notes]
143
+ if not pitches:
144
+ raise ValueError("В напеве нет нот для баса")
145
+ low, high = min(pitches), max(pitches)
146
+ shifts = range(-108, 61, 12)
147
+ sampled = [shift for shift in shifts
148
+ if BASS_SAMPLE_LOW <= low + shift
149
+ and high + shift <= BASS_SAMPLE_HIGH]
150
+ if sampled:
151
+ target = (BASS_SAMPLE_LOW + BASS_SAMPLE_HIGH) / 2
152
+ shift = min(sampled, key=lambda value: abs((low + high) / 2 + value - target))
153
+ return shift, SOUNDFONT_DIR / "finger-bass.sf2", 0, "FreePats Finger Bass YR"
154
+ available = [shift for shift in shifts if 0 <= low + shift and high + shift <= 127]
155
+ shift = min(available, key=lambda value: abs((low + high) / 2 + value - 40))
156
+ return shift, GM_SOUNDFONT, 33, "FluidR3 Fingered Bass fallback"
157
+
158
+
159
  def apply_effect(source: Path, output: Path, kind: str, duration: float,
160
  log: Path) -> Path:
161
  # Both chains use the same sampled notes and retain the source duration.
 
358
  yue_timing = {}
359
  drum_count = 0
360
  drums_midi = None
361
+ bass_shift = None
362
+ bass_source = None
363
 
364
  def deliver(name: str, raw: Path, *, normalize: bool = True,
365
  max_gain: float = 4.) -> None:
 
375
  "yue_requested": include_yue, "yue_timing": dict(yue_timing),
376
  "errors": dict(errors), "complete": complete,
377
  "source_timing_preserved": list(SAMPLER_VARIANTS),
378
+ "bass_transpose_semitones": bass_shift,
379
+ "bass_source": bass_source,
380
  "yue_vocal_absence_guaranteed": False,
381
  "elapsed_seconds": round(time.monotonic() - began, 2)}
382
  return (*(outputs[name] for name in VARIANTS), str(melody_midi),
 
389
  if progress:
390
  progress(.12, desc="SheetSage2 извлекает мелодию и сетку долей")
391
  midi, melody_abc, full_abc, beat_lab = transcribe(pcm24, folder, log)
392
+ bass_shift, bass_font, bass_program, bass_source = choose_bass_route(midi)
393
  if progress:
394
  progress(.30, desc="Синтезирую инструмент и барабаны")
395
  melody_raw, melody_midi = render_instrument(midi, folder, instrument, duration, log)
 
414
  yield snapshot()
415
  if progress:
416
  progress(.48, desc="Сэмплирую электрогитару и бас")
417
+ for name, font, program, transpose, effect in (
418
+ ("guitar_clean", SOUNDFONT_DIR / "electric-clean.sf2", 0, 0, "guitar"),
419
+ ("bass_clean", bass_font, bass_program, bass_shift, "bass"),
420
  ):
421
  try:
422
+ raw, _ = render_preset(midi, folder, name, program, duration, log,
423
+ soundfont=font,
424
  transpose=transpose)
425
  deliver(name, raw, max_gain=20.)
426
  deliver(name.replace("clean", "amp"),
test_sampler.py CHANGED
@@ -14,7 +14,7 @@ class SamplerTest(unittest.TestCase):
14
  import numpy as np
15
  import pretty_midi
16
  import soundfile as sf
17
- from pipeline import BASS_TRANSPOSE, SOUNDFONT_DIR, render_preset
18
 
19
  font = SOUNDFONT_DIR / "finger-bass.sf2"
20
  if not font.is_file():
@@ -30,8 +30,11 @@ class SamplerTest(unittest.TestCase):
30
  score.instruments.append(voice)
31
  midi = folder / "notes.mid"
32
  score.write(str(midi))
 
 
 
33
  wav, _ = render_preset(midi, folder, "bass", 0, 1., folder / "log",
34
- soundfont=font, transpose=BASS_TRANSPOSE)
35
  audio, rate = sf.read(wav, dtype="float32", always_2d=True)
36
  for start in (.12, .62):
37
  span = audio[round(start * rate):round((start + .1) * rate)]
@@ -70,6 +73,7 @@ class SamplerTest(unittest.TestCase):
70
 
71
  with (patch("pipeline.decode", return_value=(wave, wave, 1.)),
72
  patch("pipeline.transcribe", return_value=(midi, abc, abc, beat)),
 
73
  patch("pipeline.render_instrument", return_value=(wave, midi)),
74
  patch("pipeline.render_drums", return_value=(wave, midi, 3)),
75
  patch("pipeline.pair_with_drums", return_value=(wave, wave)),
@@ -83,7 +87,7 @@ class SamplerTest(unittest.TestCase):
83
 
84
  def test_bass_register_preserves_note_times(self):
85
  import pretty_midi
86
- from pipeline import BASS_TRANSPOSE, render_preset
87
 
88
  with tempfile.TemporaryDirectory() as directory:
89
  folder = Path(directory)
@@ -96,15 +100,37 @@ class SamplerTest(unittest.TestCase):
96
  original.instruments.append(voice)
97
  source = folder / "source.mid"
98
  original.write(str(source))
 
 
99
  with patch("pipeline.render_midi", return_value=folder / "bass.wav"):
100
  _, path = render_preset(source, folder, "bass", 0, 1.,
101
- folder / "log", transpose=BASS_TRANSPOSE)
102
  notes = pretty_midi.PrettyMIDI(str(path)).instruments[0].notes
103
  self.assertEqual([n.pitch for n in notes], [36, 40])
104
  for before, after in zip(voice.notes, notes):
105
  self.assertAlmostEqual(before.start, after.start, places=4)
106
  self.assertAlmostEqual(before.end, after.end, places=4)
107
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
108
  def test_effects_preserve_duration_and_change_waveform(self):
109
  import numpy as np
110
  import soundfile as sf
 
14
  import numpy as np
15
  import pretty_midi
16
  import soundfile as sf
17
+ from pipeline import SOUNDFONT_DIR, choose_bass_route, render_preset
18
 
19
  font = SOUNDFONT_DIR / "finger-bass.sf2"
20
  if not font.is_file():
 
30
  score.instruments.append(voice)
31
  midi = folder / "notes.mid"
32
  score.write(str(midi))
33
+ shift, chosen_font, program, source = choose_bass_route(midi)
34
+ self.assertEqual((shift, chosen_font, program), (-24, font, 0))
35
+ self.assertEqual(source, "FreePats Finger Bass YR")
36
  wav, _ = render_preset(midi, folder, "bass", 0, 1., folder / "log",
37
+ soundfont=chosen_font, transpose=shift)
38
  audio, rate = sf.read(wav, dtype="float32", always_2d=True)
39
  for start in (.12, .62):
40
  span = audio[round(start * rate):round((start + .1) * rate)]
 
73
 
74
  with (patch("pipeline.decode", return_value=(wave, wave, 1.)),
75
  patch("pipeline.transcribe", return_value=(midi, abc, abc, beat)),
76
+ patch("pipeline.choose_bass_route", return_value=(-24, folder / "bass.sf2", 0, "test")),
77
  patch("pipeline.render_instrument", return_value=(wave, midi)),
78
  patch("pipeline.render_drums", return_value=(wave, midi, 3)),
79
  patch("pipeline.pair_with_drums", return_value=(wave, wave)),
 
87
 
88
  def test_bass_register_preserves_note_times(self):
89
  import pretty_midi
90
+ from pipeline import choose_bass_route, render_preset
91
 
92
  with tempfile.TemporaryDirectory() as directory:
93
  folder = Path(directory)
 
100
  original.instruments.append(voice)
101
  source = folder / "source.mid"
102
  original.write(str(source))
103
+ shift, _, _, _ = choose_bass_route(source)
104
+ self.assertEqual(shift, -24)
105
  with patch("pipeline.render_midi", return_value=folder / "bass.wav"):
106
  _, path = render_preset(source, folder, "bass", 0, 1.,
107
+ folder / "log", transpose=shift)
108
  notes = pretty_midi.PrettyMIDI(str(path)).instruments[0].notes
109
  self.assertEqual([n.pitch for n in notes], [36, 40])
110
  for before, after in zip(voice.notes, notes):
111
  self.assertAlmostEqual(before.start, after.start, places=4)
112
  self.assertAlmostEqual(before.end, after.end, places=4)
113
 
114
+ def test_bass_route_automatically_shifts_or_uses_full_bank(self):
115
+ import pretty_midi
116
+ from pipeline import choose_bass_route, GM_SOUNDFONT
117
+
118
+ with tempfile.TemporaryDirectory() as directory:
119
+ midi = Path(directory) / "notes.mid"
120
+ for pitches, expected in (((42, 54), (-12, "FreePats Finger Bass YR")),
121
+ ((50, 80), (-24, "FluidR3 Fingered Bass fallback"))):
122
+ score = pretty_midi.PrettyMIDI()
123
+ track = pretty_midi.Instrument(0)
124
+ track.notes = [pretty_midi.Note(90, pitch, index * .3,
125
+ index * .3 + .2)
126
+ for index, pitch in enumerate(pitches)]
127
+ score.instruments.append(track)
128
+ score.write(str(midi))
129
+ shift, font, program, source = choose_bass_route(midi)
130
+ self.assertEqual((shift, source), expected)
131
+ if source.endswith("fallback"):
132
+ self.assertEqual((font, program), (GM_SOUNDFONT, 33))
133
+
134
  def test_effects_preserve_duration_and_change_waveform(self):
135
  import numpy as np
136
  import soundfile as sf