batoon commited on
Commit
fcf203b
·
verified ·
1 Parent(s): bbcd177

Improve short-hum note articulation fallback

Browse files
Files changed (3) hide show
  1. README.md +6 -3
  2. f0_fallback.py +21 -10
  3. test_f0_fallback.py +36 -0
README.md CHANGED
@@ -30,9 +30,12 @@ guitar and fingered bass SoundFonts, with FFmpeg compression, filtering,
30
  oversampled soft clipping, and a short guitar room echo in processed versions.
31
 
32
  If SheetSage2 returns an empty vocal melody, a conservative pYIN fallback
33
- extracts source-timed notes from sustained voiced pitch and supplies a simple
34
- instrumental ABC for YuE2. It rejects unpitched noise. Fallback note boundaries
35
- are approximate, and the run details identify which transcription was used.
 
 
 
36
  SheetSage2's beat grid remains the source for sampled drums when available.
37
 
38
  YuE2 is on by default, so the same click also starts three experiments from
 
30
  oversampled soft clipping, and a short guitar room echo in processed versions.
31
 
32
  If SheetSage2 returns an empty vocal melody, a conservative pYIN fallback
33
+ extracts source-timed notes from sustained voiced pitch. Spectral attacks can
34
+ separate repeated syllables at the same pitch; stable abrupt pitch changes can
35
+ separate legato notes without turning a smooth glide into repeated keypresses.
36
+ It supplies a simple instrumental ABC for YuE2 and rejects unpitched noise.
37
+ Fallback note boundaries remain approximate, and the run details identify
38
+ which transcription was used.
39
  SheetSage2's beat grid remains the source for sampled drums when available.
40
 
41
  YuE2 is on by default, so the same click also starts three experiments from
f0_fallback.py CHANGED
@@ -11,6 +11,7 @@ def transcribe_f0(source: Path, target: Path) -> int:
11
  import numpy as np
12
  import pretty_midi
13
  from scipy.ndimage import binary_closing, median_filter
 
14
  import soundfile as sf
15
 
16
  audio, rate = sf.read(source, dtype="float32")
@@ -19,6 +20,11 @@ def transcribe_f0(source: Path, target: Path) -> int:
19
  reduced = librosa.resample(audio, orig_sr=rate, target_sr=16000)
20
  f0, voiced, _ = librosa.pyin(reduced, sr=16000, fmin=55., fmax=1000.,
21
  frame_length=1024, hop_length=160, center=True)
 
 
 
 
 
22
  rms = librosa.feature.rms(y=reduced, frame_length=1024,
23
  hop_length=160, center=True)[0][:len(f0)]
24
  threshold = max(.001, float(np.percentile(rms, 95)) * 10 ** (-35 / 20))
@@ -31,23 +37,28 @@ def transcribe_f0(source: Path, target: Path) -> int:
31
  for first, last in zip(boundaries[::2], boundaries[1::2]):
32
  if last - first < 18: # Reject short accidental pitched patches in noise.
33
  continue
34
- pitch = median_filter(69 + 12 * np.log2(np.maximum(f0[first:last], 55.) / 440),
35
- size=9, mode="nearest")
36
  # pYIN may be unvoiced for a few interpolated frames.
37
  good = np.isfinite(pitch)
38
  if not good.any():
39
  continue
40
  pitch = np.interp(np.arange(len(pitch)), np.flatnonzero(good), pitch[good])
41
- rounded = median_filter(np.rint(pitch), size=11, mode="nearest").astype(int)
42
- changes = np.r_[0, np.flatnonzero(np.diff(rounded)) + 1, len(rounded)]
43
  cuts = [0]
44
- for index in range(1, len(changes) - 1):
45
- point, next_point = int(changes[index]), int(changes[index + 1])
46
- if (point - cuts[-1] >= 12 and next_point - point >= 12
47
- and abs(np.median(pitch[cuts[-1]:point])
48
- - np.median(pitch[point:next_point])) >= 1.5):
49
  cuts.append(point)
50
- cuts.append(len(rounded))
 
 
 
 
 
 
 
 
51
  for start, end in zip(cuts, cuts[1:]):
52
  if end - start < 8:
53
  continue
 
11
  import numpy as np
12
  import pretty_midi
13
  from scipy.ndimage import binary_closing, median_filter
14
+ from scipy.signal import find_peaks
15
  import soundfile as sf
16
 
17
  audio, rate = sf.read(source, dtype="float32")
 
20
  reduced = librosa.resample(audio, orig_sr=rate, target_sr=16000)
21
  f0, voiced, _ = librosa.pyin(reduced, sr=16000, fmin=55., fmax=1000.,
22
  frame_length=1024, hop_length=160, center=True)
23
+ flux = librosa.onset.onset_strength(y=reduced, sr=16000, hop_length=160,
24
+ n_fft=512, n_mels=64)[:len(f0)]
25
+ attack_threshold = max(2.5, float(np.percentile(flux, 95)) * .8)
26
+ attacks, _ = find_peaks(flux, height=attack_threshold,
27
+ prominence=attack_threshold * .8, distance=12)
28
  rms = librosa.feature.rms(y=reduced, frame_length=1024,
29
  hop_length=160, center=True)[0][:len(f0)]
30
  threshold = max(.001, float(np.percentile(rms, 95)) * 10 ** (-35 / 20))
 
37
  for first, last in zip(boundaries[::2], boundaries[1::2]):
38
  if last - first < 18: # Reject short accidental pitched patches in noise.
39
  continue
40
+ pitch = 69 + 12 * np.log2(np.maximum(f0[first:last], 55.) / 440)
 
41
  # pYIN may be unvoiced for a few interpolated frames.
42
  good = np.isfinite(pitch)
43
  if not good.any():
44
  continue
45
  pitch = np.interp(np.arange(len(pitch)), np.flatnonzero(good), pitch[good])
46
+ pitch = median_filter(pitch, size=5, mode="nearest")
 
47
  cuts = [0]
48
+ for attack in attacks:
49
+ point = int(attack - first)
50
+ # The first/last spectral burst is the span edge, not another keypress.
51
+ if 12 <= point <= len(pitch) - 12 and point - cuts[-1] >= 12:
 
52
  cuts.append(point)
53
+ # An abrupt stable F0 step can be a legato note even without a strong
54
+ # energy attack. A continuous glide never has stable plateaus on both sides.
55
+ for point in range(12, len(pitch) - 12):
56
+ before, after = pitch[point - 10:point - 2], pitch[point + 2:point + 10]
57
+ if (abs(np.median(after) - np.median(before)) >= 1.5
58
+ and np.ptp(before) < .65 and np.ptp(after) < .65
59
+ and all(abs(point - cut) >= 12 for cut in cuts)):
60
+ cuts.append(point)
61
+ cuts = sorted(cuts) + [len(pitch)]
62
  for start, end in zip(cuts, cuts[1:]):
63
  if end - start < 8:
64
  continue
test_f0_fallback.py CHANGED
@@ -38,3 +38,39 @@ class F0FallbackTest(unittest.TestCase):
38
  with self.assertRaisesRegex(ValueError, "устойчивую высоту"):
39
  transcribe_f0(source, target)
40
  self.assertFalse(target.exists())
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38
  with self.assertRaisesRegex(ValueError, "устойчивую высоту"):
39
  transcribe_f0(source, target)
40
  self.assertFalse(target.exists())
41
+
42
+ def test_glide_stays_one_note_but_connected_syllables_retrigger(self):
43
+ import numpy as np
44
+ import pretty_midi
45
+ import soundfile as sf
46
+ from f0_fallback import transcribe_f0
47
+
48
+ rate = 24000
49
+ times = np.arange(round(2.5 * rate)) / rate
50
+ active = (times >= .2) & (times < 2.2)
51
+ for name in ("glide", "repeats"):
52
+ pitch = np.linspace(57, 62, active.sum()) if name == "glide" else np.full(active.sum(), 57)
53
+ frequency = 440 * 2 ** ((pitch - 69) / 12)
54
+ phase = 2 * np.pi * np.cumsum(frequency) / rate
55
+ signal = np.zeros(len(times), dtype=np.float32)
56
+ signal[active] = .13 * (np.sin(phase) + .28 * np.sin(2 * phase))
57
+ if name == "repeats":
58
+ rng = np.random.default_rng(818)
59
+ for boundary in (.8, 1.5):
60
+ center = round(boundary * rate)
61
+ width = round(.035 * rate)
62
+ indices = np.arange(-width, width)
63
+ dip = .82 * np.exp(-.5 * (indices / (width / 2.5)) ** 2)
64
+ signal[center - width:center + width] *= 1 - dip
65
+ signal[center:center + round(.018 * rate)] += rng.normal(
66
+ 0, .018, round(.018 * rate)).astype(np.float32)
67
+ with self.subTest(name=name), tempfile.TemporaryDirectory() as directory:
68
+ source, target = (Path(directory) / filename for filename in ("hum.wav", "notes.mid"))
69
+ sf.write(source, signal, rate)
70
+ expected = 1 if name == "glide" else 3
71
+ self.assertEqual(transcribe_f0(source, target), expected)
72
+ notes = pretty_midi.PrettyMIDI(str(target)).instruments[0].notes
73
+ if name == "repeats":
74
+ for note, start in zip(notes, (.2, .8, 1.5)):
75
+ self.assertEqual(note.pitch, 57)
76
+ self.assertAlmostEqual(note.start, start, delta=.06)