Improve short-hum note articulation fallback
Browse files- README.md +6 -3
- f0_fallback.py +21 -10
- test_f0_fallback.py +36 -0
README.md
CHANGED
|
@@ -30,9 +30,12 @@ guitar and fingered bass SoundFonts, with FFmpeg compression, filtering,
|
|
| 30 |
oversampled soft clipping, and a short guitar room echo in processed versions.
|
| 31 |
|
| 32 |
If SheetSage2 returns an empty vocal melody, a conservative pYIN fallback
|
| 33 |
-
extracts source-timed notes from sustained voiced pitch
|
| 34 |
-
|
| 35 |
-
|
|
|
|
|
|
|
|
|
|
| 36 |
SheetSage2's beat grid remains the source for sampled drums when available.
|
| 37 |
|
| 38 |
YuE2 is on by default, so the same click also starts three experiments from
|
|
|
|
| 30 |
oversampled soft clipping, and a short guitar room echo in processed versions.
|
| 31 |
|
| 32 |
If SheetSage2 returns an empty vocal melody, a conservative pYIN fallback
|
| 33 |
+
extracts source-timed notes from sustained voiced pitch. Spectral attacks can
|
| 34 |
+
separate repeated syllables at the same pitch; stable abrupt pitch changes can
|
| 35 |
+
separate legato notes without turning a smooth glide into repeated keypresses.
|
| 36 |
+
It supplies a simple instrumental ABC for YuE2 and rejects unpitched noise.
|
| 37 |
+
Fallback note boundaries remain approximate, and the run details identify
|
| 38 |
+
which transcription was used.
|
| 39 |
SheetSage2's beat grid remains the source for sampled drums when available.
|
| 40 |
|
| 41 |
YuE2 is on by default, so the same click also starts three experiments from
|
f0_fallback.py
CHANGED
|
@@ -11,6 +11,7 @@ def transcribe_f0(source: Path, target: Path) -> int:
|
|
| 11 |
import numpy as np
|
| 12 |
import pretty_midi
|
| 13 |
from scipy.ndimage import binary_closing, median_filter
|
|
|
|
| 14 |
import soundfile as sf
|
| 15 |
|
| 16 |
audio, rate = sf.read(source, dtype="float32")
|
|
@@ -19,6 +20,11 @@ def transcribe_f0(source: Path, target: Path) -> int:
|
|
| 19 |
reduced = librosa.resample(audio, orig_sr=rate, target_sr=16000)
|
| 20 |
f0, voiced, _ = librosa.pyin(reduced, sr=16000, fmin=55., fmax=1000.,
|
| 21 |
frame_length=1024, hop_length=160, center=True)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 22 |
rms = librosa.feature.rms(y=reduced, frame_length=1024,
|
| 23 |
hop_length=160, center=True)[0][:len(f0)]
|
| 24 |
threshold = max(.001, float(np.percentile(rms, 95)) * 10 ** (-35 / 20))
|
|
@@ -31,23 +37,28 @@ def transcribe_f0(source: Path, target: Path) -> int:
|
|
| 31 |
for first, last in zip(boundaries[::2], boundaries[1::2]):
|
| 32 |
if last - first < 18: # Reject short accidental pitched patches in noise.
|
| 33 |
continue
|
| 34 |
-
pitch =
|
| 35 |
-
size=9, mode="nearest")
|
| 36 |
# pYIN may be unvoiced for a few interpolated frames.
|
| 37 |
good = np.isfinite(pitch)
|
| 38 |
if not good.any():
|
| 39 |
continue
|
| 40 |
pitch = np.interp(np.arange(len(pitch)), np.flatnonzero(good), pitch[good])
|
| 41 |
-
|
| 42 |
-
changes = np.r_[0, np.flatnonzero(np.diff(rounded)) + 1, len(rounded)]
|
| 43 |
cuts = [0]
|
| 44 |
-
for
|
| 45 |
-
point
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
- np.median(pitch[point:next_point])) >= 1.5):
|
| 49 |
cuts.append(point)
|
| 50 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 51 |
for start, end in zip(cuts, cuts[1:]):
|
| 52 |
if end - start < 8:
|
| 53 |
continue
|
|
|
|
| 11 |
import numpy as np
|
| 12 |
import pretty_midi
|
| 13 |
from scipy.ndimage import binary_closing, median_filter
|
| 14 |
+
from scipy.signal import find_peaks
|
| 15 |
import soundfile as sf
|
| 16 |
|
| 17 |
audio, rate = sf.read(source, dtype="float32")
|
|
|
|
| 20 |
reduced = librosa.resample(audio, orig_sr=rate, target_sr=16000)
|
| 21 |
f0, voiced, _ = librosa.pyin(reduced, sr=16000, fmin=55., fmax=1000.,
|
| 22 |
frame_length=1024, hop_length=160, center=True)
|
| 23 |
+
flux = librosa.onset.onset_strength(y=reduced, sr=16000, hop_length=160,
|
| 24 |
+
n_fft=512, n_mels=64)[:len(f0)]
|
| 25 |
+
attack_threshold = max(2.5, float(np.percentile(flux, 95)) * .8)
|
| 26 |
+
attacks, _ = find_peaks(flux, height=attack_threshold,
|
| 27 |
+
prominence=attack_threshold * .8, distance=12)
|
| 28 |
rms = librosa.feature.rms(y=reduced, frame_length=1024,
|
| 29 |
hop_length=160, center=True)[0][:len(f0)]
|
| 30 |
threshold = max(.001, float(np.percentile(rms, 95)) * 10 ** (-35 / 20))
|
|
|
|
| 37 |
for first, last in zip(boundaries[::2], boundaries[1::2]):
|
| 38 |
if last - first < 18: # Reject short accidental pitched patches in noise.
|
| 39 |
continue
|
| 40 |
+
pitch = 69 + 12 * np.log2(np.maximum(f0[first:last], 55.) / 440)
|
|
|
|
| 41 |
# pYIN may be unvoiced for a few interpolated frames.
|
| 42 |
good = np.isfinite(pitch)
|
| 43 |
if not good.any():
|
| 44 |
continue
|
| 45 |
pitch = np.interp(np.arange(len(pitch)), np.flatnonzero(good), pitch[good])
|
| 46 |
+
pitch = median_filter(pitch, size=5, mode="nearest")
|
|
|
|
| 47 |
cuts = [0]
|
| 48 |
+
for attack in attacks:
|
| 49 |
+
point = int(attack - first)
|
| 50 |
+
# The first/last spectral burst is the span edge, not another keypress.
|
| 51 |
+
if 12 <= point <= len(pitch) - 12 and point - cuts[-1] >= 12:
|
|
|
|
| 52 |
cuts.append(point)
|
| 53 |
+
# An abrupt stable F0 step can be a legato note even without a strong
|
| 54 |
+
# energy attack. A continuous glide never has stable plateaus on both sides.
|
| 55 |
+
for point in range(12, len(pitch) - 12):
|
| 56 |
+
before, after = pitch[point - 10:point - 2], pitch[point + 2:point + 10]
|
| 57 |
+
if (abs(np.median(after) - np.median(before)) >= 1.5
|
| 58 |
+
and np.ptp(before) < .65 and np.ptp(after) < .65
|
| 59 |
+
and all(abs(point - cut) >= 12 for cut in cuts)):
|
| 60 |
+
cuts.append(point)
|
| 61 |
+
cuts = sorted(cuts) + [len(pitch)]
|
| 62 |
for start, end in zip(cuts, cuts[1:]):
|
| 63 |
if end - start < 8:
|
| 64 |
continue
|
test_f0_fallback.py
CHANGED
|
@@ -38,3 +38,39 @@ class F0FallbackTest(unittest.TestCase):
|
|
| 38 |
with self.assertRaisesRegex(ValueError, "устойчивую высоту"):
|
| 39 |
transcribe_f0(source, target)
|
| 40 |
self.assertFalse(target.exists())
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 38 |
with self.assertRaisesRegex(ValueError, "устойчивую высоту"):
|
| 39 |
transcribe_f0(source, target)
|
| 40 |
self.assertFalse(target.exists())
|
| 41 |
+
|
| 42 |
+
def test_glide_stays_one_note_but_connected_syllables_retrigger(self):
|
| 43 |
+
import numpy as np
|
| 44 |
+
import pretty_midi
|
| 45 |
+
import soundfile as sf
|
| 46 |
+
from f0_fallback import transcribe_f0
|
| 47 |
+
|
| 48 |
+
rate = 24000
|
| 49 |
+
times = np.arange(round(2.5 * rate)) / rate
|
| 50 |
+
active = (times >= .2) & (times < 2.2)
|
| 51 |
+
for name in ("glide", "repeats"):
|
| 52 |
+
pitch = np.linspace(57, 62, active.sum()) if name == "glide" else np.full(active.sum(), 57)
|
| 53 |
+
frequency = 440 * 2 ** ((pitch - 69) / 12)
|
| 54 |
+
phase = 2 * np.pi * np.cumsum(frequency) / rate
|
| 55 |
+
signal = np.zeros(len(times), dtype=np.float32)
|
| 56 |
+
signal[active] = .13 * (np.sin(phase) + .28 * np.sin(2 * phase))
|
| 57 |
+
if name == "repeats":
|
| 58 |
+
rng = np.random.default_rng(818)
|
| 59 |
+
for boundary in (.8, 1.5):
|
| 60 |
+
center = round(boundary * rate)
|
| 61 |
+
width = round(.035 * rate)
|
| 62 |
+
indices = np.arange(-width, width)
|
| 63 |
+
dip = .82 * np.exp(-.5 * (indices / (width / 2.5)) ** 2)
|
| 64 |
+
signal[center - width:center + width] *= 1 - dip
|
| 65 |
+
signal[center:center + round(.018 * rate)] += rng.normal(
|
| 66 |
+
0, .018, round(.018 * rate)).astype(np.float32)
|
| 67 |
+
with self.subTest(name=name), tempfile.TemporaryDirectory() as directory:
|
| 68 |
+
source, target = (Path(directory) / filename for filename in ("hum.wav", "notes.mid"))
|
| 69 |
+
sf.write(source, signal, rate)
|
| 70 |
+
expected = 1 if name == "glide" else 3
|
| 71 |
+
self.assertEqual(transcribe_f0(source, target), expected)
|
| 72 |
+
notes = pretty_midi.PrettyMIDI(str(target)).instruments[0].notes
|
| 73 |
+
if name == "repeats":
|
| 74 |
+
for note, start in zip(notes, (.2, .8, 1.5)):
|
| 75 |
+
self.assertEqual(note.pitch, 57)
|
| 76 |
+
self.assertAlmostEqual(note.start, start, delta=.06)
|