File size: 3,926 Bytes
21c2288 fcf203b 21c2288 fcf203b 21c2288 fcf203b 21c2288 fcf203b 21c2288 fcf203b 21c2288 fcf203b 21c2288 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 | """Source-timed monophonic fallback for a hum SheetSage cannot notate."""
from __future__ import annotations
from pathlib import Path
def transcribe_f0(source: Path, target: Path) -> int:
"""Write MIDI from voiced F0 spans; reject inputs without a sustained pitch."""
import librosa
import numpy as np
import pretty_midi
from scipy.ndimage import binary_closing, median_filter
from scipy.signal import find_peaks
import soundfile as sf
audio, rate = sf.read(source, dtype="float32")
if audio.ndim != 1 or rate != 24000:
raise ValueError("Для резервной транскрипции нужен моно PCM 24 кГц")
reduced = librosa.resample(audio, orig_sr=rate, target_sr=16000)
f0, voiced, _ = librosa.pyin(reduced, sr=16000, fmin=55., fmax=1000.,
frame_length=1024, hop_length=160, center=True)
flux = librosa.onset.onset_strength(y=reduced, sr=16000, hop_length=160,
n_fft=512, n_mels=64)[:len(f0)]
attack_threshold = max(2.5, float(np.percentile(flux, 95)) * .8)
attacks, _ = find_peaks(flux, height=attack_threshold,
prominence=attack_threshold * .8, distance=12)
rms = librosa.feature.rms(y=reduced, frame_length=1024,
hop_length=160, center=True)[0][:len(f0)]
threshold = max(.001, float(np.percentile(rms, 95)) * 10 ** (-35 / 20))
active = voiced & np.isfinite(f0) & (rms >= threshold)
# Bridge only pitch-tracker dropouts shorter than a consonant or breath.
active = binary_closing(active, structure=np.ones(4), border_value=0)
boundaries = np.flatnonzero(np.diff(np.r_[False, active, False]))
score = pretty_midi.PrettyMIDI(initial_tempo=120)
track = pretty_midi.Instrument(program=0, name="F0 fallback")
for first, last in zip(boundaries[::2], boundaries[1::2]):
if last - first < 18: # Reject short accidental pitched patches in noise.
continue
pitch = 69 + 12 * np.log2(np.maximum(f0[first:last], 55.) / 440)
# pYIN may be unvoiced for a few interpolated frames.
good = np.isfinite(pitch)
if not good.any():
continue
pitch = np.interp(np.arange(len(pitch)), np.flatnonzero(good), pitch[good])
pitch = median_filter(pitch, size=5, mode="nearest")
cuts = [0]
for attack in attacks:
point = int(attack - first)
# The first/last spectral burst is the span edge, not another keypress.
if 12 <= point <= len(pitch) - 12 and point - cuts[-1] >= 12:
cuts.append(point)
# An abrupt stable F0 step can be a legato note even without a strong
# energy attack. A continuous glide never has stable plateaus on both sides.
for point in range(12, len(pitch) - 12):
before, after = pitch[point - 10:point - 2], pitch[point + 2:point + 10]
if (abs(np.median(after) - np.median(before)) >= 1.5
and np.ptp(before) < .65 and np.ptp(after) < .65
and all(abs(point - cut) >= 12 for cut in cuts)):
cuts.append(point)
cuts = sorted(cuts) + [len(pitch)]
for start, end in zip(cuts, cuts[1:]):
if end - start < 8:
continue
begin = max(0., (first + start) * .01 - .005)
finish = min(len(audio) / rate, (first + end) * .01 - .005)
if finish - begin < .075:
continue
note = int(np.clip(np.rint(np.median(pitch[start:end])), 0, 127))
track.notes.append(pretty_midi.Note(90, note, begin, finish))
if not track.notes:
raise ValueError("Не удалось выделить устойчивую высоту напева")
score.instruments.append(track)
score.write(str(target))
return len(track.notes)
|