Download Models/Russian_CosyVoice3/align_pitch.py from Random118/GLaDOS_TTS: direct link, hf CLI and curl.
- Browser
- Download file 5.2 kB
-
https://huggingface.co/Random118/GLaDOS_TTS/resolve/main/Models/Russian_CosyVoice3/align_pitch.py
- Command line
-
hf download hf://Random118/GLaDOS_TTS/Models/Russian_CosyVoice3/align_pitch.py
-
curl -L -o align_pitch.py https://huggingface.co/Random118/GLaDOS_TTS/resolve/main/Models/Russian_CosyVoice3/align_pitch.py
5.2 kB
| """Align a generated clip's median F0 to an English GLaDOS reference.""" | |
| import argparse | |
| from dataclasses import dataclass | |
| import math | |
| import shutil | |
| import subprocess | |
| import tempfile | |
| from pathlib import Path | |
| import librosa | |
| import numpy as np | |
| import soundfile as sf | |
| class PitchAlignmentResult: | |
| source_f0: float | |
| reference_f0: float | |
| output_f0: float | |
| semitones: float | |
| relative_error: float | |
| def median_f0(path: Path) -> float: | |
| audio, sr = librosa.load(str(path), sr=16000, mono=True) | |
| f0, voiced, _ = librosa.pyin( | |
| audio, | |
| fmin=65, | |
| fmax=500, | |
| sr=sr, | |
| frame_length=1024, | |
| hop_length=160, | |
| ) | |
| values = f0[np.isfinite(f0) & voiced] | |
| if len(values) == 0: | |
| raise ValueError(f"No voiced F0 found in {path}") | |
| return float(np.median(values)) | |
| def render_pitch_shift(input_path: Path, output_path: Path, semitones: float) -> None: | |
| """Render one formant-preserving Rubber Band pitch candidate.""" | |
| subprocess.run( | |
| [ | |
| "rubberband", | |
| "-3", | |
| "-F", | |
| "-p", | |
| f"{semitones:.4f}", | |
| str(input_path), | |
| str(output_path), | |
| ], | |
| check=True, | |
| stdout=subprocess.DEVNULL, | |
| stderr=subprocess.DEVNULL, | |
| ) | |
| write_normalized_copy(output_path, output_path) | |
| def write_normalized_copy(input_path: Path, output_path: Path) -> None: | |
| """Write PCM16 audio with enough headroom to avoid clipped candidates.""" | |
| audio, sr = sf.read(input_path, always_2d=False) | |
| peak = float(np.max(np.abs(audio))) | |
| if peak > 0.95: | |
| audio = audio * (0.95 / peak) | |
| sf.write(output_path, audio, sr, subtype="PCM_16") | |
| def _clamp_shift(semitones: float, maximum: float) -> float: | |
| return max(-maximum, min(maximum, semitones)) | |
| def align_median_f0( | |
| input_path: Path, | |
| reference_path: Path, | |
| output_path: Path, | |
| *, | |
| tolerance: float = 0.025, | |
| max_abs_semitones: float = 4.0, | |
| ) -> PitchAlignmentResult: | |
| """Align median F0 and verify the rendered result. | |
| A theoretical semitone ratio is normally sufficient. GLaDOS clips can make | |
| pYIN switch between nearby pitch modes after rendering, though, so a failed | |
| first pass is calibrated by rendering nearby shifts from the original clip. | |
| The best candidate is copied once; pitch shifts are never stacked. | |
| """ | |
| source_f0 = median_f0(input_path) | |
| reference_f0 = median_f0(reference_path) | |
| source_error = abs(source_f0 - reference_f0) / reference_f0 | |
| output_path.parent.mkdir(parents=True, exist_ok=True) | |
| if source_error <= tolerance: | |
| write_normalized_copy(input_path, output_path) | |
| return PitchAlignmentResult( | |
| source_f0=source_f0, | |
| reference_f0=reference_f0, | |
| output_f0=source_f0, | |
| semitones=0.0, | |
| relative_error=source_error, | |
| ) | |
| predicted = _clamp_shift( | |
| 12.0 * math.log2(reference_f0 / source_f0), | |
| max_abs_semitones, | |
| ) | |
| with tempfile.TemporaryDirectory(prefix="glados-pitch-") as temp_dir: | |
| temp_root = Path(temp_dir) | |
| attempts: list[tuple[float, float, float, Path]] = [] | |
| def evaluate(semitones: float) -> tuple[float, float, float, Path]: | |
| semitones = _clamp_shift(semitones, max_abs_semitones) | |
| for attempt in attempts: | |
| if math.isclose(attempt[0], semitones, abs_tol=1e-6): | |
| return attempt | |
| candidate = temp_root / f"candidate-{len(attempts):02d}.wav" | |
| render_pitch_shift(input_path, candidate, semitones) | |
| candidate_f0 = median_f0(candidate) | |
| relative_error = abs(candidate_f0 - reference_f0) / reference_f0 | |
| attempt = (semitones, candidate_f0, relative_error, candidate) | |
| attempts.append(attempt) | |
| return attempt | |
| best = evaluate(predicted) | |
| if best[2] > tolerance: | |
| for offset in (-1.5, -1.0, -0.5, 0.5, 1.0, 1.5): | |
| candidate = evaluate(predicted + offset) | |
| if candidate[2] < best[2]: | |
| best = candidate | |
| if best[2] > tolerance: | |
| center = best[0] | |
| for offset in (-0.25, 0.25): | |
| candidate = evaluate(center + offset) | |
| if candidate[2] < best[2]: | |
| best = candidate | |
| shutil.copyfile(best[3], output_path) | |
| return PitchAlignmentResult( | |
| source_f0=source_f0, | |
| reference_f0=reference_f0, | |
| output_f0=best[1], | |
| semitones=best[0], | |
| relative_error=best[2], | |
| ) | |
| def main() -> None: | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("input", type=Path) | |
| parser.add_argument("reference", type=Path) | |
| parser.add_argument("output", type=Path) | |
| args = parser.parse_args() | |
| result = align_median_f0(args.input, args.reference, args.output) | |
| print( | |
| f"source_f0={result.source_f0:.1f} reference_f0={result.reference_f0:.1f} " | |
| f"output_f0={result.output_f0:.1f} shift={result.semitones:.2f}st " | |
| f"error={result.relative_error:.2%}" | |
| ) | |
| if __name__ == "__main__": | |
| main() | |