Random118's picture
Add Russian CosyVoice 3 dubbing pack and audio gallery
8c2913a verified
Raw History Blame Contribute Delete
5.2 kB
"""Align a generated clip's median F0 to an English GLaDOS reference."""
import argparse
from dataclasses import dataclass
import math
import shutil
import subprocess
import tempfile
from pathlib import Path
import librosa
import numpy as np
import soundfile as sf
@dataclass(frozen=True)
class PitchAlignmentResult:
source_f0: float
reference_f0: float
output_f0: float
semitones: float
relative_error: float
def median_f0(path: Path) -> float:
audio, sr = librosa.load(str(path), sr=16000, mono=True)
f0, voiced, _ = librosa.pyin(
audio,
fmin=65,
fmax=500,
sr=sr,
frame_length=1024,
hop_length=160,
)
values = f0[np.isfinite(f0) & voiced]
if len(values) == 0:
raise ValueError(f"No voiced F0 found in {path}")
return float(np.median(values))
def render_pitch_shift(input_path: Path, output_path: Path, semitones: float) -> None:
"""Render one formant-preserving Rubber Band pitch candidate."""
subprocess.run(
[
"rubberband",
"-3",
"-F",
"-p",
f"{semitones:.4f}",
str(input_path),
str(output_path),
],
check=True,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
)
write_normalized_copy(output_path, output_path)
def write_normalized_copy(input_path: Path, output_path: Path) -> None:
"""Write PCM16 audio with enough headroom to avoid clipped candidates."""
audio, sr = sf.read(input_path, always_2d=False)
peak = float(np.max(np.abs(audio)))
if peak > 0.95:
audio = audio * (0.95 / peak)
sf.write(output_path, audio, sr, subtype="PCM_16")
def _clamp_shift(semitones: float, maximum: float) -> float:
return max(-maximum, min(maximum, semitones))
def align_median_f0(
input_path: Path,
reference_path: Path,
output_path: Path,
*,
tolerance: float = 0.025,
max_abs_semitones: float = 4.0,
) -> PitchAlignmentResult:
"""Align median F0 and verify the rendered result.
A theoretical semitone ratio is normally sufficient. GLaDOS clips can make
pYIN switch between nearby pitch modes after rendering, though, so a failed
first pass is calibrated by rendering nearby shifts from the original clip.
The best candidate is copied once; pitch shifts are never stacked.
"""
source_f0 = median_f0(input_path)
reference_f0 = median_f0(reference_path)
source_error = abs(source_f0 - reference_f0) / reference_f0
output_path.parent.mkdir(parents=True, exist_ok=True)
if source_error <= tolerance:
write_normalized_copy(input_path, output_path)
return PitchAlignmentResult(
source_f0=source_f0,
reference_f0=reference_f0,
output_f0=source_f0,
semitones=0.0,
relative_error=source_error,
)
predicted = _clamp_shift(
12.0 * math.log2(reference_f0 / source_f0),
max_abs_semitones,
)
with tempfile.TemporaryDirectory(prefix="glados-pitch-") as temp_dir:
temp_root = Path(temp_dir)
attempts: list[tuple[float, float, float, Path]] = []
def evaluate(semitones: float) -> tuple[float, float, float, Path]:
semitones = _clamp_shift(semitones, max_abs_semitones)
for attempt in attempts:
if math.isclose(attempt[0], semitones, abs_tol=1e-6):
return attempt
candidate = temp_root / f"candidate-{len(attempts):02d}.wav"
render_pitch_shift(input_path, candidate, semitones)
candidate_f0 = median_f0(candidate)
relative_error = abs(candidate_f0 - reference_f0) / reference_f0
attempt = (semitones, candidate_f0, relative_error, candidate)
attempts.append(attempt)
return attempt
best = evaluate(predicted)
if best[2] > tolerance:
for offset in (-1.5, -1.0, -0.5, 0.5, 1.0, 1.5):
candidate = evaluate(predicted + offset)
if candidate[2] < best[2]:
best = candidate
if best[2] > tolerance:
center = best[0]
for offset in (-0.25, 0.25):
candidate = evaluate(center + offset)
if candidate[2] < best[2]:
best = candidate
shutil.copyfile(best[3], output_path)
return PitchAlignmentResult(
source_f0=source_f0,
reference_f0=reference_f0,
output_f0=best[1],
semitones=best[0],
relative_error=best[2],
)
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("input", type=Path)
parser.add_argument("reference", type=Path)
parser.add_argument("output", type=Path)
args = parser.parse_args()
result = align_median_f0(args.input, args.reference, args.output)
print(
f"source_f0={result.source_f0:.1f} reference_f0={result.reference_f0:.1f} "
f"output_f0={result.output_f0:.1f} shift={result.semitones:.2f}st "
f"error={result.relative_error:.2%}"
)
if __name__ == "__main__":
main()