from __future__ import annotations import copy import bisect import io import json import math import re import tempfile import unicodedata import zipfile import xml.etree.ElementTree as ElementTree from dataclasses import dataclass from fractions import Fraction from pathlib import Path from statistics import median from typing import Any QUANTIZATION_DIVISORS = { "1/8": 2, "1/16": 4, "1/32": 8, } # music21 accepts several candidate grids at once and selects the closest one # note by note. Keeping a ternary divisor beside each binary resolution is a # much better fit for performances containing triplets or compound meter than # forcing every onset onto a single straight grid. QUANTIZATION_GRIDS = { "1/8": (2, 3), "1/16": (4, 3, 6), "1/32": (8, 6, 12), } CLEANUP_PROFILES = { "faithful": { "label": "Faithful", "merge_gap_beats": 0.025, "minimum_note_beats": 0.035, "drop_ghost_beats": 0.0, }, "balanced": { "label": "Balanced", "merge_gap_beats": 0.07, "minimum_note_beats": 0.09, "drop_ghost_beats": 0.0, }, "readable": { "label": "Readable", "merge_gap_beats": 0.12, "minimum_note_beats": 0.14, "drop_ghost_beats": 0.045, }, } AMBIGUOUS_INSTRUMENT_GROUPS = { "brass_section", "chromatic_percussion", "flutes", "orchestra_hit", "soprano_and_alto_sax", "string_ensemble", "strings", "synth_lead", "synth_pad", "synth_strings", "voice", } KEY_PITCH_NAMES = { "major": ("C", "D♭", "D", "E♭", "E", "F", "F♯", "G", "A♭", "A", "B♭", "B"), "minor": ("C", "C♯", "D", "E♭", "E", "F", "F♯", "G", "G♯", "A", "B♭", "B"), } SOLFEGE_NAMES = ("Do", "Do♯", "Ré", "Mi♭", "Mi", "Fa", "Fa♯", "Sol", "La♭", "La", "Si♭", "Si") @dataclass(frozen=True) class NotationPartResult: track_id: str track_key: str name: str slug: str musicxml: Path pdf: Path | None svg_pages: tuple[Path, ...] render_error: str @dataclass(frozen=True) class NotationResult: musicxml: Path pdf: Path | None bundle: Path svg_pages: tuple[Path, ...] preview_svg: str parts: tuple[NotationPartResult, ...] warnings: tuple[str, ...] export_stem: str report: Path diagnostics: dict[str, Any] def _default_score_analysis(onset_count: int = 0) -> dict[str, Any]: return { "analysis_version": 3, "tempo_bpm": 120.0, "time_signature": "4/4", "quantization": "1/16", "key_signature": "C major", "key_confidence": 0.0, "confidence": "low", "onset_count": int(onset_count), "tempo_confidence": 0.0, "quantization_fit": 0.0, "beat_phase_seconds": 0.0, "first_downbeat_seconds": 0.0, "pickup_beats": 0.0, "beat_quarter_length": 1.0, "beat_times_seconds": [], "beat_origin_index": 0, "grid_family": "binary + ternary", "analysis_source": "symbolic notes", "review_flags": [ "Tempo, meter and key need manual confirmation.", "Dynamics are not predicted by MuScriptor and are intentionally left neutral.", ], } def _estimate_key_signature( tracks: list[dict[str, Any]], np: Any, ) -> tuple[str, float]: """Estimate a concert key with Krumhansl-Schmuckler profiles. This is deliberately a suggestion, not a promise: chromatic music and short clips can make relative major/minor keys mathematically ambiguous. """ histogram = np.zeros(12, dtype=float) for track in tracks: if track.get("is_drum"): continue for item in track.get("notes") or []: pitch = max(0, min(127, int(item.get("pitch", 60)))) start = max(0.0, float(item.get("start", 0.0))) end = max(start + 0.015, float(item.get("end", start + 0.1))) # Long pedal tails should not dominate the harmonic analysis. histogram[pitch % 12] += min(2.0, end - start) ** 0.72 if float(histogram.sum()) <= 0: return "C major", 0.0 major_profile = np.asarray( [6.35, 2.23, 3.48, 2.33, 4.38, 4.09, 2.52, 5.19, 2.39, 3.66, 2.29, 2.88], dtype=float, ) minor_profile = np.asarray( [6.33, 2.68, 3.52, 5.38, 2.60, 3.53, 2.54, 4.75, 3.98, 2.69, 3.34, 3.17], dtype=float, ) scores: list[tuple[float, int, str]] = [] for root in range(12): for mode, profile in (("major", major_profile), ("minor", minor_profile)): shifted = np.roll(profile, root) if float(histogram.std()) <= 1e-9: score = 0.0 else: score = float(np.corrcoef(histogram, shifted)[0, 1]) scores.append((score, root, mode)) scores.sort(reverse=True) best, second = scores[0], scores[1] separation = max(0.0, min(1.0, (best[0] - second[0]) / 0.24)) return f"{KEY_PITCH_NAMES[best[2]][best[1]]} {best[2]}", round(separation, 3) def estimate_score_settings(tracks: list[dict[str, Any]]) -> dict[str, Any]: """Estimate notation presets from the decoded note onsets. The estimate intentionally runs after transcription: instrument-separated note onsets are substantially cleaner than a broadband audio onset curve, and this keeps the Space free from another heavyweight audio model. """ try: import numpy as np except ImportError: # pragma: no cover - numpy is an app dependency return _default_score_analysis() events: list[tuple[float, float, bool]] = [] durations: list[float] = [] for track in tracks: is_drum = bool(track.get("is_drum")) for item in track.get("notes") or []: start = max(0.0, float(item.get("start", 0.0))) end = max(start + 0.015, float(item.get("end", start + 0.1))) velocity = max(1.0, min(127.0, float(item.get("velocity", 90.0)))) / 127.0 events.append((start, velocity * (1.35 if is_drum else 1.0), is_drum)) durations.append(end - start) if len(events) < 4: analysis = _default_score_analysis(len(events)) key_signature, key_confidence = _estimate_key_signature(tracks, np) analysis.update( { "key_signature": key_signature, "key_confidence": key_confidence, } ) return analysis # Collapse chords and near-simultaneous detections into one accented onset. events.sort(key=lambda event: event[0]) clusters: list[list[tuple[float, float, bool]]] = [] for event in events: if not clusters or event[0] - clusters[-1][-1][0] > 0.035: clusters.append([event]) else: clusters[-1].append(event) onsets = np.asarray( [sum(event[0] * event[1] for event in cluster) / sum(event[1] for event in cluster) for cluster in clusters], dtype=float, ) accents = np.asarray( [max(event[1] for event in cluster) + 0.18 * sum(event[1] for event in cluster[1:]) for cluster in clusters], dtype=float, ) bpm_axis = np.arange(45.0, 221.0, 0.5) tempo_scores = np.zeros_like(bpm_axis) beat_hypotheses = ((0.25, 0.22), (0.5, 0.82), (1.0, 1.0), (1.5, 0.62), (2.0, 0.7), (3.0, 0.42), (4.0, 0.3)) for left in range(len(onsets) - 1): for right in range(left + 1, min(len(onsets), left + 9)): interval = onsets[right] - onsets[left] if interval > 4.5: break if interval < 0.08: continue pair_weight = math.sqrt(accents[left] * accents[right]) / math.sqrt(right - left) for beats, hypothesis_weight in beat_hypotheses: bpm = 60.0 * beats / interval if bpm_axis[0] <= bpm <= bpm_axis[-1]: distance = (bpm_axis - bpm) / 1.15 tempo_scores += pair_weight * hypothesis_weight * np.exp(-0.5 * distance * distance) tempo_prior = 0.72 + 0.28 * np.exp(-0.5 * ((bpm_axis - 112.0) / 52.0) ** 2) tempo_scores *= tempo_prior best_index = int(np.argmax(tempo_scores)) best_bpm = float(bpm_axis[best_index]) onset_span = max(1e-6, float(onsets[-1] - onsets[0])) onset_rate_hz = len(onsets) / onset_span tempo_octave_adjusted = False if best_bpm < 74 and best_bpm * 2 <= bpm_axis[-1]: double_index = int(np.argmin(np.abs(bpm_axis - best_bpm * 2))) # Dense arrangements are very often detected at half-time. Prefer the # musician-friendly octave when the double-tempo candidate has useful # evidence, while retaining genuinely sparse adagios at their slow BPM. double_threshold = 0.58 if onset_rate_hz >= 1.35 else 0.74 if tempo_scores[double_index] >= tempo_scores[best_index] * double_threshold: best_index = double_index best_bpm = float(bpm_axis[best_index]) tempo_octave_adjusted = True if best_bpm > 178: half_index = int(np.argmin(np.abs(bpm_axis - best_bpm / 2))) if tempo_scores[half_index] >= tempo_scores[best_index] * 0.88: best_index = half_index best_bpm = float(bpm_axis[best_index]) neighborhood = np.abs(bpm_axis - best_bpm) <= 2.0 if float(tempo_scores[neighborhood].sum()) > 0: best_bpm = float(np.average(bpm_axis[neighborhood], weights=tempo_scores[neighborhood])) beat_seconds = 60.0 / max(1.0, best_bpm) strong_threshold = float(np.quantile(accents, 0.55)) strong = accents >= strong_threshold phase_candidates = np.linspace(0.0, beat_seconds, 96, endpoint=False) phase_scores = [] for phase in phase_candidates: distance = np.abs(((onsets[strong] - phase + beat_seconds / 2) % beat_seconds) - beat_seconds / 2) phase_scores.append(float(np.sum(accents[strong] * np.exp(-0.5 * (distance / (beat_seconds * 0.11)) ** 2)))) beat_phase = float(phase_candidates[int(np.argmax(phase_scores))]) beat_indices = np.rint((onsets - beat_phase) / beat_seconds).astype(int) minimum_beat = int(beat_indices.min()) beat_indices -= minimum_beat beat_energy = np.zeros(int(beat_indices.max()) + 1, dtype=float) for index, accent in zip(beat_indices, accents): beat_energy[index] += accent meter_scores: dict[int, float] = {} meter_phases: dict[int, int] = {} # Common time is deliberately the tie-breaker: sparse arrangements often # make 2- and 4-beat periodicities mathematically indistinguishable. meter_priors = {2: 0.0, 3: 0.08, 4: 0.32} energy_mean = float(beat_energy.mean()) + 1e-9 for meter_value in (2, 3, 4): best_meter_score = -1e9 for bar_phase in range(meter_value): downbeats = beat_energy[bar_phase::meter_value] others = np.asarray( [value for index, value in enumerate(beat_energy) if index % meter_value != bar_phase], dtype=float, ) contrast = (float(downbeats.mean()) - float(others.mean() if len(others) else 0.0)) / energy_mean periodicity = 0.0 if len(beat_energy) > meter_value * 2: left = beat_energy[:-meter_value] right = beat_energy[meter_value:] if float(left.std()) > 1e-6 and float(right.std()) > 1e-6: periodicity = float(np.corrcoef(left, right)[0, 1]) best_meter_score = max(best_meter_score, contrast * 0.72 + periodicity * 0.28) if best_meter_score == contrast * 0.72 + periodicity * 0.28: meter_phases[meter_value] = bar_phase meter_scores[meter_value] = best_meter_score + meter_priors[meter_value] meter = max(meter_scores, key=meter_scores.get) bar_phase = int(meter_phases.get(meter, 0)) relative = (onsets - beat_phase) / beat_seconds offbeat = np.abs(relative - np.rint(relative)) > 0.08 compound = False if int(np.count_nonzero(offbeat)) >= max(3, len(onsets) // 8): triple_error = np.abs(relative[offbeat] * 3 - np.rint(relative[offbeat] * 3)) binary_error = np.minimum( np.abs(relative[offbeat] * 2 - np.rint(relative[offbeat] * 2)), np.abs(relative[offbeat] * 4 - np.rint(relative[offbeat] * 4)), ) compound = float(np.median(triple_error)) + 0.035 < float(np.median(binary_error)) if meter == 3: time_signature = "3/4" elif compound and meter == 2: time_signature = "6/8" elif compound and meter == 4: time_signature = "12/8" else: time_signature = f"{meter}/4" # Use chord-collapsed onsets here. Small inter-player and chord-note # offsets should not force a 1/32 grid merely because the source was a # human performance rather than step-entered notation. starts = onsets note_durations = np.asarray(durations, dtype=float) quantization_fits: dict[str, float] = {} for label, divisors in QUANTIZATION_GRIDS.items(): start_error = np.ones_like(starts) duration_error = np.ones_like(note_durations) for divisor in divisors: grid = beat_seconds / divisor start_units = (starts - beat_phase) / grid duration_units = note_durations / grid start_error = np.minimum(start_error, np.abs(start_units - np.rint(start_units))) duration_error = np.minimum(duration_error, np.abs(duration_units - np.rint(duration_units))) # Note-off timing is expressive in real performances; onset placement # is the much stronger signal for choosing an engraving grid. fit = 0.9 * float(np.mean(start_error <= 0.2)) + 0.1 * float(np.mean(duration_error <= 0.22)) quantization_fits[label] = fit if quantization_fits["1/8"] >= 0.82: quantization = "1/8" elif quantization_fits["1/16"] >= 0.72: quantization = "1/16" elif ( quantization_fits["1/32"] >= 0.82 and quantization_fits["1/32"] >= quantization_fits["1/16"] + 0.07 ): quantization = "1/32" else: # False precision is substantially harder for a musician to correct # than a few deliberately simplified ornaments. quantization = "1/16" peak = float(tempo_scores[best_index]) baseline = float(np.quantile(tempo_scores, 0.75)) + 1e-9 tempo_separation = max(0.0, min(1.0, (peak - baseline) / (peak + 1e-9))) confidence = "high" if len(onsets) >= 24 and tempo_separation >= 0.55 else "medium" if len(onsets) >= 10 else "low" # Preserve a lightweight symbolic tempo map. Onsets close to an integer # pulse gently move that pulse; missing pulses retain the global estimate. # This lets the engraving follow moderate rubato without letting every # ornamental note distort the bar structure. raw_beat_indices = np.rint((onsets - beat_phase) / beat_seconds).astype(int) first_index = int(raw_beat_indices.min()) - 1 last_index = int(raw_beat_indices.max()) + 2 beat_times: list[float] = [] for beat_index in range(first_index, last_index + 1): predicted = beat_phase + beat_index * beat_seconds distances = np.abs(onsets - predicted) nearby = distances <= beat_seconds * 0.16 if int(np.count_nonzero(nearby)): local_weights = accents[nearby] * np.exp( -0.5 * (distances[nearby] / (beat_seconds * 0.09)) ** 2 ) observed = float(np.average(onsets[nearby], weights=local_weights)) beat_time = predicted * 0.38 + observed * 0.62 else: beat_time = predicted if beat_times: beat_time = max(beat_times[-1] + beat_seconds * 0.35, beat_time) beat_times.append(round(beat_time, 6)) first_downbeat_index = minimum_beat + bar_phase first_onset = float(onsets[0]) bar_seconds = meter * beat_seconds while beat_phase + first_downbeat_index * beat_seconds < first_onset - beat_seconds * 0.18: first_downbeat_index += meter candidate_downbeat = beat_phase + first_downbeat_index * beat_seconds previous_downbeat = candidate_downbeat - bar_seconds # Avoid labelling almost a whole opening bar as a pickup when the clip # simply starts just after a downbeat. if candidate_downbeat - first_onset > bar_seconds * 0.62: candidate_downbeat = previous_downbeat first_downbeat_index -= meter pickup_beats = max(0.0, (candidate_downbeat - first_onset) / beat_seconds) pickup_beats = round(pickup_beats * 4) / 4 if pickup_beats < 0.25 or pickup_beats >= meter: pickup_beats = 0.0 while candidate_downbeat > first_onset + beat_seconds * 0.18: candidate_downbeat -= bar_seconds first_downbeat_index -= meter beat_quarter_length = 1.5 if time_signature in {"6/8", "9/8", "12/8"} else 1.0 key_signature, key_confidence = _estimate_key_signature(tracks, np) review_flags: list[str] = [] if confidence == "low": review_flags.append("Tempo and meter have low confidence; verify them against the recording.") if key_confidence < 0.34: review_flags.append("The detected key is ambiguous; confirm pitch spelling and accidentals.") if quantization_fits[quantization] < 0.68: review_flags.append( "The rhythmic grid has low confidence; 1/16 was preferred over unreadable false precision." ) ambiguous = sorted( { str(track.get("key") or "") for track in tracks if str(track.get("key") or "") in AMBIGUOUS_INSTRUMENT_GROUPS } ) if ambiguous: review_flags.append( "Broad instrument groups need an exact player/instrument assignment: " + ", ".join(name.replace("_", " ") for name in ambiguous) + "." ) review_flags.append( "Dynamics are not predicted by MuScriptor and are intentionally left neutral." ) return { "analysis_version": 3, "tempo_bpm": round(best_bpm, 1), "time_signature": time_signature, "quantization": quantization, "key_signature": key_signature, "key_confidence": key_confidence, "confidence": confidence, "onset_count": int(len(onsets)), "tempo_confidence": round(tempo_separation, 3), "tempo_octave_adjusted": tempo_octave_adjusted, "onset_rate_hz": round(onset_rate_hz, 3), "quantization_fit": round(quantization_fits[quantization], 3), "beat_phase_seconds": round(beat_phase, 6), "first_downbeat_seconds": round(candidate_downbeat, 6), "pickup_beats": pickup_beats, "beat_quarter_length": beat_quarter_length, "beat_times_seconds": beat_times, "beat_origin_index": int(first_downbeat_index - first_index), "grid_family": "binary + ternary", "analysis_source": "symbolic notes", "review_flags": review_flags, } _INVALID_XML_CHARACTERS = re.compile(r"[\x00-\x08\x0b\x0c\x0e-\x1f]") _MUSICXML_NUMERIC_VOICE = re.compile(r"()(\d+)()") def _clean_display_text(value: Any, fallback: str) -> str: cleaned = _INVALID_XML_CHARACTERS.sub("", str(value or "")) cleaned = " ".join(cleaned.split()) return cleaned or fallback def safe_filename_stem(value: str, fallback: str = "muscriptor-score") -> str: ascii_value = ( unicodedata.normalize("NFKD", str(value or "")) .encode("ascii", "ignore") .decode("ascii") .lower() ) return re.sub(r"[^a-z0-9]+", "-", ascii_value).strip("-") or fallback def _safe_slug(value: str) -> str: return safe_filename_stem(value, "instrument") def _parse_time_signature(value: str) -> tuple[int, int]: try: numerator, denominator = value.split("/", 1) parsed = int(numerator), int(denominator) except (AttributeError, TypeError, ValueError): return 4, 4 if parsed[0] <= 0 or parsed[1] not in {1, 2, 4, 8, 16, 32}: return 4, 4 return parsed def _beat_quarter_length(time_signature: str) -> float: numerator, denominator = _parse_time_signature(time_signature) return 1.5 if denominator == 8 and numerator in {6, 9, 12} else 1.0 def _quantization_grid(label: str) -> tuple[int, ...]: return QUANTIZATION_GRIDS.get(str(label), QUANTIZATION_GRIDS["1/16"]) def _nearest_grid_value( value: float, divisors: tuple[int, ...], *, for_duration: bool = False, ) -> float: value = max(0.0, float(value)) candidates: list[float] = [] for divisor in divisors: numerator = max(0, round(value * divisor)) fraction = Fraction(numerator, divisor) remainder = fraction - int(fraction) # Durations such as 5/6 or 7/8 of a quarter are technically # representable but produce distracting 6:5 / 8:7 tuplets. Keep # ternary positions for onsets while favouring musician-readable note # values and ties for durations. if for_duration and remainder and remainder.numerator > 4: continue candidates.append(float(fraction)) if not candidates: candidates = [round(value * QUANTIZATION_DIVISORS["1/16"]) / QUANTIZATION_DIVISORS["1/16"]] return min(candidates, key=lambda candidate: (abs(candidate - value), -candidate)) def _key_for_label(value: str, key_module: Any) -> Any: text = _clean_display_text(value, "C major").replace("♭", "-").replace("b", "-").replace("♯", "#") match = re.fullmatch(r"([A-Ga-g])([#-]?)[ ]+(major|minor)", text, flags=re.IGNORECASE) if not match: return key_module.Key("C", "major") tonic = match.group(1).upper() + match.group(2) return key_module.Key(tonic, match.group(3).lower()) def _normalized_key_label(value: str) -> str: text = _clean_display_text(value, "C major").replace("♭", "-").replace("b", "-").replace("♯", "#") match = re.fullmatch(r"([A-Ga-g])([#-]?)[ ]+(major|minor)", text, flags=re.IGNORECASE) if not match: return "C major" accidental = {"-": "♭", "#": "♯", "": ""}[match.group(2)] return f"{match.group(1).upper()}{accidental} {match.group(3).lower()}" @dataclass(frozen=True) class _ScoreTiming: bpm: float beat_quarter_length: float first_downbeat_seconds: float pickup_beats: float beat_times_seconds: tuple[float, ...] beat_origin_index: int use_tempo_map: bool @property def beat_seconds(self) -> float: return 60.0 / self.bpm def _pulse_position(self, seconds: float) -> float: if not self.use_tempo_map or len(self.beat_times_seconds) < 2: return (seconds - self.first_downbeat_seconds) / self.beat_seconds values = self.beat_times_seconds index = bisect.bisect_right(values, seconds) - 1 if index < 0: spacing = max(1e-6, values[1] - values[0]) position = (seconds - values[0]) / spacing elif index >= len(values) - 1: spacing = max(1e-6, values[-1] - values[-2]) position = len(values) - 1 + (seconds - values[-1]) / spacing else: spacing = max(1e-6, values[index + 1] - values[index]) position = index + (seconds - values[index]) / spacing return position - self.beat_origin_index def offset_quarters(self, seconds: float) -> float: pulses = self._pulse_position(seconds) + self.pickup_beats return max(0.0, pulses * self.beat_quarter_length) def duration_quarters(self, start: float, end: float) -> float: start_pulse = self._pulse_position(start) end_pulse = self._pulse_position(end) return max(0.0, (end_pulse - start_pulse) * self.beat_quarter_length) def _score_timing( tracks: list[dict[str, Any]], *, bpm: float, time_signature: str, pickup_beats: float, timing_analysis: dict[str, Any] | None, ) -> _ScoreTiming: note_starts = [ max(0.0, float(item.get("start", 0.0))) for track in tracks for item in track.get("notes") or [] ] first_note = min(note_starts, default=0.0) beat_ql = _beat_quarter_length(time_signature) pickup = max(0.0, float(pickup_beats or 0.0)) numerator, denominator = _parse_time_signature(time_signature) bar_quarters = numerator * 4.0 / denominator pickup = min(pickup, max(0.0, bar_quarters / beat_ql - 0.25)) analysis = timing_analysis or {} suggested_bpm = float(analysis.get("tempo_bpm") or 0.0) suggested_pickup = float(analysis.get("pickup_beats") or 0.0) analysis_matches = ( suggested_bpm > 0 and abs(suggested_bpm - bpm) / bpm <= 0.04 and str(analysis.get("time_signature") or "") == time_signature and abs(suggested_pickup - pickup) <= 0.13 ) if analysis_matches: first_downbeat = float(analysis.get("first_downbeat_seconds") or first_note) beat_times = tuple( float(value) for value in analysis.get("beat_times_seconds") or [] if math.isfinite(float(value)) ) origin = int(analysis.get("beat_origin_index") or 0) valid_map = ( len(beat_times) >= 2 and 0 <= origin < len(beat_times) and all(right > left for left, right in zip(beat_times, beat_times[1:])) ) if valid_map: first_downbeat = beat_times[origin] else: first_downbeat = first_note + pickup * (60.0 / bpm) beat_times = () origin = 0 valid_map = False if pickup <= 0 and first_downbeat > first_note + 60.0 / bpm * 0.25: first_downbeat = first_note valid_map = False return _ScoreTiming( bpm=bpm, beat_quarter_length=beat_ql, first_downbeat_seconds=first_downbeat, pickup_beats=pickup, beat_times_seconds=beat_times, beat_origin_index=origin, use_tempo_map=valid_map, ) def _prepare_tracks( tracks: list[dict[str, Any]], *, bpm: float, time_signature: str, cleanup_profile: str, ) -> tuple[list[dict[str, Any]], dict[str, Any]]: profile_key = cleanup_profile if cleanup_profile in CLEANUP_PROFILES else "readable" profile = CLEANUP_PROFILES[profile_key] beat_seconds = 60.0 / bpm merge_gap = float(profile["merge_gap_beats"]) * beat_seconds minimum_duration = float(profile["minimum_note_beats"]) * beat_seconds ghost_duration = float(profile["drop_ghost_beats"]) * beat_seconds totals = { "input_notes": 0, "output_notes": 0, "merged_notes": 0, "lengthened_notes": 0, "dropped_notes": 0, "trimmed_overlaps": 0, } prepared: list[dict[str, Any]] = [] track_reports: list[dict[str, Any]] = [] for track in tracks: source_notes = list(track.get("notes") or []) totals["input_notes"] += len(source_notes) is_drum = bool(track.get("is_drum")) by_pitch: dict[int, list[dict[str, Any]]] = {} for item in source_notes: pitch = max(0, min(127, int(item.get("pitch", 60)))) start = max(0.0, float(item.get("start", 0.0))) end = max(start + 0.01, float(item.get("end", start + 0.1))) normalized = dict(item) normalized.update({"pitch": pitch, "start": start, "end": end}) by_pitch.setdefault(pitch, []).append(normalized) cleaned: list[dict[str, Any]] = [] track_merged = 0 track_lengthened = 0 track_dropped = 0 track_trimmed = 0 for pitch_notes in by_pitch.values(): pitch_notes.sort(key=lambda item: (item["start"], item["end"])) merged: list[dict[str, Any]] = [] for item in pitch_notes: if merged and item["start"] <= merged[-1]["end"] + merge_gap: merged[-1]["end"] = max(merged[-1]["end"], item["end"]) track_merged += 1 else: merged.append(item) for item in merged: duration = item["end"] - item["start"] if not is_drum and ghost_duration and duration < ghost_duration: track_dropped += 1 continue if not is_drum and duration < minimum_duration: item["end"] = item["start"] + minimum_duration track_lengthened += 1 cleaned.append(item) cleaned.sort(key=lambda item: (item["start"], item["pitch"], item["end"])) distinct_onsets = sorted({float(item["start"]) for item in cleaned}) trim_tolerance = max(0.02, merge_gap * 2.0) for item in cleaned: next_index = bisect.bisect_right(distinct_onsets, float(item["start"]) + 1e-9) if next_index >= len(distinct_onsets): continue next_onset = distinct_onsets[next_index] overlap = float(item["end"]) - next_onset if 0 < overlap <= trim_tolerance: item["end"] = max(float(item["start"]) + 0.01, next_onset) track_trimmed += 1 updated_track = dict(track) updated_track["notes"] = cleaned updated_track["note_count"] = len(cleaned) prepared.append(updated_track) totals["output_notes"] += len(cleaned) totals["merged_notes"] += track_merged totals["lengthened_notes"] += track_lengthened totals["dropped_notes"] += track_dropped totals["trimmed_overlaps"] += track_trimmed track_reports.append( { "track": str(track.get("name") or track.get("key") or "Instrument"), "input_notes": len(source_notes), "output_notes": len(cleaned), "merged_notes": track_merged, "lengthened_notes": track_lengthened, "dropped_notes": track_dropped, "trimmed_overlaps": track_trimmed, } ) return prepared, { "profile": profile_key, "profile_label": profile["label"], **totals, "tracks": track_reports, } def _split_keyboard_hands(notes: list[dict[str, Any]]) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: """Split piano notes with chord-aware, range-aware hand continuity.""" ordered = sorted(notes, key=lambda item: (float(item.get("start", 0.0)), int(item.get("pitch", 60)))) groups: list[list[dict[str, Any]]] = [] for item in ordered: if not groups or float(item.get("start", 0.0)) - float(groups[-1][0].get("start", 0.0)) > 0.045: groups.append([item]) else: groups[-1].append(item) lower: list[dict[str, Any]] = [] upper: list[dict[str, Any]] = [] lower_center = 48.0 upper_center = 72.0 for group in groups: group = sorted(group, key=lambda item: int(item.get("pitch", 60))) pitches = [int(item.get("pitch", 60)) for item in group] best_split = 0 best_cost = float("inf") for split in range(len(group) + 1): low_pitches = pitches[:split] high_pitches = pitches[split:] cost = 0.0 cost += sum(max(0, pitch - 64) ** 2 * 0.45 for pitch in low_pitches) cost += sum(max(0, 55 - pitch) ** 2 * 0.45 for pitch in high_pitches) if low_pitches: cost += abs(sum(low_pitches) / len(low_pitches) - lower_center) * 0.55 elif min(pitches) < 58: cost += 18 if high_pitches: cost += abs(sum(high_pitches) / len(high_pitches) - upper_center) * 0.55 elif max(pitches) > 64: cost += 18 if low_pitches and high_pitches: gap = high_pitches[0] - low_pitches[-1] cost -= min(8, max(0, gap)) * 0.6 if cost < best_cost: best_cost = cost best_split = split low_group = group[:best_split] high_group = group[best_split:] lower.extend(low_group) upper.extend(high_group) if low_group: lower_center = lower_center * 0.62 + median([int(item["pitch"]) for item in low_group]) * 0.38 if high_group: upper_center = upper_center * 0.62 + median([int(item["pitch"]) for item in high_group]) * 0.38 # A true grand staff needs usable material in both hands. The caller can # fall back to one staff when a strongly one-sided performance is found. return upper, lower def _instrument_for_track(track: dict[str, Any], instrument_module: Any, interval_module: Any) -> Any: if track.get("is_drum"): result = instrument_module.UnpitchedPercussion() result.midiChannel = 9 else: program = max(0, min(127, int(track.get("program") or 0))) try: result = instrument_module.instrumentFromMidiProgram(program) except Exception: result = instrument_module.Instrument() result.midiProgram = program octave_transposers = tuple( cls for cls in ( getattr(instrument_module, "Guitar", None), getattr(instrument_module, "AcousticGuitar", None), getattr(instrument_module, "ElectricGuitar", None), getattr(instrument_module, "AcousticBass", None), getattr(instrument_module, "ElectricBass", None), getattr(instrument_module, "Contrabass", None), ) if cls is not None ) if octave_transposers and isinstance(result, octave_transposers): result.transposition = interval_module.Interval("P-8") name = str(track.get("name") or "Instrument") result.partName = name result.instrumentName = name return result def _clef_for_track( track: dict[str, Any], midi_instrument: Any, pitches: list[int], clef_module: Any, instrument_module: Any, staff_role: str | None = None, ) -> Any: if track.get("is_drum"): return clef_module.PercussionClef() if staff_role == "upper": return clef_module.TrebleClef() if staff_role == "lower": return clef_module.BassClef() if isinstance(midi_instrument, instrument_module.Viola): return clef_module.AltoClef() bass_classes = tuple( cls for cls in ( getattr(instrument_module, "AcousticBass", None), getattr(instrument_module, "ElectricBass", None), getattr(instrument_module, "Contrabass", None), getattr(instrument_module, "Violoncello", None), getattr(instrument_module, "Bassoon", None), getattr(instrument_module, "Trombone", None), getattr(instrument_module, "BassTrombone", None), getattr(instrument_module, "Tuba", None), ) if cls is not None ) if bass_classes and isinstance(midi_instrument, bass_classes): return clef_module.BassClef() guitar_classes = tuple( cls for cls in ( getattr(instrument_module, "Guitar", None), getattr(instrument_module, "AcousticGuitar", None), getattr(instrument_module, "ElectricGuitar", None), ) if cls is not None ) if guitar_classes and isinstance(midi_instrument, guitar_classes): return clef_module.Treble8vbClef() return clef_module.BassClef() if median(pitches) < 52 else clef_module.TrebleClef() def _drum_note(pitch: int, duration: float, note_module: Any, instrument_module: Any) -> Any: if pitch in {35, 36}: display, class_name = "F4", "BassDrum" elif pitch in {37, 38, 39, 40}: display, class_name = "C5", "SnareDrum" elif pitch in {41, 43, 45, 47, 48, 50}: display, class_name = "A4", "TomTom" elif pitch in {42, 44, 46}: display, class_name = "G5", "HiHatCymbal" elif pitch in {49, 51, 52, 53, 55, 57, 59}: display, class_name = "A5", "Cymbals" elif pitch == 54: display, class_name = "E5", "Tambourine" elif pitch == 56: display, class_name = "D5", "Cowbell" elif pitch in {60, 61}: display, class_name = "F5", "BongoDrums" elif pitch in {62, 63, 64}: display, class_name = "E5", "CongaDrum" else: display, class_name = "B4", "UnpitchedPercussion" stored_instrument = getattr(instrument_module, class_name)() return note_module.Unpitched( displayName=display, storedInstrument=stored_instrument, quarterLength=duration, ) def _build_music21_staff( track: dict[str, Any], notes: list[dict[str, Any]], *, staff_id: str, staff_role: str | None, display_name: str, numerator: int, denominator: int, divisors: tuple[int, ...], timing: _ScoreTiming, key_signature: str, show_solfege: bool, add_tempo: bool, modules: dict[str, Any], ) -> Any: chord = modules["chord"] clef = modules["clef"] duration_module = modules["duration"] instrument = modules["instrument"] interval = modules["interval"] key_module = modules["key"] meter = modules["meter"] note = modules["note"] percussion = modules["percussion"] stream = modules["stream"] tempo = modules["tempo"] part_class = stream.PartStaff if staff_role else stream.Part part = part_class(id=staff_id) part.partName = display_name midi_instrument = _instrument_for_track(track, instrument, interval) midi_instrument.partName = display_name part.insert(0, midi_instrument) if add_tempo: part.insert( 0, tempo.MetronomeMark( number=timing.bpm, referent=duration_module.Duration(timing.beat_quarter_length), ), ) part.insert(0, meter.TimeSignature(f"{numerator}/{denominator}")) part.insert(0, _key_for_label(key_signature, key_module)) pitches = [max(0, min(127, int(item.get("pitch", 60)))) for item in notes] part.insert(0, _clef_for_track(track, midi_instrument, pitches, clef, instrument, staff_role)) grouped: dict[float, list[tuple[int, float]]] = {} bar_quarters = numerator * 4.0 / denominator pickup_quarters = timing.pickup_beats * timing.beat_quarter_length pickup_padding = max(0.0, bar_quarters - pickup_quarters) if pickup_quarters else 0.0 minimum_duration = 1.0 / max(divisors) for item in notes: start_seconds = max(0.0, float(item.get("start", 0.0))) end_seconds = max(start_seconds + 0.03, float(item.get("end", start_seconds + 0.1))) raw_offset = timing.offset_quarters(start_seconds) + pickup_padding raw_duration = timing.duration_quarters(start_seconds, end_seconds) offset = _nearest_grid_value(raw_offset, divisors) duration = max( minimum_duration, _nearest_grid_value(raw_duration, divisors, for_duration=True), ) grouped.setdefault(round(offset, 6), []).append( ( max(0, min(127, int(item.get("pitch", 60)))), round(duration, 6), ) ) for offset, values in sorted(grouped.items()): # A performed chord rarely has identical note-off times. Group by the # quantized onset and use a representative duration instead of asking # makeVoices() to create one notated voice per key release. durations_by_pitch: dict[int, float] = {} for pitch, duration in values: durations_by_pitch[pitch] = max(durations_by_pitch.get(pitch, 0.0), duration) group_pitches = list(durations_by_pitch) duration = max(minimum_duration, float(median(durations_by_pitch.values()))) if track.get("is_drum"): unpitched = [_drum_note(pitch, duration, note, instrument) for pitch in group_pitches] musical_item = ( unpitched[0] if len(unpitched) == 1 else percussion.PercussionChord(unpitched, quarterLength=duration) ) elif len(group_pitches) == 1: musical_item = note.Note(group_pitches[0], quarterLength=duration) if show_solfege: musical_item.addLyric(SOLFEGE_NAMES[group_pitches[0] % 12]) else: unique_pitches = list(dict.fromkeys(group_pitches)) musical_item = chord.Chord(unique_pitches, quarterLength=duration) if show_solfege: musical_item.addLyric(" ".join(SOLFEGE_NAMES[pitch % 12] for pitch in unique_pitches)) part.insert(offset, musical_item) part.quantize( quarterLengthDivisors=divisors, processOffsets=True, processDurations=True, inPlace=True, ) voiced = part.makeVoices(inPlace=False, fillGaps=False) measured = voiced.makeMeasures(inPlace=False) measured.makeNotation(inPlace=True) if pickup_quarters: first_measure = measured.measure(1) if first_measure is not None: containers = list(first_measure.voices) or [first_measure] for container in containers: leading_rests = [ item for item in container.notesAndRests if isinstance(item, note.Rest) and float(item.offset) < pickup_padding ] if leading_rests: container.remove(leading_rests) for item in container.notesAndRests: item.offset = max(0.0, float(item.offset) - pickup_padding) first_measure.paddingLeft = pickup_padding first_measure.number = 0 first_measure.showNumber = stream.enums.ShowNumber.NEVER # MuScriptor pitches are concert/sounding pitches. Converting here gives # clarinet, saxophone, trumpet and horn players proper written parts while # keeping MusicXML playback at the original sounding pitch. measured.atSoundingPitch = True measured.toWrittenPitch(inPlace=True) return measured def _music21_score( tracks: list[dict[str, Any]], *, title: str, tempo_bpm: float, time_signature: str, quantization: str, show_solfege: bool, key_signature: str = "C major", pickup_beats: float = 0.0, cleanup_profile: str = "readable", timing_analysis: dict[str, Any] | None = None, tracks_prepared: bool = False, ) -> tuple[Any, list[dict[str, Any]]]: try: from music21 import ( chord, clef, duration, instrument, interval, key, layout, metadata, meter, note, percussion, stream, tempo, ) except ImportError as exc: # pragma: no cover - deployment configuration raise RuntimeError("La dépendance music21 est nécessaire pour générer la partition.") from exc modules = { "chord": chord, "clef": clef, "duration": duration, "instrument": instrument, "interval": interval, "key": key, "meter": meter, "note": note, "percussion": percussion, "stream": stream, "tempo": tempo, } bpm = max(20.0, min(300.0, float(tempo_bpm))) numerator, denominator = _parse_time_signature(time_signature) divisors = _quantization_grid(quantization) if not tracks_prepared: tracks, _ = _prepare_tracks( tracks, bpm=bpm, time_signature=time_signature, cleanup_profile=cleanup_profile, ) timing = _score_timing( tracks, bpm=bpm, time_signature=time_signature, pickup_beats=pickup_beats, timing_analysis=timing_analysis, ) score = stream.Score(id="muscriptor-score") score.metadata = metadata.Metadata() score.metadata.title = _clean_display_text(title, "MuScriptor transcription") score.metadata.composer = "Transcription automatique MuScriptor" built_tracks: list[dict[str, Any]] = [] for track_index, track in enumerate(tracks): track_notes = list(track.get("notes") or []) if not track_notes: continue track_name = _clean_display_text(track.get("name"), f"Instrument {track_index + 1}") key = str(track.get("key") or "").lower() pitches = [int(item.get("pitch", 60)) for item in track_notes] grand_staff = ( any(token in key for token in ("piano", "keyboard", "organ", "harp")) and min(pitches) < 60 <= max(pitches) and len(track_notes) >= 4 ) parts: list[Any] = [] if grand_staff: upper_notes, lower_notes = _split_keyboard_hands(track_notes) staff_specs = ( ("upper", upper_notes, track_name), ("lower", lower_notes, ""), ) if not upper_notes or not lower_notes: staff_specs = ((None, track_notes, track_name),) else: staff_specs = ((None, track_notes, track_name),) for staff_number, (staff_role, staff_notes, display_name) in enumerate(staff_specs, start=1): if not staff_notes: continue measured = _build_music21_staff( track, staff_notes, staff_id=f"track-{track_index + 1}-staff-{staff_number}", staff_role=staff_role, display_name=display_name, numerator=numerator, denominator=denominator, divisors=divisors, timing=timing, key_signature=key_signature, show_solfege=show_solfege, add_tempo=staff_number == 1, modules=modules, ) score.insert(0, measured) parts.append(measured) if len(parts) > 1: score.insert( 0, layout.StaffGroup( *parts, name=track_name, abbreviation=track_name[:8], symbol="brace", barTogether=True, ), ) built_tracks.append({"track": track, "name": track_name, "parts": tuple(parts)}) if not score.parts: raise ValueError("Aucune note n’est disponible pour générer une partition.") return score, built_tracks def _write_musicxml(score: Any, output_path: Path) -> Path: """Write well-formed UTF-8 MusicXML and remove filename control chars. music21 preserves control characters found in an uploaded filename. Those characters make the document invalid XML for strict readers even though some Verovio builds accept them, so sanitize and validate every export. """ score.write("musicxml", fp=str(output_path)) xml_text = output_path.read_text(encoding="utf-8-sig") cleaned = _INVALID_XML_CHARACTERS.sub("", xml_text) # music21 numbers Voice objects from zero in some polyphonic exports. # MusicXML technically accepts a token here, but Verovio maps numeric # voices to one-based MEI layers and reports every 0 as a # missing layer. If a zero-based voice is present, shift the complete # numeric voice set together so voices remain distinct and portable. if "0" in cleaned: cleaned = _MUSICXML_NUMERIC_VOICE.sub( lambda match: ( f"{match.group(1)}{int(match.group(2)) + 1}{match.group(3)}" ), cleaned, ) if cleaned != xml_text: output_path.write_text(cleaned, encoding="utf-8") try: ElementTree.fromstring(cleaned.encode("utf-8")) except ElementTree.ParseError as exc: raise RuntimeError( f"The generated MusicXML is not well formed ({output_path.name}): {exc}" ) from exc return output_path def _verovio_options(layout_kind: str) -> dict[str, Any]: is_part = layout_kind == "part" return { "adjustPageHeight": False, "breaks": "auto", "footer": "none", "header": "none", # Full scores use A4 landscape; individual parts use A4 portrait. "pageHeight": 2970 if is_part else 2100, "pageWidth": 2100 if is_part else 2970, "pageMarginBottom": 80, "pageMarginLeft": 80, "pageMarginRight": 80, "pageMarginTop": 80, "mmOutput": True, "scale": 100, # Verovio's documented raster guidance uses about 9 units for parts # and a smaller staff for a multi-instrument conductor score. "unit": 9.0 if is_part else 6.875, "spacingStaff": 12 if is_part else 10, "spacingSystem": 18 if is_part else 15, "minLastJustification": 0.62, } def _verovio_toolkit(verovio_module: Any, layout_kind: str = "score") -> Any: """Create a toolkit with an explicit wheel resource path when available.""" module_file = getattr(verovio_module, "__file__", "") resource_path = Path(module_file).resolve().parent / "data" if module_file else None toolkit = None if resource_path and resource_path.is_dir(): try: toolkit = verovio_module.toolkit(False) if not toolkit.setResourcePath(str(resource_path)): toolkit = None except (AttributeError, TypeError): toolkit = None if toolkit is None: toolkit = verovio_module.toolkit() toolkit.setOptions(_verovio_options(layout_kind)) return toolkit def _verovio_log(toolkit: Any) -> str: try: raw_log = toolkit.getLog() except Exception: raw_log = "" lines = [line.strip() for line in str(raw_log or "").splitlines() if line.strip()] return " · ".join(lines[-3:])[-600:] if lines else "no parser details" def _load_verovio_musicxml( verovio_module: Any, musicxml: Path, xml_text: str, *, layout_kind: str = "score", ) -> Any: """Load MusicXML through both native Python routes before giving up. `loadFile` is Verovio's documented Python path and avoids copying a large score through the binding. `loadData` remains a useful independent retry for unusual temporary-file or filesystem behaviour on hosted Spaces. """ diagnostics: list[str] = [] for method in ("file", "data"): toolkit = _verovio_toolkit(verovio_module, layout_kind) try: loaded = ( toolkit.loadFile(str(musicxml.resolve())) if method == "file" else toolkit.loadData(xml_text) ) except Exception as exc: diagnostics.append(f"{method} loader: {type(exc).__name__}: {exc}") continue if loaded: return toolkit diagnostics.append(f"{method} loader: {_verovio_log(toolkit)}") root_name = ElementTree.fromstring(xml_text.encode("utf-8")).tag detail = " · ".join(diagnostics)[-1200:] raise RuntimeError( f"Verovio could not read {musicxml.name} after file and data loading " f"({root_name}, {len(xml_text.encode('utf-8'))} bytes): {detail}" ) def _render_svg_pages( musicxml: Path, output_dir: Path, *, page_stem: str, layout_kind: str = "score", ) -> tuple[tuple[Path, ...], str]: try: import verovio except ImportError as exc: # pragma: no cover - deployment configuration raise RuntimeError("La dépendance verovio est nécessaire pour afficher la partition.") from exc xml_text = _INVALID_XML_CHARACTERS.sub("", musicxml.read_text(encoding="utf-8-sig")) try: ElementTree.fromstring(xml_text.encode("utf-8")) except ElementTree.ParseError as exc: raise RuntimeError(f"Invalid MusicXML in {musicxml.name}: {exc}") from exc toolkit = _load_verovio_musicxml( verovio, musicxml, xml_text, layout_kind=layout_kind, ) pages: list[Path] = [] preview = "" for page_number in range(1, toolkit.getPageCount() + 1): svg = toolkit.renderToSVG(page_number) if page_number == 1: preview = svg path = output_dir / f"{page_stem}-page-{page_number:02d}.svg" path.write_text(svg, encoding="utf-8") pages.append(path) return tuple(pages), preview def _render_pdf(svg_pages: tuple[Path, ...], output_path: Path) -> Path | None: try: import cairosvg from pypdf import PdfReader, PdfWriter except ImportError: return None writer = PdfWriter() for svg in svg_pages: pdf_bytes = cairosvg.svg2pdf(bytestring=svg.read_bytes()) reader = PdfReader(io.BytesIO(pdf_bytes)) for page in reader.pages: writer.add_page(page) with output_path.open("wb") as handle: writer.write(handle) return output_path def generate_notation( tracks: list[dict[str, Any]], *, title: str, tempo_bpm: float = 120, time_signature: str = "4/4", quantization: str = "1/16", key_signature: str = "C major", pickup_beats: float = 0.0, cleanup_profile: str = "readable", show_solfege: bool = False, midi_files: list[str] | None = None, timing_analysis: dict[str, Any] | None = None, ) -> NotationResult: output_dir = Path(tempfile.mkdtemp(prefix="muscriptor-notation-")) clean_title = _clean_display_text(title, "MuScriptor transcription") export_stem = safe_filename_stem(clean_title) bpm = max(20.0, min(300.0, float(tempo_bpm))) numerator, denominator = _parse_time_signature(time_signature) time_signature = f"{numerator}/{denominator}" quantization = quantization if quantization in QUANTIZATION_GRIDS else "1/16" key_signature = _normalized_key_label(key_signature) profile_key = cleanup_profile if cleanup_profile in CLEANUP_PROFILES else "readable" prepared_tracks, cleanup_diagnostics = _prepare_tracks( tracks, bpm=bpm, time_signature=time_signature, cleanup_profile=profile_key, ) timing = _score_timing( prepared_tracks, bpm=bpm, time_signature=time_signature, pickup_beats=pickup_beats, timing_analysis=timing_analysis, ) score, built_tracks = _music21_score( prepared_tracks, title=clean_title, tempo_bpm=bpm, time_signature=time_signature, quantization=quantization, key_signature=key_signature, pickup_beats=pickup_beats, cleanup_profile=profile_key, timing_analysis=timing_analysis, tracks_prepared=True, show_solfege=show_solfege, ) musicxml = output_dir / f"{export_stem}-full-score.musicxml" _write_musicxml(score, musicxml) warnings: list[str] = [] svg_pages: tuple[Path, ...] = () preview_svg = "" pdf: Path | None = None try: svg_pages, preview_svg = _render_svg_pages( musicxml, output_dir, page_stem=f"{export_stem}-full-score", ) except Exception as exc: warnings.append(f"Full score preview: {type(exc).__name__}: {exc}") if svg_pages: try: pdf = _render_pdf(svg_pages, output_dir / f"{export_stem}-full-score.pdf") except Exception as exc: warnings.append(f"Full score PDF: {type(exc).__name__}: {exc}") parts_dir = output_dir / "parts" parts_dir.mkdir() from music21 import layout as music21_layout from music21 import stream as music21_stream part_results: list[NotationPartResult] = [] for part_index, built_track in enumerate(built_tracks, start=1): source_track = built_track["track"] part_name = _clean_display_text(built_track["name"], f"Instrument {part_index}") part_slug = f"{export_stem}-{part_index:02d}-{_safe_slug(part_name)}" part_score = music21_stream.Score(id=f"score-track-{part_index}") part_score.metadata = copy.deepcopy(score.metadata) copied_parts = [copy.deepcopy(part) for part in built_track["parts"]] for copied_part in copied_parts: part_score.insert(0, copied_part) if len(copied_parts) > 1: part_score.insert( 0, music21_layout.StaffGroup( *copied_parts, name=part_name, abbreviation=part_name[:8], symbol="brace", barTogether=True, ), ) part_musicxml = parts_dir / f"{part_slug}.musicxml" _write_musicxml(part_score, part_musicxml) part_svg_dir = parts_dir / f"{part_slug}-svg" part_svg_dir.mkdir() part_svg_pages: tuple[Path, ...] = () part_pdf: Path | None = None part_warnings: list[str] = [] try: part_svg_pages, _ = _render_svg_pages( part_musicxml, part_svg_dir, page_stem=part_slug, layout_kind="part", ) except Exception as exc: part_warnings.append(f"{part_name} preview: {type(exc).__name__}: {exc}") if part_svg_pages: try: part_pdf = _render_pdf(part_svg_pages, parts_dir / f"{part_slug}.pdf") except Exception as exc: part_warnings.append(f"{part_name} PDF: {type(exc).__name__}: {exc}") warnings.extend(part_warnings) source_id = source_track.get("id") if source_id is None: source_id = source_track.get("key") or part_index part_results.append( NotationPartResult( track_id=str(source_id), track_key=str(source_track.get("key") or source_track.get("id") or part_slug), name=part_name, slug=part_slug, musicxml=part_musicxml, pdf=part_pdf, svg_pages=part_svg_pages, render_error=" · ".join(part_warnings), ) ) review_flags = list(dict.fromkeys((timing_analysis or {}).get("review_flags") or [])) dynamics_flag = "Dynamics are not predicted by MuScriptor and are intentionally left neutral." if dynamics_flag not in review_flags: review_flags.append(dynamics_flag) if cleanup_diagnostics["dropped_notes"]: review_flags.append( f"{cleanup_diagnostics['dropped_notes']} very short note(s) were removed by the Readable profile." ) if cleanup_diagnostics["lengthened_notes"]: review_flags.append( f"{cleanup_diagnostics['lengthened_notes']} short note(s) were lengthened to remain engravable." ) if cleanup_diagnostics["trimmed_overlaps"]: review_flags.append( f"{cleanup_diagnostics['trimmed_overlaps']} small note-off overlap(s) were trimmed to avoid spurious voices." ) diagnostics: dict[str, Any] = { "format_version": 1, "title": clean_title, "settings": { "tempo_bpm": round(bpm, 3), "tempo_referent": ( "dotted quarter" if timing.beat_quarter_length == 1.5 else "quarter" ), "time_signature": time_signature, "key_signature": key_signature, "quantization": quantization, "quantization_divisors": list(_quantization_grid(quantization)), "grid_family": "binary + ternary", "pickup_beats": round(timing.pickup_beats, 3), "cleanup_profile": profile_key, "solfege_lyrics": bool(show_solfege), "dynamics": "neutral; MuScriptor does not predict note velocity", }, "timing": { "first_downbeat_seconds": round(timing.first_downbeat_seconds, 6), "symbolic_tempo_map_used": timing.use_tempo_map, "beat_anchor_count": len(timing.beat_times_seconds) if timing.use_tempo_map else 0, }, "cleanup": cleanup_diagnostics, "analysis": { key: value for key, value in (timing_analysis or {}).items() if key != "beat_times_seconds" }, "outputs": { "part_count": len(part_results), "full_score_pages": len(svg_pages), }, "review_flags": list(dict.fromkeys(review_flags)), "render_warnings": list(warnings), } report = output_dir / f"{export_stem}-score-preparation.json" report.write_text( json.dumps(diagnostics, ensure_ascii=False, indent=2, sort_keys=True), encoding="utf-8", ) bundle = output_dir / f"{export_stem}-exports.zip" with zipfile.ZipFile(bundle, "w", compression=zipfile.ZIP_DEFLATED) as archive: archive.write(musicxml, musicxml.name) archive.write(report, report.name) for part_result in part_results: archive.write(part_result.musicxml, f"parts/{part_result.musicxml.name}") if part_result.pdf: archive.write(part_result.pdf, f"parts/{part_result.pdf.name}") for path in part_result.svg_pages: archive.write(path, f"parts/svg/{part_result.slug}/{path.name}") for path in svg_pages: archive.write(path, f"svg/{path.name}") if pdf: archive.write(pdf, pdf.name) for value in midi_files or []: midi_path = Path(value) if midi_path.is_file(): archive.write(midi_path, f"midi/{midi_path.name}") return NotationResult( musicxml=musicxml, pdf=pdf, bundle=bundle, svg_pages=svg_pages, preview_svg=preview_svg, parts=tuple(part_results), warnings=tuple(warnings), export_stem=export_stem, report=report, diagnostics=diagnostics, ) def session_json(value: dict[str, Any]) -> str: return json.dumps(value, ensure_ascii=False, separators=(",", ":"))