"""Audio processing — normalize, silence stats, threshold auto-tune. Three related operations that don't belong in engine.py (which is about splicing segments) but also don't deserve their own module each. - normalize: Apply EBU R128 two-pass loudness normalization so different recordings have consistent volume. - silence_stats: Quantitative summary of silence distribution — useful for picking the right threshold for a given recording. - threshold_tune: Auto-pick the silence threshold that best separates speech from breath pauses for the given audio. """ from __future__ import annotations import math import re import subprocess from dataclasses import dataclass, asdict from pathlib import Path from typing import List, Tuple # --------------------------------------------------------------------------- # normalize # --------------------------------------------------------------------------- def normalize_loudness( src: Path, out: Path, *, target_lufs: float = -16.0, true_peak_db: float = -1.5, loudness_range_lu: float = 11.0, ) -> dict: """Normalize audio to broadcast-standard EBU R128 loudness. Two-pass: first pass measures the integrated loudness, second pass applies the corrective gain. Default target is -16 LUFS (podcast/ YouTube standard). For speech-only sermon audio, -19 LUFS (Spotify standard) is also reasonable. Returns a dict with measured/achieved loudness metadata so callers can log it. """ src = Path(src) out = Path(out) # Pass 1: measure. ffmpeg's loudnorm filter in print-only mode. measure = subprocess.run([ "ffmpeg", "-v", "info", "-i", str(src), "-af", f"loudnorm=I={target_lufs}:TP={true_peak_db}:LRA={loudness_range_lu}:print_format=json", "-f", "null", "-", ], capture_output=True, text=True) # Parse the JSON block ffmpeg prints at the end of stderr. # Looks like: # { # "input_i" : "-23.81", # "input_tp" : "-3.02", # ... # } measured = _parse_loudnorm_json(measure.stderr) # Pass 2: apply. Reuse the measured values for linear mode (single-pass # would do, but linear mode produces slightly better results for highly # dynamic content). apply = subprocess.run([ "ffmpeg", "-y", "-v", "error", "-i", str(src), "-af", ( f"loudnorm=I={target_lufs}:TP={true_peak_db}:LRA={loudness_range_lu}" f":measured_I={measured['input_i']}" f":measured_TP={measured['input_tp']}" f":measured_LRA={measured['input_lra']}" f":measured_thresh={measured['input_thresh']}" f":offset={measured['target_offset']}" f":linear=true:print_format=summary" ), "-ar", "48000", # output 48kHz (industry standard for normalized audio) str(out), ], capture_output=True, text=True) if apply.returncode != 0: raise RuntimeError(f"normalize failed: {apply.stderr}") return { "target_lufs": target_lufs, "true_peak_db": true_peak_db, "measured_input_i": measured.get("input_i"), "measured_input_tp": measured.get("input_tp"), "measured_input_lra": measured.get("input_lra"), "applied_offset": measured.get("target_offset"), "output_path": str(out), } _LOUDNORM_JSON_RE = re.compile(r"\{[^{}]*\"input_i\"[^{}]*\}", re.DOTALL) def _parse_loudnorm_json(stderr: str) -> dict: """Extract the loudnorm JSON block from ffmpeg stderr. ffmpeg prints this at the end after the audio is 'rendered' to null. """ m = _LOUDNORM_JSON_RE.search(stderr) if not m: raise RuntimeError( f"could not find loudnorm JSON in ffmpeg output. " f"stderr tail: {stderr[-500:]!r}" ) import json return json.loads(m.group(0)) # --------------------------------------------------------------------------- # silence_stats # --------------------------------------------------------------------------- @dataclass class SilenceStats: """Quantitative summary of silence distribution in an audio file.""" audio_path: str duration_seconds: float threshold_db: float min_duration_seconds: float n_silence_runs: int total_silence_seconds: float silence_fraction: float # total_silence / duration mean_silence_seconds: float median_silence_seconds: float longest_silence_seconds: float longest_silence_range: Tuple[float, float] # (start, end) of longest silence_per_minute: float # mean silence runs per 60 seconds def to_dict(self) -> dict: d = asdict(self) d["longest_silence_range"] = list(self.longest_silence_range) return d def silence_stats( src: Path, *, threshold_db: float = -35.0, min_duration_seconds: float = 0.3, ) -> SilenceStats: """Compute quantitative silence statistics for the audio. Useful for: - Comparing recordings (consistent silence fraction = consistent recording setup) - Picking the right silence threshold (dense audio with mean silence <0.5s probably needs a lower threshold than sparse audio with mean silence >2s) - Detecting "dead air" recordings where the speaker lost their place """ from .waveform import find_silence_runs, _extract_pcm src = Path(src) samples = _extract_pcm(src) duration = len(samples) / 8000.0 if duration <= 0: raise ValueError(f"audio {src} has zero duration") runs = find_silence_runs( src, threshold_db=threshold_db, min_duration_seconds=min_duration_seconds, ) if not runs: return SilenceStats( audio_path=str(src), duration_seconds=duration, threshold_db=threshold_db, min_duration_seconds=min_duration_seconds, n_silence_runs=0, total_silence_seconds=0.0, silence_fraction=0.0, mean_silence_seconds=0.0, median_silence_seconds=0.0, longest_silence_seconds=0.0, longest_silence_range=(0.0, 0.0), silence_per_minute=0.0, ) durations = sorted(b - a for a, b in runs) total = sum(durations) longest = durations[-1] longest_range = next((a, b) for a, b in runs if b - a == longest) return SilenceStats( audio_path=str(src), duration_seconds=duration, threshold_db=threshold_db, min_duration_seconds=min_duration_seconds, n_silence_runs=len(runs), total_silence_seconds=total, silence_fraction=total / duration, mean_silence_seconds=total / len(runs), median_silence_seconds=_median(durations), longest_silence_seconds=longest, longest_silence_range=longest_range, silence_per_minute=len(runs) / (duration / 60.0), ) def _median(xs: List[float]) -> float: if not xs: return 0.0 n = len(xs) mid = n // 2 if n % 2 == 0: return (xs[mid - 1] + xs[mid]) / 2.0 return xs[mid] # --------------------------------------------------------------------------- # threshold_tune # --------------------------------------------------------------------------- def threshold_tune( src: Path, *, candidates: List[float] = (-25.0, -30.0, -35.0, -40.0, -45.0, -50.0), target_silence_per_minute: float = 4.0, target_min_duration: float = 0.3, ) -> dict: """Auto-pick the silence threshold that best matches expected pause density. For typical sermon audio, a speaker pauses ~3-5 times per minute for breath/sentence-end. This scans a set of candidate thresholds and picks the one whose silence count is closest to `target_silence_per_minute`. Returns the picked threshold plus the per-candidate stats so callers can show a tuning report. """ from .waveform import find_silence_runs src = Path(src) rows = [] for thresh in candidates: runs = find_silence_runs( src, threshold_db=thresh, min_duration_seconds=target_min_duration, ) # We need duration to compute silence_per_minute from .waveform import _extract_pcm samples = _extract_pcm(src) duration = len(samples) / 8000.0 spm = len(runs) / (duration / 60.0) if duration > 0 else 0.0 rows.append({ "threshold_db": thresh, "n_runs": len(runs), "silence_per_minute": round(spm, 2), "distance_from_target": round(abs(spm - target_silence_per_minute), 2), }) rows.sort(key=lambda r: r["distance_from_target"]) best = rows[0] return { "picked_threshold_db": best["threshold_db"], "picked_silence_per_minute": best["silence_per_minute"], "target_silence_per_minute": target_silence_per_minute, "candidates": rows, }