| from __future__ import annotations |
|
|
| from dataclasses import dataclass |
|
|
| import librosa |
| import numpy as np |
| import pyworld |
| from scipy import signal |
|
|
|
|
| TARGET_SAMPLE_RATE = 16_000 |
| MAX_DURATION_SECONDS = 15 |
| MAX_SAMPLES = TARGET_SAMPLE_RATE * MAX_DURATION_SECONDS |
| MIN_LOW_CUT_HZ = 20 |
| MAX_HIGH_CUT_HZ = 7_900 |
| MIN_BANDWIDTH_HZ = 100 |
|
|
|
|
| class AudioProcessingError(ValueError): |
| """Raised for user-facing audio processing errors.""" |
|
|
|
|
| @dataclass(frozen=True) |
| class ProcessingSettings: |
| pitch_semitones: int |
| low_cut_hz: float |
| high_cut_hz: float |
|
|
|
|
| def convert_voice( |
| audio_input: tuple[int, np.ndarray] | None, |
| pitch_semitones: int, |
| low_cut_hz: float, |
| high_cut_hz: float, |
| ) -> tuple[int, np.ndarray, str]: |
| settings = ProcessingSettings( |
| pitch_semitones=int(pitch_semitones), |
| low_cut_hz=float(low_cut_hz), |
| high_cut_hz=float(high_cut_hz), |
| ) |
| _validate_filter_settings(settings.low_cut_hz, settings.high_cut_hz) |
|
|
| audio = _prepare_audio(audio_input) |
| converted = _change_pitch_with_world(audio, settings.pitch_semitones) |
| filtered = _apply_band_filter( |
| converted, |
| TARGET_SAMPLE_RATE, |
| settings.low_cut_hz, |
| settings.high_cut_hz, |
| ) |
| normalized = _normalize_audio(filtered) |
|
|
| if normalized.size == 0: |
| raise AudioProcessingError("出力音声が空です。") |
| if not np.all(np.isfinite(normalized)): |
| raise AudioProcessingError("出力音声に不正な値が含まれています。") |
|
|
| message = _build_success_message(settings) |
| return TARGET_SAMPLE_RATE, normalized.astype(np.float32), message |
|
|
|
|
| def _prepare_audio(audio_input: tuple[int, np.ndarray] | None) -> np.ndarray: |
| if audio_input is None: |
| raise AudioProcessingError("音声をアップロードするか、マイクで録音してください。") |
|
|
| sample_rate, waveform = audio_input |
| if sample_rate is None or int(sample_rate) <= 0: |
| raise AudioProcessingError("サンプリング周波数が不正です。") |
| if waveform is None: |
| raise AudioProcessingError("音声データが空です。") |
|
|
| audio = np.asarray(waveform) |
| if audio.size == 0: |
| raise AudioProcessingError("音声データが空です。") |
| if not np.all(np.isfinite(audio)): |
| raise AudioProcessingError("音声データにNaNまたはInfが含まれています。") |
|
|
| audio = _to_float_audio(audio) |
| audio = _to_mono(audio) |
|
|
| if int(sample_rate) != TARGET_SAMPLE_RATE: |
| audio = librosa.resample( |
| y=audio, |
| orig_sr=int(sample_rate), |
| target_sr=TARGET_SAMPLE_RATE, |
| ) |
|
|
| audio = audio[:MAX_SAMPLES] |
| if audio.size == 0: |
| raise AudioProcessingError("音声データが空です。") |
|
|
| return np.ascontiguousarray(audio, dtype=np.float64) |
|
|
|
|
| def _to_float_audio(audio: np.ndarray) -> np.ndarray: |
| if np.issubdtype(audio.dtype, np.integer): |
| info = np.iinfo(audio.dtype) |
| scale = max(abs(info.min), info.max) |
| return audio.astype(np.float64) / scale |
|
|
| audio = audio.astype(np.float64) |
| peak = float(np.max(np.abs(audio))) |
| if peak > 1.0: |
| audio = audio / peak |
| return audio |
|
|
|
|
| def _to_mono(audio: np.ndarray) -> np.ndarray: |
| if audio.ndim == 1: |
| return audio |
| if audio.ndim != 2: |
| raise AudioProcessingError("音声データの形式が不正です。") |
|
|
| if audio.shape[1] in (1, 2): |
| return np.mean(audio, axis=1) |
| if audio.shape[0] in (1, 2): |
| return np.mean(audio, axis=0) |
|
|
| raise AudioProcessingError("音声チャンネル数が不正です。") |
|
|
|
|
| def _change_pitch_with_world(audio: np.ndarray, pitch_semitones: int) -> np.ndarray: |
| try: |
| f0, time_axis = pyworld.harvest(audio, TARGET_SAMPLE_RATE) |
| f0 = pyworld.stonemask(audio, f0, time_axis, TARGET_SAMPLE_RATE) |
| spectral_envelope = pyworld.cheaptrick(audio, f0, time_axis, TARGET_SAMPLE_RATE) |
| aperiodicity = pyworld.d4c(audio, f0, time_axis, TARGET_SAMPLE_RATE) |
|
|
| pitch_ratio = 2 ** (pitch_semitones / 12) |
| converted_f0 = f0 * pitch_ratio |
|
|
| synthesized = pyworld.synthesize( |
| converted_f0, |
| spectral_envelope, |
| aperiodicity, |
| TARGET_SAMPLE_RATE, |
| ) |
| except Exception as error: |
| raise AudioProcessingError(f"WORLD分析または再合成に失敗しました: {error}") from error |
|
|
| if synthesized.size == 0: |
| raise AudioProcessingError("WORLD再合成後の音声が空です。") |
|
|
| synthesized = np.nan_to_num(synthesized, nan=0.0, posinf=0.0, neginf=0.0) |
| return np.ascontiguousarray(synthesized[: audio.size], dtype=np.float64) |
|
|
|
|
| def _apply_band_filter( |
| audio: np.ndarray, |
| sample_rate: int, |
| low_cut_hz: float, |
| high_cut_hz: float, |
| order: int = 4, |
| ) -> np.ndarray: |
| if low_cut_hz <= MIN_LOW_CUT_HZ and high_cut_hz >= MAX_HIGH_CUT_HZ: |
| return audio |
|
|
| nyquist = sample_rate / 2 |
|
|
| if low_cut_hz <= MIN_LOW_CUT_HZ: |
| sos = signal.butter( |
| order, |
| high_cut_hz, |
| btype="lowpass", |
| fs=sample_rate, |
| output="sos", |
| ) |
| elif high_cut_hz >= MAX_HIGH_CUT_HZ: |
| sos = signal.butter( |
| order, |
| low_cut_hz, |
| btype="highpass", |
| fs=sample_rate, |
| output="sos", |
| ) |
| else: |
| sos = signal.butter( |
| order, |
| [low_cut_hz, min(high_cut_hz, nyquist - 1)], |
| btype="bandpass", |
| fs=sample_rate, |
| output="sos", |
| ) |
|
|
| try: |
| return signal.sosfiltfilt(sos, audio) |
| except ValueError: |
| return signal.sosfilt(sos, audio) |
|
|
|
|
| def _normalize_audio(audio: np.ndarray) -> np.ndarray: |
| audio = np.nan_to_num(audio, nan=0.0, posinf=0.0, neginf=0.0) |
| peak = float(np.max(np.abs(audio))) if audio.size else 0.0 |
| if peak > 0: |
| audio = audio / peak * 0.95 |
| return audio |
|
|
|
|
| def _validate_filter_settings(low_cut_hz: float, high_cut_hz: float) -> None: |
| if high_cut_hz - low_cut_hz < MIN_BANDWIDTH_HZ: |
| raise AudioProcessingError( |
| "高域カットオフ周波数は、低域カットオフ周波数より100 Hz以上高く設定してください。" |
| ) |
|
|
|
|
| def _build_success_message(settings: ProcessingSettings) -> str: |
| sign = "+" if settings.pitch_semitones > 0 else "" |
| return ( |
| "変換が完了しました。\n" |
| f"声の高さ:{sign}{settings.pitch_semitones}半音\n" |
| f"通過帯域:{int(settings.low_cut_hz)}~{int(settings.high_cut_hz)} Hz" |
| ) |
|
|