diff --git a/codex_notifier/channels/voice.py b/codex_notifier/channels/voice.py index 78a6823..048c4ca 100644 --- a/codex_notifier/channels/voice.py +++ b/codex_notifier/channels/voice.py @@ -14,6 +14,7 @@ from ..state import text SPEECH_TIMEOUT_SECONDS = 30 +PLATFORM_NAME = os.name SPANISH_WORDS = {"al", "cambios", "completado", "con", "corregido", "de", "el", "en", "esta", "este", "fue", "la", "las", "listo", "los", "para", "por", "pruebas", "que", "se", "sin", "una", "y", "ya"} ENGLISH_WORDS = {"a", "and", "changes", "completed", "done", "for", "from", "has", "fixed", "in", "is", "of", "on", "that", "the", "tests", "this", "to", "was", "ready", "with", "without"} @@ -58,13 +59,13 @@ def _creation_flags() -> int: def speak_sapi_windows(message: str, *, language: str, preferred_voice: str = "", volume: float | None = None) -> None: - if os.name != "nt": + if PLATFORM_NAME != "nt": raise RuntimeError("La voz local solo está disponible en Windows.") environment = os.environ.copy() environment["CODEX_NOTIFIER_SPEECH_B64"] = base64.b64encode(message.encode("utf-8")).decode("ascii") environment["CODEX_NOTIFIER_SPEECH_LANGUAGE"] = language environment["CODEX_NOTIFIER_SPEECH_VOICE"] = preferred_voice - environment["CODEX_NOTIFIER_SPEECH_VOLUME"] = "" if volume is None else str(round(min(1.0, max(0.0, volume)) * 100)) + environment["CODEX_NOTIFIER_SPEECH_VOLUME"] = "" if volume is None else str(round(volume * 100)) script = ( "$bytes=[Convert]::FromBase64String($env:CODEX_NOTIFIER_SPEECH_B64);" "$text=[Text.Encoding]::UTF8.GetString($bytes);$voice=New-Object -ComObject SAPI.SpVoice;" @@ -84,9 +85,8 @@ def speak_sapi_windows(message: str, *, language: str, preferred_voice: str = "" timeout=SPEECH_TIMEOUT_SECONDS) -def play_wav_windows(wav_path: Path, *, powershell: str | None = None) -> None: - executable = powershell or shutil.which("powershell.exe") or "powershell.exe" - encoded_path = base64.b64encode(str(wav_path).encode("utf-8")).decode("ascii") +def _play_wav_powershell(windows_path: str, powershell: str) -> None: + encoded_path = base64.b64encode(windows_path.encode("utf-8")).decode("ascii") script = ( "$ErrorActionPreference='Stop';$bytes=[Convert]::FromBase64String($env:CODEX_NOTIFIER_WAV_B64);" "$path=[Text.Encoding]::UTF8.GetString($bytes);$player=New-Object System.Media.SoundPlayer $path;" @@ -94,23 +94,20 @@ def play_wav_windows(wav_path: Path, *, powershell: str | None = None) -> None: ) environment = os.environ.copy() environment["CODEX_NOTIFIER_WAV_B64"] = encoded_path - subprocess.run([executable, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script], + subprocess.run([powershell, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script], env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, creationflags=_creation_flags(), check=True, timeout=SPEECH_TIMEOUT_SECONDS) +def play_wav_windows(wav_path: Path, *, powershell: str | None = None) -> None: + _play_wav_powershell(str(wav_path), powershell or "powershell.exe") + + def _play_wav_wsl(wav_path: Path, powershell: str, wslpath: str) -> None: result = subprocess.run([wslpath, "-w", str(wav_path)], stdin=subprocess.DEVNULL, capture_output=True, text=True, check=True, timeout=5) - windows_path = result.stdout.strip() - encoded_path = base64.b64encode(windows_path.encode("utf-8")).decode("ascii") - script = ("$ErrorActionPreference='Stop';$bytes=[Convert]::FromBase64String('" + encoded_path + "');" - "$path=[Text.Encoding]::UTF8.GetString($bytes);$player=New-Object System.Media.SoundPlayer $path;" - "$player.Load();$player.PlaySync()") - subprocess.run([powershell, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script], - stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, - creationflags=_creation_flags(), check=True, timeout=SPEECH_TIMEOUT_SECONDS) + _play_wav_powershell(result.stdout.strip(), powershell) def generate_piper_wav(message: str, *, language: str = "es", voice_config: dict[str, Any], @@ -142,25 +139,23 @@ def generate_piper_wav(message: str, *, language: str = "es", voice_config: dict return wav_path -def speak_piper(message: str, *, language: str = "es", voice_config: dict[str, Any], - windows: bool | None = None) -> None: +def speak_piper(message: str, *, language: str = "es", voice_config: dict[str, Any]) -> None: with tempfile.TemporaryDirectory(prefix="codex-notifier-") as temporary_dir: wav_path = generate_piper_wav(message, language=language, voice_config=voice_config, temporary_dir=temporary_dir) - use_windows = os.name == "nt" if windows is None else windows - if use_windows: - play_wav_windows(wav_path) - return - play_wav_linux(wav_path) + play_wav(wav_path) -def play_wav_linux(wav_path: Path) -> None: - powershell = shutil.which("powershell.exe") - wslpath = shutil.which("wslpath") - if powershell and wslpath: - _play_wav_wsl(wav_path, powershell, wslpath) - return - player = next(((name, shutil.which(name)) for name in ("paplay", "pw-play", "aplay", "ffplay") if shutil.which(name)), None) +def find_executable(*names: str) -> tuple[str, str] | None: + for name in names: + executable = shutil.which(name) + if executable: + return name, executable + return None + + +def _play_wav_linux(wav_path: Path) -> None: + player = find_executable("paplay", "pw-play", "aplay", "ffplay") if not player: raise RuntimeError("No se encontró un reproductor compatible: instala paplay, pw-play, aplay o ffplay. En WSL también se admite powershell.exe.") name, executable = player @@ -170,19 +165,41 @@ def play_wav_linux(wav_path: Path) -> None: creationflags=_creation_flags()) -def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, Any] | None = None) -> None: - settings = {} if voice_config is None else voice_config - if text(settings.get("piper_executable")): - speak_piper(message, language=language, voice_config=settings, windows=False) +def _play_wav_posix(wav_path: Path) -> None: + powershell = shutil.which("powershell.exe") + wslpath = shutil.which("wslpath") + if powershell and wslpath: + _play_wav_wsl(wav_path, powershell, wslpath) + return + + _play_wav_linux(wav_path) + + +def play_wav(wav_path: Path) -> None: + if PLATFORM_NAME == "nt": + _play_wav_powershell(str(wav_path), "powershell.exe") return - dispatcher = shutil.which("spd-say") + if PLATFORM_NAME == "posix": + _play_wav_posix(wav_path) + return + raise RuntimeError(f"La reproducción de audio no es compatible con la plataforma {PLATFORM_NAME}.") + + +def play_wav_linux(wav_path: Path) -> None: + _play_wav_posix(wav_path) + + +def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, Any] | None = None) -> None: + dispatcher = find_executable("spd-say") if dispatcher: - command = [dispatcher, "-w", "-l", language, message] + _, executable = dispatcher + command = [executable, "-w", "-l", language, message] else: - espeak = shutil.which("espeak-ng") or shutil.which("espeak") + espeak = find_executable("espeak-ng", "espeak") if not espeak: raise RuntimeError("Instala speech-dispatcher (spd-say) o espeak-ng para usar voz en Linux.") - command = [espeak, "-v", language, message] + _, executable = espeak + command = [executable, "-v", language, message] subprocess.run(command, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=True, timeout=SPEECH_TIMEOUT_SECONDS, creationflags=_creation_flags()) @@ -191,17 +208,17 @@ def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, A def speak(message: str, *, language: str = "es", preferred_voice: str = "", voice_config: dict[str, Any] | None = None) -> None: settings = {} if voice_config is None else voice_config - if os.name == "nt": - if text(settings.get("piper_executable")): - speak_piper(message, language=language, voice_config=settings, windows=True) - else: - speak_sapi_windows(message, language=language, preferred_voice=preferred_voice, - volume=configured_volume(settings)) + if text(settings.get("piper_executable")): + speak_piper(message, language=language, voice_config=settings) + return + if PLATFORM_NAME == "nt": + speak_sapi_windows(message, language=language, preferred_voice=preferred_voice, + volume=configured_volume(settings)) return - if os.name == "posix": - speak_linux(message, language=language, voice_config=settings) + if PLATFORM_NAME == "posix": + speak_linux(message, language=language) return - raise RuntimeError(f"La voz local no es compatible con la plataforma {os.name}.") + raise RuntimeError(f"La voz local no es compatible con la plataforma {PLATFORM_NAME}.") speak_windows = speak_sapi_windows diff --git a/tests/test_hardening.py b/tests/test_hardening.py index 88e66be..a6e2ba0 100644 --- a/tests/test_hardening.py +++ b/tests/test_hardening.py @@ -85,7 +85,7 @@ def test_retry_after_is_used_for_transient_http_error(self, urlopen, sleep) -> N @patch("codex_notifier.channels.voice.speak_sapi_windows") @patch("codex_notifier.channels.voice.speak_piper") - @patch("codex_notifier.channels.voice.os.name", "nt") + @patch("codex_notifier.channels.voice.PLATFORM_NAME", "nt") def test_windows_voice_defaults_to_sapi_and_uses_piper_when_configured(self, piper, sapi) -> None: voice.speak("hello", language="en", voice_config={}) sapi.assert_called_once() @@ -97,7 +97,7 @@ def test_windows_voice_defaults_to_sapi_and_uses_piper_when_configured(self, pip @patch("codex_notifier.channels.voice.speak_sapi_windows") @patch("codex_notifier.channels.voice.speak_piper", side_effect=RuntimeError("bad Piper")) - @patch("codex_notifier.channels.voice.os.name", "nt") + @patch("codex_notifier.channels.voice.PLATFORM_NAME", "nt") def test_explicit_piper_failure_does_not_fallback_to_sapi(self, piper, sapi) -> None: with self.assertRaisesRegex(RuntimeError, "bad Piper"): voice.speak("hello", voice_config={"piper_executable": "piper"}) diff --git a/tests/test_notifier.py b/tests/test_notifier.py index b00ac23..cbfa3be 100644 --- a/tests/test_notifier.py +++ b/tests/test_notifier.py @@ -3,7 +3,6 @@ import base64 import json import os -import re import sys import tempfile import unittest @@ -329,7 +328,7 @@ def test_voice_message_explains_unknown_duration(self) -> None: self.assertIn("No se pudo determinar la duración", message) self.assertNotIn("Cambio listo", message) - @patch("notifier.os.name", "nt") + @patch("notifier._voice.PLATFORM_NAME", "nt") @patch("notifier.subprocess.run") def test_speech_waits_and_passes_language_and_preferred_voice(self, run) -> None: notifier.speak_windows( @@ -351,6 +350,32 @@ def test_speech_waits_and_passes_language_and_preferred_voice(self, run) -> None run.call_args.kwargs["timeout"], notifier.SPEECH_TIMEOUT_SECONDS ) + @patch("notifier.tempfile.TemporaryDirectory") + @patch("notifier.subprocess.run") + @patch("notifier.shutil.which") + @patch("notifier._voice.PLATFORM_NAME", "nt") + def test_windows_piper_uses_powershell_soundplayer( + self, which, run, temporary_directory + ) -> None: + model = Path(self.temp_dir.name) / "es_ES-test-medium.onnx" + model.touch() + temporary_directory.return_value.__enter__.return_value = self.temp_dir.name + which.side_effect = lambda name: "/opt/piper/bin/piper" if name == "piper" else None + + notifier.speak( + "Tarea terminada", + language="es", + voice_config={ + "piper_executable": "piper", + "spanish_voice": str(model), + }, + ) + + self.assertEqual(run.call_count, 2) + self.assertEqual(run.call_args_list[1].args[0][0], "powershell.exe") + self.assertIn("System.Media.SoundPlayer", run.call_args_list[1].args[0][-1]) + self.assertIn("$player.Load();$player.PlaySync()", run.call_args_list[1].args[0][-1]) + @patch("notifier.speak") @patch("notifier.send_teams_notification") def test_completion_uses_file_webhook_and_spanish_voice( @@ -652,6 +677,15 @@ def test_linux_speech_falls_back_to_espeak(self, which, run) -> None: ["/usr/bin/espeak-ng", "-v", "en", "Task complete"], ) + @patch("notifier._voice.speak_piper") + @patch("notifier._voice.PLATFORM_NAME", "posix") + def test_speak_selects_piper_before_linux_fallback(self, piper) -> None: + config = {"piper_executable": "piper"} + + notifier.speak("Task complete", language="en", voice_config=config) + + piper.assert_called_once_with("Task complete", language="en", voice_config=config) + @patch("notifier.tempfile.TemporaryDirectory") @patch("notifier.subprocess.run") @patch("notifier.shutil.which") @@ -667,14 +701,16 @@ def test_linux_piper_uses_language_model_and_pw_play( "pw-play": "/usr/bin/pw-play", }.get(name) - notifier.speak_linux( - "Tarea terminada", - language="es", - voice_config={ - "piper_executable": "piper", - "spanish_voice": str(model), - }, - ) + with patch("notifier._voice.play_wav") as play_wav: + notifier.speak_piper( + "Tarea terminada", + language="es", + voice_config={ + "piper_executable": "piper", + "spanish_voice": str(model), + }, + ) + play_wav.assert_called_once_with(wav) self.assertEqual( run.call_args_list[0].args[0], @@ -688,6 +724,8 @@ def test_linux_piper_uses_language_model_and_pw_play( "Tarea terminada", ], ) + with patch("notifier._voice.PLATFORM_NAME", "posix"): + notifier._voice.play_wav(wav) self.assertEqual( run.call_args_list[1].args[0], ["/usr/bin/pw-play", str(wav)], @@ -695,7 +733,7 @@ def test_linux_piper_uses_language_model_and_pw_play( def test_linux_piper_requires_the_voice_for_the_requested_language(self) -> None: with self.assertRaisesRegex(RuntimeError, "english_voice"): - notifier.speak_linux( + notifier.speak( "Task complete", language="en", voice_config={"piper_executable": "piper"}, @@ -715,15 +753,16 @@ def test_linux_piper_passes_configured_volume( "pw-play": "/usr/bin/pw-play", }.get(name) - notifier.speak_linux( - "Tarea terminada", - language="es", - voice_config={ - "piper_executable": "piper", - "spanish_voice": str(model), - "volume": 0.65, - }, - ) + with patch("notifier._voice.play_wav"): + notifier.speak_piper( + "Tarea terminada", + language="es", + voice_config={ + "piper_executable": "piper", + "spanish_voice": str(model), + "volume": 0.65, + }, + ) self.assertIn("--volume", run.call_args_list[0].args[0]) self.assertIn("0.65", run.call_args_list[0].args[0]) @@ -745,18 +784,23 @@ def test_linux_piper_uses_windows_audio_in_wsl( }.get(name) run.side_effect = [ SimpleNamespace(), - SimpleNamespace(stdout=r"C:\\wsl.localhost\Debian\tmp\speech.wav" + "\n"), + SimpleNamespace(stdout=r"C:\\wsl.localhost\Debian\tmp\speech 'demo'.wav" + "\n"), SimpleNamespace(), ] - notifier.speak_linux( - "Task complete", - language="en", - voice_config={ - "piper_executable": "piper", - "english_voice": str(model), - }, - ) + with patch("notifier._voice.play_wav") as play_wav: + notifier.speak_piper( + "Task complete", + language="en", + voice_config={ + "piper_executable": "piper", + "english_voice": str(model), + }, + ) + play_wav.assert_called_once_with(wav) + + with patch("notifier._voice.PLATFORM_NAME", "posix"): + notifier._voice.play_wav(wav) self.assertEqual( run.call_args_list[1].args[0], @@ -768,13 +812,21 @@ def test_linux_piper_uses_windows_audio_in_wsl( "/mnt/c/Windows/System32/powershell.exe", ) powershell_script = powershell_call.args[0][-1] + self.assertIn("FromBase64String($env:CODEX_NOTIFIER_WAV_B64)", powershell_script) self.assertEqual( base64.b64decode( - re.search(r"FromBase64String\('([^']+)'\)", powershell_script).group(1) + powershell_call.kwargs["env"]["CODEX_NOTIFIER_WAV_B64"] ).decode("utf-8"), - r"C:\\wsl.localhost\Debian\tmp\speech.wav", + r"C:\\wsl.localhost\Debian\tmp\speech 'demo'.wav", ) + @patch("notifier.shutil.which", return_value=None) + def test_linux_piper_reports_missing_wav_player(self, which) -> None: + wav_path = Path(self.temp_dir.name) / "speech.wav" + with patch("notifier._voice.PLATFORM_NAME", "posix"): + with self.assertRaisesRegex(RuntimeError, "reproductor compatible"): + notifier._voice.play_wav(wav_path) + if __name__ == "__main__": unittest.main()