Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
107 changes: 62 additions & 45 deletions codex_notifier/channels/voice.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from ..state import text

SPEECH_TIMEOUT_SECONDS = 30
PLATFORM_NAME = os.name
SPANISH_WORDS = {"al", "cambios", "completado", "con", "corregido", "de", "el", "en", "esta", "este", "fue", "la", "las", "listo", "los", "para", "por", "pruebas", "que", "se", "sin", "una", "y", "ya"}
ENGLISH_WORDS = {"a", "and", "changes", "completed", "done", "for", "from", "has", "fixed", "in", "is", "of", "on", "that", "the", "tests", "this", "to", "was", "ready", "with", "without"}

Expand Down Expand Up @@ -58,13 +59,13 @@ def _creation_flags() -> int:

def speak_sapi_windows(message: str, *, language: str, preferred_voice: str = "",
volume: float | None = None) -> None:
if os.name != "nt":
if PLATFORM_NAME != "nt":
raise RuntimeError("La voz local solo está disponible en Windows.")
environment = os.environ.copy()
environment["CODEX_NOTIFIER_SPEECH_B64"] = base64.b64encode(message.encode("utf-8")).decode("ascii")
environment["CODEX_NOTIFIER_SPEECH_LANGUAGE"] = language
environment["CODEX_NOTIFIER_SPEECH_VOICE"] = preferred_voice
environment["CODEX_NOTIFIER_SPEECH_VOLUME"] = "" if volume is None else str(round(min(1.0, max(0.0, volume)) * 100))
environment["CODEX_NOTIFIER_SPEECH_VOLUME"] = "" if volume is None else str(round(volume * 100))
script = (
"$bytes=[Convert]::FromBase64String($env:CODEX_NOTIFIER_SPEECH_B64);"
"$text=[Text.Encoding]::UTF8.GetString($bytes);$voice=New-Object -ComObject SAPI.SpVoice;"
Expand All @@ -84,33 +85,29 @@ def speak_sapi_windows(message: str, *, language: str, preferred_voice: str = ""
timeout=SPEECH_TIMEOUT_SECONDS)


def play_wav_windows(wav_path: Path, *, powershell: str | None = None) -> None:
executable = powershell or shutil.which("powershell.exe") or "powershell.exe"
encoded_path = base64.b64encode(str(wav_path).encode("utf-8")).decode("ascii")
def _play_wav_powershell(windows_path: str, powershell: str) -> None:
encoded_path = base64.b64encode(windows_path.encode("utf-8")).decode("ascii")
script = (
"$ErrorActionPreference='Stop';$bytes=[Convert]::FromBase64String($env:CODEX_NOTIFIER_WAV_B64);"
"$path=[Text.Encoding]::UTF8.GetString($bytes);$player=New-Object System.Media.SoundPlayer $path;"
"$player.Load();$player.PlaySync()"
)
environment = os.environ.copy()
environment["CODEX_NOTIFIER_WAV_B64"] = encoded_path
subprocess.run([executable, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script],
subprocess.run([powershell, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script],
env=environment, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL, creationflags=_creation_flags(), check=True,
timeout=SPEECH_TIMEOUT_SECONDS)


def play_wav_windows(wav_path: Path, *, powershell: str | None = None) -> None:
_play_wav_powershell(str(wav_path), powershell or "powershell.exe")


def _play_wav_wsl(wav_path: Path, powershell: str, wslpath: str) -> None:
result = subprocess.run([wslpath, "-w", str(wav_path)], stdin=subprocess.DEVNULL,
capture_output=True, text=True, check=True, timeout=5)
windows_path = result.stdout.strip()
encoded_path = base64.b64encode(windows_path.encode("utf-8")).decode("ascii")
script = ("$ErrorActionPreference='Stop';$bytes=[Convert]::FromBase64String('" + encoded_path + "');"
"$path=[Text.Encoding]::UTF8.GetString($bytes);$player=New-Object System.Media.SoundPlayer $path;"
"$player.Load();$player.PlaySync()")
subprocess.run([powershell, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", script],
stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
creationflags=_creation_flags(), check=True, timeout=SPEECH_TIMEOUT_SECONDS)
_play_wav_powershell(result.stdout.strip(), powershell)


def generate_piper_wav(message: str, *, language: str = "es", voice_config: dict[str, Any],
Expand Down Expand Up @@ -142,25 +139,23 @@ def generate_piper_wav(message: str, *, language: str = "es", voice_config: dict
return wav_path


def speak_piper(message: str, *, language: str = "es", voice_config: dict[str, Any],
windows: bool | None = None) -> None:
def speak_piper(message: str, *, language: str = "es", voice_config: dict[str, Any]) -> None:
with tempfile.TemporaryDirectory(prefix="codex-notifier-") as temporary_dir:
wav_path = generate_piper_wav(message, language=language, voice_config=voice_config,
temporary_dir=temporary_dir)
use_windows = os.name == "nt" if windows is None else windows
if use_windows:
play_wav_windows(wav_path)
return
play_wav_linux(wav_path)
play_wav(wav_path)


def play_wav_linux(wav_path: Path) -> None:
powershell = shutil.which("powershell.exe")
wslpath = shutil.which("wslpath")
if powershell and wslpath:
_play_wav_wsl(wav_path, powershell, wslpath)
return
player = next(((name, shutil.which(name)) for name in ("paplay", "pw-play", "aplay", "ffplay") if shutil.which(name)), None)
def find_executable(*names: str) -> tuple[str, str] | None:
for name in names:
executable = shutil.which(name)
if executable:
return name, executable
return None


def _play_wav_linux(wav_path: Path) -> None:
player = find_executable("paplay", "pw-play", "aplay", "ffplay")
if not player:
raise RuntimeError("No se encontró un reproductor compatible: instala paplay, pw-play, aplay o ffplay. En WSL también se admite powershell.exe.")
name, executable = player
Expand All @@ -170,19 +165,41 @@ def play_wav_linux(wav_path: Path) -> None:
creationflags=_creation_flags())


def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, Any] | None = None) -> None:
settings = {} if voice_config is None else voice_config
if text(settings.get("piper_executable")):
speak_piper(message, language=language, voice_config=settings, windows=False)
def _play_wav_posix(wav_path: Path) -> None:
powershell = shutil.which("powershell.exe")
wslpath = shutil.which("wslpath")
if powershell and wslpath:
_play_wav_wsl(wav_path, powershell, wslpath)
return

_play_wav_linux(wav_path)


def play_wav(wav_path: Path) -> None:
if PLATFORM_NAME == "nt":
_play_wav_powershell(str(wav_path), "powershell.exe")
return
dispatcher = shutil.which("spd-say")
if PLATFORM_NAME == "posix":
_play_wav_posix(wav_path)
return
raise RuntimeError(f"La reproducción de audio no es compatible con la plataforma {PLATFORM_NAME}.")


def play_wav_linux(wav_path: Path) -> None:
_play_wav_posix(wav_path)


def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, Any] | None = None) -> None:
dispatcher = find_executable("spd-say")
if dispatcher:
command = [dispatcher, "-w", "-l", language, message]
_, executable = dispatcher
command = [executable, "-w", "-l", language, message]
else:
espeak = shutil.which("espeak-ng") or shutil.which("espeak")
espeak = find_executable("espeak-ng", "espeak")
if not espeak:
raise RuntimeError("Instala speech-dispatcher (spd-say) o espeak-ng para usar voz en Linux.")
command = [espeak, "-v", language, message]
_, executable = espeak
command = [executable, "-v", language, message]
subprocess.run(command, stdin=subprocess.DEVNULL, stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL, check=True, timeout=SPEECH_TIMEOUT_SECONDS,
creationflags=_creation_flags())
Expand All @@ -191,17 +208,17 @@ def speak_linux(message: str, *, language: str = "es", voice_config: dict[str, A
def speak(message: str, *, language: str = "es", preferred_voice: str = "",
voice_config: dict[str, Any] | None = None) -> None:
settings = {} if voice_config is None else voice_config
if os.name == "nt":
if text(settings.get("piper_executable")):
speak_piper(message, language=language, voice_config=settings, windows=True)
else:
speak_sapi_windows(message, language=language, preferred_voice=preferred_voice,
volume=configured_volume(settings))
if text(settings.get("piper_executable")):
speak_piper(message, language=language, voice_config=settings)
return
if PLATFORM_NAME == "nt":
speak_sapi_windows(message, language=language, preferred_voice=preferred_voice,
volume=configured_volume(settings))
return
if os.name == "posix":
speak_linux(message, language=language, voice_config=settings)
if PLATFORM_NAME == "posix":
speak_linux(message, language=language)
return
raise RuntimeError(f"La voz local no es compatible con la plataforma {os.name}.")
raise RuntimeError(f"La voz local no es compatible con la plataforma {PLATFORM_NAME}.")


speak_windows = speak_sapi_windows
4 changes: 2 additions & 2 deletions tests/test_hardening.py
Original file line number Diff line number Diff line change
Expand Up @@ -85,7 +85,7 @@ def test_retry_after_is_used_for_transient_http_error(self, urlopen, sleep) -> N

@patch("codex_notifier.channels.voice.speak_sapi_windows")
@patch("codex_notifier.channels.voice.speak_piper")
@patch("codex_notifier.channels.voice.os.name", "nt")
@patch("codex_notifier.channels.voice.PLATFORM_NAME", "nt")
def test_windows_voice_defaults_to_sapi_and_uses_piper_when_configured(self, piper, sapi) -> None:
voice.speak("hello", language="en", voice_config={})
sapi.assert_called_once()
Expand All @@ -97,7 +97,7 @@ def test_windows_voice_defaults_to_sapi_and_uses_piper_when_configured(self, pip

@patch("codex_notifier.channels.voice.speak_sapi_windows")
@patch("codex_notifier.channels.voice.speak_piper", side_effect=RuntimeError("bad Piper"))
@patch("codex_notifier.channels.voice.os.name", "nt")
@patch("codex_notifier.channels.voice.PLATFORM_NAME", "nt")
def test_explicit_piper_failure_does_not_fallback_to_sapi(self, piper, sapi) -> None:
with self.assertRaisesRegex(RuntimeError, "bad Piper"):
voice.speak("hello", voice_config={"piper_executable": "piper"})
Expand Down
114 changes: 83 additions & 31 deletions tests/test_notifier.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,6 @@
import base64
import json
import os
import re
import sys
import tempfile
import unittest
Expand Down Expand Up @@ -329,7 +328,7 @@ def test_voice_message_explains_unknown_duration(self) -> None:
self.assertIn("No se pudo determinar la duración", message)
self.assertNotIn("Cambio listo", message)

@patch("notifier.os.name", "nt")
@patch("notifier._voice.PLATFORM_NAME", "nt")
@patch("notifier.subprocess.run")
def test_speech_waits_and_passes_language_and_preferred_voice(self, run) -> None:
notifier.speak_windows(
Expand All @@ -351,6 +350,32 @@ def test_speech_waits_and_passes_language_and_preferred_voice(self, run) -> None
run.call_args.kwargs["timeout"], notifier.SPEECH_TIMEOUT_SECONDS
)

@patch("notifier.tempfile.TemporaryDirectory")
@patch("notifier.subprocess.run")
@patch("notifier.shutil.which")
@patch("notifier._voice.PLATFORM_NAME", "nt")
def test_windows_piper_uses_powershell_soundplayer(
self, which, run, temporary_directory
) -> None:
model = Path(self.temp_dir.name) / "es_ES-test-medium.onnx"
model.touch()
temporary_directory.return_value.__enter__.return_value = self.temp_dir.name
which.side_effect = lambda name: "/opt/piper/bin/piper" if name == "piper" else None

notifier.speak(
"Tarea terminada",
language="es",
voice_config={
"piper_executable": "piper",
"spanish_voice": str(model),
},
)

self.assertEqual(run.call_count, 2)
self.assertEqual(run.call_args_list[1].args[0][0], "powershell.exe")
self.assertIn("System.Media.SoundPlayer", run.call_args_list[1].args[0][-1])
self.assertIn("$player.Load();$player.PlaySync()", run.call_args_list[1].args[0][-1])

@patch("notifier.speak")
@patch("notifier.send_teams_notification")
def test_completion_uses_file_webhook_and_spanish_voice(
Expand Down Expand Up @@ -652,6 +677,15 @@ def test_linux_speech_falls_back_to_espeak(self, which, run) -> None:
["/usr/bin/espeak-ng", "-v", "en", "Task complete"],
)

@patch("notifier._voice.speak_piper")
@patch("notifier._voice.PLATFORM_NAME", "posix")
def test_speak_selects_piper_before_linux_fallback(self, piper) -> None:
config = {"piper_executable": "piper"}

notifier.speak("Task complete", language="en", voice_config=config)

piper.assert_called_once_with("Task complete", language="en", voice_config=config)

@patch("notifier.tempfile.TemporaryDirectory")
@patch("notifier.subprocess.run")
@patch("notifier.shutil.which")
Expand All @@ -667,14 +701,16 @@ def test_linux_piper_uses_language_model_and_pw_play(
"pw-play": "/usr/bin/pw-play",
}.get(name)

notifier.speak_linux(
"Tarea terminada",
language="es",
voice_config={
"piper_executable": "piper",
"spanish_voice": str(model),
},
)
with patch("notifier._voice.play_wav") as play_wav:
notifier.speak_piper(
"Tarea terminada",
language="es",
voice_config={
"piper_executable": "piper",
"spanish_voice": str(model),
},
)
play_wav.assert_called_once_with(wav)

self.assertEqual(
run.call_args_list[0].args[0],
Expand All @@ -688,14 +724,16 @@ def test_linux_piper_uses_language_model_and_pw_play(
"Tarea terminada",
],
)
with patch("notifier._voice.PLATFORM_NAME", "posix"):
notifier._voice.play_wav(wav)
self.assertEqual(
run.call_args_list[1].args[0],
["/usr/bin/pw-play", str(wav)],
)

def test_linux_piper_requires_the_voice_for_the_requested_language(self) -> None:
with self.assertRaisesRegex(RuntimeError, "english_voice"):
notifier.speak_linux(
notifier.speak(
"Task complete",
language="en",
voice_config={"piper_executable": "piper"},
Expand All @@ -715,15 +753,16 @@ def test_linux_piper_passes_configured_volume(
"pw-play": "/usr/bin/pw-play",
}.get(name)

notifier.speak_linux(
"Tarea terminada",
language="es",
voice_config={
"piper_executable": "piper",
"spanish_voice": str(model),
"volume": 0.65,
},
)
with patch("notifier._voice.play_wav"):
notifier.speak_piper(
"Tarea terminada",
language="es",
voice_config={
"piper_executable": "piper",
"spanish_voice": str(model),
"volume": 0.65,
},
)

self.assertIn("--volume", run.call_args_list[0].args[0])
self.assertIn("0.65", run.call_args_list[0].args[0])
Expand All @@ -745,18 +784,23 @@ def test_linux_piper_uses_windows_audio_in_wsl(
}.get(name)
run.side_effect = [
SimpleNamespace(),
SimpleNamespace(stdout=r"C:\\wsl.localhost\Debian\tmp\speech.wav" + "\n"),
SimpleNamespace(stdout=r"C:\\wsl.localhost\Debian\tmp\speech 'demo'.wav" + "\n"),
SimpleNamespace(),
]

notifier.speak_linux(
"Task complete",
language="en",
voice_config={
"piper_executable": "piper",
"english_voice": str(model),
},
)
with patch("notifier._voice.play_wav") as play_wav:
notifier.speak_piper(
"Task complete",
language="en",
voice_config={
"piper_executable": "piper",
"english_voice": str(model),
},
)
play_wav.assert_called_once_with(wav)

with patch("notifier._voice.PLATFORM_NAME", "posix"):
notifier._voice.play_wav(wav)

self.assertEqual(
run.call_args_list[1].args[0],
Expand All @@ -768,13 +812,21 @@ def test_linux_piper_uses_windows_audio_in_wsl(
"/mnt/c/Windows/System32/powershell.exe",
)
powershell_script = powershell_call.args[0][-1]
self.assertIn("FromBase64String($env:CODEX_NOTIFIER_WAV_B64)", powershell_script)
self.assertEqual(
base64.b64decode(
re.search(r"FromBase64String\('([^']+)'\)", powershell_script).group(1)
powershell_call.kwargs["env"]["CODEX_NOTIFIER_WAV_B64"]
).decode("utf-8"),
r"C:\\wsl.localhost\Debian\tmp\speech.wav",
r"C:\\wsl.localhost\Debian\tmp\speech 'demo'.wav",
)

@patch("notifier.shutil.which", return_value=None)
def test_linux_piper_reports_missing_wav_player(self, which) -> None:
wav_path = Path(self.temp_dir.name) / "speech.wav"
with patch("notifier._voice.PLATFORM_NAME", "posix"):
with self.assertRaisesRegex(RuntimeError, "reproductor compatible"):
notifier._voice.play_wav(wav_path)


if __name__ == "__main__":
unittest.main()
Loading