"""PC-speaker playback for the simulator. Lets the mock robot actually *say* the text out of this laptop's speakers, so the whole demo can be rehearsed - wording, pacing, the Stop button - before the A3 is on the network. This is simulation only. It has nothing to do with the real robot: the A3 synthesises its own speech on-board and no audio ever leaves the PC (see docs/AGIBOT_A3_INTEGRATION.md). Backends, in order of preference: Windows SAPI5 through comtypes - in-process, ~20 ms to start, real interrupt Windows PowerShell System.Speech - fallback, no packages needed at all macOS `say` Linux `espeak-ng` / `espeak` / `spd-say` Every backend supports three things the simulator needs: start without blocking, report whether audio is still playing, and **stop immediately**. Without that last one the Stop button would lie - the UI would say "stopped" while the laptop kept talking. """ from __future__ import annotations import logging import platform import queue import shutil import subprocess import threading from typing import List, Optional logger = logging.getLogger(__name__) # SAPI ISpVoice::Speak flags _SVSF_ASYNC = 1 _SVSF_PURGE_BEFORE_SPEAK = 2 _SVSF_IS_XML = 8 def _xml_escape(text: str) -> str: """Escape text going into SAPI/SSML markup. Without this, an operator typing `Tom & Jerry` or `a < b` would produce malformed markup and the utterance would be mangled or dropped. """ return ( text.replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) ) class LocalVoice: """A speech engine available to the simulator.""" name = "none" def prepare(self, text: str) -> bool: """Optional: do slow work (e.g. a cloud round trip) before playback. Called during the "Processing" stage so that "Speaking" is only announced once audio can really start. Local engines need nothing. """ return True def start(self, text: str) -> None: raise NotImplementedError def stop(self) -> None: raise NotImplementedError def is_speaking(self) -> bool: raise NotImplementedError def close(self) -> None: pass # --------------------------------------------------------------------------- # # Windows - SAPI5 via COM # --------------------------------------------------------------------------- # class SapiVoice(LocalVoice): """Windows SAPI5. COM objects are apartment-bound, so one dedicated thread creates the voice and is the only thread that ever touches it. Public methods just hand it work through a queue and read plain flags, which makes them safe to call from the asyncio loop. """ name = "sapi" def __init__(self, voice_hint: Optional[str] = None, rate: int = 0, volume: int = 100, pitch: int = 0) -> None: self._voice_hint = voice_hint self._rate = max(-10, min(10, rate)) self._volume = max(0, min(100, volume)) # SAPI pitch: -10..10. Raising it turns an adult male voice into a # younger-sounding one, which is the closest a stock Windows voice gets # to a teenage character voice. self._pitch = max(-10, min(10, pitch)) self._commands: "queue.Queue" = queue.Queue() self._speaking = threading.Event() self._stop_flag = threading.Event() self._ready = threading.Event() self._error: Optional[str] = None self._closing = False self._thread = threading.Thread(target=self._run, name="sapi-voice", daemon=True) self._thread.start() self._ready.wait(timeout=8) if self._error: raise RuntimeError(self._error) # -- public API ---------------------------------------------------------- # def start(self, text: str) -> None: self._stop_flag.clear() # Set the flag here, not on the worker thread: a caller that polls # is_speaking() immediately must never see "idle" before we begin. self._speaking.set() self._commands.put(text) def stop(self) -> None: self._stop_flag.set() def is_speaking(self) -> bool: return self._speaking.is_set() def close(self) -> None: self._closing = True self._stop_flag.set() self._commands.put(None) # -- worker thread ------------------------------------------------------- # def _run(self) -> None: try: import comtypes import comtypes.client except ImportError as exc: # pragma: no cover - checked by the factory self._error = "comtypes not installed: {0}".format(exc) self._ready.set() return try: comtypes.CoInitialize() except Exception: # pragma: no cover - already initialised is fine pass try: voice = comtypes.client.CreateObject("SAPI.SpVoice") self._select_voice(voice) voice.Rate = self._rate voice.Volume = self._volume except Exception as exc: self._error = "cannot create SAPI voice: {0}".format(exc) self._ready.set() return self._ready.set() logger.info("local audio: Windows SAPI ready (rate=%s volume=%s)", self._rate, self._volume) while not self._closing: text = self._commands.get() if text is None: break try: if self._pitch: voice.Speak( '{1}'.format(self._pitch, _xml_escape(text)), _SVSF_ASYNC | _SVSF_IS_XML, ) else: voice.Speak(text, _SVSF_ASYNC) # Poll rather than block, so a Stop lands within ~50 ms. while not voice.WaitUntilDone(50): if self._stop_flag.is_set(): voice.Speak("", _SVSF_PURGE_BEFORE_SPEAK) break except Exception as exc: # pragma: no cover - device may vanish logger.warning("local audio playback failed: %s", exc) finally: self._speaking.clear() try: import comtypes comtypes.CoUninitialize() except Exception: # pragma: no cover pass def _select_voice(self, voice) -> None: """Pick the voice whose description matches the configured hint.""" if not self._voice_hint: return hint = self._voice_hint.lower() try: for description, token, source in _sapi_tokens(): if hint in description.lower(): voice.Voice = token logger.info("local audio: using voice '%s' (%s)", description, source) return logger.warning( "local audio: no voice matching '%s'; using the system default. " "Run 'python scripts/voices.py' to see what is installed.", self._voice_hint, ) except Exception as exc: # pragma: no cover logger.debug("voice selection failed: %s", exc) #: Windows keeps two voice registries. The "OneCore" set is the newer, better #: sounding one and is invisible to SpVoice.GetVoices(), so look there first. _ONECORE_KEY = r"HKEY_LOCAL_MACHINE\SOFTWARE\Microsoft\Speech_OneCore\Voices" def _sapi_tokens(): """[(description, token, source)] for every SAPI voice, best set first. Must be called on a thread where COM is initialised. """ import comtypes.client found = [] seen = set() try: category = comtypes.client.CreateObject("SAPI.SpObjectTokenCategory") category.SetId(_ONECORE_KEY, False) tokens = category.EnumerateTokens() for index in range(tokens.Count): token = tokens.Item(index) description = token.GetDescription() found.append((description, token, "onecore")) seen.add(description) except Exception as exc: # pragma: no cover - older Windows logger.debug("OneCore voices unavailable: %s", exc) try: voice = comtypes.client.CreateObject("SAPI.SpVoice") tokens = voice.GetVoices() for index in range(tokens.Count): token = tokens.Item(index) description = token.GetDescription() if description not in seen: found.append((description, token, "classic")) except Exception as exc: # pragma: no cover logger.debug("classic voices unavailable: %s", exc) return found def list_local_voices(): """[(description, source)] - for `python scripts/voices.py`. Never raises.""" if not platform.system().lower().startswith("win"): return [] try: import comtypes try: comtypes.CoInitialize() except Exception: # pragma: no cover pass return [(description, source) for description, _, source in _sapi_tokens()] except Exception: # pragma: no cover return [] # --------------------------------------------------------------------------- # # Everything else - an external command we can kill # --------------------------------------------------------------------------- # class CommandVoice(LocalVoice): """Speak by running a command. Stop = kill the process.""" def __init__(self, name: str, argv_builder) -> None: self.name = name self._argv = argv_builder self._process: Optional[subprocess.Popen] = None self._lock = threading.Lock() def start(self, text: str) -> None: self.stop() argv = self._argv(text) try: with self._lock: self._process = subprocess.Popen( argv, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, stdin=subprocess.DEVNULL, ) except Exception as exc: # pragma: no cover - depends on host logger.warning("local audio command failed (%s): %s", argv[0], exc) self._process = None def stop(self) -> None: with self._lock: process = self._process self._process = None if process is not None and process.poll() is None: try: process.kill() except Exception: # pragma: no cover pass def is_speaking(self) -> bool: with self._lock: process = self._process return process is not None and process.poll() is None def close(self) -> None: self.stop() def _powershell_voice(rate: int, volume: int, pitch: int = 0, voice_hint: Optional[str] = None) -> CommandVoice: """Windows fallback using System.Speech - present on every Windows install.""" rate = max(-10, min(10, rate)) volume = max(0, min(100, volume)) pitch = max(-10, min(10, pitch)) def argv(text: str) -> List[str]: select = "" if voice_hint: # SelectVoiceByHints has no substring form; fall back silently when # the named voice is absent rather than throwing. select = ( "try {{ $s.SelectVoice(($s.GetInstalledVoices() | " "Where-Object {{ $_.VoiceInfo.Name -like '*{0}*' }} | " "Select-Object -First 1).VoiceInfo.Name) }} catch {{}};" ).format(voice_hint.replace("'", "''")) if pitch: # System.Speech exposes pitch only through SSML prosody. body = ( "{1}" ).format(pitch * 5, _xml_escape(text)) speak = "$s.SpeakSsml('{0}')".format(body.replace("'", "''")) else: speak = "$s.Speak('{0}')".format(text.replace("'", "''")) script = ( "Add-Type -AssemblyName System.Speech;" "$s = New-Object System.Speech.Synthesis.SpeechSynthesizer;" "{select}$s.Rate = {rate}; $s.Volume = {volume};{speak}" ).format(select=select, rate=rate, volume=volume, speak=speak) return ["powershell", "-NoProfile", "-NonInteractive", "-Command", script] return CommandVoice("powershell", argv) def _macos_voice(rate: int) -> CommandVoice: # `say` takes words per minute; map the -10..10 scale onto a sane range. wpm = max(90, min(320, 180 + rate * 12)) def argv(text: str) -> List[str]: return ["say", "-r", str(wpm), text] return CommandVoice("say", argv) def _linux_voice(rate: int, volume: int) -> Optional[CommandVoice]: if shutil.which("espeak-ng") or shutil.which("espeak"): binary = "espeak-ng" if shutil.which("espeak-ng") else "espeak" wpm = max(80, min(320, 175 + rate * 12)) amplitude = max(0, min(200, int(volume * 2))) def argv(text: str) -> List[str]: return [binary, "-s", str(wpm), "-a", str(amplitude), text] return CommandVoice(binary, argv) if shutil.which("spd-say"): def argv(text: str) -> List[str]: return ["spd-say", "-w", "-r", str(max(-100, min(100, rate * 10))), text] return CommandVoice("spd-say", argv) return None # --------------------------------------------------------------------------- # # factory # --------------------------------------------------------------------------- # def create_local_voice( voice_hint: Optional[str] = None, rate: int = 0, volume: int = 100, pitch: int = 0, pronouncer=None, ) -> Optional[LocalVoice]: """Best available PC voice, or None if this machine cannot speak. Never raises: local audio is a convenience, and losing it must not stop the simulator from running. """ voice = _create(voice_hint, rate, volume, pitch) if voice is not None and pronouncer is not None: # OS voices mispronounce brand names; fix the text at the last moment. return Respelling(voice, pronouncer) return voice class Respelling(LocalVoice): """Wraps a voice, respelling text on its way to the synthesiser.""" def __init__(self, inner: LocalVoice, pronouncer) -> None: self._inner = inner self._pronouncer = pronouncer self.name = inner.name def prepare(self, text: str) -> bool: return self._inner.prepare(self._pronouncer.apply(text)) def start(self, text: str) -> None: self._inner.start(self._pronouncer.apply(text)) def stop(self) -> None: self._inner.stop() def is_speaking(self) -> bool: return self._inner.is_speaking() def close(self) -> None: self._inner.close() def _create(voice_hint, rate, volume, pitch): system = platform.system().lower() if system.startswith("win"): try: return SapiVoice(voice_hint, rate, volume, pitch) except Exception as exc: logger.info("SAPI unavailable (%s); falling back to PowerShell", exc) if shutil.which("powershell"): return _powershell_voice(rate, volume, pitch, voice_hint) logger.warning("local audio: no speech engine found on this Windows PC") return None if system == "darwin": if shutil.which("say"): return _macos_voice(rate) logger.warning("local audio: 'say' not found") return None voice = _linux_voice(rate, volume) if voice is None: logger.warning( "local audio: install espeak-ng (apt install espeak-ng) to hear the simulator" ) return voice