From 6e939ce6c3f40c73cae1d1b418feb2fdac3bf1ae Mon Sep 17 00:00:00 2001 From: Sidra Date: Thu, 3 Sep 2026 00:10:18 +0400 Subject: [PATCH] Update 2026-09-03 00:10 --- .env.example | 276 ++++++++++ .gitignore | 13 + README.md | 392 ++++++++++++++ backend/__init__.py | 0 backend/api/__init__.py | 0 backend/api/routes.py | 172 ++++++ backend/api/schemas.py | 32 ++ backend/api/websocket.py | 114 ++++ backend/config/__init__.py | 0 backend/config/settings.py | 540 +++++++++++++++++++ backend/core/__init__.py | 0 backend/core/events.py | 125 +++++ backend/core/logging.py | 43 ++ backend/core/pronunciation.py | 112 ++++ backend/core/text.py | 39 ++ backend/main.py | 175 ++++++ backend/robot/__init__.py | 0 backend/robot/agibot_a3.py | 299 +++++++++++ backend/robot/base.py | 242 +++++++++ backend/robot/factory.py | 46 ++ backend/robot/gemini_voice.py | 433 +++++++++++++++ backend/robot/local_audio.py | 453 ++++++++++++++++ backend/robot/manager.py | 197 +++++++ backend/robot/mock_robot.py | 365 +++++++++++++ backend/robot/transports/__init__.py | 67 +++ backend/robot/transports/aimdk_transport.py | 354 ++++++++++++ backend/robot/transports/base.py | 136 +++++ backend/robot/transports/http_transport.py | 232 ++++++++ backend/robot/transports/ros2_transport.py | 236 ++++++++ backend/robot/transports/ssh_transport.py | 192 +++++++ backend/robot/transports/ws_transport.py | 226 ++++++++ backend/services/__init__.py | 0 backend/services/audio_library.py | 257 +++++++++ backend/services/history.py | 63 +++ backend/services/speech_service.py | 319 +++++++++++ docs/AGIBOT_A3_INTEGRATION.md | 382 +++++++++++++ docs/NETWORK.md | 196 +++++++ frontend/index.html | 168 ++++++ frontend/js/api.js | 69 +++ frontend/js/app.js | 373 +++++++++++++ frontend/js/socket.js | 111 ++++ frontend/js/ui.js | 367 +++++++++++++ frontend/styles/main.css | 565 ++++++++++++++++++++ pronunciation.json | 6 + requirements.txt | 28 + scripts/discover_robot.py | 303 +++++++++++ scripts/fake_a3_server.py | 194 +++++++ scripts/selftest.py | 267 +++++++++ scripts/voices.py | 130 +++++ scripts/warm_voice.py | 125 +++++ start.bat | 63 +++ start.sh | 32 ++ 52 files changed, 9529 insertions(+) create mode 100644 .env.example create mode 100644 .gitignore create mode 100644 README.md create mode 100644 backend/__init__.py create mode 100644 backend/api/__init__.py create mode 100644 backend/api/routes.py create mode 100644 backend/api/schemas.py create mode 100644 backend/api/websocket.py create mode 100644 backend/config/__init__.py create mode 100644 backend/config/settings.py create mode 100644 backend/core/__init__.py create mode 100644 backend/core/events.py create mode 100644 backend/core/logging.py create mode 100644 backend/core/pronunciation.py create mode 100644 backend/core/text.py create mode 100644 backend/main.py create mode 100644 backend/robot/__init__.py create mode 100644 backend/robot/agibot_a3.py create mode 100644 backend/robot/base.py create mode 100644 backend/robot/factory.py create mode 100644 backend/robot/gemini_voice.py create mode 100644 backend/robot/local_audio.py create mode 100644 backend/robot/manager.py create mode 100644 backend/robot/mock_robot.py create mode 100644 backend/robot/transports/__init__.py create mode 100644 backend/robot/transports/aimdk_transport.py create mode 100644 backend/robot/transports/base.py create mode 100644 backend/robot/transports/http_transport.py create mode 100644 backend/robot/transports/ros2_transport.py create mode 100644 backend/robot/transports/ssh_transport.py create mode 100644 backend/robot/transports/ws_transport.py create mode 100644 backend/services/__init__.py create mode 100644 backend/services/audio_library.py create mode 100644 backend/services/history.py create mode 100644 backend/services/speech_service.py create mode 100644 docs/AGIBOT_A3_INTEGRATION.md create mode 100644 docs/NETWORK.md create mode 100644 frontend/index.html create mode 100644 frontend/js/api.js create mode 100644 frontend/js/app.js create mode 100644 frontend/js/socket.js create mode 100644 frontend/js/ui.js create mode 100644 frontend/styles/main.css create mode 100644 pronunciation.json create mode 100644 requirements.txt create mode 100644 scripts/discover_robot.py create mode 100644 scripts/fake_a3_server.py create mode 100644 scripts/selftest.py create mode 100644 scripts/voices.py create mode 100644 scripts/warm_voice.py create mode 100644 start.bat create mode 100644 start.sh diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..a857d1e --- /dev/null +++ b/.env.example @@ -0,0 +1,276 @@ +# ============================================================================= +# AGIBOT A3 - Voice Control | configuration +# ============================================================================= +# Copy this file to `.env` and edit that copy. Nothing in the source code needs +# to change to switch between the simulator and the real robot. +# +# Restart the server after editing, or press "Reload config" in the dashboard. +# ============================================================================= + + +# ----------------------------------------------------------------------------- +# 1. WEB SERVER (this PC) +# ----------------------------------------------------------------------------- +# 127.0.0.1 = only this PC can open the dashboard (recommended). +# 0.0.0.0 = other devices on your LAN can open it too (tablet, phone). +HOST=127.0.0.1 +PORT=8000 +LOG_LEVEL=info + +# Only needed if you serve the frontend from a different origin. Usually empty. +CORS_ORIGINS= + + +# ----------------------------------------------------------------------------- +# 2. ROBOT MODE <<< THIS IS THE SWITCH >>> +# ----------------------------------------------------------------------------- +# mock = simulated robot, no hardware required (default; use this today) +# real = talk to a physical AGIBOT A3 over the network +ROBOT_MODE=mock + +ROBOT_NAME=AGIBOT A3 +ROBOT_MODEL=AgiBot A3 + + +# ----------------------------------------------------------------------------- +# 3. ROBOT ADDRESS <<< PUT THE ROBOT IP HERE WHEN YOU HAVE IT >>> +# ----------------------------------------------------------------------------- +# Leave ROBOT_IP empty while ROBOT_MODE=mock. +# When the robot is powered on and on the same LAN: +# ROBOT_MODE=real +# ROBOT_IP=192.168.1.50 <- the robot's actual IP on YOUR network +# ROBOT_PORT=59301 <- AgiBot's documented A3 TTS RPC port (HDU) +# +# NOTE: AgiBot's own examples use 10.42.10.10 - that is the robot's INTERNAL +# address for its head unit (HDU) and is not reachable from your PC. Use the +# HDU's address on your WiFi/LAN. Find it with: python scripts/discover_robot.py +ROBOT_IP= +ROBOT_PORT=59301 +ROBOT_USE_TLS=false + +# Timing (seconds). Keep connect timeout short so the UI never feels stuck. +ROBOT_CONNECT_TIMEOUT=3.0 +ROBOT_REQUEST_TIMEOUT=8.0 +ROBOT_HEALTH_INTERVAL=5.0 +ROBOT_RECONNECT_MIN_DELAY=1.0 +ROBOT_RECONNECT_MAX_DELAY=15.0 + + +# ----------------------------------------------------------------------------- +# 4. HOW TO TALK TO THE A3 +# ----------------------------------------------------------------------------- +# Only used when ROBOT_MODE=real. +# +# aimdk - AgiBot's documented A3 speech RPC <<< USE THIS, it is the default >>> +# http - a generic HTTP/REST endpoint you specify yourself +# ws - a WebSocket endpoint you specify yourself +# ros2 - a ROS 2 topic or service (requires ROS 2 on this PC) +# ssh - run a command on the robot over SSH (last-resort fallback) +# +# The `aimdk` transport implements the interface AgiBot documents for the A3: +# POST http://:59301/rpc/aimdk.protocol.TTSService/PlayTTS +# Content-Type: application/json +# {"text": "...", "priority_level": "INTERACTION_L6", "domain": "...", +# "trace_id": "...", "is_interrupted": true} +# Docs: https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play +# +# The defaults below come from that documentation, but they are still settings +# rather than hard-coded values: AgiBot does not guarantee ports or names across +# firmware, and there is no endpoint-discovery API. If your unit differs, change +# it here - never in the source. See docs/AGIBOT_A3_INTEGRATION.md. +A3_TRANSPORT=aimdk + + +# --- 4a. AimDK transport (recommended) --------------------------------------- +# Uses ROBOT_IP and ROBOT_PORT above. +A3_AIMDK_SERVICE=aimdk.protocol.TTSService +A3_AIMDK_PLAY_METHOD=PlayTTS +A3_AIMDK_STOP_METHOD=StopTTSTraceId +A3_AIMDK_STATUS_METHOD=GetAudioStatus + +# Playback priority. Only INTERACTION_L6 appears in AgiBot's A3 examples. +A3_AIMDK_PRIORITY=INTERACTION_L6 + +# Free-form caller tag sent with each request, so robot-side logs show who spoke. +A3_AIMDK_DOMAIN=voice_control + +# true = a new utterance interrupts whatever is currently playing. +A3_AIMDK_INTERRUPT=true + +# Hard request limit documented by AgiBot: 1024 BYTES of UTF-8 (not characters). +# Longer text is split automatically on sentence boundaries and sent in order. +A3_AIMDK_MAX_BYTES=1024 + + +# --- 4b. Generic HTTP / REST transport --------------------------------------- +# Only needed if your unit does NOT use the AimDK interface above. +# Path is relative to http://ROBOT_IP:ROBOT_PORT +# Example shape only - replace with the documented endpoint: +# A3_HTTP_SPEAK_PATH=/api/v1/tts/speak +A3_HTTP_SPEAK_PATH= +A3_HTTP_SPEAK_METHOD=POST + +# JSON body template. {text} is replaced with the operator's text. +# Placeholders: {text} {id} {voice} {language} {volume} {speed} +# A key whose value resolves to nothing is dropped from the request. +A3_HTTP_SPEAK_PAYLOAD={"text": "{text}"} + +# Optional: endpoint that interrupts speech. +A3_HTTP_STOP_PATH= +A3_HTTP_STOP_METHOD=POST +A3_HTTP_STOP_PAYLOAD={} + +# Optional: cheap endpoint polled for the connection indicator. +# If left empty the app falls back to a TCP connect test on ROBOT_PORT. +A3_HTTP_STATUS_PATH= +A3_HTTP_STATUS_METHOD=GET + +# Optional: extra headers and auth. +A3_HTTP_HEADERS={} +A3_HTTP_AUTH_TOKEN= + +# Optional: dotted path to a success flag in the JSON response, e.g. "code" or +# "result.success". Leave empty to trust the HTTP status code alone. +A3_HTTP_SUCCESS_FIELD= + + +# --- 4c. WebSocket transport -------------------------------------------------- +A3_WS_PATH= +A3_WS_SPEAK_PAYLOAD={"text": "{text}"} +A3_WS_STOP_PAYLOAD={} +A3_WS_PING_INTERVAL=20 + +# Optional: how the robot announces "finished speaking". With these set, the +# dashboard shows real completion instead of an estimate. +# A3_WS_DONE_FIELD=event +# A3_WS_DONE_VALUE=speech_end +A3_WS_DONE_FIELD= +A3_WS_DONE_VALUE= + + +# --- 4d. ROS 2 transport ------------------------------------------------------ +# Requires ROS 2 installed on this PC and the same ROS_DOMAIN_ID as the robot. +A3_ROS_DOMAIN_ID=0 +A3_ROS_SPEAK_TOPIC= +A3_ROS_SPEAK_MSG_TYPE=std_msgs/msg/String +A3_ROS_SPEAK_MSG_FIELD=data +A3_ROS_STOP_TOPIC= +A3_ROS_USE_SERVICE=false +A3_ROS_SERVICE_NAME= +A3_ROS_SERVICE_TYPE= + + +# --- 4e. SSH transport (last resort) -------------------------------------------- +# Runs a command on the robot's own Linux computer. {text} is shell-quoted. +# A3_SSH_SPEAK_COMMAND= {text} +A3_SSH_USER=root +A3_SSH_PORT=22 +A3_SSH_KEY_PATH= +A3_SSH_PASSWORD= +A3_SSH_SPEAK_COMMAND= +A3_SSH_STOP_COMMAND= +A3_SSH_PROBE_COMMAND=true + + +# --- 4f. Optional TTS parameters (sent only when set) ------------------------- +A3_VOICE= +A3_LANGUAGE= +A3_VOLUME= +A3_SPEED= + + +# ----------------------------------------------------------------------------- +# 5. SPEECH BEHAVIOUR +# ----------------------------------------------------------------------------- +SPEECH_MAX_LENGTH=1000 +SPEECH_MIN_LENGTH=1 +# true = pressing Speak while talking interrupts and says the new text +# false = pressing Speak while talking is rejected with "already speaking" +SPEECH_ALLOW_INTERRUPT=true +SPEECH_HISTORY_LIMIT=100 + + +# ----------------------------------------------------------------------------- +# 6. MOCK ROBOT (ROBOT_MODE=mock only) +# ----------------------------------------------------------------------------- +MOCK_CONNECT_DELAY_MS=350 +MOCK_NETWORK_LATENCY_MS=45 +MOCK_PROCESSING_MS=180 +MOCK_WORDS_PER_MINUTE=150 + +# Set to 0.2 to make 1 in 5 utterances fail, for testing error handling. +MOCK_FAILURE_RATE=0.0 +# Set to true to make the simulated link drop occasionally, for testing reconnect. +MOCK_FLAKY_CONNECTION=false +# --- PC speaker playback (simulator only) ------------------------------------ +# true = the simulator actually SPEAKS the text through this laptop's speakers, +# so you can rehearse a demo before the robot is on the network. +# When this is on, the real audio drives the UI: "Completed" appears exactly when +# the sound stops, and Stop cuts the voice mid-word. +# +# Windows uses the built-in SAPI voices - nothing to install. +# macOS uses `say`. Linux needs: sudo apt install espeak-ng +MOCK_LOCAL_AUDIO=true + +# Which PC voice to use (matched against the voice name, case-insensitive). +# python scripts/voices.py list what is installed +# python scripts/voices.py --demo hear each one +# +# NOTE: this is the SIMULATOR's voice, not the robot's - the real A3 synthesises +# speech on-board. The robot's configured voice is "Yunxiao": a MALE TEENAGER, +# multi-language. Windows ships no teenage voice, so the closest approximation is +# its lighter adult male voice (Mark) pitched up a little. It is a stand-in, not +# a match. Run `python scripts/voices.py --demo` to hear the alternatives. +MOCK_VOICE=Mark + +# Speaking rate, -10 (slowest) to 10 (fastest). Try -1 or -2 for a noisy room. +MOCK_SPEECH_RATE=0 + +# Volume, 0-100. +MOCK_SPEECH_VOLUME=100 + +# Pitch, -10 (deepest) to 10 (highest). Left at 0: raising it makes the voice +# sound affected rather than natural, which is the opposite of what a service +# robot should sound like. +MOCK_SPEECH_PITCH=0 + + +# ----------------------------------------------------------------------------- +# 7. GEMINI CLOUD VOICE (simulator only - nothing here touches the real robot) +# ----------------------------------------------------------------------------- +# The built-in Windows voice is instant but robotic. Gemini gives the simulator a +# natural neural voice, at the cost of a network round trip. +# +# system = built-in OS voice - instant, robotic +# gemini = Gemini neural voice - natural, ~4s to synthesise new text +MOCK_VOICE_ENGINE=system + +# Get a key at https://aistudio.google.com/apikey +GEMINI_API_KEY= + +GEMINI_TTS_MODEL=gemini-3.1-flash-tts-preview + +# Prebuilt voice. Iapetus ("Clear") is the default: professional and straight, +# which is how a tour-guide robot actually speaks. Charon ("Informative") is the +# same register but deeper and more adult - swap it in if you prefer that. +# Puck / Fenrir read younger but bouncier. Hear them all: +# python scripts/voices.py +GEMINI_VOICE=Iapetus + +# Optional acting direction. LEAVE THIS EMPTY. Anything here makes the model +# *perform* the line instead of simply saying it, which sounds theatrical and +# fake for a service robot. Only set it for a deliberate stage character. +GEMINI_TTS_STYLE= + +# 0 = never split. Every line is spoken as ONE clip, in one continuous take. +# Set a character count (e.g. 280) only if you want very long paragraphs to +# start speaking before the whole thing is synthesised - the pieces are then +# synthesised separately, and the join between them can be audible. +GEMINI_CHUNK_CHARS=0 + +# LATENCY: new text takes ~4s (short) to ~8s (paragraph) to synthesise. Audio is +# cached on disk in .voice-cache/, so a repeated line is INSTANT and stays +# instant across restarts. Before a live demo, warm your lines: +# python scripts/warm_voice.py --file demo_lines.txt +# If Gemini is unreachable the simulator falls back to the built-in voice, so a +# network problem never leaves you with silence. diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..7817409 --- /dev/null +++ b/.gitignore @@ -0,0 +1,13 @@ +# Local configuration - contains the robot address and any credentials. +.env + +__pycache__/ +*.py[cod] +.venv/ +venv/ +*.log +.DS_Store + + +# Saved speech audio (regenerated on demand). +audio_library/ diff --git a/README.md b/README.md new file mode 100644 index 0000000..201ae52 --- /dev/null +++ b/README.md @@ -0,0 +1,392 @@ +# AGIBOT A3 · Voice Control + +A local web dashboard for making an **AGIBOT A3** humanoid speak typed text through +its own built-in speaker. + +``` +PC Browser → Local Web App → Robot Service Layer → A3 Adapter → AGIBOT A3 → Speaker +``` + +It runs today, with **no robot attached**, against a built-in simulator. When the +robot arrives you change the IP in one file and switch one setting. + +**The A3's speech interface is already implemented.** AgiBot publicly documents a +native text-to-speech HTTP endpoint on the robot, and this project speaks it: + +```http +POST http://:59301/rpc/aimdk.protocol.TTSService/PlayTTS +Content-Type: application/json +{"text": "...", "priority_level": "INTERACTION_L6", "domain": "voice_control", + "trace_id": "...", "is_interrupted": true} +``` + +Text goes in, the robot's own TTS comes out of its own speaker. **No audio is +generated or transferred by the PC**, which is what makes it fast. Full detail, +sources and open questions: **[docs/AGIBOT_A3_INTEGRATION.md](docs/AGIBOT_A3_INTEGRATION.md)**. + +
+ +## Quick start + +```bat +start.bat +``` + +…then open **http://localhost:8000**. + +Or manually: + +```bash +pip install -r requirements.txt +python backend/main.py +``` + +The app starts in **Mock Mode**: the dashboard, the API, the status pipeline, the +history and every error path work exactly as they will with the real robot — the +robot itself is simulated. + +**You will hear it.** The simulator speaks each utterance through this PC's +speakers, so you can rehearse a demo — wording, pacing, the Stop button — before +the A3 is on the network. The real audio drives the UI: *Completed* appears when +the sound actually stops, and **Stop** cuts the voice mid-word. + +Two voice engines, set by `MOCK_VOICE_ENGINE` in `.env`: + +| | `system` | `gemini` | +| --- | --- | --- | +| Sound | robotic, built-in OS voice | **natural neural voice** | +| New text | instant | ~4 s (sentence) – 8 s (paragraph) | +| Repeated text | instant | **instant** — cached on disk | +| Needs | nothing | `GEMINI_API_KEY` + internet | + +```bash +python scripts/voices.py # list both engines' voices +python scripts/voices.py --demo # hear the system ones +``` + +### Saved audio + +Every line spoken with the neural voice is saved to **`audio_library/`** as an +ordinary `.wav` named after its text: + +``` +audio_library/ + welcome-to-our-showroom-78e24feb.wav + please-follow-me-to-the-first-exhibit-653c14ff.wav + index.json +``` + +That means a line spoken once **replays instantly** — no synthesis, no network, +no wait — and the files are yours: play them in any media player, drop them into +a video edit, or put them on a stand's playlist. + +The dashboard's **Saved audio** panel lists them with play and download buttons, +and history rows that already have audio get a ▶ for instant replay. + +**Before a live demo, warm your lines** so nothing waits on synthesis: + +```bash +python scripts/warm_voice.py --file demo_lines.txt +python scripts/warm_voice.py --stats +``` + +If Gemini is unreachable — no network, bad key, quota — the simulator silently +falls back to the built-in voice, so a network problem never leaves you with +silence. + +Pick one with `MOCK_VOICE=` in `.env`; silence it entirely with +`MOCK_LOCAL_AUDIO=false`. + +This is the *simulator's* voice — the real A3 synthesises its own speech +on-board. The robot's configured voice is **"Yunxiao"** (teenager, male, +multi-language). Windows ships no teenage voice, so the default here is its +lighter adult male voice (**Mark**) pitched up (`MOCK_SPEECH_PITCH=3`) to read +younger. It is an approximation for rehearsal, not a match. + +Verify the whole stack end to end at any time: + +```bash +python scripts/selftest.py # 31 checks: REST, WebSocket, speak, stop, errors, history +``` + +### Testing the real robot path before the robot exists + +Mock mode tests the *application*. To exercise the actual A3 wire protocol — +the URL, the JSON body, the `trace_id` round trip, byte-limit chunking, Stop — +run the local stand-in: + +```bash +python scripts/fake_a3_server.py # implements the documented A3 contract on :59301 +``` + +then set `ROBOT_MODE=real`, `ROBOT_IP=127.0.0.1` in `.env`. Every utterance +prints in the stand-in's console. This project's `aimdk` transport passes all 31 +checks against it. + +
+ +## Using the dashboard + +| Action | How | +| --- | --- | +| Speak | Type, then click **Speak** or press **Ctrl + Enter** | +| Stop | **Stop** button or **Esc** | +| Clear the box | **Clear** button | +| Repeat something | Click any entry in **History** — it goes back in the box | +| Reconnect | The ↻ button in the header | + +`Enter` inserts a newline and does **not** send — a half-typed sentence should +never reach the robot's speaker mid-demo. + +The header always states the truth about the connection: green *Robot Connected*, +amber *Connecting…*, red *Robot Disconnected*, plus the measured round-trip +latency to the robot. + +
+ +## Project layout + +``` +A3_text_to_speach/ +│ +├── frontend/ no build step - plain ES modules, served by the backend +│ ├── index.html +│ ├── styles/main.css +│ └── js/ +│ ├── app.js wiring + speech lifecycle +│ ├── api.js REST client +│ ├── socket.js WebSocket client (auto-reconnect, RTT ping) +│ └── ui.js all DOM rendering +│ +├── backend/ +│ ├── main.py entry point: app factory, static serving, error mapping +│ ├── api/ +│ │ ├── routes.py REST endpoints +│ │ ├── websocket.py live status/lifecycle channel +│ │ └── schemas.py +│ ├── services/ +│ │ ├── speech_service.py validation, single-flight, interrupt, latency +│ │ └── history.py +│ ├── robot/ +│ │ ├── base.py ← the only contract the app depends on +│ │ ├── factory.py mode → adapter +│ │ ├── manager.py connect / health-poll / backoff reconnect +│ │ ├── mock_robot.py the simulator +│ │ ├── agibot_a3.py ← THE INTEGRATION POINT +│ │ └── transports/ +│ │ ├── aimdk_transport.py ← AgiBot's documented A3 speech RPC (default) +│ │ └── http · ws · ros2 · ssh fallbacks, fully config-driven +│ ├── core/ event bus, text utils, logging +│ └── config/settings.py every tunable, loaded from .env +│ +├── docs/ +│ ├── AGIBOT_A3_INTEGRATION.md how A3 speech works + what to fill in +│ └── NETWORK.md PC ↔ robot networking and troubleshooting +│ +├── scripts/ +│ ├── voices.py list / audition the simulator's voices +│ ├── warm_voice.py pre-synthesise demo lines (instant playback) +│ ├── selftest.py end-to-end test (31 checks) +│ ├── discover_robot.py probe a robot IP for its speech interface +│ └── fake_a3_server.py local stand-in for the A3's RPC, to test the real path +│ +├── audio_library/ saved .wav files + index (instant replay) +├── pronunciation.json how the simulator says tricky names +├── .env ← YOUR CONFIGURATION (robot IP goes here) +├── .env.example documented template +├── requirements.txt +└── start.bat / start.sh +``` + +### Why it is layered this way + +Each arrow is a seam you can replace without touching the others: + +``` +Browser + │ REST for commands, WebSocket for state +Backend API (api/) + │ domain objects only - no HTTP, no sockets +Speech Service (services/) + │ RobotAdapter interface +Robot Adapter (robot/mock_robot.py | robot/agibot_a3.py) + │ SpeechTransport interface +Transport (robot/transports/*) + │ +AGIBOT A3 +``` + +Nothing above `robot/` knows a robot SDK exists. Adding an AgiBot X2, a second +robot, or a different TTS engine means writing one adapter and adding one line to +`factory.py` — the UI, the API and the service layer are untouched. + +
+ +## Configuration + +Everything lives in **`.env`** (created from `.env.example` on first run). The +source code contains no IP address, port or endpoint. + +```env +ROBOT_MODE=mock # mock | real ← the switch +ROBOT_IP= # ← the robot's IP goes here +ROBOT_PORT=59301 # AgiBot's documented A3 TTS RPC port +A3_TRANSPORT=aimdk # aimdk | http | ws | ros2 | ssh +``` + +After editing `.env`, either restart the server or call: + +```bash +curl -X POST http://localhost:8000/api/config/reload +``` + +which rebuilds the robot adapter in place — no restart, no code change. + +### Mock-mode knobs worth knowing + +| Setting | Effect | +| --- | --- | +| `MOCK_FAILURE_RATE=0.3` | 3 in 10 utterances fail — exercises the error UI | +| `MOCK_FLAKY_CONNECTION=true` | the link drops periodically — exercises reconnect | +| `MOCK_LOCAL_AUDIO=true` | **on by default** — actually speaks through this PC's speakers | +| `MOCK_VOICE_ENGINE=gemini` | natural neural voice instead of the robotic built-in one | +| `GEMINI_VOICE=Puck` | which neural voice (Puck/Fenrir read youngest + male) | +| `MOCK_VOICE=Mark` | which *system* voice to use, when engine is `system` | +| `MOCK_SPEECH_PITCH=3` | raise the pitch — approximates the robot's young male voice | +| `MOCK_SPEECH_RATE=-2` | slow the simulator down for a noisy room (−10…10) | +| `SPEECH_ALLOW_INTERRUPT=false` | a second Speak while talking is rejected instead of interrupting | + +
+ +## API + +| Method | Path | Purpose | +| --- | --- | --- | +| `GET` | `/api/health` | web app liveness (always 200, even with the robot offline) | +| `GET` | `/api/robot/status` | connection state, latency, uptime, config problems | +| `POST` | `/api/robot/speak` | `{"text": "..."}` → speak it | +| `POST` | `/api/robot/stop` | interrupt the current utterance | +| `POST` | `/api/robot/reconnect` | retry the connection immediately | +| `GET` | `/api/robot/diagnostics` | adapter + transport detail, for troubleshooting | +| `GET` | `/api/speech/history` | recent utterances | +| `DELETE` | `/api/speech/history` | clear it | +| `GET` | `/api/audio` | saved clips + library stats | +| `GET` | `/api/audio/{id}/file` | the `.wav` itself (play or download) | +| `DELETE` | `/api/audio/{id}` | delete one clip | +| `DELETE` | `/api/audio` | delete all saved audio | +| `GET` | `/api/config` | non-secret configuration the UI needs | +| `POST` | `/api/config/reload` | re-read `.env` and rebuild the adapter | +| `WS` | `/ws` | live status, speech lifecycle, history updates | + +Interactive docs: **http://localhost:8000/api/docs** + +```bash +curl -X POST http://localhost:8000/api/robot/speak \ + -H "Content-Type: application/json" \ + -d "{\"text\": \"Hello, welcome to our showroom.\"}" +``` + +```json +{ "success": true, "status": "processing", + "requestId": "3c7cfe41cf7d", "ackLatencyMs": 56 } +``` + +**`/api/robot/speak` returns when the robot has *accepted* the utterance, not when +it stops talking.** A 12-second sentence must not look like a 12-second-slow +button. The rest of the lifecycle — `speaking`, `completed`, failures — arrives on +the WebSocket. + +
+ +## How the speed is achieved + +| Decision | Why | +| --- | --- | +| Text is sent to the robot; audio is never generated on the PC | no synthesis, no file, no upload — one small HTTP request | +| One persistent connection, opened at startup and reused | no TCP/TLS handshake per utterance | +| HTTP response returns at acknowledgement | button feels instant; speaking progress streams separately | +| WebSocket push, zero polling | the UI changes the moment the robot does | +| Health probe is separate from the speech path | a slow health check can never delay a Speak | +| Short connect timeout (3 s), long request timeout (8 s) | an offline robot fails fast; a busy one is not cut off | + +
+ +## Error handling + +| Situation | What you see | +| --- | --- | +| Robot offline | `Robot is offline. Check the robot IP address and network connection.` | +| `ROBOT_IP` empty while `ROBOT_MODE=real` | a configuration banner naming the exact `.env` key | +| Invalid IP / hostname | configuration error before any connection is attempted | +| Connection timeout | fails after 3 s, UI stays responsive, retry continues in the background | +| TTS failure | `Speech request failed.` + the robot's own reason in the server log | +| Empty text | Speak is disabled; the API returns `400` | +| Text too long | rejected with the actual limit named | +| Disconnect mid-utterance | status flips immediately and the utterance is marked failed | +| Backend stopped | the page says so rather than showing a stale green light | + +The robot being unreachable is treated as normal, not exceptional: the server +keeps serving the dashboard and retries with capped exponential backoff. + +
+ +## Security + +The browser never talks to the robot. + +``` +Browser → local backend (127.0.0.1) → robot +``` + +- The server binds to `127.0.0.1` by default — nothing is exposed to the LAN. +- Robot credentials stay in `.env` on the PC; `/api/config` returns a filtered + view with no tokens, passwords or key paths. +- No robot port is proxied or forwarded to the page. + +Set `HOST=0.0.0.0` only if you deliberately want to open the dashboard from a +tablet on the same network. + +
+ +## Connecting the real robot + +Full procedure: **[docs/AGIBOT_A3_INTEGRATION.md](docs/AGIBOT_A3_INTEGRATION.md)** +· Networking and troubleshooting: **[docs/NETWORK.md](docs/NETWORK.md)** + +Short version, once you have the robot's IP: + +```bash +ping 192.168.1.50 # 1. is it reachable +python scripts/discover_robot.py 192.168.1.50 # 2. what does it expose +``` + +```bash +# 3. the decisive test - this makes the robot talk +python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3" +``` + +then in `.env` — **three lines, no code changes**: + +```env +ROBOT_MODE=real +ROBOT_IP=192.168.1.50 +ROBOT_PORT=59301 +``` + +and restart (or `curl -X POST http://localhost:8000/api/config/reload`). + +> Do **not** use `10.42.10.10`. It appears throughout AgiBot's examples but is +> the robot's *internal* address — your PC cannot reach it. Use the head unit's +> address on your own network. + +
+ +## Requirements + +- **Python 3.9+** (developed and tested on 3.11) +- Windows, macOS or Linux +- A browser from the last few years + +Runtime dependencies: `fastapi`, `uvicorn`, `python-dotenv`, `pydantic`, `httpx`, +`websockets`. Optional: `pyttsx3` (mock audio), `paramiko` (SSH password auth), +`rclpy` (ROS 2 transport — comes from a ROS 2 install, not pip). diff --git a/backend/__init__.py b/backend/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/api/__init__.py b/backend/api/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/api/routes.py b/backend/api/routes.py new file mode 100644 index 0000000..2aee4ee --- /dev/null +++ b/backend/api/routes.py @@ -0,0 +1,172 @@ +"""HTTP API. + +The browser talks only to this backend; the backend talks to the robot. No robot +address, port or credential ever reaches the page. +""" + +from __future__ import annotations + +import logging +import platform +import time +from typing import Any, Dict + +from fastapi import APIRouter, HTTPException, Request +from fastapi.responses import FileResponse + +from ..config.settings import reload_settings +from ..core.events import EventType +from ..robot.manager import RobotManager +from ..services.audio_library import get_audio_library +from ..services.speech_service import SpeechService +from .schemas import SimpleResponse, SpeakRequest, SpeakResponse + +logger = logging.getLogger(__name__) + +router = APIRouter(prefix="/api", tags=["robot"]) + +STARTED_AT = time.time() + + +def _manager(request: Request) -> RobotManager: + return request.app.state.robot_manager + + +def _speech(request: Request) -> SpeechService: + return request.app.state.speech_service + + +# --------------------------------------------------------------------------- # +# health / config +# --------------------------------------------------------------------------- # +@router.get("/health") +async def health(request: Request) -> Dict[str, Any]: + """Liveness of the *web app* - always 200, even with the robot offline.""" + return { + "status": "ok", + "uptimeSeconds": round(time.time() - STARTED_AT, 1), + "python": platform.python_version(), + "mode": request.app.state.settings.robot.mode, + } + + +@router.get("/config") +async def get_config(request: Request) -> Dict[str, Any]: + return request.app.state.settings.public_dict() + + +@router.post("/config/reload") +async def post_config_reload(request: Request) -> Dict[str, Any]: + """Re-read .env and rebuild the robot adapter without restarting the server. + + This is what you call after pasting the robot's IP into .env. + """ + settings = reload_settings() + request.app.state.settings = settings + _speech(request).update_settings(settings) + await _manager(request).rebuild(settings) + request.app.state.event_bus.publish(EventType.CONFIG_UPDATED, settings.public_dict()) + logger.info("configuration reloaded: mode=%s address=%s", + settings.robot.mode, settings.robot.address) + return {"success": True, "config": settings.public_dict()} + + +# --------------------------------------------------------------------------- # +# robot +# --------------------------------------------------------------------------- # +@router.get("/robot/status") +async def robot_status(request: Request) -> Dict[str, Any]: + status = _manager(request).status_dict() + status["busy"] = _speech(request).is_busy + status["activeRequestId"] = _speech(request).active_request_id + return status + + +@router.get("/robot/diagnostics") +async def robot_diagnostics(request: Request) -> Dict[str, Any]: + manager = _manager(request) + detail = await manager.adapter.describe() + return { + "status": manager.status_dict(), + "adapter": detail, + "config": request.app.state.settings.public_dict(), + } + + +@router.post("/robot/speak", response_model=SpeakResponse) +async def robot_speak(payload: SpeakRequest, request: Request) -> Dict[str, Any]: + """Send text to the robot. + + Returns once the robot has *acknowledged* the utterance. Speaking progress and + completion arrive over the WebSocket at /ws. + """ + return await _speech(request).speak(payload.text, payload.voice, payload.language) + + +@router.post("/robot/stop", response_model=SimpleResponse) +async def robot_stop(request: Request) -> Dict[str, Any]: + return await _speech(request).stop() + + +@router.post("/robot/reconnect") +async def robot_reconnect(request: Request) -> Dict[str, Any]: + manager = _manager(request) + manager.request_reconnect() + return {"success": True, "status": manager.state.value} + + +# --------------------------------------------------------------------------- # +# history +# --------------------------------------------------------------------------- # +@router.get("/speech/history") +async def speech_history(request: Request) -> Dict[str, Any]: + service = _speech(request) + items = service.annotate_saved(service.history.list()) + return {"items": items, "count": len(items)} + + +@router.delete("/speech/history") +async def speech_history_delete(request: Request) -> Dict[str, Any]: + return _speech(request).clear_history() + + +@router.post("/speech/history/clear") +async def speech_history_clear(request: Request) -> Dict[str, Any]: + """Same as DELETE - kept because it is easier to call from a plain form/curl.""" + return _speech(request).clear_history() + + +# --------------------------------------------------------------------------- # +# saved audio +# --------------------------------------------------------------------------- # +@router.get("/audio") +async def audio_list() -> Dict[str, Any]: + """Every line that has been synthesised and saved as a .wav.""" + library = get_audio_library() + return {"items": library.list(), "stats": library.stats()} + + +@router.get("/audio/{audio_id}/file") +async def audio_file(audio_id: str): + """The .wav itself - playable in the browser, or downloadable.""" + entry = get_audio_library().get(audio_id) + if entry is None: + raise HTTPException(status_code=404, detail="No saved audio with that id.") + path = get_audio_library().root / entry["file"] + return FileResponse( + str(path), media_type="audio/wav", filename=entry["file"], + headers={"Cache-Control": "public, max-age=31536000"}, + ) + + +@router.delete("/audio/{audio_id}") +async def audio_delete(audio_id: str) -> Dict[str, Any]: + if not get_audio_library().delete(audio_id): + raise HTTPException(status_code=404, detail="No saved audio with that id.") + return {"success": True} + + +@router.delete("/audio") +async def audio_clear() -> Dict[str, Any]: + """Delete every saved clip. They are re-created on demand.""" + return {"success": True, "removed": get_audio_library().clear()} diff --git a/backend/api/schemas.py b/backend/api/schemas.py new file mode 100644 index 0000000..77d7e92 --- /dev/null +++ b/backend/api/schemas.py @@ -0,0 +1,32 @@ +"""Request/response models for the HTTP API.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import BaseModel, Field + + +class SpeakRequest(BaseModel): + text: str = Field(..., description="What the robot should say.") + voice: Optional[str] = Field(None, description="Optional voice id, if the robot supports it.") + language: Optional[str] = Field(None, description="Optional language/locale hint.") + + +class SpeakResponse(BaseModel): + success: bool + status: str + requestId: str + text: str + ackLatencyMs: Optional[int] = None + + +class SimpleResponse(BaseModel): + success: bool + status: Optional[str] = None + + +class ErrorResponse(BaseModel): + success: bool = False + error: str + errorCode: str diff --git a/backend/api/websocket.py b/backend/api/websocket.py new file mode 100644 index 0000000..b6e7c55 --- /dev/null +++ b/backend/api/websocket.py @@ -0,0 +1,114 @@ +"""Real-time channel to the dashboard. + +One WebSocket replaces all polling: connection state, speech lifecycle, errors and +history updates are pushed the moment they happen. The socket also answers +client pings so the UI can display an honest browser->backend round-trip time. +""" + +from __future__ import annotations + +import asyncio +import json +import logging +from typing import Any, Dict + +from fastapi import APIRouter, WebSocket, WebSocketDisconnect + +from ..core.events import Event, EventBus, EventType + +logger = logging.getLogger(__name__) + +ws_router = APIRouter() + + +@ws_router.websocket("/ws") +async def dashboard_socket(websocket: WebSocket) -> None: + await websocket.accept() + app = websocket.app + bus: EventBus = app.state.event_bus + manager = app.state.robot_manager + speech = app.state.speech_service + settings = app.state.settings + + subscription = bus.subscribe(name="ws") + logger.debug("dashboard socket opened (subscribers=%d)", bus.subscriber_count) + + try: + # Prime the page with everything it needs for a correct first paint. + await _send( + websocket, + { + "type": "hello", + "data": { + "config": settings.public_dict(), + "status": _status(manager, speech), + "history": speech.annotate_saved(speech.history.list()), + }, + }, + ) + + pump = asyncio.create_task( + _pump(websocket, subscription, speech), name="ws-pump" + ) + try: + while True: + raw = await websocket.receive_text() + await _handle_client_message(websocket, raw, manager, speech) + finally: + pump.cancel() + try: + await pump + except (asyncio.CancelledError, Exception): + pass + + except WebSocketDisconnect: + logger.debug("dashboard socket closed by client") + except Exception as exc: # pragma: no cover - transport level + logger.debug("dashboard socket error: %s", exc) + finally: + subscription.close() + + +async def _pump(websocket: WebSocket, subscription, speech) -> None: + """Forward bus events to this browser until cancelled.""" + while True: + event: Event = await subscription.get() + if event is None: # pragma: no cover + continue + payload = event.to_dict() + if event.type == EventType.ROBOT_STATUS: + # The robot layer does not know about the speech queue; add it here so + # every status the page sees carries a current busy flag. + payload["data"] = dict(payload["data"]) + payload["data"]["busy"] = speech.is_busy + payload["data"]["activeRequestId"] = speech.active_request_id + await _send(websocket, payload) + + +async def _handle_client_message(websocket: WebSocket, raw: str, manager, speech) -> None: + try: + message: Dict[str, Any] = json.loads(raw) + except json.JSONDecodeError: + return + + kind = message.get("type") + if kind == "ping": + # Echo the client's timestamp so it can compute RTT without clock sync. + await _send(websocket, {"type": "pong", "data": {"t": message.get("t")}}) + elif kind == "status": + await _send( + websocket, {"type": EventType.ROBOT_STATUS, "data": _status(manager, speech)} + ) + elif kind == "reconnect": + manager.request_reconnect() + + +def _status(manager, speech) -> Dict[str, Any]: + status = manager.status_dict() + status["busy"] = speech.is_busy + status["activeRequestId"] = speech.active_request_id + return status + + +async def _send(websocket: WebSocket, payload: Dict[str, Any]) -> None: + await websocket.send_text(json.dumps(payload, ensure_ascii=False, default=str)) diff --git a/backend/config/__init__.py b/backend/config/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/config/settings.py b/backend/config/settings.py new file mode 100644 index 0000000..a1a3fc0 --- /dev/null +++ b/backend/config/settings.py @@ -0,0 +1,540 @@ +"""Typed application configuration, loaded once from the environment / .env file. + +Everything that could differ between "my laptop with no robot" and "the show floor +with a live AGIBOT A3" lives here. No module outside this package reads os.environ +directly, and no robot address is ever hard-coded in source. +""" + +from __future__ import annotations + +import ipaddress +import json +import os +import re +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, List, Literal, Optional + +from dotenv import load_dotenv + +PROJECT_ROOT = Path(__file__).resolve().parents[2] +ENV_FILE = PROJECT_ROOT / ".env" + +RobotMode = Literal["mock", "real"] +Transport = Literal["aimdk", "http", "ws", "ros2", "ssh"] + +TRANSPORTS = ("aimdk", "http", "ws", "ros2", "ssh") + + +# --------------------------------------------------------------------------- # +# env helpers +# --------------------------------------------------------------------------- # +def _raw(key: str) -> Optional[str]: + value = os.environ.get(key) + if value is None: + return None + value = value.strip() + # Treat `KEY=` and quoted-empty as "unset" rather than "empty string". + if value in ('', '""', "''"): + return None + if len(value) >= 2 and value[0] == value[-1] and value[0] in ('"', "'"): + value = value[1:-1] + return value or None + + +def env_str(key: str, default: str = "") -> str: + value = _raw(key) + return default if value is None else value + + +def env_opt(key: str) -> Optional[str]: + return _raw(key) + + +def env_int(key: str, default: int) -> int: + value = _raw(key) + if value is None: + return default + try: + return int(value) + except ValueError: + return default + + +def env_float(key: str, default: float) -> float: + value = _raw(key) + if value is None: + return default + try: + return float(value) + except ValueError: + return default + + +def env_bool(key: str, default: bool) -> bool: + value = _raw(key) + if value is None: + return default + return value.lower() in ("1", "true", "yes", "on", "y") + + +def env_json(key: str, default: Any) -> Any: + """Parse a JSON-valued env var; fall back to `default` on malformed input.""" + value = _raw(key) + if value is None: + return default + try: + return json.loads(value) + except json.JSONDecodeError: + return default + + +def env_list(key: str, default: Optional[List[str]] = None) -> List[str]: + value = _raw(key) + if value is None: + return list(default or []) + return [item.strip() for item in value.split(",") if item.strip()] + + +# --------------------------------------------------------------------------- # +# configuration issues (surfaced in the UI instead of crashing the app) +# --------------------------------------------------------------------------- # +@dataclass(frozen=True) +class ConfigIssue: + level: Literal["error", "warning"] + key: str + message: str + + def to_dict(self) -> Dict[str, str]: + return {"level": self.level, "key": self.key, "message": self.message} + + +# --------------------------------------------------------------------------- # +# settings groups +# --------------------------------------------------------------------------- # +@dataclass(frozen=True) +class ServerSettings: + host: str + port: int + log_level: str + cors_origins: List[str] + + +@dataclass(frozen=True) +class RobotSettings: + """Vendor-neutral robot connection settings.""" + + mode: RobotMode + name: str + model: str + ip: Optional[str] + port: int + use_tls: bool + connect_timeout: float + request_timeout: float + health_interval: float + reconnect_min_delay: float + reconnect_max_delay: float + + @property + def scheme(self) -> str: + return "https" if self.use_tls else "http" + + @property + def ws_scheme(self) -> str: + return "wss" if self.use_tls else "ws" + + @property + def base_url(self) -> str: + return "{0}://{1}:{2}".format(self.scheme, self.ip, self.port) + + @property + def ws_base_url(self) -> str: + return "{0}://{1}:{2}".format(self.ws_scheme, self.ip, self.port) + + @property + def address(self) -> str: + return "{0}:{1}".format(self.ip, self.port) if self.ip else "(not configured)" + + +@dataclass(frozen=True) +class A3Settings: + """AGIBOT A3 adapter settings. + + IMPORTANT: every endpoint below is *configuration*, never a guess baked into + code. See docs/AGIBOT_A3_INTEGRATION.md - once the interface for your unit is + confirmed you fill these in and the adapter works unchanged. + """ + + transport: Transport + + # -- AimDK transport (AgiBot's documented A3 speech RPC) ------------------ + # POST http://:/rpc// + # https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play + aimdk_service: str + aimdk_play_method: str + aimdk_stop_method: str + aimdk_status_method: str + aimdk_priority: str + aimdk_domain: str + aimdk_interrupt: bool + aimdk_max_bytes: int + + # -- HTTP / REST transport ------------------------------------------------ + http_speak_path: str + http_speak_method: str + http_speak_payload: Any + http_stop_path: str + http_stop_method: str + http_stop_payload: Any + http_status_path: str + http_status_method: str + http_headers: Dict[str, str] + http_auth_token: Optional[str] + http_success_field: Optional[str] + + # -- WebSocket transport -------------------------------------------------- + ws_path: str + ws_speak_payload: Any + ws_stop_payload: Any + ws_ping_interval: float + ws_done_field: Optional[str] + ws_done_value: Optional[str] + + # -- ROS 2 transport ------------------------------------------------------ + ros_domain_id: int + ros_speak_topic: str + ros_speak_msg_type: str + ros_speak_msg_field: str + ros_stop_topic: str + ros_use_service: bool + ros_service_name: str + ros_service_type: str + + # -- SSH / on-robot command transport ------------------------------------- + ssh_user: str + ssh_password: Optional[str] + ssh_key_path: Optional[str] + ssh_port: int + ssh_speak_command: str + ssh_stop_command: str + ssh_probe_command: str + + # -- shared TTS parameters (only sent when set) --------------------------- + voice: Optional[str] + language: Optional[str] + volume: Optional[float] + speed: Optional[float] + + +@dataclass(frozen=True) +class MockSettings: + connect_delay_ms: int + network_latency_ms: int + processing_ms: int + words_per_minute: int + failure_rate: float + local_audio: bool + flaky_connection: bool + voice: Optional[str] + speech_rate: int + speech_volume: int + speech_pitch: int + voice_engine: str + gemini_api_key: Optional[str] + gemini_model: str + gemini_voice: str + gemini_style: str + gemini_chunk_chars: int + pronunciation: bool + + +@dataclass(frozen=True) +class SpeechSettings: + max_length: int + min_length: int + allow_interrupt: bool + history_limit: int + + +@dataclass(frozen=True) +class Settings: + server: ServerSettings + robot: RobotSettings + a3: A3Settings + mock: MockSettings + speech: SpeechSettings + env_file: str + issues: List[ConfigIssue] = field(default_factory=list) + + @property + def is_mock(self) -> bool: + return self.robot.mode == "mock" + + @property + def has_blocking_issue(self) -> bool: + return any(issue.level == "error" for issue in self.issues) + + def public_dict(self) -> Dict[str, Any]: + """Safe-to-expose subset for the browser. Never leaks credentials.""" + return { + "mode": self.robot.mode, + "robotName": self.robot.name, + "robotModel": self.robot.model, + "transport": self.a3.transport if self.robot.mode == "real" else "mock", + "address": self.robot.address, + "ipConfigured": bool(self.robot.ip), + "maxLength": self.speech.max_length, + "allowInterrupt": self.speech.allow_interrupt, + "historyLimit": self.speech.history_limit, + "healthInterval": self.robot.health_interval, + "envFile": self.env_file, + "issues": [issue.to_dict() for issue in self.issues], + } + + +# --------------------------------------------------------------------------- # +# validation +# --------------------------------------------------------------------------- # +def _looks_like_host(value: str) -> bool: + try: + ipaddress.ip_address(value) + return True + except ValueError: + pass + # Permissive hostname check - mDNS names such as "agibot-a3.local" are valid. + return bool(re.fullmatch(r"[A-Za-z0-9]([A-Za-z0-9\-._]{0,251}[A-Za-z0-9])?", value)) + + +def _validate(robot: RobotSettings, a3: A3Settings, speech: SpeechSettings) -> List[ConfigIssue]: + issues: List[ConfigIssue] = [] + + if robot.mode not in ("mock", "real"): + issues.append( + ConfigIssue("error", "ROBOT_MODE", "Unknown ROBOT_MODE '%s'. Use 'mock' or 'real'." % robot.mode) + ) + return issues + + if robot.mode == "real": + if not robot.ip: + issues.append( + ConfigIssue( + "error", + "ROBOT_IP", + "ROBOT_MODE=real but ROBOT_IP is empty. Set the robot's LAN IP address in .env.", + ) + ) + elif not _looks_like_host(robot.ip): + issues.append( + ConfigIssue("error", "ROBOT_IP", "'%s' is not a valid IP address or hostname." % robot.ip) + ) + + if not 0 < robot.port < 65536: + issues.append( + ConfigIssue("error", "ROBOT_PORT", "ROBOT_PORT must be 1-65535, got %s." % robot.port) + ) + + if a3.transport not in TRANSPORTS: + issues.append( + ConfigIssue( + "error", + "A3_TRANSPORT", + "Unknown A3_TRANSPORT '%s'. Use one of: %s." + % (a3.transport, ", ".join(TRANSPORTS)), + ) + ) + elif a3.transport == "aimdk": + if not a3.aimdk_service or not a3.aimdk_play_method: + issues.append( + ConfigIssue( + "error", + "A3_AIMDK_SERVICE", + "AimDK transport needs a service name and a play method.", + ) + ) + if robot.port != 59301: + issues.append( + ConfigIssue( + "warning", + "ROBOT_PORT", + "AgiBot documents the A3 TTS RPC on port 59301; ROBOT_PORT is %s. " + "Confirm the port for your firmware." % robot.port, + ) + ) + elif a3.transport == "http" and not a3.http_speak_path: + issues.append( + ConfigIssue( + "error", + "A3_HTTP_SPEAK_PATH", + "HTTP transport selected but no speak endpoint is configured. " + "Fill it in from the robot's API reference - see docs/AGIBOT_A3_INTEGRATION.md.", + ) + ) + elif a3.transport == "ws" and not a3.ws_path: + issues.append( + ConfigIssue("error", "A3_WS_PATH", "WebSocket transport selected but A3_WS_PATH is empty.") + ) + elif a3.transport == "ros2" and not (a3.ros_speak_topic or a3.ros_service_name): + issues.append( + ConfigIssue( + "error", + "A3_ROS_SPEAK_TOPIC", + "ROS 2 transport selected but neither a topic nor a service name is configured.", + ) + ) + elif a3.transport == "ssh": + if not a3.ssh_speak_command: + issues.append( + ConfigIssue( + "error", + "A3_SSH_SPEAK_COMMAND", + "SSH transport selected but A3_SSH_SPEAK_COMMAND is empty.", + ) + ) + if not a3.ssh_password and not a3.ssh_key_path: + issues.append( + ConfigIssue( + "warning", + "A3_SSH_PASSWORD", + "No SSH password or key configured; authentication will rely on an ssh agent.", + ) + ) + + if speech.max_length < 1: + issues.append(ConfigIssue("error", "SPEECH_MAX_LENGTH", "SPEECH_MAX_LENGTH must be >= 1.")) + + return issues + + +# --------------------------------------------------------------------------- # +# loading +# --------------------------------------------------------------------------- # +def load_settings(env_file: Optional[Path] = None, override: bool = True) -> Settings: + """Read .env + process environment into an immutable Settings object.""" + path = Path(env_file) if env_file else ENV_FILE + if path.exists(): + load_dotenv(path, override=override) + + server = ServerSettings( + host=env_str("HOST", "127.0.0.1"), + port=env_int("PORT", 8000), + log_level=env_str("LOG_LEVEL", "info").lower(), + cors_origins=env_list("CORS_ORIGINS", []), + ) + + robot = RobotSettings( + mode=env_str("ROBOT_MODE", "mock").lower(), # type: ignore[arg-type] + name=env_str("ROBOT_NAME", "AGIBOT A3"), + model=env_str("ROBOT_MODEL", "AgiBot A3"), + ip=env_opt("ROBOT_IP"), + port=env_int("ROBOT_PORT", 59301), + use_tls=env_bool("ROBOT_USE_TLS", False), + connect_timeout=env_float("ROBOT_CONNECT_TIMEOUT", 3.0), + request_timeout=env_float("ROBOT_REQUEST_TIMEOUT", 8.0), + health_interval=env_float("ROBOT_HEALTH_INTERVAL", 5.0), + reconnect_min_delay=env_float("ROBOT_RECONNECT_MIN_DELAY", 1.0), + reconnect_max_delay=env_float("ROBOT_RECONNECT_MAX_DELAY", 15.0), + ) + + a3 = A3Settings( + transport=env_str("A3_TRANSPORT", "aimdk").lower(), # type: ignore[arg-type] + aimdk_service=env_str("A3_AIMDK_SERVICE", "aimdk.protocol.TTSService"), + aimdk_play_method=env_str("A3_AIMDK_PLAY_METHOD", "PlayTTS"), + aimdk_stop_method=env_str("A3_AIMDK_STOP_METHOD", "StopTTSTraceId"), + aimdk_status_method=env_str("A3_AIMDK_STATUS_METHOD", "GetAudioStatus"), + aimdk_priority=env_str("A3_AIMDK_PRIORITY", "INTERACTION_L6"), + aimdk_domain=env_str("A3_AIMDK_DOMAIN", "voice_control"), + aimdk_interrupt=env_bool("A3_AIMDK_INTERRUPT", True), + aimdk_max_bytes=env_int("A3_AIMDK_MAX_BYTES", 1024), + http_speak_path=env_str("A3_HTTP_SPEAK_PATH", ""), + http_speak_method=env_str("A3_HTTP_SPEAK_METHOD", "POST").upper(), + http_speak_payload=env_json("A3_HTTP_SPEAK_PAYLOAD", {"text": "{text}"}), + http_stop_path=env_str("A3_HTTP_STOP_PATH", ""), + http_stop_method=env_str("A3_HTTP_STOP_METHOD", "POST").upper(), + http_stop_payload=env_json("A3_HTTP_STOP_PAYLOAD", {}), + http_status_path=env_str("A3_HTTP_STATUS_PATH", ""), + http_status_method=env_str("A3_HTTP_STATUS_METHOD", "GET").upper(), + http_headers=env_json("A3_HTTP_HEADERS", {}), + http_auth_token=env_opt("A3_HTTP_AUTH_TOKEN"), + http_success_field=env_opt("A3_HTTP_SUCCESS_FIELD"), + ws_path=env_str("A3_WS_PATH", ""), + ws_speak_payload=env_json("A3_WS_SPEAK_PAYLOAD", {"text": "{text}"}), + ws_stop_payload=env_json("A3_WS_STOP_PAYLOAD", {}), + ws_ping_interval=env_float("A3_WS_PING_INTERVAL", 20.0), + ws_done_field=env_opt("A3_WS_DONE_FIELD"), + ws_done_value=env_opt("A3_WS_DONE_VALUE"), + ros_domain_id=env_int("A3_ROS_DOMAIN_ID", 0), + ros_speak_topic=env_str("A3_ROS_SPEAK_TOPIC", ""), + ros_speak_msg_type=env_str("A3_ROS_SPEAK_MSG_TYPE", "std_msgs/msg/String"), + ros_speak_msg_field=env_str("A3_ROS_SPEAK_MSG_FIELD", "data"), + ros_stop_topic=env_str("A3_ROS_STOP_TOPIC", ""), + ros_use_service=env_bool("A3_ROS_USE_SERVICE", False), + ros_service_name=env_str("A3_ROS_SERVICE_NAME", ""), + ros_service_type=env_str("A3_ROS_SERVICE_TYPE", ""), + ssh_user=env_str("A3_SSH_USER", "root"), + ssh_password=env_opt("A3_SSH_PASSWORD"), + ssh_key_path=env_opt("A3_SSH_KEY_PATH"), + ssh_port=env_int("A3_SSH_PORT", 22), + ssh_speak_command=env_str("A3_SSH_SPEAK_COMMAND", ""), + ssh_stop_command=env_str("A3_SSH_STOP_COMMAND", ""), + ssh_probe_command=env_str("A3_SSH_PROBE_COMMAND", "true"), + voice=env_opt("A3_VOICE"), + language=env_opt("A3_LANGUAGE"), + volume=env_float("A3_VOLUME", 1.0) if env_opt("A3_VOLUME") else None, + speed=env_float("A3_SPEED", 1.0) if env_opt("A3_SPEED") else None, + ) + + mock = MockSettings( + connect_delay_ms=env_int("MOCK_CONNECT_DELAY_MS", 350), + network_latency_ms=env_int("MOCK_NETWORK_LATENCY_MS", 45), + processing_ms=env_int("MOCK_PROCESSING_MS", 180), + words_per_minute=env_int("MOCK_WORDS_PER_MINUTE", 150), + failure_rate=env_float("MOCK_FAILURE_RATE", 0.0), + local_audio=env_bool("MOCK_LOCAL_AUDIO", False), + flaky_connection=env_bool("MOCK_FLAKY_CONNECTION", False), + voice=env_opt("MOCK_VOICE"), + speech_rate=env_int("MOCK_SPEECH_RATE", 0), + speech_volume=env_int("MOCK_SPEECH_VOLUME", 100), + speech_pitch=env_int("MOCK_SPEECH_PITCH", 0), + voice_engine=env_str("MOCK_VOICE_ENGINE", "system").lower(), + gemini_api_key=env_opt("GEMINI_API_KEY"), + gemini_model=env_str("GEMINI_TTS_MODEL", "gemini-3.1-flash-tts-preview"), + gemini_voice=env_str("GEMINI_VOICE", "Puck"), + gemini_style=env_str("GEMINI_TTS_STYLE", ""), + gemini_chunk_chars=env_int("GEMINI_CHUNK_CHARS", 0), + pronunciation=env_bool("SPEECH_PRONUNCIATION", True), + ) + + speech = SpeechSettings( + max_length=env_int("SPEECH_MAX_LENGTH", 1000), + min_length=env_int("SPEECH_MIN_LENGTH", 1), + allow_interrupt=env_bool("SPEECH_ALLOW_INTERRUPT", True), + history_limit=env_int("SPEECH_HISTORY_LIMIT", 100), + ) + + return Settings( + server=server, + robot=robot, + a3=a3, + mock=mock, + speech=speech, + env_file=str(path), + issues=_validate(robot, a3, speech), + ) + + +_settings: Optional[Settings] = None + + +def get_settings() -> Settings: + global _settings + if _settings is None: + _settings = load_settings() + return _settings + + +def reload_settings() -> Settings: + """Re-read .env from disk (used by the /api/config/reload endpoint).""" + global _settings + _settings = load_settings(override=True) + return _settings diff --git a/backend/core/__init__.py b/backend/core/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/core/events.py b/backend/core/events.py new file mode 100644 index 0000000..aac6378 --- /dev/null +++ b/backend/core/events.py @@ -0,0 +1,125 @@ +"""A tiny in-process publish/subscribe bus. + +The robot layer publishes; the WebSocket layer subscribes and forwards to the +browser. Keeping this in the middle means the robot adapters never know that a +browser exists, and the API layer never polls the robot. + +Delivery is best-effort and non-blocking: a slow or wedged browser tab can never +stall the robot loop. If a subscriber's queue overflows we drop its oldest event +and mark the stream as lossy rather than applying backpressure upstream. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from dataclasses import dataclass, field +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + +MAX_QUEUE = 256 + + +class EventType(str): + """String constants - kept as plain strings so they serialise for free.""" + + ROBOT_STATUS = "robot.status" + SPEECH_PROGRESS = "speech.progress" + SPEECH_RESULT = "speech.result" + HISTORY_UPDATED = "history.updated" + CONFIG_UPDATED = "config.updated" + NOTICE = "notice" + + +@dataclass +class Event: + type: str + data: Dict[str, Any] = field(default_factory=dict) + at: float = field(default_factory=time.time) + + def to_dict(self) -> Dict[str, Any]: + return {"type": self.type, "data": self.data, "at": self.at} + + +class Subscription: + """An async iterator over bus events, scoped to one subscriber.""" + + def __init__(self, bus: "EventBus", name: str) -> None: + self._bus = bus + self._queue: asyncio.Queue = asyncio.Queue(maxsize=MAX_QUEUE) + self.name = name + self.dropped = 0 + + def _offer(self, event: Event) -> None: + try: + self._queue.put_nowait(event) + except asyncio.QueueFull: + self.dropped += 1 + try: # make room, keep the newest - stale status is worse than none + self._queue.get_nowait() + self._queue.put_nowait(event) + except (asyncio.QueueEmpty, asyncio.QueueFull): # pragma: no cover + pass + + async def get(self, timeout: Optional[float] = None) -> Optional[Event]: + if timeout is None: + return await self._queue.get() + try: + return await asyncio.wait_for(self._queue.get(), timeout=timeout) + except asyncio.TimeoutError: + return None + + def __enter__(self) -> "Subscription": + return self + + def __exit__(self, *exc: Any) -> None: + self.close() + + def close(self) -> None: + self._bus.unsubscribe(self) + + +class EventBus: + def __init__(self) -> None: + self._subscribers: List[Subscription] = [] + self._last: Dict[str, Event] = {} + + # -- subscription -------------------------------------------------------- # + def subscribe(self, name: str = "anonymous") -> Subscription: + sub = Subscription(self, name) + self._subscribers.append(sub) + logger.debug("event subscriber added: %s (total=%d)", name, len(self._subscribers)) + return sub + + def unsubscribe(self, sub: Subscription) -> None: + if sub in self._subscribers: + self._subscribers.remove(sub) + logger.debug("event subscriber removed: %s (total=%d)", sub.name, len(self._subscribers)) + + @property + def subscriber_count(self) -> int: + return len(self._subscribers) + + # -- publishing ---------------------------------------------------------- # + def publish(self, event_type: str, data: Optional[Dict[str, Any]] = None) -> Event: + event = Event(type=event_type, data=data or {}) + self._last[event_type] = event + for sub in list(self._subscribers): + sub._offer(event) + return event + + def last(self, event_type: str) -> Optional[Event]: + """Most recent event of a type - used to prime a newly-opened socket.""" + return self._last.get(event_type) + + +_bus: Optional[EventBus] = None + + +def get_event_bus() -> EventBus: + global _bus + if _bus is None: + _bus = EventBus() + return _bus diff --git a/backend/core/logging.py b/backend/core/logging.py new file mode 100644 index 0000000..fe569e5 --- /dev/null +++ b/backend/core/logging.py @@ -0,0 +1,43 @@ +"""Console logging setup.""" + +from __future__ import annotations + +import logging +import sys + +_FORMAT = "%(asctime)s %(levelname)-7s %(name)-28s %(message)s" +_DATEFMT = "%H:%M:%S" + + +def ensure_utf8_stdout() -> None: + """Make stdout/stderr able to print non-ASCII on a Windows console. + + Windows terminals default to a legacy code page (cp1252 in Western locales), + and printing a single Chinese character raises UnicodeEncodeError. The robot's + own documentation and error strings are Chinese, so an error response would + otherwise crash the log call that was trying to report it - turning a handled + failure into an unhandled one. + """ + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") # type: ignore[union-attr] + except Exception: # pragma: no cover - never let logging setup fail + pass + + +def configure_logging(level: str = "info") -> None: + ensure_utf8_stdout() + numeric = getattr(logging, level.upper(), logging.INFO) + handler = logging.StreamHandler(sys.stdout) + handler.setFormatter(logging.Formatter(_FORMAT, datefmt=_DATEFMT)) + + root = logging.getLogger() + root.handlers = [handler] + root.setLevel(numeric) + + # Access logs for a single-user local dashboard are noise. + logging.getLogger("uvicorn.access").setLevel(logging.WARNING) + logging.getLogger("httpx").setLevel(logging.WARNING) + logging.getLogger("httpcore").setLevel(logging.WARNING) + logging.getLogger("websockets").setLevel(logging.WARNING) diff --git a/backend/core/pronunciation.py b/backend/core/pronunciation.py new file mode 100644 index 0000000..58b2e61 --- /dev/null +++ b/backend/core/pronunciation.py @@ -0,0 +1,112 @@ +"""Pronunciation fixes for the simulator's speech. + +The built-in OS voices are clear and instant, but they mispronounce brand names +and initialisms - "AGIBOT" is the obvious one. Rather than abandoning a voice +that otherwise sounds good, we respell the problem words phonetically just +before they are spoken. + +The text the *robot* receives is never touched. This rewrite happens at the very +last step, inside the PC audio layer, so: + + * the dashboard, history and API all keep the operator's original wording; + * the real AGIBOT A3 always gets the real text - it has its own TTS and its + own idea of how its name sounds. + +Rules live in `pronunciation.json` at the project root so they can be corrected +without touching code: + + { + "AGIBOT": "Ah-jee-bot", + "A3": "A three" + } + +Matching is case-insensitive and respects word boundaries, so "AGIBOT" will not +corrupt a longer word that happens to contain it. +""" + +from __future__ import annotations + +import json +import logging +import re +from pathlib import Path +from typing import Dict, List, Optional, Tuple + +logger = logging.getLogger(__name__) + +RULES_FILE = "pronunciation.json" + +#: Shipped defaults. Deliberately small - guessing at pronunciations nobody +#: asked for is worse than leaving a word alone. Extend via pronunciation.json. +DEFAULT_RULES: Dict[str, str] = { + "AGIBOT": "Ah-jee-bot", + "AgiBot": "Ah-jee-bot", +} + + +class Pronouncer: + """Applies respelling rules to text just before synthesis.""" + + def __init__(self, rules: Optional[Dict[str, str]] = None) -> None: + self._compiled: List[Tuple[re.Pattern, str]] = [] + self.rules: Dict[str, str] = {} + self.load(rules if rules is not None else DEFAULT_RULES) + + def load(self, rules: Dict[str, str]) -> None: + self.rules = dict(rules or {}) + self._compiled = [] + # Longest first, so a specific phrase wins over a word inside it. + for phrase in sorted(self.rules, key=len, reverse=True): + replacement = self.rules[phrase] + if not phrase: + continue + # \b does not anchor against digits/symbols the way we need for + # names like "A3", so use explicit look-around on word characters. + pattern = re.compile( + r"(? str: + if not text or not self._compiled: + return text + result = text + for pattern, replacement in self._compiled: + result = pattern.sub(replacement, result) + return result + + def __len__(self) -> int: + return len(self.rules) + + +def load_rules(project_root: Path) -> Dict[str, str]: + """Read pronunciation.json, falling back to the shipped defaults.""" + path = Path(project_root) / RULES_FILE + if not path.exists(): + return dict(DEFAULT_RULES) + try: + data = json.loads(path.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError) as exc: + logger.warning("%s is not valid JSON (%s); using default pronunciations", path, exc) + return dict(DEFAULT_RULES) + + if not isinstance(data, dict): + logger.warning("%s must contain a JSON object of {word: respelling}", path) + return dict(DEFAULT_RULES) + + # Keys starting with "_" are comments, so the file can document itself. + rules = { + str(k): str(v) + for k, v in data.items() + if str(k).strip() and not str(k).startswith("_") + } + logger.info("loaded %d pronunciation rule(s) from %s", len(rules), path.name) + return rules + + +def build(project_root: Path, enabled: bool = True) -> Optional[Pronouncer]: + """A Pronouncer, or None when the feature is switched off.""" + if not enabled: + return None + return Pronouncer(load_rules(project_root)) diff --git a/backend/core/text.py b/backend/core/text.py new file mode 100644 index 0000000..86ec978 --- /dev/null +++ b/backend/core/text.py @@ -0,0 +1,39 @@ +"""Text helpers shared by the adapters.""" + +from __future__ import annotations + +import re + +_WHITESPACE = re.compile(r"\s+") + + +def normalise(text: str) -> str: + """Collapse whitespace and trim. Keeps the robot from reading stray newlines.""" + return _WHITESPACE.sub(" ", text or "").strip() + + +def is_cjk(char: str) -> bool: + code = ord(char) + return ( + 0x4E00 <= code <= 0x9FFF # CJK unified ideographs + or 0x3400 <= code <= 0x4DBF # extension A + or 0x3040 <= code <= 0x30FF # kana + or 0xAC00 <= code <= 0xD7AF # hangul + ) + + +def estimate_speech_seconds(text: str, words_per_minute: int = 150) -> float: + """Rough spoken duration. + + Used only when the robot does not report utterance completion itself, so the + dashboard can still return to "Ready" at a believable moment instead of + hanging on "Speaking..." forever. + """ + cleaned = normalise(text) + if not cleaned: + return 0.0 + cjk_chars = sum(1 for ch in cleaned if is_cjk(ch)) + latin_words = len([w for w in cleaned.split() if any(not is_cjk(c) for c in w)]) + # ~2.5 CJK characters per second is a typical TTS cadence. + seconds = (latin_words / max(60, words_per_minute)) * 60.0 + (cjk_chars / 2.5) + return max(0.8, seconds) diff --git a/backend/main.py b/backend/main.py new file mode 100644 index 0000000..ae92ba5 --- /dev/null +++ b/backend/main.py @@ -0,0 +1,175 @@ +"""Application entry point. + + python backend/main.py -> http://localhost:8000 + +Serves the dashboard, the JSON API and the WebSocket from one process, so there +is nothing to orchestrate on the demo machine. +""" + +from __future__ import annotations + +import logging +import os +import sys +from contextlib import asynccontextmanager +from pathlib import Path + +# Allow `python backend/main.py` as well as `python -m backend.main`. +if __package__ in (None, ""): # pragma: no cover + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from fastapi import FastAPI, Request # noqa: E402 +from fastapi.middleware.cors import CORSMiddleware # noqa: E402 +from fastapi.responses import JSONResponse # noqa: E402 +from fastapi.staticfiles import StaticFiles # noqa: E402 + +from backend.api.routes import router as api_router # noqa: E402 +from backend.api.websocket import ws_router # noqa: E402 +from backend.config.settings import PROJECT_ROOT, get_settings # noqa: E402 +from backend.core.events import get_event_bus # noqa: E402 +from backend.core.logging import configure_logging # noqa: E402 +from backend.robot.base import ( # noqa: E402 + RobotBusy, + RobotError, + RobotNotConfigured, + RobotTimeout, + RobotUnreachable, + SpeechFailed, + TransportNotAvailable, +) +from backend.robot.manager import RobotManager # noqa: E402 +from backend.services.speech_service import SpeechService, ValidationError # noqa: E402 + +logger = logging.getLogger("agibot.app") + +FRONTEND_DIR = PROJECT_ROOT / "frontend" + +# HTTP status for each robot-layer failure. Anything unmapped becomes 500. +STATUS_FOR_ERROR = { + ValidationError: 400, + RobotBusy: 409, + TransportNotAvailable: 501, + SpeechFailed: 502, + RobotNotConfigured: 503, + RobotUnreachable: 503, + RobotTimeout: 504, +} + + +def _status_for(exc: RobotError) -> int: + for error_type, status in STATUS_FOR_ERROR.items(): + if isinstance(exc, error_type): + return status + return 500 + + +@asynccontextmanager +async def lifespan(app: FastAPI): + settings = app.state.settings + manager: RobotManager = app.state.robot_manager + + banner(settings) + await manager.start() + try: + yield + finally: + await app.state.speech_service.shutdown() + await manager.stop() + logger.info("shutdown complete") + + +def create_app() -> FastAPI: + settings = get_settings() + configure_logging(settings.server.log_level) + + app = FastAPI( + title="AGIBOT A3 Voice Control", + description="Local dashboard for sending speech to an AGIBOT A3 humanoid.", + version="1.0.0", + lifespan=lifespan, + docs_url="/api/docs", + openapi_url="/api/openapi.json", + ) + + bus = get_event_bus() + manager = RobotManager(settings, bus) + app.state.settings = settings + app.state.event_bus = bus + app.state.robot_manager = manager + app.state.speech_service = SpeechService(settings, manager, bus) + + if settings.server.cors_origins: + app.add_middleware( + CORSMiddleware, + allow_origins=settings.server.cors_origins, + allow_methods=["*"], + allow_headers=["*"], + ) + + @app.exception_handler(RobotError) + async def robot_error_handler(request: Request, exc: RobotError) -> JSONResponse: + status = _status_for(exc) + if status >= 500: + logger.warning("%s -> %s (%s)", request.url.path, exc.code, exc) + return JSONResponse( + status_code=status, + content={"success": False, "error": exc.user_message, "errorCode": exc.code}, + ) + + app.include_router(api_router) + app.include_router(ws_router) + + if FRONTEND_DIR.is_dir(): + # Mounted last so /api/* and /ws win. + app.mount("/", StaticFiles(directory=str(FRONTEND_DIR), html=True), name="frontend") + else: # pragma: no cover + logger.error("frontend directory missing: %s", FRONTEND_DIR) + + return app + + +def banner(settings) -> None: + mode = settings.robot.mode.upper() + target = "local simulation" if settings.is_mock else "{0} via {1}".format( + settings.robot.address, settings.a3.transport + ) + url = "http://{0}:{1}".format( + "localhost" if settings.server.host in ("0.0.0.0", "127.0.0.1") else settings.server.host, + settings.server.port, + ) + lines = [ + "", + " AGIBOT A3 - Voice Control", + " " + "-" * 46, + " Mode : {0}".format(mode), + " Robot : {0}".format(target), + " Dashboard : {0}".format(url), + " Config : {0}".format(settings.env_file), + ] + for issue in settings.issues: + lines.append(" {0:<10}: [{1}] {2}".format("Config", issue.level.upper(), issue.message)) + if settings.is_mock: + lines.append(" Note : running against the MOCK robot - no hardware needed.") + lines.append("") + print("\n".join(lines)) + + +app = create_app() + + +def main() -> None: + import uvicorn + + settings = get_settings() + uvicorn.run( + "backend.main:app", + host=settings.server.host, + port=settings.server.port, + reload=os.environ.get("DEV_RELOAD", "").lower() in ("1", "true", "yes"), + log_level=settings.server.log_level, + access_log=False, + ) + + +if __name__ == "__main__": + main() diff --git a/backend/robot/__init__.py b/backend/robot/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/robot/agibot_a3.py b/backend/robot/agibot_a3.py new file mode 100644 index 0000000..843e318 --- /dev/null +++ b/backend/robot/agibot_a3.py @@ -0,0 +1,299 @@ +"""AGIBOT A3 adapter. + +============================================================================ +THIS FILE IS THE INTEGRATION POINT. +============================================================================ +Everything else in this project is finished and testable today. When the robot's +speech interface is confirmed, the change is limited to: + + 1. .env - transport, address, endpoint/topic/command and payload + 2. (only if the wire format is unusual) a new file under robot/transports/ + +No endpoint, port, ROS topic or SDK symbol is invented in this code. The adapter +deliberately holds *no* opinion about the A3's protocol; it holds the parts that +are true regardless of protocol: + + - one persistent connection, reused across utterances (latency) + - ack-latency measurement, so "how fast is it really" is answerable + - the SENDING -> PROCESSING -> SPEAKING -> COMPLETED lifecycle the UI renders + - completion by robot event when the robot reports it, by estimate when it does not + - interruption, timeouts and error mapping that never wedge the web app + +See docs/AGIBOT_A3_INTEGRATION.md for the discovery procedure to run against the +real unit, and for the exact questions to put to AgiBot support. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any, Dict, Optional + +from ..config.settings import Settings +from ..core.text import estimate_speech_seconds +from .base import ( + ProgressCallback, + RobotAdapter, + RobotCapabilities, + RobotError, + RobotInfo, + RobotNotConfigured, + RobotUnreachable, + SpeechProgress, + SpeechRequest, + SpeechResult, + SpeechStage, +) +from .transports import create_transport +from .transports.base import SpeechTransport + +logger = logging.getLogger(__name__) + +_TICK = 0.05 + + +class AgibotA3(RobotAdapter): + """Real-robot adapter, delegating the wire protocol to a configured transport.""" + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._robot = settings.robot + self._a3 = settings.a3 + self._transport: Optional[SpeechTransport] = None + self._connected = False + self._speaking = False + self._cancel = asyncio.Event() + self._last_error: Optional[str] = None + + # -- identity ------------------------------------------------------------ # + @property + def info(self) -> RobotInfo: + transport = self._transport + return RobotInfo( + name=self._robot.name, + model=self._robot.model, + mode="real", + transport=self._a3.transport, + address=self._robot.address, + capabilities=RobotCapabilities( + native_tts=True, + stop=self._can_stop(), + progress_events=True, + reports_completion=bool(transport and transport.reports_completion), + voice_selection=bool(self._a3.voice), + volume_control=self._a3.volume is not None, + ), + ) + + def _can_stop(self) -> bool: + """Only advertise Stop when a stop path is actually configured.""" + transport = self._a3.transport + if transport == "aimdk": + return bool(self._a3.aimdk_stop_method) + if transport == "http": + return bool(self._a3.http_stop_path) + if transport == "ws": + return bool(self._a3.ws_stop_payload) + if transport == "ros2": + return bool(self._a3.ros_stop_topic) + if transport == "ssh": + return bool(self._a3.ssh_stop_command) + return False + + @property + def is_connected(self) -> bool: + return self._connected + + @property + def is_speaking(self) -> bool: + return self._speaking + + @property + def last_error(self) -> Optional[str]: + return self._last_error + + # -- lifecycle ----------------------------------------------------------- # + async def connect(self) -> None: + if not self._robot.ip: + raise RobotNotConfigured() + if self._settings.has_blocking_issue: + first = next(i for i in self._settings.issues if i.level == "error") + raise RobotError(first.message, user_message=first.message) + + if self._transport is None: + self._transport = create_transport(self._settings) + + try: + await self._transport.open() + ok = await self._transport.probe() + except RobotError: + self._connected = False + raise + except Exception as exc: + self._connected = False + raise RobotUnreachable( + "transport {0} failed to open: {1}".format(self._a3.transport, exc) + ) from exc + + if not ok: + self._connected = False + raise RobotUnreachable( + "probe against {0} failed".format(self._robot.address) + ) + + self._connected = True + self._last_error = None + logger.info( + "AGIBOT A3 connected via %s at %s", self._a3.transport, self._robot.address + ) + + async def disconnect(self) -> None: + self._cancel.set() + self._connected = False + self._speaking = False + if self._transport is not None: + try: + await self._transport.close() + except Exception: # pragma: no cover + logger.debug("transport close raised", exc_info=True) + + async def health_check(self) -> bool: + if self._transport is None: + return False + try: + ok = await self._transport.probe() + except Exception as exc: + logger.debug("health check raised: %s", exc) + ok = False + self._connected = ok + return ok + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult: + if self._transport is None or not self._connected: + raise RobotUnreachable("robot is not connected") + + self._cancel.clear() + self._speaking = True + started = time.perf_counter() + + try: + ack = await self._transport.speak(request, on_progress) + ack_ms = ack.ack_latency_ms + + await on_progress( + SpeechProgress( + request.id, + SpeechStage.PROCESSING, + "Robot accepted the request", + detail=ack.detail, + elapsed_ms=ack_ms, + ) + ) + + estimated = estimate_speech_seconds(request.text) + await on_progress( + SpeechProgress( + request.id, + SpeechStage.SPEAKING, + "Speaking...", + detail=None if self._reports_completion else "~{0:.1f}s".format(estimated), + elapsed_ms=int((time.perf_counter() - started) * 1000), + ) + ) + + cancelled = await self._await_end(estimated) + total_ms = int((time.perf_counter() - started) * 1000) + + if cancelled: + await on_progress( + SpeechProgress(request.id, SpeechStage.CANCELLED, "Speech stopped.") + ) + return SpeechResult( + request_id=request.id, + success=False, + stage=SpeechStage.CANCELLED, + ack_latency_ms=ack_ms, + total_ms=total_ms, + error_code="cancelled", + error="Stopped by operator.", + ) + + await on_progress( + SpeechProgress(request.id, SpeechStage.COMPLETED, "Ready", elapsed_ms=total_ms) + ) + return SpeechResult( + request_id=request.id, + success=True, + stage=SpeechStage.COMPLETED, + ack_latency_ms=ack_ms, + total_ms=total_ms, + ) + except RobotError as exc: + self._last_error = str(exc) + raise + finally: + self._speaking = False + + @property + def _reports_completion(self) -> bool: + return bool(self._transport and self._transport.reports_completion) + + async def _await_end(self, estimated_seconds: float) -> bool: + """Wait for the utterance to finish. Returns True if it was cancelled. + + If the transport delivers a real completion event we use it; otherwise we + fall back to the estimated duration so the UI still returns to Ready. + """ + transport = self._transport + if self._reports_completion and hasattr(transport, "await_completion"): + budget = max(estimated_seconds * 3, 10.0) + done_task = asyncio.ensure_future(transport.await_completion(budget)) # type: ignore[union-attr] + cancel_task = asyncio.ensure_future(self._cancel.wait()) + try: + await asyncio.wait( + {done_task, cancel_task}, return_when=asyncio.FIRST_COMPLETED + ) + return cancel_task.done() and not done_task.done() + finally: + for task in (done_task, cancel_task): + if not task.done(): + task.cancel() + + if self._reports_completion: + # e.g. SSH: transport.speak() already blocked until playback ended. + return self._cancel.is_set() + + deadline = time.perf_counter() + estimated_seconds + while True: + remaining = deadline - time.perf_counter() + if remaining <= 0: + return False + if self._cancel.is_set(): + return True + await asyncio.sleep(min(_TICK, remaining)) + + async def stop_speaking(self) -> bool: + self._cancel.set() + if self._transport is None: + return False + try: + return await self._transport.stop() + except Exception as exc: + logger.warning("stop failed: %s", exc) + return False + + async def describe(self) -> Dict[str, Any]: + detail: Dict[str, Any] = { + "transport": self._a3.transport, + "address": self._robot.address, + "connected": self._connected, + "lastError": self._last_error, + } + if self._transport is not None: + try: + detail["transportDetail"] = await self._transport.describe() + except Exception: # pragma: no cover + pass + return detail diff --git a/backend/robot/base.py b/backend/robot/base.py new file mode 100644 index 0000000..5b2e663 --- /dev/null +++ b/backend/robot/base.py @@ -0,0 +1,242 @@ +"""Robot abstraction layer. + +This is the ONLY contract the rest of the application knows about. Nothing above +this layer (services, API, frontend) may import a vendor SDK, an HTTP client or a +ROS package. Adding a new robot means adding one file that implements +`RobotAdapter` and registering it in `factory.py`. +""" + +from __future__ import annotations + +import abc +import time +import uuid +from dataclasses import dataclass, field +from enum import Enum +from typing import Any, Awaitable, Callable, Dict, Optional + + +# --------------------------------------------------------------------------- # +# enums +# --------------------------------------------------------------------------- # +class RobotState(str, Enum): + DISCONNECTED = "disconnected" + CONNECTING = "connecting" + CONNECTED = "connected" + ERROR = "error" + + +class SpeechStage(str, Enum): + """Lifecycle of a single utterance, mirrored 1:1 in the UI.""" + + QUEUED = "queued" + SENDING = "sending" # request leaving the PC + PROCESSING = "processing" # robot accepted it, synthesising + SPEAKING = "speaking" # audio is coming out of the speaker + COMPLETED = "completed" + CANCELLED = "cancelled" + FAILED = "failed" + + +TERMINAL_STAGES = {SpeechStage.COMPLETED, SpeechStage.CANCELLED, SpeechStage.FAILED} + + +# --------------------------------------------------------------------------- # +# errors +# --------------------------------------------------------------------------- # +class RobotError(Exception): + """Base class for every robot-layer failure. + + `user_message` is safe to show in the browser; `str(exc)` may contain detail + useful in the server log but noisy for an operator during a live demo. + """ + + code = "robot_error" + user_message = "The robot reported an error." + + def __init__(self, message: str = "", user_message: Optional[str] = None) -> None: + super().__init__(message or self.user_message) + if user_message: + self.user_message = user_message + + +class RobotNotConfigured(RobotError): + code = "not_configured" + user_message = "Robot is not configured. Set ROBOT_IP in the .env file." + + +class RobotUnreachable(RobotError): + code = "unreachable" + user_message = "Robot is offline. Check the robot IP address and network connection." + + +class RobotTimeout(RobotError): + code = "timeout" + user_message = "The robot did not respond in time." + + +class RobotBusy(RobotError): + code = "busy" + user_message = "The robot is already speaking." + + +class SpeechFailed(RobotError): + code = "speech_failed" + user_message = "Speech request failed." + + +class TransportNotAvailable(RobotError): + code = "transport_unavailable" + user_message = "The selected robot transport is not available on this PC." + + +# --------------------------------------------------------------------------- # +# value objects +# --------------------------------------------------------------------------- # +@dataclass(frozen=True) +class RobotCapabilities: + """What this adapter can actually do, so the UI never offers a dead control.""" + + native_tts: bool = True + stop: bool = True + progress_events: bool = True + reports_completion: bool = True + voice_selection: bool = False + volume_control: bool = False + + +@dataclass(frozen=True) +class RobotInfo: + name: str + model: str + mode: str + transport: str + address: str + capabilities: RobotCapabilities + + def to_dict(self) -> Dict[str, Any]: + return { + "name": self.name, + "model": self.model, + "mode": self.mode, + "transport": self.transport, + "address": self.address, + "capabilities": { + "nativeTts": self.capabilities.native_tts, + "stop": self.capabilities.stop, + "progressEvents": self.capabilities.progress_events, + "reportsCompletion": self.capabilities.reports_completion, + "voiceSelection": self.capabilities.voice_selection, + "volumeControl": self.capabilities.volume_control, + }, + } + + +@dataclass +class SpeechRequest: + text: str + id: str = field(default_factory=lambda: uuid.uuid4().hex[:12]) + voice: Optional[str] = None + language: Optional[str] = None + created_at: float = field(default_factory=time.time) + + @property + def preview(self) -> str: + text = " ".join(self.text.split()) + return text if len(text) <= 80 else text[:77] + "..." + + +@dataclass +class SpeechProgress: + """One lifecycle update for an utterance, pushed straight to the browser.""" + + request_id: str + stage: SpeechStage + message: str = "" + detail: Optional[str] = None + error_code: Optional[str] = None + elapsed_ms: Optional[int] = None + at: float = field(default_factory=time.time) + + def to_dict(self) -> Dict[str, Any]: + return { + "requestId": self.request_id, + "stage": self.stage.value, + "message": self.message, + "detail": self.detail, + "errorCode": self.error_code, + "elapsedMs": self.elapsed_ms, + "at": self.at, + } + + +@dataclass +class SpeechResult: + request_id: str + success: bool + stage: SpeechStage + # Time from "user pressed Speak" to the robot acknowledging the request. This + # is the number that matters for a live demo - it is the perceived delay. + ack_latency_ms: Optional[int] = None + total_ms: Optional[int] = None + error_code: Optional[str] = None + error: Optional[str] = None + + def to_dict(self) -> Dict[str, Any]: + return { + "requestId": self.request_id, + "success": self.success, + "stage": self.stage.value, + "ackLatencyMs": self.ack_latency_ms, + "totalMs": self.total_ms, + "errorCode": self.error_code, + "error": self.error, + } + + +ProgressCallback = Callable[[SpeechProgress], Awaitable[None]] + + +# --------------------------------------------------------------------------- # +# adapter contract +# --------------------------------------------------------------------------- # +class RobotAdapter(abc.ABC): + """Every robot backend implements exactly this.""" + + @property + @abc.abstractmethod + def info(self) -> RobotInfo: + """Static description of the robot / connection.""" + + @abc.abstractmethod + async def connect(self) -> None: + """Establish (or verify) the connection. Raises RobotError on failure.""" + + @abc.abstractmethod + async def disconnect(self) -> None: + """Release sockets/clients. Must never raise.""" + + @abc.abstractmethod + async def health_check(self) -> bool: + """Cheap liveness probe used by the connection manager.""" + + @abc.abstractmethod + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult: + """Send text to the robot and report progress until the utterance ends.""" + + @abc.abstractmethod + async def stop_speaking(self) -> bool: + """Interrupt the current utterance. Returns True if a stop was issued.""" + + @property + @abc.abstractmethod + def is_connected(self) -> bool: + """Last known connection state - must not perform I/O.""" + + @property + def is_speaking(self) -> bool: + return False + + async def describe(self) -> Dict[str, Any]: + """Optional richer diagnostics for the /api/robot/diagnostics endpoint.""" + return {} diff --git a/backend/robot/factory.py b/backend/robot/factory.py new file mode 100644 index 0000000..22f4f35 --- /dev/null +++ b/backend/robot/factory.py @@ -0,0 +1,46 @@ +"""Robot adapter registry. + +Adding a robot (AgiBot X2, another vendor, a different TTS backend) means writing +one adapter and adding one line here. Nothing above this layer changes. +""" + +from __future__ import annotations + +from typing import Callable, Dict + +from ..config.settings import Settings +from .base import RobotAdapter, RobotError + + +def _mock(settings: Settings) -> RobotAdapter: + from .mock_robot import MockRobot + + return MockRobot(settings) + + +def _agibot_a3(settings: Settings) -> RobotAdapter: + from .agibot_a3 import AgibotA3 + + return AgibotA3(settings) + + +_REGISTRY: Dict[str, Callable[[Settings], RobotAdapter]] = { + "mock": _mock, + "real": _agibot_a3, +} + + +def available_modes() -> list: + return sorted(_REGISTRY) + + +def create_robot(settings: Settings) -> RobotAdapter: + builder = _REGISTRY.get(settings.robot.mode) + if builder is None: + raise RobotError( + "unknown ROBOT_MODE '{0}'".format(settings.robot.mode), + user_message="Unknown ROBOT_MODE '{0}'. Valid values: {1}.".format( + settings.robot.mode, ", ".join(available_modes()) + ), + ) + return builder(settings) diff --git a/backend/robot/gemini_voice.py b/backend/robot/gemini_voice.py new file mode 100644 index 0000000..3296527 --- /dev/null +++ b/backend/robot/gemini_voice.py @@ -0,0 +1,433 @@ +"""Gemini cloud voice for the simulator. + +Gives the mock robot a natural neural voice instead of the flat built-in Windows +one, so a rehearsal sounds like the real thing. + +**Simulator only.** The real AGIBOT A3 synthesises its own speech on-board; +nothing here is ever used against the robot. + +API (verified against a live call on 2026-09-02, not copied from a doc summary): + + POST https://generativelanguage.googleapis.com/v1beta/interactions + x-goog-api-key: + {"model": "...-tts-preview", "input": "...", + "response_format": {"type": "audio"}, + "generation_config": {"speech_config": [{"voice": "Puck"}]}} + + -> steps[0].content[0].data base64 PCM + steps[0].content[0].mime_type audio/l16; rate=24000; channels=1 + +The audio lives under `steps[]`, not the `output_audio` field some docs describe +- that path was checked against a real response. + +THE LATENCY PROBLEM, AND THE TWO THINGS DONE ABOUT IT +----------------------------------------------------- +Measured on this account: ~4 s to synthesise a short sentence, ~8-10 s for a long +paragraph. Unusable for a live demo if taken naively. So: + +1. **Saved audio.** Every line is written to `audio_library/` as an ordinary + .wav named after its text, so a repeated line replays instantly, survives + restarts, works with no internet, and can be played outside this app. + Rehearsed lines can be built ahead of time - see scripts/warm_voice.py. +2. **One clip per utterance.** Text is NOT split by default: the whole line is + synthesised in a single take, so the delivery is continuous and the library + holds one file per thing you said. Optional sentence pipelining is available + for very long text via GEMINI_CHUNK_CHARS, at the cost of an audible seam. + +And it must never break the demo: any failure - no network, bad key, quota - +falls back to the built-in system voice rather than producing silence. +""" + +from __future__ import annotations + +import base64 +import json +import logging +import os +import platform +import re +import struct +import subprocess +import threading +import time +import urllib.error +import urllib.request +from pathlib import Path +from typing import Dict, List, Optional, Tuple + +from ..services.audio_library import AudioLibrary +from .local_audio import LocalVoice + +logger = logging.getLogger(__name__) + +ENDPOINT = "https://generativelanguage.googleapis.com/v1beta/interactions" + +#: Prebuilt voices with the tone Google documents for each. +VOICES: Dict[str, str] = { + "Zephyr": "Bright", "Puck": "Upbeat", "Charon": "Informative", "Kore": "Firm", + "Fenrir": "Excitable", "Leda": "Youthful", "Orus": "Firm", "Aoede": "Breezy", + "Callirrhoe": "Easy-going", "Autonoe": "Bright", "Enceladus": "Breathy", + "Iapetus": "Clear", "Umbriel": "Easy-going", "Algieba": "Smooth", + "Despina": "Smooth", "Erinome": "Clear", "Algenib": "Gravelly", + "Rasalgethi": "Informative", "Laomedeia": "Upbeat", "Achernar": "Soft", + "Alnilam": "Firm", "Schedar": "Even", "Gacrux": "Mature", + "Pulcherrima": "Forward", "Achird": "Friendly", "Zubenelgenubi": "Casual", + "Vindemiatrix": "Gentle", "Sadachbia": "Lively", "Sadaltager": "Knowledgeable", + "Sulafat": "Warm", +} + +#: 0 = never split. Every utterance is synthesised as ONE clip, in one take. +#: +#: Splitting was an optimisation: a request costs ~3.5 s fixed plus ~45 ms per +#: character, so starting playback after the first sentence reaches audio sooner +#: on long text. But each piece is a separate synthesis - separate delivery, +#: separate file - and that is audible. One continuous take is worth more than +#: a second or two of head start. +#: +#: Set GEMINI_CHUNK_CHARS in .env to a character count to re-enable pipelining +#: for very long text. +#: +#: NOTE: unrelated to the robot's own 1024-BYTE limit in aimdk_transport.py. +#: That one is a hard API constraint; this is a latency choice. +_CHUNK_TARGET_CHARS = 0 + +_SENTENCE_END = re.compile(r"(?<=[.!?。!?;:])\s+|\n+") + + +def wav_from_pcm(pcm: bytes, sample_rate: int = 24000, channels: int = 1, + bits: int = 16) -> bytes: + """Wrap raw PCM in a 44-byte WAV header so ordinary players accept it.""" + byte_rate = sample_rate * channels * bits // 8 + block_align = channels * bits // 8 + return b"".join([ + b"RIFF", struct.pack(" List[str]: + """Split into speakable chunks on sentence boundaries. + + `target <= 0` disables splitting entirely: the whole utterance is synthesised + as ONE clip, in one continuous take. That is the default, because a split + line is synthesised as separate requests - each piece gets its own delivery + and its own file, and the join between them can be audible. + """ + text = " ".join((text or "").split()) + if not text: + return [] + if target <= 0 or len(text) <= target: + return [text] + + pieces = [p.strip() for p in _SENTENCE_END.split(text) if p and p.strip()] + chunks: List[str] = [] + current = "" + for piece in pieces: + if not current: + current = piece + elif len(current) + 1 + len(piece) <= target: + current += " " + piece + else: + chunks.append(current) + current = piece + if current: + chunks.append(current) + return chunks + + +# --------------------------------------------------------------------------- # +# playback +# --------------------------------------------------------------------------- # +class _Player: + """Plays one WAV without blocking; can be stopped mid-playback.""" + + def __init__(self) -> None: + self._windows = platform.system().lower().startswith("win") + self._process: Optional[subprocess.Popen] = None + self._ends_at = 0.0 + self._lock = threading.Lock() + + def play(self, path: str, duration: float) -> None: + if self._windows: + import winsound + + winsound.PlaySound(path, winsound.SND_FILENAME | winsound.SND_ASYNC) + # winsound has no completion callback, but the exact duration is + # known from the PCM length, so the deadline is precise. + with self._lock: + self._ends_at = time.monotonic() + duration + return + + binary = "afplay" if platform.system().lower() == "darwin" else "aplay" + try: + with self._lock: + self._process = subprocess.Popen( + [binary, path], stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, stdin=subprocess.DEVNULL) + except FileNotFoundError: + logger.warning("no audio player found (%s); install it to hear the simulator", binary) + + def stop(self) -> None: + if self._windows: + try: + import winsound + + winsound.PlaySound(None, winsound.SND_PURGE) + except Exception: # pragma: no cover + pass + with self._lock: + self._ends_at = 0.0 + return + with self._lock: + process, self._process = self._process, None + if process is not None and process.poll() is None: + try: + process.kill() + except Exception: # pragma: no cover + pass + + def is_playing(self) -> bool: + if self._windows: + with self._lock: + return time.monotonic() < self._ends_at + with self._lock: + process = self._process + return process is not None and process.poll() is None + + +# --------------------------------------------------------------------------- # +# the voice +# --------------------------------------------------------------------------- # +class GeminiVoice(LocalVoice): + """Neural TTS through the Gemini API, saved to disk, with a local fallback.""" + + name = "gemini" + + def __init__(self, api_key: str, model: str, voice: str, style: str = "", + timeout: float = 60.0, cache_dir: Optional[str] = None, + library: Optional[AudioLibrary] = None, + chunk_chars: int = _CHUNK_TARGET_CHARS, + fallback: Optional[LocalVoice] = None) -> None: + if not api_key: + raise ValueError("GEMINI_API_KEY is empty") + self._key = api_key + self._model = model + self._voice = voice + self._style = style + self._timeout = timeout + # 0 (the default) means never split - one utterance, one clip. + self._chunk_chars = max(0, int(chunk_chars)) + self._fallback = fallback + self._using_fallback = False + self._warned = False + + # Audio is saved as ordinary .wav files, so a line spoken once can be + # replayed instantly, played outside this app, or used offline. + if library is not None: + self.library = library + else: + root = cache_dir or os.path.join( + os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))), + "audio_library", + ) + self.library = AudioLibrary(Path(root)) + + self._player = _Player() + self._sequence_stop = threading.Event() + self._sequence: Optional[threading.Thread] = None + self._active = threading.Event() + + # -- LocalVoice ---------------------------------------------------------- # + def prepare(self, text: str) -> bool: + """Synthesise the first chunk (and warm the rest in the background). + + Returns False when the cloud voice is unusable, so playback falls back + to the system voice. + """ + chunks = split_sentences(text, self._chunk_chars) + if not chunks: + return False + + # Start the later chunks BEFORE synthesising the first one, so they are + # fetched in parallel. Doing it afterwards leaves an audible gap between + # sentences on a cold run, because chunk 2 only starts once chunk 1 is + # already playing and finishes after the audio has run out. + prefetch: Optional[threading.Thread] = None + if len(chunks) > 1: + prefetch = threading.Thread( + target=self._warm_rest, args=(chunks[1:],), + name="gemini-prefetch", daemon=True, + ) + prefetch.start() + + try: + self._ensure(chunks[0]) + except Exception as exc: + self._using_fallback = True + if not self._warned: + self._warned = True + logger.warning( + "Gemini voice unavailable (%s) - falling back to the built-in " + "system voice. The demo continues.", exc) + else: + logger.debug("Gemini synthesis failed: %s", exc) + return False + + self._using_fallback = False + return True + + def start(self, text: str) -> None: + self.stop() + chunks = split_sentences(text, self._chunk_chars) + if not chunks: + return + + if self._using_fallback or not self._is_cached(chunks[0]): + if not self.prepare(text): + if self._fallback is not None: + self._using_fallback = True + self._fallback.start(text) + return + + self._sequence_stop.clear() + self._active.set() + self._sequence = threading.Thread( + target=self._play_sequence, args=(chunks,), + name="gemini-playback", daemon=True, + ) + self._sequence.start() + + def stop(self) -> None: + self._sequence_stop.set() + self._player.stop() + self._active.clear() + if self._fallback is not None: + try: + self._fallback.stop() + except Exception: # pragma: no cover + pass + + def is_speaking(self) -> bool: + if self._using_fallback and self._fallback is not None: + return self._fallback.is_speaking() + return self._active.is_set() + + def close(self) -> None: + self.stop() + if self._fallback is not None: + try: + self._fallback.close() + except Exception: # pragma: no cover + pass + + # -- playback sequencing ------------------------------------------------- # + def _play_sequence(self, chunks: List[str]) -> None: + """Play each chunk in turn, waiting for later ones to finish synthesis.""" + try: + for chunk in chunks: + if self._sequence_stop.is_set(): + return + try: + path, duration = self._ensure(chunk) + except Exception as exc: + logger.warning("Gemini synthesis failed mid-utterance: %s", exc) + return + if self._sequence_stop.is_set(): + return + self._player.play(path, duration) + while self._player.is_playing(): + if self._sequence_stop.is_set(): + self._player.stop() + return + time.sleep(0.03) + finally: + self._active.clear() + + def _warm_rest(self, chunks: List[str]) -> None: + for chunk in chunks: + if self._sequence_stop.is_set(): + return + try: + self._ensure(chunk) + except Exception as exc: # pragma: no cover - retried at play time + logger.debug("prefetch failed: %s", exc) + return + + # -- saved audio + synthesis --------------------------------------------- # + def _is_cached(self, text: str) -> bool: + return self.library.find(text, self._voice, self._model, self._style) is not None + + def _ensure(self, text: str) -> Tuple[str, float]: + """Return (wav path, duration), synthesising only if not already saved.""" + entry = self.library.find(text, self._voice, self._model, self._style) + if entry is None: + pcm, sample_rate, channels = self._synthesise(text) + entry = self.library.save( + text, wav_from_pcm(pcm, sample_rate, channels), + self._voice, self._model, self._style, + ) + return str(self.library.root / entry["file"]), entry["durationSeconds"] + + def _synthesise(self, text: str) -> Tuple[bytes, int, int]: + prompt = "{0} {1}".format(self._style, text).strip() if self._style else text + body = json.dumps({ + "model": self._model, + "input": prompt, + "response_format": {"type": "audio"}, + "generation_config": {"speech_config": [{"voice": self._voice}]}, + }).encode("utf-8") + + request = urllib.request.Request( + ENDPOINT, data=body, + headers={"x-goog-api-key": self._key, "Content-Type": "application/json"}, + ) + started = time.perf_counter() + try: + with urllib.request.urlopen(request, timeout=self._timeout) as response: + payload = json.loads(response.read()) + except urllib.error.HTTPError as exc: + detail = exc.read().decode("utf-8", "replace")[:300] + raise RuntimeError("HTTP {0}: {1}".format(exc.code, detail)) from exc + + pcm, sample_rate, channels = _extract_audio(payload) + logger.info( + "Gemini '%s': %.1fs of audio in %.2fs for %d chars", + self._voice, len(pcm) / float(sample_rate * channels * 2), + time.perf_counter() - started, len(text), + ) + return pcm, sample_rate, channels + + # -- helpers ------------------------------------------------------------- # + def warm(self, text: str) -> int: + """Pre-synthesise every chunk of `text`. Returns how many were fetched.""" + fetched = 0 + for chunk in split_sentences(text, self._chunk_chars): + if not self._is_cached(chunk): + self._ensure(chunk) + fetched += 1 + return fetched + + def cache_stats(self) -> Dict[str, object]: + stats = self.library.stats() + return {"entries": stats["count"], "bytes": stats["bytes"], "dir": stats["dir"]} + + def clear_cache(self) -> None: + self.library.clear() + + + +def _extract_audio(payload: dict) -> Tuple[bytes, int, int]: + """Pull PCM out of an interactions response. + + Walks the steps rather than indexing a fixed path, so an extra step or a + reordered response does not break playback. + """ + for step in payload.get("steps", []) or []: + for part in step.get("content", []) or []: + if part.get("type") != "audio" or not part.get("data"): + continue + pcm = base64.b64decode(part["data"]) + return pcm, int(part.get("sample_rate") or 24000), int(part.get("channels") or 1) + raise RuntimeError("no audio in response (status={0})".format(payload.get("status"))) diff --git a/backend/robot/local_audio.py b/backend/robot/local_audio.py new file mode 100644 index 0000000..4757e97 --- /dev/null +++ b/backend/robot/local_audio.py @@ -0,0 +1,453 @@ +"""PC-speaker playback for the simulator. + +Lets the mock robot actually *say* the text out of this laptop's speakers, so the +whole demo can be rehearsed - wording, pacing, the Stop button - before the A3 is +on the network. + +This is simulation only. It has nothing to do with the real robot: the A3 +synthesises its own speech on-board and no audio ever leaves the PC (see +docs/AGIBOT_A3_INTEGRATION.md). + +Backends, in order of preference: + + Windows SAPI5 through comtypes - in-process, ~20 ms to start, real interrupt + Windows PowerShell System.Speech - fallback, no packages needed at all + macOS `say` + Linux `espeak-ng` / `espeak` / `spd-say` + +Every backend supports three things the simulator needs: start without blocking, +report whether audio is still playing, and **stop immediately**. Without that last +one the Stop button would lie - the UI would say "stopped" while the laptop kept +talking. +""" + +from __future__ import annotations + +import logging +import platform +import queue +import shutil +import subprocess +import threading +from typing import List, Optional + +logger = logging.getLogger(__name__) + +# SAPI ISpVoice::Speak flags +_SVSF_ASYNC = 1 +_SVSF_PURGE_BEFORE_SPEAK = 2 +_SVSF_IS_XML = 8 + + +def _xml_escape(text: str) -> str: + """Escape text going into SAPI/SSML markup. + + Without this, an operator typing `Tom & Jerry` or `a < b` would produce + malformed markup and the utterance would be mangled or dropped. + """ + return ( + text.replace("&", "&") + .replace("<", "<") + .replace(">", ">") + .replace('"', """) + ) + + +class LocalVoice: + """A speech engine available to the simulator.""" + + name = "none" + + def prepare(self, text: str) -> bool: + """Optional: do slow work (e.g. a cloud round trip) before playback. + + Called during the "Processing" stage so that "Speaking" is only + announced once audio can really start. Local engines need nothing. + """ + return True + + def start(self, text: str) -> None: + raise NotImplementedError + + def stop(self) -> None: + raise NotImplementedError + + def is_speaking(self) -> bool: + raise NotImplementedError + + def close(self) -> None: + pass + + +# --------------------------------------------------------------------------- # +# Windows - SAPI5 via COM +# --------------------------------------------------------------------------- # +class SapiVoice(LocalVoice): + """Windows SAPI5. + + COM objects are apartment-bound, so one dedicated thread creates the voice + and is the only thread that ever touches it. Public methods just hand it work + through a queue and read plain flags, which makes them safe to call from the + asyncio loop. + """ + + name = "sapi" + + def __init__(self, voice_hint: Optional[str] = None, rate: int = 0, + volume: int = 100, pitch: int = 0) -> None: + self._voice_hint = voice_hint + self._rate = max(-10, min(10, rate)) + self._volume = max(0, min(100, volume)) + # SAPI pitch: -10..10. Raising it turns an adult male voice into a + # younger-sounding one, which is the closest a stock Windows voice gets + # to a teenage character voice. + self._pitch = max(-10, min(10, pitch)) + + self._commands: "queue.Queue" = queue.Queue() + self._speaking = threading.Event() + self._stop_flag = threading.Event() + self._ready = threading.Event() + self._error: Optional[str] = None + self._closing = False + + self._thread = threading.Thread(target=self._run, name="sapi-voice", daemon=True) + self._thread.start() + self._ready.wait(timeout=8) + if self._error: + raise RuntimeError(self._error) + + # -- public API ---------------------------------------------------------- # + def start(self, text: str) -> None: + self._stop_flag.clear() + # Set the flag here, not on the worker thread: a caller that polls + # is_speaking() immediately must never see "idle" before we begin. + self._speaking.set() + self._commands.put(text) + + def stop(self) -> None: + self._stop_flag.set() + + def is_speaking(self) -> bool: + return self._speaking.is_set() + + def close(self) -> None: + self._closing = True + self._stop_flag.set() + self._commands.put(None) + + # -- worker thread ------------------------------------------------------- # + def _run(self) -> None: + try: + import comtypes + import comtypes.client + except ImportError as exc: # pragma: no cover - checked by the factory + self._error = "comtypes not installed: {0}".format(exc) + self._ready.set() + return + + try: + comtypes.CoInitialize() + except Exception: # pragma: no cover - already initialised is fine + pass + + try: + voice = comtypes.client.CreateObject("SAPI.SpVoice") + self._select_voice(voice) + voice.Rate = self._rate + voice.Volume = self._volume + except Exception as exc: + self._error = "cannot create SAPI voice: {0}".format(exc) + self._ready.set() + return + + self._ready.set() + logger.info("local audio: Windows SAPI ready (rate=%s volume=%s)", self._rate, self._volume) + + while not self._closing: + text = self._commands.get() + if text is None: + break + try: + if self._pitch: + voice.Speak( + '{1}'.format(self._pitch, _xml_escape(text)), + _SVSF_ASYNC | _SVSF_IS_XML, + ) + else: + voice.Speak(text, _SVSF_ASYNC) + # Poll rather than block, so a Stop lands within ~50 ms. + while not voice.WaitUntilDone(50): + if self._stop_flag.is_set(): + voice.Speak("", _SVSF_PURGE_BEFORE_SPEAK) + break + except Exception as exc: # pragma: no cover - device may vanish + logger.warning("local audio playback failed: %s", exc) + finally: + self._speaking.clear() + + try: + import comtypes + + comtypes.CoUninitialize() + except Exception: # pragma: no cover + pass + + def _select_voice(self, voice) -> None: + """Pick the voice whose description matches the configured hint.""" + if not self._voice_hint: + return + hint = self._voice_hint.lower() + try: + for description, token, source in _sapi_tokens(): + if hint in description.lower(): + voice.Voice = token + logger.info("local audio: using voice '%s' (%s)", description, source) + return + logger.warning( + "local audio: no voice matching '%s'; using the system default. " + "Run 'python scripts/voices.py' to see what is installed.", + self._voice_hint, + ) + except Exception as exc: # pragma: no cover + logger.debug("voice selection failed: %s", exc) + + +#: Windows keeps two voice registries. The "OneCore" set is the newer, better +#: sounding one and is invisible to SpVoice.GetVoices(), so look there first. +_ONECORE_KEY = r"HKEY_LOCAL_MACHINE\SOFTWARE\Microsoft\Speech_OneCore\Voices" + + +def _sapi_tokens(): + """[(description, token, source)] for every SAPI voice, best set first. + + Must be called on a thread where COM is initialised. + """ + import comtypes.client + + found = [] + seen = set() + + try: + category = comtypes.client.CreateObject("SAPI.SpObjectTokenCategory") + category.SetId(_ONECORE_KEY, False) + tokens = category.EnumerateTokens() + for index in range(tokens.Count): + token = tokens.Item(index) + description = token.GetDescription() + found.append((description, token, "onecore")) + seen.add(description) + except Exception as exc: # pragma: no cover - older Windows + logger.debug("OneCore voices unavailable: %s", exc) + + try: + voice = comtypes.client.CreateObject("SAPI.SpVoice") + tokens = voice.GetVoices() + for index in range(tokens.Count): + token = tokens.Item(index) + description = token.GetDescription() + if description not in seen: + found.append((description, token, "classic")) + except Exception as exc: # pragma: no cover + logger.debug("classic voices unavailable: %s", exc) + + return found + + +def list_local_voices(): + """[(description, source)] - for `python scripts/voices.py`. Never raises.""" + if not platform.system().lower().startswith("win"): + return [] + try: + import comtypes + + try: + comtypes.CoInitialize() + except Exception: # pragma: no cover + pass + return [(description, source) for description, _, source in _sapi_tokens()] + except Exception: # pragma: no cover + return [] + + +# --------------------------------------------------------------------------- # +# Everything else - an external command we can kill +# --------------------------------------------------------------------------- # +class CommandVoice(LocalVoice): + """Speak by running a command. Stop = kill the process.""" + + def __init__(self, name: str, argv_builder) -> None: + self.name = name + self._argv = argv_builder + self._process: Optional[subprocess.Popen] = None + self._lock = threading.Lock() + + def start(self, text: str) -> None: + self.stop() + argv = self._argv(text) + try: + with self._lock: + self._process = subprocess.Popen( + argv, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + stdin=subprocess.DEVNULL, + ) + except Exception as exc: # pragma: no cover - depends on host + logger.warning("local audio command failed (%s): %s", argv[0], exc) + self._process = None + + def stop(self) -> None: + with self._lock: + process = self._process + self._process = None + if process is not None and process.poll() is None: + try: + process.kill() + except Exception: # pragma: no cover + pass + + def is_speaking(self) -> bool: + with self._lock: + process = self._process + return process is not None and process.poll() is None + + def close(self) -> None: + self.stop() + + +def _powershell_voice(rate: int, volume: int, pitch: int = 0, + voice_hint: Optional[str] = None) -> CommandVoice: + """Windows fallback using System.Speech - present on every Windows install.""" + rate = max(-10, min(10, rate)) + volume = max(0, min(100, volume)) + pitch = max(-10, min(10, pitch)) + + def argv(text: str) -> List[str]: + select = "" + if voice_hint: + # SelectVoiceByHints has no substring form; fall back silently when + # the named voice is absent rather than throwing. + select = ( + "try {{ $s.SelectVoice(($s.GetInstalledVoices() | " + "Where-Object {{ $_.VoiceInfo.Name -like '*{0}*' }} | " + "Select-Object -First 1).VoiceInfo.Name) }} catch {{}};" + ).format(voice_hint.replace("'", "''")) + + if pitch: + # System.Speech exposes pitch only through SSML prosody. + body = ( + "{1}" + ).format(pitch * 5, _xml_escape(text)) + speak = "$s.SpeakSsml('{0}')".format(body.replace("'", "''")) + else: + speak = "$s.Speak('{0}')".format(text.replace("'", "''")) + + script = ( + "Add-Type -AssemblyName System.Speech;" + "$s = New-Object System.Speech.Synthesis.SpeechSynthesizer;" + "{select}$s.Rate = {rate}; $s.Volume = {volume};{speak}" + ).format(select=select, rate=rate, volume=volume, speak=speak) + return ["powershell", "-NoProfile", "-NonInteractive", "-Command", script] + + return CommandVoice("powershell", argv) + + +def _macos_voice(rate: int) -> CommandVoice: + # `say` takes words per minute; map the -10..10 scale onto a sane range. + wpm = max(90, min(320, 180 + rate * 12)) + + def argv(text: str) -> List[str]: + return ["say", "-r", str(wpm), text] + + return CommandVoice("say", argv) + + +def _linux_voice(rate: int, volume: int) -> Optional[CommandVoice]: + if shutil.which("espeak-ng") or shutil.which("espeak"): + binary = "espeak-ng" if shutil.which("espeak-ng") else "espeak" + wpm = max(80, min(320, 175 + rate * 12)) + amplitude = max(0, min(200, int(volume * 2))) + + def argv(text: str) -> List[str]: + return [binary, "-s", str(wpm), "-a", str(amplitude), text] + + return CommandVoice(binary, argv) + + if shutil.which("spd-say"): + def argv(text: str) -> List[str]: + return ["spd-say", "-w", "-r", str(max(-100, min(100, rate * 10))), text] + + return CommandVoice("spd-say", argv) + + return None + + +# --------------------------------------------------------------------------- # +# factory +# --------------------------------------------------------------------------- # +def create_local_voice( + voice_hint: Optional[str] = None, rate: int = 0, volume: int = 100, pitch: int = 0, + pronouncer=None, +) -> Optional[LocalVoice]: + """Best available PC voice, or None if this machine cannot speak. + + Never raises: local audio is a convenience, and losing it must not stop the + simulator from running. + """ + voice = _create(voice_hint, rate, volume, pitch) + if voice is not None and pronouncer is not None: + # OS voices mispronounce brand names; fix the text at the last moment. + return Respelling(voice, pronouncer) + return voice + + +class Respelling(LocalVoice): + """Wraps a voice, respelling text on its way to the synthesiser.""" + + def __init__(self, inner: LocalVoice, pronouncer) -> None: + self._inner = inner + self._pronouncer = pronouncer + self.name = inner.name + + def prepare(self, text: str) -> bool: + return self._inner.prepare(self._pronouncer.apply(text)) + + def start(self, text: str) -> None: + self._inner.start(self._pronouncer.apply(text)) + + def stop(self) -> None: + self._inner.stop() + + def is_speaking(self) -> bool: + return self._inner.is_speaking() + + def close(self) -> None: + self._inner.close() + + +def _create(voice_hint, rate, volume, pitch): + system = platform.system().lower() + + if system.startswith("win"): + try: + return SapiVoice(voice_hint, rate, volume, pitch) + except Exception as exc: + logger.info("SAPI unavailable (%s); falling back to PowerShell", exc) + if shutil.which("powershell"): + return _powershell_voice(rate, volume, pitch, voice_hint) + logger.warning("local audio: no speech engine found on this Windows PC") + return None + + if system == "darwin": + if shutil.which("say"): + return _macos_voice(rate) + logger.warning("local audio: 'say' not found") + return None + + voice = _linux_voice(rate, volume) + if voice is None: + logger.warning( + "local audio: install espeak-ng (apt install espeak-ng) to hear the simulator" + ) + return voice diff --git a/backend/robot/manager.py b/backend/robot/manager.py new file mode 100644 index 0000000..76034ce --- /dev/null +++ b/backend/robot/manager.py @@ -0,0 +1,197 @@ +"""Connection supervisor. + +Owns the adapter's lifecycle so the web app never has to care whether the robot +is up. One background task: + + disconnected -> connecting -> connected -> (health poll) -> disconnected -> ... + +Connection failures are normal, not exceptional: they are published as status +events and retried with capped exponential backoff. The HTTP server keeps serving +the dashboard throughout, which is the whole point of the offline-first design. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any, Dict, Optional + +from ..config.settings import Settings +from ..core.events import EventBus, EventType +from .base import RobotAdapter, RobotError, RobotState +from .factory import create_robot + +logger = logging.getLogger(__name__) + + +class RobotManager: + def __init__(self, settings: Settings, bus: EventBus) -> None: + self._settings = settings + self._bus = bus + self._adapter: RobotAdapter = create_robot(settings) + self._state = RobotState.DISCONNECTED + self._error: Optional[str] = None + self._error_code: Optional[str] = None + self._latency_ms: Optional[float] = None + self._connected_since: Optional[float] = None + self._attempts = 0 + self._task: Optional[asyncio.Task] = None + self._wake = asyncio.Event() + self._closing = False + + # -- accessors ----------------------------------------------------------- # + @property + def adapter(self) -> RobotAdapter: + return self._adapter + + @property + def state(self) -> RobotState: + return self._state + + @property + def is_connected(self) -> bool: + return self._state == RobotState.CONNECTED and self._adapter.is_connected + + def status_dict(self) -> Dict[str, Any]: + info = self._adapter.info + return { + "state": self._state.value, + "connected": self.is_connected, + "speaking": self._adapter.is_speaking, + "robot": info.to_dict(), + "latencyMs": round(self._latency_ms, 1) if self._latency_ms is not None else None, + "uptimeSeconds": ( + round(time.time() - self._connected_since, 1) if self._connected_since else None + ), + "attempts": self._attempts, + "error": self._error, + "errorCode": self._error_code, + "configIssues": [issue.to_dict() for issue in self._settings.issues], + } + + # -- lifecycle ----------------------------------------------------------- # + async def start(self) -> None: + self._closing = False + self._task = asyncio.create_task(self._supervise(), name="robot-supervisor") + self._publish() + + async def stop(self) -> None: + self._closing = True + self._wake.set() + if self._task is not None: + self._task.cancel() + try: + await self._task + except (asyncio.CancelledError, Exception): + pass + self._task = None + try: + await self._adapter.disconnect() + except Exception: # pragma: no cover + logger.debug("adapter disconnect raised", exc_info=True) + self._set_state(RobotState.DISCONNECTED) + + def request_reconnect(self) -> None: + """Ask the supervisor to retry immediately (used by the API).""" + self._wake.set() + + async def rebuild(self, settings: Settings) -> None: + """Swap in a new configuration without restarting the process.""" + await self.stop() + self._settings = settings + self._adapter = create_robot(settings) + self._error = None + self._error_code = None + self._latency_ms = None + self._attempts = 0 + await self.start() + + # -- supervisor ---------------------------------------------------------- # + async def _supervise(self) -> None: + delay = self._settings.robot.reconnect_min_delay + while not self._closing: + try: + if not self._adapter.is_connected: + self._attempts += 1 + self._set_state(RobotState.CONNECTING) + started = time.perf_counter() + await self._adapter.connect() + self._latency_ms = (time.perf_counter() - started) * 1000 + self._connected_since = time.time() + self._error = None + self._error_code = None + delay = self._settings.robot.reconnect_min_delay + self._set_state(RobotState.CONNECTED) + logger.info("robot connected (%.0f ms)", self._latency_ms) + + await self._sleep_or_wake(self._settings.robot.health_interval) + if self._closing: + break + + started = time.perf_counter() + healthy = await self._adapter.health_check() + elapsed = (time.perf_counter() - started) * 1000 + if healthy: + # Smooth the latency reading so the UI badge does not flicker. + self._latency_ms = ( + elapsed if self._latency_ms is None else self._latency_ms * 0.7 + elapsed * 0.3 + ) + if self._state != RobotState.CONNECTED: + self._set_state(RobotState.CONNECTED) + else: + self._publish() # keep latency fresh in the dashboard + else: + logger.warning("robot health check failed - reconnecting") + self._connected_since = None + self._error = "Lost connection to the robot." + self._error_code = "unreachable" + self._set_state(RobotState.DISCONNECTED) + await self._safe_disconnect() + + except asyncio.CancelledError: + raise + except RobotError as exc: + self._on_failure(exc.user_message, exc.code) + delay = await self._backoff(delay) + except Exception as exc: # never let the supervisor die + logger.exception("unexpected supervisor error") + self._on_failure("Unexpected robot error: {0}".format(exc), "internal") + delay = await self._backoff(delay) + + def _on_failure(self, message: str, code: str) -> None: + self._connected_since = None + self._latency_ms = None + self._error = message + self._error_code = code + self._set_state(RobotState.ERROR if code == "not_configured" else RobotState.DISCONNECTED) + + async def _backoff(self, delay: float) -> float: + await self._sleep_or_wake(delay) + return min(delay * 2, self._settings.robot.reconnect_max_delay) + + async def _sleep_or_wake(self, seconds: float) -> None: + """Sleep, but return early if someone calls request_reconnect().""" + try: + await asyncio.wait_for(self._wake.wait(), timeout=max(0.1, seconds)) + except asyncio.TimeoutError: + return + finally: + self._wake.clear() + + async def _safe_disconnect(self) -> None: + try: + await self._adapter.disconnect() + except Exception: # pragma: no cover + logger.debug("disconnect during recovery raised", exc_info=True) + + # -- events -------------------------------------------------------------- # + def _set_state(self, state: RobotState) -> None: + changed = state != self._state + self._state = state + if changed: + logger.info("robot state -> %s", state.value) + self._publish() + + def _publish(self) -> None: + self._bus.publish(EventType.ROBOT_STATUS, self.status_dict()) diff --git a/backend/robot/mock_robot.py b/backend/robot/mock_robot.py new file mode 100644 index 0000000..a20e85b --- /dev/null +++ b/backend/robot/mock_robot.py @@ -0,0 +1,365 @@ +"""Mock AGIBOT A3. + +A stand-in that reproduces the *shape and timing* of a real robot TTS exchange: +a network hop, a synthesis pause, a speaking window proportional to the text, then +completion - all interruptible. The frontend cannot tell the difference, which is +the point: everything except the wire protocol gets tested before the robot lands. + +Optional: set MOCK_LOCAL_AUDIO=true to hear the text through the PC's own +speakers, so the whole demo can be rehearsed before the robot is on the network. +When that is on, the *real* playback drives the timeline - "Completed" appears +exactly when the sound stops, and Stop cuts the audio mid-word. See local_audio.py. +""" + +from __future__ import annotations + +import asyncio +import logging +import random +import time +from typing import Any, Dict, Optional + +from ..config.settings import Settings +from .local_audio import LocalVoice, create_local_voice +from .base import ( + ProgressCallback, + RobotAdapter, + RobotCapabilities, + RobotInfo, + RobotUnreachable, + SpeechFailed, + SpeechProgress, + SpeechRequest, + SpeechResult, + SpeechStage, +) + +logger = logging.getLogger(__name__) + +# Granularity of the simulated speaking window. Small enough that Stop feels +# instant, large enough not to spin the event loop. +_TICK = 0.05 + + +class MockRobot(RobotAdapter): + """Simulated robot. Never touches the network.""" + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._mock = settings.mock + self._connected = False + self._speaking = False + self._cancel = asyncio.Event() + self._health_calls = 0 + self._rng = random.Random(20260902) + self._local_voice: Optional[LocalVoice] = None + self._voice_ready = False + + # -- identity ------------------------------------------------------------ # + @property + def info(self) -> RobotInfo: + return RobotInfo( + name=self._settings.robot.name, + model=self._settings.robot.model + " (simulated)", + mode="mock", + transport="mock", + address="local simulation", + capabilities=RobotCapabilities( + native_tts=True, + stop=True, + progress_events=True, + reports_completion=True, + voice_selection=False, + volume_control=False, + ), + ) + + @property + def is_connected(self) -> bool: + return self._connected + + @property + def is_speaking(self) -> bool: + return self._speaking + + # -- lifecycle ----------------------------------------------------------- # + async def connect(self) -> None: + await asyncio.sleep(self._mock.connect_delay_ms / 1000) + if self._mock.flaky_connection and self._rng.random() < 0.25: + self._connected = False + raise RobotUnreachable("simulated connection failure") + self._connected = True + logger.info("MockRobot connected (simulated)") + + async def disconnect(self) -> None: + self._cancel.set() + self._connected = False + self._speaking = False + if self._local_voice is not None: + try: + self._local_voice.close() + except Exception: # pragma: no cover + pass + self._local_voice = None + self._voice_ready = False + logger.info("MockRobot disconnected (simulated)") + + async def health_check(self) -> bool: + self._health_calls += 1 + # Cost a plausible round trip so the dashboard's latency readout looks + # like a network number rather than a suspicious 0 ms. + await asyncio.sleep(self._mock.network_latency_ms / 2000) + # Simulate an intermittent link so reconnect logic gets exercised offline. + if self._mock.flaky_connection and self._health_calls % 17 == 0: + self._connected = False + return False + return self._connected + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> SpeechResult: + if not self._connected: + raise RobotUnreachable("mock robot is not connected") + + self._cancel.clear() + self._speaking = True + started = time.perf_counter() + ack_ms: Optional[int] = None + + try: + # 1. request leaves the PC + await on_progress( + SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...") + ) + await self._sleep(self._mock.network_latency_ms / 1000) + + if self._mock.failure_rate and self._rng.random() < self._mock.failure_rate: + raise SpeechFailed( + "simulated TTS failure", + user_message="Speech request failed. The robot rejected the utterance.", + ) + + # 2. robot has the text and is synthesising + ack_ms = int((time.perf_counter() - started) * 1000) + await on_progress( + SpeechProgress( + request.id, + SpeechStage.PROCESSING, + "Robot is preparing speech...", + elapsed_ms=ack_ms, + ) + ) + await self._sleep(self._mock.processing_ms / 1000) + + # A cloud voice needs a network round trip. Do it here, while the UI + # still says "preparing speech", so that "Speaking..." is only shown + # once audio can actually start. + voice = self._voice() + if voice is not None and not self._cancel.is_set(): + await asyncio.get_running_loop().run_in_executor( + None, voice.prepare, request.text + ) + + # 3. audio starts + duration = self._estimate_duration(request.text) + await on_progress( + SpeechProgress( + request.id, + SpeechStage.SPEAKING, + "Speaking...", + detail="~{0:.1f}s".format(duration), + elapsed_ms=int((time.perf_counter() - started) * 1000), + ) + ) + + # With PC audio on, the real playback drives the timeline, so the UI + # returns to Ready exactly when the sound stops. Otherwise fall back + # to the estimate. + if self._start_local_audio(request.text): + cancelled = await self._wait_for_audio(duration) + else: + cancelled = await self._sleep(duration) + + if cancelled: + await on_progress( + SpeechProgress(request.id, SpeechStage.CANCELLED, "Speech stopped.") + ) + return SpeechResult( + request_id=request.id, + success=False, + stage=SpeechStage.CANCELLED, + ack_latency_ms=ack_ms, + total_ms=int((time.perf_counter() - started) * 1000), + error_code="cancelled", + error="Stopped by operator.", + ) + + total_ms = int((time.perf_counter() - started) * 1000) + await on_progress( + SpeechProgress(request.id, SpeechStage.COMPLETED, "Ready", elapsed_ms=total_ms) + ) + return SpeechResult( + request_id=request.id, + success=True, + stage=SpeechStage.COMPLETED, + ack_latency_ms=ack_ms, + total_ms=total_ms, + ) + finally: + self._speaking = False + + async def stop_speaking(self) -> bool: + if not self._speaking: + return False + self._cancel.set() + # Cut the audio now rather than waiting for the polling loop, so the + # speakers go quiet the instant Stop is pressed. + if self._local_voice is not None: + try: + self._local_voice.stop() + except Exception: # pragma: no cover + pass + return True + + async def describe(self) -> Dict[str, Any]: + return { + "simulation": { + "connectDelayMs": self._mock.connect_delay_ms, + "networkLatencyMs": self._mock.network_latency_ms, + "processingMs": self._mock.processing_ms, + "wordsPerMinute": self._mock.words_per_minute, + "failureRate": self._mock.failure_rate, + "flakyConnection": self._mock.flaky_connection, + "localAudio": self._mock.local_audio, + "voiceBackend": self._local_voice.name if self._local_voice else None, + "voiceHint": self._mock.voice, + "voicePitch": self._mock.speech_pitch, + }, + "healthChecks": self._health_calls, + } + + # -- internals ----------------------------------------------------------- # + async def _sleep(self, seconds: float) -> bool: + """Sleep, but wake immediately on Stop. Returns True if interrupted.""" + deadline = time.perf_counter() + seconds + while True: + remaining = deadline - time.perf_counter() + if remaining <= 0: + return False + if self._cancel.is_set(): + return True + await asyncio.sleep(min(_TICK, remaining)) + + def _estimate_duration(self, text: str) -> float: + """Approximate how long a human-sounding voice would take to read `text`.""" + words = len(text.split()) + cjk = sum(1 for ch in text if "一" <= ch <= "鿿") + if cjk > words: # Chinese/Japanese text has few whitespace-delimited words + words = max(words, cjk // 2) + wpm = max(60, self._mock.words_per_minute) + return max(0.8, (words / wpm) * 60.0) + + def _start_local_audio(self, text: str) -> bool: + """Begin PC-speaker playback. Returns True if audio actually started.""" + voice = self._voice() + if voice is None: + return False + try: + voice.start(text) + return True + except Exception as exc: # pragma: no cover - depends on host audio + logger.warning("local audio failed, falling back to silent timing: %s", exc) + return False + + def _voice(self) -> Optional[LocalVoice]: + """Create the PC voice on first use; never fail the utterance over it.""" + if not self._mock.local_audio: + return None + if self._voice_ready: + return self._local_voice + self._voice_ready = True + try: + system_voice = create_local_voice( + voice_hint=self._mock.voice, + rate=self._mock.speech_rate, + volume=self._mock.speech_volume, + pitch=self._mock.speech_pitch, + ) + # Respell once, around whichever engine ends up in front, so both + # the cloud voice and the fallback say names the same way. + self._local_voice = self._wrap_pronunciation(self._wrap_cloud(system_voice)) + if self._local_voice is None: + logger.warning( + "MOCK_LOCAL_AUDIO is on but this PC has no usable speech engine" + ) + except Exception as exc: # pragma: no cover + logger.warning("local audio unavailable: %s", exc) + self._local_voice = None + return self._local_voice + + def _wrap_pronunciation(self, voice): + """Fix brand-name pronunciation just before synthesis, for any engine.""" + if voice is None or not self._mock.pronunciation: + return voice + from ..config.settings import PROJECT_ROOT + from ..core.pronunciation import build as build_pronouncer + from .local_audio import Respelling + + pronouncer = build_pronouncer(PROJECT_ROOT, True) + if pronouncer is None or not len(pronouncer): + return voice + return Respelling(voice, pronouncer) + + def _wrap_cloud(self, system_voice): + """Put the Gemini voice in front of the system voice, if configured. + + The system voice stays as the fallback so a network or quota problem + degrades to a robotic voice rather than to silence mid-demo. + """ + if self._mock.voice_engine != "gemini": + return system_voice + if not self._mock.gemini_api_key: + logger.warning( + "MOCK_VOICE_ENGINE=gemini but GEMINI_API_KEY is not set; " + "using the built-in system voice" + ) + return system_voice + try: + from ..services.audio_library import get_audio_library + from .gemini_voice import GeminiVoice + + voice = GeminiVoice( + library=get_audio_library(), + api_key=self._mock.gemini_api_key, + model=self._mock.gemini_model, + voice=self._mock.gemini_voice, + style=self._mock.gemini_style, + chunk_chars=self._mock.gemini_chunk_chars, + fallback=system_voice, + ) + logger.info( + "local audio: Gemini voice '%s' (%s), fallback=%s", + self._mock.gemini_voice, self._mock.gemini_model, + system_voice.name if system_voice else "none", + ) + return voice + except Exception as exc: + logger.warning("Gemini voice unavailable (%s); using the system voice", exc) + return system_voice + + async def _wait_for_audio(self, fallback_seconds: float) -> bool: + """Wait until the speakers go quiet. Returns True if Stop interrupted it.""" + voice = self._local_voice + assert voice is not None + # Never wait forever if a backend misreports its state. + deadline = time.perf_counter() + max(fallback_seconds * 4, 30.0) + while time.perf_counter() < deadline: + if self._cancel.is_set(): + try: + voice.stop() + except Exception: # pragma: no cover + pass + return True + if not voice.is_speaking(): + return False + await asyncio.sleep(_TICK) + return False diff --git a/backend/robot/transports/__init__.py b/backend/robot/transports/__init__.py new file mode 100644 index 0000000..b111198 --- /dev/null +++ b/backend/robot/transports/__init__.py @@ -0,0 +1,67 @@ +"""Transport registry for the AGIBOT A3 adapter.""" + +from __future__ import annotations + +from typing import Callable, Dict + +from ...config.settings import Settings +from ..base import RobotError +from .base import SpeechTransport, TransportAck + +_BUILDERS: Dict[str, Callable[[Settings], SpeechTransport]] = {} + + +def _aimdk(settings: Settings) -> SpeechTransport: + from .aimdk_transport import AimdkTransport + + return AimdkTransport(settings) + + +def _http(settings: Settings) -> SpeechTransport: + from .http_transport import HttpTransport + + return HttpTransport(settings) + + +def _ws(settings: Settings) -> SpeechTransport: + from .ws_transport import WebSocketTransport + + return WebSocketTransport(settings) + + +def _ros2(settings: Settings) -> SpeechTransport: + from .ros2_transport import Ros2Transport + + return Ros2Transport(settings) + + +def _ssh(settings: Settings) -> SpeechTransport: + from .ssh_transport import SshTransport + + return SshTransport(settings) + + +_BUILDERS.update( + {"aimdk": _aimdk, "http": _http, "ws": _ws, "ros2": _ros2, "ssh": _ssh} +) + + +def available_transports() -> list: + return sorted(_BUILDERS) + + +def create_transport(settings: Settings) -> SpeechTransport: + """Instantiate the transport named by A3_TRANSPORT (imported lazily).""" + name = settings.a3.transport + builder = _BUILDERS.get(name) + if builder is None: + raise RobotError( + "unknown transport '{0}'".format(name), + user_message="Unknown A3_TRANSPORT '{0}'. Valid values: {1}.".format( + name, ", ".join(available_transports()) + ), + ) + return builder(settings) + + +__all__ = ["SpeechTransport", "TransportAck", "create_transport", "available_transports"] diff --git a/backend/robot/transports/aimdk_transport.py b/backend/robot/transports/aimdk_transport.py new file mode 100644 index 0000000..e1f3465 --- /dev/null +++ b/backend/robot/transports/aimdk_transport.py @@ -0,0 +1,354 @@ +"""AimDK HTTP JSON-RPC transport - the documented AGIBOT A3 speech path. + +WHAT THIS IMPLEMENTS +-------------------- +AgiBot's A3 developer guide documents an HTTP JSON-RPC interface served by the +robot's HDU (head unit) through their AimRT runtime: + + POST http://:/rpc// + Content-Type: application/json + +and, on that interface, a text-to-speech service: + + POST http://:59301/rpc/aimdk.protocol.TTSService/PlayTTS + {"text": "...", "priority_level": "INTERACTION_L6", "domain": "...", + "trace_id": "...", "is_interrupted": true} + + -> {"trace_id": "_", "is_sucess": true, "error_message": "", ...} + +Docs (public, no login required): + https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play + https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview + +This is the fastest available path: text in, robot's own TTS out. No audio is +generated or transferred by this PC. + +FOUR TRAPS THIS FILE HANDLES DELIBERATELY +----------------------------------------- +1. `is_sucess` - the success flag is spelled with one 'c' in AgiBot's docs. Both + spellings are read, so a corrected firmware does not break us. +2. The **returned** trace_id differs from the one you send (a random suffix is + appended). Stop only works with the returned value, so it is what we keep. +3. A 1024-byte request limit on `text`. Long input is chunked on sentence + boundaries by UTF-8 byte length - a 1000-character Chinese sentence is ~3000 + bytes and would otherwise be rejected. +4. AgiBot's constraints page limits status polling to <= 0.2 Hz and warns that + high-frequency RPC can destabilise the robot. So the liveness probe is a TCP + connect, never an RPC, and completion is estimated rather than polled. + +EVERYTHING IS STILL CONFIGURABLE. Service and method names, the port, the +priority level and the byte limit are all .env keys, because AimRT publishes no +endpoint-discovery route and AgiBot does not guarantee port stability across +firmware. If your unit differs, you edit .env - not this file. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any, Dict, List, Optional, Tuple + +from ...config.settings import Settings +from ..base import ( + ProgressCallback, + RobotTimeout, + RobotUnreachable, + SpeechFailed, + SpeechProgress, + SpeechRequest, + SpeechStage, + TransportNotAvailable, +) +from .base import SpeechTransport, TransportAck, now_ms + +logger = logging.getLogger(__name__) + +try: + import httpx +except ImportError: # pragma: no cover + httpx = None # type: ignore[assignment] + +# Sentence-ish boundaries, most preferred first. Covers CJK punctuation because +# the robot's own examples are Chinese. +_BREAKS = ("\n", "。", "!", "?", ". ", "! ", "? ", ";", "; ", ",", ", ", " ") + +# Values AgiBot's docs treat as "no error" in the RPC envelope. +_OK_CODES = ("", "0", 0, None) + + +class AimdkTransport(SpeechTransport): + """AGIBOT A3 native TTS over AimRT HTTP JSON-RPC.""" + + name = "aimdk" + reports_completion = False # PlayTTS acks acceptance, not end-of-audio + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._robot = settings.robot + self._a3 = settings.a3 + self._client: Optional["httpx.AsyncClient"] = None + #: trace_id the robot gave us for the utterance in flight - Stop needs it + self._active_trace: Optional[str] = None + self._last_response: Any = None + + # -- lifecycle ----------------------------------------------------------- # + async def open(self) -> None: + if httpx is None: # pragma: no cover + raise TransportNotAvailable( + "httpx is not installed", + user_message="This transport requires 'httpx'. Run: pip install -r requirements.txt", + ) + if self._client is not None: + return + + headers = { + "Content-Type": "application/json", # mandatory per AgiBot's docs + "Connection": "keep-alive", + # AimRT honours a whole-second Timeout header; its own default is 5 s. + "Timeout": str(max(1, int(self._robot.request_timeout))), + } + headers.update(self._a3.http_headers or {}) + if self._a3.http_auth_token: + headers.setdefault("Authorization", "Bearer " + self._a3.http_auth_token) + + self._client = httpx.AsyncClient( + base_url=self._robot.base_url, + headers=headers, + timeout=httpx.Timeout(self._robot.request_timeout, connect=self._robot.connect_timeout), + limits=httpx.Limits(max_keepalive_connections=4, max_connections=8), + ) + logger.info( + "AimDK transport ready: %s/rpc/%s/%s", + self._robot.base_url, self._a3.aimdk_service, self._a3.aimdk_play_method, + ) + + async def close(self) -> None: + if self._client is not None: + try: + await self._client.aclose() + except Exception: # pragma: no cover + pass + self._client = None + + # -- health -------------------------------------------------------------- # + async def probe(self) -> bool: + """TCP connect only. + + Deliberately not an RPC call: AgiBot's constraints page caps status RPC + polling at 0.2 Hz, and this runs on the health-check interval. A socket + connect proves the service port is listening without adding RPC load. + """ + if not self._robot.ip: + return False + try: + fut = asyncio.open_connection(self._robot.ip, self._robot.port) + _, writer = await asyncio.wait_for(fut, timeout=self._robot.connect_timeout) + writer.close() + try: + await writer.wait_closed() + except Exception: # pragma: no cover + pass + return True + except Exception as exc: + logger.debug("AimDK probe to %s failed: %s", self._robot.address, exc) + return False + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck: + if self._client is None: + await self.open() + + chunks = chunk_text(request.text, self._a3.aimdk_max_bytes) + if not chunks: + raise SpeechFailed("nothing to say", user_message="Please enter some text.") + + await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...")) + started = time.perf_counter() + ack_ms: Optional[int] = None + detail: Optional[str] = None + + for index, chunk in enumerate(chunks): + payload = { + "text": chunk, + "priority_level": self._a3.aimdk_priority, + "domain": self._a3.aimdk_domain, + "trace_id": "{0}{1}".format(request.id, "" if index == 0 else "-%d" % index), + # Only the first chunk may interrupt whatever was playing; the + # rest must queue behind it or the utterance cuts itself off. + "is_interrupted": self._a3.aimdk_interrupt if index == 0 else False, + } + trace, chunk_detail = await self._call( + self._a3.aimdk_play_method, payload, what="PlayTTS" + ) + if index == 0: + ack_ms = now_ms(started) + detail = chunk_detail + if len(chunks) > 1: + detail = "{0} parts".format(len(chunks)) + if trace: + self._active_trace = trace + + return TransportAck( + accepted=True, + ack_latency_ms=ack_ms if ack_ms is not None else now_ms(started), + detail=detail, + raw=self._last_response, + ) + + async def stop(self) -> bool: + """Stop the utterance using the trace_id the robot returned to us.""" + trace = self._active_trace + if not trace: + logger.info("no active trace_id - nothing to stop") + return False + if self._client is None: + await self.open() + try: + await self._call(self._a3.aimdk_stop_method, {"trace_id": trace}, what="StopTTS") + return True + except Exception as exc: + logger.warning("StopTTS failed: %s", exc) + return False + finally: + self._active_trace = None + + async def describe(self) -> Dict[str, Any]: + return { + "baseUrl": self._robot.base_url, + "service": self._a3.aimdk_service, + "playMethod": self._a3.aimdk_play_method, + "stopMethod": self._a3.aimdk_stop_method, + "priority": self._a3.aimdk_priority, + "domain": self._a3.aimdk_domain, + "maxBytesPerChunk": self._a3.aimdk_max_bytes, + "activeTraceId": self._active_trace, + "lastResponse": self._last_response, + "docs": "https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play", + } + + # -- internals ----------------------------------------------------------- # + def _route(self, method: str) -> str: + # AimRT strips the "pb:/" func-name prefix when building the HTTP route - + # the URL is /rpc// and must never contain "pb:/". + return "/rpc/{0}/{1}".format(self._a3.aimdk_service.lstrip("/"), method) + + async def _call(self, method: str, payload: Dict[str, Any], what: str) -> Tuple[Optional[str], Optional[str]]: + """One RPC. Returns (trace_id, human detail). Raises on any failure.""" + assert self._client is not None + route = self._route(method) + + try: + response = await self._client.post(route, json=payload) + except Exception as exc: + if httpx is not None and isinstance(exc, httpx.TimeoutException): + raise RobotTimeout( + "timeout calling {0}{1}".format(self._robot.base_url, route) + ) from exc + raise RobotUnreachable( + "cannot reach {0}{1}: {2}".format(self._robot.base_url, route, exc) + ) from exc + + # AimRT maps an RPC-level failure to HTTP 500 and an unknown route to 404. + if response.status_code == 404: + raise SpeechFailed( + "route {0} not found (HTTP 404)".format(route), + user_message=( + "The robot does not expose {0} at {1}. Check A3_AIMDK_SERVICE / " + "A3_AIMDK_PLAY_METHOD and the robot's firmware version." + ).format(method, route), + ) + if response.status_code >= 400: + raise SpeechFailed( + "{0} returned HTTP {1}: {2}".format(what, response.status_code, response.text[:300]), + user_message="Speech request failed (robot returned HTTP {0}).".format( + response.status_code + ), + ) + + try: + body = response.json() + except Exception: + body = {"raw": response.text[:300]} + self._last_response = body + + if not isinstance(body, dict): + return None, None + + # Envelope check, when the firmware wraps replies in a header block. + header = body.get("header") + if isinstance(header, dict): + code = header.get("code") + if code not in _OK_CODES: + raise SpeechFailed( + "{0} rejected: code={1} msg={2}".format(what, code, header.get("msg")), + user_message="Speech request failed: {0}".format( + header.get("msg") or "robot returned error code {0}".format(code) + ), + ) + + # AgiBot's documented field is `is_sucess` (one 'c'). Read both spellings + # so a future firmware that fixes the typo keeps working. + success = body.get("is_sucess", body.get("is_success")) + if success is False: + raise SpeechFailed( + "{0} rejected: {1}".format(what, body.get("error_message")), + user_message="Speech request failed: {0}".format( + body.get("error_message") or "the robot rejected the utterance." + ), + ) + + trace = body.get("trace_id") + return (trace if isinstance(trace, str) else None), None + + +# --------------------------------------------------------------------------- # +# chunking +# --------------------------------------------------------------------------- # +def chunk_text(text: str, max_bytes: int) -> List[str]: + """Split `text` into pieces of at most `max_bytes` UTF-8 bytes. + + The limit AgiBot documents is a *byte* limit, so counting characters is not + enough - Chinese text is three bytes per character. Splits prefer sentence + ends, then clause punctuation, then spaces, and finally cut mid-word rather + than emit an over-length chunk. + """ + text = (text or "").strip() + if not text: + return [] + budget = max(64, int(max_bytes) - 2) # small margin for encoding edge cases + if len(text.encode("utf-8")) <= budget: + return [text] + + chunks: List[str] = [] + remaining = text + while remaining: + if len(remaining.encode("utf-8")) <= budget: + chunks.append(remaining.strip()) + break + + window = _prefix_within_bytes(remaining, budget) + split_at = -1 + for marker in _BREAKS: + found = window.rfind(marker) + if found > budget // 4: # avoid pathologically tiny chunks + split_at = found + len(marker) + break + if split_at <= 0: + split_at = len(window) + + piece = remaining[:split_at].strip() + if piece: + chunks.append(piece) + remaining = remaining[split_at:].lstrip() + + return [chunk for chunk in chunks if chunk] + + +def _prefix_within_bytes(text: str, budget: int) -> str: + """Longest prefix of `text` that fits in `budget` UTF-8 bytes.""" + encoded = text.encode("utf-8") + if len(encoded) <= budget: + return text + # Decode back, dropping any partial trailing character. + return encoded[:budget].decode("utf-8", "ignore") diff --git a/backend/robot/transports/base.py b/backend/robot/transports/base.py new file mode 100644 index 0000000..77ac8f4 --- /dev/null +++ b/backend/robot/transports/base.py @@ -0,0 +1,136 @@ +"""Transport contract for the AGIBOT A3 adapter. + +A *transport* knows how to move one "speak this text" command onto the wire and +back. It knows nothing about history, the UI, or the event bus. + +Why the payloads are templates rather than code +----------------------------------------------- +The exact request body an A3 expects is defined by AgiBot's SDK/API reference for +your specific unit and firmware. Inventing one would be worse than useless, so the +shape is supplied as configuration: + + A3_HTTP_SPEAK_PATH=/some/documented/path + A3_HTTP_SPEAK_PAYLOAD={"text": "{text}", "voice": "{voice}"} + +Placeholders available in any template string: + {text} {id} {voice} {language} {volume} {speed} + +A template string that is *exactly* one placeholder keeps the value's native type +(so {volume} becomes a JSON number, not the string "0.8"). Keys whose rendered +value is null are dropped, so unset optional parameters are simply not sent. +""" + +from __future__ import annotations + +import abc +import time +from dataclasses import dataclass +from typing import Any, Dict, Optional +from urllib.parse import quote + +from ..base import ProgressCallback, SpeechRequest + + +@dataclass +class TransportAck: + """What the robot said when it accepted (or rejected) the utterance.""" + + accepted: bool + ack_latency_ms: int + detail: Optional[str] = None + raw: Optional[Any] = None + + +class SpeechTransport(abc.ABC): + """One wire protocol for delivering text to the robot.""" + + name: str = "transport" + + #: True when the robot itself tells us the utterance finished. When False the + #: adapter estimates the speaking duration instead of guessing completion. + reports_completion: bool = False + + @abc.abstractmethod + async def open(self) -> None: + """Set up any persistent client/socket. Raises RobotError on failure.""" + + @abc.abstractmethod + async def close(self) -> None: + """Tear down. Must never raise.""" + + @abc.abstractmethod + async def probe(self) -> bool: + """Cheap liveness check against the robot.""" + + @abc.abstractmethod + async def speak( + self, request: SpeechRequest, on_progress: ProgressCallback + ) -> TransportAck: + """Deliver the utterance. Return as soon as the robot acknowledges it.""" + + @abc.abstractmethod + async def stop(self) -> bool: + """Ask the robot to stop speaking. Returns True if a stop was delivered.""" + + async def describe(self) -> Dict[str, Any]: + return {} + + +# --------------------------------------------------------------------------- # +# payload templating +# --------------------------------------------------------------------------- # +def build_context(request: SpeechRequest, defaults: Dict[str, Any]) -> Dict[str, Any]: + ctx: Dict[str, Any] = { + "text": request.text, + "id": request.id, + "voice": request.voice if request.voice is not None else defaults.get("voice"), + "language": request.language if request.language is not None else defaults.get("language"), + "volume": defaults.get("volume"), + "speed": defaults.get("speed"), + } + return ctx + + +def render_payload(template: Any, ctx: Dict[str, Any], drop_null: bool = True) -> Any: + """Recursively substitute placeholders into a parsed-JSON template.""" + if isinstance(template, str): + stripped = template.strip() + if stripped.startswith("{") and stripped.endswith("}"): + key = stripped[1:-1] + if key in ctx: + return ctx[key] # preserve native type + rendered = template + for key, value in ctx.items(): + token = "{" + key + "}" + if token in rendered: + rendered = rendered.replace(token, "" if value is None else str(value)) + return rendered + + if isinstance(template, dict): + out: Dict[str, Any] = {} + for key, value in template.items(): + resolved = render_payload(value, ctx, drop_null) + if drop_null and resolved is None: + continue # unset optional parameter - do not send the key at all + out[key] = resolved + return out + + if isinstance(template, list): + items = [render_payload(item, ctx, drop_null) for item in template] + return [item for item in items if not (drop_null and item is None)] + + return template + + +def render_path(template: str, ctx: Dict[str, Any]) -> str: + """Render a URL path/query template with percent-encoded values.""" + rendered = template + for key, value in ctx.items(): + token = "{" + key + "}" + if token in rendered: + rendered = rendered.replace(token, quote("" if value is None else str(value), safe="")) + return rendered + + +def now_ms(since: float) -> int: + return int((time.perf_counter() - since) * 1000) diff --git a/backend/robot/transports/http_transport.py b/backend/robot/transports/http_transport.py new file mode 100644 index 0000000..3b3d4c2 --- /dev/null +++ b/backend/robot/transports/http_transport.py @@ -0,0 +1,232 @@ +"""HTTP/REST transport. + +Lowest-friction path when the robot exposes an HTTP API: one keep-alive +connection is opened at startup and reused for every utterance, so a Speak press +costs a single request/response round trip with no TCP or TLS handshake. + +The endpoint, method and body come from .env - see transports/base.py for why. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any, Dict, Optional + +from ...config.settings import Settings +from ..base import ( + ProgressCallback, + RobotTimeout, + RobotUnreachable, + SpeechFailed, + SpeechProgress, + SpeechRequest, + SpeechStage, + TransportNotAvailable, +) +from .base import SpeechTransport, TransportAck, build_context, now_ms, render_path, render_payload + +logger = logging.getLogger(__name__) + +try: # httpx is a hard requirement for this transport only + import httpx +except ImportError: # pragma: no cover + httpx = None # type: ignore[assignment] + + +class HttpTransport(SpeechTransport): + name = "http" + reports_completion = False # a REST ack rarely means "finished speaking" + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._robot = settings.robot + self._a3 = settings.a3 + self._client: Optional["httpx.AsyncClient"] = None + + # -- lifecycle ----------------------------------------------------------- # + async def open(self) -> None: + if httpx is None: + raise TransportNotAvailable( + "httpx is not installed", + user_message="HTTP transport requires the 'httpx' package. Run: pip install -r requirements.txt", + ) + if self._client is not None: + return + headers = {"Content-Type": "application/json", "Connection": "keep-alive"} + headers.update(self._a3.http_headers or {}) + if self._a3.http_auth_token: + headers.setdefault("Authorization", "Bearer " + self._a3.http_auth_token) + + self._client = httpx.AsyncClient( + base_url=self._robot.base_url, + headers=headers, + timeout=httpx.Timeout( + self._robot.request_timeout, connect=self._robot.connect_timeout + ), + # Keep the socket hot between utterances - this is the latency win. + limits=httpx.Limits(max_keepalive_connections=4, max_connections=8), + verify=False if self._robot.use_tls else True, + ) + logger.info("HTTP transport ready for %s", self._robot.base_url) + + async def close(self) -> None: + if self._client is not None: + try: + await self._client.aclose() + except Exception: # pragma: no cover + pass + self._client = None + + # -- health -------------------------------------------------------------- # + async def probe(self) -> bool: + """Prefer a documented status endpoint; fall back to a TCP connect.""" + if self._a3.http_status_path and self._client is not None: + try: + response = await self._client.request( + self._a3.http_status_method, + self._a3.http_status_path, + timeout=self._robot.connect_timeout, + ) + return response.status_code < 500 + except Exception as exc: + logger.debug("HTTP probe failed: %s", exc) + return False + return await self._tcp_probe() + + async def _tcp_probe(self) -> bool: + """No status endpoint configured - at least verify the port is open.""" + if not self._robot.ip: + return False + try: + fut = asyncio.open_connection(self._robot.ip, self._robot.port) + reader, writer = await asyncio.wait_for(fut, timeout=self._robot.connect_timeout) + writer.close() + try: + await writer.wait_closed() + except Exception: # pragma: no cover + pass + return True + except Exception as exc: + logger.debug("TCP probe %s failed: %s", self._robot.address, exc) + return False + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck: + if self._client is None: + await self.open() + assert self._client is not None + + ctx = build_context(request, self._defaults()) + path = render_path(self._a3.http_speak_path, ctx) + payload = render_payload(self._a3.http_speak_payload, ctx) + method = self._a3.http_speak_method + + await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...")) + started = time.perf_counter() + + try: + if method in ("GET", "DELETE", "HEAD"): + response = await self._client.request(method, path) + else: + response = await self._client.request(method, path, json=payload) + except Exception as exc: # httpx timeout/connect errors + if httpx is not None and isinstance(exc, httpx.TimeoutException): + raise RobotTimeout( + "timeout calling {0}{1}: {2}".format(self._robot.base_url, path, exc) + ) from exc + raise RobotUnreachable( + "cannot reach {0}{1}: {2}".format(self._robot.base_url, path, exc) + ) from exc + + ack_ms = now_ms(started) + + if response.status_code >= 400: + raise SpeechFailed( + "robot returned HTTP {0}: {1}".format(response.status_code, response.text[:400]), + user_message="Speech request failed (robot returned HTTP {0}).".format( + response.status_code + ), + ) + + raw: Any = None + try: + raw = response.json() + except Exception: + raw = response.text[:400] + + # Optional: a documented success flag inside the JSON body. + if self._a3.http_success_field and isinstance(raw, dict): + value = _dig(raw, self._a3.http_success_field) + if value is not None and not _truthy(value): + raise SpeechFailed( + "robot reported failure: {0}".format(raw), + user_message="Speech request failed. The robot rejected the utterance.", + ) + + return TransportAck(accepted=True, ack_latency_ms=ack_ms, raw=raw) + + async def stop(self) -> bool: + if not self._a3.http_stop_path: + logger.info("no A3_HTTP_STOP_PATH configured - cannot stop robot-side playback") + return False + if self._client is None: + await self.open() + assert self._client is not None + + ctx = build_context(SpeechRequest(text=""), self._defaults()) + path = render_path(self._a3.http_stop_path, ctx) + payload = render_payload(self._a3.http_stop_payload, ctx) + try: + if self._a3.http_stop_method in ("GET", "DELETE", "HEAD"): + response = await self._client.request(self._a3.http_stop_method, path) + else: + response = await self._client.request(self._a3.http_stop_method, path, json=payload) + return response.status_code < 400 + except Exception as exc: + logger.warning("stop request failed: %s", exc) + return False + + async def describe(self) -> Dict[str, Any]: + return { + "baseUrl": self._robot.base_url, + "speak": "{0} {1}".format(self._a3.http_speak_method, self._a3.http_speak_path or "(unset)"), + "stop": "{0} {1}".format(self._a3.http_stop_method, self._a3.http_stop_path or "(unset)"), + "status": "{0} {1}".format(self._a3.http_status_method, self._a3.http_status_path or "(TCP probe)"), + "keepAlive": True, + } + + def _defaults(self) -> Dict[str, Any]: + return { + "voice": self._a3.voice, + "language": self._a3.language, + "volume": self._a3.volume, + "speed": self._a3.speed, + } + + +def _dig(data: Dict[str, Any], dotted: str) -> Any: + node: Any = data + for part in dotted.split("."): + if not isinstance(node, dict) or part not in node: + return None + node = node[part] + return node + + +def _truthy(value: Any) -> bool: + """Interpret a success field. + + Deliberately liberal: robot APIs variously signal success as `true`, `"ok"`, + `0` (a result code) or `200` (an embedded HTTP status). + """ + if isinstance(value, bool): + return value + if isinstance(value, int): + return value in (0, 200, 1) + if isinstance(value, float): + return value in (0.0, 1.0, 200.0) + if isinstance(value, str): + return value.strip().lower() in ("0", "1", "200", "true", "ok", "success", "succeeded", "yes") + return bool(value) diff --git a/backend/robot/transports/ros2_transport.py b/backend/robot/transports/ros2_transport.py new file mode 100644 index 0000000..6d4640d --- /dev/null +++ b/backend/robot/transports/ros2_transport.py @@ -0,0 +1,236 @@ +"""ROS 2 transport. + +Used when the robot exposes its speech interface as a ROS 2 topic or service on +the shared DDS network rather than as a web API. + +`rclpy` is imported lazily: the rest of this project runs on a plain Windows PC +with no ROS installation, and only this file needs one. The message/service type +is resolved from its ROS type string at runtime, so no interface is hard-coded: + + A3_ROS_SPEAK_TOPIC=/the/documented/topic + A3_ROS_SPEAK_MSG_TYPE=std_msgs/msg/String + A3_ROS_SPEAK_MSG_FIELD=data + +DDS discovery is multicast-based, so the PC and the robot must be on the same L2 +network segment and share ROS_DOMAIN_ID. See docs/NETWORK.md. +""" + +from __future__ import annotations + +import asyncio +import importlib +import logging +import os +import threading +import time +from typing import Any, Dict, Optional + +from ...config.settings import Settings +from ..base import ( + ProgressCallback, + RobotUnreachable, + SpeechFailed, + SpeechProgress, + SpeechRequest, + SpeechStage, + TransportNotAvailable, +) +from .base import SpeechTransport, TransportAck, build_context, now_ms + +logger = logging.getLogger(__name__) + + +def _resolve_type(type_string: str) -> Any: + """'std_msgs/msg/String' -> the std_msgs.msg.String class.""" + parts = type_string.replace(".", "/").split("/") + if len(parts) == 2: # tolerate 'pkg/Type' + package, name = parts + module_path = package + ".msg" + elif len(parts) == 3: + package, kind, name = parts + module_path = package + "." + kind + else: + raise SpeechFailed( + "cannot parse ROS type '{0}'".format(type_string), + user_message="Invalid ROS message type in configuration: " + type_string, + ) + module = importlib.import_module(module_path) + return getattr(module, name) + + +class Ros2Transport(SpeechTransport): + name = "ros2" + reports_completion = False + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._a3 = settings.a3 + self._robot = settings.robot + self._rclpy: Any = None + self._node: Any = None + self._publisher: Any = None + self._stop_publisher: Any = None + self._client: Any = None + self._executor: Any = None + self._spin_thread: Optional[threading.Thread] = None + + # -- lifecycle ----------------------------------------------------------- # + async def open(self) -> None: + if self._node is not None: + return + await asyncio.get_running_loop().run_in_executor(None, self._open_blocking) + + def _open_blocking(self) -> None: + try: + import rclpy + from rclpy.executors import SingleThreadedExecutor + from rclpy.qos import QoSProfile, ReliabilityPolicy + except ImportError as exc: + raise TransportNotAvailable( + "rclpy not importable: {0}".format(exc), + user_message=( + "ROS 2 transport selected but rclpy is not installed on this PC. " + "Install ROS 2 and run the app from a sourced ROS environment, " + "or choose a different A3_TRANSPORT." + ), + ) from exc + + os.environ.setdefault("ROS_DOMAIN_ID", str(self._a3.ros_domain_id)) + self._rclpy = rclpy + if not rclpy.ok(): + rclpy.init(args=None) + + self._node = rclpy.create_node("agibot_voice_bridge") + qos = QoSProfile(depth=10, reliability=ReliabilityPolicy.RELIABLE) + + if self._a3.ros_use_service: + service_type = _resolve_type(self._a3.ros_service_type) + self._client = self._node.create_client(service_type, self._a3.ros_service_name) + else: + msg_type = _resolve_type(self._a3.ros_speak_msg_type) + self._publisher = self._node.create_publisher(msg_type, self._a3.ros_speak_topic, qos) + if self._a3.ros_stop_topic: + self._stop_publisher = self._node.create_publisher( + msg_type, self._a3.ros_stop_topic, qos + ) + + self._executor = SingleThreadedExecutor() + self._executor.add_node(self._node) + self._spin_thread = threading.Thread( + target=self._executor.spin, name="rclpy-spin", daemon=True + ) + self._spin_thread.start() + logger.info("ROS 2 transport ready (domain %s)", os.environ.get("ROS_DOMAIN_ID")) + + async def close(self) -> None: + node, executor, rclpy = self._node, self._executor, self._rclpy + self._node = self._executor = self._publisher = self._client = None + if executor is not None: + try: + executor.shutdown() + except Exception: # pragma: no cover + pass + if node is not None: + try: + node.destroy_node() + except Exception: # pragma: no cover + pass + if rclpy is not None: + try: + if rclpy.ok(): + rclpy.shutdown() + except Exception: # pragma: no cover + pass + + # -- health -------------------------------------------------------------- # + async def probe(self) -> bool: + if self._node is None: + return False + try: + if self._client is not None: + return bool(self._client.service_is_ready()) + if self._publisher is not None: + # A subscriber on the speech topic means the robot side is alive. + return self._publisher.get_subscription_count() > 0 + except Exception as exc: + logger.debug("ros2 probe failed: %s", exc) + return False + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck: + if self._node is None: + await self.open() + + ctx = build_context(request, self._defaults()) + await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Publishing to ROS 2...")) + started = time.perf_counter() + + loop = asyncio.get_running_loop() + try: + await loop.run_in_executor(None, self._publish_blocking, ctx) + except TransportNotAvailable: + raise + except Exception as exc: + raise RobotUnreachable("ROS 2 publish failed: {0}".format(exc)) from exc + + return TransportAck(accepted=True, ack_latency_ms=now_ms(started)) + + def _publish_blocking(self, ctx: Dict[str, Any]) -> None: + if self._client is not None: + request_type = _resolve_type(self._a3.ros_service_type).Request + req = request_type() + setattr(req, self._a3.ros_speak_msg_field, ctx["text"]) + if not self._client.wait_for_service(timeout_sec=self._robot.connect_timeout): + raise RobotUnreachable( + "ROS service {0} not available".format(self._a3.ros_service_name) + ) + future = self._client.call_async(req) + deadline = time.time() + self._robot.request_timeout + while not future.done() and time.time() < deadline: + time.sleep(0.01) + return + + assert self._publisher is not None + msg_type = _resolve_type(self._a3.ros_speak_msg_type) + msg = msg_type() + field = self._a3.ros_speak_msg_field + if not hasattr(msg, field): + raise SpeechFailed( + "message type {0} has no field '{1}'".format(self._a3.ros_speak_msg_type, field), + user_message="Configured ROS message field '{0}' does not exist on {1}.".format( + field, self._a3.ros_speak_msg_type + ), + ) + setattr(msg, field, ctx["text"]) + self._publisher.publish(msg) + + async def stop(self) -> bool: + if self._stop_publisher is None: + return False + try: + msg_type = _resolve_type(self._a3.ros_speak_msg_type) + msg = msg_type() + setattr(msg, self._a3.ros_speak_msg_field, "") + self._stop_publisher.publish(msg) + return True + except Exception as exc: + logger.warning("ros2 stop failed: %s", exc) + return False + + async def describe(self) -> Dict[str, Any]: + return { + "domainId": self._a3.ros_domain_id, + "mode": "service" if self._a3.ros_use_service else "topic", + "topic": self._a3.ros_speak_topic, + "messageType": self._a3.ros_speak_msg_type, + "service": self._a3.ros_service_name, + "nodeUp": self._node is not None, + } + + def _defaults(self) -> Dict[str, Any]: + return { + "voice": self._a3.voice, + "language": self._a3.language, + "volume": self._a3.volume, + "speed": self._a3.speed, + } diff --git a/backend/robot/transports/ssh_transport.py b/backend/robot/transports/ssh_transport.py new file mode 100644 index 0000000..a4dda23 --- /dev/null +++ b/backend/robot/transports/ssh_transport.py @@ -0,0 +1,192 @@ +"""SSH / on-robot command transport. + +The universal fallback. Every AgiBot humanoid runs a Linux compute module, so if +no network speech API is exposed, the robot can still be made to speak by running +its own TTS or audio-playback command over SSH. + +Nothing here is A3-specific - you supply the command that actually works on your +unit, discovered by logging in once and trying it by hand: + + A3_SSH_SPEAK_COMMAND= {text} + +`{text}` is shell-quoted before substitution, so arbitrary operator input cannot +break out of the argument. Authentication uses the OpenSSH client (key or agent); +password auth is used only when `paramiko` is installed. +""" + +from __future__ import annotations + +import asyncio +import logging +import shlex +import time +from typing import Any, Dict, List, Optional + +from ...config.settings import Settings +from ..base import ( + ProgressCallback, + RobotTimeout, + RobotUnreachable, + SpeechFailed, + SpeechProgress, + SpeechRequest, + SpeechStage, +) +from .base import SpeechTransport, TransportAck, now_ms + +logger = logging.getLogger(__name__) + +try: + import paramiko # optional - only needed for password authentication +except ImportError: # pragma: no cover + paramiko = None # type: ignore[assignment] + + +class SshTransport(SpeechTransport): + name = "ssh" + reports_completion = True # the command returns when playback ends + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._robot = settings.robot + self._a3 = settings.a3 + self._current: Optional[asyncio.subprocess.Process] = None + + # -- lifecycle ----------------------------------------------------------- # + async def open(self) -> None: + if not await self.probe(): + raise RobotUnreachable( + "SSH probe to {0}@{1} failed".format(self._a3.ssh_user, self._robot.ip) + ) + + async def close(self) -> None: + process = self._current + self._current = None + if process is not None and process.returncode is None: + try: + process.terminate() + except Exception: # pragma: no cover + pass + + async def probe(self) -> bool: + try: + code, _, _ = await self._run( + self._a3.ssh_probe_command or "true", timeout=self._robot.connect_timeout + 2 + ) + return code == 0 + except Exception as exc: + logger.debug("ssh probe failed: %s", exc) + return False + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck: + command = self._a3.ssh_speak_command.replace("{text}", shlex.quote(request.text)) + command = command.replace("{id}", shlex.quote(request.id)) + + await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...")) + started = time.perf_counter() + + code, out, err = await self._run(command, timeout=self._robot.request_timeout + 60) + if code != 0: + raise SpeechFailed( + "remote command exited {0}: {1}".format(code, (err or out)[:400]), + user_message="Speech request failed. The robot's speech command returned an error.", + ) + return TransportAck(accepted=True, ack_latency_ms=now_ms(started), detail=(out or "")[:200]) + + async def stop(self) -> bool: + if not self._a3.ssh_stop_command: + return False + try: + code, _, _ = await self._run(self._a3.ssh_stop_command, timeout=10) + return code == 0 + except Exception as exc: + logger.warning("ssh stop failed: %s", exc) + return False + + async def describe(self) -> Dict[str, Any]: + return { + "host": "{0}@{1}:{2}".format(self._a3.ssh_user, self._robot.ip, self._a3.ssh_port), + "auth": "paramiko-password" if self._use_paramiko else "openssh-key/agent", + "speakCommand": self._a3.ssh_speak_command or "(unset)", + } + + # -- internals ----------------------------------------------------------- # + @property + def _use_paramiko(self) -> bool: + return bool(self._a3.ssh_password) and paramiko is not None + + async def _run(self, command: str, timeout: float): + if not self._robot.ip: + raise RobotUnreachable("ROBOT_IP is not set") + if self._use_paramiko: + return await asyncio.get_running_loop().run_in_executor( + None, self._run_paramiko, command, timeout + ) + return await self._run_openssh(command, timeout) + + def _ssh_argv(self, command: str) -> List[str]: + argv = [ + "ssh", + "-p", + str(self._a3.ssh_port), + "-o", + "BatchMode=yes", + "-o", + "StrictHostKeyChecking=accept-new", + "-o", + "ConnectTimeout={0}".format(int(max(1, self._robot.connect_timeout))), + ] + if self._a3.ssh_key_path: + argv += ["-i", self._a3.ssh_key_path] + argv.append("{0}@{1}".format(self._a3.ssh_user, self._robot.ip)) + argv.append(command) + return argv + + async def _run_openssh(self, command: str, timeout: float): + try: + process = await asyncio.create_subprocess_exec( + *self._ssh_argv(command), + stdout=asyncio.subprocess.PIPE, + stderr=asyncio.subprocess.PIPE, + ) + except FileNotFoundError as exc: + raise RobotUnreachable( + "the 'ssh' client was not found on this PC", + ) from exc + + self._current = process + try: + stdout, stderr = await asyncio.wait_for(process.communicate(), timeout=timeout) + except asyncio.TimeoutError as exc: + process.kill() + raise RobotTimeout("remote command timed out after {0}s".format(timeout)) from exc + finally: + self._current = None + return ( + process.returncode, + stdout.decode("utf-8", "replace"), + stderr.decode("utf-8", "replace"), + ) + + def _run_paramiko(self, command: str, timeout: float): # pragma: no cover - needs a host + assert paramiko is not None + client = paramiko.SSHClient() + client.set_missing_host_key_policy(paramiko.AutoAddPolicy()) + try: + client.connect( + hostname=self._robot.ip, + port=self._a3.ssh_port, + username=self._a3.ssh_user, + password=self._a3.ssh_password, + key_filename=self._a3.ssh_key_path, + timeout=self._robot.connect_timeout, + allow_agent=False, + look_for_keys=False, + ) + _, stdout, stderr = client.exec_command(command, timeout=timeout) + out = stdout.read().decode("utf-8", "replace") + err = stderr.read().decode("utf-8", "replace") + return stdout.channel.recv_exit_status(), out, err + finally: + client.close() diff --git a/backend/robot/transports/ws_transport.py b/backend/robot/transports/ws_transport.py new file mode 100644 index 0000000..f15febb --- /dev/null +++ b/backend/robot/transports/ws_transport.py @@ -0,0 +1,226 @@ +"""WebSocket transport. + +The lowest-latency option when the robot offers it: the socket is opened once at +startup and held, so pressing Speak costs one frame write - no handshake, no +connection setup. It is also the only transport that can naturally deliver +*unsolicited* robot events ("speaking started", "speech finished") back to the UI. + +If the robot's event schema is known, set: + + A3_WS_DONE_FIELD=event # dotted path inside the JSON message + A3_WS_DONE_VALUE=speech_end # value that means "utterance finished" + +and the dashboard will show true completion instead of an estimate. +""" + +from __future__ import annotations + +import asyncio +import json +import logging +import time +from typing import Any, Dict, Optional + +from ...config.settings import Settings +from ..base import ( + ProgressCallback, + RobotTimeout, + RobotUnreachable, + SpeechProgress, + SpeechRequest, + SpeechStage, + TransportNotAvailable, +) +from .base import SpeechTransport, TransportAck, build_context, now_ms, render_payload + +logger = logging.getLogger(__name__) + +try: + from websockets.asyncio.client import connect as ws_connect # websockets >= 13 +except Exception: # pragma: no cover - older releases + try: + from websockets.client import connect as ws_connect # type: ignore[no-redef] + except Exception: + ws_connect = None # type: ignore[assignment] + + +class WebSocketTransport(SpeechTransport): + name = "ws" + + def __init__(self, settings: Settings) -> None: + self._settings = settings + self._robot = settings.robot + self._a3 = settings.a3 + self._ws: Any = None + self._reader: Optional[asyncio.Task] = None + self._completion: Optional[asyncio.Future] = None + self._lock = asyncio.Lock() + self._last_message: Any = None + + @property + def reports_completion(self) -> bool: # type: ignore[override] + return bool(self._a3.ws_done_field) + + # -- lifecycle ----------------------------------------------------------- # + @property + def url(self) -> str: + path = self._a3.ws_path or "/" + if not path.startswith("/"): + path = "/" + path + return self._robot.ws_base_url + path + + async def open(self) -> None: + if ws_connect is None: # pragma: no cover + raise TransportNotAvailable( + "websockets package missing", + user_message="WebSocket transport requires the 'websockets' package.", + ) + if self._ws is not None: + return + try: + self._ws = await asyncio.wait_for( + ws_connect( + self.url, + ping_interval=self._a3.ws_ping_interval or None, + ping_timeout=self._robot.request_timeout, + open_timeout=self._robot.connect_timeout, + max_queue=32, + ), + timeout=self._robot.connect_timeout + 1, + ) + except asyncio.TimeoutError as exc: + raise RobotTimeout("timed out opening {0}".format(self.url)) from exc + except Exception as exc: + raise RobotUnreachable("cannot open {0}: {1}".format(self.url, exc)) from exc + + self._reader = asyncio.create_task(self._read_loop(), name="a3-ws-reader") + logger.info("WebSocket transport connected to %s", self.url) + + async def close(self) -> None: + if self._reader is not None: + self._reader.cancel() + try: + await self._reader + except (asyncio.CancelledError, Exception): + pass + self._reader = None + if self._ws is not None: + try: + await self._ws.close() + except Exception: # pragma: no cover + pass + self._ws = None + + async def probe(self) -> bool: + if self._ws is None: + return False + # A closed socket surfaces here without an extra round trip. + reader = self._reader + if reader is not None and reader.done(): + return False + try: + pong = await self._ws.ping() + await asyncio.wait_for(pong, timeout=self._robot.connect_timeout) + return True + except Exception as exc: + logger.debug("ws probe failed: %s", exc) + return False + + # -- speech -------------------------------------------------------------- # + async def speak(self, request: SpeechRequest, on_progress: ProgressCallback) -> TransportAck: + async with self._lock: + if self._ws is None: + await self.open() + assert self._ws is not None + + ctx = build_context(request, self._defaults()) + payload = render_payload(self._a3.ws_speak_payload, ctx) + message = json.dumps(payload, ensure_ascii=False) + + loop = asyncio.get_running_loop() + self._completion = loop.create_future() if self.reports_completion else None + + await on_progress(SpeechProgress(request.id, SpeechStage.SENDING, "Sending to robot...")) + started = time.perf_counter() + try: + await self._ws.send(message) + except Exception as exc: + await self.close() + raise RobotUnreachable("websocket send failed: {0}".format(exc)) from exc + + return TransportAck(accepted=True, ack_latency_ms=now_ms(started)) + + async def await_completion(self, timeout: float) -> bool: + """Block until the robot reports the utterance finished (if it does).""" + if self._completion is None: + return False + try: + await asyncio.wait_for(asyncio.shield(self._completion), timeout=timeout) + return True + except asyncio.TimeoutError: + return False + except Exception: + return False + + async def stop(self) -> bool: + if self._ws is None or not self._a3.ws_stop_payload: + return False + ctx = build_context(SpeechRequest(text=""), self._defaults()) + payload = render_payload(self._a3.ws_stop_payload, ctx) + try: + await self._ws.send(json.dumps(payload, ensure_ascii=False)) + return True + except Exception as exc: + logger.warning("ws stop failed: %s", exc) + return False + + async def describe(self) -> Dict[str, Any]: + return { + "url": self.url, + "connected": self._ws is not None, + "reportsCompletion": self.reports_completion, + "lastMessage": self._last_message, + } + + # -- internals ----------------------------------------------------------- # + async def _read_loop(self) -> None: + assert self._ws is not None + try: + async for raw in self._ws: + self._last_message = raw if isinstance(raw, str) else "" + if not self._a3.ws_done_field: + continue + try: + message = json.loads(raw) + except Exception: + continue + value = _dig(message, self._a3.ws_done_field) + if value is None: + continue + expected = self._a3.ws_done_value + if expected is None or str(value) == expected: + fut = self._completion + if fut is not None and not fut.done(): + fut.set_result(True) + except asyncio.CancelledError: + raise + except Exception as exc: + logger.info("websocket reader closed: %s", exc) + self._ws = None + + def _defaults(self) -> Dict[str, Any]: + return { + "voice": self._a3.voice, + "language": self._a3.language, + "volume": self._a3.volume, + "speed": self._a3.speed, + } + + +def _dig(data: Any, dotted: str) -> Any: + node = data + for part in dotted.split("."): + if not isinstance(node, dict) or part not in node: + return None + node = node[part] + return node diff --git a/backend/services/__init__.py b/backend/services/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/backend/services/audio_library.py b/backend/services/audio_library.py new file mode 100644 index 0000000..d1396cd --- /dev/null +++ b/backend/services/audio_library.py @@ -0,0 +1,257 @@ +"""Saved speech audio. + +Every line the cloud voice synthesises is written here as an ordinary `.wav` +file with a readable name, plus an `index.json` describing it. That gives three +things a hidden cache could not: + + * **Instant replay.** A saved line plays with no synthesis, no network, no + wait - just read the file and play it. + * **Files you can actually use.** Open the folder, double-click a `.wav`, drop + one into a video edit or a stand's playlist. Nothing is locked in. + * **Works offline.** Once a line is saved, the demo no longer needs internet + for that line. + +Layout: + + audio_library/ + index.json + good-afternoon-and-welcome-to-our-a1b2c3d4.wav + please-follow-me-to-the-first-9f8e7d6c.wav + +The id is a hash of (text, voice, model, style), so the same line spoken by a +different voice is a different entry, and re-saving an identical line is free. + +NOTE: only the cloud voice produces files. The built-in OS voices play straight +to the sound card and never hand us audio, so utterances spoken by them are not +saved. `MOCK_VOICE_ENGINE=gemini` is what fills this library. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import os +import re +import struct +import threading +import time +from pathlib import Path +from typing import Any, Dict, List, Optional + +logger = logging.getLogger(__name__) + +INDEX_FILE = "index.json" +_SLUG_STRIP = re.compile(r"[^a-z0-9]+") + + +def slugify(text: str, limit: int = 40) -> str: + """A short, filesystem-safe, human-readable stem for a filename.""" + ascii_text = text.encode("ascii", "ignore").decode("ascii").lower() + slug = _SLUG_STRIP.sub("-", ascii_text).strip("-") + if len(slug) > limit: + slug = slug[:limit].rsplit("-", 1)[0] or slug[:limit] + return slug or "utterance" + + +def wav_duration(path: Path) -> float: + """Duration of a 16-bit PCM WAV, from its header.""" + try: + with open(path, "rb") as handle: + header = handle.read(44) + if len(header) < 44: + return 0.0 + channels, sample_rate = struct.unpack(" None: + self.root = Path(root) + self.root.mkdir(parents=True, exist_ok=True) + self._lock = threading.RLock() + self._index: Dict[str, Dict[str, Any]] = {} + self._stamp: Optional[tuple] = None + self._load_index() + + # -- identity ------------------------------------------------------------ # + @staticmethod + def make_id(text: str, voice: str, model: str, style: str = "") -> str: + raw = "\x00".join([model or "", voice or "", style or "", text or ""]) + return hashlib.sha256(raw.encode("utf-8")).hexdigest()[:16] + + # -- index --------------------------------------------------------------- # + @property + def index_path(self) -> Path: + return self.root / INDEX_FILE + + def _stamp_of_index(self) -> Optional[tuple]: + try: + stat = self.index_path.stat() + return (stat.st_mtime_ns, stat.st_size) + except OSError: + return None + + def _refresh(self) -> None: + """Re-read the index if another process changed it. + + `warm_voice.py` and the server both open this library. Without this the + server would keep a stale copy in memory and write it back over the + other process's changes - resurrecting entries whose files were deleted. + """ + if self._stamp_of_index() != self._stamp: + self._load_index() + + def _load_index(self) -> None: + self._stamp = self._stamp_of_index() + if not self.index_path.exists(): + self._index = {} + return + try: + data = json.loads(self.index_path.read_text(encoding="utf-8")) + entries = data.get("entries", data) if isinstance(data, dict) else {} + if isinstance(entries, dict): + # Drop entries whose file was deleted by hand. + self._index = { + key: value for key, value in entries.items() + if (self.root / value.get("file", "")).exists() + } + except (json.JSONDecodeError, OSError) as exc: + logger.warning("audio index unreadable (%s); starting a fresh one", exc) + self._index = {} + + def _save_index(self) -> None: + payload = {"version": 1, "entries": self._index} + temporary = self.index_path.with_suffix(".tmp") + try: + temporary.write_text( + json.dumps(payload, indent=1, ensure_ascii=False), encoding="utf-8" + ) + os.replace(temporary, self.index_path) + self._stamp = self._stamp_of_index() + except OSError as exc: # pragma: no cover + logger.warning("could not write audio index: %s", exc) + + # -- lookup -------------------------------------------------------------- # + def get(self, audio_id: str) -> Optional[Dict[str, Any]]: + with self._lock: + self._refresh() + entry = self._index.get(audio_id) + if entry is None: + return None + if not (self.root / entry["file"]).exists(): + self._index.pop(audio_id, None) + self._save_index() + return None + return dict(entry) + + def find(self, text: str, voice: str, model: str, style: str = "") -> Optional[Dict[str, Any]]: + return self.get(self.make_id(text, voice, model, style)) + + def path_of(self, audio_id: str) -> Optional[Path]: + entry = self.get(audio_id) + return self.root / entry["file"] if entry else None + + def list(self, newest_first: bool = True) -> List[Dict[str, Any]]: + with self._lock: + self._refresh() + # Only list clips whose file is really there - the panel offers a + # play button per row, and a row that 404s is worse than no row. + alive, missing = [], [] + for key, value in self._index.items(): + (alive if (self.root / value.get("file", "")).exists() else missing).append( + (key, value) + ) + if missing: + for key, _ in missing: + self._index.pop(key, None) + self._save_index() + items = [dict(value) for _, value in alive] + items.sort(key=lambda item: item.get("createdAt", 0), reverse=newest_first) + return items + + # -- writing ------------------------------------------------------------- # + def save(self, text: str, wav_bytes: bytes, voice: str, model: str, + style: str = "") -> Dict[str, Any]: + audio_id = self.make_id(text, voice, model, style) + with self._lock: + self._refresh() + existing = self.get(audio_id) + if existing is not None: + return existing + + filename = "{0}-{1}.wav".format(slugify(text), audio_id[:8]) + destination = self.root / filename + temporary = destination.with_suffix(".part") + temporary.write_bytes(wav_bytes) + os.replace(temporary, destination) # never index a half-written file + + entry = { + "id": audio_id, + "text": text, + "file": filename, + "voice": voice, + "model": model, + "style": style, + "createdAt": time.time(), + "durationSeconds": wav_duration(destination), + "bytes": destination.stat().st_size, + } + with self._lock: + self._index[audio_id] = entry + self._save_index() + return dict(entry) + + # -- housekeeping -------------------------------------------------------- # + def delete(self, audio_id: str) -> bool: + with self._lock: + self._refresh() + entry = self._index.pop(audio_id, None) + if entry is None: + return False + self._save_index() + try: + (self.root / entry["file"]).unlink() + except OSError: # pragma: no cover + pass + return True + + def clear(self) -> int: + with self._lock: + self._refresh() + entries = list(self._index.values()) + self._index = {} + self._save_index() + for entry in entries: + try: + (self.root / entry["file"]).unlink() + except OSError: # pragma: no cover + pass + return len(entries) + + def stats(self) -> Dict[str, Any]: + items = self.list() + return { + "count": len(items), + "bytes": sum(item.get("bytes", 0) for item in items), + "seconds": round(sum(item.get("durationSeconds", 0) for item in items), 1), + "dir": str(self.root), + } + + +_library: Optional[AudioLibrary] = None + + +def get_audio_library(root: Optional[Path] = None) -> AudioLibrary: + """Process-wide library, so the API and the voice share one index.""" + global _library + if _library is None: + if root is None: + root = Path(__file__).resolve().parents[2] / "audio_library" + _library = AudioLibrary(root) + return _library diff --git a/backend/services/history.py b/backend/services/history.py new file mode 100644 index 0000000..f8981fd --- /dev/null +++ b/backend/services/history.py @@ -0,0 +1,63 @@ +"""Speech history. + +Deliberately in-memory and bounded: this is an operator convenience for a live +demo ("say that again"), not an audit log. It resets when the server restarts. +""" + +from __future__ import annotations + +import threading +import time +from collections import deque +from typing import Any, Deque, Dict, List, Optional + +from ..robot.base import SpeechStage + + +class SpeechHistory: + def __init__(self, limit: int = 100) -> None: + self._items: Deque[Dict[str, Any]] = deque(maxlen=max(1, limit)) + self._index: Dict[str, Dict[str, Any]] = {} + self._lock = threading.Lock() + + def add(self, request_id: str, text: str) -> Dict[str, Any]: + entry: Dict[str, Any] = { + "id": request_id, + "text": text, + "at": time.time(), + "stage": SpeechStage.SENDING.value, + "success": None, + "ackLatencyMs": None, + "totalMs": None, + "error": None, + } + with self._lock: + if len(self._items) == self._items.maxlen: + # Newest is at the left, so the deque drops the rightmost entry. + self._index.pop(self._items[-1]["id"], None) + self._items.appendleft(entry) + self._index[request_id] = entry + return entry + + def update(self, request_id: str, **fields: Any) -> Optional[Dict[str, Any]]: + with self._lock: + entry = self._index.get(request_id) + if entry is None: + return None + entry.update(fields) + return dict(entry) + + def list(self) -> List[Dict[str, Any]]: + with self._lock: + return [dict(item) for item in self._items] + + def clear(self) -> int: + with self._lock: + count = len(self._items) + self._items.clear() + self._index.clear() + return count + + def __len__(self) -> int: + with self._lock: + return len(self._items) diff --git a/backend/services/speech_service.py b/backend/services/speech_service.py new file mode 100644 index 0000000..e598dde --- /dev/null +++ b/backend/services/speech_service.py @@ -0,0 +1,319 @@ +"""Speech orchestration. + +The one deliberate design decision worth stating: POST /api/robot/speak returns as +soon as the *robot has acknowledged* the utterance - not when speaking finishes. +The remaining lifecycle (speaking -> completed) streams over the WebSocket. + +That matters for a live demo. Blocking the HTTP response until the robot stops +talking would make a 12-second sentence look like a 12-second-slow button, and it +would tie a browser request to the speaking duration for no benefit. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from typing import Any, Dict, Optional + +from ..config.settings import Settings +from ..core.events import EventBus, EventType +from ..core.text import normalise +from ..robot.base import ( + RobotBusy, + RobotError, + RobotUnreachable, + SpeechProgress, + SpeechRequest, + SpeechResult, + SpeechStage, +) +from ..robot.manager import RobotManager +from .history import SpeechHistory + +logger = logging.getLogger(__name__) + + +class ValidationError(RobotError): + code = "invalid_text" + user_message = "Please enter some text for the robot to say." + + +def _swallow_exception(task: "asyncio.Task") -> None: + """Retrieve a finished task's exception so asyncio stops complaining.""" + if not task.cancelled(): + task.exception() + + +class SpeechService: + def __init__(self, settings: Settings, manager: RobotManager, bus: EventBus) -> None: + self._settings = settings + self._manager = manager + self._bus = bus + self.history = SpeechHistory(settings.speech.history_limit) + self._lock = asyncio.Lock() + self._active_task: Optional[asyncio.Task] = None + self._active_id: Optional[str] = None + + def update_settings(self, settings: Settings) -> None: + """Adopt reloaded configuration (limits, interrupt policy) in place.""" + self._settings = settings + + # -- state --------------------------------------------------------------- # + @property + def is_busy(self) -> bool: + return self._active_task is not None and not self._active_task.done() + + @property + def active_request_id(self) -> Optional[str]: + return self._active_id if self.is_busy else None + + # -- validation ---------------------------------------------------------- # + def validate(self, text: Optional[str]) -> str: + cleaned = normalise(text or "") + limits = self._settings.speech + if len(cleaned) < max(1, limits.min_length): + raise ValidationError("empty text") + if len(cleaned) > limits.max_length: + raise ValidationError( + "text too long: {0} chars".format(len(cleaned)), + user_message=( + "Text is too long ({0} characters). The limit is {1}. " + "Split it into shorter sentences.".format(len(cleaned), limits.max_length) + ), + ) + return cleaned + + # -- speak --------------------------------------------------------------- # + async def speak(self, text: Optional[str], voice: Optional[str] = None, + language: Optional[str] = None) -> Dict[str, Any]: + cleaned = self.validate(text) + + if not self._manager.is_connected: + status = self._manager.status_dict() + raise RobotUnreachable( + "robot not connected (state={0})".format(status["state"]), + user_message=( + status.get("error") + or "Robot is offline. Check the robot IP address and network connection." + ), + ) + + async with self._lock: + if self.is_busy: + if not self._settings.speech.allow_interrupt: + raise RobotBusy() + logger.info("interrupting utterance %s", self._active_id) + await self._cancel_active() + + request = SpeechRequest(text=cleaned, voice=voice, language=language) + entry = self.history.add(request.id, cleaned) + self._bus.publish(EventType.HISTORY_UPDATED, {"entry": entry, "action": "add"}) + + ack = asyncio.Event() + state: Dict[str, Any] = {"ackLatencyMs": None, "stage": SpeechStage.SENDING} + + async def on_progress(progress: SpeechProgress) -> None: + state["stage"] = progress.stage + if progress.stage is not SpeechStage.SENDING: + if state["ackLatencyMs"] is None: + state["ackLatencyMs"] = progress.elapsed_ms + ack.set() + self._bus.publish(EventType.SPEECH_PROGRESS, progress.to_dict()) + self.history.update(request.id, stage=progress.stage.value) + + started = time.perf_counter() + self._active_id = request.id + self._active_task = asyncio.create_task( + self._run(request, on_progress), name="speak-" + request.id + ) + # The HTTP call usually returns before the task finishes, so nobody is + # left to await it. Consume any exception here to keep asyncio from + # logging "Task exception was never retrieved" - it is already + # reported to the browser as a speech.progress / speech.result event. + self._active_task.add_done_callback(_swallow_exception) + + # Wait only for the acknowledgement, then hand control back to the UI. + ack_wait = asyncio.ensure_future(ack.wait()) + done, _ = await asyncio.wait( + {ack_wait, self._active_task}, + timeout=self._settings.robot.request_timeout + 2, + return_when=asyncio.FIRST_COMPLETED, + ) + if not ack_wait.done(): + ack_wait.cancel() + + # The utterance failed before it was ever acknowledged - surface it now + # as an HTTP error rather than only as a WebSocket event. + if self._active_task in done: + exc = self._active_task.exception() + if exc is not None: + raise exc + result = self._active_task.result() + return self._response(request, result.stage, result.ack_latency_ms, started) + + if not ack.is_set(): + await self._cancel_active() + raise RobotError( + "no acknowledgement within timeout", + user_message="The robot did not respond in time. The request was cancelled.", + ) + + return self._response(request, state["stage"], state["ackLatencyMs"], started) + + def _response(self, request: SpeechRequest, stage: Any, ack_ms: Optional[int], + started: float) -> Dict[str, Any]: + stage_value = stage.value if hasattr(stage, "value") else str(stage) + return { + "success": stage_value not in (SpeechStage.FAILED.value, SpeechStage.CANCELLED.value), + "status": stage_value, + "requestId": request.id, + "text": request.text, + "ackLatencyMs": ack_ms if ack_ms is not None else int((time.perf_counter() - started) * 1000), + } + + async def _run(self, request: SpeechRequest, on_progress) -> SpeechResult: + """Drive the utterance to completion and record the outcome.""" + try: + result = await self._manager.adapter.speak(request, on_progress) + except asyncio.CancelledError: + entry = self.history.update( + request.id, stage=SpeechStage.CANCELLED.value, success=False, error="Stopped." + ) + self._publish_history(entry) + raise + except RobotError as exc: + logger.warning("speech %s failed: %s", request.id, exc) + await on_progress( + SpeechProgress( + request.id, + SpeechStage.FAILED, + exc.user_message, + error_code=exc.code, + ) + ) + entry = self.history.update( + request.id, + stage=SpeechStage.FAILED.value, + success=False, + error=exc.user_message, + ) + self._publish_history(entry) + self._bus.publish( + EventType.SPEECH_RESULT, + { + "requestId": request.id, + "success": False, + "stage": SpeechStage.FAILED.value, + "errorCode": exc.code, + "error": exc.user_message, + }, + ) + raise + except Exception as exc: # pragma: no cover - defensive + logger.exception("unexpected speech failure") + await on_progress( + SpeechProgress(request.id, SpeechStage.FAILED, "Speech request failed.", + error_code="internal", detail=str(exc)) + ) + entry = self.history.update( + request.id, stage=SpeechStage.FAILED.value, success=False, + error="Speech request failed.", + ) + self._publish_history(entry) + raise + + entry = self.history.update( + request.id, + stage=result.stage.value, + success=result.success, + ackLatencyMs=result.ack_latency_ms, + totalMs=result.total_ms, + error=result.error, + ) + self._publish_history(entry) + self._bus.publish(EventType.SPEECH_RESULT, result.to_dict()) + return result + + def _publish_history(self, entry: Optional[Dict[str, Any]]) -> None: + if entry is not None: + self._bus.publish(EventType.HISTORY_UPDATED, {"entry": entry, "action": "update"}) + + # -- stop ---------------------------------------------------------------- # + async def stop(self) -> Dict[str, Any]: + stopped_robot = False + try: + stopped_robot = await self._manager.adapter.stop_speaking() + except Exception as exc: + logger.warning("robot stop failed: %s", exc) + + had_active = self.is_busy + if had_active: + # Give the adapter a moment to unwind cleanly before force-cancelling. + try: + await asyncio.wait_for(asyncio.shield(self._active_task), timeout=1.5) + except (asyncio.TimeoutError, Exception): + await self._cancel_active() + + return { + "success": True, + "status": "stopped", + "hadActiveSpeech": had_active, + "robotAcknowledged": stopped_robot, + } + + async def _cancel_active(self) -> None: + task = self._active_task + if task is None or task.done(): + self._active_task = None + return + task.cancel() + try: + await task + except (asyncio.CancelledError, Exception): + pass + finally: + self._active_task = None + + # -- history ------------------------------------------------------------- # + def annotate_saved(self, items): + """Mark history entries whose audio is already on disk. + + Lets the dashboard show a replay button only where playback really is + instant, instead of promising something it cannot deliver. + """ + try: + from .audio_library import get_audio_library + + library = get_audio_library() + mock = self._settings.mock + if mock.voice_engine != "gemini": + return items + from ..core.pronunciation import build as build_pronouncer + from ..config.settings import PROJECT_ROOT + from ..robot.gemini_voice import split_sentences + + pronouncer = build_pronouncer(PROJECT_ROOT, mock.pronunciation) + for item in items: + text = item.get("text") or "" + if pronouncer is not None: + text = pronouncer.apply(text) + chunks = split_sentences(text) + entries = [ + library.find(c, mock.gemini_voice, mock.gemini_model, mock.gemini_style) + for c in chunks + ] + found = [e for e in entries if e] + item["audioSaved"] = bool(chunks) and len(found) == len(chunks) + item["audioIds"] = [e["id"] for e in found] + except Exception: # pragma: no cover - never break history over this + logger.debug("could not annotate saved audio", exc_info=True) + return items + + def clear_history(self) -> Dict[str, Any]: + removed = self.history.clear() + self._bus.publish(EventType.HISTORY_UPDATED, {"action": "clear"}) + return {"success": True, "removed": removed} + + async def shutdown(self) -> None: + await self._cancel_active() diff --git a/docs/AGIBOT_A3_INTEGRATION.md b/docs/AGIBOT_A3_INTEGRATION.md new file mode 100644 index 0000000..2a39ab5 --- /dev/null +++ b/docs/AGIBOT_A3_INTEGRATION.md @@ -0,0 +1,382 @@ +# AGIBOT A3 — how the speech integration works + +**Short answer: the A3 has a documented, LAN-callable, native text-to-speech HTTP +endpoint, and this project talks to it directly.** No SDK, no login, no ROS, no +audio file transfer. + +``` +Browser → this backend → HTTP JSON-RPC → A3 head unit (HDU) → built-in TTS → speaker +``` + +Everything below is sourced. Claims that could not be verified are tagged +**[UNVERIFIED]** or **[UNKNOWN]** rather than smoothed over, and anything found on +a *different* AgiBot model is labelled as such — an A2/X2/A3-Ultra fact is not an +A3 fact. + +> Research date: **2 September 2026**, against AgiBot doc tree **A3 v3.1 / v3.2**. +> If your unit's firmware differs, re-run the discovery procedure in §6. + +--- + +## 1. The interface + +AgiBot's A3 developer guide documents two developer transports on the robot: + +| Transport | AgiBot's stated use | Notes | +| --- | --- | --- | +| **HTTP JSON-RPC** | "low-frequency, many-to-one calls" | `POST http://{IP}:{PORT}/rpc/{service}/{method}`, `Content-Type: application/json` **mandatory** | +| **ROS 2 topic** | "high-frequency, many-to-many calls" | ROS 2 Jazzy on Fast DDS; default QoS `keep_last`/depth 10/**`best_effort`** | + +Source: [03-second_develop_interface_overview](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview) + +The same page explicitly sanctions calling it from another machine: + +> *"For remote calls, replace it with the reachable IP of the corresponding +> device in the current network."* +> *"…there are no restrictions [on language]. As long as you can send and receive +> HTTP requests, you can call all HTTP JSON RPC interfaces."* + +**There is no documented gRPC or WebSocket interface on the robot.** (WebSocket +appears only in AgiBot's separate *cloud* LinkSoul agent SDK — see §3.) + +### The TTS call + +On the **HDU** (the head unit — it owns all audio), port **59301**: + +```http +POST http://:59301/rpc/aimdk.protocol.TTSService/PlayTTS +Content-Type: application/json + +{"text": "Hello, welcome to our showroom.", + "priority_level": "INTERACTION_L6", + "domain": "voice_control", + "trace_id": "abc123", + "is_interrupted": true} +``` + +```json +{"text": "...", "priority_level": "INTERACTION_L6", "priority_weight": 0, + "domain": "voice_control", "trace_id": "abc123_18bZZLTk5VfJGSy8Cylsu4", + "is_sucess": true, "error_message": "", "estimated_duration": 0} +``` + +Source: [07-02-audio_play](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play) +· runnable `curl` in AgiBot's own [quick start](https://open.agibot.com/docs/en/aimdk/a3/v3_1/dev_guide/06-second_develop_quick_start) + +Companion methods on the same host and port: + +| Method | Body | Purpose | +| --- | --- | --- | +| `StopTTSTraceId` | `{"trace_id": "..."}` | interrupt an utterance | +| `GetAudioStatus` | `{"trace_id": "..."}` | `TTSStatusType_Begin\|_Playing\|_End\|_Stop\|_Error\|_InQue\|_NOTInQue` | +| `PlayMediaFile` | `{"file_name": "..."}` | play a file from `/agibot/data/var/interaction/audio/` | + +Volume lives on a **different port, 56666**, service `aimdk.protocol.HalAudioService` +(`GetAudioVolume` / `SetAudioVolume`, range 0–100). + +> ⚠️ **AgiBot's own hardware warning: do not set volume above 70 — "exceeding +> this range may damage the speaker."** This project does not change volume, so +> it cannot trip that. If you add volume control later, clamp it at 70. + +--- + +## 2. Four traps, and how this project handles each + +These are the things that make a naive client fail. All four are handled in +[backend/robot/transports/aimdk_transport.py](../backend/robot/transports/aimdk_transport.py). + +| # | The trap | What we do | +| --- | --- | --- | +| 1 | The success flag is spelled **`is_sucess`** — one `c` — in AgiBot's documentation. | Read `is_sucess` **and** `is_success`, so a corrected firmware still works. | +| 2 | The **returned `trace_id` is not the one you sent** — the robot appends a random suffix. Stop only works with the returned value. | The returned id is captured and is what `Stop` uses. | +| 3 | **1024 *bytes*** of UTF-8 is a hard limit (~200 characters). A 1000-character Chinese sentence is ~3000 bytes and is rejected outright. | Text is chunked on sentence boundaries by **byte** length; only the first chunk may interrupt, the rest queue behind it. | +| 4 | Status polling is capped at **≤ 0.2 Hz**; AgiBot warns high-frequency RPC can destabilise the robot. | The liveness probe is a **TCP connect, never an RPC**, and completion is estimated rather than polled. | + +Two more behaviours worth knowing, from reading AimRT's own source +([http_rpc_backend.cc](https://raw.githubusercontent.com/AimRT/AimRT/main/src/plugins/net_plugin/http/http_rpc_backend.cc)): + +- The route is `"/rpc" + ` — the `pb:/` prefix seen in + AimRT function names is **stripped**. Never put `pb:/` in a URL. +- **An RPC-level failure is HTTP 500, not a JSON error.** Unknown route is 404. + **HTTP 200 means the transport worked, not that the robot spoke** — + `is_sucess` only confirms the priority check passed. + +--- + +## 3. Why this path, and not the others + +| Path | Latency | Reliability | Complexity | Verdict | +| --- | --- | --- | --- | --- | +| **1. HTTP JSON-RPC `PlayTTS`** (this project) | 1 request RTT to enqueue (< 20 ms LAN); time-to-first-audio dominated by the robot's own synthesis | Highest — AgiBot's own primary documented path | Near zero — one HTTP POST | **✅ Chosen** | +| 2. Own TTS → WAV → SCP → `PlayFile` | 1–4 s cold; ~1 RTT for pre-cached phrases | Good; documented offline fallback | Moderate — SSH creds, file lifecycle | Fallback if #1 needs internet and you are air-gapped | +| 3. Own TTS → PCM → ROS 2 `/audiohal/audio/playback` | Best possible — tens of ms, true streaming | High ceiling, high risk (best-effort QoS, focus arbitration) | **High** — ROS 2 Jazzy on the PC, login-gated aarch64 packages, robot reboot into `only_voice` | Only if you need barge-in streaming | +| 4. LinkSoul cloud agent SDK (WebSocket) | Worst — PC → AgiBot cloud → robot | n/a | Commercial credentials, cloud dependency | Wrong tool: a conversational agent layer, not a "say this now" primitive | +| 5. SSH + direct ALSA write | — | — | — | **Do not ship.** No AgiBot document mentions ALSA. `hal_audio` owns the device behind an explicit focus/priority protocol; bypassing it also bypasses the volume damage limit. | + +Path 1 wins on every axis the brief asked for: lowest latency, highest +reliability, simplest deployment, best-documented A3 compatibility, and it +scales because it is stateless HTTP. + +**Critically, it sends *text*.** The robot synthesises. Nothing is rendered, +encoded or transferred by the PC — which is exactly the architecture the +"fast speech" requirement asks for. + +--- + +## 4. What this project does with it + +`A3_TRANSPORT=aimdk` (the default) selects +[aimdk_transport.py](../backend/robot/transports/aimdk_transport.py): + +- one **keep-alive** `httpx` client opened at startup and reused for every + utterance — no handshake per Speak; +- a `Timeout:` header (AimRT honours whole seconds) **and** a client-side + timeout, so a wedged robot cannot hang the dashboard; +- ack latency measured and reported to the UI; +- `Stop` via `StopTTSTraceId` using the robot-returned `trace_id`; +- errors mapped to messages an operator can act on. + +**Every documented value is a setting, not a literal**, because AimRT has no +service-discovery endpoint and AgiBot does not guarantee port or name stability +across firmware: + +```env +A3_AIMDK_SERVICE=aimdk.protocol.TTSService +A3_AIMDK_PLAY_METHOD=PlayTTS +A3_AIMDK_STOP_METHOD=StopTTSTraceId +A3_AIMDK_PRIORITY=INTERACTION_L6 +A3_AIMDK_MAX_BYTES=1024 +``` + +If your firmware renames something, you edit `.env`. If it uses a wholly +different shape, `A3_TRANSPORT=http` gives you a fully templated HTTP client, and +`ws` / `ros2` / `ssh` are there as further fallbacks. + +--- + +## 5. Connecting the robot — the exact steps + +### 5.1 Where the IP goes + +**One file: `.env`, in the project root.** Three lines: + +```env +ROBOT_MODE=real +ROBOT_IP=192.168.1.50 ← the robot's IP on YOUR network +ROBOT_PORT=59301 +``` + +> **Do not use `10.42.10.10`.** That address appears throughout AgiBot's +> examples, but it is the HDU's address on the robot's *internal* `eth_hdu` +> network. Your PC cannot reach it. Use the HDU's address on your WiFi/LAN. + +Then either restart the server, or without restarting: + +```bash +curl -X POST http://localhost:8000/api/config/reload +``` + +### 5.2 The order to do things in + +```bash +# 1. reachable at all? +ping 192.168.1.50 + +# 2. what does it expose? (probes the documented ports and routes, read-only) +python scripts/discover_robot.py 192.168.1.50 + +# 3. the decisive test - does it actually speak? +python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3" + +# 4. point the app at it, then +python scripts/selftest.py +``` + +### 5.3 What each failure means + +| Result of step 3 | Meaning | Next step | +| --- | --- | --- | +| **The robot speaks** | Path 1 works. Done. | Set `.env`, restart, demo. | +| Connection refused / timeout | 59301 is not bound on the WLAN interface | §6 Step 2 — SSH in and check `ss -ltnp` | +| **HTTP 404** | Port is right, route is wrong | Check `A3_AIMDK_SERVICE` / `A3_AIMDK_PLAY_METHOD`; confirm firmware version | +| **HTTP 500** | Route exists, the RPC failed — **the port is reachable** | Read the response body; check `priority_level` | +| `is_sucess: false` | The robot rejected it | Read `error_message` — usually priority or text length | +| 200 + success but silence | Accepted but not played | Check robot volume, and whether TTS needs internet (§7) | + +--- + +## 6. Discovery procedure for the live robot + +`scripts/discover_robot.py` automates the PC-side checks. The rest needs an SSH +session on the robot. + +```bash +# From the PC - which of the documented ports answer? +python scripts/discover_robot.py + +# Equivalent by hand. A 404 proves an AimRT HTTP server is listening; +# connection-refused proves it is not. +for p in 59301 56666 51049; do + curl -s -o /dev/null -w "$p -> %{http_code}\n" -m 3 \ + "http://:$p/rpc/does.not.Exist/Nope" +done +``` + +On the robot (username `agi` is **[UNVERIFIED]** — confirm with AgiBot): + +```bash +# Is the RPC server bound to 0.0.0.0 or only to the internal 10.42.10.10? +ss -ltnp | grep -E '59301|56666|51049' +ip -br a # the HDU's real WLAN address + +# The authoritative port map, from firmware config +grep -rn 'listen_port' /agibot/software/v0/ 2>/dev/null | head -50 + +# ROS environment (resolves ROS_DOMAIN_ID, only needed for the ros2 transport) +source /agibot/software/v0/entry/env/env.sh || source /agibot/software/v0/entry/env.sh +echo "DOMAIN=$ROS_DOMAIN_ID RMW=$RMW_IMPLEMENTATION" + +# The real audio surface — this substitutes for the missing discovery API +ros2 topic list | grep -Ei 'audio|tts|interaction' +ros2 topic echo --once /interaction/tts_status # while a PlayTTS is running +``` + +> **Do not put any of these in a loop.** AgiBot's +> [constraints page](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/04-second_develop_constraints_overview) +> caps status RPCs at ≤ 0.2 Hz, forbids deploying on the MDU ("risks the robot +> falling"), forbids `apt` install/uninstall, and states violations "may void +> warranty." + +--- + +## 7. Open questions — send these to AgiBot + +The five that actually affect this project, in priority order. (Chinese versions +follow, since AgiBot support answers faster in Chinese.) + +1. **Is port 59301 on the HDU bound to all interfaces and reachable from a PC on + the same WiFi/LAN using the HDU's WLAN IP — or only on the internal + `eth_hdu` 10.42.10.x segment?** + *HDU 的 59301 端口是否监听在所有网络接口上?能否从同一局域网的 PC 通过 HDU 的 WLAN IP 访问?还是仅绑定在内部 eth_hdu (10.42.10.x) 网段?* + +2. **Does `TTSService/PlayTTS` require internet access?** Your English v3.2 page + says it "requires network connection". Which engine performs synthesis, and is + there a fully on-device/offline TTS mode? What happens with LAN but no internet? + *`PlayTTS` 是否必须联网?语音合成由哪家引擎提供?是否有完全离线的端侧 TTS 方案?只有局域网、没有外网时会怎样?* + +3. **Is there any authentication, token or IP allow-list on the `/rpc/` + endpoints?** If not, what do you recommend on a shared network? (AimRT sets + `Access-Control-Allow-Origin: *` unconditionally, so any page on the LAN can + call these.) + *`/rpc/` 接口是否有鉴权、token 或 IP 白名单?如果没有,在共享网络下推荐如何防护?* + +4. **Can we select voice/timbre, language, speaking rate or emotion per + `PlayTTS` call?** Is `SetTtsParameters` or an equivalent available on the A3, + and at which endpoint? **Our unit's voice is currently set to "Yunxiao" + (teenager, male, multi-language) — is that selectable per request, or only in + the console?** + *能否在每次 PlayTTS 调用时指定音色、语言、语速或情感?A3 是否提供 SetTtsParameters?接口地址是什么? + 我们的机器人当前音色为"Yunxiao"(少年、男声、多语言),该音色能否在每次调用时指定,还是只能在控制台配置?* + + > **Operator-reported, 2026-09-02:** the robot's configured voice is + > **"Yunxiao"** — labelled *teenager | male*, *multi-language*, and marked as + > the default (默). It was shown in a voice-library UI offering per-voice + > "Speech Synthesis", which is consistent with AgiBot's LinkSoul timbre + > library (30–40 timbres + voice cloning) — though that link is inference, not + > something any published document states. This is **not** in AgiBot's docs; + > it comes from the console for this specific unit. It suggests voice + > selection happens on a platform/console rather than in the `PlayTTS` body, + > which is exactly what question 4 needs to settle. + +5. **Please confirm the complete `priority_level` enum** accepted by A3 + `PlayTTS`, and the semantics of `priority_weight` and `is_interrupted`. Your + X2 documentation publishes seven levels — do the same apply to the A3? + *请提供 A3 PlayTTS 支持的完整 priority_level 枚举,以及 priority_weight 和 is_interrupted 的语义。X2 文档公布了七个等级,A3 是否相同?* + +Lower priority, worth asking in the same message: the response envelope +(is there a `header.code`/`header.msg` wrapper?); whether the AimDK SDK download +needs a verified-customer account; whether ports 59301/56666/51049 are stable +across firmware; and the `PlayMediaFile` sample rate — **your English v3.2 page +says 24 kHz and your Chinese v3.2 page says 16 kHz for the same call.** + +--- + +## 8. Known contradictions in AgiBot's own documentation + +Recorded so nobody wastes an afternoon on them. + +| # | Contradiction | Impact here | +| --- | --- | --- | +| 1 | `PlayMediaFile` sample rate: **EN v3.2 says 24 kHz, ZH v3.2 says 16 kHz**, same page | None — we send text, not files | +| 2 | Microphone: v3.1 and v3.2 dev guides say the built-in mic is "not yet configured or enabled"; the User Manual and product page advertise an 8-mic array | None — output only | +| 3 | `env.sh` path: `/agibot/software/v0/entry/env.sh` vs `…/entry/env/env.sh` | Only matters for the `ros2` transport | +| 4 | ROS 2 distro: AgiBot says the A3 is **Jazzy**; AimRT's `ros2_plugin` doc says Humble | Only matters for the `ros2` transport | +| 5 | SDK wheel: the A3 v3.2 guide instructs `a2_aimdk-3.0.0…whl` in most places and `a3_aimdk-3.0.0…whl` in one — an **A2 artifact inside the A3 doc tree** | None — Path 1 needs no SDK | + +**`ROS_DOMAIN_ID` for the A3 is [UNKNOWN]** — it is not printed in any A3 +document. A2 uses 232 (**different model**), and one third-party A3-Ultra +deployment also uses 232 — suggestive, not proof. Read it from `env.sh` on the +robot. + +--- + +## 9. Security note worth raising internally + +No authentication scheme is documented for the `/rpc/` endpoints, and AimRT sets +`Access-Control-Allow-Origin: *` unconditionally with OPTIONS preflight handled. +**Taken together: any host — or any web page opened in any browser — on the same +LAN as the robot can make it speak.** + +This project does not widen that exposure. The browser talks only to +`127.0.0.1:8000`; robot credentials and addresses stay server-side. But on a +shared or public network, treat the robot's RPC ports as unauthenticated and +segment the network accordingly. + +--- + +## 10. Testing the real path before the robot arrives + +`ROBOT_MODE=mock` tests the application. To test **this transport** — the URL +shape, the JSON body, the `trace_id` round trip, chunking, Stop, and the error +paths — run the local stand-in: + +```bash +python scripts/fake_a3_server.py # implements the documented contract on :59301 +``` + +then in `.env`: + +```env +ROBOT_MODE=real +ROBOT_IP=127.0.0.1 +ROBOT_PORT=59301 +``` + +and use the dashboard normally. Every utterance prints in the stand-in's console. + +> It is a **test double written from public documentation**, not the robot. +> Passing against it proves the client is well-formed — not that the firmware +> behaves identically. Verify on the real unit. + +--- + +## 11. Sources + +**Official AgiBot, A3-specific** +- Audio interfaces (TTS, status, stop, volume, ROS 2 audio) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play) · [ZH](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/07-02-audio_play) +- Interface overview (the two transports, URL shape, remote-call sanction) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview) +- Quick start (the runnable `PlayTTS` curl) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_1/dev_guide/06-second_develop_quick_start) +- Resource manager (`PlayFile`, WAV/PCM rules, resource dirs) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-05-resource_manager) +- Interactive guide (`audio_msgs/msg/AudioPlayback`, focus services, `only_voice`) — [EN](https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/08-interactive_guide) +- Development constraints (polling limits, MDU, apt, warranty) — [ZH](https://open.agibot.com/docs/aimdk/a3/v3_2/dev_guide/04-second_develop_constraints_overview) +- Hardware/software overview (HDU/MDU, internal IPs, Ubuntu 24.04) — [ZH](https://open.agibot.com/docs/aimdk/a3/v3_1/dev_guide/01-a3_overview) +- AimDK docs root — https://open.agibot.com/docs/aimdk (SDK downloads are login-gated) + +**AimRT framework** (transport semantics, read from source) +- [net_plugin docs](https://docs.aimrt.org/tutorials/plugins/net_plugin.html) · [http_rpc_backend.cc](https://raw.githubusercontent.com/AimRT/AimRT/main/src/plugins/net_plugin/http/http_rpc_backend.cc) + +**Different model — do not assume it applies to the A3** +- X2 voice interface, full 7-value priority enum — https://x2-aimdk.agibot.com/en/dev/Interface/interactor/voice.html + +**Third-party corroboration** (an A3 **Ultra** deployment, not a specification) +- [tzj-hub/agibot_A3_sells_cars](https://github.com/tzj-hub/agibot_A3_sells_cars) — same base URL, methods and payload shape; reads `is_sucess` with an `is_success` fallback; its `a3_interfaces.py` is the only source anywhere for the `header.code`/`header.msg` error envelope, which is why this project reads that envelope defensively rather than depending on it. diff --git a/docs/NETWORK.md b/docs/NETWORK.md new file mode 100644 index 0000000..7541db2 --- /dev/null +++ b/docs/NETWORK.md @@ -0,0 +1,196 @@ +# Network setup: PC ↔ AGIBOT A3 + +Everything here is standard LAN networking. Nothing in this file assumes anything +about the A3's software — that lives in +[AGIBOT_A3_INTEGRATION.md](AGIBOT_A3_INTEGRATION.md). + +``` + ┌──────────────┐ ┌──────────────┐ + │ Your PC │ │ AGIBOT A3 │ + │ │ same LAN / subnet │ │ + │ browser │ ───────────────────────► │ speech svc │ + │ ↓ │ 192.168.x.x │ ↓ │ + │ backend │ │ speaker │ + └──────────────┘ └──────────────┘ +``` + +The PC and the robot must be able to reach each other by IP. That is the whole +requirement. + +--- + +## 1. Put both machines on the same network + +Ranked by how well they work in practice: + +| Setup | Latency | Notes | +| --- | --- | --- | +| **Wired Ethernet, same switch** | best, ~0.2–1 ms | Ideal for a demo. Nothing to go wrong. | +| **Both on the same Wi-Fi AP** | ~2–20 ms, variable | Fine, but a crowded venue Wi-Fi is the #1 cause of a bad demo. | +| **PC wired, robot Wi-Fi (same subnet)** | mixed | Works. Check they are on the same subnet, not two VLANs. | +| **Direct cable PC ↔ robot** | best | Needs static IPs on both, or link-local. Useful when there is no venue network. | +| Different subnets / guest Wi-Fi | — | Usually blocked. Guest networks isolate clients from each other. | + +> **Demo advice:** if you can run a cable, run a cable. Client isolation on a +> conference Wi-Fi will silently block PC→robot traffic while both devices show +> "connected to the internet". + +--- + +## 2. Find the robot's IP address + +Any of these, easiest first: + +1. **The robot's own screen / app / teach pendant** — usually shows the IP in a + network or system settings page. Most reliable. +2. **Your router's DHCP client list** — log into the router (often + `192.168.1.1`), look for a newly-connected device. +3. **Scan the subnet from the PC.** First find your own subnet: + + ```powershell + ipconfig # look at IPv4 Address, e.g. 192.168.1.23 + ``` + + Then sweep it: + + ```powershell + # Windows: ping every host, then read the ARP table + 1..254 | ForEach-Object { Start-Process -WindowStyle Hidden ping "192.168.1.$_" -ArgumentList "-n 1 -w 200" } + Start-Sleep 5 + arp -a + ``` + + ```bash + # if you have nmap (any OS) - much better + nmap -sn 192.168.1.0/24 + ``` + +4. **mDNS**, if the robot advertises itself: + + ```bash + ping agibot-a3.local + ``` + + A hostname works anywhere this project asks for `ROBOT_IP`. + +> Ask whoever commissions the robot to give it a **DHCP reservation** (a fixed +> IP tied to its MAC address). Otherwise the IP can change on reboot and you will +> be editing `.env` before every demo. + +--- + +## 3. Test reachability + +```bash +ping 192.168.1.50 +``` + +| Result | Meaning | +| --- | --- | +| Replies with a time | The robot is reachable. Continue. | +| `Request timed out` | Wrong IP, robot off, different subnet, or ICMP blocked (see below). | +| `Destination host unreachable` | No route — you are on a different subnet. | + +**ICMP being blocked does not mean the robot is unreachable.** Some robots drop +ping but still answer on their service port. Test the port directly: + +```powershell +Test-NetConnection 192.168.1.50 -Port 8080 +``` + +```bash +nc -vz 192.168.1.50 8080 # Linux/macOS +``` + +Then let this project probe it properly: + +```bash +python scripts/discover_robot.py 192.168.1.50 +``` + +--- + +## 4. Configure this app + +In `.env`: + +```env +ROBOT_MODE=real +ROBOT_IP=192.168.1.50 +ROBOT_PORT=8080 +``` + +Restart the server, or: + +```bash +curl -X POST http://localhost:8000/api/config/reload +``` + +The header pill turns green within a few seconds if the connection succeeds. If +it does not, `GET /api/robot/diagnostics` says exactly what was attempted and +what failed. + +--- + +## 5. Firewall + +The PC makes **outbound** connections to the robot, so a Windows Firewall inbound +rule is usually unnecessary. Two exceptions: + +- **`HOST=0.0.0.0`** — you want to open the dashboard from another device. + Windows will prompt to allow Python on private networks the first time; allow + it for *Private* networks only, never *Public*. +- **ROS 2 / DDS transport** — DDS uses multicast discovery and a wide range of + UDP ports **inbound** to the PC. This is the one case where the firewall + usually needs a rule: + + ```powershell + # run as Administrator, only if you use A3_TRANSPORT=ros2 + New-NetFirewallRule -DisplayName "ROS2 DDS" -Direction Inbound -Protocol UDP ` + -LocalPort 7400-7600 -Action Allow -Profile Private + ``` + + DDS discovery is multicast and does **not** cross subnets or most Wi-Fi APs. + ROS 2 realistically requires the PC and robot on the same wired L2 segment, + with the same `ROS_DOMAIN_ID`. + +--- + +## 6. Latency expectations + +Measured by this app and shown in the header. What to expect on a healthy LAN: + +| Hop | Typical | +| --- | --- | +| Browser → local backend | < 2 ms (loopback) | +| Backend → robot, wired | 0.5–3 ms | +| Backend → robot, Wi-Fi | 2–30 ms, occasionally spiking | +| Robot's own TTS synthesis start | the dominant term — tens to hundreds of ms | + +The network is almost never the bottleneck; the robot's speech synthesis is. That +is exactly why this project sends **text** and lets the robot synthesise, instead +of generating audio on the PC and transferring it. + +If the header shows latency above ~50 ms on Wi-Fi, move to Ethernet before +blaming the software. + +--- + +## 7. Troubleshooting + +| Symptom | Cause | Fix | +| --- | --- | --- | +| Pill stuck on *Connecting…* | Wrong IP or port, robot booting | `ping`, then `discover_robot.py` | +| `Robot is offline` immediately | Nothing listening on `ROBOT_PORT` | Confirm the port with the discovery script | +| Ping works, app says offline | Right host, wrong port or path | Check `ROBOT_PORT` and `A3_HTTP_SPEAK_PATH` | +| Works, then drops after minutes | Wi-Fi roaming, DHCP lease change | Wired connection + DHCP reservation | +| Connects but nothing is heard | Robot volume, muted speaker, wrong audio sink | Check the robot's own volume first | +| Fine on the bench, fails at the venue | Client isolation on guest Wi-Fi | Bring your own switch/router | +| ROS 2 sees no topics | Different `ROS_DOMAIN_ID`, multicast blocked | Same domain ID, same wired segment | + +Two commands answer most questions: + +```bash +python scripts/discover_robot.py # what the robot exposes +curl http://localhost:8000/api/robot/diagnostics # what this app tried and saw +``` diff --git a/frontend/index.html b/frontend/index.html new file mode 100644 index 0000000..8011c69 --- /dev/null +++ b/frontend/index.html @@ -0,0 +1,168 @@ + + + + + + + AGIBOT A3 · Voice Control + + + + + + +
+
+ +
+

AGIBOT A3

+

Voice Control

+
+ -- +
+ +
+ +
+ + Starting… +
+ + +
+
+ + + + + +
+ + +
+
+

Say something

+ The robot speaks through its built-in speaker +
+ +
+ + +
+ Ctrl+Enter to speak · Esc to stop + 0 / 1000 +
+
+ +
+ + + + + +
+ + +
+ +
    +
  1. Sending
  2. +
  3. Processing
  4. +
  5. Playing
  6. +
  7. Completed
  8. +
+
+ + +
+ + Ready + +
+
+ + + +
+ +
+ + + + diff --git a/frontend/js/api.js b/frontend/js/api.js new file mode 100644 index 0000000..50b1fa3 --- /dev/null +++ b/frontend/js/api.js @@ -0,0 +1,69 @@ +/** + * Thin REST client for the local backend. + * The page never talks to the robot directly - only to this origin. + */ + +class ApiError extends Error { + constructor(message, code, status) { + super(message); + this.name = 'ApiError'; + this.code = code || 'error'; + this.status = status || 0; + } +} + +async function request(path, options = {}) { + let response; + try { + response = await fetch(path, { + headers: { 'Content-Type': 'application/json' }, + ...options, + }); + } catch (err) { + // The backend itself is unreachable - a different failure from "robot offline". + throw new ApiError( + 'Cannot reach the local server. Is backend/main.py still running?', + 'backend_down', + 0, + ); + } + + let body = null; + const text = await response.text(); + if (text) { + try { body = JSON.parse(text); } catch { body = { error: text }; } + } + + if (!response.ok) { + const message = + (body && (body.error || body.detail)) || + `Request failed (HTTP ${response.status})`; + throw new ApiError( + typeof message === 'string' ? message : JSON.stringify(message), + (body && body.errorCode) || 'http_error', + response.status, + ); + } + return body; +} + +export const api = { + status: () => request('/api/robot/status'), + config: () => request('/api/config'), + diagnostics: () => request('/api/robot/diagnostics'), + reloadConfig: () => request('/api/config/reload', { method: 'POST' }), + reconnect: () => request('/api/robot/reconnect', { method: 'POST' }), + speak: (text) => request('/api/robot/speak', { + method: 'POST', + body: JSON.stringify({ text }), + }), + stop: () => request('/api/robot/stop', { method: 'POST' }), + history: () => request('/api/speech/history'), + clearHistory: () => request('/api/speech/history', { method: 'DELETE' }), + audio: () => request('/api/audio'), + audioUrl: (id) => `/api/audio/${id}/file`, + deleteAudio: (id) => request(`/api/audio/${id}`, { method: 'DELETE' }), + clearAudio: () => request('/api/audio', { method: 'DELETE' }), +}; + +export { ApiError }; diff --git a/frontend/js/app.js b/frontend/js/app.js new file mode 100644 index 0000000..f6cd9ac --- /dev/null +++ b/frontend/js/app.js @@ -0,0 +1,373 @@ +/** + * Application wiring. + * + * Flow: user text -> POST /api/robot/speak -> backend -> robot + * robot lifecycle -> backend event bus -> WebSocket -> this file -> UI + * + * The POST returns as soon as the robot acknowledges the utterance; everything + * after that ("speaking", "completed", failures) arrives over the socket. + */ + +import { api, ApiError } from './api.js'; +import { RobotSocket } from './socket.js'; +import * as ui from './ui.js'; + +const DRAFT_KEY = 'agibot-a3:draft'; +const RESET_DELAY = 2200; + +const state = { + config: { maxLength: 1000, mode: 'mock' }, + status: { state: 'connecting', connected: false }, + history: [], + audio: { items: [], stats: {} }, + auditioning: null, + activeRequestId: null, + stage: 'idle', + resetTimer: null, +}; + +const socket = new RobotSocket(); + +// ========================================================================= +// derived UI state +// ========================================================================= +function syncControls() { + const length = ui.el.textInput.value.trim().length; + const busy = Boolean(state.activeRequestId); + const connected = Boolean(state.status.connected); + + ui.setControls({ + canSpeak: connected && length > 0 && length <= state.config.maxLength, + busy, + canStop: busy, + }); + ui.renderCounter(ui.el.textInput.value.length, state.config.maxLength); +} + +function applyStatus(status) { + state.status = { ...state.status, ...status }; + ui.renderConnection(state.status); + ui.renderFacts(state.status); + + // The robot vanished mid-utterance: stop pretending it is still speaking. + if (!state.status.connected && state.activeRequestId) { + finishUtterance('error', 'Robot disconnected during speech.'); + } + + // Page opened (or reloaded) while the robot is mid-sentence - adopt it rather + // than showing "Ready" over a talking robot. + if (state.status.busy && state.status.activeRequestId && !state.activeRequestId) { + state.activeRequestId = state.status.activeRequestId; + state.stage = 'speaking'; + clearTimeout(state.resetTimer); + ui.renderPipeline('speaking'); + ui.renderLive('speaking', 'Speaking…'); + } + + syncControls(); +} + +// ========================================================================= +// speech lifecycle +// ========================================================================= +const STAGE_VIEW = { + queued: { tone: 'busy', text: 'Queued…' }, + sending: { tone: 'busy', text: 'Sending to robot…' }, + processing: { tone: 'busy', text: 'Robot is preparing speech…' }, + speaking: { tone: 'speaking', text: 'Speaking…' }, + completed: { tone: 'ok', text: 'Completed' }, + cancelled: { tone: 'error', text: 'Stopped' }, + failed: { tone: 'error', text: 'Speech failed' }, +}; + +function onProgress(progress) { + if (state.activeRequestId && progress.requestId !== state.activeRequestId) return; + state.activeRequestId = progress.requestId; + + const stage = progress.stage; + state.stage = stage; + ui.renderPipeline(stage); + + const view = STAGE_VIEW[stage] || { tone: 'busy', text: stage }; + const detail = progress.detail + || (typeof progress.elapsedMs === 'number' ? `${progress.elapsedMs} ms` : ''); + ui.renderLive(view.tone, progress.message || view.text, detail); + + if (['completed', 'failed', 'cancelled'].includes(stage)) { + if (stage === 'completed') refreshAudio(); + if (stage === 'failed') ui.toast(progress.message || 'Speech request failed.', 'error'); + finishUtterance(view.tone, progress.message || view.text, detail); + } + syncControls(); +} + +function finishUtterance(tone, message, detail = '') { + state.activeRequestId = null; + ui.renderLive(tone, message, detail); + syncControls(); + + clearTimeout(state.resetTimer); + state.resetTimer = setTimeout(() => { + if (state.activeRequestId) return; // a new utterance already started + state.stage = 'idle'; + ui.renderPipeline('idle'); + ui.renderLive('idle', state.status.connected ? 'Ready' : 'Robot offline'); + }, RESET_DELAY); +} + +// ========================================================================= +// history +// ========================================================================= +function applyHistoryEvent(payload) { + if (payload.action === 'clear') { + state.history = []; + } else if (payload.entry) { + const index = state.history.findIndex((item) => item.id === payload.entry.id); + if (index === -1) state.history.unshift(payload.entry); + else state.history[index] = payload.entry; + const limit = state.config.historyLimit || 100; + if (state.history.length > limit) state.history.length = limit; + } + ui.renderHistory(state.history, useHistoryText, replayLine); +} + +/** Speak a history line again. Its audio is saved, so this is instant. */ +async function replayLine(item) { + ui.el.textInput.value = item.text; + saveDraft(); + await speak(); +} + +function useHistoryText(text) { + ui.el.textInput.value = text; + ui.el.textInput.focus(); + ui.el.textInput.setSelectionRange(text.length, text.length); + saveDraft(); + syncControls(); +} + +// ========================================================================= +// actions +// ========================================================================= +async function speak() { + const text = ui.el.textInput.value.trim(); + if (!text) { + ui.toast('Type something for the robot to say first.', 'error'); + ui.el.textInput.focus(); + return; + } + if (!state.status.connected) { + ui.toast( + state.status.error || 'Robot is offline. Check the robot IP and network connection.', + 'error', + ); + return; + } + + clearTimeout(state.resetTimer); + ui.renderPipeline('sending'); + ui.renderLive('busy', 'Sending to robot…'); + ui.setControls({ canSpeak: false, busy: true, canStop: true }); + + try { + const result = await api.speak(text); + state.activeRequestId = result.requestId; + syncControls(); + } catch (err) { + state.activeRequestId = null; + const message = err instanceof ApiError ? err.message : 'Speech request failed.'; + ui.toast(message, 'error'); + ui.renderPipeline('failed'); + finishUtterance('error', message); + } +} + +async function stop() { + try { + await api.stop(); + ui.renderLive('error', 'Stopping…'); + } catch (err) { + ui.toast(err.message || 'Could not stop the robot.', 'error'); + } +} + +function clearText() { + ui.el.textInput.value = ''; + saveDraft(); + ui.el.textInput.focus(); + syncControls(); +} + +// ========================================================================= +// saved audio +// ========================================================================= +async function refreshAudio() { + try { + const data = await api.audio(); + state.audio = { items: data.items || [], stats: data.stats || {} }; + ui.renderAudio(state.audio.items, state.audio.stats, { + fileUrl: api.audioUrl, + onPlayState: (audio) => { + // Only one clip auditions at a time. + if (state.auditioning && state.auditioning !== audio) { + state.auditioning.pause(); + state.auditioning.currentTime = 0; + } + state.auditioning = audio; + }, + }); + } catch { + /* the panel is a convenience; never let it break the page */ + } +} + +async function clearAudio() { + try { + const result = await api.clearAudio(); + ui.toast(`Deleted ${result.removed} saved clip(s).`, 'ok', 3000); + await refreshAudio(); + // Rows can no longer promise instant replay. + state.history = state.history.map((item) => ({ ...item, audioSaved: false })); + ui.renderHistory(state.history, useHistoryText, replayLine); + } catch (err) { + ui.toast(err.message || 'Could not clear saved audio.', 'error'); + } +} + +async function clearHistory() { + try { + await api.clearHistory(); + state.history = []; + ui.renderHistory(state.history, useHistoryText, replayLine); + } catch (err) { + ui.toast(err.message || 'Could not clear history.', 'error'); + } +} + +async function reconnect() { + ui.el.reconnectBtn.dataset.spin = 'true'; + try { + await api.reconnect(); + ui.toast('Reconnecting to the robot…', 'info', 2500); + } catch (err) { + ui.toast(err.message || 'Reconnect failed.', 'error'); + } finally { + setTimeout(() => { ui.el.reconnectBtn.dataset.spin = 'false'; }, 900); + } +} + +// ========================================================================= +// draft persistence (per-browser convenience only) +// ========================================================================= +function saveDraft() { + try { localStorage.setItem(DRAFT_KEY, ui.el.textInput.value); } catch { /* private mode */ } +} + +function loadDraft() { + try { + const draft = localStorage.getItem(DRAFT_KEY); + if (draft) ui.el.textInput.value = draft; + } catch { /* ignore */ } +} + +// ========================================================================= +// events +// ========================================================================= +function bindDom() { + ui.el.speakBtn.addEventListener('click', speak); + ui.el.stopBtn.addEventListener('click', stop); + ui.el.clearBtn.addEventListener('click', clearText); + ui.el.clearHistoryBtn.addEventListener('click', clearHistory); + ui.el.clearAudioBtn.addEventListener('click', clearAudio); + ui.el.reconnectBtn.addEventListener('click', reconnect); + + ui.el.textInput.addEventListener('input', () => { saveDraft(); syncControls(); }); + + // Ctrl/Cmd+Enter speaks. A bare Enter inserts a newline, on purpose - nobody + // wants a half-typed sentence going out of the robot's speaker mid-demo. + ui.el.textInput.addEventListener('keydown', (event) => { + if ((event.ctrlKey || event.metaKey) && event.key === 'Enter') { + event.preventDefault(); + speak(); + } + }); + + document.addEventListener('keydown', (event) => { + if (event.key === 'Escape' && state.activeRequestId) { + event.preventDefault(); + stop(); + } + }); +} + +function bindSocket() { + socket.addEventListener('hello', (event) => { + const { config, status, history } = event.detail; + state.config = { ...state.config, ...config }; + ui.renderConfig(state.config); + applyStatus(status); + state.history = history || []; + ui.renderHistory(state.history, useHistoryText, replayLine); + refreshAudio(); + document.documentElement.removeAttribute('data-loading'); + }); + + socket.addEventListener('robot.status', (event) => applyStatus(event.detail)); + socket.addEventListener('speech.progress', (event) => onProgress(event.detail)); + socket.addEventListener('history.updated', (event) => applyHistoryEvent(event.detail)); + socket.addEventListener('config.updated', (event) => { + state.config = { ...state.config, ...event.detail }; + ui.renderConfig(state.config); + }); + + socket.addEventListener('rtt', (event) => { + const rtt = event.detail.rtt; + if (typeof rtt === 'number') { + ui.el.latencyChip.title = `Robot round-trip. Browser to backend: ${rtt} ms`; + } + }); + + socket.addEventListener('link', (event) => { + if (event.detail.up) return; + // The backend went away - say so rather than leaving a stale green light. + applyStatus({ + state: 'disconnected', + connected: false, + error: 'Lost connection to the local server.', + }); + }); +} + +// ========================================================================= +// boot +// ========================================================================= +async function boot() { + bindDom(); + bindSocket(); + loadDraft(); + syncControls(); + socket.connect(); + + // Fallback for the rare case the socket cannot open at all. + setTimeout(async () => { + if (socket.isOpen) return; + try { + const [config, status, history] = await Promise.all([ + api.config(), api.status(), api.history(), + ]); + state.config = { ...state.config, ...config }; + ui.renderConfig(state.config); + applyStatus(status); + state.history = history.items || []; + ui.renderHistory(state.history, useHistoryText, replayLine); + refreshAudio(); + document.documentElement.removeAttribute('data-loading'); + } catch { + ui.toast('Cannot reach the local server. Is backend/main.py running?', 'error', 12000); + } + }, 1200); + + ui.el.textInput.focus(); +} + +boot(); diff --git a/frontend/js/socket.js b/frontend/js/socket.js new file mode 100644 index 0000000..43ba136 --- /dev/null +++ b/frontend/js/socket.js @@ -0,0 +1,111 @@ +/** + * WebSocket client for live robot state. + * + * Replaces polling entirely: the backend pushes connection state, speech + * lifecycle and history updates. Reconnects on its own with capped backoff, so + * restarting the server does not require reloading the page. + */ + +const PING_INTERVAL = 5000; +const MIN_BACKOFF = 500; +const MAX_BACKOFF = 8000; + +export class RobotSocket extends EventTarget { + constructor() { + super(); + this.ws = null; + this.backoff = MIN_BACKOFF; + this.pingTimer = null; + this.rtt = null; + this.closedByUs = false; + } + + get url() { + const scheme = location.protocol === 'https:' ? 'wss' : 'ws'; + return `${scheme}://${location.host}/ws`; + } + + get isOpen() { + return this.ws && this.ws.readyState === WebSocket.OPEN; + } + + connect() { + this.closedByUs = false; + if (this.ws && (this.ws.readyState === WebSocket.OPEN || this.ws.readyState === WebSocket.CONNECTING)) { + return; + } + + let ws; + try { + ws = new WebSocket(this.url); + } catch { + this.#scheduleReconnect(); + return; + } + this.ws = ws; + + ws.onopen = () => { + this.backoff = MIN_BACKOFF; + this.#emit('link', { up: true }); + this.#startPing(); + }; + + ws.onmessage = (event) => { + let message; + try { message = JSON.parse(event.data); } catch { return; } + + if (message.type === 'pong') { + const sent = message.data && message.data.t; + if (typeof sent === 'number') this.rtt = Math.max(0, Math.round(performance.now() - sent)); + this.#emit('rtt', { rtt: this.rtt }); + return; + } + this.#emit(message.type, message.data || {}); + }; + + ws.onclose = () => { + this.#stopPing(); + this.#emit('link', { up: false }); + if (!this.closedByUs) this.#scheduleReconnect(); + }; + + ws.onerror = () => { /* onclose always follows; handled there */ }; + } + + close() { + this.closedByUs = true; + this.#stopPing(); + if (this.ws) this.ws.close(); + } + + send(type, data = {}) { + if (!this.isOpen) return false; + this.ws.send(JSON.stringify({ type, ...data })); + return true; + } + + requestStatus() { this.send('status'); } + requestReconnect() { this.send('reconnect'); } + + // -- internals ------------------------------------------------------------ + #emit(type, detail) { + this.dispatchEvent(new CustomEvent(type, { detail })); + } + + #startPing() { + this.#stopPing(); + const ping = () => this.send('ping', { t: performance.now() }); + ping(); + this.pingTimer = setInterval(ping, PING_INTERVAL); + } + + #stopPing() { + if (this.pingTimer) { clearInterval(this.pingTimer); this.pingTimer = null; } + } + + #scheduleReconnect() { + const delay = this.backoff; + this.backoff = Math.min(this.backoff * 2, MAX_BACKOFF); + setTimeout(() => this.connect(), delay); + } +} diff --git a/frontend/js/ui.js b/frontend/js/ui.js new file mode 100644 index 0000000..b4cd9a9 --- /dev/null +++ b/frontend/js/ui.js @@ -0,0 +1,367 @@ +/** DOM rendering. All element lookups and mutations live here. */ + +const $ = (id) => document.getElementById(id); + +export const el = { + modeBadge: $('modeBadge'), + connPill: $('connPill'), + connLabel: $('connLabel'), + latencyChip: $('latencyChip'), + latencyValue: $('latencyValue'), + reconnectBtn: $('reconnectBtn'), + + banner: $('configBanner'), + bannerTitle: $('bannerTitle'), + bannerText: $('bannerText'), + + textInput: $('textInput'), + counter: $('counter'), + counterNow: $('counterNow'), + counterMax: $('counterMax'), + + speakBtn: $('speakBtn'), + speakLabel: $('speakLabel'), + stopBtn: $('stopBtn'), + clearBtn: $('clearBtn'), + + pipeline: $('pipeline'), + pipelineFill: $('pipelineFill'), + + live: $('live'), + liveText: $('liveText'), + liveDetail: $('liveDetail'), + + factState: $('factState'), + factMode: $('factMode'), + factTransport: $('factTransport'), + factAddress: $('factAddress'), + factLatency: $('factLatency'), + factUptime: $('factUptime'), + statusNote: $('statusNote'), + + audioList: $('audioList'), + audioCount: $('audioCount'), + audioNote: $('audioNote'), + clearAudioBtn: $('clearAudioBtn'), + + historyList: $('historyList'), + historyEmpty: $('historyEmpty'), + historyCount: $('historyCount'), + clearHistoryBtn: $('clearHistoryBtn'), + + toasts: $('toasts'), +}; + +const CONNECTION_LABEL = { + connected: 'Robot Connected', + connecting: 'Connecting…', + disconnected: 'Robot Disconnected', + error: 'Connection Error', +}; + +const PIPELINE_PROGRESS = { + idle: 0, queued: 8, sending: 26, processing: 52, speaking: 80, + completed: 100, failed: 100, cancelled: 100, +}; + +const STEP_ORDER = ['sending', 'processing', 'speaking', 'completed']; + +// ------------------------------------------------------------------------- +// header / connection +// ------------------------------------------------------------------------- +export function renderConnection(status) { + const state = status.state || 'disconnected'; + el.connPill.dataset.state = state; + el.connLabel.textContent = CONNECTION_LABEL[state] || state; + + const latency = status.latencyMs; + if (typeof latency === 'number' && state === 'connected') { + el.latencyChip.hidden = false; + el.latencyValue.textContent = `${latency < 10 ? latency.toFixed(1) : Math.round(latency)} ms`; + } else { + el.latencyChip.hidden = true; + } +} + +export function renderConfig(config) { + if (!config) return; + const mode = (config.mode || 'mock').toLowerCase(); + el.modeBadge.textContent = mode === 'real' ? 'Live Robot' : 'Mock Mode'; + el.modeBadge.dataset.mode = mode; + + const max = config.maxLength || 1000; + el.counterMax.textContent = String(max); + el.textInput.maxLength = max; + + const issues = config.issues || []; + if (issues.length) { + const worst = issues.find((i) => i.level === 'error') || issues[0]; + el.banner.hidden = false; + el.banner.dataset.level = worst.level; + el.bannerTitle.textContent = + worst.level === 'error' ? 'Configuration error' : 'Configuration warning'; + el.bannerText.textContent = `${worst.message} (${worst.key} in ${config.envFile || '.env'})`; + } else { + el.banner.hidden = true; + } +} + +export function renderFacts(status) { + const robot = status.robot || {}; + const state = status.state || 'disconnected'; + + el.factState.textContent = CONNECTION_LABEL[state] || state; + el.factState.dataset.tone = + state === 'connected' ? 'ok' : state === 'connecting' ? 'warn' : 'error'; + + el.factMode.textContent = robot.mode === 'real' ? 'Live robot' : 'Simulation'; + el.factTransport.textContent = (robot.transport || '–').toUpperCase(); + el.factAddress.textContent = robot.address || '–'; + el.factAddress.title = robot.address || ''; + + el.factLatency.textContent = + typeof status.latencyMs === 'number' ? `${Math.round(status.latencyMs)} ms` : '–'; + el.factUptime.textContent = + typeof status.uptimeSeconds === 'number' ? formatDuration(status.uptimeSeconds) : '–'; + + if (status.error) { + el.statusNote.hidden = false; + el.statusNote.textContent = status.error; + } else { + el.statusNote.hidden = true; + } +} + +// ------------------------------------------------------------------------- +// composer +// ------------------------------------------------------------------------- +export function renderCounter(length, max) { + el.counterNow.textContent = String(length); + const ratio = max ? length / max : 0; + el.counter.dataset.level = ratio >= 1 ? 'over' : ratio > 0.85 ? 'warn' : 'ok'; +} + +export function setControls({ canSpeak, busy, canStop }) { + el.speakBtn.disabled = !canSpeak; + el.speakBtn.dataset.busy = busy ? 'true' : 'false'; + el.speakLabel.textContent = busy ? 'Speaking…' : 'Speak'; + el.stopBtn.disabled = !canStop; +} + +// ------------------------------------------------------------------------- +// pipeline + live status +// ------------------------------------------------------------------------- +export function renderPipeline(stage) { + const key = PIPELINE_PROGRESS[stage] === undefined ? 'idle' : stage; + el.pipeline.dataset.stage = key; + el.pipelineFill.style.inset = `0 ${100 - PIPELINE_PROGRESS[key]}% 0 0`; + + const activeIndex = STEP_ORDER.indexOf( + key === 'queued' ? 'sending' : key === 'failed' || key === 'cancelled' ? 'speaking' : key, + ); + el.pipeline.querySelectorAll('li').forEach((li, index) => { + if (key === 'idle') { li.removeAttribute('data-on'); return; } + if (index < activeIndex) li.dataset.on = 'done'; + else if (index === activeIndex) li.dataset.on = 'active'; + else li.removeAttribute('data-on'); + }); +} + +export function renderLive(tone, text, detail = '') { + el.live.dataset.tone = tone; + el.liveText.textContent = text; + el.liveDetail.textContent = detail || ''; +} + +// ------------------------------------------------------------------------- +// history +// ------------------------------------------------------------------------- +export function renderHistory(items, onPick, onReplay = () => {}) { + el.historyCount.textContent = String(items.length); + el.historyEmpty.hidden = items.length > 0; + el.historyList.innerHTML = ''; + + for (const item of items) { + const li = document.createElement('li'); + li.className = 'hitem'; + li.dataset.ok = item.success === null || item.success === undefined + ? (isTerminal(item.stage) ? 'false' : 'pending') + : String(Boolean(item.success)); + li.title = 'Click to put this text back in the box'; + li.tabIndex = 0; + + const bar = document.createElement('span'); + bar.className = 'hitem__bar'; + + const body = document.createElement('div'); + body.className = 'hitem__body'; + + const text = document.createElement('div'); + text.className = 'hitem__text'; + text.textContent = item.text; + + const meta = document.createElement('div'); + meta.className = 'hitem__meta'; + meta.append(spanOf(formatTime(item.at))); + if (typeof item.ackLatencyMs === 'number') meta.append(spanOf(`${item.ackLatencyMs} ms`)); + if (item.error) { + const err = spanOf(item.error); + err.className = 'err'; + meta.append(err); + } else if (!isTerminal(item.stage)) { + meta.append(spanOf(item.stage)); + } + + body.append(text, meta); + li.append(bar, body); + + if (item.audioSaved) { + const replay = document.createElement('button'); + replay.className = 'hitem__replay'; + replay.title = 'Play the saved audio instantly'; + replay.setAttribute('aria-label', 'Replay this line'); + replay.innerHTML = ''; + replay.addEventListener('click', (event) => { + event.stopPropagation(); // do not also load it into the box + onReplay(item); + }); + li.append(replay); + } + + const pick = () => onPick(item.text); + li.addEventListener('click', pick); + li.addEventListener('keydown', (event) => { + if (event.key === 'Enter' || event.key === ' ') { event.preventDefault(); pick(); } + }); + + el.historyList.append(li); + } +} + +function isTerminal(stage) { + return ['completed', 'failed', 'cancelled'].includes(stage); +} + +function spanOf(text) { + const span = document.createElement('span'); + span.textContent = text; + return span; +} + +// ------------------------------------------------------------------------- +// toasts +// ------------------------------------------------------------------------- +export function toast(message, tone = 'info', ttl = 5000) { + const node = document.createElement('div'); + node.className = 'toast'; + node.dataset.tone = tone; + + const dot = document.createElement('span'); + dot.className = 'toast__dot'; + const text = document.createElement('div'); + text.textContent = message; + + node.append(dot, text); + el.toasts.append(node); + + const remove = () => { + node.classList.add('is-out'); + setTimeout(() => node.remove(), 220); + }; + const timer = setTimeout(remove, ttl); + node.addEventListener('click', () => { clearTimeout(timer); remove(); }); + return node; +} + +// ------------------------------------------------------------------------- +// helpers +// ------------------------------------------------------------------------- +export function formatTime(epochSeconds) { + if (!epochSeconds) return ''; + return new Date(epochSeconds * 1000).toLocaleTimeString([], { + hour: '2-digit', minute: '2-digit', second: '2-digit', + }); +} + +export function formatDuration(seconds) { + const total = Math.max(0, Math.round(seconds)); + if (total < 60) return `${total}s`; + const minutes = Math.floor(total / 60); + if (minutes < 60) return `${minutes}m ${total % 60}s`; + return `${Math.floor(minutes / 60)}h ${minutes % 60}m`; +} + + +// ------------------------------------------------------------------------- +// saved audio +// ------------------------------------------------------------------------- +const PLAY_ICON = ''; +const STOP_ICON = ''; +const DOWNLOAD_ICON = + '' + + ''; + +export function renderAudio(items, stats, { fileUrl, onPlayState }) { + el.audioCount.textContent = String(items.length); + el.audioList.innerHTML = ''; + + if (!items.length) { + el.audioNote.textContent = + 'Nothing saved yet. Speak a line with the neural voice and its audio is ' + + 'kept here as a .wav, so it replays instantly with no synthesis.'; + return; + } + // Keep the note to one short line; the full path would overflow the panel, + // so it goes in the tooltip instead. + el.audioNote.textContent = + `${items.length} clip(s) · ${(stats.seconds || 0).toFixed(0)}s · ` + + `${((stats.bytes || 0) / 1048576).toFixed(1)} MB · audio_library/`; + el.audioNote.title = stats.dir || ''; + + for (const item of items) { + const li = document.createElement('li'); + li.className = 'aitem'; + + const button = document.createElement('button'); + button.className = 'aitem__play'; + button.innerHTML = PLAY_ICON; + button.title = 'Play in this browser'; + + // Played in the page, not through the backend: this is you auditioning a + // saved clip, not the robot speaking. Keeping them separate means the + // dashboard's Speaking state never lies. + const audio = new Audio(fileUrl(item.id)); + button.addEventListener('click', () => { + if (!audio.paused) { audio.pause(); audio.currentTime = 0; return; } + onPlayState(audio); + audio.play().catch(() => {}); + }); + audio.addEventListener('play', () => { button.dataset.playing = 'true'; button.innerHTML = STOP_ICON; }); + const reset = () => { button.dataset.playing = 'false'; button.innerHTML = PLAY_ICON; }; + audio.addEventListener('pause', reset); + audio.addEventListener('ended', reset); + + const body = document.createElement('div'); + body.className = 'aitem__body'; + const text = document.createElement('div'); + text.className = 'aitem__text'; + text.textContent = item.text; + text.title = item.text; + const meta = document.createElement('div'); + meta.className = 'aitem__meta'; + meta.textContent = `${(item.durationSeconds || 0).toFixed(1)}s · ${item.voice} · ${item.file}`; + meta.title = item.file; + body.append(text, meta); + + const download = document.createElement('a'); + download.className = 'aitem__dl'; + download.href = fileUrl(item.id); + download.download = item.file; + download.title = 'Download this .wav'; + download.innerHTML = DOWNLOAD_ICON; + + li.append(button, body, download); + el.audioList.append(li); + } +} diff --git a/frontend/styles/main.css b/frontend/styles/main.css new file mode 100644 index 0000000..267e24b --- /dev/null +++ b/frontend/styles/main.css @@ -0,0 +1,565 @@ +/* ========================================================================= + AGIBOT A3 · Voice Control + A dark, single-purpose control surface. Committed to one theme on purpose: + this is meant to be projected or shown on a stand next to the robot. + ========================================================================= */ + +:root { + /* surfaces */ + --bg: #080a0f; + --bg-glow-a: rgba(34, 211, 238, 0.10); + --bg-glow-b: rgba(59, 130, 246, 0.08); + --surface: #11151f; + --surface-2: #161b28; + --surface-3: #1c2230; + --border: rgba(255, 255, 255, 0.075); + --border-strong: rgba(255, 255, 255, 0.14); + + /* text */ + --text: #e8ecf4; + --text-dim: #9aa4bb; + --text-faint: #626c85; + + /* brand + state */ + --accent: #22d3ee; + --accent-deep: #0891b2; + --accent-soft: rgba(34, 211, 238, 0.14); + --ok: #34d399; + --warn: #fbbf24; + --danger: #f87171; + --busy: #818cf8; + + --radius: 14px; + --radius-sm: 10px; + --shadow: 0 1px 2px rgba(0,0,0,.4), 0 12px 32px rgba(0,0,0,.32); + + --font: "Inter", "Segoe UI Variable Text", "Segoe UI", system-ui, -apple-system, + "Helvetica Neue", Arial, sans-serif; + --mono: "JetBrains Mono", "Cascadia Mono", "SF Mono", Consolas, monospace; +} + +* { box-sizing: border-box; } + +/* Author `display` rules below outrank the UA stylesheet's [hidden], so restate + it with priority - otherwise .banner / .chip stay visible when hidden=true. */ +[hidden] { display: none !important; } + +html, body { height: 100%; } + +body { + margin: 0; + font-family: var(--font); + font-size: 15px; + line-height: 1.5; + color: var(--text); + background: + radial-gradient(1100px 620px at 12% -10%, var(--bg-glow-a), transparent 60%), + radial-gradient(900px 520px at 92% 0%, var(--bg-glow-b), transparent 55%), + var(--bg); + background-attachment: fixed; + -webkit-font-smoothing: antialiased; + text-rendering: optimizeLegibility; +} + +h1, h2 { margin: 0; font-weight: 600; letter-spacing: -0.015em; } +.mono { font-family: var(--mono); font-size: 13px; } + +/* ========================================================================= + HEADER + ========================================================================= */ +.topbar { + position: sticky; top: 0; z-index: 40; + display: flex; align-items: center; justify-content: space-between; gap: 16px; + padding: 14px clamp(16px, 4vw, 40px); + background: rgba(8, 10, 15, 0.72); + backdrop-filter: blur(14px); + -webkit-backdrop-filter: blur(14px); + border-bottom: 1px solid var(--border); +} + +.topbar__brand { display: flex; align-items: center; gap: 13px; min-width: 0; } + +.brand__mark { + display: grid; place-items: center; + width: 40px; height: 40px; flex: none; + border-radius: 11px; + color: var(--accent); + background: linear-gradient(160deg, var(--surface-3), var(--surface)); + border: 1px solid var(--border-strong); + box-shadow: inset 0 1px 0 rgba(255,255,255,.05); +} +.brand__mark svg { width: 23px; height: 23px; } + +.brand__text h1 { font-size: 16px; letter-spacing: 0.02em; } +.brand__text p { margin: 0; font-size: 12px; color: var(--text-faint); letter-spacing: .06em; text-transform: uppercase; } + +.mode-badge { + margin-left: 6px; padding: 3px 9px; + font-size: 10.5px; font-weight: 700; letter-spacing: .1em; text-transform: uppercase; + border-radius: 6px; + color: var(--warn); + background: rgba(251, 191, 36, .12); + border: 1px solid rgba(251, 191, 36, .3); + white-space: nowrap; +} +.mode-badge[data-mode="real"] { color: var(--ok); background: rgba(52,211,153,.12); border-color: rgba(52,211,153,.3); } + +.topbar__status { display: flex; align-items: center; gap: 10px; } + +/* connection pill */ +.conn { + display: inline-flex; align-items: center; gap: 9px; + padding: 7px 14px 7px 12px; + border-radius: 999px; + font-size: 13px; font-weight: 500; white-space: nowrap; + background: var(--surface-2); + border: 1px solid var(--border); + transition: border-color .25s, background .25s, color .25s; +} +.conn__dot { + width: 8px; height: 8px; border-radius: 50%; + background: var(--text-faint); + box-shadow: 0 0 0 0 transparent; + transition: background .25s; +} +.conn[data-state="connected"] { color: var(--ok); border-color: rgba(52,211,153,.32); background: rgba(52,211,153,.09); } +.conn[data-state="connected"] .conn__dot { background: var(--ok); box-shadow: 0 0 10px rgba(52,211,153,.75); } +.conn[data-state="connecting"] { color: var(--warn); border-color: rgba(251,191,36,.32); background: rgba(251,191,36,.09); } +.conn[data-state="connecting"] .conn__dot { background: var(--warn); animation: pulse 1.1s ease-in-out infinite; } +.conn[data-state="disconnected"] { color: var(--danger); border-color: rgba(248,113,113,.3); background: rgba(248,113,113,.08); } +.conn[data-state="disconnected"] .conn__dot { background: var(--danger); } +.conn[data-state="error"] { color: var(--danger); border-color: rgba(248,113,113,.45); background: rgba(248,113,113,.12); } +.conn[data-state="error"] .conn__dot { background: var(--danger); animation: pulse 1.4s ease-in-out infinite; } + +@keyframes pulse { + 0%, 100% { opacity: 1; transform: scale(1); } + 50% { opacity: .35; transform: scale(.82); } +} + +.chip { + display: inline-flex; align-items: center; gap: 6px; + padding: 6px 11px; border-radius: 999px; + font-family: var(--mono); font-size: 12px; color: var(--text-dim); + background: var(--surface-2); border: 1px solid var(--border); +} +.chip svg { width: 12px; height: 12px; color: var(--accent); } + +.icon-btn { + display: grid; place-items: center; + width: 36px; height: 36px; flex: none; + border-radius: 10px; cursor: pointer; + color: var(--text-dim); + background: var(--surface-2); + border: 1px solid var(--border); + transition: color .18s, border-color .18s, background .18s, transform .18s; +} +.icon-btn svg { width: 17px; height: 17px; } +.icon-btn:hover { color: var(--accent); border-color: var(--border-strong); background: var(--surface-3); } +.icon-btn:active { transform: scale(.94); } +.icon-btn[data-spin="true"] svg { animation: spin .9s linear infinite; } +@keyframes spin { to { transform: rotate(360deg); } } + +/* ========================================================================= + CONFIG BANNER + ========================================================================= */ +.banner { + display: flex; align-items: flex-start; gap: 11px; + margin: 16px clamp(16px, 4vw, 40px) 0; + padding: 13px 16px; + border-radius: var(--radius-sm); + font-size: 13.5px; + color: #fde68a; + background: rgba(251, 191, 36, .07); + border: 1px solid rgba(251, 191, 36, .28); +} +.banner[data-level="error"] { color: #fecaca; background: rgba(248,113,113,.08); border-color: rgba(248,113,113,.3); } +.banner svg { width: 17px; height: 17px; flex: none; margin-top: 1px; } +.banner__body { display: flex; flex-direction: column; gap: 2px; min-width: 0; } +.banner__body strong { font-weight: 650; } +.banner__body span { color: inherit; opacity: .88; } + +/* ========================================================================= + LAYOUT + ========================================================================= */ +.layout { + display: grid; + grid-template-columns: minmax(0, 1.65fr) minmax(300px, 0.85fr); + gap: 20px; + max-width: 1240px; + margin: 0 auto; + padding: 22px clamp(16px, 4vw, 40px) 48px; + align-items: start; +} +.sidebar { display: flex; flex-direction: column; gap: 20px; } + +.panel { + background: linear-gradient(180deg, var(--surface-2), var(--surface)); + border: 1px solid var(--border); + border-radius: var(--radius); + box-shadow: var(--shadow); + padding: 20px; +} +.panel--speech { padding: 22px; } + +.panel__head { + display: flex; align-items: baseline; justify-content: space-between; gap: 12px; + margin-bottom: 14px; +} +.panel__head h2 { font-size: 15px; letter-spacing: -0.01em; } +.panel__hint { font-size: 12.5px; color: var(--text-faint); } +.panel__note { + margin: 12px 0 0; font-size: 12.5px; color: var(--text-faint); line-height: 1.5; + overflow-wrap: anywhere; /* long file paths must not push the panel wide */ +} + +.count { + display: inline-block; margin-left: 6px; padding: 1px 7px; + font-size: 11px; font-weight: 600; color: var(--text-dim); + background: var(--surface-3); border-radius: 999px; border: 1px solid var(--border); +} + +/* ========================================================================= + COMPOSER + ========================================================================= */ +.composer { + border: 1px solid var(--border); + border-radius: var(--radius-sm); + background: #0d111a; + transition: border-color .2s, box-shadow .2s; +} +.composer:focus-within { + border-color: rgba(34, 211, 238, .45); + box-shadow: 0 0 0 3px var(--accent-soft); +} + +.composer__input { + display: block; width: 100%; + padding: 16px 17px 10px; + border: 0; outline: none; resize: vertical; + min-height: 150px; + font-family: inherit; font-size: 16.5px; line-height: 1.6; + color: var(--text); background: transparent; +} +.composer__input::placeholder { color: var(--text-faint); } + +.composer__meta { + display: flex; align-items: center; justify-content: space-between; gap: 12px; + padding: 9px 15px 11px; + font-size: 12px; color: var(--text-faint); + border-top: 1px solid var(--border); +} +.kbd-hint { display: inline-flex; align-items: center; gap: 4px; flex-wrap: wrap; } +kbd { + padding: 2px 6px; border-radius: 5px; + font-family: var(--mono); font-size: 10.5px; color: var(--text-dim); + background: var(--surface-3); border: 1px solid var(--border-strong); + box-shadow: 0 1px 0 rgba(0,0,0,.4); +} +.counter { font-family: var(--mono); font-size: 12px; white-space: nowrap; } +.counter b { color: var(--text-dim); font-weight: 500; } +.counter[data-level="warn"] b { color: var(--warn); } +.counter[data-level="over"] b { color: var(--danger); } + +/* ========================================================================= + ACTIONS + ========================================================================= */ +.actions { display: flex; gap: 10px; margin-top: 16px; flex-wrap: wrap; } + +.btn { + display: inline-flex; align-items: center; justify-content: center; gap: 9px; + padding: 0 20px; height: 46px; + font-family: inherit; font-size: 14.5px; font-weight: 550; + color: var(--text); + border-radius: 11px; border: 1px solid var(--border-strong); + background: var(--surface-3); + cursor: pointer; + transition: background .18s, border-color .18s, color .18s, transform .12s, opacity .18s; +} +.btn svg { width: 17px; height: 17px; } +.btn:hover:not(:disabled) { background: #232a3a; border-color: rgba(255,255,255,.2); } +.btn:active:not(:disabled) { transform: translateY(1px); } +.btn:disabled { opacity: .42; cursor: not-allowed; } +.btn--ghost { color: var(--text-dim); } +.btn--ghost:hover:not(:disabled) { color: var(--text); } + +/* the one button that matters */ +.btn--speak { + flex: 1 1 260px; + height: 54px; + padding: 0 30px; + font-size: 16px; font-weight: 650; letter-spacing: .01em; + color: #04222a; + border: 0; + background: linear-gradient(135deg, #67e8f9 0%, var(--accent) 45%, var(--accent-deep) 100%); + box-shadow: 0 6px 22px rgba(34, 211, 238, .26), inset 0 1px 0 rgba(255,255,255,.35); +} +.btn--speak:hover:not(:disabled) { + filter: brightness(1.07); + box-shadow: 0 8px 30px rgba(34, 211, 238, .36), inset 0 1px 0 rgba(255,255,255,.4); +} +.btn--speak:disabled { background: var(--surface-3); color: var(--text-faint); box-shadow: none; } +.btn__icon { display: inline-grid; place-items: center; } +.btn__spinner { display: none; width: 16px; height: 16px; border-radius: 50%; + border: 2px solid rgba(4,34,42,.28); border-top-color: #04222a; animation: spin .7s linear infinite; } + +.btn--speak[data-busy="true"] .btn__icon { display: none; } +.btn--speak[data-busy="true"] .btn__spinner { display: block; } + +/* ========================================================================= + PIPELINE + ========================================================================= */ +.pipeline { margin-top: 22px; } + +.pipeline__track { + position: relative; height: 3px; border-radius: 3px; + background: var(--surface-3); overflow: hidden; +} +.pipeline__track i { + position: absolute; inset: 0 100% 0 0; + background: linear-gradient(90deg, var(--accent-deep), var(--accent)); + transition: inset .45s cubic-bezier(.4, 0, .2, 1); +} +.pipeline[data-stage="failed"] .pipeline__track i, +.pipeline[data-stage="cancelled"] .pipeline__track i { background: var(--danger); } + +.pipeline__steps { + display: grid; grid-template-columns: repeat(4, 1fr); + margin: 11px 0 0; padding: 0; list-style: none; + font-size: 12.5px; color: var(--text-faint); +} +.pipeline__steps li { display: flex; align-items: center; gap: 7px; transition: color .3s; } +.pipeline__steps li:last-child { justify-content: flex-end; } +.pipeline__steps li:nth-child(2) { justify-content: center; } +.pipeline__steps li:nth-child(3) { justify-content: center; } +.pipeline__steps .dot { + width: 7px; height: 7px; border-radius: 50%; + background: var(--surface-3); border: 1px solid var(--border-strong); + transition: background .3s, border-color .3s, box-shadow .3s; +} +.pipeline__steps li[data-on="done"] { color: var(--text-dim); } +.pipeline__steps li[data-on="done"] .dot { background: var(--accent-deep); border-color: var(--accent-deep); } +.pipeline__steps li[data-on="active"] { color: var(--accent); font-weight: 550; } +.pipeline__steps li[data-on="active"] .dot { background: var(--accent); border-color: var(--accent); + box-shadow: 0 0 0 4px var(--accent-soft); } +.pipeline[data-stage="failed"] li[data-on="active"], +.pipeline[data-stage="cancelled"] li[data-on="active"] { color: var(--danger); } +.pipeline[data-stage="failed"] li[data-on="active"] .dot, +.pipeline[data-stage="cancelled"] li[data-on="active"] .dot { + background: var(--danger); border-color: var(--danger); box-shadow: 0 0 0 4px rgba(248,113,113,.16); +} + +/* ========================================================================= + LIVE STATUS + ========================================================================= */ +.live { + display: flex; align-items: center; gap: 11px; + margin-top: 18px; padding: 13px 16px; + border-radius: var(--radius-sm); + background: var(--surface); border: 1px solid var(--border); + font-size: 14px; + transition: border-color .3s, background .3s; +} +.live__text { font-weight: 550; } +.live__detail { color: var(--text-faint); font-size: 12.5px; font-family: var(--mono); margin-left: auto; } + +.live[data-tone="idle"] { color: var(--text-dim); } +.live[data-tone="busy"] { color: var(--accent); border-color: rgba(34,211,238,.3); background: rgba(34,211,238,.05); } +.live[data-tone="speaking"] { color: var(--accent); border-color: rgba(34,211,238,.45); background: rgba(34,211,238,.07); } +.live[data-tone="ok"] { color: var(--ok); border-color: rgba(52,211,153,.28); background: rgba(52,211,153,.05); } +.live[data-tone="error"] { color: var(--danger); border-color: rgba(248,113,113,.32); background: rgba(248,113,113,.06); } + +/* speaking waveform */ +.live__wave { display: none; align-items: flex-end; gap: 3px; height: 18px; } +.live[data-tone="speaking"] .live__wave { display: flex; } +.live__wave i { + display: block; width: 3px; border-radius: 2px; + background: currentColor; + animation: wave 900ms ease-in-out infinite; +} +.live__wave i:nth-child(1) { height: 7px; animation-delay: 0ms; } +.live__wave i:nth-child(2) { height: 14px; animation-delay: 120ms; } +.live__wave i:nth-child(3) { height: 18px; animation-delay: 240ms; } +.live__wave i:nth-child(4) { height: 11px; animation-delay: 360ms; } +.live__wave i:nth-child(5) { height: 6px; animation-delay: 480ms; } +@keyframes wave { + 0%, 100% { transform: scaleY(.35); opacity: .55; } + 50% { transform: scaleY(1); opacity: 1; } +} + +/* ========================================================================= + FACTS + ========================================================================= */ +.facts { margin: 0; display: flex; flex-direction: column; gap: 1px; } +.fact { + display: flex; align-items: center; justify-content: space-between; gap: 12px; + padding: 8px 0; + border-bottom: 1px solid rgba(255,255,255,.045); +} +.fact:last-child { border-bottom: 0; } +.fact dt { font-size: 12.5px; color: var(--text-faint); } +.fact dd { margin: 0; font-size: 13px; font-weight: 500; text-align: right; + overflow: hidden; text-overflow: ellipsis; white-space: nowrap; max-width: 62%; } +.fact dd[data-tone="ok"] { color: var(--ok); } +.fact dd[data-tone="warn"] { color: var(--warn); } +.fact dd[data-tone="error"] { color: var(--danger); } + +/* ========================================================================= + HISTORY + ========================================================================= */ +.link-btn { + padding: 3px 8px; border: 0; border-radius: 6px; cursor: pointer; + font-family: inherit; font-size: 12.5px; color: var(--text-faint); + background: transparent; transition: color .18s, background .18s; +} +.link-btn:hover { color: var(--danger); background: rgba(248,113,113,.09); } + +.history { + list-style: none; margin: 0; padding: 0; + display: flex; flex-direction: column; gap: 7px; + max-height: 340px; overflow-y: auto; + scrollbar-width: thin; scrollbar-color: var(--surface-3) transparent; +} +.history::-webkit-scrollbar { width: 7px; } +.history::-webkit-scrollbar-thumb { background: var(--surface-3); border-radius: 4px; } + +.hitem { + display: flex; align-items: flex-start; gap: 10px; + padding: 9px 11px; + border-radius: var(--radius-sm); + background: var(--surface); border: 1px solid var(--border); + cursor: pointer; + transition: border-color .16s, background .16s, transform .16s; +} +.hitem:hover { border-color: var(--border-strong); background: var(--surface-3); } +.hitem:active { transform: scale(.99); } +.hitem__bar { width: 3px; align-self: stretch; border-radius: 2px; background: var(--text-faint); flex: none; } +.hitem[data-ok="true"] .hitem__bar { background: var(--ok); } +.hitem[data-ok="false"] .hitem__bar { background: var(--danger); } +.hitem[data-ok="pending"] .hitem__bar { background: var(--accent); animation: pulse 1s ease-in-out infinite; } + +.hitem__body { min-width: 0; flex: 1; } +.hitem__text { + font-size: 13px; line-height: 1.45; color: var(--text); + display: -webkit-box; -webkit-line-clamp: 2; -webkit-box-orient: vertical; overflow: hidden; +} +.hitem__meta { + display: flex; align-items: center; gap: 8px; margin-top: 4px; + font-family: var(--mono); font-size: 11px; color: var(--text-faint); +} +.hitem__meta .err { color: var(--danger); font-family: var(--font); } + +.empty { margin: 4px 0 0; font-size: 12.5px; color: var(--text-faint); line-height: 1.55; } + +/* ========================================================================= + TOASTS + ========================================================================= */ +.toasts { + position: fixed; right: 20px; bottom: 20px; z-index: 90; + display: flex; flex-direction: column; gap: 9px; + max-width: min(380px, calc(100vw - 40px)); +} +.toast { + display: flex; align-items: flex-start; gap: 10px; + padding: 12px 14px; + border-radius: var(--radius-sm); + font-size: 13.5px; line-height: 1.45; + color: var(--text); + background: var(--surface-2); + border: 1px solid var(--border-strong); + box-shadow: var(--shadow); + animation: toast-in .25s cubic-bezier(.2, .8, .3, 1); +} +.toast[data-tone="error"] { border-color: rgba(248,113,113,.42); background: #1d1316; } +.toast[data-tone="ok"] { border-color: rgba(52,211,153,.38); background: #0f1a17; } +.toast__dot { width: 7px; height: 7px; border-radius: 50%; margin-top: 6px; flex: none; background: var(--text-faint); } +.toast[data-tone="error"] .toast__dot { background: var(--danger); } +.toast[data-tone="ok"] .toast__dot { background: var(--ok); } +.toast.is-out { animation: toast-out .2s ease-in forwards; } + +@keyframes toast-in { from { opacity: 0; transform: translateY(10px) scale(.97); } } +@keyframes toast-out { to { opacity: 0; transform: translateY(6px) scale(.98); } } + +/* ========================================================================= + RESPONSIVE + ========================================================================= */ +@media (max-width: 980px) { + .layout { grid-template-columns: 1fr; } + .sidebar { order: 2; } +} +@media (max-width: 620px) { + .topbar { flex-wrap: wrap; gap: 10px; } + .brand__text p { display: none; } + .panel__hint { display: none; } + .actions { flex-direction: column; } + .btn { width: 100%; } + .pipeline__steps { font-size: 11.5px; } + .toasts { right: 12px; left: 12px; bottom: 12px; max-width: none; } +} + +@media (prefers-reduced-motion: reduce) { + *, *::before, *::after { animation-duration: .001ms !important; transition-duration: .001ms !important; } +} + + +/* ========================================================================= + SAVED AUDIO + ========================================================================= */ +.audio-list { + list-style: none; margin: 12px 0 0; padding: 0; + display: flex; flex-direction: column; gap: 6px; + max-height: 260px; overflow-y: auto; + scrollbar-width: thin; scrollbar-color: var(--surface-3) transparent; +} +.audio-list::-webkit-scrollbar { width: 7px; } +.audio-list::-webkit-scrollbar-thumb { background: var(--surface-3); border-radius: 4px; } + +.aitem { + display: flex; align-items: center; gap: 9px; + padding: 7px 9px; + border-radius: var(--radius-sm); + background: var(--surface); border: 1px solid var(--border); + transition: border-color .16s, background .16s; +} +.aitem:hover { border-color: var(--border-strong); background: var(--surface-3); } + +.aitem__play { + display: grid; place-items: center; + width: 26px; height: 26px; flex: none; + border-radius: 50%; cursor: pointer; + color: #04222a; background: var(--accent); border: 0; + transition: filter .15s, transform .12s; +} +.aitem__play svg { width: 12px; height: 12px; } +.aitem__play:hover { filter: brightness(1.1); } +.aitem__play:active { transform: scale(.92); } +.aitem__play[data-playing="true"] { background: var(--warn); } + +.aitem__body { min-width: 0; flex: 1; } +.aitem__text { + font-size: 12.5px; line-height: 1.4; color: var(--text); + overflow: hidden; text-overflow: ellipsis; white-space: nowrap; +} +.aitem__meta { + font-family: var(--mono); font-size: 10.5px; color: var(--text-faint); margin-top: 2px; + overflow: hidden; text-overflow: ellipsis; white-space: nowrap; +} +.aitem__dl { + flex: none; display: grid; place-items: center; + width: 24px; height: 24px; border-radius: 6px; + color: var(--text-faint); text-decoration: none; + transition: color .15s, background .15s; +} +.aitem__dl svg { width: 13px; height: 13px; } +.aitem__dl:hover { color: var(--accent); background: var(--surface-3); } + +/* replay marker on history rows whose audio is already saved */ +.hitem__replay { + flex: none; display: grid; place-items: center; + width: 24px; height: 24px; border-radius: 50%; + border: 0; cursor: pointer; + color: var(--accent); background: var(--accent-soft); + transition: filter .15s, transform .12s; +} +.hitem__replay svg { width: 11px; height: 11px; } +.hitem__replay:hover { filter: brightness(1.25); } +.hitem__replay:active { transform: scale(.9); } diff --git a/pronunciation.json b/pronunciation.json new file mode 100644 index 0000000..bd00290 --- /dev/null +++ b/pronunciation.json @@ -0,0 +1,6 @@ +{ + "_comment": "How the SIMULATOR should pronounce tricky words. Edit freely, then restart. The real robot never sees these - it always gets your original text. Matching is case-insensitive, on whole words only.", + + "AGIBOT": "A G I bot", + "AgiBot": "A G I bot" +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..fee31aa --- /dev/null +++ b/requirements.txt @@ -0,0 +1,28 @@ +# AGIBOT A3 - Voice Control : Python dependencies +# Install with: pip install -r requirements.txt + +# --- core (always needed) --------------------------------------------------- +fastapi>=0.110 +uvicorn[standard]>=0.27 +python-dotenv>=1.0 +pydantic>=2.5 + +# --- transports ------------------------------------------------------------- +httpx>=0.27 # HTTP/REST transport (keep-alive client) +websockets>=12.0 # WebSocket transport + uvicorn ws support + +# --- simulator audio (MOCK_LOCAL_AUDIO=true) -------------------------------- +# Lets the simulator speak through this PC's speakers. Windows only; on other +# systems the app uses built-in commands and needs nothing installed. +comtypes>=1.2 ; sys_platform == "win32" + +# --- optional --------------------------------------------------------------- +# paramiko>=3.4 # A3_TRANSPORT=ssh with password authentication +# +# Simulator audio without comtypes: Windows falls back to PowerShell's +# System.Speech (always present, just slower to start). macOS uses `say`. +# Linux needs: sudo apt install espeak-ng +# +# ROS 2 transport (A3_TRANSPORT=ros2) needs `rclpy`, which is NOT installed via +# pip - it comes from a ROS 2 installation. Run the app from a sourced ROS 2 +# environment if you use that transport. diff --git a/scripts/discover_robot.py b/scripts/discover_robot.py new file mode 100644 index 0000000..5edabe5 --- /dev/null +++ b/scripts/discover_robot.py @@ -0,0 +1,303 @@ +"""Probe an AGIBOT A3 to find out what it actually exposes. + +Run this the moment you have the robot's IP, before touching .env: + + python scripts/discover_robot.py 192.168.1.50 + +It is READ-ONLY by default: it pings, checks the ports AgiBot documents, and asks +each one for a deliberately nonexistent route. A **404 proves an AimRT HTTP +server is listening** on that port; connection-refused proves it is not. Nothing +is played and nothing is changed. + +To run the decisive end-to-end test - which makes the robot actually talk: + + python scripts/discover_robot.py 192.168.1.50 --speak "Hello, I am Expedition A3" + +Port and route expectations come from AgiBot's A3 developer guide: + https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play + https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/03-second_develop_interface_overview + +They are documented facts for A3 v3.1/v3.2 firmware, not guarantees: AgiBot does +not promise port stability, and AimRT has no service-discovery endpoint. Anything +this script cannot confirm, it reports as unknown rather than guessing. +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import platform +import subprocess +import sys +import time +from typing import Any, Dict, List, Optional, Tuple + +try: + import httpx +except ImportError: # pragma: no cover + print("Install dependencies first: pip install -r requirements.txt") + raise SystemExit(2) + +def _use_utf8_console() -> None: + """Windows consoles default to a legacy code page; the robot speaks Chinese.""" + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + +_use_utf8_console() + + +# Ports with published evidence. Audio ones first; the motion/mapping ports are +# probed only so the report is an honest map - they have nothing to do with TTS. +PORTS: List[Tuple[int, str, str]] = [ + (59301, "TTS / agent RPC (HDU)", "audio"), + (56666, "HalAudioService - volume, PlayFile (HDU)", "audio"), + (51049, "ResourceService - audio resources (HDU)", "audio"), + (56444, "MotionCommandService (MDU)", "other"), + (56322, "motion control (MDU)", "other"), + (22, "SSH", "other"), + (8080, "common HTTP alternative", "other"), +] + +TTS_SERVICE = "aimdk.protocol.TTSService" +NONEXISTENT_ROUTE = "/rpc/does.not.Exist/Nope" + +OK = " [ OK ]" +NO = " [ -- ]" +WARN = " [ !! ]" + + +def head(title: str) -> None: + print("\n" + title) + print(" " + "-" * (len(title) + 2)) + + +# --------------------------------------------------------------------------- # +# 1. ICMP +# --------------------------------------------------------------------------- # +def ping(host: str) -> Optional[float]: + flag = "-n" if platform.system().lower().startswith("win") else "-c" + started = time.perf_counter() + try: + result = subprocess.run( + ["ping", flag, "2", host], + stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=12, + ) + except Exception: + return None + if result.returncode != 0: + return None + return (time.perf_counter() - started) * 1000 / 2 + + +# --------------------------------------------------------------------------- # +# 2. TCP +# --------------------------------------------------------------------------- # +async def tcp_open(host: str, port: int, timeout: float = 2.0) -> Optional[float]: + started = time.perf_counter() + try: + fut = asyncio.open_connection(host, port) + _, writer = await asyncio.wait_for(fut, timeout=timeout) + writer.close() + try: + await writer.wait_closed() + except Exception: + pass + return (time.perf_counter() - started) * 1000 + except Exception: + return None + + +# --------------------------------------------------------------------------- # +# 3. Is it an AimRT HTTP server? +# --------------------------------------------------------------------------- # +async def aimrt_probe(client: "httpx.AsyncClient", host: str, port: int) -> Dict[str, Any]: + """Ask for a route that cannot exist. + + AimRT answers 404 for an unknown route, which is a positive identification + without invoking anything. + """ + url = "http://{0}:{1}{2}".format(host, port, NONEXISTENT_ROUTE) + try: + response = await client.post(url, json={}, headers={"Content-Type": "application/json"}) + except Exception as exc: + return {"reachable": False, "detail": type(exc).__name__} + return { + "reachable": True, + "status": response.status_code, + "aimrt": response.status_code in (404, 500), + "cors": response.headers.get("access-control-allow-origin"), + "server": response.headers.get("server"), + "body": response.text[:160], + } + + +# --------------------------------------------------------------------------- # +# 4. The decisive test +# --------------------------------------------------------------------------- # +async def try_speak(client: "httpx.AsyncClient", host: str, port: int, text: str) -> Dict[str, Any]: + url = "http://{0}:{1}/rpc/{2}/PlayTTS".format(host, port, TTS_SERVICE) + payload = { + "text": text, + "priority_level": "INTERACTION_L6", + "domain": "discovery_probe", + "trace_id": "probe{0}".format(int(time.time())), + "is_interrupted": True, + } + started = time.perf_counter() + try: + response = await client.post(url, json=payload, headers={"Content-Type": "application/json"}) + except Exception as exc: + return {"ok": False, "error": "{0}: {1}".format(type(exc).__name__, exc)} + + elapsed = (time.perf_counter() - started) * 1000 + try: + body = response.json() + except Exception: + body = {"raw": response.text[:300]} + success = None + if isinstance(body, dict): + success = body.get("is_sucess", body.get("is_success")) + return { + "ok": response.status_code == 200 and success is not False, + "status": response.status_code, + "ms": round(elapsed, 1), + "success_flag": success, + "trace_id": body.get("trace_id") if isinstance(body, dict) else None, + "sent_trace": payload["trace_id"], + "has_header_envelope": isinstance(body, dict) and isinstance(body.get("header"), dict), + "body": body, + } + + +# --------------------------------------------------------------------------- # +# main +# --------------------------------------------------------------------------- # +async def run(host: str, text: Optional[str], tts_port: int) -> int: + print("\n" + "=" * 72) + print(" AGIBOT A3 discovery -> {0}".format(host)) + print("=" * 72) + + # ---- 1. ping ----------------------------------------------------------- + head("1. Reachability") + rtt = ping(host) + if rtt is None: + print(NO + " ping got no reply.") + print(" Not conclusive - some robots drop ICMP but still serve their ports.") + else: + print(OK + " ping replies (~{0:.0f} ms round trip)".format(rtt)) + + # ---- 2. ports ---------------------------------------------------------- + head("2. Ports") + open_ports: List[int] = [] + results = await asyncio.gather(*[tcp_open(host, port) for port, _, _ in PORTS]) + for (port, label, kind), elapsed in zip(PORTS, results): + if elapsed is None: + print("{0} {1:<6} closed / filtered {2}".format(NO, port, label)) + else: + open_ports.append(port) + print("{0} {1:<6} OPEN ({2:.0f} ms) {3}".format(OK, port, elapsed, label)) + + if not open_ports: + print("\n" + WARN + " Nothing answered. Check the IP, that the robot has finished") + print(" booting, and that both machines are on the same subnet.") + print(" See docs/NETWORK.md.") + return 1 + + # ---- 3. AimRT identification ------------------------------------------ + head("3. Which ports speak AimRT HTTP JSON-RPC?") + aimrt_ports: List[int] = [] + async with httpx.AsyncClient(timeout=httpx.Timeout(5.0, connect=3.0)) as client: + for port in open_ports: + if port == 22: + continue + info = await aimrt_probe(client, host, port) + if not info.get("reachable"): + print("{0} {1:<6} no HTTP response ({2})".format(NO, port, info.get("detail"))) + continue + if info.get("aimrt"): + aimrt_ports.append(port) + extra = "" + if info.get("cors"): + extra = " CORS: {0}".format(info["cors"]) + print("{0} {1:<6} AimRT RPC server (HTTP {2} for an unknown route){3}".format( + OK, port, info["status"], extra)) + else: + print("{0} {1:<6} HTTP {2}, but not an AimRT RPC route".format( + NO, port, info.get("status"))) + + # ---- 4. summary / speak ------------------------------------------- + head("4. Verdict") + if tts_port in aimrt_ports: + print(OK + " Port {0} looks like the A3 TTS RPC service.".format(tts_port)) + print("") + print(" Put this in .env:") + print(" ROBOT_MODE=real") + print(" ROBOT_IP={0}".format(host)) + print(" ROBOT_PORT={0}".format(tts_port)) + print(" A3_TRANSPORT=aimdk") + elif aimrt_ports: + print(WARN + " AimRT RPC found on {0}, but not on the documented TTS port {1}.".format( + ", ".join(str(p) for p in aimrt_ports), tts_port)) + print(" Try each with --speak, and set ROBOT_PORT to whichever talks.") + else: + print(WARN + " No AimRT RPC server found.") + print(" Most likely the RPC ports are bound only to the robot's internal") + print(" 10.42.10.x network. SSH in and check: ss -ltnp | grep 59301") + print(" See docs/AGIBOT_A3_INTEGRATION.md section 6.") + + if text: + head("5. Speech test (the robot should talk now)") + print(' Sending: "{0}"'.format(text)) + result = await try_speak(client, host, tts_port, text) + if result.get("ok"): + print(OK + " Accepted in {0} ms.".format(result["ms"])) + print(" is_sucess : {0}".format(result["success_flag"])) + print(" trace_id sent : {0}".format(result["sent_trace"])) + print(" trace_id back : {0}".format(result["trace_id"])) + if result["trace_id"] and result["trace_id"] != result["sent_trace"]: + print(" (differs, as documented - Stop must use the returned id)") + if result["has_header_envelope"]: + print(" header envelope: present") + print("") + print(" If you heard the robot, the integration is confirmed.") + print(" If it was silent, check the robot's volume and whether its") + print(" TTS needs internet access - see docs/AGIBOT_A3_INTEGRATION.md section 7.") + else: + print(WARN + " Speech request failed.") + for key in ("status", "error", "success_flag"): + if result.get(key) is not None: + print(" {0:<14}: {1}".format(key, result[key])) + if result.get("status") == 404: + print(" 404 = the port is right but the route is wrong. Check the") + print(" service/method names against your firmware's documentation.") + if result.get("body"): + print(" body : {0}".format( + json.dumps(result["body"], ensure_ascii=False)[:300])) + return 1 + + print("\n" + "=" * 72 + "\n") + return 0 + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Probe an AGIBOT A3 for its speech interface.", + epilog="Read-only unless --speak is given.", + ) + parser.add_argument("ip", help="The robot's IP address on your network.") + parser.add_argument("--speak", metavar="TEXT", default=None, + help="Also send this text - THE ROBOT WILL TALK.") + parser.add_argument("--port", type=int, default=59301, + help="TTS RPC port (default: 59301, AgiBot's documented port).") + args = parser.parse_args() + raise SystemExit(asyncio.run(run(args.ip, args.speak, args.port))) + + +if __name__ == "__main__": + main() diff --git a/scripts/fake_a3_server.py b/scripts/fake_a3_server.py new file mode 100644 index 0000000..402e0af --- /dev/null +++ b/scripts/fake_a3_server.py @@ -0,0 +1,194 @@ +"""A stand-in for the AGIBOT A3's speech RPC, for testing the REAL adapter path. + +The mock robot (ROBOT_MODE=mock) tests the app. This tests the *wire protocol* - +it speaks the interface AgiBot documents for the A3, so you can run the app in +ROBOT_MODE=real against 127.0.0.1 and exercise `aimdk_transport.py` itself: +the URL shape, the JSON body, the trace_id round trip, chunking, Stop, and the +error paths. When the real robot arrives, only ROBOT_IP changes. + +It mirrors the documented contract, including its quirks: + * route POST /rpc//, Content-Type: application/json + * success flag `is_sucess` (one 'c' - as printed in AgiBot's docs) + * trace_id the reply appends a random suffix to the one you sent + * size limit 1024 bytes of UTF-8 on `text` + * unknown route 404; RPC-level failure 500 + +Docs: https://open.agibot.com/docs/en/aimdk/a3/v3_2/dev_guide/07-02-audio_play + +Run: + python scripts/fake_a3_server.py # listens on 127.0.0.1:59301 + +Then in .env: + ROBOT_MODE=real + ROBOT_IP=127.0.0.1 + ROBOT_PORT=59301 + +NOTE: this is a TEST DOUBLE written from public documentation. It is not the +robot, and passing against it proves the client is well-formed - not that the +robot's firmware behaves identically. Verify against the real unit. +""" + +from __future__ import annotations + +import argparse +import json +import secrets +import sys +import time +from typing import Any, Dict, Optional + +try: + import uvicorn + from fastapi import FastAPI, Request, Response + from fastapi.responses import JSONResponse +except ImportError: # pragma: no cover + print("Install dependencies first: pip install -r requirements.txt") + raise SystemExit(2) + +def _use_utf8_console() -> None: + """Windows consoles default to a legacy code page; the robot speaks Chinese.""" + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + +_use_utf8_console() + + +SERVICE = "aimdk.protocol.TTSService" +MAX_BYTES = 1024 + +app = FastAPI(title="Fake AGIBOT A3 speech RPC", docs_url=None, openapi_url=None) + +STATE: Dict[str, Any] = {"utterances": {}, "active": None, "count": 0} + + +def _log(kind: str, message: str) -> None: + print(" {0} {1:<9} {2}".format(time.strftime("%H:%M:%S"), kind, message), flush=True) + + +@app.post("/rpc/{service}/{method}") +async def rpc(service: str, method: str, request: Request) -> Response: + if service != SERVICE: + _log("404", "unknown service {0}".format(service)) + return JSONResponse({"error": "no such service"}, status_code=404) + + # AimRT requires this header and rejects anything else. + content_type = (request.headers.get("content-type") or "").split(";")[0].strip() + if content_type != "application/json": + _log("500", "bad Content-Type: {0!r}".format(content_type)) + return JSONResponse({"error": "unsupported content type"}, status_code=500) + + try: + body = json.loads(await request.body() or b"{}") + except json.JSONDecodeError: + return JSONResponse({"error": "malformed json"}, status_code=500) + + handler = { + "PlayTTS": _play_tts, + "StopTTSTraceId": _stop_tts, + "GetAudioStatus": _get_status, + }.get(method) + + if handler is None: + _log("404", "unknown method {0}".format(method)) + return JSONResponse({"error": "no such method"}, status_code=404) + + return handler(body) + + +def _play_tts(body: Dict[str, Any]) -> Response: + text = body.get("text") or "" + size = len(text.encode("utf-8")) + + if size > MAX_BYTES: + _log("REJECT", "text is {0} bytes (limit {1})".format(size, MAX_BYTES)) + return JSONResponse(_reply(body, False, "text exceeds 1024 bytes"), status_code=200) + if not text.strip(): + _log("REJECT", "empty text") + return JSONResponse(_reply(body, False, "empty text"), status_code=200) + + trace = "{0}_{1}".format(body.get("trace_id") or "trace", secrets.token_urlsafe(16)) + STATE["utterances"][trace] = {"text": text, "started": time.time(), "stopped": False} + STATE["active"] = trace + STATE["count"] += 1 + + _log("SPEAK", '"{0}" [{1} bytes, priority={2}, interrupt={3}]'.format( + text if len(text) <= 70 else text[:67] + "...", + size, body.get("priority_level"), body.get("is_interrupted"), + )) + return JSONResponse(_reply(body, True, "", trace), status_code=200) + + +def _stop_tts(body: Dict[str, Any]) -> Response: + trace = body.get("trace_id") + entry = STATE["utterances"].get(trace) + if entry is None: + _log("STOP", "unknown trace_id {0!r}".format(trace)) + return JSONResponse({"is_sucess": False, "error_message": "unknown trace_id"}, status_code=200) + entry["stopped"] = True + if STATE["active"] == trace: + STATE["active"] = None + _log("STOP", "stopped {0}".format(trace)) + return JSONResponse({"is_sucess": True, "error_message": "", "trace_id": trace}, status_code=200) + + +def _get_status(body: Dict[str, Any]) -> Response: + trace = body.get("trace_id") + entry = STATE["utterances"].get(trace) + if entry is None: + status = "TTSStatusType_NOTInQue" + elif entry["stopped"]: + status = "TTSStatusType_Stop" + elif time.time() - entry["started"] < _estimate(entry["text"]): + status = "TTSStatusType_Playing" + else: + status = "TTSStatusType_End" + return JSONResponse({"trace_id": trace, "tts_status": status, "is_sucess": True}, status_code=200) + + +def _reply(body: Dict[str, Any], ok: bool, error: str, trace: Optional[str] = None) -> Dict[str, Any]: + return { + "text": body.get("text", ""), + "priority_level": body.get("priority_level", ""), + "priority_weight": 0, + "domain": body.get("domain", ""), + "trace_id": trace or body.get("trace_id", ""), + "is_sucess": ok, # the documented spelling - do not "fix" it + "error_message": error, + "estimated_duration": 0, + } + + +def _estimate(text: str) -> float: + words = max(1, len(text.split())) + return max(1.0, words / 150 * 60) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Fake AGIBOT A3 speech RPC endpoint.") + parser.add_argument("--host", default="127.0.0.1") + parser.add_argument("--port", type=int, default=59301) + args = parser.parse_args() + + print("") + print(" Fake AGIBOT A3 speech RPC") + print(" " + "-" * 52) + print(" Listening : http://{0}:{1}".format(args.host, args.port)) + print(" Endpoint : POST /rpc/{0}/PlayTTS".format(SERVICE)) + print("") + print(" Point the app at it with:") + print(" ROBOT_MODE=real") + print(" ROBOT_IP={0}".format(args.host)) + print(" ROBOT_PORT={0}".format(args.port)) + print("") + print(" This is a TEST DOUBLE built from public docs - not the robot.") + print("") + uvicorn.run(app, host=args.host, port=args.port, log_level="warning", access_log=False) + + +if __name__ == "__main__": + main() diff --git a/scripts/selftest.py b/scripts/selftest.py new file mode 100644 index 0000000..0b2b8d4 --- /dev/null +++ b/scripts/selftest.py @@ -0,0 +1,267 @@ +"""End-to-end self test. + +Exercises the whole stack the way the browser does - REST for commands, WebSocket +for the lifecycle - and prints a pass/fail report. + + 1. start the server: python backend/main.py + 2. in another window: python scripts/selftest.py + +Works against mock mode out of the box. Against a real robot it also works, but +the robot will actually speak, so only run it when that is fine. +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import sys +import time +from typing import Any, Dict, List, Optional + +try: + import httpx +except ImportError: # pragma: no cover + print("Install dependencies first: pip install -r requirements.txt") + raise SystemExit(2) + +try: + from websockets.asyncio.client import connect as ws_connect +except Exception: # pragma: no cover + from websockets.client import connect as ws_connect # type: ignore + +def _use_utf8_console() -> None: + """Windows consoles default to a legacy code page; the robot speaks Chinese.""" + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + +_use_utf8_console() + + +PASS = " [PASS]" +FAIL = " [FAIL]" +INFO = " [ .. ]" + +results: List[tuple] = [] + + +def check(name: str, condition: bool, detail: str = "") -> bool: + results.append((name, condition, detail)) + print("{0} {1}{2}".format(PASS if condition else FAIL, name, " - " + detail if detail else "")) + return condition + + +class Listener: + """Collects WebSocket events in the background, like the dashboard does.""" + + def __init__(self, url: str) -> None: + self.url = url + self.events: List[Dict[str, Any]] = [] + self.hello: Optional[Dict[str, Any]] = None + self._ws: Any = None + self._task: Optional[asyncio.Task] = None + + async def __aenter__(self) -> "Listener": + self._ws = await ws_connect(self.url) + self._task = asyncio.create_task(self._read()) + for _ in range(50): + if self.hello is not None: + break + await asyncio.sleep(0.05) + return self + + async def __aexit__(self, *exc: Any) -> None: + if self._task: + self._task.cancel() + if self._ws: + await self._ws.close() + + async def _read(self) -> None: + async for raw in self._ws: + message = json.loads(raw) + if message.get("type") == "hello": + self.hello = message["data"] + self.events.append(message) + + def stages(self, request_id: Optional[str] = None) -> List[str]: + out = [] + for event in self.events: + if event.get("type") != "speech.progress": + continue + data = event["data"] + if request_id and data.get("requestId") != request_id: + continue + out.append(data["stage"]) + return out + + async def wait_for_event(self, event_type: str, timeout: float = 10.0) -> bool: + """Wait for the next event of a type to arrive (ignoring earlier ones).""" + seen = sum(1 for e in self.events if e.get("type") == event_type) + deadline = time.time() + timeout + while time.time() < deadline: + if sum(1 for e in self.events if e.get("type") == event_type) > seen: + return True + await asyncio.sleep(0.1) + return False + + async def wait_for_stage(self, stage: str, request_id: str, timeout: float = 30.0) -> bool: + deadline = time.time() + timeout + while time.time() < deadline: + if stage in self.stages(request_id): + return True + await asyncio.sleep(0.05) + return False + + +async def run(base: str) -> int: + ws_url = base.replace("http://", "ws://").replace("https://", "wss://") + "/ws" + + async with httpx.AsyncClient(base_url=base, timeout=20.0) as client: + print("\n=== 1. Server & configuration " + "=" * 40) + try: + health = (await client.get("/api/health")).json() + except Exception as exc: + print(FAIL + " server unreachable at {0} ({1})".format(base, exc)) + print("\n Start it first: python backend/main.py\n") + return 1 + check("GET /api/health returns ok", health.get("status") == "ok", "mode=" + str(health.get("mode"))) + + config = (await client.get("/api/config")).json() + check("GET /api/config returns a mode", config.get("mode") in ("mock", "real")) + check("no blocking config errors", + not [i for i in config.get("issues", []) if i["level"] == "error"], + str(config.get("issues"))) + + print("\n=== 2. Dashboard is served " + "=" * 43) + page = await client.get("/") + check("GET / serves the dashboard", page.status_code == 200 and "AGIBOT" in page.text) + for asset in ("/styles/main.css", "/js/app.js", "/js/api.js", "/js/socket.js", "/js/ui.js"): + response = await client.get(asset) + check("GET {0}".format(asset), response.status_code == 200) + + print("\n=== 3. Robot connection " + "=" * 46) + status = None + for _ in range(40): # allow the supervisor a moment to connect + status = (await client.get("/api/robot/status")).json() + if status.get("connected"): + break + await asyncio.sleep(0.25) + check("robot reports connected", bool(status and status.get("connected")), + "state=" + str(status.get("state") if status else "?")) + + async with Listener(ws_url) as listener: + check("WebSocket accepts connection and sends hello", listener.hello is not None) + if listener.hello: + check("hello carries config + status + history", + all(k in listener.hello for k in ("config", "status", "history"))) + + print("\n=== 4. Speak (the main workflow) " + "=" * 37) + text = "Hello, welcome to our company." + started = time.perf_counter() + response = await client.post("/api/robot/speak", json={"text": text}) + http_ms = (time.perf_counter() - started) * 1000 + check("POST /api/robot/speak accepted", response.status_code == 200, + "HTTP {0}".format(response.status_code)) + if response.status_code != 200: + print(" body:", response.text[:300]) + return 1 + + body = response.json() + request_id = body["requestId"] + check("response reports success", body.get("success") is True, json.dumps(body)) + check("HTTP call returns before speech ends (< 3s)", http_ms < 3000, + "{0:.0f} ms".format(http_ms)) + check("ack latency reported", isinstance(body.get("ackLatencyMs"), int), + "{0} ms".format(body.get("ackLatencyMs"))) + + got_speaking = await listener.wait_for_stage("speaking", request_id, timeout=10) + check("lifecycle reaches 'speaking'", got_speaking) + got_completed = await listener.wait_for_stage("completed", request_id, timeout=60) + check("lifecycle reaches 'completed'", got_completed) + stages = listener.stages(request_id) + check("full stage sequence observed", + ["sending", "processing", "speaking", "completed"] == [ + s for s in stages if s in ("sending", "processing", "speaking", "completed") + ], + " -> ".join(stages)) + + print("\n=== 5. Stop " + "=" * 58) + long_text = "This is a much longer sentence used to verify that the stop button " \ + "interrupts an utterance while the robot is still speaking it out loud." + body2 = (await client.post("/api/robot/speak", json={"text": long_text})).json() + rid2 = body2["requestId"] + await listener.wait_for_stage("speaking", rid2, timeout=10) + stop_response = await client.post("/api/robot/stop") + check("POST /api/robot/stop accepted", stop_response.status_code == 200) + cancelled = await listener.wait_for_stage("cancelled", rid2, timeout=10) + check("utterance reports 'cancelled'", cancelled, " -> ".join(listener.stages(rid2))) + + print("\n=== 6. Error handling " + "=" * 48) + empty = await client.post("/api/robot/speak", json={"text": " "}) + check("empty text rejected with 400", empty.status_code == 400, + "HTTP {0} {1}".format(empty.status_code, empty.text[:120])) + + too_long = await client.post( + "/api/robot/speak", json={"text": "x " * (config.get("maxLength", 1000) + 50)} + ) + check("over-long text rejected with 400", too_long.status_code == 400, + "HTTP {0}".format(too_long.status_code)) + + missing = await client.post("/api/robot/speak", json={}) + check("malformed body rejected", missing.status_code in (400, 422), + "HTTP {0}".format(missing.status_code)) + + print("\n=== 7. History " + "=" * 55) + history = (await client.get("/api/speech/history")).json() + check("history contains the utterances", history.get("count", 0) >= 2, + "count={0}".format(history.get("count"))) + first = history["items"][0] if history.get("items") else {} + check("history entries carry stage + latency", + "stage" in first and "ackLatencyMs" in first) + + cleared = (await client.delete("/api/speech/history")).json() + check("history cleared", cleared.get("success") is True, + "removed={0}".format(cleared.get("removed"))) + after = (await client.get("/api/speech/history")).json() + check("history is empty afterwards", after.get("count") == 0) + + print("\n=== 8. Diagnostics & recovery " + "=" * 40) + diagnostics = (await client.get("/api/robot/diagnostics")).json() + check("GET /api/robot/diagnostics works", "status" in diagnostics and "adapter" in diagnostics) + reconnect = (await client.post("/api/robot/reconnect")).json() + check("POST /api/robot/reconnect works", reconnect.get("success") is True) + + # Status is pushed once per health interval, so wait for one rather + # than assuming the earlier checks took long enough to see it. + budget = float(config.get("healthInterval") or 5.0) + 4.0 + saw_status_event = await listener.wait_for_event("robot.status", timeout=budget) + check("robot.status events pushed over WebSocket", saw_status_event, + "waited up to {0:.0f}s".format(budget)) + + passed = sum(1 for _, ok, _ in results if ok) + total = len(results) + print("\n" + "=" * 72) + print(" {0}/{1} checks passed".format(passed, total)) + if passed != total: + print("\n Failures:") + for name, ok, detail in results: + if not ok: + print(" - {0} {1}".format(name, detail)) + print("=" * 72 + "\n") + return 0 if passed == total else 1 + + +def main() -> None: + parser = argparse.ArgumentParser(description="End-to-end test for the A3 voice dashboard.") + parser.add_argument("--url", default="http://127.0.0.1:8000", help="Base URL of the running app.") + args = parser.parse_args() + raise SystemExit(asyncio.run(run(args.url.rstrip("/")))) + + +if __name__ == "__main__": + main() diff --git a/scripts/voices.py b/scripts/voices.py new file mode 100644 index 0000000..055bc3c --- /dev/null +++ b/scripts/voices.py @@ -0,0 +1,130 @@ +"""List and audition the PC voices available to the simulator. + + python scripts/voices.py # list what is installed + python scripts/voices.py --demo # speak a sample line in each voice + python scripts/voices.py --try Zira # hear one voice, with .env's rate/volume + +Whichever you prefer goes in .env: + + MOCK_LOCAL_AUDIO=true + MOCK_VOICE=Zira + +IMPORTANT - this is the *simulator's* voice, not the robot's. The real AGIBOT A3 +synthesises speech on-board; AgiBot does not publish which engine or timbre it +uses, so no PC voice can be claimed to match it. This is a stand-in for +rehearsing a demo. See docs/AGIBOT_A3_INTEGRATION.md. +""" + +from __future__ import annotations + +import argparse +import platform +import sys +import time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + + +def _use_utf8_console() -> None: + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + +_use_utf8_console() + +from backend.config.settings import get_settings # noqa: E402 +from backend.robot.local_audio import create_local_voice, list_local_voices # noqa: E402 + +SAMPLE = "Good afternoon, and welcome to our showroom. I am the AGIBOT A3 humanoid robot." + + +def speak_with(hint, text, rate, volume, pitch=0) -> None: + voice = create_local_voice(voice_hint=hint, rate=rate, volume=volume, pitch=pitch) + if voice is None: + print(" No speech engine available on this PC.") + return + try: + voice.start(text) + while voice.is_speaking(): + time.sleep(0.05) + finally: + voice.close() + + +def main() -> None: + parser = argparse.ArgumentParser(description="List and audition simulator voices.") + parser.add_argument("--demo", action="store_true", help="Speak a sample in every voice.") + parser.add_argument("--try", dest="which", metavar="NAME", help="Speak a sample in one voice.") + parser.add_argument("--text", default=SAMPLE, help="Say this instead of the default sample.") + args = parser.parse_args() + + settings = get_settings() + rate, volume = settings.mock.speech_rate, settings.mock.speech_volume + pitch = settings.mock.speech_pitch + + if args.which: + print('\n Voice "{0}" (rate {1}, volume {2}, pitch {3})'.format( + args.which, rate, volume, pitch)) + print(" " + args.text + "\n") + speak_with(args.which, args.text, rate, volume, pitch) + return + + if settings.mock.voice_engine == "gemini": + from backend.robot.gemini_voice import VOICES as GEMINI_VOICES + + current = (settings.mock.gemini_voice or "").lower() + print("\n Gemini neural voices (active: MOCK_VOICE_ENGINE=gemini)") + print(" " + "-" * 58) + for name, tone in sorted(GEMINI_VOICES.items()): + mark = "*" if name.lower() == current else " " + note = " <- young + male, closest to the robot's Yunxiao" \ + if name in ("Puck", "Fenrir") else "" + print(" {0} {1:<16} {2}{3}".format(mark, name, tone, note)) + print("") + print(" * = current GEMINI_VOICE. Change it in .env, then warm your lines:") + print(' python scripts/warm_voice.py "Welcome to our showroom."') + + voices = list_local_voices() + print("\n System voices on this PC (used when MOCK_VOICE_ENGINE=system,") + print(" and as the fallback if Gemini fails)") + print(" " + "-" * 58) + if not voices: + if platform.system().lower().startswith("win"): + print(" None found.") + else: + print(" Listing is Windows-only; on macOS the simulator uses `say`,") + print(" on Linux `espeak-ng`.") + return + + for description, source in voices: + marker = "*" if settings.mock.voice and settings.mock.voice.lower() in description.lower() else " " + print(" {0} {1:<48} [{2}]".format(marker, description, source)) + + print("") + print(" * = currently selected by MOCK_VOICE={0}".format(settings.mock.voice or "(unset)")) + print(" 'onecore' voices are Windows' newer, better-sounding set.") + print("") + print(" Hear one: python scripts/voices.py --try Zira") + print(" Hear all: python scripts/voices.py --demo") + print("") + print(" Want more voices (including Mandarin, to match the A3's demo language)?") + print(" Windows Settings > Time & language > Speech > Manage voices > Add voices.") + print("") + + if args.demo: + for description, _ in voices: + short = description.replace("Microsoft ", "").split(" - ")[0] + print(" ->", description) + speak_with(description, "{0} speaking. {1}".format(short, args.text), + rate, volume, pitch) + time.sleep(0.4) + print("") + + +if __name__ == "__main__": + main() diff --git a/scripts/warm_voice.py b/scripts/warm_voice.py new file mode 100644 index 0000000..965db2d --- /dev/null +++ b/scripts/warm_voice.py @@ -0,0 +1,125 @@ +"""Pre-synthesise demo lines so the Gemini voice speaks instantly. + +Cloud TTS costs a network round trip - roughly 4 s for a sentence, 8 s for a +paragraph. That is fine while rehearsing and painful in front of an audience. +This warms the on-disk cache ahead of time, so a warmed line plays with **no +delay at all**. + + python scripts/warm_voice.py "Good afternoon, welcome to our showroom." + python scripts/warm_voice.py --file demo_lines.txt + python scripts/warm_voice.py --stats + python scripts/warm_voice.py --clear + +Audio is saved in audio_library/ as ordinary .wav files with readable names, so +it survives restarts, plays in any media player, and needs no internet once +saved. The server reads the same folder - warm before you present. + +Only relevant when MOCK_VOICE_ENGINE=gemini. +""" + +from __future__ import annotations + +import argparse +import sys +import time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + + +def _use_utf8_console() -> None: + for stream in (sys.stdout, sys.stderr): + try: + if stream is not None and hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="replace") + except Exception: + pass + + +_use_utf8_console() + +from backend.config.settings import get_settings # noqa: E402 +from backend.robot.gemini_voice import GeminiVoice # noqa: E402 + + +def build(settings) -> GeminiVoice: + return GeminiVoice( + api_key=settings.mock.gemini_api_key or "", + model=settings.mock.gemini_model, + voice=settings.mock.gemini_voice, + style=settings.mock.gemini_style, + chunk_chars=settings.mock.gemini_chunk_chars, + fallback=None, + ) + + +def main() -> None: + parser = argparse.ArgumentParser(description="Pre-synthesise lines for the Gemini voice.") + parser.add_argument("lines", nargs="*", help="Lines to warm.") + parser.add_argument("--file", help="Text file, one line per utterance.") + parser.add_argument("--stats", action="store_true", help="Show library size and exit.") + parser.add_argument("--clear", action="store_true", help="Delete all saved audio and exit.") + args = parser.parse_args() + + settings = get_settings() + if not settings.mock.gemini_api_key: + print("\n GEMINI_API_KEY is not set in .env - nothing to warm.\n") + raise SystemExit(1) + + voice = build(settings) + + if args.clear: + voice.clear_cache() + print("\n Cache cleared.\n") + return + + if args.stats or not (args.lines or args.file): + stats = voice.cache_stats() + print("\n Saved audio") + print(" " + "-" * 52) + print(" Entries : {0}".format(stats["entries"])) + print(" Size : {0:.1f} MB".format(stats["bytes"] / 1_048_576)) + print(" Location: {0}".format(stats["dir"])) + print(" Voice : {0} ({1})".format(settings.mock.gemini_voice, settings.mock.gemini_model)) + if not (args.lines or args.file): + print("") + print(' Warm a line: python scripts/warm_voice.py "Welcome to our showroom."') + print(" Warm a file: python scripts/warm_voice.py --file demo_lines.txt") + print("") + return + + lines = list(args.lines) + if args.file: + path = Path(args.file) + if not path.exists(): + print("\n No such file: {0}\n".format(path)) + raise SystemExit(1) + lines += [ln.strip() for ln in path.read_text(encoding="utf-8").splitlines() if ln.strip()] + + print("\n Warming {0} line(s) with voice '{1}'...".format(len(lines), settings.mock.gemini_voice)) + print(" " + "-" * 60) + total_new = 0 + started = time.time() + for index, line in enumerate(lines, 1): + t0 = time.time() + try: + fetched = voice.warm(line) + except Exception as exc: + print(" {0:>3}. FAILED {1}".format(index, exc)) + continue + total_new += fetched + state = "cached already" if fetched == 0 else "fetched {0} chunk(s)".format(fetched) + preview = line if len(line) <= 48 else line[:45] + "..." + print(" {0:>3}. {1:<50} {2:>16} {3:5.1f}s".format( + index, preview, state, time.time() - t0)) + + stats = voice.cache_stats() + print(" " + "-" * 60) + print(" {0} new clip(s) in {1:.1f}s. Library now holds {2} clips ({3:.1f} MB).".format( + total_new, time.time() - started, stats["entries"], stats["bytes"] / 1_048_576)) + print(" These lines will now speak instantly.\n") + voice.close() + + +if __name__ == "__main__": + main() diff --git a/start.bat b/start.bat new file mode 100644 index 0000000..8a3afaf --- /dev/null +++ b/start.bat @@ -0,0 +1,63 @@ +@echo off +REM =========================================================================== +REM AGIBOT A3 - Voice Control | Windows launcher +REM Double-click this file, then open http://localhost:8000 +REM =========================================================================== +setlocal +cd /d "%~dp0" + +echo. +echo AGIBOT A3 - Voice Control +echo ---------------------------------------------- + +REM --- locate Python --------------------------------------------------------- +set "PY=python" +where python >nul 2>&1 +if errorlevel 1 ( + where py >nul 2>&1 + if errorlevel 1 ( + echo [ERROR] Python was not found on this PC. + echo Install Python 3.9+ from https://python.org and tick + echo "Add python.exe to PATH" during setup. + echo. + pause + exit /b 1 + ) + set "PY=py -3" +) + +REM --- first-run config ------------------------------------------------------ +if not exist ".env" ( + echo Creating .env from .env.example ... + copy /y ".env.example" ".env" >nul +) + +REM --- dependencies ---------------------------------------------------------- +%PY% -c "import fastapi, uvicorn, dotenv, httpx" >nul 2>&1 +if errorlevel 1 ( + echo Installing dependencies ^(first run only^) ... + %PY% -m pip install --disable-pip-version-check -q -r requirements.txt + if errorlevel 1 ( + echo [ERROR] Dependency installation failed. Run this by hand: + echo %PY% -m pip install -r requirements.txt + echo. + pause + exit /b 1 + ) +) + +REM --- run ------------------------------------------------------------------- +echo Starting server ... http://localhost:8000 +echo Press Ctrl+C to stop. +echo. + +REM Open the browser a few seconds late, so the page loads against a server that +REM is already listening instead of showing a connection error. +start "" /min cmd /c "timeout /t 4 /nobreak >nul & start "" http://localhost:8000" + +%PY% backend\main.py + +echo. +echo Server stopped. +pause +endlocal diff --git a/start.sh b/start.sh new file mode 100644 index 0000000..981683c --- /dev/null +++ b/start.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash +# =========================================================================== +# AGIBOT A3 - Voice Control | Linux / macOS launcher +# ./start.sh then open http://localhost:8000 +# =========================================================================== +set -euo pipefail +cd "$(dirname "$0")" + +echo +echo " AGIBOT A3 - Voice Control" +echo " ----------------------------------------------" + +PY="${PYTHON:-python3}" +if ! command -v "$PY" >/dev/null 2>&1; then + echo " [ERROR] python3 not found. Install Python 3.9 or newer." + exit 1 +fi + +if [ ! -f .env ]; then + echo " Creating .env from .env.example ..." + cp .env.example .env +fi + +if ! "$PY" -c "import fastapi, uvicorn, dotenv, httpx" >/dev/null 2>&1; then + echo " Installing dependencies (first run only) ..." + "$PY" -m pip install --disable-pip-version-check -q -r requirements.txt +fi + +echo " Starting server ... open http://localhost:8000" +echo " Press Ctrl+C to stop." +echo +exec "$PY" backend/main.py